From 39153510d2f5ed47b98f6dd5475be165b4c9a29f Mon Sep 17 00:00:00 2001 From: PromptEngineer <134474669+PromtEngineer@users.noreply.github.com> Date: Sun, 9 Aug 2026 09:37:44 -0700 Subject: [PATCH 01/37] Rearchitect for doc/code parity; evidence-gated defaults; E2E fixes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Full rebuild driven by a verified audit of ~180 doc/code discrepancies and 52 dead-code findings, then hardened by a browser-driven end-to-end test and an evidence-gated adoption process (see eval/ and Documentation/research/ in the follow-up commits). Core rearchitecture: - Single RAG API server (api_server_with_progress.py deleted); factory.py is the only factory; main.py is master config + a thin working CLI - Backend gateway reads RAG_API_URL; every documented env var is actually read (GENERATION_MODEL, ENRICHMENT_MODEL, EMBEDDING_MODEL, RERANKER_MODEL, LANCEDB_PATH, DB_PATH, NEXT_PUBLIC_*); Docker wiring fixed end to end - Hybrid retrieval = LanceDB FTS + dense fused by RRF; retrieval_mode honored; no-op knobs (denseWeight, chunkOverlap, BM25 config, vision path) removed - Streamed turns persisted via POST /sessions//messages/save with sources and the pipeline step cascade in message metadata; backend is the sole writer of chat rows - Failed document conversion is a hard error, never a silent empty index; index status advances created->built; thinking-model JSON calls fixed (top-level think:false — chat_template_kwargs is ignored by /api/generate) Evidence-gated defaults (measurements in eval/DECISIONS.md): - Embedder: microsoft/harrier-oss-v1-0.6b with query-side instruction prefix (mixed-corpus first-stage nDCG@10 0.915 vs 0.875 for the 8GB Qwen3-4B) - Default profile reranking OFF: bge-reranker-v2-m3 measured net-negative on this first stage; Qwen3-Reranker-4B (0.977) is the lazy opt-in via a custom scorer (rerankers 0.10.0 silently mis-scores Qwen3 rerankers) - Per-table embedder-identity markers + L2 normalization (text_pages_v4); same-width model swaps now refuse instead of corrupting - Gateway routing is a deterministic gate (~750ms/message saved; 155/155 regression tests); evidence-sufficiency retrieval retry; decomposition moved to the rerank stage; VERIFIER_MODEL seam; graph module removed Dead code removed throughout (frontend components, dup requirements, broken scripts); all Documentation/ rewritten to describe only shipped behavior. Co-Authored-By: Claude Fable 5 --- .env.example | 96 ++ .github/ISSUE_TEMPLATE/bug_report.md | 2 +- .gitignore | 12 +- CONTRIBUTING.md | 380 ++++---- DOCKER_README.md | 327 ++++--- DOCKER_TROUBLESHOOTING.md | 429 +++++---- Dockerfile.backend | 13 +- Dockerfile.frontend | 18 +- Dockerfile.rag-api | 5 +- Documentation/api_reference.md | 421 ++++++--- Documentation/architecture_overview.md | 224 +++-- Documentation/deployment_guide.md | 619 +++++++------ Documentation/docker_usage.md | 462 +++++----- Documentation/improvement_plan.md | 141 +-- Documentation/indexing_pipeline.md | 862 ++++++------------ Documentation/installation_guide.md | 315 ++++--- Documentation/prompt_inventory.md | 99 ++- Documentation/quick_start.md | 293 ++++--- Documentation/retrieval_pipeline.md | 796 ++++++----------- Documentation/system_overview.md | 674 ++++++++------- Documentation/triage_system.md | 137 +-- Documentation/verifier.md | 129 ++- README.md | 825 +++++++++--------- WATSONX_README.md | 263 +++--- backend/README.md | 212 +++-- backend/database.py | 56 +- backend/ollama_client.py | 75 +- backend/requirements.txt | 1 - backend/server.py | 865 +++++++++--------- backend/simple_pdf_processor.py | 214 ----- backend/test_backend.py | 153 ---- backend/test_gateway_routing.py | 164 ++++ backend/test_ollama_connectivity.py | 37 - batch_indexing_config.json | 19 - create_index_script.py | 193 +++-- demo_batch_indexing.py | 386 --------- docker-compose.local-ollama.yml | 41 +- docker-compose.yml | 38 +- docker.env | 28 +- env.example.watsonx | 66 +- package-lock.json | 632 -------------- package.json | 3 - public/.gitkeep | 0 public/file.svg | 1 - public/globe.svg | 1 - public/next.svg | 1 - public/vercel.svg | 1 - public/window.svg | 1 - rag_system/DOCUMENTATION.md | 474 +++++++++- rag_system/README.md | 333 +++++-- rag_system/agent/loop.py | 358 ++++---- rag_system/agent/verifier.py | 179 +++- rag_system/api_server.py | 962 ++++++++------------- rag_system/api_server_with_progress.py | 443 ---------- rag_system/factory.py | 108 +-- rag_system/indexing/contextualizer.py | 37 - rag_system/indexing/embedders.py | 193 ++++- rag_system/indexing/graph_extractor.py | 86 -- rag_system/indexing/latechunk.py | 30 +- rag_system/indexing/multimodal.py | 124 --- rag_system/indexing/overview_builder.py | 6 +- rag_system/indexing/representations.py | 117 ++- rag_system/ingestion/chunking.py | 18 +- rag_system/ingestion/docling_chunker.py | 38 +- rag_system/ingestion/document_converter.py | 141 ++- rag_system/main.py | 404 +++------ rag_system/pipelines/indexing_pipeline.py | 205 +++-- rag_system/pipelines/retrieval_pipeline.py | 626 ++++++++++---- rag_system/requirements.txt | 7 +- rag_system/rerankers/reranker.py | 148 +++- rag_system/retrieval/query_transformer.py | 46 +- rag_system/retrieval/retrievers.py | 320 ++++--- rag_system/utils/batch_processor.py | 70 +- rag_system/utils/logging_utils.py | 11 - rag_system/utils/ollama_client.py | 51 +- rag_system/utils/validate_model_config.py | 219 ----- rag_system/utils/watsonx_client.py | 21 - requirements-docker.txt | 17 +- requirements.txt | 18 +- run_system.py | 234 ++++- setup_rag_system.sh | 520 ----------- simple_create_index.sh | 228 ----- src/components/IndexForm.tsx | 29 +- src/components/IndexWizard.tsx | 72 -- src/components/ModelSelect.tsx | 9 +- src/components/SessionIndexInfo.tsx | 9 +- src/components/demo.tsx | 29 +- src/components/ui/GlassSelect.tsx | 13 - src/components/ui/badge.tsx | 46 - src/components/ui/chat-bubble-demo.tsx | 103 --- src/components/ui/chat-bubble.tsx | 98 --- src/components/ui/chat-input.tsx | 2 +- src/components/ui/chat-settings-modal.tsx | 2 +- src/components/ui/conversation-page.tsx | 11 +- src/components/ui/dropdown-menu.tsx | 257 ------ src/components/ui/empty-chat-state.tsx | 292 ------- src/components/ui/localgpt-chat.tsx | 170 ---- src/components/ui/message-loading.tsx | 48 - src/components/ui/quick-chat.tsx | 4 +- src/components/ui/separator.tsx | 28 - src/components/ui/session-chat.tsx | 100 ++- src/components/ui/sidebar.tsx | 260 ------ src/components/ui/skeleton.tsx | 13 - src/lib/api.ts | 169 ++-- src/test-upload.html | 54 -- src/utils/textNormalization.ts | 25 +- start-docker.sh | 47 +- system_health_check.py | 47 +- tailwind.config.js | 11 - test_docker_build.sh | 2 +- test_markdown_streaming.js | 80 -- 111 files changed, 8404 insertions(+), 11148 deletions(-) create mode 100644 .env.example delete mode 100644 backend/simple_pdf_processor.py delete mode 100644 backend/test_backend.py create mode 100644 backend/test_gateway_routing.py delete mode 100644 backend/test_ollama_connectivity.py delete mode 100644 batch_indexing_config.json delete mode 100644 demo_batch_indexing.py create mode 100644 public/.gitkeep delete mode 100644 public/file.svg delete mode 100644 public/globe.svg delete mode 100644 public/next.svg delete mode 100644 public/vercel.svg delete mode 100644 public/window.svg delete mode 100644 rag_system/api_server_with_progress.py delete mode 100644 rag_system/indexing/graph_extractor.py delete mode 100644 rag_system/indexing/multimodal.py delete mode 100644 rag_system/utils/validate_model_config.py delete mode 100644 setup_rag_system.sh delete mode 100755 simple_create_index.sh delete mode 100644 src/components/IndexWizard.tsx delete mode 100644 src/components/ui/GlassSelect.tsx delete mode 100644 src/components/ui/badge.tsx delete mode 100644 src/components/ui/chat-bubble-demo.tsx delete mode 100644 src/components/ui/dropdown-menu.tsx delete mode 100644 src/components/ui/empty-chat-state.tsx delete mode 100644 src/components/ui/localgpt-chat.tsx delete mode 100644 src/components/ui/message-loading.tsx delete mode 100644 src/components/ui/separator.tsx delete mode 100644 src/components/ui/sidebar.tsx delete mode 100644 src/components/ui/skeleton.tsx delete mode 100644 src/test-upload.html delete mode 100644 tailwind.config.js delete mode 100644 test_markdown_streaming.js diff --git a/.env.example b/.env.example new file mode 100644 index 00000000..b97e2afe --- /dev/null +++ b/.env.example @@ -0,0 +1,96 @@ +# localGPT environment configuration +# +# Copy to .env and edit. Every variable below is read by code; the value shown +# after "default:" is what the code uses when the variable is unset. +# +# cp .env.example .env +# +# For Docker use docker.env instead (it is passed with --env-file and also +# supplies build-time values for the frontend). + +# --------------------------------------------------------------------------- +# Services +# --------------------------------------------------------------------------- + +# Ollama server. Read by backend/ollama_client.py and rag_system/main.py. +# In Docker this becomes http://host.docker.internal:11434. +# default: http://localhost:11434 +OLLAMA_HOST=http://localhost:11434 + +# Base URL of the RAG API. backend/server.py builds /chat and /index from it. +# In Docker compose this is http://rag-api:8001. +# default: http://localhost:8001 +RAG_API_URL=http://localhost:8001 + +# Browser-facing URLs, read by the Next.js frontend (src/lib/api.ts). +# NEXT_PUBLIC_* values are inlined at build time, so change them before `npm run build`. +# default: http://localhost:8000 +NEXT_PUBLIC_API_URL=http://localhost:8000 +# default: http://localhost:8001 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 + +# --------------------------------------------------------------------------- +# Storage +# --------------------------------------------------------------------------- + +# SQLite database holding sessions, messages and index metadata. +# Set this to a shared path so the backend and the RAG API use one file. +# default: backend/chat_data.db (local) or /app/backend/chat_data.db (Docker) +# DB_PATH=backend/chat_data.db + +# LanceDB vector store. Defaults to the `storage.lancedb_uri` of the active +# pipeline config in rag_system/main.py. +# default: ./lancedb +# LANCEDB_PATH=./lancedb + +# --------------------------------------------------------------------------- +# Models +# --------------------------------------------------------------------------- + +# Answer generation (Ollama). Options: qwen3.6:27b (high-end, ~17GB), qwen3.5:4b (light). +# default: qwen3.5:9b +GENERATION_MODEL=qwen3.5:9b + +# Routing, triage, query decomposition, contextual enrichment and verification +# (Ollama). Light option: qwen3.5:2b. +# default: qwen3.5:4b +ENRICHMENT_MODEL=qwen3.5:4b + +# Embeddings (HuggingFace). The default is MIT-licensed, 1.2GB, 1024-dim, and +# measured best on our gold set (eval/DECISIONS.md). +# Option: Qwen/Qwen3-Embedding-4B for multilingual / long-context corpora. +# Changing this requires re-indexing every existing index - the stored vectors +# belong to the old model's vector space. localGPT records the embedding model +# on each table and refuses to query it with a different one. +# default: microsoft/harrier-oss-v1-0.6b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b + +# Reranker (HuggingFace). Only loaded when reranking is switched on - the +# default profile ships with it OFF, because the first stage above already +# outranks the cheap cross-encoder (eval/DECISIONS.md). When you do switch it +# on (UI "AI reranker" toggle, or reranker.enabled in the profile) this is the +# model that gets loaded, lazily. +# Options: BAAI/bge-reranker-v2-m3 (low latency, only pays off with a weaker +# embedder), answerdotai/answerai-colbert-small-v1, Qwen/Qwen3-Reranker-0.6B. +# default: Qwen/Qwen3-Reranker-4B +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B + +# --------------------------------------------------------------------------- +# Optional tuning +# --------------------------------------------------------------------------- + +# Seconds the backend waits for a RAG API chat response. +# default: 600 +# RAG_API_TIMEOUT=600 + +# Seconds the backend waits for a RAG API indexing run. +# default: 3600 +# RAG_API_INDEX_TIMEOUT=3600 + +# LLM backend selector for rag_system (`ollama` or `watsonx`). +# WatsonX additionally needs the WATSONX_* variables - see env.example.watsonx. +# default: ollama +# LLM_BACKEND=ollama + +# HuggingFace token, only needed for gated model downloads. +# HF_TOKEN= diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md index 3610c914..c57e4945 100644 --- a/.github/ISSUE_TEMPLATE/bug_report.md +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -53,7 +53,7 @@ Please include relevant error messages or logs: ## 🔧 Configuration - Deployment method: [Docker / Direct Python] -- Models used: [e.g. qwen3:0.6b, qwen3:8b] +- Models used: [e.g. qwen3.5:9b, qwen3.5:4b] - Document types: [e.g. PDF, DOCX, TXT] ## 📎 Additional Context diff --git a/.gitignore b/.gitignore index b3358283..a384b19b 100644 --- a/.gitignore +++ b/.gitignore @@ -72,7 +72,17 @@ backend/chroma_db/** rag_system/documents/ *.pdf -# Ensure docker.env remains tracked +# Ensure docker.env and .env.example remain tracked !docker.env +!.env.example !backend/chat_data.db +# Phase 0 evaluation harness (eval/) — rebuildable artefacts only. +# The corpora, gold set, scripts and BASELINE.md are tracked. +eval/.eval_indexes/ +eval/results/ +!eval/corpora/*.pdf + + +# local virtualenv +.venv/ diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 7f281a60..469f6559 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,12 +1,13 @@ # Contributing to LocalGPT -Thank you for your interest in contributing to LocalGPT! This guide will help you get started with contributing to our private document intelligence platform. +Thank you for your interest in contributing to LocalGPT! This guide will help you get +started with contributing to our private document intelligence platform. ## 🚀 Quick Start for Contributors ### Prerequisites -- Python 3.8+ (we test with 3.11.5) -- Node.js 16+ (we test with 23.10.0) +- Python 3.10+ (3.11 recommended) +- Node.js 20+ - Git - Ollama (for local AI models) @@ -15,44 +16,51 @@ Thank you for your interest in contributing to LocalGPT! This guide will help yo 1. **Fork and Clone** ```bash # Fork the repository on GitHub, then clone your fork - git clone https://github.com/YOUR_USERNAME/multimodal_rag.git - cd multimodal_rag - + git clone https://github.com/YOUR_USERNAME/localGPT.git + cd localGPT + # Add upstream remote - git remote add upstream https://github.com/PromtEngineer/multimodal_rag.git + git remote add upstream https://github.com/PromtEngineer/localGPT.git ``` 2. **Set Up Development Environment** ```bash # Install Python dependencies pip install -r requirements.txt - + # Install Node.js dependencies npm install - - # Install Ollama and models + + # Install Ollama and the two default models curl -fsSL https://ollama.ai/install.sh | sh - ollama pull qwen3:0.6b - ollama pull qwen3:8b + ollama pull qwen3.5:9b # generation + ollama pull qwen3.5:4b # routing / enrichment / verification ``` + Defaults work without any configuration. To override model names, service URLs or + `DB_PATH`, put them in a `.env` at the repository root — `rag_system/main.py` calls + `load_dotenv()` on import, and `backend/server.py` imports it. `.env.example` lists + every variable the code actually reads. + 3. **Verify Setup** ```bash - # Run health check + # Config, agent construction, embedding model and LanceDB access python system_health_check.py - - # Start development system + + # Start Ollama + RAG API + backend + frontend python run_system.py --mode dev + + # In another shell: are all four services actually healthy? + python run_system.py --health ``` ## 📋 Development Workflow ### Branch Strategy -We use a feature branch workflow: +We use a feature branch workflow off `main`, which is the only long-lived branch: - `main` - Production-ready code -- `docker` - Docker deployment features and documentation - `feature/*` - New features - `fix/*` - Bug fixes - `docs/*` - Documentation updates @@ -64,27 +72,17 @@ We use a feature branch workflow: # Update your main branch git checkout main git pull upstream main - + # Create feature branch git checkout -b feature/your-feature-name ``` 2. **Make Your Changes** - - Follow our [coding standards](#coding-standards) - - Write tests for new functionality - - Update documentation as needed + - Follow our [coding standards](#-coding-standards) + - Update documentation as needed — docs are expected to describe what the code does + today, not what it might do later -3. **Test Your Changes** - ```bash - # Run health checks - python system_health_check.py - - # Test specific components - python -m pytest tests/ -v - - # Test system integration - python run_system.py --health - ``` +3. **Check Your Changes** — see [Verifying changes](#-verifying-changes) 4. **Commit Your Changes** ```bash @@ -103,25 +101,23 @@ We use a feature branch workflow: ### 🐛 Bug Fixes - Check existing issues first - Include reproduction steps -- Add tests to prevent regression +- Describe how you verified the fix ### ✨ New Features - Discuss in issues before implementing - Follow existing architecture patterns -- Include comprehensive tests - Update documentation ### 📚 Documentation - Fix typos and improve clarity - Add examples and use cases -- Update API documentation -- Improve setup guides +- **Verify every command, port, endpoint, config key, model name and default against the + code before writing it.** A doc that describes a feature the code does not have is worse + than no doc. ### 🧪 Testing -- Add unit tests -- Improve integration tests -- Add performance benchmarks -- Test edge cases +- There is no automated test suite yet; adding one is welcome +- Until then, describe the manual verification you ran in the PR ## 📝 Coding Standards @@ -131,30 +127,38 @@ We follow PEP 8 with some modifications: ```python # Use type hints -def process_document(file_path: str, config: Dict[str, Any]) -> ProcessingResult: - """Process a document with the given configuration. - +def convert_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, Any]]]: + """Convert a document to Markdown, preserving layout and tables. + Args: file_path: Path to the document file - config: Processing configuration dictionary - + Returns: - ProcessingResult object with metadata and chunks + A list of (markdown, metadata) tuples, optionally with the + DoclingDocument as a third element. """ - pass + ... # Use descriptive variable names -embedding_model_name = "Qwen/Qwen3-Embedding-0.6B" -retrieval_results = retriever.search(query, top_k=20) - -# Use dataclasses for structured data -@dataclass -class IndexingConfig: - embedding_batch_size: int = 50 - enable_late_chunking: bool = True - chunk_size: int = 512 +embedding_model_name = "microsoft/harrier-oss-v1-0.6b" +retrieved_docs = retriever.retrieve(text_query=query, table_name=table, k=20) ``` +Conventions specific to this codebase: + +- **Configuration is plain dicts**, defined once in `rag_system/main.py` and handed out as + deep copies by `rag_system/factory.py::get_pipeline_config()`. Do not introduce a second + place where a model name or a default lives. +- **Every config key must be read by code.** If you add a key, wire it; if you find a key + nothing reads, delete it rather than documenting it. +- **No hardcoded model names or embedding dimensions.** Resolve models from + `OLLAMA_CONFIG` / `EXTERNAL_MODELS` (which are environment-overridable), and derive + vector width from the embeddings the loaded model produced. +- **Fail loudly on misconfiguration, degrade quietly on optional components.** A missing + `embedding_model_name` raises; a reranker that cannot be loaded logs a warning and the + query proceeds without reranking. +- Keep comments purposeful. No decorative banners for trivial code. + ### TypeScript/React Code Style ```typescript @@ -162,19 +166,18 @@ class IndexingConfig: interface ChatMessage { id: string; content: string; - role: 'user' | 'assistant'; - timestamp: Date; - sources?: DocumentSource[]; + sender: 'user' | 'assistant'; + timestamp: string; } // Use functional components with hooks const ChatInterface: React.FC = ({ sessionId }) => { const [messages, setMessages] = useState([]); - + const handleSendMessage = useCallback(async (content: string) => { // Implementation }, [sessionId]); - + return (
{/* Component JSX */} @@ -183,126 +186,122 @@ const ChatInterface: React.FC = ({ sessionId }) => { }; ``` +- All API calls go through `src/lib/api.ts`. Base URLs come from `NEXT_PUBLIC_API_URL` and + `NEXT_PUBLIC_RAG_API_URL`; never hardcode a host. +- Response types in `src/lib/api.ts` must match what the server actually returns. + ### File Organization ``` rag_system/ -├── agent/ # ReAct agent implementation -├── indexing/ # Document processing and indexing -├── retrieval/ # Search and retrieval components -├── pipelines/ # End-to-end processing pipelines -├── rerankers/ # Result reranking implementations -└── utils/ # Shared utilities +├── main.py # Master configuration + CLI +├── factory.py # The single factory (get_agent / get_indexing_pipeline) +├── api_server.py # HTTP API on :8001 +├── agent/ # Triage, decomposition, orchestration and verification loop +├── ingestion/ # Document conversion (Docling) and chunking +├── indexing/ # Embedding, LanceDB writing, enrichment, overviews +├── retrieval/ # Retrievers and query transformation +├── pipelines/ # End-to-end indexing and retrieval pipelines +├── rerankers/ # Reranking and sentence pruning +└── utils/ # LLM clients and shared helpers + +backend/ # Gateway on :8000 (sessions, uploads, SQLite) src/ ├── components/ # React components -├── lib/ # Utility functions and API clients -└── app/ # Next.js app router pages +├── lib/ # API client and shared types +├── utils/ # Small helpers +└── app/ # Next.js app router pages ``` -## 🧪 Testing Guidelines +## 🧪 Verifying changes -### Unit Tests -```python -# Test file: tests/test_embeddings.py -import pytest -from rag_system.indexing.embedders import HuggingFaceEmbedder - -def test_embedding_generation(): - embedder = HuggingFaceEmbedder("sentence-transformers/all-MiniLM-L6-v2") - embeddings = embedder.create_embeddings(["test text"]) - - assert embeddings.shape[0] == 1 - assert embeddings.shape[1] == 384 # Model dimension - assert embeddings.dtype == np.float32 +There is no `tests/` directory and no pytest suite. Use these checks: + +### Python +```bash +# Syntax check everything +find rag_system backend -name '*.py' -exec python -m py_compile {} + + +# Config, agent construction, embedding model, LanceDB access, sample query +python system_health_check.py + +# Are the running services healthy? (exits non-zero if not) +python run_system.py --health + +# Exercise a pipeline directly +python -m rag_system.main index ./path/to/docs --mode fast +python -m rag_system.main chat "test question" --mode fast ``` -### Integration Tests -```python -# Test file: tests/test_integration.py -def test_end_to_end_indexing(): - """Test complete document indexing pipeline.""" - agent = get_agent("test") - result = agent.index_documents(["test_document.pdf"]) - - assert result.success - assert len(result.indexed_chunks) > 0 +### Frontend +```bash +npx tsc --noEmit # type check +npm run lint # Next.js ESLint +npm run build # production build ``` -### Frontend Tests -```typescript -// Test file: src/components/__tests__/ChatInterface.test.tsx -import { render, screen, fireEvent } from '@testing-library/react'; -import { ChatInterface } from '../ChatInterface'; - -test('sends message when form is submitted', async () => { - render(); - - const input = screen.getByPlaceholderText('Type your message...'); - const button = screen.getByRole('button', { name: /send/i }); - - fireEvent.change(input, { target: { value: 'test message' } }); - fireEvent.click(button); - - expect(screen.getByText('test message')).toBeInTheDocument(); -}); +`next.config.ts` sets `eslint.ignoreDuringBuilds` and `typescript.ignoreBuildErrors`, so a +successful `npm run build` does **not** imply the code type-checks. Run `npx tsc --noEmit` +separately. + +### Docker +```bash +./test_docker_build.sh +``` + +### End-to-end smoke test +```bash +curl http://localhost:8000/health +curl http://localhost:8001/health +curl -X POST http://localhost:8001/chat \ + -H 'Content-Type: application/json' \ + -d '{"query": "what is this document about?"}' ``` ## 📖 Documentation Standards ### Code Documentation ```python -def create_index( - documents: List[str], - config: IndexingConfig, - progress_callback: Optional[Callable[[float], None]] = None -) -> IndexingResult: - """Create a searchable index from documents. - - This function processes documents through the complete indexing pipeline: - 1. Text extraction and chunking - 2. Embedding generation - 3. Vector database storage - 4. BM25 index creation - +def run(self, file_paths: List[str] | None = None, *, documents: List[str] | None = None): + """Process and index documents according to the pipeline configuration. + + Steps: Docling conversion -> chunking -> optional contextual enrichment -> + embedding into LanceDB (plus the native FTS index) -> optional late chunking. + Args: - documents: List of document file paths to index - config: Indexing configuration with model settings and parameters - progress_callback: Optional callback function for progress updates - - Returns: - IndexingResult containing success status, metrics, and any errors - + file_paths: Absolute paths of the documents to index + documents: Legacy alias for file_paths + Raises: - IndexingError: If document processing fails - ModelLoadError: If embedding model cannot be loaded - - Example: - >>> config = IndexingConfig(embedding_batch_size=32) - >>> result = create_index(["doc1.pdf", "doc2.pdf"], config) - >>> print(f"Indexed {result.chunk_count} chunks") + TypeError: If neither argument is supplied + ValueError: If the embeddings do not match the target table's vector width """ ``` -### API Documentation +### HTTP handler documentation + +Both servers are built on the standard library's `http.server`; there is no FastAPI, no +Pydantic models and no generated OpenAPI schema. Document a route with a handler docstring +and keep the field list in the relevant README in sync: + ```python -# Use OpenAPI/FastAPI documentation -@app.post("/chat", response_model=ChatResponse) -async def chat_endpoint(request: ChatRequest) -> ChatResponse: - """Chat with indexed documents. - - Send a natural language query and receive an AI-generated response - based on the indexed document collection. - - - **query**: The user's question or prompt - - **session_id**: Chat session identifier - - **search_type**: Type of search (vector, hybrid, bm25) - - **retrieval_k**: Number of documents to retrieve - - Returns a response with the AI-generated answer and source documents. +def handle_chat(self): + """POST /chat — answer a query with the agentic RAG pipeline. + + Body (camelCase accepted, normalised to snake_case): + query (required), session_id, table_name, model, + retrieval_mode (hybrid|vector_only|fts_only), force_rag, + query_decompose, ai_rerank, context_expand, verify, + retrieval_k, context_window_size, reranker_top_k. + + Returns {"answer": str, "source_documents": list}. """ ``` +Route tables live in [`backend/README.md`](backend/README.md) (port 8000) and +[`rag_system/DOCUMENTATION.md`](rag_system/DOCUMENTATION.md) (port 8001). + ## 🔧 Development Tools ### Recommended VS Code Extensions @@ -310,44 +309,20 @@ async def chat_endpoint(request: ChatRequest) -> ChatResponse: { "recommendations": [ "ms-python.python", - "ms-python.pylint", - "ms-python.black-formatter", "bradlc.vscode-tailwindcss", - "esbenp.prettier-vscode", "ms-vscode.vscode-typescript-next" ] } ``` -### Pre-commit Hooks -```bash -# Install pre-commit -pip install pre-commit - -# Set up hooks -pre-commit install - -# Run manually -pre-commit run --all-files -``` - -### Development Scripts -```bash -# Lint Python code -python -m pylint rag_system/ - -# Format Python code -python -m black rag_system/ - -# Type check -python -m mypy rag_system/ +### Formatters and linters -# Lint TypeScript -npm run lint +The repository ships an ESLint flat config (`eslint.config.mjs`, used by `npm run lint`) +and nothing else — there is no Black, pylint, mypy, Prettier or pre-commit configuration +checked in, and `npm run format` does not exist. If you run a Python formatter locally, +keep the diff limited to the lines you actually changed. -# Format TypeScript -npm run format -``` +Available npm scripts (`package.json`): `dev`, `build`, `start`, `lint`. ## 🐛 Issue Reporting @@ -356,10 +331,11 @@ When reporting bugs, please include: 1. **Environment Information** ``` - - OS: macOS 13.4 + - OS: macOS 15.5 - Python: 3.11.5 - - Node.js: 23.10.0 + - Node.js: 20.11.0 - Ollama: 0.9.5 + - LLM_BACKEND: ollama ``` 2. **Steps to Reproduce** @@ -371,7 +347,8 @@ When reporting bugs, please include: ``` 3. **Expected vs Actual Behavior** -4. **Error Messages and Logs** +4. **Error Messages and Logs** — the RAG API prints most of its pipeline trace to stdout; + `RAG_LOG_LEVEL=DEBUG` adds more 5. **Screenshots (if applicable)** ### Feature Requests @@ -391,11 +368,12 @@ We use semantic versioning (semver): - Patch: Bug fixes ### Release Checklist -- [ ] All tests pass -- [ ] Documentation updated +- [ ] `system_health_check.py` and `run_system.py --health` pass +- [ ] `npx tsc --noEmit` and `npm run build` pass +- [ ] Documentation updated and re-verified against the code - [ ] Version bumped in relevant files - [ ] Changelog updated -- [ ] Docker images built and tested +- [ ] Docker images built and tested (`./test_docker_build.sh`) - [ ] Release notes prepared ## 🤝 Community Guidelines @@ -418,8 +396,8 @@ We use semantic versioning (semver): 1. **Performance Optimization**: Improving indexing and retrieval speed 2. **Model Support**: Adding more embedding and generation models 3. **User Experience**: Enhancing the web interface -4. **Documentation**: Improving setup and usage guides -5. **Testing**: Expanding test coverage +4. **Documentation**: Keeping setup and usage guides truthful +5. **Testing**: Establishing an automated test suite ### Architecture Goals - **Modularity**: Components should be loosely coupled @@ -430,23 +408,29 @@ We use semantic versioning (semver): ## 📚 Additional Resources -### Learning Resources -- [RAG System Architecture Overview](Documentation/architecture_overview.md) +### Project documentation +- [RAG package overview](rag_system/README.md) +- [RAG reference: pipelines, config keys, HTTP API](rag_system/DOCUMENTATION.md) +- [Backend gateway](backend/README.md) +- [Watson X backend](WATSONX_README.md) +- [Architecture Overview](Documentation/architecture_overview.md) - [API Reference](Documentation/api_reference.md) - [Deployment Guide](Documentation/deployment_guide.md) -- [Troubleshooting Guide](DOCKER_TROUBLESHOOTING.md) +- [Docker Troubleshooting](DOCKER_TROUBLESHOOTING.md) ### External References -- [LangChain Documentation](https://python.langchain.com/) -- [Ollama Documentation](https://ollama.ai/docs) +- [Ollama](https://github.com/ollama/ollama) +- [Docling](https://github.com/docling-project/docling) +- [LanceDB](https://github.com/lancedb/lancedb) +- [rerankers](https://github.com/AnswerDotAI/rerankers) - [Next.js Documentation](https://nextjs.org/docs) -- [FastAPI Documentation](https://fastapi.tiangolo.com/) --- ## 🙏 Thank You! -Thank you for contributing to LocalGPT! Your contributions help make private document intelligence accessible to everyone. +Thank you for contributing to LocalGPT! Your contributions help make private document +intelligence accessible to everyone. For questions about contributing, please: 1. Check existing documentation @@ -454,4 +438,4 @@ For questions about contributing, please: 3. Create a new issue with the `question` label 4. Join our community discussions -Happy coding! 🚀 \ No newline at end of file +Happy coding! 🚀 diff --git a/DOCKER_README.md b/DOCKER_README.md index bc78e8b5..8b9d9648 100644 --- a/DOCKER_README.md +++ b/DOCKER_README.md @@ -1,10 +1,11 @@ # 🐳 LocalGPT Docker Deployment Guide -This guide covers running LocalGPT using Docker containers with local Ollama for optimal performance. +This guide covers running LocalGPT in Docker containers, with Ollama either on the +host (default, best performance) or as a container. ## 🚀 Quick Start -### Complete Setup (5 Minutes) +### Complete Setup ```bash # 1. Install Ollama locally curl -fsSL https://ollama.ai/install.sh | sh @@ -12,121 +13,215 @@ curl -fsSL https://ollama.ai/install.sh | sh # 2. Start Ollama server ollama serve -# 3. Install required models (in another terminal) -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# 3. Install the models (in another terminal) +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification # 4. Clone and start LocalGPT -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT ./start-docker.sh # 5. Access the application open http://localhost:3000 ``` +The first `up --build` installs the Python and Node dependencies and then the +`rag-api` container downloads its embedding model from HuggingFace (the reranker follows lazily on the first reranked query), +so allow several minutes before the UI is reachable. + +Running this from a script or CI? Add `-y` so the fallback prompt never blocks: + +```bash +./start-docker.sh local -y # or: NONINTERACTIVE=1 ./start-docker.sh +``` + ## 📋 Prerequisites -- **Docker Desktop** installed and running -- **Ollama** installed locally (required for best performance) -- **8GB+ RAM** (16GB recommended for larger models) -- **10GB+ free disk space** +- **Docker Desktop** (or Docker Engine 24+ with the Compose plugin), running +- **Ollama** on the host (recommended) or the containerized fallback below +- **8GB+ RAM** (16GB recommended) +- **20GB+ free disk space** — images, plus ~10GB of HuggingFace weights inside + `rag-api`, plus the Ollama models ## 🏗️ Architecture -### Current Setup (Local Ollama + Docker Containers) +### Default Setup (Local Ollama + Docker Containers) ``` ┌─────────────────┐ ┌─────────────────┐ ┌─────────────────┐ │ Frontend │────│ Backend │────│ RAG API │ │ (Container) │ │ (Container) │ │ (Container) │ │ Port: 3000 │ │ Port: 8000 │ │ Port: 8001 │ └─────────────────┘ └─────────────────┘ └─────────────────┘ - │ - │ API calls - ▼ - ┌─────────────────┐ - │ Ollama │ - │ (Local/Host) │ - │ Port: 11434 │ - └─────────────────┘ + │ │ + │ browser streams directly to :8001 │ + └──────────────────────────────────────────────┤ + │ API calls + ▼ + ┌─────────────────┐ + │ Ollama │ + │ (Host, default) │ + │ Port: 11434 │ + └─────────────────┘ ``` -**Why Local Ollama?** +**Why host Ollama by default?** - ✅ Better performance (direct GPU access) - ✅ Simpler setup (one less container) - ✅ Easier model management -- ✅ More reliable connection +- ✅ Models survive `docker system prune` + +### Containerized Ollama + +`docker-compose.yml` also defines an `ollama` service behind the `with-ollama` +profile, for machines where you would rather not install Ollama: + +```bash +./start-docker.sh container +# equivalently: +# OLLAMA_HOST=http://ollama:11434 \ +# docker compose --env-file docker.env --profile with-ollama up --build -d + +# Pull the models inside the container the first time +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +Its models persist in the named volume `ollama_data`. + +### Startup Order + +``` +rag-api (healthy: GET /health) → backend (healthy: GET /health) → frontend +``` + +`backend` declares `depends_on: rag-api: condition: service_healthy`, and `rag-api` +does not answer `/health` until the agent and the embedding model are loaded (the +reranker is fetched lazily on the first query that reranks). +Seeing `backend` sit in `created` for a few minutes on a cold start is normal — +follow `docker compose logs -f rag-api`. ## 🛠️ Container Details ### Frontend Container (rag-frontend) -- **Image**: Custom Node.js 18 build +- **Image**: `node:20-alpine`, built by `Dockerfile.frontend` - **Port**: 3000 - **Purpose**: Next.js web interface -- **Health Check**: HTTP GET to / -- **Memory**: ~500MB +- **Health Check**: busybox `wget -qO- http://localhost:3000` (the alpine image has no curl) +- **Build args**: `NEXT_PUBLIC_API_URL`, `NEXT_PUBLIC_RAG_API_URL` — inlined by `next build` -### Backend Container (rag-backend) -- **Image**: Custom Python 3.11 build +### Backend Container (rag-backend) +- **Image**: `python:3.11-slim`, built by `Dockerfile.backend` - **Port**: 8000 -- **Purpose**: Session management, chat history, API gateway -- **Health Check**: HTTP GET to /health -- **Memory**: ~300MB +- **Purpose**: sessions, indexes, uploads, chat history, gateway to the RAG API +- **Health Check**: `curl -f http://localhost:8000/health` +- **Working directory**: `/app`, started with `python backend/server.py` +- **Extras**: `sqlite3` CLI installed for database troubleshooting ### RAG API Container (rag-api) -- **Image**: Custom Python 3.11 build +- **Image**: `python:3.11-slim`, built by `Dockerfile.rag-api` - **Port**: 8001 -- **Purpose**: Document indexing, retrieval, AI processing -- **Health Check**: HTTP GET to /models -- **Memory**: ~2GB (varies with model usage) +- **Purpose**: document indexing, retrieval, the agent loop +- **Health Check**: `curl -f http://localhost:8001/health` +- **Memory**: dominated by the resident embedding and reranker models +- **Concurrency**: single-threaded — one chat or indexing run at a time ## 📂 Volume Mounts & Data -### Persistent Data -- `./lancedb/` → Vector database storage -- `./index_store/` → Document indexes and metadata -- `./shared_uploads/` → Uploaded document files -- `./backend/chat_data.db` → SQLite chat history database +### Persistent Data (bind mounts to host directories) +- `./lancedb/` → `/app/lancedb` — vectors and the native full-text index +- `./index_store/` → `/app/index_store` — document overviews +- `./shared_uploads/` → `/app/shared_uploads` — uploaded documents +- `./backend/` → `/app/backend` — `chat_data.db` + +### Named volumes +- `ollama_data` — only used by the optional containerized Ollama ### Shared Between Containers -All containers share access to document storage and databases through bind mounts. +`backend` and `rag-api` mount the same four directories and both set +`DB_PATH=/app/backend/chat_data.db`, so they share one SQLite file and one vector +store. HuggingFace model weights are **not** mounted — they live in the container's +writable layer and are re-downloaded whenever `rag-api` is recreated. ## 🔧 Configuration ### Environment Variables (docker.env) ```bash -# Ollama Configuration +# Ollama on the host. Every service declares +# extra_hosts: ["host.docker.internal:host-gateway"], so this resolves on Linux too. OLLAMA_HOST=http://host.docker.internal:11434 +# Containerized alternative: OLLAMA_HOST=http://ollama:11434 -# Service Configuration +# Service wiring NODE_ENV=production RAG_API_URL=http://rag-api:8001 + +# Browser-facing URLs, inlined into the frontend at build time NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 -# Database Paths (inside containers) -DATABASE_PATH=/app/backend/chat_data.db +# Shared SQLite database and vector store +DB_PATH=/app/backend/chat_data.db LANCEDB_PATH=/app/lancedb -UPLOADS_PATH=/app/shared_uploads + +# Models +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B +``` + +Pass it explicitly (`start-docker.sh` does this for you): + +```bash +docker compose --env-file docker.env up --build -d +``` + +Variables already exported in your shell take precedence over `--env-file`. + +`NEXT_PUBLIC_*` are **build-time** for Next.js. `docker-compose.yml` forwards them +to `Dockerfile.frontend` as build args; changing them at runtime does nothing to an +already-built image, so rebuild: + +```bash +NEXT_PUBLIC_API_URL=https://gpt.example.com/api \ +NEXT_PUBLIC_RAG_API_URL=https://gpt.example.com/rag \ +docker compose --env-file docker.env up -d --build frontend ``` ### Model Configuration -The system uses these models by default: -- **Embedding**: `Qwen/Qwen3-Embedding-0.6B` (1024 dimensions) -- **Generation**: `qwen3:0.6b` (fast) or `qwen3:8b` (high quality) -- **Reranking**: Built-in cross-encoder + +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation | `qwen3.5:9b` (Ollama) | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility | `qwen3.5:4b` (Ollama) | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` — HuggingFace, MIT, 1024 dims | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranking (**off by default**) | `Qwen/Qwen3-Reranker-4B` — HuggingFace, loaded lazily when switched on | `BAAI/bge-reranker-v2-m3`, `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | + +Only the two Ollama models need `ollama pull`. Embedding dimensions are read from +the loaded model, never hardcoded — **changing `EMBEDDING_MODEL` requires rebuilding +existing indexes**, and appending mismatched vectors to a LanceDB table raises an +explicit error. If the reranker cannot be loaded, the pipeline logs a warning and +continues without reranking. ## 🎯 Management Commands ### Start/Stop Services ```bash -# Start all services +# Start all services (local Ollama) ./start-docker.sh +# Start with containerized Ollama +./start-docker.sh container + # Stop all services ./start-docker.sh stop # Restart services ./start-docker.sh stop && ./start-docker.sh + +# Non-interactive (CI) +./start-docker.sh local -y ``` ### Monitor Services @@ -153,20 +248,23 @@ docker compose --env-file docker.env up --build -d # Stop manually docker compose down -# Rebuild specific service +# Rebuild a specific service (code is COPY-ed in, not mounted - +# a restart alone will not pick up source changes) docker compose build --no-cache rag-api docker compose up -d rag-api ``` ### Health Checks ```bash -# Test all endpoints curl -f http://localhost:3000 && echo "✅ Frontend OK" curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" +curl -f http://localhost:8001/health && echo "✅ RAG API OK" curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" ``` +`test_docker_build.sh` builds and smoke-tests each image individually against the +same endpoints. + ## 🐞 Debugging ### Access Container Shells @@ -177,7 +275,7 @@ docker compose exec rag-api bash # Backend container docker compose exec backend bash -# Frontend container +# Frontend container (alpine -> sh, not bash) docker compose exec frontend sh ``` @@ -185,31 +283,32 @@ docker compose exec frontend sh ```bash # Test RAG system initialization docker compose exec rag-api python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') print('✅ RAG System OK') " -# Test Ollama connection from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +# Test Ollama connection from the container +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags -# Check environment variables -docker compose exec rag-api env | grep OLLAMA +# Check the wiring the containers actually got +docker compose exec rag-api env | grep -E "OLLAMA|MODEL|DB_PATH|LANCEDB" +docker compose exec backend env | grep RAG_API_URL + +# Verify the backend can reach the RAG API by service name +docker compose exec backend curl -s http://rag-api:8001/health # View Python packages -docker compose exec rag-api pip list | grep -E "(torch|transformers|lancedb)" +docker compose exec rag-api pip list | grep -E "(torch|transformers|lancedb|docling|rerankers)" ``` ### Resource Monitoring ```bash -# Monitor container resources docker stats -# Check disk usage docker system df -df -h ./lancedb ./shared_uploads +du -sh lancedb shared_uploads index_store -# Check memory usage by service docker stats --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}\t{{.MemPerc}}" ``` @@ -219,7 +318,7 @@ docker stats --format "table {{.Name}}\t{{.CPUPerc}}\t{{.MemUsage}}\t{{.MemPerc} #### Container Won't Start ```bash -# Check logs for specific error +# Check logs for the specific error docker compose logs [service-name] # Rebuild from scratch @@ -231,42 +330,60 @@ docker system prune -f lsof -i :3000 -i :8000 -i :8001 ``` +#### `backend` never starts +It is waiting for `rag-api` to report healthy. Check: +```bash +docker compose ps +docker compose logs -f rag-api +docker compose exec rag-api curl -s http://localhost:8001/health +``` + #### Can't Connect to Ollama ```bash -# Verify Ollama is running +# Verify Ollama is running on the host curl http://localhost:11434/api/tags # Restart Ollama pkill ollama ollama serve -# Test from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +# Test from the container +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags +``` + +#### Every chat answers "Could not connect to the RAG API server" +The backend builds its URLs from `RAG_API_URL`; inside compose that must be +`http://rag-api:8001`, since `localhost` there is the backend container itself. +```bash +docker compose exec backend env | grep RAG_API_URL +docker compose exec backend curl -s http://rag-api:8001/health ``` #### Memory Issues ```bash -# Check memory usage docker stats --no-stream -free -h # On host +free -h # on the host -# Increase Docker memory limit # Docker Desktop → Settings → Resources → Memory → 8GB+ -# Use smaller models -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Lighter configuration +GENERATION_MODEL=qwen3.5:4b \ +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ +docker compose --env-file docker.env up -d rag-api backend +# (changing the embedding model requires rebuilding your indexes) ``` #### Frontend Build Errors ```bash -# Clean build docker compose build --no-cache frontend docker compose up -d frontend - -# Check frontend logs docker compose logs frontend ``` +#### Frontend talks to the wrong host +`NEXT_PUBLIC_API_URL` / `NEXT_PUBLIC_RAG_API_URL` are baked in at build time. +Rebuild the frontend image after changing them. + #### Database/Storage Issues ```bash # Check file permissions @@ -277,64 +394,60 @@ ls -la lancedb/ chmod 664 backend/chat_data.db chmod -R 755 lancedb/ shared_uploads/ -# Test database access +# Inspect the database (sqlite3 is installed in the image) docker compose exec backend sqlite3 /app/backend/chat_data.db ".tables" ``` -### Performance Issues - -#### Slow Response Times -- Use faster models: `qwen3:0.6b` instead of `qwen3:8b` -- Increase Docker memory allocation -- Ensure SSD storage for databases -- Monitor with `docker stats` +### Performance Notes -#### High Memory Usage -- Reduce batch sizes in configuration -- Use smaller embedding models -- Clear unused Docker resources: `docker system prune` +- The RAG API is single-threaded: a chat request queued behind an indexing run + looks like a hang. Check `docker compose logs -f rag-api`. +- Contextual enrichment makes one LLM call per chunk; it dominates indexing time. + Turn it off in the index build options for the fastest ingest. +- `RAG_CONFIG_MODE=fast` switches the RAG API to vector-only retrieval with no + reranking, decomposition or verification. ### Complete Reset ```bash -# Nuclear option - reset everything +# Nuclear option - resets everything, including your documents and chat history ./start-docker.sh stop +docker compose down +rm -rf lancedb/* index_store/* shared_uploads/* backend/chat_data.db docker system prune -a --volumes -rm -rf lancedb/* shared_uploads/* backend/chat_data.db ./start-docker.sh ``` +`docker compose down -v` alone only drops the `ollama_data` volume — application +data is in host directories and must be deleted explicitly. + ## 🏆 Success Criteria Your Docker deployment is successful when: -- ✅ `./start-docker.sh status` shows all containers healthy -- ✅ All health checks pass (see commands above) +- ✅ `docker compose ps` shows all services healthy +- ✅ All health checks pass (see commands above) - ✅ You can access http://localhost:3000 - ✅ You can upload documents and create indexes - ✅ You can chat with your documents - ✅ No errors in container logs -### Performance Benchmarks - -**Good Performance:** -- Container startup: < 2 minutes -- Index creation: < 2 min per 100MB document -- Query response: < 30 seconds -- Memory usage: < 4GB total containers +### What to Expect -**Optimal Performance:** -- Container startup: < 1 minute -- Index creation: < 1 min per 100MB document -- Query response: < 10 seconds -- Memory usage: < 2GB total containers +- **First build**: installs torch/transformers/docling and builds the Next.js + bundle; then `rag-api` downloads ~10GB of HuggingFace weights before it reports + healthy +- **Restarts**: fast. **Recreating** `rag-api` re-downloads the model weights, + because no volume is mounted for the HuggingFace cache +- **Concurrency**: one RAG request at a time ## 📚 Additional Resources - **Detailed Troubleshooting**: See `DOCKER_TROUBLESHOOTING.md` -- **Complete Documentation**: See `Documentation/docker_usage.md` +- **Complete Docker Guide**: See `Documentation/docker_usage.md` +- **Deployment Guide**: See `Documentation/deployment_guide.md` - **System Architecture**: See `Documentation/architecture_overview.md` -- **Direct Development**: See main `README.md` for non-Docker setup +- **Direct Development**: See the main `README.md` --- -**Happy Dockerizing! 🐳** Need help? Check the troubleshooting guide or open an issue. \ No newline at end of file +**Happy Dockerizing! 🐳** Need help? Check the troubleshooting guide or open an issue. diff --git a/DOCKER_TROUBLESHOOTING.md b/DOCKER_TROUBLESHOOTING.md index 4d758db5..f0ab9d7b 100644 --- a/DOCKER_TROUBLESHOOTING.md +++ b/DOCKER_TROUBLESHOOTING.md @@ -1,8 +1,9 @@ # 🐳 Docker Troubleshooting Guide - LocalGPT -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide helps diagnose and fix Docker-related issues with LocalGPT's containerized deployment. +This guide helps diagnose and fix Docker-related issues with LocalGPT's +containerized deployment. --- @@ -13,7 +14,7 @@ This guide helps diagnose and fix Docker-related issues with LocalGPT's containe # Check Docker daemon docker version -# Check Ollama status +# Check Ollama status curl http://localhost:11434/api/tags # Check containers @@ -22,7 +23,7 @@ curl http://localhost:11434/api/tags # Test all endpoints curl -f http://localhost:3000 && echo "✅ Frontend OK" curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" +curl -f http://localhost:8001/health && echo "✅ RAG API OK" curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" ``` @@ -34,6 +35,18 @@ curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" ✅ Ollama OK ``` +### Know the startup order before you debug + +``` +rag-api (healthy: GET /health) → backend (healthy: GET /health) → frontend +``` + +`backend` has `depends_on: rag-api: condition: service_healthy`, and `rag-api` only +answers `/health` once the agent, the embedding model and the reranker have loaded. +On a cold start that is a multi-minute HuggingFace download. **`backend` sitting in +`created` is usually patience, not a bug** — confirm with +`docker compose logs -f rag-api`. + --- ## 🚨 Common Issues & Solutions @@ -64,16 +77,9 @@ docker version #### Solution B: Linux Docker Service ```bash -# Check Docker service status sudo systemctl status docker - -# Restart Docker service sudo systemctl restart docker - -# Enable auto-start sudo systemctl enable docker - -# Test connection docker version ``` @@ -84,7 +90,7 @@ sudo pkill -f docker # Remove socket files sudo rm -f /var/run/docker.sock -sudo rm -f /Users/prompt/.docker/run/docker.sock # macOS +sudo rm -f "$HOME/.docker/run/docker.sock" # macOS # Restart Docker Desktop open -a Docker # macOS @@ -99,37 +105,91 @@ ConnectionError: Failed to connect to Ollama at http://host.docker.internal:1143 #### Solution A: Verify Ollama is Running ```bash -# Check if Ollama is running curl http://localhost:11434/api/tags # If not running, start it ollama serve -# Install required models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# Install the required models +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` +`run_system.py` pulls these automatically, but `start-docker.sh` does not — with +host Ollama you must pull them yourself. + #### Solution B: Test from Container ```bash -# Test Ollama connection from RAG API container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +docker compose exec rag-api curl -sv http://host.docker.internal:11434/api/tags -# If this fails, check Docker network settings +# Inspect the network. The compose project name is the directory name, +# so the network is localgpt_rag-network. docker network ls -docker network inspect rag_system_old_default +docker network inspect localgpt_rag-network ``` -#### Solution C: Alternative Ollama Host +All services declare `extra_hosts: ["host.docker.internal:host-gateway"]`, so this +name resolves on Linux as well as macOS and Windows. If it still fails, check that +Ollama is listening on all interfaces rather than only the loopback: + ```bash -# Edit docker.env to use different host -echo "OLLAMA_HOST=http://172.17.0.1:11434" >> docker.env +OLLAMA_HOST=0.0.0.0:11434 ollama serve +``` -# Or use IP address -echo "OLLAMA_HOST=http://$(ipconfig getifaddr en0):11434" >> docker.env # macOS +#### Solution C: Use the containerized Ollama instead +```bash +./start-docker.sh container + +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +This exports `OLLAMA_HOST=http://ollama:11434`, which wins over `docker.env` +because shell variables take precedence over `--env-file`. + +#### Solution D: Point at a specific host address +```bash +# Edit docker.env rather than appending duplicates - the last value wins, +# but duplicate keys make the file confusing. +# macOS example: +OLLAMA_HOST=http://$(ipconfig getifaddr en0):11434 ./start-docker.sh ``` -### 3. Container Build Failures +### 3. Backend / RAG API Wiring + +#### Problem: every chat answers "Could not connect to the RAG API server" + +The backend builds `/chat` and `/index` from `RAG_API_URL`. Inside compose that +must be `http://rag-api:8001` — `localhost` there refers to the backend container. + +```bash +docker compose exec backend env | grep RAG_API_URL +docker compose exec backend curl -s http://rag-api:8001/health +``` + +#### Problem: chat returns 504 "did not complete within 600s" + +The RAG API is single-threaded, so a chat request queued behind an indexing run can +exceed the timeout. Watch `docker compose logs -f rag-api`, and raise the limits if +your hardware is genuinely slow: + +```bash +RAG_API_TIMEOUT=1200 RAG_API_INDEX_TIMEOUT=7200 \ +docker compose --env-file docker.env up -d backend +``` + +#### Problem: the frontend calls the wrong host + +`NEXT_PUBLIC_API_URL` and `NEXT_PUBLIC_RAG_API_URL` are inlined by `next build`. +Setting them only in `environment:` cannot change an already-built image. + +```bash +NEXT_PUBLIC_API_URL=http://localhost:8000 \ +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 \ +docker compose --env-file docker.env up -d --build frontend +``` + +### 4. Container Build Failures #### Problem: Frontend build fails ``` @@ -138,39 +198,40 @@ ERROR: Failed to build frontend container #### Solution: Clean Build ```bash -# Stop containers ./start-docker.sh stop -# Clean Docker cache docker system prune -f docker builder prune -f -# Rebuild frontend only docker compose build --no-cache frontend docker compose up -d frontend -# Check logs docker compose logs frontend ``` +`Dockerfile.frontend` copies `package.json`, `package-lock.json`, `src/`, +`public/`, `next.config.ts`, `tsconfig.json`, `postcss.config.mjs` and +`eslint.config.mjs`. If you deleted one of those in a fork, the `COPY` fails. +`public/` is kept non-empty by `public/.gitkeep`. + #### Problem: Python package installation fails ``` ERROR: Could not install packages due to an EnvironmentError ``` -#### Solution: Update Dependencies +#### Solution ```bash -# Check requirements file exists +# Both Python images install requirements-docker.txt ls -la requirements-docker.txt -# Test package installation locally +# Test locally pip install -r requirements-docker.txt --dry-run -# Rebuild with updated base image +# Rebuild with an updated base image docker compose build --no-cache --pull rag-api ``` -### 4. Port Conflicts +### 5. Port Conflicts #### Problem: "Port already in use" ``` @@ -179,57 +240,55 @@ Error starting userland proxy: listen tcp4 0.0.0.0:3000: bind: address already i #### Solution: Find and Kill Conflicting Processes ```bash -# Check what's using the ports lsof -i :3000 -i :8000 -i :8001 -# Kill specific processes -pkill -f "npm run dev" # Frontend -pkill -f "server.py" # Backend -pkill -f "api_server" # RAG API +# A direct (non-Docker) run of the same stack is the usual culprit +python run_system.py --stop -# Or kill by port +# Or by port sudo kill -9 $(lsof -t -i:3000) sudo kill -9 $(lsof -t -i:8000) sudo kill -9 $(lsof -t -i:8001) -# Restart containers ./start-docker.sh ``` -### 5. Memory Issues +### 6. Memory Issues -#### Problem: Containers crash due to OOM (Out of Memory) +#### Problem: Containers crash due to OOM ``` Container killed due to memory limit ``` -#### Solution: Increase Docker Memory +The `rag-api` container holds the embedding model and the reranker in memory for +the life of the process, and enabling late chunking loads a second copy of the +embedder during indexing. + +#### Solution: Increase Docker Memory or shrink the models ```bash -# Check current memory usage docker stats --no-stream -# Increase Docker Desktop memory allocation # Docker Desktop → Settings → Resources → Memory → 8GB+ -# Monitor memory usage -docker stats +# Lighter configuration (rebuild indexes after changing the embedding model) +GENERATION_MODEL=qwen3.5:4b \ +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B \ +docker compose --env-file docker.env up -d rag-api backend -# Use smaller models if needed -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Or switch the whole RAG API to the fast profile +RAG_CONFIG_MODE=fast docker compose --env-file docker.env up -d rag-api ``` #### Problem: System running slow ```bash -# Check host memory free -h # Linux vm_stat # macOS -# Clean up Docker resources docker system prune -f docker volume prune -f ``` -### 6. Volume Mount Issues +### 7. Volume Mount Issues #### Problem: Permission denied accessing files ``` @@ -238,18 +297,14 @@ Permission denied: /app/lancedb #### Solution: Fix Permissions ```bash -# Create directories if they don't exist mkdir -p lancedb index_store shared_uploads backend -# Fix permissions chmod -R 755 lancedb index_store shared_uploads chmod 664 backend/chat_data.db -# Check ownership ls -la lancedb/ shared_uploads/ backend/ -# Reset permissions if needed -sudo chown -R $USER:$USER lancedb shared_uploads backend +sudo chown -R $USER:$USER lancedb shared_uploads index_store backend ``` #### Problem: Database file not found @@ -257,24 +312,45 @@ sudo chown -R $USER:$USER lancedb shared_uploads backend No such file or directory: '/app/backend/chat_data.db' ``` -#### Solution: Initialize Database +`ChatDatabase` creates the parent directory and the file itself, so this normally +means the `./backend` bind mount is missing or `DB_PATH` points somewhere +unmounted. + ```bash -# Create empty database file -touch backend/chat_data.db +docker compose exec backend env | grep DB_PATH # expect /app/backend/chat_data.db +docker compose exec backend ls -la /app/backend -# Or initialize with schema +# Create it from the host if needed python -c " from backend.database import ChatDatabase db = ChatDatabase() -db.init_database() -print('Database initialized') +print('Database initialized at', db.db_path) " -# Restart containers ./start-docker.sh stop ./start-docker.sh ``` +#### Problem: Indexes exist in the sidebar but every answer says nothing was found + +The SQLite rows and the LanceDB tables must match. Restoring one without the other, +or changing `LANCEDB_PATH`, leaves index records pointing at tables that do not +exist. + +```bash +docker compose exec rag-api python -c " +import lancedb +print(lancedb.connect('/app/lancedb').table_names()) +" +docker compose exec backend sqlite3 /app/backend/chat_data.db \ + "select id, name, vector_table_name from indexes;" +``` + +#### Problem: `changing the embedding model requires rebuilding the index` + +Vector width is read from the loaded embedding model. Delete and rebuild the index +after changing `EMBEDDING_MODEL`. + --- ## 🔍 Advanced Debugging @@ -286,78 +362,80 @@ print('Database initialized') # RAG API container (most issues happen here) docker compose exec rag-api bash -# Check environment variables -docker compose exec rag-api env | grep -E "(OLLAMA|RAG|NODE)" +# Check the wiring +docker compose exec rag-api env | grep -E "OLLAMA|MODEL|DB_PATH|LANCEDB|RAG_CONFIG" # Test Python imports docker compose exec rag-api python -c " import sys print('Python version:', sys.version) -from rag_system.main import get_agent +from rag_system.factory import get_agent print('✅ RAG system imports work') " # Backend container docker compose exec backend bash -python -c " +docker compose exec backend python -c " from backend.database import ChatDatabase print('✅ Database imports work') " -# Frontend container -docker compose exec frontend sh -npm --version -node --version +# Frontend container (alpine -> sh, not bash) +docker compose exec frontend sh -c "node --version && npm --version" ``` #### Check Container Resources ```bash -# Monitor real-time resource usage docker stats -# Check individual container health docker compose ps docker inspect rag-api --format='{{.State.Health.Status}}' +docker inspect rag-api --format='{{json .State.Health}}' | head -c 2000 -# View container configurations docker compose config ``` #### Network Debugging + +The Python images are `python:3.11-slim` and the frontend is `node:20-alpine`; +**none of them ship `ping` or `nslookup`**. Use curl (installed in both Python +images) or busybox wget (in the alpine image): + ```bash -# Check network connectivity -docker compose exec rag-api ping backend -docker compose exec backend ping rag-api -docker compose exec rag-api ping host.docker.internal +# Backend ↔ RAG API by service name +docker compose exec backend curl -sv http://rag-api:8001/health +docker compose exec rag-api curl -sv http://backend:8000/health + +# Container → host Ollama +docker compose exec rag-api curl -sv http://host.docker.internal:11434/api/tags -# Check DNS resolution -docker compose exec rag-api nslookup host.docker.internal +# From the frontend (alpine) +docker compose exec frontend wget -qO- http://backend:8000/health -# Test HTTP connections -docker compose exec rag-api curl -v http://backend:8000/health -docker compose exec rag-api curl -v http://host.docker.internal:11434/api/tags +# If you really want ping/nslookup, install them in the running container first +docker compose exec rag-api sh -c "apt-get update && apt-get install -y iputils-ping dnsutils" ``` ### Log Analysis #### Container Logs ```bash -# View all logs ./start-docker.sh logs -# Follow specific service logs docker compose logs -f rag-api docker compose logs -f backend docker compose logs -f frontend -# Search for errors docker compose logs rag-api 2>&1 | grep -i error docker compose logs backend 2>&1 | grep -i "traceback\|error" -# Save logs to file docker compose logs > docker-debug.log 2>&1 ``` +The RAG API logs each stage of a query, so `docker compose logs -f rag-api` while +you ask a question shows exactly where time is going: retrieval, reranking, context +expansion, pruning, synthesis, verification. + #### System Logs ```bash # Docker daemon logs (Linux) @@ -373,72 +451,57 @@ journalctl -u docker.service -f ### Manual Container Testing -#### Test Individual Containers +`test_docker_build.sh` automates exactly this — building each image and probing its +health endpoint: + ```bash -# Test RAG API alone -docker build -f Dockerfile.rag-api -t test-rag-api . -docker run --rm -p 8001:8001 -e OLLAMA_HOST=http://host.docker.internal:11434 test-rag-api & -sleep 30 -curl http://localhost:8001/models -pkill -f test-rag-api +./test_docker_build.sh +``` + +By hand: -# Test Backend alone +```bash +# RAG API alone +docker build -f Dockerfile.rag-api -t test-rag-api . +docker run -d --name test-rag-api -p 8001:8001 \ + -e OLLAMA_HOST=http://host.docker.internal:11434 \ + --add-host host.docker.internal:host-gateway test-rag-api +sleep 60 # models load before /health answers +curl http://localhost:8001/health +docker rm -f test-rag-api + +# Backend alone docker build -f Dockerfile.backend -t test-backend . -docker run --rm -p 8000:8000 test-backend & -sleep 30 +docker run -d --name test-backend -p 8000:8000 test-backend +sleep 15 curl http://localhost:8000/health -pkill -f test-backend +docker rm -f test-backend ``` -#### Integration Testing +### Integration Testing ```bash -# Full system test -./start-docker.sh +./start-docker.sh -y -# Wait for all services to be ready -sleep 60 +# Wait for the health-gated chain to come up +until curl -sf http://localhost:8000/health >/dev/null; do sleep 5; done -# Test complete workflow +# Create a session curl -X POST http://localhost:8000/sessions \ -H "Content-Type: application/json" \ -d '{"title": "Test Session"}' -# Test document upload (if you have a test PDF) -# curl -X POST http://localhost:8000/upload -F "file=@test.pdf" +# Create an index, upload a document, build it +IDX=$(curl -s -X POST http://localhost:8000/indexes \ + -H "Content-Type: application/json" \ + -d '{"name":"Smoke Test"}' | python3 -c "import sys,json;print(json.load(sys.stdin)['index_id'])") +curl -X POST "http://localhost:8000/indexes/$IDX/upload" -F "files=@test.pdf" +curl -X POST "http://localhost:8000/indexes/$IDX/build" \ + -H "Content-Type: application/json" -d '{"chunk_size": 512}' -# Clean up ./start-docker.sh stop ``` -### Automated Testing Script - -Create `test-docker-health.sh`: -```bash -#!/bin/bash -set -e - -echo "🐳 Docker Health Test Starting..." - -# Start containers -./start-docker.sh - -# Wait for services -echo "⏳ Waiting for services to start..." -sleep 60 - -# Test endpoints -echo "🔍 Testing endpoints..." -curl -f http://localhost:3000 && echo "✅ Frontend OK" || echo "❌ Frontend FAIL" -curl -f http://localhost:8000/health && echo "✅ Backend OK" || echo "❌ Backend FAIL" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" || echo "❌ RAG API FAIL" -curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" || echo "❌ Ollama FAIL" - -# Test container health -echo "🔍 Checking container health..." -docker compose ps - -echo "🎉 Health test complete!" -``` +The upload form field must be named `files`. --- @@ -448,46 +511,38 @@ echo "🎉 Health test complete!" #### Soft Reset ```bash -# Stop containers ./start-docker.sh stop - -# Clean up Docker resources docker system prune -f - -# Restart containers ./start-docker.sh ``` #### Hard Reset (⚠️ Deletes all data) ```bash -# Stop everything ./start-docker.sh stop +docker compose down -# Remove all containers, images, and volumes -docker system prune -a --volumes - -# Remove local data (CAUTION: This deletes all your documents and chat history) -rm -rf lancedb/* shared_uploads/* backend/chat_data.db +# Application data lives in host directories, not volumes - +# `docker compose down -v` alone will NOT remove it. +rm -rf lancedb/* index_store/* shared_uploads/* backend/chat_data.db -# Rebuild from scratch +docker system prune -a --volumes ./start-docker.sh ``` #### Selective Reset - -Reset only specific components: ```bash -# Reset just the database +# Just the chat/index database ./start-docker.sh stop rm backend/chat_data.db ./start-docker.sh -# Reset just vector storage +# Just vector storage (indexes must be rebuilt; delete the DB rows too or they +# will point at tables that no longer exist) ./start-docker.sh stop rm -rf lancedb/* ./start-docker.sh -# Reset just uploaded documents +# Just uploaded documents rm -rf shared_uploads/* ``` @@ -497,31 +552,36 @@ rm -rf shared_uploads/* ### Resource Monitoring ```bash -# Monitor containers continuously watch -n 5 'docker stats --no-stream' -# Check disk usage docker system df -du -sh lancedb shared_uploads backend +du -sh lancedb shared_uploads index_store backend -# Monitor host resources htop # Linux -top # macOS/Windows +top # macOS ``` ### Performance Tuning ```bash -# Use smaller models for better performance -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Fast profile: vector-only retrieval, no reranking, decomposition or verification +RAG_CONFIG_MODE=fast docker compose --env-file docker.env up -d rag-api -# Reduce Docker memory if needed -# Docker Desktop → Settings → Resources → Memory +# Smaller generation model +GENERATION_MODEL=qwen3.5:4b docker compose --env-file docker.env up -d rag-api backend -# Clean up regularly -docker system prune -f -docker volume prune -f +# Smaller embedding model (rebuild indexes afterwards) +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B docker compose --env-file docker.env up -d rag-api + +# Faster indexing: turn off contextual enrichment in the index build options +# (it makes one LLM call per chunk) ``` +Structural limits worth knowing before you tune: +- The RAG API is single-threaded — requests are serialised. +- Indexing is synchronous; `POST /indexes/{id}/build` stays open until it finishes. +- Model weights are downloaded into the container's writable layer, so recreating + `rag-api` re-downloads them unless you mount a cache and set `HF_HOME`. + --- ## 🆘 When All Else Fails @@ -530,29 +590,28 @@ docker volume prune -f #### 1. Direct Development (No Docker) ```bash -# Stop Docker containers ./start-docker.sh stop - -# Use direct development instead python run_system.py ``` #### 2. Minimal Docker (RAG API only) ```bash -# Run only RAG API in Docker docker build -f Dockerfile.rag-api -t rag-api . -docker run -p 8001:8001 rag-api +docker run -p 8001:8001 \ + -e OLLAMA_HOST=http://host.docker.internal:11434 \ + --add-host host.docker.internal:host-gateway rag-api -# Run other components directly -cd backend && python server.py & +# Run the rest directly, from the repository root +python backend/server.py & npm run dev ``` #### 3. Hybrid Approach ```bash -# Run some services in Docker, others directly -docker compose up -d rag-api -cd backend && python server.py & +docker compose --env-file docker.env up -d rag-api + +# Point the host backend at the container +RAG_API_URL=http://localhost:8001 python backend/server.py & npm run dev ``` @@ -560,27 +619,23 @@ npm run dev #### Diagnostic Information to Collect ```bash -# System information docker version docker compose version uname -a -# Container information docker compose ps docker compose config -# Resource information docker stats --no-stream docker system df -# Error logs docker compose logs > docker-errors.log 2>&1 ``` #### Support Channels 1. **Check GitHub Issues**: Search existing issues for similar problems -2. **Documentation**: Review the complete documentation in `Documentation/` -3. **Create Issue**: Include diagnostic information above +2. **Documentation**: Review `Documentation/` and `DOCKER_README.md` +3. **Create Issue**: Include the diagnostic information above --- @@ -590,8 +645,8 @@ Your Docker deployment is working correctly when: - ✅ `docker version` shows Docker is running - ✅ `curl http://localhost:11434/api/tags` shows Ollama is accessible -- ✅ `./start-docker.sh status` shows all containers healthy -- ✅ All health check URLs return 200 OK +- ✅ `docker compose ps` shows all services healthy +- ✅ `curl -f http://localhost:8000/health` and `.../8001/health` both return 200 - ✅ You can access the frontend at http://localhost:3000 - ✅ You can create document indexes successfully - ✅ You can chat with your documents @@ -601,4 +656,4 @@ Your Docker deployment is working correctly when: --- -**Still having issues?** Check the main `DOCKER_README.md` or create an issue with your diagnostic information. \ No newline at end of file +**Still having issues?** Check `DOCKER_README.md` or create an issue with your diagnostic information. diff --git a/Dockerfile.backend b/Dockerfile.backend index 9aeac269..3d67fbaa 100644 --- a/Dockerfile.backend +++ b/Dockerfile.backend @@ -3,9 +3,10 @@ FROM python:3.11-slim # Set working directory WORKDIR /app -# Install system dependencies +# Install system dependencies (sqlite3 CLI is used for database troubleshooting) RUN apt-get update && apt-get install -y \ curl \ + sqlite3 \ && rm -rf /var/lib/apt/lists/* # Copy requirements and install Python dependencies (using Docker-specific requirements) @@ -16,8 +17,8 @@ RUN pip install --no-cache-dir -r requirements.txt COPY backend/ ./backend/ COPY rag_system/ ./rag_system/ -# Create necessary directories and initialize database -RUN mkdir -p shared_uploads logs backend +# Create necessary directories +RUN mkdir -p shared_uploads index_store lancedb logs # Expose port EXPOSE 8000 @@ -26,6 +27,6 @@ EXPOSE 8000 HEALTHCHECK --interval=30s --timeout=10s --start-period=30s --retries=3 \ CMD curl -f http://localhost:8000/health || exit 1 -# Run the backend server -WORKDIR /app/backend -CMD ["python", "server.py"] \ No newline at end of file +# Run the backend server from the repo root so relative paths +# (shared_uploads/, index_store/, lancedb/) match local development +CMD ["python", "backend/server.py"] diff --git a/Dockerfile.frontend b/Dockerfile.frontend index 938d5c6c..39101163 100644 --- a/Dockerfile.frontend +++ b/Dockerfile.frontend @@ -1,4 +1,4 @@ -FROM node:18-alpine +FROM node:20-alpine # Set working directory WORKDIR /app @@ -12,20 +12,24 @@ COPY src/ ./src/ COPY public/ ./public/ COPY next.config.ts ./ COPY tsconfig.json ./ -COPY tailwind.config.js ./ COPY postcss.config.mjs ./ COPY eslint.config.mjs ./ -# Build the application (skip linting for Docker) -ENV NEXT_LINT=false +# NEXT_PUBLIC_* values are inlined by `next build`, so they must be set before it runs +ARG NEXT_PUBLIC_API_URL=http://localhost:8000 +ARG NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 +ENV NEXT_PUBLIC_API_URL=$NEXT_PUBLIC_API_URL +ENV NEXT_PUBLIC_RAG_API_URL=$NEXT_PUBLIC_RAG_API_URL + +# Build the application RUN npm run build # Expose port EXPOSE 3000 -# Health check +# Health check (node:20-alpine ships busybox wget, not curl) HEALTHCHECK --interval=30s --timeout=10s --start-period=60s --retries=3 \ - CMD curl -f http://localhost:3000 || exit 1 + CMD wget -qO- http://localhost:3000 >/dev/null 2>&1 || exit 1 # Start the application -CMD ["npm", "start"] \ No newline at end of file +CMD ["npm", "start"] diff --git a/Dockerfile.rag-api b/Dockerfile.rag-api index 940a2d0c..11969e5b 100644 --- a/Dockerfile.rag-api +++ b/Dockerfile.rag-api @@ -6,6 +6,7 @@ WORKDIR /app # Install system dependencies RUN apt-get update && apt-get install -y \ curl \ + sqlite3 \ build-essential \ && rm -rf /var/lib/apt/lists/* @@ -25,7 +26,7 @@ EXPOSE 8001 # Health check HEALTHCHECK --interval=30s --timeout=10s --start-period=60s --retries=3 \ - CMD curl -f http://localhost:8001/models || exit 1 + CMD curl -f http://localhost:8001/health || exit 1 # Run the RAG API server -CMD ["python", "-m", "rag_system.api_server"] \ No newline at end of file +CMD ["python", "-m", "rag_system.api_server"] diff --git a/Documentation/api_reference.md b/Documentation/api_reference.md index 1e995bdc..4dcba670 100644 --- a/Documentation/api_reference.md +++ b/Documentation/api_reference.md @@ -1,161 +1,350 @@ -# 📚 API Reference (Backend & RAG API) +# 📚 API Reference (Backend Gateway & RAG API) -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ + +Two HTTP services, both reachable from the browser: + +| Service | Base URL | Source | +|---------|----------|--------| +| Backend gateway | `http://localhost:8000` | `backend/server.py` | +| RAG API | `http://localhost:8001` | `rag_system/api_server.py` | + +Both send `Access-Control-Allow-Origin: *` on every response and answer `OPTIONS` preflights. Neither implements authentication. + +**Wire format** is snake_case. Both services additionally accept camelCase and normalise it to the same canonical key at parse time, so `rerankerTopK` and `reranker_top_k` are interchangeable. When both spellings are present, the explicit snake_case value wins. The gateway uses an explicit alias table that covers every option it accepts (plus a few legacy names such as `latechunk` and `decompose`); the RAG API converts any camelCase key generically. --- -## Backend HTTP API (Python `backend/server.py`) -**Base URL**: `http://localhost:8000` +## 1. Backend Gateway — `http://localhost:8000` + +### 1.1 Route table -| Endpoint | Method | Description | Request Body | Success Response | +| Endpoint | Method | Description | Request body | Success response | |----------|--------|-------------|--------------|------------------| -| `/health` | GET | Health probe incl. Ollama status & DB stats | – | 200 JSON `{ status, ollama_running, available_models, database_stats }` | -| `/chat` | POST | Stateless chat (no session) | `{ message:str, model?:str, conversation_history?:[{role,content}]}` | 200 `{ response:str, model:str, message_count:int }` | -| `/sessions` | GET | List all sessions | – | `{ sessions:ChatSession[], total:int }` | -| `/sessions` | POST | Create session | `{ title?:str, model?:str }` | 201 `{ session:ChatSession, session_id }` | -| `/sessions/` | GET | Get session + msgs | – | `{ session, messages }` | -| `/sessions/` | DELETE | Delete session | – | `{ message, deleted_session_id }` | -| `/sessions//rename` | POST | Rename session | `{ title:str }` | `{ message, session }` | -| `/sessions//messages` | POST | Session chat (builds history) | See ChatRequest + retrieval opts ▼ | `{ response, session, user_message_id, ai_message_id }` | -| `/sessions//documents` | GET | List uploaded docs | – | `{ files:string[], file_count:int, session }` | -| `/sessions//upload` | POST multipart | Upload docs to session | field `files[]` | `{ message, uploaded_files, processing_results?, session_documents?, total_session_documents? }` | -| `/sessions//index` | POST | Trigger RAG indexing for session | `{ latechunk?, doclingChunk?, chunkSize?, ... }` | `{ message }` | -| `/sessions//indexes` | GET | List indexes linked to session | – | `{ indexes, total }` | -| `/sessions//indexes/` | POST | Link index to session | – | `{ message }` | -| `/sessions/cleanup` | GET | Remove empty sessions | – | `{ message, cleanup_count }` | -| `/models` | GET | List generation / embedding models | – | `{ generation_models:str[], embedding_models:str[] }` | +| `/health` | GET | Health probe with Ollama status and DB stats | – | `{ status, ollama_running, available_models, database_stats }` | +| `/chat` | POST | Stateless chat, no session, no retrieval | `{ message, model?, conversation_history? }` | `{ response, model, message_count }` | +| `/models` | GET | Available generation / embedding models | – | `{ generation_models, embedding_models }` | +| `/sessions` | GET | List sessions | – | `{ sessions, total }` | +| `/sessions` | POST | Create a session | `{ title?, model? }` | 201 `{ session, session_id }` | +| `/sessions/cleanup` | GET | Delete sessions that have no messages | – | `{ message, cleanup_count }` | +| `/sessions/` | GET | Session plus its messages | – | `{ session, messages }` | +| `/sessions/` | DELETE | Delete a session and its messages | – | `{ deleted: true }` | +| `/sessions//rename` | POST | Rename a session | `{ title }` | `{ message, session }` | +| `/sessions//messages` | POST | Session chat (persisted) | [Session chat request](#12-session-chat-request) | `{ response, session, source_documents, used_rag }` | +| `/sessions//messages/save` | POST | Persist a completed streamed turn (the browser calls this after `POST :8001/chat/stream` finishes) | `{ user_message, assistant_message, source_documents? }` | `{ session, user_message_id, ai_message_id }` — sources are stored in the assistant message's `metadata.source_documents` | +| `/sessions//documents` | GET | Files uploaded to a session | – | `{ session, files, file_count }` | +| `/sessions//upload` | POST | Upload files to a session | multipart, field `files` | `{ message, uploaded_files }` | +| `/sessions//index` | POST | Index the session's documents | [Index options](#14-index-options) (optional) | the RAG API `/index` response | +| `/sessions//indexes` | GET | Indexes linked to a session | – | `{ indexes, total }` | +| `/sessions//indexes/` | POST | Link an index to a session | – | `{ message }` | | `/indexes` | GET | List all indexes | – | `{ indexes, total }` | -| `/indexes` | POST | Create index | `{ name:str, description?:str, metadata?:dict }` | `{ index_id }` | -| `/indexes/` | GET | Get single index | – | `{ index }` | -| `/indexes/` | DELETE | Delete index | – | `{ message, index_id }` | -| `/indexes//upload` | POST multipart | Upload docs to index | field `files[]` | `{ message, uploaded_files }` | -| `/indexes//build` | POST | Build / rebuild index (RAG) | `{ latechunk?, doclingChunk?, ...}` | 200 `{ response?, message?}` (idempotent) | +| `/indexes` | POST | Create a named index | `{ name, description?, metadata? }` | 201 `{ index_id }` | +| `/indexes/` | GET | One index | – | the index object (see below) | +| `/indexes/` | DELETE | Delete an index, its links and its LanceDB table | – | `{ message, index_id }` | +| `/indexes//upload` | POST | Upload files to an index | multipart, field `files` | `{ message, uploaded_files }` | +| `/indexes//build` | POST | Build / rebuild the index | [Index options](#14-index-options) (optional) | `{ response, ...echoed options }` | ---- +`uploaded_files` entries are `{ filename, stored_path }`. The index object returned by `GET /indexes/` is `{ id, name, description, created_at, updated_at, vector_table_name, metadata, documents[] }` — it is **not** wrapped in `{ index: … }`. -## RAG API (Python `rag_system/api_server.py`) -**Base URL**: `http://localhost:8001` +> **Body required.** `POST /chat`, `POST /sessions`, `POST /sessions//messages` and `POST /indexes` read `Content-Length` unconditionally; send a JSON body (at minimum `{}`) or the request fails with a 500. `POST /sessions//rename` returns a clean `400 { "error": "Request body required" }`. `POST /sessions//index` and `POST /indexes//build` treat the body as optional. -| Endpoint | Method | Description | Request Body | Success Response | -|----------|--------|-------------|--------------|------------------| -| `/chat` | POST | Run RAG query with full pipeline | See RAG ChatRequest ▼ | `{ answer:str, source_documents:[], reasoning?:str, confidence?:float }` | -| `/chat/stream` | POST | Run RAG query with SSE streaming | Same as /chat | Server-Sent Events stream | -| `/index` | POST | Index documents with full configuration | See Index Request ▼ | `{ message:str, indexed_files:[], table_name:str }` | -| `/models` | GET | List available models | – | `{ generation_models:str[], embedding_models:str[] }` | +### 1.2 Session chat request + +`POST /sessions//messages` -### RAG ChatRequest (Advanced Options) ```jsonc { - "query": "string", // Required – user question - "session_id": "string", // Optional – for session context - "table_name": "string", // Optional – specific index table - "compose_sub_answers": true, // Optional – compose sub-answers - "query_decompose": true, // Optional – decompose complex queries - "ai_rerank": false, // Optional – AI-powered reranking - "context_expand": false, // Optional – context expansion - "verify": true, // Optional – answer verification - "retrieval_k": 20, // Optional – number of chunks to retrieve - "context_window_size": 1, // Optional – context window size - "reranker_top_k": 10, // Optional – top-k after reranking - "search_type": "hybrid", // Optional – "hybrid|dense|fts" - "dense_weight": 0.7, // Optional – dense search weight (0-1) - "force_rag": false, // Optional – bypass triage, force RAG - "provence_prune": false, // Optional – sentence-level pruning - "provence_threshold": 0.8, // Optional – pruning threshold - "model": "qwen3:8b" // Optional – generation model override + "message": "string", // required + "model": "qwen3.5:9b", // optional – generation model for this request + "force_rag": false, // optional – skip gateway routing, always call the RAG API + + // Retrieval options, forwarded to the RAG API when the RAG route is taken. + // Omitted options are not forwarded, so the pipeline profile default applies. + "compose_sub_answers": true, + "query_decompose": true, // alias: "decompose" + "ai_rerank": true, // profile default when omitted is OFF (eval/DECISIONS.md) + "context_expand": true, + "verify": true, + "retrieval_k": 20, + "context_window_size": 1, + "reranker_top_k": 10, + "retrieval_mode": "hybrid", // "hybrid" | "vector_only" | "fts_only" (alias: "search_type") + "provence_prune": false, + "provence_threshold": 0.1 } ``` -### Index Request (Document Indexing) +camelCase aliases are accepted for all of them (`composeSubAnswers`, `queryDecompose`, `aiRerank`, `contextExpand`, `retrievalK`, `contextWindowSize`, `rerankerTopK`, `retrievalMode`, `searchType`, `provencePrune`, `provenceThreshold`, `forceRag`). + +Response: + ```jsonc { - "file_paths": ["path1.pdf", "path2.pdf"], // Required – files to index - "session_id": "string", // Required – session identifier - "chunk_size": 512, // Optional – chunk size (default: 512) - "chunk_overlap": 64, // Optional – chunk overlap (default: 64) - "enable_latechunk": true, // Optional – enable late chunking - "enable_docling_chunk": false, // Optional – enable DocLing chunking - "retrieval_mode": "hybrid", // Optional – "hybrid|dense|fts" - "window_size": 2, // Optional – context window - "enable_enrich": true, // Optional – enable enrichment - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", // Optional – embedding model - "enrich_model": "qwen3:0.6b", // Optional – enrichment model - "overview_model_name": "qwen3:0.6b", // Optional – overview model - "batch_size_embed": 50, // Optional – embedding batch size - "batch_size_enrich": 25 // Optional – enrichment batch size + "response": "string", // assistant answer, tags stripped + "session": { /* ChatSession */ }, + "source_documents": [], // empty on the direct-LLM route + "used_rag": true } ``` -> **Note on CORS** – All endpoints include `Access-Control-Allow-Origin: *` header. +This endpoint persists the turn itself: it writes the user message, derives a session title from the first message, then writes the assistant message with its sources in `metadata.source_documents`. (Streamed turns are persisted separately via `/messages/save`.) ---- +`force_rag` decides gateway routing **and** is forwarded to the RAG API, so the agent's own triage is skipped too. + +### 1.3 Gateway routing + +`POST /sessions//messages` picks its path with a deterministic gate — no model call, no retrieval, sub-millisecond: + +1. `force_rag: true` (or `forceRag`) → RAG, unconditionally. +2. Session has no linked indexes → answered directly by Ollama, no retrieval. +3. The whole message is smalltalk (`hello`, `thanks!`, `bye`, `ok` — an allowlist regex capped at six words) or a question about the assistant itself (`who are you`, `what model are you`) → direct. +4. Anything else → RAG. + +`used_rag` in the response tells you which way it went. The gate errs toward RAG on purpose: the RAG API's own agent triage still runs on every forwarded request and can answer directly without retrieving, so a false "use RAG" costs one triage call, not a wrong answer. If you need the answer grounded in documents regardless, send `force_rag`. + +The streaming path (`:8001/chat/stream`, the UI default) never touches this gate — the browser calls the RAG API directly and only the agent triage applies. + +### 1.4 Index options -## Frontend Wrapper (`src/lib/api.ts`) -The React/Next.js frontend calls the backend via a typed wrapper. Important methods & payloads: - -| Method | Backend Endpoint | Payload Shape | -|--------|------------------|---------------| -| `checkHealth()` | `/health` | – | -| `sendMessage({ message, model?, conversation_history? })` | `/chat` | ChatRequest | -| `getSessions()` | `/sessions` | – | -| `createSession(title?, model?)` | `/sessions` | – | -| `getSession(sessionId)` | `/sessions/` | – | -| `sendSessionMessage(sessionId, message, opts)` | `/sessions//messages` | `ChatRequest + retrieval opts` | -| `uploadFiles(sessionId, files[])` | `/sessions//upload` | multipart | -| `indexDocuments(sessionId)` | `/sessions//index` | opts similar to buildIndex | -| `buildIndex(indexId, opts)` | `/indexes//build` | Index build options | -| `linkIndexToSession` | `/sessions//indexes/` | – | +Accepted by `POST /sessions//index` and `POST /indexes//build`, normalised and forwarded to the RAG API `/index`. **Options you omit are not sent**, so the RAG API's own defaults apply (§2.5). + +```jsonc +{ + "chunk_size": 512, + "window_size": 2, + "retrieval_mode": "hybrid", // "hybrid" | "vector_only" | "fts_only" + "enable_enrich": true, + "enable_latechunk": false, // aliases: "latechunk", "enableLatechunk" + "enable_docling_chunk": true, // aliases: "doclingChunk", "enableDoclingChunk"; false = legacy chunker — see §2.5 + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", // aliases: "overviewModel", "overview_model" + "batch_size_embed": 50, + "batch_size_enrich": 25 +} +``` + +`POST /sessions//index` returns the RAG API's `/index` response unchanged, or `200 { "message": "No documents to index for this session." }` when the session has no uploaded files. + +`POST /indexes//build` echoes the canonical options back alongside the RAG API response, renaming `enable_latechunk` → `latechunk`, `enable_docling_chunk` → `docling_chunk` and `overview_model_name` → `overview_model`, and stores the same values in the index's metadata. If the RAG API reports that the table already exists, the build is treated as idempotent and returns `{ message: "Index already built – skipping rebuild.", note }`. + +### 1.5 Error responses + +| Status | When | +|--------|------| +| 400 | Missing required field, invalid JSON, no files in a multipart upload | +| 404 | Unknown route, unknown session, unknown index | +| 500 | Unhandled server error, or the RAG API returned a non-200 | +| 502 | Could not connect to the RAG API (`RAG_API_URL`) | +| 503 | Ollama is not reachable (`POST /chat` only) | +| 504 | The RAG API did not answer within `RAG_API_TIMEOUT` (chat, default 600 s) or `RAG_API_INDEX_TIMEOUT` (indexing, default 3600 s) | + +Error bodies are `{ "error": "..." }`. --- -## Payload Definitions (Canonical) +## 2. RAG API — `http://localhost:8001` + +| Endpoint | Method | Description | Request body | Success response | +|----------|--------|-------------|--------------|------------------| +| `/health` | GET | Liveness probe | – | `{ "status": "ok" }` | +| `/models` | GET | Models available to the active LLM backend | – | `{ generation_models, embedding_models }` | +| `/chat` | POST | Run the full agent pipeline | [Chat request](#21-chat-request) | `{ answer, source_documents }` | +| `/chat/stream` | POST | Same pipeline, streamed as SSE | [Chat request](#21-chat-request) | `text/event-stream` | +| `/index` | POST | Index documents | [Index request](#25-index-request) | see §2.6 | + +Unknown routes return 404 `{ "error": "Not Found" }`. + +### 2.1 Chat request + +```jsonc +{ + "query": "string", // required + "session_id": "string", // optional – loads the session's overviews and index metadata + "table_name": "string", // optional – LanceDB table; otherwise resolved from session_id + "model": "qwen3.5:9b", // optional – generation model for this request only + + "compose_sub_answers": true, // optional – profile default when omitted + "query_decompose": true, // optional – profile default when omitted + "ai_rerank": true, // optional – profile default when omitted, which is OFF (eval/DECISIONS.md) + "context_expand": true, // optional – false forces a context window of 0 + "verify": true, // optional – profile default when omitted + "force_rag": false, // optional – skip triage and go straight to retrieval + + "retrieval_k": 20, // optional – profile default when omitted + "context_window_size": 1, // optional – profile default when omitted + "reranker_top_k": 10, // optional – profile default when omitted + + "retrieval_mode": "hybrid", // "hybrid" | "vector_only" | "fts_only"; alias "search_type" + "provence_prune": false, // optional – sentence-level pruning + "provence_threshold": 0.1 // optional – pruning threshold +} +``` + +Notes: + +* Every option, including `retrieval_k`, `context_window_size` and `reranker_top_k`, falls back to the pipeline profile when omitted; only options you actually send override the profile. +* An unsupported `retrieval_mode` is rejected with `400 { "error": "Unsupported retrieval mode '…'. Supported: hybrid, vector_only, fts_only." }`. +* `model` is applied for the duration of the request and then restored. A model id that does not match the active `LLM_BACKEND` (for example an Ollama tag while `LLM_BACKEND=watsonx`) is ignored with a warning. +* If the table's index metadata records an `embedding_model`, the retrieval pipeline switches to it before searching. + +### 2.2 Chat response -### ChatRequest (frontend ⇄ backend) ```jsonc { - "message": "string", // Required – raw user text - "model": "string", // Optional – generation model id - "conversation_history": [ // Optional – prior turn list - { "role": "user|assistant", "content": "string" } + "answer": "string", + "source_documents": [ + { + "chunk_id": "string", + "text": "string", + "score": 0.0164, // higher is better; RRF score in hybrid mode + "document_id": "report.pdf", + "chunk_index": 12, + "metadata": { }, + "rerank_score": 0.87, // present only when reranking ran + "bm25": 4.21 // present only when the full-text leg matched this chunk + } ] } ``` -### Session Chat Extended Options +There is no top-level `confidence` or `reasoning` field. When verification is enabled the confidence is appended to `answer` as `" [Confidence: N%]"`, plus `" [Warning: Low confidence. Groundedness: ]"` when the answer is judged ungrounded or scores below 50. Nothing is appended when the verifier's score parses as 0. + +When nothing is retrieved, `answer` is `"I could not find an answer in the documents."` and `source_documents` is empty. + +### 2.3 `POST /chat/stream` (SSE) + +Same request body. The response is `text/event-stream`; each event is a single line: + +``` +data: {"type": "", "data": } +``` + +| Event | Payload | Emitted when | +|-------|---------|--------------| +| `analyze` | `{query}` | Start of the run | +| `direct_answer` | `{}` | Triage chose a direct answer | +| `decomposition` | `{sub_queries}` | Query decomposition produced sub-queries | +| `retrieval_started` | `{mode}` or `{count}` | `{mode}` from the retrieval pipeline, `{count}` from the decomposition branch | +| `retrieval_done` | `{count}` | Retrieval finished | +| `rerank_started` / `rerank_done` | `{count}` | Reranking | +| `context_expand_started` / `context_expand_done` | `{count}` | Context expansion | +| `prune_started` / `prune_done` | `{count}` | Provence pruning (only when enabled) | +| `token` | `{text}` | Streamed answer tokens | +| `sub_query_token` | `{index, text, question}` | Tokens from a parallel sub-query | +| `sub_query_result` | `{index, query, answer, source_documents}` | A sub-query finished | +| `single_query_result` | the pipeline result | Decomposition produced exactly one sub-query | +| `final_answer` | `{answer, source_documents}` | Composed answer ready | +| `complete` | `{answer, source_documents}` | Final event; clients may close here | +| `error` | `{error}` | Failure after the stream opened | + +> This endpoint does **not** write to SQLite. The browser's default chat path calls it directly and, once the `complete` event arrives, persists the finished turn via `POST :8000/sessions//messages/save`. Clients that consume the stream directly must do the same if they want the turn in the session history. + +### 2.4 `GET /models` + ```jsonc { - "composeSubAnswers": true, - "decompose": true, - "aiRerank": false, - "contextExpand": false, - "verify": true, - "retrievalK": 10, - "contextWindowSize": 5, - "rerankerTopK": 20, - "searchType": "fts|hybrid|dense", - "denseWeight": 0.75, - "force_rag": false + "generation_models": ["qwen3.5:4b", "qwen3.5:9b"], + "embedding_models": ["Qwen/Qwen3-Embedding-0.6B", "Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-8B", "microsoft/harrier-oss-v1-0.6b"] +} +``` + +With `LLM_BACKEND=ollama` the generation list comes from `GET {OLLAMA_HOST}/api/tags` (5 s timeout; on failure the list is simply shorter), split by a substring match on `embed` / `bge` / `embedding`. With `LLM_BACKEND=watsonx` it lists the configured WatsonX generation and enrichment models. The embedding list always includes the pipeline's configured embedding model, the default `microsoft/harrier-oss-v1-0.6b` and the three Qwen3-Embedding sizes; it is returned sorted and de-duplicated. + +### 2.5 Index request + +```jsonc +{ + "file_paths": ["/abs/path/a.pdf", "/abs/path/b.docx"], // required + "session_id": "string", // optional – overview file + metadata target + "table_name": "string", // optional – otherwise resolved from session_id + + "chunk_size": 512, // default 512 – token budget per chunk + "window_size": 2, // default 2 – contextual-enrichment window + "enable_enrich": true, // default true + "enable_latechunk": false, // default false + "enable_docling_chunk": true, // default true – false selects the legacy fixed-size chunker + "retrieval_mode": "hybrid", // optional – validated, recorded on the index config + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, // default 50 + "batch_size_enrich": 25 // default 25 } ``` -### Index Build Options +* `session_id` is **optional**. Without it the default LanceDB table is used and overviews go to the global `index_store/overviews/overviews.jsonl` instead of a per-index file. +* `retrieval_mode` cannot change the artifacts written at index time; it is validated (400 on an unsupported value) and stored in the index config as `retrieval.search_type`. The mode that matters is the one you send at query time. +* `embedding_model` is applied to this build **and** recorded in the index metadata (when `session_id` is present), which is what makes queries against that index use the same embedder. +* `enable_latechunk` defaults to `false` here, so an HTTP build without the flag writes no `_lc` table even though the `default` profile enables late chunking. +* `enable_docling_chunk` defaults to `true`; sending `false` selects the legacy fixed-size chunker instead of Docling's structure-aware chunker. +* Unknown fields are ignored silently. + +### 2.6 Index response + ```jsonc { - "latechunk": true, - "doclingChunk": false, - "chunkSize": 512, - "chunkOverlap": 64, - "retrievalMode": "hybrid|dense|fts", - "windowSize": 2, - "enableEnrich": true, - "embeddingModel": "Qwen/Qwen3-Embedding-0.6B", - "enrichModel": "qwen3:0.6b", - "overviewModel": "qwen3:0.6b", - "batchSizeEmbed": 64, - "batchSizeEnrich": 32 + "message": "Indexing process for 2 file(s) completed successfully.", + "table_name": "text_pages_", + "latechunk": false, + "docling_chunk": true, + "indexing_config": { + "chunk_size": 512, + "retrieval_mode": "hybrid", + "window_size": 2, + "enable_enrich": true, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, + "batch_size_enrich": 25 + } } ``` +`indexing_config.embedding_model` reports the model actually used for the build, not the raw request value. There is no `indexed_files` field. + +Errors: `400 { "error": "A 'file_paths' list is required." }`, `400 { "error": "Invalid JSON" }`, `400` for an unsupported `retrieval_mode`, and `500 { "error": "Failed to start indexing: …" }`. + +--- + +## 3. Frontend Wrapper (`src/lib/api.ts`) + +The typed client exported as `chatAPI`. Base URLs come from `NEXT_PUBLIC_API_URL` (default `http://localhost:8000`) and `NEXT_PUBLIC_RAG_API_URL` (default `http://localhost:8001`), both inlined at build time. **The browser talks to both origins**, so a deployment must expose both. + +| Method | Target | +|--------|--------| +| `checkHealth()` | `GET :8000/health` | +| `sendMessage({message, model?, conversation_history?})` | `POST :8000/chat` | +| `getSessions()` | `GET :8000/sessions` | +| `createSession(title?, model?)` | `POST :8000/sessions` | +| `getSession(sessionId)` | `GET :8000/sessions/` | +| `sendSessionMessage(sessionId, message, opts)` | `POST :8000/sessions//messages` | +| `saveStreamedTurn(sessionId, userMessage, assistantMessage, sourceDocuments?)` | `POST :8000/sessions//messages/save` | +| `deleteSession(sessionId)` | `DELETE :8000/sessions/` | +| `renameSession(sessionId, title)` | `POST :8000/sessions//rename` | +| `uploadFiles(sessionId, files)` | `POST :8000/sessions//upload` | +| `indexDocuments(sessionId)` | `POST :8000/sessions//index` (no options) | +| `getModels()` | `GET :8000/models` | +| `createIndex(name, description?, metadata?)` | `POST :8000/indexes` | +| `uploadFilesToIndex(indexId, files)` | `POST :8000/indexes//upload` | +| `buildIndex(indexId, opts)` | `POST :8000/indexes//build` | +| `listIndexes()` | `GET :8000/indexes` | +| `getSessionIndexes(sessionId)` | `GET :8000/sessions//indexes` | +| `deleteIndex(indexId)` | `DELETE :8000/indexes/` | +| `linkIndexToSession(sessionId, indexId)` | `POST :8000/sessions//indexes/` | +| **`streamSessionMessage(params, onEvent)`** | **`POST :8001/chat/stream`** — the default chat path | + +The camelCase argument names on these methods (`retrievalK`, `rerankerTopK`, `doclingChunk`, …) are TypeScript parameter names. `sendSessionMessage` and `streamSessionMessage` serialise them to snake_case JSON; `buildIndex` sends camelCase, which both servers normalise. + +Exported model defaults, kept in step with `rag_system/main.py`: + +```ts +export const DEFAULT_GENERATION_MODEL = 'qwen3.5:9b'; +export const DEFAULT_ENRICHMENT_MODEL = 'qwen3.5:4b'; +export const DEFAULT_EMBEDDING_MODEL = 'microsoft/harrier-oss-v1-0.6b'; +``` + --- -_This reference is derived from static code analysis of `backend/server.py`, `rag_system/api_server.py`, and `src/lib/api.ts`. Keep it in sync with route or type changes._ \ No newline at end of file +_Derived by reading `backend/server.py`, `rag_system/api_server.py` and `src/lib/api.ts`. Keep it in sync with route, option and response-shape changes._ diff --git a/Documentation/architecture_overview.md b/Documentation/architecture_overview.md index 02bb4ef6..51af54da 100644 --- a/Documentation/architecture_overview.md +++ b/Documentation/architecture_overview.md @@ -1,83 +1,205 @@ # 🏗️ System Architecture Overview -_Last updated: 2025-07-06_ +_Last updated: 2026-08-08_ -This document explains how data and control flow through the Advanced **RAG System** — from a user's browser all the way to model inference and back. It is intended as the **ground-truth reference** for engineers and integrators. +This document explains how data and control flow through **localGPT** — from a user's browser to model inference and back. It is the ground-truth reference for engineers and integrators; every port, path and endpoint below was read out of the current source. --- -## 1. Bird's-Eye Diagram +## 1. Process Topology + +The system is four separate OS processes. Nothing in `rag_system` runs inside the backend gateway — the gateway talks to the RAG API over HTTP. ```mermaid flowchart LR subgraph Client - U["👤 User (Browser)"] - FE["Next.js Front-end\nReact Components"] + U["👤 User (Browser)"] + FE["Next.js frontend
:3000"] U --> FE end - subgraph Network - FE -->|HTTP/JSON| BE["Python HTTP Server\nbackend/server.py"] - end - - subgraph Core["rag_system core package"] - BE --> LOOP["Agent Loop\n(rag_system/agent/loop.py)"] - BE --> IDX["Indexing Pipeline\n(pipelines/indexing_pipeline.py)"] - - LOOP --> RP["Retrieval Pipeline\n(pipelines/retrieval_pipeline.py)"] - LOOP --> VER["Verifier (Grounding Check)"] - RP --> RET["Retrievers\nBM25 | Dense | Hybrid"] - RP --> RER["AI Reranker"] - RP --> SYNT["Answer Synthesiser"] + subgraph Services + BE["Backend gateway
backend/server.py
:8000"] + RAG["RAG API
rag_system/api_server.py
:8001"] + OL["Ollama
:11434"] end subgraph Storage - LDB[("LanceDB Vector Tables")] - SQL[("SQLite – chat & metadata")] + SQL[("SQLite
backend/chat_data.db")] + LDB[("LanceDB
./lancedb")] + FS["File system
shared_uploads/ · index_store/"] end - subgraph Models - OLLAMA["Ollama Server\n(qwen3, etc.)"] - HF["HuggingFace Hosted\nEmbedding/Reranker Models"] + FE -->|"REST: sessions, indexes, uploads, non-streaming chat"| BE + FE -->|"POST /chat/stream (SSE) — default chat path"| RAG + BE -->|"POST /chat, POST /index"| RAG + BE -->|"direct answers only (routing is local)"| OL + RAG -->|"generation, enrichment, verification"| OL + + BE -->|"sessions, messages, indexes"| SQL + RAG -->|"index metadata only"| SQL + RAG --> LDB + RAG --> FS +``` + +| Process | Entry point | Port | Server type | +|---------|-------------|------|-------------| +| Frontend | `npm run dev` (or `npm run build && npm run start`) | 3000 | Next.js 15 / React 19 | +| Backend gateway | `python backend/server.py` | 8000 | `socketserver.ThreadingTCPServer` (concurrent) | +| RAG API | `python -m rag_system.api_server` or `python -m rag_system.main api --port 8001` | 8001 | `socketserver.TCPServer` (**requests are serialized**) | +| Ollama | `ollama serve` | 11434 | external | + +`python run_system.py` starts all four and aggregates their logs. + +The backend port is the module constant `PORT = 8000` in `backend/server.py` and the RAG API port defaults to `8001` in `start_server()`; neither is read from an environment variable. What *is* configurable is where each side looks for the others — see [§6](#6-configuration-entry-points). + +### The browser talks to two origins + +The chat UI defaults to streaming (`enableStream` is `true` in `src/components/ui/session-chat.tsx`), and streaming goes **directly** from the browser to `:8001/chat/stream`, bypassing the gateway. Session CRUD, uploads, index management and the non-streaming chat fallback go to `:8000`. Any deployment must therefore expose **both** ports to the browser, and the frontend build must know both URLs (`NEXT_PUBLIC_API_URL`, `NEXT_PUBLIC_RAG_API_URL`). Both servers send `Access-Control-Allow-Origin: *`. + +--- + +## 2. Request Paths + +### 2.1 Streaming chat (the default UI path) + +```mermaid +sequenceDiagram + participant B as Browser + participant R as RAG API :8001 + participant O as Ollama :11434 + B->>R: POST /chat/stream {query, session_id, table_name, options} + R->>R: Agent.run(...) — triage → retrieve → rerank → prune → synthesize + R->>O: enrichment model (routing, decomposition, verification) + R->>O: generation model (answer, streamed) + R-->>B: SSE: analyze, decomposition, retrieval_*, rerank_*, token, complete +``` + +`Agent.run()` executes the whole pipeline and the handler forwards each phase as an SSE event. See [§2.4](#24-in-process-pipeline-inside-the-rag-api). + +### 2.2 Non-streaming chat + +```mermaid +sequenceDiagram + participant B as Browser + participant G as Backend :8000 + participant R as RAG API :8001 + participant O as Ollama :11434 + B->>G: POST /sessions/{id}/messages {message, options} + G->>G: persist user message (SQLite) + G->>G: should_use_rag() — deterministic gate, no model call + alt route = RAG + G->>R: POST /chat {query, session_id, table_name, options} + R-->>G: {answer, source_documents} + else route = direct + G->>O: generation model, thinking disabled end + G->>G: persist assistant message (SQLite) + G-->>B: {response, session, source_documents, used_rag} +``` + +**The gateway gate.** `should_use_rag(message, idx_ids, force_rag)` in `backend/server.py` is a pure function: `force_rag` → RAG; no linked indexes → direct; a whole-message smalltalk/assistant-meta allowlist regex (`hello`, `thanks!`, `who are you`, capped at six words) → direct; everything else → RAG. No LLM call, no file reads, no network. Measured at ~0.002 ms per message against ~750 ms for the enrichment-model router it replaced (Phase 2.3). + +It is deliberately biased toward RAG. The agent-side triage in [§2.4](#24-in-process-pipeline-inside-the-rag-api) still runs on every RAG request and can still return a direct answer, so the gate never has to decide correctly on its own — it only has to avoid sending "hi" through a retrieval pipeline. `used_rag` in the response reports which way the gate went. - %% data edges - IDX -->|chunks & embeddings| LDB - RET -->|vector search| LDB - LOOP -->|LLM calls| OLLAMA - RP -->|LLM calls| OLLAMA - VER -->|LLM calls| OLLAMA - RP -->|rerank| HF +### 2.3 Indexing - BE -->|CRUD| SQL +Uploads land in `shared_uploads/` (backend). Indexing is then triggered either per session (`POST :8000/sessions/{id}/index`) or per named index (`POST :8000/indexes/{id}/build`); both forward to `POST :8001/index`, which runs `IndexingPipeline.run(file_paths)` in the RAG API process. + +```mermaid +flowchart LR + UP["shared_uploads/*"] --> DC["DocumentConverter
(Docling, OCR when available)"] + DC --> CH["DoclingChunker
(token-budgeted)"] + CH --> OV["OverviewBuilder
index_store/overviews/<id>.jsonl"] + CH --> EN["ContextualEnricher
(optional, enrichment model)"] + EN --> EM["Embeddings
harrier-oss-v1-0.6b"] + EM --> LT["LanceDB table
+ native FTS index"] + CH --> LC["Late-chunk encoder
(optional) → <table>_lc"] ``` +Contextual enrichment returns copies of the chunks, so the late-chunk leg encodes the **original** chunk text while the main table stores the enriched text (with the original preserved in `metadata.original_text`). + +### 2.4 In-process pipeline inside the RAG API + +```mermaid +flowchart TD + Q["Query"] --> T{"Triage"} + T -->|direct_answer| DA["Generation model, streamed"] + T -->|rag_query| DEC{"Query decomposition
(enabled in 'default')"} + DEC -->|n sub-queries| PAR["Parallel retrieval (≤3 threads)"] + DEC -->|single query| RP["RetrievalPipeline.run"] + PAR --> RP + RP --> RET["MultiVectorRetriever
FTS + vector, fused with RRF"] + RET --> LCM["Late-chunk retrieval + ±1 merge (optional)"] + LCM --> RR["Reranker (off by default)
Qwen3-Reranker-4B"] + RR --> CE["Context expansion (±window chunks)"] + CE --> PR["Provence sentence pruning (opt-in)"] + PR --> SY["Answer synthesis (generation model, streamed)"] + SY --> VF["Verifier → appends [Confidence: N%]"] +``` + +Triage order in `Agent._triage_query_async`: the overview router runs first (an LLM call on the enrichment model, grounded in the document overviews loaded for the session); if there is conversation history the query short-circuits to `rag_query`; otherwise a fallback LLM triage picks `rag_query` / `direct_answer`. `force_rag=true` on the RAG API skips triage entirely. + +There is no GraphRAG path. `GraphExtractor`, `GraphRetriever`, `GraphQueryTranslator`, the `graph_query` triage outcome and the `retrieval.graph` / `graph_strategy` config blocks were **deleted on 2026-08-09** (roadmap item 2.5): the path was unreachable, GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, and it costs 41–57× at indexing and up to ~377× in query tokens ([`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) §6). + --- -### Data-flow Narrative -1. **User** interacts with the Next.js UI; messages are posted via `src/lib/api.ts`. -2. **backend/server.py** receives JSON over HTTP, applies CORS, and proxies the request into `rag_system`. -3. **Agent Loop** decides (via _Triage_) whether to perform Retrieval-Augmented Generation (RAG) or direct LLM answering. -4. If RAG is chosen: - 1. **Retrieval Pipeline** fetches candidates from **LanceDB** using BM25 + dense vectors. - 2. **AI Reranker** (HF model) sorts snippets. - 3. **Answer Synthesiser** calls **Ollama** to write the final answer. -5. Answers can be **Verified** for grounding (optional flag). -6. Index-building is an offline path triggered from the UI — PDF/📄 files are chunked, embedded and stored in LanceDB. +## 3. Where State Lives + +| Store | Location | Written by | Contents | +|-------|----------|-----------|----------| +| SQLite | `backend/chat_data.db` (`DB_PATH`) | backend (all tables), RAG API (index metadata only) | `sessions`, `messages`, `session_documents`, `indexes`, `index_documents`, `session_indexes` | +| LanceDB | `./lancedb` (`storage.lancedb_uri`, `LANCEDB_PATH`) | RAG API | `text_pages_v4` (default table), `text_pages_` (per named index), `
_lc` (late-chunk vectors) | +| Overviews | `index_store/overviews/.jsonl`, plus `index_store/overviews/overviews.jsonl` as a global fallback | RAG API (indexing) | one-paragraph document summaries used by the agent's triage router (the gateway gate no longer reads them) | +| Uploads | `shared_uploads/` | backend | original uploaded files, prefixed with a UUID | + +**Chat messages are written by `backend/server.py` only** (`handle_session_chat`). The RAG API holds a `ChatDatabase` handle but never touches the `messages` or `sessions` tables. --- -## 2. Component Documents -The table below links to deep-dives for each major component. +## 4. Threading and Shared State -| **Component** | **Documentation** | -|---------------|-------------------| -| Agent Loop | [`system_overview.md`](system_overview.md) | -| Indexing Pipeline | [`indexing_pipeline.md`](indexing_pipeline.md) | -| Retrieval Pipeline | [`retrieval_pipeline.md`](retrieval_pipeline.md) | -| Verifier | [`verifier.md`](verifier.md) | -| Triage System | [`triage_system.md`](triage_system.md) | +* The **backend** is threaded (`ThreadingTCPServer`, `daemon_threads = True`). `backend/database.py` opens a fresh SQLite connection per call, so concurrent handlers are safe. +* The **RAG API is single-threaded**. One `/chat`, `/chat/stream` or `/index` request is served at a time; everything else queues. This is deliberate: the process holds one `RAG_AGENT` singleton whose pipeline config is mutated by per-request options. +* Because that config is shared, per-request retrieval options persist into later requests unless the next request overrides them. The exception is the per-request generation `model`, which is applied through a context manager (`_generation_model_override`) that restores the previous value and rejects model ids that do not match the active `LLM_BACKEND`. +* `factory.get_pipeline_config()` returns a `copy.deepcopy` of the profile, so the module-level `PIPELINE_CONFIGS` in `rag_system/main.py` is never mutated. --- -> **Change-management**: whenever architecture changes (new micro-service, different DB, etc.) update this overview diagram first, then individual component docs. \ No newline at end of file +## 5. Known Architectural Limitations + +These are real, current behaviours — not planned work. See [`improvement_plan.md`](improvement_plan.md) for the fixes on the roadmap. + +1. **Streamed turns are persisted by a client callback, not by the stream itself.** `POST :8001/chat/stream` writes nothing to SQLite; when the stream completes, the browser posts the finished turn to `POST :8000/sessions/{id}/messages/save`, which stores both messages and derives the session title. A client that consumes the stream directly and skips that call gets no history. +2. **The RAG API serializes requests.** Concurrency is one in-flight RAG request per process. +3. **Per-request options leak across requests** on the RAG API (see §4). +5. **`enable_docling_chunk: false` has no effect.** `rag_system/api_server.py` sets `chunker_mode = "docling"` when the flag is true and leaves it unset otherwise, and `IndexingPipeline` defaults `chunker_mode` to `"docling"`. The Docling chunker always runs over HTTP. (`create_index_script.py` does set `"legacy"`.) +6. **Service ports are not configurable by environment variable** — only the URLs used to reach them are. + +--- + +## 6. Configuration Entry Points + +| Concern | Where | +|---------|-------| +| Model defaults, pipeline profiles | `rag_system/main.py` (`OLLAMA_CONFIG`, `WATSONX_CONFIG`, `EXTERNAL_MODELS`, `PIPELINE_CONFIGS`) | +| Agent / pipeline construction | `rag_system/factory.py` (`get_agent`, `get_indexing_pipeline`, `get_pipeline_config`) | +| Active profile for the RAG API | `RAG_CONFIG_MODE` (default `default`; an unknown value silently falls back to `default`) | +| Service URLs and model overrides | environment variables — see [`.env.example`](../.env.example) and [`system_overview.md`](system_overview.md#7-configuration) | + +--- + +## 7. Component Documents + +| Component | Documentation | +|-----------|---------------| +| Whole system, models, configuration | [`system_overview.md`](system_overview.md) | +| Why each component is built this way (evidence + eval numbers; "deliberately not implemented") | [`design_rationale.md`](design_rationale.md) | +| HTTP APIs (backend + RAG API) | [`api_reference.md`](api_reference.md) | +| Indexing pipeline | [`indexing_pipeline.md`](indexing_pipeline.md) | +| Retrieval pipeline | [`retrieval_pipeline.md`](retrieval_pipeline.md) | +| Verifier | [`verifier.md`](verifier.md) | +| Triage / routing | [`triage_system.md`](triage_system.md) | +| Roadmap | [`improvement_plan.md`](improvement_plan.md) | + +> **Change management**: when the topology changes (new process, new store, a new browser-facing origin), update this overview first, then the component docs. diff --git a/Documentation/deployment_guide.md b/Documentation/deployment_guide.md index 37a54e14..633b13f3 100644 --- a/Documentation/deployment_guide.md +++ b/Documentation/deployment_guide.md @@ -1,22 +1,22 @@ -# 🚀 RAG System Deployment Guide +# 🚀 LocalGPT Deployment Guide -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide provides comprehensive instructions for deploying the RAG system using both Docker and direct development approaches. +This guide covers deploying LocalGPT with Docker or as directly-run processes. --- ## 🎯 Deployment Options -### Option 1: Docker Deployment (Production) 🐳 -- **Best for**: Production environments, containerized deployments, scaling -- **Pros**: Isolated, reproducible, easy to manage -- **Cons**: Slightly more complex setup, resource overhead +### Option 1: Docker Deployment 🐳 +- **Best for**: reproducible environments, keeping dependencies isolated +- **Pros**: one command, pinned base images, health-gated startup order +- **Cons**: slower first build, GPU passthrough is extra work -### Option 2: Direct Development (Development) 💻 -- **Best for**: Development, debugging, customization -- **Pros**: Direct access to code, faster iteration, easier debugging -- **Cons**: More dependencies to manage +### Option 2: Direct Processes 💻 +- **Best for**: development, debugging, GPU/MPS access +- **Pros**: direct access to code, faster iteration, native acceleration +- **Cons**: more dependencies to manage on the host --- @@ -32,13 +32,12 @@ This guide provides comprehensive instructions for deploying the RAG system usin #### **Recommended Requirements** - **CPU**: 8+ cores, 3.0GHz+ -- **RAM**: 32GB+ (for large models) +- **RAM**: 32GB+ - **Storage**: 200GB+ SSD -- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional, for acceleration) +- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional; Apple Silicon uses MPS) ### 1.2 Common Dependencies -**Both deployment methods require:** ```bash # Ollama (required for both approaches) curl -fsSL https://ollama.ai/install.sh | sh @@ -49,21 +48,17 @@ git 2.30+ ### 1.3 Docker-Specific Dependencies -**For Docker deployment:** ```bash -# Docker & Docker Compose Docker Engine 24.0+ -Docker Compose 2.20+ +Docker Compose plugin 2.20+ ``` -### 1.4 Direct Development Dependencies +### 1.4 Direct Deployment Dependencies -**For direct development:** ```bash -# Python & Node.js -Python 3.8+ -Node.js 16+ -npm 8+ +Python 3.10+ # 3.11 recommended; the images use python:3.11-slim +Node.js 20+ +npm 10+ ``` --- @@ -76,307 +71,385 @@ npm 8+ **Ubuntu/Debian:** ```bash -# Install Docker curl -fsSL https://get.docker.com -o get-docker.sh sudo sh get-docker.sh sudo usermod -aG docker $USER newgrp docker -# Install Docker Compose V2 sudo apt-get update sudo apt-get install docker-compose-plugin ``` **macOS:** ```bash -# Install Docker Desktop brew install --cask docker # Or download from: https://www.docker.com/products/docker-desktop ``` **Windows:** ```bash -# Install Docker Desktop with WSL2 backend +# Install Docker Desktop with the WSL2 backend # Download from: https://www.docker.com/products/docker-desktop ``` #### **Step 2: Clone Repository** ```bash -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT ``` #### **Step 3: Install Ollama** ```bash -# Install Ollama (runs locally even with Docker) +# Runs on the host by default, even with Docker curl -fsSL https://ollama.ai/install.sh | sh -# Start Ollama ollama serve -# In another terminal, install models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# In another terminal +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` -#### **Step 4: Launch Docker System** +#### **Step 4: Launch** ```bash -# Start all containers using the convenience script +# Convenience script (local Ollama) ./start-docker.sh -# Or manually: +# Containerized Ollama instead +./start-docker.sh container + +# Or manually docker compose --env-file docker.env up --build -d ``` #### **Step 5: Verify Deployment** ```bash -# Check container status docker compose ps -# Test all endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API +curl http://localhost:3000 # Frontend +curl http://localhost:8000/health # Backend +curl http://localhost:8001/health # RAG API curl http://localhost:11434/api/tags # Ollama ``` -### 2.2 Docker Management +### 2.2 Startup Order -#### **Container Operations** -```bash -# Start system -./start-docker.sh - -# Stop system -./start-docker.sh stop +`docker-compose.yml` gates the stack on health checks: -# View logs -./start-docker.sh logs +``` +rag-api (healthy: GET /health) -> backend (healthy: GET /health) -> frontend +``` -# Check status -./start-docker.sh status +`rag-api` loads the embedding and reranker models before it answers `/health`, so +its check uses a 60s start period. Until it passes, `backend` stays in `created` +and `frontend` after it. This is expected on a cold start, not a hang — watch +`docker compose logs -f rag-api`. -# Manual Docker Compose commands -docker compose ps # Check status -docker compose logs -f # Follow logs -docker compose down # Stop all containers -docker compose up --build -d # Rebuild and restart -``` +### 2.3 Docker Management -#### **Individual Container Management** ```bash -# Restart specific service -docker compose restart rag-api +# Convenience script +./start-docker.sh # start (local Ollama) +./start-docker.sh container # start (containerized Ollama) +./start-docker.sh stop # stop +./start-docker.sh logs # follow logs +./start-docker.sh status # container status +./start-docker.sh help # usage -# View specific service logs -docker compose logs -f backend +# Compose directly +docker compose ps +docker compose logs -f +docker compose down +docker compose --env-file docker.env up --build -d -# Execute commands in container -docker compose exec rag-api python -c "print('Hello')" +# One service +docker compose restart rag-api +docker compose logs -f backend +docker compose exec rag-api python -c "print('hello')" ``` +### 2.4 Compose Files + +| File | Contents | +|------|----------| +| `docker-compose.yml` | The full stack: `rag-api`, `backend`, `frontend`, and an optional `ollama` service behind the `with-ollama` profile | +| `docker-compose.local-ollama.yml` | The same three application services without the optional `ollama` service | + +`docker-compose.yml` already defaults to host Ollama and only adds the `ollama` +container when you pass `--profile with-ollama`, so the second file is a +convenience, not a requirement. + --- -## 3. 💻 Direct Development +## 3. 💻 Direct Deployment ### 3.1 Installation -#### **Step 1: Install Dependencies** - **Python Dependencies:** ```bash -# Clone repository -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT -# Create virtual environment (recommended) python -m venv venv source venv/bin/activate # On Windows: venv\Scripts\activate -# Install Python packages pip install -r requirements.txt ``` **Node.js Dependencies:** ```bash -# Install Node.js dependencies npm install ``` -#### **Step 2: Install and Configure Ollama** +### 3.2 Install and Configure Ollama ```bash -# Install Ollama curl -fsSL https://ollama.ai/install.sh | sh - -# Start Ollama ollama serve -# In another terminal, install models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +# In another terminal +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` -#### **Step 3: Launch System** +### 3.3 Launch **Option A: Integrated Launcher (Recommended)** ```bash -# Start all components with one command +# Development python run_system.py + +# Production: runs `npm run build`, then `next start` +python run_system.py --mode prod ``` -**Option B: Manual Component Startup** +The launcher records the launcher PID plus each child PID in +`logs/run_system.pid`, aggregates every service's stdout into `logs/.log`, +and restarts a required service that exits unexpectedly (checked every 30s). + +**Option B: Manual Component Startup — all from the repository root** ```bash # Terminal 1: RAG API python -m rag_system.api_server # Terminal 2: Backend -cd backend && python server.py +python backend/server.py # Terminal 3: Frontend -npm run dev +npm run build && npm run start # or: npm run dev # Access at http://localhost:3000 ``` -#### **Step 4: Verify Installation** -```bash -# Check system health -python system_health_check.py - -# Test endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API -``` +> Relative paths (`backend/chat_data.db`, `lancedb/`, `index_store/`, +> `shared_uploads/`) resolve against the working directory. Always start from the +> repository root. -### 3.2 Direct Development Management +### 3.4 Direct Deployment Management -#### **System Operations** ```bash -# Start system -python run_system.py - -# Check system health -python system_health_check.py - -# Stop system -# Press Ctrl+C in terminal running run_system.py +python run_system.py --health # HTTP checks; exit 1 if a required service fails +python run_system.py --logs-only # tail logs/*.log from another shell +python run_system.py --stop # terminate everything in logs/run_system.pid +python run_system.py --no-frontend # Ollama + RAG API + backend only +python system_health_check.py # deep check incl. a real embedding + query ``` -#### **Individual Component Management** -```bash -# Start components individually -python -m rag_system.api_server # RAG API on port 8001 -cd backend && python server.py # Backend on port 8000 -npm run dev # Frontend on port 3000 - -# Development tools -npm run build # Build frontend for production -pip install -r requirements.txt --upgrade # Update Python packages -``` +`--stop` reads `logs/run_system.pid`, kills the launcher first (so its monitor +cannot restart anything), then each service and its descendants with SIGTERM, +escalating to SIGKILL after 10s. It exits non-zero if there is no pidfile. --- -## 4. Architecture Comparison +## 4. Architecture ### 4.1 Docker Architecture ```mermaid graph TB subgraph "Docker Containers" - Frontend[Frontend Container
Next.js
Port 3000] - Backend[Backend Container
Python API
Port 8000] - RAG[RAG API Container
Document Processing
Port 8001] + Frontend[frontend
Next.js
Port 3000] + Backend[backend
Gateway + SQLite
Port 8000] + RAG[rag-api
Indexing + Retrieval
Port 8001] end - - subgraph "Local System" + + subgraph "Host" Ollama[Ollama Server
Port 11434] end - + + Browser --> Frontend + Browser -. "SSE /chat/stream" .-> RAG Frontend --> Backend Backend --> RAG RAG --> Ollama + Backend --> Ollama ``` -### 4.2 Direct Development Architecture +`backend` and `rag-api` both mount `./backend`, `./lancedb` and `./index_store`, +and both point `DB_PATH` at `/app/backend/chat_data.db`, so they share one SQLite +file and one vector store. + +### 4.2 Direct Architecture ```mermaid graph TB subgraph "Local Processes" - Frontend[Next.js Dev Server
Port 3000] - Backend[Python Backend
Port 8000] - RAG[RAG API
Port 8001] + Frontend[Next.js
Port 3000] + Backend[backend/server.py
Port 8000] + RAG[rag_system.api_server
Port 8001] Ollama[Ollama Server
Port 11434] end - + + Browser --> Frontend + Browser -. "SSE /chat/stream" .-> RAG Frontend --> Backend Backend --> RAG RAG --> Ollama + Backend --> Ollama ``` +### 4.3 Concurrency + +- `backend/server.py` uses a `ThreadingTCPServer`, so it handles requests in + parallel and stays responsive during a long RAG call. +- `rag_system/api_server.py` uses a plain single-threaded `TCPServer`. **RAG + requests are serialised** — one chat or indexing run at a time. Plan capacity + for a single concurrent user of the RAG API, or put a queue in front of it. +- The backend allows `RAG_API_TIMEOUT` (default 600s) for a chat call and + `RAG_API_INDEX_TIMEOUT` (default 3600s) for indexing, returning 504 on timeout + and 502 when the RAG API is unreachable. + --- ## 5. Configuration ### 5.1 Environment Variables +Every variable below is read by code; the value shown is the default when unset. +`.env.example` carries the same list. + +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` (inlined at build time) | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` (inlined at build time) | +| `DB_PATH` | `backend/chat_data.db` | `backend/database.py` | +| `LANCEDB_PATH` | `./lancedb` | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | same | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (only loaded when reranking is switched on) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` | +| `RAG_API_TIMEOUT` | `600` | `backend/server.py` | +| `RAG_API_INDEX_TIMEOUT` | `3600` | `backend/server.py` | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py` | +| `HF_TOKEN` | unset | HuggingFace downloads | + #### **Docker Configuration (`docker.env`)** ```bash -# Ollama Configuration OLLAMA_HOST=http://host.docker.internal:11434 - -# Service Configuration NODE_ENV=production RAG_API_URL=http://rag-api:8001 NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B ``` -#### **Direct Development Configuration** +All compose services declare +`extra_hosts: ["host.docker.internal:host-gateway"]`, so `host.docker.internal` +resolves on Linux as well as macOS and Windows. + +`NEXT_PUBLIC_*` are **build-time** values for Next.js. `docker-compose.yml` passes +them as build args to `Dockerfile.frontend`; setting them only at runtime has no +effect on an already-built image. If the browser must reach the services under a +different hostname, set them and rebuild: + +```bash +NEXT_PUBLIC_API_URL=https://gpt.example.com/api \ +NEXT_PUBLIC_RAG_API_URL=https://gpt.example.com/rag \ +docker compose --env-file docker.env up --build -d frontend +``` + +#### **Direct Deployment Configuration** ```bash -# Environment variables are set automatically by run_system.py -# Override in environment if needed: +# run_system.py inherits your shell environment unchanged and adds only +# NODE_ENV=production to the Python services in --mode prod. export OLLAMA_HOST=http://localhost:11434 export RAG_API_URL=http://localhost:8001 +python run_system.py ``` ### 5.2 Model Configuration -#### **Default Models** -```python -# Embedding Models -EMBEDDING_MODELS = [ - "Qwen/Qwen3-Embedding-0.6B", # Fast, 1024 dimensions - "Qwen/Qwen3-Embedding-4B", # High quality, 2048 dimensions -] +Defaults live in `rag_system/main.py` and are overridable by environment variable. -# Generation Models -GENERATION_MODELS = [ - "qwen3:0.6b", # Fast responses - "qwen3:8b", # High quality -] -``` +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation | `qwen3.5:9b` | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility | `qwen3.5:4b` | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024 dims, ~1.2 GB) | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context — multilingual / long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranker (**off by default**) | `Qwen/Qwen3-Reranker-4B` | `BAAI/bge-reranker-v2-m3` (low latency), `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | -### 5.3 Performance Tuning - -#### **Memory Settings** -```bash -# For Docker: Increase memory allocation -# Docker Desktop → Settings → Resources → Memory → 16GB+ +`GET /models` on either service reports what is actually selectable: +Ollama tags are split into generation and embedding lists by a substring match +(`embed`, `bge`, `embedding` on the RAG API; the backend also matches `text`), and +the HuggingFace embedding models are appended. A tag whose name happens to contain +one of those substrings will be classified as an embedding model. -# For Direct Development: Monitor with -htop # or top on macOS -``` +**Changing the embedding model requires re-indexing.** Vector width is measured +from the loaded model; writing a different width into an existing LanceDB table +fails with an explicit error. -#### **Model Settings** -```python -# Batch sizes (adjust based on available RAM) -EMBEDDING_BATCH_SIZE = 50 # Reduce if OOM -ENRICHMENT_BATCH_SIZE = 25 # Reduce if OOM +### 5.3 Performance Tuning -# Chunk settings -CHUNK_SIZE = 512 # Text chunk size -CHUNK_OVERLAP = 64 # Overlap between chunks -``` +There are no `SEARCH_CONFIG` / `CHUNK_OVERLAP` globals. The knobs are pipeline +config keys and per-request fields. + +**Indexing throughput** — `PIPELINE_CONFIGS[]["indexing"]` in +`rag_system/main.py`, or `batch_size_embed` / `batch_size_enrich` on +`POST /index`: + +| Key | `default` | `fast` | Request field | +|-----|-----------|--------|---------------| +| `embedding_batch_size` | 50 | 100 | `batch_size_embed` (default 50) | +| `enrichment_batch_size` | 10 | 50 | `batch_size_enrich` (default 25) | + +**Chunking** — `chunk_size` on `POST /index` (default 512) feeds the Docling +chunker's `max_tokens`, or the legacy chunker's `max_chunk_size` with +`min_chunk_size = chunk_size // 4`. When no `chunking.chunk_size` is present at +all (for example the `python -m rag_system.main index` CLI path), the pipeline +falls back to 1500. There is no `chunk_overlap` setting. + +**Contextual enrichment** is the most expensive part of indexing: one LLM call per +chunk. Disable it with `enable_enrich: false` for the fastest ingest. + +**Query cost** — the biggest lever is the profile: + +| | `default` | `fast` | +|---|---|---| +| `retrieval.search_type` | `hybrid` | `vector_only` | +| `retrieval_k` | 20 | 10 | +| `reranker.enabled` | true | false | +| `query_decomposition.enabled` | true | false | +| `verification.enabled` | true | false | +| `retrieval.latechunk.enabled` | true | false | + +Select the profile with `RAG_CONFIG_MODE=fast` for the RAG API, or `--mode fast` +for the CLI. Individual toggles (`ai_rerank`, `verify`, `query_decompose`, +`context_expand`, `retrieval_k`, `reranker_top_k`) can also be sent per request. + +**Memory** — the embedding model stays resident in the RAG API process +(`microsoft/harrier-oss-v1-0.6b`, ~1.2 GB). The reranker is **not loaded at all** +unless reranking is switched on, and switching it on pulls ~7.5 GB of +`Qwen/Qwen3-Reranker-4B` weights alongside it — see +[`../eval/DECISIONS.md`](../eval/DECISIONS.md) for the quality/latency trade that +decision rests on. Enabling late chunking loads a second copy of the embedding +model. --- @@ -384,77 +457,84 @@ CHUNK_OVERLAP = 64 # Overlap between chunks ### 6.1 System Monitoring -#### **Health Checks** ```bash -# Comprehensive system check +# Health curl -f http://localhost:3000 && echo "✅ Frontend OK" curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" +curl -f http://localhost:8001/health && echo "✅ RAG API OK" curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" + +# Or, for a direct deployment +python run_system.py --health ``` -#### **Performance Monitoring** -```bash -# Docker monitoring -docker stats +`GET /health` on the backend returns Ollama reachability, the model list and +database stats; on the RAG API it returns `{"status": "ok"}` as soon as the agent +has finished loading. -# Direct development monitoring -htop # Overall system -nvidia-smi # GPU usage (if available) +```bash +# Resource usage +docker stats # Docker +htop # host +nvidia-smi # GPU, if present ``` ### 6.2 Log Management #### **Docker Logs** ```bash -# All services docker compose logs -f - -# Specific service docker compose logs -f rag-api - -# Save logs to file docker compose logs > system.log 2>&1 ``` -#### **Direct Development Logs** +#### **Direct Deployment Logs** ```bash -# Logs are printed to terminal -# Redirect to file if needed: -python run_system.py > system.log 2>&1 +# run_system.py writes per-service files +tail -f logs/system.log logs/rag-api.log logs/backend.log logs/frontend.log + +# or use the launcher's own tailer from a second shell +python run_system.py --logs-only ``` ### 6.3 Backup and Restore +Everything persistent is a host directory, including under Docker (all mounts are +bind mounts; the only named volume is `ollama_data` for the optional Ollama +container). + #### **Data Backup** ```bash -# Create backup directory -mkdir -p backups/$(date +%Y%m%d) +# Stop first so SQLite and LanceDB are not mid-write +./start-docker.sh stop # Docker +# or: python run_system.py --stop -# Backup databases and indexes -cp -r backend/chat_data.db backups/$(date +%Y%m%d)/ -cp -r lancedb backups/$(date +%Y%m%d)/ -cp -r index_store backups/$(date +%Y%m%d)/ +mkdir -p backups/$(date +%Y%m%d) +cp backend/chat_data.db backups/$(date +%Y%m%d)/ +tar czf backups/$(date +%Y%m%d)/lancedb.tar.gz lancedb/ +tar czf backups/$(date +%Y%m%d)/index_store.tar.gz index_store/ +tar czf backups/$(date +%Y%m%d)/shared_uploads.tar.gz shared_uploads/ -# For Docker: also backup volumes -docker compose down -docker run --rm -v rag_system_old_ollama_data:/data -v $(pwd)/backups:/backup alpine tar czf /backup/ollama_models_$(date +%Y%m%d).tar.gz -C /data . +# Only if you use the containerized Ollama, back up its named volume too. +# The project name is the directory name, so the volume is localgpt_ollama_data. +docker run --rm -v localgpt_ollama_data:/data -v $(pwd)/backups:/backup \ + alpine tar czf /backup/ollama_models_$(date +%Y%m%d).tar.gz -C /data . ``` #### **Data Restore** ```bash -# Stop system -./start-docker.sh stop # Docker -# Or Ctrl+C for direct development +./start-docker.sh stop # or: python run_system.py --stop -# Restore files -cp -r backups/YYYYMMDD/* ./ +cp backups/YYYYMMDD/chat_data.db backend/ +tar xzf backups/YYYYMMDD/lancedb.tar.gz +tar xzf backups/YYYYMMDD/index_store.tar.gz -# Restart system -./start-docker.sh # Docker -python run_system.py # Direct development +./start-docker.sh # or: python run_system.py ``` +Restore the SQLite file and `lancedb/` together — the database rows point at +LanceDB table names, so a mismatched pair leaves indexes that resolve to nothing. + --- ## 7. Troubleshooting @@ -463,109 +543,94 @@ python run_system.py # Direct development #### **Port Conflicts** ```bash -# Check what's using ports lsof -i :3000 -i :8000 -i :8001 -i :11434 -# For Docker: Stop conflicting containers +# Docker ./start-docker.sh stop -# For Direct: Kill processes -pkill -f "npm run dev" -pkill -f "server.py" -pkill -f "api_server" +# Direct +python run_system.py --stop ``` +If a required port (8001 or 8000) is already taken, `run_system.py` logs +`Port … already in use, skipping …` and aborts with "System startup failed". Port +11434 in use is treated as "Ollama already running" and reused; port 3000 in use is +tolerated because the frontend is optional. + #### **Docker Issues** ```bash -# Docker daemon not running -docker version # Check if daemon responds - -# Restart Docker Desktop (macOS/Windows) -# Or restart docker service (Linux) -sudo systemctl restart docker - -# Clear Docker cache -docker system prune -f +docker version # daemon reachable? +sudo systemctl restart docker # Linux +docker system prune -f # clear build cache ``` #### **Ollama Issues** ```bash -# Check Ollama status curl http://localhost:11434/api/tags -# Restart Ollama pkill ollama ollama serve -# Reinstall models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` +#### **Backend returns 502/504 for every chat** +The backend could not reach the RAG API. Check `RAG_API_URL` (must be +`http://rag-api:8001` inside Docker, not `localhost`) and that `rag-api` is +healthy. + ### 7.2 Performance Issues #### **Memory Problems** ```bash -# Check memory usage free -h # Linux vm_stat # macOS -docker stats # Docker containers +docker stats # containers -# Solutions: -# 1. Increase system RAM -# 2. Reduce batch sizes in configuration -# 3. Use smaller models (qwen3:0.6b instead of qwen3:8b) +# Options: +# 1. Switch to RAG_CONFIG_MODE=fast +# 2. Leave reranking off (the default) so no reranker weights are loaded +# 3. Use a smaller generation model (GENERATION_MODEL=qwen3.5:4b) +# 4. Disable late chunking so a second embedding model is not loaded ``` #### **Slow Response Times** ```bash -# Check model loading -curl http://localhost:11434/api/tags - -# Monitor component response times -time curl http://localhost:8001/models +# Where is the time going? +docker compose logs -f rag-api # stage-by-stage output -# Solutions: -# 1. Use SSD storage -# 2. Increase CPU cores -# 3. Use GPU acceleration (if available) +time curl -s http://localhost:8001/health # process responsive? ``` +Remember requests to the RAG API are serialised — a query that appears slow may be +queued behind an indexing run. + --- ## 8. Production Considerations ### 8.1 Security -#### **Network Security** -```bash -# Use reverse proxy (nginx/traefik) for production -# Enable HTTPS/TLS -# Restrict port access with firewall -``` +LocalGPT ships with **no authentication** and permissive CORS +(`Access-Control-Allow-Origin: *`) on both HTTP services. Before exposing it: -#### **Data Security** -```bash -# Enable authentication in production -# Encrypt sensitive data -# Regular security updates -``` +- Put a reverse proxy (nginx, Caddy, Traefik) in front and terminate TLS there +- Add authentication at the proxy +- Publish only port 3000; keep 8000 and 8001 on an internal network +- Note the browser calls the RAG API directly for streaming, so port 8001 must be + reachable by clients (or proxied) if you leave streaming enabled ### 8.2 Scaling -#### **Horizontal Scaling** -```bash -# Use Docker Swarm or Kubernetes -# Load balance frontend and backend -# Scale RAG API instances based on load -``` +`docker compose up --scale` does **not** work with the shipped compose file: every +service sets a fixed `container_name` and publishes a fixed host port. Scaling +requires removing those first. -#### **Resource Optimization** -```bash -# Use dedicated GPU nodes for AI workloads -# Implement model caching -# Optimize batch processing -``` +Beyond that, the RAG API is single-threaded and holds mutable state (the resident +agent, its per-session in-memory chat history and the semantic cache), so running +several replicas behind a load balancer needs sticky sessions at minimum. The +realistic path is a single RAG API with a queue in front of it. --- @@ -573,26 +638,24 @@ time curl http://localhost:8001/models ### 9.1 Deployment Verification -Your deployment is successful when: - -- ✅ All health checks pass +- ✅ `docker compose ps` shows all services healthy (or `python run_system.py --health` exits 0) - ✅ Frontend loads at http://localhost:3000 -- ✅ You can create document indexes +- ✅ You can create a document index - ✅ You can chat with uploaded documents -- ✅ No error messages in logs +- ✅ No errors in `docker compose logs` / `logs/` -### 9.2 Performance Benchmarks +### 9.2 What to Expect -**Acceptable Performance:** -- Index creation: < 2 minutes per 100MB document -- Query response: < 30 seconds for complex questions -- Memory usage: < 8GB total system memory +Throughput depends on hardware and model size, so treat these as shape rather than +guarantees: -**Optimal Performance:** -- Index creation: < 1 minute per 100MB document -- Query response: < 10 seconds for complex questions -- Memory usage: < 16GB total system memory +- Cold start is dominated by model downloads and the RAG API loading the embedding + and reranker models +- Indexing is dominated by contextual enrichment (one LLM call per chunk) +- Query latency is dominated by generation; `fast` mode removes reranking, + decomposition and verification +- Concurrency is one RAG request at a time --- -**Happy Deploying! 🚀** \ No newline at end of file +**Happy Deploying! 🚀** diff --git a/Documentation/docker_usage.md b/Documentation/docker_usage.md index 101307fb..47d57e52 100644 --- a/Documentation/docker_usage.md +++ b/Documentation/docker_usage.md @@ -1,101 +1,125 @@ -# 🐳 Docker Usage Guide - RAG System +# 🐳 Docker Usage Guide - LocalGPT -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide provides practical Docker commands and procedures for running the RAG system in containerized environments with local Ollama. +Practical Docker commands and procedures for running LocalGPT in containers. --- ## 📋 Prerequisites ### Required Setup -- Docker Desktop installed and running -- Ollama installed locally (even for Docker deployment) +- Docker Desktop (or Docker Engine 24+ with the Compose plugin) running +- Ollama, either on the host (default) or as a container - 8GB+ RAM available ### Architecture Overview ``` -┌─────────────────────────────────────┐ -│ Docker Containers │ -├─────────────────────────────────────┤ -│ Frontend (Port 3000) │ -│ Backend (Port 8000) │ -│ RAG API (Port 8001) │ -└─────────────────────────────────────┘ - │ - ▼ -┌─────────────────────────────────────┐ -│ Local System │ -├─────────────────────────────────────┤ -│ Ollama Server (Port 11434) │ -└─────────────────────────────────────┘ +┌─────────────────────────────────────────────────┐ +│ Docker Containers │ +├─────────────────────────────────────────────────┤ +│ frontend (3000) → backend (8000) → rag-api │ +│ (8001) │ +└─────────────────────────────────────────────────┘ + │ │ + │ browser streams │ + │ directly to :8001 ▼ + └──────────────────► ┌────────────────────┐ + │ Ollama (11434) │ + │ host.docker.internal│ + └────────────────────┘ ``` +`backend` and `rag-api` share `./backend` (SQLite), `./lancedb`, `./index_store` +and `./shared_uploads` as bind mounts. + --- ## 1. Quick Start Commands -### Step 1: Clone and Setup +### Step 1: Clone ```bash -# Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Verify Docker is running docker version ``` -### Step 2: Install and Configure Ollama (Required) +### Step 2: Ollama -**⚠️ Important**: Even with Docker, Ollama must be installed locally for optimal performance. +The compose files point the containers at `host.docker.internal:11434` by default, +and declare `extra_hosts: ["host.docker.internal:host-gateway"]` so that also +resolves on Linux. ```bash -# Install Ollama +# Install Ollama on the host curl -fsSL https://ollama.ai/install.sh | sh # Start Ollama (in one terminal) ollama serve -# Install required models (in another terminal) -ollama pull qwen3:0.6b # Fast model (650MB) -ollama pull qwen3:8b # High-quality model (4.7GB) +# Install the models (in another terminal) +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification -# Verify models are installed ollama list - -# Test Ollama connection curl http://localhost:11434/api/tags ``` -### Step 3: Start Docker Containers +**Or run Ollama in a container.** `docker-compose.yml` defines an `ollama` service +behind the `with-ollama` profile: ```bash -# Start all containers -./start-docker.sh - -# Stop all containers -./start-docker.sh stop +./start-docker.sh container +# equivalently: +# OLLAMA_HOST=http://ollama:11434 \ +# docker compose --env-file docker.env --profile with-ollama up --build -d + +# Models must be pulled inside the container the first time +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` -# View logs -./start-docker.sh logs +The embedding model (`microsoft/harrier-oss-v1-0.6b`) is a HuggingFace download +inside `rag-api`, not an Ollama model — nothing to pull for it. The reranker +(`Qwen/Qwen3-Reranker-4B`) is only downloaded if you switch reranking on, which +is off by default. -# Check status -./start-docker.sh status +### Step 3: Start Containers -# Restart containers +```bash +./start-docker.sh # local Ollama +./start-docker.sh container # containerized Ollama ./start-docker.sh stop -./start-docker.sh +./start-docker.sh logs +./start-docker.sh status +./start-docker.sh help ``` +`./start-docker.sh` with no argument probes port 11434. If nothing is listening it +offers the containerized fallback; pass `-y`/`--yes` (or set `NONINTERACTIVE=1`) to +take it without a prompt, which is what CI should do. With no TTY and no `-y` it +exits 1 with instructions instead of hanging. + ### 1.2 Service Access -Once running, access the system at: - **Frontend**: http://localhost:3000 -- **Backend API**: http://localhost:8000 +- **Backend API**: http://localhost:8000 - **RAG API**: http://localhost:8001 - **Ollama**: http://localhost:11434 +### 1.3 Startup Order + +``` +rag-api (healthy) → backend (healthy) → frontend +``` + +`rag-api` loads the embedding and reranker models before `/health` answers, so its +check has a 60s start period and `backend` intentionally waits in `created` until +then. `docker compose logs -f rag-api` shows the progress. + --- ## 2. Container Management @@ -103,21 +127,13 @@ Once running, access the system at: ### 2.1 Using the Convenience Script ```bash -# Start all containers ./start-docker.sh - -# Stop all containers ./start-docker.sh stop - -# View logs ./start-docker.sh logs - -# Check status ./start-docker.sh status -# Restart containers -./start-docker.sh stop -./start-docker.sh +# Restart +./start-docker.sh stop && ./start-docker.sh ``` ### 2.2 Manual Docker Compose Commands @@ -137,24 +153,22 @@ docker compose down # Force rebuild docker compose build --no-cache -docker compose up --build -d +docker compose --env-file docker.env up --build -d ``` +Always pass `--env-file docker.env` (as `start-docker.sh` does). Without it the +compose defaults still apply — they mirror `docker.env` — but any value you edit in +`docker.env` is ignored. + ### 2.3 Individual Service Management ```bash -# Start specific service docker compose up -d frontend docker compose up -d backend docker compose up -d rag-api -# Restart specific service docker compose restart rag-api - -# Stop specific service docker compose stop backend - -# View specific service logs docker compose logs -f rag-api ``` @@ -164,46 +178,47 @@ docker compose logs -f rag-api ### 3.1 Code Changes -```bash -# After frontend changes -docker compose restart frontend - -# After backend changes -docker compose restart backend +No source is bind-mounted — code is `COPY`-ed into the images at build time, so a +restart alone will not pick up an edit. Rebuild the affected service: -# After RAG system changes -docker compose restart rag-api +```bash +docker compose up -d --build frontend +docker compose up -d --build backend +docker compose up -d --build rag-api -# Rebuild after dependency changes +# After a dependency change docker compose build --no-cache rag-api docker compose up -d rag-api ``` +Editing `NEXT_PUBLIC_API_URL` or `NEXT_PUBLIC_RAG_API_URL` also needs a frontend +**rebuild** — Next.js inlines those at `next build` time and they are passed to +`Dockerfile.frontend` as build args. + ### 3.2 Debugging Containers ```bash -# Access container shell -docker compose exec frontend sh -docker compose exec backend bash -docker compose exec rag-api bash +# Shells +docker compose exec frontend sh # node:20-alpine +docker compose exec backend bash # python:3.11-slim +docker compose exec rag-api bash # python:3.11-slim -# Run commands in container -docker compose exec rag-api python -c "from rag_system.main import get_agent; print('✅ RAG System OK')" -docker compose exec backend curl http://localhost:8000/health +# Run commands in a container +docker compose exec rag-api python -c "from rag_system.factory import get_agent; get_agent('default'); print('✅ RAG System OK')" +docker compose exec backend curl -s http://localhost:8000/health -# Check environment variables -docker compose exec rag-api env | grep OLLAMA +# Environment +docker compose exec rag-api env | grep -E "OLLAMA|MODEL|DB_PATH|LANCEDB" ``` -### 3.3 Development vs Production +### 3.3 Compose File Variants -```bash -# Development mode (if docker-compose.dev.yml exists) -docker compose -f docker-compose.yml -f docker-compose.dev.yml up -d +| File | Contents | +|------|----------| +| `docker-compose.yml` | `rag-api`, `backend`, `frontend`, plus an optional `ollama` service behind the `with-ollama` profile | +| `docker-compose.local-ollama.yml` | The same three application services, no optional `ollama` service | -# Production mode (default) -docker compose --env-file docker.env up -d -``` +There is no `docker-compose.dev.yml`. --- @@ -212,131 +227,123 @@ docker compose --env-file docker.env up -d ### 4.1 Log Management ```bash -# View all logs docker compose logs - -# View specific service logs docker compose logs frontend docker compose logs backend docker compose logs rag-api -# Follow logs in real-time docker compose logs -f - -# View last N lines docker compose logs --tail=100 - -# View logs with timestamps docker compose logs -t - -# Save logs to file docker compose logs > system.log 2>&1 - -# View logs since specific time docker compose logs --since=2h -docker compose logs --since=2025-01-01T00:00:00 ``` ### 4.2 System Monitoring ```bash -# Monitor resource usage docker stats - -# Monitor specific containers docker stats rag-frontend rag-backend rag-api -# Check container health docker compose ps +docker inspect rag-api --format='{{.State.Health.Status}}' -# System information docker system info docker system df ``` +Container names are fixed by `container_name`: `rag-frontend`, `rag-backend`, +`rag-api`, and `rag-ollama` for the optional Ollama service. + --- ## 5. Ollama Integration -### 5.1 Ollama Setup +### 5.1 Host Ollama ```bash -# Install Ollama (one-time setup) curl -fsSL https://ollama.ai/install.sh | sh - -# Start Ollama server ollama serve - -# Check Ollama status curl http://localhost:11434/api/tags -# Install models -ollama pull qwen3:0.6b # Fast model -ollama pull qwen3:8b # High-quality model - -# List installed models +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ollama list ``` -### 5.2 Ollama Management +### 5.2 From Inside a Container ```bash -# Check model status from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags -# Test Ollama connection curl -X POST http://localhost:11434/api/generate \ -H "Content-Type: application/json" \ - -d '{"model": "qwen3:0.6b", "prompt": "Hello", "stream": false}' - -# Monitor Ollama logs (if running with logs) -# Ollama logs appear in the terminal where you ran 'ollama serve' + -d '{"model": "qwen3.5:4b", "prompt": "Hello", "stream": false}' ``` +Ollama logs appear in the terminal running `ollama serve`; for the containerized +variant use `docker compose --profile with-ollama logs -f ollama`. + ### 5.3 Model Management ```bash -# Update models -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b +ollama pull qwen3.6:27b # optional high-end generation model -# Remove unused models ollama rm old-model-name +ollama show qwen3.5:9b +``` + +Point the containers at a different model without editing code: -# Check model information -ollama show qwen3:0.6b +```bash +GENERATION_MODEL=qwen3.6:27b docker compose --env-file docker.env up -d rag-api backend ``` --- ## 6. Data Management -### 6.1 Volume Management +### 6.1 Volumes and Mounts + +Every application path is a **bind mount to a host directory**, so ordinary file +tools work. The only named volume is `ollama_data`, used by the optional Ollama +container. + +| Host path | Container path | Contents | +|-----------|----------------|----------| +| `./lancedb` | `/app/lancedb` | Vectors and the native full-text index | +| `./index_store` | `/app/index_store` | Document overviews | +| `./shared_uploads` | `/app/shared_uploads` | Uploaded source documents | +| `./backend` | `/app/backend` | `chat_data.db` (shared by backend and rag-api) | ```bash -# List volumes docker volume ls - -# View volume usage docker system df -v -# Backup volumes -docker run --rm -v rag_system_old_lancedb:/data -v $(pwd)/backup:/backup alpine tar czf /backup/lancedb_backup.tar.gz -C /data . +# Back up the host directories directly +tar czf backup/lancedb_backup.tar.gz lancedb/ +tar czf backup/index_store_backup.tar.gz index_store/ + +# Only the containerized Ollama uses a named volume. +# The compose project name is the directory name, so it is localgpt_ollama_data. +docker run --rm -v localgpt_ollama_data:/data -v $(pwd)/backup:/backup \ + alpine tar czf /backup/ollama_models.tar.gz -C /data . -# Clean unused volumes docker volume prune ``` ### 6.2 Database Management ```bash -# Access SQLite database -docker compose exec backend sqlite3 /app/backend/chat_data.db +# sqlite3 is installed in both Python images +docker compose exec backend sqlite3 /app/backend/chat_data.db ".tables" -# Backup database +# Back up the database cp backend/chat_data.db backup/chat_data_$(date +%Y%m%d).db -# Check LanceDB tables from container +# Check LanceDB tables from the container docker compose exec rag-api python -c " import lancedb db = lancedb.connect('/app/lancedb') @@ -344,20 +351,23 @@ print('Tables:', db.table_names()) " ``` +`backend` and `rag-api` both set `DB_PATH=/app/backend/chat_data.db` and mount the +same host directory, so they read and write one file. + ### 6.3 File Management ```bash -# Access shared files docker compose exec rag-api ls -la /app/shared_uploads -# Copy files to/from containers docker cp local_file.pdf rag-api:/app/shared_uploads/ docker cp rag-api:/app/shared_uploads/file.pdf ./local_file.pdf -# Check disk usage docker compose exec rag-api df -h ``` +Because `shared_uploads/` is a bind mount, copying a file into `./shared_uploads` +on the host is equivalent and simpler. + --- ## 7. Troubleshooting @@ -366,43 +376,41 @@ docker compose exec rag-api df -h #### Container Won't Start ```bash -# Check Docker daemon docker version - -# Check for port conflicts lsof -i :3000 -i :8000 -i :8001 - -# Check container logs docker compose logs [service-name] - -# Restart Docker Desktop -# macOS/Windows: Restart Docker Desktop -# Linux: sudo systemctl restart docker ``` +#### `backend` stays in `created` +That is `depends_on: rag-api: condition: service_healthy` doing its job. Watch +`docker compose logs -f rag-api` — on a cold start it is downloading and loading +the embedding and reranker models. + #### Ollama Connection Issues ```bash -# Check Ollama is running curl http://localhost:11434/api/tags -# Restart Ollama pkill ollama ollama serve -# Check from container -docker compose exec rag-api curl http://host.docker.internal:11434/api/tags +docker compose exec rag-api curl -s http://host.docker.internal:11434/api/tags +``` + +#### Chats return "Could not connect to the RAG API server" +The backend builds its URLs from `RAG_API_URL`, which must be +`http://rag-api:8001` inside compose (`localhost` there means the backend +container itself). +```bash +docker compose exec backend env | grep RAG_API_URL +docker compose exec backend curl -s http://rag-api:8001/health ``` #### Performance Issues ```bash -# Check resource usage docker stats - -# Increase Docker memory (Docker Desktop Settings) -# Recommended: 8GB+ for Docker - -# Check container health docker compose ps + +# Docker Desktop → Settings → Resources → Memory → 8GB+ ``` ### 7.2 Reset and Clean @@ -414,34 +422,37 @@ docker compose ps # Clean containers and images docker system prune -a -# Clean volumes (⚠️ deletes data) -docker volume prune - -# Complete reset (⚠️ deletes everything) -docker compose down -v -docker system prune -a --volumes +# Complete reset (⚠️ deletes indexes, uploads and chat history) +docker compose down +rm -rf lancedb/* index_store/* shared_uploads/* backend/chat_data.db +docker system prune -a ``` +`docker compose down -v` only removes the named `ollama_data` volume — application +data lives in host directories and must be deleted explicitly. + ### 7.3 Health Checks ```bash -# Comprehensive health check curl -f http://localhost:3000 && echo "✅ Frontend OK" curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" +curl -f http://localhost:8001/health && echo "✅ RAG API OK" curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" -# Check all container status docker compose ps -# Test model loading +# Test model loading inside the container docker compose exec rag-api python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') print('✅ RAG System initialized successfully') " ``` +These are the same endpoints the container health checks use: `curl -f /health` +for `backend` and `rag-api`, and busybox `wget -qO- http://localhost:3000` for +`frontend` (the alpine image has no curl). + --- ## 8. Advanced Usage @@ -449,36 +460,37 @@ print('✅ RAG System initialized successfully') ### 8.1 Production Deployment ```bash -# Use production environment -export NODE_ENV=production - -# Start with resource limits -docker compose --env-file docker.env up -d +# docker.env already sets NODE_ENV=production +docker compose --env-file docker.env up --build -d -# Enable automatic restarts -docker update --restart unless-stopped $(docker ps -q) +# All services already declare restart: unless-stopped +docker compose ps ``` +There is no authentication and CORS is wide open on both APIs. Put a reverse proxy +in front and publish only what you need. Note the browser streams directly from +port 8001, so that port must be reachable by clients (or proxied) unless you turn +off "Stream phases" in the chat UI. + ### 8.2 Scaling -```bash -# Scale specific services -docker compose up -d --scale backend=2 --scale rag-api=2 +`docker compose up -d --scale backend=2 --scale rag-api=2` **does not work with the +shipped compose file**: each service sets a fixed `container_name` +(`rag-backend`, `rag-api`) and publishes a fixed host port, both of which conflict +on the second replica. Remove `container_name` and the `ports:` mappings (or switch +to a random host port) first. -# Use Docker Swarm for clustering -docker swarm init -docker stack deploy -c docker-compose.yml rag-system -``` +Even then, the RAG API is single-threaded and holds per-process state — the +resident agent, its in-memory chat history and the semantic cache — so replicas +need sticky sessions at minimum. ### 8.3 Security ```bash -# Scan images for vulnerabilities docker scout cves rag-frontend docker scout cves rag-backend docker scout cves rag-api -# Update base images docker compose build --no-cache --pull ``` @@ -488,56 +500,74 @@ docker compose build --no-cache --pull ### 9.1 Environment Variables -The system uses `docker.env` for configuration: +`docker.env` is passed with `--env-file` and supplies both runtime environment and +compose-level substitution: ```bash -# Ollama configuration +# Ollama on the host; extra_hosts makes this resolve on Linux too OLLAMA_HOST=http://host.docker.internal:11434 +# Containerized alternative: OLLAMA_HOST=http://ollama:11434 -# Service configuration NODE_ENV=production RAG_API_URL=http://rag-api:8001 + +# Browser-facing; inlined into the frontend bundle at build time NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 + +# Shared SQLite + vector store +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb + +# Models +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B ``` +Values already exported in your shell win over `--env-file` — that is how +`start-docker.sh container` overrides `OLLAMA_HOST`. + +Changing `EMBEDDING_MODEL` invalidates existing indexes: every table records the +embedding model that wrote it (and its vector width), and writing or querying it +with a different model fails with an explicit error. Rebuild your indexes after +switching. + ### 9.2 Custom Configuration ```bash -# Create custom environment file cp docker.env docker.custom.env - -# Edit custom configuration nano docker.custom.env - -# Use custom configuration docker compose --env-file docker.custom.env up -d + +# Remember to rebuild if you changed a NEXT_PUBLIC_* value +docker compose --env-file docker.custom.env up -d --build frontend ``` --- ## 10. Success Checklist -Your Docker deployment is successful when: - -- ✅ All containers are running: `docker compose ps` -- ✅ Ollama is accessible: `curl http://localhost:11434/api/tags` +- ✅ All containers healthy: `docker compose ps` +- ✅ Ollama reachable: `curl http://localhost:11434/api/tags` - ✅ Frontend loads: `curl http://localhost:3000` - ✅ Backend responds: `curl http://localhost:8000/health` -- ✅ RAG API works: `curl http://localhost:8001/models` -- ✅ You can create indexes and chat with documents - -### Performance Expectations +- ✅ RAG API responds: `curl http://localhost:8001/health` +- ✅ You can create an index and chat with your documents -**Acceptable Performance:** -- Container startup: < 2 minutes -- Memory usage: < 4GB Docker containers + Ollama -- Response time: < 30 seconds for complex queries +### What to Expect -**Optimal Performance:** -- Container startup: < 1 minute -- Memory usage: < 2GB Docker containers + Ollama -- Response time: < 10 seconds for complex queries +- **First `up --build`** is slow: it installs the Python dependencies (torch, + transformers, docling) and builds the Next.js bundle, and `rag-api` then + downloads ~10GB of HuggingFace weights before it reports healthy +- **Restarting** an existing container is fast. The HuggingFace weights are + downloaded at runtime into the container's writable layer, and no volume is + mounted for them, so **recreating** `rag-api` (any `up --build`, `down` + `up`, or + image change) downloads them again. Mount a cache directory and set `HF_HOME` if + that matters to you. +- **One RAG request at a time** — the RAG API is single-threaded --- -**Happy Containerizing! 🐳** \ No newline at end of file +**Happy Containerizing! 🐳** diff --git a/Documentation/improvement_plan.md b/Documentation/improvement_plan.md index 1c84c5df..f36c3024 100644 --- a/Documentation/improvement_plan.md +++ b/Documentation/improvement_plan.md @@ -1,87 +1,126 @@ -# RAG System – Improvement Road-map +# localGPT — Improvement Road-map -_Revision: 2025-07-05_ +_Revision: 2026-08-09_ -This document captures high-impact enhancements identified during the July 2025 code-review. Items are grouped by theme and include a short rationale plus suggested implementation notes. **No code has been changed – this file is planning only.** +Planned work only. Nothing in the **Open** sections is implemented. The **Landed** section records changes that were verified in the current working tree at the revision date — every entry there names the file that proves it, so the list can be re-checked rather than trusted. + +> An evidence-based, phased extension of this plan lives in +> [research_roadmap.md](research_roadmap.md), grounded in the August 2026 +> research sweeps under [research/](research/). Items graduate from there into +> this file's Landed table as they ship. + +--- + +## 0. Landed (verified in-tree, 2026-08-08; Phase 1 rows added 2026-08-09) + +| Area | Change | Verify at | +|------|--------|-----------| +| Architecture | One RAG API server; the parallel `api_server_with_progress.py` is gone | `rag_system/` has a single `api_server.py` | +| Architecture | `factory.py` is the only agent/pipeline factory; `main.py` is config + a thin `index` / `chat` / `api` CLI | `rag_system/factory.py`, `rag_system/main.py` | +| Architecture | Backend gateway is threaded and every RAG API call has a timeout (`RAG_API_TIMEOUT`, `RAG_API_INDEX_TIMEOUT`) | `backend/server.py` | +| Ops | RAG API exposes `GET /health`; `run_system.py --health`, `--stop`, `--logs-only` and `--mode prod` all work | `rag_system/api_server.py`, `run_system.py` | +| Config | Service URLs and model ids come from environment variables, documented in `.env.example` | `.env.example`, `rag_system/main.py`, `backend/server.py`, `src/lib/api.ts` | +| Retrieval | Hybrid search fuses the full-text and vector legs with reciprocal rank fusion instead of a broken weighted blend; `retrieval_mode` (`hybrid` / `vector_only` / `fts_only`) is honoured | `rag_system/retrieval/retrievers.py` | +| Retrieval | **1.1 Late-chunk result merging** — retrieved late-chunks are merged with their ±1 siblings before reranking | `rag_system/pipelines/retrieval_pipeline.py` | +| Retrieval | A reranker that fails to load logs a warning and is skipped instead of throwing | `rag_system/pipelines/retrieval_pipeline.py::_get_ai_reranker` | +| Indexing | **3.3 Auto GPU dtype selection** — CUDA > MPS > CPU with fp16 off CPU, for both the embedder and the late-chunk encoder | `rag_system/indexing/representations.py`, `rag_system/indexing/latechunk.py` | +| Indexing | Vector width is derived from the loaded model, and appending mismatched vectors to an existing table raises with a "rebuild the index" message (part of **3.4**) | `rag_system/indexing/embedders.py::VectorIndexer` | +| Indexing | OCR engine is chosen by probing which backend is installed, so a missing macOS-only engine no longer breaks every conversion | `rag_system/ingestion/document_converter.py` | +| Storage | The LanceDB path is resolved from `LANCEDB_PATH` / the pipeline config instead of three hard-coded literals, so index deletion targets the store indexing writes to | `backend/database.py::resolve_lancedb_path` | +| Privacy | The semantic cache defaults to `cache_scope: "session"`, closing the cross-session answer leak | `rag_system/main.py`, `rag_system/agent/loop.py` | +| Hygiene | DSPy modules and the non-functional ReAct agent are gone from the code; `rank_bm25`, `scikit-learn`, `nltk`, `sentence_transformers`, `colpali-engine` and `matplotlib` were dropped from `requirements.txt` | `grep` returns no code hits for any of them | +| Retrieval | **Phase 1.2 embedder** (`research_roadmap.md` §1.2) — default embedder is `microsoft/harrier-oss-v1-0.6b` (MIT, 1024-dim) with the query-side `Instruct: … Query: …` prefix on for instruction-tuned families; documents stay unprefixed. Measured mixed-corpus first-stage nDCG@10 0.915 vs 0.875 for the previous `Qwen3-Embedding-4B` default, at ~3× lower latency (0.911 for the shipped stack once vectors are normalized — see the row below). `EMBEDDING_MODEL` still overrides | `rag_system/main.py::EXTERNAL_MODELS`, `rag_system/indexing/representations.py::default_query_instruction`, `rag_system/pipelines/retrieval_pipeline.py::_query_instruction`, `eval/DECISIONS.md` | +| Retrieval | **Phase 1.1 reranker** (`research_roadmap.md` §1.1) — the `default` profile ships `reranker.enabled = False`: with this first stage the cheap cross-encoder measures net-negative (0.915 → 0.892 mixed) and the reranker that wins costs ~12.7 s/query. The toggle (UI "AI reranker" / `reranker.enabled`) now loads `Qwen/Qwen3-Reranker-4B` lazily through the in-repo `QwenRerankerScorer` | `rag_system/main.py::PIPELINE_CONFIGS`, `rag_system/pipelines/retrieval_pipeline.py::_get_ai_reranker`, `src/components/ui/session-chat.tsx`, `eval/DECISIONS.md` | +| Indexing | **Embedder-identity guard** — the width-only check cannot catch a swap between two 1024-dim models, so every table records the embedding model that wrote it (Arrow schema metadata, sidecar fallback). Indexing into or querying a table with a different embedder raises `EmbedderMismatchError` instead of returning nonsense | `rag_system/indexing/embedders.py::read_table_marker`/`assert_embedder_matches`, `rag_system/retrieval/retrievers.py::MultiVectorRetriever._check_table_identity` | +| Indexing | **Cosine normalization** — vectors are L2-normalized at write and query time, so LanceDB's default L2 ordering is the cosine ordering both model cards specify. Gated per table on the `normalized` marker, so legacy tables keep working unnormalized with a warning; the default table moved to `text_pages_v4`. Measured a wash on the gold set (mixed nDCG@10 0.915 → 0.911, recall@5 0.917 → 0.931): adopted for card-conformance, not for a number | `rag_system/indexing/embedders.py::l2_normalize`, `rag_system/retrieval/retrievers.py`, `rag_system/main.py` | +| Evaluation | **Phase 0 harness** (`research_roadmap.md` §0) — three corpora (two planted-fact PDFs + `Documentation/*.md`), a 72-query gold set labelled by answer-bearing text rather than chunk id, an in-process recall@5/10/20 + nDCG@10 runner, a binary groundedness judge validated at TPR 1.00 / TNR 1.00 on 20 hand-labelled cases, and a scripted end-to-end smoke that passes 25/25 assertions. Baseline recorded; every Phase 1/2 item now has something to A/B against | `eval/run_eval.py`, `eval/judge.py`, `eval/smoke_e2e.py`, `eval/goldset/*.jsonl`, `eval/BASELINE.md` | +| Models | **Roadmap 1.1/1.2 — evidence-gated defaults**: embedder `microsoft/harrier-oss-v1-0.6b` + query instruction prefix (mixed nDCG@10 0.915 vs 0.875 for the 8 GB Qwen3-4B); default profile reranker **off** (bge measured net-negative on this first stage; Qwen3-Reranker-4B is the lazy opt-in at +0.06 nDCG for ~12.7 s/query); per-table embedder-identity markers + L2 normalization, `text_pages_v4` | `rag_system/main.py`, `rag_system/indexing/embedders.py`, `eval/DECISIONS.md` | +| Routing | **Roadmap 2.3 — gateway routing is a deterministic gate.** The per-message enrichment-model router and the `_simple_pattern_routing` keyword/length fallback are deleted; `should_use_rag()` routes on `force_rag` → linked indexes → a whole-message smalltalk/assistant-meta allowlist → RAG. ~750 ms/message saved; agent triage is now the only LLM routing layer | `backend/server.py::should_use_rag`, `backend/test_gateway_routing.py` (155/155), `eval/decisions/phase2-gateway.md` | +| Retrieval | **Roadmap 2.5 — graph module removed**: `GraphExtractor`, `GraphRetriever`, `GraphQueryTranslator`, the `graph_query` triage outcome and the `retrieval.graph` / `graph_strategy` config keys are gone; `networkx`, `fuzzywuzzy`, `python-Levenshtein` dropped from all requirements files. Contested gains, 41–57× indexing and up to ~377× query-token cost (`research/academic-evidence-2026.md` §6) | `rag_system/indexing/` has no `graph_extractor.py`; `eval/decisions/phase2-pipeline.md` §1 | +| Retrieval | **Roadmap 2.1 — evidence-sufficiency retry**: one conditional second retrieval on weak evidence (candidate-set contrast signal — raw top similarity measured anti-correlated), on in `default`, off in `fast`. Fires on ~10% of queries, +0.008–0.017 nDCG@10, zero per-query regressions in four runs | `rag_system/pipelines/retrieval_pipeline.py::retrieve_candidates`, `eval/decisions/phase2-pipeline.md` §2 | +| Retrieval | **Roadmap 2.2 — decomposition at rerank**: the first stage always runs once on the full query; sub-queries score candidates at rerank (`query_decomposition.rerank_aggregate`), and first-stage fan-out survives only behind `compose_from_sub_answers`. Measured negative on truly-decomposing queries — no shipped profile enables rerank-decomposition | `rag_system/pipelines/retrieval_pipeline.py`, `eval/decisions/phase2-pipeline.md` §3 | +| Verification | **Roadmap 2.4 — verifier model seam**: `VERIFIER_MODEL` / `verification.model` swaps the LLM-prompt verifier for a local NLI model (MiniCheck-DeBERTa 19/20, DeBERTa-MNLI 18/20 on the judge set); default unchanged. ThinknCheck has no public weights | `rag_system/agent/verifier.py::LocalNLIVerifier`, `eval/decisions/phase2-pipeline.md` §4 | +| Eval | `run_eval.py` drives `RetrievalPipeline.retrieve_candidates()` — first stage, retry and rerank are the shipped code path; `--retry`, `--decompose`, `--aggregate` flags added; gold rows `docs_d10` + the triage-model row re-anchored after 2.5/2.3 deleted their source text (recorded per-row) | `eval/run_eval.py`, `eval/goldset/docs.jsonl` | + +The previous revision of this file marked five cleanup items ✅ COMPLETED. Two of those claims (removing unused imports/dependencies, consolidating configuration files) were not true when written and are only partly true now — the surviving work is tracked in §9 below. --- -## 1. Retrieval Accuracy & Speed +## 1. Retrieval accuracy & speed | ID | Item | Rationale | Notes | |----|------|-----------|-------| -| 1.1 | Late-chunk result merging | Returned snippets can be single late-chunks → fragmented. | After retrieval, gather sibling chunks (±1) and concatenate before reranking / display. | -| 1.2 | Tiered retrieval (ANN pre-filter) | Large indexes → LanceDB full scan can be slow. | Use in-memory FAISS/HNSW to narrow to top-N, then exact LanceDB search. | -| 1.3 | Dynamic fusion weights | Different corpora favour dense vs BM25 differently. | Learn weight on small validation set; store in index `metadata`. | -| 1.4 | Query expansion via KG | Use extracted entities to enrich queries. | Requires Graph-RAG path clean-up first. | +| 1.2 | Tiered retrieval (ANN pre-filter) | Large tables make LanceDB scans slow. | Narrow to top-N with an in-memory index, then exact search. | +| 1.3 | Corpus-tuned fusion | RRF is weight-free and safe, but a tuned fusion could beat it per corpus. | Would need a validation set and a place to store the setting per index. | +| 1.4 | Query expansion via extracted entities | Richer queries for entity-heavy corpora. | Depends on the Graph-RAG path (§9) being finished or removed. | +| 1.5 | Deduplicate the late-chunk leg | Late-chunk hits are appended to the base hits, so the same passage can occupy two slots before reranking. | Dedupe on `chunk_id` across both legs, or rank the legs jointly. | -## 2. Routing / Triage +## 2. Routing / triage | ID | Item | Rationale | |----|------|-----------| -| 2.1 | Embed + cache document overviews | LLM router costs tokens; cosine-similarity pre-check is cheaper. | -| 2.2 | Session-level routing memo | Avoid repeated LLM triage for follow-up queries. | -| 2.3 | Remove legacy pattern rules | Simplifies maintenance once overview & ML routing mature. | +| 2.1 | Embed and cache document overviews | The agent router (now the only LLM routing layer) makes an LLM call per query; a cosine pre-check would be far cheaper. | +| 2.2 | Session-level routing memo | Today the only shortcut is "history exists → `rag_query`". Cache the decision instead. | -## 3. Indexing Pipeline +## 3. Indexing pipeline | ID | Item | Rationale | |----|------|-----------| -| 3.1 | Parallel document conversion | PDF→MD + chunking is serial today; speed gains possible. | -| 3.2 | Incremental indexing | Re-embedding whole corpus wastes time. | -| 3.3 | Auto GPU dtype selection | Use FP16 on CUDA / MPS for memory and speed. | -| 3.4 | Post-build health check | Catch broken indexes (dim mismatch etc.) early. | +| 3.1 | Parallel document conversion | Conversion and chunking are serial per file. | +| 3.2 | Incremental indexing | Re-embedding the whole corpus to add one document is wasteful. | +| 3.4 | Post-build health check | The dimension guard exists; a build should also assert the FTS index, the row count and the late-chunk table when requested. | +| 3.5 | Align chunk-size defaults | No profile sets `chunking.chunk_size`, so CLI indexing chunks at 1500 tokens while `POST /index` defaults to 512 (`api_server.py:414`) — same class of split as 3.6. Documented in `system_overview.md` §5.3. | +| 3.6 | Align late-chunk defaults | `POST /index` defaults `enable_latechunk` to `false` while the `default` profile enables late-chunk retrieval, so HTTP builds and CLI builds silently differ (retrieval degrades gracefully when the `_lc` table is absent). | -## 4. Embedding Model Management +## 4. Model management -* **Registry file** mapping tag → dims/source/license. UI & backend validate against it. -* **Embedder pool** caches loaded HF/Ollama weights per model to save RAM. +* Registry mapping model tag → dimensions, source and license, validated by the UI and both servers. +* An embedder pool that keeps one copy of each loaded model in memory (a module-level cache exists in `representations.py`; the reranker and pruner have their own ad-hoc singletons). +* Warn — or refuse — when the embedding model configured for a query differs from the one recorded in the index metadata. -## 5. Database & Storage +## 5. Database & storage -* LanceDB table GC for orphaned tables. -* Scheduled SQLite `VACUUM` when fragmentation > X %. +* Garbage-collect orphaned LanceDB tables (an index row deleted outside the API leaves its table behind). +* Delete files from `shared_uploads/` when their session or index is deleted. +* Scheduled SQLite `VACUUM` when fragmentation is high. -## 6. Observability & Ops +## 6. Observability & ops -* JSON structured logging. -* `/metrics` endpoint for Prometheus. -* Deep health-probe (`/health/deep`) exercising end-to-end query. +* JSON structured logging and log rotation in `run_system.setup_logging` (today: plain text, no rotation). +* Move the agent's and pipelines' `print()` progress output onto the `logging` module. +* A `/metrics` endpoint for Prometheus. +* A deep health probe (`/health/deep`) that runs a real end-to-end query. +* Per-request pipeline configuration on the RAG API, so options stop leaking between requests, followed by switching the RAG API to a threading server. ## 7. Front-end UX -* SSE-driven progress bar for indexing. +* SSE-driven progress for indexing (chat already streams phases; indexing is a blocking POST). * Matched-term highlighting in retrieved snippets. -* Preset buttons (Fast / Balanced / High-Recall) for retrieval settings. +* Preset buttons (Fast / Balanced / High-Recall) over the retrieval settings. +* Surface the verifier's confidence as a field rather than parsing it out of the answer string. ## 8. Testing & CI -* Replace deleted BM25 tests with LanceDB hybrid tests. -* Integration test: build → query → assert ≥1 doc. -* GitHub Action that spins up Ollama, pulls small embedding model, runs smoke test. +There is no automated test suite. The only checks are `python system_health_check.py`, `python run_system.py --health` and `./test_docker_build.sh`. -## 9. Codebase Hygiene +* Unit tests for `MultiVectorRetriever.retrieve` across all three modes, including the RRF ordering. +* Integration test: build an index → query → assert at least one source document. +* A GitHub Action that starts Ollama, pulls a small embedding model and runs the smoke test. +* Re-enable the type/lint gates that `next.config.ts` currently disables (`eslint.ignoreDuringBuilds`, `typescript.ignoreBuildErrors`). -* Graph-RAG integration (currently disabled, can be implemented if needed). -* Consolidate duplicate config keys (`embedding_model_name`, etc.). -* Run `mypy --strict`, pylint, and black in CI. +## 9. Codebase hygiene ---- - -### 🧹 System Cleanup (Priority: **HIGH**) -Reduce complexity and improve maintainability. +* `docker-compose.local-ollama.yml` now duplicates `docker-compose.yml` (which already defaults to host Ollama). Pick one. +* Run `mypy`, `pylint` and `black` in CI. -* **✅ COMPLETED**: Remove experimental DSPy integration and unused modules (35+ files removed) -* **✅ COMPLETED**: Clean up duplicate or obsolete documentation files -* **✅ COMPLETED**: Remove unused import statements and dependencies -* **✅ COMPLETED**: Consolidate similar configuration files -* **✅ COMPLETED**: Remove broken or non-functional ReAct agent implementation +--- -### Priority Matrix (suggested order) +### Priority matrix (suggested order) -1. **Critical reliability**: 3.4, 5.1, 9.2 -2. **User-visible wins**: 1.1, 7.1, 7.2 -3. **Performance**: 1.2, 3.1, 3.3 -4. **Long-term maintainability**: 2.3, 9.1, 9.3 +1. **Correctness / data loss**: 3.6 +2. **User-visible wins**: 7.1, 7.2, 2.4 +3. **Reliability**: 3.4, 5.1, 8.2 +4. **Performance**: 1.2, 1.5, 3.1 +5. **Long-term maintainability**: 2.1, 4, 9 -Feel free to rearrange based on team objectives and resource availability. \ No newline at end of file +Rearrange to suit team objectives and available time. diff --git a/Documentation/indexing_pipeline.md b/Documentation/indexing_pipeline.md index 009ee497..e4c8fc76 100644 --- a/Documentation/indexing_pipeline.md +++ b/Documentation/indexing_pipeline.md @@ -1,665 +1,299 @@ # 🗂️ Indexing Pipeline -_Implementation entry-point: `rag_system/pipelines/indexing_pipeline.py` + helpers in `indexing/` & `ingestion/`._ +_Implementation entry-point: `rag_system/pipelines/indexing_pipeline.py`, with helpers in `rag_system/ingestion/` and `rag_system/indexing/`._ ## Overview -Transforms raw documents (PDF, TXT, etc.) into search-ready **chunks** with embeddings, storing them in LanceDB and generating auxiliary assets (overviews, context summaries). +Turns documents (PDF, DOCX, HTML/HTM, MD, TXT) into search-ready chunks with embeddings, writes them to LanceDB, builds a full-text index over the same table, and generates a per-document overview used by the triage routers. + +Public entry point: + +```python +IndexingPipeline.run(file_paths: list[str]) -> None +``` + +`run()` also accepts a legacy `documents=` keyword as an alias for `file_paths` (`indexing_pipeline.py:157-167`). It returns `None`; progress and results are printed. + +## High-level diagram -## High-Level Diagram ```mermaid flowchart TD - A["Uploaded Files"] --> B{Converter} - B -->|PDF→text| C["Plain Text"] - C --> D{Chunker} - D -->|docling| D1[DocLing Chunking] - D -->|latechunk| D2[Late Chunking] - D -->|standard| D3[Fixed-size] - D1 & D2 & D3 --> E["Contextual Enricher"] - E -->|local ctx summary| F["Embedding Generator"] - F -->|vectors| G[(LanceDB Table)] - E --> H["Overview Builder"] - H -->|JSONL| OVR[[`index_store/overviews/.jsonl`]] + A["Files"] --> B["DocumentConverter (docling)"] + B --> C["Markdown + DoclingDocument"] + C --> D{"chunker_mode"} + D -- docling --> D1["DoclingChunker.chunk_document()"] + D -- legacy --> D2["MarkdownRecursiveChunker.chunk()"] + D1 --> OV["OverviewBuilder (per document)"] + D2 --> OV + OV --> OVF[["index_store/overviews/<id>.jsonl"]] + D1 --> E["ContextualEnricher (optional)"] + D2 --> E + E --> F["EmbeddingGenerator"] + F --> G[("LanceDB table")] + G --> H["create_fts_index('text')"] + F --> LC["LateChunkEncoder (optional)"] + LC --> GLC[("LanceDB table <table>_lc")] ``` -## Steps in Detail -| Step | Module | Key Classes | Notes | -|------|--------|------------|-------| -| Conversion | `ingestion/pdf_converter.py` | `PDFConverter` | Uses `Docling` library to extract text with structure preservation. | -| Chunking | `ingestion/chunking.py`, `indexing/latechunk.py`, `ingestion/docling_chunker.py` | `MarkdownRecursiveChunker`, `DoclingChunker` | Controlled by flags `latechunk`, `doclingChunk`, `chunkSize`, `chunkOverlap`. | -| Contextual Enrichment | `indexing/contextualizer.py` | `ContextualEnricher` | Generates per-chunk summaries (LLM call). | -| Embedding | `indexing/embedders.py`, `indexing/representations.py` | `QwenEmbedder`, `EmbeddingGenerator` | Batch size tunable (`batchSizeEmbed`). Uses Qwen3-Embedding models. | -| LanceDB Ingest | `index_store/lancedb/…` | – | Each index has a dedicated table `text_pages_`. | -| Overview | `indexing/overview_builder.py` | `OverviewBuilder` | First-N chunks summarised for triage routing. | - -### Control Flow (Code) -1. **backend/server.py → handle_build_index()** collects files + opts and POSTs to `/index` endpoint on advanced RAG API (local process). -2. **indexing_pipeline.IndexingPipeline.run()** orchestrates conversion → chunking → enrichment → embedding → storage. -3. Metadata (chunk_size, models, etc.) stored in SQLite `indexes` table. - -## Configuration Flags -| Flag | Description | Default | -|------|-------------|---------| -| `latechunk` | Merge k adjacent sibling chunks at query time | false | -| `doclingChunk` | Use DocLing structural chunking | false | -| `chunkSize` / `chunkOverlap` | Standard fixed slicing | 512 / 64 | -| `enableEnrich` | Run contextual summaries | true | -| `embeddingModel` | Override embedder | `Qwen/Qwen3-Embedding-0.6B` | -| `overviewModel` | Model used in `OverviewBuilder` | `qwen3:0.6b` | -| `batchSizeEmbed / Enrich` | Batch sizes | 50 / 25 | - -## Error Handling -* Duplicate LanceDB table ➟ now idempotent (commit `af99b38`). -* Failed PDF parse ➟ chunker skips file, logs warning. - -## Extension Ideas -* Add OCR layer before PDF conversion. -* Store embeddings in Remote LanceDB instance (update URL in config). - -## Detailed Implementation Analysis - -### Pipeline Architecture Pattern -The `IndexingPipeline` uses a **sequential processing pattern** with parallel batch operations. Each stage processes all documents before moving to the next stage, enabling efficient memory usage and progress tracking. +## Steps in detail -```python -def run(self, file_paths: List[str]): - with timer("Complete Indexing Pipeline"): - # Stage 1: Document Processing & Chunking - all_chunks = [] - doc_chunks_map = {} - - # Stage 2: Contextual Enrichment (optional) - if self.contextual_enricher: - all_chunks = self.contextual_enricher.enrich_batch(all_chunks) - - # Stage 3: Dense Indexing (embedding + storage) - if self.vector_indexer: - self.vector_indexer.index_chunks(all_chunks, table_name) - - # Stage 4: Graph Extraction (optional) - if self.graph_extractor: - self.graph_extractor.extract_and_store(all_chunks) -``` +| Step | Module | Key classes | Notes | +|------|--------|-------------|-------| +| Conversion | `ingestion/document_converter.py` | `DocumentConverter` | docling for PDF/DOCX/HTML/MD; plain read for `.txt`. PyMuPDF (`fitz`) is used **only** to test whether a PDF already has a text layer. | +| Chunking | `ingestion/docling_chunker.py`, `ingestion/chunking.py` | `DoclingChunker`, `MarkdownRecursiveChunker` | Docling is the default. Both are token-aware via the embedding model's tokenizer. | +| Overview | `indexing/overview_builder.py` | `OverviewBuilder` | One LLM call per document, from its first N chunks. On by default. | +| Contextual enrichment | `indexing/contextualizer.py` | `ContextualEnricher` | One LLM call per chunk. On in the `default` profile. | +| Embedding | `indexing/representations.py` | `QwenEmbedder`, `OllamaEmbedder`, `EmbeddingGenerator` | `select_embedder()` picks HuggingFace vs Ollama. | +| LanceDB write | `indexing/embedders.py` | `LanceDBManager`, `VectorIndexer` | Appends to an existing table, creates it otherwise. Vector width is taken from the produced embeddings. | +| Full-text index | `pipelines/indexing_pipeline.py:264-284` | – | `tbl.create_fts_index("text", use_tantivy=False)`. This is what makes `hybrid` retrieval work. | +| Late chunking | `indexing/latechunk.py` | `LateChunkEncoder` | Optional second embedding pass into `
_lc`. | -### Document Processing Deep-Dive +Storage location is `storage.lancedb_uri` (also accepted as `storage.db_path` / `storage.lancedb_path`), which both profiles set to `./lancedb`. Per-index tables are named `text_pages_` (`backend/database.py:351`); the profile fallback table is `text_pages_v4`. -#### PDF Conversion Strategy -```python -# PDFConverter uses Docling for robust text extraction with structure -def convert_to_markdown(self, file_path: str) -> List[Tuple[str, Dict, Any]]: - # Quick heuristic: if PDF has text layer, skip OCR for speed - use_ocr = not self._pdf_has_text(file_path) - converter = self.converter_ocr if use_ocr else self.converter_no_ocr - - result = converter.convert(file_path) - markdown_content = result.document.export_to_markdown() - - metadata = {"source": file_path} - # Return DoclingDocument object for advanced chunkers - return [(markdown_content, metadata, result.document)] -``` +## Control flow -**Benefits**: -- Preserves document structure (headings, lists, tables) -- Automatic OCR fallback for image-based PDFs -- Maintains page-level metadata for source attribution -- Structured output supports advanced chunking strategies +**Through the UI** -#### Chunking Strategy Selection -```python -# Dynamic chunker selection based on config -chunker_mode = config.get("chunker_mode", "legacy") - -if chunker_mode == "docling": - self.chunker = DoclingChunker( - max_tokens=chunk_size, - overlap=overlap_sentences, - tokenizer_model="Qwen/Qwen3-Embedding-0.6B" - ) -else: - self.chunker = MarkdownRecursiveChunker( - max_chunk_size=chunk_size, - min_chunk_size=min(chunk_overlap, chunk_size // 4) - ) +1. `IndexForm` → `chatAPI.buildIndex()` → `POST :8000/indexes//build` (or `POST :8000/sessions//index` for session uploads). +2. `backend/server.py` normalises the option names (`INDEX_OPTIONS`, `server.py:58-70`), omits anything the caller did not send so pipeline defaults apply, and POSTs to `${RAG_API_URL}/index`. +3. `rag_system/api_server.py:392-473` validates the body, builds a per-request config from a deep copy of the pipeline profile (`_build_index_config_override`, `:197-236`), constructs a fresh `IndexingPipeline` with it, and calls `run(file_paths)`. +4. The chosen embedding model is written into the index's SQLite metadata (`api_server.py:445-449`); retrieval later re-applies it (`api_server.py:116-134`). + +**From the command line** + +```bash +python -m rag_system.main index /path/to/docs --mode default ``` -#### Recursive Markdown Chunking Algorithm -```python -def chunk(self, text: str, document_id: str, metadata: Dict) -> List[Dict]: - # Priority hierarchy for splitting - separators = [ - "\n\n# ", # H1 headers (highest priority) - "\n\n## ", # H2 headers - "\n\n### ", # H3 headers - "\n\n", # Paragraph breaks - "\n", # Line breaks - ". ", # Sentence boundaries - " " # Word boundaries (last resort) - ] - - chunks = [] - current_chunk = "" - - for separator in separators: - if len(current_chunk) <= self.max_chunk_size: - continue - - # Split on current separator - parts = current_chunk.split(separator) - - # Reassemble with overlap - for i, part in enumerate(parts): - if len(part) > self.max_chunk_size: - # Recursively split large parts - continue - - # Add overlap from previous chunk - if i > 0 and len(chunks) > 0: - overlap_text = chunks[-1]["text"][-self.chunk_overlap:] - part = overlap_text + separator + part - - chunks.append({ - "text": part, - "document_id": document_id, - "metadata": {**metadata, "chunk_index": len(chunks)} - }) +`_collect_file_paths` (`rag_system/main.py:133-146`) walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, `.md`, `.txt`, or accepts a single file; the `index` branch (`main.py:171-188`) then calls `factory.get_indexing_pipeline(mode).run(file_paths)`. `--mode` accepts `default` or `fast`. `python rag_system/main.py …` as a file path does not work — run it as a module from the project root. + +`create_index_script.py` is a separate interactive/batch tool that creates a SQLite index row and points the pipeline at that index's own table (`--batch `, `--config `, `--create-sample`). + +## Request fields (`POST /index`) + +Both camelCase and snake_case are accepted; the RAG API normalises to snake_case once at parse time (`api_server.py:51-76`). + +| Field | Default | Effect | +|-------|---------|--------| +| `file_paths` | **required** | List of absolute paths. A missing or non-list value returns HTTP 400. | +| `session_id` | – | Sets `overview_path` to `index_store/overviews/.jsonl`. | +| `table_name` | resolved from the session's index | Sets `storage.text_table_name` and `retrieval.dense.lancedb_table_name`. | +| `enable_latechunk` | `false` | Writes a second `
_lc` table of late-chunked vectors. | +| `enable_docling_chunk` | `false` | Sets `chunker_mode: "docling"` when true. See the caveat below. | +| `chunk_size` | `512` | Token budget per chunk, for both chunkers. | +| `retrieval_mode` (alias `search_type`) | – | Validated against `hybrid` / `vector_only` / `fts_only` (HTTP 400 otherwise) and recorded on the index config as `retrieval.search_type`. It cannot change the artifacts written; it takes effect at query time. | +| `window_size` | `2` | Neighbour window for contextual enrichment. | +| `enable_enrich` | `true` | Contextual enrichment on/off. | +| `embedding_model` | profile value | Overrides `embedding_model_name` and is stamped into the index metadata. | +| `enrich_model` | – | Model for contextual enrichment. | +| `overview_model_name` (aliases `overviewModel`, `overview_model`) | – | Model for document overviews. | +| `batch_size_embed` | `50` | Embedding batch size. | +| `batch_size_enrich` | `25` | Enrichment batch size. | + +Response (`api_server.py:451-467`): + +```jsonc +{ + "message": "Indexing process for N file(s) completed successfully.", + "table_name": "text_pages_", + "latechunk": false, + "docling_chunk": true, + "indexing_config": { + "chunk_size": 512, "retrieval_mode": "hybrid", "window_size": 2, + "enable_enrich": true, "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": null, "overview_model_name": null, + "batch_size_embed": 50, "batch_size_enrich": 25 + } +} ``` -### DocLing Chunking Implementation +`indexing_config.embedding_model` reports the model actually used, not the raw request value. There is no `indexed_files` field. -#### Token-Aware Sentence Packing -```python -class DoclingChunker: - def __init__(self, max_tokens: int = 512, overlap: int = 1, - tokenizer_model: str = "Qwen/Qwen3-Embedding-0.6B"): - self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_model) - self.max_tokens = max_tokens - self.overlap = overlap # sentences of overlap - - def split_markdown(self, markdown: str, document_id: str, metadata: Dict): - sentences = self._sentence_split(markdown) - chunks = [] - window = [] - - while sentences: - # Add sentences until token limit - while (sentences and - self._token_len(" ".join(window + [sentences[0]])) <= self.max_tokens): - window.append(sentences.pop(0)) - - if not window: # Single sentence > limit - window.append(sentences.pop(0)) - - # Create chunk - chunk_text = " ".join(window) - chunks.append({ - "chunk_id": f"{document_id}_{len(chunks)}", - "text": chunk_text, - "metadata": { - **metadata, - "chunk_index": len(chunks), - "heading_path": metadata.get("heading_path", []), - "block_type": metadata.get("block_type", "paragraph") - } - }) - - # Add overlap for next chunk - if self.overlap and sentences: - overlap_sentences = window[-self.overlap:] - sentences = overlap_sentences + sentences - window = [] - - return chunks -``` +Unknown fields are ignored rather than rejected. -#### Document Structure Preservation -```python -def chunk_document(self, doc, document_id: str, metadata: Dict): - """Walk DoclingDocument tree and emit structured chunks.""" - chunks = [] - current_heading_path = [] - buffer = [] - - # Process document elements in reading order - for txt_item in doc.texts: - role = getattr(txt_item, "role", None) - - if role == "heading": - self._flush_buffer(buffer, chunks, current_heading_path) - level = getattr(txt_item, "level", 1) - # Update heading hierarchy - current_heading_path = current_heading_path[:level-1] - current_heading_path.append(txt_item.text.strip()) - continue - - # Accumulate text in token-aware buffer - text_piece = txt_item.text - if self._buffer_would_exceed_limit(buffer, text_piece): - self._flush_buffer(buffer, chunks, current_heading_path) - - buffer.append(text_piece) - - self._flush_buffer(buffer, chunks, current_heading_path) - return chunks -``` +## Pipeline config keys -### Contextual Enrichment Implementation +Read by `IndexingPipeline.__init__` and by `run()` for the table paths: -#### Batch Processing Pattern -```python -class ContextualEnricher: - def enrich_batch(self, chunks: List[Dict]) -> List[Dict]: - enriched_chunks = [] - - # Process in batches to manage memory - for i in range(0, len(chunks), self.batch_size): - batch = chunks[i:i + self.batch_size] - - # Parallel enrichment within batch - with concurrent.futures.ThreadPoolExecutor(max_workers=4) as executor: - futures = [ - executor.submit(self._enrich_single_chunk, chunk) - for chunk in batch - ] - - for future in concurrent.futures.as_completed(futures): - enriched_chunks.append(future.result()) - - return enriched_chunks -``` +| Key | Default in code | `default` profile | `fast` profile | +|-----|-----------------|-------------------|----------------| +| `chunker_mode` | `docling` | not set | not set | +| `chunking.chunk_size` (aliases `chunk_size`, `max_tokens`) | `1500` | not set | not set | +| `overlap_sentences` | `1` | not set | not set | +| `embedding_model_name` | `EXTERNAL_MODELS["embedding_model"]` | `microsoft/harrier-oss-v1-0.6b` | same | +| `storage.lancedb_uri` / `db_path` / `lancedb_path` | – (raises if all absent) | `./lancedb` | `./lancedb` | +| `storage.text_table_name` | falls back to `retrieval.dense.lancedb_table_name`, then `default_text_table` | `text_pages_v4` | `text_pages_v4` | +| `indexing.embedding_batch_size` | `50` | `50` | `100` | +| `indexing.enrichment_batch_size` | `10` | `10` | `50` | +| `retrieval.dense.enabled` | `true` | `true` | `true` | +| `retrieval.latechunk.enabled` (also read as `late_chunking`) | `false` | `true` | `false` | +| `retrieval.latechunk.lancedb_table_name` / `table_suffix` | suffix `_lc` | not set | not set | +| `contextual_enricher.enabled` / `window_size` | `false` / `1` | `true` / `1` | `false` / `1` | +| `enrich_model` (then `enrichment_model_name`, then `OLLAMA_CONFIG["enrichment_model"]`, then `generation_model`) | – | – | – | +| `overview.enabled` | `true` | not set ⇒ on | not set ⇒ on | +| `overview_model_name` / `overview.model` | falls back to enrichment then generation model | not set | not set | +| `overview_first_n_chunks` / `overview.max_chunks` | `5` | not set | not set | +| `overview_path` | `index_store/overviews/overviews.jsonl` | not set | not set | -#### Contextual Prompt Engineering -```python -def _generate_context_summary(self, chunk_text: str, surrounding_context: str) -> str: - prompt = f""" - Analyze this text chunk and provide a concise summary that captures: - 1. Main topics and key information - 2. Context within the broader document - 3. Relevance for search and retrieval - - Document Context: - {surrounding_context} - - Chunk to Analyze: - {chunk_text} - - Summary (max 2 sentences): - """ - - response = self.llm_client.complete( - prompt=prompt, - model=self.ollama_config["enrichment_model"] # qwen3:0.6b - ) - - return response.strip() -``` +Note the layering: the CLI/profile path uses the code default `chunk_size` of **1500** tokens because `PIPELINE_CONFIGS` sets no `chunk_size`, while the HTTP path always sends **512**. Likewise `enrichment_batch_size` is 10 from the profile but 25 over HTTP. -### Embedding Generation Pipeline +## Conversion (`ingestion/document_converter.py`) -#### Model Selection Strategy -```python -def select_embedder(model_name: str, ollama_host: str = None): - """Select appropriate embedder based on model name.""" - if "Qwen3-Embedding" in model_name: - return QwenEmbedder(model_name=model_name) - elif "bge-" in model_name: - return BGEEmbedder(model_name=model_name) - elif ollama_host and model_name in ["nomic-embed-text"]: - return OllamaEmbedder(model_name=model_name, host=ollama_host) - else: - # Default to Qwen embedder - return QwenEmbedder(model_name="Qwen/Qwen3-Embedding-0.6B") -``` +Three docling converters are built independently at construction time, so a failure in one does not disable the others (`:67-104`): -#### Batch Embedding Generation -```python -class QwenEmbedder: - def create_embeddings(self, texts: List[str]) -> np.ndarray: - """Generate embeddings in batches for efficiency.""" - embeddings = [] - - for i in range(0, len(texts), self.batch_size): - batch = texts[i:i + self.batch_size] - - # Tokenize and encode - inputs = self.tokenizer( - batch, - padding=True, - truncation=True, - max_length=512, - return_tensors='pt' - ) - - with torch.no_grad(): - outputs = self.model(**inputs) - # Mean pooling over token embeddings - batch_embeddings = outputs.last_hidden_state.mean(dim=1) - embeddings.append(batch_embeddings.cpu().numpy()) - - return np.vstack(embeddings) -``` +* **no-OCR** PDF converter, +* **OCR** PDF converter, +* **general** converter for DOCX/HTML/MD. -### LanceDB Storage Implementation +`.txt` files bypass docling entirely and are wrapped in a fenced code block (`:152-166`). -#### Table Management Strategy -```python -class LanceDBManager: - def create_table_if_not_exists(self, table_name: str, schema: Schema): - """Create LanceDB table with proper schema.""" - try: - table = self.db.open_table(table_name) - print(f"Table {table_name} already exists") - return table - except FileNotFoundError: - # Table doesn't exist, create it - table = self.db.create_table( - table_name, - schema=schema, - mode="create" - ) - print(f"Created new table: {table_name}") - return table - - def index_chunks(self, chunks: List[Dict], table_name: str): - """Store chunks with embeddings in LanceDB.""" - table = self.get_table(table_name) - - # Prepare data for insertion - records = [] - for chunk in chunks: - record = { - "chunk_id": chunk["chunk_id"], - "text": chunk["text"], - "vector": chunk["embedding"].tolist(), - "metadata": json.dumps(chunk["metadata"]), - "document_id": chunk["metadata"]["document_id"], - "chunk_index": chunk["metadata"]["chunk_index"] - } - records.append(record) - - # Batch insert - table.add(records) - - # Create vector index for fast similarity search - table.create_index("vector", config=IvfPq(num_partitions=256)) -``` +For PDFs, `_pdf_has_text()` opens the file with PyMuPDF and checks for any extractable text; if there is none, the OCR converter is used, otherwise the fast no-OCR converter (`:125-150`). When a PDF has no text layer but no OCR converter is available, the pipeline logs and retries without OCR rather than skipping the file. -### Overview Building for Query Routing +**OCR engine selection** (`build_ocr_options`, `:22-48`) picks the first engine whose backend is actually installed: -#### Document Summarization Strategy -```python -class OverviewBuilder: - def build_overview(self, chunks: List[Dict], document_id: str) -> Dict: - """Generate document overview for query routing.""" - # Take first N chunks for overview (usually most important) - sample_chunks = chunks[:self.max_chunks_for_overview] - combined_text = "\n\n".join([c["text"] for c in sample_chunks]) - - overview_prompt = f""" - Analyze this document and create a brief overview that includes: - 1. Main topic and purpose - 2. Key themes and concepts - 3. Document type and domain - 4. Relevant search keywords - - Document text: - {combined_text} - - Overview (max 3 sentences): - """ - - overview = self.llm_client.complete( - prompt=overview_prompt, - model=self.overview_model # qwen3:0.6b for speed - ) - - return { - "document_id": document_id, - "overview": overview.strip(), - "chunk_count": len(chunks), - "keywords": self._extract_keywords(combined_text), - "created_at": datetime.now().isoformat() - } - - def save_overview(self, overview: Dict): - """Save overview to JSONL file for query routing.""" - overview_path = f"./index_store/overviews/{overview['document_id']}.jsonl" - - with open(overview_path, 'w') as f: - json.dump(overview, f) -``` +| Order | docling options class | Requires | +|-------|----------------------|----------| +| 1 | `OcrMacOptions` | macOS only (`platform.system() == "Darwin"`) plus the `ocrmac` package | +| 2 | `EasyOcrOptions` | `easyocr` | +| 3 | `RapidOcrOptions` | `rapidocr_onnxruntime` | +| 4 | `TesseractOcrOptions` | `tesserocr` | +| 5 | `TesseractCliOcrOptions` | a `tesseract` binary on `PATH` | -### Performance Optimizations +If none is available it logs "No OCR engine available; using docling's default OCR settings." On Linux/Docker, install one of the non-macOS backends if you need scanned-PDF support — `ocrmac` is macOS-only and is excluded from `requirements-docker.txt`. -#### Memory Management -```python -class IndexingPipeline: - def __init__(self, config: Dict, ollama_client: OllamaClient, ollama_config: Dict): - # Lazy initialization to save memory - self._pdf_converter = None - self._chunker = None - self._embedder = None - - def _get_embedder(self): - """Lazy load embedder to avoid memory overhead.""" - if self._embedder is None: - model_name = self.config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B") - self._embedder = select_embedder(model_name) - return self._embedder - - def process_document_batch(self, file_paths: List[str]): - """Process documents in batches to manage memory.""" - for batch_start in range(0, len(file_paths), self.batch_size): - batch = file_paths[batch_start:batch_start + self.batch_size] - - # Process batch - self._process_batch(batch) - - # Cleanup to free memory - if hasattr(self, '_embedder') and self._embedder: - self._embedder.cleanup() -``` +Conversion returns `[(markdown, metadata, DoclingDocument)]` for docling paths and `[(markdown, metadata)]` for `.txt`. A conversion error is caught per document and returns `[]` (`:190-192`), so one bad file does not abort the run. -#### Parallel Processing -```python -def run_parallel_processing(self, file_paths: List[str]): - """Process multiple documents in parallel.""" - with concurrent.futures.ProcessPoolExecutor(max_workers=4) as executor: - futures = [] - - for file_path in file_paths: - future = executor.submit(self._process_single_file, file_path) - futures.append(future) - - # Collect results - results = [] - for future in concurrent.futures.as_completed(futures): - try: - result = future.result(timeout=300) # 5 minute timeout - results.append(result) - except Exception as e: - print(f"Error processing file: {e}") - - return results -``` +## Chunking -### Error Handling and Recovery +`chunker_mode` defaults to `"docling"` (`indexing_pipeline.py:27`). `MarkdownRecursiveChunker` is reached when `chunker_mode` is anything else, or when `DoclingChunker` construction raises (`:48-54`). -#### Graceful Degradation -```python -def run(self, file_paths: List[str], table_name: str): - """Main pipeline with comprehensive error handling.""" - processed_files = [] - failed_files = [] - - for file_path in file_paths: - try: - # Attempt processing - chunks = self._process_single_file(file_path) - - if chunks: - # Store successfully processed chunks - self._store_chunks(chunks, table_name) - processed_files.append(file_path) - else: - print(f"⚠️ No chunks generated from {file_path}") - failed_files.append((file_path, "No chunks generated")) - - except Exception as e: - print(f"❌ Error processing {file_path}: {e}") - failed_files.append((file_path, str(e))) - continue # Continue with other files - - # Return summary - return { - "processed": len(processed_files), - "failed": len(failed_files), - "processed_files": processed_files, - "failed_files": failed_files - } -``` +> **Caveat:** the RAG API only ever sets `chunker_mode` when `enable_docling_chunk` is true (`api_server.py:214-215`) — there is no `else` branch. Sending `doclingChunk: false` over HTTP therefore leaves the key unset and the docling default still applies. The legacy chunker is selectable only by setting `chunker_mode: "legacy"` in a pipeline config (which `create_index_script.py:74` does). -#### Recovery Mechanisms -```python -def recover_from_partial_failure(self, table_name: str, document_id: str): - """Recover from partial indexing failures.""" - try: - # Check what was already processed - table = self.db_manager.get_table(table_name) - existing_chunks = table.search().where(f"document_id = '{document_id}'").to_list() - - if existing_chunks: - print(f"Found {len(existing_chunks)} existing chunks for {document_id}") - return True - - # Cleanup partial data - self._cleanup_partial_data(table_name, document_id) - return False - - except Exception as e: - print(f"Recovery failed: {e}") - return False -``` +**DoclingChunker** (`ingestion/docling_chunker.py`) -### Configuration and Customization +* `chunk_document(doc, …)` walks the `DoclingDocument` element tree in reading order (`:88-246`). Tables are emitted as atomic markdown chunks; headings maintain a `heading_path` and are not emitted as content; paragraph text is packed until `max_tokens`. +* A second consolidation pass merges consecutive paragraph chunks that share a page and heading path, up to `max_tokens` (`:199-246`). +* `split_markdown()` is the fallback when only Markdown is available: it runs the legacy chunker with a 10,000-token ceiling, then repacks sentences to `max_tokens`, carrying `overlap` sentences (default 1) into the next window (`:47-83`). +* `chunk_document()` is what runs for documents converted by docling; the pipeline picks it via `hasattr(self.chunker, "chunk_document")` (`indexing_pipeline.py:192-195`). -#### Pipeline Configuration Options -```python -DEFAULT_CONFIG = { - "chunking": { - "strategy": "docling", # "docling", "recursive", "fixed" - "max_tokens": 512, - "overlap": 64, - "min_chunk_size": 100 - }, - "embedding": { - "model_name": "Qwen/Qwen3-Embedding-0.6B", - "batch_size": 32, - "max_length": 512 - }, - "enrichment": { - "enabled": True, - "model": "qwen3:0.6b", - "batch_size": 16 - }, - "overview": { - "enabled": True, - "max_chunks": 5, - "model": "qwen3:0.6b" - }, - "storage": { - "create_index": True, - "index_type": "IvfPq", - "num_partitions": 256 - } -} +**MarkdownRecursiveChunker** (`ingestion/chunking.py`) + +Recursively splits on `\n## `, `\n### `, `\n#### `, ```` ``` ````, `\n\n`, then on word boundaries, and merges adjacent pieces up to `max_chunk_size` while respecting `min_chunk_size` (`:34-124`). The pipeline passes `min_chunk_size = max(1, chunk_size // 4)`. + +Both chunkers count tokens with `AutoTokenizer.from_pretrained(embedding_model_name)`. If the tokenizer cannot be loaded they log a warning and fall back to a 4-characters-per-token approximation. + +There is **no chunk-overlap knob**. The docling path carries `overlap_sentences` (default 1) only in `split_markdown`; `chunk_document` emits non-overlapping consolidated blocks; the legacy path has no overlap logic at all. + +After chunking, the pipeline stamps a sequential `metadata.chunk_index` on every chunk of the document (`indexing_pipeline.py:202-205`) — this is what context expansion and late-chunk merging use at query time. + +## Document overviews + +`OverviewBuilder.build_and_store(doc_id, chunks)` (`indexing/overview_builder.py:33-49`) runs once per document, inside the chunking loop and **before** enrichment. It sends the first `first_n_chunks` chunks (default 5), truncated to 5000 characters, to the overview model and appends one JSON line per document: + +```jsonc +{"doc_id": "report.pdf", "overview": "…"} ``` -#### Custom Processing Hooks -```python -class IndexingPipeline: - def __init__(self, config: Dict, hooks: Dict = None): - self.hooks = hooks or {} - - def _run_hook(self, hook_name: str, *args, **kwargs): - """Execute custom processing hooks.""" - if hook_name in self.hooks: - return self.hooks[hook_name](*args, **kwargs) - return None - - def process_chunk(self, chunk: Dict) -> Dict: - """Process single chunk with custom hooks.""" - # Pre-processing hook - chunk = self._run_hook("pre_chunk_process", chunk) or chunk - - # Standard processing - if self.contextual_enricher: - chunk = self.contextual_enricher.enrich_chunk(chunk) - - # Post-processing hook - chunk = self._run_hook("post_chunk_process", chunk) or chunk - - return chunk +Written in **append** mode to `overview_path`, which is one file per index (`index_store/overviews/.jsonl` when the API supplies a `session_id`), not one file per document. Failures are caught per document and logged (`indexing_pipeline.py:208-212`). Set `overview.enabled: false` in the pipeline config to skip the stage; there is no HTTP field for it. + +## Contextual enrichment + +`ContextualEnricher.enrich_chunks(chunks, window_size)` (`indexing/contextualizer.py:82-144`) prepends an LLM-written summary to each chunk's embedded text and preserves the untouched original in `metadata.original_text`: + ``` +Context: <2-5 sentence summary> --- -## Current Implementation Status - -### Completed Features ✅ -- DocLing-based PDF processing with OCR fallback -- Multiple chunking strategies (DocLing, Recursive, Fixed-size) -- Qwen3-Embedding-0.6B integration -- Contextual enrichment with qwen3:0.6b -- LanceDB storage with vector indexing -- Overview generation for query routing -- Batch processing and parallel execution -- Comprehensive error handling - -### In Development 🚧 -- Graph extraction and knowledge graph building -- Multimodal processing for images and tables -- Advanced late-chunking optimization -- Distributed processing support - -### Planned Features 📋 -- Custom model fine-tuning pipeline -- Real-time incremental indexing -- Cross-document relationship extraction -- Advanced metadata enrichment + +``` ---- +Processing is **sequential**: `BatchProcessor.process_in_batches` is a plain `for` loop over slices with progress reporting and a `gc.collect()` every fifth batch (`utils/batch_processor.py:105-125`). "Batch size" controls reporting granularity and memory, not concurrency. Budget one LLM round-trip per chunk when sizing a run — there is no thread or process pool anywhere in the indexing path. -## Performance Benchmarks +Chain-of-thought markers (`…`), assistant tags and a leading `Answer:` are stripped from the summary; a summary shorter than 5 characters is discarded and the chunk is indexed unenriched (`contextualizer.py:56-74`). -| Document Type | Processing Speed | Memory Usage | Storage Efficiency | -|---------------|------------------|--------------|-------------------| -| Text PDFs | 2-5 pages/sec | 2-4GB | 1MB/100 pages | -| Image PDFs | 0.5-1 page/sec | 4-8GB | 2MB/100 pages | -| Technical Docs | 1-3 pages/sec | 3-6GB | 1.5MB/100 pages | -| Research Papers | 2-4 pages/sec | 2-4GB | 1.2MB/100 pages | +## Embedding -## Extension Points +`select_embedder(model_name, ollama_host)` (`indexing/representations.py:167-173`) is a two-way dispatch: -### Custom Chunkers -```python -class CustomChunker(BaseChunker): - def chunk(self, text: str, document_id: str, metadata: Dict) -> List[Dict]: - # Implement custom chunking logic - pass -``` +* the name contains `/` or starts with `http` → `QwenEmbedder` (HuggingFace `AutoModel`, loaded in-process); +* anything else → `OllamaEmbedder`, one HTTP call to `/api/embeddings` per text. -### Custom Embedders -```python -class CustomEmbedder(BaseEmbedder): - def create_embeddings(self, texts: List[str]) -> np.ndarray: - # Implement custom embedding generation - pass -``` +`QwenEmbedder` (`representations.py:15-94`): -### Custom Enrichers -```python -class CustomEnricher(BaseEnricher): - def enrich_chunk(self, chunk: Dict) -> Dict: - # Implement custom enrichment logic - pass -``` \ No newline at end of file +* device order CUDA → MPS → CPU; `float16` off CPU; +* weights cached per model name in a module-level dict, so repeated construction reuses them; +* tokenizer is loaded with `padding_side="left"`, and pooling takes the **last real token**, not a mean — with left padding that is simply the final column, and the right-padding case is handled explicitly (`:66-76`); +* `max_length` is `min(tokenizer.model_max_length, 8192)` so `truncation=True` actually truncates; +* NaN/Inf values are replaced with zeros and logged. + +`LateChunkEncoder` mean-pools over each span instead (`indexing/latechunk.py:82`) — the two paths deliberately use different pooling and write to different tables. + +**Embedding dimensions are never hard-coded.** `VectorIndexer.index()` reads the width from the first produced vector (`indexing/embedders.py:47`) and builds the pyarrow schema from it. If the target table already exists with a different width, indexing raises: + +> Table '…' stores N-dim vectors but the current embedding model produced M-dim vectors. Changing the embedding model requires rebuilding the index. + +**Changing `EMBEDDING_MODEL` (or the per-request `embedding_model`) requires re-indexing.** `Qwen/Qwen3-Embedding-4B` produces 2560-dim vectors; `microsoft/harrier-oss-v1-0.6b` and `Qwen/Qwen3-Embedding-0.6B` both produce 1024-dim. They are not interchangeable against an existing table — and because a matching width does not make two models compatible, `VectorIndexer` also stamps the embedding model name (and an L2-`normalized` flag) into each table's metadata and refuses to write or read it with a different one. + +## LanceDB write and full-text index + +`VectorIndexer.index(table_name, chunks, embeddings)` (`indexing/embedders.py:39-135`) writes one row per chunk: + +| Column | Content | +|--------|---------| +| `vector` | fixed-size float32 list, width from the model | +| `text` | the text that was embedded (enriched, if enrichment ran) | +| `chunk_id`, `document_id`, `chunk_index` | flattened identifiers used by context expansion | +| `metadata` | the whole chunk dict as a JSON string, including `original_text` | + +Chunks whose vector contains NaN or Inf are skipped with a warning. The table is appended to when it exists and created otherwise, which makes re-running a build idempotent at the table level; `tbl.add(..., on_bad_vectors='drop')` is retried with a zero-fill strategy on failure. + +Immediately after the vector write, the pipeline ensures a Lance native full-text index on the `text` column (`indexing_pipeline.py:264-284`), guarding against both the LanceDB default name `text_idx` and this project's older name `fts_text` so a rebuild does not raise. This index is what the `hybrid` and `fts_only` retrieval modes query — there is no separate BM25 store on disk and no `bm25_path` config key. + +No ANN index is created. `create_index` / IVF-PQ appears nowhere in `rag_system/`, so vector search is an exhaustive scan. That is fine for the corpus sizes this project targets and avoids the training-set-size requirements of IVF-PQ. + +## Late chunking (optional) + +When `retrieval.latechunk.enabled` is true, a second pass runs per document (`indexing_pipeline.py:286-327`): + +1. Concatenate the document's chunk texts with newlines and record each chunk's character span. +2. Feed the whole document through `LateChunkEncoder.encode()` — one forward pass, truncated at 8192 tokens — and mean-pool the token hidden states inside each span. +3. Write those vectors to `
_lc` (or `latechunk.lancedb_table_name` when set) with the same chunk rows. + +Each chunk vector is therefore produced with knowledge of the whole document. A per-document encode failure, or a vector/chunk count mismatch, logs a warning and skips that document. The cost is a second full copy of the embedding model in memory and roughly double the vectors written. + +The retrieval side reads `
_lc` with the same default suffix — see `retrieval_pipeline.md`. + +## Knowledge graph — removed 2026-08-09 + +There is no knowledge-graph step any more. `indexing/graph_extractor.py`, the +`retrieval.graph.*` config keys and the `.gml` writer were deleted at roadmap +item 2.5. The path had never been armed (both shipped profiles set +`enabled: false`), and the evidence argues against reviving it: GraphRAG *loses* +on single-hop retrieval, its multi-hop gains span +3 to +27 points depending on +how well the vector baseline is tuned, and it costs **41–57× at indexing** and up +to **~377× in query tokens** — see +[`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) §6. +`networkx` was dropped from `requirements.txt` in the same change. + +An index built before this change is unaffected: the graph lived in a standalone +`.gml` file that nothing else read. A stale `index_store/graph/` directory can be +deleted by hand. + +## Error handling + +* **Per file** — conversion or chunking errors are caught, logged as `❌ Error processing `, counted by the progress tracker, and the run continues (`indexing_pipeline.py:219-222`). +* **No chunks at all** — the run raises `RuntimeError` ("No text chunks were generated from the supplied documents…"), which surfaces as a 500 from `POST /index` so a failed conversion is never reported as a successful build (`:226-231`). +* **Duplicate table** — `VectorIndexer` appends instead of recreating (`embedders.py:105-120`); the backend additionally treats an "already exists" error from the RAG API as non-fatal and reports "Index already built – skipping rebuild." (`backend/server.py`, `handle_build_index`). +* **Dimension mismatch** — hard `ValueError`, on purpose. Silently dropping or recreating an index would corrupt it. +* **FTS index** — creation failures are logged, not raised; the vectors are already written and `vector_only` retrieval still works. +* **Overview / late-chunk / enrichment failures** — logged and skipped; the main vector index is still produced. + +At the end, `_print_final_statistics` reports files processed, chunks generated, average chunks per file, which components ran, and the batch sizes used (`indexing_pipeline.py:349-372`). + +## Not integrated + +* **Vision / multimodal.** There is no vision model anywhere in the pipeline: no image embeddings are produced and no image table is written. PDF understanding is docling's layout parsing plus OCR. Models such as GLM-OCR or Qwen3-VL would be reasonable extensions, but no code path exists for them today. +* **Parallel document processing.** Documents are processed one at a time and enrichment is one sequential LLM call per chunk. There is no `ProcessPoolExecutor` or `ThreadPoolExecutor` in the indexing path. + +--- +_Keep this document updated when stages, config keys, or the `/index` contract change._ diff --git a/Documentation/installation_guide.md b/Documentation/installation_guide.md index 2bb881ab..44c966fa 100644 --- a/Documentation/installation_guide.md +++ b/Documentation/installation_guide.md @@ -1,22 +1,23 @@ -# 📦 RAG System Installation Guide +# 📦 LocalGPT Installation Guide -_Last updated: 2025-01-07_ +_Last updated: 2026-08-08_ -This guide provides step-by-step instructions for installing and setting up the RAG system using either Docker or direct development approaches. +This guide provides step-by-step instructions for installing and setting up +LocalGPT using either Docker or a direct development install. --- ## 🎯 Installation Options -### Option 1: Docker Deployment (Production Ready) 🐳 -- **Best for**: Production environments, isolated setups, easy management -- **Requirements**: Docker Desktop + Local Ollama -- **Setup time**: ~10 minutes +### Option 1: Docker Deployment 🐳 +- **Best for**: reproducible environments, keeping Python dependencies isolated +- **Requirements**: Docker Desktop + Ollama (host or containerized) +- **Setup time**: ~10 minutes plus image build and model downloads -### Option 2: Direct Development (Developer Friendly) 💻 -- **Best for**: Development, customization, debugging +### Option 2: Direct Development 💻 +- **Best for**: development, customization, debugging - **Requirements**: Python + Node.js + Ollama -- **Setup time**: ~15 minutes +- **Setup time**: ~15 minutes plus model downloads --- @@ -28,26 +29,31 @@ This guide provides step-by-step instructions for installing and setting up the - **CPU**: 4 cores, 2.5GHz+ - **RAM**: 8GB (16GB recommended) - **Storage**: 50GB free space -- **OS**: macOS 10.15+, Ubuntu 20.04+, Windows 10+ +- **OS**: macOS 10.15+, Ubuntu 20.04+, Windows 10+ (WSL2) #### **Recommended Requirements** - **CPU**: 8+ cores, 3.0GHz+ -- **RAM**: 32GB+ (for large models) +- **RAM**: 32GB+ (for the larger generation models) - **Storage**: 200GB+ SSD -- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional) +- **GPU**: NVIDIA GPU with 8GB+ VRAM (optional; Apple Silicon uses MPS) + +Disk budget for the defaults: `qwen3.5:9b` and `qwen3.5:4b` in Ollama, plus +`microsoft/harrier-oss-v1-0.6b` (~1.2GB) in the HuggingFace cache +(`~/.cache/huggingface`). Reranking is off by default, so no reranker weights +are fetched unless you switch it on (`Qwen/Qwen3-Reranker-4B`, ~7.5GB). ### 1.2 Common Dependencies **Required for both approaches:** -- **Ollama**: AI model runtime (always required) -- **Git**: 2.30+ for cloning repository +- **Ollama**: model runtime (always required) +- **Git**: 2.30+ for cloning the repository **Docker-specific:** -- **Docker Desktop**: 24.0+ with Docker Compose +- **Docker Desktop**: 24.0+ with the Compose plugin **Direct Development-specific:** -- **Python**: 3.8+ -- **Node.js**: 16+ with npm +- **Python**: 3.10+ (3.11 recommended — the Docker images use `python:3.11-slim`) +- **Node.js**: 20+ with npm --- @@ -70,23 +76,26 @@ ollama --version # Run the installer and follow setup wizard ``` -### 2.2 Configure Ollama +### 2.2 Pull the Models ```bash # Start Ollama server ollama serve -# In another terminal, install required models -ollama pull qwen3:0.6b # Fast model (650MB) -ollama pull qwen3:8b # High-quality model (4.7GB) +# In another terminal, install the two default models +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, decomposition, enrichment, verification # Verify models are installed ollama list # Test Ollama -ollama run qwen3:0.6b "Hello, how are you?" +ollama run qwen3.5:4b "Hello, how are you?" ``` +The embedding and reranker models are **not** Ollama models — they are downloaded +from HuggingFace automatically the first time the pipeline loads them. + **⚠️ Important**: Keep Ollama running (`ollama serve`) for the entire setup process. --- @@ -135,33 +144,40 @@ docker compose version 3. Restart computer and start Docker Desktop 4. Verify in PowerShell: `docker --version` -### 3.2 Clone and Setup RAG System +### 3.2 Clone and Start ```bash # Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT -# Verify Ollama is running +# Verify Ollama is running on the host curl http://localhost:11434/api/tags -# Start Docker containers +# Start Docker containers (uses local Ollama) ./start-docker.sh -# Wait for containers to start (2-3 minutes) -sleep 120 +# Or, without host Ollama: +# ./start-docker.sh container +# For scripts and CI, add -y so the script never prompts: +# ./start-docker.sh local -y (or: NONINTERACTIVE=1 ./start-docker.sh) # Verify deployment ./start-docker.sh status ``` +The first build compiles the frontend and installs the Python dependencies, and +`rag-api` loads the embedding and reranker models before it reports healthy. The +`backend` service has `depends_on: rag-api: service_healthy`, so it deliberately +waits. Expect several minutes on a cold start. + ### 3.3 Test Docker Deployment ```bash # Test all endpoints curl -f http://localhost:3000 && echo "✅ Frontend OK" curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" +curl -f http://localhost:8001/health && echo "✅ RAG API OK" curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" # Access the application @@ -177,8 +193,8 @@ open http://localhost:3000 #### **Python Setup:** ```bash # Clone repository -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Create virtual environment (recommended) python -m venv venv @@ -194,17 +210,27 @@ pip install -r requirements.txt python -c "import torch; print('✅ PyTorch OK')" python -c "import transformers; print('✅ Transformers OK')" python -c "import lancedb; print('✅ LanceDB OK')" +python -c "import docling; print('✅ Docling OK')" ``` +`requirements.txt` at the repository root is the full install used by +`run_system.py`. Two variants exist for narrower cases: + +| File | Purpose | +|------|---------| +| `requirements.txt` | Everything the local stack needs | +| `requirements-docker.txt` | Used by `Dockerfile.backend` and `Dockerfile.rag-api` (no macOS-only packages) | +| `rag_system/requirements.txt` | RAG core only; adds macOS `ocrmac` and `ibm-watsonx-ai` | +| `backend/requirements.txt` | Gateway only (`requests`, `python-dotenv`) | + #### **Node.js Setup:** ```bash # Install Node.js dependencies npm install # Verify Node.js setup -node --version # Should be 16+ +node --version # Should be 20+ npm --version -npm list --depth=0 ``` ### 4.2 Start Direct Development @@ -216,25 +242,30 @@ curl http://localhost:11434/api/tags # Start all components with one command python run_system.py -# Or start components manually in separate terminals: +# Or start components manually, each from the repository root: # Terminal 1: python -m rag_system.api_server -# Terminal 2: cd backend && python server.py +# Terminal 2: python backend/server.py # Terminal 3: npm run dev ``` +> Always run from the repository root. `backend/chat_data.db`, `lancedb/`, +> `index_store/` and `shared_uploads/` are resolved relative to the working +> directory, so `cd backend && python server.py` would create a second database at +> `backend/backend/chat_data.db`. + ### 4.3 Test Direct Development ```bash -# Check system health +# Deep check: imports, config, LanceDB, embedding model, sample query python system_health_check.py +# HTTP health check per service +python run_system.py --health + # Test endpoints curl -f http://localhost:3000 && echo "✅ Frontend OK" curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" - -# Access the application -open http://localhost:3000 +curl -f http://localhost:8001/health && echo "✅ RAG API OK" ``` --- @@ -245,54 +276,106 @@ open http://localhost:3000 ```bash # Clone repository -git clone https://github.com/your-org/rag-system.git -cd rag-system +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Check repository structure ls -la -# Create required directories +# These are created on demand, but you can pre-create them mkdir -p lancedb index_store shared_uploads logs backend -touch backend/chat_data.db # Set permissions chmod -R 755 lancedb index_store shared_uploads -chmod 664 backend/chat_data.db ``` +The SQLite file is created automatically the first time `ChatDatabase` is +constructed — you do not need to `touch` it. + ### 5.2 Configuration +LocalGPT runs with no configuration file. Every variable below has a default that +matches the code. Override them in the shell, in a `.env` at the repository root +(`load_dotenv()` runs on import of `rag_system/main.py` and +`rag_system/factory.py`), or in `docker.env` for containers. `.env.example` ships +the same list. + #### **Environment Variables** -For Docker (automatic via `docker.env`): + +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` (build-time) | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` (build-time) | +| `DB_PATH` | `backend/chat_data.db` | `backend/database.py` | +| `LANCEDB_PATH` | `./lancedb` | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | same | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (only loaded when reranking is switched on) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` (`default` or `fast`) | +| `RAG_API_TIMEOUT` | `600` | `backend/server.py` | +| `RAG_API_INDEX_TIMEOUT` | `3600` | `backend/server.py` | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py` (`ollama` or `watsonx`) | +| `HF_TOKEN` | unset | HuggingFace downloads | + +For Docker these are set in `docker.env` and passed with +`docker compose --env-file docker.env`: + ```bash OLLAMA_HOST=http://host.docker.internal:11434 NODE_ENV=production RAG_API_URL=http://rag-api:8001 NEXT_PUBLIC_API_URL=http://localhost:8000 +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B ``` -For Direct Development (set automatically by `run_system.py`): -```bash -OLLAMA_HOST=http://localhost:11434 -RAG_API_URL=http://localhost:8001 -NEXT_PUBLIC_API_URL=http://localhost:8000 -``` +`run_system.py` does **not** invent environment variables. It inherits your shell +environment unchanged and only adds `NODE_ENV=production` to the Python services +in `--mode prod`. If you want a non-default `OLLAMA_HOST` or `RAG_API_URL`, export +it before launching. + +`NEXT_PUBLIC_*` values are inlined into the frontend bundle by `next build`, so +they are build-time settings. In Docker they are passed as build args in +`docker-compose.yml`; changing them means `docker compose build frontend`. #### **Model Configuration** -The system defaults to these models: -- **Embedding**: `Qwen/Qwen3-Embedding-0.6B` (1024 dimensions) -- **Generation**: `qwen3:0.6b` for fast responses, `qwen3:8b` for quality -- **Reranking**: Built-in cross-encoder + +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation | `qwen3.5:9b` | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility | `qwen3.5:4b` | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024 dims) | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranker (**off by default**) | `Qwen/Qwen3-Reranker-4B`, loaded lazily by the in-repo `QwenRerankerScorer` | `BAAI/bge-reranker-v2-m3`, `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | + +Notes: +- Embedding dimensions are read from the loaded model, never hardcoded. + **Changing `EMBEDDING_MODEL` requires rebuilding every existing index** — + appending vectors of a different width to a LanceDB table raises an explicit + error telling you to re-index. +- If the reranker fails to load, the pipeline logs a warning and continues + **without reranking**. There is no secondary reranker to fall back to. +- Any name containing `/` is treated as a HuggingFace model; anything else is + treated as an Ollama tag, so an Ollama embedding model such as + `nomic-embed-text` also works if you have pulled it. +- Vision / multimodal models are not wired into any pipeline. PDF parsing and OCR + are handled by Docling. ### 5.3 Database Initialization ```bash -# Initialize SQLite database +# Initialize SQLite database (also happens automatically at first use) python -c " from backend.database import ChatDatabase db = ChatDatabase() -db.init_database() -print('✅ Database initialized') +print('✅ Database initialized at', db.db_path) " # Verify database @@ -313,26 +396,27 @@ docker compose ps # For Direct development python system_health_check.py +python run_system.py --health # Universal health check curl -f http://localhost:3000 && echo "✅ Frontend OK" curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" +curl -f http://localhost:8001/health && echo "✅ RAG API OK" curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" ``` #### **RAG System Test:** ```bash -# Test RAG system initialization +# Test RAG system initialization (factory.py is the single entry point) python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') print('✅ RAG System initialized successfully') " -# Test embedding generation +# Test embedding generation and report the real dimension python -c " -from rag_system.main import get_agent +from rag_system.factory import get_agent agent = get_agent('default') embedder = agent.retrieval_pipeline._get_text_embedder() test_emb = embedder.create_embeddings(['Hello world']) @@ -356,7 +440,8 @@ curl -X POST http://localhost:8000/sessions \ -H "Content-Type: application/json" \ -d '{"title": "Test Session"}' -# Test models endpoint +# Test models endpoints +curl http://localhost:8000/models curl http://localhost:8001/models # Test health endpoints @@ -375,13 +460,13 @@ curl http://localhost:8001/health # Ollama not responding curl http://localhost:11434/api/tags -# If fails, restart Ollama +# If it fails, restart Ollama pkill ollama ollama serve # Reinstall models if needed -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` #### **Docker Issues:** @@ -400,7 +485,7 @@ docker system prune -f #### **Python Issues:** ```bash # Check Python version -python --version # Should be 3.8+ +python --version # 3.10+ required, 3.11 recommended # Check virtual environment which python @@ -413,13 +498,23 @@ pip install -r requirements.txt --force-reinstall #### **Node.js Issues:** ```bash # Check Node version -node --version # Should be 16+ +node --version # Should be 20+ # Clear and reinstall rm -rf node_modules package-lock.json npm install ``` +#### **Scanned PDFs produce no text:** +A PDF with no text layer is re-run through Docling's OCR pipeline, but only with an +engine that is actually installed. `rag_system/ingestion/document_converter.py` +probes, in order: `ocrmac` (macOS only), `easyocr`, `rapidocr_onnxruntime`, +`tesserocr`, then the `tesseract` binary. Install whichever suits your platform — +`pip install ocrmac` on macOS (it is listed in `rag_system/requirements.txt` but not +in the root `requirements.txt`), or `pip install easyocr` / `apt install tesseract-ocr` +elsewhere. At startup the converter prints the engine it chose (`OCR engine: …`) or +`No OCR engine available; using docling's default OCR settings.` + ### 7.2 Performance Issues #### **Memory Problems:** @@ -431,35 +526,45 @@ vm_stat # macOS # For Docker: Increase memory allocation # Docker Desktop → Settings → Resources → Memory → 8GB+ -# Use smaller models -ollama pull qwen3:0.6b # Instead of qwen3:8b +# Use lighter models +export GENERATION_MODEL=qwen3.5:4b +# (the default embedder is already the small one, ~1.2GB) ``` #### **Slow Performance:** -- Use SSD storage for databases (`lancedb/`, `shared_uploads/`) -- Increase CPU cores if possible -- Close unnecessary applications -- Use smaller batch sizes in configuration +- Use SSD storage for `lancedb/` and `shared_uploads/` +- Switch to `RAG_CONFIG_MODE=fast` (vector-only retrieval, no reranking, + no decomposition, no verification) +- Lower `indexing.embedding_batch_size` / `enrichment_batch_size` if you are + swapping rather than compute-bound +- Remember the RAG API serialises requests — one query or indexing run at a time --- ## 8. Post-Installation Setup -### 8.1 Model Optimization +### 8.1 Model Experiments ```bash -# Install additional models (optional) -ollama pull nomic-embed-text # Alternative embedding model -ollama pull llama3.1:8b # Alternative generation model +# Install additional generation models +ollama pull qwen3.6:27b # highest quality, ~17GB +ollama pull qwen3.5:2b # lightest utility model -# Test model switching +# Try one for a single request curl -X POST http://localhost:8001/chat \ -H "Content-Type: application/json" \ - -d '{"query": "Hello", "model": "qwen3:8b"}' + -d '{"query": "Hello", "model": "qwen3.5:4b"}' ``` +A per-request `model` is applied only for that request and then restored, and is +rejected with a warning if it is not valid for the active `LLM_BACKEND`. + ### 8.2 Security Configuration +Neither server implements authentication, and both send +`Access-Control-Allow-Origin: *`. Do not expose ports 8000/8001 outside a trusted +network. + ```bash # Set proper file permissions chmod 600 backend/chat_data.db # Restrict database access @@ -492,32 +597,34 @@ EOF chmod +x backup_system.sh ``` +All four paths are plain host directories (bind-mounted into the containers), so +a file copy is a complete backup. Stop the services first so SQLite and LanceDB +are not mid-write. + --- ## 9. Success Criteria ### 9.1 Installation Complete When: -- ✅ All health checks pass without errors +- ✅ `python run_system.py --health` exits 0 - ✅ Frontend loads at http://localhost:3000 -- ✅ All models are installed and responding -- ✅ You can create document indexes +- ✅ `ollama list` shows `qwen3.5:9b` and `qwen3.5:4b` +- ✅ You can create a document index - ✅ You can chat with uploaded documents -- ✅ No error messages in logs/terminal +- ✅ No error messages in `logs/` or the terminal -### 9.2 Performance Benchmarks +### 9.2 Performance Expectations -**Acceptable Performance:** -- System startup: < 5 minutes -- Index creation: < 2 minutes per 100MB document -- Query response: < 30 seconds -- Memory usage: < 8GB total +Numbers depend heavily on hardware, model size and document length. On a machine +matching the recommended requirements, expect: -**Optimal Performance:** -- System startup: < 2 minutes -- Index creation: < 1 minute per 100MB document -- Query response: < 10 seconds -- Memory usage: < 4GB total +- First startup dominated by model downloads (Ollama tags + ~10GB of HuggingFace + weights); subsequent startups load from cache +- Indexing dominated by contextual enrichment — it runs one LLM call per chunk, so + turn it off (`enable_enrich: false`) for the fastest ingest +- Query latency dominated by generation; `RAG_CONFIG_MODE=fast` removes reranking, + decomposition and verification --- @@ -525,18 +632,20 @@ chmod +x backup_system.sh ### 10.1 Getting Started -1. **Upload Documents**: Create your first index with PDF documents -2. **Explore Features**: Try different query types and models -3. **Customize**: Adjust model settings and chunk sizes +1. **Upload Documents**: Create your first index +2. **Explore Features**: Try different retrieval modes and models +3. **Customize**: Adjust chunk size, enrichment and verification per request 4. **Scale**: Add more documents and create multiple indexes ### 10.2 Additional Resources - **Quick Start**: See `Documentation/quick_start.md` - **Docker Usage**: See `Documentation/docker_usage.md` +- **Deployment**: See `Documentation/deployment_guide.md` - **System Architecture**: See `Documentation/architecture_overview.md` - **API Reference**: See `Documentation/api_reference.md` +- **WatsonX backend**: See `WATSONX_README.md` --- -**Congratulations! 🎉** Your RAG system is now ready to use. Visit http://localhost:3000 to start chatting with your documents. \ No newline at end of file +**Congratulations! 🎉** Visit http://localhost:3000 to start chatting with your documents. diff --git a/Documentation/prompt_inventory.md b/Documentation/prompt_inventory.md index a1f14a0d..f5e051e9 100644 --- a/Documentation/prompt_inventory.md +++ b/Documentation/prompt_inventory.md @@ -1,70 +1,81 @@ # 📜 Prompt Inventory (Ground-Truth) -_All generation / verification prompts currently hard-coded in the codebase._ -_Last updated: 2025-07-06_ +_Every prompt hard-coded in the codebase, re-derived from the current source._ -> Edit process: if you change a prompt in code, please **update this file** or, once we migrate to the central registry, delete the entry here. +> Edit process: if you change a prompt in code, update the line range here in the same commit. + +## Which model runs which prompt + +There are two model roles (`rag_system/main.py` `OLLAMA_CONFIG`): + +| Role | Config key | Default | Env override | +|------|-----------|---------|--------------| +| Generation — user-facing answers | `generation_model` | `qwen3.5:9b` | `GENERATION_MODEL` | +| Utility — routing, triage, decomposition, enrichment, overviews, verification | `enrichment_model` | `qwen3.5:4b` | `ENRICHMENT_MODEL` | + +Inside the agent the utility model is resolved by `Agent._utility_model()` (`rag_system/agent/loop.py:56-58`), which returns `ollama_config["enrichment_model"]` and falls back to `generation_model` when that key is absent. No prompt hard-codes a model name. --- -## 1. Indexing / Context Enrichment +## 1. Indexing / context enrichment -| ID | File & Lines | Variable / Builder | Purpose | -|----|--------------|--------------------|---------| -| `overview_builder.default` | `rag_system/indexing/overview_builder.py` `12-21` | `DEFAULT_PROMPT` | Generate 1-paragraph document overview for search-time routing. -| `contextualizer.system` | `rag_system/indexing/contextualizer.py` `11` | `SYSTEM_PROMPT` | System instruction: explain summarisation role. -| `contextualizer.local_context` | same file `13-15` | `LOCAL_CONTEXT_PROMPT_TEMPLATE` | Human message – wraps neighbouring chunks. -| `contextualizer.chunk` | same file `17-19` | `CHUNK_PROMPT_TEMPLATE` | Human message – shows the target chunk. -| `graph_extractor.entities` | `rag_system/indexing/graph_extractor.py` `20-31` | `entity_prompt` | Ask LLM to list entities. -| `graph_extractor.relationships` | same file `53-64` | `relationship_prompt` | Ask LLM to list relationships. +| ID | File & lines | Variable / builder | Model | Purpose | +|----|--------------|--------------------|-------|---------| +| `overview_builder.default` | `rag_system/indexing/overview_builder.py` `13-19` | `OverviewBuilder.DEFAULT_PROMPT` | overview model (resolved at `indexing_pipeline.py:128-133`; utility model by default) | One-paragraph document overview used by the triage routers. Input is the first `first_n_chunks` chunks (default 5), truncated to 5000 characters at `overview_builder.py:37`. | +| `contextualizer.system` | `rag_system/indexing/contextualizer.py` `12` | `SYSTEM_PROMPT` | enrichment model | Role instruction for the summariser. | +| `contextualizer.local_context` | same file `14-16` | `LOCAL_CONTEXT_PROMPT_TEMPLATE` | — | Wraps the neighbouring-chunk window in `` tags. | +| `contextualizer.chunk` | same file `18-26` | `CHUNK_PROMPT_TEMPLATE` | — | Shows the target chunk and carries the actual instruction: a 2-5 sentence context summary, "Answer *only* with the succinct context and nothing else." | -## 2. Retrieval / Query Transformation +The three contextualizer parts are concatenated into a single `/api/generate` completion prompt at `contextualizer.py:42-51` (no chat roles) and sent at `contextualizer.py:53` with `enable_thinking=False`. -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `query_transformer.expand` | `rag_system/retrieval/query_transformer.py` `10-26` | Produce query rewrites (keywords, boolean). | -| `hyde.hypothetical_doc` | same `115-122` | HyDE hypothetical document generator. | -| `graph_query.translate` | same `124-140` | Translate user question to JSON KG query. | +## 2. Retrieval / query transformation -## 3. Pipeline Answer Synthesis +| ID | File & lines | Model | Purpose | +|----|--------------|-------|---------| +| `query_transformer.decompose` | `rag_system/retrieval/query_transformer.py` `38-84` (system) + `87-196` (few-shot examples) + `250-265` (assembly) | utility model (`loop.py:30`) | Resolve pronouns/ellipsis against the last 5 turns, then split the query into standalone sub-queries. Returns RFC-8259 JSON `{requires_decomposition, reasoning, resolved_query, sub_queries}`; sent with `format="json"` at `:268`; the list is deduplicated and capped by `max_sub_queries` at `:290` (default 10). | -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `retrieval_pipeline.synth_final` | `rag_system/pipelines/retrieval_pipeline.py` `217-256` | Turn verified facts into answer (with directives 1-6). | +| `retrieval.retry.reformulate` | `rag_system/pipelines/retrieval_pipeline.py`, `_reformulate_query` | enrichment/utility model, `format="json"` | Fires **only** when the evidence-sufficiency retry triggers (roadmap 2.1). Asks for one rewrite of a weak-evidence query into the concrete nouns and synonyms a document would use, preserving every entity and constraint. Returns `{"query": "…"}`; `format="json"` is what keeps a small model's thinking preamble out of the rewritten query. | -## 4. Agent – Classical Loop +Two legacy example blocks still live in the file at `199-203` and `205-248`, but they are excluded from the assembled prompt — the two lines that would concatenate them are commented out at `:253-254`. -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `agent.loop.initial_thought` | `rag_system/agent/loop.py` `157-180` | First LLM call to think about query. | -| `agent.loop.verify_path` | same `190-205` | Secondary thought loop. | -| `agent.loop.compose_sub` | same `506-542` | Compose answer from sub-answers. | -| `agent.loop.router` | same `648-660` | Decide which subsystem handles query. | +## 3. Answer synthesis + +| ID | File & lines | Model | Purpose | +|----|--------------|-------|---------| +| `retrieval_pipeline.synth_final` | `rag_system/pipelines/retrieval_pipeline.py` `225-250` | generation model | Turn the retrieved snippets into the final answer (6 numbered directives; instructs the model to reply exactly "I could not find that information in the provided documents." when the snippets do not cover the question). Streamed token-by-token via `stream_completion` at `:253-256`; each token is forwarded to the `event_callback` as a `token` event. | + +## 4. Agent loop (`rag_system/agent/loop.py`) + +| ID | Lines | Model | Purpose | +|----|-------|-------|---------| +| `agent.loop.history_wrapper` | `163-171` | none — template only | `_format_query_with_history` builds the `contextual_query` string that is embedded into downstream prompts. It makes no LLM call. | +| `agent.loop.overview_router` | `615-630` | utility model (call at `632-634`, `format="json"`) | First routing pass. Interpolates the loaded document overviews (first 40, `:612-613`) under a `DOCUMENT OVERVIEWS:` header and returns `{"category": "direct_answer"}` or `{"category": "rag_query"}`. | +| `agent.loop.triage_fallback` | `agent/loop.py`, `_triage_query_async` | utility model, `format="json"` | Last-resort routing. Reached only when the overview router returns `None` (no overviews loaded) **and** there is no chat history. Two-way vocabulary: `rag_query` / `direct_answer`; `_normalize_triage()` maps anything else to `rag_query`. | +| `agent.loop.direct_answer` | `331-336` | generation model (streamed at `342-344`) | Answers on the `direct_answer` route from conversation history or general knowledge. Caps the reply at 1-2 sentences. | +| `agent.loop.compose_sub` | `490-515` | generation model (streamed at `519-522`) | Compose one final answer from the JSON list of sub-question/sub-answer pairs. Used when `query_decomposition.compose_from_sub_answers` is true and decomposition produced more than one sub-query. | ## 5. Verifier -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `verifier.fact_check` | `rag_system/agent/verifier.py` `18-58` | Strict JSON-format grounding verifier. | +| ID | File & lines | Model | Purpose | +|----|--------------|-------|---------| +| `verifier.fact_check` | `rag_system/agent/verifier.py` `25-85` | utility model (`loop.py:29`) | Grounding check with three few-shot examples and a `# TASK` block. The prompt is built in four appends: the base f-string ends at `:75`, the context is appended clamped to 4000 characters at `:76`, the answer at `:81`, and the `` tag at `:82-85`. Sent asynchronously with `format="json"`. Verdict labels: `SUPPORTED` / `NOT_SUPPORTED` / `NEEDS_CLARIFICATION`. **Skipped entirely when `VERIFIER_MODEL` / `verification.model` names a local NLI verifier** — that backend makes no LLM call at all (roadmap 2.4, `verifier.md`). | -## 6. Backend Router (Fast path) +## 6. Backend router (fast path) -| ID | File & Lines | Purpose | -|----|--------------|---------| -| `backend.router` | `backend/server.py` `435-448` | Decide "RAG vs direct LLM" before heavy processing. | +| ID | File & lines | Model | Purpose | +|----|--------------|-------|---------| +| `backend.router` | `backend/server.py` `534-560` | `ENRICHMENT_MODEL` (call at `:564-568`, `enable_thinking=False`) | Decides RAG vs direct LLM inside the backend gateway before it calls the RAG API. Unlike the agent routers this one returns **plain text**, not JSON: "Respond with exactly one word: USE_RAG or DIRECT_LLM" (`:560`), substring-matched at `:574-579` with an "unclear ⇒ RAG" default at `:580-582`. | ## 7. Miscellaneous -| ID | File & Lines | Purpose | +| ID | File & lines | Purpose | |----|--------------|---------| -| `vision.placeholder` | `rag_system/utils/ollama_client.py` `169` | Dummy prompt for VLM colour check. | +| `vision.placeholder` | `rag_system/utils/ollama_client.py` `158` | `prompt="What color is this image?"` inside that module's `if __name__ == '__main__'` demo block. Not part of any pipeline. | --- -### Missing / To-Do -1. Verify whether **ReActAgent.PROMPT_TEMPLATE** captures every placeholder – some earlier lines may need explicit ID when we move to central registry. -2. Search TS/JS code once the backend prompts are ported (currently none). - ---- +### Notes -**Next step:** create `rag_system/prompts/registry.yaml` and start moving each prompt above into a key–value entry with identical IDs. Update callers gradually using the helper proposed earlier. \ No newline at end of file +* `rag_system/utils/watsonx_client.py:222` contains the string `prompt="What is AI?"`, but it is literal text inside a `print()` usage banner — it is never sent to a model and is therefore not inventoried. +* There is no prompt registry module; every prompt above is an inline literal at the cited location. +* There is no ReAct-style think/act/observe prompt anywhere. The agent's stages are triage → (optional) decomposition → retrieval → rerank → expand → prune → synthesis → verification. diff --git a/Documentation/quick_start.md b/Documentation/quick_start.md index 3a68f48d..661717de 100644 --- a/Documentation/quick_start.md +++ b/Documentation/quick_start.md @@ -1,34 +1,37 @@ -# ⚡ Quick Start Guide - RAG System +# ⚡ Quick Start Guide - LocalGPT -_Get up and running in 5 minutes!_ +_Get up and running in about 10 minutes (plus model download time)._ --- ## 🚀 Choose Your Deployment Method -### Option 1: Docker Deployment (Production Ready) 🐳 +### Option 1: Docker Deployment 🐳 -Best for: Production deployments, isolated environments, easy scaling +Best for: isolated environments, running the same stack everywhere. -### Option 2: Direct Development (Developer Friendly) 💻 +### Option 2: Direct Development 💻 -Best for: Development, customization, debugging, faster iteration +Best for: development, customization, debugging, faster iteration. + +Both need Ollama. The Docker path runs Ollama on the host by default (better GPU +access), but can also run it as a container. --- ## 🐳 Docker Deployment ### Prerequisites -- Docker Desktop installed and running +- Docker Desktop (or Docker Engine 24+ with the Compose plugin) installed and running - 8GB+ RAM available -- Internet connection +- Internet connection (first build downloads Python and Node packages; first query + downloads the embedding model from HuggingFace; the reranker follows on first use) -### Step 1: Clone and Setup +### Step 1: Clone ```bash -# Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Ensure Docker is running docker version @@ -36,7 +39,7 @@ docker version ### Step 2: Install Ollama Locally -**Even with Docker, Ollama runs locally for better performance:** +**By default the containers talk to Ollama on the host:** ```bash # Install Ollama @@ -46,32 +49,53 @@ curl -fsSL https://ollama.ai/install.sh | sh ollama serve # Install models (in another terminal) -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification ``` +Prefer not to install Ollama on the host? Skip this step and use +`./start-docker.sh container` below. + ### Step 3: Start Docker Containers ```bash -# Start all containers +# Start all containers against local Ollama ./start-docker.sh # Or manually: docker compose --env-file docker.env up --build -d ``` +Containerized Ollama instead: + +```bash +./start-docker.sh container +# Pull the models inside the container the first time +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +`./start-docker.sh` with no argument checks port 11434. If nothing is listening it +offers to switch to the containerized Ollama; pass `-y` (or set `NONINTERACTIVE=1`) +to accept that without a prompt in scripts. + ### Step 4: Verify Deployment ```bash -# Check container status +# Check container status (backend waits for rag-api to report healthy) docker compose ps # Test endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API +curl http://localhost:3000 # Frontend +curl http://localhost:8000/health # Backend +curl http://localhost:8001/health # RAG API ``` +The `rag-api` container loads the agent and the embedding model at startup, so its +health check has a 60s start period and the first `docker compose up` can take +several minutes before `backend` starts. The reranker (~2 GB) is fetched lazily on +the first query that reranks — expect one slow first answer. + ### Step 5: Access Application Open your browser to: **http://localhost:3000** @@ -81,21 +105,20 @@ Open your browser to: **http://localhost:3000** ## 💻 Direct Development ### Prerequisites -- Python 3.8+ -- Node.js 16+ and npm +- Python 3.10+ (3.11 recommended) +- Node.js 20+ and npm - 8GB+ RAM available ### Step 1: Clone and Install Dependencies ```bash -# Clone repository -git clone -cd rag_system_old +git clone https://github.com/PromtEngineer/localGPT.git +cd localGPT # Install Python dependencies pip install -r requirements.txt -# Install Node.js dependencies +# Install Node.js dependencies npm install ``` @@ -109,8 +132,8 @@ curl -fsSL https://ollama.ai/install.sh | sh ollama serve # Install models (in another terminal) -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` ### Step 3: Start the System @@ -120,29 +143,39 @@ ollama pull qwen3:8b python run_system.py ``` -**Or start components manually in separate terminals:** +`run_system.py` reuses an already-running Ollama, pulls any missing model, then +starts the RAG API, the backend and the frontend. + +**Or start components manually in separate terminals — all from the repository root:** ```bash # Terminal 1: RAG API python -m rag_system.api_server # Terminal 2: Backend -cd backend && python server.py +python backend/server.py # Terminal 3: Frontend npm run dev ``` +> Do not `cd backend` first. The SQLite path defaults to `backend/chat_data.db` +> relative to the working directory, so running from inside `backend/` creates a +> second database at `backend/backend/chat_data.db`. + ### Step 4: Verify Installation ```bash -# Check system health +# Loads the models and runs a sample query against the first LanceDB table python system_health_check.py +# HTTP health check per service (exits non-zero if a required one is unhealthy) +python run_system.py --health + # Test endpoints -curl http://localhost:3000 # Frontend -curl http://localhost:8000/health # Backend -curl http://localhost:8001/models # RAG API +curl http://localhost:3000 # Frontend +curl http://localhost:8000/health # Backend +curl http://localhost:8001/health # RAG API ``` ### Step 5: Access Application @@ -158,14 +191,16 @@ Open your browser to: **http://localhost:3000** - Give your session a descriptive name ### 2. Upload Documents -- Click "Create New Index" button -- Upload PDF files from your computer +- Click "Create New Index" +- Upload PDF, DOCX, TXT, MD or HTML files - Configure processing options: - - **Chunk Size**: 512 (recommended) - - **Embedding Model**: Qwen/Qwen3-Embedding-0.6B + - **Chunk Size**: 512 (default) + - **Embedding Model**: `microsoft/harrier-oss-v1-0.6b` (default) - **Enable Enrichment**: Yes - Click "Build Index" and wait for processing +Indexing is synchronous: the request stays open until the pipeline finishes. + ### 3. Start Chatting - Select your built index - Ask questions about your documents: @@ -174,6 +209,15 @@ Open your browser to: **http://localhost:3000** - "What are the main findings?" - "Compare the arguments in section 3 and 5" +The chat settings panel exposes the same knobs the API does: search type +(`hybrid` / `vector_only` / `fts_only`), retrieval_k, reranker top-k, context +window, decomposition, verification, Provence pruning, and "Stream phases". + +> **Stream phases is on by default**, and that path streams straight from the RAG +> API to the browser. Those turns are kept in the agent's memory for follow-up +> questions but are **not** written to the chat history database. Turn it off if you +> want the conversation saved. + --- ## 🔧 Management Commands @@ -182,31 +226,38 @@ Open your browser to: **http://localhost:3000** ```bash # Container management -./start-docker.sh # Start all containers -./start-docker.sh stop # Stop all containers -./start-docker.sh logs # View logs -./start-docker.sh status # Check status +./start-docker.sh # Start (local Ollama) +./start-docker.sh container # Start (containerized Ollama) +./start-docker.sh stop # Stop all containers +./start-docker.sh logs # View logs +./start-docker.sh status # Check status +./start-docker.sh help # Usage # Manual Docker Compose docker compose ps # Check status -docker compose logs -f # Follow logs -docker compose down # Stop containers -docker compose up --build -d # Rebuild and start +docker compose logs -f # Follow logs +docker compose down # Stop containers +docker compose --env-file docker.env up --build -d # Rebuild and start ``` ### Direct Development Commands ```bash # System management -python run_system.py # Start all services -python system_health_check.py # Check system health - -# Individual components -python -m rag_system.api_server # RAG API only -cd backend && python server.py # Backend only -npm run dev # Frontend only - -# Stop: Press Ctrl+C in terminal running services +python run_system.py # Start all services +python run_system.py --mode prod # `npm run build` then `next start` +python run_system.py --no-frontend # Ollama + RAG API + backend only +python run_system.py --health # HTTP health checks +python run_system.py --logs-only # Tail logs/*.log from another shell +python run_system.py --stop # Stop everything in logs/run_system.pid +python system_health_check.py # Deep check: models, LanceDB, sample query + +# Individual components (from the repository root) +python -m rag_system.api_server # RAG API only +python backend/server.py # Backend only +npm run dev # Frontend only + +# Stop: Ctrl+C in the terminal running the services, or `python run_system.py --stop` ``` --- @@ -220,8 +271,9 @@ npm run dev # Frontend only # Check Docker daemon docker version -# Restart Docker Desktop and try again -./start-docker.sh +# The backend will not start until rag-api reports healthy +docker compose ps +docker compose logs -f rag-api ``` **Port conflicts?** @@ -238,7 +290,7 @@ lsof -i :3000 -i :8000 -i :8001 **Import errors?** ```bash # Check Python installation -python --version # Should be 3.8+ +python --version # 3.10+ required, 3.11 recommended # Reinstall dependencies pip install -r requirements.txt --force-reinstall @@ -247,7 +299,7 @@ pip install -r requirements.txt --force-reinstall **Node.js errors?** ```bash # Check Node version -node --version # Should be 16+ +node --version # Should be 20+ # Reinstall dependencies rm -rf node_modules package-lock.json @@ -272,20 +324,25 @@ ollama serve docker stats # For Docker htop # For direct development -# Recommended: 16GB+ RAM for optimal performance +# Use lighter models +export GENERATION_MODEL=qwen3.5:4b +# (the default embedder is already the small one, ~1.2GB) ``` +**Answers ignore my documents?** +Check that the session is linked to a built index — the backend only routes to the +RAG API when the session has one. Sending `"force_rag": true` on +`POST /sessions/{id}/messages` bypasses the router. + --- ## 📊 System Verification -Run this comprehensive check: - ```bash # Check all endpoints curl -f http://localhost:3000 && echo "✅ Frontend OK" -curl -f http://localhost:8000/health && echo "✅ Backend OK" -curl -f http://localhost:8001/models && echo "✅ RAG API OK" +curl -f http://localhost:8000/health && echo "✅ Backend OK" +curl -f http://localhost:8001/health && echo "✅ RAG API OK" curl -f http://localhost:11434/api/tags && echo "✅ Ollama OK" # For Docker: Check containers @@ -298,82 +355,116 @@ docker compose ps If you see: - ✅ All services responding -- ✅ Frontend accessible at http://localhost:3000 +- ✅ Frontend accessible at http://localhost:3000 - ✅ No error messages You're ready to start using LocalGPT! ### What's Next? -1. **📚 Upload Documents**: Add your PDF files to create indexes +1. **📚 Upload Documents**: Add files to create an index 2. **💬 Start Chatting**: Ask questions about your documents -3. **🔧 Customize**: Explore different models and settings -4. **📖 Learn More**: Check the full documentation below +3. **🔧 Customize**: Try `RAG_CONFIG_MODE=fast`, different models, other retrieval modes +4. **📖 Learn More**: Check the documentation below ### 📁 Key Files ``` -rag-system/ +localGPT/ ├── 🐳 start-docker.sh # Docker deployment script ├── 🏃 run_system.py # Direct development launcher -├── 🩺 system_health_check.py # System verification +├── 🩺 system_health_check.py # Deep system verification +├── 🛠️ create_index_script.py # Interactive / batch index creation ├── 📋 requirements.txt # Python dependencies ├── 📦 package.json # Node.js dependencies +├── ⚙️ .env.example # Every environment variable with its default ├── 📁 Documentation/ # Complete documentation -└── 📁 rag_system/ # Core system code +├── 📁 rag_system/ # RAG API, agent, pipelines (config in main.py) +├── 📁 backend/ # Gateway server + SQLite database +└── 📁 src/ # Next.js frontend ``` ### 📖 Additional Resources - **🏗️ Architecture**: See `Documentation/architecture_overview.md` -- **🔧 Configuration**: See `Documentation/system_overview.md` +- **🔧 Configuration**: See `Documentation/system_overview.md` - **🚀 Deployment**: See `Documentation/deployment_guide.md` +- **🐳 Docker**: See `Documentation/docker_usage.md` - **🐛 Troubleshooting**: See `DOCKER_TROUBLESHOOTING.md` --- -**Happy RAG-ing! 🚀** - ---- +## 🛠️ Indexing Without the UI -## 🛠️ Indexing Scripts - -The repository includes several convenient scripts for document indexing: - -### Simple Index Creation Script - -For quick document indexing without the UI: +### Built-in CLI ```bash -# Basic usage -./simple_create_index.sh "Index Name" "document.pdf" +# Index a file or a whole directory with the 'default' profile +python -m rag_system.main index ./my_documents + +# Speed-optimised profile +python -m rag_system.main index ./my_documents --mode fast -# Multiple documents -./simple_create_index.sh "Research Papers" "paper1.pdf" "paper2.pdf" "notes.txt" +# Ask one question and print the JSON result +python -m rag_system.main chat "What are the key findings?" --mode default -# Using wildcards -./simple_create_index.sh "Invoice Collection" ./invoices/*.pdf +# Start the RAG API (same as `python -m rag_system.api_server`) +python -m rag_system.main api --port 8001 ``` -**Supported file types**: PDF, TXT, DOCX, MD +`index` walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, `.md` and `.txt`. +It writes into the profile's shared table (`text_pages_v4`), not into a per-index +table, so indexes built this way are not listed in the web UI. -### Batch Indexing Script +### Interactive / Batch Script -For processing large document collections: +For an index the web UI can see, use `create_index_script.py` — it creates the +database row, uploads the document records and writes to `text_pages_`: ```bash -# Using the Python batch indexing script -python demo_batch_indexing.py - -# Or using the direct indexing script +# Guided prompts python create_index_script.py + +# Write a template, edit it, then run it +python create_index_script.py --create-sample # writes index_config.sample.json +python create_index_script.py --batch index_config.sample.json + +# Use a custom pipeline config instead of PIPELINE_CONFIGS["default"] +python create_index_script.py --config my_pipeline.json +``` + +The batch file looks like this — replace the placeholder paths with your own +absolute paths: + +```json +{ + "index_name": "Sample Batch Index", + "index_description": "Example batch index configuration", + "documents": [ + "/absolute/path/to/first.pdf", + "/absolute/path/to/second.pdf" + ], + "processing": { + "chunk_size": 512, + "enable_enrich": true, + "enable_latechunk": true, + "enable_docling": true, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "retrieval_mode": "hybrid", + "window_size": 2 + } +} ``` -These scripts automatically: -- ✅ Check prerequisites (Ollama running, Python dependencies) -- ✅ Validate document formats -- ✅ Create database entries -- ✅ Process documents with the RAG pipeline -- ✅ Generate searchable indexes +Both paths: +- ✅ Parse documents with Docling (OCR fallback for scanned PDFs) +- ✅ Chunk, optionally enrich, and embed +- ✅ Write vectors plus a native full-text index into LanceDB +- ✅ Generate a document overview used by the query router + +The script exits non-zero on failure and deletes the half-created index row. + +--- ---- \ No newline at end of file +**Happy RAG-ing! 🚀** diff --git a/Documentation/retrieval_pipeline.md b/Documentation/retrieval_pipeline.md index d4ae7168..9dff9d8e 100644 --- a/Documentation/retrieval_pipeline.md +++ b/Documentation/retrieval_pipeline.md @@ -1,616 +1,320 @@ # 📥 Retrieval Pipeline -_Maps to `rag_system/pipelines/retrieval_pipeline.py` and helpers in `retrieval/`, `rerankers/`._ +_Maps to `rag_system/pipelines/retrieval_pipeline.py`, orchestrated by `rag_system/agent/loop.py`, with helpers in `retrieval/` and `rerankers/`._ ## Role -Given a **user query** and one or more indexed tables, retrieve the most relevant text chunks and synthesise an answer. +Given a user query and one LanceDB table, retrieve the most relevant chunks and synthesise an answer with its source documents. + +Two objects share the work: + +* **`Agent`** (`rag_system/agent/loop.py`) owns triage, the semantic cache, query decomposition, sub-answer composition, conversation history and verification. +* **`RetrievalPipeline`** (`rag_system/pipelines/retrieval_pipeline.py`) owns everything from retrieval to synthesis for a *single* query string. `Agent` may call it once, or once per sub-query in parallel. ## Sub-components -| Stage | Module | Key Classes / Fns | Notes | -|-------|--------|-------------------|-------| -| Query Pre-processing | `retrieval/query_transformer.py` | `QueryTransformer`, `HyDEGenerator`, `GraphQueryTranslator` | Expands, rewrites, or translates the raw query. | -| Retrieval | `retrieval/retrievers.py` | `BM25Retriever`, `DenseRetriever`, `HybridRetriever` | Abstract over LanceDB vector + FTS search. | -| Reranking | `rerankers/reranker.py` | `ColBERTSmall`, fallback `bge-reranker` | Optionally improves result ordering. | -| Synthesis | `pipelines/retrieval_pipeline.py` | `_synthesize_final_answer()` | Calls LLM with evidence snippets. | -## End-to-End Flow +| Stage | Module | Key classes / functions | Notes | +|-------|--------|-------------------------|-------| +| Query decomposition | `retrieval/query_transformer.py` | `QueryDecomposer.decompose()` | Optional. Splits a query into standalone sub-queries; runs on the utility model. | +| Retrieval | `retrieval/retrievers.py` | `MultiVectorRetriever.retrieve()` | Runs LanceDB full-text and/or vector search over one table. There is no separate BM25 retriever class — lexical search is LanceDB's native FTS index. | +| Reranking | `pipelines/retrieval_pipeline.py`, `rerankers/reranker.py` | `_get_ai_reranker()`, `QwenRerankerScorer`, `rerankers.Reranker`, `CrossEncoderReranker` | Off by default. Qwen3-Reranker names go to `QwenRerankerScorer`; otherwise the model is loaded through the `rerankers` library. `CrossEncoderReranker` is the non-library fallback branch. | +| Sentence pruning | `rerankers/sentence_pruner.py` | `SentencePruner.prune_documents()` | Provence (`naver/provence-reranker-debertav3-v1`). Off unless requested. | +| Synthesis | `pipelines/retrieval_pipeline.py` | `_synthesize_final_answer()` | Streams an LLM completion on the generation model. | +| Verification | `agent/verifier.py` | `Verifier.verify_async()` | See `verifier.md`. | + +### Removed: the graph path + +`GraphQueryTranslator`, `GraphRetriever` and `GraphExtractor` were **deleted on +2026-08-09** (roadmap item 2.5). They were unreachable — no shipped profile ever +set `graph_strategy` — and the evidence is against re-adding them: GraphRAG +*loses* on single-hop retrieval, its multi-hop gains range from +3 points to +27 +depending on how well the vector baseline is tuned, and it costs **41–57× at +indexing** and up to **~377× in query tokens**. See +[`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) §6. +`networkx` and `fuzzywuzzy` left `requirements.txt` with them. + +## End-to-end flow ```mermaid -flowchart LR - Q["User Query"] --> XT["Query Transformer"] - XT -->|variants| RETRIEVE - subgraph Retrieval - RET_BM25[BM25] --> MERGE - RET_DENSE[Dense Vector] --> MERGE - style RET_BM25 fill:#444,stroke:#ccc,color:#fff - style RET_DENSE fill:#444,stroke:#ccc,color:#fff +flowchart TD + Q["User query"] --> T["Triage (see triage_system.md)"] + T -- direct_answer --> DA["Direct LLM answer"] + T -- rag_query --> C{"Semantic cache hit?"} + C -- yes --> OUT["answer + source_documents"] + C -- no --> D{"Decomposition enabled?"} + D -- yes --> SUB["1..N sub-queries in parallel"] + D -- no --> ONE["Single query"] + SUB --> RP + ONE --> RP + subgraph RP [RetrievalPipeline.run] + R1["Retrieve (hybrid RRF / vector_only / fts_only)"] --> R2["Late-chunk table + sibling merge (optional)"] + R2 --> R3["AI rerank (optional, off by default)"] + R3 --> R4["Context expansion (optional)"] + R4 --> R5["Provence pruning (optional)"] + R5 --> R6["Synthesis (streamed)"] end - MERGE --> RERANK - RERANK --> K[["Top-K Chunks"]] - K --> SYNTH["Answer Synthesiser\n(LLM)"] - SYNTH --> A["Answer + Sources"] + RP --> COMP["Compose sub-answers (only when decomposed)"] + COMP --> V["Verification (optional)"] + DA -- "no source_documents, skipped" --> V + V --> OUT ``` -### Narrative -1. **Query Transformer** may expand the query (keyword list, HyDE doc, KG translation) depending on `searchType`. -2. **Retrievers** execute BM25 and/or dense similarity against LanceDB. Combination controlled by `retrievalMode` and `denseWeight`. -3. **Reranker** (if `aiRerank=true` or hybrid search) scores snippets; top `rerankerTopK` chosen. -4. **Synthesiser** streams an LLM completion using the prompt described in `prompt_inventory.md` (`retrieval_pipeline.synth_final`). - -## Configuration Flags (passed from UI → backend) -| Flag | Default | Effect | -|------|---------|--------| -| `searchType` | `fts` | UI label (FTS / Dense / Hybrid). | -| `retrievalK` | 10 | Initial candidate count per retriever. | -| `contextWindowSize` | 5 | How many adjacent chunks to merge (late-chunk). | -| `rerankerTopK` | 20 | How many docs to pass into AI reranker. | -| `denseWeight` | 0.5 | When `hybrid`, linear mix weight. | -| `aiRerank` | bool | Toggle reranker. | -| `verify` | bool | If true, pass answer to **Verifier** component. | +## Stage detail -## Interfaces -* Reads from **LanceDB** tables `text_pages_`. -* Calls **Ollama** generation model specified in `PIPELINE_CONFIGS`. -* Exposes `RetrievalPipeline.answer_stream()` iterator consumed by SSE API. +### 1. Retrieval — `MultiVectorRetriever.retrieve()` (`retrievers.py:91-222`) -## Extension Points -* Plug new retriever by inheriting `BaseRetriever` and registering in `retrievers.py`. -* Swap reranker model via `EXTERNAL_MODELS['reranker_model']`. -* Custom answer prompt can be overridden by passing `prompt_override` to `_synthesize_final_answer()` (not yet surfaced in UI). +```python +retrieve(text_query: str, table_name: str, k: int, search_type: str = "hybrid") -> List[Dict] +``` -## Detailed Implementation Analysis +`search_type` selects which LanceDB legs run. Unknown values log a warning and degrade to `hybrid` (`retrievers.py:99-102`). -### Core Architecture Pattern -The `RetrievalPipeline` uses **lazy initialization** for all components to avoid heavy memory usage during startup. Each component (embedder, retrievers, rerankers) is only loaded when first accessed via private `_get_*()` methods. +| Mode | Legs | Score returned | +|------|------|----------------| +| `hybrid` (default) | FTS and vector, run concurrently on a 2-worker thread pool (`retrievers.py:139-143`), fused by reciprocal rank fusion | RRF score | +| `vector_only` | vector only | `1 / (1 + distance)` | +| `fts_only` | FTS only | LanceDB's BM25 score | -```python -def _get_text_embedder(self): - if self.text_embedder is None: - self.text_embedder = select_embedder( - self.config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B"), - self.ollama_config.get("host") - ) - return self.text_embedder -``` +Details: -### Thread Safety Implementation -**Critical Issue**: ColBERT reranker and model loading are not thread-safe. The system uses multiple locks: +* **FTS leg** — `tbl.search(query=..., query_type="fts").limit(k)`. Single-word queries are rewritten to `"* OR ~"` to add prefix and fuzzy matching (`retrievers.py:117-118`). This is LanceDB's built-in full-text index (created at index time, see `indexing_pipeline.md`) — no SQLite, no Porter stemming, no configurable stop-word or n-gram handling. +* **Vector leg** — the query is embedded once and memoised in a 256-entry `lru_cache` per retriever instance, then `tbl.search(vector).limit(k)`. Before searching, the retriever reads the table's embedder marker: a table written by a different embedding model raises `EmbedderMismatchError` (which is deliberately *not* swallowed by the catch-all below), and the query vector is L2-normalized only when the table's own vectors are, so LanceDB's default L2 ordering matches the cosine ordering the model cards specify. A table with no marker is searched the legacy, unnormalized way with a one-time warning. +* **Fusion** — each leg contributes `1 / (60 + rank)` (`_RRF_K = 60`, `retrievers.py:20`). Rows are deduplicated on `chunk_id`, falling back to `_rowid` then `text` (`retrievers.py:83-89`), summed, sorted, and truncated to `k`. There is no weighted linear blend and no `dense_weight` knob. +* Each leg fetches `k` rows, so hybrid examines up to `2k` candidates and returns `k`. +* Every returned doc carries a finite, higher-is-better `score`. The raw per-leg values `bm25` and `_distance` are attached **only** when that leg actually hit the row. +* `metadata` is accepted as either a dict or a JSON string; `text` falls back through `metadata.original_text` → `row.text` → `""` (`retrievers.py:179-202`). +* On any exception other than `EmbedderMismatchError` the method logs and returns `[]`, so a missing table degrades to zero results rather than an error. An embedder mismatch propagates instead — a wrong-model answer is worse than an error. -```python -# Global locks to prevent race conditions -_rerank_lock: Lock = Lock() # Protects .rank() calls -_ai_reranker_init_lock: Lock = Lock() # Prevents concurrent model loading -_sentence_pruner_lock: Lock = Lock() # Serializes Provence model init -``` +### 2. Late chunking at query time (`retrieval_pipeline.py:289-341`) -When multiple queries run in parallel, only one thread can initialize heavy models or perform reranking operations. +When the merged late-chunk config is enabled, two extra things happen: -### Retrieval Strategy Deep-Dive +1. A second `retrieve()` runs against the late-chunk table and its hits are appended to the candidate list (`:293-305`). The table name is `latechunk.lancedb_table_name` if set, otherwise `
` with `table_suffix` defaulting to `_lc` (`:85-92`) — the same name `IndexingPipeline` writes. +2. **Sibling merging** (`:316-341`): for every retrieved chunk, the ±1 neighbouring chunks from the same document are fetched from the main table and their text is concatenated into that chunk's `text`. This runs off the config flag alone, whether or not the `_lc` table exists. -#### 1. Multi-Vector Dense Retrieval (`_get_dense_retriever()`) -```python -self.dense_retriever = MultiVectorRetriever( - db_manager, # LanceDB connection - text_embedder, # Qwen3-Embedding embedder - vision_model=None, # Optional multimodal - fusion_config={} # Score combination rules -) -``` +The config block is merged across both container names and both spellings — `retrieval.late_chunking`, `retrieval.latechunk`, `retrievers.late_chunking`, `retrievers.latechunk`, later writes winning (`:70-83`) — so a profile setting and a runtime API override can come from different places and both apply. -**Process**: -1. Query → embedding vector (1024D for Qwen3-Embedding-0.6B) -2. LanceDB ANN search using IVF-PQ index -3. Cosine similarity scoring -4. Returns top-K with metadata +The `default` profile enables late chunking at query time (`main.py:57-59`), while the RAG API's `/index` endpoint defaults `enable_latechunk` to `false` (`api_server.py:410`). Unless you explicitly index with late chunking on, the `_lc` table does not exist: the extra `retrieve()` logs "Could not search table …" and returns nothing, while sibling merging still applies. + +### 2b. Evidence-sufficiency retry (`RetrievalPipeline.retrieve_candidates`) + +_Roadmap item 2.1, shipped 2026-08-09. On in the `default` profile, off in `fast`._ + +One conditional second retrieval, and only one — the evidence for iterative +retrieval says iteration 1→2 captures ~95% of the gains and iterations ≥3 are +noise. It wraps the first stage **and** the rerank stage, so it sees the ranking +the caller would actually have got. -#### 2. BM25 Full-Text Search (`_get_bm25_retriever()`) ```python -# Uses SQLite FTS5 under the hood -SELECT chunk_id, text, bm25(fts_table) as score -FROM fts_table -WHERE fts_table MATCH ? -ORDER BY bm25(fts_table) -LIMIT ? +"retrieval": {"retry": {"enabled": True, "min_top_score": 0.12, "max_attempts": 1}} ``` -**Token Processing**: -- Stemming via Porter algorithm -- Stop-word removal -- N-gram tokenization (configurable) +**The signal.** Not the raw top similarity. That was measured on the gold set and +is *anti*-correlated with success: the three `mixed` first-stage misses each +scored a **higher** top cosine than the median successful query, because absolute +similarity mostly encodes how close a query's phrasing sits to the corpus's +register. What carries signal is contrast — how far the best candidate stands +above everything else the query dragged in: -#### 3. Hybrid Score Fusion -When both retrievers are enabled: -```python -final_score = (1 - dense_weight) * bm25_score + dense_weight * dense_score ``` -Default `dense_weight = 0.7` favors semantic over lexical matching (updated from 0.5). +evidence = (cos_top − cos_background) / (1 − cos_background) +``` -### Late-Chunk Merging Algorithm +`cos_background` is the mean cosine of candidates from rank 6 down; the +denominator rescales against this query's reachable headroom, keeping the result +in 0–1 and comparable across queries. Cosine comes from LanceDB's squared-L2 +`_distance` on the L2-normalized v4 tables (`cos = 1 − d/2`), so the retry is only +armed on a normalized table — on a legacy table, or in `fts_only` mode, the score +is `None` and nothing fires. **RRF scores are never used**: they encode rank, not +confidence, and are near-identical for every query. -**Problem**: Small chunks lose context; large chunks dilute relevance. -**Solution**: Retrieve small chunks, then expand with neighbors. +When reranking is on, the top reranker score is preferred instead, but only when +it is a genuine 0–1 probability (`QwenRerankerScorer` returns P("yes")); an +arbitrary logit is rejected rather than compared against a probability threshold. +The threshold for that path is `min_rerank_score`, defaulting to `min_top_score`. -```python -def _get_surrounding_chunks_lancedb(self, chunk, window_size): - start_index = max(0, chunk_index - window_size) - end_index = chunk_index + window_size - - sql_filter = f"document_id = '{document_id}' AND chunk_index >= {start_index} AND chunk_index <= {end_index}" - results = tbl.search().where(sql_filter).to_list() - - # Sort by chunk_index to maintain document order - return sorted(results, key=lambda x: x.get("chunk_index", 0)) -``` +**What happens on a fire.** `_reformulate_query()` makes one `format="json"` call +on the **enrichment** model asking for a rewrite in the vocabulary a document +would use, the whole first stage + rerank runs again on it, and **the better of +the two result sets by the same score is kept** — a retry that does not improve +the evidence is discarded, never merged. A `retrieval_retry` event goes out +through `event_callback`, so the RAG API's SSE stream carries it and the UI +cascade shows a "Rechecking weak evidence" step. -**Benefits**: -- Maintains granular search precision -- Provides richer context for answer generation -- Configurable window size (default: 5 chunks = ~2500 tokens) +**Measured** (`../eval/decisions/phase2-pipeline.md`): fires on **9.7% of `mixed` +queries** (7/72) and 20.8% of `docs`, and moved `mixed` first-stage nDCG@10 from +0.889 to 0.901/0.906 across two runs with **zero per-query regressions**. It +repaired `docs_d16`, a genuine recall@10 miss. -### AI Reranker Implementation +### 3. AI reranking (`retrieval_pipeline.py`, `_rerank_stage`) -#### ColBERT Strategy (via rerankers-lib) -```python -from rerankers import Reranker -self.ai_reranker = Reranker("answerdotai/answerai-colbert-small-v1", model_type="colbert") +Loaded lazily behind `_ai_reranker_init_lock` so only one thread performs the heavy `from_pretrained()`; the `.rank()` call itself is serialised behind `_rerank_lock` because the `rerankers` backends are not thread-safe. -# Usage -scores = reranker.rank(query, [doc.text for doc in candidates]) -``` +Config keys read from `reranker`: -**ColBERT Architecture**: -- **Query encoding**: Each token → 128D vector -- **Document encoding**: Each token → 128D vector -- **Interaction**: MaxSim between all query-doc token pairs -- **Advantage**: Fine-grained token-level matching +| Key | Default in code | `default` profile | +|-----|-----------------|-------------------| +| `enabled` | falsy | **`false`** — reranking is off by default ([`../eval/DECISIONS.md`](../eval/DECISIONS.md)) | +| `model_name` | none — missing logs a warning and skips reranking (`:145-148`) | `EXTERNAL_MODELS["reranker_model"]` = `Qwen/Qwen3-Reranker-4B`, loaded only if `enabled` is turned on | +| `strategy` | `rerankers-lib` | `rerankers-lib` | +| `model_type` | `cross-encoder` | not set ⇒ `cross-encoder` | +| `top_k` | all retrieved docs | `10` | +| `top_percent` | unset | unset — when set (0 < p ≤ 1) it overrides `top_k` with `max(1, len(docs) * p)` | -#### Fallback: BGE Cross-Encoder -```python -# When ColBERT fails/unavailable -from sentence_transformers import CrossEncoder -model = CrossEncoder('BAAI/bge-reranker-base') -scores = model.predict([(query, doc.text) for doc in candidates]) -``` +A model whose `model_type` is `qwen3`, or whose name contains `qwen3-reranker`, is routed to the in-repo `QwenRerankerScorer` — the `rerankers` library builds Qwen3-Reranker with a randomly initialised score head, so this route is what makes the default reranker model correct rather than merely loadable. Otherwise `strategy: "rerankers-lib"` loads `rerankers.Reranker(model_name, model_type=model_type)`, and any other value constructs the local `CrossEncoderReranker` (`rerankers/reranker.py:5`), an `AutoModelForSequenceClassification` cross-encoder with batched scoring and an early-exit heuristic. -### Answer Synthesis Pipeline +**If the reranker fails to load, the pipeline logs `⚠️ Could not load reranker '' (). Continuing without reranking.` and continues with the unranked candidates.** There is no second fallback model. -#### Prompt Engineering Pattern -```python -def _synthesize_final_answer(self, query: str, facts: str, *, event_callback=None): - prompt = f""" -You are an AI assistant specialised in answering questions from retrieved context. - -Context you receive -• VERIFIED FACTS – text snippets retrieved from the user's documents. -• ORIGINAL QUESTION – the user's actual query. - -Instructions -1. Evaluate each snippet for relevance to the ORIGINAL QUESTION -2. Synthesise an answer **using only information from relevant snippets** -3. If snippets contradict, mention the contradiction explicitly -4. If insufficient information: "I could not find that information in the provided documents." -5. Provide thorough, well-structured answer with relevant numbers/names -6. Do **not** introduce external knowledge - -––––– Retrieved Snippets ––––– -{facts} -–––––––––––––––––––––––––––––– - -ORIGINAL QUESTION: "{query}" -""" - - response = self.llm_client.complete_stream( - prompt=prompt, - model=self.ollama_config["generation_model"] # qwen3:8b - ) - - for chunk in response: - if event_callback: - event_callback({"type": "answer_chunk", "content": chunk}) - yield chunk -``` +#### Decomposition applies here, not at the first stage -**Advanced Features**: -- **Source Attribution**: Automatic citation generation -- **Confidence Scoring**: Based on retrieval scores and snippet relevance -- **Answer Verification**: Optional grounding check via Verifier component +_Roadmap item 2.2, shipped 2026-08-09._ -### Query Processing and Transformation +Decomposing the **first stage** dilutes it semantically; the 2026 evidence +(MultiConIR/SSRB) puts the win at the reranking stage instead. So: -#### Query Decomposition -```python -class QueryDecomposer: - def decompose_query(self, query: str) -> List[str]: - """Break complex queries into simpler sub-queries.""" - decomposition_prompt = f""" - Break down this complex question into 2-4 simpler sub-questions that would help answer the original question. - - Original question: {query} - - Sub-questions: - 1. - 2. - 3. - 4. - """ - - response = self.llm_client.complete( - prompt=decomposition_prompt, - model=self.enrichment_model # qwen3:0.6b for speed - ) - - # Parse response into list of sub-queries - return self._parse_subqueries(response) -``` +* The first stage **always** runs once, on the full original query. +* When sub-queries are supplied *and* the reranker is on, every candidate is + scored against **every** sub-query and the per-sub-query scores are combined + with `query_decomposition.rerank_aggregate` — `"mean"` (default) or `"max"`. + `mean` is the default because it measured better than `max` on both halves of + the A/B ([`../eval/decisions/phase2-pipeline.md`](../eval/decisions/phase2-pipeline.md) §3). +* When the reranker is off — the shipped default — there is no rerank stage, so + the sub-queries are simply unused and this is plain single-query retrieval. -#### HyDE (Hypothetical Document Embeddings) -```python -class HyDEGenerator: - def generate_hypothetical_doc(self, query: str) -> str: - """Generate hypothetical document that would answer the query.""" - hyde_prompt = f""" - Generate a hypothetical document passage that would perfectly answer this question: - - Question: {query} - - Hypothetical passage: - """ - - response = self.llm_client.complete( - prompt=hyde_prompt, - model=self.enrichment_model - ) - - return response.strip() -``` +One path still fans the first stage out over sub-queries, and it is gated behind +its own pre-existing flag: `query_decomposition.compose_from_sub_answers` (true +in the `default` profile). That path needs a separate *answer* per sub-question +to compose from, which a single shared candidate set cannot produce, so it runs a +full `RetrievalPipeline.run()` per sub-query in parallel. Exactly what runs: -### Caching and Performance Optimization +| `query_decomposition` | reranker | First stage | Rerank scored against | +|---|---|---|---| +| off | either | once, full query | full query | +| on, `compose_from_sub_answers: true` (profile default) | either | **once per sub-query**, in parallel | that sub-query | +| on, `compose_from_sub_answers: false` | on | once, full query | **all sub-queries, aggregated** | +| on, `compose_from_sub_answers: false` | off | once, full query | — (no rerank stage; sub-queries unused) | +| on, one sub-query after decomposition | either | once, the resolved query | the resolved query | -#### Semantic Query Caching -```python -class RetrievalPipeline: - def __init__(self, config, ollama_client, ollama_config): - # TTL cache for embeddings and results - self.query_cache = TTLCache(maxsize=100, ttl=300) # 5 min TTL - self.embedding_cache = LRUCache(maxsize=500) - self.semantic_threshold = 0.98 # Similarity threshold for cache hits - - def get_cached_result(self, query: str, session_id: str = None) -> Optional[Dict]: - """Check for semantically similar cached queries.""" - query_embedding = self._get_text_embedder().create_embeddings([query])[0] - - for cached_query, cached_data in self.query_cache.items(): - cached_embedding = cached_data["embedding"] - similarity = cosine_similarity([query_embedding], [cached_embedding])[0][0] - - if similarity > self.semantic_threshold: - # Check session scope if configured - if self.cache_scope == "session" and cached_data.get("session_id") != session_id: - continue - - print(f"🎯 Cache hit: {similarity:.3f} similarity") - return cached_data["result"] - - return None -``` +### 4. Context expansion (`retrieval_pipeline.py:400-441`) -#### Batch Processing Optimizations -```python -def process_query_batch(self, queries: List[str]) -> List[Dict]: - """Process multiple queries efficiently.""" - # Batch embed all queries - query_embeddings = self._get_text_embedder().create_embeddings(queries) - - # Batch search - results = [] - for i, query in enumerate(queries): - embedding = query_embeddings[i] - - # Search with pre-computed embedding - dense_results = self._search_dense_with_embedding(embedding) - bm25_results = self._search_bm25(query) - - # Combine and rerank - combined = self._combine_results(dense_results, bm25_results) - reranked = self._rerank_batch([query], [combined])[0] - - results.append(reranked) - - return results -``` +When the effective window size is greater than 0, each surviving doc is expanded with its neighbours from the same document via a metadata-only LanceDB filter (`document_id = ... AND chunk_index BETWEEN ...`), run across a thread pool. The union is deduplicated on `chunk_id` and re-sorted by `rerank_score`, then `_distance`, then `score`, then document order. -### Advanced Search Features +Caveat worth knowing: immediately afterwards, `if any('rerank_score' in d for d in final_docs): final_docs = [d for d in final_docs if 'rerank_score' in d]` (`:445-446`). Only the reranked seed chunks carry a `rerank_score`, so **when the AI reranker ran, the freshly added neighbours are filtered back out**. Context expansion therefore only changes the context when reranking is disabled — which, since the Phase 1 adoption, is the default. -#### Conversational Context Integration -```python -def answer_with_history(self, query: str, conversation_history: List[Dict], **kwargs): - """Answer query with conversation context.""" - # Build conversational context - context_prompt = self._build_conversation_context(conversation_history) - - # Expand query with context - expanded_query = f"{context_prompt}\n\nCurrent question: {query}" - - # Process with expanded context - return self.answer_stream(expanded_query, **kwargs) - -def _build_conversation_context(self, history: List[Dict]) -> str: - """Build context from conversation history.""" - context_parts = [] - - for turn in history[-3:]: # Last 3 turns for context - if turn.get("role") == "user": - context_parts.append(f"Previous question: {turn['content']}") - elif turn.get("role") == "assistant": - # Extract key points from previous answers - context_parts.append(f"Previous context: {turn['content'][:200]}...") - - return "\n".join(context_parts) -``` +### 5. Provence sentence pruning (`retrieval_pipeline.py:448-462`) -#### Multi-Index Search -```python -def search_multiple_indexes(self, query: str, index_ids: List[str], **kwargs): - """Search across multiple document indexes.""" - all_results = [] - - for index_id in index_ids: - table_name = f"text_pages_{index_id}" - - try: - # Search individual index - index_results = self._search_single_index(query, table_name, **kwargs) - - # Add index metadata - for result in index_results: - result["source_index"] = index_id - - all_results.extend(index_results) - - except Exception as e: - print(f"⚠️ Error searching index {index_id}: {e}") - continue - - # Global reranking across all indexes - if len(all_results) > kwargs.get("retrieval_k", 20): - all_results = self._rerank_global(query, all_results, **kwargs) - - return all_results -``` +Runs between context expansion and synthesis when `provence.enabled` is set. Loads `naver/provence-reranker-debertav3-v1` once behind `_sentence_pruner_lock`, drops sentences scoring below `provence.threshold` (default `0.1`, `:455`), and then removes any chunk whose text was pruned to nothing (`:460`). If the model cannot be downloaded or loaded, `SentencePruner` logs and `prune_documents()` echoes its input unchanged. -### Error Handling and Resilience +No shipped profile contains a `provence` block, so pruning is off unless a request enables it. -#### Graceful Degradation -```python -def answer_stream(self, query: str, **kwargs): - """Main answer method with comprehensive error handling.""" - try: - # Try full pipeline - return self._answer_stream_full_pipeline(query, **kwargs) - - except Exception as e: - print(f"⚠️ Full pipeline failed: {e}") - - try: - # Fallback: Dense-only search - kwargs["search_type"] = "dense" - kwargs["ai_rerank"] = False - return self._answer_stream_fallback(query, **kwargs) - - except Exception as e2: - print(f"⚠️ Fallback failed: {e2}") - - # Last resort: Direct LLM answer - return self._direct_llm_answer(query) - -def _direct_llm_answer(self, query: str): - """Direct LLM answer as last resort.""" - prompt = f""" - The document retrieval system is temporarily unavailable. - Please provide a helpful response acknowledging this limitation. - - User question: {query} - - Response: - """ - - response = self.llm_client.complete_stream( - prompt=prompt, - model=self.ollama_config["generation_model"] - ) - - yield "⚠️ Document search unavailable. Providing general response:\n\n" - - for chunk in response: - yield chunk -``` +### 6. Synthesis (`retrieval_pipeline.py:223-261, 502-514`) -#### Recovery Mechanisms -```python -def recover_from_embedding_failure(self, query: str, **kwargs): - """Recover when embedding model fails.""" - print("🔄 Attempting embedding model recovery...") - - # Try to reinitialize embedder - try: - self.text_embedder = None # Clear failed instance - embedder = self._get_text_embedder() # Reinitialize - - # Test with simple query - test_embedding = embedder.create_embeddings(["test"]) - - if test_embedding is not None: - print("✅ Embedding model recovered") - return True - - except Exception as e: - print(f"❌ Recovery failed: {e}") - - # Fallback to BM25-only search - kwargs["search_type"] = "bm25" - kwargs["ai_rerank"] = False - print("🔄 Falling back to keyword search only") - - return False -``` +The surviving chunk texts are joined with blank lines and passed to `_synthesize_final_answer()`, which streams a completion on the **generation** model. Each token is pushed to `event_callback("token", {"text": ...})` — synthesis is push-based, not an iterator. -### Performance Monitoring and Metrics +Before serialisation, `vector` and `_distance` are removed from every doc and NaN/Inf floats are nulled (`:479-500`). The return value is: -#### Query Performance Tracking -```python -class PerformanceTracker: - def __init__(self): - self.metrics = { - "query_count": 0, - "avg_response_time": 0, - "cache_hit_rate": 0, - "error_rate": 0, - "embedding_time": 0, - "retrieval_time": 0, - "reranking_time": 0, - "synthesis_time": 0 - } - - @contextmanager - def track_query(self, query: str): - """Context manager for tracking query performance.""" - start_time = time.time() - - try: - yield - - # Success metrics - duration = time.time() - start_time - self.metrics["query_count"] += 1 - self.metrics["avg_response_time"] = ( - (self.metrics["avg_response_time"] * (self.metrics["query_count"] - 1) + duration) - / self.metrics["query_count"] - ) - - except Exception as e: - # Error metrics - self.metrics["error_rate"] = ( - self.metrics["error_rate"] * self.metrics["query_count"] + 1 - ) / (self.metrics["query_count"] + 1) - - raise e - - finally: - self.metrics["query_count"] += 1 +```jsonc +{ "answer": "...", "source_documents": [ /* chunk dicts */ ] } ``` -#### Resource Usage Monitoring -```python -def monitor_memory_usage(self): - """Monitor memory usage of pipeline components.""" - import psutil - import gc - - process = psutil.Process() - memory_info = process.memory_info() - - print(f"Memory Usage: {memory_info.rss / 1024 / 1024:.1f} MB") - - # Component-specific monitoring - if hasattr(self, 'text_embedder') and self.text_embedder: - print(f"Embedder loaded: {type(self.text_embedder).__name__}") - - if hasattr(self, 'ai_reranker') and self.ai_reranker: - print(f"Reranker loaded: {type(self.ai_reranker).__name__}") - - # Suggest cleanup if memory usage is high - if memory_info.rss > 8 * 1024 * 1024 * 1024: # 8GB - print("⚠️ High memory usage detected - consider cleanup") - gc.collect() -``` +with `{"answer": "I could not find an answer in the documents.", "source_documents": []}` when nothing survives (`:476-477`). ---- +There are no inline citation markers. Sources are returned as the `source_documents` array and rendered by the UI as a collapsible list. -## Configuration Reference +### 7. Semantic cache (`agent/loop.py:130-154, 305-324, 587-594`) -### Default Pipeline Configuration -```python -RETRIEVAL_CONFIG = { - "retriever": "multivector", - "search_type": "hybrid", - "retrieval_k": 20, - "reranker_top_k": 10, - "dense_weight": 0.7, - "late_chunking": { - "enabled": True, - "window_size": 5 - }, - "ai_rerank": True, - "verify_answers": False, - "cache_enabled": True, - "cache_ttl": 300, - "semantic_cache_threshold": 0.98 -} -``` +Owned by `Agent`, not by the pipeline: -### Model Configuration -```python -MODEL_CONFIG = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:8b", - "enrichment_model": "qwen3:0.6b", - "reranker_model": "answerdotai/answerai-colbert-small-v1", - "fallback_reranker": "BAAI/bge-reranker-base" -} -``` +* `TTLCache(maxsize=100, ttl=300)` (`loop.py:33`) keyed by raw query text, storing `{embedding, result, session_id}`. +* Looked up by cosine similarity against the freshly embedded query; a hit requires similarity ≥ `semantic_cache_threshold` (`0.98` in both profiles). +* `cache_scope` defaults to `"session"`: entries from a different `session_id` are skipped (`loop.py:141`). Set it to `"global"` to share cached answers across sessions — note that this can return one session's document-derived answer to another. +* Skipped entirely on the `direct_answer` route. +* Per-query embeddings are additionally memoised by the retriever's own 256-entry `lru_cache`. -### Performance Tuning -```python -PERFORMANCE_CONFIG = { - "batch_sizes": { - "embedding": 32, - "reranking": 16, - "synthesis": 1 - }, - "timeouts": { - "embedding": 30, - "retrieval": 60, - "reranking": 30, - "synthesis": 120 - }, - "memory_limits": { - "max_cache_size": 1000, - "max_results_per_query": 100, - "chunk_size_limit": 2048 - } -} -``` +## Configuration flags -## Extension Examples +Both camelCase and snake_case are accepted; the RAG API normalises them to snake_case once at parse time (`api_server.py:51-76`). + +| Wire field | RAG API default when absent | `default` profile | `fast` profile | Effect | +|------------|-----------------------------|-------------------|----------------|--------| +| `retrieval_mode` (alias `search_type`) | not set ⇒ profile value | `hybrid` | `vector_only` | `hybrid` / `vector_only` / `fts_only`. Anything else is rejected with HTTP 400 (`api_server.py:161-168`). | +| `retrieval_k` | `20` | `20` | `10` | Rows fetched per leg, and the size of the fused candidate list. | +| `reranker_top_k` | `10` | `10` (only applies when reranking is switched on) | reranker off | Docs kept after reranking. | +| `context_window_size` | `1` | `0` | `0` | Neighbouring chunks merged around each hit. Because the API always sends a value, HTTP traffic effectively uses `1`; the profile's `0` applies only to direct programmatic use. | +| `ai_rerank` | not sent ⇒ profile value | reranker **disabled** | reranker disabled | Toggles `reranker.enabled`. | +| `query_decompose` | not sent ⇒ profile value | `true` | `false` | Toggles `query_decomposition.enabled`. | +| `compose_sub_answers` | not sent ⇒ profile value | `true` | — | Compose one answer from sub-answers vs. aggregating all sub-query documents into a single synthesis. | +| `context_expand` | not sent | — | — | `false` forces `window_size_override=0` for this request. | +| `verify` | not sent ⇒ profile value | `true` | `false` | See `verifier.md`. | +| `force_rag` | `false` | — | — | Skip triage, force the RAG path; all other toggles still apply. | +| `provence_prune` | not sent ⇒ disabled | absent | absent | Enable Provence sentence pruning. | +| `provence_threshold` | not sent ⇒ `0.1` | absent | absent | Pruning threshold. Not exposed in the UI. | +| `model` | not sent | — | — | Per-request generation model. Applied through a context manager that restores the previous value afterwards, and ignored with a warning when the id does not suit the active backend (`api_server.py:88-113`). | + +UI defaults (`src/components/ui/session-chat.tsx:44-60`): compose `true`, decompose `true`, aiRerank `false`, contextExpand `true`, stream `true`, verify `true`, forceDocs `false`, provencePrune `false`, retrievalK `20`, contextWindowSize `1`, rerankerTopK `10`, searchType `hybrid`. + +There is no `dense_weight` / `denseWeight` knob anywhere in the stack, and no `fusion` config block. + +## Entry points -### Custom Retriever Implementation ```python -class CustomRetriever(BaseRetriever): - def search(self, query: str, k: int = 10) -> List[Dict]: - """Implement custom search logic.""" - # Your custom retrieval implementation - pass - - def get_embeddings(self, texts: List[str]) -> np.ndarray: - """Generate embeddings for custom retrieval.""" - # Your custom embedding logic - pass +# rag_system/pipelines/retrieval_pipeline.py:263 +RetrievalPipeline.run(query, table_name=None, window_size_override=None, event_callback=None) -> Dict + +# rag_system/agent/loop.py:237 +Agent.run(query, table_name=None, session_id=None, compose_sub_answers=None, query_decompose=None, + ai_rerank=None, context_expand=None, verify=None, retrieval_k=None, context_window_size=None, + reranker_top_k=None, retrieval_mode=None, force_rag=False, event_callback=None) -> Dict ``` -### Custom Reranker Implementation +`Agent.run` is a synchronous wrapper around `_run_async`. Both `/chat` and `/chat/stream` go through `Agent.run` — including the `force_rag` path (`api_server.py:354-390`), so no request shape bypasses the toggles. `RetrievalPipeline` has no iterator/`answer_stream` entry point. + +Build them with the factory, never by hand: + ```python -class CustomReranker(BaseReranker): - def rank(self, query: str, documents: List[Dict]) -> List[Dict]: - """Implement custom reranking logic.""" - # Your custom reranking implementation - pass +from rag_system.factory import get_agent +agent = get_agent("default") # deep copy of PIPELINE_CONFIGS["default"] +result = agent.run("What does the contract say about termination?") ``` -### Custom Query Transformer -```python -class CustomQueryTransformer: - def transform(self, query: str, context: Dict = None) -> str: - """Transform query based on context.""" - # Your custom query transformation logic - pass -``` \ No newline at end of file +## Streaming event protocol + +`POST /chat/stream` returns `text/event-stream`; every frame is `data: {"type": , "data": }\n\n` (`api_server.py:329-333`). + +| Event | Payload | Emitted by | +|-------|---------|-----------| +| `analyze` | `{query}` | `loop.py:250-251` | +| `direct_answer` | `{}` | `loop.py:328-329` | +| `decomposition` | `{sub_queries}` | `loop.py:378-379` | +| `retrieval_started` | `{mode}` (pipeline) or `{count}` (decomposition) | `retrieval_pipeline.py:276-277`, `loop.py:384-385` | +| `retrieval_done` | `{count}` | `retrieval_pipeline.py:307-308`, `loop.py:401, 485` | +| `rerank_started` / `rerank_done` | `{count}` | `retrieval_pipeline.py:346-347, 394-395`, `loop.py:402-403, 418-419, 486` | +| `context_expand_started` / `context_expand_done` | `{count}` | `retrieval_pipeline.py:404-405, 438-439` | +| `prune_started` / `prune_done` | `{count}` | `retrieval_pipeline.py:453-454, 461-462` | +| `token` | `{text}` | synthesis and composition streams | +| `sub_query_token` | `{index, text, question}` | `loop.py:430` | +| `sub_query_result` | `{index, query, answer, source_documents}` | `loop.py:453-459` | +| `single_query_result` / `final_answer` | the result dict | `loop.py:397-398, 533-534` | +| `complete` | the final result dict | `api_server.py:338` | +| `error` | `{error}` | `api_server.py:344` | + +## Interfaces + +* Reads LanceDB tables at `storage.lancedb_uri` (`./lancedb`), table `text_pages_` (`backend/database.py:351`) or the profile's `storage.text_table_name` (`text_pages_v4`) when a session has no linked index. +* Calls Ollama at `OLLAMA_CONFIG["host"]` — generation model for answers, utility model for routing/decomposition/verification. +* Embeddings come from `select_embedder()`: a model name containing `/` is loaded from HuggingFace in-process, anything else is treated as an Ollama tag. `_get_text_embedder()` **raises** when `embedding_model_name` is missing rather than substituting a default, because a wrong-dimensionality embedder silently returns nothing useful against an existing index (`retrieval_pipeline.py:103-116`). +* Vector search is a brute-force scan: nothing in `rag_system/` ever calls `create_index` for an ANN/IVF-PQ index. Only the full-text index is built. + +## Extension points + +* **New retriever** — the contract is duck-typed, not an ABC. Provide an object with `retrieve(text_query: str, table_name: str, k: int, search_type: str = "hybrid") -> List[Dict]` returning dicts with at least `chunk_id`, `text`, `score`, `document_id`, `chunk_index`, `metadata`, and return it from `RetrievalPipeline._get_dense_retriever()` (`retrieval_pipeline.py:118-133`). There is no `BaseRetriever` and no registry. +* **New reranker** — either point `reranker.model_name` / `reranker.model_type` at another `rerankers`-library model, or set `reranker.strategy` to something other than `rerankers-lib` and swap the class constructed in `_get_ai_reranker()` (`retrieval_pipeline.py:135-165`). There is no `BaseReranker`. +* **Answer prompt** — the synthesis prompt is an inline f-string at `retrieval_pipeline.py:225-250`. `_synthesize_final_answer(query, facts, *, event_callback=None)` takes no prompt-override argument; edit the literal. + +## Operational notes + +* The RAG API is a single-threaded `socketserver.TCPServer` (`api_server.py:527-530`), so requests are handled one at a time. Treat it as a single-concurrent-user service. +* `RAG_AGENT` and its `RetrievalPipeline` are process-wide singletons created once at startup (`api_server.py:34-37`). Per-request overrides write into that shared config object. Fields the API always sends (`retrieval_k`, `context_window_size`, `reranker_top_k`) are refreshed on every request; fields it only sends when present (`ai_rerank`, `provence_prune`, `provence_threshold`, `retrieval_mode`) persist until another request changes them. +* Changing the embedding model requires re-indexing. `VectorIndexer` raises a clear error if you try to append vectors of a different width to an existing table (`indexing/embedders.py:110-116`), and the query side will simply fail to match if the dimensions differ. + +--- +_Keep this document updated when stages, config keys, or the event protocol change._ diff --git a/Documentation/system_overview.md b/Documentation/system_overview.md index 7c6aeac0..86561b4a 100644 --- a/Documentation/system_overview.md +++ b/Documentation/system_overview.md @@ -1,429 +1,487 @@ -# 🏗️ RAG System - Complete System Overview +# 🏗️ localGPT — Complete System Overview -_Last updated: 2025-01-09_ +_Last updated: 2026-08-08_ -This document provides a comprehensive overview of the Advanced Retrieval-Augmented Generation (RAG) System, covering its architecture, components, data flow, and operational characteristics. +A comprehensive overview of the localGPT Retrieval-Augmented Generation system: architecture, components, data flow, configuration and operational characteristics. Everything here was verified against the source in this repository; where a feature exists but is not wired up, it says so. --- ## 1. System Architecture -### 1.1 High-Level Architecture +### 1.1 High-level architecture -The RAG system implements a sophisticated 4-tier microservices architecture: +Four processes. The browser talks to **two** of them. ```mermaid graph TB subgraph "Client Layer" - Browser[👤 User Browser] - UI[Next.js Frontend
React/TypeScript] + Browser["👤 User Browser"] + UI["Next.js Frontend
React / TypeScript
Port 3000"] Browser --> UI end - - subgraph "API Gateway Layer" - Backend[Backend Server
Python HTTP Server
Port 8000] - UI -->|REST API| Backend + + subgraph "Gateway Layer" + Backend["Backend Server
backend/server.py
Port 8000"] end - + subgraph "Processing Layer" - RAG[RAG API Server
Document Processing
Port 8001] - Backend -->|Internal API| RAG + RAG["RAG API Server
rag_system/api_server.py
Port 8001"] end - + subgraph "LLM Service Layer" - Ollama[Ollama Server
LLM Inference
Port 11434] - RAG -->|Model Calls| Ollama + Ollama["Ollama
Port 11434"] end - + subgraph "Storage Layer" - SQLite[(SQLite Database
Sessions & Metadata)] - LanceDB[(LanceDB
Vector Embeddings)] - FileSystem[File System
Documents & Indexes] - - Backend --> SQLite - RAG --> LanceDB - RAG --> FileSystem + SQLite[("SQLite
sessions, messages, index metadata")] + LanceDB[("LanceDB
chunk vectors + native FTS")] + FileSystem["File system
shared_uploads/ · index_store/"] end + + UI -->|"REST"| Backend + UI -->|"SSE POST /chat/stream (default chat path)"| RAG + Backend -->|"POST /chat · POST /index"| RAG + Backend -->|"routing + direct answers"| Ollama + RAG -->|"generation · enrichment · verification"| Ollama + Backend --> SQLite + RAG -->|"index metadata only"| SQLite + RAG --> LanceDB + RAG --> FileSystem ``` -### 1.2 Component Breakdown +### 1.2 Component breakdown | Component | Technology | Port | Purpose | |-----------|------------|------|---------| -| **Frontend** | Next.js 15, React 19, TypeScript | 3000 | User interface, chat interactions | -| **Backend** | Python 3.11, HTTP Server | 8000 | API gateway, session management, routing | -| **RAG API** | Python 3.11, Advanced NLP | 8001 | Document processing, retrieval, generation | -| **Ollama** | Go-based LLM server | 11434 | Local LLM inference (embedding, generation) | -| **SQLite** | Embedded database | - | Sessions, messages, index metadata | -| **LanceDB** | Vector database | - | Document embeddings, similarity search | +| **Frontend** | Next.js 15, React 19, TypeScript, Tailwind v4 | 3000 | Chat UI, index management, retrieval settings | +| **Backend gateway** | Python 3.10+, `http.server` on `ThreadingTCPServer` | 8000 | Sessions, messages, uploads, index CRUD, first-layer routing | +| **RAG API** | Python 3.10+, `http.server` on `TCPServer` (serialized) | 8001 | Agent, retrieval pipeline, indexing pipeline | +| **Ollama** | External LLM server | 11434 | Generation and enrichment model inference | +| **SQLite** | Embedded | – | Sessions, messages, documents, index metadata | +| **LanceDB** | Embedded vector store | – | Chunk vectors + native full-text (BM25) index | + +For the process topology, request sequences and threading model see [`architecture_overview.md`](architecture_overview.md). --- ## 2. Core Functionality -### 2.1 Intelligent Dual-Layer Routing +### 2.1 Two-layer routing -The system's key innovation is its **dual-layer routing architecture** that optimizes both speed and intelligence: +Routing happens twice, in two different processes. Layer 1 is a deterministic gate with no model call; layer 2 is the system's single LLM routing layer. -#### **Layer 1: Speed Optimization Routing** -- **Location**: `backend/server.py` -- **Purpose**: Route simple queries to Direct LLM (~1.3s) vs complex queries to RAG Pipeline (~20s) -- **Decision Logic**: Pattern matching, keyword detection, query complexity analysis +#### Layer 1 — gateway routing (`backend/server.py`, non-streaming path only) -```python -# Example routing decisions -"Hello!" → Direct LLM (greeting pattern) -"What does the document say about pricing?" → RAG Pipeline (document keyword) -"What's 2+2?" → Direct LLM (simple + short) -"Summarize the key findings from the report" → RAG Pipeline (complex + indicators) -``` +`should_use_rag(message, idx_ids, force_rag)` — a module-level function, no LLM call, no network I/O: + +1. `force_rag` → RAG, unconditionally. +2. Session has **no linked indexes** → direct LLM, no RAG. +3. Message is unmistakable smalltalk (`hello`, `thanks!`, `bye`, `ok` — a whole-message allowlist regex capped at six words) or a question about the assistant itself (`who are you`, `what model are you`) → direct LLM. +4. Everything else → RAG. + +This is retrieval-first: escalate rather than pre-decide. The gate deliberately over-sends to RAG because layer 2 can still answer directly — the cost of a wrong "send to RAG" is one agent triage call, while the cost of a wrong "answer directly" is an unanswerable question. Pre-retrieval LLM routing was removed here in Phase 2.3: it is the weakest measured routing pattern (`Documentation/research/`), and it duplicated layer 2. The old `_simple_pattern_routing` keyword/length fallback (which misrouted any question containing the word "test") is gone with it. + +A `force_rag: true` field on the request skips the gate and always calls the RAG API. This layer does **not** run on the streaming path, because the browser calls the RAG API directly. + +#### Layer 2 — agent triage (`rag_system/agent/loop.py`, always) + +`_triage_query_async()`: + +1. `_route_via_overviews()` — the enrichment model classifies the query against the overviews loaded for the session, returning `direct_answer` or `rag_query`. +2. If conversation history exists for the session, short-circuit to `rag_query`. +3. Otherwise a fallback triage prompt (also on the enrichment model) picks `rag_query` or `direct_answer`. + +`force_rag=true` on the RAG API pins `query_type = "rag_query"` and skips all three steps, while still honouring the reranking / decomposition / verification toggles. + +Triage is two-way since the graph module was removed on 2026-08-09 (roadmap 2.5). `Agent._normalize_triage()` collapses anything that is not an explicit `direct_answer` — including a stray `graph_query` from a small utility model — to `rag_query`. + +### 2.2 Indexing + +1. **Upload** — files are stored under `shared_uploads/` with a UUID prefix and recorded in SQLite. +2. **Conversion** — `DocumentConverter` (Docling) produces markdown plus structure. OCR options are chosen by probing which backend is actually installed (OcrMac on macOS, then EasyOCR, RapidOCR, tesserocr, the `tesseract` CLI); when none is available Docling's defaults are used and text-layer PDFs still convert. +3. **Chunking** — `DoclingChunker` packs sentences up to a token budget (`chunk_size`, default 512 over HTTP) using the embedding model's tokenizer. A legacy `MarkdownRecursiveChunker` exists and is selected by `chunker_mode: "legacy"` (used by `create_index_script.py`; not reachable through the HTTP API — see [§11](#11-known-limitations)). +4. **Overviews** — the first *n* chunks of each document (default 5) are summarised by the enrichment model into `index_store/overviews/.jsonl`. The agent's triage router reads these; the gateway gate (§2.1) does not. +5. **Contextual enrichment** (optional) — the enrichment model summarises a window of surrounding chunks and prepends it to each chunk. The untouched text is kept in `metadata.original_text`. +6. **Embedding + indexing** — chunks are embedded and written to a LanceDB table; a native FTS index is created on the `text` column. +7. **Late chunking** (optional) — each document is re-encoded in one pass and per-chunk vectors are pooled from it, written to `
_lc`. Enrichment produces *copies* of the chunks, so this leg encodes the original text, not the enriched text. + +The vector width is taken from the embeddings actually produced. If you point an existing table at a different-dimensional model, `VectorIndexer` raises with a message telling you to rebuild — it will not silently corrupt the table. Because two models can share a width, each table also records the embedding model that wrote it and whether its vectors are L2-normalized; a mismatch raises `EmbedderMismatchError` at index time and at query time, and a table with no marker (built before this existed) keeps working unnormalized with a warning. + +### 2.3 Retrieval + +1. **Query embedding** — the same embedding model used at index time (an LRU cache holds 256 single-query embeddings). Instruction-tuned families (harrier-oss-v1, Qwen3-Embedding) get the query-side `Instruct: … \nQuery: …` prefix their cards require; documents never do. The query vector is L2-normalized when the table's marker says its vectors are, so LanceDB's L2 ordering is the cosine ordering the cards specify. +2. **Search** — `MultiVectorRetriever.retrieve()` runs LanceDB full-text search and vector search **in parallel** and fuses them with **reciprocal rank fusion** (`1/(60 + rank)` per leg). `search_type` selects `hybrid` (both legs), `vector_only` or `fts_only`; an unrecognised value logs a warning and falls back to `hybrid`. Every returned document carries a finite, higher-is-better `score`. +3. **Late-chunk leg** (optional) — when late chunking is enabled the same query also runs against `
_lc` and those hits are appended. Every retrieved chunk (from either leg) then has its text replaced by the concatenation of its ±1 neighbours in the main table, so a hit on one sub-vector still yields readable context. Hits are not de-duplicated across the two legs. +4. **Reranking** (**off by default** — `reranker.enabled: False`; the measured reason is in [`../eval/DECISIONS.md`](../eval/DECISIONS.md)) — when switched on, `Qwen/Qwen3-Reranker-4B` goes to the in-repo `QwenRerankerScorer`, and any other model is loaded through the `rerankers` library (`reranker.strategy: "rerankers-lib"`, `reranker.model_type: "cross-encoder"`); any other `strategy` value uses the in-repo `CrossEncoderReranker` (`transformers` `AutoModelForSequenceClassification`). If the model fails to load, a warning is printed and **reranking is skipped** — there is no second reranker to fall back to. +5. **Context expansion** — each surviving chunk is widened to its neighbours within `context_window_size`. +6. **Sentence pruning** (opt-in, `provence.enabled`) — `naver/provence-reranker-debertav3-v1` drops sentences below `provence.threshold` (default `0.1`); chunks pruned to nothing are removed. +7. **Synthesis** — the generation model writes the answer, streamed token by token. +8. **Verification** (see §2.4). + +Retrieval matches against the stored (possibly enriched) text, and chunks coming out of the retriever expose `metadata.original_text` as their `text` when enrichment ran — so enrichment improves recall without pushing its own prefix into the answer's context. Neighbour chunks pulled in by context expansion are returned exactly as stored, so those still carry the enrichment prefix. + +### 2.4 Verification + +When `verification.enabled` is on (or `verify: true` is sent) **and** the result has source documents, `Verifier.verify_async()` asks the enrichment model for a JSON verdict and the confidence is appended to the answer **string**: + +* `" [Confidence: N%]"` for any non-zero score; +* additionally `" [Warning: Low confidence. Groundedness: ]"` when the answer is not grounded or the score is below 50; +* nothing at all when the score parses as 0 (treated as a parser failure). -#### **Layer 2: Intelligence Optimization Routing** -- **Location**: `rag_system/agent/loop.py` -- **Purpose**: Within RAG pipeline, route to optimal processing method -- **Methods**: - - `direct_answer`: General knowledge queries - - `rag_query`: Document-specific queries requiring retrieval - - `graph_query`: Entity relationship queries (future feature) - -### 2.2 Document Processing Pipeline - -#### **Indexing Process** -1. **Document Upload**: PDF files uploaded via web interface -2. **Text Extraction**: Docling library extracts text with layout preservation -3. **Chunking**: Intelligent chunking with configurable strategies (DocLing, Late Chunking, Standard) -4. **Embedding**: Text converted to vector embeddings using Qwen models -5. **Storage**: Vectors stored in LanceDB with metadata in SQLite - -#### **Retrieval Process** -1. **Query Processing**: User query analyzed and contextualized -2. **Embedding**: Query converted to vector embedding -3. **Search**: Hybrid search combining vector similarity and BM25 keyword matching -4. **Reranking**: AI-powered reranking for relevance optimization -5. **Synthesis**: LLM generates final answer using retrieved context - -### 2.3 Advanced Features - -#### **Query Decomposition** -- Complex queries automatically broken into sub-queries -- Parallel processing of sub-queries for efficiency -- Intelligent composition of final answers - -#### **Contextual Enrichment** -- Conversation history integration -- Context-aware query expansion -- Session-based memory management - -#### **Verification System** -- Answer verification against source documents -- Confidence scoring and grounding checks -- Source attribution and citation +There is no top-level `confidence` field in the response. Responses are `{answer, source_documents}`. + +### 2.5 Query decomposition + +Enabled in the `default` profile. The enrichment model splits the raw user query (plus up to 5 recent turns for pronoun resolution) into sub-queries, capped by `query_decomposition.max_sub_queries` (default 10). Sub-queries are retrieved in parallel with at most 3 worker threads. With `compose_from_sub_answers: true` the generation model composes a final answer from the sub-answers; otherwise the unique source chunks are aggregated and synthesised in one pass. A decomposition that yields a single sub-query skips the parallel machinery. + +### 2.6 Semantic cache and conversation memory + +* `TTLCache(maxsize=100, ttl=300)` keyed on the query embedding. A cached answer is reused when cosine similarity ≥ `semantic_cache_threshold` (`0.98`). +* `cache_scope` is `"session"` in both profiles: an entry is only reused inside the session that produced it. Setting it to `"global"` re-enables cross-session reuse — including answers derived from another session's documents. +* Conversation history for triage and query rewriting lives in an in-process `LRUCache(maxsize=100)` on the agent, keyed by `session_id`. It is **not** the SQLite message history and is lost on restart. --- ## 3. Data Architecture -### 3.1 Storage Systems +### 3.1 SQLite (`backend/chat_data.db`, override with `DB_PATH`) -#### **SQLite Database** (`backend/chat_data.db`) ```sql --- Core tables -sessions -- Chat sessions with metadata -messages -- Individual messages and responses -indexes -- Document index metadata -session_indexes -- Links sessions to their indexes +sessions -- id, title, created_at, updated_at, model_used, message_count +messages -- id, session_id, content, sender('user'|'assistant'), timestamp, metadata +session_documents -- files uploaded to a session +indexes -- id, name, description, created_at, updated_at, vector_table_name, metadata +index_documents -- files belonging to a named index +session_indexes -- links sessions to indexes ``` -#### **LanceDB Vector Store** (`./lancedb/`) -``` -tables/ -├── text_pages_[uuid] -- Document text embeddings -├── image_pages_[uuid] -- Image embeddings (future) -└── metadata_[uuid] -- Document metadata -``` +Written by `backend/server.py`. The RAG API opens the same database but only reads/writes the `indexes` rows (index metadata); it never writes `messages`. + +### 3.2 LanceDB (`./lancedb`, override with `LANCEDB_PATH`) -#### **File System** (`./index_store/`) ``` -index_store/ -├── overviews/ -- Document summaries for routing -├── bm25/ -- BM25 keyword indexes -└── graph/ -- Knowledge graph data +lancedb/ +├── text_pages_v4 -- default table (storage.text_table_name) +├── text_pages_ -- one table per index created via POST /indexes +└──
_lc -- late-chunk vectors for the table above ``` -### 3.2 Data Flow +Each table stores `chunk_id`, `text`, `document_id`, `chunk_index`, `metadata` (JSON) and `vector`, plus a native full-text index on `text`. -1. **Document Upload** → File System (`shared_uploads/`) -2. **Processing** → Embeddings stored in LanceDB -3. **Metadata** → Index info stored in SQLite -4. **Query** → Search LanceDB + SQLite coordination -5. **Response** → Message history stored in SQLite +### 3.3 File system ---- +``` +shared_uploads/ -- uploaded documents (_) +index_store/overviews/.jsonl -- per-index / per-session document overviews +index_store/overviews/overviews.jsonl -- global fallback overview file +logs/ -- run_system.py service logs + run_system.pid +``` -## 4. Model Architecture +--- -### 4.1 Configurable Model Pipeline +## 4. Models -The system supports multiple embedding and generation models with automatic switching: +### 4.1 Configured defaults (`rag_system/main.py`) -#### **Current Model Configuration** ```python -EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # 1024D - "reranker_model": "answerdotai/answerai-colbert-small-v1", # ColBERT reranker - "vision_model": "Qwen/Qwen-VL-Chat", # Vision model for multimodal - "fallback_reranker": "BAAI/bge-reranker-base", # Backup reranker +OLLAMA_CONFIG = { + "host": os.getenv("OLLAMA_HOST", "http://localhost:11434"), + "generation_model": os.getenv("GENERATION_MODEL", "qwen3.5:9b"), + "enrichment_model": os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b"), } -OLLAMA_CONFIG = { - "generation_model": "qwen3:8b", # High-quality generation - "enrichment_model": "qwen3:0.6b", # Fast enrichment/routing - "host": "http://localhost:11434" +EXTERNAL_MODELS = { + "embedding_model": os.getenv("EMBEDDING_MODEL", "microsoft/harrier-oss-v1-0.6b"), + "reranker_model": os.getenv("RERANKER_MODEL", "Qwen/Qwen3-Reranker-4B"), } ``` -#### **Model Switching** -- **Per-Session**: Each chat session can use different embedding models -- **Automatic**: System automatically switches models based on index metadata -- **Dynamic**: Models loaded just-in-time to optimize memory usage +| Role | Default | Used for | +|------|---------|----------| +| Generation | `qwen3.5:9b` (Ollama) | Final answers, sub-answer composition, direct answers | +| Enrichment / utility | `qwen3.5:4b` (Ollama) | Agent triage (the only LLM router), query decomposition, contextual enrichment, document overviews, verification | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (HuggingFace, MIT, 1024 dims) | Index and query embeddings | +| Reranker | `Qwen/Qwen3-Reranker-4B` (HuggingFace, own yes/no-logit scorer) | Reranking retrieved chunks — **off by default**, loaded lazily only when switched on ([`../eval/DECISIONS.md`](../eval/DECISIONS.md)) | +| Sentence pruner | `naver/provence-reranker-debertav3-v1` (HuggingFace) | Opt-in sentence-level pruning | + +Approximate footprints published by the model authors (not measured here): `qwen3.5:9b` ≈ 6.6 GB at Q4, `qwen3.5:4b` ≈ 3.4 GB, `qwen3.6:27b` ≈ 17 GB, `microsoft/harrier-oss-v1-0.6b` ≈ 1.2 GB, `Qwen/Qwen3-Embedding-4B` ≈ 8 GB in bf16, `Qwen/Qwen3-Embedding-0.6B` ≈ 1.2 GB, `Qwen/Qwen3-Reranker-4B` ≈ 7.5 GB. + +### 4.2 Documented alternatives + +| Role | Options | +|------|---------| +| Generation | `qwen3.6:27b` (high-end), `qwen3.5:4b` (light) | +| Enrichment | `qwen3.5:2b` (light) | +| Embedding | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context — for multilingual / long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024 dims, light) | +| Reranker | `BAAI/bge-reranker-v2-m3` (cross-encoder, low latency — only pays off with a weaker embedder than the default), `answerdotai/answerai-colbert-small-v1` (late interaction — also set `reranker.model_type: "colbert"`), `Qwen/Qwen3-Reranker-0.6B` | -### 4.2 Supported Models +Set them with the `GENERATION_MODEL` / `ENRICHMENT_MODEL` / `EMBEDDING_MODEL` / `RERANKER_MODEL` environment variables, or edit `rag_system/main.py`. -#### **Embedding Models** -- `Qwen/Qwen3-Embedding-0.6B` (1024D) - Default, fast and high-quality +> ⚠️ **Changing the embedding model requires re-indexing.** Vector width is derived from the loaded model, and `VectorIndexer` raises rather than appending mismatched vectors to an existing LanceDB table. Width alone is not a sufficient check — `harrier-oss-v1-0.6b` and `Qwen3-Embedding-0.6B` are both 1024 dims — so every table also records the embedding model that wrote it, and indexing into or querying it with a different one raises `EmbedderMismatchError`. Ollama embedding tags are also supported: `select_embedder()` treats a name containing `/` as a HuggingFace repo and anything else as an Ollama tag. -#### **Generation Models** (via Ollama) -- `qwen3:8b` - Primary generation model (high quality) -- `qwen3:0.6b` - Fast enrichment and routing model +### 4.3 Model selection at runtime -#### **Reranking Models** -- `answerdotai/answerai-colbert-small-v1` - Primary ColBERT reranker -- `BAAI/bge-reranker-base` - Fallback cross-encoder reranker +* **Per request** — `model` on `POST :8000/sessions/{id}/messages` and on the RAG API chat endpoints overrides the generation model for that request only. The RAG API rejects ids that do not match the active backend (an Ollama tag will not be forced onto a WatsonX deployment). +* **Per index** — when an index records an `embedding_model` in its metadata, the RAG API switches the retrieval pipeline's embedder to it before querying that index. +* Generation model precedence on the gateway's direct-LLM path: request `model` → the session's `model_used` → `GENERATION_MODEL`. -#### **Vision Models** (Multimodal) -- `Qwen/Qwen-VL-Chat` - Vision-language model for image processing +### 4.4 Vision / multimodal — not integrated + +There is **no** vision model in the configuration and no multimodal path in the pipelines: PDF parsing and OCR are handled entirely by Docling. Models such as GLM-OCR or Qwen3-VL could be added as an extension; wiring them up is not done today. + +### 4.5 Alternative LLM backend: WatsonX + +`LLM_BACKEND=watsonx` swaps the Ollama client for `WatsonXClient` (`WATSONX_CONFIG`: `WATSONX_API_KEY`, `WATSONX_PROJECT_ID`, `WATSONX_URL`, `WATSONX_GENERATION_MODEL`, `WATSONX_ENRICHMENT_MODEL`). It requires `pip install ibm-watsonx-ai` — the root `requirements.txt` lists it as an optional, commented dependency. Embedding and reranking still run locally through HuggingFace. See [`../WATSONX_README.md`](../WATSONX_README.md). --- ## 5. Pipeline Configurations -### 5.1 Default Production Pipeline +`PIPELINE_CONFIGS` in `rag_system/main.py` contains exactly two profiles, `default` and `fast`. `RAG_CONFIG_MODE` selects the one the RAG API server uses (default `default`); an unknown value silently falls back to `default`. `factory.get_pipeline_config()` hands out a deep copy, so runtime overrides never mutate the master config. + +### 5.1 `default` ```python -PIPELINE_CONFIGS = { - "default": { - "description": "Production-ready pipeline with hybrid search, AI reranking, and verification", - "storage": { - "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "bm25_path": "./index_store/bm25", - "graph_path": "./index_store/graph/knowledge_graph.gml" - }, - "retrieval": { - "retriever": "multivector", - "search_type": "hybrid", - "late_chunking": { - "enabled": True, - "table_suffix": "_lc_v3" - }, - "dense": { - "enabled": True, - "weight": 0.7 - }, - "bm25": { - "enabled": True, - "index_name": "rag_bm25_index" - } - }, - "embedding_model_name": "Qwen/Qwen3-Embedding-0.6B", - "reranker": { - "enabled": True, - "model_name": "answerdotai/answerai-colbert-small-v1", - "top_k": 20 - } +"default": { + "description": "Production-ready pipeline with hybrid search, query decomposition, and verification", + "storage": { + "lancedb_uri": "./lancedb", + "text_table_name": "text_pages_v4" + }, + "retrieval": { + "search_type": "hybrid", + "latechunk": {"enabled": True}, + "dense": {"enabled": True}, + "retry": {"enabled": True, "min_top_score": 0.12, "max_attempts": 1} + }, + "embedding_model_name": EXTERNAL_MODELS["embedding_model"], + "reranker": { + "enabled": False, # see eval/DECISIONS.md + "model_type": "cross-encoder", + "strategy": "rerankers-lib", + "model_name": EXTERNAL_MODELS["reranker_model"], + "top_k": 10 + }, + "query_decomposition": {"enabled": True, "compose_from_sub_answers": True}, + "verification": {"enabled": True}, + "retrieval_k": 20, + "context_window_size": 0, + "semantic_cache_threshold": 0.98, + "cache_scope": "session", + "contextual_enricher": {"enabled": True, "window_size": 1}, + "indexing": { + "embedding_batch_size": 50, + "enrichment_batch_size": 10, + "enable_progress_tracking": True + } +} +``` + +### 5.2 `fast` + +```python +"fast": { + "description": "Speed-optimized pipeline with minimal overhead", + "storage": {"lancedb_uri": "./lancedb", "text_table_name": "text_pages_v4"}, + "retrieval": { + "search_type": "vector_only", + "latechunk": {"enabled": False}, + "dense": {"enabled": True} + }, + "embedding_model_name": EXTERNAL_MODELS["embedding_model"], + "reranker": {"enabled": False}, + "query_decomposition": {"enabled": False}, + "verification": {"enabled": False}, + "retrieval_k": 10, + "context_window_size": 0, + "semantic_cache_threshold": 0.98, + "cache_scope": "session", + "contextual_enricher": {"enabled": False, "window_size": 1}, + "indexing": { + "embedding_batch_size": 100, + "enrichment_batch_size": 50, + "enable_progress_tracking": False } } ``` -### 5.2 Processing Options +Three keys in the blocks above currently have no consumer and are inert: the profile's `description`, `reranker.type` (the loader reads `strategy` and `model_type`), and `indexing.enable_progress_tracking` (assigned to an attribute that is never checked — progress is always tracked). -#### **Chunking Strategies** -- **Standard**: Fixed-size chunks with overlap -- **DocLing**: Structure-aware chunking using DocLing library -- **Late Chunking**: Small chunks expanded at query time +### 5.3 Keys read at runtime but absent from the profiles -#### **Enrichment Options** -- **Contextual Enrichment**: AI-generated chunk summaries -- **Overview Building**: Document-level summaries for routing -- **Graph Extraction**: Entity and relationship extraction +These have code defaults and can be added to a profile if you want to change them: ---- +| Key | Default | Effect | +|-----|---------|--------| +| `chunking.chunk_size` | `1500` (profile absent) / `512` (HTTP requests) | Token budget per chunk | +| `chunker_mode` | `"docling"` | `"docling"` or `"legacy"` | +| `query_decomposition.max_sub_queries` | `10` | Cap on sub-queries | +| `query_decomposition.rerank_aggregate` | `"mean"` | `mean` or `max`; how per-sub-query rerank scores combine (roadmap 2.2) | +| `retrieval.retry.min_rerank_score` | falls back to `min_top_score` | Retry threshold used when the reranker returns a 0–1 probability | +| `verification.model` / `VERIFIER_MODEL` | unset | HuggingFace NLI/verifier model; unset keeps the LLM-prompt verifier (roadmap 2.4) | +| `verification.threshold` | `0.5` | Grounded/ungrounded cut for the local verifier | +| `reranker.model_type` | `"cross-encoder"` | `rerankers` library model type | +| `reranker.top_percent` | – | Keep a fraction of candidates instead of `top_k` | +| `provence.enabled` / `provence.threshold` | `False` / `0.1` | Sentence-level pruning | +| `overview.enabled` / `overview.model` / `overview.max_chunks` | `True` / enrichment model / `5` | Document overview generation | +| `enrich_model` | enrichment model | Overrides the model used for contextual enrichment | +| `overview_path` | `index_store/overviews/overviews.jsonl` | Where overviews are written | -## 6. Performance Characteristics +### 5.4 Profile values the HTTP API always overrides -### 6.1 Response Times +`rag_system/api_server.py` always sends `retrieval_k` (default `20`), `context_window_size` (default `1`) and `reranker_top_k` (default `10`) to `Agent.run()`, even when the client omits them. The profile values for those three keys therefore apply only to the CLI and programmatic paths. Everything else (`verify`, `ai_rerank`, `query_decompose`, `compose_sub_answers`, `context_expand`, `retrieval_mode`) is passed as `None` when omitted, so the profile wins. -| Operation | Time Range | Notes | -|-----------|------------|-------| -| Simple Chat | 1-3 seconds | Direct LLM, no retrieval | -| Document Query | 5-15 seconds | Includes retrieval and reranking | -| Complex Analysis | 15-30 seconds | Multi-step reasoning | -| Document Indexing | 2-5 min/100MB | Depends on enrichment settings | +Note the practical consequence: over HTTP, context expansion of ±1 chunk is on by default even though both profiles set `context_window_size: 0`. -### 6.2 Memory Usage +--- + +## 6. Resource Notes -| Component | Memory Usage | Notes | -|-----------|--------------|-------| -| Embedding Model | 1-2GB | Qwen3-Embedding-0.6B | -| Generation Model | 8-16GB | qwen3:8b | -| Reranker Model | 500MB-1GB | ColBERT reranker | -| Database Cache | 500MB-2GB | LanceDB and SQLite | +There are no benchmarks in this repository, so no latency or throughput figures are published here. What determines cost: -### 6.3 Scalability +* **Memory** is dominated by the models you load: the Ollama generation model, plus the embedding model and (if enabled) the reranker and Provence pruner, which run in the RAG API process via `transformers`. +* **Concurrency** is bounded by the RAG API's single-threaded server: one RAG request at a time per process. The backend gateway is threaded, so session and index CRUD stay responsive while a query runs. +* **Indexing cost** scales with contextual enrichment (one LLM call per chunk batch) and late chunking (a second full encode of every document, plus a second vector table). +* **Query cost** scales with query decomposition (one retrieval + one synthesis per sub-query), reranking and verification. The `fast` profile turns all of these off. -- **Concurrent Users**: 5-10 users with 16GB RAM -- **Document Capacity**: 10,000+ documents per index -- **Query Throughput**: 10-20 queries/minute per instance -- **Storage**: Approximately 1MB per 100 pages indexed +Use `python system_health_check.py` to print the resolved configuration, the embedding dimension of the loaded model, and the LanceDB tables that actually exist. --- -## 7. Security & Privacy +## 7. Configuration -### 7.1 Data Privacy +### 7.1 Environment variables -- **Local Processing**: All AI models run locally via Ollama -- **No External Calls**: No data sent to external APIs -- **Document Isolation**: Documents stored locally with session-based access -- **User Isolation**: Each session maintains separate context +Every variable below is read by this repository's code, except `HF_TOKEN` which is consumed by the HuggingFace client libraries. See [`.env.example`](../.env.example) for the annotated file. ---- +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` (all calls to the RAG API) | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` — **inlined at build time** | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` — **inlined at build time** | +| `DB_PATH` | `backend/chat_data.db` (`/app/backend/chat_data.db` in Docker) | `backend/database.py` | +| `LANCEDB_PATH` | `storage.lancedb_uri`, else `./lancedb` | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | same | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (only loaded when reranking is switched on) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py`, `rag_system/factory.py` | +| `RAG_API_TIMEOUT` | `600` (seconds) | `backend/server.py` — chat calls | +| `RAG_API_INDEX_TIMEOUT` | `3600` (seconds) | `backend/server.py` — indexing calls | +| `HF_TOKEN` | – | `huggingface_hub` (library) — gated model downloads | -## 8. Configuration & Customization +`NEXT_PUBLIC_*` values are baked into the JavaScript bundle by `next build`; changing them at runtime has no effect on an already-built frontend. The compose files pass them as build args as well as runtime environment. -### 8.1 Model Configuration -Models can be configured in `rag_system/main.py`: +Service **ports** are not environment-configurable: `PORT = 8000` in `backend/server.py`, `8001` in `start_server()`, `3000` from Next.js. -```python -# Embedding model configuration -EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # Your preferred model - "reranker_model": "answerdotai/answerai-colbert-small-v1", -} +### 7.2 Per-request options -# Generation model configuration -OLLAMA_CONFIG = { - "generation_model": "qwen3:8b", # Your LLM model - "enrichment_model": "qwen3:0.6b", # Your fast model -} -``` +Retrieval and indexing behaviour is controlled per request, not by editing a config file. See [`api_reference.md`](api_reference.md) for the full field list. Both casings are accepted end to end: the frontend historically sent camelCase, the gateway sends snake_case, and both the gateway and the RAG API normalise every option to one canonical snake_case key at parse time. -### 8.2 Pipeline Configuration -Processing behavior configured in `PIPELINE_CONFIGS`: +### 7.3 Command-line entry points -```python -PIPELINE_CONFIGS = { - "retrieval": { - "search_type": "hybrid", - "dense": {"weight": 0.7}, - "bm25": {"enabled": True} - }, - "chunking": { - "chunk_size": 512, - "chunk_overlap": 64, - "enable_latechunk": True, - "enable_docling": True - } -} +```bash +# Index a file or a directory (walks for .pdf .docx .html .htm .md .txt) +python -m rag_system.main index /path/to/docs --mode default + +# One-shot query, prints JSON +python -m rag_system.main chat "What does the contract say about termination?" --mode default + +# Start the RAG API +python -m rag_system.main api --port 8001 ``` -### 8.3 UI Configuration -Frontend behavior configured in environment variables: +`python rag_system/main.py …` does **not** work — the module must be run with `-m` from the project root. Programmatically, `from rag_system.factory import get_agent, get_indexing_pipeline` is the supported entry point; `IndexingPipeline.run(file_paths)` is the public indexing call. + +--- + +## 8. Operations + +### 8.0 Prerequisites + +* **Python 3.10+** (3.11 recommended — both Docker images are `python:3.11-slim`). +* **Node 20+** for the frontend (`Dockerfile.frontend` is `node:20-alpine`). +* **Ollama** installed and running, with the generation and enrichment models pulled. +* Python dependencies: `pip install -r requirements.txt`. `backend/requirements.txt` is the minimal set for running only the gateway. `pip install ibm-watsonx-ai` is additionally required for `LLM_BACKEND=watsonx`. + +### 8.1 Local launcher ```bash -NEXT_PUBLIC_API_URL=http://localhost:8000 -NEXT_PUBLIC_ENABLE_STREAMING=true -NEXT_PUBLIC_MAX_FILE_SIZE=50MB +python run_system.py # dev mode: all four services +python run_system.py --mode prod # runs `npm run build` before `npm run start` +python run_system.py --no-frontend # backend stack only +python run_system.py --health # HTTP probes each service, exits non-zero if unhealthy +python run_system.py --stop # terminates the processes recorded in logs/run_system.pid +python run_system.py --logs-only # tails logs/*.log without starting anything ``` ---- +`--health` probes `http://localhost:11434/api/tags`, `:8001/health`, `:8000/health` and `:3000/`. On startup the launcher checks that `GENERATION_MODEL` and `ENRICHMENT_MODEL` are present in Ollama. + +### 8.2 Docker -## 9. Monitoring & Observability +`docker compose --env-file docker.env up -d --build` brings up `rag-api`, `backend` and `frontend`; Ollama runs on the host by default (`OLLAMA_HOST=http://host.docker.internal:11434`, with `extra_hosts: host.docker.internal:host-gateway` so it also resolves on Linux). A containerised Ollama is available behind the `with-ollama` profile. `backend` and `rag-api` bind-mount `./backend`, `./lancedb`, `./index_store` and `./shared_uploads`, so both processes share one SQLite file and one vector store. Health checks use `/health` on both Python services and busybox `wget` for the frontend. See [`docker_usage.md`](docker_usage.md) and [`../DOCKER_README.md`](../DOCKER_README.md). -### 9.1 Logging System -- **Structured Logging**: JSON-formatted logs with timestamps -- **Log Levels**: DEBUG, INFO, WARNING, ERROR -- **Log Rotation**: Automatic log file rotation -- **Component Isolation**: Separate logs per service +### 8.3 Health and logging -### 9.2 Health Monitoring -- **Health Endpoints**: `/health` on all services -- **Service Dependencies**: Cascading health checks -- **Performance Metrics**: Response times, error rates -- **Resource Monitoring**: Memory, CPU, disk usage +| Endpoint | Response | +|----------|----------| +| `GET :8000/health` | `{status, ollama_running, available_models, database_stats}` | +| `GET :8001/health` | `{"status": "ok"}` | -### 9.3 Debugging Features -- **Debug Mode**: Detailed operation tracing -- **Query Inspection**: Step-by-step query processing -- **Model Switching Logs**: Embedding model change tracking -- **Error Reporting**: Comprehensive error context +`run_system.py` writes per-service logs to `logs/.log` plus `logs/system.log`, with a coloured console formatter. Logging is plain text — there is no JSON formatter and no log rotation (see [`improvement_plan.md`](improvement_plan.md)). The RAG API routes its handler output through the `logging` module; the agent and both pipelines still print progress with `print()`, which is what you see in `logs/rag-api.log`. --- -## ⚙️ Configuration Modes - -The system supports multiple configuration modes optimized for different use cases: - -### **Default Mode** (`"default"`) -- **Description**: Production-ready pipeline with full features -- **Search**: Hybrid (dense + BM25) with 0.7 dense weight -- **Reranking**: AI-powered ColBERT reranker -- **Query Processing**: Query decomposition enabled -- **Verification**: Grounding verification enabled -- **Performance**: ~3-8 seconds per query -- **Memory**: ~10-16GB (with models loaded) - -### **Fast Mode** (`"fast"`) -- **Description**: Speed-optimized pipeline with minimal overhead -- **Search**: Vector-only (no BM25, no late chunking) -- **Reranking**: Disabled -- **Query Processing**: Single-pass, no decomposition -- **Verification**: Disabled -- **Performance**: ~1-3 seconds per query -- **Memory**: ~8-12GB (with models loaded) - -### **BM25 Mode** (`"bm25"`) -- **Description**: Traditional keyword-based search -- **Search**: BM25 only -- **Use Case**: Exact keyword matching, legacy compatibility - -### **Graph RAG Mode** (`"graph_rag"`) -- **Description**: Knowledge graph integration (currently disabled) -- **Status**: Available for future implementation -- **Use Case**: Relationship-aware retrieval +## 9. Security & Privacy + +* **Local by default** — generation, embedding, reranking and pruning all run locally (Ollama + HuggingFace models). Nothing leaves the machine unless you set `LLM_BACKEND=watsonx`, which sends prompts to IBM Cloud. +* **Model downloads** — HuggingFace models are fetched on first use and cached; that is the only outbound traffic in the default setup. +* **No authentication** — neither server implements auth, and both send `Access-Control-Allow-Origin: *`. Ports 8000 and 8001 must not be exposed to an untrusted network. +* **Session isolation** — retrieval is scoped to the tables of the indexes linked to a session, and the semantic cache is session-scoped by default. Setting `cache_scope: "global"` allows one session's document-derived answer to be returned in another. +* **Deletion** — deleting an index removes its rows and drops its LanceDB table. Uploaded files in `shared_uploads/` are not deleted. --- ## 10. Development & Extension -### 10.1 Architecture Principles -- **Modular Design**: Clear separation of concerns -- **Configuration-Driven**: Behavior controlled via config files -- **Lazy Loading**: Components loaded on-demand -- **Thread Safety**: Proper synchronization for concurrent access +### 10.1 Principles + +* Configuration-driven: profiles in `rag_system/main.py`, construction in `rag_system/factory.py`. +* Lazy loading: embedders, rerankers and the pruner are built on first use and cached on the pipeline instance. +* One factory, one RAG API server, one owner per store. + +### 10.2 Extension points + +| To add… | Do this | +|---------|---------| +| A retriever | Implement the duck-typed contract `retrieve(text_query: str, table_name: str, k: int, search_type: str = "hybrid") -> List[Dict]` and return it from `RetrievalPipeline._get_dense_retriever()`. There is no `BaseRetriever` ABC. | +| A reranker | Plug it into `RetrievalPipeline._get_ai_reranker()`; a `rerankers`-library model only needs `reranker.model_name` + `reranker.model_type`. | +| A chunker | Add a `chunker_mode` branch in `IndexingPipeline.__init__`. | +| An embedding model | Point `EMBEDDING_MODEL` at a HuggingFace repo (contains `/`) or an Ollama tag, then re-index. | +| A pipeline profile | Add an entry to `PIPELINE_CONFIGS` and select it with `RAG_CONFIG_MODE` or `--mode`. | + +### 10.3 Validation + +There is no automated test suite in this repository. What exists: + +* `python system_health_check.py` — imports, configuration dump, LanceDB connectivity, agent construction, embedding dimension, and a sample query against the first available table. +* `python run_system.py --health` — HTTP health probes of all four services. +* `./test_docker_build.sh` — builds the images and probes the container health endpoints. + +Building the automated tests is tracked in [`improvement_plan.md`](improvement_plan.md) §8. + +--- -### 10.2 Extension Points -- **Custom Retrievers**: Implement `BaseRetriever` interface -- **Custom Chunkers**: Extend chunking strategies -- **Custom Models**: Add new embedding or generation models -- **Custom Pipelines**: Create specialized processing workflows +## 11. Known Limitations -### 10.3 Testing Strategy -- **Unit Tests**: Individual component testing -- **Integration Tests**: End-to-end workflow testing -- **Performance Tests**: Load and stress testing -- **Health Checks**: Automated system validation +1. **Streamed chat turns are persisted via a follow-up call.** The stream itself (`POST :8001/chat/stream`) writes nothing to SQLite; the UI posts the completed turn to `POST :8000/sessions/{id}/messages/save` when the stream finishes. Direct stream consumers must do the same to get history. +2. **The RAG API serializes requests** and shares one mutable agent config, so per-request retrieval options persist into subsequent requests. +5. **`enable_latechunk` defaults to `false` on `POST :8001/index`**, so an HTTP index build without that flag produces no late-chunk table even though the `default` profile enables late chunking. The CLI (`python -m rag_system.main index`) uses the profile value. +6. **A reranker that fails to load is skipped**, not replaced — there is no fallback reranker. +7. **`requirements-docker.txt` has drifted** from `requirements.txt` and still lists packages with no importers (see [`improvement_plan.md`](improvement_plan.md) §9). --- -> **Note**: This overview reflects the current implementation as of 2025-01-09. For the latest changes, check the git history and individual component documentation. \ No newline at end of file +> This overview describes the implementation as of 2026-08-08. When behaviour changes, update [`architecture_overview.md`](architecture_overview.md) and this file together. diff --git a/Documentation/triage_system.md b/Documentation/triage_system.md index bed44d4b..6d474890 100644 --- a/Documentation/triage_system.md +++ b/Documentation/triage_system.md @@ -1,60 +1,99 @@ # 🔀 Triage / Routing System -_Maps to `rag_system/agent/loop.Agent._should_use_rag`, `_route_using_overviews`, and the fast-path router in `backend/server.py`._ +_One deterministic gate and one LLM router, in two processes:_ +* _`should_use_rag()` in `backend/server.py` (gateway, port 8000) — deterministic, no LLM call; see "Backend gate" below._ +* _`Agent._triage_query_async` in `rag_system/agent/loop.py` (RAG API, port 8001) — the only LLM routing layer._ ## Purpose -Determine, for every incoming query, whether it should be answered by: -1. **Direct LLM Generation** (no retrieval) — faster, cheaper. -2. **Retrieval-Augmented Generation (RAG)** — when the answer likely requires document context. - -## Decision Signals -| Signal | Source | Notes | -|--------|--------|-------| -| Keyword/regex check | `backend/server.py` (fast path) | Hard-coded quick wins (`what time`, `define`, etc.). | -| Index presence | SQLite (session → indexes) | If no indexes linked, direct LLM. | -| Overview routing | `_route_using_overviews()` | Uses document overviews and enrichment model to predict relevance. | -| LLM router prompt | `agent/loop.py` lines 648-665 | Final arbitrator (Ollama call, JSON output). | - -## High-level Flow +Decide, per query, whether to answer with: +1. **Direct LLM generation** — no retrieval, faster and cheaper; or +2. **Retrieval-Augmented Generation** — search the indexed documents first. + +## Which router actually runs + +| Request path | Router(s) involved | +|--------------|--------------------| +| Streaming chat (UI default): browser → `POST :8001/chat/stream` (`src/lib/api.ts:509`, toggle at `session-chat.tsx:48`, default on) | Agent router only. The backend gateway is not in this path. | +| Non-streaming chat: browser → `POST :8000/sessions//messages` → `POST :8001/chat` | Backend router first (`server.py:382`), then the agent router again inside the RAG API. | +| `POST :8000/chat` (`handle_chat`) | Neither. That endpoint always calls Ollama directly. | + +On the non-streaming path the backend gate decides `use_rag` locally; when it routes to RAG it forwards the query (and `force_rag`, when set) to the RAG API, where the agent triages again and may still choose `direct_answer`. Over-sending to RAG is therefore safe. + +## Agent router (`rag_system/agent/loop.py`) + +Order of evaluation in `_triage_query_async` (`loop.py:175-223`): + +1. **Overview routing** — `_route_via_overviews(query)` (`loop.py:602-644`). Returns `None` immediately when no overviews are loaded (`loop.py:605-607`); otherwise it builds a `DOCUMENT OVERVIEWS:` block from the first 40 loaded overviews (`loop.py:612-613`), interpolates it into the router prompt (`loop.py:615-630`) and calls the utility model with `format="json"`. Parses `{"category": ...}`, defaulting to `rag_query` on a parse failure. +2. **History short-circuit** — if the overview router returned `None` **and** the session already has chat history, the query is treated as a follow-up and routed to `rag_query` without any LLM call (`loop.py:188-193`). +3. **LLM fallback triage** — a two-way classifier (`rag_query` / `direct_answer`) on the utility model, defaulting to `rag_query` if the JSON cannot be parsed. `Agent._normalize_triage()` runs on every verdict and collapses anything that is not an explicit `direct_answer` to `rag_query`, so a small model that emits the retired `graph_query` label still lands on the RAG path. + +`force_rag=true` skips all three: `query_type` is pinned to `rag_query` (`loop.py:268-270`) while the `verify` / `ai_rerank` / `query_decompose` / `compose_sub_answers` / `context_expand` toggles all still apply. + +There is no regex or keyword stage in the agent. + +## Backend gate (`backend/server.py`) + +Since the Phase-2 routing change (see `eval/decisions/phase2-gateway.md`), the gateway makes **no LLM call and reads no files** to route. `should_use_rag()` (module-level, unit-tested in `backend/test_gateway_routing.py`) evaluates in order: + +1. **`force_rag`** ⇒ RAG, unconditionally (also forwarded, so agent triage is skipped too). +2. **No indexes linked** to the session ⇒ direct LLM (nothing to retrieve from). +3. **Smalltalk / assistant-meta** — a whole-message anchored allowlist (greetings, thanks, goodbyes, "who are you?"-style meta) capped at ~6 words ⇒ direct LLM. +4. **Everything else** ⇒ RAG. + +The old per-message enrichment-model router (`_route_using_overviews`) and the keyword/length fallback (`_simple_pattern_routing`) were deleted — the fallback's substring matching misrouted most real document questions (`'hi'` matched *this* and *machine*). Over-sending to RAG is safe because the agent-side triage above can still answer directly; the gateway gate exists only to skip obvious non-retrieval turns at zero cost (~750 ms saved per routed message). + +## Flow + ```mermaid flowchart TD - Q["Incoming Query"] --> S1{Session\nHas Indexes?} - S1 -- no --> LLM["Direct LLM Generation"] - S1 -- yes --> S2{Fast Regex\nHeuristics} - S2 -- match--> LLM - S2 -- no --> S3{Overview\nRelevance > τ?} - S3 -- low --> LLM - S3 -- high --> S4[LLM Router\n(prompt @648)] - S4 -- "route: RAG" --> RAG["Retrieval Pipeline"] - S4 -- "route: DIRECT" --> LLM + Q["Incoming query"] --> FR{force_rag?} + FR -- yes --> RAG["Retrieval pipeline"] + FR -- no --> OV{Overviews loaded?} + OV -- no --> H{Chat history?} + OV -- yes --> R1["Overview router LLM
(utility model, JSON)"] + R1 -- rag_query --> RAG + R1 -- direct_answer --> LLM["Direct LLM answer"] + H -- yes --> RAG + H -- no --> R2["Fallback triage LLM
(rag_query / direct_answer)"] + R2 -- rag_query --> RAG + R2 -- direct_answer --> LLM ``` -## Detailed Sequence (Code-level) -1. **backend/server.py** - * `handle_session_chat()` builds `router_prompt` (line ~435) and makes a **first pass** decision before calling the heavy agent code. -2. **agent.loop._should_use_rag()** - * Re-evaluates using richer features (e.g., token count, query type). -3. **Overviews Phase** (`_route_using_overviews()`) - * Loads JSONL overviews file per index. - * Calls enrichment model (`qwen3:0.6b`) with prompt: _"Does this overview mention … ? "_ → returns yes/no. -4. **LLM Router** (prompt lines 648-665) - * JSON-only response `{ "route": "RAG" | "DIRECT" }`. - -## Interfaces & Dependencies -| Component | Calls / Data | -|-----------|--------------| -| SQLite `chat_sessions` | Reads `indexes` column to know linked index IDs. | -| LanceDB Overviews | Reads `index_store/overviews/.jsonl`. | -| `OllamaClient` | Generates LLM router decision. | - -## Config Flags -* `PIPELINE_CONFIGS.triage.enabled` – global toggle. -* Env var `TRIAGE_OVERVIEW_THRESHOLD` – min similarity score to prefer RAG (default 0.35). - -## Failure / Fallback Modes -1. If overview file missing → skip to LLM router. -2. If LLM router errors → default to RAG (safer) but log warning. +The backend gate is not in this diagram: it is a deterministic pre-filter (force_rag → indexes → smalltalk allowlist) with no LLM call, described above. + +## Overviews: where they come from + +| Step | Code | +|------|------| +| Written at index time, one JSON line per document: `{"doc_id": ..., "overview": ...}` | `rag_system/indexing/overview_builder.py:33-49` | +| Default file `index_store/overviews/overviews.jsonl`; the RAG API overrides it to `index_store/overviews/.jsonl` | `overview_builder.py:24`, `api_server.py:233-234` | +| Loaded per request by the RAG API before the agent runs | `api_server.py:362-366` → `Agent.load_overviews_for_indexes` (`loop.py:79-107`) | +| Falls back to the global `overviews.jsonl` when no per-index file exists | `loop.py:104-107` | + +If no overview file exists for the session, the agent's overview router returns `None` and routing falls through to the history short-circuit or the fallback triage prompt. (The backend gate does not read overview files at all.) + +## Models + +Only the agent-side router costs an LLM call, on the utility model — `Agent._utility_model()` resolves `ENRICHMENT_MODEL` env var → `OLLAMA_CONFIG["enrichment_model"]` → `qwen3.5:4b`. The gateway gate is pure Python. Routing is never charged to the generation model, and a per-request `model` override does not change the routing model: the RAG API applies that override only for the duration of the request via a context manager, and the router reads `enrichment_model`, not `generation_model`. + +## Configuration + +| Knob | Where | Effect | +|------|-------|--------| +| `force_rag` (`forceRag`) | request body on `/chat`, `/chat/stream` (`api_server.py:191`) and on the backend's `/sessions//messages` (`server.py:381`) | Skips triage entirely and forces the RAG path. Surfaced in the UI as the "Always search documents" toggle (`session-chat.tsx:51`, default off). | + +The third outcome, `graph_query`, and the `graph_strategy` config block that armed it were **removed on 2026-08-09** (roadmap item 2.5) along with the rest of the graph module. Evidence: GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, and it costs 41–57× at indexing and up to ~377× in query tokens — [`research/academic-evidence-2026.md`](research/academic-evidence-2026.md) §6. + +There is no global triage on/off switch and no similarity threshold. `PIPELINE_CONFIGS` has no `triage` key, and no `TRIAGE_OVERVIEW_THRESHOLD` environment variable is read anywhere. + +## Failure / fallback modes + +| Failure | Agent | Backend | +|---------|-------|---------| +| No overviews on disk | `_route_via_overviews` returns `None`; history short-circuit or fallback triage decides | n/a — gateway gate reads no files | +| Router LLM returns unparseable JSON / unexpected text | defaults to `rag_query` | n/a — gateway gate makes no LLM call | +| Router LLM call raises | exception propagates to the API handler (500 / SSE `error`) | n/a | --- -_Keep this document updated whenever routing heuristics, thresholds, or prompt wording change._ \ No newline at end of file +_Keep this document updated whenever routing order, prompts, or fallback behaviour change._ diff --git a/Documentation/verifier.md b/Documentation/verifier.md index a1c5bf7d..416b449a 100644 --- a/Documentation/verifier.md +++ b/Documentation/verifier.md @@ -1,49 +1,128 @@ # ✅ Answer Verifier -_File: `rag_system/agent/verifier.py`_ +_File: `rag_system/agent/verifier.py`. Sole caller: `rag_system/agent/loop.py:560-578`._ ## Objective -Assess whether an answer produced by RAG is **grounded** in the retrieved context snippets. +Assess whether an answer produced by the RAG path is **grounded** in the retrieved context snippets, and annotate the answer with the model's self-reported confidence. + +Two interchangeable backends implement it. The **LLM-prompt verifier below is what +ships**; a local NLI/verifier model is opt-in via `VERIFIER_MODEL` (see +[Local verifier model](#local-verifier-model-opt-in)). + +> **`[Confidence: N%]` is UX, not a measurement.** It is whatever the verifier +> emitted, rescaled to a percent. Neither backend is calibrated: an 80% does not +> mean the answer is right four times in five. Swapping the LLM prompt for an NLI +> model changes where the number comes from, not that caveat. + +## Prompt +See `prompt_inventory.md` → `verifier.fact_check` (`verifier.py:25-85`). The prompt carries three few-shot examples and then a `# TASK` block into which the query, the context (clamped to the first 4000 characters at `verifier.py:76`) and the answer are injected. It is sent asynchronously with `format="json"` (`verifier.py:86`). + +Expected response, one line of JSON: -## Prompt (see `prompt_inventory.md` `verifier.fact_check`) -Strict JSON schema: ```jsonc { "verdict": "SUPPORTED" | "NOT_SUPPORTED" | "NEEDS_CLARIFICATION", "is_grounded": true | false, - "reasoning": "< ≤30 words >", + "reasoning": "", "confidence_score": 0-100 } ``` -## Sequence Diagram +It is parsed into a `VerificationResult` (`verifier.py:4-9`) with those four fields. + +## Sequence + ```mermaid sequenceDiagram - participant RP as Retrieval Pipeline + participant A as Agent._run_async participant V as Verifier - participant LLM as Ollama + participant LLM as Ollama (utility model) - RP->>V: query, context, answer - V->>LLM: verification prompt + A->>A: build context_str from result["source_documents"] + A->>V: verify_async(contextual_query, context_str, answer) + V->>LLM: fact-check prompt (format=json) LLM-->>V: JSON verdict - V-->>RP: VerificationResult + V-->>A: VerificationResult + A->>A: append confidence tag to result["answer"] ``` -## Usage Sites -| Caller | Code | When | -|--------|------|------| -| `RetrievalPipeline.answer_stream()` | `pipelines/retrieval_pipeline.py` | If `verify=true` flag from frontend. | -| `Agent.loop.run()` | fallback path | Experimental for composed answers. | +## Call site + +| Caller | Code | When it runs | +|--------|------|--------------| +| `Agent._run_async()` | `rag_system/agent/loop.py`, end of `_run_async` | After every branch (direct answer, decomposed/composed, single-query RAG), when verification is enabled **and** `result["source_documents"]` is non-empty. | + +There is exactly one call site in the repository. `rag_system/pipelines/retrieval_pipeline.py` does not import or reference `Verifier`. Only the async `verify_async()` exists — the synchronous `verify()` was removed (`verifier.py:20`). + +Because the check is gated on non-empty `source_documents`, the `direct_answer` route (which returns `source_documents: []`) is never verified. + +## Configuration + +| Knob | Where | Default | Meaning | +|------|-------|---------|---------| +| `verification.enabled` | `rag_system/main.py:80` (`default` profile) | `true` | Profile-level switch. | +| `verification.enabled` | `rag_system/main.py:109` (`fast` profile) | `false` | Verification off in the speed profile. | +| — | `loop.py:560` | `true` | Fallback used when the profile has no `verification` block. | +| `verify` | HTTP request field on `/chat` and `/chat/stream` (`api_server.py:186`) | not sent ⇒ profile value wins | Per-request override; forwarded to `Agent.run(verify=...)`. Also accepted by the backend gateway as `verify` (`backend/server.py:48`). | +| model | `loop.py`, `Agent.__init__` | utility model (`enrichment_model`, default `qwen3.5:4b`) | Which Ollama model runs the LLM-prompt verifier. Verification runs on the small model, not the answer model. | +| `verification.model` / `VERIFIER_MODEL` | pipeline config, or the env var | unset ⇒ LLM-prompt verifier | A HuggingFace model name switches the backend to a local NLI/verifier model. | +| `verification.threshold` | pipeline config | `0.5` | Score at or above which the local verifier calls an answer grounded. Ignored by the LLM-prompt backend. | +| `VERIFIER_TRUST_REMOTE_CODE` | env var | unset | Must be `1` to load a verifier that ships custom modelling code (e.g. Vectara HHEM). | + +## Local verifier model (opt-in) + +_Roadmap item 2.4, shipped 2026-08-09 as a **seam**: the default is unchanged._ + +```bash +VERIFIER_MODEL=MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli python -m rag_system.main api +``` + +`LocalNLIVerifier` (`rag_system/agent/verifier.py`) loads any HuggingFace +sequence-classification model **lazily on first use**, splits the answer into +sentences, scores each one against the retrieved evidence as the premise, and +takes the **minimum** — one unsupported sentence makes the answer ungrounded, +matching the binary semantics `eval/judge.py` already uses. The "supported" logit +is resolved from `id2label` (`entailment` / `consistent` / `supported` / `1`), +falling back to the last class for binary checkers. + +A model that cannot be loaded **raises** with the list of names that were +checked; it does not silently fall back to the LLM prompt. A verifier that +quietly is not the verifier you configured is worse than an error. + +### Availability, checked 2026-08-09 + +| Candidate | Verdict | +|---|---| +| **ThinknCheck** (arXiv 2604.01652, UPenn, 1B, 78.1 BAcc) | **No public weights.** The paper is real, but a HuggingFace Hub search for `thinkncheck` returns zero models and the paper links no release. Cannot be wired. | +| `ibm-granite/granite-guardian-3.3-8b` | Exists, Apache-2.0 — but 8B / ~16 GB, far over the budget this seam is for. | +| `ibm-granite/granite-guardian-hap-38m` | Exists, 38M, Apache-2.0 — but it is a **hate/abuse/profanity** RoBERTa classifier. Wrong task: it does not score answer-vs-evidence entailment. | +| `MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli` | ✅ MIT, 369 MB, no custom code. Generic NLI. | +| `lytang/MiniCheck-DeBERTa-v3-Large` | ✅ MIT, 1.74 GB, no custom code. Purpose-built grounded claim verification (the baseline ThinknCheck benchmarks against). | +| `vectara/hallucination_evaluation_model` (HHEM-2.1-open) | Apache-2.0, 438 MB, but ships custom modelling code — needs `VERIFIER_TRUST_REMOTE_CODE=1`. | + +The same table is embedded in the code as `VERIFIER_AVAILABILITY_NOTES` and is +printed verbatim when a configured verifier fails to load. + +The UI initialises its verify toggle to `true` (`src/components/ui/session-chat.tsx:49`), so verification is on by default for chat traffic. + +## Effect on the answer + +The verifier does **not** add a field to the response. It mutates the answer string (`loop.py:568-578`): + +* `confidence_score > 0` → appends `" [Confidence: N%]"`. +* Additionally, when `is_grounded` is false **or** the score is below 50 → appends `" [Warning: Low confidence. Groundedness: ]"`. +* `confidence_score == 0` → nothing is appended (0 is treated as a parse failure) and a warning is logged to stdout. + +The API response shape is unchanged: `{"answer": ..., "source_documents": [...]}`. + +## Failure modes + +* Invalid JSON or a missing `response` key → the `except (json.JSONDecodeError, AttributeError)` at `verifier.py:95` returns `VerificationResult(False, "Failed async parse", "NOT_SUPPORTED", 0)`, and because the score is 0 no tag is appended — the answer is returned unannotated. +* If the LLM call itself raises, the exception propagates out of `_run_async` to the API handler, which returns a 500 (or an SSE `error` event on the streaming endpoint). There is no try/except around the `verify_async` call. -## Config -| Flag | Default | Meaning | -|------|---------|---------| -| `verify` | false | Frontend toggle; if true verifier runs. | -| `generation_model` | `qwen3:8b` | Same model as answer generation. +## Cost -## Failure Modes -* If LLM returns invalid JSON → parse exception handled, result = NOT_SUPPORTED. -* If verification call times out → pipeline logs but still returns answer (unverified). +Verification is one extra LLM round-trip per answered query, on the utility model, with a prompt containing up to 4000 characters of context. Set `verify: false` on the request, or run the `fast` profile, to skip it. --- -_Keep updated when schema or usage flags change._ \ No newline at end of file +_Keep updated when the schema, the gating conditions, or the answer annotations change._ diff --git a/README.md b/README.md index 702dcd0d..237da30f 100644 --- a/README.md +++ b/README.md @@ -10,7 +10,7 @@ [![GitHub Forks](https://img.shields.io/github/forks/PromtEngineer/localGPT?style=flat-square)](https://github.com/PromtEngineer/localGPT/network/members) [![GitHub Issues](https://img.shields.io/github/issues/PromtEngineer/localGPT?style=flat-square)](https://github.com/PromtEngineer/localGPT/issues) [![GitHub Pull Requests](https://img.shields.io/github/issues-pr/PromtEngineer/localGPT?style=flat-square)](https://github.com/PromtEngineer/localGPT/pulls) -[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg?style=flat-square)](https://www.python.org/downloads/) +[![Python 3.10+](https://img.shields.io/badge/python-3.10+-blue.svg?style=flat-square)](https://www.python.org/downloads/) [![License](https://img.shields.io/badge/license-MIT-green.svg?style=flat-square)](LICENSE) [![Docker](https://img.shields.io/badge/docker-supported-blue.svg?style=flat-square)](https://www.docker.com/) @@ -28,12 +28,12 @@ LocalGPT is a **fully private, on-premise Document Intelligence platform**. Ask questions, summarise, and uncover insights from your files with state-of-the-art AI—no data ever leaves your machine. -More than a traditional RAG (Retrieval-Augmented Generation) tool, LocalGPT features a **hybrid search engine** that blends semantic similarity, keyword matching, and [Late Chunking](https://jina.ai/news/late-chunking-in-long-context-embedding-models/) for long-context precision. A **smart router** automatically selects between RAG and direct LLM answering for every query, while **contextual enrichment** and sentence-level [Context Pruning](https://huggingface.co/naver/provence-reranker-debertav3-v1) surface only the most relevant content. An independent **verification** pass adds an extra layer of accuracy. +More than a traditional RAG (Retrieval-Augmented Generation) tool, LocalGPT features a **hybrid search engine** that fuses dense vector search with LanceDB's native full-text search, plus [Late Chunking](https://jina.ai/news/late-chunking-in-long-context-embedding-models/) for long-context precision. A **smart router** picks between RAG and direct LLM answering for every query, while **contextual enrichment** and sentence-level [Context Pruning](https://huggingface.co/naver/provence-reranker-debertav3-v1) surface only the most relevant content. An independent **verification** pass adds an extra layer of accuracy. -The architecture is **modular and lightweight**—enable only the components you need. With a pure-Python core and minimal dependencies, LocalGPT is simple to deploy, run, and maintain on any infrastructure.The system has minimal dependencies on frameworks and libraries, making it easy to deploy and maintain. The RAG system is pure python and does not require any additional dependencies. +The architecture is **modular and lightweight**—enable only the components you need. The RAG core is plain Python built on the standard library's HTTP server, with no web framework and no agent framework in the way. ## ▶️ Video -Watch this [video](https://youtu.be/JTbtGH3secI) to get started with LocalGPT. +Watch this [video](https://youtu.be/JTbtGH3secI) to get started with LocalGPT. | Home | Create Index | Chat | |------|--------------|------| @@ -42,73 +42,64 @@ Watch this [video](https://youtu.be/JTbtGH3secI) to get started with LocalGPT. ## ✨ Features - **Utmost Privacy**: Your data remains on your computer, ensuring 100% security. -- **Versatile Model Support**: Seamlessly integrate a variety of open-source models via Ollama. -- **Diverse Embeddings**: Choose from a range of open-source embeddings. +- **Versatile Model Support**: Swap generation models freely via Ollama. +- **Diverse Embeddings**: HuggingFace embedding models (harrier-oss-v1, the Qwen3-Embedding family) or any Ollama embedding tag. - **Reuse Your LLM**: Once downloaded, reuse your LLM without the need for repeated downloads. -- **Chat History**: Remembers your previous conversations (in a session). -- **API**: LocalGPT has an API that you can use for building RAG Applications. -- **GPU, CPU, HPU & MPS Support**: Supports multiple platforms out of the box, Chat with your data using `CUDA`, `CPU`, `HPU (Intel® Gaudi®)` or `MPS` and more! +- **API**: A REST gateway on port 8000 and the RAG API on port 8001 for building your own applications. +- **CUDA, MPS & CPU**: Embedding and reranking pick CUDA, then Apple MPS, then CPU automatically. ### 📖 Document Processing -- **Multi-format Support**: PDF, DOCX, TXT, Markdown, and more (Currently only PDF is supported) -- **Contextual Enrichment**: Enhanced document understanding with AI-generated context, inspired by [Contextual Retrieval](https://www.anthropic.com/news/contextual-retrieval) -- **Batch Processing**: Handle multiple documents simultaneously +- **Formats**: PDF, DOCX, HTML/HTM, Markdown, and TXT, parsed by [Docling](https://github.com/docling-project/docling) +- **OCR fallback**: PDFs with no text layer are re-run through Docling's OCR pipeline; the engine is chosen from whatever is installed (OcrMac on macOS, then EasyOCR, RapidOCR, tesserocr, or the `tesseract` CLI) +- **Contextual Enrichment**: Chunk-level context generated by a small LLM, inspired by [Contextual Retrieval](https://www.anthropic.com/news/contextual-retrieval) +- **Late Chunking**: A second, document-level embedding pass stored in a companion `
_lc` table +- **Document Overviews**: A short per-document summary written to `index_store/overviews/.jsonl` and used by the router ### 🤖 AI-Powered Chat - **Natural Language Queries**: Ask questions in plain English -- **Source Attribution**: Every answer includes document references -- **Smart Routing**: Automatically chooses between RAG and direct LLM responses -- **Query Decomposition**: Breaks complex queries into sub-questions for better answers -- **Semantic Caching**: TTL-based caching with similarity matching for faster responses -- **Session-Aware History**: Maintains conversation context across interactions -- **Answer Verification**: Independent verification pass for accuracy -- **Multiple AI Models**: Ollama for inference, HuggingFace for embeddings and reranking - +- **Source Attribution**: Answers come back with the chunks they were grounded in +- **Smart Routing**: Chooses between RAG and a direct LLM answer per query +- **Query Decomposition**: Splits complex questions into sub-questions, answers each, then composes +- **Reciprocal Rank Fusion**: Vector and full-text hits are fused with RRF — no weights to tune +- **Optional Reranking**: A cross-encoder pass over the fused candidate set, off by default — the first stage already outranks it ([`eval/DECISIONS.md`](eval/DECISIONS.md)) +- **Sentence Pruning**: Optional Provence pruning drops irrelevant sentences from each chunk +- **Semantic Caching**: TTL cache with a 0.98 similarity threshold, scoped to the session +- **Answer Verification**: A second pass that appends `[Confidence: N%]` to the answer ### 🛠️ Developer-Friendly -- **RESTful APIs**: Complete API access for integration -- **Real-time Progress**: Live updates during document processing -- **Flexible Configuration**: Customize models, chunk sizes, and search parameters -- **Extensible Architecture**: Plugin system for custom components +- **RESTful APIs**: Every UI action is a documented HTTP call +- **Streaming phases**: Server-Sent Events expose each pipeline stage as it runs +- **Flexible Configuration**: Models, chunk size, retrieval mode and toggles per request +- **One master config**: `rag_system/main.py` holds every default, overridable by environment variable ### 🎨 Modern Interface - **Intuitive Web UI**: Clean, responsive design - **Session Management**: Organize conversations by topic - **Index Management**: Easy document collection management -- **Real-time Chat**: Streaming responses for immediate feedback +- **Live Progress**: Retrieval, reranking and synthesis stages stream into the chat as they happen --- ## 🚀 Quick Start -Note: The installation is currently only tested on macOS. - ### Prerequisites -- Python 3.8 or higher (tested with Python 3.11.5) -- Node.js 16+ and npm (tested with Node.js 23.10.0, npm 10.9.2) +- Python 3.10+ (3.11 recommended — the Docker images use `python:3.11-slim`) +- Node.js 20+ and npm - Docker (optional, for containerized deployment) - 8GB+ RAM (16GB+ recommended) - Ollama (required for both deployment approaches) -### ***NOTE*** -Before this brach is moved to the main branch, please clone this branch for instalation: - -```bash -git clone -b localgpt-v2 https://github.com/PromtEngineer/localGPT.git -cd localGPT -``` - -### Option 1: Docker Deployment +### Option 1: Docker Deployment ```bash # Clone the repository git clone https://github.com/PromtEngineer/localGPT.git cd localGPT -# Install Ollama locally (required even for Docker) +# Install Ollama locally (recommended even for Docker) curl -fsSL https://ollama.ai/install.sh | sh -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b # Start Ollama ollama serve @@ -120,6 +111,19 @@ ollama serve open http://localhost:3000 ``` +If you would rather not install Ollama on the host, run it as a container instead: + +```bash +./start-docker.sh container +# then pull the models inside the container +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:9b +docker compose --profile with-ollama exec ollama ollama pull qwen3.5:4b +``` + +`./start-docker.sh` (with no argument) uses local Ollama. If nothing is listening on +port 11434 it offers to switch to the containerized Ollama; add `-y` (or set +`NONINTERACTIVE=1`) to take that fallback without a prompt in scripts and CI. + **Docker Management Commands:** ```bash # Check container status @@ -143,20 +147,19 @@ cd localGPT pip install -r requirements.txt # Key dependencies installed: -# - torch==2.4.1, transformers==4.51.0 (AI models) -# - lancedb (vector database) -# - rank_bm25, fuzzywuzzy (search algorithms) -# - sentence_transformers, rerankers (embedding/reranking) -# - docling (document processing) -# - colpali-engine (multimodal processing - support coming soon) +# - torch==2.4.1, transformers==4.51.0 (embedding + reranker models) +# - lancedb (vector store and full-text search) +# - rerankers (cross-encoder reranking) +# - docling (document parsing) +# - fuzzywuzzy, python-Levenshtein (fuzzy matching helpers) # Install Node.js dependencies npm install # Install and start Ollama curl -fsSL https://ollama.ai/install.sh | sh -ollama pull qwen3:0.6b -ollama pull qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ollama serve # Start the system (in a new terminal) @@ -168,32 +171,35 @@ open http://localhost:3000 **System Management:** ```bash -# Check system health (comprehensive diagnostics) +# Check system health (loads the models and runs a sample query) python system_health_check.py -# Check service status and health +# Real HTTP health checks against each service; exits non-zero if one is unhealthy python run_system.py --health -# Start in production mode +# Start in production mode (runs `npm run build` before `next start`) python run_system.py --mode prod -# Skip frontend (backend + RAG API only) +# Skip frontend (Ollama + RAG API + backend only) python run_system.py --no-frontend -# View aggregated logs +# Tail logs/*.log from another shell python run_system.py --logs-only -# Stop all services +# Stop everything recorded in logs/run_system.pid python run_system.py --stop # Or press Ctrl+C in the terminal running python run_system.py ``` **Service Architecture:** -The `run_system.py` launcher manages four key services: -- **Ollama Server** (port 11434): AI model serving -- **RAG API Server** (port 8001): Document processing and retrieval -- **Backend Server** (port 8000): Session management and API endpoints -- **Frontend Server** (port 3000): React/Next.js web interface +The `run_system.py` launcher manages four services and writes their PIDs to `logs/run_system.pid`: +- **Ollama Server** (port 11434): model serving — reused if already running +- **RAG API Server** (port 8001): indexing, retrieval and the agent loop +- **Backend Server** (port 8000): sessions, indexes, uploads, chat history +- **Frontend Server** (port 3000): Next.js web interface (optional — skipped if `npm` is missing) + +On startup the launcher checks that `qwen3.5:9b` and `qwen3.5:4b` are present and +runs `ollama pull` for anything missing. ### Option 3: Manual Component Startup @@ -203,9 +209,10 @@ ollama serve # Terminal 2: Start RAG API python -m rag_system.api_server +# equivalently: python -m rag_system.main api --port 8001 # Terminal 3: Start Backend -cd backend && python server.py +python backend/server.py # Terminal 4: Start Frontend npm run dev @@ -213,6 +220,11 @@ npm run dev # Access at http://localhost:3000 ``` +> Run every command from the repository root. Relative paths (`backend/chat_data.db`, +> `lancedb/`, `index_store/`, `shared_uploads/`) resolve against the current working +> directory, so `cd backend && python server.py` would create a second database at +> `backend/backend/chat_data.db`. + --- ### Detailed Installation @@ -222,62 +234,69 @@ npm run dev **Ubuntu/Debian:** ```bash sudo apt update -sudo apt install python3.8 python3-pip nodejs npm docker.io docker-compose +sudo apt install python3.11 python3-pip nodejs npm docker.io docker-compose-plugin ``` **macOS:** ```bash -brew install python@3.8 node npm docker docker-compose +brew install python@3.11 node docker ``` **Windows:** ```bash -# Install Python 3.8+, Node.js, and Docker Desktop +# Install Python 3.10+, Node.js 20+, and Docker Desktop # Then use PowerShell or WSL2 ``` #### 2. Install AI Models -**Install Ollama (Recommended):** +Only the two Ollama models need an explicit pull. The embedding model +(`microsoft/harrier-oss-v1-0.6b`, 1.2 GB) is downloaded from HuggingFace the +first time it is used; the reranker is only downloaded if you switch reranking +on, which is off by default. + ```bash # Install Ollama curl -fsSL https://ollama.ai/install.sh | sh -# Pull recommended models -ollama pull qwen3:0.6b # Fast generation model -ollama pull qwen3:8b # High-quality generation model -``` - -#### 3. Configure Environment - -```bash -# Copy environment template -cp .env.example .env - -# Edit configuration -nano .env -``` - -**Key Configuration Options:** -```env -# AI Models (referenced in rag_system/main.py) -OLLAMA_HOST=http://localhost:11434 - -# Database Paths (used by backend and RAG system) -DATABASE_PATH=./backend/chat_data.db -VECTOR_DB_PATH=./lancedb - -# Server Settings (used by run_system.py) -BACKEND_PORT=8000 -FRONTEND_PORT=3000 -RAG_API_PORT=8001 - -# Optional: Override default models -GENERATION_MODEL=qwen3:8b -ENRICHMENT_MODEL=qwen3:0.6b -EMBEDDING_MODEL=Qwen/Qwen3-Embedding-0.6B -RERANKER_MODEL=answerdotai/answerai-colbert-small-v1 -``` +# Pull the default models +ollama pull qwen3.5:9b # answer generation +ollama pull qwen3.5:4b # routing, triage, enrichment, verification +``` + +#### 3. Configure Environment (optional) + +Every setting has a working default, so LocalGPT runs with no `.env` at all. +To override one, create a `.env` in the repository root (`rag_system/main.py` +calls `load_dotenv()` at import, before its config constants are evaluated; the +factory calls it again defensively). `.env.example` lists the same variables +with their code defaults. + +| Variable | Default | Read by | +|----------|---------|---------| +| `OLLAMA_HOST` | `http://localhost:11434` | `rag_system/main.py`, `backend/ollama_client.py` | +| `RAG_API_URL` | `http://localhost:8001` | `backend/server.py` (builds `/chat` and `/index`) | +| `NEXT_PUBLIC_API_URL` | `http://localhost:8000` | `src/lib/api.ts` — inlined at `npm run build` | +| `NEXT_PUBLIC_RAG_API_URL` | `http://localhost:8001` | `src/lib/api.ts` — inlined at `npm run build` | +| `DB_PATH` | `backend/chat_data.db` | `backend/database.py` | +| `LANCEDB_PATH` | `storage.lancedb_uri` (`./lancedb`) | `rag_system/main.py` (pipeline profiles), `backend/database.py`, `system_health_check.py` | +| `GENERATION_MODEL` | `qwen3.5:9b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | `rag_system/main.py`, `backend/server.py`, `run_system.py` | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | `rag_system/main.py` | +| `RERANKER_MODEL` | `Qwen/Qwen3-Reranker-4B` (only loaded when reranking is switched on) | `rag_system/main.py` | +| `RAG_CONFIG_MODE` | `default` | `rag_system/api_server.py` (`default` or `fast`) | +| `RAG_API_TIMEOUT` | `600` | `backend/server.py` (seconds to wait for a chat answer) | +| `RAG_API_INDEX_TIMEOUT` | `3600` | `backend/server.py` (seconds to wait for an indexing run) | +| `LLM_BACKEND` | `ollama` | `rag_system/main.py` (`ollama` or `watsonx`) | +| `HF_TOKEN` | unset | HuggingFace, for gated model downloads | + +`NEXT_PUBLIC_*` values are baked into the frontend bundle by `next build`. +Changing them requires a rebuild (`npm run build`, or `docker compose build frontend`). + +> **Changing `EMBEDDING_MODEL` invalidates existing indexes.** Vector width is read +> from the loaded model, and appending vectors of a different width to an existing +> LanceDB table raises an error telling you to rebuild. Re-create your indexes after +> switching embedding models. #### 4. Initialize the System @@ -285,13 +304,13 @@ RERANKER_MODEL=answerdotai/answerai-colbert-small-v1 # Run system health check python system_health_check.py -# Initialize databases +# Initialize the SQLite database python -c "from backend.database import ChatDatabase; ChatDatabase().init_database()" -# Test installation -python -c "from rag_system.main import get_agent; print('✅ Installation successful!')" +# Test the RAG imports +python -c "from rag_system.factory import get_agent; print('✅ Installation successful!')" -# Validate complete setup +# Validate the running services python run_system.py --health ``` @@ -306,32 +325,51 @@ An **index** is a collection of processed documents that you can chat with. #### Using the Web Interface: 1. Open http://localhost:3000 2. Click "Create New Index" -3. Upload your documents (PDF, DOCX, TXT) +3. Upload your documents (PDF, DOCX, TXT, MD, HTML) 4. Configure processing options 5. Click "Build Index" -#### Using Scripts: +#### Using the CLI: ```bash -# Simple script approach -./simple_create_index.sh "My Documents" "path/to/document.pdf" +# Index a single file or a whole directory with the 'default' profile +python -m rag_system.main index ./my_documents + +# Use the speed-optimised profile instead +python -m rag_system.main index ./my_documents --mode fast + +# Ask one question and print the JSON result +python -m rag_system.main chat "What are the key findings?" +``` -# Interactive script +`index` walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, `.md` and `.txt` +files. It writes into the profile's `storage.text_table_name` (`text_pages_v4`), +which is *not* the per-index table the web UI creates. + +#### Using the interactive script (creates a UI-visible index): +```bash +# Guided prompts: name, documents, chunk size, models python create_index_script.py + +# Non-interactive, from a JSON file +python create_index_script.py --create-sample # writes index_config.sample.json +python create_index_script.py --batch index_config.sample.json ``` -#### Using API: +#### Using the HTTP API: ```bash # Create index curl -X POST http://localhost:8000/indexes \ -H "Content-Type: application/json" \ -d '{"name": "My Index", "description": "My documents"}' -# Upload documents +# Upload documents (form field name must be "files") curl -X POST http://localhost:8000/indexes/INDEX_ID/upload \ -F "files=@document.pdf" # Build index -curl -X POST http://localhost:8000/indexes/INDEX_ID/build +curl -X POST http://localhost:8000/indexes/INDEX_ID/build \ + -H "Content-Type: application/json" \ + -d '{"chunk_size": 512, "enable_enrich": true, "enable_latechunk": true}' ``` ### 2. Start Chatting @@ -345,96 +383,116 @@ Once your index is built: ### 3. Advanced Features -#### Custom Model Configuration +#### Per-session and per-request model choice ```bash -# Use different models for different tasks +# The session's default generation model curl -X POST http://localhost:8000/sessions \ -H "Content-Type: application/json" \ - -d '{ - "title": "High Quality Session", - "model": "qwen3:8b", - "embedding_model": "Qwen/Qwen3-Embedding-4B" - }' -``` + -d '{"title": "High Quality Session", "model": "qwen3.6:27b"}' -#### Batch Document Processing -```bash -# Process multiple documents at once -python demo_batch_indexing.py --config batch_indexing_config.json +# Override it for one message +curl -X POST http://localhost:8000/sessions/SESSION_ID/messages \ + -H "Content-Type: application/json" \ + -d '{"message": "Summarise section 3", "model": "qwen3.5:4b"}' ``` +The embedding model is a property of the index, not the session — choose it when +you build the index. + #### API Integration ```python import requests -# Chat with your documents via API -response = requests.post('http://localhost:8000/chat', json={ +# Talk to the RAG API directly +response = requests.post('http://localhost:8001/chat', json={ 'query': 'What are the key findings in the research papers?', 'session_id': 'your-session-id', - 'search_type': 'hybrid', - 'retrieval_k': 20 + 'retrieval_mode': 'hybrid', + 'retrieval_k': 20, }) -print(response.json()['response']) +print(response.json()['answer']) ``` --- ## 🔧 Configuration +All defaults live in `rag_system/main.py`. Every model name there can be +overridden with the environment variables listed above. + ### Model Configuration -LocalGPT supports multiple AI model providers with centralized configuration: +| Role | Default | Documented options | +|------|---------|--------------------| +| Generation (answers) | `qwen3.5:9b` | `qwen3.6:27b` (high-end, ~17GB), `qwen3.5:4b` (light) | +| Enrichment / utility (routing, triage, decomposition, verification) | `qwen3.5:4b` | `qwen3.5:2b` (light) | +| Embedding | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024 dims) | `Qwen/Qwen3-Embedding-4B` (2560 dims, 32K context, for multilingual / long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024 dims) | +| Reranker (off by default) | `Qwen/Qwen3-Reranker-4B` | `BAAI/bge-reranker-v2-m3` (low latency), `answerdotai/answerai-colbert-small-v1`, `Qwen/Qwen3-Reranker-0.6B` | -#### Ollama Models (Local Inference) ```python +# rag_system/main.py OLLAMA_CONFIG = { - "host": "http://localhost:11434", - "generation_model": "qwen3:8b", # Main text generation - "enrichment_model": "qwen3:0.6b" # Lightweight routing/enrichment + "host": os.getenv("OLLAMA_HOST", "http://localhost:11434"), + "generation_model": os.getenv("GENERATION_MODEL", "qwen3.5:9b"), + "enrichment_model": os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b"), } -``` -#### External Models (HuggingFace Direct) -```python EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # 1024 dimensions - "reranker_model": "answerdotai/answerai-colbert-small-v1", # ColBERT reranker - "fallback_reranker": "BAAI/bge-reranker-base" # Backup reranker + "embedding_model": os.getenv("EMBEDDING_MODEL", "microsoft/harrier-oss-v1-0.6b"), + "reranker_model": os.getenv("RERANKER_MODEL", "Qwen/Qwen3-Reranker-4B"), } ``` +Embedding dimensions are never hardcoded — they are measured from the vectors the +loaded model produces. If the reranker fails to load, the pipeline logs a warning +and continues **without** reranking rather than falling back to another model. + +Vision / multimodal models are not part of the pipeline. PDF parsing and OCR are +handled by Docling. Models such as GLM-OCR or Qwen3-VL could be added as a +pre-processing step, but **they are not integrated today**. + ### Pipeline Configuration -LocalGPT offers two main pipeline configurations: +`PIPELINE_CONFIGS` has exactly two profiles. Select one with `RAG_CONFIG_MODE` +(RAG API) or `--mode` (CLI). #### Default Pipeline (Production-Ready) ```python "default": { - "description": "Production-ready pipeline with hybrid search, AI reranking, and verification", + "description": "Production-ready pipeline with hybrid search, query decomposition, and verification", "storage": { "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "bm25_path": "./index_store/bm25" + "text_table_name": "text_pages_v4" }, "retrieval": { - "retriever": "multivector", "search_type": "hybrid", - "late_chunking": {"enabled": True}, - "dense": {"enabled": True, "weight": 0.7}, - "bm25": {"enabled": True} + "latechunk": {"enabled": True}, + "dense": {"enabled": True}, + "retry": {"enabled": True, "min_top_score": 0.12, "max_attempts": 1} }, + "embedding_model_name": EXTERNAL_MODELS["embedding_model"], + # Off by default: the first stage already outranks the cheap cross-encoder, + # and the reranker that does win costs ~12.7s/query (eval/DECISIONS.md). "reranker": { - "enabled": True, - "type": "ai", + "enabled": False, + "model_type": "cross-encoder", "strategy": "rerankers-lib", - "model_name": "answerdotai/answerai-colbert-small-v1", + "model_name": EXTERNAL_MODELS["reranker_model"], "top_k": 10 }, - "query_decomposition": {"enabled": True, "max_sub_queries": 3}, + "query_decomposition": {"enabled": True, "compose_from_sub_answers": True}, "verification": {"enabled": True}, "retrieval_k": 20, - "contextual_enricher": {"enabled": True, "window_size": 1} + "context_window_size": 0, + "semantic_cache_threshold": 0.98, + "cache_scope": "session", + "contextual_enricher": {"enabled": True, "window_size": 1}, + "indexing": { + "embedding_batch_size": 50, + "enrichment_batch_size": 10, + "enable_progress_tracking": True + } } ``` @@ -444,28 +502,35 @@ LocalGPT offers two main pipeline configurations: "description": "Speed-optimized pipeline with minimal overhead", "retrieval": { "search_type": "vector_only", - "late_chunking": {"enabled": False} + "latechunk": {"enabled": False}, + "dense": {"enabled": True} }, "reranker": {"enabled": False}, "query_decomposition": {"enabled": False}, "verification": {"enabled": False}, "retrieval_k": 10, - "contextual_enricher": {"enabled": False} + "contextual_enricher": {"enabled": False}, + "indexing": { + "embedding_batch_size": 100, + "enrichment_batch_size": 50, + "enable_progress_tracking": False + } } ``` -### Search Configuration +### Retrieval Modes + +`retrieval_mode` (wire name; `search_type` inside the pipeline config) accepts: + +| Value | Behaviour | +|-------|-----------| +| `hybrid` *(default)* | Vector and LanceDB full-text legs run in parallel and are fused with Reciprocal Rank Fusion | +| `vector_only` | Dense vector search only | +| `fts_only` | LanceDB full-text search only | + +Anything else is rejected with HTTP 400 by the RAG API. There is no +`dense_weight` / `denseWeight` knob — RRF needs no weights. -```python -SEARCH_CONFIG = { - 'hybrid': { - 'dense_weight': 0.7, - 'sparse_weight': 0.3, - 'retrieval_k': 20, - 'reranker_top_k': 10 - } -} -``` --- ## 🛠️ Troubleshooting @@ -475,10 +540,10 @@ SEARCH_CONFIG = { #### Installation Problems ```bash # Check Python version -python --version # Should be 3.8+ +python --version # 3.10+ required, 3.11 recommended # Check dependencies -pip list | grep -E "(torch|transformers|lancedb)" +pip list | grep -E "(torch|transformers|lancedb|docling|rerankers)" # Reinstall dependencies pip install -r requirements.txt --force-reinstall @@ -491,7 +556,8 @@ ollama list curl http://localhost:11434/api/tags # Pull missing models -ollama pull qwen3:0.6b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b ``` #### Database Issues @@ -499,11 +565,17 @@ ollama pull qwen3:0.6b # Check database connectivity python -c "from backend.database import ChatDatabase; db = ChatDatabase(); print('✅ Database OK')" -# Reset database (WARNING: This deletes all data) +# Reset database (WARNING: This deletes all sessions, messages and index metadata) rm backend/chat_data.db python -c "from backend.database import ChatDatabase; ChatDatabase().init_database()" ``` +#### Dimension mismatch after changing the embedding model +``` +ValueError: ... changing the embedding model requires rebuilding the index +``` +Delete the affected index in the UI (or `DELETE /indexes/{id}`) and rebuild it. + #### Performance Issues ```bash # Check system resources @@ -512,263 +584,239 @@ python system_health_check.py # Monitor memory usage htop # or Task Manager on Windows -# Optimize for low-memory systems -export PYTORCH_CUDA_ALLOC_CONF=max_split_size_mb:512 +# Use lighter models (the default embedder is already the small one at 1.2GB) +export GENERATION_MODEL=qwen3.5:4b ``` ### Getting Help -1. **Check Logs**: The system creates structured logs in the `logs/` directory: - - `logs/system.log`: Main system events and errors - - `logs/ollama.log`: Ollama server logs - - `logs/rag-api.log`: RAG API processing logs - - `logs/backend.log`: Backend server logs - - `logs/frontend.log`: Frontend build and runtime logs +1. **Check Logs**: `run_system.py` writes structured logs to `logs/`: + - `logs/system.log`: launcher events + - `logs/ollama.log`, `logs/rag-api.log`, `logs/backend.log`, `logs/frontend.log`: per-service output + - `logs/run_system.pid`: PIDs used by `--stop` -2. **System Health**: Run comprehensive diagnostics: +2. **System Health**: Run diagnostics: ```bash - python system_health_check.py # Full system diagnostics - python run_system.py --health # Service status check + python system_health_check.py # loads models, runs a sample query + python run_system.py --health # HTTP checks, non-zero exit on failure ``` -3. **Health Endpoints**: Check individual service health: +3. **Health Endpoints**: - Backend: `http://localhost:8000/health` - RAG API: `http://localhost:8001/health` - Ollama: `http://localhost:11434/api/tags` -4. **Documentation**: Check the [Technical Documentation](TECHNICAL_DOCS.md) +4. **Documentation**: See [Documentation/system_overview.md](Documentation/system_overview.md), and [Documentation/design_rationale.md](Documentation/design_rationale.md) for why each component is built the way it is — with the evidence and the eval numbers behind every default, plus a "deliberately not implemented" list 5. **GitHub Issues**: Report bugs and request features -6. **Community**: Join our Discord/Slack community +6. **Community**: Join our Discord --- ## 🔗 API Reference -### Core Endpoints +Two HTTP services. The backend gateway on **:8000** owns sessions, indexes, +uploads and chat history; the RAG API on **:8001** owns retrieval and indexing. +Both accept `snake_case` and `camelCase` spellings of every option and normalise +them to one canonical key. + +### Backend gateway — http://localhost:8000 + +```http +GET /health # {status, ollama_running, available_models, database_stats} +GET /models # {generation_models, embedding_models} + +GET /sessions # {sessions, total} +POST /sessions # {title?, model?} -> 201 {session, session_id} +GET /sessions/{id} # {session, messages} +DELETE /sessions/{id} # {deleted: true} +GET /sessions/cleanup # removes empty sessions +POST /sessions/{id}/rename # {title} -> {message, session} +GET /sessions/{id}/documents # {session, files, file_count} +GET /sessions/{id}/indexes # {indexes, total} +POST /sessions/{id}/indexes/{index_id} # link an index to a session +POST /sessions/{id}/upload # multipart/form-data, field "files" +POST /sessions/{id}/index # index this session's uploads +POST /sessions/{id}/messages # chat (see below) + +GET /indexes # {indexes, total} +POST /indexes # {name, description?, metadata?} -> 201 {index_id} +GET /indexes/{id} +DELETE /indexes/{id} # also drops the LanceDB table +POST /indexes/{id}/upload # multipart/form-data, field "files" +POST /indexes/{id}/build # build/rebuild from uploaded documents + +POST /chat # session-less Ollama chat, no retrieval +``` + +#### Session chat -#### Chat API ```http -# Session-based chat (recommended) -POST /sessions/{session_id}/chat +POST /sessions/{session_id}/messages Content-Type: application/json { - "query": "What are the main topics discussed?", - "search_type": "hybrid", + "message": "What are the main topics discussed?", + "model": "qwen3.5:9b", + "retrieval_mode": "hybrid", "retrieval_k": 20, + "reranker_top_k": 10, + "context_window_size": 1, "ai_rerank": true, - "context_window_size": 5 -} - -# Legacy chat endpoint -POST /chat -Content-Type: application/json - -{ - "query": "What are the main topics discussed?", - "session_id": "uuid", - "search_type": "hybrid", - "retrieval_k": 20 + "context_expand": true, + "query_decompose": true, + "compose_sub_answers": true, + "verify": true, + "provence_prune": false, + "provence_threshold": 0.1, + "force_rag": false } ``` -#### Index Management -```http -# Create index -POST /indexes -Content-Type: application/json -{ - "name": "My Index", - "description": "Description", - "config": "default" -} - -# Get all indexes -GET /indexes - -# Get specific index -GET /indexes/{id} - -# Upload documents to index -POST /indexes/{id}/upload -Content-Type: multipart/form-data -files: [file1.pdf, file2.pdf, ...] - -# Build index (process uploaded documents) -POST /indexes/{id}/build -Content-Type: application/json +Response: +```json { - "config_mode": "default", - "enable_enrich": true, - "chunk_size": 512 + "response": "…", + "session": { "...": "updated session row" }, + "source_documents": [], + "used_rag": true } - -# Delete index -DELETE /indexes/{id} ``` -#### Session Management -```http -# Create session -POST /sessions -Content-Type: application/json -{ - "title": "My Session", - "model": "qwen3:0.6b" -} - -# Get all sessions -GET /sessions - -# Get specific session -GET /sessions/{session_id} +The backend decides per message whether to answer directly with Ollama or to +forward to the RAG API. `force_rag: true` skips that decision and always calls the +RAG API. Both the user message and the answer are written to SQLite on this path. -# Get session documents -GET /sessions/{session_id}/documents +### RAG API — http://localhost:8001 -# Get session indexes -GET /sessions/{session_id}/indexes - -# Link index to session -POST /sessions/{session_id}/indexes/{index_id} - -# Delete session -DELETE /sessions/{session_id} - -# Rename session -POST /sessions/{session_id}/rename -Content-Type: application/json -{ - "new_title": "Updated Session Name" -} +```http +GET /health # {"status": "ok"} +GET /models # {generation_models, embedding_models} +POST /chat # {answer, source_documents} +POST /chat/stream # Server-Sent Events, terminated by a "complete" event +POST /index # run the indexing pipeline over file_paths ``` -### Advanced Features - -#### Query Decomposition -The system can break complex queries into sub-questions for better answers: -```http -POST /sessions/{session_id}/chat -Content-Type: application/json +#### `POST /chat` and `POST /chat/stream` +```json { - "query": "Compare the methodologies and analyze their effectiveness", + "query": "Explain the methodology", + "session_id": "uuid", + "table_name": "text_pages_", + "model": "qwen3.5:9b", + "retrieval_mode": "hybrid", + "retrieval_k": 20, + "context_window_size": 1, + "reranker_top_k": 10, + "ai_rerank": true, + "context_expand": true, "query_decompose": true, - "compose_sub_answers": true + "compose_sub_answers": true, + "verify": true, + "force_rag": false, + "provence_prune": false, + "provence_threshold": 0.1 } ``` -#### Answer Verification -Independent verification pass for accuracy using a separate verification model: -```http -POST /sessions/{session_id}/chat -Content-Type: application/json +`/chat` returns `{"answer": "...", "source_documents": [...]}`. There is no +top-level `confidence` field — when verification runs it appends +`[Confidence: N%]` (and a low-confidence warning) to the answer text itself. -{ - "query": "What are the key findings?", - "verify": true -} -``` - -#### Contextual Enrichment -Document context enrichment during indexing for better understanding: -```bash -# Enable during index building -POST /indexes/{id}/build -{ - "enable_enrich": true, - "window_size": 2 -} -``` +`/chat/stream` emits `data: {"type": "", "data": {...}}` lines and ends +with a `complete` event carrying the same object `/chat` would return. -#### Late Chunking -Better context preservation by chunking after embedding: -```bash -# Configure in pipeline -"late_chunking": {"enabled": true} -``` +`force_rag: true` skips the agent's triage step so the query always goes through +retrieval; `verify`, `ai_rerank`, `query_decompose` and `compose_sub_answers` +still apply. An unsupported `retrieval_mode` is rejected with HTTP 400. -#### Streaming Chat -```http -POST /chat/stream -Content-Type: application/json +#### `POST /index` +```json { - "query": "Explain the methodology", + "file_paths": ["/abs/path/doc1.pdf", "/abs/path/doc2.pdf"], "session_id": "uuid", - "stream": true + "table_name": "text_pages_", + "chunk_size": 512, + "window_size": 2, + "retrieval_mode": "hybrid", + "enable_enrich": true, + "enable_latechunk": false, + "enable_docling_chunk": false, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, + "batch_size_enrich": 25 } ``` -#### Batch Processing -```bash -# Using the batch indexing script -python demo_batch_indexing.py --config batch_indexing_config.json +`file_paths` is required; the values above are the defaults applied when a field +is omitted. Response: -# Example batch configuration (batch_indexing_config.json): +```json { - "index_name": "Sample Batch Index", - "index_description": "Example batch index configuration", - "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" - ], - "processing": { + "message": "Indexing process for 2 file(s) completed successfully.", + "table_name": "text_pages_", + "latechunk": false, + "docling_chunk": false, + "indexing_config": { "chunk_size": 512, - "chunk_overlap": 64, - "enable_enrich": true, - "enable_latechunk": true, - "enable_docling": true, - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:0.6b", "retrieval_mode": "hybrid", - "window_size": 2 + "window_size": 2, + "enable_enrich": true, + "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": "qwen3.5:4b", + "overview_model_name": "qwen3.5:4b", + "batch_size_embed": 50, + "batch_size_enrich": 25 } } ``` -```http -# API endpoint for batch processing -POST /batch/index -Content-Type: application/json +`retrieval_mode` at index time is validated and recorded with the index config; +it takes effect at query time. Indexing is synchronous — the call returns when +the pipeline finishes, which is why the backend allows up to +`RAG_API_INDEX_TIMEOUT` (default 3600s) for it. -{ - "file_paths": ["doc1.pdf", "doc2.pdf"], - "config": { - "chunk_size": 512, - "enable_enrich": true, - "enable_latechunk": true, - "enable_docling": true - } -} -``` +For the full route table see [Documentation/api_reference.md](Documentation/api_reference.md). -For complete API documentation, see [API_REFERENCE.md](API_REFERENCE.md). +### Known limitations + +- **The RAG API is single-threaded.** Requests are serialised: one chat or + indexing run at a time. The backend gateway is threaded, so it stays responsive, + but a long RAG call blocks the next one. +- **Streamed turns are persisted after the fact.** The chat UI streams from + `:8001/chat/stream` directly and, when the stream completes, saves the finished + turn through the gateway (`POST /sessions/{id}/messages/save`). If the browser is + closed mid-stream, that turn is not saved. +- **Index metadata is per index, not per session.** Choosing a different embedding + model requires rebuilding the index. --- ## 🏗️ Architecture -LocalGPT is built with a modular, scalable architecture: - ```mermaid graph TB - UI[Web Interface] --> API[Backend API] - API --> Agent[RAG Agent] + UI[Next.js UI :3000] --> API[Backend gateway :8000] + UI -. "SSE /chat/stream" .-> RAGAPI + API --> RAGAPI[RAG API :8001] + RAGAPI --> Agent[RAG Agent] Agent --> Retrieval[Retrieval Pipeline] - Agent --> Generation[Generation Pipeline] + Agent --> Ollama[Ollama :11434] - Retrieval --> Vector[Vector Search] - Retrieval --> BM25[BM25 Search] - Retrieval --> Rerank[Reranking] + Retrieval --> Vector[Vector search] + Retrieval --> FTS[LanceDB full-text search] + Vector --> RRF[Reciprocal Rank Fusion] + FTS --> RRF + RRF --> Rerank["Cross-encoder rerank (optional, off by default)"] Vector --> LanceDB[(LanceDB)] - BM25 --> BM25DB[(BM25 Index)] + FTS --> LanceDB - Generation --> Ollama[Ollama Models] - Generation --> HF[Hugging Face Models] - - API --> SQLite[(SQLite DB)] + API --> SQLite[(SQLite: sessions, messages, indexes)] + RAGAPI --> SQLite ``` Overview of the Retrieval Agent @@ -778,53 +826,53 @@ graph TD classDef llmcall fill:#e6f3ff,stroke:#007bff; classDef pipeline fill:#e6ffe6,stroke:#28a745; classDef cache fill:#fff3e0,stroke:#fd7e14; - classDef logic fill:#f8f9fa,stroke:#6c757d; - classDef thread stroke-dasharray: 5 5; - A(Start: Agent.run) --> B_asyncio.run(_run_async); - B --> C{_run_async}; + A(Start: Agent.run) --> C{_run_async}; - C --> C1[Get Chat History]; - C1 --> T1[Build Triage Prompt
Query + Doc Overviews ]; - T1 --> T2["(asyncio.to_thread)
LLM Triage: RAG or LLM_DIRECT?"]; class T2 llmcall,thread; + C --> C1[Get chat history]; + C1 --> T0{force_rag?}; + T0 -- Yes --> RAG_Path; + T0 -- No --> T1[Route via document overviews]; + T1 --> T2["LLM triage fallback:
rag_query | direct_answer"]; class T2 llmcall; T2 --> T3{Decision?}; - T3 -- RAG --> RAG_Path; - T3 -- LLM_DIRECT --> LLM_Path; + T3 -- rag_query --> RAG_Path; + T3 -- direct_answer --> LLM_Path; subgraph RAG Path - RAG_Path --> R1[Format Query + History]; - R1 --> R2["(asyncio.to_thread)
Generate Query Embedding"]; class R2 pipeline,thread; - R2 --> R3{{Check Semantic Cache}}; class R3 cache; - R3 -- Hit --> R_Cache_Hit(Return Cached Result); - R_Cache_Hit --> R_Hist_Update; - R3 -- Miss --> R4{Decomposition
Enabled?}; - - R4 -- Yes --> R5["(asyncio.to_thread)
Decompose Raw Query"]; class R5 llmcall,thread; - R5 --> R6{{Run Sub-Queries
Parallel RAG Pipeline}}; class R6 pipeline,thread; - R6 --> R7[Collect Results & Docs]; - R7 --> R8["(asyncio.to_thread)
Compose Final Answer"]; class R8 llmcall,thread; - R8 --> V1(RAG Answer); - - R4 -- No --> R9["(asyncio.to_thread)
Run Single Query
(RAG Pipeline)"]; class R9 pipeline,thread; + RAG_Path --> R1[Format query + history]; + R1 --> R2[Embed query]; class R2 pipeline; + R2 --> R3{{Semantic cache
threshold 0.98, session-scoped}}; class R3 cache; + R3 -- Hit --> FinalResult; + R3 -- Miss --> R4{Decomposition enabled?}; + + R4 -- Yes --> R5[Decompose query]; class R5 llmcall; + R5 --> R6{{Run sub-queries through the retrieval pipeline}}; class R6 pipeline; + R6 --> R8[Compose final answer]; class R8 llmcall; + R8 --> V1(RAG answer); + + R4 -- No --> R9[Run single query through the retrieval pipeline]; class R9 pipeline; R9 --> V1; - V1 --> V2{{Verification
await verify_async}}; class V2 llmcall; - V2 --> V3(Final RAG Result); - V3 --> R_Cache_Store{{Store in Semantic Cache}}; class R_Cache_Store cache; + V1 --> V2{{Verification}}; class V2 llmcall; + V2 --> R_Cache_Store{{Store in semantic cache}}; class R_Cache_Store cache; R_Cache_Store --> FinalResult; end subgraph Direct LLM Path - LLM_Path --> L1[Format Query + History]; - L1 --> L2["(asyncio.to_thread)
Generate Direct LLM Answer
(No RAG)"]; class L2 llmcall,thread; - L2 --> FinalResult(Final Direct Result); + LLM_Path --> L2[Generate answer without retrieval]; class L2 llmcall; + L2 --> FinalResult(Final result); end - FinalResult --> R_Hist_Update(Update Chat History); - R_Hist_Update --> ZZZ(End: Return Result); + FinalResult --> R_Hist_Update(Update in-memory chat history); + R_Hist_Update --> ZZZ["End: return answer + source_documents"]; ``` +Inside the retrieval pipeline a query runs: embed → hybrid retrieve (vector + +FTS, fused with RRF) → optional late-chunk leg → optional cross-encoder rerank +(off by default) → context window expansion → optional Provence sentence +pruning → synthesis. + --- ## 🤝 Contributing @@ -844,7 +892,8 @@ npm install # Install Ollama and models curl -fsSL https://ollama.ai/install.sh | sh -ollama pull qwen3:0.6b qwen3:8b +ollama pull qwen3.5:9b +ollama pull qwen3.5:4b # Verify setup python system_health_check.py @@ -879,7 +928,7 @@ This project is licensed under the MIT License - see the [LICENSE](LICENSE) file ## 📞 Support -- **Documentation**: [Technical Docs](TECHNICAL_DOCS.md) +- **Documentation**: [Documentation/system_overview.md](Documentation/system_overview.md) - **Issues**: [GitHub Issues](https://github.com/PromtEngineer/localGPT/issues) - **Discussions**: [GitHub Discussions](https://github.com/PromtEngineer/localGPT/discussions) - **Business Deployment and Customization**: [Contact Us](https://tally.so/r/wv6R2d) @@ -890,3 +939,5 @@ This project is licensed under the MIT License - see the [LICENSE](LICENSE) file ## Star History [![Star History Chart](https://api.star-history.com/svg?repos=PromtEngineer/localGPT&type=Date)](https://star-history.com/#PromtEngineer/localGPT&Date) + + diff --git a/WATSONX_README.md b/WATSONX_README.md index a21bcbc6..4358fcaa 100644 --- a/WATSONX_README.md +++ b/WATSONX_README.md @@ -1,91 +1,124 @@ # Watson X Integration with Granite Models -This branch adds support for IBM Watson X AI with Granite models as an alternative to Ollama for running LocalGPT. +localGPT can run its LLM calls against IBM watsonx.ai Granite models instead of a local +Ollama server. ## Overview -LocalGPT now supports two LLM backends: -1. **Ollama** (default): Run models locally using Ollama -2. **Watson X**: Use IBM's Granite models hosted on Watson X AI +`rag_system` supports two LLM backends, selected with the `LLM_BACKEND` environment +variable: -## What Changed +1. **Ollama** (`ollama`, the default) — models run locally. +2. **Watson X** (`watsonx`) — Granite models hosted on IBM watsonx.ai. -- Added `WatsonXClient` class in `rag_system/utils/watsonx_client.py` that provides an Ollama-compatible interface for Watson X -- Updated `factory.py` and `main.py` to support backend switching via environment variable -- Added `ibm-watsonx-ai` SDK dependency to `requirements.txt` -- Configuration now supports both backends through environment variables +The switch is made in `rag_system/factory.py::_build_llm_client()`, which returns either an +`OllamaClient` or a `WatsonXClient` (`rag_system/utils/watsonx_client.py`) together with +the matching config dict (`OLLAMA_CONFIG` or `WATSONX_CONFIG` from `rag_system/main.py`). +Both dicts expose the same `generation_model` / `enrichment_model` keys, so the agent, the +retrieval pipeline and the indexing pipeline are unchanged. -## Prerequisites +### What the backend switch does and does not cover + +Switched to Watson X: + +- Answer generation and sub-answer composition (`generation_model`). +- Query routing, triage, query decomposition, contextual enrichment, document overviews and + answer verification (`enrichment_model`). -To use Watson X with Granite models, you need: +**Always local, regardless of `LLM_BACKEND`:** + +- **Embeddings.** `rag_system/indexing/representations.py::select_embedder()` returns a + Hugging Face model when `EMBEDDING_MODEL` contains a `/`, and an Ollama embedder + otherwise. There is no Watson X embedding path, and `WatsonXClient` exposes no embedding + method. Point `EMBEDDING_MODEL` at a Hugging Face repo (the default + `microsoft/harrier-oss-v1-0.6b`) so no Ollama server is needed for indexing. +- **Reranking** (off by default; `Qwen/Qwen3-Reranker-4B` when switched on) and **Provence + sentence pruning** — both are local `transformers` models. +- **The backend gateway's direct-LLM path.** `backend/server.py` answers non-document + questions and makes its routing decision with `backend/ollama_client.py`, which always + talks to `OLLAMA_HOST`. Watson X only serves requests that reach the RAG API on port + 8001. + +## Prerequisites -1. IBM Cloud account with Watson X access -2. Watson X API key -3. Watson X project ID +1. IBM Cloud account with watsonx.ai access +2. A watsonx.ai API key +3. A watsonx.ai project ID -### Getting Your Credentials +### Getting your credentials 1. Go to [IBM Cloud](https://cloud.ibm.com/) -2. Navigate to Watson X AI service +2. Navigate to the watsonx.ai service 3. Create or select a project 4. Get your API key from IBM Cloud IAM -5. Copy your project ID from the Watson X project settings +5. Copy your project ID from the project settings + +## Installation + +The SDK is **not** installed by the root `requirements.txt` (it is listed there as a +commented-out optional extra). Install it explicitly: + +```bash +pip install "ibm-watsonx-ai>=1.3.39" +``` + +`rag_system/requirements.txt` — the RAG-only dependency list — does pin it, so +`pip install -r rag_system/requirements.txt` also gets you the SDK. + +Without the package, `WatsonXClient.__init__` raises +`ImportError: ibm-watsonx-ai package is required.` as soon as the agent is constructed. ## Configuration -### Environment Variables +Copy the example file and fill in your credentials: + +```bash +cp env.example.watsonx .env +``` -Create a `.env` file or set these environment variables: +The variables, with the defaults from `rag_system/main.py`: ```bash # Choose LLM backend (default: ollama) LLM_BACKEND=watsonx -# Watson X Configuration +# Watson X credentials WATSONX_API_KEY=your_api_key_here WATSONX_PROJECT_ID=your_project_id_here WATSONX_URL=https://us-south.ml.cloud.ibm.com -# Model Configuration +# Model configuration WATSONX_GENERATION_MODEL=ibm/granite-13b-chat-v2 WATSONX_ENRICHMENT_MODEL=ibm/granite-8b-japanese ``` -### Available Granite Models +`WATSONX_API_KEY` and `WATSONX_PROJECT_ID` are mandatory: `_build_llm_client()` raises +`ValueError: Watson X configuration incomplete.` when either is empty. -Watson X offers several Granite models: -- `ibm/granite-13b-chat-v2` - General purpose chat model -- `ibm/granite-13b-instruct-v2` - Instruction-following model -- `ibm/granite-20b-multilingual` - Multilingual support -- `ibm/granite-8b-japanese` - Lightweight Japanese model -- `ibm/granite-3b-code-instruct` - Code generation model +Use model ids that exist in your watsonx.ai instance — the two above are only the code +defaults, and `ibm/granite-8b-japanese` in particular is unlikely to be the utility model +you want. IBM's +[supported foundation models](https://www.ibm.com/docs/en/watsonx/saas?topic=solutions-supported-foundation-models) +page lists what is currently available. -For a full list of available models, visit the [Watson X documentation](https://www.ibm.com/docs/en/watsonx/saas?topic=solutions-supported-foundation-models). - -## Installation +## Usage -1. Install the Watson X SDK: -```bash -pip install ibm-watsonx-ai>=1.3.39 -``` +### Running with Watson X -Or install all dependencies: ```bash -pip install -r rag_system/requirements.txt +export LLM_BACKEND=watsonx +python -m rag_system.main api # RAG API on port 8001 ``` -## Usage - -### Running with Watson X - -Once configured, simply set the environment variable and run as normal: +`python -m rag_system.api_server` is equivalent. To index or ask a one-off question from +the CLI: ```bash -export LLM_BACKEND=watsonx -python -m rag_system.main api +python -m rag_system.main index ./shared_uploads +python -m rag_system.main chat "What is in these documents?" ``` -Or in Python: +Or programmatically: ```python import os @@ -93,133 +126,151 @@ os.environ['LLM_BACKEND'] = 'watsonx' from rag_system.factory import get_agent -# Get agent with Watson X backend agent = get_agent(mode="default") - -# Use as normal result = agent.run("What is artificial intelligence?") -print(result) +print(result["answer"]) ``` -### Switching Between Backends - -You can easily switch between Ollama and Watson X: +### Switching between backends ```bash -# Use Ollama (local) +# Local Ollama export LLM_BACKEND=ollama python -m rag_system.main api -# Use Watson X (cloud) +# Watson X export LLM_BACKEND=watsonx python -m rag_system.main api ``` -## Features +### Using Watson X from the web UI + +Document-grounded answers do come from Watson X, with two caveats. -The Watson X client supports all the key features used by LocalGPT: +**The model dropdown still lists Ollama tags.** The UI populates it from the *backend +gateway's* `GET :8000/models` (`src/lib/api.ts`), which always queries Ollama. Only the +RAG API's own `GET :8001/models` is backend-aware and returns the configured Granite ids +when `LLM_BACKEND=watsonx`. -- ✅ Text generation / completion -- ✅ Async generation -- ✅ Streaming responses -- ✅ Embeddings (if using Watson X embedding models) -- ✅ Custom generation parameters (temperature, max_tokens, top_p, top_k) -- ⚠️ Image/multimodal support (limited, depends on model availability) +**That mismatch is harmless.** A per-request `model` is applied only when it is valid for +the active backend — under Watson X the id must contain a `/`, so a stray Ollama tag such +as `qwen3.5:9b` is ignored with a warning and `WATSONX_GENERATION_MODEL` is used instead. +The override is also scoped to that single request and restored afterwards, so one user's +choice cannot leak into the next request. -## API Compatibility +The backend gateway also keeps using Ollama for its non-document fast path and for its +routing decision (see above), so a fully Ollama-free setup means talking to the RAG API on +port 8001 directly. -The `WatsonXClient` provides the same interface as `OllamaClient`: +## API compatibility + +`WatsonXClient` implements the three methods the RAG system uses from `OllamaClient`: ```python from rag_system.utils.watsonx_client import WatsonXClient client = WatsonXClient( api_key="your_api_key", - project_id="your_project_id" + project_id="your_project_id", ) -# Generate completion +# Blocking completion -> {"response": str, "model": str, "done": True} response = client.generate_completion( model="ibm/granite-13b-chat-v2", - prompt="Explain quantum computing" + prompt="Explain quantum computing", ) - print(response['response']) -# Stream completion +# Streaming completion -> yields text chunks for chunk in client.stream_completion( model="ibm/granite-13b-chat-v2", - prompt="Write a story about AI" + prompt="Write a story about AI", ): print(chunk, end='', flush=True) ``` +There is also `generate_completion_async()`, which runs the blocking call in an executor +(the IBM SDK has no native async API). The verifier uses it. + ## Limitations -1. **Embedding Models**: Watson X uses different embedding models than Ollama. Make sure to configure embedding models appropriately in `main.py` if needed. +1. **No JSON mode.** `OllamaClient.generate_completion()` forwards `format="json"` to + Ollama; `WatsonXClient.generate_completion()` accepts the argument and ignores it. + Triage, the overview router, query decomposition and the verifier all rely on the model + returning parseable JSON. When parsing fails the code falls back to safe defaults + (route to RAG, do not decompose, no confidence tag), so answers still work but routing + and verification are less reliable than on Ollama. -2. **Multimodal Support**: Image support varies by model availability in Watson X. Not all Granite models support multimodal inputs. +2. **Embeddings and rerankers are never Watson X.** See the section above — they load from + Hugging Face (or Ollama) in-process. -3. **Streaming**: Streaming support depends on the Watson X SDK version and may fall back to returning the full response at once. +3. **Streaming.** `stream_completion()` uses the SDK's `generate_text_stream()` and falls + back to yielding the full response as a single chunk if that method is unavailable. -4. **Rate Limits**: Watson X has API rate limits that may differ from local Ollama usage. Monitor your usage accordingly. +4. **Errors are swallowed into empty responses.** `generate_completion()` catches + exceptions and returns `{"response": "", "error": ...}`, so an authentication or quota + failure shows up as an empty answer rather than an HTTP error. Check the server log. + +5. **Rate limits and cost.** Watson X is a metered cloud service; local Ollama is not. ## Troubleshooting -### Authentication Errors +### `ImportError: ibm-watsonx-ai package is required` + +Run `pip install "ibm-watsonx-ai>=1.3.39"` in the environment that runs the RAG API. -If you see authentication errors: -- Verify your API key is correct -- Check that your project ID matches an existing Watson X project -- Ensure your IBM Cloud account has Watson X access +### `ValueError: Watson X configuration incomplete` -### Model Not Found +`WATSONX_API_KEY` or `WATSONX_PROJECT_ID` is empty. If you put them in `.env`, make sure the +process is started from the repository root — `load_dotenv()` resolves the file relative to +the working directory. -If you get model not found errors: -- Verify the model ID is correct (e.g., `ibm/granite-13b-chat-v2`) -- Check that the model is available in your Watson X instance -- Some models may require additional permissions +### Authentication errors -### Connection Errors +- Verify the API key is correct +- Check that the project ID matches an existing watsonx.ai project +- Ensure your IBM Cloud account has watsonx.ai access + +### Model not found + +- Verify the model id (e.g. `ibm/granite-13b-chat-v2`) +- Check that the model is available in your instance and region +- Some models require additional entitlements + +### Connection errors -If you experience connection issues: - Check your internet connection -- Verify the Watson X URL is correct for your region -- Check IBM Cloud status page for service outages +- Verify `WATSONX_URL` matches your region +- Check the IBM Cloud status page -## Cost Considerations +### Empty answers with no visible error -Unlike local Ollama, Watson X is a cloud service with usage-based pricing: -- Token-based pricing for generation -- Consider your query volume -- Monitor usage through IBM Cloud dashboard +See limitation 4 — look at the RAG API's stdout for `Error generating completion: …`. ## Reverting to Ollama -To switch back to local Ollama: - ```bash -unset LLM_BACKEND # or set LLM_BACKEND=ollama +unset LLM_BACKEND # or set LLM_BACKEND=ollama python -m rag_system.main api ``` +Indexes built while Watson X was active remain valid: the embeddings were produced locally +and never touched Watson X. + ## Support -For Watson X specific issues: -- [IBM Watson X Documentation](https://www.ibm.com/docs/en/watsonx/saas) -- [Watson X Developer Hub](https://www.ibm.com/watsonx/developer/) +For Watson X issues: +- [IBM watsonx Documentation](https://www.ibm.com/docs/en/watsonx/saas) +- [watsonx Developer Hub](https://www.ibm.com/watsonx/developer/) - [IBM Cloud Support](https://cloud.ibm.com/docs/get-support) -For LocalGPT issues: -- [LocalGPT GitHub Issues](https://github.com/PromtEngineer/localGPT/issues) +For localGPT issues: +- [localGPT GitHub Issues](https://github.com/PromtEngineer/localGPT/issues) ## Contributing -If you find issues with the Watson X integration or want to add features: -1. Create an issue describing the problem/feature -2. Submit a pull request with your changes -3. Ensure all tests pass +See [CONTRIBUTING.md](CONTRIBUTING.md). ## License -This integration follows the same license as LocalGPT (MIT License). +This integration follows the same license as localGPT (MIT License). diff --git a/backend/README.md b/backend/README.md index 0a74e565..baa32bdd 100644 --- a/backend/README.md +++ b/backend/README.md @@ -1,93 +1,185 @@ # localGPT Backend -Simple Python backend that connects your frontend to Ollama for local LLM chat. +The gateway between the Next.js frontend and the rest of the system. It owns chat sessions, +uploaded documents and index bookkeeping in SQLite, answers simple queries directly with +Ollama, and forwards document-grounded queries and indexing jobs to the RAG API. + +``` +frontend :3000 ──► backend/server.py :8000 ──► rag_system/api_server.py :8001 ──► Ollama :11434 + │ + └──► Ollama :11434 (direct, non-RAG answers) +``` + +Built on the standard library's `http.server` (`ThreadingTCPServer`, so requests are +handled concurrently). No web framework. ## Prerequisites -1. **Install Ollama** (if not already installed): - ```bash - # Visit https://ollama.ai or run: - curl -fsSL https://ollama.ai/install.sh | sh - ``` +1. **Python 3.10+** (3.11 recommended). -2. **Start Ollama**: +2. **Ollama** running locally: ```bash + # https://ollama.com/download, or: + curl -fsSL https://ollama.ai/install.sh | sh ollama serve ``` -3. **Pull a model** (optional, server will suggest if needed): +3. **Models** — the defaults the backend resolves to: ```bash - ollama pull llama3.2 + ollama pull qwen3.5:9b # generation + ollama pull qwen3.5:4b # enrichment / utility (used by the RAG API, not by the gateway) ``` +4. **The RAG API** (`python -m rag_system.api_server`) if you want document-grounded + answers or indexing. Without it, `/sessions//messages` still works for non-document + queries, and RAG queries return a "could not connect" message. + ## Setup -1. **Install Python dependencies**: - ```bash - pip install -r requirements.txt - ``` +```bash +# From the repository root +pip install -r backend/requirements.txt # requests, python-dotenv -2. **Test Ollama connection**: - ```bash - python ollama_client.py - ``` +# Smoke-test the Ollama connection (note: this pulls GENERATION_MODEL if it is missing) +python backend/ollama_client.py -3. **Start the backend server**: - ```bash - python server.py - ``` +# Start the server +python backend/server.py +``` -Server will run on `http://localhost:8000` +The server listens on `http://localhost:8000`. Run it from the repository root so the +relative paths it uses (`shared_uploads/`, `index_store/`, `lancedb/`, `backend/chat_data.db`) +resolve the same way they do in the Docker image. -## API Endpoints +Usually you do not start it by hand — `python run_system.py` launches Ollama, the RAG API, +the backend and the frontend together. -### Health Check -```bash -GET /health -``` -Returns server status and available models. +## Configuration -### Chat -```bash -POST /chat -Content-Type: application/json - -{ - "message": "Hello!", - "model": "llama3.2:latest", - "conversation_history": [] -} -``` +All environment variables are optional; the value shown is the code default. -Returns: -```json -{ - "response": "Hello! How can I help you?", - "model": "llama3.2:latest", - "message_count": 1 -} -``` +| Variable | Default | Used for | +| --- | --- | --- | +| `OLLAMA_HOST` | `http://localhost:11434` | Direct Ollama calls (`backend/ollama_client.py`). | +| `RAG_API_URL` | `http://localhost:8001` | Base URL for the RAG API; `/chat` and `/index` are built from it. | +| `RAG_API_TIMEOUT` | `600` | Seconds to wait for a RAG chat response before returning 504. | +| `RAG_API_INDEX_TIMEOUT` | `3600` | Seconds to wait for a RAG indexing run before returning 504. | +| `GENERATION_MODEL` | `qwen3.5:9b` | Default answer model. Resolved as env var → `rag_system.main.OLLAMA_CONFIG` → literal default. | +| `ENRICHMENT_MODEL` | `qwen3.5:4b` | Recorded in index metadata as the enrich/overview model. The gateway no longer calls it (routing is local since Phase 2.3). | +| `EMBEDDING_MODEL` | `microsoft/harrier-oss-v1-0.6b` | Recorded in the default index metadata. | +| `DB_PATH` | `backend/chat_data.db` (`/app/backend/chat_data.db` in Docker) | SQLite file for sessions, messages, indexes and documents. | +| `LANCEDB_PATH` | `storage.lancedb_uri` from the default pipeline profile (`./lancedb`) | Vector store the backend drops tables from when an index is deleted. | + +The backend imports `PIPELINE_CONFIGS` and `OLLAMA_CONFIG` from `rag_system.main` when the +package is importable, and degrades gracefully (printing a warning) when it is not. + +## Request routing + +`POST /sessions//messages` decides per message whether to use RAG. The decision is +made by `should_use_rag(message, idx_ids, force_rag)` — a module-level function in +`server.py` with no LLM call, no file reads and no network I/O: + +1. `force_rag` (or `forceRag`) in the body → RAG, unconditionally. +2. No indexes linked to the session → direct Ollama answer. +3. The whole message is smalltalk (`hello`, `thanks!`, `bye`, `ok` — a whole-message + allowlist regex, max six words) or a question about the assistant itself + (`who are you`, `what model are you`) → direct Ollama answer. +4. Everything else → RAG. + +RAG queries are forwarded to `POST {RAG_API_URL}/chat`; the answer and `source_documents` +come back from there. Direct answers call Ollama with thinking disabled. + +The gate is intentionally biased toward RAG ("escalate, don't pre-decide"). The RAG API's +agent triage runs on every forwarded request and can still answer directly, so the gateway +only has to keep greetings out of the retrieval pipeline — it does not have to be right +about which questions the documents can answer. The pre-retrieval enrichment-model router +and its keyword/length fallback (which misrouted any question containing the word "test") +were removed in Phase 2.3; `Documentation/research/` has the evidence. + +`backend/test_gateway_routing.py` covers the gate: `.venv/bin/python backend/test_gateway_routing.py`. + +**Option casing.** Chat and index options are accepted in both `camelCase` and +`snake_case` and normalised to one canonical `snake_case` key before being forwarded +(`rerankerTopK` → `reranker_top_k`, `enableLatechunk` → `enable_latechunk`, and so on). +Options the caller omits are left out of the payload entirely so the RAG pipeline's own +defaults apply. + +**Chat history.** This server is the only writer of chat message rows. The frontend's +streaming path talks to `:8001/chat/stream` directly and then persists the completed +turn here via `POST /sessions//messages/save`. + +## API Endpoints + +`Documentation/api_reference.md` has the request/response detail; this is the route table. + +### GET + +| Route | Returns | +| --- | --- | +| `/health` | `{status, ollama_running, available_models, database_stats}` | +| `/sessions` | `{sessions, total}` | +| `/sessions/cleanup` | `{message, cleanup_count}` — deletes sessions with no messages | +| `/sessions/` | `{session, messages}` | +| `/sessions//documents` | `{session, files, file_count}` | +| `/sessions//indexes` | `{indexes, total}` | +| `/models` | `{generation_models, embedding_models}` | +| `/indexes` | `{indexes, total}` | +| `/indexes/` | the index record, or 404 | + +### POST + +| Route | Body | Returns | +| --- | --- | --- | +| `/chat` | `{message, model?, conversation_history?}` | `{response, model, message_count}` — legacy sessionless path, always direct Ollama | +| `/sessions` | `{title?, model?}` | `{session, session_id}` (201) | +| `/sessions//messages` | `{message, model?, force_rag?, …chat options}` | `{response, session, source_documents, used_rag}` | +| `/sessions//upload` | `multipart/form-data`, field `files` | `{message, uploaded_files}` | +| `/sessions//index` | optional JSON index options | the RAG API's `/index` response, or `{message: "No documents to index for this session."}` | +| `/sessions//rename` | `{title}` | `{message, session}` | +| `/sessions//indexes/` | — | `{message}` — links an index to a session | +| `/indexes` | `{name, description?, metadata?}` | `{index_id}` (201) | +| `/indexes//upload` | `multipart/form-data`, field `files` | `{message, uploaded_files}` | +| `/indexes//build` | optional JSON index options | `{response, …applied options}` | + +### DELETE + +| Route | Returns | +| --- | --- | +| `/sessions/` | `{deleted: true}`, or 404 | +| `/indexes/` | `{message, index_id}`, or 404 — also drops the index's LanceDB table | + +`OPTIONS` on any path returns the CORS preflight headers (`GET, POST, DELETE, OPTIONS`). ## Testing -Test the chat endpoint: ```bash +curl http://localhost:8000/health + curl -X POST http://localhost:8000/chat \ -H "Content-Type: application/json" \ - -d '{"message": "Hello!", "model": "llama3.2:latest"}' + -d '{"message": "Hello!"}' ``` -## Frontend Integration +For an end-to-end check of the whole stack use `python run_system.py --health` or +`python system_health_check.py`. There is no automated test suite in this directory. + +## Frontend integration + +The frontend reads `NEXT_PUBLIC_API_URL` (default `http://localhost:8000`) for this +server and `NEXT_PUBLIC_RAG_API_URL` (default `http://localhost:8001`) for the streaming +endpoint. Both are inlined at `next build` time, so they must be set before the frontend +is built. -Your React frontend should connect to: -- **Backend**: `http://localhost:8000` -- **Chat endpoint**: `http://localhost:8000/chat` +## Implemented today -## What's Next +- Session-scoped chat with SQLite-persisted history and auto-generated titles. +- Document upload (per session and per index) into `shared_uploads/`. +- Indexing delegated to the RAG API, with the resulting configuration stored as index + metadata. +- Vector-store bookkeeping: index records point at their LanceDB table, and deleting an + index drops that table (the late-chunk `_lc` sibling is left behind). +- Retrieval-augmented answers with source documents, plus a direct-LLM fast path for + general questions. -This simple backend is ready for: -- ✅ **Real-time chat** with local LLMs -- 🔜 **Document upload** for RAG -- 🔜 **Vector database** integration -- 🔜 **Streaming responses** -- 🔜 **Chat history** persistence \ No newline at end of file +Streaming is **not** served by this process — the frontend streams from the RAG API's +`/chat/stream` endpoint directly. diff --git a/backend/database.py b/backend/database.py index a5d38aec..bba59240 100644 --- a/backend/database.py +++ b/backend/database.py @@ -1,20 +1,40 @@ +import os import sqlite3 import uuid import json from datetime import datetime from typing import List, Dict, Optional, Tuple + +def resolve_lancedb_path() -> str: + """LanceDB location, kept in sync with the store the indexing pipeline writes to.""" + env_path = os.getenv('LANCEDB_PATH') + if env_path: + return env_path + try: + from rag_system.main import PIPELINE_CONFIGS + uri = PIPELINE_CONFIGS.get('default', {}).get('storage', {}).get('lancedb_uri') + if uri: + return uri + except Exception: + pass + return './lancedb' + + class ChatDatabase: def __init__(self, db_path: str = None): + if db_path is None: + db_path = os.getenv("DB_PATH") if db_path is None: # Auto-detect environment and set appropriate path - import os if os.path.exists("/app"): # Docker environment - self.db_path = "/app/backend/chat_data.db" + db_path = "/app/backend/chat_data.db" else: # Local development environment - self.db_path = "backend/chat_data.db" - else: - self.db_path = db_path + db_path = "backend/chat_data.db" + self.db_path = db_path + parent_dir = os.path.dirname(os.path.abspath(self.db_path)) + if parent_dir: + os.makedirs(parent_dir, exist_ok=True) self.init_database() def init_database(self): @@ -414,9 +434,7 @@ def delete_index(self, index_id: str) -> bool: if vector_table_name: try: from rag_system.indexing.embedders import LanceDBManager - import os - db_path = os.getenv('LANCEDB_PATH') or './rag_system/index_store/lancedb' - ldb = LanceDBManager(db_path) + ldb = LanceDBManager(resolve_lancedb_path()) db = ldb.db if hasattr(db, 'table_names') and vector_table_name in db.table_names(): db.drop_table(vector_table_name) @@ -464,12 +482,10 @@ def inspect_and_populate_index_metadata(self, index_id: str) -> dict: # Try to import the RAG system modules try: from rag_system.indexing.embedders import LanceDBManager - import os - - # Use the same path as the system - db_path = os.getenv('LANCEDB_PATH') or './rag_system/index_store/lancedb' - ldb = LanceDBManager(db_path) - + + # Use the same store the indexing pipeline writes to + ldb = LanceDBManager(resolve_lancedb_path()) + # Check if table exists if not hasattr(ldb.db, 'table_names') or vector_table_name not in ldb.db.table_names(): # Table doesn't exist - this means the index was never properly built @@ -525,8 +541,14 @@ def inspect_and_populate_index_metadata(self, index_id: str) -> dict: 384: 'BAAI/bge-small-en-v1.5 (or similar)', 512: 'sentence-transformers/all-MiniLM-L6-v2 (or similar)', 768: 'BAAI/bge-base-en-v1.5 (or similar)', - 1024: 'Qwen/Qwen3-Embedding-0.6B (or similar)', - 1536: 'text-embedding-ada-002 (or similar)' + # 1024 is ambiguous by construction: harrier-oss-v1-0.6b + # (the default) and Qwen3-Embedding-0.6B share it. The + # authoritative answer is the table's own embedder marker + # (rag_system/indexing/embedders.py), not this guess. + 1024: 'microsoft/harrier-oss-v1-0.6b or Qwen/Qwen3-Embedding-0.6B (or similar)', + 1536: 'text-embedding-ada-002 (or similar)', + 2560: 'Qwen/Qwen3-Embedding-4B (or similar)', + 4096: 'Qwen/Qwen3-Embedding-8B (or similar)' } if len(vector_data) in dim_to_model: inferred_metadata['embedding_model_inferred'] = dim_to_model[len(vector_data)] @@ -671,7 +693,7 @@ def generate_session_title(first_message: str, max_length: int = 50) -> str: print("🧪 Testing database...") # Create a test session - session_id = db.create_session("Test Chat", "llama3.2:latest") + session_id = db.create_session("Test Chat", os.getenv("GENERATION_MODEL", "qwen3.5:9b")) # Add some messages db.add_message(session_id, "Hello!", "user") diff --git a/backend/ollama_client.py b/backend/ollama_client.py index 33ee103d..996ab16f 100644 --- a/backend/ollama_client.py +++ b/backend/ollama_client.py @@ -1,8 +1,12 @@ import requests import json import os +import re from typing import List, Dict, Optional +DEFAULT_GENERATION_MODEL = os.getenv("GENERATION_MODEL", "qwen3.5:9b") + + class OllamaClient: def __init__(self, base_url: Optional[str] = None): if base_url is None: @@ -54,11 +58,13 @@ def pull_model(self, model_name: str) -> bool: print(f"Error pulling model: {e}") return False - def chat(self, message: str, model: str = "llama3.2", conversation_history: List[Dict] = None, enable_thinking: bool = True) -> str: + def chat(self, message: str, model: str = None, conversation_history: List[Dict] = None, enable_thinking: bool = True) -> str: """Send a chat message to Ollama""" + if model is None: + model = DEFAULT_GENERATION_MODEL if conversation_history is None: conversation_history = [] - + # Add user message to conversation messages = conversation_history + [{"role": "user", "content": message}] @@ -96,7 +102,6 @@ def chat(self, message: str, model: str = "llama3.2", conversation_history: List # Additional cleanup: remove any thinking tokens that might slip through if not enable_thinking: # Remove common thinking token patterns - import re response_text = re.sub(r'.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE) response_text = re.sub(r'.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE) response_text = response_text.strip() @@ -108,64 +113,6 @@ def chat(self, message: str, model: str = "llama3.2", conversation_history: List except requests.exceptions.RequestException as e: return f"Connection error: {e}" - def chat_stream(self, message: str, model: str = "llama3.2", conversation_history: List[Dict] = None, enable_thinking: bool = True): - """Stream chat response from Ollama""" - if conversation_history is None: - conversation_history = [] - - messages = conversation_history + [{"role": "user", "content": message}] - - try: - payload = { - "model": model, - "messages": messages, - "stream": True, - } - - # Multiple approaches to disable thinking tokens - if not enable_thinking: - payload.update({ - "think": False, # Native Ollama parameter - "options": { - "think": False, - "thinking": False, - "temperature": 0.7, - "top_p": 0.9 - } - }) - else: - payload["think"] = True - - response = requests.post( - f"{self.api_url}/chat", - json=payload, - stream=True, - timeout=60 - ) - - if response.status_code == 200: - for line in response.iter_lines(): - if line: - try: - data = json.loads(line) - if "message" in data and "content" in data["message"]: - content = data["message"]["content"] - - # Filter out thinking tokens in streaming mode - if not enable_thinking: - # Skip content that looks like thinking tokens - if '' in content.lower() or '' in content.lower(): - continue - - yield content - except json.JSONDecodeError: - continue - else: - yield f"Error: {response.status_code} - {response.text}" - - except requests.exceptions.RequestException as e: - yield f"Connection error: {e}" - def main(): """Test the Ollama client""" client = OllamaClient() @@ -183,9 +130,9 @@ def main(): models = client.list_models() print(f"Available models: {models}") - # Try to use llama3.2, pull if needed - model_name = "llama3.2" - if model_name not in [m.split(":")[0] for m in models]: + # Try to use the configured generation model, pull if needed + model_name = DEFAULT_GENERATION_MODEL + if model_name not in models: print(f"Model {model_name} not found. Pulling...") if client.pull_model(model_name): print(f"✅ Model {model_name} pulled successfully!") diff --git a/backend/requirements.txt b/backend/requirements.txt index bbd44d99..df7458c2 100644 --- a/backend/requirements.txt +++ b/backend/requirements.txt @@ -1,3 +1,2 @@ requests python-dotenv -PyPDF2 \ No newline at end of file diff --git a/backend/server.py b/backend/server.py index 859040ef..332d1551 100644 --- a/backend/server.py +++ b/backend/server.py @@ -1,10 +1,10 @@ import json import http.server import socketserver -import cgi +import email import os import uuid -from urllib.parse import urlparse, parse_qs +from urllib.parse import urlparse import requests # 🆕 Import requests for making HTTP calls import sys from datetime import datetime @@ -14,24 +14,223 @@ # Import RAG system modules for complete metadata try: - from rag_system.main import PIPELINE_CONFIGS + from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG RAG_SYSTEM_AVAILABLE = True print("✅ RAG system modules accessible from backend") except ImportError as e: PIPELINE_CONFIGS = {} + OLLAMA_CONFIG = {} RAG_SYSTEM_AVAILABLE = False print(f"⚠️ RAG system modules not available: {e}") from ollama_client import OllamaClient from database import db, generate_session_title -import simple_pdf_processor as pdf_module -from simple_pdf_processor import initialize_simple_pdf_processor -from typing import List, Dict, Any +from typing import List, Dict, Any, Optional, Tuple import re -# 🆕 Reusable TCPServer with address reuse enabled -class ReusableTCPServer(socketserver.TCPServer): +PORT = 8000 + +# Base URL of the RAG API service. In Docker this is http://rag-api:8001. +RAG_API_URL = os.getenv("RAG_API_URL", "http://localhost:8001").rstrip("/") +RAG_API_TIMEOUT = float(os.getenv("RAG_API_TIMEOUT", "600")) +RAG_API_INDEX_TIMEOUT = float(os.getenv("RAG_API_INDEX_TIMEOUT", "3600")) + +GENERATION_MODEL = os.getenv("GENERATION_MODEL") or OLLAMA_CONFIG.get("generation_model") or "qwen3.5:9b" +ENRICHMENT_MODEL = os.getenv("ENRICHMENT_MODEL") or OLLAMA_CONFIG.get("enrichment_model") or "qwen3.5:4b" + +# Canonical snake_case option name -> (caster, accepted aliases). +# The frontend historically sent camelCase, so both spellings map to one key. +CHAT_OPTIONS: Dict[str, Tuple[Any, Tuple[str, ...]]] = { + "model": (str, ()), + "compose_sub_answers": (bool, ("composeSubAnswers",)), + "query_decompose": (bool, ("queryDecompose", "decompose")), + "ai_rerank": (bool, ("aiRerank",)), + "context_expand": (bool, ("contextExpand",)), + "verify": (bool, ()), + "retrieval_k": (int, ("retrievalK",)), + "context_window_size": (int, ("contextWindowSize",)), + "reranker_top_k": (int, ("rerankerTopK",)), + "retrieval_mode": (str, ("retrievalMode", "search_type", "searchType")), + "provence_prune": (bool, ("provencePrune",)), + "provence_threshold": (float, ("provenceThreshold",)), +} + +INDEX_OPTIONS: Dict[str, Tuple[Any, Tuple[str, ...]]] = { + "chunk_size": (int, ("chunkSize",)), + "window_size": (int, ("windowSize",)), + "retrieval_mode": (str, ("retrievalMode",)), + "enable_enrich": (bool, ("enableEnrich",)), + "enable_latechunk": (bool, ("enableLatechunk", "latechunk")), + "enable_docling_chunk": (bool, ("enableDoclingChunk", "doclingChunk")), + "embedding_model": (str, ("embeddingModel",)), + "enrich_model": (str, ("enrichModel",)), + "overview_model_name": (str, ("overviewModelName", "overviewModel", "overview_model")), + "batch_size_embed": (int, ("batchSizeEmbed",)), + "batch_size_enrich": (int, ("batchSizeEnrich",)), +} + + +def normalize_options(data: dict, spec: Dict[str, Tuple[Any, Tuple[str, ...]]]) -> Dict[str, Any]: + """Return canonical snake_case options from a body that may use either casing.""" + options: Dict[str, Any] = {} + for canonical, (caster, aliases) in spec.items(): + for key in (canonical,) + tuple(aliases): + if key not in data or data[key] is None: + continue + try: + options[canonical] = caster(data[key]) + except (TypeError, ValueError): + options[canonical] = data[key] + break + return options + + +# --------------------------------------------------------------------------- +# Gateway routing gate +# --------------------------------------------------------------------------- +# Retrieval-first cascade: escalate, don't pre-decide. When a session has +# documents linked, the gateway sends the message to the RAG API unless it is +# unmistakable smalltalk or a question about the assistant itself. There is no +# LLM call here — pre-retrieval LLM routing is measurably the weakest routing +# pattern available (see Documentation/research/), and the agent-side triage in +# rag_system/agent/loop.py remains the single LLM routing layer: it can still +# answer directly, so over-sending to RAG here is cheap and recoverable. +# +# Both regexes below are whole-message allowlists. Anything that is not an +# exact match falls through to RAG. + +SMALLTALK_MAX_WORDS = 6 + +# Phrases that, on their own, carry no retrievable intent. +_SMALLTALK_CORE = ( + # greetings + r"hi+", r"hey+", r"hello+", r"heya", r"yo", r"howdy", r"greetings", + r"good\s+(?:morning|afternoon|evening|day)", + r"how\s+are\s+(?:you|u|ya)(?:\s+doing)?", r"how'?s\s+it\s+going", + r"what'?s\s+up", r"sup", + # thanks + r"thanks", r"thank\s+you", r"thx", r"ty", r"cheers", + r"much\s+appreciated", r"appreciate\s+it", + # farewells + r"bye+", r"goodbye", r"good\s*night", r"see\s+(?:you|ya)", + r"talk\s+to\s+you\s+later", r"take\s+care", r"later", + # acknowledgements / fillers that end a turn + r"ok(?:ay)?", r"kk", r"alright", r"got\s+it", r"understood", r"i\s+see", + r"sounds\s+good", r"no\s+problem", r"np", r"never\s*mind", r"nvm", + r"cool", r"nice", r"awesome", r"perfect", r"great", + r"yes", r"yeah", r"yep", r"nope", r"no", + r"sorry", r"please", r"lol", r"haha+", +) + +# Words allowed *alongside* a core phrase but never sufficient on their own. +_SMALLTALK_FILLER = ( + r"there", r"again", r"all", r"everyone", r"folks", r"guys", r"team", + r"friend", r"buddy", r"mate", r"man", r"dude", r"bot", r"assistant", + r"a\s+lot", r"so\s+much", r"very\s+much", r"so", r"much", r"very", + r"then", r"too", r"and", r"well", r"my", r"you", +) + + +def _alternation(*groups: Tuple[str, ...]) -> str: + """Join regex phrases longest-first so the widest match is tried first.""" + phrases = [p for group in groups for p in group] + phrases.sort(key=len, reverse=True) + return "|".join(phrases) + + +# Whole message consists only of allowlisted smalltalk/filler phrases. +_SMALLTALK_RE = re.compile( + r"^\W*(?:{alt})(?:\W+(?:{alt}))*\W*$".format( + alt=_alternation(_SMALLTALK_CORE, _SMALLTALK_FILLER) + ), + re.IGNORECASE, +) + +# ...and at least one of those phrases is a *core* smalltalk phrase. +_SMALLTALK_CORE_RE = re.compile( + r"(? bool: + """True when a message is pure smalltalk or a question about the assistant. + + Deterministic and allocation-cheap: two anchored regexes and a word count. + Deliberately conservative — an unmatched message routes to RAG. + """ + text = (message or "").strip() + if not text: + return True + if _ASSISTANT_META_RE.match(text): + return True + if len(text.split()) > SMALLTALK_MAX_WORDS: + return False + return bool(_SMALLTALK_RE.match(text) and _SMALLTALK_CORE_RE.search(text)) + + +def should_use_rag(message: str, idx_ids: Optional[List[str]], force_rag: bool = False) -> bool: + """Decide whether one chat message goes to the RAG API or straight to Ollama. + + 1. ``force_rag`` → RAG, unconditionally. + 2. No indexes linked to the session → direct LLM (nothing to retrieve from). + 3. Smalltalk / assistant-meta → direct LLM. + 4. Everything else → RAG. + """ + if force_rag: + return True + if not idx_ids: + return False + return not is_smalltalk_or_meta(message) + + +def default_index_metadata() -> Dict[str, Any]: + """Index metadata defaults taken from the live RAG pipeline configuration.""" + config = PIPELINE_CONFIGS.get('default', {}) if RAG_SYSTEM_AVAILABLE else {} + retrieval = config.get('retrieval', {}) + indexing = config.get('indexing', {}) + return { + 'chunk_size': 512, + 'retrieval_mode': retrieval.get('search_type', 'hybrid'), + 'window_size': config.get('contextual_enricher', {}).get('window_size', 1), + 'embedding_model': os.getenv('EMBEDDING_MODEL') or config.get('embedding_model_name') or 'microsoft/harrier-oss-v1-0.6b', + 'enrich_model': ENRICHMENT_MODEL, + 'overview_model': ENRICHMENT_MODEL, + 'enable_enrich': config.get('contextual_enricher', {}).get('enabled', True), + 'latechunk': retrieval.get('latechunk', {}).get('enabled', False), + 'docling_chunk': True, + 'batch_size_embed': indexing.get('embedding_batch_size', 50), + 'batch_size_enrich': indexing.get('enrichment_batch_size', 25), + } + + +# 🆕 Threaded TCPServer with address reuse enabled. Threading keeps a slow RAG +# query from blocking every other request, including the RAG API's callback. +class ReusableTCPServer(socketserver.ThreadingTCPServer): allow_reuse_address = True + daemon_threads = True class ChatHandler(http.server.BaseHTTPRequestHandler): def __init__(self, *args, **kwargs): @@ -102,6 +301,9 @@ def do_POST(self): session_id = parts[2] index_id = parts[4] self.handle_link_index_to_session(session_id, index_id) + elif parsed_path.path.startswith('/sessions/') and parsed_path.path.endswith('/messages/save'): + session_id = parsed_path.path.split('/')[-3] + self.handle_save_messages(session_id) elif parsed_path.path.startswith('/sessions/') and parsed_path.path.endswith('/messages'): session_id = parsed_path.path.split('/')[-2] self.handle_session_chat(session_id) @@ -140,7 +342,7 @@ def handle_chat(self): data = json.loads(post_data.decode('utf-8')) message = data.get('message', '') - model = data.get('model', 'llama3.2:latest') + model = data.get('model', GENERATION_MODEL) conversation_history = data.get('conversation_history', []) if not message: @@ -250,8 +452,8 @@ def handle_create_session(self): data = json.loads(post_data.decode('utf-8')) title = data.get('title', 'New Chat') - model = data.get('model', 'llama3.2:latest') - + model = data.get('model', GENERATION_MODEL) + session_id = db.create_session(title, model) session = db.get_session(session_id) @@ -295,23 +497,33 @@ def handle_session_chat(self, session_id: str): # Add user message to database first user_message_id = db.add_message(session_id, message, "user") - - # 🎯 SMART ROUTING: Decide between direct LLM vs RAG + + options = normalize_options(data, CHAT_OPTIONS) + + # 🎯 ROUTING: deterministic gate, no LLM call (see should_use_rag) idx_ids = db.get_indexes_for_session(session_id) - force_rag = bool(data.get("force_rag", False)) - use_rag = True if force_rag else self._should_use_rag(message, idx_ids) - + force_rag = bool(data.get("force_rag", data.get("forceRag", False))) + if force_rag: + options["force_rag"] = True + use_rag = should_use_rag(message, idx_ids, force_rag=force_rag) + if use_rag: # 🔍 --- Use RAG Pipeline for Document-Related Queries --- print(f"🔍 Using RAG pipeline for document query: '{message[:50]}...'") - response_text, source_docs = self._handle_rag_query(session_id, message, data, idx_ids) + response_text, source_docs = self._handle_rag_query(session_id, message, options, idx_ids) else: # ⚡ --- Use Direct LLM for General Queries (FAST) --- print(f"⚡ Using direct LLM for general query: '{message[:50]}...'") - response_text, source_docs = self._handle_direct_llm_query(session_id, message, session) - - # Add AI response to database - ai_message_id = db.add_message(session_id, response_text, "assistant") + response_text, source_docs = self._handle_direct_llm_query( + session_id, message, session, options.get('model') + ) + + # Add AI response to database (sources go into metadata so reloaded + # sessions can still render attribution) + ai_message_id = db.add_message( + session_id, response_text, "assistant", + metadata={'source_documents': source_docs} if source_docs else None + ) updated_session = db.get_session(session_id) @@ -325,7 +537,8 @@ def handle_session_chat(self, session_id: str): except BrokenPipeError: # Client disconnected - this is normal for long queries, just log it - print(f"⚠️ Client disconnected during RAG processing for query: '{message[:30]}...'") + preview = message[:30] if 'message' in locals() else '' + print(f"⚠️ Client disconnected during RAG processing for query: '{preview}...'") except json.JSONDecodeError: self.send_json_response({ "error": "Invalid JSON" @@ -339,232 +552,20 @@ def handle_session_chat(self, session_id: str): except BrokenPipeError: print(f"⚠️ Client disconnected during error response") - def _should_use_rag(self, message: str, idx_ids: List[str]) -> bool: - """ - 🧠 ENHANCED: Determine if a query should use RAG pipeline using document overviews. - - Args: - message: The user's query - idx_ids: List of index IDs associated with the session - - Returns: - bool: True if should use RAG, False for direct LLM - """ - # No indexes = definitely no RAG needed - if not idx_ids: - return False - - # Load document overviews for intelligent routing - try: - doc_overviews = self._load_document_overviews(idx_ids) - if doc_overviews: - return self._route_using_overviews(message, doc_overviews) - except Exception as e: - print(f"⚠️ Overview-based routing failed, falling back to simple routing: {e}") - - # Fallback to simple pattern matching if overviews unavailable - return self._simple_pattern_routing(message, idx_ids) - - def _load_document_overviews(self, idx_ids: List[str]) -> List[str]: - """Load and aggregate overviews for the given index IDs. - - Strategy: - 1. Attempt to load each index's dedicated overview file. - 2. Aggregate all overviews found across available files (deduplicated). - 3. If none of the index files exist, fall back to the legacy global overview file. - """ - import os, json - - aggregated: list[str] = [] - - # 1️⃣ Collect overviews from per-index files - for idx in idx_ids: - candidate_paths = [ - f"../index_store/overviews/{idx}.jsonl", - f"index_store/overviews/{idx}.jsonl", - f"./index_store/overviews/{idx}.jsonl", - ] - for p in candidate_paths: - if os.path.exists(p): - print(f"📖 Loading overviews from: {p}") - try: - with open(p, "r", encoding="utf-8") as f: - for line in f: - if not line.strip(): - continue - try: - record = json.loads(line) - overview = record.get("overview", "").strip() - if overview: - aggregated.append(overview) - except json.JSONDecodeError: - continue # skip malformed lines - break # Stop after the first existing path for this idx - except Exception as e: - print(f"⚠️ Error reading {p}: {e}") - break # Don't keep trying other paths for this idx if read failed - - # 2️⃣ Fall back to legacy global file if no per-index overviews found - if not aggregated: - legacy_paths = [ - "../index_store/overviews/overviews.jsonl", - "index_store/overviews/overviews.jsonl", - "./index_store/overviews/overviews.jsonl", - ] - for p in legacy_paths: - if os.path.exists(p): - print(f"⚠️ Falling back to legacy overviews file: {p}") - try: - with open(p, "r", encoding="utf-8") as f: - for line in f: - if not line.strip(): - continue - try: - record = json.loads(line) - overview = record.get("overview", "").strip() - if overview: - aggregated.append(overview) - except json.JSONDecodeError: - continue - except Exception as e: - print(f"⚠️ Error reading legacy overviews file {p}: {e}") - break - - # Limit for performance - if aggregated: - print(f"✅ Loaded {len(aggregated)} document overviews from {len(idx_ids)} index(es)") - else: - print(f"⚠️ No overviews found for indices {idx_ids}") - return aggregated[:40] - - def _route_using_overviews(self, query: str, overviews: List[str]) -> bool: - """ - 🎯 Use document overviews and LLM to make intelligent routing decisions. - - Returns True if RAG should be used, False for direct LLM. - """ - if not overviews: - return False - - # Format overviews for the routing prompt - overviews_block = "\n".join(f"[{i+1}] {ov}" for i, ov in enumerate(overviews)) - - router_prompt = f"""You are an AI router deciding whether a user question should be answered via: -• "USE_RAG" – search the user's private documents (described below) -• "DIRECT_LLM" – reply from general knowledge (greetings, public facts, unrelated topics) - -CRITICAL PRINCIPLE: When documents exist in the KB, strongly prefer USE_RAG unless the query is purely conversational or completely unrelated to any possible document content. - -RULES: -1. If ANY overview clearly relates to the question (entities, numbers, addresses, dates, amounts, companies, technical terms) → USE_RAG -2. For document operations (summarize, analyze, explain, extract, find) → USE_RAG -3. For greetings only ("Hi", "Hello", "Thanks") → DIRECT_LLM -4. For pure math/world knowledge clearly unrelated to documents → DIRECT_LLM -5. When in doubt → USE_RAG - -DOCUMENT OVERVIEWS: -{overviews_block} - -DECISION EXAMPLES: -• "What invoice amounts are mentioned?" → USE_RAG (document-specific) -• "Who is PromptX AI LLC?" → USE_RAG (entity in documents) -• "What is the DeepSeek model?" → USE_RAG (mentioned in documents) -• "Summarize the research paper" → USE_RAG (document operation) -• "What is 2+2?" → DIRECT_LLM (pure math) -• "Hi there" → DIRECT_LLM (greeting only) - -USER QUERY: "{query}" - -Respond with exactly one word: USE_RAG or DIRECT_LLM""" - - try: - # Use Ollama to make the routing decision - response = self.ollama_client.chat( - message=router_prompt, - model="qwen3:0.6b", # Fast model for routing - enable_thinking=False # Fast routing - ) - - # The response is directly the text, not a dict - decision = response.strip().upper() - - # Parse decision - if "USE_RAG" in decision: - print(f"🎯 Overview-based routing: USE_RAG for query: '{query[:50]}...'") - return True - elif "DIRECT_LLM" in decision: - print(f"⚡ Overview-based routing: DIRECT_LLM for query: '{query[:50]}...'") - return False - else: - print(f"⚠️ Unclear routing decision '{decision}', defaulting to RAG") - return True # Default to RAG when uncertain - - except Exception as e: - print(f"❌ LLM routing failed: {e}, falling back to pattern matching") - return self._simple_pattern_routing(query, []) - - def _simple_pattern_routing(self, message: str, idx_ids: List[str]) -> bool: - """ - 📝 FALLBACK: Simple pattern-based routing (original logic). - """ - message_lower = message.lower() - - # Always use Direct LLM for greetings and casual conversation - greeting_patterns = [ - 'hello', 'hi', 'hey', 'greetings', 'good morning', 'good afternoon', 'good evening', - 'how are you', 'how do you do', 'nice to meet', 'pleasure to meet', - 'thanks', 'thank you', 'bye', 'goodbye', 'see you', 'talk to you later', - 'test', 'testing', 'check', 'ping', 'just saying', 'nevermind', - 'ok', 'okay', 'alright', 'got it', 'understood', 'i see' - ] - - # Check for greeting patterns - for pattern in greeting_patterns: - if pattern in message_lower: - return False # Use Direct LLM for greetings - - # Keywords that strongly suggest document-related queries - rag_indicators = [ - 'document', 'doc', 'file', 'pdf', 'text', 'content', 'page', - 'according to', 'based on', 'mentioned', 'states', 'says', - 'what does', 'summarize', 'summary', 'analyze', 'analysis', - 'quote', 'citation', 'reference', 'source', 'evidence', - 'explain from', 'extract', 'find in', 'search for' - ] - - # Check for strong RAG indicators - for indicator in rag_indicators: - if indicator in message_lower: - return True - - # Question words + substantial length might benefit from RAG - question_words = ['what', 'how', 'when', 'where', 'why', 'who', 'which'] - starts_with_question = any(message_lower.startswith(word) for word in question_words) - - if starts_with_question and len(message) > 40: - return True - - # Very short messages - use direct LLM - if len(message.strip()) < 20: - return False - - # Default to Direct LLM unless there's clear indication of document query - return False - - def _handle_direct_llm_query(self, session_id: str, message: str, session: dict): + def _handle_direct_llm_query(self, session_id: str, message: str, session: dict, model: Optional[str] = None): """ Handle query using direct Ollama client with thinking disabled for speed. - + Returns: tuple: (response_text, empty_source_docs) """ try: # Get conversation history for context conversation_history = db.get_conversation_history(session_id) - - # Use the session's model or default - model = session.get('model', 'qwen3:8b') # Default to fast model - + + # Per-request override wins, then the session's model, then the configured default + model = model or session.get('model') or GENERATION_MODEL + # Direct Ollama call with thinking disabled for speed response_text = self.ollama_client.chat( message=message, @@ -579,9 +580,9 @@ def _handle_direct_llm_query(self, session_id: str, message: str, session: dict) print(f"❌ Direct LLM error: {e}") return f"Error processing query: {str(e)}", [] - def _handle_rag_query(self, session_id: str, message: str, data: dict, idx_ids: List[str]): + def _handle_rag_query(self, session_id: str, message: str, options: Dict[str, Any], idx_ids: List[str]): """ - Handle query using the full RAG pipeline (delegates to the advanced RAG API running on port 8001). + Handle query using the full RAG pipeline (delegates to the RAG API at RAG_API_URL). Returns: tuple[str, List[dict]]: (response_text, source_documents) @@ -591,7 +592,7 @@ def _handle_rag_query(self, session_id: str, message: str, data: dict, idx_ids: source_docs: List[dict] = [] # Build payload for RAG API - rag_api_url = "http://localhost:8001/chat" + rag_api_url = f"{RAG_API_URL}/chat" table_name = f"text_pages_{idx_ids[-1]}" if idx_ids else None payload: Dict[str, Any] = { "query": message, @@ -600,31 +601,10 @@ def _handle_rag_query(self, session_id: str, message: str, data: dict, idx_ids: if table_name: payload["table_name"] = table_name - # Copy optional parameters from the incoming request - optional_params: Dict[str, tuple[type, str]] = { - "compose_sub_answers": (bool, "compose_sub_answers"), - "query_decompose": (bool, "query_decompose"), - "ai_rerank": (bool, "ai_rerank"), - "context_expand": (bool, "context_expand"), - "verify": (bool, "verify"), - "retrieval_k": (int, "retrieval_k"), - "context_window_size": (int, "context_window_size"), - "reranker_top_k": (int, "reranker_top_k"), - "search_type": (str, "search_type"), - "dense_weight": (float, "dense_weight"), - "provence_prune": (bool, "provence_prune"), - "provence_threshold": (float, "provence_threshold"), - } - for key, (caster, payload_key) in optional_params.items(): - val = data.get(key) - if val is not None: - try: - payload[payload_key] = caster(val) # type: ignore[arg-type] - except Exception: - payload[payload_key] = val + payload.update(options) try: - rag_response = requests.post(rag_api_url, json=payload) + rag_response = requests.post(rag_api_url, json=payload, timeout=RAG_API_TIMEOUT) if rag_response.status_code == 200: rag_data = rag_response.json() response_text = rag_data.get("answer", "No answer found.") @@ -632,15 +612,18 @@ def _handle_rag_query(self, session_id: str, message: str, data: dict, idx_ids: else: response_text = f"Error from RAG API ({rag_response.status_code}): {rag_response.text}" print(f"❌ RAG API error: {response_text}") + except requests.exceptions.Timeout: + response_text = f"The RAG API did not respond within {RAG_API_TIMEOUT:.0f}s." + print(f"❌ RAG API request timed out after {RAG_API_TIMEOUT:.0f}s ({rag_api_url}).") except requests.exceptions.ConnectionError: - response_text = "Could not connect to the RAG API server. Please ensure it is running." - print("❌ Connection to RAG API failed (port 8001).") + response_text = f"Could not connect to the RAG API server at {RAG_API_URL}. Please ensure it is running." + print(f"❌ Connection to RAG API failed ({rag_api_url}).") except Exception as e: response_text = f"Error processing RAG query: {str(e)}" print(f"❌ RAG processing error: {e}") # Strip any / tags that might slip through - response_text = re.sub(r'<(think|thinking)>.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE).strip() + response_text = re.sub(r'<(think|thinking)>.*?', '', response_text, flags=re.DOTALL | re.IGNORECASE).strip() return response_text, source_docs @@ -655,36 +638,60 @@ def handle_delete_session(self, session_id: str): except Exception as e: self.send_json_response({'error': str(e)}, status_code=500) + def parse_multipart_files(self, field_name: str = 'files') -> List[Tuple[str, bytes]]: + """Parse a multipart/form-data body and return (filename, content) for one field. + + Uses the stdlib email parser; `cgi` was removed from Python 3.13. + """ + content_type = self.headers.get('Content-Type', '') or '' + if not content_type.lower().startswith('multipart/form-data'): + return [] + + length = int(self.headers.get('Content-Length', 0) or 0) + if length <= 0: + return [] + + body = self.rfile.read(length) + prologue = f"Content-Type: {content_type}\r\nMIME-Version: 1.0\r\n\r\n".encode('utf-8') + message = email.message_from_bytes(prologue + body) + if not message.is_multipart(): + return [] + + files: List[Tuple[str, bytes]] = [] + for part in message.walk(): + if part.is_multipart(): + continue + filename = part.get_filename() + if not filename: + continue + if field_name and part.get_param('name', header='content-disposition') != field_name: + continue + payload = part.get_payload(decode=True) + if payload is None: + continue + files.append((os.path.basename(filename), payload)) + return files + def handle_file_upload(self, session_id: str): """Handle file uploads, save them, and associate with the session.""" - form = cgi.FieldStorage( - fp=self.rfile, - headers=self.headers, - environ={'REQUEST_METHOD': 'POST', 'CONTENT_TYPE': self.headers['Content-Type']} - ) - uploaded_files = [] - if 'files' in form: - files = form['files'] - if not isinstance(files, list): - files = [files] - + incoming = self.parse_multipart_files('files') + if incoming: upload_dir = "shared_uploads" os.makedirs(upload_dir, exist_ok=True) - for file_item in files: - if file_item.filename: - # Create a unique filename to avoid overwrites - unique_filename = f"{uuid.uuid4()}_{file_item.filename}" - file_path = os.path.join(upload_dir, unique_filename) - - with open(file_path, 'wb') as f: - f.write(file_item.file.read()) - - # Store the absolute path for the indexing service - absolute_file_path = os.path.abspath(file_path) - db.add_document_to_session(session_id, absolute_file_path) - uploaded_files.append({"filename": file_item.filename, "stored_path": absolute_file_path}) + for filename, content in incoming: + # Create a unique filename to avoid overwrites + unique_filename = f"{uuid.uuid4()}_{filename}" + file_path = os.path.join(upload_dir, unique_filename) + + with open(file_path, 'wb') as f: + f.write(content) + + # Store the absolute path for the indexing service + absolute_file_path = os.path.abspath(file_path) + db.add_document_to_session(session_id, absolute_file_path) + uploaded_files.append({"filename": filename, "stored_path": absolute_file_path}) if not uploaded_files: self.send_json_response({"error": "No files were uploaded"}, status_code=400) @@ -695,27 +702,44 @@ def handle_file_upload(self, session_id: str): "uploaded_files": uploaded_files }) + def read_json_body(self) -> dict: + """Read and decode an optional JSON request body. Returns {} when absent.""" + length = int(self.headers.get('Content-Length', 0) or 0) + if length <= 0: + return {} + body = self.rfile.read(length) + try: + parsed = json.loads(body.decode('utf-8')) + except (ValueError, UnicodeDecodeError): + return {} + return parsed if isinstance(parsed, dict) else {} + def handle_index_documents(self, session_id: str): """Triggers indexing for all documents in a session.""" print(f"🔥 Received request to index documents for session {session_id[:8]}...") try: + options = normalize_options(self.read_json_body(), INDEX_OPTIONS) + file_paths = db.get_documents_for_session(session_id) if not file_paths: self.send_json_response({"message": "No documents to index for this session."}, status_code=200) return print(f"Found {len(file_paths)} documents to index. Sending to RAG API...") - - rag_api_url = "http://localhost:8001/index" - rag_response = requests.post(rag_api_url, json={"file_paths": file_paths, "session_id": session_id}) + + rag_api_url = f"{RAG_API_URL}/index" + payload: Dict[str, Any] = {"file_paths": file_paths, "session_id": session_id} + payload.update(options) + rag_response = requests.post(rag_api_url, json=payload, timeout=RAG_API_INDEX_TIMEOUT) if rag_response.status_code == 200: print("✅ RAG API successfully indexed documents.") # Merge key config values into index metadata - idx_meta = { + idx_meta: Dict[str, Any] = { "session_linked": True, - "retrieval_mode": "hybrid", + "retrieval_mode": options.get("retrieval_mode", "hybrid"), } + idx_meta.update({k: v for k, v in options.items() if k != "retrieval_mode"}) try: db.update_index_metadata(session_id, idx_meta) # session_id used as index_id in text table naming except Exception as e: @@ -726,22 +750,19 @@ def handle_index_documents(self, session_id: str): print(f"❌ RAG API indexing failed ({rag_response.status_code}): {error_info}") self.send_json_response({"error": f"Indexing failed: {error_info}"}, status_code=500) + except requests.exceptions.Timeout: + print(f"❌ RAG API indexing timed out after {RAG_API_INDEX_TIMEOUT:.0f}s.") + self.send_json_response({ + "error": f"Indexing did not complete within {RAG_API_INDEX_TIMEOUT:.0f}s." + }, status_code=504) + except requests.exceptions.ConnectionError: + print(f"❌ Connection to RAG API failed ({RAG_API_URL}).") + self.send_json_response({ + "error": f"Could not connect to the RAG API server at {RAG_API_URL}." + }, status_code=502) except Exception as e: print(f"❌ Exception during indexing: {str(e)}") self.send_json_response({"error": f"An unexpected error occurred: {str(e)}"}, status_code=500) - - def handle_pdf_upload(self, session_id: str): - """ - Processes PDF files: extracts text and stores it in the database. - DEPRECATED: This is the old method. Use handle_file_upload instead. - """ - # This function is now deprecated in favor of the new indexing workflow - # but is kept for potential legacy/compatibility reasons. - # For new functionality, it should not be used. - self.send_json_response({ - "warning": "This upload method is deprecated. Use the new file upload and indexing flow.", - "message": "No action taken." - }, status_code=410) # 410 Gone def handle_get_models(self): """Get available models from both Ollama and HuggingFace, grouped by capability""" @@ -762,8 +783,9 @@ def handle_get_models(self): # Add supported HuggingFace embedding models huggingface_embedding_models = [ + "microsoft/harrier-oss-v1-0.6b", # shipped default "Qwen/Qwen3-Embedding-0.6B", - "Qwen/Qwen3-Embedding-4B", + "Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-8B" ] embedding_models.extend(huggingface_embedding_models) @@ -813,27 +835,18 @@ def handle_create_index(self): # Add complete metadata from RAG system configuration if available if RAG_SYSTEM_AVAILABLE and PIPELINE_CONFIGS.get('default'): - default_config = PIPELINE_CONFIGS['default'] complete_metadata = { 'status': 'created', 'metadata_source': 'rag_system_config', - 'created_at': json.loads(json.dumps(datetime.now().isoformat())), - 'chunk_size': 512, # From default config - 'chunk_overlap': 64, # From default config - 'retrieval_mode': 'hybrid', # From default config - 'window_size': 5, # From default config - 'embedding_model': 'Qwen/Qwen3-Embedding-0.6B', # From default config - 'enrich_model': 'qwen3:0.6b', # From default config - 'overview_model': 'qwen3:0.6b', # From default config - 'enable_enrich': True, # From default config - 'latechunk': True, # From default config - 'docling_chunk': True, # From default config - 'note': 'Default configuration from RAG system' + 'created_at': datetime.now().isoformat(), + 'note': 'Default configuration from RAG system', } + complete_metadata.update(default_index_metadata()) # Merge with any provided metadata complete_metadata.update(metadata) metadata = complete_metadata - + + idx_id = db.create_index(name, description, metadata) self.send_json_response({'index_id': idx_id}, status_code=201) except Exception as e: @@ -841,27 +854,28 @@ def handle_create_index(self): def handle_index_file_upload(self, index_id: str): """Reuse file upload logic but store docs under index.""" - form = cgi.FieldStorage(fp=self.rfile, headers=self.headers, environ={'REQUEST_METHOD':'POST', 'CONTENT_TYPE': self.headers['Content-Type']}) uploaded_files=[] - if 'files' in form: - files=form['files'] - if not isinstance(files, list): - files=[files] + incoming = self.parse_multipart_files('files') + if incoming: upload_dir='shared_uploads' os.makedirs(upload_dir, exist_ok=True) - for f in files: - if f.filename: - unique=f"{uuid.uuid4()}_{f.filename}" - path=os.path.join(upload_dir, unique) - with open(path,'wb') as out: out.write(f.file.read()) - db.add_document_to_index(index_id, f.filename, os.path.abspath(path)) - uploaded_files.append({'filename':f.filename,'stored_path':os.path.abspath(path)}) + for filename, content in incoming: + unique=f"{uuid.uuid4()}_{filename}" + path=os.path.join(upload_dir, unique) + with open(path,'wb') as out: out.write(content) + db.add_document_to_index(index_id, filename, os.path.abspath(path)) + uploaded_files.append({'filename':filename,'stored_path':os.path.abspath(path)}) if not uploaded_files: self.send_json_response({'error':'No files uploaded'}, status_code=400); return self.send_json_response({'message':f"Uploaded {len(uploaded_files)} files","uploaded_files":uploaded_files}) def handle_build_index(self, index_id: str): try: + # Parse request body for optional flags and configuration. Options the + # caller omits are left out of the payload so the RAG pipeline defaults apply. + # Read before any early return so the request body is always consumed. + options = normalize_options(self.read_json_body(), INDEX_OPTIONS) + index=db.get_index(index_id) if not index: self.send_json_response({'error':'Index not found'}, status_code=404); return @@ -869,95 +883,27 @@ def handle_build_index(self, index_id: str): if not file_paths: self.send_json_response({'error':'No documents to index'}, status_code=400); return - # Parse request body for optional flags and configuration - latechunk = False - docling_chunk = False - chunk_size = 512 - chunk_overlap = 64 - retrieval_mode = 'hybrid' - window_size = 2 - enable_enrich = True - embedding_model = None - enrich_model = None - batch_size_embed = 50 - batch_size_enrich = 25 - overview_model = None - - if 'Content-Length' in self.headers and int(self.headers['Content-Length']) > 0: - try: - length = int(self.headers['Content-Length']) - body = self.rfile.read(length) - opts = json.loads(body.decode('utf-8')) - latechunk = bool(opts.get('latechunk', False)) - docling_chunk = bool(opts.get('doclingChunk', False)) - chunk_size = int(opts.get('chunkSize', 512)) - chunk_overlap = int(opts.get('chunkOverlap', 64)) - retrieval_mode = str(opts.get('retrievalMode', 'hybrid')) - window_size = int(opts.get('windowSize', 2)) - enable_enrich = bool(opts.get('enableEnrich', True)) - embedding_model = opts.get('embeddingModel') - enrich_model = opts.get('enrichModel') - batch_size_embed = int(opts.get('batchSizeEmbed', 50)) - batch_size_enrich = int(opts.get('batchSizeEnrich', 25)) - overview_model = opts.get('overviewModel') - except Exception: - # Keep defaults on parse error - pass - - # Set per-index overview file path - overview_path = f"index_store/overviews/{index_id}.jsonl" - - # Ensure config_override includes overview_path - def ensure_overview_path(cfg: dict): - cfg["overview_path"] = overview_path - - # we'll inject later when we build config_override - - # Delegate to advanced RAG API same as session indexing - rag_api_url = "http://localhost:8001/index" - import requests, json as _json + # Delegate to the RAG API, same as session indexing + rag_api_url = f"{RAG_API_URL}/index" # Use the index's dedicated LanceDB table so retrieval matches table_name = index.get("vector_table_name") - payload = { + payload: Dict[str, Any] = { "file_paths": file_paths, "session_id": index_id, # reuse index_id for progress tracking "table_name": table_name, - "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, - "retrieval_mode": retrieval_mode, - "window_size": window_size, - "enable_enrich": enable_enrich, - "batch_size_embed": batch_size_embed, - "batch_size_enrich": batch_size_enrich } - if latechunk: - payload["enable_latechunk"] = True - if docling_chunk: - payload["enable_docling_chunk"] = True - if embedding_model: - payload["embedding_model"] = embedding_model - if enrich_model: - payload["enrich_model"] = enrich_model - if overview_model: - payload["overview_model_name"] = overview_model - - rag_resp = requests.post(rag_api_url, json=payload) + payload.update(options) + + rag_resp = requests.post(rag_api_url, json=payload, timeout=RAG_API_INDEX_TIMEOUT) if rag_resp.status_code==200: - meta_updates = { - "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, - "retrieval_mode": retrieval_mode, - "window_size": window_size, - "enable_enrich": enable_enrich, - "latechunk": latechunk, - "docling_chunk": docling_chunk, - } - if embedding_model: - meta_updates["embedding_model"] = embedding_model - if enrich_model: - meta_updates["enrich_model"] = enrich_model - if overview_model: - meta_updates["overview_model"] = overview_model + meta_updates: Dict[str, Any] = dict(options) + meta_updates["status"] = "built" + if "enable_latechunk" in meta_updates: + meta_updates["latechunk"] = meta_updates.pop("enable_latechunk") + if "enable_docling_chunk" in meta_updates: + meta_updates["docling_chunk"] = meta_updates.pop("enable_docling_chunk") + if "overview_model_name" in meta_updates: + meta_updates["overview_model"] = meta_updates.pop("overview_model_name") try: db.update_index_metadata(index_id, meta_updates) except Exception as e: @@ -982,9 +928,17 @@ def ensure_overview_path(cfg: dict): }) else: self.send_json_response({"error":f"RAG indexing failed: {rag_resp.text}"}, status_code=500) + except requests.exceptions.Timeout: + self.send_json_response({ + "error": f"Indexing did not complete within {RAG_API_INDEX_TIMEOUT:.0f}s." + }, status_code=504) + except requests.exceptions.ConnectionError: + self.send_json_response({ + "error": f"Could not connect to the RAG API server at {RAG_API_URL}." + }, status_code=502) except Exception as e: self.send_json_response({'error':str(e)}, status_code=500) - + def handle_link_index_to_session(self, session_id: str, index_id: str): try: db.link_index_to_session(session_id, index_id) @@ -1022,6 +976,50 @@ def handle_delete_index(self, index_id: str): except Exception as e: self.send_json_response({'error': str(e)}, status_code=500) + def handle_save_messages(self, session_id: str): + """Persist a completed streamed turn (the browser streams straight from the + RAG API, so the gateway never sees those messages otherwise).""" + try: + session = db.get_session(session_id) + if not session: + self.send_json_response({"error": "Session not found"}, status_code=404) + return + + content_length = int(self.headers.get('Content-Length', 0)) + if content_length == 0: + self.send_json_response({"error": "Request body required"}, status_code=400) + return + + post_data = self.rfile.read(content_length) + data = json.loads(post_data.decode('utf-8')) + user_message = (data.get('user_message') or '').strip() + assistant_message = (data.get('assistant_message') or '').strip() + source_documents = data.get('source_documents') or [] + steps = data.get('steps') + + if not user_message or not assistant_message: + self.send_json_response({"error": "user_message and assistant_message are required"}, status_code=400) + return + + if session['message_count'] == 0: + db.update_session_title(session_id, generate_session_title(user_message)) + + ai_metadata: Dict[str, Any] = {} + if source_documents: + ai_metadata['source_documents'] = source_documents + if isinstance(steps, list) and steps: + ai_metadata['steps'] = steps + user_message_id = db.add_message(session_id, user_message, "user") + ai_message_id = db.add_message(session_id, assistant_message, "assistant", metadata=ai_metadata or None) + + self.send_json_response({ + "session": db.get_session(session_id), + "user_message_id": user_message_id, + "ai_message_id": ai_message_id, + }) + except Exception as e: + self.send_json_response({"error": str(e)}, status_code=500) + def handle_rename_session(self, session_id: str): """Rename an existing session title""" try: @@ -1081,31 +1079,10 @@ def log_message(self, format, *args): def main(): """Main function to initialize and start the server""" - PORT = 8000 # 🆕 Define port try: # Initialize the database print("✅ Database initialized successfully") - # Initialize the PDF processor - try: - pdf_module.initialize_simple_pdf_processor() - print("📄 Initializing simple PDF processing...") - if pdf_module.simple_pdf_processor: - print("✅ Simple PDF processor initialized") - else: - print("⚠️ PDF processing could not be initialized.") - except Exception as e: - print(f"❌ Error initializing PDF processor: {e}") - print("⚠️ PDF processing disabled - server will run without RAG functionality") - - # Set a global reference to the initialized processor if needed elsewhere - global pdf_processor - pdf_processor = pdf_module.simple_pdf_processor - if pdf_processor: - print("✅ Global PDF processor initialized") - else: - print("⚠️ PDF processing disabled - server will run without RAG functionality") - # Cleanup empty sessions on startup print("🧹 Cleaning up empty sessions...") cleanup_count = db.cleanup_empty_sessions() @@ -1119,7 +1096,9 @@ def main(): print(f"🚀 Starting localGPT backend server on port {PORT}") print(f"📍 Chat endpoint: http://localhost:{PORT}/chat") print(f"🔍 Health check: http://localhost:{PORT}/health") - + print(f"🧠 RAG API: {RAG_API_URL}") + print(f"🤖 Default generation model: {GENERATION_MODEL}") + # Test Ollama connection client = OllamaClient() if client.is_ollama_running(): diff --git a/backend/simple_pdf_processor.py b/backend/simple_pdf_processor.py deleted file mode 100644 index e1f8dfad..00000000 --- a/backend/simple_pdf_processor.py +++ /dev/null @@ -1,214 +0,0 @@ -""" -Simple PDF Processing Service -Handles PDF upload and text extraction for RAG functionality -""" - -import uuid -from typing import List, Dict, Any -import PyPDF2 -from io import BytesIO -import sqlite3 -from datetime import datetime - -class SimplePDFProcessor: - def __init__(self, db_path: str = "chat_data.db"): - """Initialize simple PDF processor with SQLite storage""" - self.db_path = db_path - self.init_database() - print("✅ Simple PDF processor initialized") - - def init_database(self): - """Initialize SQLite database for storing PDF content""" - conn = sqlite3.connect(self.db_path) - conn.execute(''' - CREATE TABLE IF NOT EXISTS pdf_documents ( - id TEXT PRIMARY KEY, - session_id TEXT NOT NULL, - filename TEXT NOT NULL, - content TEXT NOT NULL, - created_at TEXT NOT NULL - ) - ''') - - conn.commit() - conn.close() - - def extract_text_from_pdf(self, pdf_bytes: bytes) -> str: - """Extract text from PDF bytes""" - try: - print(f"📄 Starting PDF text extraction ({len(pdf_bytes)} bytes)") - pdf_file = BytesIO(pdf_bytes) - pdf_reader = PyPDF2.PdfReader(pdf_file) - - print(f"📖 PDF has {len(pdf_reader.pages)} pages") - - text = "" - for page_num, page in enumerate(pdf_reader.pages): - print(f"📄 Processing page {page_num + 1}") - try: - page_text = page.extract_text() - if page_text.strip(): - text += f"\n--- Page {page_num + 1} ---\n" - text += page_text + "\n" - print(f"✅ Page {page_num + 1}: extracted {len(page_text)} characters") - except Exception as page_error: - print(f"❌ Error on page {page_num + 1}: {str(page_error)}") - continue - - print(f"📄 Total extracted text: {len(text)} characters") - return text.strip() - - except Exception as e: - print(f"❌ Error extracting text from PDF: {str(e)}") - print(f"❌ Error type: {type(e).__name__}") - return "" - - def process_pdf(self, pdf_bytes: bytes, filename: str, session_id: str) -> Dict[str, Any]: - """Process a PDF file and store in database""" - print(f"📄 Processing PDF: {filename}") - - # Extract text - text = self.extract_text_from_pdf(pdf_bytes) - if not text: - return { - "success": False, - "error": "Could not extract text from PDF", - "filename": filename - } - - print(f"📝 Extracted {len(text)} characters from {filename}") - - # Store in database - document_id = str(uuid.uuid4()) - now = datetime.now().isoformat() - - try: - conn = sqlite3.connect(self.db_path) - - # Store document - conn.execute(''' - INSERT INTO pdf_documents (id, session_id, filename, content, created_at) - VALUES (?, ?, ?, ?, ?) - ''', (document_id, session_id, filename, text, now)) - - conn.commit() - conn.close() - - print(f"💾 Stored document {filename} in database") - - return { - "success": True, - "filename": filename, - "file_id": document_id, - "text_length": len(text) - } - - except Exception as e: - print(f"❌ Error storing in database: {str(e)}") - return { - "success": False, - "error": f"Database storage failed: {str(e)}", - "filename": filename - } - - def get_session_documents(self, session_id: str) -> List[Dict[str, Any]]: - """Get all documents for a session""" - try: - conn = sqlite3.connect(self.db_path) - conn.row_factory = sqlite3.Row - - cursor = conn.execute(''' - SELECT id, filename, created_at - FROM pdf_documents - WHERE session_id = ? - ORDER BY created_at DESC - ''', (session_id,)) - - documents = [dict(row) for row in cursor.fetchall()] - conn.close() - - return documents - - except Exception as e: - print(f"❌ Error getting session documents: {str(e)}") - return [] - - def get_document_content(self, session_id: str) -> str: - """Get all document content for a session (for LLM context)""" - try: - conn = sqlite3.connect(self.db_path) - - cursor = conn.execute(''' - SELECT filename, content - FROM pdf_documents - WHERE session_id = ? - ORDER BY created_at ASC - ''', (session_id,)) - - rows = cursor.fetchall() - conn.close() - - if not rows: - return "" - - # Combine all document content - combined_content = "" - for filename, content in rows: - combined_content += f"\n\n=== Document: {filename} ===\n\n" - combined_content += content - - return combined_content.strip() - - except Exception as e: - print(f"❌ Error getting document content: {str(e)}") - return "" - - def delete_session_documents(self, session_id: str) -> bool: - """Delete all documents for a session""" - try: - conn = sqlite3.connect(self.db_path) - cursor = conn.execute(''' - DELETE FROM pdf_documents - WHERE session_id = ? - ''', (session_id,)) - - deleted_count = cursor.rowcount - conn.commit() - conn.close() - - if deleted_count > 0: - print(f"🗑️ Deleted {deleted_count} documents for session {session_id[:8]}...") - - return deleted_count > 0 - - except Exception as e: - print(f"❌ Error deleting session documents: {str(e)}") - return False - - -# Global instance -simple_pdf_processor = None - -def initialize_simple_pdf_processor(): - """Initialize the global PDF processor""" - global simple_pdf_processor - try: - simple_pdf_processor = SimplePDFProcessor() - print("✅ Global PDF processor initialized") - except Exception as e: - print(f"❌ Failed to initialize PDF processor: {str(e)}") - simple_pdf_processor = None - -def get_simple_pdf_processor(): - """Get the global PDF processor instance""" - global simple_pdf_processor - if simple_pdf_processor is None: - initialize_simple_pdf_processor() - return simple_pdf_processor - -if __name__ == "__main__": - # Test the simple PDF processor - print("🧪 Testing simple PDF processor...") - - processor = SimplePDFProcessor() - print("✅ Simple PDF processor test completed!") \ No newline at end of file diff --git a/backend/test_backend.py b/backend/test_backend.py deleted file mode 100644 index 0ff94f2f..00000000 --- a/backend/test_backend.py +++ /dev/null @@ -1,153 +0,0 @@ -#!/usr/bin/env python3 -""" -Simple test script for the localGPT backend -""" - -import requests - -def test_health_endpoint(): - """Test the health endpoint""" - print("🔍 Testing health endpoint...") - try: - response = requests.get("http://localhost:8000/health", timeout=5) - if response.status_code == 200: - data = response.json() - print(f"✅ Health check passed") - print(f" Ollama running: {data['ollama_running']}") - print(f" Models available: {len(data['available_models'])}") - return True - else: - print(f"❌ Health check failed: {response.status_code}") - return False - except requests.exceptions.RequestException as e: - print(f"❌ Health check failed: {e}") - return False - -def test_chat_endpoint(): - """Test the chat endpoint""" - print("\n💬 Testing chat endpoint...") - - test_message = { - "message": "Say 'Hello World' and nothing else.", - "model": "llama3.2:latest" - } - - try: - response = requests.post( - "http://localhost:8000/chat", - headers={"Content-Type": "application/json"}, - json=test_message, - timeout=30 - ) - - if response.status_code == 200: - data = response.json() - print(f"✅ Chat test passed") - print(f" Model: {data['model']}") - print(f" Response: {data['response']}") - print(f" Message count: {data['message_count']}") - return True - else: - print(f"❌ Chat test failed: {response.status_code}") - print(f" Response: {response.text}") - return False - - except requests.exceptions.RequestException as e: - print(f"❌ Chat test failed: {e}") - return False - -def test_conversation_history(): - """Test conversation with history""" - print("\n🗨️ Testing conversation history...") - - # First message - conversation = [] - - message1 = { - "message": "My name is Alice. Remember this.", - "model": "llama3.2:latest", - "conversation_history": conversation - } - - try: - response1 = requests.post( - "http://localhost:8000/chat", - headers={"Content-Type": "application/json"}, - json=message1, - timeout=30 - ) - - if response1.status_code == 200: - data1 = response1.json() - - # Add to conversation history - conversation.append({"role": "user", "content": "My name is Alice. Remember this."}) - conversation.append({"role": "assistant", "content": data1["response"]}) - - # Second message asking about the name - message2 = { - "message": "What is my name?", - "model": "llama3.2:latest", - "conversation_history": conversation - } - - response2 = requests.post( - "http://localhost:8000/chat", - headers={"Content-Type": "application/json"}, - json=message2, - timeout=30 - ) - - if response2.status_code == 200: - data2 = response2.json() - print(f"✅ Conversation history test passed") - print(f" First response: {data1['response']}") - print(f" Second response: {data2['response']}") - - # Check if the AI remembered the name - if "alice" in data2['response'].lower(): - print(f"✅ AI correctly remembered the name!") - else: - print(f"⚠️ AI might not have remembered the name") - return True - else: - print(f"❌ Second message failed: {response2.status_code}") - return False - else: - print(f"❌ First message failed: {response1.status_code}") - return False - - except requests.exceptions.RequestException as e: - print(f"❌ Conversation test failed: {e}") - return False - -def main(): - print("🧪 Testing localGPT Backend") - print("=" * 40) - - # Test health endpoint - health_ok = test_health_endpoint() - if not health_ok: - print("\n❌ Backend server is not running or not healthy") - print(" Make sure to run: python server.py") - return - - # Test basic chat - chat_ok = test_chat_endpoint() - if not chat_ok: - print("\n❌ Chat functionality is not working") - return - - # Test conversation history - conversation_ok = test_conversation_history() - - print("\n" + "=" * 40) - if health_ok and chat_ok and conversation_ok: - print("🎉 All tests passed! Backend is ready for frontend integration.") - else: - print("⚠️ Some tests failed. Check the issues above.") - - print("\n🔗 Ready to connect to frontend at http://localhost:3000") - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/backend/test_gateway_routing.py b/backend/test_gateway_routing.py new file mode 100644 index 00000000..c3b1acc8 --- /dev/null +++ b/backend/test_gateway_routing.py @@ -0,0 +1,164 @@ +"""Unit tests for the gateway's deterministic routing gate (no HTTP, no LLM). + +Covers `should_use_rag` / `is_smalltalk_or_meta` in `backend/server.py`, the +retrieval-first cascade that replaced the per-message enrichment-model router: + + force_rag → RAG · no indexes → direct · smalltalk/meta → direct · else RAG + +Run it: + + .venv/bin/python backend/test_gateway_routing.py +""" + +import os +import sys + +BACKEND_DIR = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, BACKEND_DIR) + +from server import is_smalltalk_or_meta, should_use_rag # noqa: E402 + +IDX = ["2fb7a91a-e7ae-46f5-93cd-c8ac52b3ea79"] +NO_IDX: list = [] + +# (message, expected_use_rag, why) +CASES = [ + # --- planted-fact questions from eval/smoke_e2e.py must reach retrieval --- + ("What pressure does the brew boiler operate at during extraction?", True, "planted fact"), + ("Which sensor part should be replaced when error code E11 appears?", True, "planted fact"), + ("How long is the Atlas-7 parts warranty?", True, "planted fact"), + ("Where is the serial number engraved?", True, "planted fact"), + ("Summarize the service manual.", True, "document operation"), + ("Who manufactures the Atlas-7?", True, "entity question"), + ("9.2 bar?", True, "terse but factual"), + ("descaling", True, "bare keyword, no core smalltalk phrase"), + + # --- the old defect: any message containing "test" went direct --- + ("Which test procedure applies after replacing the pump?", True, "old defect: 'test'"), + ("What is the test point voltage on the control board?", True, "old defect: 'test'"), + ("Explain the E11 diagnostic test in detail.", True, "old defect: 'test'"), + ("test", True, "bare 'test' is not an allowlisted smalltalk phrase"), + ("Is this document checked and verified?", True, "old defect: 'check'"), + ("How do I test the group head gasket?", True, "old defect: 'test'"), + + # --- smalltalk shortcuts --- + ("hello", False, "greeting"), + ("Hello!", False, "greeting, punctuated"), + ("hi there", False, "greeting + filler"), + ("hey", False, "greeting"), + ("Good morning", False, "greeting"), + ("how are you?", False, "greeting"), + ("thanks!", False, "thanks"), + ("Thank you so much", False, "thanks"), + ("thx", False, "thanks"), + ("bye", False, "farewell"), + ("goodbye, see you later", False, "farewell"), + ("ok", False, "acknowledgement"), + ("got it, thanks", False, "acknowledgement + thanks"), + ("nevermind", False, "acknowledgement"), + ("", False, "empty message never needs retrieval"), + (" ", False, "whitespace-only"), + + # --- assistant-meta shortcuts --- + ("who are you?", False, "meta"), + ("Who are you", False, "meta"), + ("what model are you", False, "meta"), + ("what model are you?", False, "meta"), + ("which model do you use?", False, "meta"), + ("what are you?", False, "meta"), + ("what is your name?", False, "meta"), + ("are you an AI?", False, "meta"), + ("who built you?", False, "meta"), + ("what can you do?", False, "meta"), + ("tell me about yourself", False, "meta"), + + # --- smalltalk words inside a real question must NOT shortcut --- + ("Hello, what is the brew boiler pressure?", True, "greeting prefix + real question"), + ("Thanks - now summarize section 4 for me", True, "thanks prefix + real question"), + ("Who are the authors of the service manual?", True, "'who are' but not about the assistant"), + ("What model number is the pressure sensor?", True, "'what model' but not about the assistant"), + ("Is the machine ok to run at 1.45 bar?", True, "'ok' inside a real question"), + ("no problem code is listed for E12?", True, "'no problem' inside a real question"), + ("What is a good night mode setting?", True, "'good night' inside a real question"), +] + + +def check(label, actual, expected, why, kind="route"): + ok = actual == expected + fmt = (lambda v: ("RAG" if v else "DIRECT")) if kind == "route" else (lambda v: str(v)) + print(f" {'PASS' if ok else 'FAIL'} {label:<62} -> {fmt(actual):<6}" + f" (expected {fmt(expected)}; {why})") + return ok + + +def main(): + failures = 0 + total = 0 + + print("\n[1] Session WITH linked indexes") + for message, expected, why in CASES: + total += 1 + label = repr(message) if len(message) < 60 else repr(message[:57] + "...") + if not check(label, should_use_rag(message, IDX), expected, why): + failures += 1 + + print("\n[2] Session with NO linked indexes -> always direct") + for message, _expected, _why in CASES: + total += 1 + label = repr(message) if len(message) < 60 else repr(message[:57] + "...") + if not check(label, should_use_rag(message, NO_IDX), False, "no indexes linked"): + failures += 1 + for empty in (None, [], ()): + total += 1 + if not check(f"idx_ids={empty!r}", should_use_rag("What is the brew pressure?", empty), + False, "no indexes linked"): + failures += 1 + + print("\n[3] force_rag is honored") + force_cases = [ + ("hello", IDX), + ("thanks!", IDX), + ("who are you?", IDX), + ("What pressure does the brew boiler operate at?", IDX), + ("hello", NO_IDX), # force_rag wins even with no indexes + ("", NO_IDX), + ] + for message, idx in force_cases: + total += 1 + if not check(f"force_rag {message!r} idx={bool(idx)}", + should_use_rag(message, idx, force_rag=True), True, "force_rag=True"): + failures += 1 + + print("\n[4] No LLM call is made by the gate") + # server.should_use_rag must not reference an Ollama client at all: the gate + # is a module-level function with no client argument and no network use. + import inspect + import server + total += 1 + src = inspect.getsource(server.should_use_rag) + inspect.getsource(server.is_smalltalk_or_meta) + clean = not any(tok in src for tok in ("ollama", "requests", "ENRICHMENT_MODEL", "http")) + if not check("gate source is free of LLM/network calls", clean, True, + "deterministic gate", kind="bool"): + failures += 1 + total += 1 + removed = not any(hasattr(server.ChatHandler, name) for name in + ("_should_use_rag", "_simple_pattern_routing", + "_route_using_overviews", "_load_document_overviews")) + if not check("old LLM router + pattern fallback deleted", removed, True, + "no _simple_pattern_routing / _route_using_overviews", kind="bool"): + failures += 1 + + print("\n[5] is_smalltalk_or_meta is index-independent") + for message, expected_rag, why in CASES: + total += 1 + # With indexes linked, should_use_rag is exactly `not is_smalltalk_or_meta`. + if not check(f"consistency {message[:40]!r}", + should_use_rag(message, IDX), not is_smalltalk_or_meta(message), why): + failures += 1 + + print(f"\n{total - failures}/{total} checks passed") + return 1 if failures else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/backend/test_ollama_connectivity.py b/backend/test_ollama_connectivity.py deleted file mode 100644 index d4e2e65c..00000000 --- a/backend/test_ollama_connectivity.py +++ /dev/null @@ -1,37 +0,0 @@ -#!/usr/bin/env python3 - -import os -import sys - -def test_ollama_connectivity(): - """Test Ollama connectivity from within Docker container""" - print("🧪 Testing Ollama Connectivity") - print("=" * 40) - - ollama_host = os.getenv('OLLAMA_HOST', 'Not set') - print(f"OLLAMA_HOST environment variable: {ollama_host}") - - try: - from ollama_client import OllamaClient - client = OllamaClient() - print(f"OllamaClient base_url: {client.base_url}") - - is_running = client.is_ollama_running() - print(f"Ollama running: {is_running}") - - if is_running: - models = client.list_models() - print(f"Available models: {models}") - print("✅ Ollama connectivity test passed!") - return True - else: - print("❌ Ollama connectivity test failed!") - return False - - except Exception as e: - print(f"❌ Error testing Ollama connectivity: {e}") - return False - -if __name__ == "__main__": - success = test_ollama_connectivity() - sys.exit(0 if success else 1) diff --git a/batch_indexing_config.json b/batch_indexing_config.json deleted file mode 100644 index 8ac66256..00000000 --- a/batch_indexing_config.json +++ /dev/null @@ -1,19 +0,0 @@ -{ - "index_name": "Sample Batch Index", - "index_description": "Example batch index configuration", - "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" - ], - "processing": { - "chunk_size": 512, - "chunk_overlap": 64, - "enable_enrich": true, - "enable_latechunk": true, - "enable_docling": true, - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:0.6b", - "retrieval_mode": "hybrid", - "window_size": 2 - } -} \ No newline at end of file diff --git a/create_index_script.py b/create_index_script.py index dc7b894e..dadd17f2 100644 --- a/create_index_script.py +++ b/create_index_script.py @@ -8,10 +8,12 @@ Usage: python create_index_script.py - python create_index_script.py --batch - python create_index_script.py --config custom_config.json + python create_index_script.py --batch index_config.json + python create_index_script.py --config custom_pipeline_config.json + python create_index_script.py --create-sample """ +import copy import os import sys import json @@ -23,7 +25,8 @@ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) try: - from rag_system.main import PIPELINE_CONFIGS, get_agent + from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG + from rag_system.factory import get_agent from rag_system.pipelines.indexing_pipeline import IndexingPipeline from rag_system.utils.ollama_client import OllamaClient from backend.database import ChatDatabase @@ -35,26 +38,14 @@ class IndexCreator: """Interactive index creation utility.""" - + def __init__(self, config_path: Optional[str] = None): """Initialize the index creator with optional custom configuration.""" self.db = ChatDatabase() - self.config = self._load_config(config_path) - - # Initialize Ollama client - self.ollama_client = OllamaClient() - self.ollama_config = { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - } - - # Initialize indexing pipeline - self.pipeline = IndexingPipeline( - self.config, - self.ollama_client, - self.ollama_config - ) - + self.base_config = self._load_config(config_path) + self.ollama_config = OLLAMA_CONFIG + self.ollama_client = OllamaClient(host=OLLAMA_CONFIG["host"]) + def _load_config(self, config_path: Optional[str] = None) -> dict: """Load configuration from file or use default.""" if config_path and os.path.exists(config_path): @@ -64,9 +55,37 @@ def _load_config(self, config_path: Optional[str] = None) -> dict: except Exception as e: print(f"⚠️ Error loading config from {config_path}: {e}") print("Using default configuration...") - + return PIPELINE_CONFIGS.get("default", {}) - + + def _build_pipeline(self, index_id: str, processing: dict) -> IndexingPipeline: + """Build a pipeline that writes into this index's own tables.""" + config = copy.deepcopy(self.base_config) + table_name = f"text_pages_{index_id}" + + config.setdefault("storage", {})["text_table_name"] = table_name + # The pipeline reads whichever of these two keys is present. + retrievers = config.get("retrievers") + if retrievers is None: + retrievers = config.setdefault("retrieval", {}) + retrievers.setdefault("dense", {})["lancedb_table_name"] = table_name + retrievers.setdefault("latechunk", {})["enabled"] = bool(processing.get("enable_latechunk", False)) + + config["chunker_mode"] = "docling" if processing.get("enable_docling", True) else "legacy" + config.setdefault("chunking", {})["chunk_size"] = int(processing.get("chunk_size", 512)) + config.setdefault("contextual_enricher", {}).update({ + "enabled": bool(processing.get("enable_enrich", True)), + "window_size": int(processing.get("window_size", 2)), + }) + config["overview_path"] = f"index_store/overviews/{index_id}.jsonl" + + if processing.get("embedding_model"): + config["embedding_model_name"] = processing["embedding_model"] + if processing.get("enrich_model"): + config["enrich_model"] = processing["enrich_model"] + + return IndexingPipeline(config, self.ollama_client, self.ollama_config) + def get_user_input(self, prompt: str, default: str = "") -> str: """Get user input with optional default value.""" if default: @@ -148,33 +167,34 @@ def configure_processing(self) -> dict: print("Configure how documents will be processed:") # Basic settings - chunk_size = int(self.get_user_input("Chunk size", "512")) - chunk_overlap = int(self.get_user_input("Chunk overlap", "64")) - + chunk_size = int(self.get_user_input("Chunk size (tokens)", "512")) + # Advanced settings print("\nAdvanced options:") enable_enrich = self.get_user_input("Enable contextual enrichment? (y/n)", "y").lower() == 'y' enable_latechunk = self.get_user_input("Enable late chunking? (y/n)", "y").lower() == 'y' enable_docling = self.get_user_input("Enable Docling chunking? (y/n)", "y").lower() == 'y' - + # Model selection print("\nModel Configuration:") - embedding_model = self.get_user_input("Embedding model", "Qwen/Qwen3-Embedding-0.6B") - generation_model = self.get_user_input("Generation model", "qwen3:0.6b") - + default_embedding = self.base_config.get("embedding_model_name", "") + embedding_model = self.get_user_input("Embedding model", default_embedding) + enrich_model = self.get_user_input( + "Enrichment model", self.ollama_config.get("enrichment_model", "") + ) + return { "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, "enable_enrich": enable_enrich, "enable_latechunk": enable_latechunk, "enable_docling": enable_docling, "embedding_model": embedding_model, - "generation_model": generation_model, + "enrich_model": enrich_model, "retrieval_mode": "hybrid", "window_size": 2 } - - def create_index_interactive(self) -> None: + + def create_index_interactive(self) -> bool: """Run the interactive index creation process.""" print("🚀 LocalGPT Index Creation Tool") print("=" * 50) @@ -201,41 +221,48 @@ def create_index_interactive(self) -> None: if self.get_user_input("\nProceed with index creation? (y/n)", "y").lower() != 'y': print("❌ Index creation cancelled.") - return - + return False + # Create the index + index_id = None try: print("\n🔥 Creating index...") - + # Create index record in database index_id = self.db.create_index( name=index_name, description=index_description, metadata=processing_config ) - + # Add documents to index for doc_path in documents: filename = os.path.basename(doc_path) self.db.add_document_to_index(index_id, filename, doc_path) - + # Process documents through pipeline print("📚 Processing documents...") - self.pipeline.process_documents(documents) - + self._build_pipeline(index_id, processing_config).run(documents) + print(f"\n✅ Index '{index_name}' created successfully!") print(f"Index ID: {index_id}") print(f"Processed {len(documents)} documents") - + # Test the index if self.get_user_input("\nTest the index with a sample query? (y/n)", "y").lower() == 'y': self.test_index(index_id) - + + return True + except Exception as e: print(f"❌ Error creating index: {e}") import traceback traceback.print_exc() - + if index_id: + print(f"🧹 Removing incomplete index record {index_id}") + self.db.delete_index(index_id) + return False + def test_index(self, index_id: str) -> None: """Test the created index with a sample query.""" try: @@ -257,58 +284,67 @@ def test_index(self, index_id: str) -> None: except Exception as e: print(f"❌ Error testing index: {e}") - def batch_create_from_config(self, config_file: str) -> None: + def batch_create_from_config(self, config_file: str) -> bool: """Create index from batch configuration file.""" + index_id = None try: with open(config_file, 'r') as f: batch_config = json.load(f) - + index_name = batch_config.get("index_name", "Batch Index") index_description = batch_config.get("index_description", "") documents = batch_config.get("documents", []) processing_config = batch_config.get("processing", {}) - + if not documents: print("❌ No documents specified in batch configuration") - return - + return False + # Validate documents exist valid_documents = [] for doc_path in documents: if os.path.exists(doc_path): - valid_documents.append(doc_path) + valid_documents.append(os.path.abspath(doc_path)) else: print(f"⚠️ Document not found: {doc_path}") - + if not valid_documents: print("❌ No valid documents found") - return - + return False + print(f"🚀 Creating batch index: {index_name}") print(f"📄 Processing {len(valid_documents)} documents...") - + # Create index index_id = self.db.create_index( name=index_name, description=index_description, metadata=processing_config ) - + # Add documents for doc_path in valid_documents: filename = os.path.basename(doc_path) self.db.add_document_to_index(index_id, filename, doc_path) - + # Process documents - self.pipeline.process_documents(valid_documents) - + self._build_pipeline(index_id, processing_config).run(valid_documents) + print(f"✅ Batch index '{index_name}' created successfully!") print(f"Index ID: {index_id}") - + return True + except Exception as e: print(f"❌ Error creating batch index: {e}") import traceback traceback.print_exc() + if index_id: + print(f"🧹 Removing incomplete index record {index_id}") + self.db.delete_index(index_id) + return False + + +SAMPLE_CONFIG_FILENAME = "index_config.sample.json" def create_sample_batch_config(): @@ -317,56 +353,59 @@ def create_sample_batch_config(): "index_name": "Sample Batch Index", "index_description": "Example batch index configuration", "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" + "/absolute/path/to/first.pdf", + "/absolute/path/to/second.pdf" ], "processing": { "chunk_size": 512, - "chunk_overlap": 64, "enable_enrich": True, "enable_latechunk": True, "enable_docling": True, - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", - "generation_model": "qwen3:0.6b", + "embedding_model": PIPELINE_CONFIGS["default"]["embedding_model_name"], + "enrich_model": OLLAMA_CONFIG["enrichment_model"], "retrieval_mode": "hybrid", "window_size": 2 } } - - with open("batch_indexing_config.json", "w") as f: + + with open(SAMPLE_CONFIG_FILENAME, "w") as f: json.dump(sample_config, f, indent=2) - - print("📄 Sample batch configuration created: batch_indexing_config.json") + + print(f"📄 Sample batch configuration created: {SAMPLE_CONFIG_FILENAME}") -def main(): +def main() -> int: """Main entry point for the script.""" parser = argparse.ArgumentParser(description="LocalGPT Index Creation Tool") parser.add_argument("--batch", help="Batch configuration file", type=str) parser.add_argument("--config", help="Custom pipeline configuration file", type=str) parser.add_argument("--create-sample", action="store_true", help="Create sample batch config") - + args = parser.parse_args() - + if args.create_sample: create_sample_batch_config() - return - + return 0 + try: creator = IndexCreator(config_path=args.config) - + if args.batch: - creator.batch_create_from_config(args.batch) + ok = creator.batch_create_from_config(args.batch) else: - creator.create_index_interactive() - + ok = creator.create_index_interactive() + + return 0 if ok else 1 + except KeyboardInterrupt: print("\n\n❌ Operation cancelled by user.") + return 130 except Exception as e: print(f"❌ Unexpected error: {e}") import traceback traceback.print_exc() + return 1 if __name__ == "__main__": - main() \ No newline at end of file + sys.exit(main()) diff --git a/demo_batch_indexing.py b/demo_batch_indexing.py deleted file mode 100644 index 06a1847b..00000000 --- a/demo_batch_indexing.py +++ /dev/null @@ -1,386 +0,0 @@ -#!/usr/bin/env python3 -""" -Demo Batch Indexing Script for LocalGPT RAG System - -This script demonstrates how to perform batch indexing of multiple documents -using configuration files. It's designed to showcase the full capabilities -of the indexing pipeline with various configuration options. - -Usage: - python demo_batch_indexing.py --config batch_indexing_config.json - python demo_batch_indexing.py --create-sample-config - python demo_batch_indexing.py --help -""" - -import os -import sys -import json -import argparse -import time -import logging -from typing import List, Dict, Any, Optional -from pathlib import Path -from datetime import datetime - -# Add the project root to the path so we can import rag_system modules -sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) - -try: - from rag_system.main import PIPELINE_CONFIGS - from rag_system.pipelines.indexing_pipeline import IndexingPipeline - from rag_system.utils.ollama_client import OllamaClient - from backend.database import ChatDatabase -except ImportError as e: - print(f"❌ Error importing required modules: {e}") - print("Please ensure you're running this script from the project root directory.") - sys.exit(1) - -# Configure logging -logging.basicConfig( - level=logging.INFO, - format="%(asctime)s | %(levelname)-7s | %(name)s | %(message)s", -) - - -class BatchIndexingDemo: - """Demonstration of batch indexing capabilities.""" - - def __init__(self, config_path: str): - """Initialize the batch indexing demo.""" - self.config_path = config_path - self.config = self._load_config() - self.db = ChatDatabase() - - # Initialize Ollama client - self.ollama_client = OllamaClient() - - # Initialize pipeline with merged configuration - self.pipeline_config = self._merge_configurations() - self.pipeline = IndexingPipeline( - self.pipeline_config, - self.ollama_client, - self.config.get("ollama_config", { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - }) - ) - - def _load_config(self) -> Dict[str, Any]: - """Load batch indexing configuration from file.""" - try: - with open(self.config_path, 'r') as f: - config = json.load(f) - print(f"✅ Loaded configuration from {self.config_path}") - return config - except FileNotFoundError: - print(f"❌ Configuration file not found: {self.config_path}") - sys.exit(1) - except json.JSONDecodeError as e: - print(f"❌ Invalid JSON in configuration file: {e}") - sys.exit(1) - - def _merge_configurations(self) -> Dict[str, Any]: - """Merge batch config with default pipeline config.""" - # Start with default pipeline configuration - merged_config = PIPELINE_CONFIGS.get("default", {}).copy() - - # Override with batch-specific settings - batch_settings = self.config.get("pipeline_settings", {}) - - # Deep merge for nested dictionaries - def deep_merge(base: dict, override: dict) -> dict: - result = base.copy() - for key, value in override.items(): - if key in result and isinstance(result[key], dict) and isinstance(value, dict): - result[key] = deep_merge(result[key], value) - else: - result[key] = value - return result - - return deep_merge(merged_config, batch_settings) - - def validate_documents(self, documents: List[str]) -> List[str]: - """Validate and filter document paths.""" - valid_documents = [] - - print(f"📋 Validating {len(documents)} documents...") - - for doc_path in documents: - # Handle relative paths - if not os.path.isabs(doc_path): - doc_path = os.path.abspath(doc_path) - - if os.path.exists(doc_path): - # Check file extension - ext = Path(doc_path).suffix.lower() - if ext in ['.pdf', '.txt', '.docx', '.md', '.html', '.htm']: - valid_documents.append(doc_path) - print(f" ✅ {doc_path}") - else: - print(f" ⚠️ Unsupported file type: {doc_path}") - else: - print(f" ❌ File not found: {doc_path}") - - print(f"📊 {len(valid_documents)} valid documents found") - return valid_documents - - def create_indexes(self) -> List[str]: - """Create multiple indexes based on configuration.""" - indexes = self.config.get("indexes", []) - created_indexes = [] - - for index_config in indexes: - index_id = self.create_single_index(index_config) - if index_id: - created_indexes.append(index_id) - - return created_indexes - - def create_single_index(self, index_config: Dict[str, Any]) -> Optional[str]: - """Create a single index from configuration.""" - try: - # Extract index metadata - index_name = index_config.get("name", "Unnamed Index") - index_description = index_config.get("description", "") - documents = index_config.get("documents", []) - - if not documents: - print(f"⚠️ No documents specified for index '{index_name}', skipping...") - return None - - # Validate documents - valid_documents = self.validate_documents(documents) - if not valid_documents: - print(f"❌ No valid documents found for index '{index_name}'") - return None - - print(f"\n🚀 Creating index: {index_name}") - print(f"📄 Processing {len(valid_documents)} documents") - - # Create index record in database - index_metadata = { - "created_by": "demo_batch_indexing.py", - "created_at": datetime.now().isoformat(), - "document_count": len(valid_documents), - "config_used": index_config.get("processing_options", {}) - } - - index_id = self.db.create_index( - name=index_name, - description=index_description, - metadata=index_metadata - ) - - # Add documents to index - for doc_path in valid_documents: - filename = os.path.basename(doc_path) - self.db.add_document_to_index(index_id, filename, doc_path) - - # Process documents through pipeline - start_time = time.time() - self.pipeline.process_documents(valid_documents) - processing_time = time.time() - start_time - - print(f"✅ Index '{index_name}' created successfully!") - print(f" Index ID: {index_id}") - print(f" Processing time: {processing_time:.2f} seconds") - print(f" Documents processed: {len(valid_documents)}") - - return index_id - - except Exception as e: - print(f"❌ Error creating index '{index_name}': {e}") - import traceback - traceback.print_exc() - return None - - def demonstrate_features(self): - """Demonstrate various indexing features.""" - print("\n🎯 Batch Indexing Demo Features:") - print("=" * 50) - - # Show configuration - print(f"📋 Configuration file: {self.config_path}") - print(f"📊 Number of indexes to create: {len(self.config.get('indexes', []))}") - - # Show pipeline settings - pipeline_settings = self.config.get("pipeline_settings", {}) - if pipeline_settings: - print("\n⚙️ Pipeline Settings:") - for key, value in pipeline_settings.items(): - print(f" {key}: {value}") - - # Show model configuration - ollama_config = self.config.get("ollama_config", {}) - if ollama_config: - print("\n🤖 Model Configuration:") - for key, value in ollama_config.items(): - print(f" {key}: {value}") - - def run_demo(self): - """Run the complete batch indexing demo.""" - print("🚀 LocalGPT Batch Indexing Demo") - print("=" * 50) - - # Show demo features - self.demonstrate_features() - - # Create indexes - print(f"\n📚 Starting batch indexing process...") - start_time = time.time() - - created_indexes = self.create_indexes() - - total_time = time.time() - start_time - - # Summary - print(f"\n📊 Batch Indexing Summary") - print("=" * 50) - print(f"✅ Successfully created {len(created_indexes)} indexes") - print(f"⏱️ Total processing time: {total_time:.2f} seconds") - - if created_indexes: - print(f"\n📋 Created Indexes:") - for i, index_id in enumerate(created_indexes, 1): - index_info = self.db.get_index(index_id) - if index_info: - print(f" {i}. {index_info['name']} ({index_id[:8]}...)") - print(f" Documents: {len(index_info.get('documents', []))}") - - print(f"\n🎉 Demo completed successfully!") - print(f"💡 You can now use these indexes in the LocalGPT interface.") - - -def create_sample_config(): - """Create a comprehensive sample configuration file.""" - sample_config = { - "description": "Demo batch indexing configuration showcasing various features", - "pipeline_settings": { - "embedding_model_name": "Qwen/Qwen3-Embedding-0.6B", - "indexing": { - "embedding_batch_size": 50, - "enrichment_batch_size": 25, - "enable_progress_tracking": True - }, - "contextual_enricher": { - "enabled": True, - "window_size": 2, - "model_name": "qwen3:0.6b" - }, - "chunking": { - "chunk_size": 512, - "chunk_overlap": 64, - "enable_latechunk": True, - "enable_docling": True - }, - "retrievers": { - "dense": { - "enabled": True, - "lancedb_table_name": "demo_text_pages" - }, - "bm25": { - "enabled": True, - "index_name": "demo_bm25_index" - } - }, - "storage": { - "lancedb_uri": "./index_store/lancedb", - "bm25_path": "./index_store/bm25" - } - }, - "ollama_config": { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - }, - "indexes": [ - { - "name": "Sample Invoice Collection", - "description": "Demo index containing sample invoice documents", - "documents": [ - "./rag_system/documents/invoice_1039.pdf", - "./rag_system/documents/invoice_1041.pdf" - ], - "processing_options": { - "chunk_size": 512, - "enable_enrichment": True, - "retrieval_mode": "hybrid" - } - }, - { - "name": "Research Papers Demo", - "description": "Demo index for research papers and whitepapers", - "documents": [ - "./rag_system/documents/Newwhitepaper_Agents2.pdf" - ], - "processing_options": { - "chunk_size": 1024, - "enable_enrichment": True, - "retrieval_mode": "dense" - } - } - ] - } - - config_filename = "batch_indexing_config.json" - with open(config_filename, "w") as f: - json.dump(sample_config, f, indent=2) - - print(f"✅ Sample configuration created: {config_filename}") - print(f"📝 Edit this file to customize your batch indexing setup") - print(f"🚀 Run: python demo_batch_indexing.py --config {config_filename}") - - -def main(): - """Main entry point for the demo script.""" - parser = argparse.ArgumentParser( - description="LocalGPT Batch Indexing Demo", - formatter_class=argparse.RawDescriptionHelpFormatter, - epilog=""" -Examples: - python demo_batch_indexing.py --config batch_indexing_config.json - python demo_batch_indexing.py --create-sample-config - -This demo showcases the advanced batch indexing capabilities of LocalGPT, -including multi-index creation, advanced configuration options, and -comprehensive processing pipelines. - """ - ) - - parser.add_argument( - "--config", - type=str, - default="batch_indexing_config.json", - help="Path to batch indexing configuration file" - ) - - parser.add_argument( - "--create-sample-config", - action="store_true", - help="Create a sample configuration file" - ) - - args = parser.parse_args() - - if args.create_sample_config: - create_sample_config() - return - - if not os.path.exists(args.config): - print(f"❌ Configuration file not found: {args.config}") - print(f"💡 Create a sample config with: python {sys.argv[0]} --create-sample-config") - sys.exit(1) - - try: - demo = BatchIndexingDemo(args.config) - demo.run_demo() - - except KeyboardInterrupt: - print("\n\n❌ Demo cancelled by user.") - except Exception as e: - print(f"❌ Demo failed: {e}") - import traceback - traceback.print_exc() - - -if __name__ == "__main__": - main() \ No newline at end of file diff --git a/docker-compose.local-ollama.yml b/docker-compose.local-ollama.yml index a51dee0a..71bc924f 100644 --- a/docker-compose.local-ollama.yml +++ b/docker-compose.local-ollama.yml @@ -8,14 +8,24 @@ services: ports: - "8001:8001" environment: - - OLLAMA_HOST=http://host.docker.internal:11434 + - OLLAMA_HOST=${OLLAMA_HOST:-http://host.docker.internal:11434} - NODE_ENV=production + - DB_PATH=${DB_PATH:-/app/backend/chat_data.db} + - LANCEDB_PATH=${LANCEDB_PATH:-/app/lancedb} + - GENERATION_MODEL=${GENERATION_MODEL:-qwen3.5:9b} + - ENRICHMENT_MODEL=${ENRICHMENT_MODEL:-qwen3.5:4b} + - EMBEDDING_MODEL=${EMBEDDING_MODEL:-microsoft/harrier-oss-v1-0.6b} + - RERANKER_MODEL=${RERANKER_MODEL:-Qwen/Qwen3-Reranker-4B} + extra_hosts: + - "host.docker.internal:host-gateway" volumes: - ./lancedb:/app/lancedb - ./index_store:/app/index_store - ./shared_uploads:/app/shared_uploads + # Shared SQLite database - same mount as the backend service + - ./backend:/app/backend healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:8001/models"] + test: ["CMD", "curl", "-f", "http://localhost:8001/health"] interval: 30s timeout: 10s retries: 3 @@ -33,10 +43,20 @@ services: - "8000:8000" environment: - NODE_ENV=production - - RAG_API_URL=http://rag-api:8001 + - RAG_API_URL=${RAG_API_URL:-http://rag-api:8001} + - OLLAMA_HOST=${OLLAMA_HOST:-http://host.docker.internal:11434} + - DB_PATH=${DB_PATH:-/app/backend/chat_data.db} + - LANCEDB_PATH=${LANCEDB_PATH:-/app/lancedb} + - GENERATION_MODEL=${GENERATION_MODEL:-qwen3.5:9b} + - ENRICHMENT_MODEL=${ENRICHMENT_MODEL:-qwen3.5:4b} + extra_hosts: + - "host.docker.internal:host-gateway" volumes: - - ./backend/chat_data.db:/app/backend/chat_data.db + # Shared SQLite database - same mount as the rag-api service + - ./backend:/app/backend - ./shared_uploads:/app/shared_uploads + - ./index_store:/app/index_store + - ./lancedb:/app/lancedb depends_on: rag-api: condition: service_healthy @@ -54,17 +74,24 @@ services: build: context: . dockerfile: Dockerfile.frontend + args: + NEXT_PUBLIC_API_URL: ${NEXT_PUBLIC_API_URL:-http://localhost:8000} + NEXT_PUBLIC_RAG_API_URL: ${NEXT_PUBLIC_RAG_API_URL:-http://localhost:8001} container_name: rag-frontend ports: - "3000:3000" environment: - NODE_ENV=production - - NEXT_PUBLIC_API_URL=http://localhost:8000 + - NEXT_PUBLIC_API_URL=${NEXT_PUBLIC_API_URL:-http://localhost:8000} + - NEXT_PUBLIC_RAG_API_URL=${NEXT_PUBLIC_RAG_API_URL:-http://localhost:8001} + extra_hosts: + - "host.docker.internal:host-gateway" depends_on: backend: condition: service_healthy healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:3000"] + # node:20-alpine ships busybox wget, not curl + test: ["CMD-SHELL", "wget -qO- http://localhost:3000 >/dev/null 2>&1 || exit 1"] interval: 30s timeout: 10s retries: 3 @@ -74,4 +101,4 @@ services: networks: rag-network: - driver: bridge \ No newline at end of file + driver: bridge diff --git a/docker-compose.yml b/docker-compose.yml index abf8b80f..394c01d0 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -32,12 +32,22 @@ services: # Use host Ollama by default, or containerized Ollama if enabled - OLLAMA_HOST=${OLLAMA_HOST:-http://host.docker.internal:11434} - NODE_ENV=production + - DB_PATH=${DB_PATH:-/app/backend/chat_data.db} + - LANCEDB_PATH=${LANCEDB_PATH:-/app/lancedb} + - GENERATION_MODEL=${GENERATION_MODEL:-qwen3.5:9b} + - ENRICHMENT_MODEL=${ENRICHMENT_MODEL:-qwen3.5:4b} + - EMBEDDING_MODEL=${EMBEDDING_MODEL:-microsoft/harrier-oss-v1-0.6b} + - RERANKER_MODEL=${RERANKER_MODEL:-Qwen/Qwen3-Reranker-4B} + extra_hosts: + - "host.docker.internal:host-gateway" volumes: - ./lancedb:/app/lancedb - ./index_store:/app/index_store - ./shared_uploads:/app/shared_uploads + # Shared SQLite database - same mount as the backend service + - ./backend:/app/backend healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:8001/models"] + test: ["CMD", "curl", "-f", "http://localhost:8001/health"] interval: 30s timeout: 10s retries: 3 @@ -55,11 +65,20 @@ services: - "8000:8000" environment: - NODE_ENV=production - - RAG_API_URL=http://rag-api:8001 - - OLLAMA_HOST=${OLLAMA_HOST:-http://172.18.0.1:11434} + - RAG_API_URL=${RAG_API_URL:-http://rag-api:8001} + - OLLAMA_HOST=${OLLAMA_HOST:-http://host.docker.internal:11434} + - DB_PATH=${DB_PATH:-/app/backend/chat_data.db} + - LANCEDB_PATH=${LANCEDB_PATH:-/app/lancedb} + - GENERATION_MODEL=${GENERATION_MODEL:-qwen3.5:9b} + - ENRICHMENT_MODEL=${ENRICHMENT_MODEL:-qwen3.5:4b} + extra_hosts: + - "host.docker.internal:host-gateway" volumes: + # Shared SQLite database - same mount as the rag-api service - ./backend:/app/backend - ./shared_uploads:/app/shared_uploads + - ./index_store:/app/index_store + - ./lancedb:/app/lancedb depends_on: rag-api: condition: service_healthy @@ -77,17 +96,24 @@ services: build: context: . dockerfile: Dockerfile.frontend + args: + NEXT_PUBLIC_API_URL: ${NEXT_PUBLIC_API_URL:-http://localhost:8000} + NEXT_PUBLIC_RAG_API_URL: ${NEXT_PUBLIC_RAG_API_URL:-http://localhost:8001} container_name: rag-frontend ports: - "3000:3000" environment: - NODE_ENV=production - - NEXT_PUBLIC_API_URL=http://localhost:8000 + - NEXT_PUBLIC_API_URL=${NEXT_PUBLIC_API_URL:-http://localhost:8000} + - NEXT_PUBLIC_RAG_API_URL=${NEXT_PUBLIC_RAG_API_URL:-http://localhost:8001} + extra_hosts: + - "host.docker.internal:host-gateway" depends_on: backend: condition: service_healthy healthcheck: - test: ["CMD", "curl", "-f", "http://localhost:3000"] + # node:20-alpine ships busybox wget, not curl + test: ["CMD-SHELL", "wget -qO- http://localhost:3000 >/dev/null 2>&1 || exit 1"] interval: 30s timeout: 10s retries: 3 @@ -101,4 +127,4 @@ volumes: networks: rag-network: - driver: bridge \ No newline at end of file + driver: bridge diff --git a/docker.env b/docker.env index 4bb29eed..6b9a57f6 100644 --- a/docker.env +++ b/docker.env @@ -1,12 +1,26 @@ -# Docker environment configuration -# Set this to use local Ollama instance running on host -# Note: Using Docker gateway IP instead of host.docker.internal for Linux compatibility -OLLAMA_HOST=http://172.18.0.1:11434 +# Docker environment configuration (passed via: docker compose --env-file docker.env ...) -# Alternative: Use containerized Ollama (uncomment and run with --profile with-ollama) +# Ollama running on the host. The compose files add +# extra_hosts: ["host.docker.internal:host-gateway"] so this also resolves on Linux. +OLLAMA_HOST=http://host.docker.internal:11434 + +# Alternative: containerized Ollama (start with: ./start-docker.sh container) # OLLAMA_HOST=http://ollama:11434 -# Other configuration +# Service wiring NODE_ENV=production +RAG_API_URL=http://rag-api:8001 + +# Browser-facing URLs. These are inlined into the frontend at build time. NEXT_PUBLIC_API_URL=http://localhost:8000 -RAG_API_URL=http://rag-api:8001 \ No newline at end of file +NEXT_PUBLIC_RAG_API_URL=http://localhost:8001 + +# Shared SQLite database, mounted into both backend and rag-api +DB_PATH=/app/backend/chat_data.db +LANCEDB_PATH=/app/lancedb + +# Models +GENERATION_MODEL=qwen3.5:9b +ENRICHMENT_MODEL=qwen3.5:4b +EMBEDDING_MODEL=microsoft/harrier-oss-v1-0.6b +RERANKER_MODEL=Qwen/Qwen3-Reranker-4B diff --git a/env.example.watsonx b/env.example.watsonx index 5c3f3f86..12fd353f 100644 --- a/env.example.watsonx +++ b/env.example.watsonx @@ -1,11 +1,18 @@ # ==================================================================== -# LocalGPT Watson X Configuration Example +# localGPT Watson X Configuration Example # ==================================================================== -# This file shows how to configure LocalGPT to use IBM Watson X AI -# with Granite models instead of local Ollama. +# This file shows how to point localGPT's LLM calls at IBM watsonx.ai +# Granite models instead of a local Ollama server. # # Copy this file to .env and fill in your credentials: -# cp .env.example.watsonx .env +# cp env.example.watsonx .env +# +# (The filename has no leading dot on purpose: .gitignore ignores .env*, +# so a dotted example file could never be committed.) +# +# See WATSONX_README.md for what the backend switch does and does not +# cover. Short version: generation and all utility LLM calls move to +# Watson X; embeddings, reranking and sentence pruning stay local. # ==================================================================== # LLM Backend Selection @@ -15,17 +22,20 @@ LLM_BACKEND=watsonx # ==================================================================== # Watson X Credentials # ==================================================================== -# Get these from your IBM Cloud Watson X project: +# Get these from your IBM Cloud watsonx.ai project: # 1. Go to https://cloud.ibm.com/ -# 2. Navigate to Watson X AI service +# 2. Navigate to the watsonx.ai service # 3. Create or select a project -# 4. Get API key from IBM Cloud IAM -# 5. Copy project ID from project settings +# 4. Get an API key from IBM Cloud IAM +# 5. Copy the project ID from the project settings +# +# Both of the next two are mandatory - rag_system/factory.py raises +# "Watson X configuration incomplete" if either is empty. # Your IBM Cloud API key WATSONX_API_KEY=your_api_key_here -# Your Watson X project ID +# Your watsonx.ai project ID WATSONX_PROJECT_ID=your_project_id_here # Watson X service URL (default: us-south region) @@ -39,23 +49,39 @@ WATSONX_URL=https://us-south.ml.cloud.ibm.com # ==================================================================== # Model Configuration # ==================================================================== -# Granite models available on Watson X +# Use model ids that exist in your watsonx.ai instance and region. The +# values below are only the defaults hardcoded in rag_system/main.py. +# IBM's "Supported foundation models" documentation lists what is +# currently available. -# Main generation model for answering queries -# Options: -# - ibm/granite-13b-chat-v2 (recommended for chat) -# - ibm/granite-13b-instruct-v2 (for instructions) -# - ibm/granite-20b-multilingual (for multilingual) -# - ibm/granite-3b-code-instruct (for code) +# Answers and sub-answer composition. +# default: ibm/granite-13b-chat-v2 WATSONX_GENERATION_MODEL=ibm/granite-13b-chat-v2 -# Lightweight model for enrichment and routing -# Use a smaller model for better performance on simple tasks +# Routing, triage, query decomposition, contextual enrichment, document +# overviews and answer verification. These prompts expect JSON back, and +# the Watson X client cannot force JSON mode - pick a model that follows +# format instructions well. +# default: ibm/granite-8b-japanese WATSONX_ENRICHMENT_MODEL=ibm/granite-8b-japanese # ==================================================================== -# Optional: Ollama Configuration (fallback) +# Local models (used no matter which LLM_BACKEND is selected) +# ==================================================================== +# Embeddings run in-process from Hugging Face when the name contains a +# "/", and through Ollama otherwise. Changing this requires re-indexing. +# default: Qwen/Qwen3-Embedding-4B +EMBEDDING_MODEL=Qwen/Qwen3-Embedding-4B + +# Reranker, loaded locally through the `rerankers` library. +# default: BAAI/bge-reranker-v2-m3 +RERANKER_MODEL=BAAI/bge-reranker-v2-m3 + +# ==================================================================== +# Optional: Ollama Configuration # ==================================================================== -# These settings are used if LLM_BACKEND=ollama +# Still used when LLM_BACKEND=ollama, and always used by the backend +# gateway's direct-LLM fast path (backend/server.py) and by any Ollama +# embedding model. OLLAMA_HOST=http://localhost:11434 diff --git a/package-lock.json b/package-lock.json index 7eb5b4df..5f4ef42b 100644 --- a/package-lock.json +++ b/package-lock.json @@ -9,13 +9,10 @@ "version": "0.1.0", "dependencies": { "@radix-ui/react-avatar": "^1.1.10", - "@radix-ui/react-dropdown-menu": "^2.1.15", "@radix-ui/react-scroll-area": "^1.2.9", - "@radix-ui/react-separator": "^1.1.7", "@radix-ui/react-slot": "^1.2.3", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", - "framer-motion": "^12.16.0", "lucide-react": "^0.513.0", "next": "15.3.3", "react": "^19.0.0", @@ -238,44 +235,6 @@ "node": "^18.18.0 || ^20.9.0 || >=21.1.0" } }, - "node_modules/@floating-ui/core": { - "version": "1.7.1", - "resolved": "https://registry.npmjs.org/@floating-ui/core/-/core-1.7.1.tgz", - "integrity": "sha512-azI0DrjMMfIug/ExbBaeDVJXcY0a7EPvPjb2xAJPa4HeimBX+Z18HK8QQR3jb6356SnDDdxx+hinMLcJEDdOjw==", - "license": "MIT", - "dependencies": { - "@floating-ui/utils": "^0.2.9" - } - }, - "node_modules/@floating-ui/dom": { - "version": "1.7.1", - "resolved": "https://registry.npmjs.org/@floating-ui/dom/-/dom-1.7.1.tgz", - "integrity": "sha512-cwsmW/zyw5ltYTUeeYJ60CnQuPqmGwuGVhG9w0PRaRKkAyi38BT5CKrpIbb+jtahSwUl04cWzSx9ZOIxeS6RsQ==", - "license": "MIT", - "dependencies": { - "@floating-ui/core": "^1.7.1", - "@floating-ui/utils": "^0.2.9" - } - }, - "node_modules/@floating-ui/react-dom": { - "version": "2.1.3", - "resolved": "https://registry.npmjs.org/@floating-ui/react-dom/-/react-dom-2.1.3.tgz", - "integrity": "sha512-huMBfiU9UnQ2oBwIhgzyIiSpVgvlDstU8CX0AF+wS+KzmYMs0J2a3GwuFHV1Lz+jlrQGeC1fF+Nv0QoumyV0bA==", - "license": "MIT", - "dependencies": { - "@floating-ui/dom": "^1.0.0" - }, - "peerDependencies": { - "react": ">=16.8.0", - "react-dom": ">=16.8.0" - } - }, - "node_modules/@floating-ui/utils": { - "version": "0.2.9", - "resolved": "https://registry.npmjs.org/@floating-ui/utils/-/utils-0.2.9.tgz", - "integrity": "sha512-MDWhGtE+eHw5JW7lq4qhc5yRLS11ERl1c7Z6Xd0a58DozHES6EnNNwUWbMiG4J9Cgj053Bhk8zvlhFYKVhULwg==", - "license": "MIT" - }, "node_modules/@humanfs/core": { "version": "0.19.1", "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", @@ -1021,29 +980,6 @@ "integrity": "sha512-XnbHrrprsNqZKQhStrSwgRUQzoCI1glLzdw79xiZPoofhGICeZRSQ3dIxAKH1gb3OHfNf4d6f+vAv3kil2eggA==", "license": "MIT" }, - "node_modules/@radix-ui/react-arrow": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-arrow/-/react-arrow-1.1.7.tgz", - "integrity": "sha512-F+M1tLhO+mlQaOWspE8Wstg+z6PwxwRd8oQ8IXceWz92kfAmalTRf0EjrouQeo7QssEPfCn05B4Ihs1K9WQ/7w==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-primitive": "2.1.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-avatar": { "version": "1.1.10", "resolved": "https://registry.npmjs.org/@radix-ui/react-avatar/-/react-avatar-1.1.10.tgz", @@ -1071,32 +1007,6 @@ } } }, - "node_modules/@radix-ui/react-collection": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-collection/-/react-collection-1.1.7.tgz", - "integrity": "sha512-Fh9rGN0MoI4ZFUNyfFVNU4y9LUz93u9/0K+yLgA2bwRojxM8JU1DyvvMBabnZPBgMWREAJvU2jjVzq+LrFUglw==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-slot": "1.2.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-compose-refs": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/@radix-ui/react-compose-refs/-/react-compose-refs-1.1.2.tgz", @@ -1142,216 +1052,6 @@ } } }, - "node_modules/@radix-ui/react-dismissable-layer": { - "version": "1.1.10", - "resolved": "https://registry.npmjs.org/@radix-ui/react-dismissable-layer/-/react-dismissable-layer-1.1.10.tgz", - "integrity": "sha512-IM1zzRV4W3HtVgftdQiiOmA0AdJlCtMLe00FXaHwgt3rAnNsIyDqshvkIW3hj/iu5hu8ERP7KIYki6NkqDxAwQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "@radix-ui/react-use-escape-keydown": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-dropdown-menu": { - "version": "2.1.15", - "resolved": "https://registry.npmjs.org/@radix-ui/react-dropdown-menu/-/react-dropdown-menu-2.1.15.tgz", - "integrity": "sha512-mIBnOjgwo9AH3FyKaSWoSu/dYj6VdhJ7frEPiGTeXCdUFHjl9h3mFh2wwhEtINOmYXWhdpf1rY2minFsmaNgVQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-id": "1.1.1", - "@radix-ui/react-menu": "2.1.15", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-controllable-state": "1.2.2" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-focus-guards": { - "version": "1.1.2", - "resolved": "https://registry.npmjs.org/@radix-ui/react-focus-guards/-/react-focus-guards-1.1.2.tgz", - "integrity": "sha512-fyjAACV62oPV925xFCrH8DR5xWhg9KYtJT4s3u54jxp+L/hbpTY2kIeEFFbFe+a/HCE94zGQMZLIpVTPVZDhaA==", - "license": "MIT", - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-focus-scope": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-focus-scope/-/react-focus-scope-1.1.7.tgz", - "integrity": "sha512-t2ODlkXBQyn7jkl6TNaw/MtVEVvIGelJDCG41Okq/KwUsJBwQ4XVZsHAVUkK4mBv3ewiAS3PGuUWuY2BoK4ZUw==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-id": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-id/-/react-id-1.1.1.tgz", - "integrity": "sha512-kGkGegYIdQsOb4XjsfM97rXsiHaBwco+hFI66oO4s9LU+PLAC5oJ7khdOVFxkhsmlbpUqDAvXw11CluXP+jkHg==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-menu": { - "version": "2.1.15", - "resolved": "https://registry.npmjs.org/@radix-ui/react-menu/-/react-menu-2.1.15.tgz", - "integrity": "sha512-tVlmA3Vb9n8SZSd+YSbuFR66l87Wiy4du+YE+0hzKQEANA+7cWKH1WgqcEX4pXqxUFQKrWQGHdvEfw00TjFiew==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-collection": "1.1.7", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-direction": "1.1.1", - "@radix-ui/react-dismissable-layer": "1.1.10", - "@radix-ui/react-focus-guards": "1.1.2", - "@radix-ui/react-focus-scope": "1.1.7", - "@radix-ui/react-id": "1.1.1", - "@radix-ui/react-popper": "1.2.7", - "@radix-ui/react-portal": "1.1.9", - "@radix-ui/react-presence": "1.1.4", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-roving-focus": "1.1.10", - "@radix-ui/react-slot": "1.2.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "aria-hidden": "^1.2.4", - "react-remove-scroll": "^2.6.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-popper": { - "version": "1.2.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-popper/-/react-popper-1.2.7.tgz", - "integrity": "sha512-IUFAccz1JyKcf/RjB552PlWwxjeCJB8/4KxT7EhBHOJM+mN7LdW+B3kacJXILm32xawcMMjb2i0cIZpo+f9kiQ==", - "license": "MIT", - "dependencies": { - "@floating-ui/react-dom": "^2.0.0", - "@radix-ui/react-arrow": "1.1.7", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "@radix-ui/react-use-layout-effect": "1.1.1", - "@radix-ui/react-use-rect": "1.1.1", - "@radix-ui/react-use-size": "1.1.1", - "@radix-ui/rect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-portal": { - "version": "1.1.9", - "resolved": "https://registry.npmjs.org/@radix-ui/react-portal/-/react-portal-1.1.9.tgz", - "integrity": "sha512-bpIxvq03if6UNwXZ+HTK71JLh4APvnXntDc6XOX8UVq4XQOVl7lwok0AvIl+b8zgCw3fSaVTZMpAPPagXbKmHQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-presence": { "version": "1.1.4", "resolved": "https://registry.npmjs.org/@radix-ui/react-presence/-/react-presence-1.1.4.tgz", @@ -1399,37 +1099,6 @@ } } }, - "node_modules/@radix-ui/react-roving-focus": { - "version": "1.1.10", - "resolved": "https://registry.npmjs.org/@radix-ui/react-roving-focus/-/react-roving-focus-1.1.10.tgz", - "integrity": "sha512-dT9aOXUen9JSsxnMPv/0VqySQf5eDQ6LCk5Sw28kamz8wSOW2bJdlX2Bg5VUIIcV+6XlHpWTIuTPCf/UNIyq8Q==", - "license": "MIT", - "dependencies": { - "@radix-ui/primitive": "1.1.2", - "@radix-ui/react-collection": "1.1.7", - "@radix-ui/react-compose-refs": "1.1.2", - "@radix-ui/react-context": "1.1.2", - "@radix-ui/react-direction": "1.1.1", - "@radix-ui/react-id": "1.1.1", - "@radix-ui/react-primitive": "2.1.3", - "@radix-ui/react-use-callback-ref": "1.1.1", - "@radix-ui/react-use-controllable-state": "1.2.2" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-scroll-area": { "version": "1.2.9", "resolved": "https://registry.npmjs.org/@radix-ui/react-scroll-area/-/react-scroll-area-1.2.9.tgz", @@ -1461,29 +1130,6 @@ } } }, - "node_modules/@radix-ui/react-separator": { - "version": "1.1.7", - "resolved": "https://registry.npmjs.org/@radix-ui/react-separator/-/react-separator-1.1.7.tgz", - "integrity": "sha512-0HEb8R9E8A+jZjvmFCy/J4xhbXy3TV+9XSnGJ3KvTtjlIUy/YQ/p6UYZvi7YbeoeXdyU9+Y3scizK6hkY37baA==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-primitive": "2.1.3" - }, - "peerDependencies": { - "@types/react": "*", - "@types/react-dom": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc", - "react-dom": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - }, - "@types/react-dom": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-slot": { "version": "1.2.3", "resolved": "https://registry.npmjs.org/@radix-ui/react-slot/-/react-slot-1.2.3.tgz", @@ -1517,61 +1163,6 @@ } } }, - "node_modules/@radix-ui/react-use-controllable-state": { - "version": "1.2.2", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-controllable-state/-/react-use-controllable-state-1.2.2.tgz", - "integrity": "sha512-BjasUjixPFdS+NKkypcyyN5Pmg83Olst0+c6vGov0diwTEo6mgdqVR6hxcEgFuh4QrAs7Rc+9KuGJ9TVCj0Zzg==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-effect-event": "0.0.2", - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-use-effect-event": { - "version": "0.0.2", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-effect-event/-/react-use-effect-event-0.0.2.tgz", - "integrity": "sha512-Qp8WbZOBe+blgpuUT+lw2xheLP8q0oatc9UpmiemEICxGvFLYmHm9QowVZGHtJlGbS6A6yJ3iViad/2cVjnOiA==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-use-escape-keydown": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-escape-keydown/-/react-use-escape-keydown-1.1.1.tgz", - "integrity": "sha512-Il0+boE7w/XebUHyBjroE+DbByORGR9KKmITzbR7MyQ4akpORYP/ZmbhAr0DG7RmmBqoOnZdy2QlvajJ2QA59g==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-callback-ref": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, "node_modules/@radix-ui/react-use-is-hydrated": { "version": "0.1.0", "resolved": "https://registry.npmjs.org/@radix-ui/react-use-is-hydrated/-/react-use-is-hydrated-0.1.0.tgz", @@ -1605,48 +1196,6 @@ } } }, - "node_modules/@radix-ui/react-use-rect": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-rect/-/react-use-rect-1.1.1.tgz", - "integrity": "sha512-QTYuDesS0VtuHNNvMh+CjlKJ4LJickCMUAqjlE3+j8w+RlRpwyX3apEQKGFzbZGdo7XNG1tXa+bQqIE7HIXT2w==", - "license": "MIT", - "dependencies": { - "@radix-ui/rect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/react-use-size": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/react-use-size/-/react-use-size-1.1.1.tgz", - "integrity": "sha512-ewrXRDTAqAXlkl6t/fkXWNAhFX9I+CkKlw6zjEwk86RSPKwZr3xpBRso655aqYafwtnbpHLj6toFzmd6xdVptQ==", - "license": "MIT", - "dependencies": { - "@radix-ui/react-use-layout-effect": "1.1.1" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8 || ^17.0 || ^18.0 || ^19.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/@radix-ui/rect": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/@radix-ui/rect/-/rect-1.1.1.tgz", - "integrity": "sha512-HPwpGIzkl28mWyZqG52jiqDJ12waP11Pa1lGoiyUkIEuMLBP0oeK/C89esbXrxsky5we7dfd8U58nm0SgAWpVw==", - "license": "MIT" - }, "node_modules/@rtsao/scc": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/@rtsao/scc/-/scc-1.1.0.tgz", @@ -2657,18 +2206,6 @@ "dev": true, "license": "Python-2.0" }, - "node_modules/aria-hidden": { - "version": "1.2.6", - "resolved": "https://registry.npmjs.org/aria-hidden/-/aria-hidden-1.2.6.tgz", - "integrity": "sha512-ik3ZgC9dY/lYVVM++OISsaYDeg1tb0VtP5uL3ouh1koGOaUMDPpbFIei4JkFimWUFPn90sbMNMXQAIVOlnYKJA==", - "license": "MIT", - "dependencies": { - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - } - }, "node_modules/aria-query": { "version": "5.3.2", "resolved": "https://registry.npmjs.org/aria-query/-/aria-query-5.3.2.tgz", @@ -3364,12 +2901,6 @@ "node": ">=8" } }, - "node_modules/detect-node-es": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/detect-node-es/-/detect-node-es-1.1.0.tgz", - "integrity": "sha512-ypdmJU/TbBby2Dxibuv7ZLW3Bs1QEmM7nHjEANfohJLvE0XVujisn1qPJcZxg+qDucsr+bP6fLD1rPS3AhJ7EQ==", - "license": "MIT" - }, "node_modules/devlop": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/devlop/-/devlop-1.1.0.tgz", @@ -4205,33 +3736,6 @@ "url": "https://github.com/sponsors/ljharb" } }, - "node_modules/framer-motion": { - "version": "12.16.0", - "resolved": "https://registry.npmjs.org/framer-motion/-/framer-motion-12.16.0.tgz", - "integrity": "sha512-xryrmD4jSBQrS2IkMdcTmiS4aSKckbS7kLDCuhUn9110SQKG1w3zlq1RTqCblewg+ZYe+m3sdtzQA6cRwo5g8Q==", - "license": "MIT", - "dependencies": { - "motion-dom": "^12.16.0", - "motion-utils": "^12.12.1", - "tslib": "^2.4.0" - }, - "peerDependencies": { - "@emotion/is-prop-valid": "*", - "react": "^18.0.0 || ^19.0.0", - "react-dom": "^18.0.0 || ^19.0.0" - }, - "peerDependenciesMeta": { - "@emotion/is-prop-valid": { - "optional": true - }, - "react": { - "optional": true - }, - "react-dom": { - "optional": true - } - } - }, "node_modules/function-bind": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/function-bind/-/function-bind-1.1.2.tgz", @@ -4298,15 +3802,6 @@ "url": "https://github.com/sponsors/ljharb" } }, - "node_modules/get-nonce": { - "version": "1.0.1", - "resolved": "https://registry.npmjs.org/get-nonce/-/get-nonce-1.0.1.tgz", - "integrity": "sha512-FJhYRoDaiatfEkUK8HKlicmu/3SGFD51q3itKDGoSTysQJBnfOcxU5GxnhE1E6soB76MbT0MBtnKJuXyAx+96Q==", - "license": "MIT", - "engines": { - "node": ">=6" - } - }, "node_modules/get-proto": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/get-proto/-/get-proto-1.0.1.tgz", @@ -6499,21 +5994,6 @@ "url": "https://github.com/sponsors/isaacs" } }, - "node_modules/motion-dom": { - "version": "12.16.0", - "resolved": "https://registry.npmjs.org/motion-dom/-/motion-dom-12.16.0.tgz", - "integrity": "sha512-Z2nGwWrrdH4egLEtgYMCEN4V2qQt1qxlKy/uV7w691ztyA41Q5Rbn0KNGbsNVDZr9E8PD2IOQ3hSccRnB6xWzw==", - "license": "MIT", - "dependencies": { - "motion-utils": "^12.12.1" - } - }, - "node_modules/motion-utils": { - "version": "12.12.1", - "resolved": "https://registry.npmjs.org/motion-utils/-/motion-utils-12.12.1.tgz", - "integrity": "sha512-f9qiqUHm7hWSLlNW8gS9pisnsN7CRFRD58vNjptKdsqFLpkVnX00TNeD6Q0d27V9KzT7ySFyK1TZ/DShfVOv6w==", - "license": "MIT" - }, "node_modules/ms": { "version": "2.1.3", "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", @@ -7075,75 +6555,6 @@ "react": ">=18" } }, - "node_modules/react-remove-scroll": { - "version": "2.7.1", - "resolved": "https://registry.npmjs.org/react-remove-scroll/-/react-remove-scroll-2.7.1.tgz", - "integrity": "sha512-HpMh8+oahmIdOuS5aFKKY6Pyog+FNaZV/XyJOq7b4YFwsFHe5yYfdbIalI4k3vU2nSDql7YskmUseHsRrJqIPA==", - "license": "MIT", - "dependencies": { - "react-remove-scroll-bar": "^2.3.7", - "react-style-singleton": "^2.2.3", - "tslib": "^2.1.0", - "use-callback-ref": "^1.3.3", - "use-sidecar": "^1.1.3" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/react-remove-scroll-bar": { - "version": "2.3.8", - "resolved": "https://registry.npmjs.org/react-remove-scroll-bar/-/react-remove-scroll-bar-2.3.8.tgz", - "integrity": "sha512-9r+yi9+mgU33AKcj6IbT9oRCO78WriSj6t/cF8DWBZJ9aOGPOTEDvdUDz1FwKim7QXWwmHqtdHnRJfhAxEG46Q==", - "license": "MIT", - "dependencies": { - "react-style-singleton": "^2.2.2", - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/react-style-singleton": { - "version": "2.2.3", - "resolved": "https://registry.npmjs.org/react-style-singleton/-/react-style-singleton-2.2.3.tgz", - "integrity": "sha512-b6jSvxvVnyptAiLjbkWLE/lOnR4lfTtDAl+eUC7RZy+QQWc6wRzIV2CE6xBuMmDxc2qIihtDCZD5NPOFl7fRBQ==", - "license": "MIT", - "dependencies": { - "get-nonce": "^1.0.0", - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, "node_modules/reflect.getprototypeof": { "version": "1.0.10", "resolved": "https://registry.npmjs.org/reflect.getprototypeof/-/reflect.getprototypeof-1.0.10.tgz", @@ -8295,49 +7706,6 @@ "punycode": "^2.1.0" } }, - "node_modules/use-callback-ref": { - "version": "1.3.3", - "resolved": "https://registry.npmjs.org/use-callback-ref/-/use-callback-ref-1.3.3.tgz", - "integrity": "sha512-jQL3lRnocaFtu3V00JToYz/4QkNWswxijDaCVNZRiRTO3HQDLsdu1ZtmIUvV4yPp+rvWm5j0y0TG/S61cuijTg==", - "license": "MIT", - "dependencies": { - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, - "node_modules/use-sidecar": { - "version": "1.1.3", - "resolved": "https://registry.npmjs.org/use-sidecar/-/use-sidecar-1.1.3.tgz", - "integrity": "sha512-Fedw0aZvkhynoPYlA5WXrMCAMm+nSWdZt6lzJQ7Ok8S6Q+VsHmHpRWndVRJ8Be0ZbkfPc5LRYH+5XrzXcEeLRQ==", - "license": "MIT", - "dependencies": { - "detect-node-es": "^1.1.0", - "tslib": "^2.0.0" - }, - "engines": { - "node": ">=10" - }, - "peerDependencies": { - "@types/react": "*", - "react": "^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 || ^19.0.0-rc" - }, - "peerDependenciesMeta": { - "@types/react": { - "optional": true - } - } - }, "node_modules/use-sync-external-store": { "version": "1.5.0", "resolved": "https://registry.npmjs.org/use-sync-external-store/-/use-sync-external-store-1.5.0.tgz", diff --git a/package.json b/package.json index 30219fb5..edb5ec0b 100644 --- a/package.json +++ b/package.json @@ -10,13 +10,10 @@ }, "dependencies": { "@radix-ui/react-avatar": "^1.1.10", - "@radix-ui/react-dropdown-menu": "^2.1.15", "@radix-ui/react-scroll-area": "^1.2.9", - "@radix-ui/react-separator": "^1.1.7", "@radix-ui/react-slot": "^1.2.3", "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", - "framer-motion": "^12.16.0", "lucide-react": "^0.513.0", "next": "15.3.3", "react": "^19.0.0", diff --git a/public/.gitkeep b/public/.gitkeep new file mode 100644 index 00000000..e69de29b diff --git a/public/file.svg b/public/file.svg deleted file mode 100644 index 004145cd..00000000 --- a/public/file.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/public/globe.svg b/public/globe.svg deleted file mode 100644 index 567f17b0..00000000 --- a/public/globe.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/public/next.svg b/public/next.svg deleted file mode 100644 index 5174b28c..00000000 --- a/public/next.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/public/vercel.svg b/public/vercel.svg deleted file mode 100644 index 77053960..00000000 --- a/public/vercel.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/public/window.svg b/public/window.svg deleted file mode 100644 index b2b2a44f..00000000 --- a/public/window.svg +++ /dev/null @@ -1 +0,0 @@ - \ No newline at end of file diff --git a/rag_system/DOCUMENTATION.md b/rag_system/DOCUMENTATION.md index bf240d59..75b8e778 100644 --- a/rag_system/DOCUMENTATION.md +++ b/rag_system/DOCUMENTATION.md @@ -1,62 +1,460 @@ # RAG System Documentation -This document provides a detailed overview of the RAG (Retrieval-Augmented Generation) system, its architecture, and how to use it. +Reference for the `rag_system` package: pipelines, configuration keys, the HTTP API and +the CLI. For the shorter tour of the package, see [`README.md`](README.md). -## System Overview +Everything below is described as the code behaves today. Where a knob exists but nothing +reads it, that is called out explicitly rather than glossed over. -This RAG system is a sophisticated, multimodal question-answering system designed to work with a variety of documents. It can understand and process both the text and the visual layout of documents, and it uses a knowledge graph to understand the relationships between the entities in the documents. +## 1. Where this runs -The system is built around an agentic workflow that allows it to: +`rag_system` is one process in a four-process system: -* **Decompose complex questions** into smaller, more manageable sub-questions. -* **Triage queries** to determine if they can be answered directly or if they require retrieval from the knowledge base. -* **Verify answers** against the retrieved context to ensure they are accurate and supported by the documents. +``` +Next.js frontend :3000 ──► backend gateway :8000 ──► RAG API :8001 ──► Ollama :11434 + └────────────────── /chat/stream (SSE) ─────────────► +``` -## Architecture +The RAG API (`rag_system/api_server.py`) is a standard-library `http.server` running on a +plain `socketserver.TCPServer`, so **requests are handled one at a time**. A long chat or +indexing call blocks every other request to port 8001 until it finishes. Plan for a single +concurrent user per RAG API process. -The system is composed of two main pipelines: an indexing pipeline and a retrieval pipeline. +The API opens the same SQLite file as the backend gateway (`backend/chat_data.db`, or +`DB_PATH`), but only to read session→index links and to read/write index metadata. +**Chat message rows are written exclusively by `backend/server.py`.** The frontend's +streaming path posts straight to `:8001/chat/stream`; the stream itself persists nothing, +and the browser saves the completed turn through the gateway +(`POST :8000/sessions//messages/save`) once the `complete` event arrives. -### Indexing Pipeline +## 2. Indexing pipeline -The indexing pipeline is responsible for processing the documents and building the knowledge base. It performs the following steps: +`rag_system/pipelines/indexing_pipeline.py`. Public entry point: -1. **Text Extraction**: The pipeline uses `PyMuPDF` to extract the text from each page of the PDF documents, preserving the original layout. -2. **Text Embedding**: The extracted text is then passed to a text embedding model (`Qwen/Qwen3-Embedding-0.6B`) to create numerical vector representations of the text. -3. **Knowledge Graph Creation**: The text is also passed to a `GraphExtractor` that uses a large language model (`qwen2.5vl:7b`) to extract entities and their relationships. This information is then used to build a knowledge graph, which is stored as a `.gml` file. -4. **Indexing**: The text embeddings and the knowledge graph are then stored in a LanceDB database. +```python +IndexingPipeline(config, llm_client, ollama_config).run(file_paths: list[str]) +``` -### Retrieval Pipeline +`run()` also accepts the legacy keyword alias `documents=`. There is no +`process_documents()` method. -The retrieval pipeline is responsible for answering user queries. It uses an agentic workflow that includes the following steps: +### Steps -1. **Triage**: The agent first triages the user's query to determine if it can be answered directly or if it requires retrieval from the knowledge base. -2. **Query Decomposition**: If the query is complex, the agent uses a `QueryDecomposer` to break it down into smaller, more manageable sub-questions. -3. **Retrieval**: The agent then uses a `MultiVectorRetriever` and a `GraphRetriever` to retrieve relevant information from the knowledge base. -4. **Verification**: The retrieved context is then passed to a `Verifier` that uses an LLM to check if the context is sufficient to answer the query. -5. **Synthesis**: Finally, the agent uses an LLM to synthesize a final answer from the verified context. +1. **Conversion** — `ingestion/document_converter.py`. Docling converts the file to a + single Markdown string plus the `DoclingDocument` object. + - `.pdf`, `.docx`, `.html`, `.htm`, `.md` go through Docling; `.txt` is read + directly and wrapped in a fenced block. + - PDFs are first probed with PyMuPDF (`fitz`). If any page yields text, the no-OCR + converter is used; otherwise the OCR converter is used. + - OCR engine selection is dynamic: `OcrMacOptions` (macOS only), then `EasyOcrOptions`, + `RapidOcrOptions`, `TesseractOcrOptions`, `TesseractCliOcrOptions` — the first one + whose backend module or binary is actually installed wins. If none is available, + Docling's default OCR settings are used and a message is printed. + - The three converters (no-OCR, OCR, general) are built independently, so a failing + OCR engine does not disable the other paths. -## API Endpoints +2. **Chunking** — `chunker_mode` selects the chunker. + - `docling` (default): `ingestion/docling_chunker.py`. Walks the DoclingDocument + element tree in reading order, emits tables as atomic Markdown chunks, tracks the + heading path, and token-packs paragraphs up to `chunk_size` using the embedding + model's tokenizer. Each chunk records `heading_path`, `heading_level`, + `block_type` and (when available) `page`. If the tree walk fails it falls back to + splitting the exported Markdown with `overlap_sentences` sentences of overlap + (default 1). + - `legacy`: `ingestion/chunking.py::MarkdownRecursiveChunker`. Recursively splits on + `\n## `, `\n### `, `\n#### `, code fences and blank lines, measuring size with the + embedding model's tokenizer, with `max_chunk_size = chunk_size` and + `min_chunk_size = max(1, chunk_size // 4)`. + - If the Docling chunker fails to initialise, the pipeline falls back to the legacy + chunker automatically. + - Each chunk gets a sequential `metadata.chunk_index` within its document. -The system provides the following command-line endpoints: +3. **Document overview** — `indexing/overview_builder.py`, on unless + `overview.enabled` is `false`. The enrichment model summarises the first + `overview_first_n_chunks` (or `overview.max_chunks`, default 5) chunks into one + paragraph, appended as JSONL to `overview_path` + (default `index_store/overviews/overviews.jsonl`). These overviews are what the query + router reads later. Failures here are logged and do not abort indexing. -* `index`: This endpoint runs the indexing pipeline to process the documents and build the knowledge base. -* `chat`: This endpoint runs the retrieval pipeline to answer a user's query. -* `show_graph`: This endpoint displays the knowledge graph in a human-readable format and also provides a visual representation of the graph. +4. **Contextual enrichment** — `indexing/contextualizer.py`, when + `contextual_enricher.enabled`. For each chunk the enrichment model writes a 2–5 + sentence summary of the surrounding `window_size` chunks; the summary is prepended to + the chunk text (`"Context: …\n\n---\n\n"`) and the untouched text is stored + in `metadata.original_text`. The enriched text is what gets embedded. -### Usage +5. **Embedding and indexing** — `indexing/representations.py` + + `indexing/embedders.py`. Chunks are embedded in batches of + `indexing.embedding_batch_size` and written to LanceDB. The Arrow schema is + `vector` (fixed-size float32 list), `text`, `chunk_id`, `document_id`, `chunk_index`, + `metadata` (JSON string of the whole chunk). + - The vector width comes from the embeddings the model produced. Appending to a table + whose stored width differs raises + `Table '' stores N-dim vectors but the current embedding model produced M-dim + vectors` — re-index instead. + - Chunks whose vector contains NaN/Inf are skipped with a warning. + - After the append, a **native LanceDB FTS index** is created on `text` (index name + `text_idx`, `use_tantivy=False`) unless `text_idx` or the historical `fts_text` + already exists. There is no separate BM25 library or sidecar index. -To run the system, use the following commands: +6. **Late chunking** (optional) — `indexing/latechunk.py`. The whole document is fed to + the embedding model once, per-token hidden states are mean-pooled inside each chunk's + character span, and the resulting vectors are written to a sibling table: + `latechunk.lancedb_table_name` if set, otherwise + `f"{table_name}{latechunk.table_suffix}"` with `table_suffix` defaulting to `_lc`. + Documents whose vector count does not match their chunk count are skipped. -```bash -# Activate the virtual environment -source rag_system/rag_venv/bin/activate +There is no step 7. Knowledge-graph extraction (`indexing/graph_extractor.py`, the +`retrieval.graph.*` keys, the NetworkX `.gml` writer) was **removed on 2026-08-09** +(roadmap item 2.5): unreachable in every shipped profile, and evidence-negative — +GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, and it costs +41–57x at indexing and up to ~377x in query tokens +(`Documentation/research/academic-evidence-2026.md` §6). + +## 3. Retrieval: agent and pipeline + +### 3.1 Agent (`rag_system/agent/loop.py`) + +```python +Agent.run( + query, + table_name=None, session_id=None, + compose_sub_answers=None, query_decompose=None, ai_rerank=None, + context_expand=None, verify=None, + retrieval_k=None, context_window_size=None, reranker_top_k=None, + retrieval_mode=None, force_rag=False, event_callback=None, +) -> {"answer": str, "source_documents": list[dict]} +``` + +`run()` is a synchronous wrapper around `_run_async()`. Every `None` argument leaves the +profile value in place; anything else is written into the live retrieval-pipeline config. + +**Routing order.** + +1. If `force_rag` is true, triage is skipped and the query type is pinned to `rag_query`. +2. Otherwise the overview router runs first: the enrichment model is shown up to 40 + document overviews and returns `{"category": "direct_answer" | "rag_query"}`. If no + overviews are loaded it returns nothing and routing continues. +3. If the session already has history, the query is treated as a follow-up → + `rag_query`. +4. Otherwise an LLM triage prompt picks `rag_query` or `direct_answer`. + A JSON parse failure defaults to `rag_query`. + +`Agent._normalize_triage()` is applied to every verdict: anything that is not an explicit +`direct_answer` becomes `rag_query`. That is what catches a small utility model still +emitting the retired `graph_query` label (the graph module was removed on 2026-08-09). + +**Semantic cache.** For non-`direct_answer` queries the raw query is embedded and compared +against a `TTLCache(maxsize=100, ttl=300)` of previous results. A cosine similarity ≥ +`semantic_cache_threshold` (0.98) is a hit. With `cache_scope` at its default `session`, +entries from other sessions are skipped; setting it to `global` shares answers — including +document-derived ones — across sessions. + +**Query decomposition.** When enabled, `retrieval/query_transformer.py::QueryDecomposer` +splits the raw query (plus the last 5 turns for pronoun resolution) into at most +`query_decomposition.max_sub_queries` (default 10) sub-queries. What happens next depends +on `compose_from_sub_answers` (default `true`), and this changed at roadmap item 2.2 +(2026-08-09): + +* **`compose_from_sub_answers: true`** — one full `RetrievalPipeline.run()` per sub-query, + in parallel on up to 3 worker threads, then the generation model composes one answer from + the sub-answers. This is the only remaining path that fans the *first stage* out over + sub-queries, and it does so because it needs a separate answer per sub-question. +* **`compose_from_sub_answers: false`** — the first stage runs **once, on the full original + query**, and the sub-queries are applied at the **rerank** stage instead: every candidate + is scored against every sub-query and the scores combined with + `query_decomposition.rerank_aggregate` (`"mean"` default, or `"max"`). Decomposing the + first stage dilutes it semantically; the 2026 evidence puts the win at reranking. With + reranking off there is no rerank stage, so the sub-queries go unused. +* One sub-query after decomposition takes the direct path with the resolved query. + +**Verification.** When `verification.enabled` (or the per-request `verify` flag) is true +*and* the result has source documents, `agent/verifier.py` grades groundedness with the +enrichment model and appends ` [Confidence: N%]` to the answer string, plus +` [Warning: Low confidence. Groundedness: ]` when the answer is not grounded or the +score is under 50. A score of 0 (parse failure) appends nothing. There is no separate +`confidence` field in the response. + +Setting `verification.model` (or the `VERIFIER_MODEL` env var) to a HuggingFace model name +swaps the LLM prompt for a local NLI/verifier model scored per answer sentence against the +evidence (roadmap 2.4, off by default). `Documentation/verifier.md` has the availability +findings and the reasons `[Confidence: N%]` is UX rather than a calibrated measurement. + +**History.** The agent keeps an in-process `LRUCache(maxsize=100)` of per-session +`{query, answer}` turns. This is not persisted and is lost on restart. + +### 3.2 Retrieval pipeline (`rag_system/pipelines/retrieval_pipeline.py`) + +`run(query, table_name=None, window_size_override=None, event_callback=None)`: + +1. **Retrieve** — `retrieval/retrievers.py::MultiVectorRetriever.retrieve(text_query, + table_name, k, search_type)`: + - `hybrid` (default): the FTS leg and the vector leg each fetch `k` rows in parallel + and are fused with **reciprocal rank fusion** (`1/(60 + rank)` per leg). There are + no tunable leg weights. + - `vector_only` / `fts_only`: a single leg; the ordering is LanceDB's own. + - Single-word FTS queries are expanded to `word* OR word~` for prefix/fuzzy recall. + - Every returned document carries a finite, higher-is-better `score`: the BM25 score + for `fts_only`, `1/(1+distance)` for `vector_only`, the RRF score for `hybrid`. + `bm25` and `_distance` are only present when that leg actually matched. + - An unknown mode logs a warning and degrades to `hybrid`. +2. **Late-chunk leg** — if late chunking is enabled, the same query also runs against the + `_lc` table and those hits are appended to the candidate list. Then "late-chunk + merging" runs over every candidate: each chunk's text is replaced by itself plus its + ±1 neighbours from the main table, joined in `chunk_index` order. +2b. **Evidence-sufficiency retry** — if `retrieval.retry.enabled` (true in `default`, false + in `fast`) and the first pass found weak evidence, the query is reformulated once on the + enrichment model and steps 1–3 run again; the better of the two result sets is kept. The + signal is the *contrast* between the top candidate's cosine similarity and the + background of the rest, not the raw top similarity (which measured anti-correlated with + success), and it is only available on L2-normalized v4+ tables. See + `Documentation/retrieval_pipeline.md` §2b. +3. **Rerank** — if `reranker.enabled` and a reranker loaded. `strategy: "rerankers-lib"` + (the default) loads the model through the `rerankers` library with + `model_type` (default `cross-encoder`); any other strategy uses the local + `rerankers/reranker.py::CrossEncoderReranker`. Results are trimmed to + `reranker.top_k` (or `reranker.top_percent` of the candidate count). **If the model + cannot be loaded the pipeline prints a warning and skips reranking** rather than + failing. +4. **Context expansion** — when the effective window is > 0, each surviving chunk pulls + `chunk_index ± window_size` siblings from the same `document_id` via a LanceDB metadata + filter. Results are re-sorted by `rerank_score`, then `_distance`, then `score`. If any + chunk carries a `rerank_score`, non-reranked chunks are dropped. +5. **Sentence pruning** — when `provence.enabled`, `rerankers/sentence_pruner.py` runs + `naver/provence-reranker-debertav3-v1` over each chunk at `provence.threshold` + (default 0.1) and chunks pruned to empty are dropped. If the Provence weights cannot be + loaded, pruning is a no-op. +6. **Synthesis** — the generation model streams the final answer from the surviving + chunks. `_distance` and `vector` are stripped from the returned documents and + NaN/Inf numerics are nulled so the payload is JSON-safe. + +Returned source documents have the shape +`{chunk_id, text, score, document_id, chunk_index, metadata}` plus `bm25` and/or +`rerank_score` when those stages ran. `metadata` is the chunk record that was serialised +into the LanceDB `metadata` column at index time, with `document_id` and `chunk_index` +filled in from the row. + +## 4. Configuration reference + +Defined in `rag_system/main.py`. `factory.get_pipeline_config(mode)` hands out a deep copy, +so per-request overrides never mutate the master dictionaries. + +### 4.1 Model configuration + +| Object | Keys | Env overrides | +| --- | --- | --- | +| `LLM_BACKEND` | `ollama` (default) or `watsonx` | `LLM_BACKEND` | +| `OLLAMA_CONFIG` | `host`, `generation_model`, `enrichment_model` | `OLLAMA_HOST`, `GENERATION_MODEL`, `ENRICHMENT_MODEL` | +| `WATSONX_CONFIG` | `api_key`, `project_id`, `url`, `generation_model`, `enrichment_model` | `WATSONX_API_KEY`, `WATSONX_PROJECT_ID`, `WATSONX_URL`, `WATSONX_GENERATION_MODEL`, `WATSONX_ENRICHMENT_MODEL` | +| `EXTERNAL_MODELS` | `embedding_model`, `reranker_model` | `EMBEDDING_MODEL`, `RERANKER_MODEL` | + +### 4.2 Pipeline profile keys + +`PIPELINE_CONFIGS` contains exactly two profiles: `default` and `fast`. + +| Key | Read by | Notes | +| --- | --- | --- | +| `storage.lancedb_uri` | both pipelines | LanceDB directory (`./lancedb`). Also accepted as `storage.db_path` / `storage.lancedb_path`. | +| `storage.text_table_name` | both pipelines | Default table (`text_pages_v4`). | +| `retrieval.search_type` | `RetrievalPipeline._retrieval_mode` | `hybrid`, `vector_only`, `fts_only`. | +| `retrieval.dense.enabled` | both pipelines | `false` skips vector indexing and disables retrieval entirely — `MultiVectorRetriever` owns both the FTS and the vector leg, so no retriever is built at all. | +| `retrieval.dense.lancedb_table_name` | indexing | Fallback table name when `storage.text_table_name` is unset. | +| `retrieval.latechunk.enabled` | both pipelines | Also accepted under `retrievers.latechunk` / `retrieval.late_chunking`. | +| `retrieval.latechunk.table_suffix` / `.lancedb_table_name` | both pipelines | Late-chunk table name; suffix defaults to `_lc` on both sides. | +| `embedding_model_name` | both pipelines | Required — the retrieval pipeline raises if it is missing rather than guessing a dimension. | +| `reranker.enabled` | retrieval | | +| `reranker.strategy` | retrieval | `rerankers-lib` (default) or anything else → local `CrossEncoderReranker`. | +| `reranker.model_type` | retrieval | `rerankers` library model type, default `cross-encoder`. Use `colbert` for late-interaction models. | +| `reranker.model_name` | retrieval | Missing value logs a warning and skips reranking. | +| `reranker.top_k` / `reranker.top_percent` | retrieval | `top_percent` (0–1) wins when set. | +| `query_decomposition.enabled` | agent | | +| `query_decomposition.compose_from_sub_answers` | agent | Default `true`. | +| `query_decomposition.max_sub_queries` | agent | Default 10; not present in the shipped profiles. | +| `query_decomposition.rerank_aggregate` | retrieval | `mean` (default) or `max`; how per-sub-query rerank scores combine. Only read when reranking runs with sub-queries. | +| `retrieval.retry.enabled` / `.min_top_score` / `.max_attempts` | retrieval | Evidence-sufficiency retry. `true` / `0.12` / `1` in `default`; disabled in `fast`. Also accepted under `retrievers.retry`. | +| `retrieval.retry.min_rerank_score` | retrieval | Threshold used instead of `min_top_score` when the reranker produced a 0–1 probability. Defaults to `min_top_score`. | +| `verification.enabled` | agent | | +| `verification.model` | agent | HuggingFace model name for the local verifier; unset ⇒ the LLM-prompt verifier. Also settable as `VERIFIER_MODEL`. | +| `verification.threshold` | agent | Default `0.5`. Local verifier only. | +| `retrieval_k` | retrieval | Rows fetched per leg; the fused list is truncated to the same value. | +| `context_window_size` | retrieval | `0` disables context expansion (the value in both shipped profiles). | +| `semantic_cache_threshold` | agent | Cosine similarity for a cache hit (0.98). | +| `cache_scope` | agent | `session` (default) or `global`. | +| `contextual_enricher.enabled` / `.window_size` | indexing | | +| `enrich_model` / `enrichment_model_name` | indexing | Per-index override of `OLLAMA_CONFIG.enrichment_model`. | +| `overview.enabled` / `.model` / `.max_chunks` | indexing | Not present in the shipped profiles; defaults are enabled, enrichment model, 5 chunks. | +| `overview_model_name` / `overview_first_n_chunks` / `overview_path` | indexing | Top-level equivalents, set per request by the API. | +| `chunker_mode` | indexing | `docling` (default) or `legacy`. | +| `chunking.chunk_size` | indexing | Token budget. Also accepted as top-level `chunk_size` or `max_tokens`. **Default when unset: 1500**; the HTTP `/index` endpoint sends 512. | +| `overlap_sentences` | indexing | Docling chunker sentence overlap, default 1. | +| `indexing.embedding_batch_size` / `.enrichment_batch_size` | indexing | | +| `provence.enabled` / `.threshold` | retrieval | Not in the profiles; set per request by the API. Threshold default 0.1. | + +Keys that exist in the shipped profiles but that **nothing reads**: `description`, +`reranker.type`, and `indexing.enable_progress_tracking`. + +`retrieval.graph.*` and `graph_strategy.*` no longer exist — they are ignored if present +in a hand-written config (see the removal note in §2). + +There is no `dense.weight`/`denseWeight`, no `bm25_path` or other BM25 index path, no +`fallback_reranker`, no `vision_model_name`, and no `chunk_overlap` — the sparse leg is +LanceDB's native FTS, hybrid fusion is weight-free, and chunk overlap was never applied by +either chunker. -# Index the documents -python rag_system/main.py index +## 5. HTTP API -# Ask a question -python rag_system/main.py chat "Your question here" +`rag_system/api_server.py`, default port 8001. Start it with +`python -m rag_system.api_server` or `python -m rag_system.main api --port 8001`. The +profile it loads comes from `RAG_CONFIG_MODE` (default `default`). -# Show the knowledge graph -python rag_system/main.py show_graph +All responses are JSON with `Access-Control-Allow-Origin: *`. Errors are +`{"error": ""}` with a 4xx/5xx status. `OPTIONS` on any path returns the CORS +preflight headers (`GET, POST, OPTIONS`); any unmatched path returns +`{"error": "Not Found"}` with 404. + +**Key casing.** Every request body is normalised once at parse time: `camelCase` keys are +converted to `snake_case`, so `rerankerTopK` and `reranker_top_k` land in the same place. +An explicit `snake_case` value always wins over its camelCase twin. `overview_model` / +`overviewModel` are additionally aliased to `overview_model_name`. Unknown fields are +ignored. + +### `GET /health` + +```json +{"status": "ok"} +``` + +### `GET /models` + +```json +{"generation_models": ["..."], "embedding_models": ["..."]} +``` + +With `LLM_BACKEND=ollama` the list comes from `GET {OLLAMA_HOST}/api/tags` (5 s timeout); +tags whose name contains `embed`, `bge` or `embedding` are classified as embedding models +and the rest as generation models. With `LLM_BACKEND=watsonx` the generation list is the +two configured granite ids. `embedding_models` always also contains the currently +configured embedding model plus `microsoft/harrier-oss-v1-0.6b` and +`Qwen/Qwen3-Embedding-4B`, `-0.6B` and `-8B`. + +### `POST /chat` + +| Field | Type | Default | Behaviour | +| --- | --- | --- | --- | +| `query` | string | — | **Required**; 400 if missing. | +| `session_id` | string | – | Loads that session's linked index (table name, embedding model, overviews) and its in-memory history. | +| `table_name` | string | – | Explicit LanceDB table; otherwise resolved from `session_id`, otherwise the profile default. | +| `model` | string | – | Per-request generation model, applied for this request only and restored afterwards. Ignored with a warning when the id is not valid for the active backend (watsonx ids contain `/`, Ollama tags do not). | +| `retrieval_mode` (alias `search_type`) | string | profile | `hybrid`, `vector_only`, `fts_only`. **Any other value is a 400.** | +| `force_rag` | bool | `false` | Skips triage and forces the RAG path; all other toggles still apply. | +| `query_decompose` | bool | profile | | +| `compose_sub_answers` | bool | profile | | +| `ai_rerank` | bool | profile | Toggles `reranker.enabled`. | +| `context_expand` | bool | profile | `false` forces the expansion window to 0. | +| `verify` | bool | profile | | +| `retrieval_k` | int | `20` | | +| `context_window_size` | int | `1` | Note: the profiles ship `0`, so an HTTP request expands context by default while the CLI does not. | +| `reranker_top_k` | int | `10` | | +| `provence_prune` | bool | off | Enables Provence sentence pruning. | +| `provence_threshold` | float | `0.1` | | + +Response: + +```json +{ + "answer": "…", + "source_documents": [ + {"chunk_id": "…", "text": "…", "score": 0.031, "document_id": "…", + "chunk_index": 4, "metadata": {"…": "…"}, "rerank_score": 6.1} + ] +} +``` + +### `POST /chat/stream` + +Same request body as `/chat`. Responds with `Content-Type: text/event-stream` and one +`data: {"type": "", "data": {...}}\n\n` frame per event. + +Event types: `analyze`, `direct_answer`, `decomposition`, `retrieval_started`, +`retrieval_done`, `rerank_started`, `rerank_done`, `context_expand_started`, +`context_expand_done`, `prune_started`, `prune_done`, `token`, `sub_query_token`, +`sub_query_result`, `single_query_result`, `final_answer`, `complete`, `error`. +`complete` carries the same object `/chat` would have returned and is the last frame. + +### `POST /index` + +| Field | Type | Default | Behaviour | +| --- | --- | --- | --- | +| `file_paths` | string[] | — | **Required** list of absolute paths; 400 otherwise. | +| `session_id` | string | – | Resolves the target table and sets `overview_path` to `index_store/overviews/.jsonl`. | +| `table_name` | string | – | Explicit LanceDB table. | +| `embedding_model` | string | profile | Overrides `embedding_model_name` for this build. Also written into the index metadata when `session_id` is supplied, so later queries reuse the same embedder. | +| `enrich_model` | string | profile | Model used for contextual enrichment. | +| `overview_model_name` | string | profile | Model used for document overviews. | +| `enable_latechunk` | bool | `false` | **Note:** the HTTP default is off even though the `default` profile enables late chunking. | +| `enable_enrich` | bool | `true` | | +| `window_size` | int | `2` | Contextual-enrichment window. | +| `chunk_size` | int | `512` | Token budget per chunk. | +| `enable_docling_chunk` | bool | `false` | `true` pins `chunker_mode` to `docling`. Passing `false` leaves the pipeline default, which is *also* `docling` — it does not select the legacy chunker. | +| `retrieval_mode` (alias `search_type`) | string | profile | Validated against the same three values (400 otherwise) and recorded on the index config. The mode only changes behaviour at query time. | +| `batch_size_embed` | int | `50` | | +| `batch_size_enrich` | int | `25` | | + +Response: + +```json +{ + "message": "Indexing process for 3 file(s) completed successfully.", + "table_name": "text_pages_", + "latechunk": false, + "docling_chunk": true, + "indexing_config": { + "chunk_size": 512, "retrieval_mode": "hybrid", "window_size": 2, + "enable_enrich": true, "embedding_model": "microsoft/harrier-oss-v1-0.6b", + "enrich_model": null, "overview_model_name": null, + "batch_size_embed": 50, "batch_size_enrich": 25 + } +} +``` + +Indexing is synchronous: the response is sent after the pipeline finishes. There is no +per-file result list and no progress endpoint. + +## 6. Command line + +Always run as a module from the repository root: + +```bash +python -m rag_system.main index [--mode default|fast] +python -m rag_system.main chat "" [--mode default|fast] +python -m rag_system.main api [--port 8001] ``` + +`index` accepts a single file or walks a directory for `.pdf`, `.docx`, `.html`, `.htm`, +`.md` and `.txt`. `chat` prints the JSON result of one `Agent.run()` call. `api` is +equivalent to `python -m rag_system.api_server`. + +`python rag_system/main.py …` does not work (the package would not be importable). + +## 7. Storage layout + +| Path | Contents | +| --- | --- | +| `./lancedb/` | LanceDB tables. `text_pages_v4` is the profile default; the UI creates `text_pages_` per index, and late chunking adds `
_lc`. Override with `LANCEDB_PATH` or `storage.lancedb_uri`. | +| `./index_store/overviews/*.jsonl` | Per-document overviews, one JSON object (`doc_id`, `overview`) per line. `overviews.jsonl` is the global fallback; per-session/index files are named `.jsonl`. | +| `backend/chat_data.db` | SQLite: sessions, messages, indexes, documents and index metadata. Override with `DB_PATH`. | +| `./shared_uploads/` | Files uploaded through the web UI. | + +## 8. Operational notes + +- **Logging** — `rag_system/__init__.py` configures the root logger; set `RAG_LOG_LEVEL` + (`DEBUG`/`INFO`/`WARNING`/`ERROR`, default `INFO`). Much of the pipeline still prints + progress directly to stdout. +- **Hugging Face auth** — `HF_TOKEN` (or `HUGGINGFACE_HUB_TOKEN`) is picked up on import + and used to log in to the Hub. +- **Model loading** — the embedder is cached per model name in-process, and the reranker + and Provence loads are guarded by locks so parallel sub-queries do not load the same + weights twice. +- **Changing the embedding model requires re-indexing.** Vector width is part of the + LanceDB table schema and a mismatch is a hard error. +- **Thread safety** — `rerankers` backends are not thread-safe, so `.rank()` calls are + serialised behind a lock. The API server itself is single-threaded. diff --git a/rag_system/README.md b/rag_system/README.md index ac2e59b4..c3b6dfba 100644 --- a/rag_system/README.md +++ b/rag_system/README.md @@ -1,104 +1,285 @@ -# Multimodal RAG System +# RAG System -This document provides a detailed overview of the multimodal Retrieval-Augmented Generation (RAG) system implemented in this directory. The system is designed to process and understand information from PDF documents, combining both textual and visual data to answer complex queries. +This directory contains the retrieval-augmented generation engine behind localGPT: the +master configuration, the indexing and retrieval pipelines, the agent loop, and the HTTP +API that the backend gateway and the frontend talk to. + +For a deeper, key-by-key reference (config tables, HTTP request/response shapes, SSE +events) see [`DOCUMENTATION.md`](DOCUMENTATION.md) next to this file. ## 1. Overview -This RAG system is a sophisticated pipeline that leverages state-of-the-art open-source models to provide accurate, context-aware answers from a document corpus. Unlike traditional RAG systems that only process text, this implementation is fully multimodal. It extracts and indexes both text and images from PDFs, allowing a Vision Language Model (VLM) to reason over both modalities when generating a final answer. +The system is **text-only**. Documents are converted to Markdown with +[Docling](https://github.com/docling-project/docling), chunked, optionally enriched with +LLM-generated context, embedded with a local Hugging Face model, and written to +**LanceDB**. Queries are answered by combining LanceDB's native full-text search with +vector search, reranking the merged candidates, and synthesising an answer with a local +Ollama model. + +Core capabilities: + +- **Docling ingestion** — PDF, DOCX, HTML/HTM and MD go through Docling; TXT is read + directly and wrapped in a fenced block. Scanned PDFs (no text layer) take an OCR + pipeline when an OCR backend is installed. +- **Hybrid retrieval** — LanceDB full-text (BM25-scored, native to LanceDB) and vector + search run in parallel and are fused with reciprocal rank fusion. `vector_only` and + `fts_only` run a single leg. +- **Cross-encoder reranking** — the merged candidates are reordered by a reranker loaded + through the [`rerankers`](https://github.com/AnswerDotAI/rerankers) library. +- **Agentic query handling** — triage (documents vs. general knowledge), optional query + decomposition with parallel sub-query retrieval, optional answer verification. +- **Late chunking** — an optional second embedding pass that encodes the whole document + and mean-pools per-chunk spans, stored in a sibling `
_lc` table. -The core capabilities include: -- **Multimodal Indexing**: Extracts text and images from PDFs and creates separate vector embeddings for each. -- **Hybrid Retrieval**: Combines dense vector search (for semantic similarity) with traditional keyword-based search (BM25) for robust retrieval. -- **Advanced Reranking**: Utilizes a powerful reranker model to improve the relevance of retrieved documents before they are passed to the generator. -- **VLM-Powered Synthesis**: Employs a Vision Language Model to synthesize the final answer, allowing it to analyze both the text and the images from the retrieved document chunks. +There is **no multimodal path**. Nothing produces image embeddings and no +vision-language model is invoked; PDF understanding is Docling's layout parsing plus +OCR. If you want page-image understanding you have to build it; GLM-OCR or Qwen3-VL +would be reasonable starting points, but neither is integrated today. ## 2. Architecture -The system is composed of several key Python modules that work together to form the RAG pipeline. - -### Key Modules: - -- `main.py`: The main entry point for the application. It contains the configuration for all models and pipelines and orchestrates the indexing and retrieval processes. -- `rag_system/pipelines/`: Contains the high-level orchestration for indexing and retrieval. - - `indexing_pipeline.py`: Manages the process of converting raw PDFs into indexed, searchable data. - - `retrieval_pipeline.py`: Handles the end-to-end process of taking a user query, retrieving relevant information, and generating a final answer. -- `rag_system/indexing/`: Contains all modules related to data processing and indexing. - - `multimodal.py`: Responsible for extracting text and images from PDFs and generating embeddings using the configured vision model (`colqwen2-v1.0`). - - `representations.py`: Defines the text embedding model (`Qwen2-7B-instruct`) and other data representation generators. - - `embedders.py`: Manages the connection to the **LanceDB** vector database and handles the indexing of vector embeddings. -- `rag_system/retrieval/`: Contains modules for retrieving and ranking documents. - - `retrievers.py`: Implements the logic for searching the vector database to find relevant text and image chunks. - - `reranker.py`: Contains the `QwenReranker` class, which re-ranks the retrieved documents for improved relevance. -- `rag_system/agent/`: Contains the `Agent` loop that interacts with the user and the RAG pipelines. -- `rag_system/utils/`: Contains utility clients, such as the `OllamaClient` for interacting with the Ollama server. - -### Data Flow: - -1. **Indexing**: - - The `MultimodalProcessor` reads a PDF and splits it into pages. - - For each page, it extracts the raw text and a full-page image. - - The `QwenEmbedder` generates a vector embedding for the text. - - The `LocalVisionModel` (using `colqwen2-v1.0`) generates a vector embedding for the image. - - The `VectorIndexer` stores these embeddings in separate tables within a **LanceDB** database. -2. **Retrieval**: - - A user submits a query to the `Agent`. - - The `RetrievalPipeline`'s `MultiVectorRetriever` searches both the text and image tables in LanceDB for relevant chunks. - - The retrieved documents are passed to the `QwenReranker`, which re-orders them based on relevance to the query. - - The top-ranked documents (containing both text and image references) are passed to the Vision Language Model (`qwen-vl`). - - The VLM analyzes the text and images to extract key facts. - - A final text generation model (`llama3`) synthesizes these facts into a coherent, human-readable answer. +### Key modules + +- `main.py` — the MASTER configuration (`OLLAMA_CONFIG`, `WATSONX_CONFIG`, + `EXTERNAL_MODELS`, `PIPELINE_CONFIGS`) plus a thin argparse CLI. It builds nothing + itself; the CLI delegates to the factory. +- `factory.py` — the single factory: `get_agent(mode)`, `get_indexing_pipeline(mode)` + and `get_pipeline_config(mode)` (which returns a deep copy so per-request overrides + cannot mutate the master config). +- `api_server.py` — the HTTP API on port 8001 (`/health`, `/models`, `/chat`, + `/chat/stream`, `/index`), built on the standard library's `http.server`. +- `ingestion/` — everything that turns a file into chunks. + - `document_converter.py`: Docling converters (no-OCR, OCR, general) and OCR-engine + selection. + - `docling_chunker.py`: token-budgeted chunker that walks the DoclingDocument tree, + keeping tables atomic and recording the heading path on every chunk. + - `chunking.py`: `MarkdownRecursiveChunker`, the heading/paragraph recursive + splitter used as the `legacy` chunker and as the Docling chunker's fallback. +- `indexing/` — everything that turns chunks into an index. + - `representations.py`: `QwenEmbedder` (local Hugging Face), `OllamaEmbedder`, + `EmbeddingGenerator` and the `select_embedder()` dispatcher. + - `embedders.py`: `LanceDBManager` and `VectorIndexer` (schema, incremental append, + vector-width guard). + - `contextualizer.py`: `ContextualEnricher`, which prepends an LLM summary of the + surrounding window to each chunk before embedding. + - `latechunk.py`: `LateChunkEncoder`. + - `overview_builder.py`: per-document overviews used by the triage router. +- `retrieval/` — `retrievers.py` (`MultiVectorRetriever`) and + `query_transformer.py` (`QueryDecomposer`). +- `rerankers/` — `reranker.py` (`CrossEncoderReranker`, the non-`rerankers`-library + fallback) and `sentence_pruner.py` (Provence sentence-level pruning). +- `pipelines/` — `indexing_pipeline.py` and `retrieval_pipeline.py`. +- `agent/` — `loop.py` (`Agent`: triage, decomposition, orchestration, semantic cache, + per-session history) and `verifier.py` (`Verifier`). +- `utils/` — `ollama_client.py`, `watsonx_client.py`, `batch_processor.py`, + `logging_utils.py`. + +### Indexing data flow + +1. `DocumentConverter.convert_to_markdown()` converts the file to Markdown. PDFs are + probed with PyMuPDF for an existing text layer; only text-layer-less PDFs take the + OCR converter. +2. `DoclingChunker` (default) or `MarkdownRecursiveChunker` splits the document into + chunks with a token budget of `chunking.chunk_size`. +3. `OverviewBuilder` writes a one-paragraph overview per document to + `index_store/overviews/*.jsonl` (used later for query routing). +4. If `contextual_enricher.enabled`, `ContextualEnricher` asks the enrichment model for a + short summary of each chunk's neighbourhood and prepends it to the chunk text. The + untouched text is kept in `metadata.original_text`. +5. `EmbeddingGenerator` embeds the chunks in batches and `VectorIndexer` writes them to + the LanceDB table, then creates the native FTS index on the `text` column. +6. If late chunking is enabled, `LateChunkEncoder` re-encodes each document as a whole and + writes per-chunk vectors to `
_lc`. + +There is no knowledge-graph step: `graph_extractor.py`, `GraphRetriever`, +`GraphQueryTranslator` and the `retrieval.graph` / `graph_strategy` config blocks were +removed on 2026-08-09 (roadmap item 2.5). The path was unreachable, and the evidence is +against it — GraphRAG loses on single-hop retrieval, its multi-hop gains are contested, +and it costs 41–57x at indexing and up to ~377x in query tokens +(`Documentation/research/academic-evidence-2026.md` §6). + +### Retrieval data flow + +1. `Agent.run()` routes the query: the overview router first, then a + "history exists → treat as follow-up" short circuit, then an LLM triage fallback that + picks `rag_query` or `direct_answer`. `force_rag=True` skips triage + entirely and pins `rag_query`. +2. Non-`direct_answer` queries are checked against the in-process semantic cache + (cosine similarity ≥ `semantic_cache_threshold`, scoped per session by default). +3. If query decomposition is on, `QueryDecomposer` splits the query. With + `compose_from_sub_answers` (the profile default) each sub-query gets its own full + retrieval in parallel, up to 3 workers; otherwise the first stage runs **once on the + full query** and the sub-queries are applied at the rerank stage instead (roadmap 2.2). +4. `RetrievalPipeline.run()` retrieves via `MultiVectorRetriever.retrieve()` in the + configured mode, optionally also querying the late-chunk table and merging neighbours. +4b. If `retrieval.retry.enabled` and the first pass found weak evidence — measured as the + *contrast* between the top candidate and the background of the rest, not the raw top + similarity — the query is reformulated once on the enrichment model and steps 4-5 run + again, keeping whichever result set scores better (roadmap 2.1). +5. The reranker (if enabled) reorders the candidates and keeps `reranker.top_k`. When + sub-queries are present each candidate is scored against all of them and the scores + combined with `query_decomposition.rerank_aggregate`. +6. Context expansion pulls ±`context_window_size` neighbouring chunks from LanceDB. +7. Provence pruning (off unless requested) drops irrelevant sentences from each chunk. +8. The generation model synthesises the answer from the surviving chunks, streaming + tokens to the caller when an event callback is supplied. +9. If verification is enabled and there are source documents, `Verifier` grades the + answer and appends ` [Confidence: N%]` (plus a low-confidence warning) to the answer + string. ## 3. Models -This system relies on a suite of powerful, open-source models. +The defaults live in `main.py` and every one of them except the Provence pruner is +overridable with an environment variable. -| Component | Model | Framework | Purpose | -| --------------------- | ----------------------------------- | -------------- | ------------------------------------------- | -| **Image Embedding** | `vidore/colqwen2-v1.0` | `colpali` | Generates vector embeddings from images. | -| **Text Embedding** | `Qwen/Qwen2-7B-instruct` | `transformers` | Generates vector embeddings from text. | -| **Reranker** | `Qwen/Qwen-reranker` | `transformers` | Re-ranks retrieved documents for relevance. | -| **Vision Language Model** | `qwen2.5vl:7b` | `Ollama` | Extracts facts from text and images. | -| **Text Generation** | `llama3` | `Ollama` | Synthesizes the final answer. | +| Role | Default | Runtime | Env override | +| ---- | ------- | ------- | ------------ | +| **Generation** (answers, sub-answer composition) | `qwen3.5:9b` | Ollama | `GENERATION_MODEL` | +| **Enrichment / utility** (routing, triage, decomposition, contextual enrichment, overviews, verification) | `qwen3.5:4b` | Ollama | `ENRICHMENT_MODEL` | +| **Embedding** | `microsoft/harrier-oss-v1-0.6b` (MIT, 1024-dim) | `transformers`, in-process | `EMBEDDING_MODEL` | +| **Reranker** (off by default) | `Qwen/Qwen3-Reranker-4B` | own yes/no-logit scorer, loaded lazily | `RERANKER_MODEL` | +| **Sentence pruning** (optional) | `naver/provence-reranker-debertav3-v1` | `transformers`, in-process | — (hardcoded in `rerankers/sentence_pruner.py`) | + +Documented alternatives: + +- Generation: `qwen3.6:27b` (high-end, ~17 GB) or `qwen3.5:4b` (light). +- Enrichment: `qwen3.5:2b` (light). +- Embedding: `Qwen/Qwen3-Embedding-4B` (2560-dim, 32K context; for multilingual or + long-context corpora), `Qwen/Qwen3-Embedding-0.6B` (1024-dim). +- Reranker: `BAAI/bge-reranker-v2-m3` (low latency; only pays off with a weaker + embedder than the default — see [`../eval/DECISIONS.md`](../eval/DECISIONS.md)), + `answerdotai/answerai-colbert-small-v1` (late interaction — also set + `reranker.model_type` to `colbert`), `Qwen/Qwen3-Reranker-0.6B`. + +Notes: + +- **Embedding dimensions are never hardcoded.** `VectorIndexer.index()` reads the width + from the vectors the loaded model actually produced and builds the LanceDB schema from + it. Appending vectors of a different width to an existing table raises an explicit + error — **changing the embedding model requires re-indexing every existing index.** +- **The width check is not enough on its own**, because two different models can share + it (harrier-oss-v1-0.6b and Qwen3-Embedding-0.6B are both 1024-dim). Every table + therefore records the embedding model that wrote it and whether its vectors are + L2-normalized; indexing into or querying a table with a different embedder raises + `EmbedderMismatchError` and names the model to rebuild with. +- **Vectors are L2-normalized at write and query time**, so LanceDB's default L2 + ordering is the cosine ordering both model cards specify. Tables written before this + existed carry no marker: they keep working with unnormalized vectors and log a + warning recommending a rebuild. +- `QwenEmbedder` truncates inputs at `min(tokenizer.model_max_length, 8192)` tokens. +- `select_embedder()` treats a name containing `/` as a Hugging Face repo id and anything + else as an Ollama tag served through `/api/embeddings`. +- If the reranker cannot be loaded, the pipeline logs a warning and continues **without** + reranking rather than failing the query. ## 4. Configuration -All system configurations are centralized in `main.py`. +All configuration lives in `main.py`. -- **`OLLAMA_CONFIG`**: Defines the models that will be run via the Ollama server. This includes the final text generation model and the Vision Language Model. -- **`PIPELINE_CONFIGS`**: Contains the configurations for both the `indexing` and `retrieval` pipelines. Here you can specify: - - The paths for the LanceDB database and source documents. - - The names of the tables to be used for text and image embeddings. - - The Hugging Face model names for the text embedder, vision model, and reranker. - - Parameters for the reranker and retrieval process (e.g., `top_k`, `retrieval_k`). +- **`LLM_BACKEND`** — `ollama` (default) or `watsonx`. See + [`../WATSONX_README.md`](../WATSONX_README.md). +- **`OLLAMA_CONFIG`** — `host`, `generation_model`, `enrichment_model`. The generation + model writes user-facing answers; the enrichment model does all the utility work + (routing, triage, decomposition, contextual enrichment, document overviews, + verification). +- **`WATSONX_CONFIG`** — credentials, URL and the two granite model ids. It has the same + `generation_model` / `enrichment_model` keys so it is a drop-in replacement. +- **`EXTERNAL_MODELS`** — `embedding_model` and `reranker_model`, the two Hugging Face + models loaded in-process. +- **`PIPELINE_CONFIGS`** — exactly two profiles, `default` and `fast`. Selected with + `--mode` on the CLI or the `RAG_CONFIG_MODE` environment variable for the API server. -To change a model, simply update the corresponding model name in this configuration file. +| | `default` | `fast` | +| --- | --- | --- | +| `retrieval.search_type` | `hybrid` | `vector_only` | +| `retrieval.latechunk.enabled` | `true` | `false` | +| `reranker.enabled` | `false` (top 10 when switched on) | `false` | +| `query_decomposition.enabled` | `true` | `false` | +| `verification.enabled` | `true` | `false` | +| `contextual_enricher.enabled` | `true` (window 1) | `false` | +| `retrieval_k` | 20 | 10 | +| `indexing` batch sizes | 50 embed / 10 enrich | 100 embed / 50 enrich | -## 5. Usage +Both profiles store vectors under `./lancedb` in the table `text_pages_v4`, use +`semantic_cache_threshold: 0.98` and `cache_scope: "session"`. -To run the system, you first need to ensure the required models are available. +A per-key table (including which keys the API can override per request) is in +[`DOCUMENTATION.md`](DOCUMENTATION.md#4-configuration-reference). -### Prerequisites: +## 5. Usage + +### Prerequisites -1. **Install Dependencies**: +1. **Python 3.10+** (3.11 recommended) and dependencies, installed from the repository + root: ```bash pip install -r requirements.txt ``` -2. **Download Ollama Models**: + The root `requirements.txt` is the complete local-run set. `rag_system/requirements.txt` + is a partial list of the heavy ML dependencies that additionally pins + `ibm-watsonx-ai` and the macOS-only `ocrmac`. The Docker images install + `requirements-docker.txt` instead. + +2. **Ollama models**: ```bash - ollama pull llama3 - ollama pull qwen2.5vl:7b + ollama pull qwen3.5:9b + ollama pull qwen3.5:4b ``` -3. **Hugging Face Models**: The `transformers` and `colpali` libraries will automatically download the required models the first time they are used. Ensure you have a stable internet connection. -### Running the System: +3. **Hugging Face models** — the embedder, reranker and (if used) Provence weights are + downloaded on first use by `transformers`. Set `HF_TOKEN` if you point at a gated + repository. -1. **Execute the Main Script**: - ```bash - python rag_system/main.py - ``` -2. **Indexing**: The script will first run the indexing pipeline, processing any documents in the `rag_system/documents` directory and storing their embeddings in LanceDB. -3. **Querying**: Once indexing is complete, the RAG agent will be ready. You can ask questions about the documents you have indexed. - ``` - > What was the revenue growth in Q3? - ``` -4. **Exit**: To stop the agent, type `quit`. +### Command line + +Run as a module from the repository root — `python rag_system/main.py` will not work +because the package needs to be importable: + +```bash +# Index one file or a whole directory (walks *.pdf, *.docx, *.html, *.htm, *.md, *.txt) +python -m rag_system.main index ./shared_uploads --mode default + +# Ask a single question and print the JSON result +python -m rag_system.main chat "What was the revenue growth in Q3?" + +# Start the HTTP API (equivalent to python -m rag_system.api_server) +python -m rag_system.main api --port 8001 +``` + +`--mode` accepts `default` or `fast`. There is no interactive REPL. + +### Programmatic + +```python +from rag_system.factory import get_agent, get_indexing_pipeline + +get_indexing_pipeline("default").run(["/abs/path/to/document.pdf"]) + +agent = get_agent("default") +result = agent.run("What does the contract say about termination?") +print(result["answer"]) +print(len(result["source_documents"])) +``` + +`Agent.run()` returns `{"answer": str, "source_documents": list[dict]}` — there is no +top-level confidence field; the verifier appends its score to the answer string. + +### HTTP + +```bash +python -m rag_system.api_server # port 8001 +curl http://localhost:8001/health # {"status": "ok"} +curl -X POST http://localhost:8001/chat \ + -H 'Content-Type: application/json' \ + -d '{"query": "What is in these documents?"}' +``` + +Endpoints, accepted fields and defaults are documented in +[`DOCUMENTATION.md`](DOCUMENTATION.md#5-http-api). + +## 6. Where this fits + +The RAG API is one of four processes. The frontend (`:3000`) talks to the backend gateway +(`:8000`), which forwards chat and indexing to this API (`:8001`), which in turn calls +Ollama (`:11434`). The frontend also streams directly from `:8001/chat/stream`. See the +repository [`README.md`](../README.md) and `Documentation/` for the system-level view. diff --git a/rag_system/agent/loop.py b/rag_system/agent/loop.py index b7da704d..e2717e2a 100644 --- a/rag_system/agent/loop.py +++ b/rag_system/agent/loop.py @@ -7,8 +7,7 @@ from rag_system.utils.ollama_client import OllamaClient from rag_system.pipelines.retrieval_pipeline import RetrievalPipeline from rag_system.agent.verifier import Verifier -from rag_system.retrieval.query_transformer import QueryDecomposer, GraphQueryTranslator -from rag_system.retrieval.retrievers import GraphRetriever +from rag_system.retrieval.query_transformer import QueryDecomposer class Agent: """ @@ -18,40 +17,46 @@ def __init__(self, pipeline_configs: Dict[str, Dict], llm_client: OllamaClient, self.pipeline_configs = pipeline_configs self.llm_client = llm_client self.ollama_config = ollama_config - - gen_model = self.ollama_config["generation_model"] - + + # Utility work (routing, triage, decomposition, verification) runs on the + # small enrichment model; only user-facing answers use the generation model. + utility_model = self._utility_model() + # Initialize the single, persistent retrieval pipeline for this agent self.retrieval_pipeline = RetrievalPipeline(pipeline_configs, self.llm_client, self.ollama_config) - - self.verifier = Verifier(llm_client, gen_model) - self.query_decomposer = QueryDecomposer(llm_client, gen_model) - + + # `verification.model` (or the VERIFIER_MODEL env var, read inside + # Verifier) swaps the LLM-prompt verifier for a local NLI/verifier model. + # Unset by default — the LLM prompt is what ships (roadmap 2.4). + verification_config = self.pipeline_configs.get("verification", {}) or {} + self.verifier = Verifier( + llm_client, + utility_model, + model_name=verification_config.get("model"), + threshold=float(verification_config.get("threshold", 0.5)), + ) + self.query_decomposer = QueryDecomposer(llm_client, utility_model) + # 🚀 OPTIMIZED: TTL cache now stores embeddings for semantic matching - self._cache_max_size = 100 # fallback size limit for manual eviction helper - self._query_cache: TTLCache = TTLCache(maxsize=self._cache_max_size, ttl=300) + self._query_cache: TTLCache = TTLCache(maxsize=100, ttl=300) self.semantic_cache_threshold = self.pipeline_configs.get("semantic_cache_threshold", 0.98) - # If set to "session", semantic-cache hits will be restricted to the same chat session. - # Otherwise (default "global") answers can be reused across sessions. - self.cache_scope = self.pipeline_configs.get("cache_scope", "global") # 'global' or 'session' - + # If set to "global", semantic-cache hits are reused across chat sessions. + # The default keeps answers (and therefore document content) inside one session. + self.cache_scope = self.pipeline_configs.get("cache_scope", "session") # 'global' or 'session' + # 🚀 NEW: In-memory store for conversational history per session self.chat_histories: LRUCache = LRUCache(maxsize=100) # Stores history for 100 recent sessions - graph_config = self.pipeline_configs.get("graph_strategy", {}) - if graph_config.get("enabled"): - self.graph_query_translator = GraphQueryTranslator(llm_client, gen_model) - self.graph_retriever = GraphRetriever(graph_config["graph_path"]) - print("Agent initialized with live GraphRAG capabilities.") - else: - print("Agent initialized (GraphRAG disabled).") - # ---- Load document overviews for fast routing ---- self._global_overview_path = os.path.join("index_store", "overviews", "overviews.jsonl") self.doc_overviews: list[str] = [] self._current_overview_session: str | None = None # cache key to avoid rereading on every query self._load_overviews(self._global_overview_path) + def _utility_model(self) -> str: + """Model used for routing, triage, decomposition and verification.""" + return self.ollama_config.get("enrichment_model") or self.ollama_config["generation_model"] + def _load_overviews(self, path: str): """Helper to load overviews from a .jsonl file into self.doc_overviews.""" import json, os @@ -133,9 +138,8 @@ def _find_in_semantic_cache(self, query_embedding: np.ndarray, session_id: Optio continue # Respect cache scoping: if scope is session-level, skip results from other sessions - if self.cache_scope == "session" and session_id is not None: - if cached_item.get("session_id") != session_id: - continue + if self.cache_scope != "global" and cached_item.get("session_id") != session_id: + continue try: similarity = self._cosine_similarity(query_embedding, cached_embedding) @@ -168,14 +172,26 @@ def _format_query_with_history(self, query: str, history: list) -> str: return prompt # ---------------- Asynchronous triage using Ollama ---------------- + @staticmethod + def _normalize_triage(decision: str) -> str: + """Collapse a triage label to one of the two categories that still exist. + + ``graph_query`` was a third outcome until the graph module was removed + (roadmap 2.5, 2026-08-09). Small utility models still emit it from time to + time — they have seen the label in their training data — so anything that + is not an explicit ``direct_answer`` is answered from the documents. + """ + return "direct_answer" if (decision or "").strip() == "direct_answer" else "rag_query" + async def _triage_query_async(self, query: str, history: list) -> str: - + print(f"🔍 ROUTING DEBUG: Starting triage for query: '{query[:100]}...'") # 1️⃣ Fast routing using precomputed overviews (if available) print(f"📖 ROUTING DEBUG: Attempting overview-based routing...") routed = self._route_via_overviews(query) if routed: + routed = self._normalize_triage(routed) print(f"✅ ROUTING DEBUG: Overview routing decided: '{routed}'") return routed else: @@ -198,8 +214,6 @@ async def _triage_query_async(self, query: str, history: list) -> str: 2. "direct_answer" – General knowledge questions, greetings, or queries unrelated to uploaded documents. Examples: "Who are the CEOs of Tesla and Amazon?", "What is the capital of France?", "Hello", "Explain quantum physics" -3. "graph_query" – Specific factual relations for knowledge-graph lookup (currently limited use) - IMPORTANT: For general world knowledge about well-known companies, people, or facts NOT related to uploaded documents, choose "direct_answer". User query: "{query}" @@ -207,57 +221,28 @@ async def _triage_query_async(self, query: str, history: list) -> str: Respond with JSON: {{"category": ""}} """ resp = self.llm_client.generate_completion( - model=self.ollama_config["generation_model"], prompt=prompt, format="json" + model=self._utility_model(), prompt=prompt, format="json" ) try: data = json.loads(resp.get("response", "{}")) - decision = data.get("category", "rag_query") + decision = self._normalize_triage(data.get("category", "rag_query")) print(f"🤖 ROUTING DEBUG: LLM fallback triage decided: '{decision}'") return decision except json.JSONDecodeError: print(f"❌ ROUTING DEBUG: LLM fallback triage JSON parsing failed, defaulting to 'rag_query'") return "rag_query" - def _run_graph_query(self, query: str, history: list) -> Dict[str, Any]: - contextual_query = self._format_query_with_history(query, history) - structured_query = self.graph_query_translator.translate(contextual_query) - if not structured_query.get("start_node"): - return self.retrieval_pipeline.run(contextual_query, window_size_override=0) - results = self.graph_retriever.retrieve(structured_query) - if not results: - return self.retrieval_pipeline.run(contextual_query, window_size_override=0) - answer = ", ".join([res['details']['node_id'] for res in results]) - return {"answer": f"From the knowledge graph: {answer}", "source_documents": results} - - def _get_cache_key(self, query: str, query_type: str) -> str: - """Generate a cache key for the query""" - # Simple cache key based on query and type - return f"{query_type}:{query.strip().lower()}" - - def _cache_result(self, cache_key: str, result: Dict[str, Any], session_id: Optional[str] = None): - """Cache a result with size limit""" - if len(self._query_cache) >= self._cache_max_size: - # Remove oldest entry (simple FIFO eviction) - oldest_key = next(iter(self._query_cache)) - del self._query_cache[oldest_key] - - self._query_cache[cache_key] = { - 'result': result, - 'timestamp': time.time(), - 'session_id': session_id - } - # ---------------- Public sync API (kept for backwards compatibility) -------------- - def run(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, search_type: Optional[str] = None, dense_weight: Optional[float] = None, max_retries: int = 1, event_callback: Optional[callable] = None) -> Dict[str, Any]: + def run(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, retrieval_mode: Optional[str] = None, force_rag: bool = False, event_callback: Optional[callable] = None) -> Dict[str, Any]: """Synchronous helper. If *event_callback* is supplied, important milestones will be forwarded to that callable as event_callback(phase:str, payload:Any) """ - return asyncio.run(self._run_async(query, table_name, session_id, compose_sub_answers, query_decompose, ai_rerank, context_expand, verify, retrieval_k, context_window_size, reranker_top_k, search_type, dense_weight, max_retries, event_callback)) + return asyncio.run(self._run_async(query, table_name, session_id, compose_sub_answers, query_decompose, ai_rerank, context_expand, verify, retrieval_k, context_window_size, reranker_top_k, retrieval_mode, force_rag, event_callback)) # ---------------- Main async implementation -------------------------------------- - async def _run_async(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, search_type: Optional[str] = None, dense_weight: Optional[float] = None, max_retries: int = 1, event_callback: Optional[callable] = None) -> Dict[str, Any]: + async def _run_async(self, query: str, table_name: str = None, session_id: str = None, compose_sub_answers: Optional[bool] = None, query_decompose: Optional[bool] = None, ai_rerank: Optional[bool] = None, context_expand: Optional[bool] = None, verify: Optional[bool] = None, retrieval_k: Optional[int] = None, context_window_size: Optional[int] = None, reranker_top_k: Optional[int] = None, retrieval_mode: Optional[str] = None, force_rag: bool = False, event_callback: Optional[callable] = None) -> Dict[str, Any]: start_time = time.time() # Emit analyze event at the start @@ -279,51 +264,41 @@ async def _run_async(self, query: str, table_name: str = None, session_id: str = # self._load_overviews(self._global_overview_path) # self._current_overview_session = "GLOBAL" - query_type = await self._triage_query_async(query, history) - print(f"🎯 ROUTING DEBUG: Final triage decision: '{query_type}'") + if force_rag: + query_type = "rag_query" + print("🎯 ROUTING DEBUG: force_rag set – triage skipped, using 'rag_query'") + else: + query_type = await self._triage_query_async(query, history) + print(f"🎯 ROUTING DEBUG: Final triage decision: '{query_type}'") print(f"Agent Triage Decision: '{query_type}'") - + # Create a contextual query that includes history for most operations contextual_query = self._format_query_with_history(query, history) raw_query = query.strip() - + # --- Apply runtime AI reranker override (must happen before any retrieval calls) --- if ai_rerank is not None: rr_cfg = self.retrieval_pipeline.config.setdefault("reranker", {}) rr_cfg["enabled"] = bool(ai_rerank) - if ai_rerank: - # Ensure the pipeline knows to use the external ColBERT reranker - rr_cfg.setdefault("type", "ai") - rr_cfg.setdefault("strategy", "rerankers-lib") - rr_cfg.setdefault( - "model_name", - # Falls back to ColBERT-small if the caller did not supply one - self.ollama_config.get("rerank_model", "answerai-colbert-small-v1"), - ) # --- Apply runtime retrieval configuration overrides --- if retrieval_k is not None: self.retrieval_pipeline.config["retrieval_k"] = retrieval_k print(f"🔍 Retrieval K set to: {retrieval_k}") - + if context_window_size is not None: self.retrieval_pipeline.config["context_window_size"] = context_window_size print(f"🔍 Context window size set to: {context_window_size}") - + if reranker_top_k is not None: rr_cfg = self.retrieval_pipeline.config.setdefault("reranker", {}) rr_cfg["top_k"] = reranker_top_k print(f"🔍 Reranker top K set to: {reranker_top_k}") - - if search_type is not None: + + if retrieval_mode is not None: retrieval_cfg = self.retrieval_pipeline.config.setdefault("retrieval", {}) - retrieval_cfg["search_type"] = search_type - print(f"🔍 Search type set to: {search_type}") - - if dense_weight is not None: - dense_cfg = self.retrieval_pipeline.config.setdefault("retrieval", {}).setdefault("dense", {}) - dense_cfg["weight"] = dense_weight - print(f"🔍 Dense search weight set to: {dense_weight}") + retrieval_cfg["search_type"] = retrieval_mode + print(f"🔍 Retrieval mode set to: {retrieval_mode}") query_embedding = None # 🚀 OPTIMIZED: Semantic Cache Check @@ -376,10 +351,6 @@ def _blocking_stream(): final_answer = await _run_stream() result = {"answer": final_answer, "source_documents": []} - - elif query_type == "graph_query" and hasattr(self, 'graph_retriever'): - print(f"✅ ROUTING DEBUG: Executing GRAPH_QUERY path") - result = self._run_graph_query(query, history) # --- RAG Query Processing with Optional Query Decomposition --- else: # Default to rag_query @@ -394,7 +365,11 @@ def _blocking_stream(): # Use the raw user query (without conversation history) for decomposition to avoid leakage of prior context # Pass the last 5 conversation turns for context resolution within the decomposer recent_history = history[-5:] if history else [] - sub_queries = self.query_decomposer.decompose(raw_query, recent_history) + sub_queries = self.query_decomposer.decompose( + raw_query, + recent_history, + max_sub_queries=query_decomp_config.get("max_sub_queries", 10), + ) if event_callback: event_callback("decomposition", {"sub_queries": sub_queries}) print(f"Original query: '{query}' (Contextual: '{contextual_query}')") @@ -426,59 +401,88 @@ def _blocking_stream(): if compose_sub_answers is not None: compose_from_sub_answers = compose_sub_answers - print(f"\n--- Processing {len(sub_queries)} sub-queries in parallel ---") - start_time_inner = time.time() - - # Shared containers - sub_answers = [] # For two-stage composition - all_source_docs = [] # For single-stage aggregation - citations_seen = set() + if not compose_from_sub_answers: + # ---- Roadmap item 2.2: decomposition applies at RERANK ---- + # The first stage runs ONCE, on the full original query. + # Fanning the *first stage* out over sub-queries dilutes + # it semantically (2026 MultiConIR/SSRB finding); the + # sub-queries earn their keep at the rerank stage, where + # every candidate is scored against every sub-query and + # the scores are aggregated with + # `query_decomposition.rerank_aggregate` ("max" or "mean"). + # With reranking off there is no rerank stage, so the + # sub-queries go unused and this is plain single-query + # retrieval — which is the shipped default. + print("\n--- Decomposition applied at rerank; first stage uses the full query ---") + if event_callback: + event_callback("retrieval_started", {"count": 1}) + result = self.retrieval_pipeline.run( + contextual_query, + table_name, + 0 if context_expand is False else None, + event_callback=event_callback, + sub_queries=sub_queries, + ) + if event_callback: + event_callback("final_answer", result) + else: + # `compose_from_sub_answers` keeps per-sub-query *retrieval* + # on purpose, and is the only thing that does: it needs a + # separate answer per sub-question to compose from, which a + # single shared candidate set cannot produce. One full + # RetrievalPipeline.run() — retrieval, rerank, synthesis — + # per sub-query, in parallel. + print(f"\n--- Processing {len(sub_queries)} sub-queries in parallel ---") + start_time_inner = time.time() + + sub_answers = [] + all_source_docs = [] + citations_seen = set() + + # Emit rerank_started before the parallel retrievals (each sub-query reranks). + if event_callback: + event_callback("rerank_started", {"count": len(sub_queries)}) + + # Emit token chunks as soon as we receive them. The UI + # keeps answers separated by `index`, so interleaving is + # harmless and gives continuous feedback. + + def make_cb(idx: int): + def _cb(ev_type: str, payload): + if event_callback is None: + return + if ev_type == "token": + event_callback("sub_query_token", {"index": idx, "text": payload.get("text", ""), "question": sub_queries[idx]}) + else: + event_callback(ev_type, payload) + return _cb + + with concurrent.futures.ThreadPoolExecutor(max_workers=min(3, len(sub_queries))) as executor: + future_to_query = { + executor.submit( + self.retrieval_pipeline.run, + sub_query, + table_name, + 0 if context_expand is False else None, + make_cb(i), + ): (i, sub_query) + for i, sub_query in enumerate(sub_queries) + } - # Emit rerank_started event before parallel retrievals (since each sub-query will rerank) - if event_callback: - event_callback("rerank_started", {"count": len(sub_queries)}) - - # Emit token chunks as soon as we receive them. The UI - # keeps answers separated by `index`, so interleaving is - # harmless and gives continuous feedback. - - def make_cb(idx: int): - def _cb(ev_type: str, payload): - if event_callback is None: - return - if ev_type == "token": - event_callback("sub_query_token", {"index": idx, "text": payload.get("text", ""), "question": sub_queries[idx]}) - else: - event_callback(ev_type, payload) - return _cb - - with concurrent.futures.ThreadPoolExecutor(max_workers=min(3, len(sub_queries))) as executor: - future_to_query = { - executor.submit( - self.retrieval_pipeline.run, - sub_query, - table_name, - 0 if context_expand is False else None, - make_cb(i), - ): (i, sub_query) - for i, sub_query in enumerate(sub_queries) - } + for future in concurrent.futures.as_completed(future_to_query): + i, sub_query = future_to_query[future] + try: + sub_result = future.result() + print(f"✅ Sub-Query {i+1} completed: '{sub_query}'") - for future in concurrent.futures.as_completed(future_to_query): - i, sub_query = future_to_query[future] - try: - sub_result = future.result() - print(f"✅ Sub-Query {i+1} completed: '{sub_query}'") - - if event_callback: - event_callback("sub_query_result", { - "index": i, - "query": sub_query, - "answer": sub_result.get("answer", ""), - "source_documents": sub_result.get("source_documents", []), - }) + if event_callback: + event_callback("sub_query_result", { + "index": i, + "query": sub_query, + "answer": sub_result.get("answer", ""), + "source_documents": sub_result.get("source_documents", []), + }) - if compose_from_sub_answers: sub_answers.append({ "question": sub_query, "answer": sub_result.get("answer", "") @@ -488,24 +492,15 @@ def _cb(ev_type: str, payload): if doc['chunk_id'] not in citations_seen: all_source_docs.append(doc) citations_seen.add(doc['chunk_id']) - else: - # Aggregate unique docs (single-stage path) - for doc in sub_result.get('source_documents', []): - if doc['chunk_id'] not in citations_seen: - all_source_docs.append(doc) - citations_seen.add(doc['chunk_id']) - except Exception as e: - print(f"❌ Sub-Query {i+1} failed: '{sub_query}' - {e}") + except Exception as e: + print(f"❌ Sub-Query {i+1} failed: '{sub_query}' - {e}") - parallel_time = time.time() - start_time_inner - print(f"🚀 Parallel processing completed in {parallel_time:.2f}s") + print(f"🚀 Parallel processing completed in {time.time() - start_time_inner:.2f}s") - # Emit retrieval_done and rerank_done after all sub-queries are processed - if event_callback: - event_callback("retrieval_done", {"count": len(sub_queries)}) - event_callback("rerank_done", {"count": len(sub_queries)}) + if event_callback: + event_callback("retrieval_done", {"count": len(sub_queries)}) + event_callback("rerank_done", {"count": len(sub_queries)}) - if compose_from_sub_answers: print("\n--- Composing final answer from sub-answers ---") compose_prompt = f""" You are an expert answer composer for a Retrieval-Augmented Generation (RAG) system. @@ -552,47 +547,11 @@ def _cb(ev_type: str, payload): } if event_callback: event_callback("final_answer", result) - else: - print(f"\n--- Aggregated {len(all_source_docs)} unique documents from all sub-queries ---") - - if all_source_docs: - aggregated_context = "\n\n".join([doc['text'] for doc in all_source_docs]) - final_answer = self.retrieval_pipeline._synthesize_final_answer(contextual_query, aggregated_context) - result = { - "answer": final_answer, - "source_documents": all_source_docs - } - if event_callback: - event_callback("final_answer", result) - else: - result = { - "answer": "I could not find relevant information to answer your question.", - "source_documents": [] - } - if event_callback: - event_callback("final_answer", result) else: # Standard retrieval (single-query) - retrieved_docs = (self.retrieval_pipeline.retriever.retrieve( - text_query=contextual_query, - table_name=table_name or self.retrieval_pipeline.storage_config["text_table_name"], - k=self.retrieval_pipeline.config.get("retrieval_k", 10), - ) if hasattr(self.retrieval_pipeline, "retriever") and self.retrieval_pipeline.retriever else []) - - print("\n=== DEBUG: Original retrieval order ===") - for i, d in enumerate(retrieved_docs[:10]): - snippet = (d.get('text','') or '')[:200].replace('\n',' ') - print(f"Orig[{i}] id={d.get('chunk_id')} dist={d.get('_distance','') or d.get('score','')} {snippet}") - result = self.retrieval_pipeline.run(contextual_query, table_name, 0 if context_expand is False else None, event_callback=event_callback) - # After run, result['source_documents'] is reranked list - reranked_docs = result.get('source_documents', []) - print("\n=== DEBUG: Reranked docs order ===") - for i, d in enumerate(reranked_docs[:10]): - snippet = (d.get('text','') or '')[:200].replace('\n',' ') - print(f"ReRank[{i}] id={d.get('chunk_id')} score={d.get('rerank_score','')} {snippet}") - + # Verification step (simplified for now) - Skip in fast mode verification_enabled = self.pipeline_configs.get("verification", {}).get("enabled", True) if verify is not None: @@ -651,22 +610,23 @@ def _route_via_overviews(self, query: str) -> str | None: router_prompt = f"""Task: Route query to correct system. -Documents available: Invoices, DeepSeek-V3 research papers +DOCUMENT OVERVIEWS: +{overviews_block} Query: "{query}" Is this query asking about: A) Greetings/social: "Hi", "Hello", "Thanks", "What's up", "How are you" -B) General knowledge: "CEO of Tesla", "capital of France", "what is 2+2" -C) Document content: invoice amounts, DeepSeek-V3 details, companies mentioned +B) General knowledge unrelated to the documents above: "CEO of Tesla", "capital of France", "what is 2+2" +C) Anything covered by, or plausibly contained in, the documents above If A or B → {{"category": "direct_answer"}} If C → {{"category": "rag_query"}} Response:""" - + resp = self.llm_client.generate_completion( - model=self.ollama_config["generation_model"], prompt=router_prompt, format="json" + model=self._utility_model(), prompt=router_prompt, format="json" ) try: raw_response = resp.get("response", "{}") diff --git a/rag_system/agent/verifier.py b/rag_system/agent/verifier.py index 89ae64c4..0cf4030c 100644 --- a/rag_system/agent/verifier.py +++ b/rag_system/agent/verifier.py @@ -1,6 +1,17 @@ +import asyncio import json +import os +import re +from threading import Lock +from typing import List, Optional + from rag_system.utils.ollama_client import OllamaClient +# Serialises the first (heavy) load of a local verifier model, the same way the +# reranker and Provence loads are serialised in retrieval_pipeline.py. +_local_verifier_lock: Lock = Lock() + + class VerificationResult: def __init__(self, is_grounded: bool, reasoning: str, verdict: str, confidence_score: int): self.is_grounded = is_grounded @@ -8,20 +19,182 @@ def __init__(self, is_grounded: bool, reasoning: str, verdict: str, confidence_s self.verdict = verdict self.confidence_score = confidence_score + +class VerifierModelUnavailable(RuntimeError): + """Raised when `VERIFIER_MODEL` names something that cannot be loaded.""" + + +# Models checked for suitability on 2026-08-09 (roadmap 2.4). Reported to the +# user verbatim when a configured verifier fails to load, so the failure names +# what was actually verified instead of hand-waving. +VERIFIER_AVAILABILITY_NOTES = """\ +Checked on 2026-08-09 (HuggingFace Hub API): + * ThinknCheck (arXiv 2604.01652, UPenn) — NO PUBLIC WEIGHTS. The paper is + real (1B, 78.1 BAcc on LLMAggreFact) but a Hub search for "thinkncheck" + returns zero models and the paper links no release. Cannot be wired. + * ibm-granite/granite-guardian-3.3-8b — exists, Apache-2.0, but 8B / + ~16 GB. Far over the "small local verifier" budget this seam is for. + * ibm-granite/granite-guardian-hap-38m — exists, 38M, Apache-2.0, but it + is a hate/abuse/profanity RoBERTa classifier. Wrong task: it does not score + answer-vs-evidence entailment at all. +Verified working (<2 GB, no trust_remote_code): + * lytang/MiniCheck-DeBERTa-v3-Large (MIT, 1.74 GB) <- smoke-tested default + * lytang/MiniCheck-RoBERTa-Large (MIT) + * MoritzLaurer/DeBERTa-v3-base-mnli-fever-anli (MIT, 369 MB, generic NLI) +Needs trust_remote_code (opt in with VERIFIER_TRUST_REMOTE_CODE=1): + * vectara/hallucination_evaluation_model (HHEM-2.1-open, Apache-2.0, 438 MB) +""" + +_SENTENCE_SPLIT = re.compile(r"(?<=[.!?])\s+") + + +class LocalNLIVerifier: + """Answer-vs-evidence scoring with a local sequence-classification model. + + The seam roadmap item 2.4 asks for. Any HuggingFace model that scores a + (premise, hypothesis) pair works: MiniCheck's grounded-claim checkers, a + generic MNLI cross-encoder, or Vectara's HHEM. The answer is split into + sentences, each is scored against the retrieved evidence as the premise, and + the **minimum** is taken — one unsupported sentence makes the answer + ungrounded, which is the semantics the binary judge already uses. + + Note on ``[Confidence: N%]``: this number is a model output, not a + calibrated probability of correctness. It is UX, and `Documentation/ + verifier.md` says so. Swapping the LLM prompt for an NLI model changes + where the number comes from; it does not make it calibrated. + """ + + def __init__(self, model_name: str, threshold: float = 0.5, + trust_remote_code: Optional[bool] = None): + self.model_name = model_name + self.threshold = threshold + if trust_remote_code is None: + trust_remote_code = os.getenv("VERIFIER_TRUST_REMOTE_CODE", "") == "1" + try: + import torch + from transformers import AutoModelForSequenceClassification, AutoTokenizer + except ImportError as e: # pragma: no cover - transformers is a hard dep + raise VerifierModelUnavailable( + f"transformers/torch are required to load VERIFIER_MODEL: {e}") + + self._torch = torch + try: + self.tokenizer = AutoTokenizer.from_pretrained( + model_name, trust_remote_code=trust_remote_code) + self.model = AutoModelForSequenceClassification.from_pretrained( + model_name, trust_remote_code=trust_remote_code) + except Exception as e: + raise VerifierModelUnavailable( + f"Could not load VERIFIER_MODEL='{model_name}': {e}\n\n" + f"{VERIFIER_AVAILABILITY_NOTES}" + "Set VERIFIER_MODEL to one of the verified names above, unset it to " + "use the default LLM-prompt verifier, or add " + "VERIFIER_TRUST_REMOTE_CODE=1 if the model ships custom code." + ) from e + + self.model.eval() + self.device = ("mps" if torch.backends.mps.is_available() + else "cuda" if torch.cuda.is_available() else "cpu") + self.model.to(self.device) + self._supported_index = self._resolve_supported_index() + print(f"✅ Local verifier '{model_name}' loaded on {self.device} " + f"(supported label index {self._supported_index}).") + + def _resolve_supported_index(self) -> int: + """Which logit means "the evidence supports this".""" + id2label = getattr(self.model.config, "id2label", None) or {} + for idx, label in id2label.items(): + if str(label).lower() in {"entailment", "consistent", "supported", "1", "true"}: + return int(idx) + # Binary checkers (MiniCheck) label their classes "0"/"1": 1 = supported. + return int(self.model.config.num_labels) - 1 + + def score(self, evidence: str, answer: str) -> float: + sentences = [s.strip() for s in _SENTENCE_SPLIT.split(answer or "") if s.strip()] + if not sentences: + return 0.0 + torch = self._torch + scores: List[float] = [] + with torch.no_grad(): + for sentence in sentences: + inputs = self.tokenizer(evidence, sentence, return_tensors="pt", + truncation=True, max_length=self.tokenizer.model_max_length + if self.tokenizer.model_max_length < 100000 else 2048) + inputs = {k: v.to(self.device) for k, v in inputs.items()} + logits = self.model(**inputs).logits[0] + if logits.numel() == 1: + probability = torch.sigmoid(logits)[0] + else: + probability = torch.softmax(logits, dim=-1)[self._supported_index] + scores.append(float(probability)) + # Weakest link: one unsupported sentence makes the answer ungrounded. + return min(scores) + + def verify(self, query: str, context: str, answer: str) -> VerificationResult: + probability = self.score(context, answer) + grounded = probability >= self.threshold + return VerificationResult( + is_grounded=grounded, + reasoning=(f"{self.model_name}: weakest answer sentence scored " + f"{probability:.3f} against the retrieved evidence " + f"(threshold {self.threshold})."), + verdict="SUPPORTED" if grounded else "NOT_SUPPORTED", + confidence_score=int(round(probability * 100)), + ) + + class Verifier: """ - Verifies if a generated answer is grounded in the provided context using Ollama. + Verifies if a generated answer is grounded in the provided context. + + Two backends, same interface: + + * **default** — an LLM prompt on the utility model (below). This is what + ships; nothing changes unless you opt in. + * **local NLI/verifier model** — set ``VERIFIER_MODEL`` (or + ``verification.model`` in the pipeline config) to a HuggingFace model name. + Loaded lazily on first use through ``LocalNLIVerifier``, so naming a model + costs nothing until a query is actually verified. A model that cannot be + loaded raises with the list of names that were checked, rather than + silently degrading to the LLM prompt — a verifier that quietly is not the + verifier you configured is worse than an error. + + Roadmap item 2.4. Availability findings are in ``VERIFIER_AVAILABILITY_NOTES``. """ - def __init__(self, llm_client: OllamaClient, llm_model: str): + + def __init__(self, llm_client: OllamaClient, llm_model: str, + model_name: Optional[str] = None, threshold: float = 0.5): self.llm_client = llm_client self.llm_model = llm_model - print(f"Initialized Verifier with Ollama model '{self.llm_model}'.") + self.local_model_name = model_name or os.getenv("VERIFIER_MODEL") or None + self.local_threshold = threshold + self._local: Optional[LocalNLIVerifier] = None + if self.local_model_name: + print(f"Initialized Verifier with local model '{self.local_model_name}' " + f"(loaded on first use); LLM fallback model '{self.llm_model}'.") + else: + print(f"Initialized Verifier with Ollama model '{self.llm_model}'.") + + def _get_local(self) -> Optional[LocalNLIVerifier]: + if not self.local_model_name: + return None + if self._local is None: + with _local_verifier_lock: + if self._local is None: + self._local = LocalNLIVerifier(self.local_model_name, + self.local_threshold) + return self._local # Synchronous verify() method removed – async version is used everywhere. # --- Async wrapper ------------------------------------------------ async def verify_async(self, query: str, context: str, answer: str) -> VerificationResult: """Async variant that calls the Ollama client asynchronously.""" + local = self._get_local() + if local is not None: + # transformers is blocking; keep the event loop responsive. + return await asyncio.to_thread(local.verify, query, context, answer) + prompt = f""" You are an automated fact-checker. Determine whether the ANSWER is fully supported by the CONTEXT and output a single line of JSON. diff --git a/rag_system/api_server.py b/rag_system/api_server.py index de148361..cb835e6a 100644 --- a/rag_system/api_server.py +++ b/rag_system/api_server.py @@ -1,8 +1,11 @@ +import copy import json import http.server import socketserver -from urllib.parse import urlparse, parse_qs +from contextlib import contextmanager +from urllib.parse import urlparse import os +import re import requests import sys import logging @@ -12,112 +15,238 @@ if backend_dir not in sys.path: sys.path.append(backend_dir) -from backend.database import ChatDatabase, generate_session_title -from rag_system.main import get_agent -from rag_system.factory import get_indexing_pipeline +from backend.database import ChatDatabase +from rag_system.factory import get_agent, get_indexing_pipeline +from rag_system.main import LLM_BACKEND, PIPELINE_CONFIGS, WATSONX_CONFIG -# Initialize database connection once at module level -# Use auto-detection for environment-appropriate path +logger = logging.getLogger(__name__) + +# The RAG API reads/writes index metadata only. Chat message rows are owned +# exclusively by backend/server.py. db = ChatDatabase() # Get the desired agent mode from environment variables, defaulting to 'default' -# This allows us to easily switch between 'default', 'fast', 'react', etc. AGENT_MODE = os.getenv("RAG_CONFIG_MODE", "default") + +# --- Global Singletons --- +# The agent and indexing pipeline are initialized once when the server starts so +# that models are not reloaded on every request. +print("🧠 Initializing RAG Agent... (This may take a moment)") RAG_AGENT = get_agent(AGENT_MODE) INDEXING_PIPELINE = get_indexing_pipeline(AGENT_MODE) +print("✅ RAG Agent initialized successfully.") + +DEFAULT_TEXT_TABLE = PIPELINE_CONFIGS.get(AGENT_MODE, PIPELINE_CONFIGS["default"])["storage"]["text_table_name"] +SUPPORTED_RETRIEVAL_MODES = ("hybrid", "vector_only", "fts_only") +OLLAMA_TIMEOUT_SECONDS = 5 + +_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])") + +# Wire aliases that are not a plain camelCase/snake_case pair. +_KEY_ALIASES = { + "overview_model": "overview_model_name", + "latechunk": "enable_latechunk", + "docling_chunk": "enable_docling_chunk", + "decompose": "query_decompose", +} + + +def normalize_request_keys(data): + """Accept camelCase and snake_case spellings of every option. + + The frontend historically sent camelCase while the backend gateway sends + snake_case. Both land in the same canonical snake_case key here, once, at + parse time. Explicit snake_case values always win over a camelCase twin. + """ + if not isinstance(data, dict): + return data + + normalized = {} + camel_derived = {} + for key, value in data.items(): + if not isinstance(key, str): + normalized[key] = value + continue + canonical = _CAMEL_BOUNDARY.sub("_", key).lower() + canonical = _KEY_ALIASES.get(canonical, canonical) + if canonical == key: + normalized[key] = value + else: + camel_derived[canonical] = value + + for key, value in camel_derived.items(): + normalized.setdefault(key, value) + return normalized + + +def _read_json_body(handler): + """Read and normalize a JSON request body. Returns an empty dict on empty body.""" + content_length = int(handler.headers.get('Content-Length') or 0) + if content_length <= 0: + return {} + post_data = handler.rfile.read(content_length) + return normalize_request_keys(json.loads(post_data.decode('utf-8'))) + + +def _model_valid_for_backend(model_name: str) -> bool: + """A watsonx deployment cannot serve an Ollama model id (and vice versa).""" + if LLM_BACKEND.lower() == "watsonx": + return "/" in model_name + return "/" not in model_name + + +@contextmanager +def _generation_model_override(requested_model): + """Apply a per-request generation model without permanently mutating the singleton.""" + applied = False + previous = RAG_AGENT.ollama_config.get("generation_model") + if isinstance(requested_model, str) and requested_model: + if _model_valid_for_backend(requested_model): + RAG_AGENT.ollama_config["generation_model"] = requested_model + applied = True + else: + logger.warning( + "Ignoring requested model '%s': not valid for the '%s' backend (using '%s').", + requested_model, LLM_BACKEND, previous, + ) + try: + yield + finally: + if applied: + RAG_AGENT.ollama_config["generation_model"] = previous -# --- Global Singleton for the RAG Agent --- -# The agent is initialized once when the server starts. -# This avoids reloading all the models on every request. -print("🧠 Initializing RAG Agent with MAXIMUM ACCURACY... (This may take a moment)") -if RAG_AGENT is None: - print("❌ Critical error: RAG Agent could not be initialized. Exiting.") - exit(1) -print("✅ RAG Agent initialized successfully with MAXIMUM ACCURACY.") -# --- - -# Add helper near top after db & agent init -# -------------- Helper ---------------- def _apply_index_embedding_model(idx_ids): - """Ensure retrieval pipeline uses the embedding model stored with the first index.""" - debug_info = f"🔧 _apply_index_embedding_model called with idx_ids: {idx_ids}\n" - + """Ensure the retrieval pipeline uses the embedding model stored with the first index.""" if not idx_ids: - debug_info += "⚠️ No index IDs provided\n" - with open("logs/embedding_debug.log", "a") as f: - f.write(debug_info) + logger.debug("No index IDs provided; keeping the configured embedding model.") return try: idx = db.get_index(idx_ids[0]) - debug_info += f"🔧 Retrieved index: {idx.get('id')} with metadata: {idx.get('metadata', {})}\n" model = (idx.get("metadata") or {}).get("embedding_model") - debug_info += f"🔧 Embedding model from metadata: {model}\n" - if model: - rp = RAG_AGENT.retrieval_pipeline - current_model = rp.config.get("embedding_model_name") - debug_info += f"🔧 Current embedding model: {current_model}\n" - rp.update_embedding_model(model) - debug_info += f"🔧 Updated embedding model to: {model}\n" - else: - debug_info += "⚠️ No embedding model found in metadata\n" + if not model: + logger.debug("Index %s has no embedding_model metadata.", idx_ids[0]) + return + rp = RAG_AGENT.retrieval_pipeline + logger.info( + "Applying index embedding model '%s' (was '%s').", + model, rp.config.get("embedding_model_name"), + ) + rp.update_embedding_model(model) except Exception as e: - debug_info += f"⚠️ Could not apply index embedding model: {e}\n" - - # Write debug info to file - with open("logs/embedding_debug.log", "a") as f: - f.write(debug_info) + logger.warning("Could not apply index embedding model: %s", e) + def _get_table_name_for_session(session_id): """Get the correct vector table name for a session by looking up its linked indexes.""" - logger = logging.getLogger(__name__) - if not session_id: - logger.info("❌ No session_id provided") return None - + try: - # Get indexes linked to this session idx_ids = db.get_indexes_for_session(session_id) - logger.info(f"🔍 Session {session_id[:8]}... has {len(idx_ids)} indexes: {idx_ids}") - if not idx_ids: - logger.warning(f"⚠️ No indexes found for session {session_id}") - # Use the default table name from config instead of session-specific name - from rag_system.main import PIPELINE_CONFIGS - default_table = PIPELINE_CONFIGS["default"]["storage"]["text_table_name"] - logger.info(f"📊 Using default table '{default_table}' for session {session_id[:8]}...") - return default_table - - # Use the first index's vector table name + logger.info("No indexes for session %s; using default table '%s'.", session_id, DEFAULT_TEXT_TABLE) + return DEFAULT_TEXT_TABLE + idx = db.get_index(idx_ids[0]) if idx and idx.get('vector_table_name'): table_name = idx['vector_table_name'] - logger.info(f"📊 Using table '{table_name}' for session {session_id[:8]}...") - print(f"📊 RAG API: Using table '{table_name}' for session {session_id[:8]}...") + logger.info("Using table '%s' for session %s.", table_name, session_id) return table_name - else: - logger.warning(f"⚠️ Index found but no vector table name for session {session_id}") - # Use the default table name from config instead of session-specific name - from rag_system.main import PIPELINE_CONFIGS - default_table = PIPELINE_CONFIGS["default"]["storage"]["text_table_name"] - logger.info(f"📊 Using default table '{default_table}' for session {session_id[:8]}...") - return default_table - + + logger.warning("Index found but no vector table name for session %s.", session_id) + return DEFAULT_TEXT_TABLE except Exception as e: - logger.error(f"❌ Error getting table name for session {session_id}: {e}") - # Use the default table name from config instead of session-specific name - from rag_system.main import PIPELINE_CONFIGS - default_table = PIPELINE_CONFIGS["default"]["storage"]["text_table_name"] - logger.info(f"📊 Using default table '{default_table}' for session {session_id[:8]}...") - return default_table + logger.error("Error getting table name for session %s: %s", session_id, e) + return DEFAULT_TEXT_TABLE + + +def _resolve_retrieval_mode(data): + """`retrieval_mode` is the wire name for the pipeline's `search_type`; both are accepted.""" + value = data.get('retrieval_mode') or data.get('search_type') + if value is None: + return None, None + if value not in SUPPORTED_RETRIEVAL_MODES: + return None, f"Unsupported retrieval mode '{value}'. Supported: {', '.join(SUPPORTED_RETRIEVAL_MODES)}." + return value, None + + +def _parse_chat_request(data): + """Extract the canonical chat options from an already-normalized body.""" + retrieval_mode, error = _resolve_retrieval_mode(data) + if error: + return None, error + + return { + "query": data.get('query'), + "session_id": data.get('session_id'), + "table_name": data.get('table_name'), + "model": data.get('model'), + "compose_sub_answers": data.get('compose_sub_answers'), + "query_decompose": data.get('query_decompose'), + "ai_rerank": data.get('ai_rerank'), + "context_expand": data.get('context_expand'), + "verify": data.get('verify'), + "retrieval_k": data.get('retrieval_k'), + "context_window_size": data.get('context_window_size'), + "reranker_top_k": data.get('reranker_top_k'), + "retrieval_mode": retrieval_mode, + "force_rag": bool(data.get('force_rag', False)), + "provence_prune": data.get('provence_prune'), + "provence_threshold": data.get('provence_threshold'), + }, None + + +def _build_index_config_override(base_config, *, table_name, options): + """Build the per-request indexing config from the pipeline profile plus request options.""" + config_override = copy.deepcopy(base_config) + retrieval_cfg = config_override.setdefault("retrieval", {}) + + if table_name: + config_override.setdefault("storage", {})["text_table_name"] = table_name + retrieval_cfg.setdefault("dense", {})["lancedb_table_name"] = table_name + + retrieval_cfg.setdefault("latechunk", {})["enabled"] = options["enable_latechunk"] + + # `retrieval_mode` is the wire name for the pipeline's `search_type`; it is + # recorded on the index config so the built index carries the mode it was + # requested with. + if options["retrieval_mode"] is not None: + retrieval_cfg["search_type"] = options["retrieval_mode"] + + config_override["chunker_mode"] = "docling" if options["enable_docling_chunk"] else "legacy" + + enricher_cfg = config_override.setdefault("contextual_enricher", {}) + enricher_cfg["enabled"] = options["enable_enrich"] + enricher_cfg["window_size"] = options["window_size"] + + indexing_cfg = config_override.setdefault("indexing", {}) + indexing_cfg["embedding_batch_size"] = options["batch_size_embed"] + indexing_cfg["enrichment_batch_size"] = options["batch_size_enrich"] + + config_override.setdefault("chunking", {})["chunk_size"] = options["chunk_size"] + + if options["embedding_model"]: + config_override["embedding_model_name"] = options["embedding_model"] + if options["enrich_model"]: + config_override["enrich_model"] = options["enrich_model"] + if options["overview_model_name"]: + config_override["overview_model_name"] = options["overview_model_name"] + if options["session_id"]: + config_override["overview_path"] = f"index_store/overviews/{options['session_id']}.jsonl" + + return config_override + class AdvancedRagApiHandler(http.server.BaseHTTPRequestHandler): + def log_message(self, format, *args): + logger.info("%s - %s", self.address_string(), format % args) + def do_OPTIONS(self): """Handle CORS preflight requests for frontend integration.""" self.send_response(200) self.send_header('Access-Control-Allow-Origin', '*') - self.send_header('Access-Control-Allow-Methods', 'POST, OPTIONS') + self.send_header('Access-Control-Allow-Methods', 'GET, POST, OPTIONS') self.send_header('Access-Control-Allow-Headers', 'Content-Type') self.end_headers() @@ -137,7 +266,9 @@ def do_POST(self): def do_GET(self): parsed_path = urlparse(self.path) - if parsed_path.path == '/models': + if parsed_path.path == '/health': + self.send_json_response({"status": "ok"}) + elif parsed_path.path == '/models': self.handle_models() else: self.send_json_response({"error": "Not Found"}, status_code=404) @@ -145,234 +276,47 @@ def do_GET(self): def handle_chat(self): """Handles a chat query by calling the agentic RAG pipeline.""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - query = data.get('query') - session_id = data.get('session_id') - compose_flag = data.get('compose_sub_answers') - decomp_flag = data.get('query_decompose') - ai_rerank_flag = data.get('ai_rerank') - ctx_expand_flag = data.get('context_expand') - verify_flag = data.get('verify') - - # ✨ NEW RETRIEVAL PARAMETERS - retrieval_k = data.get('retrieval_k', 20) - context_window_size = data.get('context_window_size', 1) - reranker_top_k = data.get('reranker_top_k', 10) - search_type = data.get('search_type', 'hybrid') - dense_weight = data.get('dense_weight', 0.7) - - # 🚩 NEW: Force RAG override from frontend - force_rag = bool(data.get('force_rag', False)) - - # 🌿 Provence sentence pruning - provence_prune = data.get('provence_prune') - provence_threshold = data.get('provence_threshold') - - # User-selected generation model - requested_model = data.get('model') - if isinstance(requested_model,str) and requested_model: - RAG_AGENT.ollama_config['generation_model']=requested_model - + data = _read_json_body(self) + params, error = _parse_chat_request(data) + if error: + self.send_json_response({"error": error}, status_code=400) + return + + query = params["query"] if not query: self.send_json_response({"error": "Query is required"}, status_code=400) return - # 🔄 UPDATE SESSION TITLE: If this is the first message in the session, update the title - if session_id: - try: - # Check if this is the first message by calling the backend server - backend_url = f"http://localhost:8000/sessions/{session_id}" - session_resp = requests.get(backend_url) - if session_resp.status_code == 200: - session_data = session_resp.json() - session = session_data.get('session', {}) - # If message_count is 0, this is the first message - if session.get('message_count', 0) == 0: - # Generate a title from the first message - title = generate_session_title(query) - # Update the session title via backend API - # We'll need to add this endpoint to the backend, for now let's make a direct database call - # This is a temporary solution until we add a proper API endpoint - db.update_session_title(session_id, title) - print(f"📝 Updated session title to: {title}") - - # 💾 STORE USER MESSAGE: Add the user message to the database - user_message_id = db.add_message(session_id, query, "user") - print(f"💾 Stored user message: {user_message_id}") - else: - # Not the first message, but still store the user message - user_message_id = db.add_message(session_id, query, "user") - print(f"💾 Stored user message: {user_message_id}") - except Exception as e: - print(f"⚠️ Failed to update session title or store user message: {e}") - # Continue with the request even if title update fails - - # Allow explicit table_name override - table_name = data.get('table_name') - if not table_name and session_id: - table_name = _get_table_name_for_session(session_id) - - # Decide execution path - print(f"🔧 Force RAG flag: {force_rag}") - if force_rag: - # --- Apply runtime overrides manually because we skip Agent.run() - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if retrieval_k is not None: - rp_cfg["retrieval_k"] = retrieval_k - if reranker_top_k is not None: - rp_cfg.setdefault("reranker", {})["top_k"] = reranker_top_k - if search_type is not None: - rp_cfg.setdefault("retrieval", {})["search_type"] = search_type - if dense_weight is not None: - rp_cfg.setdefault("retrieval", {}).setdefault("dense", {})["weight"] = dense_weight - - # Provence overrides - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # 🔄 Apply embedding model for this session (same as in agent path) - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - - # Directly invoke retrieval pipeline to bypass triage - result = RAG_AGENT.retrieval_pipeline.run( - query, - table_name=table_name, - window_size_override=context_window_size, - ) - else: - # Use full agent with smart routing - # Apply Provence overrides even in agent path - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # 🔄 Refresh document overviews for this session - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - RAG_AGENT.load_overviews_for_indexes(idx_ids) - - # 🔧 Set index-specific overview path - if session_id: - rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # 🔧 Configure late chunking - rp_cfg.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - result = RAG_AGENT.run( - query, - table_name=table_name, - session_id=session_id, - compose_sub_answers=compose_flag, - query_decompose=decomp_flag, - ai_rerank=ai_rerank_flag, - context_expand=ctx_expand_flag, - verify=verify_flag, - retrieval_k=retrieval_k, - context_window_size=context_window_size, - reranker_top_k=reranker_top_k, - search_type=search_type, - dense_weight=dense_weight, - ) - - # The result is a dict, so we need to dump it to a JSON string + session_id = params["session_id"] + table_name = params["table_name"] or _get_table_name_for_session(session_id) + + with _generation_model_override(params["model"]): + result = self._run_query(params, query, table_name, session_id, emit=None) + self.send_json_response(result) - - # 💾 STORE AI RESPONSE: Add the AI response to the database - if session_id and result and result.get("answer"): - try: - ai_message_id = db.add_message(session_id, result["answer"], "assistant") - print(f"💾 Stored AI response: {ai_message_id}") - except Exception as e: - print(f"⚠️ Failed to store AI response: {e}") - # Continue even if storage fails except json.JSONDecodeError: self.send_json_response({"error": "Invalid JSON"}, status_code=400) except Exception as e: + logger.exception("Chat request failed") self.send_json_response({"error": f"Server error: {str(e)}"}, status_code=500) def handle_chat_stream(self): """Stream internal phases and final answer using SSE (text/event-stream).""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - query = data.get('query') - session_id = data.get('session_id') - compose_flag = data.get('compose_sub_answers') - decomp_flag = data.get('query_decompose') - ai_rerank_flag = data.get('ai_rerank') - ctx_expand_flag = data.get('context_expand') - verify_flag = data.get('verify') - - # ✨ NEW RETRIEVAL PARAMETERS - retrieval_k = data.get('retrieval_k', 20) - context_window_size = data.get('context_window_size', 1) - reranker_top_k = data.get('reranker_top_k', 10) - search_type = data.get('search_type', 'hybrid') - dense_weight = data.get('dense_weight', 0.7) - - # 🚩 NEW: Force RAG override from frontend - force_rag = bool(data.get('force_rag', False)) - - # 🌿 Provence sentence pruning - provence_prune = data.get('provence_prune') - provence_threshold = data.get('provence_threshold') - - # User-selected generation model - requested_model = data.get('model') - if isinstance(requested_model,str) and requested_model: - RAG_AGENT.ollama_config['generation_model']=requested_model + data = _read_json_body(self) + params, error = _parse_chat_request(data) + if error: + self.send_json_response({"error": error}, status_code=400) + return + query = params["query"] if not query: self.send_json_response({"error": "Query is required"}, status_code=400) return - # 🔄 UPDATE SESSION TITLE: If this is the first message in the session, update the title - if session_id: - try: - # Check if this is the first message by calling the backend server - backend_url = f"http://localhost:8000/sessions/{session_id}" - session_resp = requests.get(backend_url) - if session_resp.status_code == 200: - session_data = session_resp.json() - session = session_data.get('session', {}) - # If message_count is 0, this is the first message - if session.get('message_count', 0) == 0: - # Generate a title from the first message - title = generate_session_title(query) - # Update the session title via backend API - # We'll need to add this endpoint to the backend, for now let's make a direct database call - # This is a temporary solution until we add a proper API endpoint - db.update_session_title(session_id, title) - print(f"📝 Updated session title to: {title}") - - # 💾 STORE USER MESSAGE: Add the user message to the database - user_message_id = db.add_message(session_id, query, "user") - print(f"💾 Stored user message: {user_message_id}") - else: - # Not the first message, but still store the user message - user_message_id = db.add_message(session_id, query, "user") - print(f"💾 Stored user message: {user_message_id}") - except Exception as e: - print(f"⚠️ Failed to update session title or store user message: {e}") - # Continue with the request even if title update fails - - # Allow explicit table_name override - table_name = data.get('table_name') - if not table_name and session_id: - table_name = _get_table_name_for_session(session_id) + session_id = params["session_id"] + table_name = params["table_name"] or _get_table_name_for_session(session_id) # Prepare response headers for SSE self.send_response(200) @@ -386,347 +330,188 @@ def handle_chat_stream(self): def emit(event_type: str, payload): """Send a single SSE event.""" - try: - data_str = json.dumps({"type": event_type, "data": payload}) - self.wfile.write(f"data: {data_str}\n\n".encode('utf-8')) - self.wfile.flush() - except BrokenPipeError: - # Client disconnected - raise + data_str = json.dumps({"type": event_type, "data": payload}) + self.wfile.write(f"data: {data_str}\n\n".encode('utf-8')) + self.wfile.flush() - # Run the agent synchronously, emitting checkpoints try: - if force_rag: - # Apply overrides same as above since we bypass Agent.run - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if retrieval_k is not None: - rp_cfg["retrieval_k"] = retrieval_k - if reranker_top_k is not None: - rp_cfg.setdefault("reranker", {})["top_k"] = reranker_top_k - if search_type is not None: - rp_cfg.setdefault("retrieval", {})["search_type"] = search_type - if dense_weight is not None: - rp_cfg.setdefault("retrieval", {}).setdefault("dense", {})["weight"] = dense_weight - - # Provence overrides - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # 🔄 Apply embedding model for this session (same as in agent path) - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - - # 🔧 Set index-specific overview path so each index writes separate file - if session_id: - rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # 🔧 Configure late chunking - rp_cfg.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # Straight retrieval pipeline with streaming events - final_result = RAG_AGENT.retrieval_pipeline.run( - query, - table_name=table_name, - window_size_override=context_window_size, - event_callback=emit, - ) - else: - # Provence overrides - rp_cfg = RAG_AGENT.retrieval_pipeline.config - if provence_prune is not None: - rp_cfg.setdefault("provence", {})["enabled"] = bool(provence_prune) - if provence_threshold is not None: - rp_cfg.setdefault("provence", {})["threshold"] = float(provence_threshold) - - # 🔄 Refresh overviews for this session - if session_id: - idx_ids = db.get_indexes_for_session(session_id) - _apply_index_embedding_model(idx_ids) - RAG_AGENT.load_overviews_for_indexes(idx_ids) - - # 🔧 Set index-specific overview path - if session_id: - rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # 🔧 Configure late chunking - rp_cfg.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - final_result = RAG_AGENT.run( - query, - table_name=table_name, - session_id=session_id, - compose_sub_answers=compose_flag, - query_decompose=decomp_flag, - ai_rerank=ai_rerank_flag, - context_expand=ctx_expand_flag, - verify=verify_flag, - # ✨ NEW RETRIEVAL PARAMETERS - retrieval_k=retrieval_k, - context_window_size=context_window_size, - reranker_top_k=reranker_top_k, - search_type=search_type, - dense_weight=dense_weight, - event_callback=emit, - ) - - # Ensure the final answer is sent (in case callback missed it) + with _generation_model_override(params["model"]): + final_result = self._run_query(params, query, table_name, session_id, emit=emit) emit("complete", final_result) - - # 💾 STORE AI RESPONSE: Add the AI response to the database - if session_id and final_result and final_result.get("answer"): - try: - ai_message_id = db.add_message(session_id, final_result["answer"], "assistant") - print(f"💾 Stored AI response: {ai_message_id}") - except Exception as e: - print(f"⚠️ Failed to store AI response: {e}") - # Continue even if storage fails except BrokenPipeError: - print("🔌 Client disconnected from SSE stream.") + logger.info("Client disconnected from SSE stream.") except Exception as e: - # Send error event then close - error_payload = {"error": str(e)} + logger.exception("Stream error") try: - emit("error", error_payload) - finally: - print(f"❌ Stream error: {e}") + emit("error", {"error": str(e)}) + except BrokenPipeError: + pass except json.JSONDecodeError: self.send_json_response({"error": "Invalid JSON"}, status_code=400) except Exception as e: + logger.exception("Chat stream request failed") self.send_json_response({"error": f"Server error: {str(e)}"}, status_code=500) + def _run_query(self, params, query, table_name, session_id, emit): + """Shared execution path for /chat and /chat/stream. + + Everything, including force_rag, goes through Agent.run so that the + verify / ai_rerank / decompose toggles are honored on every path. + """ + rp_cfg = RAG_AGENT.retrieval_pipeline.config + + if session_id: + idx_ids = db.get_indexes_for_session(session_id) + _apply_index_embedding_model(idx_ids) + rp_cfg["overview_path"] = f"index_store/overviews/{session_id}.jsonl" + RAG_AGENT.load_overviews_for_indexes(idx_ids) + + if params["provence_prune"] is not None: + rp_cfg.setdefault("provence", {})["enabled"] = bool(params["provence_prune"]) + if params["provence_threshold"] is not None: + rp_cfg.setdefault("provence", {})["threshold"] = float(params["provence_threshold"]) + + run_kwargs = { + "table_name": table_name, + "session_id": session_id, + "compose_sub_answers": params["compose_sub_answers"], + "query_decompose": params["query_decompose"], + "ai_rerank": params["ai_rerank"], + "context_expand": params["context_expand"], + "verify": params["verify"], + "retrieval_k": params["retrieval_k"], + "context_window_size": params["context_window_size"], + "reranker_top_k": params["reranker_top_k"], + "retrieval_mode": params["retrieval_mode"], + "force_rag": params["force_rag"], + } + if emit is not None: + run_kwargs["event_callback"] = emit + + return RAG_AGENT.run(query, **run_kwargs) + def handle_index(self): """Triggers the document indexing pipeline for specific files.""" try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - + data = _read_json_body(self) + file_paths = data.get('file_paths') - session_id = data.get('session_id') - compose_flag = data.get('compose_sub_answers') - decomp_flag = data.get('query_decompose') - ai_rerank_flag = data.get('ai_rerank') - ctx_expand_flag = data.get('context_expand') - enable_latechunk = bool(data.get("enable_latechunk", False)) - enable_docling_chunk = bool(data.get("enable_docling_chunk", False)) - - # 🆕 NEW CONFIGURATION OPTIONS: - chunk_size = int(data.get("chunk_size", 512)) - chunk_overlap = int(data.get("chunk_overlap", 64)) - retrieval_mode = data.get("retrieval_mode", "hybrid") - window_size = int(data.get("window_size", 2)) - enable_enrich = bool(data.get("enable_enrich", True)) - embedding_model = data.get('embeddingModel') - enrich_model = data.get('enrichModel') - overview_model = data.get('overviewModel') or data.get('overview_model_name') - batch_size_embed = int(data.get("batch_size_embed", 50)) - batch_size_enrich = int(data.get("batch_size_enrich", 25)) - if not file_paths or not isinstance(file_paths, list): - self.send_json_response({ - "error": "A 'file_paths' list is required." - }, status_code=400) + self.send_json_response({"error": "A 'file_paths' list is required."}, status_code=400) return - # Allow explicit table_name override - table_name = data.get('table_name') - if not table_name and session_id: - table_name = _get_table_name_for_session(session_id) - - # The INDEXING_PIPELINE is already initialized. We just need to use it. - # If a session-specific table is needed, we can override the config for this run. - if table_name: - import copy - config_override = copy.deepcopy(INDEXING_PIPELINE.config) - config_override["storage"]["text_table_name"] = table_name - config_override.setdefault("retrievers", {}).setdefault("dense", {})["lancedb_table_name"] = table_name - - # 🔧 Configure late chunking - if enable_latechunk: - config_override["retrievers"].setdefault("latechunk", {})["enabled"] = True - else: - # ensure disabled if not requested - config_override["retrievers"].setdefault("latechunk", {})["enabled"] = False - - # 🔧 Configure docling chunking - if enable_docling_chunk: - config_override["chunker_mode"] = "docling" - - # 🔧 Configure contextual enrichment (THIS WAS MISSING!) - config_override.setdefault("contextual_enricher", {}) - config_override["contextual_enricher"]["enabled"] = enable_enrich - config_override["contextual_enricher"]["window_size"] = window_size - - # 🔧 Configure indexing batch sizes - config_override.setdefault("indexing", {}) - config_override["indexing"]["embedding_batch_size"] = batch_size_embed - config_override["indexing"]["enrichment_batch_size"] = batch_size_enrich - - # 🔧 Configure chunking parameters - config_override.setdefault("chunking", {}) - config_override["chunking"]["chunk_size"] = chunk_size - config_override["chunking"]["chunk_overlap"] = chunk_overlap - - # 🔧 Configure embedding model if specified - if embedding_model: - config_override["embedding_model_name"] = embedding_model - - # 🔧 Configure enrichment model if specified - if enrich_model: - config_override["enrich_model"] = enrich_model - - # 🔧 Overview model (can differ from enrichment) - if overview_model: - config_override["overview_model_name"] = overview_model - - print(f"🔧 INDEXING CONFIG: Contextual Enrichment: {enable_enrich}, Window Size: {window_size}") - print(f"🔧 CHUNKING CONFIG: Size: {chunk_size}, Overlap: {chunk_overlap}") - print(f"🔧 MODEL CONFIG: Embedding: {embedding_model or 'default'}, Enrichment: {enrich_model or 'default'}") - - # 🔧 Set index-specific overview path so each index writes separate file - if session_id: - config_override["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # 🔧 Configure late chunking - config_override.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # Create a temporary pipeline instance with the overridden config - temp_pipeline = INDEXING_PIPELINE.__class__( - config_override, - INDEXING_PIPELINE.llm_client, - INDEXING_PIPELINE.ollama_config - ) - temp_pipeline.run(file_paths) - else: - # Use the default pipeline with overrides - import copy - config_override = copy.deepcopy(INDEXING_PIPELINE.config) - - # 🔧 Configure late chunking - if enable_latechunk: - config_override.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # 🔧 Configure docling chunking - if enable_docling_chunk: - config_override["chunker_mode"] = "docling" - - # 🔧 Configure contextual enrichment (THIS WAS MISSING!) - config_override.setdefault("contextual_enricher", {}) - config_override["contextual_enricher"]["enabled"] = enable_enrich - config_override["contextual_enricher"]["window_size"] = window_size - - # 🔧 Configure indexing batch sizes - config_override.setdefault("indexing", {}) - config_override["indexing"]["embedding_batch_size"] = batch_size_embed - config_override["indexing"]["enrichment_batch_size"] = batch_size_enrich - - # 🔧 Configure chunking parameters - config_override.setdefault("chunking", {}) - config_override["chunking"]["chunk_size"] = chunk_size - config_override["chunking"]["chunk_overlap"] = chunk_overlap - - # 🔧 Configure embedding model if specified - if embedding_model: - config_override["embedding_model_name"] = embedding_model - - # 🔧 Configure enrichment model if specified - if enrich_model: - config_override["enrich_model"] = enrich_model - - # 🔧 Overview model (can differ from enrichment) - if overview_model: - config_override["overview_model_name"] = overview_model - - print(f"🔧 INDEXING CONFIG: Contextual Enrichment: {enable_enrich}, Window Size: {window_size}") - print(f"🔧 CHUNKING CONFIG: Size: {chunk_size}, Overlap: {chunk_overlap}") - print(f"🔧 MODEL CONFIG: Embedding: {embedding_model or 'default'}, Enrichment: {enrich_model or 'default'}") - - # 🔧 Set index-specific overview path so each index writes separate file - if session_id: - config_override["overview_path"] = f"index_store/overviews/{session_id}.jsonl" - - # 🔧 Configure late chunking - config_override.setdefault("retrievers", {}).setdefault("latechunk", {})["enabled"] = True - - # Create temporary pipeline with overridden config - temp_pipeline = INDEXING_PIPELINE.__class__( - config_override, - INDEXING_PIPELINE.llm_client, - INDEXING_PIPELINE.ollama_config - ) - temp_pipeline.run(file_paths) + retrieval_mode, error = _resolve_retrieval_mode(data) + if error: + self.send_json_response({"error": error}, status_code=400) + return + + session_id = data.get('session_id') + options = { + "session_id": session_id, + "enable_latechunk": bool(data.get("enable_latechunk", False)), + "enable_docling_chunk": bool(data.get("enable_docling_chunk", True)), + "chunk_size": int(data.get("chunk_size", 512)), + "retrieval_mode": retrieval_mode, + "window_size": int(data.get("window_size", 2)), + "enable_enrich": bool(data.get("enable_enrich", True)), + "embedding_model": data.get('embedding_model'), + "enrich_model": data.get('enrich_model'), + "overview_model_name": data.get('overview_model_name'), + "batch_size_embed": int(data.get("batch_size_embed", 50)), + "batch_size_enrich": int(data.get("batch_size_enrich", 25)), + } + + table_name = data.get('table_name') or _get_table_name_for_session(session_id) + + config_override = _build_index_config_override( + INDEXING_PIPELINE.config, table_name=table_name, options=options + ) + + logger.info( + "Indexing %d file(s) | table=%s | enrich=%s (window %s) | latechunk=%s | " + "chunk_size=%s | embedding=%s | enrichment=%s", + len(file_paths), table_name or DEFAULT_TEXT_TABLE, options["enable_enrich"], + options["window_size"], options["enable_latechunk"], options["chunk_size"], + options["embedding_model"] or config_override.get("embedding_model_name"), + options["enrich_model"] or "default", + ) + + temp_pipeline = INDEXING_PIPELINE.__class__( + config_override, + INDEXING_PIPELINE.llm_client, + INDEXING_PIPELINE.ollama_config, + ) + temp_pipeline.run(file_paths) + + if options["embedding_model"] and session_id: + try: + db.update_index_metadata(session_id, {"embedding_model": options["embedding_model"]}) + except Exception as e: + logger.warning("Could not update embedding_model metadata: %s", e) self.send_json_response({ "message": f"Indexing process for {len(file_paths)} file(s) completed successfully.", - "table_name": table_name or "default_text_table", - "latechunk": enable_latechunk, - "docling_chunk": enable_docling_chunk, + "table_name": table_name or DEFAULT_TEXT_TABLE, + "latechunk": options["enable_latechunk"], + "docling_chunk": options["enable_docling_chunk"], "indexing_config": { - "chunk_size": chunk_size, - "chunk_overlap": chunk_overlap, - "retrieval_mode": retrieval_mode, - "window_size": window_size, - "enable_enrich": enable_enrich, - "embedding_model": embedding_model, - "enrich_model": enrich_model, - "batch_size_embed": batch_size_embed, - "batch_size_enrich": batch_size_enrich + "chunk_size": options["chunk_size"], + "retrieval_mode": config_override.get("retrieval", {}).get("search_type"), + "window_size": options["window_size"], + "enable_enrich": options["enable_enrich"], + "embedding_model": config_override.get("embedding_model_name"), + "enrich_model": options["enrich_model"], + "overview_model_name": options["overview_model_name"], + "batch_size_embed": options["batch_size_embed"], + "batch_size_enrich": options["batch_size_enrich"], } }) - if embedding_model: - try: - db.update_index_metadata(session_id, {"embedding_model": embedding_model}) - except Exception as e: - print(f"⚠️ Could not update embedding_model metadata: {e}") - except json.JSONDecodeError: self.send_json_response({"error": "Invalid JSON"}, status_code=400) except Exception as e: + logger.exception("Indexing request failed") self.send_json_response({"error": f"Failed to start indexing: {str(e)}"}, status_code=500) def handle_models(self): - """Return a list of locally installed Ollama models and supported HuggingFace models, grouped by capability.""" + """Return the generation and embedding models available to the active backend.""" try: generation_models = [] - embedding_models = [] - - # Get Ollama models if available - try: - resp = requests.get(f"{RAG_AGENT.ollama_config['host']}/api/tags", timeout=5) - resp.raise_for_status() - data = resp.json() - - all_ollama_models = [m.get('name') for m in data.get('models', [])] - - # Very naive classification - ollama_embedding_models = [m for m in all_ollama_models if any(k in m for k in ['embed','bge','embedding','text'])] - ollama_generation_models = [m for m in all_ollama_models if m not in ollama_embedding_models] - - generation_models.extend(ollama_generation_models) - embedding_models.extend(ollama_embedding_models) - except Exception as e: - print(f"⚠️ Could not get Ollama models: {e}") - - # Add supported HuggingFace embedding models - huggingface_embedding_models = [ + embedding_models = [RAG_AGENT.retrieval_pipeline.config.get("embedding_model_name")] + + if LLM_BACKEND.lower() == "watsonx": + generation_models.extend( + m for m in (WATSONX_CONFIG.get("generation_model"), WATSONX_CONFIG.get("enrichment_model")) if m + ) + else: + host = RAG_AGENT.ollama_config.get('host') + try: + resp = requests.get(f"{host}/api/tags", timeout=OLLAMA_TIMEOUT_SECONDS) + resp.raise_for_status() + all_ollama_models = [m.get('name') for m in resp.json().get('models', []) if m.get('name')] + + ollama_embedding_models = [ + m for m in all_ollama_models if any(k in m for k in ['embed', 'bge', 'embedding']) + ] + generation_models.extend(m for m in all_ollama_models if m not in ollama_embedding_models) + embedding_models.extend(ollama_embedding_models) + except Exception as e: + logger.warning("Could not list Ollama models from %s: %s", host, e) + + # HuggingFace embedding models loaded in-process (see rag_system/main.py EXTERNAL_MODELS). + # harrier-oss-v1-0.6b is the shipped default; the Qwen3 family stays + # selectable for multilingual / long-context corpora. + embedding_models.extend([ + "microsoft/harrier-oss-v1-0.6b", + "Qwen/Qwen3-Embedding-4B", "Qwen/Qwen3-Embedding-0.6B", - "Qwen/Qwen3-Embedding-4B", - "Qwen/Qwen3-Embedding-8B" - ] - embedding_models.extend(huggingface_embedding_models) - - # Sort models for consistent ordering - generation_models.sort() - embedding_models.sort() + "Qwen/Qwen3-Embedding-8B", + ]) self.send_json_response({ - "generation_models": generation_models, - "embedding_models": embedding_models + "generation_models": sorted(set(generation_models)), + "embedding_models": sorted({m for m in embedding_models if m}), }) except Exception as e: self.send_json_response({"error": f"Could not list models: {e}"}, status_code=500) @@ -740,6 +525,7 @@ def send_json_response(self, data, status_code=200): response = json.dumps(data, indent=2) self.wfile.write(response.encode('utf-8')) + def start_server(port=8001): """Starts the API server.""" # Use a reusable TCP server to avoid "address in use" errors on restart @@ -748,10 +534,12 @@ class ReusableTCPServer(socketserver.TCPServer): with ReusableTCPServer(("", port), AdvancedRagApiHandler) as httpd: print(f"🚀 Starting Advanced RAG API server on port {port}") + print(f"🩺 Health endpoint: http://localhost:{port}/health") print(f"💬 Chat endpoint: http://localhost:{port}/chat") print(f"✨ Indexing endpoint: http://localhost:{port}/index") httpd.serve_forever() + if __name__ == "__main__": # To run this server: python -m rag_system.api_server - start_server() \ No newline at end of file + start_server() diff --git a/rag_system/api_server_with_progress.py b/rag_system/api_server_with_progress.py deleted file mode 100644 index 438dcb34..00000000 --- a/rag_system/api_server_with_progress.py +++ /dev/null @@ -1,443 +0,0 @@ -import json -import threading -import time -from typing import Dict, List, Any -import logging -from urllib.parse import urlparse, parse_qs -import http.server -import socketserver - -# Import the core logic and batch processing utilities -from rag_system.main import get_agent -from rag_system.utils.batch_processor import ProgressTracker, timer - -# Set up logging -logging.basicConfig(level=logging.INFO) -logger = logging.getLogger(__name__) - -# Global progress tracking storage -ACTIVE_PROGRESS_SESSIONS: Dict[str, Dict[str, Any]] = {} - -# --- Global Singleton for the RAG Agent --- -print("🧠 Initializing RAG Agent... (This may take a moment)") -RAG_AGENT = get_agent() -if RAG_AGENT is None: - print("❌ Critical error: RAG Agent could not be initialized. Exiting.") - exit(1) -print("✅ RAG Agent initialized successfully.") - -class ServerSentEventsHandler: - """Handler for Server-Sent Events (SSE) for real-time progress updates""" - - active_connections: Dict[str, Any] = {} - - @classmethod - def add_connection(cls, session_id: str, response_handler): - """Add a new SSE connection""" - cls.active_connections[session_id] = response_handler - logger.info(f"SSE connection added for session: {session_id}") - - @classmethod - def remove_connection(cls, session_id: str): - """Remove an SSE connection""" - if session_id in cls.active_connections: - del cls.active_connections[session_id] - logger.info(f"SSE connection removed for session: {session_id}") - - @classmethod - def send_event(cls, session_id: str, event_type: str, data: Dict[str, Any]): - """Send an SSE event to a specific session""" - if session_id not in cls.active_connections: - return - - try: - handler = cls.active_connections[session_id] - event_data = json.dumps(data) - message = f"event: {event_type}\ndata: {event_data}\n\n" - handler.wfile.write(message.encode('utf-8')) - handler.wfile.flush() - except Exception as e: - logger.error(f"Failed to send SSE event: {e}") - cls.remove_connection(session_id) - -class RealtimeProgressTracker(ProgressTracker): - """Enhanced ProgressTracker that sends updates via Server-Sent Events""" - - def __init__(self, total_items: int, operation_name: str, session_id: str): - super().__init__(total_items, operation_name) - self.session_id = session_id - self.last_update = 0 - self.update_interval = 1 # Update every 1 second - - # Initialize session progress - ACTIVE_PROGRESS_SESSIONS[session_id] = { - "operation_name": operation_name, - "total_items": total_items, - "processed_items": 0, - "errors_encountered": 0, - "start_time": self.start_time, - "status": "running", - "current_step": "", - "eta_seconds": 0, - "throughput": 0, - "progress_percentage": 0 - } - - # Send initial progress update - self._send_progress_update() - - def update(self, items_processed: int, errors: int = 0, current_step: str = ""): - """Update progress and send notification""" - super().update(items_processed, errors) - - # Update session data - session_data = ACTIVE_PROGRESS_SESSIONS.get(self.session_id) - if session_data: - session_data.update({ - "processed_items": self.processed_items, - "errors_encountered": self.errors_encountered, - "current_step": current_step, - "progress_percentage": (self.processed_items / self.total_items) * 100, - }) - - # Calculate throughput and ETA - elapsed = time.time() - self.start_time - if elapsed > 0: - session_data["throughput"] = self.processed_items / elapsed - remaining = self.total_items - self.processed_items - session_data["eta_seconds"] = remaining / session_data["throughput"] if session_data["throughput"] > 0 else 0 - - # Send update if enough time has passed - current_time = time.time() - if current_time - self.last_update >= self.update_interval: - self._send_progress_update() - self.last_update = current_time - - def finish(self): - """Mark progress as finished and send final update""" - super().finish() - - # Update session status - session_data = ACTIVE_PROGRESS_SESSIONS.get(self.session_id) - if session_data: - session_data.update({ - "status": "completed", - "progress_percentage": 100, - "eta_seconds": 0 - }) - - # Send final update - self._send_progress_update(final=True) - - def _send_progress_update(self, final: bool = False): - """Send progress update via Server-Sent Events""" - session_data = ACTIVE_PROGRESS_SESSIONS.get(self.session_id, {}) - - event_data = { - "session_id": self.session_id, - "progress": session_data.copy(), - "final": final, - "timestamp": time.time() - } - - ServerSentEventsHandler.send_event(self.session_id, "progress", event_data) - -def run_indexing_with_progress(file_paths: List[str], session_id: str): - """Enhanced indexing function with real-time progress tracking""" - from rag_system.pipelines.indexing_pipeline import IndexingPipeline - from rag_system.utils.ollama_client import OllamaClient - import json - - try: - # Send initial status - ServerSentEventsHandler.send_event(session_id, "status", { - "message": "Initializing indexing pipeline...", - "session_id": session_id - }) - - # Load configuration - config_file = "batch_indexing_config.json" - try: - with open(config_file, 'r') as f: - config = json.load(f) - except FileNotFoundError: - # Fallback to default config - config = { - "embedding_model_name": "Qwen/Qwen3-Embedding-0.6B", - "indexing": { - "embedding_batch_size": 50, - "enrichment_batch_size": 10, - "enable_progress_tracking": True - }, - "contextual_enricher": {"enabled": True, "window_size": 1}, - "retrievers": { - "dense": {"enabled": True, "lancedb_table_name": "default_text_table"}, - "bm25": {"enabled": True, "index_name": "default_bm25_index"} - }, - "storage": { - "chunk_store_path": "./index_store/chunks/chunks.pkl", - "lancedb_uri": "./index_store/lancedb", - "bm25_path": "./index_store/bm25" - } - } - - # Initialize components - ollama_client = OllamaClient() - ollama_config = { - "generation_model": "llama3.2:1b", - "embedding_model": "mxbai-embed-large" - } - - # Create enhanced pipeline - pipeline = IndexingPipeline(config, ollama_client, ollama_config) - - # Create progress tracker for the overall process - total_steps = 6 # Rough estimate of pipeline steps - step_tracker = RealtimeProgressTracker(total_steps, "Document Indexing", session_id) - - with timer("Complete Indexing Pipeline"): - try: - # Step 1: Document Processing - step_tracker.update(1, current_step="Processing documents...") - - # Run the indexing pipeline - pipeline.run(file_paths) - - # Update progress through the steps - step_tracker.update(1, current_step="Chunking completed...") - step_tracker.update(1, current_step="BM25 indexing completed...") - step_tracker.update(1, current_step="Contextual enrichment completed...") - step_tracker.update(1, current_step="Vector embeddings completed...") - step_tracker.update(1, current_step="Indexing finalized...") - - step_tracker.finish() - - # Send completion notification - ServerSentEventsHandler.send_event(session_id, "completion", { - "message": f"Successfully indexed {len(file_paths)} file(s)", - "file_count": len(file_paths), - "session_id": session_id - }) - - except Exception as e: - # Send error notification - ServerSentEventsHandler.send_event(session_id, "error", { - "message": str(e), - "session_id": session_id - }) - raise - - except Exception as e: - logger.error(f"Indexing failed for session {session_id}: {e}") - ServerSentEventsHandler.send_event(session_id, "error", { - "message": str(e), - "session_id": session_id - }) - raise - -class EnhancedRagApiHandler(http.server.BaseHTTPRequestHandler): - """Enhanced API handler with progress tracking support""" - - def do_OPTIONS(self): - """Handle CORS preflight requests for frontend integration.""" - self.send_response(200) - self.send_header('Access-Control-Allow-Origin', '*') - self.send_header('Access-Control-Allow-Methods', 'POST, GET, OPTIONS') - self.send_header('Access-Control-Allow-Headers', 'Content-Type') - self.end_headers() - - def do_GET(self): - """Handle GET requests for progress status and SSE streams""" - parsed_path = urlparse(self.path) - - if parsed_path.path == '/progress': - self.handle_progress_status() - elif parsed_path.path == '/stream': - self.handle_progress_stream() - else: - self.send_json_response({"error": "Not Found"}, status_code=404) - - def do_POST(self): - """Handle POST requests for chat and indexing.""" - parsed_path = urlparse(self.path) - - if parsed_path.path == '/chat': - self.handle_chat() - elif parsed_path.path == '/index': - self.handle_index_with_progress() - else: - self.send_json_response({"error": "Not Found"}, status_code=404) - - def handle_chat(self): - """Handles a chat query by calling the agentic RAG pipeline.""" - try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - query = data.get('query') - if not query: - self.send_json_response({"error": "Query is required"}, status_code=400) - return - - # Use the single, persistent agent instance to run the query - result = RAG_AGENT.run(query) - - # The result is a dict, so we need to dump it to a JSON string - self.send_json_response(result) - - except json.JSONDecodeError: - self.send_json_response({"error": "Invalid JSON"}, status_code=400) - except Exception as e: - self.send_json_response({"error": f"Server error: {str(e)}"}, status_code=500) - - def handle_index_with_progress(self): - """Triggers the document indexing pipeline with real-time progress tracking.""" - try: - content_length = int(self.headers['Content-Length']) - post_data = self.rfile.read(content_length) - data = json.loads(post_data.decode('utf-8')) - - file_paths = data.get('file_paths') - session_id = data.get('session_id') - - if not file_paths or not isinstance(file_paths, list): - self.send_json_response({ - "error": "A 'file_paths' list is required." - }, status_code=400) - return - - if not session_id: - self.send_json_response({ - "error": "A 'session_id' is required for progress tracking." - }, status_code=400) - return - - # Start indexing in a separate thread to avoid blocking - def run_indexing_thread(): - try: - run_indexing_with_progress(file_paths, session_id) - except Exception as e: - logger.error(f"Indexing thread failed: {e}") - - thread = threading.Thread(target=run_indexing_thread) - thread.daemon = True - thread.start() - - # Return immediate response - self.send_json_response({ - "message": f"Indexing started for {len(file_paths)} file(s)", - "session_id": session_id, - "status": "started", - "progress_stream_url": f"http://localhost:8001/stream?session_id={session_id}" - }) - - except json.JSONDecodeError: - self.send_json_response({"error": "Invalid JSON"}, status_code=400) - except Exception as e: - self.send_json_response({"error": f"Failed to start indexing: {str(e)}"}, status_code=500) - - def handle_progress_status(self): - """Handle GET requests for current progress status""" - parsed_url = urlparse(self.path) - params = parse_qs(parsed_url.query) - session_id = params.get('session_id', [None])[0] - - if not session_id: - self.send_json_response({"error": "session_id is required"}, status_code=400) - return - - progress_data = ACTIVE_PROGRESS_SESSIONS.get(session_id) - if not progress_data: - self.send_json_response({"error": "No active progress for this session"}, status_code=404) - return - - self.send_json_response({ - "session_id": session_id, - "progress": progress_data - }) - - def handle_progress_stream(self): - """Handle Server-Sent Events stream for real-time progress""" - parsed_url = urlparse(self.path) - params = parse_qs(parsed_url.query) - session_id = params.get('session_id', [None])[0] - - if not session_id: - self.send_response(400) - self.end_headers() - return - - # Set up SSE headers - self.send_response(200) - self.send_header('Content-Type', 'text/event-stream') - self.send_header('Cache-Control', 'no-cache') - self.send_header('Connection', 'keep-alive') - self.send_header('Access-Control-Allow-Origin', '*') - self.end_headers() - - # Add this connection to the SSE handler - ServerSentEventsHandler.add_connection(session_id, self) - - # Send initial connection message - initial_message = json.dumps({ - "session_id": session_id, - "message": "Progress stream connected", - "timestamp": time.time() - }) - self.wfile.write(f"event: connected\ndata: {initial_message}\n\n".encode('utf-8')) - self.wfile.flush() - - # Keep connection alive - try: - while session_id in ServerSentEventsHandler.active_connections: - time.sleep(1) - # Send heartbeat - heartbeat = json.dumps({"type": "heartbeat", "timestamp": time.time()}) - self.wfile.write(f"event: heartbeat\ndata: {heartbeat}\n\n".encode('utf-8')) - self.wfile.flush() - except Exception as e: - logger.info(f"SSE connection closed for session {session_id}: {e}") - finally: - ServerSentEventsHandler.remove_connection(session_id) - - def send_json_response(self, data, status_code=200): - """Utility to send a JSON response with CORS headers.""" - self.send_response(status_code) - self.send_header('Content-Type', 'application/json') - self.send_header('Access-Control-Allow-Origin', '*') - self.end_headers() - response = json.dumps(data, indent=2) - self.wfile.write(response.encode('utf-8')) - -def start_enhanced_server(port=8000): - """Start the enhanced API server with a reusable TCP socket.""" - - # Use a custom TCPServer that allows address reuse - class ReusableTCPServer(socketserver.TCPServer): - allow_reuse_address = True - - with ReusableTCPServer(("", port), EnhancedRagApiHandler) as httpd: - print(f"🚀 Starting Enhanced RAG API server on port {port}") - print(f"💬 Chat endpoint: http://localhost:{port}/chat") - print(f"✨ Indexing endpoint: http://localhost:{port}/index") - print(f"📊 Progress endpoint: http://localhost:{port}/progress") - print(f"🌊 Progress stream: http://localhost:{port}/stream") - print(f"📈 Real-time progress tracking enabled via Server-Sent Events!") - httpd.serve_forever() - -if __name__ == '__main__': - # Start the server on a dedicated thread - server_thread = threading.Thread(target=start_enhanced_server) - server_thread.daemon = True - server_thread.start() - - print("🚀 Enhanced RAG API server with progress tracking is running.") - print("Press Ctrl+C to stop.") - - # Keep the main thread alive - try: - while True: - time.sleep(1) - except KeyboardInterrupt: - print("\nStopping server...") \ No newline at end of file diff --git a/rag_system/factory.py b/rag_system/factory.py index 77a79e89..6e2e2e62 100644 --- a/rag_system/factory.py +++ b/rag_system/factory.py @@ -1,82 +1,66 @@ +import copy + from dotenv import load_dotenv -def get_agent(mode: str = "default"): - """ - Factory function to get an instance of the RAG agent based on the specified mode. - This uses local imports to prevent circular dependencies. + +def _build_llm_client(): + """Create the LLM client and its config for the active LLM_BACKEND. + + Uses local imports to prevent circular dependencies with rag_system.main. """ - from rag_system.agent.loop import Agent - from rag_system.utils.ollama_client import OllamaClient - from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG, LLM_BACKEND, WATSONX_CONFIG + from rag_system.main import LLM_BACKEND, OLLAMA_CONFIG, WATSONX_CONFIG - load_dotenv() - - # Initialize the appropriate LLM client based on backend configuration if LLM_BACKEND.lower() == "watsonx": from rag_system.utils.watsonx_client import WatsonXClient - + if not WATSONX_CONFIG["api_key"] or not WATSONX_CONFIG["project_id"]: raise ValueError( "Watson X configuration incomplete. Please set WATSONX_API_KEY and WATSONX_PROJECT_ID " "environment variables." ) - - llm_client = WatsonXClient( + + client = WatsonXClient( api_key=WATSONX_CONFIG["api_key"], project_id=WATSONX_CONFIG["project_id"], - url=WATSONX_CONFIG["url"] + url=WATSONX_CONFIG["url"], ) - llm_config = WATSONX_CONFIG - else: - llm_client = OllamaClient(host=OLLAMA_CONFIG["host"]) - llm_config = OLLAMA_CONFIG - - config = PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS['default']) - - if 'storage' not in config: - config['storage'] = { - 'db_path': 'lancedb', - 'text_table_name': 'text_pages_default', - 'image_table_name': 'image_pages' - } - - agent = Agent( - pipeline_configs=config, - llm_client=llm_client, - ollama_config=llm_config + return client, WATSONX_CONFIG + + from rag_system.utils.ollama_client import OllamaClient + + return OllamaClient(host=OLLAMA_CONFIG["host"]), OLLAMA_CONFIG + + +def get_pipeline_config(mode: str = "default") -> dict: + """Return a deep copy of a pipeline profile so callers cannot mutate the master config.""" + from rag_system.main import PIPELINE_CONFIGS + + return copy.deepcopy(PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS["default"])) + + +def get_agent(mode: str = "default"): + """Factory function to get an instance of the RAG agent for the specified mode.""" + from rag_system.agent.loop import Agent + + load_dotenv() + + llm_client, llm_config = _build_llm_client() + config = get_pipeline_config(mode) + + return Agent( + pipeline_configs=config, + llm_client=llm_client, + ollama_config=llm_config, ) - return agent + def get_indexing_pipeline(mode: str = "default"): - """ - Factory function to get an instance of the Indexing Pipeline. - """ + """Factory function to get an instance of the Indexing Pipeline for the specified mode.""" from rag_system.pipelines.indexing_pipeline import IndexingPipeline - from rag_system.main import PIPELINE_CONFIGS, OLLAMA_CONFIG, LLM_BACKEND, WATSONX_CONFIG - from rag_system.utils.ollama_client import OllamaClient load_dotenv() - - # Initialize the appropriate LLM client based on backend configuration - if LLM_BACKEND.lower() == "watsonx": - from rag_system.utils.watsonx_client import WatsonXClient - - if not WATSONX_CONFIG["api_key"] or not WATSONX_CONFIG["project_id"]: - raise ValueError( - "Watson X configuration incomplete. Please set WATSONX_API_KEY and WATSONX_PROJECT_ID " - "environment variables." - ) - - llm_client = WatsonXClient( - api_key=WATSONX_CONFIG["api_key"], - project_id=WATSONX_CONFIG["project_id"], - url=WATSONX_CONFIG["url"] - ) - llm_config = WATSONX_CONFIG - else: - llm_client = OllamaClient(host=OLLAMA_CONFIG["host"]) - llm_config = OLLAMA_CONFIG - - config = PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS['default']) - - return IndexingPipeline(config, llm_client, llm_config) \ No newline at end of file + + llm_client, llm_config = _build_llm_client() + config = get_pipeline_config(mode) + + return IndexingPipeline(config, llm_client, llm_config) diff --git a/rag_system/indexing/contextualizer.py b/rag_system/indexing/contextualizer.py index 714c65d3..5228c01f 100644 --- a/rag_system/indexing/contextualizer.py +++ b/rag_system/indexing/contextualizer.py @@ -141,41 +141,4 @@ def process_chunk_batch(chunk_indices): "Contextual Enrichment" ) - return enriched_chunks - - def enrich_chunks_sequential(self, chunks: List[Dict[str, Any]], window_size: int = 1) -> List[Dict[str, Any]]: - """Sequential enrichment method (legacy) - kept for comparison""" - if not chunks: - return [] - - logger.info(f"Enriching {len(chunks)} chunks sequentially (window_size={window_size})...") - enriched_chunks = [] - - for i, chunk in enumerate(chunks): - local_context_text = create_contextual_window(chunks, chunk_index=i, window_size=window_size) - - # The summary is generated based on the original, unmodified text - original_text = chunk['text'] - summary = self._generate_summary(local_context_text, original_text) - - new_chunk = chunk.copy() - - # Ensure metadata is a dictionary - if 'metadata' not in new_chunk or not isinstance(new_chunk['metadata'], dict): - new_chunk['metadata'] = {} - - # Store original text and summary in metadata - new_chunk['metadata']['original_text'] = original_text - new_chunk['metadata']['contextual_summary'] = "N/A" - - # Prepend the context summary ONLY if it was successfully generated - if summary: - new_chunk['text'] = f"Context: {summary}\n\n---\n\n{original_text}" - new_chunk['metadata']['contextual_summary'] = summary - - enriched_chunks.append(new_chunk) - - if (i + 1) % 10 == 0 or i == len(chunks) - 1: - logger.info(f" ...processed {i+1}/{len(chunks)} chunks.") - return enriched_chunks \ No newline at end of file diff --git a/rag_system/indexing/embedders.py b/rag_system/indexing/embedders.py index b48648f2..b6c3c22f 100644 --- a/rag_system/indexing/embedders.py +++ b/rag_system/indexing/embedders.py @@ -1,10 +1,135 @@ # from rag_system.indexing.representations import BM25Generator import lancedb +import os import pyarrow as pa -from typing import List, Dict, Any +from typing import Any, Dict, List, Optional import numpy as np import json +# --------------------------------------------------------------------------- +# Per-table embedder identity + vector-normalization marker +# --------------------------------------------------------------------------- +# Two facts have to travel with a LanceDB table, because neither can be +# recovered from the vectors themselves: +# +# 1. WHICH embedding model wrote them. The vector-width check below cannot +# catch a swap between two same-width models (harrier-oss-v1-0.6b and +# Qwen3-Embedding-0.6B are both 1024-dim), and appending one model's +# vectors to the other's table silently produces nonsense rankings. +# 2. WHETHER they are L2-normalized. Both model cards specify cosine +# similarity, but LanceDB's default metric is L2; L2 ordering equals +# cosine ordering only when every vector is unit length. Normalizing +# invalidates vectors written before this existed, so it is recorded +# per table rather than assumed globally. +# +# Primary store: Arrow schema metadata on the table, which lancedb 0.36.0 +# round-trips through create_table/open_table (verified on this tree). If a +# LanceDB version ever drops it, a sidecar JSON is written next to the database +# instead — under /table_meta/
.json, *not* a single global +# directory, because different indexes legitimately reuse the same table name +# in different database directories (the eval harness does exactly that). +# +# A table with neither marker is a legacy table: its embedder is unknown and +# its vectors are unnormalized. It keeps working, unnormalized, with a warning. + +_META_MODEL_KEY = b"localgpt_embedding_model" +_META_NORMALIZED_KEY = b"localgpt_normalized" +_SIDECAR_DIRNAME = "table_meta" + + +class EmbedderMismatchError(RuntimeError): + """A table was written by a different embedding model than the configured one.""" + + +def l2_normalize(vector: np.ndarray) -> np.ndarray: + """Unit-length copy of *vector*; returned unchanged when that is impossible.""" + array = np.asarray(vector, dtype=np.float32) + if not np.isfinite(array).all(): + # Leave it alone so the NaN/Inf reporting downstream stays accurate. + return array + norm = float(np.linalg.norm(array)) + if norm == 0.0: + return array + return array / norm + + +def _sidecar_path(db_path: str, table_name: str) -> str: + return os.path.join(db_path, _SIDECAR_DIRNAME, f"{table_name}.json") + + +def _read_sidecar(db_path: Optional[str], table_name: Optional[str]) -> Optional[Dict[str, Any]]: + if not db_path or not table_name: + return None + path = _sidecar_path(db_path, table_name) + try: + with open(path, "r", encoding="utf-8") as fh: + return json.load(fh) + except (OSError, ValueError): + return None + + +def _write_sidecar(db_path: str, table_name: str, model_name: str, normalized: bool) -> None: + path = _sidecar_path(db_path, table_name) + try: + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "w", encoding="utf-8") as fh: + json.dump({"embedding_model": model_name, "normalized": bool(normalized)}, fh, indent=2) + except OSError as e: + print(f"⚠️ Could not write table marker {path}: {e}") + + +def table_schema_metadata(model_name: str, normalized: bool) -> Dict[bytes, bytes]: + return { + _META_MODEL_KEY: model_name.encode("utf-8"), + _META_NORMALIZED_KEY: (b"true" if normalized else b"false"), + } + + +def read_table_marker(tbl, db_path: Optional[str] = None, + table_name: Optional[str] = None) -> Optional[Dict[str, Any]]: + """The embedder identity recorded for *tbl*, or None for a legacy table.""" + metadata = getattr(getattr(tbl, "schema", None), "metadata", None) or {} + raw_model = metadata.get(_META_MODEL_KEY) + if raw_model: + raw_norm = metadata.get(_META_NORMALIZED_KEY, b"false") + return { + "embedding_model": raw_model.decode("utf-8"), + "normalized": raw_norm.decode("utf-8").lower() == "true", + "source": "lancedb schema metadata", + } + sidecar = _read_sidecar(db_path, table_name) + if sidecar and sidecar.get("embedding_model"): + return { + "embedding_model": sidecar["embedding_model"], + "normalized": bool(sidecar.get("normalized")), + "source": _sidecar_path(db_path, table_name), + } + return None + + +def assert_embedder_matches(table_name: str, marker: Dict[str, Any], configured_model: str) -> None: + """Raise when *marker* names a different embedder than *configured_model*.""" + recorded = marker.get("embedding_model") + if not recorded or not configured_model or recorded == configured_model: + return + raise EmbedderMismatchError( + f"Table '{table_name}' was built with embedding model '{recorded}' but the " + f"pipeline is configured for '{configured_model}'. The two produce vectors in " + f"different spaces (a matching vector width does not make them compatible), so " + f"any result from this table would be meaningless. Rebuild the index with " + f"'{configured_model}', or set EMBEDDING_MODEL='{recorded}' to keep using it." + ) + + +def legacy_table_warning(table_name: str) -> str: + return ( + f"⚠️ Table '{table_name}' carries no embedder marker — it was built before " + f"localGPT recorded one. Its embedding model cannot be verified and its vectors " + f"are unnormalized, so scores use legacy unnormalized vectors; a rebuilt index " + f"is recommended." + ) + + class LanceDBManager: def __init__(self, db_path: str): self.db_path = db_path @@ -27,15 +152,59 @@ class VectorIndexer: def __init__(self, db_manager: LanceDBManager): self.db_manager = db_manager - def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.ndarray): + @staticmethod + def _table_vector_dim(tbl) -> int | None: + """Vector width of an existing LanceDB table, or None if it can't be read.""" + try: + field = tbl.schema.field("vector") + except (KeyError, AttributeError): + return None + return getattr(field.type, "list_size", None) + + def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.ndarray, + embedding_model: Optional[str] = None): if len(chunks) != len(embeddings): raise ValueError("The number of chunks and embeddings must be the same.") if not chunks: print("No chunks to index.") return - vector_dim = embeddings[0].shape[0] - + # Dimensionality always comes from the vectors the loaded model produced. + vector_dim = int(embeddings[0].shape[0]) + + db = self.db_manager.db # underlying LanceDB connection + db_path = getattr(self.db_manager, "db_path", None) + table_exists = bool(hasattr(db, "table_names") and table_name in db.table_names()) + + # ------------------------------------------------------------------ + # Decide, before touching the vectors, which table this is: + # new table -> record the embedder, write normalized vectors + # marked table -> embedder must match; follow the table's own flag + # legacy table -> unknown embedder, unnormalized; warn and comply + # ------------------------------------------------------------------ + existing_tbl = None + normalize = bool(embedding_model) # can't claim an identity we weren't given + if table_exists: + existing_tbl = self.db_manager.get_table(table_name) + existing_dim = self._table_vector_dim(existing_tbl) + if existing_dim is not None and existing_dim != vector_dim: + raise ValueError( + f"Table '{table_name}' stores {existing_dim}-dim vectors but the current " + f"embedding model produced {vector_dim}-dim vectors. Changing the embedding " + f"model requires rebuilding the index." + ) + marker = read_table_marker(existing_tbl, db_path, table_name) + if marker is None: + print(legacy_table_warning(table_name)) + normalize = False + else: + if embedding_model: + assert_embedder_matches(table_name, marker, embedding_model) + normalize = bool(marker["normalized"]) + + if normalize: + embeddings = [l2_normalize(v) for v in embeddings] + # The schema stores the text that was used for the embedding (potentially enriched) # and the full metadata object as a JSON string. schema = pa.schema([ @@ -45,7 +214,7 @@ def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.nd pa.field("document_id", pa.string()), pa.field("chunk_index", pa.int32()), pa.field("metadata", pa.string()) - ]) + ], metadata=table_schema_metadata(embedding_model, normalize) if embedding_model else None) data = [] skipped_count = 0 @@ -93,14 +262,20 @@ def index(self, table_name: str, chunks: List[Dict[str, Any]], embeddings: np.nd return # Incremental indexing: append to existing table if present, otherwise create it - db = self.db_manager.db # underlying LanceDB connection - - if hasattr(db, "table_names") and table_name in db.table_names(): - tbl = self.db_manager.get_table(table_name) + if existing_tbl is not None: + tbl = existing_tbl print(f"Appending {len(data)} vectors to existing table '{table_name}'.") else: print(f"Creating table '{table_name}' (new) and adding {len(data)} vectors...") tbl = self.db_manager.create_table(table_name, schema=schema, mode="create") + if embedding_model: + # Trust nothing: re-read the marker off the created table. If this + # LanceDB build dropped the Arrow schema metadata, fall back to the + # sidecar so the guard still has something to compare against. + if read_table_marker(self.db_manager.get_table(table_name)) is None and db_path: + _write_sidecar(db_path, table_name, embedding_model, normalize) + print(f"🔖 Table '{table_name}' marked: embedder='{embedding_model}', " + f"normalized={str(normalize).lower()}.") # Add data with NaN handling configuration try: diff --git a/rag_system/indexing/graph_extractor.py b/rag_system/indexing/graph_extractor.py deleted file mode 100644 index 70084a44..00000000 --- a/rag_system/indexing/graph_extractor.py +++ /dev/null @@ -1,86 +0,0 @@ -from typing import List, Dict, Any -import json -from rag_system.utils.ollama_client import OllamaClient - -class GraphExtractor: - """ - Extracts entities and relationships from text chunks using a live Ollama model. - """ - def __init__(self, llm_client: OllamaClient, llm_model: str): - self.llm_client = llm_client - self.llm_model = llm_model - print(f"Initialized GraphExtractor with Ollama model '{self.llm_model}'.") - - def extract(self, chunks: List[Dict[str, Any]]) -> Dict[str, List[Dict]]: - all_entities = {} - all_relationships = set() - - print(f"Extracting graph from {len(chunks)} chunks with Ollama...") - for i, chunk in enumerate(chunks): - # Step 1: Extract Entities - entity_prompt = f""" - From the following text, extract key entities (people, companies, locations). - Return the answer as a JSON object with a single key 'entities', which is a list of strings. - Each entity should be a short, specific name, not a long string of text. - - Text: "{chunk['text']}" - """ - - entity_response = self.llm_client.generate_completion( - self.llm_model, - entity_prompt, - format="json" - ) - - entity_response_text = entity_response.get('response', '{}') - - try: - entity_data = json.loads(entity_response_text) - entities = entity_data.get('entities', []) - - if not entities: - continue - - # Clean up entities - cleaned_entities = [] - for entity in entities: - if len(entity) < 50 and not any(c in entity for c in "[]{}()"): - cleaned_entities.append(entity) - - if not cleaned_entities: - continue - - # Step 2: Extract Relationships - relationship_prompt = f""" - Given the following entities: {cleaned_entities} - And the following text: "{chunk['text']}" - Extract the relationships between the entities. - Return the answer as a JSON object with a single key 'relationships', which is a list of objects, each with 'source', 'target', and 'label'. - """ - - relationship_response = self.llm_client.generate_completion( - self.llm_model, - relationship_prompt, - format="json" - ) - - relationship_response_text = relationship_response.get('response', '{}') - relationship_data = json.loads(relationship_response_text) - - for entity_name in cleaned_entities: - all_entities[entity_name] = {"id": entity_name, "type": "Unknown"} # Placeholder type - - for rel in relationship_data.get("relationships", []): - if 'source' in rel and 'target' in rel and 'label' in rel: - all_relationships.add( - (rel['source'], rel['target'], rel['label']) - ) - - except json.JSONDecodeError: - print(f"Warning: Could not decode JSON from LLM for chunk {i+1}.") - continue - - return { - "entities": list(all_entities.values()), - "relationships": [{"source": s, "target": t, "label": l} for s, t, l in all_relationships] - } diff --git a/rag_system/indexing/latechunk.py b/rag_system/indexing/latechunk.py index 094a5243..156b8440 100644 --- a/rag_system/indexing/latechunk.py +++ b/rag_system/indexing/latechunk.py @@ -21,21 +21,25 @@ class LateChunkEncoder: """Generate late-chunked embeddings given character-offset spans.""" - def __init__(self, model_name: str = "Qwen/Qwen3-Embedding-0.6B", *, max_tokens: int = 8192) -> None: + def __init__(self, model_name: str | None = None, *, max_tokens: int = 8192) -> None: + if not model_name: + from rag_system.main import EXTERNAL_MODELS + model_name = EXTERNAL_MODELS["embedding_model"] self.model_name = model_name self.max_len = max_tokens - self.device = torch.device("cuda" if torch.cuda.is_available() else "cpu") - # Back-compat: allow short alias without repo namespace - repo_id = model_name - if "/" not in model_name and not model_name.startswith("Qwen/"): - # map common alias to official repo - alias_map = { - "qwen3-embedding-0.6b": "Qwen/Qwen3-Embedding-0.6B", - } - repo_id = alias_map.get(model_name.lower(), model_name) + if torch.cuda.is_available(): + self.device = torch.device("cuda") + elif getattr(torch.backends, "mps", None) and torch.backends.mps.is_available(): + self.device = torch.device("mps") + else: + self.device = torch.device("cpu") - self.tokenizer = AutoTokenizer.from_pretrained(repo_id, trust_remote_code=True) - self.model = AutoModel.from_pretrained(repo_id, trust_remote_code=True) + self.tokenizer = AutoTokenizer.from_pretrained(model_name, trust_remote_code=True) + self.model = AutoModel.from_pretrained( + model_name, + trust_remote_code=True, + torch_dtype=torch.float16 if self.device.type != "cpu" else None, + ) self.model.to(self.device) self.model.eval() @@ -66,7 +70,7 @@ def encode(self, text: str, chunk_spans: List[Tuple[int, int]]) -> List[np.ndarr out = self.model(**inputs) last_hidden = out.last_hidden_state.squeeze(0) # (seq_len, dim) - last_hidden = last_hidden.cpu() + last_hidden = last_hidden.float().cpu() # For each chunk span, gather token indices belonging to it vectors: List[np.ndarray] = [] diff --git a/rag_system/indexing/multimodal.py b/rag_system/indexing/multimodal.py deleted file mode 100644 index b2a89945..00000000 --- a/rag_system/indexing/multimodal.py +++ /dev/null @@ -1,124 +0,0 @@ -import fitz # PyMuPDF -from PIL import Image -import torch -import os -from typing import List, Dict, Any - -from rag_system.indexing.embedders import LanceDBManager, VectorIndexer -from rag_system.indexing.representations import QwenEmbedder - - -from transformers import ColPaliForRetrieval, ColPaliProcessor, Qwen2TokenizerFast - -class LocalVisionModel: - """ - A wrapper for a local vision model (ColPali) from the transformers library. - """ - def __init__(self, model_name: str = "vidore/colqwen2-v1.0", device: str = "cpu"): - print(f"Initializing local vision model '{model_name}' on device '{device}'.") - self.device = device - self.model = ColPaliForRetrieval.from_pretrained(model_name).to(self.device).eval() - self.tokenizer = Qwen2TokenizerFast.from_pretrained(model_name) - self.image_processor = ColPaliProcessor.from_pretrained(model_name).image_processor - self.processor = ColPaliProcessor(tokenizer=self.tokenizer, image_processor=self.image_processor) - print("Local vision model loaded successfully.") - - def embed_image(self, image: Image.Image) -> torch.Tensor: - """ - Generates a multi-vector embedding for a single image. - """ - inputs = self.processor(text="", images=image, return_tensors="pt").to(self.device) - with torch.no_grad(): - image_embeds = self.model.get_image_features(**inputs) - return image_embeds - - -class MultimodalProcessor: - """ - Processes PDFs into separate text and image embeddings using local models. - """ - def __init__(self, vision_model: LocalVisionModel, text_embedder: QwenEmbedder, db_manager: LanceDBManager): - self.vision_model = vision_model - self.text_embedder = text_embedder - self.text_vector_indexer = VectorIndexer(db_manager) - self.image_vector_indexer = VectorIndexer(db_manager) - - def process_and_index( - self, - pdf_path: str, - text_table_name: str, - image_table_name: str - ): - print(f"\n--- Processing PDF for multimodal indexing: {os.path.basename(pdf_path)} ---") - doc = fitz.open(pdf_path) - document_id = os.path.basename(pdf_path) - - all_pages_text_chunks = [] - all_pages_images = [] - - for page_num in range(len(doc)): - page = doc.load_page(page_num) - - # 1. Extract Text - text = page.get_text("text") - if not text.strip(): - text = f"Page {page_num + 1} contains no extractable text." - - all_pages_text_chunks.append({ - "chunk_id": f"{document_id}_page_{page_num+1}", - "text": text, - "metadata": {"document_id": document_id, "page_number": page_num + 1} - }) - - # 2. Extract Image - pix = page.get_pixmap() - img = Image.frombytes("RGB", [pix.width, pix.height], pix.samples) - all_pages_images.append(img) - - # --- Batch Indexing --- - # Index all text chunks - if all_pages_text_chunks: - text_embeddings = self.text_embedder.create_embeddings([c['text'] for c in all_pages_text_chunks]) - self.text_vector_indexer.index(text_table_name, all_pages_text_chunks, text_embeddings) - print(f"Indexed {len(all_pages_text_chunks)} text pages into '{text_table_name}'.") - - # Index all images - if all_pages_images: - image_embeddings = self.vision_model.create_image_embeddings(all_pages_images) - # We use the text chunks as placeholders for metadata - self.image_vector_indexer.index(image_table_name, all_pages_text_chunks, image_embeddings) - print(f"Indexed {len(all_pages_images)} image pages into '{image_table_name}'.") - -if __name__ == '__main__': - # This test requires an internet connection to download the models. - try: - # 1. Setup models and dependencies - text_embedder = QwenEmbedder() - vision_model = LocalVisionModel() - db_manager = LanceDBManager(db_path="./rag_system/index_store/lancedb") - - # 2. Create a dummy PDF - dummy_pdf_path = "multimodal_test.pdf" - doc = fitz.open() - page = doc.new_page() - page.insert_text((50, 72), "This is a test page with text and an image.") - doc.save(dummy_pdf_path) - - # 3. Run the processor - processor = MultimodalProcessor(vision_model, text_embedder, db_manager) - processor.process_and_index( - pdf_path=dummy_pdf_path, - text_table_name="test_text_pages", - image_table_name="test_image_pages" - ) - - # 4. Verify - print("\n--- Verification ---") - text_tbl = db_manager.get_table("test_text_pages") - img_tbl = db_manager.get_table("test_image_pages") - print(f"Text table has {len(text_tbl)} rows.") - print(f"Image table has {len(img_tbl)} rows.") - - except Exception as e: - print(f"\nAn error occurred during the multimodal test: {e}") - print("Please ensure you have an internet connection for model downloads.") \ No newline at end of file diff --git a/rag_system/indexing/overview_builder.py b/rag_system/indexing/overview_builder.py index 27ede9b3..ed7e64e9 100644 --- a/rag_system/indexing/overview_builder.py +++ b/rag_system/indexing/overview_builder.py @@ -18,7 +18,7 @@ class OverviewBuilder: "DOCUMENT_START:\n{text}\n\nOVERVIEW:" ) - def __init__(self, llm_client, model: str = "qwen3:0.6b", first_n_chunks: int = 5, + def __init__(self, llm_client, model: str, first_n_chunks: int = 5, out_path: str | None = None): if out_path is None: out_path = "index_store/overviews/overviews.jsonl" @@ -26,7 +26,9 @@ def __init__(self, llm_client, model: str = "qwen3:0.6b", first_n_chunks: int = self.model = model self.first_n = first_n_chunks self.out_path = out_path - os.makedirs(os.path.dirname(out_path), exist_ok=True) + out_dir = os.path.dirname(out_path) + if out_dir: + os.makedirs(out_dir, exist_ok=True) def build_and_store(self, doc_id: str, chunks: List[Dict[str, Any]]): if not chunks: diff --git a/rag_system/indexing/representations.py b/rag_system/indexing/representations.py index a3e5ce38..6cd65906 100644 --- a/rag_system/indexing/representations.py +++ b/rag_system/indexing/representations.py @@ -11,13 +11,72 @@ def create_embeddings(self, texts: List[str]) -> np.ndarray: ... # Global cache for models - use dict to cache by model name _MODEL_CACHE = {} +# --------------------------------------------------------------------------- +# Query-side instruction prefix +# --------------------------------------------------------------------------- +# Instruction-tuned decoder embedders (Qwen3-Embedding, microsoft/harrier-oss-v1) +# are trained with an instruction on the QUERY side only. Both model cards use +# the identical wire format and the identical MS-MARCO-style retrieval task +# string, and both state explicitly that documents must be embedded WITHOUT any +# instruction. +# +# query -> "Instruct: {task}\nQuery: {text}" +# document -> "{text}" (unchanged, always) +# +# The asymmetry is what keeps this change index-compatible: an index built +# before the prefix existed stays valid, because nothing on the document side +# moves. Only the query vector changes. +# +# An embedder instance carries the instruction; it does not decide per call. +# The indexing pipeline builds its embedder without one, the retrieval pipeline +# builds its embedder with one, and neither can leak into the other. + +QUERY_PROMPT_TEMPLATE = "Instruct: {instruction}\nQuery: {text}" + +# The official retrieval task description used by both model families. +DEFAULT_RETRIEVAL_INSTRUCTION = ( + "Given a web search query, retrieve relevant passages that answer the query" +) + +# Model-name fragments whose families are instruction-tuned in this format. +_INSTRUCTION_TUNED_FAMILIES = ("qwen3-embedding", "harrier") + + +def default_query_instruction(model_name: str) -> str: + """The retrieval instruction a model family expects, or "" when it wants none. + + Returning "" (not None) is deliberate: "" means "this model takes no + instruction", which is a decision, whereas None means "nobody decided yet" + and is what callers pass to ask for this default. + """ + name = (model_name or "").lower() + if any(fragment in name for fragment in _INSTRUCTION_TUNED_FAMILIES): + return DEFAULT_RETRIEVAL_INSTRUCTION + return "" + + +def apply_query_instruction(texts: List[str], instruction: str | None) -> List[str]: + """Prefix every text with the instruction block, or return them untouched.""" + if not instruction: + return texts + return [QUERY_PROMPT_TEMPLATE.format(instruction=instruction, text=t) for t in texts] + # --- New Ollama Embedder --- class QwenEmbedder(EmbeddingModel): """ An embedding model that uses a local Hugging Face transformer model. """ - def __init__(self, model_name: str = "Qwen/Qwen3-Embedding-0.6B"): + MAX_TOKENS = 8192 + + def __init__(self, model_name: str | None = None, query_instruction: str | None = None): + if not model_name: + from rag_system.main import EXTERNAL_MODELS + model_name = EXTERNAL_MODELS["embedding_model"] self.model_name = model_name + # "" / None => this instance embeds raw text (the document side). + # A non-empty string => this instance is a QUERY embedder and prefixes + # every text it is given with the instruction block. + self.query_instruction = query_instruction or "" # Auto-select the best available device: CUDA > MPS > CPU if torch.cuda.is_available(): self.device = "cuda" @@ -39,22 +98,40 @@ def __init__(self, model_name: str = "Qwen/Qwen3-Embedding-0.6B"): print(f"QwenEmbedder weights loaded and cached for {model_name}.") else: print(f"Reusing cached QwenEmbedder weights for {model_name}.") - + self.tokenizer, self.model = _MODEL_CACHE[model_name] + # Some tokenizers report a sentinel model_max_length; clamp it so that + # truncation=True actually truncates. + reported = getattr(self.tokenizer, "model_max_length", None) + self.max_length = min(reported, self.MAX_TOKENS) if isinstance(reported, int) and reported > 0 else self.MAX_TOKENS def create_embeddings(self, texts: List[str]) -> np.ndarray: print(f"Generating {len(texts)} embeddings with {self.model_name} model...") - inputs = self.tokenizer(texts, padding=True, truncation=True, return_tensors="pt").to(self.device) + texts = apply_query_instruction(texts, self.query_instruction) + inputs = self.tokenizer( + texts, + padding=True, + truncation=True, + max_length=self.max_length, + return_tensors="pt", + ).to(self.device) with torch.no_grad(): outputs = self.model(**inputs) last_hidden = outputs.last_hidden_state # [B, seq, dim] - # Pool via last valid token per sequence (recommended for Qwen3) - seq_len = inputs["attention_mask"].sum(dim=1) - 1 # index of last token - batch_indices = torch.arange(last_hidden.size(0), device=self.device) - embeddings = last_hidden[batch_indices, seq_len] - + # Last-token pooling (recommended for Qwen3-Embedding). The tokenizer + # pads on the left, in which case the final column is the last real + # token for every row; handle right padding too for safety. + attention_mask = inputs["attention_mask"] + left_padded = bool(attention_mask[:, -1].min().item()) + if left_padded: + embeddings = last_hidden[:, -1] + else: + seq_len = attention_mask.sum(dim=1) - 1 # index of last token + batch_indices = torch.arange(last_hidden.size(0), device=last_hidden.device) + embeddings = last_hidden[batch_indices, seq_len] + # Convert to numpy and validate - embeddings_np = embeddings.cpu().numpy() + embeddings_np = embeddings.float().cpu().numpy() # Check for NaN or infinite values if np.isnan(embeddings_np).any(): @@ -105,10 +182,13 @@ def process_text_batch(text_batch): class OllamaEmbedder(EmbeddingModel): """Call Ollama's /api/embeddings endpoint for each text.""" - def __init__(self, model_name: str, host: str | None = None, timeout: int = 60): + def __init__(self, model_name: str, host: str | None = None, timeout: int = 60, + query_instruction: str | None = None): self.model_name = model_name self.host = (host or os.getenv("OLLAMA_HOST") or "http://localhost:11434").rstrip("/") self.timeout = timeout + # Same contract as QwenEmbedder: set on the query-side instance only. + self.query_instruction = query_instruction or "" def _embed_single(self, text: str): import requests, numpy as np, json @@ -124,6 +204,7 @@ def _embed_single(self, text: str): def create_embeddings(self, texts: List[str]): import numpy as np + texts = apply_query_instruction(texts, self.query_instruction) vectors = [self._embed_single(t) for t in texts] embeddings_np = np.vstack(vectors) @@ -142,13 +223,21 @@ def create_embeddings(self, texts: List[str]): return embeddings_np -def select_embedder(model_name: str, ollama_host: str | None = None): - """Return appropriate EmbeddingModel implementation for the given name.""" +def select_embedder(model_name: str, ollama_host: str | None = None, + query_instruction: str | None = None): + """Return appropriate EmbeddingModel implementation for the given name. + + ``query_instruction`` is the query-side instruction prefix. Leave it unset + (the default) for document-side embedders — that is what keeps indexes + stable across this change. Callers on the query path pass the instruction + explicitly; see ``RetrievalPipeline._get_text_embedder``. + """ if "/" in model_name or model_name.startswith("http"): # Treat as HF model path - return QwenEmbedder(model_name=model_name) + return QwenEmbedder(model_name=model_name, query_instruction=query_instruction) # Otherwise assume it's an Ollama tag - return OllamaEmbedder(model_name=model_name, host=ollama_host) + return OllamaEmbedder(model_name=model_name, host=ollama_host, + query_instruction=query_instruction) if __name__ == '__main__': print("representations.py cleaned up.") diff --git a/rag_system/ingestion/chunking.py b/rag_system/ingestion/chunking.py index 6e55ff0a..b9c1f5a9 100644 --- a/rag_system/ingestion/chunking.py +++ b/rag_system/ingestion/chunking.py @@ -8,21 +8,19 @@ class MarkdownRecursiveChunker: and embeds document-level metadata into each chunk. """ - def __init__(self, max_chunk_size: int = 1500, min_chunk_size: int = 200, tokenizer_model: str = "Qwen/Qwen3-Embedding-0.6B"): + def __init__(self, max_chunk_size: int = 1500, min_chunk_size: int = 200, tokenizer_model: str | None = None): self.max_chunk_size = max_chunk_size self.min_chunk_size = min_chunk_size self.split_priority = ["\n## ", "\n### ", "\n#### ", "```", "\n\n"] - - repo_id = tokenizer_model - if "/" not in tokenizer_model and not tokenizer_model.startswith("Qwen/"): - repo_id = { - "qwen3-embedding-0.6b": "Qwen/Qwen3-Embedding-0.6B", - }.get(tokenizer_model.lower(), tokenizer_model) - + + if not tokenizer_model: + from rag_system.main import EXTERNAL_MODELS + tokenizer_model = EXTERNAL_MODELS["embedding_model"] + try: - self.tokenizer = AutoTokenizer.from_pretrained(repo_id, trust_remote_code=True) + self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_model, trust_remote_code=True) except Exception as e: - print(f"Warning: Failed to load tokenizer {repo_id}: {e}") + print(f"Warning: Failed to load tokenizer {tokenizer_model}: {e}") print("Falling back to character-based approximation (4 chars ≈ 1 token)") self.tokenizer = None diff --git a/rag_system/ingestion/docling_chunker.py b/rag_system/ingestion/docling_chunker.py index 4a27ff44..47b2b303 100644 --- a/rag_system/ingestion/docling_chunker.py +++ b/rag_system/ingestion/docling_chunker.py @@ -1,40 +1,40 @@ from __future__ import annotations -"""Docling-aware chunker (simplified). +"""Docling-aware chunker. -For now we proxy the old MarkdownRecursiveChunker but add: -• sentence-aware packing to max_tokens with overlap -• breadcrumb metadata stubs so downstream code already handles them +Two entry points: +• chunk_document(doc) walks a DoclingDocument element tree, emitting tables and + code as atomic chunks and token-packing paragraphs up to max_tokens. +• chunk()/split_markdown() fall back to MarkdownRecursiveChunker plus + sentence-aware packing when only Markdown is available. -In a follow-up we can replace the internals with true Docling element-tree -walking once the PDFConverter returns structured nodes. +Both attach heading-path / block-type metadata to every chunk. """ -from typing import List, Dict, Any, Tuple -import math +from typing import List, Dict, Any import re -from itertools import islice from rag_system.ingestion.chunking import MarkdownRecursiveChunker from transformers import AutoTokenizer class DoclingChunker: - def __init__(self, *, max_tokens: int = 512, overlap: int = 1, tokenizer_model: str = "Qwen/Qwen3-Embedding-0.6B"): + def __init__(self, *, max_tokens: int = 512, overlap: int = 1, tokenizer_model: str | None = None): self.max_tokens = max_tokens self.overlap = overlap # sentences of overlap - repo_id = tokenizer_model - if "/" not in tokenizer_model and not tokenizer_model.startswith("Qwen/"): - repo_id = { - "qwen3-embedding-0.6b": "Qwen/Qwen3-Embedding-0.6B", - }.get(tokenizer_model.lower(), tokenizer_model) - + + if not tokenizer_model: + from rag_system.main import EXTERNAL_MODELS + tokenizer_model = EXTERNAL_MODELS["embedding_model"] + try: - self.tokenizer = AutoTokenizer.from_pretrained(repo_id, trust_remote_code=True) + self.tokenizer = AutoTokenizer.from_pretrained(tokenizer_model, trust_remote_code=True) except Exception as e: - print(f"Warning: Failed to load tokenizer {repo_id}: {e}") + print(f"Warning: Failed to load tokenizer {tokenizer_model}: {e}") print("Falling back to character-based approximation (4 chars ≈ 1 token)") self.tokenizer = None # Fallback simple sentence splitter (period, question, exclamation, newline) self._sent_re = re.compile(r"(?<=[\.\!\?])\s+|\n+") - self.legacy = MarkdownRecursiveChunker(max_chunk_size=10_000, min_chunk_size=100) + self.legacy = MarkdownRecursiveChunker( + max_chunk_size=10_000, min_chunk_size=100, tokenizer_model=tokenizer_model + ) # ------------------------------------------------------------------ def _token_len(self, text: str) -> int: diff --git a/rag_system/ingestion/document_converter.py b/rag_system/ingestion/document_converter.py index 78b2a9ce..83974a6a 100644 --- a/rag_system/ingestion/document_converter.py +++ b/rag_system/ingestion/document_converter.py @@ -1,9 +1,67 @@ from typing import List, Tuple, Dict, Any +import os +import platform + +# torch.compile's inductor backend has no MPS support; docling's layout model +# calls it and crashes on Apple Silicon unless dynamo is disabled up front. +if platform.system() == "Darwin": + os.environ.setdefault("TORCHDYNAMO_DISABLE", "1") + from docling.document_converter import DocumentConverter as DoclingConverter, PdfFormatOption -from docling.datamodel.pipeline_options import PdfPipelineOptions, OcrMacOptions +from docling.datamodel import pipeline_options as docling_options +from docling.datamodel.pipeline_options import PdfPipelineOptions from docling.datamodel.base_models import InputFormat import fitz # PyMuPDF for quick text inspection -import os +import importlib.util +import shutil + +# docling options class -> the module or binary its backend needs at runtime. +OCR_BACKENDS = [ + ("OcrMacOptions", "module", ("ocrmac",)), + ("EasyOcrOptions", "module", ("easyocr",)), + # rapidocr renamed its package: >=3.x installs as `rapidocr`, older + # releases as `rapidocr_onnxruntime` — accept either. + ("RapidOcrOptions", "module", ("rapidocr", "rapidocr_onnxruntime")), + ("TesseractOcrOptions", "module", ("tesserocr",)), + ("TesseractCliOcrOptions", "binary", ("tesseract",)), +] + + +def build_ocr_options(): + """Pick an OCR engine docling can actually run on this host. + + OcrMac is only tried on macOS; the remaining engines are tried in order and + only if their backend is installed. Returns None when nothing is available, + in which case docling's own default OCR settings are used. + """ + for name, kind, dependencies in OCR_BACKENDS: + if name == "OcrMacOptions" and platform.system() != "Darwin": + continue + options_cls = getattr(docling_options, name, None) + if options_cls is None: + continue + if kind == "module" and not any(importlib.util.find_spec(d) for d in dependencies): + continue + if kind == "binary" and not any(shutil.which(d) for d in dependencies): + continue + try: + kwargs = {"force_full_page_ocr": True} + # docling's RapidOCR default is lang=['chinese']; pin an explicit + # recognition language (OCR_LANG env, comma-separated, to override). + if name == "RapidOcrOptions": + kwargs["lang"] = [ + l.strip() for l in os.getenv("OCR_LANG", "english").split(",") if l.strip() + ] + options = options_cls(**kwargs) + except Exception as e: + print(f"OCR engine {name} is not usable here: {e}") + continue + print(f"OCR engine: {name}") + return options + + print("No OCR engine available; using docling's default OCR settings.") + return None + class DocumentConverter: """ @@ -22,58 +80,63 @@ class DocumentConverter: } def __init__(self): - """Initializes the docling document converter with forced OCR enabled for macOS.""" + """Initializes one docling converter per path (no-OCR, OCR, general). + + Each converter is built independently so that a failure in one (typically + the OCR engine) does not disable the others. + """ + self.converter_no_ocr = self._build_pdf_converter(do_ocr=False) + self.converter_ocr = self._build_pdf_converter(do_ocr=True) try: - # --- Converter WITHOUT OCR (fast path) --- - pipeline_no_ocr = PdfPipelineOptions() - pipeline_no_ocr.do_ocr = False - format_no_ocr = { - InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline_no_ocr) - } - self.converter_no_ocr = DoclingConverter(format_options=format_no_ocr) - - # --- Converter WITH OCR (fallback) --- - pipeline_ocr = PdfPipelineOptions() - pipeline_ocr.do_ocr = True - ocr_options = OcrMacOptions(force_full_page_ocr=True) - pipeline_ocr.ocr_options = ocr_options - format_ocr = { - InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline_ocr) - } - self.converter_ocr = DoclingConverter(format_options=format_ocr) - self.converter_general = DoclingConverter() - - print("docling DocumentConverter(s) initialized (OCR + no-OCR + general).") except Exception as e: - print(f"Error initializing docling DocumentConverter(s): {e}") - self.converter_no_ocr = None - self.converter_ocr = None + print(f"Error initializing general docling converter: {e}") self.converter_general = None + available = [ + name for name, conv in ( + ("no-OCR", self.converter_no_ocr), + ("OCR", self.converter_ocr), + ("general", self.converter_general), + ) if conv is not None + ] + print(f"docling DocumentConverter(s) initialized ({', '.join(available) or 'none'}).") + + @staticmethod + def _build_pdf_converter(*, do_ocr: bool): + try: + pipeline = PdfPipelineOptions() + pipeline.do_ocr = do_ocr + if do_ocr: + ocr_options = build_ocr_options() + if ocr_options is not None: + pipeline.ocr_options = ocr_options + return DoclingConverter( + format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=pipeline)} + ) + except Exception as e: + print(f"Error initializing docling PDF converter (ocr={do_ocr}): {e}") + return None + def convert_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, Any]]]: """ Converts a document to a single Markdown string, preserving layout and tables. Supports PDF, DOCX, HTML, and other formats. """ - if not (self.converter_no_ocr and self.converter_ocr and self.converter_general): - print("docling converters not available. Skipping conversion.") - return [] - file_ext = os.path.splitext(file_path)[1].lower() if file_ext not in self.SUPPORTED_FORMATS: print(f"Unsupported file format: {file_ext}") return [] - + input_format = self.SUPPORTED_FORMATS[file_ext] - + if input_format == InputFormat.PDF: return self._convert_pdf_to_markdown(file_path) elif input_format == 'TXT': return self._convert_txt_to_markdown(file_path) else: return self._convert_general_to_markdown(file_path, input_format) - + def _convert_pdf_to_markdown(self, pdf_path: str) -> List[Tuple[str, Dict[str, Any]]]: """Convert PDF with OCR detection logic.""" # Quick heuristic: if the PDF already contains a text layer, skip OCR for speed @@ -88,12 +151,19 @@ def _pdf_has_text(path: str) -> bool: return False use_ocr = not _pdf_has_text(pdf_path) + if use_ocr and self.converter_ocr is None: + print(f"{pdf_path} has no text layer but no OCR converter is available; trying without OCR.") + use_ocr = False converter = self.converter_ocr if use_ocr else self.converter_no_ocr ocr_msg = "(OCR enabled)" if use_ocr else "(no OCR)" + if converter is None: + print(f"No docling PDF converter available. Skipping {pdf_path}.") + return [] + print(f"Converting {pdf_path} to Markdown using docling {ocr_msg}...") return self._perform_conversion(pdf_path, converter, ocr_msg) - + def _convert_txt_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, Any]]]: """Convert plain text files to markdown by reading content directly.""" print(f"Converting {file_path} (TXT) to Markdown...") @@ -112,6 +182,9 @@ def _convert_txt_to_markdown(self, file_path: str) -> List[Tuple[str, Dict[str, def _convert_general_to_markdown(self, file_path: str, input_format: InputFormat) -> List[Tuple[str, Dict[str, Any]]]: """Convert non-PDF formats using general converter.""" + if self.converter_general is None: + print(f"General docling converter not available. Skipping {file_path}.") + return [] print(f"Converting {file_path} ({input_format.name}) to Markdown using docling...") return self._perform_conversion(file_path, self.converter_general, f"({input_format.name})") diff --git a/rag_system/main.py b/rag_system/main.py index a1f50794..cdc2d178 100644 --- a/rag_system/main.py +++ b/rag_system/main.py @@ -1,38 +1,29 @@ import os import json -import sys import argparse from dotenv import load_dotenv # Load environment variables from .env file load_dotenv() -# The sys.path manipulation has been removed to prevent import conflicts. -# This script should be run as a module from the project root, e.g.: +# This module holds the MASTER configuration for the RAG system plus a thin CLI. +# Agent / pipeline construction lives in rag_system/factory.py. +# Run it as a module from the project root, e.g.: # python -m rag_system.main api -from rag_system.agent.loop import Agent -from rag_system.utils.ollama_client import OllamaClient -# Configuration is now defined in this file - no import needed - -# Advanced RAG System Configuration -# ================================== -# This file contains the MASTER configuration for all models used in the RAG system. -# All components should reference these configurations to ensure consistency. - # ============================================================================ -# 🎯 MASTER MODEL CONFIGURATION +# MASTER MODEL CONFIGURATION # ============================================================================ -# All model configurations are centralized here to prevent conflicts +# Every model default below can be overridden with an environment variable. -# LLM Backend Configuration +# LLM Backend Configuration ("ollama" or "watsonx") LLM_BACKEND = os.getenv("LLM_BACKEND", "ollama") # Ollama Models Configuration (for inference via Ollama) OLLAMA_CONFIG = { "host": os.getenv("OLLAMA_HOST", "http://localhost:11434"), - "generation_model": "qwen3:8b", # Main text generation model - "enrichment_model": "qwen3:0.6b", # Lightweight model for routing/enrichment + "generation_model": os.getenv("GENERATION_MODEL", "qwen3.5:9b"), + "enrichment_model": os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b"), } WATSONX_CONFIG = { @@ -40,79 +31,89 @@ "project_id": os.getenv("WATSONX_PROJECT_ID", ""), "url": os.getenv("WATSONX_URL", "https://us-south.ml.cloud.ibm.com"), "generation_model": os.getenv("WATSONX_GENERATION_MODEL", "ibm/granite-13b-chat-v2"), - "enrichment_model": os.getenv("WATSONX_ENRICHMENT_MODEL", "ibm/granite-8b-japanese"), # Lightweight model + "enrichment_model": os.getenv("WATSONX_ENRICHMENT_MODEL", "ibm/granite-8b-japanese"), } -# External Model Configuration (HuggingFace models used directly) +# External Model Configuration (HuggingFace models loaded in-process) +# +# Defaults set at the Phase 1 adoption gate (2026-08-09); the measurements and +# the reasoning are in eval/DECISIONS.md. +# +# embedding_model microsoft/harrier-oss-v1-0.6b (MIT, 1024-dim). Measured +# mixed-corpus first-stage nDCG@10 0.915 vs 0.875 for +# Qwen/Qwen3-Embedding-4B, at ~3x lower latency and ~7x less +# memory. Qwen/Qwen3-Embedding-4B remains a supported option. +# reranker_model Only loaded when reranking is switched on — the "default" +# profile ships with reranker.enabled = False (see below). +# When a user does switch it on, they get the model that +# measured a win on top of this first stage. EXTERNAL_MODELS = { - "embedding_model": "Qwen/Qwen3-Embedding-0.6B", # HuggingFace embedding model (1024 dims - fresh start) - "reranker_model": "answerdotai/answerai-colbert-small-v1", # ColBERT reranker - "vision_model": "Qwen/Qwen-VL-Chat", # Vision model for multimodal - "fallback_reranker": "BAAI/bge-reranker-base", # Backup reranker + "embedding_model": os.getenv("EMBEDDING_MODEL", "microsoft/harrier-oss-v1-0.6b"), + "reranker_model": os.getenv("RERANKER_MODEL", "Qwen/Qwen3-Reranker-4B"), } # ============================================================================ -# 🔧 PIPELINE CONFIGURATIONS +# PIPELINE CONFIGURATIONS # ============================================================================ PIPELINE_CONFIGS = { "default": { - "description": "Production-ready pipeline with hybrid search, AI reranking, and verification", + "description": "Production-ready pipeline with hybrid search, query decomposition, and verification", "storage": { - "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "image_table_name": "image_pages_v3", - "bm25_path": "./index_store/bm25", - "graph_path": "./index_store/graph/knowledge_graph.gml" + "lancedb_uri": os.getenv("LANCEDB_PATH", "./lancedb"), + # v4: vectors are L2-normalized at write and query time (cosine + # ordering). v3 tables hold unnormalized vectors — see + # rag_system/indexing/embedders.py table markers. + "text_table_name": "text_pages_v4" }, "retrieval": { - "retriever": "multivector", "search_type": "hybrid", - "late_chunking": { - "enabled": True, - "table_suffix": "_lc_v3" - }, - "dense": { - "enabled": True, - "weight": 0.7 + "latechunk": { + "enabled": True }, - "bm25": { - "enabled": True, - "index_name": "rag_bm25_index" + "dense": { + "enabled": True }, - "graph": { - "enabled": False, - "graph_path": "./index_store/graph/knowledge_graph.gml" + # Evidence-sufficiency retry (roadmap 2.1). One conditional second + # retrieval when the first pass found weak evidence; hard cap of one + # extra attempt. The signal is NOT the raw top cosine — that measured + # anti-correlated with success — but the contrast between the best + # candidate and the background of the rest; see + # RetrievalPipeline._dense_evidence_score and eval/decisions/ + # phase2-pipeline.md for the calibration. + "retry": { + "enabled": True, + "min_top_score": 0.12, + "max_attempts": 1 } }, - # 🎯 EMBEDDING MODEL: Uses HuggingFace Qwen model directly "embedding_model_name": EXTERNAL_MODELS["embedding_model"], - # 🎯 VISION MODEL: For multimodal capabilities - "vision_model_name": EXTERNAL_MODELS["vision_model"], - # 🎯 RERANKER: AI-powered reranking with ColBERT + # Reranking is OFF by default. Measured at the Phase 1 gate on the mixed + # corpus: this first stage alone scores nDCG@10 0.915; the cheap + # cross-encoder (bge-reranker-v2-m3) drops it to 0.892 for ~1.6s/query, + # and Qwen3-Reranker-4B lifts it to 0.977 for ~12.7s/query. Neither is a + # sensible always-on default, so the toggle (UI "AI reranker" / + # reranker.enabled) picks the quality model and loads it lazily. "reranker": { - "enabled": True, - "type": "ai", + "enabled": False, + "model_type": "cross-encoder", "strategy": "rerankers-lib", "model_name": EXTERNAL_MODELS["reranker_model"], "top_k": 10 }, "query_decomposition": { "enabled": True, - "max_sub_queries": 3, "compose_from_sub_answers": True }, "verification": {"enabled": True}, "retrieval_k": 20, "context_window_size": 0, "semantic_cache_threshold": 0.98, - "cache_scope": "global", - # 🔧 Contextual enrichment configuration + "cache_scope": "session", "contextual_enricher": { "enabled": True, "window_size": 1 }, - # 🔧 Indexing configuration "indexing": { "embedding_batch_size": 50, "enrichment_batch_size": 10, @@ -122,16 +123,16 @@ "fast": { "description": "Speed-optimized pipeline with minimal overhead", "storage": { - "lancedb_uri": "./lancedb", - "text_table_name": "text_pages_v3", - "image_table_name": "image_pages_v3", - "bm25_path": "./index_store/bm25" + "lancedb_uri": os.getenv("LANCEDB_PATH", "./lancedb"), + "text_table_name": "text_pages_v4" }, "retrieval": { - "retriever": "multivector", "search_type": "vector_only", - "late_chunking": {"enabled": False}, - "dense": {"enabled": True} + "latechunk": {"enabled": False}, + "dense": {"enabled": True}, + # Off in `fast`: the retry costs one enrichment-model round-trip plus + # a second retrieval, which is exactly what this profile exists to avoid. + "retry": {"enabled": False} }, "embedding_model_name": EXTERNAL_MODELS["embedding_model"], "reranker": {"enabled": False}, @@ -139,231 +140,100 @@ "verification": {"enabled": False}, "retrieval_k": 10, "context_window_size": 0, - # 🔧 Contextual enrichment (disabled for speed) + "semantic_cache_threshold": 0.98, + "cache_scope": "session", "contextual_enricher": { "enabled": False, "window_size": 1 }, - # 🔧 Indexing configuration "indexing": { "embedding_batch_size": 100, "enrichment_batch_size": 50, "enable_progress_tracking": False } - }, - "bm25": { - "enabled": True, - "index_name": "rag_bm25_index" - }, - "graph_rag": { - "enabled": False, # Keep disabled for now unless specified } } # ============================================================================ -# 🏭 FACTORY FUNCTIONS +# CLI # ============================================================================ -def get_agent(mode: str = "default") -> Agent: - """ - Factory function to get an instance of the RAG agent based on the specified mode. - - Args: - mode: Configuration mode ("default", "fast") - - Returns: - Configured Agent instance - """ - load_dotenv() - - # Initialize the appropriate LLM client based on backend configuration - if LLM_BACKEND.lower() == "watsonx": - from rag_system.utils.watsonx_client import WatsonXClient - - if not WATSONX_CONFIG["api_key"] or not WATSONX_CONFIG["project_id"]: - raise ValueError( - "Watson X configuration incomplete. Please set WATSONX_API_KEY and WATSONX_PROJECT_ID " - "environment variables." - ) - - llm_client = WatsonXClient( - api_key=WATSONX_CONFIG["api_key"], - project_id=WATSONX_CONFIG["project_id"], - url=WATSONX_CONFIG["url"] - ) - llm_config = WATSONX_CONFIG - print(f"🔧 Using Watson X backend with granite models") - else: - llm_client = OllamaClient(host=OLLAMA_CONFIG["host"]) - llm_config = OLLAMA_CONFIG - print(f"🔧 Using Ollama backend") - - # Get the configuration for the specified mode - config = PIPELINE_CONFIGS.get(mode, PIPELINE_CONFIGS['default']) - - agent = Agent( - pipeline_configs=config, - llm_client=llm_client, - ollama_config=llm_config - ) - return agent - -def validate_model_config(): - """ - Validates the model configuration for consistency and availability. - - Raises: - ValueError: If configuration conflicts are detected - """ - print("🔍 Validating model configuration...") - - # Check for embedding model consistency - default_embedding = PIPELINE_CONFIGS["default"]["embedding_model_name"] - external_embedding = EXTERNAL_MODELS["embedding_model"] - - if default_embedding != external_embedding: - raise ValueError(f"Embedding model mismatch: {default_embedding} != {external_embedding}") - - # Check reranker configuration - default_reranker = PIPELINE_CONFIGS["default"]["reranker"]["model_name"] - external_reranker = EXTERNAL_MODELS["reranker_model"] - - if default_reranker != external_reranker: - raise ValueError(f"Reranker model mismatch: {default_reranker} != {external_reranker}") - - print("✅ Model configuration validation passed!") - - return True +SUPPORTED_DOCUMENT_EXTENSIONS = (".pdf", ".docx", ".html", ".htm", ".md", ".txt") -# ============================================================================ -# 🚀 UTILITY FUNCTIONS -# ============================================================================ -def run_indexing(docs_path: str, config_mode: str = "default"): - """Runs the indexing pipeline for the specified documents.""" - print(f"📚 Starting indexing for documents in: {docs_path}") - validate_model_config() - - # Local import to avoid circular dependencies - from rag_system.pipelines.indexing_pipeline import IndexingPipeline - - # Get the appropriate indexing pipeline from the factory - indexing_pipeline = IndexingPipeline(PIPELINE_CONFIGS[config_mode]) - - # Find all PDF files in the directory - pdf_files = [os.path.join(docs_path, f) for f in os.listdir(docs_path) if f.endswith(".pdf")] - - if not pdf_files: - print("No PDF files found to index.") - return - - # Process all documents through the pipeline - indexing_pipeline.process_documents(pdf_files) - print("✅ Indexing complete.") - -def run_chat(query: str): - """ - Runs the agentic RAG pipeline for a given query. - Returns the result as a JSON string. - """ - try: - validate_model_config() - ollama_client = OllamaClient(OLLAMA_CONFIG["host"]) - except ConnectionError as e: - print(e) - return json.dumps({"error": str(e)}, indent=2) - except ValueError as e: - print(f"Configuration Error: {e}") - return json.dumps({"error": f"Configuration Error: {e}"}, indent=2) - - agent = Agent(PIPELINE_CONFIGS['default'], ollama_client, OLLAMA_CONFIG) - result = agent.run(query) - return json.dumps(result, indent=2, ensure_ascii=False) - -def show_graph(): - """ - Loads and displays the knowledge graph. - """ - import networkx as nx - import matplotlib.pyplot as plt - - graph_path = PIPELINE_CONFIGS["indexing"]["graph_path"] - if not os.path.exists(graph_path): - print("Knowledge graph not found. Please run the 'index' command first.") - return - - G = nx.read_gml(graph_path) - print("--- Knowledge Graph ---") - print("Nodes:", G.nodes(data=True)) - print("Edges:", G.edges(data=True)) - print("---------------------") - - # Optional: Visualize the graph - try: - pos = nx.spring_layout(G) - nx.draw(G, pos, with_labels=True, node_size=2000, node_color="skyblue", font_size=10, font_weight="bold") - edge_labels = nx.get_edge_attributes(G, 'label') - nx.draw_networkx_edge_labels(G, pos, edge_labels=edge_labels) - plt.title("Knowledge Graph Visualization") - plt.show() - except Exception as e: - print(f"\nCould not visualize the graph. Matplotlib might not be installed or configured for your environment.") - print(f"Error: {e}") - -def run_api_server(): - """Starts the advanced RAG API server.""" - from rag_system.api_server import start_server - start_server() - -def main(): - if len(sys.argv) < 2: - print("Usage: python main.py [index|chat|show_graph|api] [query]") - return - - command = sys.argv[1] - if command == "index": - # Allow passing file paths from the command line - files = sys.argv[2:] if len(sys.argv) > 2 else None - run_indexing(files) - elif command == "chat": - if len(sys.argv) < 3: - print("Usage: python main.py chat ") - return - query = " ".join(sys.argv[2:]) - # 🆕 Print the result for command-line usage - print(run_chat(query)) - elif command == "show_graph": - show_graph() - elif command == "api": - run_api_server() - else: - print(f"Unknown command: {command}") +def _collect_file_paths(path: str) -> list[str]: + """Expand a file or directory argument into a list of indexable file paths.""" + if os.path.isfile(path): + return [os.path.abspath(path)] -if __name__ == "__main__": - # This allows running the script from the command line to index documents. - parser = argparse.ArgumentParser(description="Main entry point for the RAG system.") - parser.add_argument( - '--index', - type=str, - help='Path to the directory containing documents to index.' - ) - parser.add_argument( - '--config', - type=str, - default='default', - help='The configuration profile to use (e.g., "default", "fast").' + if not os.path.isdir(path): + raise FileNotFoundError(f"No such file or directory: {path}") + + collected = [] + for root, _dirs, files in os.walk(path): + for name in sorted(files): + if name.lower().endswith(SUPPORTED_DOCUMENT_EXTENSIONS): + collected.append(os.path.join(root, name)) + return collected + + +def main() -> int: + parser = argparse.ArgumentParser( + prog="python -m rag_system.main", + description="localGPT RAG system: indexing, one-shot chat, and API server." ) + subparsers = parser.add_subparsers(dest="command", required=True) + + modes = sorted(PIPELINE_CONFIGS) + + index_parser = subparsers.add_parser("index", help="Index a document or a directory of documents.") + index_parser.add_argument("path", help="File or directory to index.") + index_parser.add_argument("--mode", default="default", choices=modes, help="Pipeline profile to use.") + + chat_parser = subparsers.add_parser("chat", help="Answer a single query and print the JSON result.") + chat_parser.add_argument("query", help="The question to ask.") + chat_parser.add_argument("--mode", default="default", choices=modes, help="Pipeline profile to use.") + + api_parser = subparsers.add_parser("api", help="Start the RAG API server.") + api_parser.add_argument("--port", type=int, default=8001, help="Port to listen on.") args = parser.parse_args() - # Load environment variables - load_dotenv() - - if args.index: - run_indexing(args.index, args.config) - else: - # This is where you might start a server or interactive session - print("No action specified. Use --index to process documents.") - # Example of how to get an agent instance - # agent = get_agent(args.config) - # print(f"Agent loaded with '{args.config}' config.") + if args.command == "index": + from rag_system.factory import get_indexing_pipeline + + try: + file_paths = _collect_file_paths(args.path) + except FileNotFoundError as e: + print(f"❌ {e}") + return 1 + + if not file_paths: + print(f"No indexable documents found in {args.path} " + f"(supported: {', '.join(SUPPORTED_DOCUMENT_EXTENSIONS)}).") + return 1 + + print(f"📚 Indexing {len(file_paths)} file(s) with the '{args.mode}' profile...") + get_indexing_pipeline(args.mode).run(file_paths) + print("✅ Indexing complete.") + return 0 + + if args.command == "chat": + from rag_system.factory import get_agent + + result = get_agent(args.mode).run(args.query) + print(json.dumps(result, indent=2, ensure_ascii=False)) + return 0 + + if args.command == "api": + from rag_system.api_server import start_server + + start_server(port=args.port) + return 0 + + parser.error(f"Unknown command: {args.command}") + return 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/rag_system/pipelines/indexing_pipeline.py b/rag_system/pipelines/indexing_pipeline.py index 9fc61e7e..7926484f 100644 --- a/rag_system/pipelines/indexing_pipeline.py +++ b/rag_system/pipelines/indexing_pipeline.py @@ -1,15 +1,20 @@ from typing import List, Dict, Any import os -import networkx as nx from rag_system.ingestion.document_converter import DocumentConverter from rag_system.ingestion.chunking import MarkdownRecursiveChunker from rag_system.indexing.representations import EmbeddingGenerator, select_embedder from rag_system.indexing.embedders import LanceDBManager, VectorIndexer -from rag_system.indexing.graph_extractor import GraphExtractor from rag_system.utils.ollama_client import OllamaClient from rag_system.indexing.contextualizer import ContextualEnricher from rag_system.indexing.overview_builder import OverviewBuilder + +def _default_embedding_model() -> str: + """The single source of truth for the embedding model default.""" + from rag_system.main import EXTERNAL_MODELS + return EXTERNAL_MODELS["embedding_model"] + + class IndexingPipeline: def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_config: Dict[str, str]): self.config = config @@ -18,35 +23,38 @@ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_c self.document_converter = DocumentConverter() # Chunker selection: docling (token-based) or legacy (character-based) chunker_mode = config.get("chunker_mode", "docling") - - # 🔧 Get chunking configuration from frontend parameters + + self.embedding_model_name = config.get("embedding_model_name") or _default_embedding_model() + + # Chunk size is the token budget per chunk for both chunkers. chunking_config = config.get("chunking", {}) - chunk_size = chunking_config.get("chunk_size", config.get("chunk_size", 1500)) - chunk_overlap = chunking_config.get("chunk_overlap", config.get("chunk_overlap", 200)) - - print(f"🔧 CHUNKING CONFIG: Size: {chunk_size}, Overlap: {chunk_overlap}, Mode: {chunker_mode}") - + chunk_size = chunking_config.get( + "chunk_size", config.get("chunk_size", config.get("max_tokens", 1500)) + ) + + print(f"🔧 CHUNKING CONFIG: Size: {chunk_size}, Mode: {chunker_mode}") + if chunker_mode == "docling": try: from rag_system.ingestion.docling_chunker import DoclingChunker self.chunker = DoclingChunker( - max_tokens=config.get("max_tokens", chunk_size), + max_tokens=chunk_size, overlap=config.get("overlap_sentences", 1), - tokenizer_model=config.get("embedding_model_name", "qwen3-embedding-0.6b"), + tokenizer_model=self.embedding_model_name, ) print("🪄 Using DoclingChunker for high-recall sentence packing.") except Exception as e: print(f"⚠️ Failed to initialise DoclingChunker: {e}. Falling back to legacy chunker.") self.chunker = MarkdownRecursiveChunker( max_chunk_size=chunk_size, - min_chunk_size=min(chunk_overlap, chunk_size // 4), # Sensible minimum - tokenizer_model=config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B") + min_chunk_size=max(1, chunk_size // 4), + tokenizer_model=self.embedding_model_name, ) else: self.chunker = MarkdownRecursiveChunker( max_chunk_size=chunk_size, - min_chunk_size=min(chunk_overlap, chunk_size // 4), # Sensible minimum - tokenizer_model=config.get("embedding_model_name", "Qwen/Qwen3-Embedding-0.6B") + min_chunk_size=max(1, chunk_size // 4), + tokenizer_model=self.embedding_model_name, ) retriever_configs = self.config.get("retrievers") or self.config.get("retrieval", {}) @@ -76,7 +84,7 @@ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_c self.lancedb_manager = LanceDBManager(db_path=db_path) self.vector_indexer = VectorIndexer(self.lancedb_manager) embedding_model = select_embedder( - self.config.get("embedding_model_name", "BAAI/bge-small-en-v1.5"), + self.embedding_model_name, self.ollama_config.get("host") if isinstance(self.ollama_config, dict) else None, ) self.embedding_generator = EmbeddingGenerator( @@ -84,46 +92,56 @@ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_c batch_size=self.embedding_batch_size ) - if retriever_configs.get("graph", {}).get("enabled"): - self.graph_extractor = GraphExtractor( - llm_client=self.llm_client, - llm_model=self.ollama_config["generation_model"] - ) - - if self.config.get("contextual_enricher", {}).get("enabled"): - # 🔧 Use frontend enrich_model parameter if provided + enricher_config = self.config.get("contextual_enricher", {}) + self.enricher_enabled = bool(enricher_config.get("enabled", False)) + self.enricher_window_size = enricher_config.get("window_size", 1) + self.contextual_enricher = None + if self.enricher_enabled: enrichment_model = ( - self.config.get("enrich_model") or # Frontend parameter + self.config.get("enrich_model") or # Per-request override self.config.get("enrichment_model_name") or # Alternative config key - self.ollama_config.get("enrichment_model") or # Default from ollama config + self.ollama_config.get("enrichment_model") or # Default from llm config self.ollama_config["generation_model"] # Final fallback ) print(f"🔧 ENRICHMENT MODEL: Using '{enrichment_model}' for contextual enrichment") - + self.contextual_enricher = ContextualEnricher( llm_client=self.llm_client, llm_model=enrichment_model, batch_size=self.enrichment_batch_size ) - # Overview builder always enabled for triage routing - ov_path = self.config.get("overview_path") - self.overview_builder = OverviewBuilder( - llm_client=self.llm_client, - model=self.config.get("overview_model_name", self.ollama_config.get("enrichment_model", "qwen3:0.6b")), - first_n_chunks=self.config.get("overview_first_n_chunks", 5), - out_path=ov_path if ov_path else None, - ) + # Document overviews feed the triage router; on by default. + overview_config = self.config.get("overview", {}) + self.overview_builder = None + if overview_config.get("enabled", True): + self.overview_builder = OverviewBuilder( + llm_client=self.llm_client, + model=( + self.config.get("overview_model_name") + or overview_config.get("model") + or self.ollama_config.get("enrichment_model") + or self.ollama_config["generation_model"] + ), + first_n_chunks=self.config.get( + "overview_first_n_chunks", overview_config.get("max_chunks", 5) + ), + out_path=self.config.get("overview_path") or None, + ) # ------------------------------------------------------------------ # Late-Chunk encoder initialisation (optional) # ------------------------------------------------------------------ - self.latechunk_enabled = retriever_configs.get("latechunk", {}).get("enabled", False) + self.latechunk_cfg = ( + retriever_configs.get("latechunk") + or retriever_configs.get("late_chunking") + or {} + ) + self.latechunk_enabled = bool(self.latechunk_cfg.get("enabled", False)) if self.latechunk_enabled: try: from rag_system.indexing.latechunk import LateChunkEncoder - self.latechunk_cfg = retriever_configs["latechunk"] - self.latechunk_encoder = LateChunkEncoder(model_name=self.config.get("embedding_model_name", "qwen3-embedding-0.6b")) + self.latechunk_encoder = LateChunkEncoder(model_name=self.embedding_model_name) except Exception as e: print(f"⚠️ Failed to initialise LateChunkEncoder: {e}. Disabling latechunk retrieval.") self.latechunk_enabled = False @@ -179,11 +197,12 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non chunk['metadata']['chunk_index'] = i # Build and persist document overview (non-blocking errors) - try: - self.overview_builder.build_and_store(document_id, file_chunks) - except Exception as e: - print(f" ⚠️ Failed to create overview for {document_id}: {e}") - + if self.overview_builder is not None: + try: + self.overview_builder.build_and_store(document_id, file_chunks) + except Exception as e: + print(f" ⚠️ Failed to create overview for {document_id}: {e}") + all_chunks.extend(file_chunks) doc_chunks_map[document_id] = file_chunks # save for late-chunk step print(f" Generated {len(file_chunks)} chunks from {document_id}") @@ -197,8 +216,11 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non file_tracker.finish() if not all_chunks: - print("No text chunks were generated. Skipping indexing.") - return + raise RuntimeError( + "No text chunks were generated from the supplied documents — " + "conversion or chunking failed for every file. Check the server " + "log for per-file conversion errors; nothing was indexed." + ) print(f"\n✅ Generated {len(all_chunks)} text chunks total.") memory_mb = estimate_memory_usage(all_chunks) @@ -206,53 +228,33 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non retriever_configs = self.config.get("retrievers") or self.config.get("retrieval", {}) - # Step 3: Optional Contextual Enrichment (before indexing for consistency) - enricher_config = self.config.get("contextual_enricher", {}) - enricher_enabled = enricher_config.get("enabled", False) - - print(f"\n🔍 CONTEXTUAL ENRICHMENT DEBUG:") - print(f" Config present: {bool(enricher_config)}") - print(f" Enabled: {enricher_enabled}") - print(f" Has enricher object: {hasattr(self, 'contextual_enricher')}") - - if hasattr(self, 'contextual_enricher') and enricher_enabled: + # Step 2: Optional Contextual Enrichment (before indexing for consistency) + if self.contextual_enricher is not None: with timer("Contextual Enrichment"): - window_size = enricher_config.get("window_size", 1) - print(f"\n🚀 CONTEXTUAL ENRICHMENT ACTIVE!") - print(f" Window size: {window_size}") - print(f" Model: {self.contextual_enricher.llm_model}") - print(f" Batch size: {self.contextual_enricher.batch_size}") - print(f" Processing {len(all_chunks)} chunks...") - - # Show before/after example - if all_chunks: - print(f" Example BEFORE: '{all_chunks[0]['text'][:100]}...'") - + print( + f"\n🚀 Contextual enrichment: model={self.contextual_enricher.llm_model}, " + f"window={self.enricher_window_size}, batch={self.contextual_enricher.batch_size}, " + f"chunks={len(all_chunks)}" + ) # This modifies the 'text' field in each chunk dictionary - all_chunks = self.contextual_enricher.enrich_chunks(all_chunks, window_size=window_size) - - if all_chunks: - print(f" Example AFTER: '{all_chunks[0]['text'][:100]}...'") - + all_chunks = self.contextual_enricher.enrich_chunks( + all_chunks, window_size=self.enricher_window_size + ) print(f"✅ Enriched {len(all_chunks)} chunks with context for indexing.") else: - print(f"⚠️ CONTEXTUAL ENRICHMENT SKIPPED:") - if not hasattr(self, 'contextual_enricher'): - print(f" Reason: No enricher object (config enabled={enricher_enabled})") - elif not enricher_enabled: - print(f" Reason: Disabled in config") - print(f" Chunks will be indexed without contextual enrichment.") - - # Step 4: Create BM25 Index from enriched chunks (for consistency with vector index) + print("\nℹ️ Contextual enrichment disabled; indexing chunks as-is.") + + # Step 3: Embed chunks into LanceDB and build the native FTS index if hasattr(self, 'vector_indexer') and hasattr(self, 'embedding_generator'): with timer("Vector Embedding & Indexing"): table_name = self.config["storage"].get("text_table_name") or retriever_configs.get("dense", {}).get("lancedb_table_name", "default_text_table") - print(f"\n--- Generating embeddings with {self.config.get('embedding_model_name')} ---") + print(f"\n--- Generating embeddings with {self.embedding_model_name} ---") embeddings = self.embedding_generator.generate(all_chunks) print(f"\n--- Indexing {len(embeddings)} vectors into LanceDB table: {table_name} ---") - self.vector_indexer.index(table_name, all_chunks, embeddings) + self.vector_indexer.index(table_name, all_chunks, embeddings, + embedding_model=self.embedding_model_name) print("✅ Vector embeddings indexed successfully") # Create FTS index on the 'text' field after adding data @@ -282,7 +284,9 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non # --------------------------------------------------- if self.latechunk_enabled: with timer("Late-Chunk Embedding & Indexing"): - lc_table_name = self.latechunk_cfg.get("lancedb_table_name", f"{table_name}_lc") + lc_table_name = self.latechunk_cfg.get("lancedb_table_name") or ( + f"{table_name}{self.latechunk_cfg.get('table_suffix', '_lc')}" + ) print(f"\n--- Generating late-chunk embeddings (table={lc_table_name}) ---") total_lc_vecs = 0 @@ -313,28 +317,12 @@ def run(self, file_paths: List[str] | None = None, *, documents: List[str] | Non print(f"⚠️ Mismatch LC vecs ({len(lc_vecs)}) vs chunks ({len(doc_chunks)}) for {doc_id}. Skipping.") continue - self.vector_indexer.index(lc_table_name, doc_chunks, lc_vecs) + self.vector_indexer.index(lc_table_name, doc_chunks, lc_vecs, + embedding_model=self.embedding_model_name) total_lc_vecs += len(lc_vecs) print(f"✅ Late-chunk vectors indexed: {total_lc_vecs}") - - # Step 6: Knowledge Graph Extraction (Optional) - if hasattr(self, 'graph_extractor'): - with timer("Knowledge Graph Extraction"): - graph_path = retriever_configs.get("graph", {}).get("graph_path", "./index_store/graph/default_graph.gml") - print(f"\n--- Building and saving knowledge graph to: {graph_path} ---") - - graph_data = self.graph_extractor.extract(all_chunks) - G = nx.DiGraph() - for entity in graph_data['entities']: - G.add_node(entity['id'], type=entity.get('type', 'Unknown'), properties=entity.get('properties', {})) - for rel in graph_data['relationships']: - G.add_edge(rel['source'], rel['target'], label=rel['label']) - - os.makedirs(os.path.dirname(graph_path), exist_ok=True) - nx.write_gml(G, graph_path) - print(f"✅ Knowledge graph saved successfully.") - + print("\n--- ✅ Indexing Complete ---") self._print_final_statistics(len(file_paths), len(all_chunks)) @@ -343,16 +331,19 @@ def _print_final_statistics(self, num_files: int, num_chunks: int): print(f"\n📈 Final Statistics:") print(f" Files processed: {num_files}") print(f" Chunks generated: {num_chunks}") - print(f" Average chunks per file: {num_chunks/num_files:.1f}") - + if num_files: + print(f" Average chunks per file: {num_chunks/num_files:.1f}") + # Component status components = [] - if hasattr(self, 'contextual_enricher'): + if self.contextual_enricher is not None: components.append("✅ Contextual Enrichment") if hasattr(self, 'vector_indexer'): components.append("✅ Vector & FTS Index") - if hasattr(self, 'graph_extractor'): - components.append("✅ Knowledge Graph") - + if self.latechunk_enabled: + components.append("✅ Late Chunking") + if self.overview_builder is not None: + components.append("✅ Document Overviews") + print(f" Components: {', '.join(components)}") print(f" Batch sizes: Embeddings={self.embedding_batch_size}, Enrichment={self.enrichment_batch_size}") diff --git a/rag_system/pipelines/retrieval_pipeline.py b/rag_system/pipelines/retrieval_pipeline.py index f2151273..d7b827fd 100644 --- a/rag_system/pipelines/retrieval_pipeline.py +++ b/rag_system/pipelines/retrieval_pipeline.py @@ -1,32 +1,25 @@ -import pymupdf -from typing import List, Dict, Any, Tuple, Optional -from PIL import Image +from typing import List, Dict, Any, Optional import concurrent.futures import time import json -import lancedb import logging import math +import os import numpy as np from threading import Lock from rag_system.utils.ollama_client import OllamaClient -from rag_system.retrieval.retrievers import MultiVectorRetriever, GraphRetriever -from rag_system.indexing.multimodal import LocalVisionModel -from rag_system.indexing.representations import select_embedder +from rag_system.retrieval.retrievers import MultiVectorRetriever +from rag_system.indexing.representations import default_query_instruction, select_embedder from rag_system.indexing.embedders import LanceDBManager -from rag_system.rerankers.reranker import QwenReranker +from rag_system.rerankers.reranker import CrossEncoderReranker, QwenRerankerScorer, is_qwen3_reranker from rag_system.rerankers.sentence_pruner import SentencePruner -# from rag_system.indexing.chunk_store import ChunkStore - -import os -from PIL import Image # --------------------------------------------------------------------------- # Thread-safety helpers # --------------------------------------------------------------------------- -# 1. ColBERT (via `rerankers` lib) is not thread-safe. We protect the actual +# 1. The `rerankers` lib backends are not thread-safe. We protect the actual # `.rank()` call with `_rerank_lock`. _rerank_lock: Lock = Lock() @@ -42,27 +35,77 @@ class RetrievalPipeline: """ - Orchestrates the state-of-the-art multimodal RAG pipeline. + Orchestrates retrieval, reranking, context expansion, pruning and synthesis. """ def __init__(self, config: Dict[str, Any], ollama_client: OllamaClient, ollama_config: Dict[str, Any]): self.config = config self.ollama_config = ollama_config self.ollama_client = ollama_client - - # Support both legacy "retrievers" key and newer "retrieval" key - self.retriever_configs = self.config.get("retrievers") or self.config.get("retrieval", {}) + self.storage_config = self.config["storage"] - + # Defer initialization to just-in-time methods self.db_manager = None self.text_embedder = None self.dense_retriever = None - self.bm25_retriever = None - # Use a private attribute to avoid clashing with the public property - self._graph_retriever = None - self.reranker = None self.ai_reranker = None + def _retriever_config(self, name: str, *aliases: str) -> Dict[str, Any]: + """Look a retriever sub-config up in "retrievers" then "retrieval". + + Resolved on every access instead of cached in ``__init__`` so runtime + overrides written by the API land in the pipeline. + """ + for container_key in ("retrievers", "retrieval"): + container = self.config.get(container_key) or {} + for key in (name, *aliases): + if container.get(key): + return container[key] + return {} + + def _retrieval_mode(self) -> str: + """Query-time search mode: "hybrid" (default), "vector_only" or "fts_only".""" + retrieval_cfg = self.config.get("retrieval") or {} + return retrieval_cfg.get("search_type") or self.config.get("search_type") or "hybrid" + + def _retry_config(self) -> Dict[str, Any]: + """The evidence-sufficiency retry block (roadmap item 2.1). + + Merged across both container spellings so a runtime override written by + the API under ``retrievers.retry`` beats the profile's + ``retrieval.retry``, matching how ``_latechunk_config`` behaves. + """ + merged: Dict[str, Any] = {} + for container_key in ("retrieval", "retrievers"): + block = (self.config.get(container_key) or {}).get("retry") + if isinstance(block, dict): + merged.update(block) + return merged + + def _latechunk_config(self) -> Dict[str, Any]: + """Merge the late-chunk block across both container and key spellings. + + The profile declares it under ``retrieval.late_chunking`` while the API + toggles it at runtime under ``retrievers.latechunk``; later writes win. + """ + merged: Dict[str, Any] = {} + for container_key in ("retrieval", "retrievers"): + container = self.config.get(container_key) or {} + for key in ("late_chunking", "latechunk"): + block = container.get(key) + if isinstance(block, dict): + merged.update(block) + return merged + + @staticmethod + def _latechunk_table_name(latechunk_cfg: Dict[str, Any], base_table: str) -> Optional[str]: + explicit = latechunk_cfg.get("lancedb_table_name") + if explicit: + return explicit + # "_lc" is the suffix IndexingPipeline writes when none is configured. + suffix = latechunk_cfg.get("table_suffix", "_lc") + return f"{base_table}{suffix}" if suffix and base_table else None + def _get_db_manager(self): if self.db_manager is None: # Accept either "db_path" (preferred) or legacy "lancedb_uri" @@ -72,66 +115,65 @@ def _get_db_manager(self): self.db_manager = LanceDBManager(db_path=db_path) return self.db_manager + def _query_instruction(self, model_name: str) -> str: + """The query-side instruction prefix for this pipeline's embedder. + + Resolution order, most explicit first: + + 1. ``config["embedding_instruction"]`` — set it to ``""`` to switch the + prefix off for a model whose family would otherwise get one. + 2. ``EMBEDDING_INSTRUCTION`` env var — same semantics, for A/B runs. + 3. The model family's official retrieval instruction (Qwen3-Embedding + and harrier-oss-v1), or ``""`` for everything else. + + This is the QUERY side only. Documents are embedded by + ``IndexingPipeline``, which calls ``select_embedder`` without an + instruction, so an index built before this existed remains valid. + """ + configured = self.config.get("embedding_instruction") + if configured is not None: + return configured + env = os.getenv("EMBEDDING_INSTRUCTION") + if env is not None: + return env + return default_query_instruction(model_name) + def _get_text_embedder(self): if self.text_embedder is None: - from rag_system.indexing.representations import select_embedder + model_name = self.config.get("embedding_model_name") + if not model_name: + raise ValueError( + "Config must contain 'embedding_model_name'. Falling back to a hard-coded " + "default here would silently produce vectors whose dimensionality does not " + "match the index." + ) + instruction = self._query_instruction(model_name) + if instruction: + print(f"🔧 Query-side embedding instruction active: '{instruction}'") self.text_embedder = select_embedder( - self.config.get("embedding_model_name", "BAAI/bge-small-en-v1.5"), + model_name, self.ollama_config.get("host") if isinstance(self.ollama_config, dict) else None, + query_instruction=instruction, ) return self.text_embedder def _get_dense_retriever(self): - """Ensure a dense MultiVectorRetriever is always available unless explicitly disabled.""" + """Ensure a MultiVectorRetriever is always available unless explicitly disabled.""" if self.dense_retriever is None: # If the config explicitly sets dense.enabled to False, respect it - if self.retriever_configs.get("dense", {}).get("enabled", True) is False: + if self._retriever_config("dense").get("enabled", True) is False: return None try: - db_manager = self._get_db_manager() - text_embedder = self._get_text_embedder() - fusion_cfg = self.config.get("fusion", {}) self.dense_retriever = MultiVectorRetriever( - db_manager, - text_embedder, - vision_model=None, - fusion_config=fusion_cfg, + self._get_db_manager(), + self._get_text_embedder(), ) except Exception as e: print(f"❌ Failed to initialise dense retriever: {e}") self.dense_retriever = None return self.dense_retriever - def _get_bm25_retriever(self): - if self.bm25_retriever is None and self.retriever_configs.get("bm25", {}).get("enabled"): - try: - print(f"🔧 Lazily initializing BM25 retriever...") - self.bm25_retriever = BM25Retriever( - index_path=self.storage_config["bm25_path"], - index_name=self.retriever_configs["bm25"]["index_name"] - ) - print("✅ BM25 retriever initialized successfully") - except Exception as e: - print(f"❌ Failed to initialize BM25 retriever on demand: {e}") - # Keep it None so we don't try again - return self.bm25_retriever - - def _get_graph_retriever(self): - if self._graph_retriever is None and self.retriever_configs.get("graph", {}).get("enabled"): - self._graph_retriever = GraphRetriever(graph_path=self.storage_config["graph_path"]) - return self._graph_retriever - - def _get_reranker(self): - """Initializes the reranker for hybrid search score fusion.""" - reranker_config = self.config.get("reranker", {}) - # This is for the LanceDB internal reranker, not the AI one. - if self.reranker is None and reranker_config.get("type") == "linear_combination": - rerank_weight = reranker_config.get("weight", 0.5) - self.reranker = lancedb.rerankers.LinearCombinationReranker(weight=rerank_weight) - print(f"✅ Initialized LinearCombinationReranker with weight {rerank_weight}") - return self.reranker - def _get_ai_reranker(self): """Initializes a dedicated AI-based reranker.""" reranker_config = self.config.get("reranker", {}) @@ -142,22 +184,34 @@ def _get_ai_reranker(self): with _ai_reranker_init_lock: # Another thread may have completed init while we waited if self.ai_reranker is None: + model_name = reranker_config.get("model_name") + if not model_name: + print("⚠️ Reranking is enabled but 'reranker.model_name' is not configured; skipping reranking.") + return None try: - model_name = reranker_config.get("model_name") - strategy = reranker_config.get("strategy", "qwen") - - if strategy == "rerankers-lib": - print(f"🔧 Initialising Answer.AI ColBERT reranker ({model_name}) via rerankers lib…") + strategy = reranker_config.get("strategy", "rerankers-lib") + model_type = reranker_config.get("model_type", "cross-encoder") + + # Qwen3-Reranker is a causal-LM yes/no-logit scorer, not a + # SequenceClassification model. The rerankers lib silently + # loads it with a randomly-initialised score head, so route + # the whole family to our own scorer — either by explicit + # `reranker.model_type: "qwen3"` or by model name. + if model_type == "qwen3" or is_qwen3_reranker(model_name): + print(f"🔧 Initialising Qwen3 yes/no-logit reranker ({model_name})…") + self.ai_reranker = QwenRerankerScorer(model_name=model_name) + elif strategy == "rerankers-lib": + print(f"🔧 Initialising {model_type} reranker ({model_name}) via rerankers lib…") from rerankers import Reranker - self.ai_reranker = Reranker(model_name, model_type="colbert") + self.ai_reranker = Reranker(model_name, model_type=model_type) else: - print(f"🔧 Lazily initializing Qwen reranker ({model_name})…") - self.ai_reranker = QwenReranker(model_name=model_name) + print(f"🔧 Lazily initializing local cross-encoder reranker ({model_name})…") + self.ai_reranker = CrossEncoderReranker(model_name=model_name) print("✅ AI reranker initialized successfully.") except Exception as e: # Leave as None so the pipeline can proceed without reranking - print(f"❌ Failed to initialize AI reranker: {e}") + print(f"⚠️ Could not load reranker '{model_name}' ({e}). Continuing without reranking.") return self.ai_reranker def _get_sentence_pruner(self): @@ -167,6 +221,113 @@ def _get_sentence_pruner(self): self._sentence_pruner = SentencePruner() return self._sentence_pruner + # ------------------------------------------------------------------ + # Evidence-sufficiency retry (roadmap item 2.1) + # ------------------------------------------------------------------ + + # Candidates from this rank onwards are treated as "background": chunks the + # query pulled in because they are documents in this corpus, not because + # they answer it. Rank 6 of 20 leaves the plausible answers out of the + # background estimate while still averaging over enough rows to be stable. + _EVIDENCE_BACKGROUND_FROM = 5 + + @classmethod + def _dense_evidence_score(cls, docs: List[Dict[str, Any]]) -> Optional[float]: + """A 0–1 "did we actually find something" score from the dense leg. + + The naive choice — the raw top cosine similarity — was **measured and + rejected**: on the gold set it is *anti*-correlated with success, + because absolute similarity mostly encodes how close the query's + phrasing sits to the corpus's register, not whether the answer-bearing + chunk was found. The three `mixed` first-stage misses all scored a + *higher* top cosine than the median successful query. + + What does carry signal is **contrast**: how far the best candidate + stands above the background of everything else the query pulled in. + + score = (cos_top − cos_background) / (1 − cos_background) + + where ``cos_background`` is the mean cosine of the candidates from rank + ``_EVIDENCE_BACKGROUND_FROM`` down. The denominator rescales against the + headroom that is actually reachable for this query, keeping the result + in 0–1 and comparable across queries whose background level differs. + + Returns ``None`` when the dense leg did not run (``fts_only``) or the + table predates cosine normalization, in which case the caller must not + retry — the number would not mean anything. Requires L2-normalized + vectors (v4+ tables), where LanceDB's squared-L2 ``_distance`` maps to + cosine as ``cos = 1 − d/2``. + """ + sims = [] + for doc in docs: + distance = doc.get("_distance") + if distance is None: + continue + try: + sims.append(1.0 - float(distance) / 2.0) + except (TypeError, ValueError): + continue + if len(sims) < 2: + return None + sims.sort(reverse=True) + tail = sims[cls._EVIDENCE_BACKGROUND_FROM:] or sims[1:] + background = sum(tail) / len(tail) + headroom = 1.0 - background + if headroom <= 1e-6: + return None + return max(0.0, min(1.0, (sims[0] - background) / headroom)) + + @staticmethod + def _rerank_evidence_score(docs: List[Dict[str, Any]]) -> Optional[float]: + """Top reranker score, when the reranker produces a calibrated 0–1 one. + + ``QwenRerankerScorer`` returns P("yes") per candidate, which is directly + interpretable. Other backends return arbitrary logits, so anything + outside 0–1 is rejected rather than silently compared to a probability + threshold. + """ + scores = [d.get("rerank_score") for d in docs if d.get("rerank_score") is not None] + if not scores: + return None + try: + top = float(max(scores)) + except (TypeError, ValueError): + return None + if not (0.0 <= top <= 1.0): + return None + return top + + def _reformulate_query(self, query: str) -> Optional[str]: + """One rewrite of a query whose first pass found weak evidence. + + Runs on the enrichment (utility) model, not the generation model, and + asks for JSON so the "thinking" preamble small models emit cannot leak + into the rewritten query. + """ + model = (self.ollama_config.get("enrichment_model") + or self.ollama_config.get("generation_model")) + if not model: + return None + prompt = ( + "A document search for the question below returned weak matches.\n" + "Rewrite it once as a single self-contained search query that uses the " + "concrete nouns, technical terms and synonyms a document would actually " + "use, instead of the asker's phrasing. Keep every entity, number and " + "constraint from the original. Do not answer the question.\n\n" + f'Question: "{query}"\n\n' + 'Respond with JSON: {"query": ""}' + ) + try: + resp = self.ollama_client.generate_completion(model=model, prompt=prompt, format="json") + data = json.loads(resp.get("response", "{}")) + except Exception as e: + print(f"⚠️ Retry reformulation failed ({e}); keeping the original query.") + return None + rewritten = (data.get("query") or "").strip() if isinstance(data, dict) else "" + if not rewritten or rewritten.lower() == query.strip().lower(): + return None + return rewritten + def _get_surrounding_chunks_lancedb(self, chunk: Dict[str, Any], window_size: int) -> List[Dict[str, Any]]: """ Retrieves a window of chunks around a central chunk using LanceDB. @@ -256,45 +417,39 @@ def _synthesize_final_answer(self, query: str, facts: str, *, event_callback=Non return "".join(answer_parts) - def run(self, query: str, table_name: str = None, window_size_override: Optional[int] = None, event_callback=None) -> Dict[str, Any]: - start_time = time.time() - retrieval_k = self.config.get("retrieval_k", 10) + def _first_stage(self, query: str, base_table: str, retrieval_k: int, retrieval_mode: str, + event_callback=None) -> List[Dict[str, Any]]: + """Hybrid/vector/FTS retrieval plus the optional late-chunk table and merge. + Split out of ``run()`` so the evidence-sufficiency retry can call it a + second time with a reformulated query without duplicating any of it. + """ + start_time = time.time() logger = logging.getLogger(__name__) - logger.debug("--- Running Hybrid Search for query '%s' (table=%s) ---", query, table_name or self.storage_config.get("text_table_name")) - - # If a custom table_name is provided, propagate it to storage config so helper methods use it - if table_name: - self.storage_config["text_table_name"] = table_name - - if event_callback: - event_callback("retrieval_started", {}) - # Unified retrieval using the refactored MultiVectorRetriever dense_retriever = self._get_dense_retriever() - # Get the LanceDB reranker for initial score fusion - lancedb_reranker = self._get_reranker() - + retrieved_docs = [] if dense_retriever: retrieved_docs = dense_retriever.retrieve( text_query=query, - table_name=table_name or self.storage_config["text_table_name"], + table_name=base_table, k=retrieval_k, - reranker=lancedb_reranker # Pass the reranker to enable hybrid search + search_type=retrieval_mode, ) # --------------------------------------------------------------- # Late-Chunk retrieval (optional) # --------------------------------------------------------------- - if self.retriever_configs.get("latechunk", {}).get("enabled"): - lc_table = self.retriever_configs["latechunk"].get("lancedb_table_name") + latechunk_cfg = self._latechunk_config() + if dense_retriever and latechunk_cfg.get("enabled"): + lc_table = self._latechunk_table_name(latechunk_cfg, base_table) if lc_table: try: lc_docs = dense_retriever.retrieve( text_query=query, table_name=lc_table, k=retrieval_k, - reranker=lancedb_reranker, + search_type=retrieval_mode, ) retrieved_docs.extend(lc_docs) except Exception as e: @@ -302,14 +457,13 @@ def run(self, query: str, table_name: str = None, window_size_override: Optional if event_callback: event_callback("retrieval_done", {"count": len(retrieved_docs)}) - - retrieval_time = time.time() - start_time - logger.debug("Retrieved %s chunks in %.2fs", len(retrieved_docs), retrieval_time) + + logger.debug("Retrieved %s chunks in %.2fs", len(retrieved_docs), time.time() - start_time) # ----------------------------------------------------------- # LATE-CHUNK MERGING (merge ±1 sub-vector into central hit) # ----------------------------------------------------------- - if self.retriever_configs.get("latechunk", {}).get("enabled") and retrieved_docs: + if latechunk_cfg.get("enabled") and retrieved_docs: merged_count = 0 for doc in retrieved_docs: try: @@ -336,62 +490,200 @@ def run(self, query: str, table_name: str = None, window_size_override: Optional if merged_count: print(f"🪄 Late-chunk merging applied to {merged_count} retrieved chunks.") - # --- AI Reranking Step --- - ai_reranker = self._get_ai_reranker() - if ai_reranker and retrieved_docs: - if event_callback: - event_callback("rerank_started", {"count": len(retrieved_docs)}) - print(f"\n--- Reranking top {len(retrieved_docs)} docs with AI model... ---") - start_rerank_time = time.time() - - rerank_cfg = self.config.get("reranker", {}) - top_k_cfg = rerank_cfg.get("top_k") - top_percent = rerank_cfg.get("top_percent") # value in range 0–1 - - if top_percent is not None: - try: - pct = float(top_percent) - assert 0 < pct <= 1 - top_k = max(1, int(len(retrieved_docs) * pct)) - except Exception: - print("⚠️ Invalid top_percent value; falling back to top_k") - top_k = top_k_cfg or len(retrieved_docs) - else: - top_k = top_k_cfg or len(retrieved_docs) + return retrieved_docs - strategy = self.config.get("reranker", {}).get("strategy", "qwen") + # ------------------------------------------------------------------ + # Reranking (roadmap item 2.2: decomposition applies HERE, not first stage) + # ------------------------------------------------------------------ + @staticmethod + def _score_pairs(ai_reranker, strategy: str, query: str, texts: List[str]) -> Dict[int, float]: + """Score every candidate against one query. Returns {candidate index: score}.""" + # Some rerankers-lib backends are not thread-safe; serialise calls. + with _rerank_lock: if strategy == "rerankers-lib": - texts = [d['text'] for d in retrieved_docs] - # ColBERT's Rust backend isn't Sync; serialise calls. - with _rerank_lock: - ranked = ai_reranker.rank(query=query, docs=texts) - # ranked is RankedResults; convert to list of (score, idx) + ranked = ai_reranker.rank(query=query, docs=texts) try: pairs = [(r.score, r.document.doc_id) for r in ranked.results] - if any(p[1] is None for p in pairs): + if any(not isinstance(p[1], int) for p in pairs): pairs = [(r.score, i) for i, r in enumerate(ranked.results)] except Exception: pairs = ranked - # Keep only top_k results if requested - if top_k is not None and len(pairs) > top_k: - pairs = pairs[:top_k] - reranked_docs = [retrieved_docs[idx] | {"rerank_score": score} for score, idx in pairs] else: - try: - reranked_docs = ai_reranker.rerank(query, retrieved_docs, top_k=top_k) - except TypeError: - texts = [d['text'] for d in retrieved_docs] - pairs = ai_reranker.rank(query, texts, top_k=top_k) - reranked_docs = [retrieved_docs[idx] | {"rerank_score": score} for score, idx in pairs] - - rerank_time = time.time() - start_rerank_time - print(f"✅ Reranking completed in {rerank_time:.2f}s. Refined to {len(reranked_docs)} docs.") - if event_callback: - event_callback("rerank_done", {"count": len(reranked_docs)}) + pairs = ai_reranker.rank(query, texts) + return {int(idx): float(score) for score, idx in pairs} + + def _rerank_stage(self, query: str, retrieved_docs: List[Dict[str, Any]], + sub_queries: Optional[List[str]] = None, + event_callback=None) -> List[Dict[str, Any]]: + """Reorder the first-stage candidates. No-op when reranking is off. + + When *sub_queries* is supplied (query decomposition is on **and** the + reranker is on), each candidate is scored against every sub-query and + the per-sub-query scores are aggregated with + ``query_decomposition.rerank_aggregate`` (``"mean"``, the default, or + ``"max"``). This is the whole of roadmap item 2.2: the first stage + always ran on the full original query, because decomposing *there* + dilutes the query semantically, while decomposition applied at reranking + is where the 2026 evidence puts the win. + """ + ai_reranker = self._get_ai_reranker() + if not ai_reranker or not retrieved_docs: + return retrieved_docs + + if event_callback: + event_callback("rerank_started", {"count": len(retrieved_docs)}) + print(f"\n--- Reranking top {len(retrieved_docs)} docs with AI model... ---") + start_rerank_time = time.time() + + rerank_cfg = self.config.get("reranker", {}) + top_k_cfg = rerank_cfg.get("top_k") + top_percent = rerank_cfg.get("top_percent") # value in range 0–1 + + if top_percent is not None: + try: + pct = float(top_percent) + assert 0 < pct <= 1 + top_k = max(1, int(len(retrieved_docs) * pct)) + except Exception: + print("⚠️ Invalid top_percent value; falling back to top_k") + top_k = top_k_cfg or len(retrieved_docs) else: - # If no AI reranker, proceed with the initially retrieved docs - reranked_docs = retrieved_docs + top_k = top_k_cfg or len(retrieved_docs) + + strategy = rerank_cfg.get("strategy", "rerankers-lib") + texts = [d["text"] for d in retrieved_docs] + + queries = [q for q in (sub_queries or []) if q and q.strip()] or [query] + # "mean" is the default because it measured better than "max" on both + # subsets of the item-2.2 A/B (eval/decisions/phase2-pipeline.md §3). + aggregate = (self.config.get("query_decomposition", {}) or {}).get( + "rerank_aggregate", "mean") + if len(queries) > 1: + print(f"🔀 Scoring candidates against {len(queries)} sub-queries " + f"(aggregate={aggregate}).") + + per_query = [self._score_pairs(ai_reranker, strategy, q, texts) for q in queries] + + scores: Dict[int, float] = {} + for idx in range(len(texts)): + values = [m[idx] for m in per_query if idx in m] + if not values: + continue + scores[idx] = (sum(values) / len(values)) if aggregate == "mean" else max(values) + + ordered = sorted(scores.items(), key=lambda kv: kv[1], reverse=True) + if top_k is not None and len(ordered) > top_k: + ordered = ordered[:top_k] + reranked_docs = [retrieved_docs[idx] | {"rerank_score": score} for idx, score in ordered] + + rerank_time = time.time() - start_rerank_time + print(f"✅ Reranking completed in {rerank_time:.2f}s. Refined to {len(reranked_docs)} docs.") + if event_callback: + event_callback("rerank_done", {"count": len(reranked_docs)}) + return reranked_docs + + # ------------------------------------------------------------------ + + def retrieve_candidates(self, query: str, table_name: Optional[str] = None, + sub_queries: Optional[List[str]] = None, + event_callback=None) -> Dict[str, Any]: + """First stage + rerank + the evidence-sufficiency retry around both. + + This is the whole candidate-selection path, factored out of ``run()`` so + that ``eval/run_eval.py`` measures exactly what ships rather than a + reimplementation of it. + + Returns:: + + {"first_stage": [...], # post-retry first-stage ordering + "documents": [...], # after reranking, or == first_stage + "query_used": str, + "retry": {...} | None} + """ + retrieval_k = self.config.get("retrieval_k", 10) + retrieval_mode = self._retrieval_mode() + base_table = table_name or self.storage_config["text_table_name"] + + if event_callback: + event_callback("retrieval_started", {"mode": retrieval_mode}) + + first_stage = self._first_stage(query, base_table, retrieval_k, retrieval_mode, + event_callback) + documents = self._rerank_stage(query, first_stage, sub_queries, event_callback) + + result = {"first_stage": first_stage, "documents": documents, + "query_used": query, "retry": None} + + retry_cfg = self._retry_config() + if not retry_cfg.get("enabled") or not first_stage: + return result + + # Prefer the reranker's calibrated probability when it produced one; + # otherwise fall back to the dense contrast score. RRF ranks are + # deliberately never used — they carry no absolute information. + score = self._rerank_evidence_score(documents) + signal = "rerank" + threshold = retry_cfg.get("min_rerank_score", retry_cfg.get("min_top_score")) + if score is None: + score = self._dense_evidence_score(first_stage) + signal = "dense_contrast" + threshold = retry_cfg.get("min_top_score") + if score is None or threshold is None: + # fts_only, or a legacy unnormalized table: no meaningful signal. + return result + + if score >= float(threshold): + return result + + max_attempts = int(retry_cfg.get("max_attempts", 1) or 0) + if max_attempts < 1: + return result + + print(f"\n🔁 Evidence-sufficiency retry: {signal} score {score:.3f} " + f"< {float(threshold):.3f} — reformulating once.") + reformulated = self._reformulate_query(query) + info = {"signal": signal, "threshold": float(threshold), + "score_before": round(score, 4), "reformulated": reformulated, + "attempted": reformulated is not None, "kept": "original", + "score_after": None} + + if reformulated: + retry_first = self._first_stage(reformulated, base_table, retrieval_k, + retrieval_mode, event_callback) + retry_docs = self._rerank_stage(reformulated, retry_first, sub_queries, + event_callback) + retry_score = (self._rerank_evidence_score(retry_docs) if signal == "rerank" + else self._dense_evidence_score(retry_first)) + info["score_after"] = None if retry_score is None else round(retry_score, 4) + # Keep whichever attempt scored better on the same signal. A retry + # that did not improve the evidence is discarded, not merged. + if retry_score is not None and retry_score > score: + result["first_stage"] = retry_first + result["documents"] = retry_docs + result["query_used"] = reformulated + info["kept"] = "retry" + + result["retry"] = info + if event_callback: + event_callback("retrieval_retry", info) + print(f"🔁 Retry kept the {info['kept']} result set " + f"(score_after={info['score_after']}).") + return result + + def run(self, query: str, table_name: str = None, window_size_override: Optional[int] = None, + event_callback=None, sub_queries: Optional[List[str]] = None) -> Dict[str, Any]: + base_table = table_name or self.storage_config["text_table_name"] + + logger = logging.getLogger(__name__) + logger.debug("--- Running search for query '%s' (table=%s) ---", query, base_table) + + # If a custom table_name is provided, propagate it to storage config so helper methods use it + if table_name: + self.storage_config["text_table_name"] = table_name + + candidates = self.retrieve_candidates(query, base_table, sub_queries, event_callback) + reranked_docs = candidates["documents"] window_size = self.config.get("context_window_size", 1) if window_size_override is not None: @@ -509,57 +801,13 @@ def _clean_val(v): return {"answer": final_answer, "source_documents": final_docs} - # ------------------------------------------------------------------ - # Public utility - # ------------------------------------------------------------------ - def list_document_titles(self, max_items: int = 25) -> List[str]: - """Return up to *max_items* distinct document titles (or IDs). - - This is used only for prompt-routing, so we favour robustness over - perfect recall. If anything goes wrong we return an empty list so - the caller can degrade gracefully. - """ - try: - tbl_name = self.storage_config.get("text_table_name") - if not tbl_name: - return [] - - tbl = self._get_db_manager().get_table(tbl_name) - - field_name = "document_title" if "document_title" in tbl.schema.names else "document_id" - - # Use a cheap SQL filter to grab distinct values; fall back to a - # simple scan if the driver lacks DISTINCT support. - try: - sql = f"SELECT DISTINCT {field_name} FROM tbl LIMIT {max_items}" - rows = tbl.search().where("true").sql(sql).to_list() # type: ignore - titles = [r[field_name] for r in rows if r.get(field_name)] - except Exception: - # Fallback: scan first N rows - rows = tbl.search().select(field_name).limit(max_items * 4).to_list() - seen = set() - titles = [] - for r in rows: - val = r.get(field_name) - if val and val not in seen: - titles.append(val) - seen.add(val) - if len(titles) >= max_items: - break - - # Ensure we don't exceed max_items - return titles[:max_items] - except Exception: - # Any issues (missing table, bad schema, etc.) –> just return [] - return [] - # -------------------- Public helper properties -------------------- @property def retriever(self): - """Lazily exposes the main (dense) retriever so external components - like the ReAct agent tools can call `.retrieve()` directly without - reaching into private helpers. If the retriever has not yet been - instantiated, it is created on first access via `_get_dense_retriever`.""" + """Lazily exposes the MultiVectorRetriever so external components can + call `.retrieve()` directly without reaching into private helpers. If + the retriever has not yet been instantiated, it is created on first + access via `_get_dense_retriever`.""" return self._get_dense_retriever() def update_embedding_model(self, model_name: str): diff --git a/rag_system/requirements.txt b/rag_system/requirements.txt index 1387f755..be434807 100644 --- a/rag_system/requirements.txt +++ b/rag_system/requirements.txt @@ -1,15 +1,10 @@ -colpali-engine PyMuPDF Pillow transformers==4.51.0 torch==2.4.1 torchvision==0.19.1 -lancedb -rank_bm25 -fuzzywuzzy -python-Levenshtein torchaudio -transformers +lancedb sentencepiece accelerate docling diff --git a/rag_system/rerankers/reranker.py b/rag_system/rerankers/reranker.py index 54332d36..886e80cb 100644 --- a/rag_system/rerankers/reranker.py +++ b/rag_system/rerankers/reranker.py @@ -1,12 +1,12 @@ -from transformers import AutoModelForSequenceClassification, AutoTokenizer +from transformers import AutoModelForCausalLM, AutoModelForSequenceClassification, AutoTokenizer import torch -from typing import List, Dict, Any +from typing import List, Dict, Any, Optional, Tuple -class QwenReranker: +class CrossEncoderReranker: """ - A reranker that uses a local Hugging Face transformer model. + A cross-encoder reranker backed by a local Hugging Face sequence-classification model. """ - def __init__(self, model_name: str = "BAAI/bge-reranker-base"): + def __init__(self, model_name: str = "BAAI/bge-reranker-v2-m3"): # Auto-select the best available device: CUDA > MPS > CPU if torch.cuda.is_available(): self.device = "cuda" @@ -14,18 +14,14 @@ def __init__(self, model_name: str = "BAAI/bge-reranker-base"): self.device = "mps" else: self.device = "cpu" - print(f"Initializing BGE Reranker with model '{model_name}' on device '{self.device}'.") + print(f"Initializing cross-encoder reranker with model '{model_name}' on device '{self.device}'.") self.tokenizer = AutoTokenizer.from_pretrained(model_name) self.model = AutoModelForSequenceClassification.from_pretrained( model_name, torch_dtype=torch.float16 if self.device != "cpu" else None, ).to(self.device).eval() - print("BGE Reranker loaded successfully.") - - def _format_instruction(self, query: str, doc: str): - instruction = 'Given a web search query, retrieve relevant passages that answer the query' - return f": {instruction}\n: {query}\n: {doc}" + print("Cross-encoder reranker loaded successfully.") def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = 5, *, early_exit: bool = True, margin: float = 0.4, min_scored: int = 8, batch_size: int = 8) -> List[Dict[str, Any]]: """ @@ -80,10 +76,136 @@ def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = 5, *, return reranked_docs +def is_qwen3_reranker(model_name: str) -> bool: + """True for the Qwen3-Reranker family (causal-LM yes/no scorers).""" + return "qwen3-reranker" in (model_name or "").lower() + + +class QwenRerankerScorer: + """ + Reranker for the ``Qwen/Qwen3-Reranker-*`` family. + + These are **causal LMs**, not ``AutoModelForSequenceClassification`` models. + Loading them through the `rerankers` library's cross-encoder path builds a + ``Qwen3ForSequenceClassification`` with a **randomly initialised** ``score`` + head, which produces meaningless (untrained) scores. This class implements + the scoring scheme published on the model card instead: the query/document + pair is wrapped in the model's chat template, the model is asked whether the + document satisfies the query, and the score is the probability of the "yes" + token against the "no" token at the final position. + + Interface mirrors what ``RetrievalPipeline`` and ``eval/run_eval.py`` expect + from the `rerankers` lib branch: ``rank(query=..., docs=[...])`` returns a + list of ``(score, original_index)`` tuples sorted by score, descending. + """ + + DEFAULT_INSTRUCTION = ( + "Given a web search query, retrieve relevant passages that answer the query" + ) + PREFIX = ( + "<|im_start|>system\nJudge whether the Document meets the requirements " + 'based on the Query and the Instruct provided. Note that the answer can ' + 'only be "yes" or "no".<|im_end|>\n<|im_start|>user\n' + ) + SUFFIX = "<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + + def __init__( + self, + model_name: str = "Qwen/Qwen3-Reranker-0.6B", + *, + instruction: Optional[str] = None, + max_length: int = 2048, + batch_size: int = 8, + device: Optional[str] = None, + ): + if device: + self.device = device + elif torch.cuda.is_available(): + self.device = "cuda" + elif getattr(torch.backends, "mps", None) and torch.backends.mps.is_available(): + self.device = "mps" + else: + self.device = "cpu" + + self.model_name = model_name + self.instruction = instruction or self.DEFAULT_INSTRUCTION + self.max_length = max_length + self.batch_size = batch_size + + print(f"Initializing Qwen3 reranker '{model_name}' on device '{self.device}'.") + self.tokenizer = AutoTokenizer.from_pretrained(model_name, padding_side="left") + self.model = AutoModelForCausalLM.from_pretrained( + model_name, + torch_dtype=torch.float16 if self.device != "cpu" else torch.float32, + ).to(self.device).eval() + + self.token_true_id = self.tokenizer.convert_tokens_to_ids("yes") + self.token_false_id = self.tokenizer.convert_tokens_to_ids("no") + self.prefix_tokens = self.tokenizer.encode(self.PREFIX, add_special_tokens=False) + self.suffix_tokens = self.tokenizer.encode(self.SUFFIX, add_special_tokens=False) + print("Qwen3 reranker loaded successfully.") + + # -- internals --------------------------------------------------------- + + def _format_pair(self, query: str, doc: str) -> str: + return f": {self.instruction}\n: {query}\n: {doc}" + + def _score_batch(self, pairs: List[str]) -> List[float]: + budget = self.max_length - len(self.prefix_tokens) - len(self.suffix_tokens) + enc = self.tokenizer( + pairs, + padding=False, + truncation="longest_first", + return_attention_mask=False, + max_length=max(budget, 16), + ) + enc["input_ids"] = [ + self.prefix_tokens + ids + self.suffix_tokens for ids in enc["input_ids"] + ] + enc = self.tokenizer.pad(enc, padding=True, return_tensors="pt") + enc = {k: v.to(self.device) for k, v in enc.items()} + + with torch.no_grad(): + logits = self.model(**enc).logits[:, -1, :].float() + stacked = torch.stack( + [logits[:, self.token_false_id], logits[:, self.token_true_id]], dim=1 + ) + probs = torch.nn.functional.log_softmax(stacked, dim=1)[:, 1].exp() + return probs.cpu().tolist() + + # -- public API -------------------------------------------------------- + + def score(self, query: str, docs: List[str]) -> List[float]: + """Relevance probability in [0, 1], one per document, in input order.""" + scores: List[float] = [] + for start in range(0, len(docs), self.batch_size): + batch = docs[start : start + self.batch_size] + scores.extend(self._score_batch([self._format_pair(query, d) for d in batch])) + return scores + + def rank(self, query: str, docs: List[str], top_k: Optional[int] = None + ) -> List[Tuple[float, int]]: + """``[(score, original_index), …]`` sorted by score, descending.""" + if not docs: + return [] + scores = self.score(query, docs) + pairs = sorted(enumerate(scores), key=lambda x: x[1], reverse=True) + out = [(score, idx) for idx, score in pairs] + return out[:top_k] if top_k else out + + def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = 5, + **_ignored) -> List[Dict[str, Any]]: + """Dict-in/dict-out form, matching ``CrossEncoderReranker.rerank``.""" + if not documents: + return [] + pairs = self.rank(query, [d["text"] for d in documents], top_k=top_k) + return [documents[idx] | {"rerank_score": score} for score, idx in pairs] + + if __name__ == '__main__': # This test requires an internet connection to download the models. try: - reranker = QwenReranker(model_name="BAAI/bge-reranker-base") + reranker = CrossEncoderReranker(model_name="BAAI/bge-reranker-v2-m3") query = "What is the capital of France?" documents = [ @@ -101,5 +223,5 @@ def rerank(self, query: str, documents: List[Dict[str, Any]], top_k: int = 5, *, print(f" - Score: {doc['rerank_score']:.4f}, Text: {doc['text']}") except Exception as e: - print(f"\nAn error occurred during the QwenReranker test: {e}") + print(f"\nAn error occurred during the CrossEncoderReranker test: {e}") print("Please ensure you have an internet connection for model downloads.") diff --git a/rag_system/retrieval/query_transformer.py b/rag_system/retrieval/query_transformer.py index 77ab5165..0bc63898 100644 --- a/rag_system/retrieval/query_transformer.py +++ b/rag_system/retrieval/query_transformer.py @@ -7,7 +7,7 @@ def __init__(self, llm_client: OllamaClient, llm_model: str): self.llm_client = llm_client self.llm_model = llm_model - def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None) -> List[str]: + def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None, max_sub_queries: int = 10) -> List[str]: """Decompose *query* into standalone sub-queries. Parameters @@ -18,6 +18,8 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None Recent conversation turns (each item should contain at least the original user query under the key ``"query"``). Only the **last 5** turns are included to keep the prompt short. + max_sub_queries : int + Hard cap on the returned sub-queries (``query_decomposition.max_sub_queries``). """ # ---- Limit history to last 5 user turns and extract the queries ---- @@ -285,43 +287,13 @@ def decompose(self, query: str, chat_history: List[Dict[str, Any]] | None = None # Deduplicate while preserving order sub_queries = list(dict.fromkeys(sub_queries)) - # Enforce 10 sub-query limit per new requirements - return sub_queries[:10] + return sub_queries[:max(1, int(max_sub_queries))] except json.JSONDecodeError: print(f"Failed to decode JSON from query decomposer: {response_text}") return [query] -class HyDEGenerator: - def __init__(self, llm_client: OllamaClient, llm_model: str): - self.llm_client = llm_client - self.llm_model = llm_model - - def generate(self, query: str) -> str: - prompt = f"Generate a short, hypothetical document that answers the following question. The document should be dense with keywords and concepts related to the query.\n\nQuery: {query}\n\nHypothetical Document:" - response = self.llm_client.generate_completion(self.llm_model, prompt) - return response.get('response', '') - -class GraphQueryTranslator: - def __init__(self, llm_client: OllamaClient, llm_model: str): - self.llm_client = llm_client - self.llm_model = llm_model - - def _generate_translation_prompt(self, query: str) -> str: - return f""" -You are an expert query planner. Convert the user's question into a structured JSON query for a knowledge graph. -The JSON should contain a 'start_node' (the known entity in the query) and an 'edge_label' (the relationship being asked about). -The graph has nodes (entities) and directed edges (relationships). For example, (Tim Cook) -[IS_CEO_OF]-> (Apple). -Return ONLY the JSON object. - -User Question: "{query}" - -JSON Output: -""" - - def translate(self, query: str) -> Dict[str, Any]: - prompt = self._generate_translation_prompt(query) - response = self.llm_client.generate_completion(self.llm_model, prompt, format="json") - try: - return json.loads(response.get('response', '{}')) - except json.JSONDecodeError: - return {} \ No newline at end of file +# GraphQueryTranslator was removed on 2026-08-09 (roadmap item 2.5) together with +# GraphExtractor and GraphRetriever. Evidence: Documentation/research/ +# academic-evidence-2026.md §6 — GraphRAG loses on single-hop, its multi-hop +# gains are contested, and it costs 41–57x at indexing and up to ~377x in query +# tokens. Nothing in this repo ever armed it. \ No newline at end of file diff --git a/rag_system/retrieval/retrievers.py b/rag_system/retrieval/retrievers.py index 2b418d8d..c6e45a91 100644 --- a/rag_system/retrieval/retrievers.py +++ b/rag_system/retrieval/retrievers.py @@ -1,200 +1,246 @@ -import lancedb -import pickle import json -from typing import List, Dict, Any -import numpy as np -import networkx as nx -import os -from PIL import Image -from transformers import CLIPProcessor, CLIPModel -import torch +from typing import Any, Dict, List, Optional, Tuple import logging -import pandas as pd import math import concurrent.futures from functools import lru_cache -from rag_system.indexing.embedders import LanceDBManager +from rag_system.indexing.embedders import ( + EmbedderMismatchError, + LanceDBManager, + assert_embedder_matches, + l2_normalize, + legacy_table_warning, + read_table_marker, +) from rag_system.indexing.representations import QwenEmbedder -from rag_system.indexing.multimodal import LocalVisionModel from rag_system.utils.logging_utils import log_retrieval_results -# BM25Retriever is no longer needed. -# class BM25Retriever: ... - -from fuzzywuzzy import process - -class GraphRetriever: - def __init__(self, graph_path: str): - self.graph = nx.read_gml(graph_path) - - def retrieve(self, query: str, k: int = 5, score_cutoff: int = 80) -> List[Dict[str, Any]]: - print(f"\n--- Performing Graph Retrieval for query: '{query}' ---") - - query_parts = query.split() - entities = [] - for part in query_parts: - match = process.extractOne(part, self.graph.nodes(), score_cutoff=score_cutoff) - if match and isinstance(match[0], str): - entities.append(match[0]) - - retrieved_docs = [] - for entity in set(entities): - for neighbor in self.graph.neighbors(entity): - retrieved_docs.append({ - 'chunk_id': f"graph_{entity}_{neighbor}", - 'text': f"Entity: {entity}, Neighbor: {neighbor}", - 'score': 1.0, - 'metadata': {'source': 'graph'} - }) - - print(f"Retrieved {len(retrieved_docs)} documents from the graph.") - return retrieved_docs[:k] +# Retrieval modes accepted by MultiVectorRetriever.retrieve(). +RETRIEVAL_MODES = ("hybrid", "vector_only", "fts_only") + +# Reciprocal-rank-fusion constant (Cormack et al. 2009); dampens the influence +# of the top ranks so a single leg cannot dominate the fused ordering. +_RRF_K = 60 + + +def _is_nan(value: Any) -> bool: + return isinstance(value, float) and math.isnan(value) + + +def _finite(value: Any) -> Optional[float]: + """Return *value* as a float, or None when it is missing/NaN/Inf.""" + if value is None or _is_nan(value): + return None + try: + num = float(value) + except (TypeError, ValueError): + return None + if math.isnan(num) or math.isinf(num): + return None + return num + + +# GraphRetriever was removed on 2026-08-09 (roadmap item 2.5). GraphRAG loses on +# single-hop retrieval, its multi-hop gains are contested, and it costs 41–57x at +# indexing and up to ~377x in query tokens — see Documentation/research/ +# academic-evidence-2026.md §6. It was also unreachable: no shipped profile ever +# set `graph_strategy`. + # region === MultiVectorRetriever === class MultiVectorRetriever: """ - Performs hybrid (vector + FTS) or vector-only retrieval. + Runs LanceDB full-text and/or vector search over a single table. """ - def __init__(self, db_manager: LanceDBManager, text_embedder: QwenEmbedder, vision_model: LocalVisionModel = None, *, fusion_config: Dict[str, Any] | None = None): + def __init__(self, db_manager: LanceDBManager, text_embedder: QwenEmbedder): self.db_manager = db_manager self.text_embedder = text_embedder - self.vision_model = vision_model - self.fusion_config = fusion_config or {"method": "linear", "bm25_weight": 0.5, "vec_weight": 0.5} - # Lightweight in-memory LRU cache for single-query embeddings (256 entries) + # Lightweight in-memory LRU cache for single-query embeddings (256 entries). + # The cache holds the raw model output; normalization is applied after the + # lookup because it is a property of the table being searched, not of the + # query (a legacy table must be searched with legacy vectors). @lru_cache(maxsize=256) def _embed_single(q: str): return self.text_embedder.create_embeddings([q])[0] self._embed_single = _embed_single - def retrieve(self, text_query: str, table_name: str, k: int, reranker=None) -> List[Dict[str, Any]]: + # Tables already reported as unmarked, so the warning is printed once each. + self._legacy_tables_warned: set = set() + + def _check_table_identity(self, tbl, table_name: str) -> bool: + """Verify the table's embedder against ours; return whether to normalize. + + Raises ``EmbedderMismatchError`` when the table was written by a + different embedding model — a same-width swap (harrier-oss-v1-0.6b vs + Qwen3-Embedding-0.6B, both 1024-dim) is otherwise undetectable and would + silently return nonsense. """ - Performs a search on a single LanceDB table. - If a reranker is provided, it performs a hybrid search. - Otherwise, it performs a standard vector search. + marker = read_table_marker( + tbl, getattr(self.db_manager, "db_path", None), table_name + ) + if marker is None: + if table_name not in self._legacy_tables_warned: + self._legacy_tables_warned.add(table_name) + print(legacy_table_warning(table_name)) + return False + configured = getattr(self.text_embedder, "model_name", None) + if configured: + assert_embedder_matches(table_name, marker, configured) + return bool(marker["normalized"]) + + @staticmethod + def _dedup_key(row: Dict[str, Any]) -> Tuple[str, Any]: + for field in ("chunk_id", "_rowid"): + value = row.get(field) + if value is not None and not _is_nan(value): + return (field, value) + return ("text", row.get("text")) + + def retrieve(self, text_query: str, table_name: str, k: int, search_type: str = "hybrid") -> List[Dict[str, Any]]: + """ + Retrieves up to *k* chunks from *table_name*. + + `search_type` selects which LanceDB query legs run: "hybrid" (full-text + + vector, fused with reciprocal rank fusion), "vector_only", or + "fts_only". Unknown values fall back to "hybrid". """ - print(f"\n--- Performing Retrieval for query: '{text_query}' on table '{table_name}' ---") - + mode = (search_type or "hybrid").lower() + if mode not in RETRIEVAL_MODES: + print(f"⚠️ Unknown retrieval mode '{search_type}'; falling back to 'hybrid'.") + mode = "hybrid" + + print(f"\n--- Performing {mode} retrieval for query: '{text_query}' on table '{table_name}' ---") + try: if table_name is None: table_name = "default_text_table" tbl = self.db_manager.get_table(table_name) - - # Create / fetch cached text embedding for the query - text_query_embedding = self._embed_single(text_query) - - logger = logging.getLogger(__name__) + # Vectors are stored L2-normalized on v4+ tables, so LanceDB's default + # L2 ordering equals the cosine ordering both model cards specify. The + # query vector has to be normalized the same way — and must *not* be + # for a legacy table, whose documents are unnormalized. + normalize_query = self._check_table_identity(tbl, table_name) - # Always perform hybrid lexical + vector search - logger.debug( - "Running hybrid search on table '%s' (k=%s, have_reranker=%s)", - table_name, - k, - bool(reranker), - ) - - if reranker: - logger.debug("Hybrid + reranker path not yet implemented with manual fusion; proceeding without extra reranker.") - - # Manual two-leg hybrid: take half from each modality - fts_k = k // 2 - vec_k = k - fts_k + logger = logging.getLogger(__name__) + logger.debug("Running %s search on table '%s' (k=%s)", mode, table_name, k) - # Run FTS and vector search in parallel to cut latency def _run_fts(): # Very short queries often underperform → add fuzzy wildcard fts_query = text_query if len(text_query.split()) == 1: fts_query = f"{text_query}* OR {text_query}~" return ( - tbl.search(query=fts_query, query_type="fts") - .limit(fts_k) - .to_df() - ) + tbl.search(query=fts_query, query_type="fts") + .limit(k) + .to_df() + ) def _run_vec(): - if vec_k == 0: - return None + vector = self._embed_single(text_query) + if normalize_query: + vector = l2_normalize(vector) return ( - tbl.search(text_query_embedding) - .limit(vec_k * 2) # fetch extra to allow for dedup + tbl.search(vector) + .limit(k) .to_df() ) - with concurrent.futures.ThreadPoolExecutor(max_workers=2) as executor: - fts_future = executor.submit(_run_fts) - vec_future = executor.submit(_run_vec) - fts_df = fts_future.result() - vec_df = vec_future.result() - - if vec_df is not None: - combined = pd.concat([fts_df, vec_df]) + fts_df = None + vec_df = None + if mode == "fts_only": + fts_df = _run_fts() + elif mode == "vector_only": + vec_df = _run_vec() else: - combined = fts_df - - # Remove duplicates preserving first occurrence, then trim to k - dedup_subset = ["_rowid"] if "_rowid" in combined.columns else (["chunk_id"] if "chunk_id" in combined.columns else None) - if dedup_subset: - combined = combined.drop_duplicates(subset=dedup_subset, keep="first") - combined = combined.head(k) + with concurrent.futures.ThreadPoolExecutor(max_workers=2) as executor: + fts_future = executor.submit(_run_fts) + vec_future = executor.submit(_run_vec) + fts_df = fts_future.result() + vec_df = vec_future.result() + + def _records(df) -> List[Dict[str, Any]]: + if df is None or len(df) == 0: + return [] + return df.to_dict("records") + + fts_rows = _records(fts_df) + vec_rows = _records(vec_df) + + # Reciprocal rank fusion: each leg contributes 1/(_RRF_K + rank). + # A single-leg run keeps its native ordering because RRF is + # monotonically decreasing in rank. + fused: Dict[Tuple[str, Any], Dict[str, Any]] = {} + for rank, row in enumerate(fts_rows, start=1): + entry = fused.setdefault(self._dedup_key(row), {"row": row, "rrf": 0.0, "bm25": None, "distance": None}) + entry["rrf"] += 1.0 / (_RRF_K + rank) + entry["bm25"] = _finite(row.get("score")) + for rank, row in enumerate(vec_rows, start=1): + entry = fused.setdefault(self._dedup_key(row), {"row": row, "rrf": 0.0, "bm25": None, "distance": None}) + entry["rrf"] += 1.0 / (_RRF_K + rank) + entry["distance"] = _finite(row.get("_distance")) + + ordered = sorted(fused.values(), key=lambda e: e["rrf"], reverse=True)[:k] - results_df = combined logger.debug( - "Hybrid (fts=%s, vec=%s) → %s unique chunks", - len(fts_df), - 0 if vec_df is None else len(vec_df), - len(results_df), + "%s search (fts=%s, vec=%s) → %s unique chunks", + mode, + len(fts_rows), + len(vec_rows), + len(ordered), ) - + retrieved_docs = [] - for _, row in results_df.iterrows(): - metadata = json.loads(row.get('metadata', '{}')) + for entry in ordered: + row = entry["row"] + raw_metadata = row.get('metadata') + if isinstance(raw_metadata, dict): + metadata = dict(raw_metadata) + else: + try: + metadata = json.loads(raw_metadata or '{}') + except (TypeError, ValueError): + metadata = {} # Add top-level fields back into metadata for consistency if they don't exist metadata.setdefault('document_id', row.get('document_id')) metadata.setdefault('chunk_index', row.get('chunk_index')) - - # Determine score (vector distance or FTS). Replace NaN with 0.0 - raw_score = row.get('_distance') if '_distance' in row else row.get('score') - try: - if raw_score is None or (isinstance(raw_score, float) and math.isnan(raw_score)): - raw_score = 0.0 - except Exception: - raw_score = 0.0 - - combined_score = raw_score - # Optional linear-weight fusion if both FTS & vector scores exist - if '_distance' in row and 'score' in row: - try: - bm25 = row.get('score', 0.0) - vec_sim = 1.0 / (1.0 + row.get('_distance', 1.0)) # convert distance to similarity - w_bm25 = float(self.fusion_config.get('bm25_weight', 0.5)) - w_vec = float(self.fusion_config.get('vec_weight', 0.5)) - combined_score = w_bm25 * bm25 + w_vec * vec_sim - except Exception: - pass - - retrieved_docs.append({ + + # A single score field per mode, always "higher is better". + if mode == "fts_only": + score = entry["bm25"] if entry["bm25"] is not None else 0.0 + elif mode == "vector_only": + distance = entry["distance"] + score = 1.0 / (1.0 + distance) if distance is not None else 0.0 + else: + score = entry["rrf"] + + doc = { 'chunk_id': row.get('chunk_id'), - 'text': metadata.get('original_text', row.get('text')), - 'score': combined_score, - 'bm25': row.get('score'), - '_distance': row.get('_distance'), + 'text': metadata.get('original_text') or row.get('text') or '', + 'score': score, 'document_id': row.get('document_id'), 'chunk_index': row.get('chunk_index'), 'metadata': metadata - }) + } + # Only carry the per-leg raw scores when that leg actually hit, + # so downstream sorting never sees a None. + if entry["bm25"] is not None: + doc['bm25'] = entry["bm25"] + if entry["distance"] is not None: + doc['_distance'] = entry["distance"] + retrieved_docs.append(doc) - logger.debug("Hybrid search returned %s results", len(retrieved_docs)) log_retrieval_results(retrieved_docs, k) print(f"Retrieved {len(retrieved_docs)} documents.") return retrieved_docs - + + except EmbedderMismatchError: + # Never degrade an embedder mismatch into "0 results" — the whole + # point of the guard is that the user has to see it. + raise except Exception as e: print(f"Could not search table '{table_name}': {e}") return [] # endregion - -if __name__ == '__main__': - print("retrievers.py updated for LanceDB FTS Hybrid Search.") diff --git a/rag_system/utils/batch_processor.py b/rag_system/utils/batch_processor.py index 25d1eaeb..d0a318c8 100644 --- a/rag_system/utils/batch_processor.py +++ b/rag_system/utils/batch_processor.py @@ -1,6 +1,6 @@ import time import logging -from typing import List, Dict, Any, Callable, Optional, Iterator +from typing import List, Dict, Any, Callable from contextlib import contextmanager import gc @@ -126,76 +126,8 @@ def process_in_batches( tracker.finish() return results - - def batch_iterator(self, items: List[Any]) -> Iterator[List[Any]]: - """Generate batches as an iterator for memory-efficient processing""" - for i in range(0, len(items), self.batch_size): - yield items[i:i + self.batch_size] - -class StreamingProcessor: - """Process items one at a time with minimal memory usage""" - - def __init__(self, enable_gc_interval: int = 100): - self.enable_gc_interval = enable_gc_interval - - def process_streaming( - self, - items: List[Any], - process_func: Callable, - operation_name: str = "Streaming Processing", - **kwargs - ) -> List[Any]: - """ - Process items one at a time with minimal memory footprint - - Args: - items: List of items to process - process_func: Function to process each item - operation_name: Name for progress reporting - **kwargs: Additional arguments passed to process_func - - Returns: - List of results - """ - if not items: - logger.info(f"{operation_name}: No items to process") - return [] - - tracker = ProgressTracker(len(items), operation_name) - results = [] - - logger.info(f"Starting {operation_name} for {len(items)} items (streaming)") - - with timer(f"{operation_name} (streaming)"): - for i, item in enumerate(items): - try: - result = process_func(item, **kwargs) - results.append(result) - tracker.update(1) - - except Exception as e: - logger.error(f"Error processing item {i}: {e}") - tracker.update(1, errors=1) - continue - - # Periodic garbage collection - if self.enable_gc_interval and (i + 1) % self.enable_gc_interval == 0: - gc.collect() - - tracker.finish() - return results # Utility functions for common batch operations -def batch_chunks_by_document(chunks: List[Dict[str, Any]]) -> Dict[str, List[Dict[str, Any]]]: - """Group chunks by document_id for document-level batch processing""" - document_batches = {} - for chunk in chunks: - doc_id = chunk.get('metadata', {}).get('document_id', 'unknown') - if doc_id not in document_batches: - document_batches[doc_id] = [] - document_batches[doc_id].append(chunk) - return document_batches - def estimate_memory_usage(chunks: List[Dict[str, Any]]) -> float: """Estimate memory usage of chunks in MB""" if not chunks: diff --git a/rag_system/utils/logging_utils.py b/rag_system/utils/logging_utils.py index 88336584..e4bb1406 100644 --- a/rag_system/utils/logging_utils.py +++ b/rag_system/utils/logging_utils.py @@ -12,17 +12,6 @@ ) -def log_query(query: str, sub_queries: List[str] | None = None) -> None: - """Emit a nicely-formatted block describing the incoming query and any - decomposition.""" - border = "=" * 60 - logger.info("\n%s\nUSER QUERY: %s", border, query) - if sub_queries: - for i, q in enumerate(sub_queries, 1): - logger.info(" sub-%d → %s", i, q) - logger.info("%s", border) - - def log_retrieval_results(results: List[Dict], k: int) -> None: """Show chunk_id, truncated text and score for the first *k* rows.""" if not results: diff --git a/rag_system/utils/ollama_client.py b/rag_system/utils/ollama_client.py index ea979d5c..e1674e70 100644 --- a/rag_system/utils/ollama_client.py +++ b/rag_system/utils/ollama_client.py @@ -21,18 +21,6 @@ def _image_to_base64(self, image: Image.Image) -> str: image.save(buffered, format="PNG") return base64.b64encode(buffered.getvalue()).decode('utf-8') - def generate_embedding(self, model: str, text: str) -> List[float]: - try: - response = requests.post( - f"{self.api_url}/embeddings", - json={"model": model, "prompt": text} - ) - response.raise_for_status() - return response.json().get("embedding", []) - except requests.exceptions.RequestException as e: - print(f"Error generating embedding: {e}") - return [] - def generate_completion( self, model: str, @@ -64,9 +52,14 @@ def generate_completion( if images: payload["images"] = [self._image_to_base64(img) for img in images] - # Optional: disable thinking mode for Qwen3 / DeepSeek models + # Thinking models put JSON into the `thinking` field and leave + # `response` empty when format=json, so default thinking off there. + # `think` is the top-level knob /api/generate actually honors + # (chat_template_kwargs is silently ignored by the generate API). + if enable_thinking is None and format == "json": + enable_thinking = False if enable_thinking is not None: - payload["chat_template_kwargs"] = {"enable_thinking": enable_thinking} + payload["think"] = enable_thinking response = requests.post( f"{self.api_url}/generate", @@ -103,8 +96,10 @@ async def generate_completion_async( if images: payload["images"] = [self._image_to_base64(img) for img in images] + if enable_thinking is None and format == "json": + enable_thinking = False if enable_thinking is not None: - payload["chat_template_kwargs"] = {"enable_thinking": enable_thinking} + payload["think"] = enable_thinking try: async with httpx.AsyncClient(timeout=timeout) as client: @@ -137,7 +132,7 @@ def stream_completion( if images: payload["images"] = [self._image_to_base64(img) for img in images] if enable_thinking is not None: - payload["chat_template_kwargs"] = {"enable_thinking": enable_thinking} + payload["think"] = enable_thinking with requests.post(f"{self.api_url}/generate", json=payload, stream=True) as resp: resp.raise_for_status() @@ -155,27 +150,3 @@ def stream_completion( yield chunk if data.get("done"): break - -if __name__ == '__main__': - # This test now requires a VLM model like 'llava' or 'qwen-vl' to be pulled. - print("Ollama client updated for multimodal (VLM) support.") - try: - client = OllamaClient() - # Create a dummy black image for testing - dummy_image = Image.new('RGB', (100, 100), 'black') - - # Test VLM completion - vlm_response = client.generate_completion( - model="llava", # Make sure you have run 'ollama pull llava' - prompt="What color is this image?", - images=[dummy_image] - ) - - if vlm_response and 'response' in vlm_response: - print("\n--- VLM Test Response ---") - print(vlm_response['response']) - else: - print("\nFailed to get VLM response. Is 'llava' model pulled and running?") - - except Exception as e: - print(f"An error occurred: {e}") \ No newline at end of file diff --git a/rag_system/utils/validate_model_config.py b/rag_system/utils/validate_model_config.py deleted file mode 100644 index 6516b497..00000000 --- a/rag_system/utils/validate_model_config.py +++ /dev/null @@ -1,219 +0,0 @@ -#!/usr/bin/env python3 -""" -Model Configuration Validation Script -===================================== - -This script validates the consolidated model configuration system to ensure: -1. No configuration conflicts exist -2. All model names are consistent across components -3. Models are accessible and properly configured -4. The configuration validation system works correctly - -Run this after making configuration changes to catch issues early. -""" - -import sys -import os -# Add parent directories to path for imports -sys.path.append(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))) - -from rag_system.main import ( - PIPELINE_CONFIGS, - OLLAMA_CONFIG, - EXTERNAL_MODELS, - validate_model_config -) - -def print_header(title: str): - """Print a formatted header.""" - print(f"\n{'='*60}") - print(f"🔍 {title}") - print(f"{'='*60}") - -def print_section(title: str): - """Print a formatted section header.""" - print(f"\n{'─'*40}") - print(f"📋 {title}") - print(f"{'─'*40}") - -def validate_configuration_consistency(): - """Validate that all configurations are consistent.""" - print_header("CONFIGURATION CONSISTENCY VALIDATION") - - errors = [] - - # 1. Check embedding model consistency - print_section("Embedding Model Consistency") - default_embedding = PIPELINE_CONFIGS["default"]["embedding_model_name"] - external_embedding = EXTERNAL_MODELS["embedding_model"] - fast_embedding = PIPELINE_CONFIGS["fast"]["embedding_model_name"] - - print(f"Default Config: {default_embedding}") - print(f"External Models: {external_embedding}") - print(f"Fast Config: {fast_embedding}") - - if default_embedding != external_embedding: - errors.append(f"❌ Embedding model mismatch: default={default_embedding}, external={external_embedding}") - elif default_embedding != fast_embedding: - errors.append(f"❌ Embedding model mismatch: default={default_embedding}, fast={fast_embedding}") - else: - print("✅ Embedding models are consistent") - - # 2. Check reranker model consistency - print_section("Reranker Model Consistency") - default_reranker = PIPELINE_CONFIGS["default"]["reranker"]["model_name"] - external_reranker = EXTERNAL_MODELS["reranker_model"] - - print(f"Default Config: {default_reranker}") - print(f"External Models: {external_reranker}") - - if default_reranker != external_reranker: - errors.append(f"❌ Reranker model mismatch: default={default_reranker}, external={external_reranker}") - else: - print("✅ Reranker models are consistent") - - # 3. Check vision model consistency - print_section("Vision Model Consistency") - default_vision = PIPELINE_CONFIGS["default"]["vision_model_name"] - external_vision = EXTERNAL_MODELS["vision_model"] - - print(f"Default Config: {default_vision}") - print(f"External Models: {external_vision}") - - if default_vision != external_vision: - errors.append(f"❌ Vision model mismatch: default={default_vision}, external={external_vision}") - else: - print("✅ Vision models are consistent") - - return errors - -def print_model_usage_map(): - """Print a comprehensive map of which models are used where.""" - print_header("MODEL USAGE MAP") - - print_section("🤖 Ollama Models (Local Inference)") - for model_type, model_name in OLLAMA_CONFIG.items(): - if model_type != "host": - print(f" {model_type.replace('_', ' ').title()}: {model_name}") - - print_section("🔗 External Models (HuggingFace/Direct)") - for model_type, model_name in EXTERNAL_MODELS.items(): - print(f" {model_type.replace('_', ' ').title()}: {model_name}") - - print_section("📍 Model Usage by Component") - usage_map = { - "🔤 Text Embedding": { - "Model": EXTERNAL_MODELS["embedding_model"], - "Used In": ["Retrieval Pipeline", "Semantic Cache", "Dense Retrieval", "Late Chunking"], - "Component": "QwenEmbedder (representations.py)" - }, - "🧠 Text Generation": { - "Model": OLLAMA_CONFIG["generation_model"], - "Used In": ["Agent Loop", "Answer Synthesis", "Query Decomposition", "Verification"], - "Component": "OllamaClient" - }, - "🚀 Enrichment/Routing": { - "Model": OLLAMA_CONFIG["enrichment_model"], - "Used In": ["Query Routing", "Document Overview Analysis"], - "Component": "Agent Loop (_route_via_overviews)" - }, - "🔀 Reranking": { - "Model": EXTERNAL_MODELS["reranker_model"], - "Used In": ["Hybrid Search", "Document Reranking", "AI Reranker"], - "Component": "ColBERT (rerankers-lib) or QwenReranker" - }, - "👁️ Vision": { - "Model": EXTERNAL_MODELS["vision_model"], - "Used In": ["Multimodal Processing", "Image Embeddings"], - "Component": "Vision Pipeline (when enabled)" - } - } - - for model_name, details in usage_map.items(): - print(f"\n{model_name}") - print(f" Model: {details['Model']}") - print(f" Component: {details['Component']}") - print(f" Used In: {', '.join(details['Used In'])}") - -def test_validation_function(): - """Test the built-in validation function.""" - print_header("VALIDATION FUNCTION TEST") - - try: - result = validate_model_config() - if result: - print("✅ validate_model_config() passed successfully!") - else: - print("❌ validate_model_config() returned False") - except Exception as e: - print(f"❌ validate_model_config() failed with error: {e}") - return False - - return True - -def check_pipeline_configurations(): - """Check all pipeline configurations for completeness.""" - print_header("PIPELINE CONFIGURATION COMPLETENESS") - - required_keys = { - "default": ["storage", "retrieval", "embedding_model_name", "reranker"], - "fast": ["storage", "retrieval", "embedding_model_name"] - } - - errors = [] - - for config_name, required in required_keys.items(): - print_section(f"{config_name.title()} Configuration") - config = PIPELINE_CONFIGS.get(config_name, {}) - - for key in required: - if key in config: - print(f" ✅ {key}: {type(config[key]).__name__}") - else: - error_msg = f"❌ Missing required key '{key}' in {config_name} config" - errors.append(error_msg) - print(f" {error_msg}") - - return errors - -def main(): - """Run all validation checks.""" - print("🚀 Starting Model Configuration Validation") - print(f"Python Path: {sys.path[0]}") - - all_errors = [] - - # Run all validation checks - all_errors.extend(validate_configuration_consistency()) - all_errors.extend(check_pipeline_configurations()) - - # Print model usage map - print_model_usage_map() - - # Test validation function - validation_passed = test_validation_function() - - # Final summary - print_header("VALIDATION SUMMARY") - - if all_errors: - print("❌ VALIDATION FAILED - Issues Found:") - for error in all_errors: - print(f" {error}") - return 1 - elif not validation_passed: - print("❌ VALIDATION FAILED - validate_model_config() function failed") - return 1 - else: - print("✅ ALL VALIDATIONS PASSED!") - print("\n🎉 Your model configuration is consistent and properly structured!") - print("\n📋 Summary:") - print(f" • Embedding Model: {EXTERNAL_MODELS['embedding_model']}") - print(f" • Generation Model: {OLLAMA_CONFIG['generation_model']}") - print(f" • Enrichment Model: {OLLAMA_CONFIG['enrichment_model']}") - print(f" • Reranker Model: {EXTERNAL_MODELS['reranker_model']}") - print(f" • Vision Model: {EXTERNAL_MODELS['vision_model']}") - return 0 - -if __name__ == "__main__": - sys.exit(main()) \ No newline at end of file diff --git a/rag_system/utils/watsonx_client.py b/rag_system/utils/watsonx_client.py index 1c926346..91377c5c 100644 --- a/rag_system/utils/watsonx_client.py +++ b/rag_system/utils/watsonx_client.py @@ -58,27 +58,6 @@ def _image_to_base64(self, image: Image.Image) -> str: image.save(buffered, format="PNG") return base64.b64encode(buffered.getvalue()).decode('utf-8') - def generate_embedding(self, model: str, text: str) -> List[float]: - """ - Generate embeddings using Watson X embedding models. - Note: This requires using Watson X embedding models through the embeddings API. - """ - try: - from ibm_watsonx_ai.foundation_models import Embeddings - - embedding_model = Embeddings( - model_id=model, - credentials=self.credentials, - project_id=self.project_id - ) - - result = embedding_model.embed_query(text) - return result if isinstance(result, list) else [] - - except Exception as e: - print(f"Error generating embedding: {e}") - return [] - def generate_completion( self, model: str, diff --git a/requirements-docker.txt b/requirements-docker.txt index a1f57884..cb1f423e 100644 --- a/requirements-docker.txt +++ b/requirements-docker.txt @@ -1,31 +1,20 @@ requests python-dotenv -PyPDF2 -colpali-engine PyMuPDF Pillow transformers==4.51.0 torch==2.4.1 torchvision==0.19.1 -lancedb -rank_bm25 -fuzzywuzzy -python-Levenshtein torchaudio +lancedb sentencepiece accelerate docling cachetools numpy -networkx -matplotlib psutil httpx -scikit-learn pandas -sentence_transformers rerankers -nltk -# Standard library modules (no need to install) -# asyncio, logging, json, os, sys, typing, threading, itertools, math, re -# ocrmac - removed for Docker compatibility (macOS-specific) +# ocrmac - removed for Docker compatibility (macOS-specific); docling's bundled +# EasyOCR backend handles OCR on Linux. diff --git a/requirements.txt b/requirements.txt index f768bb66..e1834647 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,32 +1,20 @@ requests python-dotenv -PyPDF2 -colpali-engine -requests -python-dotenv -PyPDF2 -colpali-engine PyMuPDF Pillow transformers==4.51.0 torch==2.4.1 torchvision==0.19.1 -lancedb -rank_bm25 -fuzzywuzzy -python-Levenshtein torchaudio +lancedb sentencepiece accelerate docling cachetools numpy -networkx -matplotlib psutil httpx -scikit-learn pandas -sentence_transformers rerankers -nltk +# Optional: IBM watsonx.ai backend, enabled with LLM_BACKEND=watsonx +# ibm-watsonx-ai>=1.3.39 diff --git a/run_system.py b/run_system.py index 8064d6be..660771a5 100644 --- a/run_system.py +++ b/run_system.py @@ -18,6 +18,8 @@ Usage: python run_system.py [--mode dev|prod] [--logs-only] [--no-frontend] + python run_system.py --health + python run_system.py --stop """ import subprocess @@ -36,6 +38,10 @@ from dataclasses import dataclass import psutil +GENERATION_MODEL = os.getenv("GENERATION_MODEL", "qwen3.5:9b") +ENRICHMENT_MODEL = os.getenv("ENRICHMENT_MODEL", "qwen3.5:4b") + + @dataclass class ServiceConfig: name: str @@ -46,6 +52,7 @@ class ServiceConfig: health_check_path: str = "/health" startup_delay: int = 2 required: bool = True + build_command: Optional[List[str]] = None class ColoredFormatter(logging.Formatter): """Custom formatter with colors for different log levels and services.""" @@ -89,11 +96,12 @@ def __init__(self, mode: str = "dev", logs_dir: str = "logs"): self.mode = mode self.logs_dir = Path(logs_dir) self.logs_dir.mkdir(exist_ok=True) - + self.pidfile = self.logs_dir / 'run_system.pid' + self.processes: Dict[str, subprocess.Popen] = {} self.log_threads: Dict[str, threading.Thread] = {} self.running = False - + # Setup logging self.setup_logging() @@ -129,6 +137,7 @@ def _get_service_configs(self) -> Dict[str, ServiceConfig]: name='ollama', command=['ollama', 'serve'], port=11434, + health_check_path='/api/tags', startup_delay=5, required=True ), @@ -150,19 +159,21 @@ def _get_service_configs(self) -> Dict[str, ServiceConfig]: name='frontend', command=['npm', 'run', 'dev' if self.mode == 'dev' else 'start'], port=3000, + health_check_path='/', startup_delay=5, required=False # Optional in case Node.js not available ) } - + # Production mode adjustments if self.mode == 'prod': - # Use production build for frontend + # `next start` needs an existing .next build, so build before starting base_configs['frontend'].command = ['npm', 'run', 'start'] + base_configs['frontend'].build_command = ['npm', 'run', 'build'] # Add production environment variables base_configs['rag-api'].env = {'NODE_ENV': 'production'} base_configs['backend'].env = {'NODE_ENV': 'production'} - + return base_configs def _signal_handler(self, signum, frame): @@ -223,8 +234,8 @@ def ensure_models(self): """Ensure required Ollama models are available.""" self.logger.info("📥 Checking required models...") - required_models = ['qwen3:8b', 'qwen3:0.6b'] - + required_models = [GENERATION_MODEL, ENRICHMENT_MODEL] + try: # Get list of installed models result = subprocess.run(['ollama', 'list'], @@ -256,14 +267,17 @@ def start_service(self, service_name: str, config: ServiceConfig) -> bool: self.logger.warning(f"⚠️ Port {config.port} already in use, skipping {service_name}") return not config.required + if config.build_command and not self._run_build(service_name, config): + return False + self.logger.info(f"🔄 Starting {service_name} on port {config.port}...") - + try: # Setup environment env = os.environ.copy() if config.env: env.update(config.env) - + # Start process process = subprocess.Popen( config.command, @@ -293,15 +307,56 @@ def start_service(self, service_name: str, config: ServiceConfig) -> bool: # Check if process is still running if process.poll() is None: self.logger.info(f"✅ {service_name} started successfully (PID: {process.pid})") + self._write_pidfile() return True else: self.logger.error(f"❌ {service_name} failed to start") + del self.processes[service_name] return False - + except Exception as e: self.logger.error(f"❌ Failed to start {service_name}: {e}") + self.processes.pop(service_name, None) return False - + + def _run_build(self, service_name: str, config: ServiceConfig) -> bool: + """Run a service's build step before starting it (prod frontend needs `next build`).""" + self.logger.info(f"🏗️ Building {service_name}: {' '.join(config.build_command)}") + try: + result = subprocess.run(config.build_command, cwd=config.cwd) + except FileNotFoundError as e: + self.logger.error(f"❌ Build command for {service_name} not found: {e}") + return False + if result.returncode != 0: + self.logger.error(f"❌ Build failed for {service_name} (exit {result.returncode})") + return False + self.logger.info(f"✅ {service_name} build complete") + return True + + def _write_pidfile(self): + """Persist launcher + child PIDs so `--stop` can find them in another shell.""" + data = { + 'launcher': os.getpid(), + 'services': {name: proc.pid for name, proc in self.processes.items()}, + } + try: + self.pidfile.write_text(json.dumps(data, indent=2)) + except OSError as e: + self.logger.warning(f"⚠️ Could not write pidfile {self.pidfile}: {e}") + + def _read_pidfile(self) -> Dict: + try: + return json.loads(self.pidfile.read_text()) + except (OSError, ValueError): + return {} + + def _clear_pidfile(self): + try: + self.pidfile.unlink() + except OSError: + pass + + def _monitor_service_logs(self, service_name: str, process: subprocess.Popen): """Monitor service logs and forward to main logger.""" service_logger = logging.getLogger(service_name) @@ -335,14 +390,43 @@ def _monitor_service_logs(self, service_name: str, process: subprocess.Popen): self.logger.error(f"Error monitoring {service_name} logs: {e}") def health_check(self, service_name: str, config: ServiceConfig) -> bool: - """Perform health check on a service.""" + """Perform an HTTP health check against a service.""" + url = f"http://localhost:{config.port}{config.health_check_path}" try: - url = f"http://localhost:{config.port}{config.health_check_path}" response = requests.get(url, timeout=5) return response.status_code == 200 - except: + except requests.exceptions.RequestException: return False - + + def print_health_report(self) -> bool: + """Run a real health check per service and report pass/fail.""" + self.logger.info("🏥 Health check:") + all_required_healthy = True + + for service_name, config in self.services.items(): + url = f"http://localhost:{config.port}{config.health_check_path}" + if not self.is_port_in_use(config.port): + status = "❌ Not running" + healthy = False + elif self.health_check(service_name, config): + status = "✅ Healthy" + healthy = True + else: + status = "⚠️ Port open, health check failed" + healthy = False + + self.logger.info(f" • {service_name.capitalize():<10}: {status:<32} {url}") + if config.required and not healthy: + all_required_healthy = False + + self.logger.info("") + if all_required_healthy: + self.logger.info("✅ All required services are healthy") + else: + self.logger.error("❌ One or more required services are unhealthy") + return all_required_healthy + + def start_all(self, skip_frontend: bool = False) -> bool: """Start all services in order.""" self.logger.info("🚀 Starting RAG System Components...") @@ -421,24 +505,112 @@ def _print_status_summary(self): self.logger.info("🌐 Access your RAG system at: http://localhost:3000") self.logger.info("") self.logger.info("📋 Useful commands:") - self.logger.info(" • Stop system: Ctrl+C") + self.logger.info(" • Stop system: Ctrl+C (or: python run_system.py --stop)") self.logger.info(" • Check logs: tail -f logs/*.log") self.logger.info(" • Health check: python run_system.py --health") - + + def stop_all(self) -> bool: + """Stop services recorded in the pidfile (works from a separate shell).""" + record = self._read_pidfile() + if not record: + self.logger.warning(f"⚠️ No pidfile at {self.pidfile} – nothing to stop.") + return False + + stopped = 0 + + # Stop the launcher first so its monitor loop cannot restart anything + launcher_pid = record.get('launcher') + if launcher_pid and launcher_pid != os.getpid(): + self._terminate_pid('launcher', launcher_pid) + + for service_name, pid in (record.get('services') or {}).items(): + if self._terminate_pid(service_name, pid): + stopped += 1 + + self._clear_pidfile() + self.logger.info(f"✅ Stopped {stopped} service(s)") + return stopped > 0 + + def _terminate_pid(self, label: str, pid: int) -> bool: + """Terminate a PID, escalating to kill, and reap its children.""" + try: + process = psutil.Process(pid) + except psutil.NoSuchProcess: + self.logger.info(f" • {label}: already stopped (pid {pid})") + return False + except psutil.Error as e: + self.logger.warning(f"⚠️ Could not inspect {label} (pid {pid}): {e}") + return False + + targets = [process] + try: + targets.extend(process.children(recursive=True)) + except psutil.Error: + pass + + for target in reversed(targets): + try: + target.terminate() + except psutil.Error: + pass + + _, alive = psutil.wait_procs(targets, timeout=10) + for target in alive: + try: + target.kill() + except psutil.Error: + pass + + self.logger.info(f" • {label}: stopped (pid {pid})") + return True + + def tail_logs(self): + """Follow every logs/*.log file, like `tail -f logs/*.log`.""" + log_files = sorted(self.logs_dir.glob('*.log')) + if not log_files: + self.logger.warning(f"⚠️ No log files in {self.logs_dir}/ – start the system first.") + return + + self.logger.info(f"📋 Following {len(log_files)} log file(s): {', '.join(f.name for f in log_files)}") + handles = {} + try: + for path in log_files: + handle = path.open('r', errors='replace') + handle.seek(0, os.SEEK_END) + handles[path.stem] = handle + + while True: + emitted = False + for name, handle in handles.items(): + line = handle.readline() + while line: + print(f"[{name.upper()}] {line.rstrip()}") + emitted = True + line = handle.readline() + if not emitted: + time.sleep(0.5) + except KeyboardInterrupt: + self.logger.info("Log tailing stopped by user") + finally: + for handle in handles.values(): + handle.close() + def shutdown(self): """Gracefully shutdown all services.""" if not self.running: return - + self.logger.info("🛑 Shutting down RAG system...") self.running = False - + # Stop services in reverse order for service_name in reversed(list(self.processes.keys())): self._stop_service(service_name) - + + self._clear_pidfile() self.logger.info("✅ All services stopped") - + + def _stop_service(self, service_name: str): """Stop a single service.""" if service_name not in self.processes: @@ -509,22 +681,20 @@ def main(): try: if args.health: - # Health check mode - manager._print_status_summary() - return - + # Health check mode - real HTTP checks against each /health endpoint + sys.exit(0 if manager.print_health_report() else 1) + if args.stop: - # Stop mode - kill any running processes + # Stop mode - terminate the processes recorded in the pidfile manager.logger.info("🛑 Stopping all RAG system processes...") - # Implementation for stopping would go here - return - + sys.exit(0 if manager.stop_all() else 1) + if args.logs_only: - # Logs only mode - just tail existing logs + # Logs only mode - tail the log files written by a running launcher manager.logger.info("📋 Showing aggregated logs... (Press Ctrl+C to stop)") - manager.monitor() + manager.tail_logs() return - + # Normal startup mode if manager.start_all(skip_frontend=args.no_frontend): manager.monitor() diff --git a/setup_rag_system.sh b/setup_rag_system.sh deleted file mode 100644 index aab81518..00000000 --- a/setup_rag_system.sh +++ /dev/null @@ -1,520 +0,0 @@ -#!/bin/bash -# setup_rag_system.sh - Complete RAG System Setup Script -# This script handles Docker installation, system setup, and initial configuration - -set -e - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -BLUE='\033[0;34m' -NC='\033[0m' # No Color - -# Logging function -log() { - echo -e "${GREEN}[$(date +'%Y-%m-%d %H:%M:%S')] $1${NC}" -} - -warn() { - echo -e "${YELLOW}[$(date +'%Y-%m-%d %H:%M:%S')] WARNING: $1${NC}" -} - -error() { - echo -e "${RED}[$(date +'%Y-%m-%d %H:%M:%S')] ERROR: $1${NC}" -} - -info() { - echo -e "${BLUE}[$(date +'%Y-%m-%d %H:%M:%S')] INFO: $1${NC}" -} - -# Check if running as root -if [[ $EUID -eq 0 ]]; then - error "This script should not be run as root (except for package installation steps)" - exit 1 -fi - -echo "================================================================" -echo "🚀 RAG System Complete Setup Script" -echo "================================================================" -echo "" - -# Step 1: System Requirements Check -log "Step 1: Checking system requirements..." - -# Check OS -if [[ "$OSTYPE" == "darwin"* ]]; then - OS="macos" - info "Detected macOS" -elif [[ -f /etc/os-release ]]; then - . /etc/os-release - OS=$ID - info "Detected Linux: $OS" -else - error "Unsupported operating system" - exit 1 -fi - -# Check available memory -MEMORY_GB=$(free -g 2>/dev/null | grep '^Mem:' | awk '{print $2}' || sysctl -n hw.memsize 2>/dev/null | awk '{print int($1/1024/1024/1024)}' || echo "unknown") -if [[ "$MEMORY_GB" != "unknown" && "$MEMORY_GB" -lt 8 ]]; then - warn "System has ${MEMORY_GB}GB RAM. Recommended: 16GB+ for optimal performance" -else - info "Memory check passed: ${MEMORY_GB}GB RAM" -fi - -# Check available disk space -DISK_GB=$(df -BG . | tail -1 | awk '{print $4}' | sed 's/G//' || echo "unknown") -if [[ "$DISK_GB" != "unknown" && "$DISK_GB" -lt 50 ]]; then - warn "Available disk space: ${DISK_GB}GB. Recommended: 50GB+ free space" -else - info "Disk space check passed: ${DISK_GB}GB available" -fi - -# Step 2: Install Dependencies -log "Step 2: Installing system dependencies..." - -# Install Git if not present -if ! command -v git &> /dev/null; then - info "Installing Git..." - case $OS in - "macos") - if command -v brew &> /dev/null; then - brew install git - else - error "Git not found. Please install Git first or install Homebrew" - exit 1 - fi - ;; - "ubuntu"|"debian") - sudo apt-get update - sudo apt-get install -y git - ;; - "centos"|"rhel"|"fedora") - if command -v dnf &> /dev/null; then - sudo dnf install -y git - else - sudo yum install -y git - fi - ;; - esac -else - info "Git is already installed: $(git --version)" -fi - -# Install curl if not present -if ! command -v curl &> /dev/null; then - info "Installing curl..." - case $OS in - "macos") - # curl is usually pre-installed on macOS - ;; - "ubuntu"|"debian") - sudo apt-get install -y curl - ;; - "centos"|"rhel"|"fedora") - if command -v dnf &> /dev/null; then - sudo dnf install -y curl - else - sudo yum install -y curl - fi - ;; - esac -else - info "curl is already installed" -fi - -# Step 3: Install Docker -log "Step 3: Installing Docker..." - -if command -v docker &> /dev/null; then - info "Docker is already installed: $(docker --version)" -else - info "Docker not found. Installing Docker..." - - case $OS in - "macos") - # Check if Homebrew is installed - if ! command -v brew &> /dev/null; then - info "Installing Homebrew..." - /bin/bash -c "$(curl -fsSL https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)" - fi - - # Install Docker Desktop - info "Installing Docker Desktop..." - brew install --cask docker - - warn "Docker Desktop installed. Please:" - warn "1. Start Docker Desktop from Applications" - warn "2. Wait for Docker to start completely" - warn "3. Run this script again" - exit 0 - ;; - - "ubuntu"|"debian") - # Update package index - sudo apt-get update - - # Install dependencies - sudo apt-get install -y \ - ca-certificates \ - curl \ - gnupg \ - lsb-release - - # Add Docker's official GPG key - sudo mkdir -p /etc/apt/keyrings - curl -fsSL https://download.docker.com/linux/$OS/gpg | sudo gpg --dearmor -o /etc/apt/keyrings/docker.gpg - - # Set up repository - echo \ - "deb [arch=$(dpkg --print-architecture) signed-by=/etc/apt/keyrings/docker.gpg] https://download.docker.com/linux/$OS \ - $(lsb_release -cs) stable" | sudo tee /etc/apt/sources.list.d/docker.list > /dev/null - - # Install Docker Engine - sudo apt-get update - sudo apt-get install -y docker-ce docker-ce-cli containerd.io docker-compose-plugin - - # Add user to docker group - sudo usermod -aG docker $USER - - # Start Docker service - sudo systemctl enable docker - sudo systemctl start docker - - info "Docker installed successfully!" - warn "Please log out and log back in for group changes to take effect, then run this script again" - warn "Or run: newgrp docker && $0" - exit 0 - ;; - - "centos"|"rhel"|"fedora") - # Install required packages - if command -v dnf &> /dev/null; then - sudo dnf install -y yum-utils - sudo dnf config-manager --add-repo https://download.docker.com/linux/centos/docker-ce.repo - sudo dnf install -y docker-ce docker-ce-cli containerd.io docker-compose-plugin - else - sudo yum install -y yum-utils - sudo yum-config-manager --add-repo https://download.docker.com/linux/centos/docker-ce.repo - sudo yum install -y docker-ce docker-ce-cli containerd.io docker-compose-plugin - fi - - # Add user to docker group - sudo usermod -aG docker $USER - - # Start Docker service - sudo systemctl enable docker - sudo systemctl start docker - - info "Docker installed successfully!" - warn "Please log out and log back in for group changes to take effect, then run this script again" - exit 0 - ;; - esac -fi - -# Verify Docker is working -if ! docker --version &> /dev/null; then - error "Docker is not working properly. Please check Docker installation" - exit 1 -fi - -if ! docker compose version &> /dev/null; then - error "Docker Compose is not working properly. Please check Docker Compose installation" - exit 1 -fi - -info "Docker verification passed: $(docker --version)" -info "Docker Compose verification passed: $(docker compose version)" - -# Test Docker daemon -if ! docker ps &> /dev/null; then - error "Cannot connect to Docker daemon. Please ensure Docker is running" - exit 1 -fi - -# Step 4: Setup RAG System -log "Step 4: Setting up RAG System..." - -# Create project directory structure -info "Creating directory structure..." -mkdir -p {lancedb,shared_uploads,logs,ollama_data} -mkdir -p index_store/{overviews,bm25,graph} -mkdir -p backups - -# Set proper permissions -chmod 755 {lancedb,shared_uploads,logs,ollama_data} -chmod 755 index_store/{overviews,bm25,graph} -chmod 755 backups - -# Create environment file -if [[ ! -f ".env" ]]; then - info "Creating environment configuration..." - cat > .env << 'EOF' -# System Configuration -NODE_ENV=production -LOG_LEVEL=info -DEBUG=false - -# Service URLs -FRONTEND_URL=http://localhost:3000 -BACKEND_URL=http://localhost:8000 -RAG_API_URL=http://localhost:8001 -OLLAMA_URL=http://localhost:11434 - -# Database Configuration -DATABASE_PATH=./backend/chat_data.db -LANCEDB_PATH=./lancedb -UPLOADS_PATH=./shared_uploads -INDEX_STORE_PATH=./index_store - -# Model Configuration -DEFAULT_EMBEDDING_MODEL=sentence-transformers/all-mpnet-base-v2 -# Default model names - updated to current versions -DEFAULT_GENERATION_MODEL=qwen3:8b -DEFAULT_RERANKER_MODEL=answerdotai/answerai-colbert-small-v1 -DEFAULT_ENRICHMENT_MODEL=qwen3:0.6b - -# Performance Configuration -MAX_CONCURRENT_REQUESTS=5 -REQUEST_TIMEOUT=300 -EMBEDDING_BATCH_SIZE=32 -MAX_CONTEXT_LENGTH=4096 - -# Security Configuration -CORS_ORIGINS=http://localhost:3000 -API_KEY_REQUIRED=false -RATE_LIMIT_REQUESTS=100 -RATE_LIMIT_WINDOW=60 - -# Storage Configuration -MAX_FILE_SIZE=50MB -MAX_UPLOAD_FILES=10 -CLEANUP_INTERVAL=3600 -BACKUP_RETENTION_DAYS=30 -EOF - info "Environment file created: .env" -else - info "Environment file already exists: .env" -fi - -# Step 5: Build and Start Services -log "Step 5: Building and starting services..." - -info "Building Docker containers (this may take 10-15 minutes)..." -docker compose build --no-cache - -info "Starting services..." -docker compose up -d - -# Wait for services to start -info "Waiting for services to initialize..." -sleep 30 - -# Check service status -info "Checking service status..." -docker compose ps - -# Step 6: Install AI Models -log "Step 6: Installing AI models..." - -# Wait for Ollama to be ready -info "Waiting for Ollama to be ready..." -max_attempts=30 -attempt=0 -while ! docker compose exec ollama ollama list &> /dev/null; do - if [ $attempt -ge $max_attempts ]; then - error "Ollama failed to start after $max_attempts attempts" - exit 1 - fi - info "Waiting for Ollama... (attempt $((attempt+1))/$max_attempts)" - sleep 10 - ((attempt++)) -done - -# Download Ollama models -info "Downloading required Ollama models..." -docker compose exec ollama ollama pull qwen3:8b -docker compose exec ollama ollama pull qwen3:0.6b - -info "Verifying model installation..." -docker compose exec ollama ollama list - -# Step 7: System Verification -log "Step 7: Verifying system installation..." - -# Check service health -info "Checking service health..." -services=("frontend:3000" "backend:8000" "rag-api:8001" "ollama:11434") -for service in "${services[@]}"; do - name="${service%:*}" - port="${service#*:}" - - if curl -s -f "http://localhost:$port" &> /dev/null || curl -s -f "http://localhost:$port/health" &> /dev/null || curl -s -f "http://localhost:$port/api/tags" &> /dev/null || curl -s -f "http://localhost:$port/models" &> /dev/null; then - info "✅ $name service is healthy" - else - warn "⚠️ $name service may not be ready yet" - fi -done - -# Step 8: Create Helper Scripts -log "Step 8: Creating helper scripts..." - -# Create start script -cat > start_rag_system.sh << 'EOF' -#!/bin/bash -# Start RAG System -echo "Starting RAG System..." -docker compose up -d -echo "RAG System started. Access at: http://localhost:3000" -EOF -chmod +x start_rag_system.sh - -# Create stop script -cat > stop_rag_system.sh << 'EOF' -#!/bin/bash -# Stop RAG System -echo "Stopping RAG System..." -docker compose down -echo "RAG System stopped." -EOF -chmod +x stop_rag_system.sh - -# Create status script -cat > status_rag_system.sh << 'EOF' -#!/bin/bash -# Check RAG System Status -echo "=== RAG System Status ===" -docker compose ps -echo "" -echo "=== Service Health ===" -curl -s -f http://localhost:3000 && echo "✅ Frontend: OK" || echo "❌ Frontend: FAIL" -curl -s -f http://localhost:8000/health && echo "✅ Backend: OK" || echo "❌ Backend: FAIL" -curl -s -f http://localhost:8001/models && echo "✅ RAG API: OK" || echo "❌ RAG API: FAIL" -curl -s -f http://localhost:11434/api/tags && echo "✅ Ollama: OK" || echo "❌ Ollama: FAIL" -EOF -chmod +x status_rag_system.sh - -# Create backup script -cat > backup_rag_system.sh << 'EOF' -#!/bin/bash -# Backup RAG System Data -BACKUP_DIR="./backups/$(date +%Y%m%d_%H%M%S)" -mkdir -p "$BACKUP_DIR" - -echo "Creating backup in $BACKUP_DIR..." - -# Stop services -docker compose down - -# Backup data -cp -r ./backend/chat_data.db "$BACKUP_DIR/" 2>/dev/null || true -cp -r ./lancedb "$BACKUP_DIR/" 2>/dev/null || true -cp -r ./shared_uploads "$BACKUP_DIR/" 2>/dev/null || true -cp -r ./index_store "$BACKUP_DIR/" 2>/dev/null || true - -# Backup configuration -cp .env "$BACKUP_DIR/" -cp docker-compose.yml "$BACKUP_DIR/" - -# Restart services -docker compose up -d - -echo "Backup completed: $BACKUP_DIR" -EOF -chmod +x backup_rag_system.sh - -# Create update script -cat > update_rag_system.sh << 'EOF' -#!/bin/bash -# Update RAG System -echo "Updating RAG System..." - -# Backup first -./backup_rag_system.sh - -# Pull latest changes -git pull origin main - -# Rebuild containers -docker compose build --no-cache - -# Restart services -docker compose up -d - -echo "Update completed!" -EOF -chmod +x update_rag_system.sh - -info "Helper scripts created:" -info " - start_rag_system.sh: Start the system" -info " - stop_rag_system.sh: Stop the system" -info " - status_rag_system.sh: Check system status" -info " - backup_rag_system.sh: Backup system data" -info " - update_rag_system.sh: Update the system" - -# Step 9: Final Setup -log "Step 9: Final setup and verification..." - -# Create initial database if it doesn't exist -if [[ ! -f "./backend/chat_data.db" ]]; then - info "Creating initial database..." - docker compose exec backend python -c " -import sqlite3 -conn = sqlite3.connect('/app/backend/chat_data.db') -conn.execute('CREATE TABLE IF NOT EXISTS sessions (id TEXT PRIMARY KEY, title TEXT, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)') -conn.execute('CREATE TABLE IF NOT EXISTS messages (id INTEGER PRIMARY KEY, session_id TEXT, content TEXT, role TEXT, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)') -conn.execute('CREATE TABLE IF NOT EXISTS indexes (id TEXT PRIMARY KEY, name TEXT, metadata TEXT, created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)') -conn.execute('CREATE TABLE IF NOT EXISTS session_indexes (session_id TEXT, index_id TEXT, PRIMARY KEY (session_id, index_id))') -conn.commit() -conn.close() -print('Database initialized') -" 2>/dev/null || warn "Database initialization may have failed" -fi - -# Final health check -info "Performing final health check..." -sleep 10 -./status_rag_system.sh - -echo "" -echo "================================================================" -echo "🎉 RAG System Setup Complete!" -echo "================================================================" -echo "" -echo "✅ System Status:" -echo " - Frontend: http://localhost:3000" -echo " - Backend API: http://localhost:8000" -echo " - RAG API: http://localhost:8001" -echo " - Ollama: http://localhost:11434" -echo "" -echo "📚 Documentation:" -echo " - System Overview: Documentation/system_overview.md" -echo " - Deployment Guide: Documentation/deployment_guide.md" -echo " - Docker Usage: Documentation/docker_usage.md" -echo " - Installation Guide: Documentation/installation_guide.md" -echo "" -echo "🔧 Helper Scripts:" -echo " - Start system: ./start_rag_system.sh" -echo " - Stop system: ./stop_rag_system.sh" -echo " - Check status: ./status_rag_system.sh" -echo " - Backup data: ./backup_rag_system.sh" -echo " - Update system: ./update_rag_system.sh" -echo "" -echo "🚀 Next Steps:" -echo " 1. Open http://localhost:3000 in your browser" -echo " 2. Create a new chat session" -echo " 3. Upload some PDF documents" -echo " 4. Start asking questions about your documents!" -echo "" -echo "📋 System Information:" -echo " - OS: $OS" -echo " - Memory: ${MEMORY_GB}GB" -echo " - Disk Space: ${DISK_GB}GB available" -echo " - Docker: $(docker --version)" -echo " - Docker Compose: $(docker compose version)" -echo "" -echo "For support and troubleshooting, check the documentation in the" -echo "Documentation/ folder or run ./status_rag_system.sh to check system health." -echo "" \ No newline at end of file diff --git a/simple_create_index.sh b/simple_create_index.sh deleted file mode 100755 index ebe5b845..00000000 --- a/simple_create_index.sh +++ /dev/null @@ -1,228 +0,0 @@ -#!/bin/bash - -# Simple Index Creation Script for LocalGPT RAG System -# Usage: ./simple_create_index.sh "Index Name" "path/to/document.pdf" [additional_files...] - -set -e # Exit on any error - -# Colors for output -RED='\033[0;31m' -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -BLUE='\033[0;34m' -NC='\033[0m' # No Color - -# Function to print colored output -print_status() { - echo -e "${BLUE}[INFO]${NC} $1" -} - -print_success() { - echo -e "${GREEN}[SUCCESS]${NC} $1" -} - -print_warning() { - echo -e "${YELLOW}[WARNING]${NC} $1" -} - -print_error() { - echo -e "${RED}[ERROR]${NC} $1" -} - -# Function to check if a command exists -command_exists() { - command -v "$1" >/dev/null 2>&1 -} - -# Function to check prerequisites -check_prerequisites() { - print_status "Checking prerequisites..." - - # Check Python - if ! command_exists python3; then - print_error "Python 3 is required but not installed." - exit 1 - fi - - # Check if we're in the right directory - if [ ! -f "run_system.py" ] || [ ! -d "rag_system" ]; then - print_error "This script must be run from the LocalGPT project root directory." - exit 1 - fi - - # Check if Ollama is running - if ! curl -s http://localhost:11434/api/tags >/dev/null 2>&1; then - print_error "Ollama is not running. Please start Ollama first:" - echo " ollama serve" - exit 1 - fi - - print_success "Prerequisites check passed" -} - -# Function to validate documents -validate_documents() { - local documents=("$@") - local valid_docs=() - - print_status "Validating documents..." - - for doc in "${documents[@]}"; do - if [ -f "$doc" ]; then - # Check file extension - case "${doc##*.}" in - pdf|txt|docx|md|html|htm) - valid_docs+=("$doc") - print_status "✓ Valid document: $doc" - ;; - *) - print_warning "Unsupported file type: $doc (skipping)" - ;; - esac - else - print_warning "File not found: $doc (skipping)" - fi - done - - if [ ${#valid_docs[@]} -eq 0 ]; then - print_error "No valid documents found." - exit 1 - fi - - echo "${valid_docs[@]}" -} - -# Function to create index using Python -create_index() { - local index_name="$1" - shift - local documents=("$@") - - print_status "Creating index: $index_name" - print_status "Documents: ${documents[*]}" - - # Create a temporary Python script to create the index - cat > /tmp/create_index_temp.py << EOF -#!/usr/bin/env python3 -import sys -import os -import json -sys.path.insert(0, os.getcwd()) - -from rag_system.main import PIPELINE_CONFIGS -from rag_system.pipelines.indexing_pipeline import IndexingPipeline -from rag_system.utils.ollama_client import OllamaClient -from backend.database import ChatDatabase -import uuid - -def create_index_simple(): - try: - # Initialize database - db = ChatDatabase() - - # Create index record - index_id = db.create_index( - name="$index_name", - description="Created with simple_create_index.sh", - metadata={ - "chunk_size": 512, - "chunk_overlap": 64, - "enable_enrich": True, - "enable_latechunk": True, - "retrieval_mode": "hybrid", - "created_by": "simple_create_index.sh" - } - ) - - # Add documents to index - documents = [${documents[@]/#/\"} ${documents[@]/%/\"}] - for doc_path in documents: - if doc_path.strip(): # Skip empty strings - filename = os.path.basename(doc_path.strip()) - db.add_document_to_index(index_id, filename, os.path.abspath(doc_path.strip())) - - # Initialize pipeline - config = PIPELINE_CONFIGS.get("default", {}) - ollama_client = OllamaClient() - ollama_config = { - "generation_model": "qwen3:0.6b", - "embedding_model": "qwen3:0.6b" - } - - pipeline = IndexingPipeline(config, ollama_client, ollama_config) - - # Process documents - valid_docs = [doc.strip() for doc in documents if doc.strip() and os.path.exists(doc.strip())] - if valid_docs: - pipeline.process_documents(valid_docs) - - print(f"✅ Index '{index_name}' created successfully!") - print(f"Index ID: {index_id}") - print(f"Processed {len(valid_docs)} documents") - - return index_id - - except Exception as e: - print(f"❌ Error creating index: {e}") - import traceback - traceback.print_exc() - return None - -if __name__ == "__main__": - create_index_simple() -EOF - - # Run the Python script - python3 /tmp/create_index_temp.py - - # Clean up - rm -f /tmp/create_index_temp.py -} - -# Function to show usage -show_usage() { - echo "Usage: $0 \"Index Name\" \"path/to/document.pdf\" [additional_files...]" - echo "" - echo "Examples:" - echo " $0 \"My Documents\" \"document.pdf\"" - echo " $0 \"Research Papers\" \"paper1.pdf\" \"paper2.pdf\" \"notes.txt\"" - echo " $0 \"Invoice Collection\" ./invoices/*.pdf" - echo "" - echo "Supported file types: PDF, TXT, DOCX, MD, HTML" -} - -# Main script -main() { - # Check arguments - if [ $# -lt 2 ]; then - print_error "Insufficient arguments provided." - show_usage - exit 1 - fi - - local index_name="$1" - shift - local documents=("$@") - - # Check prerequisites - check_prerequisites - - # Validate documents - local valid_documents - valid_documents=($(validate_documents "${documents[@]}")) - - if [ ${#valid_documents[@]} -eq 0 ]; then - print_error "No valid documents to process." - exit 1 - fi - - # Create the index - print_status "Starting index creation process..." - create_index "$index_name" "${valid_documents[@]}" - - print_success "Index creation completed!" - print_status "You can now use the index in the LocalGPT interface." -} - -# Run main function with all arguments -main "$@" \ No newline at end of file diff --git a/src/components/IndexForm.tsx b/src/components/IndexForm.tsx index f0e0d9af..0f88faa1 100644 --- a/src/components/IndexForm.tsx +++ b/src/components/IndexForm.tsx @@ -4,7 +4,7 @@ import { GlassInput } from '@/components/ui/GlassInput'; import { GlassToggle } from '@/components/ui/GlassToggle'; import { AccordionGroup } from '@/components/ui/AccordionGroup'; import { ModelSelect } from '@/components/ModelSelect'; -import { chatAPI, ChatSession } from '@/lib/api'; +import { chatAPI, ChatSession, DEFAULT_ENRICHMENT_MODEL } from '@/lib/api'; import { InfoTooltip } from '@/components/ui/InfoTooltip'; interface Props { @@ -16,16 +16,14 @@ export function IndexForm({ onClose, onIndexed }: Props) { const [files, setFiles] = useState(null); const [indexName, setIndexName] = useState(''); const [chunkSize, setChunkSize] = useState(512); - const [chunkOverlap, setChunkOverlap] = useState(64); const [windowSize, setWindowSize] = useState(5); const [enableEnrich, setEnableEnrich] = useState(true); - const [retrievalMode, setRetrievalMode] = useState<'hybrid' | 'vector' | 'fts'>('hybrid'); + const [retrievalMode, setRetrievalMode] = useState<'hybrid' | 'vector_only' | 'fts_only'>('hybrid'); const [embeddingModel, setEmbeddingModel] = useState(); - const DEFAULT_LLM = 'qwen3:0.6b'; - const [enrichModel, setEnrichModel] = useState(DEFAULT_LLM); - const [overviewModel, setOverviewModel] = useState(DEFAULT_LLM); - const [batchSizeEmbed, setBatchSizeEmbed] = useState(64); - const [batchSizeEnrich, setBatchSizeEnrich] = useState(64); + const [enrichModel, setEnrichModel] = useState(DEFAULT_ENRICHMENT_MODEL); + const [overviewModel, setOverviewModel] = useState(DEFAULT_ENRICHMENT_MODEL); + const [batchSizeEmbed, setBatchSizeEmbed] = useState(50); + const [batchSizeEnrich, setBatchSizeEnrich] = useState(25); const [loading, setLoading] = useState(false); const [enableLateChunk, setEnableLateChunk] = useState(false); const [enableDoclingChunk, setEnableDoclingChunk] = useState(true); @@ -45,8 +43,7 @@ export function IndexForm({ onClose, onIndexed }: Props) { latechunk: enableLateChunk, doclingChunk: enableDoclingChunk, chunkSize: chunkSize, - chunkOverlap: chunkOverlap, - retrievalMode: retrievalMode==='fts' ? 'bm25' : retrievalMode, + retrievalMode: retrievalMode, windowSize: windowSize, enableEnrich: enableEnrich, embeddingModel: embeddingModel, @@ -108,8 +105,8 @@ export function IndexForm({ onClose, onIndexed }: Props) {
- {(['hybrid','vector','fts'] as const).map((m)=>( - + {([['hybrid','hybrid'],['vector_only','vector'],['fts_only','FTS']] as const).map(([m,label])=>( + ))}
@@ -127,14 +124,6 @@ export function IndexForm({ onClose, onIndexed }: Props) { setChunkSize(parseInt(e.target.value))} />
-
- - setChunkOverlap(parseInt(e.target.value))} - /> -
{/* Embedding & Overview models */} diff --git a/src/components/IndexWizard.tsx b/src/components/IndexWizard.tsx deleted file mode 100644 index 8e14a9dd..00000000 --- a/src/components/IndexWizard.tsx +++ /dev/null @@ -1,72 +0,0 @@ -"use client"; -import { useState } from 'react'; -import { ModelSelect } from '@/components/ModelSelect'; - -interface Props { - onClose: () => void; -} - -export function IndexWizard({ onClose }: Props) { - const [files, setFiles] = useState(null); - const [chunkSize, setChunkSize] = useState(512); - const [chunkOverlap, setChunkOverlap] = useState(64); - const [embeddingModel, setEmbeddingModel] = useState(); - // TODO: more params - - const handleFile = (e: React.ChangeEvent) => { - setFiles(e.target.files); - }; - - return ( -
-
-

Create new index

- -
-
- - -
- -
-
- - setChunkSize(parseInt(e.target.value))} - className="w-full bg-gray-800 rounded px-2 py-1" - /> -
-
- - setChunkOverlap(parseInt(e.target.value))} - className="w-full bg-gray-800 rounded px-2 py-1" - /> -
-
- -
- - -
-
- -
- - -
-
-
- ); -} \ No newline at end of file diff --git a/src/components/ModelSelect.tsx b/src/components/ModelSelect.tsx index 28e4f410..b90cdbac 100644 --- a/src/components/ModelSelect.tsx +++ b/src/components/ModelSelect.tsx @@ -1,5 +1,5 @@ import { useEffect, useState } from 'react'; -import { chatAPI, ModelsResponse } from '@/lib/api'; +import { chatAPI, ModelsResponse, DEFAULT_GENERATION_MODEL, DEFAULT_EMBEDDING_MODEL } from '@/lib/api'; interface Props { value: string | undefined; @@ -22,9 +22,10 @@ export function ModelSelect({ value, onChange, type, className, placeholder }: P if (!mounted) return; const list = type === 'generation' ? res.generation_models : res.embedding_models; setModels(list); - // Auto-select default qwen3:0.6b if available and not chosen yet - if(!value && list.includes('qwen3:0.6b')){ - onChange('qwen3:0.6b'); + // Auto-select the role default when available and nothing is chosen yet + const preferred = type === 'generation' ? DEFAULT_GENERATION_MODEL : DEFAULT_EMBEDDING_MODEL; + if(!value && list.includes(preferred)){ + onChange(preferred); } setLoading(false); }) diff --git a/src/components/SessionIndexInfo.tsx b/src/components/SessionIndexInfo.tsx index 3c9098dc..82a4f0e7 100644 --- a/src/components/SessionIndexInfo.tsx +++ b/src/components/SessionIndexInfo.tsx @@ -77,8 +77,7 @@ export default function SessionIndexInfo({ sessionId, onClose }: Props) { if (indexStatus === 'functional') { // Check if we have complete configuration metadata - const hasCompleteConfig = indexMeta.chunk_size && - indexMeta.chunk_overlap !== undefined && + const hasCompleteConfig = indexMeta.chunk_size && indexMeta.retrieval_mode && indexMeta.embedding_model; @@ -185,12 +184,6 @@ export default function SessionIndexInfo({ sessionId, onClose }: Props) {

)} - {typeof indexMeta.chunk_overlap==='number' && ( -
- Chunk overlap -

{indexMeta.chunk_overlap} tokens

-
- )} {/* Context and Features */} diff --git a/src/components/demo.tsx b/src/components/demo.tsx index 970bf180..707a60f4 100644 --- a/src/components/demo.tsx +++ b/src/components/demo.tsx @@ -1,29 +1,23 @@ "use client"; import { useState, useEffect } from "react" -import { LocalGPTChat } from "@/components/ui/localgpt-chat" import { SessionSidebar } from "@/components/ui/session-sidebar" import { SessionChat } from '@/components/ui/session-chat' import { chatAPI, ChatSession } from "@/lib/api" import { LandingMenu } from "@/components/LandingMenu"; import { IndexForm } from "@/components/IndexForm"; -import SessionIndexInfo from "@/components/SessionIndexInfo"; import IndexPicker from "@/components/IndexPicker"; import { QuickChat } from '@/components/ui/quick-chat' export function Demo() { const [currentSessionId, setCurrentSessionId] = useState() - const [currentSession, setCurrentSession] = useState(null) const [showConversation, setShowConversation] = useState(false) const [backendStatus, setBackendStatus] = useState<'checking' | 'connected' | 'error'>('checking') const [sidebarRef, setSidebarRef] = useState<{ refreshSessions: () => Promise } | null>(null) const [homeMode, setHomeMode] = useState<'HOME' | 'INDEX' | 'CHAT_EXISTING' | 'QUICK_CHAT'>('HOME') - const [showIndexInfo, setShowIndexInfo] = useState(false) const [showIndexPicker, setShowIndexPicker] = useState(false) const [sidebarOpen, setSidebarOpen] = useState(true) - console.log('Demo component rendering...') - useEffect(() => { console.log('Demo component mounted') checkBackendHealth() @@ -49,14 +43,11 @@ export function Demo() { const handleNewSession = () => { // Reset state and return to landing page so user can choose chat type setCurrentSessionId(undefined) - setCurrentSession(null) setShowConversation(false) // Hide conversation view & sidebar setHomeMode('HOME') // Show landing selector (Create index / Chat with index / LLM Chat) } const handleSessionChange = async (session: ChatSession) => { - setCurrentSession(session) - // Update the current session ID if it changed (e.g., brand-new session) if (session.id !== currentSessionId) { setCurrentSessionId(session.id) @@ -72,16 +63,6 @@ export function Demo() { if (currentSessionId === deletedSessionId) { // Stay in conversation mode but show empty state setCurrentSessionId(undefined) - setCurrentSession(null) - } - } - - const handleStartConversation = () => { - if (backendStatus === 'connected') { - // Just show empty state, don't create session yet - handleNewSession() - } else { - setShowConversation(true) } } @@ -163,15 +144,13 @@ export function Demo() { {homeMode==='INDEX' && ( -
- setHomeMode('HOME')} onIndexed={(s)=>{setHomeMode('CHAT_EXISTING'); handleSessionSelect(s.id);}} /> +
+
+ setHomeMode('HOME')} onIndexed={(s)=>{setHomeMode('CHAT_EXISTING'); handleSessionSelect(s.id);}} /> +
)} - {showIndexInfo && currentSessionId && ( - setShowIndexInfo(false)} /> - )} - {showIndexPicker && ( setShowIndexPicker(false)} onSelect={async (idxId)=>{ // create session and link index then open chat diff --git a/src/components/ui/GlassSelect.tsx b/src/components/ui/GlassSelect.tsx deleted file mode 100644 index 940975ad..00000000 --- a/src/components/ui/GlassSelect.tsx +++ /dev/null @@ -1,13 +0,0 @@ -"use client"; -import React, { SelectHTMLAttributes } from 'react'; - -export function GlassSelect(props: SelectHTMLAttributes) { - return ( - - ); -} \ No newline at end of file diff --git a/src/components/ui/badge.tsx b/src/components/ui/badge.tsx deleted file mode 100644 index 02054139..00000000 --- a/src/components/ui/badge.tsx +++ /dev/null @@ -1,46 +0,0 @@ -import * as React from "react" -import { Slot } from "@radix-ui/react-slot" -import { cva, type VariantProps } from "class-variance-authority" - -import { cn } from "@/lib/utils" - -const badgeVariants = cva( - "inline-flex items-center justify-center rounded-md border px-2 py-0.5 text-xs font-medium w-fit whitespace-nowrap shrink-0 [&>svg]:size-3 gap-1 [&>svg]:pointer-events-none focus-visible:border-ring focus-visible:ring-ring/50 focus-visible:ring-[3px] aria-invalid:ring-destructive/20 dark:aria-invalid:ring-destructive/40 aria-invalid:border-destructive transition-[color,box-shadow] overflow-hidden", - { - variants: { - variant: { - default: - "border-transparent bg-primary text-primary-foreground [a&]:hover:bg-primary/90", - secondary: - "border-transparent bg-secondary text-secondary-foreground [a&]:hover:bg-secondary/90", - destructive: - "border-transparent bg-destructive text-white [a&]:hover:bg-destructive/90 focus-visible:ring-destructive/20 dark:focus-visible:ring-destructive/40 dark:bg-destructive/60", - outline: - "text-foreground [a&]:hover:bg-accent [a&]:hover:text-accent-foreground", - }, - }, - defaultVariants: { - variant: "default", - }, - } -) - -function Badge({ - className, - variant, - asChild = false, - ...props -}: React.ComponentProps<"span"> & - VariantProps & { asChild?: boolean }) { - const Comp = asChild ? Slot : "span" - - return ( - - ) -} - -export { Badge, badgeVariants } diff --git a/src/components/ui/chat-bubble-demo.tsx b/src/components/ui/chat-bubble-demo.tsx deleted file mode 100644 index 032b63f8..00000000 --- a/src/components/ui/chat-bubble-demo.tsx +++ /dev/null @@ -1,103 +0,0 @@ -"use client" - -import { - ChatBubble, - ChatBubbleAvatar, - ChatBubbleMessage -} from "@/components/ui/chat-bubble" -import { Copy, RefreshCcw } from "lucide-react" - -const messages = [ - { - id: 1, - message: "Help me with my essay.", - sender: "user", - }, - { - id: 2, - message: "I can help you with that. What do you need help with?", - sender: "bot", - }, -] - -const actionIcons = [ - { icon: Copy, type: "Copy" }, - { icon: RefreshCcw, type: "Regenerate" }, -] - -export function ChatBubbleVariants() { - return ( -
- - - - I have a question about the library. - - - - - - - Sure, I'd be happy to help! - - -
- ) -} - -export function ChatBubbleAiLayout() { - return ( -
- {messages.map((message, index) => { - const variant = message.sender === "user" ? "sent" : "received" - return ( -
-
- -
- {message.message} - {message.sender === "bot" && ( -
- {actionIcons.map(({ icon: Icon, type }) => ( - - ))} -
- )} -
-
-
- ) - })} -
- ) -} - -export function ChatBubbleStates() { - return ( -
- - - - - - - - - Error processing request - - -
- ) -} \ No newline at end of file diff --git a/src/components/ui/chat-bubble.tsx b/src/components/ui/chat-bubble.tsx index 87718f4a..659088f2 100644 --- a/src/components/ui/chat-bubble.tsx +++ b/src/components/ui/chat-bubble.tsx @@ -1,68 +1,7 @@ "use client" -import * as React from "react" import { cn } from "@/lib/utils" import { Avatar, AvatarFallback, AvatarImage } from "@/components/ui/avatar" -import { Button } from "@/components/ui/button" -import { MessageLoading } from "@/components/ui/message-loading"; - -interface ChatBubbleProps { - variant?: "sent" | "received" - layout?: "default" | "ai" - className?: string - children: React.ReactNode -} - -export function ChatBubble({ - variant = "received", - layout = "default", // eslint-disable-line @typescript-eslint/no-unused-vars - className, - children, -}: ChatBubbleProps) { - return ( -
- {children} -
- ) -} - -interface ChatBubbleMessageProps { - variant?: "sent" | "received" - isLoading?: boolean - className?: string - children?: React.ReactNode -} - -export function ChatBubbleMessage({ - variant = "received", - isLoading, - className, - children, -}: ChatBubbleMessageProps) { - return ( -
- {isLoading ? ( -
- -
- ) : ( - children - )} -
- ) -} interface ChatBubbleAvatarProps { src?: string @@ -82,40 +21,3 @@ export function ChatBubbleAvatar({ ) } - -interface ChatBubbleActionProps { - icon?: React.ReactNode - onClick?: () => void - className?: string -} - -export function ChatBubbleAction({ - icon, - onClick, - className, -}: ChatBubbleActionProps) { - return ( - - ) -} - -export function ChatBubbleActionWrapper({ - className, - children, -}: { - className?: string - children: React.ReactNode -}) { - return ( -
- {children} -
- ) -} \ No newline at end of file diff --git a/src/components/ui/chat-input.tsx b/src/components/ui/chat-input.tsx index a6ff6770..1f9b74e0 100644 --- a/src/components/ui/chat-input.tsx +++ b/src/components/ui/chat-input.tsx @@ -2,7 +2,7 @@ import * as React from "react" import { useState, useRef } from "react" -import { ArrowUp, Settings as SettingsIcon, Plus, X, FileText } from "lucide-react" +import { ArrowUp, Settings as SettingsIcon, X, FileText } from "lucide-react" import { Button } from "@/components/ui/button" import { AttachedFile } from "@/lib/types" diff --git a/src/components/ui/chat-settings-modal.tsx b/src/components/ui/chat-settings-modal.tsx index 3e9d9f69..8965d263 100644 --- a/src/components/ui/chat-settings-modal.tsx +++ b/src/components/ui/chat-settings-modal.tsx @@ -43,7 +43,7 @@ const optionHelp: Record = { 'RAG (no-triage)':'Force retrieval on every query; disables index-selection triage.', 'Verify answer':'Runs an extra LLM pass to self-critique the draft answer.', 'Streaming':'Send tokens to the UI as they are generated.', - 'AI reranker':'Re-orders retrieved chunks with a cross-encoder (higher quality, more latency).', + 'AI reranker':'Off by default. Re-orders retrieved chunks with Qwen3-Reranker-4B: measurably better ranking, but it loads 7.5GB of weights and added ~12.7s per query in our eval.', 'Expand context window':'Adds neighbour chunks around each top chunk to provide more context.', 'Context window size':'How many neighbour chunks to include on each side.', 'Retrieval chunks':'Number of chunks fetched before reranking.', diff --git a/src/components/ui/conversation-page.tsx b/src/components/ui/conversation-page.tsx index 191f1554..d6ab6eb8 100644 --- a/src/components/ui/conversation-page.tsx +++ b/src/components/ui/conversation-page.tsx @@ -8,7 +8,6 @@ import { import { Copy, RefreshCcw, ThumbsUp, ThumbsDown, Volume2, MoreHorizontal, ChevronDown, Loader2, CheckCircle, XOctagon } from "lucide-react" import { ScrollArea } from "@/components/ui/scroll-area" import { ChatMessage } from "@/lib/api" -import { cn } from "@/lib/utils" import Markdown from "@/components/Markdown" import { normalizeWhitespace } from "@/utils/textNormalization" @@ -111,7 +110,7 @@ function ThinkingText({ text }: { text: string }) { )} {visibleText.trim() && ( - + )} ); @@ -151,7 +150,7 @@ function StructuredMessageBlock({ content }: { content: Array -
+
{!hasSubAnswers && step.details.source_documents && step.details.source_documents.length > 0 && ( @@ -159,7 +158,7 @@ function StructuredMessageBlock({ content }: { content: Array ) : step.key === 'final' && step.details && typeof step.details === 'string' ? ( -
+
) : Array.isArray(step.details) ? ( @@ -309,7 +308,7 @@ export function ConversationPage({ /> )} -
+
) : ( -
+
{typeof message.content === 'string' ? : diff --git a/src/components/ui/dropdown-menu.tsx b/src/components/ui/dropdown-menu.tsx deleted file mode 100644 index ec51e9cc..00000000 --- a/src/components/ui/dropdown-menu.tsx +++ /dev/null @@ -1,257 +0,0 @@ -"use client" - -import * as React from "react" -import * as DropdownMenuPrimitive from "@radix-ui/react-dropdown-menu" -import { CheckIcon, ChevronRightIcon, CircleIcon } from "lucide-react" - -import { cn } from "@/lib/utils" - -function DropdownMenu({ - ...props -}: React.ComponentProps) { - return -} - -function DropdownMenuPortal({ - ...props -}: React.ComponentProps) { - return ( - - ) -} - -function DropdownMenuTrigger({ - ...props -}: React.ComponentProps) { - return ( - - ) -} - -function DropdownMenuContent({ - className, - sideOffset = 4, - ...props -}: React.ComponentProps) { - return ( - - - - ) -} - -function DropdownMenuGroup({ - ...props -}: React.ComponentProps) { - return ( - - ) -} - -function DropdownMenuItem({ - className, - inset, - variant = "default", - ...props -}: React.ComponentProps & { - inset?: boolean - variant?: "default" | "destructive" -}) { - return ( - - ) -} - -function DropdownMenuCheckboxItem({ - className, - children, - checked, - ...props -}: React.ComponentProps) { - return ( - - - - - - - {children} - - ) -} - -function DropdownMenuRadioGroup({ - ...props -}: React.ComponentProps) { - return ( - - ) -} - -function DropdownMenuRadioItem({ - className, - children, - ...props -}: React.ComponentProps) { - return ( - - - - - - - {children} - - ) -} - -function DropdownMenuLabel({ - className, - inset, - ...props -}: React.ComponentProps & { - inset?: boolean -}) { - return ( - - ) -} - -function DropdownMenuSeparator({ - className, - ...props -}: React.ComponentProps) { - return ( - - ) -} - -function DropdownMenuShortcut({ - className, - ...props -}: React.ComponentProps<"span">) { - return ( - - ) -} - -function DropdownMenuSub({ - ...props -}: React.ComponentProps) { - return -} - -function DropdownMenuSubTrigger({ - className, - inset, - children, - ...props -}: React.ComponentProps & { - inset?: boolean -}) { - return ( - - {children} - - - ) -} - -function DropdownMenuSubContent({ - className, - ...props -}: React.ComponentProps) { - return ( - - ) -} - -export { - DropdownMenu, - DropdownMenuPortal, - DropdownMenuTrigger, - DropdownMenuContent, - DropdownMenuGroup, - DropdownMenuLabel, - DropdownMenuItem, - DropdownMenuCheckboxItem, - DropdownMenuRadioGroup, - DropdownMenuRadioItem, - DropdownMenuSeparator, - DropdownMenuShortcut, - DropdownMenuSub, - DropdownMenuSubTrigger, - DropdownMenuSubContent, -} diff --git a/src/components/ui/empty-chat-state.tsx b/src/components/ui/empty-chat-state.tsx deleted file mode 100644 index 30462d96..00000000 --- a/src/components/ui/empty-chat-state.tsx +++ /dev/null @@ -1,292 +0,0 @@ -"use client"; - -import { useEffect, useRef, useCallback } from "react"; -import { useState } from "react"; -import { Textarea } from "@/components/ui/textarea"; -import { cn } from "@/lib/utils"; -import { - ArrowUpIcon, - Paperclip, - PlusIcon, - X, - FileText, -} from "lucide-react"; -import { AttachedFile } from "@/lib/types"; - -interface UseAutoResizeTextareaProps { - minHeight: number; - maxHeight?: number; -} - -function useAutoResizeTextarea({ - minHeight, - maxHeight, -}: UseAutoResizeTextareaProps) { - const textareaRef = useRef(null); - - const adjustHeight = useCallback( - (reset?: boolean) => { - const textarea = textareaRef.current; - if (!textarea) return; - - if (reset) { - textarea.style.height = `${minHeight}px`; - return; - } - - // Temporarily shrink to get the right scrollHeight - textarea.style.height = `${minHeight}px`; - - // Calculate new height - const newHeight = Math.max( - minHeight, - Math.min( - textarea.scrollHeight, - maxHeight ?? Number.POSITIVE_INFINITY - ) - ); - - textarea.style.height = `${newHeight}px`; - }, - [minHeight, maxHeight] - ); - - useEffect(() => { - // Set initial height - const textarea = textareaRef.current; - if (textarea) { - textarea.style.height = `${minHeight}px`; - } - }, [minHeight]); - - // Adjust height on window resize - useEffect(() => { - const handleResize = () => adjustHeight(); - window.addEventListener("resize", handleResize); - return () => window.removeEventListener("resize", handleResize); - }, [adjustHeight]); - - return { textareaRef, adjustHeight }; -} - -interface EmptyChatStateProps { - onSendMessage: (message: string, attachedFiles?: AttachedFile[]) => void; - disabled?: boolean; - placeholder?: string; -} - -export function EmptyChatState({ - onSendMessage, - disabled = false, - placeholder = "Ask localgpt a question..." -}: EmptyChatStateProps) { - const [value, setValue] = useState(""); - const [attachedFiles, setAttachedFiles] = useState([]); - const fileInputRef = useRef(null); - const { textareaRef, adjustHeight } = useAutoResizeTextarea({ - minHeight: 60, - maxHeight: 200, - }); - - const handleSend = () => { - if ((value.trim() || attachedFiles.length > 0) && !disabled) { - onSendMessage(value.trim(), attachedFiles); - setValue(""); - setAttachedFiles([]); - adjustHeight(true); - } - }; - - const handleKeyDown = (e: React.KeyboardEvent) => { - if (e.key === "Enter" && !e.shiftKey) { - e.preventDefault(); - handleSend(); - } - }; - - const handleFileAttach = () => { - fileInputRef.current?.click(); - }; - - const handleFileChange = (e: React.ChangeEvent) => { - const files = e.target.files; - if (!files) return; - - const newFiles: AttachedFile[] = []; - for (let i = 0; i < files.length; i++) { - const file = files[i]; - if (file.type === 'application/pdf' || - file.type === 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' || - file.type === 'application/msword' || - file.type === 'text/html' || - file.type === 'text/markdown' || - file.type === 'text/plain' || - file.name.toLowerCase().endsWith('.pdf') || - file.name.toLowerCase().endsWith('.docx') || - file.name.toLowerCase().endsWith('.doc') || - file.name.toLowerCase().endsWith('.html') || - file.name.toLowerCase().endsWith('.htm') || - file.name.toLowerCase().endsWith('.md') || - file.name.toLowerCase().endsWith('.txt')) { - newFiles.push({ - id: crypto.randomUUID(), - name: file.name, - size: file.size, - type: file.type, - file: file, - }); - } - } - - setAttachedFiles(prev => [...prev, ...newFiles]); - - // Reset the input - if (fileInputRef.current) { - fileInputRef.current.value = ''; - } - - // --- NEW: Immediately trigger upload when files are selected --- - if (newFiles.length > 0) { - onSendMessage("", newFiles); - // Clear the local attachment state as the parent now handles it - setAttachedFiles([]); - } - }; - - const removeFile = (fileId: string) => { - setAttachedFiles(prev => prev.filter(f => f.id !== fileId)); - }; - - const formatFileSize = (bytes: number) => { - if (bytes === 0) return '0 Bytes'; - const k = 1024; - const sizes = ['Bytes', 'KB', 'MB', 'GB']; - const i = Math.floor(Math.log(bytes) / Math.log(k)); - return parseFloat((bytes / Math.pow(k, i)).toFixed(2)) + ' ' + sizes[i]; - }; - - return ( -
-

- What can I help you find? -

- -
- {/* Attached Files Display */} - {attachedFiles.length > 0 && ( -
-
Attached Files:
-
- {attachedFiles.map((file) => ( -
- -
-
{file.name}
-
{formatFileSize(file.size)}
-
- {/* The remove button is commented out as the parent will manage the state now */} - {/* */} -
- ))} -
-
- )} - -
-
-