From a3c085f457dd1deed68c714c62a1a9aacc62ed36 Mon Sep 17 00:00:00 2001 From: kartik Date: Fri, 25 Sep 2026 17:42:41 +0530 Subject: [PATCH] release: prepare 0.5.0 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes the [Unreleased] section as [0.5.0] and cuts the matching docs snapshot. 0.5 earns a snapshot rather than markers because it changes documented behaviour rather than only adding to it: `insert-without-columns` now fires on every row-inserting keyword and its message changed with it, `sqlguard explain --allow-dml` admits `REPLACE` and `UPSERT` where it previously refused them as unrecognized statements, and a query on MySQL is analyzed once per execution rather than once per attempt, which halves nothing in the N+1 counter any more. `/docs/` has to keep describing 0.4 for anyone still on it, and versions.json is now 0.5, 0.4, 0.3, 0.2 — exactly the four maxLiveVersions builds. Only the changelog and the snapshot are here. Per CONTRIBUTING the root tag comes first and the satellites are pinned to it afterwards, so the eight go.mod files still require v0.4.0 and are updated once v0.5.0 exists. --- CHANGELOG.md | 5 +- .../versioned_docs/version-0.5/analyzer.md | 219 ++++++++++++++ website/versioned_docs/version-0.5/bun.md | 61 ++++ .../version-0.5/configuration.md | 219 ++++++++++++++ website/versioned_docs/version-0.5/ent.md | 71 +++++ website/versioned_docs/version-0.5/explain.md | 225 +++++++++++++++ .../version-0.5/getting-started.md | 148 ++++++++++ website/versioned_docs/version-0.5/gorm.md | 76 +++++ .../version-0.5/integrations.md | 64 +++++ website/versioned_docs/version-0.5/intro.md | 80 ++++++ .../versioned_docs/version-0.5/middleware.md | 165 +++++++++++ .../versioned_docs/version-0.5/n-plus-one.md | 107 +++++++ .../version-0.5/noise-control.md | 104 +++++++ website/versioned_docs/version-0.5/parsers.md | 120 ++++++++ website/versioned_docs/version-0.5/pgx.md | 106 +++++++ .../versioned_docs/version-0.5/redaction.md | 157 ++++++++++ website/versioned_docs/version-0.5/rules.md | 271 ++++++++++++++++++ website/versioned_docs/version-0.5/scan.md | 169 +++++++++++ website/versioned_docs/version-0.5/sqlx.md | 71 +++++ .../version-0.5/suppressions.md | 83 ++++++ website/versioned_docs/version-0.5/xorm.md | 56 ++++ .../version-0.5-sidebars.json | 59 ++++ website/versions.json | 1 + 23 files changed, 2636 insertions(+), 1 deletion(-) create mode 100644 website/versioned_docs/version-0.5/analyzer.md create mode 100644 website/versioned_docs/version-0.5/bun.md create mode 100644 website/versioned_docs/version-0.5/configuration.md create mode 100644 website/versioned_docs/version-0.5/ent.md create mode 100644 website/versioned_docs/version-0.5/explain.md create mode 100644 website/versioned_docs/version-0.5/getting-started.md create mode 100644 website/versioned_docs/version-0.5/gorm.md create mode 100644 website/versioned_docs/version-0.5/integrations.md create mode 100644 website/versioned_docs/version-0.5/intro.md create mode 100644 website/versioned_docs/version-0.5/middleware.md create mode 100644 website/versioned_docs/version-0.5/n-plus-one.md create mode 100644 website/versioned_docs/version-0.5/noise-control.md create mode 100644 website/versioned_docs/version-0.5/parsers.md create mode 100644 website/versioned_docs/version-0.5/pgx.md create mode 100644 website/versioned_docs/version-0.5/redaction.md create mode 100644 website/versioned_docs/version-0.5/rules.md create mode 100644 website/versioned_docs/version-0.5/scan.md create mode 100644 website/versioned_docs/version-0.5/sqlx.md create mode 100644 website/versioned_docs/version-0.5/suppressions.md create mode 100644 website/versioned_docs/version-0.5/xorm.md create mode 100644 website/versioned_sidebars/version-0.5-sidebars.json diff --git a/CHANGELOG.md b/CHANGELOG.md index 2901777..67c0e31 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,8 @@ the same version in lockstep. ## [Unreleased] +## [0.5.0] - 2026-09-25 + ### Fixed - **Every query is no longer analyzed twice when the base driver returns @@ -368,7 +370,8 @@ Initial public release. `integrations/sqlxguard`, `integrations/pgxguard` (native pgx / pgxpool), `integrations/bunguard`, `integrations/xormguard`, `integrations/entguard`. -[Unreleased]: https://github.com/KARTIKrocks/sqlguard/compare/v0.4.0...HEAD +[Unreleased]: https://github.com/KARTIKrocks/sqlguard/compare/v0.5.0...HEAD +[0.5.0]: https://github.com/KARTIKrocks/sqlguard/compare/v0.4.0...v0.5.0 [0.4.0]: https://github.com/KARTIKrocks/sqlguard/compare/v0.3.0...v0.4.0 [0.3.0]: https://github.com/KARTIKrocks/sqlguard/compare/v0.2.0...v0.3.0 [0.2.0]: https://github.com/KARTIKrocks/sqlguard/compare/v0.1.1...v0.2.0 diff --git a/website/versioned_docs/version-0.5/analyzer.md b/website/versioned_docs/version-0.5/analyzer.md new file mode 100644 index 0000000..7e6fa53 --- /dev/null +++ b/website/versioned_docs/version-0.5/analyzer.md @@ -0,0 +1,219 @@ +--- +id: analyzer +title: Analyzer API +description: Use the analyzer directly, pick a rule subset, write your own rules against the normalized Statement, register them by name, and write a custom reporter. +--- + +# Analyzer API + +Everything above the driver is ordinary Go you can call yourself. The +`analyzer` package has no dependencies outside the standard library. + +```go +import "github.com/KARTIKrocks/sqlguard/analyzer" +``` + +## Analyze a query + +```go +a := analyzer.Default() // every built-in rule at its default severity + +for _, r := range a.Analyze("DELETE FROM users") { + fmt.Printf("[%s] %s: %s\n", r.Severity, r.RuleName, r.Message) +} +// [CRITICAL] delete-without-where: DELETE without WHERE clause detected. This will delete all rows. +``` + +`Analyze(query string) []analyzer.Result` parses once, runs every rule, +applies in-SQL [suppressions](suppressions) and severity overrides, and +sets `Query` (redacted) and `Fingerprint` on each result. It never errors +or panics; SQL the parser cannot understand still gets a best-effort pass +through the fallback parser. + +### `Result` + +| Field | Meaning | +| --- | --- | +| `RuleName string` | Stable rule identifier, e.g. `select-star`. | +| `Severity analyzer.Severity` | `SeverityInfo`, `SeverityWarning`, `SeverityCritical`. `String()` gives `INFO` / `WARNING` / `CRITICAL`. | +| `Query string` | The offending SQL, [redacted](redaction) unless the analyzer was built `WithRawQuery()`. | +| `Fingerprint string` | Always set. PII-free, low-cardinality query identity. | +| `Message string` | What was detected. | +| `Suggestion string` | How to fix it. May be empty. | +| `File string`, `Line int` | Set only by the static scanner. | + +## Constructors + +| Constructor | Rules | Configurable by name? | +| --- | --- | --- | +| `analyzer.Default()` | every registered rule | yes | +| `analyzer.DefaultWithProfile(p analyzer.Profile)` | registered rules filtered/tuned by the profile | yes | +| `analyzer.New(rules ...analyzer.Rule)` | exactly the functions you pass | no — anonymous rules have no name to configure | + +And two copying modifiers: + +```go +a = a.WithParser(pgparser.New()) // swap the parser; nil resets to the fallback +a = a.WithRawQuery() // keep literals in Result.Query — local debugging only +``` + +### `Profile` + +`Profile` is the resolved, parser-independent view of configuration. The +`config` package builds one from `.sqlguard.yml`; you can build one by +hand: + +```go +p := analyzer.Profile{ + Disabled: map[string]bool{"orderby-without-limit": true}, + Severity: map[string]analyzer.Severity{"select-star": analyzer.SeverityInfo}, + Settings: map[string]analyzer.Settings{ + "large-offset": {"threshold": 5000}, + }, +} +a := analyzer.DefaultWithProfile(p) +``` + +`Only` (a whitelist) and `RawQuery` are the other fields. Everything in +the profile is resolved once at construction — the per-query path does no +configuration work, which is what keeps it cheap. + +## Writing a rule + +A rule is a function over the normalized statement: + +```go +type Rule func(s *analyzer.Statement) (analyzer.Result, bool) +``` + +It returns `(result, true)` to report, `(Result{}, false)` to stay quiet. +Rules read `Statement` fields; they never re-parse or pattern-match the raw +SQL — that is the parser's job, and it is what keeps rules correct across +the fallback and the real grammars. + +```go +func checkSelectForUpdateWithoutLimit(s *analyzer.Statement) (analyzer.Result, bool) { + if s.Kind == analyzer.StmtSelect && s.HasOrderBy && !s.HasLimit && !s.HasWhere { + return analyzer.Result{ + RuleName: "unbounded-sorted-select", + Message: "Sorted SELECT with no WHERE and no LIMIT sorts the whole table.", + Suggestion: "Add a WHERE filter or a LIMIT.", + }, true + } + return analyzer.Result{}, false +} +``` + +Leave `Severity`, `Query` and `Fingerprint` unset when the rule is +registered (below): the registry's default severity, profile overrides +and the redaction policy are applied centrally. Set `Severity` yourself +only for anonymous rules passed to `analyzer.New`. + +Treat a `false` boolean as "not detected", not "proven absent" — the +fallback parser leaves a field `false` when it genuinely cannot tell, and +a rule that assumes otherwise produces false positives. `Statement.Exact` +tells you whether a real grammar produced the structural fields. + +### Registering it by name + +```go +func init() { + analyzer.Register(analyzer.RuleSpec{ + Name: "unbounded-sorted-select", + DefaultSeverity: analyzer.SeverityWarning, + Factory: func(s analyzer.Settings) analyzer.Rule { + return checkSelectForUpdateWithoutLimit + }, + }) +} +``` + +Once registered, the rule is part of `analyzer.Default()` and is +addressable by name everywhere: `rules.disable` / `rules.severity` / +`rules.settings` in [config](configuration), `sqlguard:ignore:` in +[suppressions](suppressions), and `analyzer.RuleNames()`. Registering a +name that already exists **replaces** the built-in, so you can override +one. + +The `Factory` receives the rule's `Settings` map (from +`rules.settings.` in config). Nil-safe accessors — `Int`, `Bool`, +`String`, `Duration`, each with a default — let a rule take tunables +without touching the config schema: + +```go +Factory: func(s analyzer.Settings) analyzer.Rule { + max := s.Int("max-rows", 10000) + return func(st *analyzer.Statement) (analyzer.Result, bool) { /* use max */ } +}, +``` + +### Using a subset + +```go +a := analyzer.New( + analyzer.CheckDeleteWithoutWhere, + analyzer.CheckUpdateWithoutWhere, +) +``` + +The built-in rule functions (`CheckSelectStar`, `CheckLeadingWildcard`, +`CheckDeleteWithoutWhere`, `CheckUpdateWithoutWhere`, +`CheckInsertWithoutColumns`, `CheckSelectWithoutLimit`, +`CheckOrderByWithoutLimit`, `CheckNonSargablePredicate`, +`CheckAddNotNullWithoutDefault`, `CheckImplicitJoin`, +`CheckCartesianJoin`, `CheckInListTooLarge`, `CheckLargeOffset`, +`CheckSelectDistinct`) are exported. Rules passed to `New` are anonymous: +profile overrides do not apply, and the tunable ones run at their +defaults. Prefer `DefaultWithProfile` with `Only` when you want a named, +configurable subset. + +## Helpers + +| Function | Use | +| --- | --- | +| `analyzer.Redact(sql string) string` | Literals → `?`, comments stripped, structure kept. | +| `analyzer.Fingerprint(sql string) string` | `Redact` + whitespace collapse + list fold. | +| `analyzer.IsMultiStatement(sql string) bool` | Comment- and string-aware `;` check. | +| `analyzer.ParseIgnoreComment(text string) (all bool, rules map[string]bool, found bool)` | Parse a Go comment for a suppression directive. | +| `analyzer.RuleNames() []string` | Every registered rule name, sorted. | +| `analyzer.NewFallbackParser() *FallbackParser` | The zero-dependency parser, for delegation. | +| `(*Analyzer).PrepareQuery(raw string) (display, fingerprint string)` | Apply this analyzer's redaction policy to a query outside the rule path. | + +## Reporters + +```go +import "github.com/KARTIKrocks/sqlguard/reporter" + +type Reporter interface { + Report(results []analyzer.Result) +} +``` + +| Reporter | Output | +| --- | --- | +| `reporter.NewConsoleReporter()` / `NewConsoleReporterTo(w)` | Colored, human-readable blocks; stderr by default. | +| `reporter.NewJSONReporter()` / `NewJSONReporterTo(w)` | A JSON array; stderr by default. | + +`Report` must be safe for concurrent calls — the middleware invokes it +from whichever goroutine ran the query. The result slice handed to you may +be shared with the analysis cache; treat it as read-only. + +A reporter that forwards to `slog`: + +```go +type slogReporter struct{ l *slog.Logger } + +func (s *slogReporter) Report(rs []analyzer.Result) { + for _, r := range rs { + s.l.Warn("sqlguard finding", + "rule", r.RuleName, + "severity", r.Severity.String(), + "fingerprint", r.Fingerprint, + "query", r.Query, + "message", r.Message, + ) + } +} + +sqlguard.Register("sqlguard-pg", "pgx", middleware.WithReporter(&slogReporter{l: slog.Default()})) +``` diff --git a/website/versioned_docs/version-0.5/bun.md b/website/versioned_docs/version-0.5/bun.md new file mode 100644 index 0000000..8b5a7f7 --- /dev/null +++ b/website/versioned_docs/version-0.5/bun.md @@ -0,0 +1,61 @@ +--- +id: bun +title: bun +description: bunguard — a bun.QueryHook that analyzes every statement bun renders. +--- + +# bun + +`bunguard` is a `bun.QueryHook`. bun renders SQL through its own query +builder and exposes the final text and a start timestamp in `AfterQuery`, +which is where the hook runs the static rules and the latency check. + +```bash +go get github.com/KARTIKrocks/sqlguard/integrations/bunguard +``` + +```go +import ( + "github.com/KARTIKrocks/sqlguard/integrations/bunguard" + "github.com/KARTIKrocks/sqlguard/middleware" + "github.com/uptrace/bun" + "github.com/uptrace/bun/dialect/pgdialect" + "github.com/uptrace/bun/driver/pgdriver" +) + +sqldb := sql.OpenDB(pgdriver.NewConnector(pgdriver.WithDSN(dsn))) +db := bun.NewDB(sqldb, pgdialect.New()) + +db.AddQueryHook(bunguard.New( + middleware.WithSlowQueryThreshold(500*time.Millisecond), + middleware.WithN1Detection(10, time.Second), +)) +``` + +## API + +| Symbol | Use | +| --- | --- | +| `bunguard.New(opts ...middleware.Option) *QueryHook` | Build the hook. Pass to `db.AddQueryHook`. | +| `(*QueryHook).ResetN1()` | Clear N+1 state at a request boundary. | +| `BeforeQuery`, `AfterQuery` | The `bun.QueryHook` methods; you do not call them. | + +```go +hook := bunguard.New(middleware.WithN1Detection(10, time.Second)) +db.AddQueryHook(hook) + +func handler(w http.ResponseWriter, r *http.Request) { + defer hook.ResetN1() + // ... +} +``` + +## Notes + +- Static rules run on every query; latency is reported only when the + query succeeded (`event.Err == nil`). Same semantics as every other + surface. +- bun's `pgdriver` is not a `database/sql` driver name you can wrap with + `sqlguard.Register`, but `sqlguard.OpenDB(pgdriver.NewConnector(...))` + works if you prefer driver-level coverage over `ResetN1()`. Do not use + both on one connection. diff --git a/website/versioned_docs/version-0.5/configuration.md b/website/versioned_docs/version-0.5/configuration.md new file mode 100644 index 0000000..4f871f2 --- /dev/null +++ b/website/versioned_docs/version-0.5/configuration.md @@ -0,0 +1,219 @@ +--- +id: configuration +title: Configuration +description: The .sqlguard.yml file — every key, how it is discovered, how the CLI and the middleware load it, and the config package API. +--- + +# Configuration + +One file, `.sqlguard.yml`, configures every surface: which rules run, at +what severity, with which tunables, plus the runtime thresholds and the +scanner's exclusions. It is optional — with no file, every rule runs at its +default. + +## Discovery + +sqlguard looks for `.sqlguard.yml` (or `.sqlguard.yaml`) starting in the +scanned or working directory and walking **up** until it finds one or +reaches the git root (a directory containing `.git`). Put it at the +repository root and every sub-package inherits it. + +The CLI accepts `--config ` to load a specific file and `--no-config` +to ignore any file and use built-in defaults. Both are persistent flags, so +they work with `scan` and `explain` alike. + +## Full reference + +```yaml +version: 1 + +# Turn soft problems (unknown keys, unknown rule names, bad severities) +# into hard errors. Leave false so a config written for a newer sqlguard +# still loads on an older binary. +strict: false + +rules: + # Turn rules off entirely. + disable: + - orderby-without-limit + + # Whitelist mode: when non-empty, a rule must be listed to run. It narrows + # the rules evaluated against a statement — the scanner and the runtime + # statement rules — and does NOT reach slow-query, n-plus-one or the EXPLAIN + # plan rules; switch one of those off by naming it in `disable` above. + # `disable` still applies to the rules listed here, so listing and disabling + # the same rule disables it. + # only: + # - delete-without-where + # - update-without-where + + # Override the reported severity per rule: info | warning | critical | off. + # "off" is equivalent to disabling the rule. + severity: + select-star: info + select-without-limit: "off" + + # Per-rule tunables. Keys are rule-specific; see the Rules reference. + settings: + leading-wildcard: + min-length: 3 # ignore LIKE '%x%' with a searchable term shorter than this + in-list-too-large: + max-length: 100 # flag IN (...) with more elements than this + large-offset: + threshold: 1000 # flag a literal OFFSET above this + slow-query: + threshold: 200ms # runtime: flag a query at or above this latency + n-plus-one: + threshold: 10 # runtime: this many of the same fingerprint... + window: 1m # ...within this window + +# Redact literal values out of Result.Query. ON by default. Set false ONLY +# for local debugging where the query text is trusted. +redact: true + +# Runtime middleware: report each (rule, fingerprint) at most once per +# window. "0" disables and reports every occurrence. +dedup: + window: 1m + +# Static scanner only: skip files whose path matches any of these regexes. +scan: + exclude-paths: + - "(^|/)legacy/" + - "_gen\\.go$" +``` + +| Key | Applies to | Notes | +| --- | --- | --- | +| `version` | all | Reserved for forward compatibility; always `1` today. | +| `strict` | all | Make unknown keys, unknown rule names and bad severities fatal instead of warnings. | +| `rules.disable` | every rule | Rule names to turn off. | +| `rules.only` | the scanner and the statement rules at runtime | Whitelist over the rules evaluated against a statement. `disable` still applies to the ones listed, so listing and disabling the same rule disables it. It does **not** reach `slow-query`, `n-plus-one` or the plan rules — see below. | +| `rules.severity` | every rule | `info`, `warning`, `critical`, or `off`. | +| `rules.settings` | rules with tunables | `leading-wildcard.min-length`, `in-list-too-large.max-length`, `large-offset.threshold`, `slow-query.threshold`, `n-plus-one.threshold` / `.window`. See [Rules](rules). | +| `redact` | all | `false` keeps raw literals in `Result.Query`. See [Redaction](redaction). | +| `dedup.window` | middleware, integrations | Go duration or `"0"`. Equivalent to `WithFindingDedup`. | +| `scan.exclude-paths` | scanner | Regexes matched against the scanned file path. | + +Quote `"off"` — unquoted `off` is a YAML boolean. + +_Changed in 0.3._ "Every rule" now means every rule. In 0.2 only the 14 +statement rules were addressable: naming `slow-query`, `n-plus-one` or a plan +rule (`seq-scan`, `high-cost`, `full-table-scan`, `no-index-used`, `filesort`) +warned with `unknown rule`, and failed outright under `strict: true`, even +though the [rules reference](rules) listed them. The slow-query threshold also +moved from a top-level `slow-query.threshold` key to +`rules.settings.slow-query.threshold`, so every tunable lives in one place. + +## Lenient by default + +Unknown top-level keys and unknown rule names are **warnings**, printed to +stderr by the CLI as `sqlguard: config warning: …`, so a config that names +a rule added in a newer release still loads on an older binary. Set +`strict: true` when you want CI to fail on a typo. + +_Added in 0.3._ N+1 detection can be turned on from the file. Setting both +`rules.settings.n-plus-one.threshold` and `.window` enables it; previously it +was reachable only from Go with `WithN1Detection`, which remains the way to +set it in code. An explicit Go option wins over the file for the N+1 and slow-query +_thresholds_. Turning either rule off goes the other way — see +[Precedence](#precedence). + +## Loading it from Go + +The `config` package is the only place YAML is parsed. `analyzer` and +`middleware` never import it, so users who configure in code pay for no +YAML dependency. + +```go +import "github.com/KARTIKrocks/sqlguard/config" +``` + +| Function | Returns | +| --- | --- | +| `config.Middleware(path, startDir string) ([]middleware.Option, error)` | Ready-to-use options. `path == ""` discovers from `startDir`. A missing file is not an error — it yields the defaults. | +| `config.Load(path string) (*Config, error)` | Parse one file. | +| `config.Discover(startDir string) (*Config, string, error)` | Walk up from `startDir`; returns the config and the path it came from (`""` if none). | +| `config.Default() *Config` | The empty config: every rule at its defaults. | + +And on a `*Config`: + +| Method | Use | +| --- | --- | +| `MiddlewareOptions() ([]middleware.Option, error)` | `WithAnalyzer` from the profile — which carries the rule settings, including the slow-query and N+1 tunables — plus `WithFindingDedup` when set. Append your own options after it. | +| `Analyzer() (*analyzer.Analyzer, error)` | `analyzer.DefaultWithProfile` built from this file. | +| `Profile() (analyzer.Profile, error)` | The resolved, parser-independent profile. | +| `DedupWindow()` | `(time.Duration, ok bool, error)` — `ok` is false when the key is unset. _Changed in 0.3._ `SlowQueryThreshold()` is gone; take the profile first and read the setting off it (see below). | +| `ExcludeMatcher() (func(path string) bool, error)` | The compiled `scan.exclude-paths` predicate. | +| `Warnings() []string` | Non-fatal problems found while loading. Surface them. | + +The slow-query threshold now lives in the profile with every other tunable: + +```go +p, err := cfg.Profile() +if err != nil { + return err +} +d := p.Settings["slow-query"].Duration("threshold", 200*time.Millisecond) +``` + +The common case is one line: + +```go +opts, err := config.Middleware("", ".") +if err != nil { + log.Fatal(err) +} +opts = append(opts, middleware.WithParser(pgparser.New())) +sqlguard.Register("sqlguard-pg", "pgx", opts...) +``` + +## Precedence + +_Changed in 0.3._ For the **thresholds**, an explicit Go option wins over the +file wherever it appears in the list — `WithSlowQueryThreshold` and +`WithN1Detection` record that they were called, so a config value no longer +has to be ordered around. In 0.2 this depended on option order, because the +file's threshold arrived as an option of its own. + +```go +opts, _ := cfg.MiddlewareOptions() +opts = append(opts, middleware.WithSlowQueryThreshold(time.Second)) +// 1s, whatever rules.settings.slow-query.threshold says +``` + +**Turning a rule off is the other way round: the file wins.** `disable: +[slow-query]` silences the finding even with `WithSlowQueryThreshold` set, and +`disable: [n-plus-one]` stops the tracker being built at all despite +`WithN1Detection`. That is deliberate — `disable` is an instruction, not a +tuning value, and an operator editing `.sqlguard.yml` should be able to +silence a noisy rule without a redeploy. + +`only:` narrows; it does not override. A rule has to survive both checks, so +`only: [select-star]` together with `disable: [select-star]` leaves nothing. + +## What `only:` reaches + +`only:` selects which rules run **against a statement** — the 14 in the +scanner and at runtime. It does not reach the seven findings that are not +derived from statement text: `slow-query` and `n-plus-one`, which the +middleware computes from latency and repetition, and the five plan rules +[`sqlguard explain`](explain) reads from the database's own plan. + +That is because a whitelist is nearly always written to focus a scan, and it +names statement rules. If it reached the rest, `only: [select-star]` in a +repository's config would also switch off latency and N+1 reporting in the +running application, and make `sqlguard explain` report nothing — none of +which it mentions, and none of which would produce a warning. + +To switch one of those off, name it: `disable:` and `severity: off` reach +every surface. + +```yaml +rules: + only: [select-star] # scanner + runtime statement rules + disable: [slow-query] # and this reaches the middleware too +``` + +Inline [suppressions](suppressions) win over both: they silence a finding at +one site regardless of config. diff --git a/website/versioned_docs/version-0.5/ent.md b/website/versioned_docs/version-0.5/ent.md new file mode 100644 index 0000000..bdf9ca6 --- /dev/null +++ b/website/versioned_docs/version-0.5/ent.md @@ -0,0 +1,71 @@ +--- +id: ent +title: ent +description: entguard — decorate ent's dialect.Driver so every Exec, Query and transaction is analyzed, whatever opened the underlying *sql.DB. +--- + +# ent + +ent runs on `database/sql`, so the simplest coverage is to point `entsql` +at a `*sql.DB` from `sqlguard.Register` / `OpenDB`. `entguard` is the +dedicated alternative: it decorates ent's own `dialect.Driver`, the same +seam ent's built-in `dialect.Debug` wrapper uses, so it works regardless of +how the `*sql.DB` was opened and gives you a handle for `ResetN1()`. + +```bash +go get github.com/KARTIKrocks/sqlguard/integrations/entguard +``` + +```go +import ( + "entgo.io/ent/dialect" + entsql "entgo.io/ent/dialect/sql" + + "github.com/KARTIKrocks/sqlguard/integrations/entguard" + "github.com/KARTIKrocks/sqlguard/middleware" + + "myapp/ent" +) + +drv, err := entsql.Open(dialect.Postgres, dsn) +guarded := entguard.Wrap(drv, + middleware.WithSlowQueryThreshold(500*time.Millisecond), + middleware.WithN1Detection(10, time.Second), +) +client := ent.NewClient(ent.Driver(guarded)) +``` + +## API + +| Symbol | Use | +| --- | --- | +| `entguard.Wrap(d dialect.Driver, opts ...middleware.Option) *Driver` | Decorate any `dialect.Driver` — including one already wrapped by `dialect.Debug`. | +| `(*Driver).ResetN1()` | Clear N+1 state at a request boundary. | +| `Exec`, `Query`, `Tx`, `BeginTx` | The `dialect.Driver` methods; ent calls them. | + +```go +guarded := entguard.Wrap(drv, middleware.WithN1Detection(10, time.Second)) + +func handler(w http.ResponseWriter, r *http.Request) { + defer guarded.ResetN1() + // ... +} +``` + +## What is covered + +- `Exec` and `Query` on the driver. +- `Tx` and `BeginTx`, and every `Exec` / `Query` on the transactions they + return. Statements inside a transaction are analyzed exactly like the + ones outside it. + +Every call flows through `middleware.Guard.Observe`: static rules run on +each statement, latency is reported only when it succeeded. + +## Notes + +- ent's generated client issues predictable, parameterized SQL, which is + the case the [analysis cache](noise-control) was built for — after the + first call each query shape is a cache hit. +- Do not combine `entguard` with a driver-level `sqlguard.Register` on the + same `*sql.DB`; each statement would be analyzed twice. diff --git a/website/versioned_docs/version-0.5/explain.md b/website/versioned_docs/version-0.5/explain.md new file mode 100644 index 0000000..1aaea5a --- /dev/null +++ b/website/versioned_docs/version-0.5/explain.md @@ -0,0 +1,225 @@ +--- +id: explain +title: EXPLAIN Analyzer +description: sqlguard explain — plan a query against live PostgreSQL, MySQL or MariaDB, flag sequential scans and missing indexes, and never execute it. +--- + +# EXPLAIN Analyzer + +Static rules can tell you a query _looks_ risky. Only the database can tell +you it _is_: that the `WHERE` you wrote hits no index, that the planner +chose a sequential scan over ten million rows, that the `ORDER BY` needs a +filesort. `sqlguard explain` asks the planner and reports what it finds — +without ever executing the query. + +## Usage + +```bash +# PostgreSQL +sqlguard explain --db "postgres://app:secret@localhost/app?sslmode=disable" \ + "SELECT id, total FROM orders WHERE customer_id = 42" + +# MySQL / MariaDB +sqlguard explain --dialect mysql --db "app:secret@tcp(localhost:3306)/app" \ + "SELECT id, total FROM orders WHERE customer_id = 42" + +# JSON for tooling +sqlguard explain --db "…" --format json "SELECT …" +``` + +| Flag | Default | Effect | +| --- | --- | --- | +| `--db ` | required | Connection string. Postgres DSNs use pgx's `postgres://` URL or key=value form; MySQL uses `go-sql-driver/mysql` DSN syntax. | +| `--dialect postgres\|mysql` | `postgres` | Which planner to talk to. MariaDB works through `mysql`. | +| `--format console\|json` | `console` | Output shape. | +| `--allow-dml` | off | Permit `INSERT` / `UPDATE` / `DELETE` (incl. `REPLACE` / `UPSERT` _0.5+_). Still planned only, still rolled back. | +| `--config`, `--no-config` | — | Persistent flags. `rules:` config applies — see below. | + +The whole command runs under a 30-second timeout, including the initial +connectivity check. Exit code is **1** when the plan has issues, **0** +when clean. + +As with [`scan`](scan#usage), the console rendering goes to stderr and +`--format json` goes to stdout, always as an array. _Changed in 0.3._ In 0.2 +JSON went to stderr, so redirecting it produced an empty file. + +```text +[SQLGUARD WARNING] seq-scan + Query: SELECT id, total FROM orders WHERE customer_id = 42 + Issue: Sequential scan detected (estimated 812430 rows, cost 21877.5) + Fix: Consider adding an index to avoid full table scan. + +[SQLGUARD WARNING] high-cost + Query: SELECT id, total FROM orders WHERE customer_id = 42 + Issue: High cost operation: Seq Scan (cost 21877.5) + Fix: Review query plan and consider optimization. + +2 issue(s) found in query plan +``` + +Unlike every other surface, `Query` here is the raw text you typed — +there is no log sink to protect, and you need to recognise your own query. +`Fingerprint` is still set. See [Redaction](redaction#the-one-deliberate-exception). + +## What it detects + +| Rule | Dialect | Fires on | +| --- | --- | --- | +| `seq-scan` | postgres | A `Seq Scan` node. `INFO` at ≤ 1,000 estimated rows, `WARNING` above. | +| `high-cost` | postgres | Any node with `Total Cost` > 10,000. | +| `full-table-scan` | mysql | A row with access `type = ALL`. | +| `no-index-used` | mysql | A row with empty `key` **and** empty `possible_keys`. | +| `filesort` | mysql | `Using filesort` in `Extra`. | + +_Changed in 0.3._ These five are ordinary rule names now, so +[`.sqlguard.yml`](configuration) can turn one off or re-severity it: + +```yaml +rules: + disable: [high-cost] + severity: + seq-scan: critical +``` + +In 0.2 `explain` ignored `rules:` entirely, and naming a plan rule in a config +was an `unknown rule` warning — a hard error under `strict: true`. A +`severity` override also wins over `seq-scan`'s row-count-derived severity. + +`disable:` and `severity:` apply here; **`only:` does not**. A whitelist +selects which rules run over a statement, and a plan rule is not one — see +[what `only:` reaches](configuration#what-only-reaches). Switching a plan rule +off takes naming it. + +Postgres plans are requested as `EXPLAIN (FORMAT JSON)` and walked +recursively, so nested scans inside joins and CTEs are found. MySQL plans +are requested as `EXPLAIN FORMAT=TRADITIONAL` — MySQL 9 defaults +`@@explain_format` to `TREE`, which is a single free-text column — and read +**by column name**, because MariaDB emits 10 columns where MySQL emits 12. +`UNION RESULT` and derived-table rows (``, ``) are +skipped: they name temporary tables that have no index by construction. + +If the server returns a plan shape the analyzer does not recognise — a +missing expected column, say — it **fails with an error** rather than +reporting "no issues". A plan checker that says clean when it could not +look is worse than one that says so. + +## Safety model + +`EXPLAIN` cannot take bind parameters, so the query text is necessarily +concatenated into the `EXPLAIN` statement. The defense is layered and does +not rely on parameterization: + +1. **Validation.** Empty input is refused. Multi-statement input is refused + using a comment- and string-literal-aware check + (`analyzer.IsMultiStatement`), so a `;` hidden in a `--` comment, a + `/* */` block or a string cannot smuggle a second statement. That check + takes the _narrowest_ reading of a string literal — the opposite of + [`Redact`](redaction#when-the-dialect-is-ambiguous-it-over-redacts), which + takes the widest. The two have opposite fail-safe directions: redaction + must never leave a literal byte in its output, while this check must never + miss a separator, so `'a\'; DROP TABLE t; --'` is refused rather than read + as one literal. The statement is then classified with the fallback parser: + `SELECT` / `WITH` pass; `INSERT` / `UPDATE` / `DELETE` pass only with + `--allow-dml`; DDL, `SET`, transaction control and anything unrecognised + are always refused. + + _Changed in 0.5._ The row-inserting dialect keywords `REPLACE` and + `UPSERT` now classify as `INSERT` rather than as unrecognised, so + `--allow-dml` admits them instead of refusing them outright. They are + planned and rolled back like any other DML. On a server where the + keyword is not valid you now get that server's syntax error in place of + sqlguard's refusal. + + _Changed in 0.3._ The separator check previously scanned with a single + reading of `$$`, which let some stacked input through. It now counts a `;` + as a separator whenever _any_ dialect reading leaves it outside a literal, + because each reading of `$$` is blind to a different payload. Scanning + `$tag$…$tag$` as a literal is what hides the `;` in + `UPDATE t AS $$ SET id = 1; DROP TABLE t` — on MySQL those are identifier + bytes, but that reading opens an unterminated dollar body which swallows + the rest. Leaving `$$` as ordinary bytes is what hides the `;` in + `SELECT $$'$$; DROP TABLE t` — on PostgreSQL the apostrophe is data inside + the dollar body, but that reading opens an unterminated ordinary literal + which swallows the rest. Each catches what the other misses, so both run. + The cost is that a single statement carrying a `;` inside a dollar-quoted + body, such as `SELECT $$a ; b$$`, is refused; a one-statement query + occasionally rejected is the safe error here. +2. **A transaction that is always rolled back.** Every `EXPLAIN` runs + inside `BeginTx` with a deferred `Rollback`. Nothing commits. +3. **Never `ANALYZE`.** `EXPLAIN ANALYZE` executes the statement to collect + real timings; sqlguard never uses it. The statement is planned, only. + +The transaction is additionally opened **read-only** everywhere except +MySQL/MariaDB under `--allow-dml`. Those servers reject _every_ statement +in a `READ ONLY` transaction with error 1792 — including an `EXPLAIN` that +only plans a DML statement — so there the guarantee rests on the three +layers above. This is deliberate; it is not a gap. + +## Library use + +The CLI is a thin wrapper over the `explain` package, which works with any +`*sql.DB`: + +```go +import "github.com/KARTIKrocks/sqlguard/explain" + +pa, err := explain.New(db, "postgres") // or "mysql" +pa, err = explain.New(db, "mysql", explain.WithAllowDML()) + +res, err := pa.Analyze(ctx, "SELECT id FROM orders WHERE customer_id = 42") +for _, issue := range res.Issues { // []analyzer.Result + fmt.Println(issue.RuleName, issue.Message) +} +fmt.Println(res.RawPlan) // the plan text, for humans +``` + +_Added in 0.3._ `explain.WithAnalyzer` applies a rule profile, which is how the +CLI passes your `.sqlguard.yml` through. From code you can build one without a +file: + +```go +import ( + "github.com/KARTIKrocks/sqlguard/analyzer" + "github.com/KARTIKrocks/sqlguard/explain" +) + +a := analyzer.DefaultWithProfile(analyzer.Profile{ + Disabled: map[string]bool{"high-cost": true}, + Severity: map[string]analyzer.Severity{"seq-scan": analyzer.SeverityCritical}, +}) + +pa, err := explain.New(db, "postgres", explain.WithAnalyzer(a)) +if err != nil { + return err +} +``` + +Or from a loaded config, so the same file governs the scanner, the middleware +and this: + +```go +cfg, err := config.Load(".sqlguard.yml") +if err != nil { + return err +} +a, err := cfg.Analyzer() +if err != nil { + return err +} +pa, err := explain.New(db, "postgres", explain.WithAnalyzer(a)) +``` + +Without it, every plan rule fires at its built-in severity. + +`explain.Result` carries `Query`, `RawPlan` and `Issues`. Pair it with a +test that runs your hottest queries through `Analyze` against a seeded +database — a `seq-scan` on the orders table is cheaper to find in CI than +at 3 a.m. + +## Drivers + +_Changed in 0.3._ The CLI links `github.com/jackc/pgx/v5/stdlib` and +`github.com/go-sql-driver/mysql`, so `sqlguard explain` works out of the +box. Only the `cmd/sqlguard` package imports them; a library consumer of +`analyzer` or `middleware` never compiles them in. In 0.2 the released +binary shipped without a driver and `explain` could not connect. diff --git a/website/versioned_docs/version-0.5/getting-started.md b/website/versioned_docs/version-0.5/getting-started.md new file mode 100644 index 0000000..ae10779 --- /dev/null +++ b/website/versioned_docs/version-0.5/getting-started.md @@ -0,0 +1,148 @@ +--- +id: getting-started +title: Getting Started +description: Install sqlguard, wrap a database/sql driver, and see the first finding in under a minute. +--- + +# Getting Started + +## Installation + +Requires **Go 1.27+**. _Changed in 0.2._ Previously Go 1.26+. + +```bash +go get github.com/KARTIKrocks/sqlguard +``` + +The CLI (static scanner and EXPLAIN analyzer) is a separate binary: + +```bash +go install github.com/KARTIKrocks/sqlguard/cmd/sqlguard@latest +``` + +To pin a release: + +```bash +go get github.com/KARTIKrocks/sqlguard@v0.2.0 +``` + +## Runtime: wrap a driver + +sqlguard wraps at the `database/sql` **driver** layer. You register a new +driver name that delegates to an existing one, open it as usual, and get a +plain `*sql.DB` back — nothing else in your code changes. + +```go +package main + +import ( + "database/sql" + "log" + "time" + + _ "github.com/jackc/pgx/v5/stdlib" // registers the "pgx" driver + + "github.com/KARTIKrocks/sqlguard" + "github.com/KARTIKrocks/sqlguard/middleware" +) + +func main() { + // Wrap the "pgx" driver under a new name. + if err := sqlguard.Register("sqlguard-pg", "pgx", + middleware.WithSlowQueryThreshold(500*time.Millisecond), + middleware.WithN1Detection(5, 2*time.Second), + ); err != nil { + log.Fatal(err) + } + + db, err := sql.Open("sqlguard-pg", "postgres://app@localhost/app") + if err != nil { + log.Fatal(err) + } + defer db.Close() + + // Use db exactly as before. Every query is analyzed on its way through. + rows, err := db.Query("SELECT * FROM users WHERE email = 'a@b.c'") + // ... +} +``` + +The query above produces one finding on stderr: + +```text +[SQLGUARD WARNING] select-star + Query: SELECT * FROM users WHERE email = ? + Issue: SELECT * detected. Selecting all columns can hurt performance. + Fix: Select only the columns you need. +``` + +Note the literal `'a@b.c'` became `?` — findings are +[redacted by default](redaction) so customer data never lands in a log. + +If you already hold a `driver.Connector` (for example from pgx's +`stdlib.GetConnector`), skip the registry: + +```go +db := sqlguard.OpenDB(connector, middleware.WithN1Detection(5, time.Second)) +``` + +Using an ORM? The driver wrapper already covers anything built on +`database/sql` — sqlc, ent, sqlx, GORM, pgx-stdlib. For native pgx/pgxpool +and for ORM-specific seams, see [Integrations](integrations). + +## Static: scan your source + +```bash +sqlguard scan ./... +``` + +The scanner finds calls like `db.Query(...)`, `tx.ExecContext(...)` and +`stmt.QueryRow(...)`, resolves the SQL from literals, constants and +`fmt.Sprintf` with a constant format, and runs the same static rules: + +```text +[SQLGUARD CRITICAL] delete-without-where + File: internal/repo/users.go:42 + Query: DELETE FROM sessions + Issue: DELETE without WHERE clause detected. This will delete all rows. + Fix: Add a WHERE clause to limit the scope of the delete. + +1 issue(s) found (17 file(s) scanned) +``` + +It exits **1** when it finds anything and **0** when clean, so it drops into +a CI step as-is. `--format json` emits machine-readable output. See +[Static scanner](scan). + +## Plan: EXPLAIN a query + +```bash +sqlguard explain --db "postgres://app@localhost/app?sslmode=disable" \ + "SELECT id FROM orders WHERE customer_id = 42" +``` + +The query is planned — never executed — inside a read-only transaction that +is always rolled back, and the plan is checked for sequential scans, missing +indexes, filesorts and high-cost nodes. See [EXPLAIN analyzer](explain). + +## Configure once + +Drop a [`.sqlguard.yml`](configuration) at the repository root and every +surface picks it up. Feed the same file to the middleware with one line: + +```go +opts, err := config.Middleware("", ".") // discover from the working directory +sqlguard.Register("sqlguard-pg", "pgx", opts...) +``` + +Or suppress a single query inline, no config needed: + +```sql +SELECT * FROM feature_flags -- sqlguard:ignore:select-star +``` + +## Where next + +- [Runtime middleware](middleware) — every option, and what is intercepted. +- [Rules](rules) — what each of the 21 rules catches and why it matters. +- [Integrations](integrations) — GORM, sqlx, pgx, bun, xorm, ent. diff --git a/website/versioned_docs/version-0.5/gorm.md b/website/versioned_docs/version-0.5/gorm.md new file mode 100644 index 0000000..b878c1c --- /dev/null +++ b/website/versioned_docs/version-0.5/gorm.md @@ -0,0 +1,76 @@ +--- +id: gorm +title: GORM +description: gormguard — a gorm.Plugin that runs every GORM-generated statement through sqlguard. +--- + +# GORM + +`gormguard` is a `gorm.Plugin`. Once registered, every statement GORM +renders — ORM calls and `db.Raw` / `db.Exec` alike — is analyzed by the +shared core. + +```bash +go get github.com/KARTIKrocks/sqlguard/integrations/gormguard +``` + +```go +import ( + "github.com/KARTIKrocks/sqlguard/integrations/gormguard" + "github.com/KARTIKrocks/sqlguard/middleware" + "gorm.io/driver/postgres" + "gorm.io/gorm" +) + +gormDB, err := gorm.Open(postgres.Open(dsn), &gorm.Config{}) + +// Defaults: 200ms slow-query threshold, console reporter, no N+1. +gormguard.Register(gormDB) + +// Or with options — the same middleware.Option set as everywhere else. +gormguard.Register(gormDB, + middleware.WithSlowQueryThreshold(500*time.Millisecond), + middleware.WithN1Detection(10, time.Second), +) +``` + +## API + +| Symbol | Use | +| --- | --- | +| `gormguard.Register(db *gorm.DB, opts ...middleware.Option) error` | Build the plugin and `db.Use` it in one call. | +| `gormguard.New(opts ...middleware.Option) *Plugin` | Build the plugin yourself, when you need to keep a handle for `ResetN1()`. | +| `(*Plugin).ResetN1()` | Clear N+1 state at a request boundary. No-op unless `WithN1Detection` was passed. | +| `(*Plugin).Name() string` | `"sqlguard"` — the `gorm.Plugin` name. | + +```go +plugin := gormguard.New(middleware.WithN1Detection(10, time.Second)) +gormDB.Use(plugin) + +func handler(w http.ResponseWriter, r *http.Request) { + defer plugin.ResetN1() + // ... +} +``` + +## What is hooked + +GORM v2 routes operations through six callback chains — **Create**, +**Query**, **Update**, **Delete**, **Row** (`db.Raw(…).Scan` / `.Row`) +and **Raw** (`db.Exec`). The plugin registers before/after callbacks on all +six. Hooking only the ORM chains would silently miss every `db.Raw` and +`db.Exec` in the codebase, which is exactly the SQL most worth analyzing. + +GORM has not rendered `Statement.SQL` when the before-callback fires for +the ORM chains, so analysis happens in the after-callback with the final +SQL. Behaviour matches the driver wrapper: static rules run on every call; +latency is reported only when the statement succeeded. + +## Notes + +- GORM sits on `database/sql`, so `sqlguard.Register` on the underlying + driver is an alternative that needs no plugin. Use this adapter when you + want `ResetN1()`. Do not use both on one connection — each statement + would be analyzed twice. +- Findings are [redacted](redaction) like everywhere else; GORM's own + `Logger` is unaffected. diff --git a/website/versioned_docs/version-0.5/integrations.md b/website/versioned_docs/version-0.5/integrations.md new file mode 100644 index 0000000..f18f01f --- /dev/null +++ b/website/versioned_docs/version-0.5/integrations.md @@ -0,0 +1,64 @@ +--- +id: integrations +title: Integrations Overview +description: Which ORM and driver adapters exist, what seam each one hooks, what it covers, and when the plain database/sql wrapper is enough. +--- + +# Integrations Overview + +The [runtime middleware](middleware) wraps the `database/sql` driver, which +already covers every library built on `database/sql` — sqlc, ent, sqlx, +GORM, pgx-stdlib. The integrations exist for two reasons: + +1. **The library bypasses `database/sql`.** Native pgx / pgxpool talk to + PostgreSQL directly. [`pgxguard`](pgx) hooks pgx's tracer seam instead. +2. **You want a handle.** The driver wrapper hands back a bare `*sql.DB`, + so there is nothing to call `ResetN1()` on. Every integration holds the + guard and exposes `ResetN1()`, which lets you scope + [N+1 detection](n-plus-one) to a request. + +All six are built on the same exported `middleware.Guard`, so they inherit +redaction-by-default, fingerprints, the parser seam, config, de-duplication +and the analysis cache — and they all take the same +[`middleware.Option`](middleware#options) set. There is no second surface to +learn. + +## Coverage matrix + +| Module | Hooks | Covers | Install | +| --- | --- | --- | --- | +| [`gormguard`](gorm) | `gorm.Plugin` callbacks on all six chains | Create / Query / Update / Delete / Row / Raw | `go get github.com/KARTIKrocks/sqlguard/integrations/gormguard` | +| [`sqlxguard`](sqlx) | Wrapper around `*sqlx.DB` | `Select`, `Get`, `Queryx`, `NamedExec` (+ `Context`), `Query`, `Exec` | `go get github.com/KARTIKrocks/sqlguard/integrations/sqlxguard` | +| [`pgxguard`](pgx) | `pgx.QueryTracer` + `pgx.BatchTracer` | `Query`, `QueryRow`, `Exec`, `SendBatch` on `pgx.Conn` and `pgxpool.Pool` | `go get github.com/KARTIKrocks/sqlguard/integrations/pgxguard` | +| [`bunguard`](bun) | `bun.QueryHook` | Every query bun renders | `go get github.com/KARTIKrocks/sqlguard/integrations/bunguard` | +| [`xormguard`](xorm) | xorm `contexts.Hook` | Every statement xorm executes | `go get github.com/KARTIKrocks/sqlguard/integrations/xormguard` | +| [`entguard`](ent) | Decorates ent's `dialect.Driver` | `Exec`, `Query`, and transactions it opens | `go get github.com/KARTIKrocks/sqlguard/integrations/entguard` | + +Each is its own Go module, so its ORM dependency enters your build only +when you import it. All nine modules in the repository are released in +lockstep with the same version number. + +## Choosing + +- **On `database/sql` and happy with process-wide N+1 windows?** Use + `sqlguard.Register` / `OpenDB` and nothing else. It sees more than any + adapter — every statement, including the ones an ORM issues that its + hook seam does not surface. +- **On native pgx / pgxpool?** `pgxguard`. There is no alternative; the + driver wrapper never sees those queries. +- **Need `ResetN1()` per request on an ORM?** The matching adapter. +- **On sqlx and want everything covered?** Layer sqlx over the driver + wrapper (`sqlx.NewDb(sqlDB, "postgres")`) — it covers every sqlx method. + `sqlxguard` wraps the common helpers and gives you `ResetN1()`. + +Using two at once — the driver wrapper _and_ an ORM adapter on the same +connection — analyzes each statement twice. Pick one per connection. + +## Writing your own + +Any library with a before/after hook (or a single "here is the SQL" hook) +can be integrated in a few dozen lines on top of `middleware.NewGuard`. +`pgxguard` is the reference for split start/end hooks (`Guard.Observe`), +`gormguard` for a seam that only exposes SQL after execution +(`Guard.Check` + `Guard.CheckLatency`). See +[`Guard` for integration authors](middleware#guard-for-integration-authors). diff --git a/website/versioned_docs/version-0.5/intro.md b/website/versioned_docs/version-0.5/intro.md new file mode 100644 index 0000000..493ee20 --- /dev/null +++ b/website/versioned_docs/version-0.5/intro.md @@ -0,0 +1,80 @@ +--- +id: intro +title: Introduction +description: What sqlguard is, the three places it can analyze your SQL, and how the modules fit together. +slug: / +--- + +# Introduction + +**sqlguard** is a production-safe SQL query analyzer for Go. It finds the +queries that hurt in production — `SELECT *`, a `DELETE` with no `WHERE`, an +N+1 loop, a `LIKE '%…'` that can never use an index, a query that took 800 ms +— and reports each one with a rule name, a redacted copy of the query, and a +suggested fix. Think of it as `golangci-lint` for the SQL your application +actually runs. + +## Three entry surfaces, one analyzer + +Every surface runs the same rules through the same analyzer and reporter, so +a finding looks the same whether it came from a test run, a CI job or a +production log. + +| Surface | How | When to use it | +| --- | --- | --- | +| [Runtime middleware](./middleware.md) | Wraps the `database/sql` **driver**, returns a real `*sql.DB` | Catch everything your ORM or query builder emits, plus latency and N+1 | +| [Static scanner](./scan.md) | `sqlguard scan ./...` walks Go source and resolves string constants | Gate pull requests in CI without a database | +| [EXPLAIN analyzer](./explain.md) | `sqlguard explain --db …` plans a query on a live server | Check a specific query's plan for sequential scans and missing indexes | + +Six [ORM and driver integrations](./integrations.md) — GORM, sqlx, native pgx, +bun, xorm and ent — are additional runtime surfaces. They hook each library's +own callback seam and route through the same exported +[`middleware.Guard`](./middleware.md#guard-for-integration-authors), so they inherit +every runtime behaviour without a second option surface to learn. + +## What you get out of the box + +- **21 detection rules** — 14 static SQL rules, `slow-query` and + `n-plus-one` at runtime, and 5 plan-level rules from EXPLAIN. See + [Rules](./rules.md). +- **Redaction by default** — string and numeric literals become `?` before a + finding leaves the process, and every finding carries a stable, PII-free + [fingerprint](./redaction.md) that is safe as a metric label. +- **Quiet in production** — repeated findings are + [de-duplicated](./noise-control.md) per window, and repeated queries hit an + exact-string cache instead of being re-parsed. +- **One config file** — a [`.sqlguard.yml`](./configuration.md) drives the + middleware, the scanner and the CLI. [Inline suppressions](./suppressions.md) + work without any config. +- **A pluggable parser** — a zero-dependency fallback by default, or an + [opt-in real grammar](./parsers.md) for PostgreSQL or MySQL. + +## Module layout + +sqlguard is nine Go modules released in lockstep, so a heavy dependency never +enters your build unless you import the module that needs it: + +| Module | Import path | Third-party deps | +| --- | --- | --- | +| Core | `github.com/KARTIKrocks/sqlguard` | none in `analyzer`, `middleware`, `reporter` | +| Config | `…/sqlguard/config` | `gopkg.in/yaml.v3` (only here) | +| CLI | `…/sqlguard/cmd/sqlguard` | cobra, `go/packages`, database drivers | +| Parsers | `…/sqlguard/parsers/pgparser`, `…/parsers/mysqlparser` | the respective SQL grammar | +| Integrations | `…/sqlguard/integrations/{gormguard,sqlxguard,pgxguard,bunguard,xormguard,entguard}` | the respective ORM/driver | + +Importing `analyzer` or `middleware` pulls in nothing outside the standard +library. + +## What sqlguard is not + +- **Not a query rewriter.** It observes and reports; it never alters the SQL + or the arguments on their way to the driver. +- **Not a schema linter.** Rules reason about the statement text and, via + EXPLAIN, the plan — not your migrations. +- **Not a replacement for tracing.** The [pgx integration](./pgx.md) composes with + `otelpgx` and friends rather than replacing them. + +## Next steps + +Start with [Getting Started](./getting-started.md) — it takes about a minute to +see the first finding. diff --git a/website/versioned_docs/version-0.5/middleware.md b/website/versioned_docs/version-0.5/middleware.md new file mode 100644 index 0000000..b86aae1 --- /dev/null +++ b/website/versioned_docs/version-0.5/middleware.md @@ -0,0 +1,165 @@ +--- +id: middleware +title: Runtime Middleware +description: Wrap a database/sql driver so every query is analyzed at runtime — entry points, every option, what is intercepted, and the Guard core integrations build on. +--- + +# Runtime Middleware + +The runtime middleware intercepts at the `database/sql` **driver** layer, not +with a wrapper type around `*sql.DB`. That is the difference between "analyze +the calls you remembered to route through a helper" and "analyze every +statement that reaches the database" — including the ones an ORM, sqlc or a +query builder generates for you. You get a real `*sql.DB` back, and there is +no method list to keep in sync. + +## Entry points + +```go +import ( + "github.com/KARTIKrocks/sqlguard" + "github.com/KARTIKrocks/sqlguard/middleware" +) +``` + +| Function | Use when | +| --- | --- | +| `sqlguard.Register(name, baseDriver string, opts ...middleware.Option) error` | You open by driver name. Wraps the driver registered as `baseDriver` and registers the result as `name`; then `sql.Open(name, dsn)`. Errors if `name` is taken or `baseDriver` is unknown. | +| `sqlguard.OpenDB(c driver.Connector, opts ...middleware.Option) *sql.DB` | You already hold a `driver.Connector` (e.g. pgx's `stdlib.GetConnector`). | +| `middleware.WrapDriver(base driver.Driver, opts ...Option) driver.Driver` | You want the wrapped `driver.Driver` itself, to register under your own scheme. | +| `middleware.WrapConnector(base driver.Connector, opts ...Option) driver.Connector` | You want the wrapped connector, to pass to `sql.OpenDB` yourself. | + +`sqlguard.Register` and `sqlguard.OpenDB` are thin aliases for the +`middleware` functions of the same name; import whichever reads better. + +```go +_ "github.com/jackc/pgx/v5/stdlib" + +sqlguard.Register("sqlguard-pg", "pgx", + middleware.WithSlowQueryThreshold(500*time.Millisecond), + middleware.WithN1Detection(5, 2*time.Second), +) +db, err := sql.Open("sqlguard-pg", dsn) // a plain *sql.DB +``` + +## Options + +Every option is a `middleware.Option`. The same set is accepted by every +[integration](integrations), so there is one surface to learn. + +| Option | Default | Effect | +| --- | --- | --- | +| `WithSlowQueryThreshold(d time.Duration)` | `200ms` | Report `slow-query` when a successful query's driver-measured latency reaches `d`. Takes precedence over `rules.settings.slow-query.threshold` _0.3+_. | +| `WithReporter(r reporter.Reporter)` | `reporter.NewConsoleReporter()` (stderr) | Where findings go. `reporter.NewJSONReporter()` is built in; implement `Report([]analyzer.Result)` for anything else. | +| `WithAnalyzer(a *analyzer.Analyzer)` | `analyzer.Default()` | Replace the rule set — typically `analyzer.DefaultWithProfile(...)` from config, or an analyzer built `WithRawQuery()`. | +| `WithParser(p analyzer.Parser)` | `analyzer.FallbackParser` | Swap in a real grammar from [`parsers/`](parsers). Applied to whichever analyzer is in use. | +| `WithN1Detection(threshold int, window time.Duration)` | off | Report `n-plus-one` when the same query fingerprint runs `threshold` times within `window`. Takes precedence over `rules.settings.n-plus-one` _0.3+_. See [N+1 detection](n-plus-one). | +| `WithFindingDedup(window time.Duration)` | `1m` | Report each (rule, fingerprint) pair at most once per window. `0` reports every occurrence. See [Noise control](noise-control). | +| `WithAnalysisCacheSize(n int)` | `1024` | Memoize static analysis per exact query string in an LRU of `n` entries. `0` disables the cache. | + +Load them from a [`.sqlguard.yml`](configuration) instead of hard-coding: + +```go +opts, err := config.Middleware("", ".") // "" = discover from startDir +opts = append(opts, middleware.WithParser(pgparser.New())) +sqlguard.Register("sqlguard-pg", "pgx", opts...) +``` + +## What is intercepted + +The wrapper hand-implements the standard driver chain — `Driver` / +`DriverContext` → `Connector` → `Conn` → `Stmt` / `Tx` — so every path +`database/sql` can take is covered: + +- `Query`, `QueryRow`, `Exec` and their `Context` variants, on the `*sql.DB`, + on a `*sql.Conn`, and inside a `*sql.Tx`. +- Prepared statements: `Prepare` / `PrepareContext`, then `Stmt.Query` / + `Stmt.Exec` and their `Context` variants. The statement text is analyzed + on each execution (so a prepared statement in a loop still counts toward + N+1), and the analysis cache makes the repeat essentially free. +- Anything built on top of those — sqlc, ent, sqlx, GORM, pgx-stdlib. If it + talks to `database/sql`, it goes through the wrapper. + +For each executed statement the middleware runs the static rules (through +the cache and de-duplicator), feeds the N+1 tracker, and — when the +statement succeeds — measures latency for `slow-query`. A failed query's +latency is not reported; it is not a meaningful number. + +### Base-driver behaviour is preserved + +Optional driver interfaces (`QueryerContext`, `ExecerContext`, +`ConnBeginTx`, `Pinger`, `SessionResetter`, `Validator`, +`NamedValueChecker`, `ConnPrepareContext`, `StmtExecContext`, +`StmtQueryContext`) are implemented on every wrapper type, but each method +forwards to the base only if the base implements that interface — otherwise +it returns `driver.ErrSkip` or the documented no-op, so `database/sql` falls +back exactly as it would for the bare driver. A driver that lacks +`QueryerContext`, for example, still gets analyzed exactly once, on the +prepare-then-execute path `database/sql` takes instead. + +_Changed in 0.5._ The same holds for a driver that _has_ `QueryerContext` and +declines an individual query with `driver.ErrSkip` — `go-sql-driver/mysql` +does this for every parameterized query unless `interpolateParams=true`. +Before 0.5 that query was analyzed twice: once before the base declined, once +on the fallback path. N+1 counts were doubled on MySQL as a result. Analysis +now happens once the base has answered, so a declined query is analyzed only +where it actually runs. The same holds for `driver.ErrBadConn`, which +`database/sql` answers by retrying the query on another connection: the +attempt that hit the dead connection executed nothing, so it is not counted. + +The wrapper never modifies the SQL text or the arguments. It observes. + +## Reporters + +A `reporter.Reporter` is one method: + +```go +type Reporter interface { + Report(results []analyzer.Result) +} +``` + +`Report` may be called concurrently from many goroutines; both built-in +reporters take a mutex around their writer. The console reporter prints: + +```text +[SQLGUARD WARNING] select-star + Query: SELECT * FROM users WHERE email = ? + Issue: SELECT * detected. Selecting all columns can hurt performance. + Fix: Select only the columns you need. +``` + +The JSON reporter emits an array of objects with `rule`, `severity`, +`query`, `fingerprint`, `message`, `suggestion` and — in static mode — +`file` and `line`. Both accept an `io.Writer` via `NewConsoleReporterTo` / +`NewJSONReporterTo`. + +Custom reporters typically forward to a structured logger or a metrics +client. `Result.Fingerprint` is designed to be the label: it is stable, +low-cardinality and never carries a literal value. See +[Redaction & fingerprints](redaction). + +## `Guard`: for integration authors + +`middleware.Guard` is the single analysis core. Every interception point in +the driver chain calls into one `Guard`, and every out-of-tree integration +must too — that is what makes redaction, fingerprints, the parser seam, +config, N+1, de-duplication and the cache behave identically everywhere. + +```go +g := middleware.NewGuard(opts...) +``` + +| Method | Use | +| --- | --- | +| `Check(query string)` | Run the static rules and feed the N+1 tracker. Use when you have a single before-execution hook and no latency. | +| `CheckLatency(query string, elapsed time.Duration)` | Report `slow-query` if `elapsed` reaches the threshold. Use from an after-execution hook. | +| `Observe(query string) func(err error)` | `Check`, then start a timer. Call the returned closure when the operation completes; it reports latency only when `err == nil`. Designed for split start/end hooks — stash the closure in the `context.Context`. | +| `ResetN1()` | Clear N+1 state. Call at a request boundary. No-op when N+1 detection is off. | +| `Analyzer() *analyzer.Analyzer` | The configured analyzer, for `PrepareQuery` when you need redact/fingerprint without re-deriving policy. | + +A `Guard` is safe for concurrent use. The [pgx integration](pgx) is the +reference implementation: `TraceQueryStart` calls `Observe`, stores the +closure on the context, and `TraceQueryEnd` invokes it with +`data.Err`. Hand-rolling a check/latency pair instead of using `Guard` +silently loses everything above; do not. diff --git a/website/versioned_docs/version-0.5/n-plus-one.md b/website/versioned_docs/version-0.5/n-plus-one.md new file mode 100644 index 0000000..175680c --- /dev/null +++ b/website/versioned_docs/version-0.5/n-plus-one.md @@ -0,0 +1,107 @@ +--- +id: n-plus-one +title: N+1 Detection +description: How the runtime middleware spots the same query repeating in a loop, how the window works, and how to scope detection to a request. +--- + +# N+1 Detection + +An N+1 is the loop that loads a list, then issues one more query per row: + +```go +rows, _ := db.Query("SELECT id FROM orders WHERE customer_id = $1", cid) +for rows.Next() { + var id int + rows.Scan(&id) + db.QueryRow("SELECT status FROM shipments WHERE order_id = $1", id) // ×N +} +``` + +Nothing about any single statement is wrong, which is why static rules +cannot catch it. The runtime middleware can, because it sees the sequence. + +## Enabling it + +```go +sqlguard.Register("sqlguard-pg", "pgx", + middleware.WithN1Detection(5, 2*time.Second), // 5 hits in 2s → report +) +``` + +`WithN1Detection(threshold, window)` is off by default. _Added in 0.3._ It can +also be switched on from [`.sqlguard.yml`](configuration), which is the only +way to enable it without a code change: + +```yaml +rules: + settings: + n-plus-one: + threshold: 5 # both keys are required; one alone does nothing + window: 2s +``` + +`WithN1Detection` in code wins over those values, but `disable: [n-plus-one]` +in the file switches detection off regardless. When enabled, every +executed statement is reduced to its [fingerprint](redaction) — literals +replaced, whitespace collapsed, `IN (?, ?, ?)` folded to `IN (?)` — and +counted. When the same fingerprint reaches `threshold` executions inside +`window`, one finding is emitted: + +```text +[SQLGUARD WARNING] n-plus-one + Query: SELECT status FROM shipments WHERE order_id = ? + Issue: Possible N+1 query detected: same pattern executed 5 times in 2s + Fix: Consider using a JOIN or IN clause to batch these queries. +``` + +Because the key is the fingerprint, `WHERE order_id = 1`, `= 2`, `= 3` and +`= $1` all count as the same pattern — which is exactly what an N+1 looks +like from the driver. + +## How the window behaves + +- The window starts at the first sighting of a fingerprint. The finding + fires the moment the count reaches `threshold` within it. +- Each fingerprint is reported **once per window**. When the window + expires, the count resets and the pattern can be reported again. +- The tracker is bounded at 10,000 distinct fingerprints. Past that, expired + entries are evicted first; if every entry is still live, a new pattern is + dropped rather than the map grown — a rare false negative under + pathological query-shape cardinality, never a memory leak. Patterns already + being tracked are always honored. +- N+1 findings are not subject to [finding de-duplication](noise-control); + the once-per-window rule above is their own emission policy. + +## Scoping to a request + +On the plain `database/sql` path you get back a `*sql.DB` and nothing else, +so detection is process-wide and windowed: a hot endpoint that legitimately +runs the same lookup for many different users inside two seconds will look +like an N+1. Tune `threshold` and `window` to your traffic, or use one of +the [integrations](integrations) — every one of them holds the guard and +exposes `ResetN1()`: + +```go +tracer := pgxguard.NewTracer(middleware.WithN1Detection(10, time.Second)) + +func handler(w http.ResponseWriter, r *http.Request) { + defer tracer.ResetN1() // detection scoped to this request + // ... +} +``` + +Calling `ResetN1()` at a request boundary clears the counts, so a pattern +only trips the rule when it repeats **within one unit of work**. It is a +no-op when N+1 detection is not enabled. + +If you build your own integration on [`middleware.Guard`](middleware#guard-for-integration-authors), +`Guard.ResetN1()` is the same call. + +## What it will not catch + +- Loops that use different SQL text per iteration in a way that survives + fingerprinting (for example a table name interpolated per row). That is a + separate anti-pattern; the static rules will usually flag the + interpolation site. +- Queries issued from a library that bypasses `database/sql` entirely, unless + it has an integration — native pgx/pgxpool needs [pgxguard](pgx). diff --git a/website/versioned_docs/version-0.5/noise-control.md b/website/versioned_docs/version-0.5/noise-control.md new file mode 100644 index 0000000..020e351 --- /dev/null +++ b/website/versioned_docs/version-0.5/noise-control.md @@ -0,0 +1,104 @@ +--- +id: noise-control +title: Noise Control +description: Finding de-duplication and the per-query analysis cache — what keeps sqlguard quiet and cheap on a hot query path. +--- + +# Noise Control + +A query that runs ten thousand times a minute would, naively, produce ten +thousand identical warnings and ten thousand parses. Two mechanisms make the +runtime middleware quiet and cheap: **finding de-duplication** decides +whether a finding is _reported_, and the **analysis cache** decides whether a +query is _analyzed_. They are independent and both are on by default. + +## Finding de-duplication + +A finding's identity is the pair **(rule, fingerprint)** — the same rule +firing on the same canonical query shape. By default the middleware reports +each identity **at most once per minute**: + +```go +sqlguard.Register("sqlguard-pg", "pgx", + middleware.WithFindingDedup(5*time.Minute), // quieter +) +sqlguard.Register("sqlguard-pg", "pgx", + middleware.WithFindingDedup(0), // report every occurrence +) +``` + +Or in [`.sqlguard.yml`](configuration): + +```yaml +dedup: + window: 1m # "0" disables +``` + +Because the key is the [fingerprint](redaction) rather than the raw query, +`SELECT * FROM users WHERE id = 1` and `... WHERE id = 2` are the same +finding — one `select-star` warning per window, not one per user. + +What de-duplication does **not** touch: + +- `n-plus-one` — the tracker has its own once-per-window policy; see + [N+1 detection](n-plus-one). +- `slow-query` — reported on every slow execution on purpose. A query that + is slow three times in a row is three data points, not one. +- The static scanner and EXPLAIN analyzer — they run once per invocation, + so there is nothing to de-duplicate. + +The de-duplicator tracks up to 10,000 identities. At capacity it evicts +expired entries; if every entry is still inside its window, a _new_ +identity is dropped rather than the map grown without bound. Identities +already being tracked are always updated. + +## Analysis cache + +Static analysis — parse, build the normalized `Statement`, run every rule — +is the expensive part of `Guard.Check`. The middleware memoizes it per +**exact query string** in a bounded LRU: + +```go +sqlguard.Register("sqlguard-pg", "pgx", + middleware.WithAnalysisCacheSize(4096), // default 1024 +) +sqlguard.Register("sqlguard-pg", "pgx", + middleware.WithAnalysisCacheSize(0), // analyze every execution +) +``` + +A cache hit is a mutex-guarded map lookup with zero allocations — about +20 ns on a laptop, versus roughly 22 µs for a full analysis with the +default rule set. Parameterized queries (the common case) and repeated +identical strings hit; a query whose literals vary per call misses, and has +to, because its findings can differ. + +### Why the exact string, not the fingerprint + +The fingerprint folds literals away, but three rules read facts that live +in the literals: `large-offset` (the `OFFSET` value), `in-list-too-large` +(the element count) and `leading-wildcard` (`min-length` of the search +term). Two queries can share a fingerprint and deserve different verdicts — +`OFFSET 10` and `OFFSET 100000` — so keying on the fingerprint would cache +a wrong answer. Identical strings always analyze identically, which makes +the exact string the only fully correct key. + +### Interaction with the parser + +The cache sits in front of whichever [parser](parsers) is configured. With +a real grammar the uncached cost is higher, which makes the cache matter +more, not less. + +## Putting it together + +For a request that runs the same three parameterized statements: + +1. First execution of each: full analysis, findings reported, cache and + de-duplicator populated. +2. Every execution after that, within the window: cache hit; findings are + found again but suppressed by the de-duplicator; N+1 still counted; + latency still measured. +3. After the window: the next execution re-reports the finding, once. + +The cost of the steady state is the cost of step 2 — a map lookup and a +timestamp comparison. diff --git a/website/versioned_docs/version-0.5/parsers.md b/website/versioned_docs/version-0.5/parsers.md new file mode 100644 index 0000000..0d5906d --- /dev/null +++ b/website/versioned_docs/version-0.5/parsers.md @@ -0,0 +1,120 @@ +--- +id: parsers +title: SQL Parsers +description: The zero-dependency fallback parser, the opt-in PostgreSQL and MySQL grammars, which facts each derives structurally, and how parse failures degrade. +--- + +# SQL Parsers + +Rules never read raw SQL. They read an `analyzer.Statement` — a small, +dialect-agnostic struct of facts (`Kind`, `HasWhere`, `HasLimit`, +`SelectStar`, `OffsetValue`, …) that a `Parser` produces. That seam is what +lets the same rules run on a regex-free fallback and on a real grammar. + +```go +type Parser interface { + Parse(sql string) (*Statement, error) +} +``` + +## The default: `FallbackParser` + +Zero dependencies, ships in the core module, never returns an error. It +strips comments and string-literal contents before looking at the +statement, so keywords inside comments or strings and identifiers like +`update_at` do not cause false positives. CTEs, subqueries and driver +placeholders (`$1`, `?`, `:name`) are handled well enough for every +built-in rule. + +It is best-effort by design. A fact it cannot determine is left `false` +(or zero), and rules treat that as "not detected", never "proven absent" +— so the failure mode is a missed finding, not a spurious one. Every +`Statement` it produces has `Exact == false`. + +## Opt-in real grammars + +For structural certainty, add a dialect parser. Each lives in its own Go +module so the grammar dependency never enters your build unless you import +it: + +```bash +go get github.com/KARTIKrocks/sqlguard/parsers/pgparser # PostgreSQL — auxten/postgresql-parser, pure Go +go get github.com/KARTIKrocks/sqlguard/parsers/mysqlparser # MySQL — xwb1989/sqlparser (Vitess-derived), pure Go +``` + +```go +import "github.com/KARTIKrocks/sqlguard/parsers/pgparser" + +// Runtime middleware or any integration: +sqlguard.Register("sqlguard-pg", "pgx", middleware.WithParser(pgparser.New())) + +// Standalone analyzer: +a := analyzer.Default().WithParser(pgparser.New()) +``` + +`middleware.WithParser` applies to whichever analyzer is in use, including +one loaded from config, so `append(opts, middleware.WithParser(...))` is +enough. + +Neither parser uses cgo. + +## What a real parser changes + +| Fact | Fallback | Real parser | +| --- | --- | --- | +| `Kind` (SELECT / INSERT / UPDATE / DELETE / other) | lexical | AST | +| `HasWhere`, `HasLimit`, `HasOrderBy`, `HasFrom` | lexical | AST — correct through CTEs, subqueries, dialect syntax | +| `SelectStar`, `SelectDistinct` | lexical | AST — `COUNT(*)` and `COUNT(DISTINCT x)` never confuse it | +| `InsertColumnsListed` | lexical | AST | +| `OffsetValue` | lexical | AST limit clause | +| `MaxInListLen` (`in-list-too-large`) | lexical | **still lexical** — the AST discards the literal list | +| `ImplicitCommaJoin`, `CartesianJoin` | lexical | **still lexical** — deliberately text-level | +| `LeadingWildcardLike`, `LeadingWildcardTermLen`, `NonSargablePredicate`, `AddNotNullNoDefault` | lexical | **still lexical** — they read literal values or DDL text the AST does not carry | + +The first group is the false-positive-prone set; those become exact +(`Statement.Exact == true`). The rest stay best-effort heuristics +regardless of the parser, and each field's doc comment says so. This is a +documented carve-out, not a gap waiting to be closed. + +_Added in 0.5._ **A dialect parser is meant to remove findings, never add +them.** Opting in can drop a finding the heuristics guessed wrong; it +should not report one the fallback would not have. Each parser module +pins this direction against a corpus +(`TestParser_NeverAddsFindingTheFallbackDoesNot`), so a rule that reads a +structural field the grammar fills differently from the fallback fails +there rather than reaching you. That is a tested invariant over a corpus, +not a proof: a dialect form neither the corpus nor the fallback's keyword +list knows is how it would break again. + +In 0.4 and earlier it did break, in one rule, on every statement the +grammar recognised as inserting rows and the fallback did not: + +- `pgparser` reported `insert-without-columns` on + `INSERT INTO t DEFAULT VALUES`, which names no columns and needs none, + and on `UPSERT INTO t VALUES (…)`, plain or behind a CTE. +- `mysqlparser` reported it on `REPLACE INTO t VALUES (…)` and on the + forms that omit MySQL's optional `INTO`, such as `INSERT t VALUES (…)`. + +## Degradation on parse failure + +A real grammar will reject SQL it does not know: dynamic fragments, a +dialect extension, a CTE form the MySQL grammar predates, a placeholder +style it does not model. When that happens the dialect parser **returns +the fallback parser's `Statement`** (`Exact == false`) — never an error. +And if a `Parser` you write yourself does return an error, +`analyzer.Analyze` catches it and re-parses with the fallback. Either way, +analysis never breaks the caller's `db.Query`. + +## Writing a parser + +Implement `Parse` and fill the `Statement` fields you can derive; leave the +rest zero. The contract on the runtime path: + +- **Must not panic.** It runs inside every database call. +- **Should not error** for SQL it merely does not understand — degrade + to a best-effort `Statement` instead. `analyzer.NewFallbackParser()` is + exported so you can delegate to it. +- **Set `Exact`** only for the structural fields you actually computed + from an AST. + +Then pass it with `middleware.WithParser` or `analyzer.WithParser`. diff --git a/website/versioned_docs/version-0.5/pgx.md b/website/versioned_docs/version-0.5/pgx.md new file mode 100644 index 0000000..414a67f --- /dev/null +++ b/website/versioned_docs/version-0.5/pgx.md @@ -0,0 +1,106 @@ +--- +id: pgx +title: pgx / pgxpool +description: pgxguard — analyze native pgx and pgxpool queries through pgx's tracer seam, composing with otelpgx and other tracers already installed. +--- + +# pgx / pgxpool + +The `database/sql` wrapper covers pgx's stdlib shim (`pgx/v5/stdlib`). The +**native** APIs — `pgx.Conn`, `pgxpool.Pool` — never touch `database/sql`, +so they need `pgxguard`. It hooks pgx's own `QueryTracer` and +`BatchTracer` seams, which is how every pgx ecosystem tool extends the +driver, so every `Query` / `QueryRow` / `Exec` and every `SendBatch` is +analyzed without a wrapper type or a method list. + +```bash +go get github.com/KARTIKrocks/sqlguard/integrations/pgxguard +``` + +## With a pool + +```go +import ( + "github.com/KARTIKrocks/sqlguard/integrations/pgxguard" + "github.com/KARTIKrocks/sqlguard/middleware" + "github.com/jackc/pgx/v5/pgxpool" +) + +cfg, err := pgxpool.ParseConfig(dsn) +pgxguard.ApplyPool(cfg, + middleware.WithSlowQueryThreshold(50*time.Millisecond), + middleware.WithN1Detection(10, time.Second), +) +pool, err := pgxpool.NewWithConfig(ctx, cfg) +``` + +## With a single connection + +```go +cfg, err := pgx.ParseConfig(dsn) +pgxguard.Apply(cfg) +conn, err := pgx.ConnectConfig(ctx, cfg) +``` + +## API + +| Symbol | Use | +| --- | --- | +| `pgxguard.ApplyPool(cfg *pgxpool.Config, opts ...middleware.Option) *pgxpool.Config` | Install on a pool config. Returns `cfg` for chaining. Panics on nil. | +| `pgxguard.Apply(cfg *pgx.ConnConfig, opts ...middleware.Option) *pgx.ConnConfig` | Install on a connection config. Returns `cfg`. Panics on nil. | +| `pgxguard.NewTracer(opts ...middleware.Option) *Tracer` | Build the tracer yourself, to keep a handle for `ResetN1()` or to compose manually. | +| `(*Tracer).ResetN1()` | Clear N+1 state at a request boundary. | + +## Composes with existing tracers + +pgx allows exactly one `Tracer` per config, and production services usually +already have one — `otelpgx`, `ddtrace`, a logger. `Apply` and `ApplyPool` +do not overwrite it. If `cfg.Tracer` is set, they wrap both in pgx's own +`multitracer`, which fans every event out to each tracer and routes by +interface, so the existing tracer keeps receiving exactly the events it +did before. + +```go +cfg.ConnConfig.Tracer = otelpgx.NewTracer() +pgxguard.ApplyPool(cfg) // both tracers now run +``` + +## Keeping a handle for `ResetN1()` + +`Apply` builds the tracer internally. When you need `ResetN1()`, build it +yourself and install it — with `multitracer` if you also have another +tracer: + +```go +import "github.com/jackc/pgx/v5/multitracer" + +tracer := pgxguard.NewTracer(middleware.WithN1Detection(10, time.Second)) +cfg.ConnConfig.Tracer = multitracer.New(otelpgx.NewTracer(), tracer) + +func handler(w http.ResponseWriter, r *http.Request) { + defer tracer.ResetN1() + // ... +} +``` + +## Coverage details + +| pgx call | Seam | Static rules | N+1 | Latency | +| --- | --- | --- | --- | --- | +| `Query`, `QueryRow`, `Exec` | `QueryTracer` | yes | yes | yes, on success | +| Prepared statements executed via the above | `QueryTracer` | yes | yes | yes | +| `SendBatch` | `BatchTracer`, per statement | yes | yes | **no** | +| `CopyFrom` | — | no | no | no | + +- **Batches**: pgx exposes only the whole-batch round trip, not + per-statement latency, so `slow-query` is deliberately not reported for + batch statements rather than reported wrongly. +- **`PrepareTracer` is intentionally not implemented**: execution already + routes through `QueryTracer`, so tracing `Prepare` as well would + double-report findings and inflate N+1 counts. +- **`CopyFrom`** carries no SQL statement and is out of scope. + +The tracer is the reference implementation of the split start/end pattern +on [`middleware.Guard`](middleware#guard-for-integration-authors): +`TraceQueryStart` calls `Guard.Observe` and stashes the returned closure on +the context; `TraceQueryEnd` invokes it with `data.Err`. diff --git a/website/versioned_docs/version-0.5/redaction.md b/website/versioned_docs/version-0.5/redaction.md new file mode 100644 index 0000000..6a61976 --- /dev/null +++ b/website/versioned_docs/version-0.5/redaction.md @@ -0,0 +1,157 @@ +--- +id: redaction +title: Redaction & Fingerprints +description: Why findings never carry literal values, how Redact and Fingerprint work, what a fingerprint is safe for, and how to opt out for local debugging. +--- + +# Redaction & Fingerprints + +sqlguard's findings flow into logs. A query like +`SELECT * FROM users WHERE email = 'ceo@example.com'` is exactly the query +most likely to trip a rule, and the literal in it is exactly what must not +reach a log sink. So by default **no `Result` leaves the process with a +literal value in it**. This is a security invariant, not a formatting +choice — see the repository's `SECURITY.md`. + +## What redaction does + +`analyzer.Redact` is a zero-dependency lexical pass over the SQL: + +- Comments (`--` and `/* */`) are stripped. +- Every single-quoted string literal becomes `?`, honoring both `''` and + `\'` escapes. +- Every dollar-quoted string (`$$…$$`, `$tag$…$tag$`) becomes `?`. `$1` / + `$2` bind placeholders are left alone. +- Every numeric literal becomes `?`, including hex (`0x1F`) and binary + (`0b1011`) forms. +- Keywords, structure, and identifiers — including `"double-quoted"` and + `` `backtick` `` names — are preserved. + +It never errors; unparseable input still comes out with its literals gone. + +_Changed in 0.3._ Only the `''` escape was honored before, so a literal +containing `\'` closed at the wrong quote and the scanner then copied the +_following_ literal's contents out as query structure. Numeric redaction +stopped at the radix prefix, leaving `0x4142` as `?x4142` — the payload +intact. _Added in 0.3._ Dollar-quoted strings were not recognised at all and +passed through whole. + +```text +in: SELECT * FROM "users" WHERE email = 'a@b.c' AND age > 30 -- vip +out: SELECT * FROM "users" WHERE email = ? AND age > ? +``` + +### When the dialect is ambiguous, it over-redacts + +Dialects disagree about whether a backslash escapes a quote. MySQL's default +`sql_mode` and PostgreSQL's `E'…'` strings say yes; PostgreSQL with +`standard_conforming_strings` says the backslash is an ordinary byte. The two +readings disagree about _where a literal ends_, and picking the wrong one +desynchronises the lexer — it closes a literal at the wrong quote and then +emits the next literal's contents as if they were query structure. + +So `Redact` scans under both readings and redacts the union. A byte that is +inside a literal under _either_ reading is replaced: + +```text +in: SELECT * FROM t WHERE path = 'C:\' AND secret = 'hunter2' +out: SELECT * FROM t WHERE path = ? +``` + +That output has lost the `AND secret =` structure, which is the deliberate +trade: losing structure costs readability, and under-redacting leaks a value. +SQL with no backslash in it is unambiguous, scans identically under both +readings, and is unaffected. + +One consequence is worth knowing about, because it reaches past readability: +an over-redacted query has a **different fingerprint** from the same query +with an unambiguous value. Above, `p = 'C:\'` folds to +`… WHERE p = ?` while `p = 'plain'` folds to `… WHERE p = ? AND id = ?`. So a +query whose values only _sometimes_ end in a backslash — Windows paths, +`LIKE … ESCAPE` patterns, regexes — groups under two fingerprints instead of +one. That splits [N+1](n-plus-one) counts across both (a run that would trip +a threshold of 20 may land as 12 and 8 and trip neither) and lets the +[de-duplicator](noise-control) emit the same finding once per group. Bind +parameters avoid it entirely, and are the better answer for those values +anyway. + +### One dialect gap to know about + +A `"double-quoted"` run is treated as an **identifier** and preserved. That is +correct for ANSI SQL, for PostgreSQL, and for MySQL under `ANSI_QUOTES` — but +MySQL's default `sql_mode` also accepts `"…"` as a _string literal_, and +sqlguard does not redact it. Preserving quoted identifiers is what keeps +ORM-generated SQL readable in a finding, so the trade is deliberate. On MySQL, +use `'…'` or bind parameters for values you want redacted. + +`Result.Query` is set from `Redact` centrally in `Analyzer.Analyze`, and the +findings built outside the rule path — `slow-query` and `n-plus-one` — go +through `Analyzer.PrepareQuery`, which applies the same policy. There is one +normalizer in the codebase; nothing re-implements it. + +## Fingerprints + +Every `Result` also carries `Fingerprint`: the redacted query with +whitespace collapsed, `IN (?, ?, ?)` and `VALUES (?, ?)` lists folded to +`(?)`, and a trailing `;` trimmed. + +```text +SELECT id FROM t WHERE x IN (1, 2, 3) → SELECT id FROM t WHERE x IN (?) +SELECT id FROM t WHERE x IN (4, 5) → SELECT id FROM t WHERE x IN (?) +SELECT id\n FROM t\n WHERE x = 'a'; → SELECT id FROM t WHERE x = ? +``` + +A fingerprint is: + +- **Stable** — the same query shape always yields the same string. +- **PII-free** — it is derived from the redacted form. +- **Low-cardinality** — literals and list lengths do not create new values. + +That makes it safe as a metrics label, a log key, or a grouping key in your +observability stack. It is the value the [N+1 tracker](n-plus-one) groups +on and the [de-duplicator](noise-control) keys on. The JSON reporter emits +it as `fingerprint`; `analyzer.Fingerprint(sql)` computes it directly. + +`Fingerprint` is **always** populated, whether or not redaction is on. + +## Opting out + +Only where the query text is trusted — local debugging, a test — you can +keep the raw SQL in `Result.Query`: + +```go +a := analyzer.Default().WithRawQuery() +sqlguard.Register("sqlguard-pg", "pgx", middleware.WithAnalyzer(a)) +``` + +or in [`.sqlguard.yml`](configuration): + +```yaml +redact: false +``` + +`WithRawQuery` returns a copy; the original analyzer is unchanged. Do not +ship this to production. + +## The one deliberate exception + +The [EXPLAIN analyzer](explain) keeps `Result.Query` raw. The user typed the +query on their own command line; it goes back to their own terminal and +never reaches a log or telemetry sink. `Fingerprint` is still set. The +exception is scoped to `explain` only. + +## Using it in your own reporter + +```go +type metricsReporter struct{ counter *prometheus.CounterVec } + +func (m *metricsReporter) Report(rs []analyzer.Result) { + for _, r := range rs { + m.counter.WithLabelValues(r.RuleName, r.Fingerprint).Inc() + } +} +``` + +`RuleName` and `Fingerprint` are the two fields designed to be labels. +`Query` is for humans and `Message` for context; neither should be a metric +dimension. diff --git a/website/versioned_docs/version-0.5/rules.md b/website/versioned_docs/version-0.5/rules.md new file mode 100644 index 0000000..f9f8996 --- /dev/null +++ b/website/versioned_docs/version-0.5/rules.md @@ -0,0 +1,271 @@ +--- +id: rules +title: Detection Rules +description: Reference for all 21 rules — what triggers each one, why it matters, the suggested fix, default severity, and the tunables. +--- + +# Detection Rules + +Every finding names a rule. The name is stable: it is what you disable, +re-prioritise or tune in [`.sqlguard.yml`](configuration), and what you +put after `sqlguard:ignore:` in a [suppression](suppressions). + +Severity is one of `INFO`, `WARNING`, `CRITICAL`, and every rule's default +can be overridden per project. + +## At a glance + +| Rule | Severity | Where | Fires on | +| --- | --- | --- | --- | +| `select-star` | WARNING | static, runtime | `SELECT *` / `SELECT t.*` | +| `leading-wildcard` | WARNING | static, runtime | `LIKE '%…'`, `ILIKE '%…'` | +| `non-sargable-predicate` | WARNING | static, runtime | `WHERE LOWER(col) = …`, `WHERE col::text = …` | +| `add-not-null-without-default` | WARNING | static, runtime | `ALTER TABLE … ADD COLUMN … NOT NULL` with no `DEFAULT` | +| `implicit-join` | WARNING | static, runtime | `FROM a, b` | +| `cartesian-join` | WARNING | static, runtime | Multi-table `FROM` with no join condition and no `WHERE` | +| `in-list-too-large` | WARNING | static, runtime | `IN (…)` with more than `max-length` elements (default 100) | +| `large-offset` | WARNING | static, runtime | Literal `OFFSET` above `threshold` (default 1000) | +| `select-distinct` | INFO | static, runtime | `SELECT DISTINCT` | +| `delete-without-where` | CRITICAL | static, runtime | `DELETE` with no `WHERE` | +| `update-without-where` | CRITICAL | static, runtime | `UPDATE` with no `WHERE` | +| `insert-without-columns` | WARNING | static, runtime | `INSERT INTO t VALUES (…)` with no column list | +| `select-without-limit` | WARNING | static, runtime | `SELECT … FROM` with neither `LIMIT` nor `WHERE` | +| `orderby-without-limit` | INFO | static, runtime | `ORDER BY` with no `LIMIT` | +| `n-plus-one` | WARNING | runtime | Same fingerprint `threshold` times within `window` | +| `slow-query` | WARNING | runtime | Latency at or above the threshold (default 200 ms) | +| `seq-scan` | INFO / WARNING | EXPLAIN (postgres) | A `Seq Scan` node; WARNING above 1,000 estimated rows | +| `high-cost` | WARNING | EXPLAIN (postgres) | Any plan node with total cost above 10,000 | +| `full-table-scan` | WARNING | EXPLAIN (mysql) | Access `type = ALL` | +| `no-index-used` | WARNING | EXPLAIN (mysql) | Empty `key` **and** empty `possible_keys` | +| `filesort` | INFO | EXPLAIN (mysql) | `Using filesort` in `Extra` | + +_Changed in 0.3._ Every rule in this table is addressable by name in +[`.sqlguard.yml`](configuration): `disable` and `severity` work the same for a +runtime or plan rule as for a statement rule. In 0.2 only the 14 statement +rules were — naming any of the other seven warned with `unknown rule`, and +failed under `strict: true`. + +Two qualifications. `only:` is a whitelist over the rules evaluated against a +statement, so it reaches neither the runtime findings nor the plan rules — +see [what `only:` reaches](configuration#what-only-reaches). And `settings` only +exists where a rule has a tunable: `leading-wildcard`, `in-list-too-large`, +`large-offset`, `slow-query` and `n-plus-one` have them; the five plan rules +have none and their thresholds are fixed, so a `settings` block for one is +reported as having no effect. + +The runtime and plan rules are not evaluated against parsed SQL — middleware +derives them from latency and fingerprint counts, and `explain` from the +database's own plan — so they never fire during a static `sqlguard scan`. + +"static, runtime" rules read the normalized `Statement` a [parser](parsers) +produces; they never look at raw SQL. The runtime and EXPLAIN rules are +built into the [middleware](middleware) and the [EXPLAIN analyzer](explain) +respectively and are not part of `analyzer.Default()`. + +## Static rules + +### `select-star` + +`SELECT *` couples the query to the table's current column list: a new wide +column makes every caller slower, a dropped column breaks scans into +structs, and the database cannot serve the query from a covering index. +Aggregate forms such as `COUNT(*)` are not flagged. + +> **Fix:** Select only the columns you need. + +### `leading-wildcard` + +`LIKE '%foo'` and `LIKE '%foo%'` (and Postgres `ILIKE`) cannot use a B-tree +index — the planner has nothing to seek to — so they scan the whole table. + +> **Fix:** Use prefix search or a full-text index. + +**Setting `min-length`** (default `0`, off): ignore patterns whose +searchable term — the literal with its surrounding `%` trimmed — is shorter +than this. `LIKE '%x%'` on a small lookup table is often intentional. + +```yaml +rules: + settings: + leading-wildcard: + min-length: 3 +``` + +When the term length is unknown (a real parser that did not compute it), +the rule fires rather than staying silent — an unknown is never treated as +"short". + +### `non-sargable-predicate` + +A function or cast applied to a column on the column side of a comparison — +`WHERE LOWER(email) = $1`, `WHERE created_at::date = $1` — means an ordinary +index on that column cannot be used. + +> **Fix:** Compare the bare column instead, or add a matching expression/function index. + +### `add-not-null-without-default` + +`ALTER TABLE t ADD COLUMN c int NOT NULL` fails on a populated table, or +forces a full rewrite with an implicit default, depending on the engine. + +> **Fix:** Add a `DEFAULT`, or split into: add the column nullable, backfill, then `SET NOT NULL`. + +### `implicit-join` + +`FROM orders, customers` is the old comma join. It works, until the join +condition in `WHERE` is forgotten or dropped — and then it is a cartesian +product that returns rows × rows. + +> **Fix:** Use explicit `JOIN … ON` syntax. + +### `cartesian-join` + +The high-confidence subset of the above: multiple tables (comma join, +`CROSS JOIN`, or a bare `JOIN`) with **no** `ON` / `USING` / `NATURAL` and +**no** top-level `WHERE`. Both rules can fire on the same statement. + +> **Fix:** Add a `JOIN … ON` condition (or a `WHERE` clause relating the tables). + +### `in-list-too-large` + +A very long `IN (…)` value list is slow to plan, blows past prepared- +statement parameter limits on some drivers, and usually means a set that +should have been a join. `IN (SELECT …)` subqueries are never counted. + +> **Fix:** Use a `JOIN` against a temp table / `VALUES` list, or a parameterized array such as `= ANY($1)`. + +**Setting `max-length`** (default `100`): flag lists with more than this +many elements. + +### `large-offset` + +`OFFSET 100000` makes the database produce and discard 100,000 rows before +returning any. Page 1 is fast; page 1,000 is not. Parameterized offsets +(`OFFSET $1`) cannot be evaluated statically and are never flagged; MySQL's +`LIMIT offset, count` form is recognised. + +> **Fix:** Use keyset (cursor) pagination: `WHERE id > $last ORDER BY id LIMIT n`. + +**Setting `threshold`** (default `1000`): flag a literal offset above this. + +### `select-distinct` + +`SELECT DISTINCT` is frequently a patch over a join that fans out — the +duplicates are the bug, and `DISTINCT` hides it while adding a sort. This +is `INFO` by default because it is sometimes exactly right. Postgres +`DISTINCT ON` and MySQL `DISTINCTROW` count; `COUNT(DISTINCT col)` does not. + +> **Fix:** Confirm the duplicates aren't a join fan-out; prefer fixing the join or using `EXISTS` / `GROUP BY`. + +### `delete-without-where` + +A `DELETE` with no `WHERE` deletes every row. There is almost no situation +in application code where that is intended, which is why it is `CRITICAL`. +For the rare case that it is, [suppress it](suppressions) at the call site. + +> **Fix:** Add a `WHERE` clause to limit the scope of the delete. + +### `update-without-where` + +Same as above for `UPDATE`. + +> **Fix:** Add a `WHERE` clause to limit the scope of the update. + +### `insert-without-columns` + +`INSERT INTO t VALUES (…)` and `INSERT INTO t SELECT …` bind positionally +to the table's column order. Adding, dropping or reordering a column +silently shifts every value. + +_Changed in 0.5._ Every keyword that inserts rows this way is flagged, not +just `INSERT`: MySQL/SQLite's `REPLACE`, the `UPSERT` CockroachDB accepts, +and the forms that omit the optional `INTO` (`INSERT t VALUES (…)`). +Previously only a leading `INSERT INTO` was recognised, so the others were +reported solely to users who had opted into a dialect parser. + +Forms that name their columns, or have none to name, are not flagged: +MySQL's `INSERT INTO t SET col = …`, and PostgreSQL's +`INSERT INTO t DEFAULT VALUES`, which writes no caller-supplied values at +all. + +> **Fix:** Specify columns explicitly: `INSERT INTO table (col1, col2) VALUES (…)`. + +### `select-without-limit` + +A `SELECT … FROM` with neither a `WHERE` filter nor a `LIMIT` returns the +whole table. Fine for a ten-row config table; not for one that grows. +`SELECT 1` and `SELECT version()` (no `FROM`) are not flagged. + +> **Fix:** Add a `LIMIT` clause or `WHERE` filter to restrict results. + +### `orderby-without-limit` + +`ORDER BY` with no `LIMIT` sorts the entire result set even if the caller +only reads the first few rows. `INFO` by default: a full sorted export is a +legitimate query. + +> **Fix:** Add a `LIMIT` clause if you only need a subset of rows. + +## Runtime rules + +### `n-plus-one` + +Emitted by the middleware when the same query fingerprint executes +`threshold` times inside `window`. Off unless it is switched on — with +`WithN1Detection` in Go, or by setting both +`rules.settings.n-plus-one.threshold` and `.window` in +[config](configuration) _0.3+_. Full description in +[N+1 detection](n-plus-one). + +> **Fix:** Consider using a `JOIN` or `IN` clause to batch these queries. + +### `slow-query` + +Emitted when a successful query's latency, measured at the driver, reaches +the threshold (`WithSlowQueryThreshold`, default 200 ms; or +`rules.settings.slow-query.threshold` in config — _changed in 0.3_, this was +a top-level `slow-query.threshold` key). The message includes the measured time +and the threshold. Reported on every slow execution — it is not +[de-duplicated](noise-control). + +> **Fix:** Consider adding indexes or optimizing the query. + +## EXPLAIN rules + +Produced by [`sqlguard explain`](explain) from the query plan. _Changed in +0.3._ These are configurable through `rules:` in `.sqlguard.yml` like any +other rule; in 0.2 they were not, and naming one was an `unknown rule` +warning. + +### `seq-scan` (PostgreSQL) + +Every `Seq Scan` node in the plan. `INFO` when the planner estimates 1,000 +rows or fewer (a small table scan is often the right plan), `WARNING` +above that. The message carries the estimated rows and the node's total +cost. + +### `high-cost` (PostgreSQL) + +Any node whose `Total Cost` exceeds 10,000 planner units, named by node +type. + +### `full-table-scan` (MySQL / MariaDB) + +A plan row with access `type = ALL`, with the estimated row count. + +### `no-index-used` (MySQL / MariaDB) + +A plan row where both `key` and `possible_keys` are empty — the optimizer +did not even have a candidate index. + +### `filesort` (MySQL / MariaDB) + +`Using filesort` in the `Extra` column: the `ORDER BY` is not covered by an +index and is being sorted after the fact. + +## Tuning + +Everything above can be disabled, re-prioritised or tuned per project in +[`.sqlguard.yml`](configuration), or silenced at a single site with a +[suppression](suppressions). To add rules of your own, see +[Analyzer API](analyzer). diff --git a/website/versioned_docs/version-0.5/scan.md b/website/versioned_docs/version-0.5/scan.md new file mode 100644 index 0000000..ee5efce --- /dev/null +++ b/website/versioned_docs/version-0.5/scan.md @@ -0,0 +1,169 @@ +--- +id: scan +title: Static Scanner +description: sqlguard scan — find SQL issues in Go source without running the application, wire it into CI, and understand what it can and cannot resolve. +--- + +# Static Scanner + +`sqlguard scan` walks Go source, finds the calls that send SQL to a +database, recovers the query text, and runs the same static +[rules](rules) the runtime middleware runs. No database, no running +application, no test fixtures — it fits in a pre-commit hook or a CI step. + +## Usage + +```bash +go install github.com/KARTIKrocks/sqlguard/cmd/sqlguard@latest + +sqlguard scan . # current module +sqlguard scan ./internal/repository # one package tree +sqlguard scan --format json ./... # machine-readable +``` + +The scan is always recursive, so a path may be written plainly (`./internal`) +or with the Go package-pattern suffix (`./internal/...`); both select the same +files. With no path at all it scans the current directory. + +A trailing `...` is always read as the pattern, as it is in every Go tool. If +you genuinely have a directory named `...`, add a trailing slash +(`./queries/.../`) to address it — and sqlguard says so on stderr when the +argument is ambiguous, rather than reporting a clean run for a tree it never +opened. + +_Changed in 0.3._ Pattern handling as a whole is new, including that warning. +In 0.2 the spelling was rejected outright — `sqlguard scan ./...` failed with +`lstat ./...: no such file or directory` — so the form used throughout these +docs had to be written as `sqlguard scan .`, and no path was ever ambiguous. + +| Flag | Default | Effect | +| --- | --- | --- | +| `--format console\|json` | `console` | Output shape. JSON is an array of `{rule, severity, query, fingerprint, message, suggestion, file, line}`. | +| `--config ` | auto-discover | Load a specific [`.sqlguard.yml`](configuration). | +| `--no-config` | — | Ignore any config file; run every rule at its default. | + +Exit code is **1** when any issue is found and **0** when clean. + +Output is split by audience. The **console** format writes findings and the +summary line to **stderr**, so `2>` captures the human-readable report without +disturbing whatever your CI step prints to stdout. The **JSON** format writes +to **stdout**, so `--format json > findings.json` and pipes into `jq` both +work, and it always emits an array — `[]` on a clean run — so a consumer never +has to parse an empty file. + +_Changed in 0.3._ In 0.2 JSON also went to stderr, which made +`--format json > findings.json` produce an empty file, and a clean run printed +nothing at all rather than `[]`. + +```text +[SQLGUARD CRITICAL] delete-without-where + File: internal/repo/sessions.go:42 + Query: DELETE FROM sessions + Issue: DELETE without WHERE clause detected. This will delete all rows. + Fix: Add a WHERE clause to limit the scope of the delete. + +[SQLGUARD WARNING] select-star + File: internal/repo/users.go:17 + Query: SELECT * FROM users WHERE id = $1 + Issue: SELECT * detected. Selecting all columns can hurt performance. + Fix: Select only the columns you need. + +2 issue(s) found (17 file(s) scanned) +``` + +`File` and `Line` point at the call, not at the string constant — that is +where the fix goes. + +## What it looks for + +Any call whose method name is one of `Query`, `QueryContext`, `QueryRow`, +`QueryRowContext`, `Exec`, `ExecContext`, `Prepare` or `PrepareContext`, on +any receiver — `*sql.DB`, `*sql.Tx`, `*sql.Conn`, an `sqlx.DB`, a repository +interface of your own. The first argument is taken as the SQL — the second +for the `…Context` variants, whose first argument is the `ctx`. + +`_test.go` files, hidden directories, `vendor/` and `node_modules/` are +skipped, plus anything matching `scan.exclude-paths` in your config. + +## What it can resolve + +The scanner type-checks the target with `golang.org/x/tools/go/packages`, +so the query does not have to be an inline literal: + +| Argument shape | Resolved? | +| --- | --- | +| `db.Query("SELECT …")` | yes | +| `db.Query(selectUsers)` — a `const` in the same package | yes | +| `db.Query(queries.SelectUsers)` — a `const` in another package | yes | +| `db.Query("SELECT id " + "FROM users")` — constant concatenation | yes | +| `db.Query(fmt.Sprintf("SELECT * FROM %s WHERE id = %d", table, id))` | yes — the format string is analyzed with verbs neutralized | +| `db.Query(buildQuery(filters))` — a runtime value | no | +| `db.Query(q)` where `q` is a `var` | no | + +For `fmt.Sprintf`, numeric verbs become `0` and everything else becomes a +placeholder identifier, so the SQL keeps enough structure for the rules: +`SELECT * FROM %s` still trips `select-star`; `LIMIT %d` still counts as a +`LIMIT`. + +If the target is not a loadable module (no `go.mod`, or a build failure), +the scanner falls back to a plain `go/parser` walk that still resolves +inline literals, so a broken tree is reported rather than silently skipped. + +## Suppressing a finding + +Add a comment on the call line or the line directly above: + +```go +// sqlguard:ignore:delete-without-where +db.Exec("DELETE FROM sessions") +``` + +Or inside the SQL itself, which also works at runtime. Details in +[Suppressions](suppressions). + +## In CI + +```yaml +# .github/workflows/ci.yml +- uses: actions/setup-go@v7 + with: + go-version: "1.27" +- run: go install github.com/KARTIKrocks/sqlguard/cmd/sqlguard@latest +- run: sqlguard scan ./... +``` + +To keep the findings as a build artifact, redirect the JSON and let the exit +code still fail the step: + +```yaml +- run: sqlguard scan --format json ./... > sqlguard.json +- uses: actions/upload-artifact@v4 + if: always() + with: + name: sqlguard-findings + path: sqlguard.json +``` + +The step fails on any finding. To gate only on the serious ones, lower the +noisy rules in `.sqlguard.yml`: + +```yaml +rules: + severity: + select-distinct: "off" + orderby-without-limit: "off" +``` + +Severity does not affect the exit code — any reported finding is a +non-zero exit — so use `"off"` (or `rules.disable`) for rules you do not +want to gate on, rather than `info`. + +## Limits + +- Runtime-only rules (`n-plus-one`, `slow-query`) cannot fire statically. +- Only values the type checker can fold are seen. A query assembled at + runtime is invisible here — that is what the [middleware](middleware) is + for. +- The default [fallback parser](parsers) is best-effort. The scanner does + not currently accept a parser flag; it uses the analyzer your config + produces. diff --git a/website/versioned_docs/version-0.5/sqlx.md b/website/versioned_docs/version-0.5/sqlx.md new file mode 100644 index 0000000..da0eb3b --- /dev/null +++ b/website/versioned_docs/version-0.5/sqlx.md @@ -0,0 +1,71 @@ +--- +id: sqlx +title: sqlx +description: sqlxguard — wrap a *sqlx.DB so its Select, Get, Queryx and NamedExec helpers are analyzed, or layer sqlx over the driver wrapper for full coverage. +--- + +# sqlx + +sqlx builds on `database/sql`, so there are two ways to cover it. Pick one +per connection. + +## Option A: the driver wrapper (full coverage) + +Register the wrapped driver and hand sqlx the resulting `*sql.DB`. Every +sqlx method — `QueryRowx`, `NamedQuery`, `MustExec`, `Beginx`, all of it — +is covered, because interception happens beneath sqlx: + +```go +sqlguard.Register("sqlguard-pg", "pgx", opts...) +sqlDB, err := sql.Open("sqlguard-pg", dsn) +db := sqlx.NewDb(sqlDB, "pgx") +``` + +This is the recommended path unless you need `ResetN1()`. + +## Option B: `sqlxguard` (helpers + `ResetN1`) + +```bash +go get github.com/KARTIKrocks/sqlguard/integrations/sqlxguard +``` + +```go +import ( + "github.com/KARTIKrocks/sqlguard/integrations/sqlxguard" + "github.com/KARTIKrocks/sqlguard/middleware" + "github.com/jmoiron/sqlx" +) + +sqlxDB := sqlx.MustConnect("pgx", dsn) + +db := sqlxguard.WrapSqlx(sqlxDB, + middleware.WithSlowQueryThreshold(500*time.Millisecond), + middleware.WithN1Detection(10, time.Second), +) + +var users []User +err := db.Select(&users, "SELECT * FROM users") // warns: select-star +``` + +### API + +| Symbol | Use | +| --- | --- | +| `sqlxguard.WrapSqlx(db *sqlx.DB, opts ...middleware.Option) *WrappedDB` | Wrap. Panics on a nil `*sqlx.DB`. | +| `(*WrappedDB).DB() *sqlx.DB` | The underlying connection, for methods the wrapper does not expose. | +| `(*WrappedDB).ResetN1()` | Clear N+1 state at a request boundary. | +| `Select`, `SelectContext`, `Get`, `GetContext`, `Queryx` | sqlx helpers, analyzed. | +| `NamedExec`, `NamedExecContext` | Named-parameter exec, analyzed. | +| `Query`, `QueryContext`, `Exec`, `ExecContext` | `database/sql` passthroughs, analyzed. | +| `Ping`, `Close` | Passthroughs. | + +Everything you call through `db.DB()` bypasses analysis. That is the +trade-off of a wrapper type versus the driver layer, and the reason +Option A is the default recommendation. + +## Notes + +- With Option B, do **not** also register the driver wrapper for the same + connection — each helper call would be analyzed twice. +- The wrapper's `Query`/`Exec` return the plain `*sql.Rows` / `sql.Result` + types, exactly like sqlx's own. diff --git a/website/versioned_docs/version-0.5/suppressions.md b/website/versioned_docs/version-0.5/suppressions.md new file mode 100644 index 0000000..56f7bee --- /dev/null +++ b/website/versioned_docs/version-0.5/suppressions.md @@ -0,0 +1,83 @@ +--- +id: suppressions +title: Inline Suppressions +description: Silence a finding at one call site with a comment in the SQL or in the Go source — no config file required. +--- + +# Inline Suppressions + +Sometimes the rule is right in general and wrong here: the config table +really does have nine rows and `SELECT *` is fine; the nightly job really +does `DELETE FROM sessions`. A suppression silences a specific finding at a +specific site while leaving the rule on everywhere else. No config file is +needed. + +## In the SQL + +Put a `sqlguard:ignore` directive in a SQL comment anywhere in the statement: + +```sql +SELECT * FROM feature_flags -- sqlguard:ignore +DELETE FROM sessions /* sqlguard:ignore:delete-without-where */ +SELECT * FROM t ORDER BY created_at -- sqlguard:ignore:select-star, orderby-without-limit +``` + +- A bare `sqlguard:ignore` suppresses **every** rule for that statement. +- `sqlguard:ignore:rule-a,rule-b` suppresses only the named + [rules](rules). Whitespace after the commas is fine. +- `--`, `/* */` and `#` comments all work; the directive is case-insensitive. + +In-SQL directives are honored **everywhere the statement is analyzed**: the +runtime middleware, every integration, and the static scanner. Because the +text travels with the query, the suppression follows it through an ORM or a +query builder untouched. + +### Why it must be in a comment + +The directive is matched only when it follows a comment marker. A string +literal that happens to contain the words — say a support ticket body with +`'-- sqlguard:ignore'` in it — does not suppress anything. This is a +deliberate anchoring, not an accident of the regex. + +## In the Go source + +For the [static scanner](scan), a Go comment on the call line or on the +line directly above it also works: + +```go +// sqlguard:ignore +db.Exec("DELETE FROM sessions") + +db.Query("SELECT * FROM feature_flags") // sqlguard:ignore:select-star + +rows, err := tx.QueryContext(ctx, + "SELECT * FROM t") // sqlguard:ignore ← applies: this is the call's line +``` + +The scanner reads the file's comment map, so both `//` and `/* */` +comments qualify. The scope is exactly **the comment's own line and the +next line**; a directive two lines above a call does nothing. + +Go-source suppressions are only visible to the scanner. The runtime +middleware sees SQL strings, not Go files, so a call site that must be +quiet at runtime too needs the in-SQL form. + +## What suppression does not do + +- It does not affect the `slow-query` or `n-plus-one` runtime findings. Those + are turned off in [config](configuration) instead — `disable: [slow-query]` + — which since 0.3 works for every rule name in the reference table. + Those are about behaviour, not statement text; tune their thresholds via + [options](middleware#options) or scope N+1 with `ResetN1()`. +- It does not affect the [EXPLAIN analyzer](explain), which reports on the + plan. +- It is per statement. To turn a rule off for a whole project, use + `rules.disable` in [`.sqlguard.yml`](configuration); to lower its + severity, `rules.severity`. + +## Programmatic use + +`analyzer.ParseIgnoreComment(text string) (all bool, rules map[string]bool, found bool)` +is exported for tools that want to honor the Go-source form against their +own AST walk. It expects the comment text with the `//` or `/* */` already +stripped — which is how `go/ast` hands it to you. diff --git a/website/versioned_docs/version-0.5/xorm.md b/website/versioned_docs/version-0.5/xorm.md new file mode 100644 index 0000000..b5fef5d --- /dev/null +++ b/website/versioned_docs/version-0.5/xorm.md @@ -0,0 +1,56 @@ +--- +id: xorm +title: xorm +description: xormguard — an xorm contexts.Hook that analyzes every statement the engine executes. +--- + +# xorm + +`xormguard` implements xorm's `contexts.Hook`. xorm's `AfterProcess` hook +exposes the rendered SQL and the measured execution time, which is where +the static rules and the latency check run. + +```bash +go get github.com/KARTIKrocks/sqlguard/integrations/xormguard +``` + +```go +import ( + "github.com/KARTIKrocks/sqlguard/integrations/xormguard" + "github.com/KARTIKrocks/sqlguard/middleware" + "xorm.io/xorm" +) + +engine, err := xorm.NewEngine("pgx", dsn) + +engine.AddHook(xormguard.New( + middleware.WithSlowQueryThreshold(500*time.Millisecond), + middleware.WithN1Detection(10, time.Second), +)) +``` + +## API + +| Symbol | Use | +| --- | --- | +| `xormguard.New(opts ...middleware.Option) *Hook` | Build the hook. Pass to `engine.AddHook`. | +| `(*Hook).ResetN1()` | Clear N+1 state at a request boundary. | +| `BeforeProcess`, `AfterProcess` | The `contexts.Hook` methods; you do not call them. | + +```go +hook := xormguard.New(middleware.WithN1Detection(10, time.Second)) +engine.AddHook(hook) + +func handler(w http.ResponseWriter, r *http.Request) { + defer hook.ResetN1() + // ... +} +``` + +## Notes + +- Static rules run on every statement; latency is reported only on + success. Same semantics as every other surface. +- xorm sits on `database/sql`, so `sqlguard.Register` on the driver you + pass to `xorm.NewEngine` is the alternative. Use this adapter when you + want `ResetN1()`; do not use both on one engine. diff --git a/website/versioned_sidebars/version-0.5-sidebars.json b/website/versioned_sidebars/version-0.5-sidebars.json new file mode 100644 index 0000000..2deb5f5 --- /dev/null +++ b/website/versioned_sidebars/version-0.5-sidebars.json @@ -0,0 +1,59 @@ +{ + "docsSidebar": [ + "intro", + "getting-started", + { + "type": "category", + "label": "Runtime", + "collapsed": false, + "items": [ + "middleware", + "n-plus-one", + "noise-control", + "redaction" + ] + }, + { + "type": "category", + "label": "Rules", + "collapsed": false, + "items": [ + "rules", + "suppressions", + "configuration" + ] + }, + { + "type": "category", + "label": "CLI", + "collapsed": false, + "items": [ + "scan", + "explain" + ] + }, + { + "type": "category", + "label": "Integrations", + "collapsed": false, + "items": [ + "integrations", + "gorm", + "sqlx", + "pgx", + "bun", + "xorm", + "ent" + ] + }, + { + "type": "category", + "label": "Extending", + "collapsed": false, + "items": [ + "parsers", + "analyzer" + ] + } + ] +} diff --git a/website/versions.json b/website/versions.json index 28959c2..3654c90 100644 --- a/website/versions.json +++ b/website/versions.json @@ -1,4 +1,5 @@ [ + "0.5", "0.4", "0.3", "0.2"