diff --git a/.github/actions/verify/action.yml b/.github/actions/verify/action.yml new file mode 100644 index 0000000..d6d160c --- /dev/null +++ b/.github/actions/verify/action.yml @@ -0,0 +1,67 @@ +name: Verify worker +description: >- + lockfile どおりに依存を入れて lint / typecheck / test を回し、Worker が実際に + バンドルできることを dev・production 双方の設定で確認する。検証用ワークフローと + デプロイ用ワークフローで同じ手順を踏むため composite action に切り出してある。 + +inputs: + node-version: + description: Node.js のバージョン (package.json の engines と揃えること) + required: false + default: "22" + +# ローカル action は checkout 済みでないと解決できないため、checkout は +# 呼び出し側のワークフローに置いてある。 +runs: + using: composite + steps: + - uses: actions/setup-node@v4 + with: + node-version: ${{ inputs.node-version }} + cache: npm + + # npm install と違い package-lock.json を書き換えないので、ローカルで + # 動かしたのと同じ wrangler / biome / TypeScript の版で検証できる。 + # wrangler もこの lockfile から入るため、デプロイに使う版の固定先は + # ワークフロー側ではなく package-lock.json 一箇所で済む。 + - name: Install dependencies + shell: bash + run: npm ci + + - name: Lint + shell: bash + run: npm run lint + + - name: Typecheck + shell: bash + run: npm run typecheck + + - name: Test + shell: bash + run: npm test + + # tsc は型しか見ないので、import の解決ミスや nodejs_compat で賄えない + # Node API はバンドルして初めて落ちる。--dry-run は Cloudflare API を + # 叩かないため認証情報なしで回せる。 + # + # dev と production を両方バンドルするのは、wrangler.jsonc の env.production + # 側だけが壊れている状態を master へ入れる前に捕まえるため。dev への push + # では production 設定に一切触れないまま緑になってしまう。 + - name: Build (dry-run) + shell: bash + env: + WRANGLER_SEND_METRICS: "false" + # wrangler の色付けが Total Upload 行に混ざると要約が読めなくなる + NO_COLOR: "1" + run: | + # dev は wrangler.jsonc の top-level 設定。wrangler 4 は環境が複数ある + # 状態で --env を省くと警告を出すため、空文字でも明示する。 + for target in "" production; do + label="${target:-dev}" + npx wrangler deploy --env="$target" --dry-run \ + --outdir "$RUNNER_TEMP/bundle-$label" 2>&1 | tee "$RUNNER_TEMP/$label.log" + # Workers の上限は gzip 後で 10 MiB。今は 1/10 にも届かないので + # 失敗にはせず、増え方が見えるよう要約に残すだけにする。 + size=$(grep -m1 'Total Upload' "$RUNNER_TEMP/$label.log" || true) + echo "- \`$label\`: ${size:-size unknown}" >> "$GITHUB_STEP_SUMMARY" + done diff --git a/.github/workflows/auto_fix_from_feedback.yml b/.github/workflows/auto_fix_from_feedback.yml new file mode 100644 index 0000000..e586a30 --- /dev/null +++ b/.github/workflows/auto_fix_from_feedback.yml @@ -0,0 +1,159 @@ +# アプリに届いたフィードバックのうち、原因がこのリポジトリにあると判定された +# ものを Claude Code に読ませ、直せるなら修正の Pull Request まで作らせる。 +# +# 起点はこのリポジトリのフィードバックトリアージが自分で立てるスタブ issue。 +# 原因のリポジトリは受信時に判定済みなので、ここで振り分けをやり直さない。 +# 判定・無害化・プロンプトの組み立て・結果の報告は TrainLCD/feedback-autofix +# の composite action が持っている。 +# +# 出来上がった PR は必ず人がレビューすること。フィードバックの本文はアプリの +# 利用者がそのまま書いたもので、誰かが内容を確かめる工程が無い。個人情報を +# 取り除いたうえでデータとして渡しているが、プロンプトインジェクションの +# 抜け道を完全にはふさげない。自動マージもしない。 +name: Auto Fix From Feedback + +on: + # トリアージがこのリポジトリに立てるスタブ issue を起点にする。 + issues: + types: [opened] + # 取りこぼしをやり直すときと、スタブを介さず動かすときに使う。 + workflow_dispatch: + inputs: + issue_number: + description: "TrainLCD/Issues の issue 番号" + required: true + type: string + +permissions: + contents: write + # prepare がスタブ issue を GITHUB_TOKEN で取得する。permissions を書いた + # 時点で挙げなかった権限は none になるので、これを省くとそこで止まる。 + issues: read + pull-requests: write + +# 同じ管理チケットに対する実行を直列にする。スタブ issue の番号で束ねると、 +# 同じチケットを指すスタブが 2 つあったときに別のグループへ入って同時に走る。 +# エージェントが 2 つ動き、同じ名前のブランチを取り合う。 +# +# 題名には管理チケットへの参照が入っていて、トリアージが立てるものも引き継ぎで +# 立つものも同じ形なので、これで同じチケットは同じグループに入る。 +# cancel-in-progress を false にしてあるので後続は待たされ、先行が結果を +# 書き終えてから動き出してそこで打ち切られる。 +# +# 題名にコロンが入るため、値は引用符で囲むこと。囲まないと YAML がそこを +# キーの区切りと読み、このファイルを読み込めなくなる。 +concurrency: + group: "auto-fix-from-feedback-${{ github.event.issue.title || format('フィードバック対応: {0}#{1}', 'TrainLCD/Issues', inputs.issue_number) }}" + cancel-in-progress: false + +jobs: + auto-fix: + name: Auto fix from feedback + runs-on: ubuntu-latest + timeout-minutes: 60 + + # environment を宣言しない。ci.yml と同じ理由で、宣言するとこのジョブが + # その環境へのデプロイとして履歴に載り、環境 Secret に触れる状態になる。 + steps: + # persist-credentials: false を外さないこと。既定の true だと、書き込み + # 権限付きの GITHUB_TOKEN がローカルの git 設定に残り、npm ci の + # postinstall とエージェントの Bash(git:*) から素で使える状態になる。 + # 外部 action は可変タグではなく commit SHA で固定する。このジョブは + # contents: write と pull-requests: write を持ち、secret も渡している。 + # タグが差し替えられると、その内容がここで動く。 + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 + with: + ref: ${{ github.event.repository.default_branch }} + fetch-depth: 0 + persist-credentials: false + + - uses: TrainLCD/feedback-autofix/prepare@c71e2b4abf4149b4b83e9a87cd185b1558ed8237 # feedback-autofix#4 + id: prepare + with: + stub_issue_number: ${{ github.event.issue.number }} + issue_number: ${{ inputs.issue_number }} + # スタブ issue の作成者として認めるアカウントの数値 ID。このリポジトリ + # は公開されていて issue は誰でも立てられるので、これが無いと第三者が + # 管理チケットの番号を書いた issue を立てるだけで、非公開チケットの + # 本文を取得させられる。 + allowed_authors: "32848922" + issues_repo_token: ${{ secrets.ISSUES_REPO_TOKEN }} + github_token: ${{ secrets.GITHUB_TOKEN }} + anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} + # 共通アクションの既定は P1 だけ。2026 年に入って立った Bug は P1 が + # 8 件、P2 が 43 件、P3 が 3 件(duplicate と Spam を除く)で、P1 だけだと + # 月 1 件動くかどうかになる。3 つ足しても月 6 件程度(54 件 ÷ 8.6 か月)。 + # + # P3 を外す理由は無い。作るものから決める Feature Request や Improvement は + # カテゴリの条件が弾くので、ここへ残るのは Bug と Crash だけ。 + # + # 3 つとも入れたことでこの条件はほぼ素通りになるが、外さないこと。優先度が + # 付く前の issue を先に渡さないための関門になっている。 + triage_labels: "🟠 P1 / High,🟡 P2 / Medium,🟢 P3 / Low" + handoffs: MobileApp,StationAPI + guidelines_file: AGENTS.md + # このリポジトリに PR テンプレートは無い。 + pr_template: "" + # .github/actions/verify/action.yml が回すものと同じ。バンドルまで + # 入れてあるのは、tsc が型しか見ないため、import の解決ミスや + # nodejs_compat で賄えない Node API はバンドルして初めて落ちるから。 + # これを省くと、3 つを通した PR が CI のビルドで落ちる。 + checks: | + npm run lint + npm run typecheck + npm test + npx wrangler deploy --env="" --dry-run --outdir /tmp/feedback-autofix-bundle-dev + npx wrangler deploy --env=production --dry-run --outdir /tmp/feedback-autofix-bundle-production + scope: | + - `src/**` の Worker のコード + - `test/**` のテスト + + # ここから先は対象だったときだけ走らせる。届く issue のうち条件を満たす + # ものは一部なので、判定より先に重い準備を置かない。 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + if: steps.prepare.outputs.eligible == 'true' + with: + # package.json の engines と .github/actions/verify の既定に揃えること。 + node-version: "22" + cache: npm + + - run: npm ci + if: steps.prepare.outputs.eligible == 'true' + + - uses: anthropics/claude-code-action@cfc3eb22bfed5c26ef66e3223c982af27e4524de # v1 + id: claude + if: steps.prepare.outputs.eligible == 'true' + env: + # wrangler の色付けが Total Upload 行に混ざると読めなくなる。 + NO_COLOR: "1" + WRANGLER_SEND_METRICS: "false" + with: + anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} + prompt: ${{ steps.prepare.outputs.prompt }} + claude_args: | + # Sonnet で始める。まず費用を見たいため。上げるかどうかは「直せない」と + # 誤って判断した件数で決めること。この経路がいちばん高くつく。管理チケット + # に「済み」の目印が残り、次から打ち切られるので、手つかずのまま対応済みに + # 見える。出来の悪い修正は人のレビューで止まるが、誤った辞退は止まらない。 + --model claude-sonnet-5 + --allowedTools "Edit,Read,Write,Glob,Grep,TodoWrite,Bash(npm:*),Bash(npx:*),Bash(git:*),Bash(gh:*),Bash(node:*)" + + # always() を外さないこと。エージェントが失敗した場合こそ報告が要る。 + # 何も言わずに終わると、管理チケットを見た人には「調べたうえで何もしな + # かった」のか「そもそも動かなかった」のかが分からない。 + # + # failed も外さないこと。prepare が対象だと分かったあとで落ちた場合、 + # eligible は 'true' になりません。この条件が無いとここが走らず、ジョブが + # 赤くなるだけで管理チケットには何も残りません。 + - uses: TrainLCD/feedback-autofix/report@c71e2b4abf4149b4b83e9a87cd185b1558ed8237 # feedback-autofix#4 + if: ${{ always() && (steps.prepare.outputs.eligible == 'true' || steps.prepare.outputs.failed == 'true') }} + with: + issue_number: ${{ steps.prepare.outputs.issue_number }} + issues_repo_token: ${{ secrets.ISSUES_REPO_TOKEN }} + github_token: ${{ secrets.GITHUB_TOKEN }} + branch: ${{ steps.prepare.outputs.branch }} + verdict_path: ${{ steps.prepare.outputs.verdict_path }} + claude_outcome: ${{ steps.claude.outcome }} + # 引き継ぎ先に issue を立てられるトークン。省くと引き継ぎ先の名前が + # コメントに出るだけで、向こうは動かない。 + handoff_token: ${{ secrets.HANDOFF_ISSUE_TOKEN }} diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..9e156bf --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,85 @@ +# lint / typecheck / test が通ることと、Worker がバンドルできることを検証する +# だけのワークフロー。デプロイはしない。 +# +# デプロイは環境ごとに別ファイルへ分けてある: +# dev -> deploy_dev.yml (trainlcd-worker-dev) +# master -> deploy_production.yml (trainlcd-worker) +# デプロイ先をトリガとファイルで固定することで、他のブランチが誤って +# どこかの環境へ向くことがないようにしている。 +on: + pull_request: + # GitHub Actions は YAML のアンカー / エイリアスを解釈しないため、 + # このリストは push 側とも、デプロイ用の 2 ファイルとも二重に書く必要が + # ある。片方だけ直さないこと。 + paths: + - "src/**" + - "test/**" + - "scripts/**" + - "package.json" + - "package-lock.json" + - "tsconfig.json" + - "biome.json" + - "jest.config.js" + - "wrangler.jsonc" + - ".github/actions/verify/action.yml" + - ".github/workflows/ci.yml" + # deploy 用の 2 ファイルはどちらも pull_request で起動しないため、 + # ここに載せておかないと変更した PR がどの workflow も通らないまま + # マージされ、デプロイ時に初めて動くことになる。 + - ".github/workflows/deploy_dev.yml" + - ".github/workflows/deploy_production.yml" + push: + # dev / master は deploy_dev.yml / deploy_production.yml が同じ composite + # action で検証してからデプロイするため、ここでは走らせない。 + branches-ignore: + - dev + - master + paths: + - "src/**" + - "test/**" + - "scripts/**" + - "package.json" + - "package-lock.json" + - "tsconfig.json" + - "biome.json" + - "jest.config.js" + - "wrangler.jsonc" + - ".github/actions/verify/action.yml" + - ".github/workflows/ci.yml" + # deploy 用の 2 ファイルはどちらも pull_request で起動しないため、 + # ここに載せておかないと変更した PR がどの workflow も通らないまま + # マージされ、デプロイ時に初めて動くことになる。 + - ".github/workflows/deploy_dev.yml" + - ".github/workflows/deploy_production.yml" + workflow_dispatch: + +name: Continuous integration + +# 同じ PR / ブランチに続けて push したとき、古い方は結果が要らない。 +concurrency: + group: ci-${{ github.ref }} + cancel-in-progress: true + +permissions: + contents: read + +jobs: + verify: + name: Lint, typecheck, test and build + runs-on: ubuntu-latest + + # ここでは environment を宣言しない。environment は if: と違って + # ジョブが走れば必ず適用されるため、宣言すると全ブランチ・全 PR が + # その環境へのデプロイとして履歴に載り、環境 Secret (デプロイ用の + # CLOUDFLARE_API_TOKEN を含む) が任意のブランチのビルドから触れる。 + # + # 検証は --dry-run で Cloudflare API を叩かないため、そもそも認証情報が要らない。 + steps: + # checkout は既定で GITHUB_TOKEN を .git/config に残す。後続の npm ci は + # 依存パッケージの install スクリプトを実行するため読み取られうる。 + # ここから先で git 認証は使わない。 + - uses: actions/checkout@v4 + with: + persist-credentials: false + + - uses: ./.github/actions/verify diff --git a/.github/workflows/deploy_dev.yml b/.github/workflows/deploy_dev.yml new file mode 100644 index 0000000..69b531f --- /dev/null +++ b/.github/workflows/deploy_dev.yml @@ -0,0 +1,69 @@ +# dev を dev 環境 (trainlcd-worker-dev) へデプロイする。 +# +# デプロイ先はこのファイルとトリガで固定してある。ブランチを式で判定して +# 環境を選ぶ作りにすると、environment は if: と違ってジョブが走れば必ず +# 適用されるため、意図しないブランチがこの環境の履歴と Secret に触れる。 +on: + push: + branches: + - dev + # GitHub Actions は YAML のアンカー / エイリアスを解釈しないため、 + # このリストは ci.yml / deploy_production.yml とも二重に書く必要がある。 + # 片方だけ直さないこと。 + paths: + - "src/**" + - "test/**" + - "scripts/**" + - "package.json" + - "package-lock.json" + - "tsconfig.json" + - "biome.json" + - "jest.config.js" + - "wrangler.jsonc" + - ".github/actions/verify/action.yml" + - ".github/workflows/deploy_dev.yml" + workflow_dispatch: + +name: Deploy to dev + +# 同時に流れると、先に始まった古い版が後から上書きしうる。 +concurrency: + group: deploy-dev + cancel-in-progress: false + +permissions: + contents: read + +jobs: + deploy: + name: Verify and deploy to dev + runs-on: ubuntu-latest + + # workflow_dispatch にはブランチ絞り込みが無いので、ここで塞ぐ。 + # push は on: branches で dev に限定済み。 + if: github.ref == 'refs/heads/dev' + + environment: dev + + steps: + - uses: actions/checkout@v4 + with: + persist-credentials: false + + - uses: ./.github/actions/verify + + # 直前の検証で --dry-run 済みのものと同じ入力から同じバンドルが組み上がる。 + # wrangler は node_modules から解決されるので、検証と同じ版が走る。 + # + # dev は wrangler.jsonc の top-level 設定なので環境名は空にする。 + # wrangler 4 は環境が複数あると --env の省略を警告するため、空でも明示する。 + # + # Worker の secrets (SESSION_JWT_SECRET など) はここでは触らない。 + # deploy は既存の secrets を保持するため、投入は scripts/put-secrets.sh で + # 手元から行う運用のままでよい。 + - name: Deploy + env: + CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} + CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }} + WRANGLER_SEND_METRICS: "false" + run: npx wrangler deploy --env="" diff --git a/.github/workflows/deploy_production.yml b/.github/workflows/deploy_production.yml new file mode 100644 index 0000000..0b108e8 --- /dev/null +++ b/.github/workflows/deploy_production.yml @@ -0,0 +1,68 @@ +# master を production (trainlcd-worker) へデプロイする。 +# +# デプロイ先はこのファイルとトリガで固定してある。ブランチを式で判定して +# 環境を選ぶ作りにすると、environment は if: と違ってジョブが走れば必ず +# 適用されるため、意図しないブランチがこの環境の履歴と Secret に触れる。 +on: + push: + branches: + - master + # GitHub Actions は YAML のアンカー / エイリアスを解釈しないため、 + # このリストは ci.yml / deploy_dev.yml とも二重に書く必要がある。 + # 片方だけ直さないこと。 + paths: + - "src/**" + - "test/**" + - "scripts/**" + - "package.json" + - "package-lock.json" + - "tsconfig.json" + - "biome.json" + - "jest.config.js" + - "wrangler.jsonc" + - ".github/actions/verify/action.yml" + - ".github/workflows/deploy_production.yml" + workflow_dispatch: + +name: Deploy to production + +# 同時に流れると、先に始まった古い版が後から上書きしうる。 +concurrency: + group: deploy-production + cancel-in-progress: false + +permissions: + contents: read + +jobs: + deploy: + name: Verify and deploy to production + runs-on: ubuntu-latest + + # workflow_dispatch にはブランチ絞り込みが無いので、ここで塞ぐ。 + # push は on: branches で master に限定済み。 + if: github.ref == 'refs/heads/master' + + environment: production + + steps: + - uses: actions/checkout@v4 + with: + persist-credentials: false + + - uses: ./.github/actions/verify + + # 直前の検証で --dry-run 済みのものと同じ入力から同じバンドルが組み上がる。 + # wrangler は node_modules から解決されるので、検証と同じ版が走る。 + # + # production は wrangler.jsonc の env.production を指す。 + # + # Worker の secrets (SESSION_JWT_SECRET など) はここでは触らない。 + # deploy は既存の secrets を保持するため、投入は scripts/put-secrets.sh で + # 手元から行う運用のままでよい。 + - name: Deploy + env: + CLOUDFLARE_API_TOKEN: ${{ secrets.CLOUDFLARE_API_TOKEN }} + CLOUDFLARE_ACCOUNT_ID: ${{ secrets.CLOUDFLARE_ACCOUNT_ID }} + WRANGLER_SEND_METRICS: "false" + run: npx wrangler deploy --env production diff --git a/.secrets.env.example b/.secrets.env.example index 474d447..72122d3 100644 --- a/.secrets.env.example +++ b/.secrets.env.example @@ -7,7 +7,6 @@ # ./scripts/put-secrets.sh --env production # prod(.secrets.prod.env を使うなら SECRETS_FILE=... を併用) SESSION_JWT_SECRET= -AZURE_SPEECH_KEY= # GOOGLE_PLAY_SA_KEY はここに 1 行 JSON を入れてもよいが、複数行 JSON は # GOOGLE_PLAY_SA_KEY_FILE=./sa.json ./scripts/put-secrets.sh の方が扱いやすい。 GOOGLE_PLAY_SA_KEY= @@ -21,5 +20,19 @@ DISCORD_REVIEW_WEBHOOK_URL= ANTHROPIC_API_KEY= # AGENT_MODEL が openai: のとき必須 OPENAI_API_KEY= +# AGENT_MODEL が google: のとき必須(Vertex AI 用サービスアカウント鍵 JSON)。 +# 複数行 JSON は GOOGLE_VERTEX_SA_KEY_FILE=./secrets-vertex-sa.json ./scripts/put-secrets.sh が +# 扱いやすい(リポジトリ内に置くなら secrets*.json 名にすること。.gitignore 済み)。 +GOOGLE_VERTEX_SA_KEY= # LangSmith トレーシング(dev のみ・任意) LANGSMITH_API_KEY= +# --- フィードバックのトリアージ --- +# 判定(スパム・カテゴリ・優先度・原因コンポーネント)に使う TypeSafe の API キー。 +# 計測スクリプト(npm run typesafe-spike)はこの値を環境変数から直接読むため、 +# Worker のシークレットとして投入する必要があるのは本体を切り替えてから。 +TYPESAFE_API_KEY= +# --- TTS(/tts)--- +# Cloud Text-to-Speech を呼べるサービスアカウント鍵 JSON(必須)。 +# 複数行 JSON は GOOGLE_TTS_SA_KEY_FILE=./secrets-tts-sa.json ./scripts/put-secrets.sh が +# 扱いやすい(リポジトリ内に置くなら secrets*.json 名にすること。.gitignore 済み)。 +GOOGLE_TTS_SA_KEY= diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..59ee04e --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,33 @@ +# Repository Guidelines + +## Project Structure & Module Organization + +This is the Cloudflare Worker backend for TrainLCD. `src/index.ts` wires together HTTP, queue, and scheduled handlers. Keep endpoints in `src/routes/`, queue consumers in `src/consumers/`, Cron jobs in `src/scheduled/`, and integrations in `src/lib/`. Agent logic lives in `src/agent/`, shared models in `src/models/`, and maintenance commands in `src/cli/`. Tests are colocated as `*.test.ts`; shared Jest stubs live in `test/stubs/`. Deployment configuration is in `wrangler.jsonc`. + +## Build, Test, and Development Commands + +- `npm install`: install dependencies; Node.js 22 or newer is required. +- `npm run dev`: start the Worker locally with Wrangler. +- `npm test`: run the Jest unit suite once. Use `npm test -- --watch` while developing. +- `npm run typecheck`: run strict TypeScript checks without emitting files. +- `npm run lint`: check formatting and lint rules with Biome. +- `npm run format`: rewrite supported files to Biome formatting. +- `npm run deploy:dev` / `npm run deploy:prod`: deploy the development or production environment. + +Before submitting changes, run `npm run typecheck && npm run lint && npm test`. + +## Coding Style & Naming Conventions + +Use strict TypeScript; avoid `any`, parameter reassignment, and non-null assertions. Biome enforces two-space indentation, single quotes, and ES5-style trailing commas. Use `camelCase` for functions and variables, `PascalCase` for types, and descriptive filenames (for example, `feedbackTriage.ts`). Keep handlers thin and move testable logic into focused utilities. + +## Testing Guidelines + +Jest runs through `ts-jest` in a Node environment. Name tests `*.test.ts` beside the source they cover. Mock network, AI-provider, and Cloudflare-boundary behavior; shared AI stubs already exist under `test/stubs/`. Cover success paths, invalid input, and operational failure modes. Verify runtime integrations manually with `npm run dev`; scheduled handlers can be exercised with `wrangler dev --test-scheduled`. + +## Commit & Pull Request Guidelines + +This repository uses git-flow: `dev` is the development branch and `master` the release branch. Branch from `dev` as `feature/`, `fix/`, or `release/`; reserve `master` for releases. Use concise, imperative commit summaries, often in Japanese. Pull requests must assign `@TinyKitten`, target the appropriate git-flow branch, explain the motivation and behavior, link issues, list verification commands, and call out binding, secret, queue, KV, R2, or Cron changes. Include request/response examples for API changes. + +## Security & Configuration + +Never commit `.dev.vars`, `.secrets.env`, API keys, tokens, or service-account JSON. Start from `.secrets.env.example`, use Wrangler secrets for deployed environments, and review both development and production sections of `wrangler.jsonc` when changing bindings. diff --git a/README.md b/README.md index 7e7ff5b..e8a00e6 100644 --- a/README.md +++ b/README.md @@ -6,12 +6,12 @@ single Worker. ## Features -- **TTS synthesis** (`POST /tts`): synthesizes SSML into audio via Azure Speech and caches it in KV/R2. +- **TTS synthesis** (`POST /tts`): synthesizes plain text into audio via Google Cloud Text-to-Speech and caches it in KV/R2. - **Session issuance** (`POST /auth/token`): issues a short-lived session JWT from an install ID (the replacement for Firebase anonymous auth). - **Feedback intake** (`POST /postFeedback`): enqueues feedback onto the triage queue. - **Image upload** (`POST /feedback/upload-image`): stores feedback images in R2 and returns a public URL. - **App config delivery** (`GET /config/maintenance`, `GET /config/remote`): maintenance status and GPS thresholds (the replacement for Remote Config). -- **Feedback triage** (queue `feedback-triage`): summarizes and classifies feedback with Workers AI, then creates a GitHub Issue and notifies Discord. +- **Feedback triage** (queue `feedback-triage`): summarizes and classifies feedback with Workers AI, then creates a GitHub Issue and notifies Discord. Actionable feedback whose root-cause component the AI identifies confidently also gets a linked stub Issue in the matching public repo (see [public repo routing](#public-repo-routing) for the exact conditions). - **TTS cache writes**: synthesized audio is written directly from the `/tts` handler to R2 + KV (no queue is used, because audio does not fit within the 128 KB Queues limit). - **Review notifications** (Cron, hourly): notifies Discord of new App Store / Google Play reviews. @@ -20,9 +20,12 @@ single Worker. - **Cloudflare Workers** — `fetch` / `queue` / `scheduled` handlers - **Workers KV** — TTS cache metadata, config, and review read-state - **R2** — audio binaries and feedback images -- **Cloudflare Queues** — `feedback-triage` +- **Cloudflare Queues** — `feedback-triage` (+ `feedback-triage-dlq` as its dead letter queue) - **Workers AI** — feedback triage -- **Azure Speech** — TTS synthesis (SSML) +- **Google Cloud Text-to-Speech** — TTS synthesis (`Standard` voices, service-account auth) +- **OpenAI** — the conversational agent +- **Anthropic / Google Gemini (Vertex AI)** — alternative back ends for the + conversational agent; the provider is selected by the `AGENT_MODEL` var - **Google Android Publisher API** — Google Play review retrieval (service-account JWT) - **TypeScript / Biome / Jest / Wrangler** @@ -54,14 +57,22 @@ wrangler r2 bucket create trainlcd-tts-dev wrangler r2 bucket create trainlcd-uploads-dev # Queues wrangler queues create feedback-triage-dev +wrangler queues create feedback-triage-dev-dlq # dead letter queue (no consumer) ``` +The names above are the dev ones. Production uses the same set without the +`-dev` suffix (`trainlcd-tts`, `trainlcd-uploads`, `feedback-triage`, +`feedback-triage-dlq`); create those too if you are setting up prod from +scratch. Both environments' resources already exist on the TrainLCD account. + ### Setting secrets ```bash wrangler secret put SESSION_JWT_SECRET # signing key for session JWTs (any long random string) -wrangler secret put AZURE_SPEECH_KEY # Azure Speech subscription key wrangler secret put GOOGLE_PLAY_SA_KEY # Android Publisher SA key JSON (single-line string) +wrangler secret put OPENAI_API_KEY # the conversational agent +wrangler secret put GOOGLE_TTS_SA_KEY # Cloud Text-to-Speech SA key JSON (for POST /tts) +wrangler secret put GOOGLE_VERTEX_SA_KEY # Vertex AI SA key JSON; only when AGENT_MODEL is "google:" wrangler secret put OCTOKIT_PAT wrangler secret put DISCORD_CS_WEBHOOK_URL wrangler secret put DISCORD_CRASH_WEBHOOK_URL @@ -76,9 +87,40 @@ You can also bulk-load secrets with the helper scripts: copy ### Non-secret configuration (vars) -See `vars` in `wrangler.jsonc`. Configure the Azure region, voice names, AI model -name, package name, public upload URL (the R2 public domain), and so on per -environment. +See `vars` in `wrangler.jsonc`. Configure the TTS voice names and delivery +(`TTS_SPEED` / `TTS_PITCH`), AI model name, package name, public upload URL (the +R2 public domain), and so on per environment. + +Synthesis runs on Google Cloud Text-to-Speech, which authenticates with a service +account rather than an API key: `GOOGLE_TTS_SA_KEY` holds the key JSON and the +Worker signs a JWT with Web Crypto to obtain an access token. The Cloud +Text-to-Speech API must be enabled on that project. Unlike the agent providers, +TTS is not routed through Cloudflare AI Gateway (Cloud TTS is not a supported +gateway provider). + +The conversational agent picks its provider from `AGENT_MODEL`, written as +`:`: + +| `AGENT_MODEL` | Provider | Required secret | +| ------------------------- | ----------------------- | ---------------------- | +| `openai:gpt-5.6-luna` | OpenAI | `OPENAI_API_KEY` | +| `anthropic:` | Anthropic | `ANTHROPIC_API_KEY` | +| `google:gemini-3.8-flash` | Google Vertex AI | `GOOGLE_VERTEX_SA_KEY` | + +Switching providers is a vars-only change (`wrangler deploy`) **as long as that +provider's secret is already set** — no code change is needed. If it is missing, +`/agent/chat` fails on every request (the model resolver throws +` is not configured`), so put the secret in before flipping +`AGENT_MODEL`. When `AI_GATEWAY_BASE_URL` is set, every provider is routed +through Cloudflare AI Gateway (`/anthropic/v1`, `/openai`, +`/google-vertex-ai/v1beta1`) with request bodies excluded from the gateway logs. + +Gemini runs on **Vertex AI**, which authenticates with Google Cloud credentials +(ADC) rather than an API key. Workers have no ADC, so `GOOGLE_VERTEX_SA_KEY` +holds a service-account key JSON (role: *Vertex AI User*) and the Worker signs a +JWT with Web Crypto to obtain an access token per request. Two optional vars go +with it: `GOOGLE_VERTEX_PROJECT` (defaults to the key's `project_id`) and +`GOOGLE_VERTEX_LOCATION` (defaults to `global`). ## Develop & deploy @@ -92,6 +134,45 @@ npm run deploy:prod # wrangler deploy --env production npm run tail # follow logs ``` +### CI / CD (GitHub Actions) + +Deploys run from GitHub Actions. The target is fixed by the workflow file and +its trigger rather than chosen by an expression, so no branch can point at an +environment it was not meant to reach: + +| Workflow | Trigger | Result | +| ------------------------------------ | -------------------------------- | ----------------------------------- | +| `.github/workflows/ci.yml` | PRs, and pushes to other branches | Verify only, no deploy | +| `.github/workflows/deploy_dev.yml` | push to `dev` | Deploy `trainlcd-worker-dev` | +| `.github/workflows/deploy_production.yml` | push to `master` | Deploy `trainlcd-worker` | + +All three run the same `.github/actions/verify` composite action first — `npm +ci`, lint, typecheck, tests, and a `wrangler deploy --dry-run` of **both** the +dev and the production config. The dry run bundles the Worker for real, so +import mistakes and missing `nodejs_compat` APIs fail there rather than at +deploy time, and building the production config on every run catches an +`env.production` that only breaks after the merge to `master`. `npm ci` installs +wrangler from `package-lock.json`, so the version that deploys is the version +the lockfile pins — there is no second place to bump. + +Each deploy workflow needs two secrets on its GitHub environment (`dev` and +`production` respectively): + +- `CLOUDFLARE_API_TOKEN` — the *Edit Cloudflare Workers* template plus + **Queues: Edit**, since `wrangler deploy` also applies the queue consumer + settings from `wrangler.jsonc`. +- `CLOUDFLARE_ACCOUNT_ID` — `wrangler.jsonc` carries no `account_id`. + +Keeping them on the environment rather than on the repository is what stops an +arbitrary branch from reading the production token: `ci.yml` deliberately +declares no `environment`, and it needs no credentials because `--dry-run` never +calls the Cloudflare API. + +Worker secrets (`SESSION_JWT_SECRET`, `OCTOKIT_PAT`, …) are **not** touched by +the workflows. `wrangler deploy` preserves the secrets already on a Worker, so +they stay a manual `scripts/put-secrets.sh` step — see [Setting +secrets](#setting-secrets). + ## Client wire protocol `POST /tts` and `POST /postFeedback` keep the Firebase callable-compatible wire @@ -103,10 +184,56 @@ format. A session JWT is obtained from `POST /auth/token` (body `{ "installId": "" }`). +### `POST /tts` + +Synthesis runs on Google Cloud Text-to-Speech. The client sends plain text; SSML +is **not** interpreted (stray tags are stripped server-side rather than read +aloud), and delivery is steered by the `TTS_SPEED` / `TTS_PITCH` vars. + +```json +{ + "data": { + "textJa": "次は、オオサキです", + "textEn": "The next station is Osaki, J-Y 24.", + "jaVoiceName": "ja-JP-Standard-B", + "enVoiceName": "en-US-Standard-G" + } +} +``` + +Every field is optional except that **at least one of `textJa` / `textEn` must +be present**. Synthesis is billed per character, so the app omits a language the +user has switched off; only the languages it asks for are synthesized, cached, +and returned. Voice names are checked against an allowlist of voices that are +known to exist (`ja-JP` / `en-US` in the `Standard` / `Wavenet` / `Neural2` +families) — anything else falls back to the KV config (`config:tts`) and then to +the `TTS_*` vars. That keeps a client from naming an arbitrary (far more +expensive) voice such as `Studio`, `Chirp3-HD`, or a Gemini-TTS voice, and also +keeps a well-formed but non-existent name (`ja-JP-Standard-Z`) from reaching the +API, where it would fail the whole request with a 400. Using another locale +means adding its voices to the list in `src/utils/ttsVoice.ts`. The `model` / `instructions*` +fields of the previous OpenAI-based engine are accepted but ignored, so older +app builds keep working. + +The response carries only the requested languages: + +```json +{ + "result": { + "id": "", + "jaAudioContent": "", + "jaAudioMimeType": "audio/mpeg", + "enAudioContent": "", + "enAudioMimeType": "audio/mpeg" + } +} +``` + ## Testing strategy -Unit tests cover pure functions (SSML formatting, voice-name resolution, triage -JSON normalization, review parsing) with Jest. Runtime integration for HTTP / +Unit tests cover pure functions (TTS request building, voice/model resolution, +text validation, cache writes, triage JSON normalization, review parsing) with +Jest. Runtime integration for HTTP / queue / Cron is verified with `wrangler dev` / `wrangler dev --test-scheduled`. ## few-shot data @@ -120,6 +247,16 @@ example per line (see `fewshot.example.jsonl`): {"input": "user body text", "output": "{\"title\":...,\"isSpam\":false,...}"} ``` +The `output` of each example must include `component` / `componentConfidence` +as well; the model imitates the examples, so examples without those fields make +it omit them and public repo routing never fires. + +Optional per-example fields: `weight` (a value above 1 makes the example more +likely to survive sampling) and `disabled: true` (skips the line). Only +`FEW_SHOT_LIMIT` examples are sampled per request, so a component or category +with a single example — `praise`, `functions`, `website` — must carry a +`weight`, otherwise it drops out of the prompt and the model never produces it. + Upload (the file is stored verbatim as a single KV value): ```bash @@ -132,6 +269,233 @@ wrangler kv key put --binding CONFIG_KV "config:fewshot" --path fewshot.jsonl -- If it is not present, triage fails hard with `FEW_SHOT_NOT_AVAILABLE` (a fail-hard guard that prevents mis-training). +## Triage safeguards + +Two guards keep bad triage output from degrading the backlog +(`src/consumers/feedbackTriage.ts`): + +- **Gratitude is never spam.** Praise-only feedback gets `category: "praise"` + (`💚 Praise`, P3) instead of `💩 Spam` — there is nothing to fix, but it is + still a real message from a real user. Spam means content unrelated to + improving the app: announcement transcripts, unrelated chit-chat, ads. +- **Spam heuristic is advisory, not authoritative.** `looksLikeSpam()` only + scores announcement-transcript signals when an actual announcement phrase is + present — "停車駅" / "方面" / station enumerations are core domain vocabulary + and appear in legitimate data reports. When the heuristic and the model + disagree and the model is confident (`confidence` ≥ + `SPAM_OVERRIDE_MAX_CONFIDENCE`), the model wins and the Issue is tagged + `❓ Unknown Type` for a human check instead of being buried as spam. +- **Broken titles never reach the backlog.** `findBrokenTitleReason()` rejects + titles that are missing, mojibake, foreign-script, looping, or a run of + particles. Those are filed with the `要約失敗` marker plus `❓ Unknown Type`, + and a `console.warn` records the reason so the corruption rate can be measured + with `wrangler tail`. A broken title is never propagated into the summary. + +`AI_TRIAGE_MODEL` therefore needs decent Japanese generation quality — it writes +the Issue title and summary. Before switching it, verify three things against the +real prompt (`wrangler dev` with the AI binding hits the live API even locally): + +1. **JSON schema mode is supported** — `wrangler ai models schema ` must + list `response_format` / `json_schema`. +2. **The response shape** — some models return `response`, others only + `choices[0].message.content`. `pickModelResponse()` accepts both; a model + returning neither would fail every message. +3. **`max_tokens` headroom** — reasoning models spend most of the budget on the + trace before the JSON. Watch for `finish_reason: "length"`, which truncates + the JSON and looks like a parse failure. + +Measured with the real prompt + few-shot (2026-08): `@cf/google/gemma-4-26b-a4b-it` +completes in 5–17 s at 26–62 neurons per feedback, versus ~1 s and ~4.7 neurons +for the old 8B model. Queue consumers use `max_batch_size: 5`, so a batch stays +well inside the invocation limit. + +### Dead letter queue + +`processFeedbackMessage()` rethrows on failure so the consumer can `retry()`, +but retrying does not help when the cause is permanent — a credential that no +longer grants access, a repo that was renamed. Without a dead letter queue the +message is simply dropped once `max_retries: 3` is exhausted, and the feedback +is lost for good. + +The consumers therefore declare `dead_letter_queue` (`feedback-triage-dlq`, and +`feedback-triage-dev-dlq` for dev). The DLQ intentionally has **no consumer** — +running the same handler against it would fail for the same reason. + +Do not look for the wiring on the DLQ itself. Moving a message into a DLQ is +something Cloudflare does internally, not something the Worker sends, so a +correctly configured DLQ still reports zero producers and zero consumers in +`wrangler queues list`. The link lives on the **source** queue's consumer, and it +only exists once the config has been deployed: + +```bash +wrangler queues consumer list feedback-triage-dev # or feedback-triage for prod +# dead_letter_queue must name the DLQ; "-" means this environment is still +# dropping messages once max_retries is exhausted. +``` + +**A DLQ is not archival storage.** Messages sitting in it expire on the queue's +retention period, which both DLQs inherit from the account default (4 days on a +paid plan, and not extendable beyond 24 h on the free plan). That is the replay +deadline: once it passes the feedback is gone just as surely as it was before +this queue existed. Check and extend it if an incident may outlast it: + +```bash +# 1209600 = 14 days, the paid-plan maximum. The free tier is capped at 86400 +# (24 h) and rejects anything above it, so use that value instead on free. +# Run this for the dev DLQ too — retention is per queue, and +# feedback-triage-dev-dlq inherits nothing from the prod one. +RETENTION=1209600 +for q in feedback-triage-dlq feedback-triage-dev-dlq; do + wrangler queues update "$q" --message-retention-period-secs "$RETENTION" +done +``` + +`wrangler queues info` does not print the retention period (as of wrangler 4.103), +so there is no CLI read-back for it — set it explicitly, or check the dashboard. + +To recover, fix the root cause first, then replay by temporarily attaching a +consumer to the DLQ. Give that consumer its own dead letter queue — a replay +consumer runs the same handler, so anything still failing would hit `max_retries` +and be deleted outright, which is the exact loss this section exists to prevent. + +```bash +# production. For dev: DLQ=feedback-triage-dev-dlq, SCRIPT=trainlcd-worker-dev +DLQ=feedback-triage-dlq +SCRIPT=trainlcd-worker +RETENTION=1209600 # 86400 on the free tier, as above + +wrangler queues info "$DLQ" # backlog size, current consumers + +# Catches whatever still fails on replay. Give it the same retention as the DLQ, +# otherwise it silently falls back to the account default. +wrangler queues create "$DLQ-quarantine" --message-retention-period-secs "$RETENTION" + +# --batch-size 5 matches the max_batch_size the regular consumer runs with; the +# default of 10 is a lot for one invocation at 5–17 s of inference per message. +wrangler queues consumer add "$DLQ" "$SCRIPT" \ + --dead-letter-queue "$DLQ-quarantine" --batch-size 5 + +# ...wait for the backlog to drain, then detach: +wrangler queues consumer remove "$DLQ" "$SCRIPT" +``` + +The replay consumer is deliberately not declared in `wrangler.jsonc` — it exists +only for the duration of an incident, so nothing will remove it for you. Leaving +it attached means every later failure gets reprocessed by it instead of landing +in the DLQ where you can see it. + +`$DLQ-quarantine` outlives the incident as well. `queues create` fails if it +already exists, so on the next incident either reuse it (confirm it is empty +first — anything left in it is unprocessed feedback) or `wrangler queues delete` +it once you have dealt with whatever landed there. + +Note that DLQ messages carry the full feedback payload, so the DLQ is subject to +the same handling rules as the private `TrainLCD/Issues` repo. + +### Retry idempotency + +A retry re-runs `processFeedbackMessage()` from the top, so anything that throws +*after* the Issue has been created files the same feedback again — up to four +Issues with `max_retries: 3`, plus one more for every DLQ replay. + +To prevent that, the consumer keeps a per-report marker in `STATE_KV` under +`feedbackTriage:processed:` (30-day TTL, long enough to cover a DLQ +replay). It records the created Issue number and URL, the public stub URL, the +triage result, and whether the Discord notification went out. The marker decides +what each delivery still has to do: + +- **notified** — nothing. The message is acked and dropped. +- **Issue created, not notified** — skip triage and Issue creation, re-send the + Discord notification only. The stored triage result is reused instead of being + re-inferred, so the notification matches the Issue that was already filed, and + the retry costs no Workers AI neurons. +- **no marker** — the full path, writing the marker as soon as the Issue exists. + +Nothing between the Issue being created and the marker being written may throw, +because a throw there is a retry with no marker to stop it. So a malformed +Issue-creation response and a failed marker write are logged and swallowed, and +`notifyDiscord()` turns every failure into a return value instead of an +exception — including `fetch()` itself rejecting on a network or DNS error, +which is what made this reachable in practice. + +Past that point a throw is safe, and one is deliberate. The marker records +whether Discord actually accepted the request, so a failed notification is saved +as `notified: false` and *then* rethrown as `FeedbackNotifyError`, which retries +the message: the retry reads the marker, skips straight to the notification, and +leaves the Issue alone. Retrying the handler for a Discord outage is exactly +what used to duplicate Issues — the marker is what makes it safe now. A +notification that never succeeds ends up in the DLQ after `max_retries`, which +is how a broken webhook becomes visible. + +The one case that is *not* retried is a notification failure where the marker +write also failed. Without the marker a retry would file the Issue again, so the +notification is given up and the message acked — the feedback is on GitHub +either way. + +**KV is not a lock, and the marker read is what makes this work — so the retry +has to be slow enough for the read to see it.** KV caches the *absence* of a key +at the edge for the read's `cacheTtl` (60 s by default), so a retry that runs +immediately after the failure can miss a marker that was written seconds ago and +file the Issue again. The consumer therefore retries with +`message.retry({ delaySeconds: FEEDBACK_RETRY_DELAY_SECONDS })` (90 s) so the +negative cache has expired by the time the marker is read. Changing that +constant without understanding this is how the duplicate comes back. + +The same limit applies to the writes: KV accepts at most one write per second to +a given key, and one report writes that key twice — once when the Issue exists, +once when the notification result is known. A notification that completes in +under a second would make the second write a 429, so the consumer spaces writes +to the same key ~1.1 s apart (and waits that long before its one write retry) +rather than losing the notification state and re-notifying on a replay. + +That covers the sequential retries of one message. It does **not** serialize two +deliveries of the same report racing each other — Cloudflare Queues is +at-least-once, so that race is possible in principle, and with an eventually +consistent read there is nothing to make it safe. Strict de-duplication would +take a per-report claim in a Durable Object (the only strongly consistent option +here), which is a bigger change than the failure it covers. + +One gap stays open by design: if the Issue-creation `fetch()` fails *after* +GitHub has already created the Issue, no marker was written and the retry files +a second one. Closing that would mean searching `TrainLCD/Issues` by ticket ID +before every creation, which costs a request per feedback for a case that needs +GitHub to drop the response of a request it accepted. + +## Public repo routing + +Feedback Issues are always created in the private `TrainLCD/Issues` repo with +the full report (original text, device info, reporter UID, stacktrace, image). + +On top of that, the worker opens a stub Issue in the matching public repo — but +only when **every** condition in `resolvePublicIssueRepo()` holds: + +1. `reportType` is `feedback` (crash reports are never routed — their stacktraces + are not vetted for public disclosure), +2. triage succeeded (`triageFailed === false`) and the report is not spam, +3. the heuristic did not flag it for human review (`needsSpamReview !== true`), +4. `category` is one of `PUBLIC_ISSUE_CATEGORIES` — `bug`, `improvement`, + `feature_request` (so `question` and `praise` are excluded), and +5. `component` ≠ unknown with `componentConfidence` ≥ + `PUBLIC_ISSUE_MIN_CONFIDENCE` (0.7). + +The component then selects the repo: + +| `component` | repo | +| ------------- | -------------------- | +| `mobile_app` | `TrainLCD/MobileApp` | +| `station_api` | `TrainLCD/StationAPI`| +| `functions` | `TrainLCD/Functions` | +| `website` | `TrainLCD/Website` | + +Because those repos are public, the stub carries **no feedback content at all** — +no original text, no AI summary or title, no device info. It only links back to +the private ticket (`TrainLCD/Issues#` and the ticket ID), and the +private Issue gets a comment pointing at the public one so both sides are +traceable. + +`OCTOKIT_PAT` therefore needs write access to those four repos on top of +`TrainLCD/Issues`. + ## Maintenance CLI Maintenance tools that operate on KV (TTS_KV) and R2 (the audio bucket). Both @@ -142,13 +506,13 @@ the KV namespace and R2 bucket from `wrangler.jsonc`). ### `find-tts-cache` -Searches the TTS cache by SSML body and optionally deletes the matching KV +Searches the TTS cache by spoken text and optionally deletes the matching KV document and R2 audio. KV is read via `wrangler kv key list` / `wrangler kv bulk get` and deleted via `wrangler kv key delete`; R2 audio is removed via `wrangler r2 object delete`. ```bash -npm run find-tts-cache -- "東京" --field ssmlJa +npm run find-tts-cache -- "東京" --field textJa npm run find-tts-cache -- "東京" --delete npm run find-tts-cache -- "東京" --env production --delete ``` diff --git a/agent-rerank-eval.jsonl b/agent-rerank-eval.jsonl new file mode 100644 index 0000000..244af5d --- /dev/null +++ b/agent-rerank-eval.jsonl @@ -0,0 +1,20 @@ +{"id": "ocean-nationwide", "request": "海が見える駅に行きたい", "queries": ["海", "根府川", "稲毛海岸"], "expect": ["根府川", "稲毛海岸"], "reject": ["北海道医療大学"], "note": "「海」の部分一致で海と無関係な駅が混ざるプールから拾えるか。種差海岸・松島海岸は実際に海沿いなので reject にしない"} +{"id": "ocean-from-tokyo", "request": "海が見える駅に行きたい", "from": "東京", "queries": ["根府川", "早川", "真鶴", "海"], "expect": ["根府川", "早川", "真鶴"], "reject": [], "note": "現在駅つき。東海道線の海側3駅はシステムプロンプトが例示している"} +{"id": "ocean-english", "request": "I want to go to a station with an ocean view", "from": "東京", "queries": ["根府川", "真鶴", "海"], "expect": ["根府川", "真鶴"], "reject": [], "note": "英語の要望と日本語の候補。命令文が日本語でも判定できるか"} +{"id": "onsen-nationwide", "request": "温泉がある駅に行きたい", "queries": ["温泉", "鬼怒川温泉", "大手町"], "expect": [], "reject": ["大手町"], "note": "無関係な駅を1件混ぜて閾値が効くかを見る。他の温泉駅も「温泉がある駅」として妥当なので expect は置かない(実測で全て0.94以上)"} +{"id": "onsen-atami", "request": "日帰りで温泉に入れる駅はある?", "from": "東京", "queries": ["熱海", "温泉"], "expect": ["熱海"], "reject": [], "note": "東海道線で直通。要望が「日帰り」でも駅の妥当性は変わらない"} +{"id": "atami-by-name", "request": "熱海に行きたい", "queries": ["熱海", "来宮"], "expect": ["熱海"], "reject": ["来宮", "磐梯熱海"], "note": "駅名の直接指定。隣接駅と、名前に「熱海」を含む別地域の駅を拾ってはいけない"} +{"id": "shinjuku-exact", "request": "新宿駅に行きたい", "queries": ["新宿"], "expect": ["新宿"], "reject": ["西新宿", "新宿三丁目", "新宿御苑前"], "note": "同名前方一致の取り違え。現在駅なしだと limit:10 が全部「新宿」で埋まり識別テストにならないため東京を起点にする", "from": "東京"} +{"id": "shinjuku-gyoen", "request": "新宿御苑に行きたい", "queries": ["新宿御苑", "新宿"], "expect": ["新宿御苑前"], "reject": ["新宿"], "note": "shinjuku-exact の逆方向。最短一致に倒れていないか。現在駅なしでは新宿御苑前がプールに入らなかった", "from": "東京"} +{"id": "kamakura-kokomae", "request": "鎌倉高校前に行きたい", "queries": ["鎌倉"], "expect": ["鎌倉高校前"], "reject": ["鎌倉", "北鎌倉"], "note": "部分一致で親の駅名が上位に来る典型"} +{"id": "kamakura-kokomae-roman", "request": "Kamakura-kokomae", "queries": ["鎌倉"], "expect": ["鎌倉高校前"], "reject": ["鎌倉"], "note": "ローマ字の駅名指定。候補側は日本語表記"} +{"id": "inage-kaigan", "request": "稲毛海岸", "queries": ["稲毛"], "expect": ["稲毛海岸"], "reject": ["稲毛", "京成稲毛"], "note": "駅名だけの発話。要望の語をそのまま含む駅を選べるか"} +{"id": "kinugawa-partial", "request": "鬼怒川温泉に行きたい", "queries": ["鬼怒川"], "expect": ["鬼怒川温泉"], "reject": ["鬼怒川公園"], "note": "システムプロンプトが例示している引き直し(鬼怒川温泉→鬼怒川)のプール"} +{"id": "osaka-castle", "request": "お城が見たい", "queries": ["城", "大阪城公園"], "expect": ["大阪城公園"], "reject": ["磐城太田"], "note": "「城」の部分一致は旧国名の「磐城」など城と無関係な駅名を多く含む"} +{"id": "tateyama-boso", "request": "房総の海沿いに行きたい", "from": "東京", "queries": ["館山", "海"], "expect": ["館山"], "reject": ["熱海", "東海"], "note": "地域名つきの要望。内房線はシステムプロンプトの例示。房総でない海沿いの駅を弾けるか"} +{"id": "quiet-scenery", "request": "静かで景色のいい駅に行きたい", "from": "東京", "queries": ["根府川", "新宿", "東京"], "expect": ["根府川"], "reject": ["新宿"], "note": "主観的な要望。繁華街の駅を弾けるか"} +{"id": "ambiguous-short", "request": "どこか行きたい", "queries": ["新宿", "根府川"], "expect": [], "reject": [], "note": "条件が無い発話。expect も reject も置かず、確率が横並びになるかだけを見る"} +{"id": "no-match-ski", "request": "スキー場のある駅に行きたい", "queries": ["新宿", "大手町"], "expectEmpty": true, "note": "合成。プールを意図的に無関係にした。閾値で全部落ちるべき(温泉地はスキー場が近いものがあるため検索語から外した)"} +{"id": "no-match-zoo", "request": "動物園に行きたい", "queries": ["大手町", "稲毛"], "expectEmpty": true, "note": "合成。同上"} +{"id": "no-match-airport", "request": "空港に行きたい", "queries": ["根府川", "熱海"], "expectEmpty": true, "note": "合成。要望は実在するが、プールには一致する駅が無い"} +{"id": "no-match-unrelated-onsen", "request": "美術館のある駅に行きたい", "queries": ["温泉", "鬼怒川温泉"], "expectEmpty": true, "note": "合成。温泉駅だけのプールに美術館の要望を当てる"} diff --git a/fewshot.example.jsonl b/fewshot.example.jsonl index 4f1eac6..f4d0018 100644 --- a/fewshot.example.jsonl +++ b/fewshot.example.jsonl @@ -1,3 +1,7 @@ -{"input": "電車の音声案内が途中で止まってしまうことがあります。Androidの最新版です。", "output": "{\"title\": \"自動アナウンスが途中で停止する不具合\", \"summary\": \"オートモード中に自動音声が途中で再生されなくなる事象の報告。Android端末で発生。\", \"isSpam\": false, \"labels\": [\"bug\"], \"category\": \"bug\", \"triageLevel\": \"high\", \"confidence\": 0.8, \"reason\": \"特定機能(TTS)が使えない不具合\"}"} -{"input": "路線図にダークモードがほしいです。", "output": "{\"title\": \"路線図のダークモード対応要望\", \"summary\": \"路線図表示にダークモードを追加してほしいという新機能要望。\", \"isSpam\": false, \"labels\": [\"feature\", \"ui\"], \"category\": \"feature_request\", \"triageLevel\": \"low\", \"confidence\": 0.7, \"reason\": \"新規機能の要望\"}"} -{"input": "次は東京、東京です。お出口は左側です。ご利用ありがとうございます。", "output": "{\"title\": \"内容未分類(改善要望なし)\", \"summary\": \"\", \"isSpam\": true, \"labels\": [], \"category\": \"question\", \"triageLevel\": \"low\", \"confidence\": 0.9, \"reason\": \"車内放送の書き起こしでアクション不能\"}"} +{"input": "電車の音声案内が途中で止まってしまうことがあります。Androidの最新版です。", "output": "{\"title\": \"自動アナウンスが途中で停止する不具合\", \"summary\": \"オートモード中に自動音声が途中で再生されなくなる事象の報告。Android端末で発生。\", \"isSpam\": false, \"labels\": [\"bug\"], \"category\": \"bug\", \"triageLevel\": \"high\", \"component\": \"mobile_app\", \"componentConfidence\": 0.6, \"confidence\": 0.8, \"reason\": \"特定機能(TTS)が使えない不具合。再生側か合成側か切り分け不能\"}"} +{"input": "路線図にダークモードがほしいです。", "output": "{\"title\": \"路線図のダークモード対応要望\", \"summary\": \"路線図表示にダークモードを追加してほしいという新機能要望。\", \"isSpam\": false, \"labels\": [\"feature\", \"ui\"], \"category\": \"feature_request\", \"triageLevel\": \"low\", \"component\": \"mobile_app\", \"componentConfidence\": 0.9, \"confidence\": 0.7, \"reason\": \"アプリ画面の新規機能要望\"}"} +{"input": "〇〇線の△△駅が検索しても出てきません。駅名の表記も間違っています。", "output": "{\"title\": \"特定駅が検索に出ず駅名表記も誤り\", \"summary\": \"特定路線の駅が検索結果に表示されず、駅名の表記も誤っているという報告。\", \"isSpam\": false, \"labels\": [\"bug\", \"location\"], \"category\": \"bug\", \"triageLevel\": \"high\", \"component\": \"station_api\", \"componentConfidence\": 0.9, \"confidence\": 0.85, \"reason\": \"駅データそのものの欠落・誤りで原因はデータ提供側\"}"} +{"input": "アプリの中のAIに乗り換えを聞いても「エラーが発生しました」としか返ってきません。何度やり直しても同じです。", "output": "{\"title\": \"AIチャットが常にエラーを返し応答しない\", \"summary\": \"アプリ内のAIチャットに乗り換えを質問すると毎回エラーが返り、回答が得られない。再試行しても同様。\", \"isSpam\": false, \"labels\": [\"bug\", \"network\"], \"category\": \"bug\", \"triageLevel\": \"high\", \"component\": \"functions\", \"componentConfidence\": 0.8, \"confidence\": 0.8, \"reason\": \"AI応答APIが毎回エラーで原因はバックエンド側\"}", "weight": 2} +{"input": "公式サイトのプライバシーポリシーのリンクを踏むと404になります。ご確認ください。", "output": "{\"title\": \"公式サイトのプライバシーポリシーが404になる\", \"summary\": \"公式サイト上のプライバシーポリシーへのリンクが404を返し、内容を閲覧できない。\", \"isSpam\": false, \"labels\": [\"bug\"], \"category\": \"bug\", \"triageLevel\": \"medium\", \"component\": \"website\", \"componentConfidence\": 0.9, \"confidence\": 0.85, \"reason\": \"trainlcd.app 上のリンク切れでアプリ本体とは無関係\"}", "weight": 2} +{"input": "いつも使わせてもらってます。おかげで乗り過ごさなくなりました。本当にありがとうございます!", "output": "{\"title\": \"アプリへの感謝の声\", \"summary\": \"日常的に利用しており、乗り過ごしが減ったという感謝の声。対応すべき要望は含まれない。\", \"isSpam\": false, \"labels\": [], \"category\": \"praise\", \"triageLevel\": \"low\", \"component\": \"unknown\", \"componentConfidence\": 0, \"confidence\": 0.9, \"reason\": \"感謝のみで対応不要だが、利用者からの正当なフィードバック\"}", "weight": 2} +{"input": "次は東京、東京です。お出口は左側です。ご利用ありがとうございます。", "output": "{\"title\": \"内容未分類(改善要望なし)\", \"summary\": \"\", \"isSpam\": true, \"labels\": [], \"category\": \"question\", \"triageLevel\": \"low\", \"component\": \"unknown\", \"componentConfidence\": 0, \"confidence\": 0.9, \"reason\": \"車内放送の書き起こしでアクション不能\"}"} diff --git a/jest.config.js b/jest.config.js index 8c6b161..0bcdfdf 100644 --- a/jest.config.js +++ b/jest.config.js @@ -9,6 +9,7 @@ module.exports = { moduleNameMapper: { '^ai$': '/test/stubs/ai-sdk.ts', '^@ai-sdk/(anthropic|openai)$': '/test/stubs/ai-sdk-provider.ts', + '^@ai-sdk/google-vertex/edge$': '/test/stubs/ai-sdk-provider.ts', '^langsmith$': '/test/stubs/langsmith.ts', '^langsmith/experimental/vercel$': '/test/stubs/langsmith.ts', }, diff --git a/package-lock.json b/package-lock.json index 4a53bb4..c62f54b 100644 --- a/package-lock.json +++ b/package-lock.json @@ -7,6 +7,7 @@ "name": "functions", "dependencies": { "@ai-sdk/anthropic": "^4.0.23", + "@ai-sdk/google-vertex": "^5.0.54", "@ai-sdk/openai": "^4.0.23", "ai": "^7.0.41", "dayjs": "^1.11.9", @@ -30,13 +31,44 @@ } }, "node_modules/@ai-sdk/anthropic": { - "version": "4.0.23", - "resolved": "https://registry.npmjs.org/@ai-sdk/anthropic/-/anthropic-4.0.23.tgz", - "integrity": "sha512-9sky++sOcQ3V38XWqPpiLIe4knjWREOUEv3ZOZZ9mTQepb2Ho60NUFGR/0NxQlTHEW9G0zxciEecySNE/iHdVQ==", + "version": "4.0.39", + "resolved": "https://registry.npmjs.org/@ai-sdk/anthropic/-/anthropic-4.0.39.tgz", + "integrity": "sha512-JAMGtYeEuaBzqbsPO4fkho6vQyNoVhsHASM4o59wmJRU6Vh7prjOp490Kmc7YQTY+ioU1/xYzXvWOtxZBup0Xw==", "license": "Apache-2.0", "dependencies": { - "@ai-sdk/provider": "4.0.4", - "@ai-sdk/provider-utils": "5.0.14" + "@ai-sdk/provider": "4.0.7", + "@ai-sdk/provider-utils": "5.0.27" + }, + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "zod": "^3.25.76 || ^4.1.8" + } + }, + "node_modules/@ai-sdk/anthropic/node_modules/@ai-sdk/provider": { + "version": "4.0.7", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.7.tgz", + "integrity": "sha512-6or44XprPzKbr8zkmzosowSE0pxkvJcoojBL+mCZvPUt3kvXp3XSNqeVun9golb1acEfSo6yaEBRT18h2VU+1Q==", + "license": "Apache-2.0", + "dependencies": { + "json-schema": "^0.4.0" + }, + "engines": { + "node": ">=22" + } + }, + "node_modules/@ai-sdk/anthropic/node_modules/@ai-sdk/provider-utils": { + "version": "5.0.27", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.27.tgz", + "integrity": "sha512-EzAn4pdgG5g0xXtH6lE2zyNmfjDQIDjATkfqzuidEI35g++hh4+07vnjzkT/RmGmIClPZiRj/Q2GMPV2V7mkHw==", + "license": "Apache-2.0", + "dependencies": { + "@ai-sdk/provider": "4.0.7", + "@standard-schema/spec": "^1.1.0", + "@workflow/serde": "4.1.0", + "eventsource-parser": "^3.0.8", + "undici": "^7.28.0" }, "engines": { "node": ">=22" @@ -62,6 +94,104 @@ "zod": "^3.25.76 || ^4.1.8" } }, + "node_modules/@ai-sdk/google": { + "version": "4.0.44", + "resolved": "https://registry.npmjs.org/@ai-sdk/google/-/google-4.0.44.tgz", + "integrity": "sha512-bmRTDg06jQD+eX8nf214pET9+Oe8O1+lUIRGbWsGXj9IN2UJkpl1O1x7cvtiboyTtKSLvSRdVtItUfSl8sQ2GA==", + "license": "Apache-2.0", + "dependencies": { + "@ai-sdk/provider": "4.0.7", + "@ai-sdk/provider-utils": "5.0.27" + }, + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "zod": "^3.25.76 || ^4.1.8" + } + }, + "node_modules/@ai-sdk/google-vertex": { + "version": "5.0.54", + "resolved": "https://registry.npmjs.org/@ai-sdk/google-vertex/-/google-vertex-5.0.54.tgz", + "integrity": "sha512-vcOCAEnzMXjgNJzXiNWiIMXReMfwf0Sk+yIaBLBePgxhb0ep1Uv0mI7ekG3XrnqDvMAs7eVDlwER8gyPuhGTCA==", + "license": "Apache-2.0", + "dependencies": { + "@ai-sdk/anthropic": "4.0.39", + "@ai-sdk/google": "4.0.44", + "@ai-sdk/openai-compatible": "3.0.30", + "@ai-sdk/provider": "4.0.7", + "@ai-sdk/provider-utils": "5.0.27", + "google-auth-library": "^10.6.2" + }, + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "zod": "^3.25.76 || ^4.1.8" + } + }, + "node_modules/@ai-sdk/google-vertex/node_modules/@ai-sdk/provider": { + "version": "4.0.7", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.7.tgz", + "integrity": "sha512-6or44XprPzKbr8zkmzosowSE0pxkvJcoojBL+mCZvPUt3kvXp3XSNqeVun9golb1acEfSo6yaEBRT18h2VU+1Q==", + "license": "Apache-2.0", + "dependencies": { + "json-schema": "^0.4.0" + }, + "engines": { + "node": ">=22" + } + }, + "node_modules/@ai-sdk/google-vertex/node_modules/@ai-sdk/provider-utils": { + "version": "5.0.27", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.27.tgz", + "integrity": "sha512-EzAn4pdgG5g0xXtH6lE2zyNmfjDQIDjATkfqzuidEI35g++hh4+07vnjzkT/RmGmIClPZiRj/Q2GMPV2V7mkHw==", + "license": "Apache-2.0", + "dependencies": { + "@ai-sdk/provider": "4.0.7", + "@standard-schema/spec": "^1.1.0", + "@workflow/serde": "4.1.0", + "eventsource-parser": "^3.0.8", + "undici": "^7.28.0" + }, + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "zod": "^3.25.76 || ^4.1.8" + } + }, + "node_modules/@ai-sdk/google/node_modules/@ai-sdk/provider": { + "version": "4.0.7", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.7.tgz", + "integrity": "sha512-6or44XprPzKbr8zkmzosowSE0pxkvJcoojBL+mCZvPUt3kvXp3XSNqeVun9golb1acEfSo6yaEBRT18h2VU+1Q==", + "license": "Apache-2.0", + "dependencies": { + "json-schema": "^0.4.0" + }, + "engines": { + "node": ">=22" + } + }, + "node_modules/@ai-sdk/google/node_modules/@ai-sdk/provider-utils": { + "version": "5.0.27", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.27.tgz", + "integrity": "sha512-EzAn4pdgG5g0xXtH6lE2zyNmfjDQIDjATkfqzuidEI35g++hh4+07vnjzkT/RmGmIClPZiRj/Q2GMPV2V7mkHw==", + "license": "Apache-2.0", + "dependencies": { + "@ai-sdk/provider": "4.0.7", + "@standard-schema/spec": "^1.1.0", + "@workflow/serde": "4.1.0", + "eventsource-parser": "^3.0.8", + "undici": "^7.28.0" + }, + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "zod": "^3.25.76 || ^4.1.8" + } + }, "node_modules/@ai-sdk/openai": { "version": "4.0.23", "resolved": "https://registry.npmjs.org/@ai-sdk/openai/-/openai-4.0.23.tgz", @@ -78,6 +208,53 @@ "zod": "^3.25.76 || ^4.1.8" } }, + "node_modules/@ai-sdk/openai-compatible": { + "version": "3.0.30", + "resolved": "https://registry.npmjs.org/@ai-sdk/openai-compatible/-/openai-compatible-3.0.30.tgz", + "integrity": "sha512-BB35G4fS/Ey5OHbWrVLxRLX1gkTlO+9I4YhlmdH1skNVoPaFXZlR6bQ+1C76d/ug7O6ocbIw4qcd2G4+GPdWEA==", + "license": "Apache-2.0", + "dependencies": { + "@ai-sdk/provider": "4.0.7", + "@ai-sdk/provider-utils": "5.0.27" + }, + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "zod": "^3.25.76 || ^4.1.8" + } + }, + "node_modules/@ai-sdk/openai-compatible/node_modules/@ai-sdk/provider": { + "version": "4.0.7", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.7.tgz", + "integrity": "sha512-6or44XprPzKbr8zkmzosowSE0pxkvJcoojBL+mCZvPUt3kvXp3XSNqeVun9golb1acEfSo6yaEBRT18h2VU+1Q==", + "license": "Apache-2.0", + "dependencies": { + "json-schema": "^0.4.0" + }, + "engines": { + "node": ">=22" + } + }, + "node_modules/@ai-sdk/openai-compatible/node_modules/@ai-sdk/provider-utils": { + "version": "5.0.27", + "resolved": "https://registry.npmjs.org/@ai-sdk/provider-utils/-/provider-utils-5.0.27.tgz", + "integrity": "sha512-EzAn4pdgG5g0xXtH6lE2zyNmfjDQIDjATkfqzuidEI35g++hh4+07vnjzkT/RmGmIClPZiRj/Q2GMPV2V7mkHw==", + "license": "Apache-2.0", + "dependencies": { + "@ai-sdk/provider": "4.0.7", + "@standard-schema/spec": "^1.1.0", + "@workflow/serde": "4.1.0", + "eventsource-parser": "^3.0.8", + "undici": "^7.28.0" + }, + "engines": { + "node": ">=22" + }, + "peerDependencies": { + "zod": "^3.25.76 || ^4.1.8" + } + }, "node_modules/@ai-sdk/provider": { "version": "4.0.4", "resolved": "https://registry.npmjs.org/@ai-sdk/provider/-/provider-4.0.4.tgz", @@ -2503,6 +2680,15 @@ "integrity": "sha512-pav4F2BoirECWR7Nf1TKt+2eETcBj7jj4cBefQ8VXQCA6NPkaKeLfj/zMgi+3zYV5ZIBT4GuUiphsj0/b9hPQQ==", "license": "Apache-2.0" }, + "node_modules/agent-base": { + "version": "7.1.4", + "resolved": "https://registry.npmjs.org/agent-base/-/agent-base-7.1.4.tgz", + "integrity": "sha512-MnA+YT8fwfJPgBx3m60MNqakm30XOkyIoH1y6huTQvC0PwZG7ki8NacLBcrPbNoo8vEZy7Jpuk7+jMO+CUovTQ==", + "license": "MIT", + "engines": { + "node": ">= 14" + } + }, "node_modules/ai": { "version": "7.0.41", "resolved": "https://registry.npmjs.org/ai/-/ai-7.0.41.tgz", @@ -2724,6 +2910,35 @@ "dev": true, "license": "MIT" }, + "node_modules/base64-js": { + "version": "1.5.1", + "resolved": "https://registry.npmjs.org/base64-js/-/base64-js-1.5.1.tgz", + "integrity": "sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT" + }, + "node_modules/bignumber.js": { + "version": "9.3.1", + "resolved": "https://registry.npmjs.org/bignumber.js/-/bignumber.js-9.3.1.tgz", + "integrity": "sha512-Ko0uX15oIUS7wJ3Rb30Fs6SkVbLmPBAKdlm7q9+ak9bbIeFf0MwuBsQV6z7+X768/cHsfg+WlysDWJcmthjsjQ==", + "license": "MIT", + "engines": { + "node": "*" + } + }, "node_modules/blake3-wasm": { "version": "2.1.5", "resolved": "https://registry.npmjs.org/blake3-wasm/-/blake3-wasm-2.1.5.tgz", @@ -2810,6 +3025,12 @@ "node-int64": "^0.4.0" } }, + "node_modules/buffer-equal-constant-time": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/buffer-equal-constant-time/-/buffer-equal-constant-time-1.0.1.tgz", + "integrity": "sha512-zRpUiDwd/xk6ADqPMATG8vc9VPrkck7T07OIx0gnjmJAnHnTVXNQG3vfvWNuiZIkwu9KrKdA1iJKfsfTVxE6NA==", + "license": "BSD-3-Clause" + }, "node_modules/buffer-from": { "version": "1.1.2", "resolved": "https://registry.npmjs.org/buffer-from/-/buffer-from-1.1.2.tgz", @@ -3021,6 +3242,15 @@ "node": ">= 8" } }, + "node_modules/data-uri-to-buffer": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/data-uri-to-buffer/-/data-uri-to-buffer-4.0.1.tgz", + "integrity": "sha512-0R9ikRb668HB7QDxT1vkpuUBtqc53YyAwMwGeUFKRojY/NWKvdZ+9UYtRfGmhqNbRkTSVpMbmyhXipFFv2cb/A==", + "license": "MIT", + "engines": { + "node": ">= 12" + } + }, "node_modules/dayjs": { "version": "1.11.9", "resolved": "https://registry.npmjs.org/dayjs/-/dayjs-1.11.9.tgz", @@ -3030,7 +3260,6 @@ "version": "4.3.4", "resolved": "https://registry.npmjs.org/debug/-/debug-4.3.4.tgz", "integrity": "sha512-PRWFHuSU3eDtQJPvnNY7Jcket1j0t5OuOsFzPPzsekD52Zl8qUfFIPEiswXqIvHWGVHOgX+7G/vCNNhehwxfkQ==", - "dev": true, "dependencies": { "ms": "2.1.2" }, @@ -3098,6 +3327,15 @@ "node": "^14.15.0 || ^16.10.0 || >=18.0.0" } }, + "node_modules/ecdsa-sig-formatter": { + "version": "1.0.11", + "resolved": "https://registry.npmjs.org/ecdsa-sig-formatter/-/ecdsa-sig-formatter-1.0.11.tgz", + "integrity": "sha512-nagl3RYrbNv6kQkeJIpt6NJZy8twLB/2vtz6yN9Z4vRKHN4/QZJIEbqohALSgwKdnksuY3k5Addp5lg8sVoVcQ==", + "license": "Apache-2.0", + "dependencies": { + "safe-buffer": "^5.0.1" + } + }, "node_modules/ejs": { "version": "3.1.10", "resolved": "https://registry.npmjs.org/ejs/-/ejs-3.1.10.tgz", @@ -3291,6 +3529,12 @@ "node": "^14.15.0 || ^16.10.0 || >=18.0.0" } }, + "node_modules/extend": { + "version": "3.0.2", + "resolved": "https://registry.npmjs.org/extend/-/extend-3.0.2.tgz", + "integrity": "sha512-fjquC59cD7CyW6urNXK0FBufkZcoiGG80wTuPujX590cB5Ttln20E2UB4S/WARVqhXffZl2LNgS+gQdPIIim/g==", + "license": "MIT" + }, "node_modules/fast-json-stable-stringify": { "version": "2.1.0", "resolved": "https://registry.npmjs.org/fast-json-stable-stringify/-/fast-json-stable-stringify-2.1.0.tgz", @@ -3308,6 +3552,29 @@ "bser": "2.1.1" } }, + "node_modules/fetch-blob": { + "version": "3.2.0", + "resolved": "https://registry.npmjs.org/fetch-blob/-/fetch-blob-3.2.0.tgz", + "integrity": "sha512-7yAQpD2UMJzLi1Dqv7qFYnPbaPx7ZfFK6PiIxQ4PfkGPyNyl2Ugx+a/umUonmKqjhM4DnfbMvdX6otXq83soQQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/jimmywarting" + }, + { + "type": "paypal", + "url": "https://paypal.me/jimmywarting" + } + ], + "license": "MIT", + "dependencies": { + "node-domexception": "^1.0.0", + "web-streams-polyfill": "^3.0.3" + }, + "engines": { + "node": "^12.20 || >= 14.13" + } + }, "node_modules/filelist": { "version": "1.0.4", "resolved": "https://registry.npmjs.org/filelist/-/filelist-1.0.4.tgz", @@ -3367,6 +3634,18 @@ "node": ">=8" } }, + "node_modules/formdata-polyfill": { + "version": "4.0.10", + "resolved": "https://registry.npmjs.org/formdata-polyfill/-/formdata-polyfill-4.0.10.tgz", + "integrity": "sha512-buewHzMvYL29jdeQTVILecSaZKnt/RJWjoZCF5OW60Z67/GmSLBkOFM7qh1PI3zFNtJbaZL5eQu1vLfazOwj4g==", + "license": "MIT", + "dependencies": { + "fetch-blob": "^3.1.2" + }, + "engines": { + "node": ">=12.20.0" + } + }, "node_modules/fs.realpath": { "version": "1.0.0", "resolved": "https://registry.npmjs.org/fs.realpath/-/fs.realpath-1.0.0.tgz", @@ -3398,6 +3677,34 @@ "url": "https://github.com/sponsors/ljharb" } }, + "node_modules/gaxios": { + "version": "7.3.1", + "resolved": "https://registry.npmjs.org/gaxios/-/gaxios-7.3.1.tgz", + "integrity": "sha512-kB3rzJV7d9juLZh8/56QTXCwQfxyhdOMdyYk1HdQKFtF8TJTDTZQJtixWIwXdE9Jji91mC41DUNpjleo4L4eAQ==", + "license": "Apache-2.0", + "dependencies": { + "extend": "^3.0.2", + "https-proxy-agent": "^7.0.1", + "node-fetch": "^3.3.2" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/gcp-metadata": { + "version": "8.1.2", + "resolved": "https://registry.npmjs.org/gcp-metadata/-/gcp-metadata-8.1.2.tgz", + "integrity": "sha512-zV/5HKTfCeKWnxG0Dmrw51hEWFGfcF2xiXqcA3+J90WDuP0SvoiSO5ORvcBsifmx/FoIjgQN3oNOGaQ5PhLFkg==", + "license": "Apache-2.0", + "dependencies": { + "gaxios": "^7.0.0", + "google-logging-utils": "^1.0.0", + "json-bigint": "^1.0.0" + }, + "engines": { + "node": ">=18" + } + }, "node_modules/gensync": { "version": "1.0.0-beta.2", "resolved": "https://registry.npmjs.org/gensync/-/gensync-1.0.0-beta.2.tgz", @@ -3472,6 +3779,32 @@ "node": ">=4" } }, + "node_modules/google-auth-library": { + "version": "10.9.1", + "resolved": "https://registry.npmjs.org/google-auth-library/-/google-auth-library-10.9.1.tgz", + "integrity": "sha512-i1ydyHrqcIxXkWh/uBmVkzCvIuq5yiK2ATndIe5XxKholrG/MTYP9xGYka4sQhrbIAgGjL2B6NOE7rFaiF3fXw==", + "license": "Apache-2.0", + "dependencies": { + "base64-js": "^1.3.0", + "ecdsa-sig-formatter": "^1.0.11", + "gaxios": "^7.1.4", + "gcp-metadata": "8.1.2", + "google-logging-utils": "1.1.3", + "jws": "^4.0.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/google-logging-utils": { + "version": "1.1.3", + "resolved": "https://registry.npmjs.org/google-logging-utils/-/google-logging-utils-1.1.3.tgz", + "integrity": "sha512-eAmLkjDjAFCVXg7A1unxHsLf961m6y17QFqXqAXGj/gVkKFrEICfStRfwUlGNfeCEjNRa32JEWOUTlYXPyyKvA==", + "license": "Apache-2.0", + "engines": { + "node": ">=14" + } + }, "node_modules/graceful-fs": { "version": "4.2.11", "resolved": "https://registry.npmjs.org/graceful-fs/-/graceful-fs-4.2.11.tgz", @@ -3506,6 +3839,19 @@ "dev": true, "license": "MIT" }, + "node_modules/https-proxy-agent": { + "version": "7.0.6", + "resolved": "https://registry.npmjs.org/https-proxy-agent/-/https-proxy-agent-7.0.6.tgz", + "integrity": "sha512-vK9P5/iUfdl95AI+JVyUuIcVtd4ofvtrOr3HNtM2yxC9bnMbEdp3x01OhQNnjb8IJYi38VlTE3mBXwcfvywuSw==", + "license": "MIT", + "dependencies": { + "agent-base": "^7.1.2", + "debug": "4" + }, + "engines": { + "node": ">= 14" + } + }, "node_modules/human-signals": { "version": "2.1.0", "resolved": "https://registry.npmjs.org/human-signals/-/human-signals-2.1.0.tgz", @@ -4340,6 +4686,15 @@ "node": ">=6" } }, + "node_modules/json-bigint": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/json-bigint/-/json-bigint-1.0.0.tgz", + "integrity": "sha512-SiPv/8VpZuWbvLSMtTDU8hEfrZWg/mH/nV/b4o0CYbSxu1UIQPLdwKOCIyLQX+VIPO5vrLX3i8qtqFyhdPSUSQ==", + "license": "MIT", + "dependencies": { + "bignumber.js": "^9.0.0" + } + }, "node_modules/json-parse-even-better-errors": { "version": "2.3.1", "resolved": "https://registry.npmjs.org/json-parse-even-better-errors/-/json-parse-even-better-errors-2.3.1.tgz", @@ -4372,6 +4727,27 @@ "integrity": "sha512-HUgH65KyejrUFPvHFPbqOY0rsFip3Bo5wb4ngvdi1EpCYWUQDC5V+Y7mZws+DLkr4M//zQJoanu1SP+87Dv1oQ==", "license": "MIT" }, + "node_modules/jwa": { + "version": "2.0.1", + "resolved": "https://registry.npmjs.org/jwa/-/jwa-2.0.1.tgz", + "integrity": "sha512-hRF04fqJIP8Abbkq5NKGN0Bbr3JxlQ+qhZufXVr0DvujKy93ZCbXZMHDL4EOtodSbCWxOqR8MS1tXA5hwqCXDg==", + "license": "MIT", + "dependencies": { + "buffer-equal-constant-time": "^1.0.1", + "ecdsa-sig-formatter": "1.0.11", + "safe-buffer": "^5.0.1" + } + }, + "node_modules/jws": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/jws/-/jws-4.0.1.tgz", + "integrity": "sha512-EKI/M/yqPncGUUh44xz0PxSidXFr/+r0pA70+gIYhjv+et7yxM+s29Y+VGDkovRofQem0fs7Uvf4+YmAdyRduA==", + "license": "MIT", + "dependencies": { + "jwa": "^2.0.1", + "safe-buffer": "^5.0.1" + } + }, "node_modules/kleur": { "version": "3.0.3", "resolved": "https://registry.npmjs.org/kleur/-/kleur-3.0.3.tgz", @@ -4552,8 +4928,7 @@ "node_modules/ms": { "version": "2.1.2", "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.2.tgz", - "integrity": "sha512-sGkPx+VjMtmA6MX27oA4FBFELFCZZ4S4XqeGOXCv68tT+jb3vk/RyaKWP0PTKyWtmLSM0b+adUTEvbs1PEaH2w==", - "dev": true + "integrity": "sha512-sGkPx+VjMtmA6MX27oA4FBFELFCZZ4S4XqeGOXCv68tT+jb3vk/RyaKWP0PTKyWtmLSM0b+adUTEvbs1PEaH2w==" }, "node_modules/natural-compare": { "version": "1.4.0", @@ -4562,6 +4937,44 @@ "dev": true, "license": "MIT" }, + "node_modules/node-domexception": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/node-domexception/-/node-domexception-1.0.0.tgz", + "integrity": "sha512-/jKZoMpw0F8GRwl4/eLROPA3cfcXtLApP0QzLmUT/HuPCZWyB7IY9ZrMeKw2O/nFIqPQB3PVM9aYm0F312AXDQ==", + "deprecated": "Use your platform's native DOMException instead", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/jimmywarting" + }, + { + "type": "github", + "url": "https://paypal.me/jimmywarting" + } + ], + "license": "MIT", + "engines": { + "node": ">=10.5.0" + } + }, + "node_modules/node-fetch": { + "version": "3.3.2", + "resolved": "https://registry.npmjs.org/node-fetch/-/node-fetch-3.3.2.tgz", + "integrity": "sha512-dRB78srN/l6gqWulah9SrxeYnxeddIG30+GOqK/9OlLVyLg3HPnr6SqOWTWOXKRwC2eGYCkZ59NNuSgvSrpgOA==", + "license": "MIT", + "dependencies": { + "data-uri-to-buffer": "^4.0.0", + "fetch-blob": "^3.1.4", + "formdata-polyfill": "^4.0.10" + }, + "engines": { + "node": "^12.20.0 || ^14.13.1 || >=16.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/node-fetch" + } + }, "node_modules/node-int64": { "version": "0.4.0", "resolved": "https://registry.npmjs.org/node-int64/-/node-int64-0.4.0.tgz", @@ -4948,6 +5361,26 @@ "node": ">=10" } }, + "node_modules/safe-buffer": { + "version": "5.2.1", + "resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz", + "integrity": "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT" + }, "node_modules/semver": { "version": "7.8.4", "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.4.tgz", @@ -5350,7 +5783,6 @@ "version": "7.28.0", "resolved": "https://registry.npmjs.org/undici/-/undici-7.28.0.tgz", "integrity": "sha512-cRZYrTDwWznlnRiPjggAGxZXanty6M8RV1ff8Wm4LWXBp7/IG8v5DnOm74DtUBp9OONpK75YlPnIjQqX0dBDtA==", - "dev": true, "license": "MIT", "engines": { "node": ">=20.18.1" @@ -5429,6 +5861,15 @@ "makeerror": "1.0.12" } }, + "node_modules/web-streams-polyfill": { + "version": "3.3.3", + "resolved": "https://registry.npmjs.org/web-streams-polyfill/-/web-streams-polyfill-3.3.3.tgz", + "integrity": "sha512-d2JWLCivmZYTSIoge9MsgFCZrt571BikcWGYkjC1khllbTeDlGqZ2D8vD8E/lJa8WGWbb7Plm8/XJYV7IJHZZw==", + "license": "MIT", + "engines": { + "node": ">= 8" + } + }, "node_modules/which": { "version": "2.0.2", "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz", diff --git a/package.json b/package.json index 7b0e9fa..978f854 100644 --- a/package.json +++ b/package.json @@ -12,13 +12,16 @@ "format": "biome format . --write", "cf-typegen": "wrangler types", "find-tts-cache": "tsx src/cli/find-tts-cache.ts", - "find-orphaned-tts": "tsx src/cli/find-orphaned-tts.ts" + "find-orphaned-tts": "tsx src/cli/find-orphaned-tts.ts", + "typesafe-spike": "tsx src/cli/typesafe-triage-spike.ts", + "typesafe-rerank-spike": "tsx src/cli/typesafe-rerank-spike.ts" }, "engines": { "node": ">=22" }, "dependencies": { "@ai-sdk/anthropic": "^4.0.23", + "@ai-sdk/google-vertex": "^5.0.54", "@ai-sdk/openai": "^4.0.23", "ai": "^7.0.41", "dayjs": "^1.11.9", diff --git a/scripts/build-secrets-json.mjs b/scripts/build-secrets-json.mjs index b8f1573..ca1d211 100644 --- a/scripts/build-secrets-json.mjs +++ b/scripts/build-secrets-json.mjs @@ -8,7 +8,6 @@ import { readFileSync, writeFileSync } from 'node:fs'; const SECRET_NAMES = [ 'SESSION_JWT_SECRET', - 'AZURE_SPEECH_KEY', 'GOOGLE_PLAY_SA_KEY', 'APPSTORE_CONNECT_KEY', 'OCTOKIT_PAT', @@ -18,7 +17,17 @@ const SECRET_NAMES = [ // AI エージェント(/agent/chat) 'ANTHROPIC_API_KEY', 'OPENAI_API_KEY', + 'GOOGLE_VERTEX_SA_KEY', 'LANGSMITH_API_KEY', + // TTS(/tts) + 'GOOGLE_TTS_SA_KEY', +]; + +/** _FILE でファイル内容を投入できるシークレット(サービスアカウント鍵 JSON) */ +const FILE_BACKED_SECRETS = [ + 'GOOGLE_PLAY_SA_KEY', + 'GOOGLE_VERTEX_SA_KEY', + 'GOOGLE_TTS_SA_KEY', ]; const [, , secretsFile, outFile] = process.argv; @@ -54,23 +63,37 @@ try { if (e?.code !== 'ENOENT') throw e; } -// GOOGLE_PLAY_SA_KEY はファイル指定があれば中身をそのまま採用。 -// 指定は 環境変数 GOOGLE_PLAY_SA_KEY_FILE でも .secrets.env の GOOGLE_PLAY_SA_KEY_FILE 行でも可。 -const saFile = - process.env.GOOGLE_PLAY_SA_KEY_FILE ?? values.GOOGLE_PLAY_SA_KEY_FILE; -if (saFile) { - try { - values.GOOGLE_PLAY_SA_KEY = readFileSync(saFile, 'utf8'); - } catch (e) { - process.stderr.write( - `GOOGLE_PLAY_SA_KEY_FILE を読めません: ${saFile} (${e.message})\n` + - '(パスは functions/ からの相対、または絶対パスで指定してください)\n' - ); - process.exit(2); +// 鍵 JSON はファイル指定があれば中身をそのまま採用。 +// 指定は 環境変数 _FILE でも .secrets.env の _FILE 行でも可。 +for (const name of FILE_BACKED_SECRETS) { + const fileKey = `${name}_FILE`; + const saFile = process.env[fileKey] ?? values[fileKey]; + if (saFile) { + let contents; + try { + contents = readFileSync(saFile, 'utf8'); + } catch (e) { + process.stderr.write( + `${fileKey} を読めません: ${saFile} (${e.message})\n` + + '(パスは functions/ からの相対、または絶対パスで指定してください)\n' + ); + process.exit(2); + } + // 空ファイルを黙って捨てると「投入対象のシークレットがありません」としか出ず + // 原因を追いにくい。失敗した `gcloud ... keys create` は出力先を空のまま残すため、 + // 鍵ファイルが空になる事故は実際に起こる + if (contents.trim().length === 0) { + process.stderr.write( + `${fileKey} が空です: ${saFile}\n` + + '(鍵の発行に失敗していないか確認してください)\n' + ); + process.exit(2); + } + values[name] = contents; } + // 補助キーは secret として出力しない + delete values[fileKey]; } -// 補助キーは secret として出力しない -delete values.GOOGLE_PLAY_SA_KEY_FILE; const out = {}; for (const name of SECRET_NAMES) { diff --git a/scripts/put-secrets.sh b/scripts/put-secrets.sh old mode 100644 new mode 100755 index 8034b14..98456e2 --- a/scripts/put-secrets.sh +++ b/scripts/put-secrets.sh @@ -2,8 +2,9 @@ # # Worker のシークレットを `wrangler secret bulk` で一括投入する。 # 値は KEY=VALUE 形式のファイル(既定: functions/.secrets.env, gitignore 済み)から読む。 -# GOOGLE_PLAY_SA_KEY は環境変数 GOOGLE_PLAY_SA_KEY_FILE にサービスアカウント鍵 JSON の -# パスを渡せば、その中身をそのまま投入する。 +# サービスアカウント鍵(GOOGLE_PLAY_SA_KEY / GOOGLE_VERTEX_SA_KEY / +# GOOGLE_TTS_SA_KEY)は環境変数 +# _FILE に鍵 JSON のパスを渡せば、その中身をそのまま投入する。 # # stdin パイプ(`echo ... | wrangler secret put`)は Windows/Git Bash で値が # 届かないことがあるため、Node で一時 JSON を生成して `secret bulk` に渡す方式にしている。 @@ -13,6 +14,8 @@ # ./scripts/put-secrets.sh --env production # production へ # SECRETS_FILE=.secrets.prod.env ./scripts/put-secrets.sh --env production # GOOGLE_PLAY_SA_KEY_FILE=./sa.json ./scripts/put-secrets.sh +# GOOGLE_VERTEX_SA_KEY_FILE=./secrets-vertex-sa.json ./scripts/put-secrets.sh +# GOOGLE_TTS_SA_KEY_FILE=./secrets-tts-sa.json ./scripts/put-secrets.sh # set -euo pipefail @@ -47,9 +50,9 @@ TMP_JSON=".secrets.bulk.$$.json" cleanup() { rm -f "$TMP_JSON"; } trap cleanup EXIT -# .secrets.env + GOOGLE_PLAY_SA_KEY_FILE から bulk 用 JSON を生成(エスケープは Node 任せ) +# .secrets.env + _FILE から bulk 用 JSON を生成(エスケープは Node 任せ) if ! node scripts/build-secrets-json.mjs "$SECRETS_FILE" "$TMP_JSON"; then - echo "投入対象のシークレットがありません($SECRETS_FILE / GOOGLE_PLAY_SA_KEY_FILE を確認)" >&2 + echo "投入対象のシークレットがありません($SECRETS_FILE / _FILE を確認)" >&2 exit 1 fi diff --git a/src/agent/handler.test.ts b/src/agent/handler.test.ts index 585b426..8ba1fb1 100644 --- a/src/agent/handler.test.ts +++ b/src/agent/handler.test.ts @@ -354,6 +354,29 @@ describe('runAgentTurn', () => { expect(streamText.mock.calls[1][0].providerOptions).toEqual({ anthropic: { thinking: { type: 'disabled' } }, }); + // reasoning 設定は Gemini 専用(他社モデルでは付けない) + expect(streamText.mock.calls[0][0].reasoning).toBeUndefined(); + expect(streamText.mock.calls[1][0].reasoning).toBeUndefined(); + }); + + it('Gemini の思考は AI SDK 共通の reasoning 設定で抑制する', async () => { + const streamText: AnyFn = jest.fn(async () => + streamResult({ output: { reply: 'ok', suggestions: [] } }) + ); + await runAgentTurn({ + ...baseParams, + model: 'gemini-3.8-flash' as AnyFn, + streamText, + searchStations: jest.fn(), + }); + + const options = streamText.mock.calls[0][0]; + // 3 系が受理する最小値(thinkingLevel: low 相当) + expect(options.reasoning).toBe('low'); + // thinkingLevel / thinkingBudget の出し分けは SDK に任せ、自前では組まない + expect(options.providerOptions).toEqual({ + anthropic: { thinking: { type: 'disabled' } }, + }); }); }); @@ -649,3 +672,189 @@ describe('handleAgentChatStream', () => { expect((classifyTopic as AnyFn).mock.calls[0][2].aborted).toBe(true); }); }); + +/** + * リランクの組み込み。判定そのものは rerank.test.ts が見るので、ここでは + * 「注入されるか / 何回走るか / 失敗したときに今の挙動へ戻るか」だけを見る。 + */ +describe('提案駅のリランクの組み込み', () => { + /** 検索を 1 回済ませた時点で prepareStep に渡されるメッセージ列(本番と同じ形) */ + const messagesAfterSearch: AnyFn[] = [ + { role: 'user', content: '海が見える駅に行きたい' }, + { + role: 'assistant', + content: [ + { + type: 'tool-call', + toolCallId: 'call-1', + toolName: 'search_stations_by_name', + input: { name: '熱海' }, + }, + ], + }, + { + role: 'tool', + content: [ + { + type: 'tool-result', + toolCallId: 'call-1', + toolName: 'search_stations_by_name', + output: { + type: 'json', + value: { stations: [{ id: 1, name: '熱海' }] }, + }, + }, + ], + }, + ]; + + /** + * ツールを 1 回実行して verified を埋めたあと prepareStep を呼び、 + * その結果のメッセージ列(上書きしなければ undefined)を返す。 + */ + const preparedMessages = async ( + rerank: AnyFn | undefined, + stepNumber = 1 + ): Promise => { + let injected: AnyFn[] | undefined; + const streamText: AnyFn = jest.fn(async (options: AnyFn) => { + await options.tools.search_stations_by_name.execute({ name: '熱海' }, {}); + const prepared = await options.prepareStep({ + stepNumber, + messages: messagesAfterSearch, + }); + injected = prepared?.messages; + return streamResult({ output: { reply: 'ok', suggestions: [] } }); + }); + await runAgentTurn({ + ...baseParams, + streamText, + searchStations: jest + .fn() + .mockResolvedValue([station(1, '熱海'), station(2, '来宮')]), + currentStationName: '東京', + rerank: rerank as AnyFn, + }); + return injected; + }; + + const injectedMessage = async ( + rerank: AnyFn | undefined, + stepNumber = 1 + ): Promise => + (await preparedMessages(rerank, stepNumber))?.at(-1); + + const injectedNote = async ( + rerank: AnyFn | undefined, + stepNumber = 1 + ): Promise => + (await injectedMessage(rerank, stepNumber))?.content; + + it('無効(rerank 未指定)なら何も注入しない', async () => { + await expect(injectedNote(undefined)).resolves.toBeUndefined(); + }); + + it('選ばれた駅を順序付きで注入する', async () => { + const note = await injectedNote( + jest.fn().mockResolvedValue([station(1, '熱海')]) + ); + expect(note).toContain('提案してよい駅'); + expect(note).toContain('1. 熱海'); + expect(note).not.toContain('来宮'); + }); + + // Gemini(Vertex)は会話の途中の system メッセージを受け付けず、リクエスト + // 組み立ての時点で UnsupportedFunctionalityError になる。注入は user で行う + it('注入は user ロールで行う', async () => { + const message = await injectedMessage( + jest.fn().mockResolvedValue([station(1, '熱海')]) + ); + expect(message?.role).toBe('user'); + }); + + // ツール結果より前に置くと、モデルは提案集合を知る前に本文を書き始められる。 + // 既存のメッセージ列を保ったまま末尾に足すことも併せて見る + it('既存のメッセージ列の末尾(ツール結果の後ろ)へ注入する', async () => { + const messages = await preparedMessages( + jest.fn().mockResolvedValue([station(1, '熱海')]) + ); + expect(messages?.slice(0, -1)).toEqual(messagesAfterSearch); + expect(messages?.at(-2)?.role).toBe('tool'); + expect(messages?.at(-1)?.content).toContain('提案してよい駅'); + }); + + it('要望に合う駅が無ければ、空配列にして正直に伝えるよう注入する', async () => { + const note = await injectedNote(jest.fn().mockResolvedValue([])); + expect(note).toContain('見つからなかった'); + expect(note).toContain('空配列'); + }); + + // 判定できなかったときに注入すると、モデルは提案を全部失う。今の挙動 + // (モデルが自分で選ぶ)へ戻すのがフォールバック + it('判定できなかった(null)ときは何も注入しない', async () => { + await expect( + injectedNote(jest.fn().mockResolvedValue(null)) + ).resolves.toBeUndefined(); + }); + + it('判定材料として直近のユーザ発話と現在駅を渡す', async () => { + const rerank = jest.fn().mockResolvedValue([station(1, '熱海')]); + await injectedNote(rerank); + expect(rerank).toHaveBeenCalledWith( + { request: '海が見える駅に行きたい', currentStationName: '東京' }, + [station(1, '熱海'), station(2, '来宮')] + ); + }); + + // prepareStep はツール実行のたびに走る。都度判定すると最大 3 回ぶんの往復が + // 最初の delta までのレイテンシに積み上がる + it('prepareStep が何度呼ばれてもターンに 1 回だけ判定する', async () => { + const rerank = jest.fn().mockResolvedValue([station(1, '熱海')]); + const streamText: AnyFn = jest.fn(async (options: AnyFn) => { + await options.tools.search_stations_by_name.execute({ name: '熱海' }, {}); + for (const stepNumber of [1, 2, 3]) { + await options.prepareStep({ stepNumber, messages: [] }); + } + return streamResult({ output: { reply: 'ok', suggestions: [] } }); + }); + await runAgentTurn({ + ...baseParams, + streamText, + searchStations: jest.fn().mockResolvedValue([station(1, '熱海')]), + rerank: rerank as AnyFn, + }); + expect(rerank).toHaveBeenCalledTimes(1); + }); + + it('ツール結果が無ければ判定しない(使い方の質問など)', async () => { + const rerank = jest.fn(); + const streamText: AnyFn = jest.fn(async (options: AnyFn) => { + await options.prepareStep({ stepNumber: 1, messages: [] }); + return streamResult({ output: { reply: 'ok', suggestions: [] } }); + }); + await runAgentTurn({ + ...baseParams, + streamText, + searchStations: jest.fn(), + rerank: rerank as AnyFn, + }); + expect(rerank).not.toHaveBeenCalled(); + }); + + it('イテレーション上限では注入の有無に関わらずツールを外す', async () => { + let prepared: AnyFn; + const streamText: AnyFn = jest.fn(async (options: AnyFn) => { + await options.tools.search_stations_by_name.execute({ name: '熱海' }, {}); + prepared = await options.prepareStep({ stepNumber: 3, messages: [] }); + return streamResult({ output: { reply: 'ok', suggestions: [] } }); + }); + await runAgentTurn({ + ...baseParams, + streamText, + searchStations: jest.fn().mockResolvedValue([station(1, '熱海')]), + rerank: jest.fn().mockResolvedValue([station(1, '熱海')]) as AnyFn, + }); + expect(prepared.activeTools).toEqual([]); + expect(prepared.messages.at(-1).content).toContain('提案してよい駅'); + }); +}); diff --git a/src/agent/handler.ts b/src/agent/handler.ts index ec13d2e..084c702 100644 --- a/src/agent/handler.ts +++ b/src/agent/handler.ts @@ -20,8 +20,18 @@ import { } from '../lib/callable'; import type { Env } from '../types'; import { classifyTopic } from './gate'; -import { resolveAgentModel, resolveOpenAIReasoningOptions } from './llm'; +import { + resolveAgentModel, + resolveGoogleReasoningSetting, + resolveOpenAIReasoningOptions, +} from './llm'; import { buildContextMessage, buildSystemPrompt, loadAgentFaq } from './prompt'; +import { + buildRerankNote, + createRerankSelector, + type RerankSelector, + resolveRerankThreshold, +} from './rerank'; import { type AgentChatResult, type AgentOutput, @@ -209,7 +219,9 @@ type TurnPhase = /** 並列前段のうち FAQ + 現在駅の解決の完了まで */ | 'context' /** 本体 LLM 開始から最初の delta まで */ - | 'firstDelta'; + | 'firstDelta' + /** 提案駅のリランク判定(ツール結果が出てから 1 回だけ。無効時は出ない) */ + | 'rerank'; type TurnOutcome = 'done' | 'refused' | 'error'; @@ -291,8 +303,48 @@ export interface AgentTurnParams { onDelta?: (text: string) => void | Promise; /** ツール実行の開始(SSE の tool)。入力内容は会話本文相当のため渡さない */ onToolStart?: () => void | Promise; + /** 現在駅の駅名。リランクの判定材料(「近く」などの相対表現の解釈)に使う */ + currentStationName?: string | null; + /** 提案駅のリランク。無効なら undefined(注入も判定も行わない) */ + rerank?: RerankSelector; + /** リランクの所要時間を記録する(内容は持たない) */ + onRerankDone?: (elapsedMs: number) => void; } +/** + * リランクはターンに 1 回だけ実行する。prepareStep はツール実行のたびに走るため、 + * 都度判定すると最大 3 回ぶんの往復(実測 1 回 364ms)が最初の delta までの + * レイテンシに積み上がる。複数回検索するターンでは 2 回目以降に増えた候補が + * 判定対象から漏れるが、その駅も `sanitizeSuggestions` は通す(モデルが自分で + * 選べば提案できる)ので、提案が減ることはあっても実在しない駅は出ない。 + */ +type RerankState = { done: boolean }; + +/** + * 提案してよい駅を伝えるメッセージ本文を作る。注入しない場合は null。 + * 判定できなかったとき(API 障害・レート制限・期限切れ)も null を返し、 + * モデルが自分で選ぶ今の挙動にフォールバックする。 + */ +const rerankNote = async ( + params: AgentTurnParams, + verified: ReadonlyMap, + state: RerankState +): Promise => { + if (!params.rerank || state.done || verified.size === 0) return null; + state.done = true; + + const request = params.messages[params.messages.length - 1]?.content; + if (!request) return null; + + const startedAt = Date.now(); + const picked = await params.rerank( + { request, currentStationName: params.currentStationName ?? null }, + [...verified.values()] + ); + params.onRerankDone?.(Date.now() - startedAt); + return picked === null ? null : buildRerankNote(picked); +}; + /** 構造化出力を取り出せなかったときの救済用に、最終テキストを best effort で読む */ const readFinalText = async (text: PromiseLike): Promise => { try { @@ -314,8 +366,10 @@ export const runAgentTurn = async ( params: AgentTurnParams ): Promise => { const verified = new Map(); + const rerankState: RerankState = { done: false }; const budget = { remaining: MAX_TOOL_CALLS_PER_TURN }; const openaiReasoning = resolveOpenAIReasoningOptions(params.model); + const googleReasoning = resolveGoogleReasoningSetting(params.model); const result = await params.streamText({ model: params.model, @@ -340,9 +394,23 @@ export const runAgentTurn = async ( }), }, stopWhen: stepCountIs(MAX_TOOL_ITERATIONS + 1), - // イテレーション上限に達したらツールを外し、その時点の結果で応答を生成させる - prepareStep: ({ stepNumber }) => - stepNumber >= MAX_TOOL_ITERATIONS ? { activeTools: [] } : undefined, + prepareStep: async ({ stepNumber, messages }) => { + // イテレーション上限に達したらツールを外し、その時点の結果で応答を生成させる + const tools = + stepNumber >= MAX_TOOL_ITERATIONS ? { activeTools: [] as [] } : {}; + const note = await rerankNote(params, verified, rerankState); + if (!note) return stepNumber >= MAX_TOOL_ITERATIONS ? tools : undefined; + // 本文を書く前に提案集合を渡す。reply がこの集合に条件付けられるので、 + // 本文と提案カードが食い違わない。 + // ロールは system ではなく user にする。Gemini(Vertex)は会話の途中の + // system メッセージを受け付けず、リクエスト組み立て時に + // UnsupportedFunctionalityError(system messages are only supported at + // the beginning of the conversation)で落ちる + return { + ...tools, + messages: [...messages, { role: 'user' as const, content: note }], + }; + }, output: Output.object({ schema: agentOutputSchema }), // 思考(reasoning)の抑制。プロバイダは自分のキーだけを読むため、 // 使っていない側のキーは無視される(anthropic 使用時に openai は無視、その逆も同様) @@ -357,6 +425,11 @@ export const runAgentTurn = async ( // 送るため、'none' 対応が確認できるモデル以外では指定自体を省略する ...(openaiReasoning ? { openai: openaiReasoning } : {}), }, + // Gemini の思考抑制だけは providerOptions ではなく AI SDK 共通の reasoning 設定で + // 指定する。世代ごとに送るパラメータ(thinkingLevel / thinkingBudget)も + // 受理される最小値も異なるため、その解決は llm.ts と SDK に任せる + // (google 以外のモデルでは付けない) + ...(googleReasoning ? { reasoning: googleReasoning } : {}), timeout: { stepMs: LLM_CALL_TIMEOUT_MS }, abortSignal: params.signal, // LLM 呼び出しはコスト重複を避けるため自動再試行しない(設計値) @@ -411,6 +484,8 @@ type PreparedTurn = currentStation: StationSuggestion | null; /** 本体ターン開始時に消費する日次カウンタ(commitDailyTurn に渡す) */ usage: DailyUsage; + /** config:remote の内容(リランクの閾値をここから読む) */ + remoteConfig: Record; }; /** @@ -501,7 +576,7 @@ const prepareAgentTurn = async ( const [faq, currentStation] = await contextPromise; markPhase(metrics, 'context', parallelStartedAt); - return { chatReq, faq, currentStation, usage }; + return { chatReq, faq, currentStation, usage, remoteConfig }; }; /** 前段処理の結果から runAgentTurn の引数を組み立てる(両エンドポイントで共通) */ @@ -513,7 +588,9 @@ const buildTurnParams = ( metrics: TurnMetrics, callbacks: Pick = {} ): AgentTurnParams => { - const { chatReq, faq, currentStation } = prepared; + const { chatReq, faq, currentStation, remoteConfig } = prepared; + // 閾値が KV に入っていなければリランクは丸ごと無効(既定は無効) + const threshold = resolveRerankThreshold(remoteConfig); return { streamText: runtime.streamText, model: resolveAgentModel(env), @@ -535,6 +612,19 @@ const buildTurnParams = ( searchStations: (name) => searchStationsByName(env, name, chatReq.currentStationGroupId, signal), signal, + currentStationName: currentStation?.name ?? null, + rerank: + threshold !== null && env.TYPESAFE_API_KEY + ? createRerankSelector({ + apiKey: env.TYPESAFE_API_KEY, + model: env.TYPESAFE_MODEL, + threshold, + signal, + }) + : undefined, + onRerankDone: (elapsedMs) => { + metrics.phases.rerank = elapsedMs; + }, // 計測は内容を一切持たず、最初の delta までの所要時間とツール実行回数だけを数える onDelta: async (text) => { if (metrics.phases.firstDelta === undefined) { diff --git a/src/agent/llm.test.ts b/src/agent/llm.test.ts new file mode 100644 index 0000000..d13d41e --- /dev/null +++ b/src/agent/llm.test.ts @@ -0,0 +1,177 @@ +/** + * AGENT_MODEL からのプロバイダ解決テスト。 + * @ai-sdk/* は ESM 専用のため Jest ではスタブへ差し替わる(jest.config.js)。 + * スタブは生成時オプションをそのまま持ち回るため、AI Gateway の URL 組み立てと + * サービスアカウント資格情報の受け渡しを検証できる。 + */ +import type { LanguageModel } from 'ai'; +import type { Env } from '../types'; +import { + resolveAgentModel, + resolveGoogleReasoningSetting, + resolveOpenAIReasoningOptions, +} from './llm'; + +/** スタブ(test/stubs/ai-sdk-provider.ts)が返す形。実型には無いので読むときだけ被せる */ +type StubModel = { + modelId: string; + provider: string; + options?: { + baseURL?: string; + headers?: Record; + project?: string; + location?: string; + googleCredentials?: { + clientEmail?: string; + privateKey?: string; + privateKeyId?: string; + }; + }; +}; +const asStub = (model: LanguageModel): StubModel => + model as unknown as StubModel; + +const makeEnv = (env: Partial): Env => env as Env; + +/** サービスアカウント鍵 JSON(1 行 JSON 投入を想定し改行はエスケープ済み) */ +const SA_KEY = JSON.stringify({ + type: 'service_account', + project_id: 'sa-project', + private_key_id: 'kid-1', + private_key: + '-----BEGIN PRIVATE KEY-----\\nMIIB\\n-----END PRIVATE KEY-----\\n', + client_email: 'agent@sa-project.iam.gserviceaccount.com', +}); + +describe('resolveAgentModel', () => { + it('google: はサービスアカウント鍵から Vertex AI を解決する', () => { + const model = asStub( + resolveAgentModel( + makeEnv({ + AGENT_MODEL: 'google:gemini-3.8-flash', + GOOGLE_VERTEX_SA_KEY: SA_KEY, + }) + ) + ); + expect(model.modelId).toBe('gemini-3.8-flash'); + // プロジェクトは鍵の project_id を既定にし、ロケーションは global + expect(model.options?.project).toBe('sa-project'); + expect(model.options?.location).toBe('global'); + expect(model.options?.googleCredentials?.clientEmail).toBe( + 'agent@sa-project.iam.gserviceaccount.com' + ); + expect(model.options?.googleCredentials?.privateKeyId).toBe('kid-1'); + // 1 行 JSON でエスケープされたままの改行は復元してから SDK へ渡す + expect(model.options?.googleCredentials?.privateKey).toBe( + '-----BEGIN PRIVATE KEY-----\nMIIB\n-----END PRIVATE KEY-----\n' + ); + // Gateway 未設定なら Vertex へ直行(baseURL を上書きしない) + expect(model.options?.baseURL).toBeUndefined(); + }); + + it('project / location の var が鍵と既定より優先される', () => { + const model = asStub( + resolveAgentModel( + makeEnv({ + AGENT_MODEL: 'google:gemini-3.8-flash', + GOOGLE_VERTEX_SA_KEY: SA_KEY, + GOOGLE_VERTEX_PROJECT: 'other-project', + GOOGLE_VERTEX_LOCATION: 'asia-northeast1', + }) + ) + ); + expect(model.options?.project).toBe('other-project'); + expect(model.options?.location).toBe('asia-northeast1'); + }); + + it('AI Gateway 経由では google-vertex-ai のモデルパスを baseURL にする', () => { + const model = asStub( + resolveAgentModel( + makeEnv({ + AGENT_MODEL: 'google:gemini-3.8-flash', + GOOGLE_VERTEX_SA_KEY: SA_KEY, + GOOGLE_VERTEX_LOCATION: 'asia-northeast1', + // 末尾スラッシュの揺れも吸収する + AI_GATEWAY_BASE_URL: 'https://gateway.example/v1/acct/gw/', + }) + ) + ); + expect(model.options?.baseURL).toBe( + 'https://gateway.example/v1/acct/gw/google-vertex-ai/v1beta1/projects/sa-project/locations/asia-northeast1/publishers/google' + ); + // 会話本文を Gateway のログに残さない設定は他プロバイダと同じ + expect(model.options?.headers).toEqual({ + 'cf-aig-collect-log-payload': 'false', + }); + }); + + it('google: で鍵が無ければエラーにする', () => { + expect(() => + resolveAgentModel(makeEnv({ AGENT_MODEL: 'google:gemini-3.8-flash' })) + ).toThrow('GOOGLE_VERTEX_SA_KEY is not configured'); + }); + + it('鍵の形式が不正ならエラーにする', () => { + expect(() => + resolveAgentModel( + makeEnv({ + AGENT_MODEL: 'google:gemini-3.8-flash', + GOOGLE_VERTEX_SA_KEY: 'not-json', + }) + ) + ).toThrow('GOOGLE_VERTEX_SA_KEY is not valid JSON'); + + expect(() => + resolveAgentModel( + makeEnv({ + AGENT_MODEL: 'google:gemini-3.8-flash', + GOOGLE_VERTEX_SA_KEY: JSON.stringify({ project_id: 'p' }), + }) + ) + ).toThrow(/client_email and private_key/); + }); + + it('project を鍵からも var からも決められなければエラーにする', () => { + expect(() => + resolveAgentModel( + makeEnv({ + AGENT_MODEL: 'google:gemini-3.8-flash', + GOOGLE_VERTEX_SA_KEY: JSON.stringify({ + client_email: 'a@b.iam.gserviceaccount.com', + private_key: 'pk', + }), + }) + ) + ).toThrow('GOOGLE_VERTEX_PROJECT is not configured'); + }); + + it('未対応のプロバイダ指定はエラーにする', () => { + expect(() => + resolveAgentModel(makeEnv({ AGENT_MODEL: 'gemini-3.8-flash' })) + ).toThrow(/unsupported AGENT_MODEL/); + }); +}); + +describe('resolveGoogleReasoningSetting', () => { + it('思考を完全に止められる 2.5 系では none を返す', () => { + expect(resolveGoogleReasoningSetting('gemini-2.5-flash')).toBe('none'); + }); + + it('3 系は受理される最小値の low まで下げる', () => { + // 'none' は thinkingLevel: minimal に変換され、Vertex に 400 で拒否される + expect(resolveGoogleReasoningSetting('gemini-3.8-flash')).toBe('low'); + expect(resolveGoogleReasoningSetting('gemini-3-flash-preview')).toBe('low'); + }); + + it('thinkingConfig 非対応の世代・他社モデルには何も指定しない', () => { + expect(resolveGoogleReasoningSetting('gemini-2.0-flash')).toBeUndefined(); + expect( + resolveGoogleReasoningSetting('gemini-flash-latest') + ).toBeUndefined(); + expect(resolveGoogleReasoningSetting('gpt-5.1')).toBeUndefined(); + }); + + it('OpenAI 向けの抑制指定は Gemini に反応しない', () => { + expect(resolveOpenAIReasoningOptions('gemini-3.8-flash')).toBeUndefined(); + }); +}); diff --git a/src/agent/llm.ts b/src/agent/llm.ts index d5854fa..4a805ba 100644 --- a/src/agent/llm.ts +++ b/src/agent/llm.ts @@ -1,10 +1,11 @@ /** * 対話本体の LLM プロバイダ解決。 - * AGENT_MODEL("anthropic:" | "openai:")でモデルを切り替え、 - * AI_GATEWAY_BASE_URL が設定されていれば Cloudflare AI Gateway を経由させる - * (ログ・コスト集計・レート制限を Cloudflare 側に集約)。 + * AGENT_MODEL("anthropic:" | "openai:" | "google:")で + * モデルを切り替え、AI_GATEWAY_BASE_URL が設定されていれば Cloudflare AI Gateway を + * 経由させる(ログ・コスト集計・レート制限を Cloudflare 側に集約)。 */ import { createAnthropic } from '@ai-sdk/anthropic'; +import { createVertex } from '@ai-sdk/google-vertex/edge'; import { createOpenAI } from '@ai-sdk/openai'; import type { LanguageModel } from 'ai'; import type { Env } from '../types'; @@ -12,6 +13,48 @@ import type { Env } from '../types'; // AI Gateway 経由時も会話本文をゲートウェイのログに保存させない(設計: プライバシー) const GATEWAY_HEADERS = { 'cf-aig-collect-log-payload': 'false' } as const; +/** Vertex のロケーション既定。global はモデルの提供範囲が最も広く、Gateway 経由でも中継される */ +const DEFAULT_VERTEX_LOCATION = 'global'; + +interface VertexCredentials { + clientEmail: string; + privateKey: string; + privateKeyId?: string; + projectId?: string; +} + +/** + * GOOGLE_VERTEX_SA_KEY(サービスアカウント鍵 JSON)を資格情報に変換する。 + * Vertex AI は API キーではなく ADC(サービスアカウント)認証が前提だが、Workers に + * ADC は無いため鍵の中身を渡し、JWT 署名とトークン交換は SDK の edge 版に行わせる。 + */ +const parseVertexCredentials = (keyJson: string): VertexCredentials => { + let parsed: Record; + try { + parsed = JSON.parse(keyJson) as Record; + } catch { + throw new Error('GOOGLE_VERTEX_SA_KEY is not valid JSON'); + } + const clientEmail = parsed.client_email; + const privateKey = parsed.private_key; + if (typeof clientEmail !== 'string' || typeof privateKey !== 'string') { + throw new Error( + 'GOOGLE_VERTEX_SA_KEY must contain client_email and private_key' + ); + } + return { + clientEmail, + // 1 行 JSON で投入された鍵は改行がエスケープされたままのことがあるため戻す + privateKey: privateKey.replace(/\\n/g, '\n'), + privateKeyId: + typeof parsed.private_key_id === 'string' + ? parsed.private_key_id + : undefined, + projectId: + typeof parsed.project_id === 'string' ? parsed.project_id : undefined, + }; +}; + /** AGENT_MODEL の指定から AI SDK のモデルを生成する。 */ export const resolveAgentModel = (env: Env): LanguageModel => { const spec = env.AGENT_MODEL ?? ''; @@ -49,13 +92,50 @@ export const resolveAgentModel = (env: Env): LanguageModel => { }); return openai(modelId); } + // Gemini は Vertex AI(サービスアカウント認証)経由で使う。 + // Google AI Studio の API キー方式は使わないため "google:" は Vertex を指す + case 'google': { + if (!env.GOOGLE_VERTEX_SA_KEY) { + throw new Error('GOOGLE_VERTEX_SA_KEY is not configured'); + } + const credentials = parseVertexCredentials(env.GOOGLE_VERTEX_SA_KEY); + // プロジェクトは鍵の project_id を既定にし、別プロジェクトを使うときだけ var で上書きする + const project = env.GOOGLE_VERTEX_PROJECT || credentials.projectId; + if (!project) { + throw new Error('GOOGLE_VERTEX_PROJECT is not configured'); + } + const location = env.GOOGLE_VERTEX_LOCATION || DEFAULT_VERTEX_LOCATION; + const vertex = createVertex({ + project, + location, + googleCredentials: { + clientEmail: credentials.clientEmail, + privateKey: credentials.privateKey, + privateKeyId: credentials.privateKeyId, + }, + ...(gateway + ? { + // Gateway は google-vertex-ai 配下のパスをそのまま Vertex へ中継する。 + // API バージョンは直行時の SDK 既定(v1beta1)に合わせ、Gateway の + // 有無で挙動が変わらないようにする + baseURL: `${gateway}/google-vertex-ai/v1beta1/projects/${project}/locations/${location}/publishers/google`, + headers: { ...GATEWAY_HEADERS }, + } + : {}), + }); + return vertex(modelId); + } default: throw new Error( - `unsupported AGENT_MODEL: "${spec}" (expected "anthropic:" or "openai:")` + `unsupported AGENT_MODEL: "${spec}" (expected "anthropic:", "openai:" or "google:")` ); } }; +/** LanguageModel(文字列指定・インスタンスのどちらも来る)からモデル ID を取り出す。 */ +const modelIdOf = (model: LanguageModel): string => + typeof model === 'string' ? model : model.modelId; + /** * reasoningEffort: 'none' を受け付ける OpenAI モデル(GPT-5.1 系以降の * マイナーバージョン付き GPT-5)。gpt-5 無印('minimal' まで)・o 系 @@ -72,9 +152,31 @@ const OPENAI_REASONING_NONE_MODELS = /^gpt-5\.[1-9]/; */ export const resolveOpenAIReasoningOptions = ( model: LanguageModel -): { reasoningEffort: 'none' } | undefined => { - const modelId = typeof model === 'string' ? model : model.modelId; - return OPENAI_REASONING_NONE_MODELS.test(modelId) +): { reasoningEffort: 'none' } | undefined => + OPENAI_REASONING_NONE_MODELS.test(modelIdOf(model)) ? { reasoningEffort: 'none' } : undefined; + +/** 思考を完全に止められる Gemini(2.5 系は thinkingBudget: 0 が通る) */ +const GEMINI_THINKING_OFF_MODELS = /^gemini-2\.5/; +/** thinkingLevel で思考量を指定する Gemini(3 系以降) */ +const GEMINI_THINKING_LEVEL_MODELS = /^gemini-[3-9]/; + +/** + * Gemini 向けの reasoning 抑制を AI SDK 共通の reasoning 設定として解決する。 + * Gemini は世代で送るパラメータが異なる(3 系: thinkingLevel / 2.5 系: + * thinkingBudget)ため、providerOptions.google を自前で組まず SDK に変換させる。 + * + * 3 系で 'none' を使わないのは、SDK が thinkingLevel: 'minimal' に変換する一方、 + * Vertex がこれを拒否する(400 "Thinking level is unsupported: + * THINKING_LEVEL_MINIMAL")ため。受理される最小値の 'low' まで下げる。 + * 2.0 系や gemini-flash-latest のようなエイリアスは指定自体を省略する。 + */ +export const resolveGoogleReasoningSetting = ( + model: LanguageModel +): 'none' | 'low' | undefined => { + const modelId = modelIdOf(model); + if (GEMINI_THINKING_OFF_MODELS.test(modelId)) return 'none'; + if (GEMINI_THINKING_LEVEL_MODELS.test(modelId)) return 'low'; + return undefined; }; diff --git a/src/agent/prompt.test.ts b/src/agent/prompt.test.ts index b36a568..9b17449 100644 --- a/src/agent/prompt.test.ts +++ b/src/agent/prompt.test.ts @@ -72,9 +72,7 @@ describe('buildSystemPrompt', () => { expect(prompt).toContain( 'search_stations_by_name の結果は現在駅からの直通到達性しか保証しない' ); - expect(prompt).toContain( - 'それだけを根拠に駅を乗換地点として扱わない' - ); + expect(prompt).toContain('それだけを根拠に駅を乗換地点として扱わない'); expect(prompt).toContain( '最終目的地への接続を確認できない場合は suggestions を空配列' ); diff --git a/src/agent/rerank.test.ts b/src/agent/rerank.test.ts new file mode 100644 index 0000000..bf1bcaf --- /dev/null +++ b/src/agent/rerank.test.ts @@ -0,0 +1,496 @@ +import { + buildBatchedRequest, + buildIsolatedRequests, + buildRerankNote, + type CandidateScore, + createRerankSelector, + judgeCandidates, + MAX_JUDGED_CANDIDATES, + questionId, + type RerankInput, + resolveRerankThreshold, + selectSuggestions, +} from './rerank'; +import type { StationSuggestion } from './schema'; + +const station = (id: number, name: string): StationSuggestion => ({ + stationId: id, + stationGroupId: id, + name, + nameRoman: name, + lineNames: ['テスト線'], +}); + +const input: RerankInput = { + request: '海が見える駅に行きたい', + currentStationName: '東京', +}; + +const options = { + apiKey: 'key', + model: 'jev-latest', + shape: 'batched' as const, +}; + +const originalFetch = global.fetch; +const warn = jest.spyOn(console, 'warn').mockImplementation(() => {}); + +afterEach(() => { + global.fetch = originalFetch; + jest.clearAllMocks(); +}); + +const jsonResponse = (body: unknown, status = 200) => + new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' }, + }); + +const mockFetch = (handler: () => Response | Promise) => { + const fetchMock = jest.fn(async () => handler()); + global.fetch = fetchMock as unknown as typeof fetch; + return fetchMock; +}; + +const noulAnswers = (values: number[]) => + Object.fromEntries( + values.map((noul, index) => [questionId(index), { type: 'noul', noul }]) + ); + +describe('selectSuggestions', () => { + const scores = (values: number[]): CandidateScore[] => + values.map((fits, index) => ({ + station: station(index, `駅${index}`), + fits, + })); + + it('確率の降順に並べる', () => { + const picked = selectSuggestions(scores([0.3, 0.9, 0.6]), 0); + expect(picked.map((s) => s.name)).toEqual(['駅1', '駅2', '駅0']); + }); + + it('閾値未満は落とす', () => { + const picked = selectSuggestions(scores([0.8, 0.4, 0.75]), 0.7); + expect(picked.map((s) => s.name)).toEqual(['駅0', '駅2']); + }); + + it('閾値と同値は残す', () => { + expect(selectSuggestions(scores([0.7]), 0.7)).toHaveLength(1); + }); + + it('全候補が閾値未満なら空配列(「合う駅が無い」を表現できる)', () => { + expect(selectSuggestions(scores([0.2, 0.1]), 0.7)).toEqual([]); + }); + + it('同じ確率なら判定に渡した順を保つ', () => { + const picked = selectSuggestions(scores([0.5, 0.5, 0.5]), 0); + expect(picked.map((s) => s.name)).toEqual(['駅0', '駅1', '駅2']); + }); + + it('上限件数で切り詰める', () => { + const picked = selectSuggestions(scores([0.9, 0.8, 0.7, 0.6, 0.5, 0.4]), 0); + expect(picked).toHaveLength(5); + expect(picked.map((s) => s.name)).toEqual([ + '駅0', + '駅1', + '駅2', + '駅3', + '駅4', + ]); + }); + + // stationsByName は同一物理駅を路線別レコード(別 stationId・同一 groupId)で + // 返す。畳まずに確率順で切ると枠が同じ駅で埋まる(実測で「海が見える駅」の + // 上位5件が熱海4レコード+真鶴になった)。 + describe('同一物理駅は groupId で 1 件に畳む', () => { + /** 熱海の路線別 3 レコード(groupId 同じ)+ 根府川・早川 */ + const atamiAndOthers: CandidateScore[] = [ + { station: { ...station(1, '熱海'), stationGroupId: 100 }, fits: 0.9 }, + { station: { ...station(2, '熱海'), stationGroupId: 100 }, fits: 0.89 }, + { station: { ...station(3, '熱海'), stationGroupId: 100 }, fits: 0.88 }, + { station: { ...station(4, '根府川'), stationGroupId: 200 }, fits: 0.8 }, + { station: { ...station(5, '早川'), stationGroupId: 300 }, fits: 0.7 }, + ]; + + it('枠を同じ駅で埋めず、別の駅に回す', () => { + const picked = selectSuggestions(atamiAndOthers, 0, 3); + expect(picked.map((s) => s.name)).toEqual(['熱海', '根府川', '早川']); + }); + + it('残すのは同一グループで最も確率の高いレコード', () => { + const picked = selectSuggestions(atamiAndOthers, 0, 1); + expect(picked).toEqual([{ ...station(1, '熱海'), stationGroupId: 100 }]); + }); + + it('畳んだ結果が上限未満なら、その件数で返す', () => { + const picked = selectSuggestions(atamiAndOthers.slice(0, 3), 0); + expect(picked).toHaveLength(1); + }); + }); +}); + +describe('buildBatchedRequest', () => { + it('候補数ぶんの質問を作り、state は 1 つにまとめる', () => { + const { state, questions } = buildBatchedRequest(input, [ + station(1, '根府川'), + station(2, '海老名'), + ]); + + expect(state).toEqual({ + request: '海が見える駅に行きたい', + current_station: '東京', + candidates: [ + { name: '根府川', name_roman: '根府川', lines: ['テスト線'] }, + { name: '海老名', name_roman: '海老名', lines: ['テスト線'] }, + ], + }); + expect(Object.keys(questions)).toEqual(['fits_0', 'fits_1']); + expect(questions.fits_0.instructions).toContain('`candidates[0]`'); + expect(questions.fits_1.instructions).toContain('`candidates[1]`'); + }); + + it('基準は全候補で同一にする(候補ごとに基準が変わると比較できない)', () => { + const { questions } = buildBatchedRequest(input, [ + station(1, 'A'), + station(2, 'B'), + ]); + expect(questions.fits_0.criteria).toEqual(questions.fits_1.criteria); + }); + + it('現在駅が不明なら state に載せない', () => { + const { state } = buildBatchedRequest( + { request: '温泉', currentStationName: null }, + [station(1, '鬼怒川温泉')] + ); + expect(state).not.toHaveProperty('current_station'); + }); +}); + +describe('buildIsolatedRequests', () => { + it('候補ごとに 1 リクエストへ分け、各 state は 1 候補だけを載せる', () => { + const requests = buildIsolatedRequests(input, [ + station(1, '根府川'), + station(2, '海老名'), + ]); + + expect(requests).toHaveLength(2); + expect(requests[0].state.candidate).toEqual({ + name: '根府川', + name_roman: '根府川', + lines: ['テスト線'], + }); + expect(requests[0].state).not.toHaveProperty('candidates'); + expect(Object.keys(requests[0].questions)).toEqual(['fits']); + expect(requests[0].questions.fits.instructions).toContain('`candidate`'); + }); +}); + +describe('judgeCandidates', () => { + it('候補が無ければ API を呼ばずに空配列を返す', async () => { + const fetchMock = mockFetch(() => jsonResponse({})); + await expect(judgeCandidates(input, [], options)).resolves.toEqual([]); + expect(fetchMock).not.toHaveBeenCalled(); + }); + + it('案 Y は 1 リクエストで、回答を候補へ添字で戻す', async () => { + const fetchMock = mockFetch(() => + jsonResponse({ answers: noulAnswers([0.2, 0.95]) }) + ); + + const scores = await judgeCandidates( + input, + [station(1, '海老名'), station(2, '根府川')], + options + ); + + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(scores).toEqual([ + { station: station(1, '海老名'), fits: 0.2 }, + { station: station(2, '根府川'), fits: 0.95 }, + ]); + }); + + // 候補ごとに違う確率を返させ、送った state.candidate と戻ってきた station の + // 対応まで固定する。全候補同じ値だと添字ズレの退行を検出できない。 + it('案 X は候補ごとにリクエストを出し、応答を送った候補に対応づける', async () => { + const byName: Record = { A: 0.1, B: 0.5, C: 0.9 }; + const fetchMock = jest.fn(async (_url: string, init: RequestInit) => { + const body = JSON.parse(init.body as string); + const name = body.state.candidate.name as string; + return jsonResponse({ + answers: { fits: { type: 'noul', noul: byName[name] } }, + }); + }); + global.fetch = fetchMock as unknown as typeof fetch; + + const scores = await judgeCandidates( + input, + [station(1, 'A'), station(2, 'B'), station(3, 'C')], + { ...options, shape: 'isolated' } + ); + + expect(fetchMock).toHaveBeenCalledTimes(3); + expect(scores).toEqual([ + { station: station(1, 'A'), fits: 0.1 }, + { station: station(2, 'B'), fits: 0.5 }, + { station: station(3, 'C'), fits: 0.9 }, + ]); + }); + + // 判定できなかった候補を黙って落とすと、戻り値が「完全な判定結果」として扱われ、 + // その候補は閾値以上でも提案から確実に除外される。全候補を判定できたときだけ + // 結果を返す。 + describe('1 件でも判定できなければ null を返す', () => { + it('案 Y で一部の回答が型不一致・範囲外', async () => { + mockFetch(() => + jsonResponse({ + answers: { + fits_0: { type: 'noul', noul: 0.9 }, + fits_1: { type: 'choice', choice: 'yes' }, + fits_2: { type: 'noul', noul: 1.4 }, + }, + }) + ); + + await expect( + judgeCandidates( + input, + [station(1, 'A'), station(2, 'B'), station(3, 'C')], + options + ) + ).resolves.toBeNull(); + }); + + it('案 Y で回答が 1 件も読めない(200 だが answers が空)', async () => { + mockFetch(() => jsonResponse({ answers: {} })); + + await expect( + judgeCandidates(input, [station(1, 'A'), station(2, 'B')], options) + ).resolves.toBeNull(); + }); + + it('案 X で一部のリクエストだけ失敗(429 の着順で提案が揺れないように)', async () => { + let call = 0; + mockFetch(() => { + call += 1; + return call === 1 + ? jsonResponse({}, 429) + : jsonResponse({ answers: { fits: { type: 'noul', noul: 0.7 } } }); + }); + + await expect( + judgeCandidates(input, [station(1, 'A'), station(2, 'B')], { + ...options, + shape: 'isolated', + }) + ).resolves.toBeNull(); + }); + }); + + it('判定に渡す候補数には天井があり、送る質問も切り詰め後の数になる', async () => { + const many = Array.from({ length: MAX_JUDGED_CANDIDATES + 10 }, (_, i) => + station(i, `駅${i}`) + ); + const fetchMock = jest.fn(async (_url: string, _init: RequestInit) => + jsonResponse({ + answers: noulAnswers(Array(MAX_JUDGED_CANDIDATES).fill(0.5)), + }) + ); + global.fetch = fetchMock as unknown as typeof fetch; + + const scores = await judgeCandidates(input, many, options); + + expect(scores).toHaveLength(MAX_JUDGED_CANDIDATES); + // 戻り値の長さだけを見ると、切り詰め前の候補で質問を組む退行を見逃す + // (モックが 30 問分しか答えないので scores は 30 のまま通ってしまう) + const body = JSON.parse(fetchMock.mock.calls[0]?.[1].body as string); + expect(Object.keys(body.questions)).toHaveLength(MAX_JUDGED_CANDIDATES); + expect(body.state.candidates).toHaveLength(MAX_JUDGED_CANDIDATES); + }); + + it('使用トークンを onUsage で返す', async () => { + mockFetch(() => + jsonResponse({ + answers: noulAnswers([0.5]), + usage: { input_tokens: 120, output_tokens: 3 }, + }) + ); + const onUsage = jest.fn(); + + await judgeCandidates(input, [station(1, 'A')], { ...options, onUsage }); + + expect(onUsage).toHaveBeenCalledWith({ inputTokens: 120, outputTokens: 3 }); + }); + + // 失敗しても応答を壊さないのがこのモジュールの前提。例外を外に出さず null を返し、 + // 呼び出し側は LLM 側の順序に倒す。再試行もしない(待つ時間は本文に使うべき)。 + describe('失敗は握って null を返す', () => { + it('HTTP エラー', async () => { + const fetchMock = mockFetch(() => jsonResponse({}, 500)); + await expect( + judgeCandidates(input, [station(1, 'A')], options) + ).resolves.toBeNull(); + expect(fetchMock).toHaveBeenCalledTimes(1); + expect(warn).toHaveBeenCalled(); + }); + + it('ネットワーク断', async () => { + mockFetch(() => { + throw new Error('network down'); + }); + await expect( + judgeCandidates(input, [station(1, 'A')], options) + ).resolves.toBeNull(); + }); + + // SyntaxError.message は応答本文の先頭を含む(`Unexpected token 'o', + // "SECRET" is not valid JSON`)。エラーをそのまま出すと、!res.ok 側で + // 本文を出さないようにした意図がここで破れる。 + it('JSON が壊れている(本文をログに載せない)', async () => { + mockFetch(() => new Response('SENSITIVE-BODY', { status: 200 })); + + await expect( + judgeCandidates(input, [station(1, 'A')], options) + ).resolves.toBeNull(); + + const logged = warn.mock.calls.flat().map(String).join(' '); + expect(logged).toContain('SyntaxError'); + expect(logged).not.toContain('SENSITIVE-BODY'); + }); + + it('案 X で全候補が失敗した場合', async () => { + mockFetch(() => jsonResponse({}, 503)); + await expect( + judgeCandidates(input, [station(1, 'A'), station(2, 'B')], { + ...options, + shape: 'isolated', + }) + ).resolves.toBeNull(); + }); + + // 部分失敗(案 X の一部だけ 429 など)も null。 + // 「1 件でも判定できなければ null を返す」に集約してある。 + + // 計測でコストを取り落とさないよう、null を返す経路でも使用トークンは報告する + it('null を返す場合でも onUsage は呼ぶ', async () => { + mockFetch(() => + jsonResponse({ answers: {}, usage: { input_tokens: 90 } }) + ); + const onUsage = jest.fn(); + + await expect( + judgeCandidates(input, [station(1, 'A')], { ...options, onUsage }) + ).resolves.toBeNull(); + expect(onUsage).toHaveBeenCalledWith({ + inputTokens: 90, + outputTokens: 0, + }); + }); + }); + + it('signal を fetch へ渡す(ターン全体の期限で中断させる)', async () => { + const controller = new AbortController(); + const fetchMock = jest.fn(async (_url: string, _init: RequestInit) => + jsonResponse({ answers: noulAnswers([0.5]) }) + ); + global.fetch = fetchMock as unknown as typeof fetch; + + await judgeCandidates(input, [station(1, 'A')], { + ...options, + signal: controller.signal, + }); + + expect(fetchMock.mock.calls[0]?.[1].signal).toBe(controller.signal); + }); +}); + +describe('resolveRerankThreshold', () => { + // 既定値をコードに持たない。KV に値が入るまでリランクは丸ごと無効 + it('未設定なら無効', () => { + expect(resolveRerankThreshold({})).toBeNull(); + }); + + it('0 より大きく 1 以下の数値だけ受ける', () => { + expect(resolveRerankThreshold({ agent_rerank_threshold: 0.7 })).toBe(0.7); + expect(resolveRerankThreshold({ agent_rerank_threshold: 1 })).toBe(1); + expect(resolveRerankThreshold({ agent_rerank_threshold: 0 })).toBeNull(); + expect(resolveRerankThreshold({ agent_rerank_threshold: -0.1 })).toBeNull(); + expect(resolveRerankThreshold({ agent_rerank_threshold: 1.5 })).toBeNull(); + }); + + it('数値でない値・非有限値は無効に倒す', () => { + expect(resolveRerankThreshold({ agent_rerank_threshold: 'x' })).toBeNull(); + expect(resolveRerankThreshold({ agent_rerank_threshold: null })).toBeNull(); + expect( + resolveRerankThreshold({ + agent_rerank_threshold: Number.POSITIVE_INFINITY, + }) + ).toBeNull(); + }); + + // Number() に直接かけると true が 1(ほぼ全部の候補を棄却する閾値)になり、 + // 「とりあえず true で有効化」という書き方で提案が静かに消える。 + // 配列も Number([0.7]) === 0.7 で通ってしまう。 + it('真偽値・配列・オブジェクトは無効に倒す', () => { + expect(resolveRerankThreshold({ agent_rerank_threshold: true })).toBeNull(); + expect( + resolveRerankThreshold({ agent_rerank_threshold: false }) + ).toBeNull(); + expect( + resolveRerankThreshold({ agent_rerank_threshold: [0.7] }) + ).toBeNull(); + expect( + resolveRerankThreshold({ agent_rerank_threshold: { value: 0.7 } }) + ).toBeNull(); + }); + + // KV は文字列で入ることがある(wrangler kv key put) + it('数値として読める文字列は受ける', () => { + expect(resolveRerankThreshold({ agent_rerank_threshold: '0.7' })).toBe(0.7); + }); +}); + +describe('buildRerankNote', () => { + it('順序と路線名つきで並べ、集合外を禁じる', () => { + const note = buildRerankNote([ + { ...station(1, '熱海'), lineNames: ['東海道線', '伊東線'] }, + station(2, '真鶴'), + ]); + expect(note).toContain('1. 熱海(東海道線・伊東線)'); + expect(note).toContain('2. 真鶴(テスト線)'); + expect(note).toContain('ここに無い駅を入れてはならない'); + }); + + it('0 件なら「見つからなかった」と伝えるよう書く', () => { + const note = buildRerankNote([]); + expect(note).toContain('見つからなかった'); + expect(note).toContain('空配列'); + expect(note).toContain('埋め合わせに提案してはならない'); + }); +}); + +describe('createRerankSelector', () => { + it('判定して閾値で絞った駅を返す', async () => { + mockFetch(() => jsonResponse({ answers: noulAnswers([0.9, 0.3]) })); + const select = createRerankSelector({ + apiKey: 'key', + model: 'jev-latest', + threshold: 0.7, + }); + + await expect( + select(input, [station(1, '熱海'), station(2, '来宮')]) + ).resolves.toEqual([station(1, '熱海')]); + }); + + it('判定できなければ null をそのまま返す(呼び出し側がフォールバックする)', async () => { + mockFetch(() => jsonResponse({}, 500)); + const select = createRerankSelector({ + apiKey: 'key', + model: 'jev-latest', + threshold: 0.7, + }); + + await expect(select(input, [station(1, '熱海')])).resolves.toBeNull(); + }); +}); diff --git a/src/agent/rerank.ts b/src/agent/rerank.ts new file mode 100644 index 0000000..d5d892f --- /dev/null +++ b/src/agent/rerank.ts @@ -0,0 +1,410 @@ +/** + * 提案駅のリランク — ツール結果(実在確認済みの候補)のうち、どれがユーザの要望に + * 合っているかを TypeSafe(System One / Jev)の noul で判定する。 + * + * TypeSafe は判定しか返さず文章生成をしない()ので、 + * ここが決めるのは「どの駅か」だけ。候補を思いつく(世界知識)のと本文を書くのは + * 対話本体の LLM が担い続ける。 + * + * # フィードバックのトリアージとは別物 + * + * 同じ API を叩くが、失敗したときにすべきことが正反対のため実装を共有しない。 + * トリアージはキューのコンシューマで動き、1 件も落とせないので粘って再試行し、 + * 最後は throw して DLQ に残す。こちらはライブの HTTP リクエスト(ターン全体 + * 25 秒)の中で動き、遅延がそのままユーザの待ち時間になる。判定は「あると嬉しい」 + * ものでしかないので、 + * + * - 再試行しない(待つならその時間は本文のストリーミングに使うべき) + * - 例外を外に出さない。失敗は null で返し、呼び出し側は LLM 側の順序に倒す + * + * という設計にする。トリアージ側の再試行方針をこちらに持ち込んではならない。 + * + * # 閾値は未フィッティング + * + * 採用する閾値は計測(src/cli/typesafe-rerank-spike.ts)で決める。既定値を置くと + * 測る前に本番へ出る道ができてしまうため、selectSuggestions は閾値を必須引数で + * 受け取る。 + */ +import { AGENT_MAX_SUGGESTIONS, type StationSuggestion } from './schema'; + +const API_URL = 'https://api.typesafe.ai/v1/systemone'; + +/** 判定に渡す候補の上限。1 ターンのツール結果が数十件になり得るため天井を置く */ +export const MAX_JUDGED_CANDIDATES = 30; + +/** + * このモジュールが使う TypeSafe の型は noul だけ。choice / score は使わないので + * 定義も持たない(使う形だけを持つことで、他用途への流用を誘わない)。 + */ +type NoulQuestion = { + type: 'noul'; + instructions: string; + criteria: { true: string; false: string }; +}; + +type SystemOneResult = { + answers: Record; + usage?: { input_tokens?: number; output_tokens?: number }; +}; + +/** 判定の材料。直近のユーザ発話と、相対表現の解釈に必要な現在駅 */ +export type RerankInput = { + /** 直近のユーザ発話(要望そのもの) */ + request: string; + /** 現在駅の駅名。「近く」「ここから」の解釈に使う。不明なら null */ + currentStationName: string | null; +}; + +export type CandidateScore = { + station: StationSuggestion; + /** 要望を満たす確率(0〜1) */ + fits: number; +}; + +/** + * 候補 1 件が要望に合っているかを問う。 + * + * 命令・基準は日本語で書く。判定対象が日本語の要望文と日本の駅名で、英語に + * 置き換えると「海が見える」「下町の雰囲気」のような要望の輪郭がぼやけるため + * (トリアージ側の質問定義と同じ判断)。 + * + * 基準は候補同士を比べない書き方にしてある。候補ごとに独立して評価される + * (案 X)場合、比較を求める基準は答えられないため。 + */ +const FITS_INSTRUCTIONS = + '`candidate` の駅は、`request` の要望に対する行き先として妥当か。'; + +const FITS_CRITERIA = { + true: '要望が挙げている条件(地名・目的・景色・施設・雰囲気)に、この駅が実際に当てはまる。要望が特定の駅を指しているなら、その駅そのもの。', + false: + '要望の条件に当てはまらない。名前や読みが要望の語と似ているだけで別の場所にある駅、要望と無関係な地域の駅、条件を満たさない駅。', +} as const; + +const fitsQuestion = (candidatePath: string): NoulQuestion => ({ + type: 'noul', + // 参照する候補の位置だけを差し替える。命令と基準は全候補で同一にする + instructions: FITS_INSTRUCTIONS.replace( + '`candidate`', + `\`${candidatePath}\`` + ), + criteria: FITS_CRITERIA, +}); + +/** state に載せる候補の形。判定に効かないフィールド(ID)は渡さない */ +const toCandidateState = (station: StationSuggestion) => ({ + name: station.name, + name_roman: station.nameRoman, + lines: station.lineNames, +}); + +const baseState = (input: RerankInput): Record => ({ + request: input.request, + ...(input.currentStationName + ? { current_station: input.currentStationName } + : {}), +}); + +/** + * 案 X: 候補ごとに 1 リクエスト。 + * rerank cookbook()準拠で、 + * 各問が 1 候補しか見ないため候補同士の比較効果が入らない。state を候補数ぶん + * 払うのでトークンは案 Y より高い。 + */ +export const buildIsolatedRequests = ( + input: RerankInput, + candidates: readonly StationSuggestion[] +): { + state: Record; + questions: Record; +}[] => + candidates.map((station) => ({ + state: { ...baseState(input), candidate: toCandidateState(station) }, + questions: { fits: fitsQuestion('candidate') }, + })); + +/** + * 案 Y: 全候補を 1 リクエストに集約。 + * state を 1 回しか払わないためトークンが安い。質問は互いに独立で並列に評価 + * されるが、各問が候補一覧全体を見る点が案 X と違う。どちらが精度で勝つかは + * 計測で決める。 + */ +export const buildBatchedRequest = ( + input: RerankInput, + candidates: readonly StationSuggestion[] +): { + state: Record; + questions: Record; +} => { + const questions: Record = {}; + candidates.forEach((_, index) => { + questions[questionId(index)] = fitsQuestion(`candidates[${index}]`); + }); + return { + state: { + ...baseState(input), + candidates: candidates.map(toCandidateState), + }, + questions, + }; +}; + +/** 案 Y の質問 ID。回答を候補の添字へ戻すために使う */ +export const questionId = (index: number): string => `fits_${index}`; + +export type JudgeOptions = { + apiKey: string; + model: string; + /** 案 X(候補ごと)か案 Y(1 リクエストに集約)か */ + shape: 'isolated' | 'batched'; + signal?: AbortSignal; + /** 計測用。使用トークンを受け取る */ + onUsage?: (usage: { inputTokens: number; outputTokens: number }) => void; +}; + +/** + * TypeSafe へ 1 リクエスト投げる。再試行はしない。失敗(HTTP エラー・ネットワーク + * 断・JSON 破損・中断)はすべて null に落とし、例外を外へ出さない。 + */ +const requestSystemOne = async ( + body: unknown, + options: JudgeOptions +): Promise => { + try { + const res = await fetch(API_URL, { + method: 'POST', + headers: { + Authorization: `Bearer ${options.apiKey}`, + 'Content-Type': 'application/json', + }, + body: JSON.stringify(body), + signal: options.signal, + }); + if (!res.ok) { + // 本文はログに出さない(会話本文由来の情報が混ざり得る) + console.warn(`agent rerank: TypeSafe API ${res.status}`); + return null; + } + return (await res.json()) as SystemOneResult; + } catch (e) { + // エラーオブジェクトをそのまま出さない。JSON 破損時の SyntaxError.message は + // 応答本文の先頭を含む(例: `Unexpected token 'o', "not json" is not valid JSON`) + // ため、!res.ok 側で本文を出さないようにした意図がここで破れる。 + // 種別(AbortError / TypeError / SyntaxError)だけで運用上は足りる。 + console.warn( + `agent rerank: TypeSafe API の呼び出しに失敗した (${kindOf(e)})` + ); + return null; + } +}; + +/** + * エラーの種別だけを取り出す。`instanceof Error` は使わない。fetch / Response の + * 実装が別 realm の Error を投げると false になり(Jest の node 環境で実際に + * SyntaxError が取れず `object` に落ちた)、種別が分からないログになる。 + */ +const kindOf = (e: unknown): string => { + const name = (e as { name?: unknown } | null)?.name; + return typeof name === 'string' && name ? name : typeof e; +}; + +/** 回答から noul を取り出す。型が合わない・範囲外は null */ +const readNoul = (answer: { type?: string; noul?: number } | undefined) => { + if (!answer || answer.type !== 'noul') return null; + const { noul } = answer; + return typeof noul === 'number' && noul >= 0 && noul <= 1 ? noul : null; +}; + +/** + * 候補を判定する。 + * + * **全候補を判定できたときだけ結果を返す。1 件でも判定できなければ null。** + * 判定できなかった候補を黙って落とすと、戻り値は「完全な判定結果」として扱われ、 + * その候補は閾値以上でも提案から確実に除外される。案 X は候補ごとに並列で投げる + * ので、レート制限(429)に当たるのがどの候補かは着順で決まり、同じ会話でも + * 提案が揺れることになる。再試行しない設計なので回復経路も無い。 + * + * null のときの代償はリランクを丸ごと捨てて LLM 側の順序に倒すことで、これは + * 今の本番挙動そのものなので劣化にならない。「落としてよい」のはリランクの結果 + * 全体であって、個々の候補ではない。 + * + * 戻り値の意味: + * null … 判定できなかった(呼び出し側は LLM 側の順序に倒す) + * [] … 判定した結果、候補が 0 件だった(要望に合う駅が無いとは別。閾値は + * selectSuggestions が当てる) + */ +export const judgeCandidates = async ( + input: RerankInput, + candidates: readonly StationSuggestion[], + options: JudgeOptions +): Promise => { + const targets = candidates.slice(0, MAX_JUDGED_CANDIDATES); + if (targets.length === 0) return []; + + const usage = { inputTokens: 0, outputTokens: 0 }; + const collect = (results: readonly (SystemOneResult | null)[]) => { + for (const result of results) { + usage.inputTokens += result?.usage?.input_tokens ?? 0; + usage.outputTokens += result?.usage?.output_tokens ?? 0; + } + }; + + /** 候補の添字 → その候補の回答を含む応答。判定できない候補があれば null */ + let fitsByIndex: (number | null)[]; + + if (options.shape === 'batched') { + const { state, questions } = buildBatchedRequest(input, targets); + const result = await requestSystemOne( + { state, model: options.model, questions }, + options + ); + collect([result]); + // 使用トークンは null を返す場合でも報告する(計測でコストを取り落とさない) + options.onUsage?.(usage); + if (!result) return null; + fitsByIndex = targets.map((_, index) => + readNoul(result.answers?.[questionId(index)]) + ); + } else { + const requests = buildIsolatedRequests(input, targets); + const results = await Promise.all( + requests.map(({ state, questions }) => + requestSystemOne({ state, model: options.model, questions }, options) + ) + ); + collect(results); + options.onUsage?.(usage); + fitsByIndex = results.map((result) => readNoul(result?.answers?.fits)); + } + + const scores: CandidateScore[] = []; + for (const [index, station] of targets.entries()) { + const fits = fitsByIndex[index]; + if (fits === null || fits === undefined) return null; + scores.push({ station, fits }); + } + return scores; +}; + +/** + * 判定結果から提案駅を選ぶ。確率の降順に並べ、閾値未満を落とし、同一物理駅を + * 1 件に畳んでから上限件数で切る。同じ確率のときは判定に渡した順(=ツール結果の + * 順)を保つ。 + * + * `stationGroupId` で畳むのが要点。`stationsByName` は同一物理駅を路線別レコード + * (別 stationId・同一 groupId)で返すため、畳まずに確率順で切ると枠が同じ駅で + * 埋まる。実測(案 Y)では「海が見える駅」の上位 5 件が熱海の 4 レコードと真鶴に + * なり、根府川と早川が押し出された。アプリ側は受け取った提案を groupId で畳む + * (`dedupeStationsByGroupId`)ので、そのままでは提案カードが 2 枚に減る。 + * 実在性検証の `sanitizeSuggestions` は stationId しか見ないため、ここで畳む。 + * + * threshold は既定値を持たない。計測で決めるまで本番に出せないようにするため。 + */ +export const selectSuggestions = ( + scores: readonly CandidateScore[], + threshold: number, + max: number = AGENT_MAX_SUGGESTIONS +): StationSuggestion[] => { + const ranked = scores + .map((score, index) => ({ score, index })) + .filter(({ score }) => score.fits >= threshold) + .sort((a, b) => b.score.fits - a.score.fits || a.index - b.index); + + const seenGroups = new Set(); + const picked: StationSuggestion[] = []; + for (const { score } of ranked) { + const { stationGroupId } = score.station; + // 同一物理駅は最も確率の高いレコードだけを残す + if (seenGroups.has(stationGroupId)) continue; + seenGroups.add(stationGroupId); + picked.push(score.station); + if (picked.length >= max) break; + } + return picked; +}; + +/** + * 有効化と閾値は `config:remote` の `agent_rerank_threshold` で決める + * (キルスイッチ・日次上限と同じ運用感でデプロイなしに切り替えられる)。 + * 0 より大きく 1 以下の数値のときだけ有効。未設定・不正値は無効(null)。 + * + * 既定値をコードに持たない。実測(2026-09-18・案 Y・20 項目)では 0.70 が + * reject 違反 0・recall 17/19・「合う駅なし」4/4 を満たす最小値だったが、 + * 20 項目に対するグリッド最良値なので、値は KV 側で持って調整する。 + */ +export const resolveRerankThreshold = ( + remoteConfig: Record +): number | null => { + const raw = remoteConfig.agent_rerank_threshold; + // 数値・数値形式の文字列だけを通す。Number() に直接かけると + // `true` が 1(=ほぼ全部の候補を棄却する閾値)として有効になり、 + // 「とりあえず true にして有効化する」という書き方で提案が消える。 + // 配列(`[0.7]` → 0.7)も同様に通ってしまう。 + if (typeof raw !== 'number' && typeof raw !== 'string') return null; + const value = Number(raw); + return Number.isFinite(value) && value > 0 && value <= 1 ? value : null; +}; + +/** 判定結果を提案駅へ落とすところまでをまとめた関数。null は「判定できなかった」 */ +export type RerankSelector = ( + input: RerankInput, + candidates: readonly StationSuggestion[] +) => Promise; + +/** 対話ターンから使う形。無効時は呼び出し側が null を渡して分岐を消す */ +export const createRerankSelector = (config: { + apiKey: string; + model: string; + threshold: number; + signal?: AbortSignal; +}): RerankSelector => { + return async (input, candidates) => { + const scores = await judgeCandidates(input, candidates, { + apiKey: config.apiKey, + model: config.model, + // 実測で案 X(候補ごと)に全指標で勝ち、トークンは 55%、レイテンシは 65% だった + shape: 'batched', + signal: config.signal, + }); + return scores === null ? null : selectSuggestions(scores, config.threshold); + }; +}; + +/** + * 提案してよい駅を対話本体へ伝えるメッセージ本文(ツール結果のあとに user として + * 差し込む。Gemini は会話の途中の system メッセージを受け付けないため)。 + * + * 判定結果でモデルの出力スキーマを置き換えるのではなく、**本文を書く前に + * 提案集合を渡す**形にしてある。こうすると reply が提案集合に条件付けられるので + * 本文と提案カードが食い違わず、判定できなかったとき(null)は何も注入せず + * 今と同じ挙動(モデルが自分で選ぶ)にフォールバックできる。 + */ +export const buildRerankNote = ( + picked: readonly StationSuggestion[] +): string => { + if (picked.length === 0) { + return [ + '# 提案してよい駅', + '', + 'ツール結果を精査したが、ユーザの要望を満たす駅は見つからなかった。', + 'suggestions は空配列にし、見つからなかったことを正直に伝えるか、', + 'ユーザが答えられる具体的な確認を 1 つだけ返すこと。', + '要望に合わない駅を埋め合わせに提案してはならない。', + ].join('\n'); + } + const lines = picked.map((station, index) => { + const lines_ = station.lineNames.length + ? `(${station.lineNames.join('・')})` + : ''; + return `${index + 1}. ${station.name}${lines_}`; + }); + return [ + '# 提案してよい駅(要望に合う順)', + '', + ...lines, + '', + 'suggestions はこの中からこの順で入れること。ここに無い駅を入れてはならない。', + 'reply もこの駅について書くこと(ここに無い駅名を本文で挙げない)。', + '提案が不要な応答(使い方の質問、確認質問を返す場合)では suggestions を', + '空配列にしてよい。', + ].join('\n'); +}; diff --git a/src/agent/tools.test.ts b/src/agent/tools.test.ts index 35c6d77..6fa3d57 100644 --- a/src/agent/tools.test.ts +++ b/src/agent/tools.test.ts @@ -115,7 +115,7 @@ describe('buildStationNameVariants', () => { describe('searchStationsByName', () => { const makeEnv = (fetchImpl: jest.Mock): Env => - ({ SAPI_BFF: { fetch: fetchImpl } }) as unknown as Env; + ({ STATION_API: { fetch: fetchImpl } }) as unknown as Env; const queriedNames = (fetchMock: jest.Mock): string[] => fetchMock.mock.calls.map( @@ -138,6 +138,34 @@ describe('searchStationsByName', () => { limit: 10, fromStationGroupId: 1130205, }); + // stationapi は GraphQL をサブドメイン直下(POST /)でのみ受ける + expect(fetchMock.mock.calls[0][0]).toBe('https://stationapi/'); + expect(init.method).toBe('POST'); + }); + + it('STATION_API があれば STATION_API_GRAPHQL_URL より優先する', async () => { + const bindingFetch = jest + .fn() + .mockResolvedValue(gqlResponse([gqlStation(1)])); + // 分岐順が入れ替わっても実ネットワークへ出ないようにしておく + const globalFetch = jest + .spyOn(globalThis, 'fetch') + .mockResolvedValue(gqlResponse([gqlStation(99)])); + try { + const result = await searchStationsByName( + { + STATION_API: { fetch: bindingFetch }, + STATION_API_GRAPHQL_URL: 'https://gql.example/graphql', + } as unknown as Env, + '鎌倉', + undefined + ); + expect(result[0].stationId).toBe(1); + expect(bindingFetch.mock.calls[0][0]).toBe('https://stationapi/'); + expect(globalFetch).not.toHaveBeenCalled(); + } finally { + globalFetch.mockRestore(); + } }); it('失敗時に 1 回だけ再試行する', async () => { @@ -181,7 +209,7 @@ describe('searchStationsByName', () => { it('バインディングも URL も無ければエラー', async () => { await expect( searchStationsByName({} as unknown as Env, '海', undefined) - ).rejects.toThrow('SAPI_BFF'); + ).rejects.toThrow('STATION_API'); }); it('0 件なら表記ゆれ候補で引き直す(分かち書きローマ字の救済)', async () => { @@ -260,7 +288,7 @@ describe('searchStationsByName', () => { describe('fetchStationByGroupId', () => { const makeEnv = (fetchImpl: jest.Mock): Env => - ({ SAPI_BFF: { fetch: fetchImpl } }) as unknown as Env; + ({ STATION_API: { fetch: fetchImpl } }) as unknown as Env; const groupResponse = (stations: unknown[]) => new Response(JSON.stringify({ data: { stationGroupStations: stations } }), { diff --git a/src/agent/tools.ts b/src/agent/tools.ts index c6dabe4..cfe9aa2 100644 --- a/src/agent/tools.ts +++ b/src/agent/tools.ts @@ -1,7 +1,7 @@ /** - * 駅検索ツール — sapi-bff(BFF ルートワーカー)の GraphQL stationsByName で - * 駅名の実在確認を行う。Service Binding(SAPI_BFF)を優先し、 - * 未設定なら SAPI_BFF_GRAPHQL_URL への fetch にフォールバックする。 + * 駅検索ツール — stationapi の GraphQL stationsByName で + * 駅名の実在確認を行う。Service Binding(STATION_API)を優先し、 + * 未設定なら STATION_API_GRAPHQL_URL への fetch にフォールバックする。 * 検索結果は verified マップへ蓄積し、最終応答のサーバ側検証 * (validate.ts の sanitizeSuggestions)の突合元になる。 */ @@ -15,12 +15,12 @@ import { /** stationsByName へ渡す件数(設計値。全量を返すとツール結果でトークンを浪費する) */ const STATION_SEARCH_LIMIT = 10; -/** sapi-bff 呼び出しの 1 試行あたり期限 */ +/** 駅検索 API 呼び出しの 1 試行あたり期限 */ const TOOL_TIMEOUT_MS = 5_000; /** 1 ターン合計のツール呼び出し上限 */ export const MAX_TOOL_CALLS_PER_TURN = 5; /** - * ツール 1 回あたりの sapi-bff 呼び出し上限(表記ゆれ候補 + 一過性エラーの再試行の合計)。 + * ツール 1 回あたりの駅検索 API 呼び出し上限(表記ゆれ候補 + 一過性エラーの再試行の合計)。 * 1 試行 5 秒のため、全体 25 秒の予算内に収まる値にする。 */ const MAX_SEARCH_ATTEMPTS = 3; @@ -102,14 +102,15 @@ const postGraphQL = async ( body, signal, }; - if (env.SAPI_BFF) { - // Service Binding はホスト名を解決しないため URL はダミーでよい - return env.SAPI_BFF.fetch('https://sapi-bff/graphql', init); + if (env.STATION_API) { + // Service Binding はホスト名を解決しないため URL はダミーでよい。 + // stationapi は GraphQL をサブドメイン直下(POST /)で受ける + return env.STATION_API.fetch('https://stationapi/', init); } - if (env.SAPI_BFF_GRAPHQL_URL) { - return fetch(env.SAPI_BFF_GRAPHQL_URL, init); + if (env.STATION_API_GRAPHQL_URL) { + return fetch(env.STATION_API_GRAPHQL_URL, init); } - throw new Error('SAPI_BFF binding or SAPI_BFF_GRAPHQL_URL is required'); + throw new Error('STATION_API binding or STATION_API_GRAPHQL_URL is required'); }; const queryStationsOnce = async ( @@ -341,7 +342,7 @@ export interface StationSearchToolResult { } export interface StationSearchToolOptions { - /** 駅名 → 実在駅リスト(sapi-bff 呼び出し。テストでは差し替え可能) */ + /** 駅名 → 実在駅リスト(駅検索 API 呼び出し。テストでは差し替え可能) */ search: (name: string) => Promise; /** このターンで実在確認済みの駅(stationId → 駅)。突合検証の元データ */ verified: Map; diff --git a/src/cli/find-tts-cache.ts b/src/cli/find-tts-cache.ts index e4cebd3..d26e699 100644 --- a/src/cli/find-tts-cache.ts +++ b/src/cli/find-tts-cache.ts @@ -1,5 +1,5 @@ /** - * KV(TTS_KV) の voice:* メタを SSML 本文で検索し、必要なら KV ドキュメントと + * KV(TTS_KV) の voice:* メタを読み上げ本文で検索し、必要なら KV ドキュメントと * R2 上の音声ファイルを削除する。旧 Firestore+GCS 版の Cloudflare 移植。 * * KV の一覧・値取得・削除、R2 の削除、バケット名解決はすべて wrangler @@ -7,7 +7,7 @@ * ネームスペース ID、R2 認証情報を環境変数で渡す必要はない(要 `wrangler login`)。 * * 例: - * npm run find-tts-cache -- "東京" --field ssmlJa + * npm run find-tts-cache -- "東京" --field textJa * npm run find-tts-cache -- "東京" --delete * npm run find-tts-cache -- "東京" --env production --delete */ @@ -26,7 +26,7 @@ const R2_BINDING = 'TTS_BUCKET'; interface CliArgs { searchTerm: string; - field?: 'ssmlJa' | 'ssmlEn'; + field?: 'textJa' | 'textEn'; exact: boolean; delete: boolean; env?: string; @@ -34,12 +34,12 @@ interface CliArgs { function printUsage(): void { console.error( - 'Usage: npm run find-tts-cache -- [--field ssmlJa|ssmlEn] [--exact] [--delete] [--env ]' + 'Usage: npm run find-tts-cache -- [--field textJa|textEn] [--exact] [--delete] [--env ]' ); console.error(''); console.error('Options:'); console.error( - ' --field 検索対象フィールド(省略時は両方)' + ' --field 検索対象フィールド(省略時は両方)' ); console.error(' --exact 部分一致ではなく完全一致で検索'); console.error(' --delete KV ドキュメントと R2 音声を削除'); @@ -57,7 +57,7 @@ function parseArgs(argv: string[]): CliArgs | null { if (args.length === 0) return null; let searchTerm = ''; - let field: 'ssmlJa' | 'ssmlEn' | undefined; + let field: 'textJa' | 'textEn' | undefined; let exact = false; let deleteMode = false; let env: string | undefined; @@ -66,8 +66,8 @@ function parseArgs(argv: string[]): CliArgs | null { switch (args[i]) { case '--field': { const value = args[++i]; - if (value !== 'ssmlJa' && value !== 'ssmlEn') { - console.error('Error: --field は "ssmlJa" か "ssmlEn" を指定'); + if (value !== 'textJa' && value !== 'textEn') { + console.error('Error: --field は "textJa" か "textEn" を指定'); process.exit(1); } field = value; @@ -141,9 +141,14 @@ async function main(): Promise { if (typeof rec.id !== 'string' || rec.id.length === 0) { continue; } + // ssmlJa/ssmlEn は Azure 時代のレコード。旧エントリも掃除できるよう併せて見る const hit = field - ? matchValue(rec[field]) - : matchValue(rec.ssmlJa) || matchValue(rec.ssmlEn); + ? matchValue(rec[field]) || + matchValue(field === 'textJa' ? rec.ssmlJa : rec.ssmlEn) + : matchValue(rec.textJa) || + matchValue(rec.textEn) || + matchValue(rec.ssmlJa) || + matchValue(rec.ssmlEn); if (hit) matches.push(rec); } @@ -155,8 +160,9 @@ async function main(): Promise { console.log(`${matches.length}件のドキュメントが見つかりました:\n`); for (const rec of matches) { console.log(`ID: ${rec.id}`); - console.log(`SSML (JA): ${rec.ssmlJa ?? ''}`); - console.log(`SSML (EN): ${rec.ssmlEn ?? ''}`); + console.log(`Text (JA): ${rec.textJa ?? rec.ssmlJa ?? ''}`); + console.log(`Text (EN): ${rec.textEn ?? rec.ssmlEn ?? ''}`); + console.log(`Model: ${rec.model ?? ''}`); console.log(`Path (JA): ${rec.pathJa ?? ''}`); console.log(`Path (EN): ${rec.pathEn ?? ''}`); console.log(`Voice (JA): ${rec.voiceJa ?? ''}`); diff --git a/src/cli/lib/wrangler.ts b/src/cli/lib/wrangler.ts index 5a5da49..d264767 100644 --- a/src/cli/lib/wrangler.ts +++ b/src/cli/lib/wrangler.ts @@ -219,6 +219,10 @@ export function confirm(prompt: string): Promise { // --- 共有: voice メタの型 --- export interface VoiceCacheRecord { id: string; + model?: string; + textJa?: string; + textEn?: string; + /** Azure/SSML 時代のレコード。旧エントリを検索・削除できるよう残している */ ssmlJa?: string; ssmlEn?: string; pathJa?: string; diff --git a/src/cli/typesafe-rerank-spike.ts b/src/cli/typesafe-rerank-spike.ts new file mode 100644 index 0000000..9ae225d --- /dev/null +++ b/src/cli/typesafe-rerank-spike.ts @@ -0,0 +1,403 @@ +/** + * 提案駅のリランク(src/agent/rerank.ts)をオフラインで実測するスパイク。 + * + * 出したい数字は 3 つ。 + * 1. 正解の駅を上位 5 件で拾えるか(recall@5) + * 2. 無関係な駅を上位 5 件に入れないか(reject 違反) + * 3. 「合う駅が無い」を空配列で表現できるか、そのための閾値が引けるか(分離幅) + * あわせて案 X(候補ごとに 1 リクエスト)と案 Y(1 リクエストに集約)の + * 精度・トークン・レイテンシを比べ、どちらを採るかを決める。 + * + * # 2 段で動かす + * + * 判定の入力になる候補プールは手で書かない。実際の駅検索 + * (src/agent/tools.ts の searchStationsByName)を評価セットの検索語で叩いて作る。 + * + * STATION_API_GRAPHQL_URL=https://... \ + * npm run typesafe-rerank-spike -- --record --out /tmp/rerank-pool.json + * TYPESAFE_API_KEY=... \ + * npm run typesafe-rerank-spike -- --pool /tmp/rerank-pool.json + * + * プールはリポジトリ外(/tmp など)へ置く。生成物であり、コミットしない。 + * + * 対話本体の LLM は通さない。ここで測るのは「プールが与えられたときの選択」なので、 + * LLM の揺れをプールに混ぜると測りたいものが見えなくなる。LLM の選択との比較は + * 本番シャドー(Phase 2)で実トラフィック上で行う。 + * + * # 評価セット(agent-rerank-eval.jsonl)の形 + * + * id 識別子 + * request ユーザ発話(要望そのもの) + * from 現在駅の駅名(任意)。検索は「そこから直通で行ける駅」に絞られる + * queries 候補プールを作る検索語。正解だけでなく紛らわしい語も入れる + * expect 上位 5 件に入るべき駅名(完全一致)。網羅的な正解集合ではない + * reject 上位 5 件に入ってはいけない駅名(完全一致) + * expectEmpty true なら、閾値で全候補が落ちるべき + * note 意図(人間向け。実行では使わない) + * + * expect / reject は手で書いた期待値である。--record で実際のプールを見て、 + * 期待が現実と食い違っていたら評価セット側を直す。プールに存在しない expect は + * 「判定の外し」ではなく「プールに無い」として分母から除く。 + */ +import { readFileSync, writeFileSync } from 'node:fs'; +import { resolve } from 'node:path'; +import { parse as parseJsonc } from 'jsonc-parser'; +import { + type CandidateScore, + judgeCandidates, + type RerankInput, + selectSuggestions, +} from '../agent/rerank'; +import type { StationSuggestion } from '../agent/schema'; +import { searchStationsByName } from '../agent/tools'; +import type { Env } from '../types'; + +const EVAL_PATH = resolve(process.cwd(), 'agent-rerank-eval.jsonl'); + +/** 閾値のグリッド探索の範囲 */ +const THRESHOLDS = [ + 0.2, 0.25, 0.3, 0.35, 0.4, 0.45, 0.5, 0.55, 0.6, 0.65, 0.7, 0.75, 0.8, 0.85, + 0.9, +]; + +type EvalItem = { + id: string; + request: string; + from?: string; + queries: string[]; + expect?: string[]; + reject?: string[]; + expectEmpty?: boolean; + note?: string; +}; + +type PoolItem = EvalItem & { + currentStationName: string | null; + candidates: StationSuggestion[]; +}; + +/** + * モデル名は wrangler.jsonc の vars から読む。ここで文字列を持つと、本体を + * 切り替えたあと var を変えてもスパイクだけ別のモデルを測り続けることになる。 + */ +function resolveModel(): string { + const cfg = parseJsonc( + readFileSync(resolve(process.cwd(), 'wrangler.jsonc'), 'utf8') + ) as { vars?: Record }; + const model = cfg.vars?.TYPESAFE_MODEL; + if (!model) { + throw new Error('wrangler.jsonc の vars に TYPESAFE_MODEL がありません'); + } + return model; +} + +function loadEval(limit: number): EvalItem[] { + const items = readFileSync(EVAL_PATH, 'utf8') + .split(/\r?\n/) + .filter(Boolean) + .map((line) => JSON.parse(line) as EvalItem); + return limit > 0 ? items.slice(0, limit) : items; +} + +// ---- --record: 実際の駅検索でプールを作る ---- + +/** 現在駅の名前から groupId を解決する。見つからなければ nationwide 検索に倒す */ +async function resolveFrom( + env: Env, + name: string | undefined +): Promise<{ groupId: number | undefined; resolved: string | null }> { + if (!name) return { groupId: undefined, resolved: null }; + const hits = await searchStationsByName(env, name, undefined); + const exact = hits.find((s) => s.name === name) ?? hits[0]; + if (!exact) { + console.warn(` 現在駅「${name}」を解決できなかった。全国検索で続行する`); + return { groupId: undefined, resolved: null }; + } + return { groupId: exact.stationGroupId, resolved: exact.name }; +} + +async function record(limit: number, outPath: string | null): Promise { + const url = process.env.STATION_API_GRAPHQL_URL; + if (!url) { + console.error( + 'STATION_API_GRAPHQL_URL が未設定。--record は駅検索 API を直接叩くため必須。' + ); + process.exit(1); + } + const env = { STATION_API_GRAPHQL_URL: url } as unknown as Env; + const items = loadEval(limit); + const pool: PoolItem[] = []; + + for (const item of items) { + const { groupId, resolved } = await resolveFrom(env, item.from); + // 同じ駅が複数の検索語で返るため stationId で重複を除く。順序は検索語の順 + const seen = new Set(); + const candidates: StationSuggestion[] = []; + for (const query of item.queries) { + const hits = await searchStationsByName(env, query, groupId); + for (const hit of hits) { + if (seen.has(hit.stationId)) continue; + seen.add(hit.stationId); + candidates.push(hit); + } + } + pool.push({ ...item, currentStationName: resolved, candidates }); + console.log( + `${item.id.padEnd(28)} 候補 ${String(candidates.length).padStart(3)} 件` + + (resolved ? `(現在駅 ${resolved})` : '') + ); + warnMissingExpectations(item, candidates); + } + + const json = JSON.stringify(pool, null, 2); + if (outPath) { + writeFileSync(outPath, json, 'utf8'); + console.log(`\nプールを ${outPath} に書き出した`); + } else { + console.log(json); + } +} + +/** 期待した駅がプールに入っていなければ、評価セット側の問題として先に知らせる */ +function warnMissingExpectations( + item: EvalItem, + candidates: readonly StationSuggestion[] +): void { + const names = new Set(candidates.map((c) => c.name)); + const missing = (item.expect ?? []).filter((n) => !names.has(n)); + if (missing.length) { + console.warn(` expect がプールに無い: ${missing.join('・')}`); + } +} + +// ---- --eval: 判定して採点する ---- + +type ItemResult = { + item: PoolItem; + scores: CandidateScore[]; + elapsedMs: number; + inputTokens: number; +}; + +async function judgeAll( + pool: PoolItem[], + shape: 'isolated' | 'batched', + apiKey: string, + model: string +): Promise { + const results: ItemResult[] = []; + for (const item of pool) { + // 候補 0 件の項目を通すと judgeCandidates は判定せず [] を返し、 + // picked も [] になるので expectEmpty が無条件で正解になる(閾値を + // 変えても結果が変わらない項目が「空配列の正解」を水増しする)。 + if (item.candidates.length === 0) { + console.warn(`${item.id}: 候補が 0 件(スキップ)`); + continue; + } + const input: RerankInput = { + request: item.request, + currentStationName: item.currentStationName, + }; + let inputTokens = 0; + const startedAt = Date.now(); + const scores = await judgeCandidates(input, item.candidates, { + apiKey, + model, + shape, + onUsage: (usage) => { + inputTokens = usage.inputTokens; + }, + }); + const elapsedMs = Date.now() - startedAt; + if (!scores) { + console.warn(`${item.id}: 判定できなかった(スキップ)`); + continue; + } + results.push({ item, scores, elapsedMs, inputTokens }); + } + return results; +} + +type Grade = { + /** プールに存在する expect のうち上位 5 件で拾えた数 / 分母 */ + recall: { hit: number; total: number }; + /** 上位 5 件に入ってしまった reject の数 */ + violations: number; + /** expectEmpty の判定が正しかったか。対象外なら null */ + emptyOk: boolean | null; + /** min(expect の確率) − max(reject の確率)。両方あるときのみ */ + sep: number | null; + /** プールに無くて分母から除いた expect の数(recall が良く見える分) */ + excludedExpect: number; +}; + +function grade(result: ItemResult, threshold: number): Grade { + const { item, scores } = result; + const picked = selectSuggestions(scores, threshold); + const pickedNames = new Set(picked.map((s) => s.name)); + const poolNames = new Set(item.candidates.map((c) => c.name)); + + const expected = (item.expect ?? []).filter((n) => poolNames.has(n)); + const rejected = (item.reject ?? []).filter((n) => poolNames.has(n)); + + /** + * 同名の別レコード(同一物理駅の路線別レコード。例: JR / 小田急 / 京王の「新宿」) + * は stationId が別なのでプールに全部残る。先頭 1 件だけを見ると、片方が低く + * 片方が高いときに「分離幅は正・違反あり」が同時に出て、質問文を直すべき項目を + * 取りこぼす。expect は最小、reject は最大(どちらも最悪側)を取る。 + */ + const fitsOf = (name: string) => + scores.filter((s) => s.station.name === name).map((s) => s.fits); + const expectedFits = expected.flatMap(fitsOf); + const rejectedFits = rejected.flatMap(fitsOf); + + return { + recall: { + hit: expected.filter((n) => pickedNames.has(n)).length, + total: expected.length, + }, + violations: rejected.filter((n) => pickedNames.has(n)).length, + emptyOk: item.expectEmpty ? picked.length === 0 : null, + sep: + expectedFits.length && rejectedFits.length + ? Math.min(...expectedFits) - Math.max(...rejectedFits) + : null, + excludedExpect: (item.expect ?? []).length - expected.length, + }; +} + +function reportShape(shape: string, results: ItemResult[]): void { + console.log(`\n=== ${shape} ===`); + + console.log('\n--- 閾値ごとの成績 ---'); + console.log('閾値 recall@5 reject違反 空配列の正解'); + for (const threshold of THRESHOLDS) { + const grades = results.map((r) => grade(r, threshold)); + const hit = grades.reduce((a, g) => a + g.recall.hit, 0); + const total = grades.reduce((a, g) => a + g.recall.total, 0); + const violations = grades.reduce((a, g) => a + g.violations, 0); + const emptyTargets = grades.filter((g) => g.emptyOk !== null); + const emptyOk = emptyTargets.filter((g) => g.emptyOk).length; + console.log( + `${threshold.toFixed(2)} ${String(hit).padStart(3)}/${String(total).padEnd(3)} ` + + `${String(violations).padStart(3)} ${emptyOk}/${emptyTargets.length}` + ); + } + + // recall の分母が縮んだ分を明示する。--record の警告を見落としても、 + // 期待した駅がプールに無かったことに気づけるようにする。 + const excluded = results.reduce((a, r) => a + grade(r, 0).excludedExpect, 0); + if (excluded > 0) { + console.log( + `※ プールに無いため分母から除いた expect: ${excluded} 件(評価セットを直すこと)` + ); + } + + // 分離幅は閾値に依存しないので 1 回だけ出す。ここが 0 以下の項目は + // 「どの閾値でも正解と不正解を分けられない」=質問文を直すべき項目。 + console.log('\n--- 分離幅(min(expect) − max(reject))---'); + for (const result of results) { + const { sep } = grade(result, 0); + if (sep === null) continue; + const flag = sep > 0 ? ' ' : '!!'; + console.log(`${flag} ${result.item.id.padEnd(28)} ${sep.toFixed(3)}`); + } + + console.log('\n--- 確率の内訳(上位 5 件)---'); + for (const result of results) { + const top = [...result.scores] + .sort((a, b) => b.fits - a.fits) + .slice(0, 5) + .map((s) => `${s.station.name}:${s.fits.toFixed(2)}`) + .join(' '); + console.log(`${result.item.id.padEnd(28)} ${top}`); + } + + const tokens = results.reduce((a, r) => a + r.inputTokens, 0); + const elapsed = results.reduce((a, r) => a + r.elapsedMs, 0); + const candidates = results.reduce((a, r) => a + r.item.candidates.length, 0); + console.log('\n--- コスト・レイテンシ ---'); + console.log(`候補 ${candidates} 件 / 入力 ${tokens} tok`); + console.log(`概算コスト: $${((tokens * 42) / 1e9).toFixed(6)}($42/Btok)`); + console.log( + `1 項目あたり平均 ${Math.round(elapsed / Math.max(1, results.length))} ms` + ); +} + +// ---- 実行 ---- + +/** --limit は有限の非負整数だけ受ける。0 / 未指定は全件 */ +function parseLimit(raw: string | undefined): number { + if (raw === undefined) return 0; + const limit = Number(raw); + if (!Number.isInteger(limit) || limit < 0) { + console.error(`--limit は 0 以上の整数のみ。受け取った値: ${raw}`); + process.exit(1); + } + return limit; +} + +/** --shape は x / y のみ。未指定は両方測る */ +function parseShapes( + raw: string | undefined, + present: boolean +): readonly ('isolated' | 'batched')[] { + if (!present) return ['batched', 'isolated']; + if (raw === 'x') return ['isolated']; + if (raw === 'y') return ['batched']; + console.error( + `--shape は x(候補ごと)か y(1 リクエスト集約)のみ。受け取った値: ${raw ?? '(なし)'}` + ); + process.exit(1); +} + +async function main(): Promise { + const args = process.argv.slice(2); + const flag = (name: string) => args.includes(name); + const value = (name: string) => { + const index = args.indexOf(name); + return index >= 0 ? args[index + 1] : undefined; + }; + // 引数は有料 API を叩く前に検証する。黙って既定へ倒すと、--limit のタイポ 1 つで + // 全項目 × 両案(案 X は項目ごとに最大 MAX_JUDGED_CANDIDATES 並列)が走る。 + const limit = parseLimit(value('--limit')); + const shapes = parseShapes(value('--shape'), flag('--shape')); + + if (flag('--record')) { + await record(limit, value('--out') ?? null); + return; + } + + const poolPath = value('--pool'); + if (!poolPath) { + console.error( + '--pool が必要(先に --record --out でプールを作る)。' + ); + process.exit(1); + } + const apiKey = process.env.TYPESAFE_API_KEY; + if (!apiKey) { + console.error('TYPESAFE_API_KEY が未設定。'); + process.exit(1); + } + + const pool = (JSON.parse(readFileSync(poolPath, 'utf8')) as PoolItem[]).slice( + 0, + limit > 0 ? limit : undefined + ); + const model = resolveModel(); + + console.log(`対象 ${pool.length} 項目 / model=${model}`); + for (const shape of shapes) { + const results = await judgeAll(pool, shape, apiKey, model); + reportShape( + shape === 'batched' ? '案 Y(1 リクエスト集約)' : '案 X(候補ごと)', + results + ); + } +} + +main().catch((e) => { + console.error(e); + process.exit(1); +}); diff --git a/src/cli/typesafe-triage-spike.ts b/src/cli/typesafe-triage-spike.ts new file mode 100644 index 0000000..9f50925 --- /dev/null +++ b/src/cli/typesafe-triage-spike.ts @@ -0,0 +1,329 @@ +/** + * TypeSafe(System One / Jev)でフィードバックのトリアージ判定ができるかを、 + * fewshot.jsonl の 18 件を正解ラベルとして実測するスパイク。 + * + * 目的は「日本語のフィードバックに対して型付き判定がどれだけ一致するか」と + * 「confidence が公開リポジトリ起票の門(PUBLIC_ISSUE_MIN_CONFIDENCE)として + * 機能する分布になっているか」の 2 点を数字で出すこと。実装の置き換えはしない。 + * + * 注意: fewshot.jsonl は現行 Workers AI 経路に few-shot として与えている例そのもの + * なので、現行モデルにとっては既出であり、ここで出る一致率を現行経路の精度と + * 直接比較してはいけない。TypeSafe にとっては未見のため、TypeSafe 側の数字だけが + * 意味を持つ。 + * + * TypeSafe は判定のみを返し、文章生成は行わない。したがって title / summary は + * 対象外で、判定(isSpam / category / triageLevel / component)だけを見る。 + * + * 例: + * TYPESAFE_API_KEY=... npm run typesafe-spike + * TYPESAFE_API_KEY=... npm run typesafe-spike -- --limit 3 --json + */ +import { readFileSync, writeFileSync } from 'node:fs'; +import { resolve } from 'node:path'; +import { parse as parseJsonc } from 'jsonc-parser'; +import { + compose, + QUESTIONS, + type SystemOneResponse, +} from '../consumers/typesafeTriage'; + +const API_URL = 'https://api.typesafe.ai/v1/systemone'; +const FEWSHOT_PATH = resolve(process.cwd(), 'fewshot.jsonl'); + +/** + * モデル名は wrangler.jsonc の vars から読む。ここで文字列を持つと、本体を + * 切り替えたあと var を変えてもスパイクだけ別のモデルを測り続けることになる。 + */ +function resolveModel(): string { + const cfgPath = resolve(process.cwd(), 'wrangler.jsonc'); + const cfg = parseJsonc(readFileSync(cfgPath, 'utf8')) as { + vars?: Record; + }; + const model = cfg.vars?.TYPESAFE_MODEL; + if (!model) { + throw new Error('wrangler.jsonc の vars に TYPESAFE_MODEL がありません'); + } + return model; +} + +// ---- TypeSafe API の型(docs.typesafe.ai/api) ---- + +// ---- 実行 ---- + +type GoldenItem = { + input: string; + golden: { + isSpam: boolean; + category: string; + triageLevel: string; + /** loadGolden が null を 'unknown' に正規化するため常に string */ + component: string; + }; +}; + +function loadGolden(limit: number): GoldenItem[] { + const lines = readFileSync(FEWSHOT_PATH, 'utf8') + .split(/\r?\n/) + .filter(Boolean); + const items: GoldenItem[] = []; + for (const line of lines) { + const o = JSON.parse(line); + if (!o?.input || !o?.output || o.disabled) continue; + const out = JSON.parse(o.output); + items.push({ + input: String(o.input), + golden: { + isSpam: Boolean(out.isSpam), + category: String(out.category), + triageLevel: String(out.triageLevel), + component: out.component == null ? 'unknown' : String(out.component), + }, + }); + } + return limit > 0 ? items.slice(0, limit) : items; +} + +/** 時間を置けば通る可能性があるステータス(レート制限と過負荷) */ +const RETRYABLE_STATUSES = new Set([429, 529]); + +async function ask( + apiKey: string, + model: string, + feedback: string +): Promise { + const body = JSON.stringify({ + // 本番では report_type / app_version / os / has_stacktrace も名前付きで渡す。 + // fewshot.jsonl には本文しか無いため、ここでは本文のみ。 + state: { feedback }, + model, + questions: QUESTIONS, + }); + let lastError: Error | null = null; + for (let attempt = 0; attempt < 4; attempt++) { + const res = await fetch(API_URL, { + method: 'POST', + headers: { + Authorization: `Bearer ${apiKey}`, + 'Content-Type': 'application/json', + }, + body, + }); + if (res.ok) return (await res.json()) as SystemOneResponse; + const text = await res.text().catch(() => ''); + lastError = new Error(`TypeSafe API ${res.status}: ${text.slice(0, 500)}`); + // 逐次実行なので、後半で落ちるとそこまでの計測が無駄になる。 + // 429 と 529 は時間を置けば通る。 + if (!RETRYABLE_STATUSES.has(res.status)) break; + const retryAfter = Number(res.headers.get('retry-after')); + // ヘッダが無いと get() は null を返し、Number(null) は 0。 + // isFinite(0) は true なので、正値であることまで確かめないと待機しない。 + const waitMs = + Number.isFinite(retryAfter) && retryAfter > 0 + ? retryAfter * 1000 + : 500 * 2 ** attempt; + console.warn(` ${res.status} のため ${waitMs}ms 待って再試行する`); + await new Promise((resolve) => setTimeout(resolve, waitMs)); + } + throw lastError ?? new Error('TypeSafe API の呼び出しに失敗した'); +} + +function pad(s: string, n: number): string { + // 日本語は 2 幅として揃える + const w = [...s].reduce( + (a, c) => a + ((c.codePointAt(0) ?? 0) < 0x80 ? 1 : 2), + 0 + ); + return s + ' '.repeat(Math.max(0, n - w)); +} + +async function main(): Promise { + const apiKey = process.env.TYPESAFE_API_KEY; + if (!apiKey) { + console.error( + 'TYPESAFE_API_KEY が未設定。TYPESAFE_API_KEY=... npm run typesafe-spike のように渡すこと。' + ); + process.exit(1); + } + const args = process.argv.slice(2); + const dumpJson = args.includes('--json'); + const limitIdx = args.indexOf('--limit'); + const rawLimit = limitIdx >= 0 ? args[limitIdx + 1] : undefined; + const limit = rawLimit === undefined ? 0 : Number(rawLimit); + if (limitIdx >= 0 && (!Number.isInteger(limit) || limit <= 0)) { + // Number('foo') も Number(undefined) も 0 になり、loadGolden(0) は全件を返す。 + // 件数を絞ったつもりで全件分の API を呼ぶことになるため、ここで落とす。 + console.error('--limit には正の整数を指定すること'); + process.exit(1); + } + const outIdx = args.indexOf('--out'); + const outPath = outIdx >= 0 ? args[outIdx + 1] : undefined; + if (outIdx >= 0 && (!outPath || outPath.startsWith('--'))) { + // 値が無いと --out が黙って無視され、`--out --json` は `--json` という名前の + // ファイルを作ってしまう。API を呼ぶ前に落とす。 + console.error('--out には書き出し先のパスを指定すること'); + process.exit(1); + } + + const model = resolveModel(); + const items = loadGolden(limit); + console.log(`対象 ${items.length} 件 / model=${model}\n`); + + let okSpam = 0; + let okCat = 0; + let okLevel = 0; + let okComp = 0; + let inTok = 0; + let outTok = 0; + let elapsed = 0; + const compConfidences: number[] = []; + const rows: string[] = []; + /** 閾値のフィッティングを API 再実行なしで行うための生データ */ + const records: unknown[] = []; + const misses: { field: string; input: string; got: string; want: string }[] = + []; + + for (const [i, item] of items.entries()) { + const started = Date.now(); + const res = await ask(apiKey, model, item.input); + elapsed += Date.now() - started; + inTok += res.usage.input_tokens; + outTok += res.usage.output_tokens; + + const v = compose(res.answers); + const g = item.golden; + const mSpam = v.isSpam === g.isSpam; + const mCat = v.category === g.category; + const mLevel = v.triageLevel === g.triageLevel; + const mComp = v.component === g.component; + if (mSpam) okSpam++; + if (mCat) okCat++; + if (mLevel) okLevel++; + if (mComp) okComp++; + if (!v.isSpam) compConfidences.push(v.componentConfidence); + + records.push({ + input: item.input, + golden: g, + answers: res.answers, + verdict: v, + }); + // 1 件ごとに書き出す。逐次実行なので、後半で失敗したときに + // それまでの計測(API を叩いて得た高価なデータ)を失わないようにする。 + if (outPath) + writeFileSync(outPath, JSON.stringify(records, null, 2), 'utf8'); + const head = item.input.slice(0, 40).replace(/\n/g, ' '); + if (!mSpam) { + misses.push({ + field: 'isSpam', + input: head, + got: String(v.isSpam), + want: String(g.isSpam), + }); + } + if (!mCat) { + misses.push({ + field: 'category', + input: head, + got: v.category, + want: g.category, + }); + } + if (!mLevel) { + misses.push({ + field: 'triageLevel', + input: head, + got: v.triageLevel, + want: g.triageLevel, + }); + } + if (!mComp) { + misses.push({ + field: 'component', + input: head, + got: v.component, + want: g.component, + }); + } + + const mark = (ok: boolean) => (ok ? ' ' : '×'); + rows.push( + [ + String(i + 1).padStart(2), + pad(`${item.input.slice(0, 22)}…`, 26), + `${mark(mSpam)}spam=${v.isSpam ? 'T' : 'F'}/${g.isSpam ? 'T' : 'F'}`, + `${mark(mCat)}${pad(`${v.category}/${g.category}`, 30)}`, + `${mark(mLevel)}${pad(`${v.triageLevel}/${g.triageLevel}`, 16)}`, + `${mark(mComp)}${pad(`${v.component}/${g.component}`, 26)}`, + `conf=${v.componentConfidence.toFixed(2)}`, + `sev=${v.severity.toFixed(2)} act=${v.actionability.toFixed(2)}`, + v.needsSpamReview ? 'REVIEW' : '', + ].join(' ') + ); + console.log(rows[rows.length - 1]); + + if (dumpJson) { + console.log( + JSON.stringify( + { input: item.input, answers: res.answers, verdict: v }, + null, + 2 + ) + ); + } + } + + const n = items.length; + const pct = (k: number) => `${((k / n) * 100).toFixed(1)}% (${k}/${n})`; + console.log('\n--- 一致率(正解 = fewshot.jsonl の手当てラベル) ---'); + console.log(`isSpam : ${pct(okSpam)}`); + console.log(`category : ${pct(okCat)}`); + console.log(`triageLevel : ${pct(okLevel)} ※暫定閾値での結果`); + console.log(`component : ${pct(okComp)}`); + + const sorted = [...compConfidences].sort((a, b) => a - b); + const q = (p: number) => + sorted.length + ? sorted[Math.floor((sorted.length - 1) * p)].toFixed(2) + : '-'; + const over = compConfidences.filter((c) => c >= 0.7).length; + console.log('\n--- component の confidence 分布(非スパムのみ) ---'); + console.log( + `min=${q(0)} p25=${q(0.25)} median=${q(0.5)} p75=${q(0.75)} max=${q(1)}` + ); + console.log( + `PUBLIC_ISSUE_MIN_CONFIDENCE(0.7) 以上: ${over}/${compConfidences.length} 件` + ); + + console.log('\n--- 不一致の内訳 ---'); + for (const f of ['isSpam', 'category', 'triageLevel', 'component']) { + const rows = misses.filter((m) => m.field === f); + if (rows.length === 0) continue; + console.log(`\n[${f}] ${rows.length} 件`); + const pairs = new Map(); + for (const m of rows) { + const k = `${m.want} → ${m.got}`; + pairs.set(k, (pairs.get(k) ?? 0) + 1); + console.log(` 正解=${m.want} 判定=${m.got} ${m.input}…`); + } + console.log( + ` 混同: ${[...pairs.entries()] + .sort((a, b) => b[1] - a[1]) + .map(([k, c]) => `${k}(${c})`) + .join(', ')}` + ); + } + + if (outPath) { + console.log(`\n生データを ${outPath} に書き出した(閾値の再計算に使う)`); + } + + console.log('\n--- コスト・レイテンシ ---'); + console.log(`入力 ${inTok} tok / 出力 ${outTok} tok(出力は無課金)`); + console.log(`概算コスト: $${((inTok * 42) / 1e9).toFixed(6)}($42/Btok)`); + console.log(`1 件あたり平均 ${Math.round(elapsed / n)} ms(逐次実行)`); +} + +main().catch((err) => { + console.error(err instanceof Error ? err.message : String(err)); + process.exit(1); +}); diff --git a/src/consumers/feedbackTriage.test.ts b/src/consumers/feedbackTriage.test.ts index 7520ae4..a83fa50 100644 --- a/src/consumers/feedbackTriage.test.ts +++ b/src/consumers/feedbackTriage.test.ts @@ -1,9 +1,25 @@ +import type { AIReport } from '../models/ai'; +import type { Report } from '../models/feedback'; +import type { FeedbackQueueMessage } from '../types'; import { + applySpamHeuristic, buildFailedReport, + buildPublicIssueBody, + buildPublicIssueTitle, coerceReport, extractReportJson, + findBrokenTitleReason, + isUnusableTitle, looksLikeSpam, + MISSING_TITLE, + NON_ACTIONABLE_TITLE, + PUBLIC_ISSUE_MIN_CONFIDENCE, + pickModelResponse, + processFeedbackMessage, + resolvePublicIssueRepo, + SPAM_OVERRIDE_MAX_CONFIDENCE, TRIAGE_FAILED_SUMMARY, + triageMarkerKey, } from './feedbackTriage'; describe('coerceReport', () => { @@ -102,8 +118,14 @@ describe('coerceReport', () => { it('falls back to the default title when both title and summary are empty', () => { const r = coerceReport({}); - expect(r.title).toBe('要約未取得'); - expect(r.summary).toBe('要約未取得'); + expect(r.title).toBe(MISSING_TITLE); + // タイトル未取得を要約に伝播させると両方が同時に無意味になるため、失敗を明示する + expect(r.summary).toBe(TRIAGE_FAILED_SUMMARY); + }); + + it('does not propagate a broken title into the summary', () => { + const r = coerceReport({ title: 'ををををを', summary: '' }); + expect(r.summary).toBe(TRIAGE_FAILED_SUMMARY); }); it('supports question and improvement synonyms', () => { @@ -112,6 +134,14 @@ describe('coerceReport', () => { 'improvement' ); }); + + it('感謝・称賛は praise として扱う(スパムに落とさない)', () => { + for (const raw of ['praise', 'Thanks', ' compliment ', 'gratitude']) { + const r = coerceReport({ title: 't', summary: 's', category: raw }); + expect(r.category).toBe('praise'); + expect(r.isSpam).toBe(false); + } + }); }); describe('looksLikeSpam', () => { @@ -198,3 +228,875 @@ describe('buildFailedReport', () => { expect(r.summary).toBe(TRIAGE_FAILED_SUMMARY); }); }); + +describe('coerceReport(原因コンポーネント)', () => { + it('canonical な component と信頼度を読み取る', () => { + const r = coerceReport({ + title: 't', + summary: 's', + category: 'bug', + component: 'station_api', + componentConfidence: 0.9, + }); + expect(r.component).toBe('station_api'); + expect(r.componentConfidence).toBe(0.9); + }); + + it('表記ゆれを正規化する', () => { + expect(coerceReport({ component: 'MobileApp' }).component).toBe( + 'mobile_app' + ); + expect(coerceReport({ component: ' iOS ' }).component).toBe('mobile_app'); + expect(coerceReport({ component: 'Web' }).component).toBe('website'); + expect(coerceReport({ component: 'workers' }).component).toBe('functions'); + }); + + it('unknown・未知の値・欠落は null にする', () => { + expect(coerceReport({ component: 'unknown' }).component).toBeNull(); + expect(coerceReport({ component: 'なにか' }).component).toBeNull(); + expect(coerceReport({}).component).toBeNull(); + }); + + it('component が特定できないときは信頼度を 0 に落とす', () => { + const r = coerceReport({ component: 'unknown', componentConfidence: 0.95 }); + expect(r.component).toBeNull(); + expect(r.componentConfidence).toBe(0); + }); + + it('component はあるが信頼度が欠落しているときは 0 とみなす', () => { + const r = coerceReport({ component: 'functions' }); + expect(r.componentConfidence).toBe(0); + }); + + it('0..1 の範囲外の信頼度は信用せず既定値に倒す', () => { + // パーセント表記(90)や負値は、そのまま通すと閾値判定をすり抜けて + // 公開リポジトリへ起票されてしまう + for (const bad of [90, 1.2, -0.5]) { + const r = coerceReport({ + component: 'station_api', + // category を省くと question 扱いになり、信頼度に到達する前に弾かれてしまう + category: 'bug', + componentConfidence: bad, + confidence: bad, + }); + expect(r.componentConfidence).toBe(0); + expect(r.confidence).toBe(0.5); + expect( + resolvePublicIssueRepo(r, { + reportType: 'feedback', + triageFailed: false, + }) + ).toBeNull(); + } + }); + + it('数値でない信頼度は既定値に倒す(Number() の型強制を通さない)', () => { + // Number(true) === 1、Number([1]) === 1 なので、型で絞らないと閾値を通過する + for (const bad of [true, [1], null, '', ' ', {}]) { + const r = coerceReport({ + component: 'station_api', + category: 'bug', + componentConfidence: bad, + confidence: bad, + }); + expect(r.componentConfidence).toBe(0); + expect(r.confidence).toBe(0.5); + expect( + resolvePublicIssueRepo(r, { + reportType: 'feedback', + triageFailed: false, + }) + ).toBeNull(); + } + }); + + it('数値だけの文字列は受け付ける', () => { + const r = coerceReport({ + component: 'station_api', + category: 'bug', + componentConfidence: '0.9', + }); + expect(r.componentConfidence).toBe(0.9); + }); + + it('境界値の 0 と 1 は受け付ける', () => { + expect( + coerceReport({ component: 'website', componentConfidence: 1 }) + .componentConfidence + ).toBe(1); + expect( + coerceReport({ component: 'website', componentConfidence: 0 }) + .componentConfidence + ).toBe(0); + }); +}); + +const baseReport = (overrides: Partial = {}): AIReport => ({ + title: 'タイトル', + summary: 'サマリ', + isSpam: false, + labels: [], + confidence: 0.9, + reason: 'reason', + category: 'bug', + triageLevel: 'high', + component: 'mobile_app', + componentConfidence: 0.9, + ...overrides, +}); + +describe('resolvePublicIssueRepo', () => { + const opts = { reportType: 'feedback' as const, triageFailed: false }; + + it('原因コンポーネントに対応する公開リポジトリを返す', () => { + expect(resolvePublicIssueRepo(baseReport(), opts)).toBe( + 'TrainLCD/MobileApp' + ); + expect( + resolvePublicIssueRepo(baseReport({ component: 'station_api' }), opts) + ).toBe('TrainLCD/StationAPI'); + expect( + resolvePublicIssueRepo(baseReport({ component: 'functions' }), opts) + ).toBe('TrainLCD/Functions'); + expect( + resolvePublicIssueRepo(baseReport({ component: 'website' }), opts) + ).toBe('TrainLCD/Website'); + }); + + it('改善・要望も対象にする', () => { + expect( + resolvePublicIssueRepo(baseReport({ category: 'improvement' }), opts) + ).toBe('TrainLCD/MobileApp'); + expect( + resolvePublicIssueRepo(baseReport({ category: 'feature_request' }), opts) + ).toBe('TrainLCD/MobileApp'); + }); + + it('原因が特定できていなければ起票しない', () => { + expect( + resolvePublicIssueRepo( + baseReport({ component: null, componentConfidence: 0 }), + opts + ) + ).toBeNull(); + }); + + it('信頼度が閾値未満なら起票しない', () => { + expect( + resolvePublicIssueRepo( + baseReport({ componentConfidence: PUBLIC_ISSUE_MIN_CONFIDENCE - 0.01 }), + opts + ) + ).toBeNull(); + expect( + resolvePublicIssueRepo( + baseReport({ componentConfidence: PUBLIC_ISSUE_MIN_CONFIDENCE }), + opts + ) + ).toBe('TrainLCD/MobileApp'); + }); + + it('スパム・質問・トリアージ失敗・クラッシュは起票しない', () => { + expect( + resolvePublicIssueRepo(baseReport({ isSpam: true }), opts) + ).toBeNull(); + expect( + resolvePublicIssueRepo(baseReport({ category: 'question' }), opts) + ).toBeNull(); + expect( + resolvePublicIssueRepo(baseReport(), { ...opts, triageFailed: true }) + ).toBeNull(); + expect( + resolvePublicIssueRepo(baseReport(), { ...opts, reportType: 'crash' }) + ).toBeNull(); + }); +}); + +describe('公開リポジトリ用の Issue 本文', () => { + const ticketId = 'a1b2c3d4-0000-4444-8888-abcdefabcdef'; + const body = buildPublicIssueBody({ internalIssueNumber: 123, ticketId }); + + it('管理 Issue 番号とチケットIDを紐づける', () => { + expect(buildPublicIssueTitle(123)).toContain('TrainLCD/Issues#123'); + expect(body).toContain('TrainLCD/Issues#123'); + expect(body).toContain(ticketId); + }); + + it('フィードバック由来の情報を一切含めない', () => { + const report = baseReport({ + title: '駅名が誤って表示される', + summary: '山手線で駅名が1つずれて表示されるという報告', + }); + const rendered = `${buildPublicIssueTitle(123)}\n${body}`; + expect(rendered).not.toContain(report.title); + expect(rendered).not.toContain(report.summary); + expect(rendered).not.toContain(report.reason); + }); +}); + +describe('looksLikeSpam(正当な報告の誤判定)', () => { + // 実フィードバックは非公開のため、同等の語彙構成を持つ合成文で確認する + it('停車駅・方面・路線名を含むデータ不備の報告をスパムにしない', () => { + expect( + looksLikeSpam('架空線の停車駅が違います。仮駅と例駅にも停車するはずです') + ).toBe(false); + expect(looksLikeSpam('行き先方面の案内が実際と異なります')).toBe(false); + expect(looksLikeSpam('架空線の停車駅、仮駅・例駅が抜けています')).toBe( + false + ); + expect(looksLikeSpam('種別が反映されていないようです')).toBe(false); + expect(looksLikeSpam('乗り換え路線を追加してほしいです')).toBe(false); + expect(looksLikeSpam('駅ナンバリングの表記がおかしいです')).toBe(false); + }); + + it('放送定型句を伴わない停車駅・方面の言及だけでは加点しない', () => { + // ACTIONABLE に一致しない書き方でも、放送の書き起こしでなければスパムにしない + expect(looksLikeSpam('架空線の停車駅と方面の情報について')).toBe(false); + }); + + it('車内放送の書き起こしは引き続きスパムとして扱う', () => { + expect( + looksLikeSpam( + '次は仮駅、仮駅です。お出口は左側です。ご利用ありがとうございます。' + ) + ).toBe(true); + expect( + looksLikeSpam( + '次は仮駅方面、停車駅は例駅、見本駅です。お乗り換えのご案内' + ) + ).toBe(true); + }); +}); + +describe('applySpamHeuristic', () => { + const notSpam = (confidence: number): AIReport => ({ + title: '停車駅の誤りについて', + summary: 'サマリ', + isSpam: false, + labels: ['bug'], + confidence, + reason: 'reason', + category: 'bug', + triageLevel: 'high', + component: 'station_api', + componentConfidence: 0.9, + }); + const transcript = + '次は仮駅、仮駅です。お出口は左側です。ご利用ありがとうございます。'; + + it('モデルが確信を持って非スパムと判定していれば分類を維持し、人手確認に回す', () => { + const { report, needsSpamReview } = applySpamHeuristic( + notSpam(SPAM_OVERRIDE_MAX_CONFIDENCE), + transcript, + { triageFailed: false } + ); + expect(needsSpamReview).toBe(true); + expect(report.isSpam).toBe(false); + expect(report.title).toBe('停車駅の誤りについて'); + expect(report.labels).toEqual(['bug']); + expect(report.category).toBe('bug'); + }); + + it('モデルの確信度が低い場合はヒューリスティックでスパムに倒す', () => { + const { report, needsSpamReview } = applySpamHeuristic( + notSpam(SPAM_OVERRIDE_MAX_CONFIDENCE - 0.01), + transcript, + { triageFailed: false } + ); + expect(needsSpamReview).toBe(false); + expect(report.isSpam).toBe(true); + expect(report.title).toBe(NON_ACTIONABLE_TITLE); + expect(report.labels).toEqual([]); + }); + + it('正当な報告には何もしない', () => { + const input = notSpam(0.9); + const { report, needsSpamReview } = applySpamHeuristic( + input, + '架空線の停車駅が違います', + { triageFailed: false } + ); + expect(needsSpamReview).toBe(false); + expect(report).toBe(input); + }); + + it('トリアージ失敗のレポートは上書きしない(失敗の事実を残す)', () => { + const failed = buildFailedReport(transcript, 72); + const { report, needsSpamReview } = applySpamHeuristic(failed, transcript, { + triageFailed: true, + }); + expect(needsSpamReview).toBe(false); + expect(report).toBe(failed); + expect(report.summary).toBe(TRIAGE_FAILED_SUMMARY); + }); + + it('モデル自身がスパムと判定したものはそのまま', () => { + const spam = { ...notSpam(0.9), isSpam: true }; + const { report, needsSpamReview } = applySpamHeuristic(spam, transcript, { + triageFailed: false, + }); + expect(needsSpamReview).toBe(false); + expect(report).toBe(spam); + }); +}); + +describe('findBrokenTitleReason', () => { + it('正常な日本語タイトルは通す', () => { + for (const title of [ + '自動アナウンスが途中で停止する不具合', + // 異なる助詞の連結は正常な日本語(誤検知の回帰テスト) + '駅名が反映されないのでは?という報告', + 'そのものには問題がない旨の報告', + '路線図のダークモード対応要望', + '特定駅が検索に出ず駅名表記も誤り', + 'オートモード時に駅ナンバリングがずれる', + 'Auto mode stops announcing station names', + ]) { + expect(findBrokenTitleReason(title)).toBeNull(); + } + }); + + it('未取得・空・記号のみを検知する', () => { + expect(findBrokenTitleReason(MISSING_TITLE)).toBe('missing'); + expect(findBrokenTitleReason('')).toBe('empty'); + expect(findBrokenTitleReason(' ')).toBe('empty'); + expect(findBrokenTitleReason('!!!…')).toBe('no_word_char'); + }); + + it('破損した生成結果を検知する', () => { + expect(findBrokenTitleReason('駅名が\uFFFD示される')).toBe( + 'replacement_char' + ); + expect(findBrokenTitleReason('繧医↓縺ゅk陦ィ遉コ')).toBe('mojibake_kanji'); + expect(findBrokenTitleReason('駅名ををををが変')).toBe('char_repeat'); + expect(findBrokenTitleReason('表示表示表示がおかしい')).toBe( + 'phrase_repeat' + ); + expect(findBrokenTitleReason('駅名ををを変わる')).toBe('particle_run'); + expect(findBrokenTitleReason('역명이 잘못 표시됨')).toBe('foreign_script'); + }); + + it('isUnusableTitle は真偽値を返す', () => { + expect(isUnusableTitle('正常なタイトル')).toBe(false); + expect(isUnusableTitle(MISSING_TITLE)).toBe(true); + }); +}); + +describe('pickModelResponse', () => { + it('従来の Workers AI 形式(response)を取り出す', () => { + expect(pickModelResponse({ response: { title: 't' } })).toEqual({ + title: 't', + }); + expect(pickModelResponse({ response: '{"title":"t"}' })).toBe( + '{"title":"t"}' + ); + }); + + it('OpenAI 互換形式(choices[0].message.content)を取り出す', () => { + // gemma-4 系は response を返さず choices のみ。ここを見落とすと全件失敗する + const raw = { + choices: [ + { + finish_reason: 'stop', + message: { content: '{"title":"駅名がずれる"}', reasoning: '...' }, + }, + ], + usage: { completion_tokens: 892 }, + }; + expect(pickModelResponse(raw)).toBe('{"title":"駅名がずれる"}'); + }); + + it('response があればそちらを優先する', () => { + const raw = { + response: { title: 'A' }, + choices: [{ message: { content: '{"title":"B"}' } }], + }; + expect(pickModelResponse(raw)).toEqual({ title: 'A' }); + }); + + it('取り出せない形は null', () => { + expect(pickModelResponse(null)).toBeNull(); + expect(pickModelResponse('text')).toBeNull(); + expect(pickModelResponse({})).toBeNull(); + expect(pickModelResponse({ choices: [] })).toBeNull(); + expect(pickModelResponse({ choices: [{ message: {} }] })).toBeNull(); + }); + + it('取り出した文字列は既存の JSON 抽出でパースできる', () => { + const content = '{\n "title": "駅名がずれる",\n "isSpam": false\n}'; + const picked = pickModelResponse({ + choices: [{ message: { content } }], + }) as string; + expect(extractReportJson(picked)).toEqual({ + title: '駅名がずれる', + isSpam: false, + }); + }); +}); + +describe('processFeedbackMessage(再試行時の冪等化)', () => { + const ISSUES_API = 'https://api.github.com/repos/TrainLCD/Issues/issues'; + const CS_WEBHOOK = 'https://discord.example.com/webhooks/cs'; + + const AI_JSON = JSON.stringify({ + title: 'タイトル', + summary: '要約', + isSpam: false, + labels: [], + confidence: 0.9, + reason: '理由', + category: 'question', + triageLevel: 'medium', + component: null, + componentConfidence: 0, + }); + + const report: Report = { + id: 'report-1', + reportType: 'feedback', + description: '駅の表示がおかしいので直してほしいです', + stacktrace: undefined, + resolved: false, + resolvedReason: '', + language: 'ja-JP', + appVersion: '1.0.0', + deviceInfo: null, + resolverUid: '', + createdAt: 1_700_000_000_000, + updatedAt: 1_700_000_000_000, + reporterUid: 'uid-1', + imageUrl: null, + appEdition: 'production', + appClip: false, + autoModeEnabled: false, + }; + + // KV の同一キー書き込み制限を守るための待ちがテスト間に持ち越されないよう、 + // レポートIDはテストごとに変える。 + let seq = 0; + const makeMessage = (): FeedbackQueueMessage => { + seq += 1; + return { + id: `msg-${seq}`, + receivedAt: '2024-01-01T00:00:00.000Z', + report: { ...report, id: `report-${seq}` }, + version: 1, + }; + }; + + // biome-ignore lint/suspicious/noExplicitAny: テスト用の最小 Env スタブ + type TestEnv = any; + + const createEnv = (): { env: TestEnv; store: Map } => { + const store = new Map(); + return { + env: { + AI: { run: jest.fn().mockResolvedValue({ response: AI_JSON }) }, + CONFIG_KV: { + get: jest + .fn() + .mockResolvedValue('{"input":"入力例","output":"出力例"}'), + }, + STATE_KV: { + get: jest.fn(async (key: string) => store.get(key) ?? null), + put: jest.fn(async (key: string, value: string) => { + store.set(key, value); + }), + }, + AI_TRIAGE_MODEL: 'test-model', + FEW_SHOT_KV_KEY: 'fewshot', + FEW_SHOT_LIMIT: '1', + FEW_SHOT_PER_EX_MAX: '800', + OCTOKIT_PAT: 'pat', + DISCORD_CS_WEBHOOK_URL: CS_WEBHOOK, + DISCORD_CRASH_WEBHOOK_URL: '', + }, + store, + }; + }; + + const marker = (overrides: Record = {}) => + JSON.stringify({ + version: 1, + issueNumber: 42, + issueUrl: 'https://github.com/TrainLCD/Issues/issues/42', + publicIssueUrl: null, + aiReport: JSON.parse(AI_JSON), + triageFailed: false, + needsSpamReview: false, + notified: false, + updatedAt: '2024-01-01T00:00:00.000Z', + ...overrides, + }); + + const originalFetch = global.fetch; + let errorSpy: jest.SpyInstance; + let warnSpy: jest.SpyInstance; + + beforeEach(() => { + errorSpy = jest.spyOn(console, 'error').mockImplementation(() => {}); + warnSpy = jest.spyOn(console, 'warn').mockImplementation(() => {}); + }); + + afterEach(() => { + global.fetch = originalFetch; + errorSpy.mockRestore(); + warnSpy.mockRestore(); + }); + + const githubCalls = (fetchMock: jest.Mock) => + fetchMock.mock.calls.filter((c) => String(c[0]) === ISSUES_API); + const discordCalls = (fetchMock: jest.Mock) => + fetchMock.mock.calls.filter((c) => String(c[0]) === CS_WEBHOOK); + + it('起票直後と通知後のマーカー保存を 1 秒以上空ける(KV の同一キー制限)', async () => { + const msg = makeMessage(); + const { env } = createEnv(); + const writeAt: number[] = []; + const originalPut = env.STATE_KV.put; + env.STATE_KV.put = jest.fn(async (key: string, value: string) => { + writeAt.push(Date.now()); + return originalPut(key, value); + }); + const fetchMock = jest.fn(async (input: unknown) => { + if (String(input) === ISSUES_API) { + return new Response( + JSON.stringify({ + html_url: 'https://github.com/TrainLCD/Issues/issues/42', + number: 42, + }), + { status: 201 } + ); + } + return new Response(null, { status: 204 }); + }); + global.fetch = fetchMock as unknown as typeof fetch; + + await processFeedbackMessage(msg, env); + + expect(writeAt).toHaveLength(2); + expect(writeAt[1] - writeAt[0]).toBeGreaterThanOrEqual(1000); + }); + + it('Discord への fetch が throw したら未通知として再試行に回す(起票は 1 回だけ)', async () => { + const msg = makeMessage(); + const { env, store } = createEnv(); + const fetchMock = jest.fn(async (input: unknown) => { + if (String(input) === ISSUES_API) { + return new Response( + JSON.stringify({ + html_url: 'https://github.com/TrainLCD/Issues/issues/42', + number: 42, + }), + { status: 201 } + ); + } + throw new TypeError('network error'); + }); + global.fetch = fetchMock as unknown as typeof fetch; + + await expect(processFeedbackMessage(msg, env)).rejects.toThrow( + 'Discord notification failed' + ); + + expect(githubCalls(fetchMock)).toHaveLength(1); + const saved = JSON.parse(store.get(triageMarkerKey(msg.report.id)) ?? '{}'); + expect(saved.issueNumber).toBe(42); + // 未通知のまま残し、再試行では起票を飛ばして通知だけやり直す + expect(saved.notified).toBe(false); + }); + + it('通知にもマーカー保存にも失敗したら再試行しない(重複起票に戻るため)', async () => { + const msg = makeMessage(); + const { env } = createEnv(); + env.STATE_KV.put = jest.fn(async () => { + throw new Error('KV unavailable'); + }); + const fetchMock = jest.fn(async (input: unknown) => { + if (String(input) === ISSUES_API) { + return new Response( + JSON.stringify({ + html_url: 'https://github.com/TrainLCD/Issues/issues/42', + number: 42, + }), + { status: 201 } + ); + } + throw new TypeError('network error'); + }); + global.fetch = fetchMock as unknown as typeof fetch; + + await expect(processFeedbackMessage(msg, env)).resolves.toBeUndefined(); + + expect(githubCalls(fetchMock)).toHaveLength(1); + }); + + it('Discord が HTTP エラーを返したときも未通知のまま記録する', async () => { + const msg = makeMessage(); + const { env, store } = createEnv(); + store.set(triageMarkerKey(msg.report.id), marker()); + const fetchMock = jest.fn( + async () => new Response('rate limited', { status: 429 }) + ); + global.fetch = fetchMock as unknown as typeof fetch; + + await expect(processFeedbackMessage(msg, env)).rejects.toThrow( + 'Discord notification failed' + ); + + expect(discordCalls(fetchMock)).toHaveLength(1); + expect( + JSON.parse(store.get(triageMarkerKey(msg.report.id)) ?? '{}').notified + ).toBe(false); + }); + + it('起票済みマーカーがあれば Issue を作り直さず、通知だけやり直す', async () => { + const msg = makeMessage(); + const { env, store } = createEnv(); + store.set(triageMarkerKey(msg.report.id), marker()); + const fetchMock = jest.fn(async () => new Response(null, { status: 204 })); + global.fetch = fetchMock as unknown as typeof fetch; + + await processFeedbackMessage(msg, env); + + expect(githubCalls(fetchMock)).toHaveLength(0); + // 再試行でトリアージをやり直すと Issue と通知の内容がずれるため、AI も呼ばない + expect(env.AI.run).not.toHaveBeenCalled(); + expect(discordCalls(fetchMock)).toHaveLength(1); + expect( + JSON.parse(store.get(triageMarkerKey(msg.report.id)) ?? '{}').notified + ).toBe(true); + }); + + it('マーカー保存が一度失敗しても書き直し、throw しない', async () => { + const msg = makeMessage(); + const { env, store } = createEnv(); + store.set(triageMarkerKey(msg.report.id), marker()); + let puts = 0; + env.STATE_KV.put = jest.fn(async (key: string, value: string) => { + puts += 1; + if (puts === 1) throw new Error('KV unavailable'); + store.set(key, value); + }); + const fetchMock = jest.fn(async () => new Response(null, { status: 204 })); + global.fetch = fetchMock as unknown as typeof fetch; + + await expect(processFeedbackMessage(msg, env)).resolves.toBeUndefined(); + + expect(puts).toBe(2); + expect( + JSON.parse(store.get(triageMarkerKey(msg.report.id)) ?? '{}').notified + ).toBe(true); + }); + + it('通知まで完了したマーカーがあれば何もしない', async () => { + const msg = makeMessage(); + const { env, store } = createEnv(); + store.set(triageMarkerKey(msg.report.id), marker({ notified: true })); + const fetchMock = jest.fn(async () => new Response(null, { status: 204 })); + global.fetch = fetchMock as unknown as typeof fetch; + + await processFeedbackMessage(msg, env); + + expect(fetchMock).not.toHaveBeenCalled(); + expect(env.AI.run).not.toHaveBeenCalled(); + expect(env.STATE_KV.put).not.toHaveBeenCalled(); + }); + + it('起票前の失敗は再送出し、マーカーを残さない(メッセージを失わないため)', async () => { + const msg = makeMessage(); + const { env, store } = createEnv(); + const fetchMock = jest.fn( + async () => new Response('{"message":"boom"}', { status: 500 }) + ); + global.fetch = fetchMock as unknown as typeof fetch; + + await expect(processFeedbackMessage(msg, env)).rejects.toThrow( + 'GitHub API failed with status 500' + ); + expect(store.has(triageMarkerKey(msg.report.id))).toBe(false); + }); +}); + +describe('TypeSafe の判定を起票に反映する', () => { + const ISSUES_API = 'https://api.github.com/repos/TrainLCD/Issues/issues'; + const TYPESAFE_API = 'https://api.typesafe.ai/v1/systemone'; + + const report: Report = { + id: 'ts-report', + reportType: 'feedback', + description: '特急の停車駅が実際と違います', + stacktrace: undefined, + resolved: false, + resolvedReason: '', + language: 'ja-JP', + appVersion: '1.0.0', + deviceInfo: null, + resolverUid: '', + createdAt: 1_700_000_000_000, + updatedAt: 1_700_000_000_000, + reporterUid: 'uid-1', + imageUrl: null, + appEdition: 'production', + appClip: false, + autoModeEnabled: false, + }; + + const noul = (v: number) => ({ type: 'noul', noul: v }); + const typesafeBody = (over: Record = {}) => + JSON.stringify({ + model: 'jev-1.13.0', + answers: { + is_spam: noul(0.02), + is_announcement_transcript: noul(0.03), + is_praise_only: noul(0.01), + is_crash_or_data_loss: noul(0.02), + mentions_station_data: noul(0.95), + category: { + type: 'choice', + choice: 'bug', + probabilities: {}, + confidence: 0.91, + }, + component: { + type: 'choice', + choice: 'station_api', + probabilities: {}, + confidence: 0.86, + }, + severity: { + type: 'score', + score: 1.9, + legend: {}, + probabilities: {}, + confidence: 0.8, + }, + actionability: { + type: 'score', + score: 1.8, + legend: {}, + probabilities: {}, + confidence: 0.8, + }, + ...over, + }, + usage: { input_tokens: 1, output_tokens: 1 }, + }); + + // biome-ignore lint/suspicious/noExplicitAny: テスト用の最小 Env スタブ + type TestEnv = any; + const createEnv = (): TestEnv => ({ + AI: { + run: jest.fn().mockResolvedValue({ + response: JSON.stringify({ title: '停車駅の誤り', summary: '要約' }), + }), + }, + CONFIG_KV: { + get: jest + .fn() + .mockResolvedValue( + '{"input":"入力例","output":"{\\"title\\":\\"t\\",\\"summary\\":\\"s\\"}"}' + ), + }, + STATE_KV: { get: jest.fn().mockResolvedValue(null), put: jest.fn() }, + AI_TRIAGE_MODEL: 'model', + FEW_SHOT_KV_KEY: 'config:fewshot', + FEW_SHOT_LIMIT: '4', + FEW_SHOT_PER_EX_MAX: '800', + TYPESAFE_API_KEY: 'key', + TYPESAFE_MODEL: 'jev-latest', + OCTOKIT_PAT: 'pat', + DISCORD_CS_WEBHOOK_URL: 'https://discord.example.com/webhooks/cs', + }); + + const originalFetch = global.fetch; + afterEach(() => { + global.fetch = originalFetch; + jest.clearAllMocks(); + }); + + const run = async (env: TestEnv, typesafe: () => Response) => { + const created: Record[] = []; + global.fetch = jest.fn(async (input: unknown, init?: RequestInit) => { + const url = String(input); + if (url === TYPESAFE_API) return typesafe(); + if (url === ISSUES_API) { + created.push(JSON.parse(String(init?.body))); + return new Response( + JSON.stringify({ html_url: 'https://example.test/1', number: 1 }), + { status: 201 } + ); + } + return new Response(null, { status: 204 }); + }) as unknown as typeof fetch; + await processFeedbackMessage( + { + id: 'msg-ts', + receivedAt: '2024-01-01T00:00:00.000Z', + report, + version: 1, + } as FeedbackQueueMessage, + env + ); + return created; + }; + + it('判定したカテゴリと優先度をラベルに反映する', async () => { + const created = await run( + createEnv(), + () => new Response(typesafeBody(), { status: 200 }) + ); + const labels = created[0]?.labels as string[]; + expect(labels).toContain('🐛 Bug'); + expect(labels).toContain('🟠 P1 / High'); + expect(created[0]?.title).toBe('停車駅の誤り'); + }); + + it('判定を取得できなくても、生成したタイトルを保ったまま起票する', async () => { + const created = await run( + createEnv(), + () => new Response('boom', { status: 500 }) + ); + // 原文もタイトルも失わない。原因を絞り込めなかった扱いにするだけ。 + expect(created[0]?.title).toBe('停車駅の誤り'); + expect(String(created[0]?.body)).toContain('特急の停車駅が実際と違います'); + expect(created[0]?.labels as string[]).toContain('❓ Unknown Type'); + }); + + it('スパム確定ではないが確認が必要な判定は、公開リポジトリへ出さない', async () => { + // compose は「スパム確定ではないが人手確認に回す」場合に needsSpamReview を立てる。 + // これを落とすと resolvePublicIssueRepo のガードが素通りし、確認前の + // フィードバックが公開リポジトリのスタブ Issue になる。 + const created = await run( + createEnv(), + () => + new Response( + typesafeBody({ + // スパム確定(0.5)には届かないが確認下限(0.3)は超える + is_spam: noul(0.4), + is_praise_only: noul(0.05), + }), + { status: 200 } + ) + ); + expect(created[0]?.labels as string[]).toContain('❓ Unknown Type'); + // 非公開リポジトリへの起票だけで、公開リポジトリへは出ていない + expect(created).toHaveLength(1); + }); + + it('スパムと判定されたらタイトルを伏せてスパムラベルを付ける', async () => { + const created = await run( + createEnv(), + () => + new Response( + typesafeBody({ + is_spam: noul(0.95), + is_announcement_transcript: noul(0.9), + }), + { status: 200 } + ) + ); + expect(created[0]?.title).toBe(NON_ACTIONABLE_TITLE); + expect(created[0]?.labels as string[]).toContain('💩 Spam'); + }); +}); diff --git a/src/consumers/feedbackTriage.ts b/src/consumers/feedbackTriage.ts index 555a755..e497266 100644 --- a/src/consumers/feedbackTriage.ts +++ b/src/consumers/feedbackTriage.ts @@ -5,12 +5,55 @@ import dayjs from 'dayjs'; import type { AICategory, + AIComponent, AIReport, AITriageLevel, FewShotItem, } from '../models/ai'; import type { DiscordEmbed } from '../models/common'; +import type { Report } from '../models/feedback'; import type { Env, FeedbackQueueMessage } from '../types'; +import { judgeFeedback, type Verdict } from './typesafeTriage'; + +/** フィードバック原本を保管する非公開リポジトリ */ +const INTERNAL_REPO = 'TrainLCD/Issues'; + +/** + * 原因コンポーネント → 起票先の公開リポジトリ。 + * 公開リポジトリなのでフィードバックの内容は一切載せず、非公開の管理チケットへの + * 参照(Issue 番号・チケットID)だけを持つスタブ Issue を立てる。 + */ +const COMPONENT_REPOS: Record = { + mobile_app: 'TrainLCD/MobileApp', + station_api: 'TrainLCD/StationAPI', + functions: 'TrainLCD/Functions', + website: 'TrainLCD/Website', +}; + +/** 原因コンポーネントを信用して公開リポジトリに起票する最低信頼度 */ +export const PUBLIC_ISSUE_MIN_CONFIDENCE = 0.7; + +/** TypeSafe の判定取得を諦めるまでの試行回数 */ +const MAX_JUDGE_ATTEMPTS = 3; + +/** カテゴリと原因コンポーネントから GitHub のラベル名以外の分類ラベルを導く */ +export function deriveLabels(judgment: Verdict): string[] { + if (judgment.isSpam) return []; + const labels: string[] = []; + if (judgment.category === 'bug') labels.push('bug'); + if (judgment.category === 'improvement') labels.push('improvement'); + if (judgment.category === 'feature_request') labels.push('feature'); + if (judgment.component === 'station_api') labels.push('location'); + if (judgment.component === 'functions') labels.push('network'); + return labels; +} + +/** 公開リポジトリへ転記する対象カテゴリ(質問は原因特定の対象外) */ +const PUBLIC_ISSUE_CATEGORIES: readonly AICategory[] = [ + 'bug', + 'improvement', + 'feature_request', +]; const GITHUB_LABELS = { PLATFORM_IOS: '🍎 iOS', @@ -29,6 +72,7 @@ const GITHUB_LABELS = { CATEGORY_FEATURE_REQUEST: '✨ Feature Request', CATEGORY_IMPROVEMENT: '🛠️ Improvement', CATEGORY_QUESTION: '❓ Question', + CATEGORY_PRAISE: '💚 Praise', TRIAGE_URGENT: '🔴 P0 / Urgent', TRIAGE_HIGH: '🟠 P1 / High', TRIAGE_MEDIUM: '🟡 P2 / Medium', @@ -40,6 +84,7 @@ const CATEGORY_LABELS: Record = { feature_request: GITHUB_LABELS.CATEGORY_FEATURE_REQUEST, improvement: GITHUB_LABELS.CATEGORY_IMPROVEMENT, question: GITHUB_LABELS.CATEGORY_QUESTION, + praise: GITHUB_LABELS.CATEGORY_PRAISE, }; const TRIAGE_LABELS: Record = { @@ -63,6 +108,35 @@ const CATEGORY_SYNONYMS: Record = { question: 'question', support: 'question', help: 'question', + praise: 'praise', + thanks: 'praise', + thankyou: 'praise', + gratitude: 'praise', + compliment: 'praise', + kudos: 'praise', + positive: 'praise', +}; + +// モデルは enum 外の表記("MobileApp" / "ios" / "web" など)を返すことがあるため、 +// 区切り文字を除去したキーで正規化する。未知の値は null(=原因未特定)に落とす。 +const COMPONENT_SYNONYMS: Record = { + mobileapp: 'mobile_app', + mobile: 'mobile_app', + app: 'mobile_app', + client: 'mobile_app', + ios: 'mobile_app', + android: 'mobile_app', + stationapi: 'station_api', + station: 'station_api', + stationdata: 'station_api', + functions: 'functions', + function: 'functions', + worker: 'functions', + workers: 'functions', + website: 'website', + web: 'website', + site: 'website', + homepage: 'website', }; const TRIAGE_SYNONYMS: Record = { @@ -79,25 +153,44 @@ const TRIAGE_SYNONYMS: Record = { p3: 'low', }; +/** + * 不具合・要望の報告で使われる語彙。これが出たら車内放送の書き起こしではないと + * 断定してよいので、スコアリングに入る前に非スパムとして返す。 + * 「〜が違います」「反映されない」「〜してほしい」のように、報告者が「不具合」「要望」と + * いう語を使わずに書くケースを取りこぼさないことを重視している。 + */ +const ACTIONABLE = + /(修正|改善|追加|希望|要望|不具合|バグ|誤|間違|違い|違う|反映|表示|保存|再生|遅|遅延|できない|出来ない|できません|出来ません|されない|されません|しない|しません|エラー|落ちる|クラッシュ|重複|ズレ|ずれ|おかしい|ほしい|欲しい|直し|なおし|音がない|読み上げない)/; + +/** + * 車内放送でも報告文でも使われる言い回し。単独では判断できないため早期リターンには + * 使わず、スパムスコアの減点に留める(正当な報告を握りつぶす方が、スパムを 1 件 + * 通すより損失が大きいという方針)。 + */ +const WEAK_ACTIONABLE = /(になります|になっています|になってます|されています)/; + +/** 車内放送の定型句。書き起こし判定の主シグナル */ +const ANNOUNCEMENT_PHRASE = + /(次は|まもなく|この(列車|電車)は|行きです|ご利用ありがとうございます|お出口は(左|右)側です|各駅に(停ま|止ま)ります|お乗り換え)/; + export function looksLikeSpam(text: string): boolean { if (!text) return false; const t = String(text).replace(/\s+/g, ' ').trim(); - const ACTIONABLE = - /(修正|改善|追加|希望|要望|不具合|バグ|誤|間違い|表示|保存|再生|遅|遅延|できない|出来ない|エラー|落ちる|クラッシュ|重複|ズレ|音がない|読み上げない)/; if (ACTIONABLE.test(t)) return false; let score = 0; - if ( - /(次は|まもなく|この(列車|電車)は|行きです|ご利用ありがとうございます|お出口は(左|右)側です|各駅に(停ま|止ま)ります|お乗り換え)/.test( - t - ) - ) { + const hasAnnouncementPhrase = ANNOUNCEMENT_PHRASE.test(t); + if (hasAnnouncementPhrase) { score += 1; } - if (/(停車駅|方面)/.test(t)) score += 1; + // 「停車駅」「方面」「駅名・路線名の併記」はいずれも本アプリのドメイン語彙そのもので、 + // データ不備の報告に普通に現れる。放送の定型句と共起したときだけ書き起こしの + // シグナルとして扱う(単独加点だと、正確な報告ほどスパム判定されてしまう)。 + if (hasAnnouncementPhrase && /(停車駅|方面)/.test(t)) score += 1; if ( + hasAnnouncementPhrase && /([一-龥ァ-ヶー]{2,})(、|,|・|\s)([一-龥ァ-ヶー]{2,})/.test(t) && /(停車|次は|方面)/.test(t) ) { @@ -118,9 +211,76 @@ export function looksLikeSpam(text: string): boolean { if (/[🚃🚇🚈♪🎵]/u.test(t)) { score += 0.5; } + if (WEAK_ACTIONABLE.test(t)) score -= 1; return score >= 2; } +/** モデルがタイトルを返さなかったときの穴埋め文言 */ +export const MISSING_TITLE = '要約未取得'; + +/** + * 日本語として成立しない生成タイトルのパターン。 + * 小型モデルは日本語生成が破綻することがあり、破損タイトルのまま起票すると + * Issue 一覧から内容を判別できなくなる(= バックログの一次スクリーニングが機能しない)。 + * 誤検知するとまともなタイトルまで「要約失敗」に落としてしまうため、 + * 正常な日本語では起こり得ないものだけを列挙する。 + */ +const BROKEN_TITLE_PATTERNS: readonly { + name: string; + test: (title: string) => boolean; +}[] = [ + // 文字化け(U+FFFD)・制御文字 + { name: 'replacement_char', test: (t) => /\uFFFD/.test(t) }, + { + name: 'control_char', + // biome-ignore lint/suspicious/noControlCharactersInRegex: 制御文字の混入そのものを検知する + test: (t) => /[\u0000-\u0008\u000B\u000C\u000E-\u001F]/.test(t), + }, + // 日本語アプリのタイトルに現れない文字体系(ギリシャ・キリル・ヘブライ・アラビア・ + // デーヴァナーガリー・タイ・ハングル) + { + name: 'foreign_script', + test: (t) => + /[\u0370-\u03FF\u0400-\u04FF\u0590-\u05FF\u0600-\u06FF\u0900-\u097F\u0E00-\u0E7F\uAC00-\uD7AF]/.test( + t + ), + }, + // UTF-8 を Shift_JIS として解釈したときに出る典型的な文字化け漢字。現代日本語では + // ほぼ使われない字なので、連続していなくても 2 文字あれば破損とみなす + { + name: 'mojibake_kanji', + test: (t) => + (t.match(/[縺繧繝蜿蛻髢讌荳陦莠譌蟄蜀隱蠢遘蜷コサ]/g) ?? []).length >= 2, + }, + // 同一文字の 4 連続・同一語の 3 連続(生成ループ) + { name: 'char_repeat', test: (t) => /(.)\1{3,}/u.test(t) }, + { name: 'phrase_repeat', test: (t) => /(.{2,4})\1{2,}/u.test(t) }, + // 同一助詞の 3 連続。「のでは」「のにも」「ものには」のように異なる助詞が連なるのは + // 正常な日本語なので対象にしない(誤検知するとトリアージ結果ごと捨ててしまう) + { name: 'particle_run', test: (t) => /([はがのにをでとへも])\1{2,}/.test(t) }, +]; + +/** + * 生成タイトルが起票に使えない(未取得 or 日本語として破損している)かを判定する。 + * true のときは破損タイトルのまま起票せず、失敗を明示したレポートに倒す。 + */ +export function findBrokenTitleReason(title: string): string | null { + const t = String(title ?? '').trim(); + if (!t) return 'empty'; + if (t === MISSING_TITLE) return 'missing'; + // 記号・空白だけのタイトル + if (!/[\p{L}\p{N}]/u.test(t)) return 'no_word_char'; + for (const { name, test } of BROKEN_TITLE_PATTERNS) { + if (test(t)) return name; + } + return null; +} + +/** findBrokenTitleReason の真偽値版 */ +export function isUnusableTitle(title: string): boolean { + return findBrokenTitleReason(title) !== null; +} + export function coerceReport(raw: unknown, titleMax = 72): AIReport { const norm = (k: string) => String(k).toLowerCase().replace(/\s+/g, '').trim(); @@ -129,9 +289,22 @@ export function coerceReport(raw: unknown, titleMax = 72): AIReport { for (const [k, v] of entries) map.set(norm(k), v); const getStr = (k: string, d = '') => String(map.get(k) ?? d).trim(); - const getNum = (k: string, d = 0.5) => { - const n = Number(map.get(k)); - return Number.isFinite(n) ? n : d; + /** + * 0..1 の信頼度を読む。数値(または数値だけの文字列)以外と範囲外は、 + * スキーマに従っていない応答なので値を信用せず既定値に倒す。 + * Number() 任せにすると true や [1] が 1 に化けてしまうため型で絞る。 + * 特に componentConfidence は公開リポジトリへの起票判定に使うため、 + * 壊れた値をそのまま通すと内容を公開すべきでないものが流出しうる。 + */ + const getRatio = (k: string, d: number) => { + const raw = map.get(k); + const n = + typeof raw === 'number' + ? raw + : typeof raw === 'string' && raw.trim() !== '' + ? Number(raw) + : Number.NaN; + return Number.isFinite(n) && n >= 0 && n <= 1 ? n : d; }; const getBool = (...keys: string[]) => keys.some((k) => { @@ -146,7 +319,7 @@ export function coerceReport(raw: unknown, titleMax = 72): AIReport { const labels: string[] = Array.isArray(rawLabels) ? rawLabels.filter((l): l is string => typeof l === 'string') : []; - const confidence = getNum('confidence', 0.5); + const confidence = getRatio('confidence', 0.5); const reason = getStr('reason'); const categoryKey = getStr('category') .toLowerCase() @@ -156,12 +329,25 @@ export function coerceReport(raw: unknown, titleMax = 72): AIReport { .toLowerCase() .replaceAll(/[\s-]+/g, ''); const triageLevel: AITriageLevel = TRIAGE_SYNONYMS[triageKey] ?? 'medium'; + const componentKey = getStr('component') + .toLowerCase() + .replaceAll(/[\s_-]+/g, ''); + const component: AIComponent | null = + COMPONENT_SYNONYMS[componentKey] ?? null; + // 原因が特定できていないのに信頼度だけ高い、という応答を弾くため component とセットで扱う + const componentConfidence = component + ? getRatio('componentconfidence', 0) + : 0; - if (!title) title = '要約未取得'; + if (!title) title = MISSING_TITLE; if (title.length > titleMax) title = `${title.slice(0, titleMax - 1)}…`; // 要約が空だと Issue 本文の節が空になり、Discord embed も value 空でリジェクトされるため - // タイトルにフォールバックして常に何らかのテキストを入れる - if (!summary) summary = title; + // 常に何らかのテキストを入れる。ただしタイトルが未取得・破損しているときにそれを + // 要約へ伝播させると、タイトルと要約の両方が同時に壊れて内容が判別できなくなるため、 + // その場合は失敗を明示する文言に倒す。 + if (!summary) { + summary = isUnusableTitle(title) ? TRIAGE_FAILED_SUMMARY : title; + } return { title, @@ -172,6 +358,8 @@ export function coerceReport(raw: unknown, titleMax = 72): AIReport { reason, category, triageLevel, + component, + componentConfidence, }; } @@ -256,60 +444,89 @@ export function buildFailedReport( reason: 'triage_failed', category: 'question', triageLevel: 'medium', + component: null, + componentConfidence: 0, + }; +} + +/** + * ヒューリスティックがモデルの非スパム判定を覆せる、モデル側 confidence の上限。 + * これ以上の確信度でモデルが「スパムではない」と言っているときは、ヒューリスティックは + * 上書きせず人手確認のマーカーだけを付ける。 + */ +export const SPAM_OVERRIDE_MAX_CONFIDENCE = 0.5; + +/** スパム上書き時に使う、内容を判別できないことを示すタイトル */ +export const NON_ACTIONABLE_TITLE = '内容未分類(改善要望なし)'; + +/** + * looksLikeSpam の結果をレポートに反映する。 + * + * ヒューリスティックは補助でしかなく、正当な報告を握りつぶすと利用者の声が + * 完全に失われる(ラベルもカテゴリも消えて候補プールから脱落する)。そのため + * モデルが確信を持って「スパムではない」と判定しているときは分類をそのまま残し、 + * 人手確認用のマーカー(needsSpamReview)だけを立てる。 + */ +export function applySpamHeuristic( + aiReport: AIReport, + description: string, + opts: { triageFailed: boolean } +): { report: AIReport; needsSpamReview: boolean } { + // トリアージ自体が失敗しているレポートは、そもそもモデルの判定が無い。 + // ここでスパムに倒すと「要約失敗」の事実が消えるため触らない。 + if (opts.triageFailed) return { report: aiReport, needsSpamReview: false }; + if (aiReport.isSpam) return { report: aiReport, needsSpamReview: false }; + if (!looksLikeSpam(description)) { + return { report: aiReport, needsSpamReview: false }; + } + if (aiReport.confidence >= SPAM_OVERRIDE_MAX_CONFIDENCE) { + return { report: aiReport, needsSpamReview: true }; + } + return { + report: { + ...aiReport, + title: NON_ACTIONABLE_TITLE, + isSpam: true, + labels: [], + reason: 'non-actionable', + }, + needsSpamReview: false, }; } const SYSTEM_PROMPT = ` -You are a precise issue triager for TrainLCD. +You are a precise issue summarizer for TrainLCD. + Task: 1. Summarize the user's message into a ONE-LINE issue title in Japanese (≤72 chars). 2. Also create a 1–3 sentence summary in Japanese that concisely describes the feedback content. -3. Classify spam. -4. If NOT spam, pick ONE primary "category" from ["bug","feature_request","improvement","question"]: - - bug: 不具合・誤動作・クラッシュ・表示崩れ - - feature_request: まだ存在しない機能の新規要望 - - improvement: 既存機能の改善・調整 - - question: 質問・使い方の確認・情報要求 -5. If NOT spam, pick ONE "triageLevel" from ["urgent","high","medium","low"]: - - urgent: クラッシュ・データ消失・広範な実用不能 - - high: 特定機能が使えない/重要機能要望 - - medium: 通常の改善・軽微なバグ - - low: 体裁の問題・質問・軽い要望 - If spam, omit "category" and "triageLevel" entirely. Rules: - Newspaper-style headline: [症状/論点]+[対象](助詞は最小限) - No device/OS/version/URL/stack unless essential - Prefer Japanese if input has Japanese -- If no actionable content (announcement transcript, chit-chat, praise-only), mark spam -- If not spam, pick labels from: - ["bug","improvement","feature","localization","location","ui","performance","network","settings"] +- Summarize whatever the message says, even if it looks like spam, an announcement + transcript or gibberish. Classification is decided elsewhere; do not refuse. Output JSON only: -{"title": "...", "summary": "...", "isSpam": true|false, "labels": [], "category": "...", "triageLevel": "...", "confidence": 0..1, "reason": "..."} +{"title": "...", "summary": "..."} Return ONLY that JSON. No prose, no markdown. `.trim(); // Workers AI の JSON Mode(response_format)に渡すスキーマ。 // summary を required にして「フィールド欠落で要約が空」になるのを構造的に防ぐ。 -// category / triageLevel はスパム時に省ける運用なので optional のまま。 +// スパム時も含めて全フィールドを必須にし、フィールド欠落による既定値落ちを防ぐ +// (スパム判定時の category / triageLevel は起票側で無視する)。 const TRIAGE_JSON_SCHEMA = { type: 'object', properties: { title: { type: 'string' }, summary: { type: 'string' }, - isSpam: { type: 'boolean' }, - labels: { type: 'array', items: { type: 'string' } }, - category: { - type: 'string', - enum: ['bug', 'feature_request', 'improvement', 'question'], - }, - triageLevel: { type: 'string', enum: ['urgent', 'high', 'medium', 'low'] }, - confidence: { type: 'number' }, - reason: { type: 'string' }, }, - required: ['title', 'summary', 'isSpam', 'labels', 'confidence', 'reason'], + // 判定は TypeSafe が担当するため、モデルに出させるのはこの 2 つだけ。 + // いずれも required にして、フィールド欠落で要約が空になるのを構造的に防ぐ。 + required: ['title', 'summary'], } as const; // ---- Few-shot loader(CONFIG_KV) ---- @@ -342,12 +559,33 @@ async function loadFewShot(env: Env): Promise { .map(({ it }) => it); const blocks = shuffled.map((it) => { - const block = `Input:\n${String(it.input)}\nOutput:\n${String(it.output)}`; + const block = `Input:\n${String(it.input)}\nOutput:\n${projectExample(it.output)}`; return block.length > perExMax ? `${block.slice(0, perExMax - 1)}…` : block; }); return blocks.join('\n\n'); } +/** + * few-shot の output を、モデルに出させるフィールドだけに絞る。 + * + * KV に置く few-shot は判定(category / component など)も含んだ完全な形で保つ。 + * トリアージの正解データとして計測(src/cli/typesafe-triage-spike.ts)にも使うため。 + * 一方、Workers AI に担当させるのはタイトルと要約だけなので、例にだけ余分な + * フィールドがあるとモデルがそれを真似て出力し、スキーマと食い違う。 + */ +export function projectExample(output: string): string { + try { + const o = JSON.parse(output); + return JSON.stringify({ + title: String(o?.title ?? ''), + summary: String(o?.summary ?? ''), + }); + } catch { + // 壊れた例はそのまま渡す(従来どおり、モデル側で無視されることを期待する) + return output; + } +} + async function getFewShotText(env: Env): Promise { const now = Date.now(); if (fewShotCache && now - fewShotCache.loadedAt < FEW_SHOT_TTL_MS) { @@ -376,17 +614,37 @@ async function runTriage( ? '\n\nIMPORTANT: Output exactly ONE minified JSON object and nothing else. Do not repeat the examples. Do not add prose, comments, or code fences.' : ''; const prompt = `${fewshot}\n\nNow process this message:\n\n<>\n${userText}`; - const result = (await env.AI.run(env.AI_TRIAGE_MODEL, { + const result = await env.AI.run(env.AI_TRIAGE_MODEL, { messages: [ { role: 'system', content: SYSTEM_PROMPT + strictNudge }, { role: 'user', content: prompt }, ], - max_tokens: 768, + // 推論トレースを出すモデル(gemma-4 など)は本文の前に思考を吐くため、 + // 768 だと JSON が途中で切れる(finish_reason: "length")。実測で完了まで + // 900〜1100 トークン使うので余裕を持たせる。 + max_tokens: 2048, temperature: strict ? 0 : 0.2, // JSON Schema を強制し、summary などのフィールド欠落を防ぐ。 response_format: { type: 'json_schema', json_schema: TRIAGE_JSON_SCHEMA }, - })) as { response?: unknown }; - return result?.response ?? null; + }); + return pickModelResponse(result); +} + +/** + * Workers AI の応答から本文を取り出す。モデルによって形が 2 通りある。 + * - `response`: 従来の Workers AI 形式(パース済みオブジェクト or 文字列) + * - `choices[0].message.content`: OpenAI 互換形式(gemma-4 などはこちらのみ) + * 片方しか見ないとモデル差し替え時に全件トリアージ失敗になるため、両方を受ける。 + */ +export function pickModelResponse(result: unknown): unknown | null { + if (!result || typeof result !== 'object') return null; + const r = result as { + response?: unknown; + choices?: { message?: { content?: unknown } }[]; + }; + if (r.response !== undefined && r.response !== null) return r.response; + const content = r.choices?.[0]?.message?.content; + return content ?? null; } /** runTriage の戻り(オブジェクト or 文字列)からトリアージ JSON を取り出す。 */ @@ -405,13 +663,345 @@ function responseLength(resp: unknown): number { return s.length; } -export const processFeedbackMessage = async ( - data: FeedbackQueueMessage, - env: Env -): Promise => { - if (!data?.report) return; - const { report } = data; +// ---- GitHub Issue 作成 ---- + +/** GitHub REST API への POST(Issue 作成・コメント投稿の共通処理)。 */ +function githubPost(env: Env, path: string, body: unknown): Promise { + return fetch(`https://api.github.com/repos/${path}`, { + method: 'post', + headers: { + Accept: 'application/vnd.github+json', + Authorization: `Bearer ${env.OCTOKIT_PAT ?? ''}`, + 'X-GitHub-Api-Version': '2022-11-28', + 'User-Agent': 'trainlcd-worker', + }, + body: JSON.stringify(body), + }); +} + +/** + * 原因が特定できているフィードバックについて、起票先の公開リポジトリを返す。 + * 特定できていない・公開に適さない場合は null(=非公開リポジトリのみに起票)。 + * + * - クラッシュレポートは対象外(スタックトレースを含み、内容の公開範囲が読めないため) + * - スパム/スパム疑い/トリアージ失敗は原因を特定できていないので対象外 + * - 質問カテゴリは修正対象のコンポーネントが定まらないので対象外 + */ +export function resolvePublicIssueRepo( + aiReport: AIReport, + opts: { + reportType: Report['reportType']; + triageFailed: boolean; + needsSpamReview?: boolean; + } +): string | null { + if (opts.reportType !== 'feedback') return null; + if (opts.triageFailed || aiReport.isSpam) return null; + // スパム疑いで人手確認待ちのものを公開リポジトリに出さない + if (opts.needsSpamReview) return null; + if (!PUBLIC_ISSUE_CATEGORIES.includes(aiReport.category)) return null; + if (!aiReport.component) return null; + if (aiReport.componentConfidence < PUBLIC_ISSUE_MIN_CONFIDENCE) return null; + return COMPONENT_REPOS[aiReport.component]; +} + +/** + * 公開リポジトリに立てるスタブ Issue のタイトル。 + * 公開範囲にフィードバックの内容を出さないため、AI 要約もタイトルも使わず、 + * 非公開の管理 Issue 番号だけで表現する。 + */ +export function buildPublicIssueTitle(internalIssueNumber: number): string { + return `フィードバック対応: ${INTERNAL_REPO}#${internalIssueNumber}`; +} + +/** + * 公開リポジトリに立てるスタブ Issue の本文。 + * 意図的に、フィードバックの原文・AI 要約・タイトル・端末情報を一切含めない。 + * 内容を追うための手掛かりは非公開の管理チケットへの参照だけに限定する。 + */ +export function buildPublicIssueBody(params: { + internalIssueNumber: number; + ticketId: string; +}): string { + const { internalIssueNumber, ticketId } = params; + return ` +アプリから届いたフィードバックのトリアージで、原因が本リポジトリにあると推定されたため起票しています。 + +フィードバックの内容は公開リポジトリには掲載していません。原文・要約・端末情報などの詳細は、下記の非公開の管理チケットを参照してください。 + +## 管理チケット +- Issue: ${INTERNAL_REPO}#${internalIssueNumber} +- チケットID: \`${ticketId}\` +`.trim(); +} + +/** + * 公開リポジトリへスタブ Issue を作成し、その URL を返す(失敗時は null)。 + * ここで throw すると queue が再試行して非公開 Issue が重複作成されるため、 + * 失敗はログに留める。 + */ +async function createPublicIssue( + env: Env, + params: { repo: string; internalIssueNumber: number; ticketId: string } +): Promise { + const { repo, internalIssueNumber, ticketId } = params; + try { + // ラベルは公開リポジトリ側に存在しないと自動生成されてしまうため付けない。 + // 分類・優先度は非公開の管理 Issue 側のラベルで管理する。 + const res = await githubPost(env, `${repo}/issues`, { + title: buildPublicIssueTitle(internalIssueNumber), + body: buildPublicIssueBody({ internalIssueNumber, ticketId }), + assignees: ['TinyKitten'], + }); + if (res.status !== 201) { + console.error('公開リポジトリへの起票に失敗', { + repo, + status: res.status, + internalIssueNumber, + }); + return null; + } + const created = (await res.json()) as { html_url: string }; + return created.html_url; + } catch (err) { + console.error('公開リポジトリへの起票に失敗', { repo, err }); + return null; + } +} + +/** 非公開の管理 Issue 側に、公開 Issue へのリンクをコメントで残す(失敗しても無視)。 */ +async function linkPublicIssue( + env: Env, + internalIssueNumber: number, + publicIssueUrl: string +): Promise { + try { + const res = await githubPost( + env, + `${INTERNAL_REPO}/issues/${internalIssueNumber}/comments`, + { body: `公開リポジトリに対応 Issue を起票しました: ${publicIssueUrl}` } + ); + if (res.status !== 201) { + console.error('管理 Issue への相互リンクコメントに失敗', { + status: res.status, + internalIssueNumber, + }); + } + } catch (err) { + console.error('管理 Issue への相互リンクコメントに失敗', { err }); + } +} + +// ---- 冪等化マーカー(STATE_KV) ---- + +/** + * レポート 1 件の処理状態を STATE_KV に残すマーカー。 + * + * GitHub Issue の作成後に例外が出ると queue が再試行し、同じフィードバックで + * Issue がもう 1 件作られてしまう。report.id をキーに「どこまで終わったか」を + * 永続化しておき、再試行では済んだ工程を飛ばす。 + * + * 再試行のたびに AI を呼び直すとトリアージ結果がぶれ、起票済み Issue と通知の + * 内容がずれるため、トリアージ結果もマーカーに含めて再利用する。 + */ +export type TriageMarker = { + version: 1; + /** 非公開リポジトリに作成した Issue 番号(レスポンスの解析に失敗したときは null) */ + issueNumber: number | null; + /** 作成した Issue の URL(同上) */ + issueUrl: string | null; + /** 公開リポジトリに作成したスタブ Issue の URL(作っていなければ null) */ + publicIssueUrl: string | null; + aiReport: AIReport; + triageFailed: boolean; + needsSpamReview: boolean; + /** Discord 通知まで完了しているか */ + notified: boolean; + updatedAt: string; +}; + +/** + * マーカーの保持期間。queue の再試行自体は数分で終わるが、DLQ に落ちたメッセージを + * 後日手動で流し直すことがあるため長めに取る。 + */ +export const TRIAGE_MARKER_TTL_SECONDS = 60 * 60 * 24 * 30; + +/** + * Discord 通知だけが失敗したことを示す。queue ハンドラはこれを受けて再試行し、 + * 再試行はマーカーを見て通知から再開する(Issue は作り直さない)。 + */ +export class FeedbackNotifyError extends Error { + constructor(reportId: string) { + super(`Discord notification failed for report ${reportId}`); + this.name = 'FeedbackNotifyError'; + } +} + +/** 処理済みマーカーの KV キー。 */ +export const triageMarkerKey = (reportId: string): string => + `feedbackTriage:processed:${reportId}`; + +/** + * 処理済みマーカーを読む。KV 障害は握り潰さず上位へ伝播させる(=再試行させる)。 + * ここで null に倒すと重複起票を防ぐという目的そのものを損なうため。 + * まだ副作用を出していない地点なので、throw しても Issue は重複しない。 + */ +async function loadTriageMarker( + env: Env, + reportId: string +): Promise { + const raw = await env.STATE_KV.get(triageMarkerKey(reportId), 'text'); + if (!raw) return null; + + let parsed: unknown; + try { + parsed = JSON.parse(raw); + } catch { + console.error('feedbackTriage: 処理済みマーカーが壊れているため無視する', { + reportId, + }); + return null; + } + if (!parsed || typeof parsed !== 'object') return null; + + const marker = parsed as Partial; + // aiReport を失っているマーカーは通知を組み立て直せないので無効扱いにする。 + if (!marker.aiReport || typeof marker.aiReport !== 'object') { + console.error( + 'feedbackTriage: 処理済みマーカーの内容が不正なため無視する', + { + reportId, + } + ); + return null; + } + + return { + version: 1, + issueNumber: + typeof marker.issueNumber === 'number' ? marker.issueNumber : null, + issueUrl: typeof marker.issueUrl === 'string' ? marker.issueUrl : null, + publicIssueUrl: + typeof marker.publicIssueUrl === 'string' ? marker.publicIssueUrl : null, + aiReport: marker.aiReport, + triageFailed: marker.triageFailed === true, + needsSpamReview: marker.needsSpamReview === true, + notified: marker.notified === true, + updatedAt: + typeof marker.updatedAt === 'string' + ? marker.updatedAt + : new Date().toISOString(), + }; +} + +/** マーカー保存の試行回数。KV の一過性エラーで冪等化の記録を落とさないため。 */ +const SAVE_MARKER_ATTEMPTS = 2; + +/** + * 同一マーカーキーへの書き込みを空ける間隔。KV は同一キーへの書き込みを 1 秒に + * 1 回までしか受け付けず、超えると 429 になる。 + */ +const SAVE_MARKER_MIN_INTERVAL_MS = 1100; + +/** 同一キーに最後に書き込めた時刻。次の書き込みを 1 秒以上空けるために持つ。 */ +const lastMarkerWriteAt = new Map(); + +/** + * 失敗したメッセージを再試行に回すまでの待ち時間。 + * + * KV はキーが無かったという結果も cacheTtl(既定 60 秒)の間エッジにキャッシュ + * するため、遅延なしで再試行すると、起票直後に書いたマーカーを読めずに + * Issue を作り直してしまう。ネガティブキャッシュが切れてから再試行させる。 + */ +export const FEEDBACK_RETRY_DELAY_SECONDS = 90; + +/** 同一キーへの書き込み間隔が 1 秒未満にならないよう、必要なぶんだけ待つ。 */ +async function waitForMarkerWriteWindow( + key: string, + extraWaitMs = 0 +): Promise { + const lastAt = lastMarkerWriteAt.get(key); + const sinceLastWrite = + lastAt === undefined ? Number.POSITIVE_INFINITY : Date.now() - lastAt; + const waitMs = Math.max( + SAVE_MARKER_MIN_INTERVAL_MS - sinceLastWrite, + extraWaitMs + ); + if (waitMs <= 0) return; + await new Promise((resolve) => setTimeout(resolve, waitMs)); +} + +/** + * 処理済みマーカーを書く。ここで throw すると「Issue は作成済みなのに再試行される」 + * という、まさに防ぎたい状態を作ってしまうため、失敗はログに留めて false を返す。 + * 呼び出し側は、マーカーを残せたかどうかで再試行してよいかを判断する。 + * + * 書けなかったマーカーはそのまま重複起票の窓になるので、諦める前に一度だけ + * 書き直す(KV の書き込み失敗は一過性のことが多い)。 + * + * 1 件のレポートでは、起票直後(notified: false)と通知後(notified の実結果)の + * 2 回、同じキーに書く。通知が 1 秒以内に終わると KV の同一キー書き込み制限に + * かかるため、間隔が足りなければ待ってから書く。 + */ +async function saveTriageMarker( + env: Env, + reportId: string, + marker: Omit +): Promise { + const key = triageMarkerKey(reportId); + for (let attempt = 1; attempt <= SAVE_MARKER_ATTEMPTS; attempt++) { + await waitForMarkerWriteWindow( + key, + attempt > 1 ? SAVE_MARKER_MIN_INTERVAL_MS : 0 + ); + const value: TriageMarker = { + version: 1, + ...marker, + updatedAt: new Date().toISOString(), + }; + try { + await env.STATE_KV.put(key, JSON.stringify(value), { + expirationTtl: TRIAGE_MARKER_TTL_SECONDS, + }); + lastMarkerWriteAt.set(key, Date.now()); + pruneMarkerWriteTimes(); + return true; + } catch (err) { + console.error('feedbackTriage: 処理済みマーカーの保存に失敗', { + reportId, + attempt, + maxAttempts: SAVE_MARKER_ATTEMPTS, + error: err instanceof Error ? err.message : String(err), + }); + } + } + return false; +} +/** 書き込み時刻の記録が isolate に溜まり続けないよう、間隔を過ぎたものを捨てる。 */ +function pruneMarkerWriteTimes(): void { + const now = Date.now(); + for (const [key, at] of lastMarkerWriteAt) { + if (now - at >= SAVE_MARKER_MIN_INTERVAL_MS) lastMarkerWriteAt.delete(key); + } +} + +// ---- トリアージ ---- + +type TriageOutcome = { + aiReport: AIReport; + triageFailed: boolean; + needsSpamReview: boolean; +}; + +/** + * フィードバック本文を AI でトリアージする。生成に失敗しても throw せず、 + * 「要約失敗」レポートに倒して原文を保全する(フィードバックを捨てないため)。 + */ +async function triageFeedback( + env: Env, + report: Report +): Promise { const fewshot = await getFewShotText(env); // 生成 → 最初のバランスした JSON を抽出。失敗したら厳格モードで数回まで再生成する。 @@ -420,7 +1010,22 @@ export const processFeedbackMessage = async ( let raw: unknown = null; let lastResponseLength = 0; for (let attempt = 1; attempt <= MAX_TRIAGE_ATTEMPTS; attempt++) { - const resp = await runTriage(env, fewshot, report.description, attempt > 1); + let resp: unknown = null; + try { + resp = await runTriage(env, fewshot, report.description, attempt > 1); + } catch (err) { + // JSON Mode を満たせない場合や AI 側の一時障害では env.AI.run が throw する。 + // ここで抜けると queue が再試行し、max_retries を使い切った時点でフィードバックが + // 消えるため、生成失敗として扱って最終的に「要約失敗」で起票する(原文は残る)。 + console.warn('feedbackTriage: トリアージの推論呼び出しが失敗(再試行)', { + reportId: report.id, + attempt, + maxAttempts: MAX_TRIAGE_ATTEMPTS, + model: env.AI_TRIAGE_MODEL, + error: err instanceof Error ? err.message : String(err), + }); + continue; + } lastResponseLength = responseLength(resp); raw = normalizeTriageResponse(resp); if (raw !== null) break; @@ -432,7 +1037,7 @@ export const processFeedbackMessage = async ( }); } - const triageFailed = raw === null; + let triageFailed = raw === null; let aiReport: AIReport; if (triageFailed) { // 生成に失敗しても破棄しない。原文を保全したまま、要約欄に失敗を明示して起票する。 @@ -451,23 +1056,138 @@ export const processFeedbackMessage = async ( )?.[1] : undefined; if (!aiReport.isSpam && String(rawSummary ?? '').trim() === '') { + console.warn('feedbackTriage: モデルが summary を空/欠落で返却', { + reportId: report.id, + responseLength: lastResponseLength, + }); + } + + // タイトルが未取得・日本語として破損している場合は、そのまま起票すると + // Issue 一覧から内容を判別できない。失敗を明示するレポートに倒したうえで + // ❓ Unknown Type を付け、破損率を追えるようにログを残す。 + const brokenTitleReason = findBrokenTitleReason(aiReport.title); + if (brokenTitleReason) { console.warn( - 'feedbackTriage: モデルが summary を空/欠落で返却(title にフォールバック)', - { reportId: report.id, responseLength: lastResponseLength } + 'feedbackTriage: 生成タイトルが使用不能(要約失敗として起票)', + { + reportId: report.id, + reason: brokenTitleReason, + model: env.AI_TRIAGE_MODEL, + responseLength: lastResponseLength, + } ); + aiReport = buildFailedReport(report.description, 72); + triageFailed = true; } } - if (!aiReport.isSpam && looksLikeSpam(report.description)) { + // 判定は TypeSafe が担当する。タイトル・要約の生成とは独立なので、失敗しても + // 生成結果は捨てない。取得できなかった場合は、原因を絞り込めなかった扱いに倒す + // (componentConfidence 0 なので公開リポジトリへの起票は行われない)。 + let judgment: Verdict | null = null; + for (let attempt = 1; attempt <= MAX_JUDGE_ATTEMPTS; attempt++) { + try { + judgment = await judgeFeedback(env, report); + break; + } catch (err) { + console.warn('feedbackTriage: TypeSafe の判定取得に失敗(再試行)', { + reportId: report.id, + attempt, + maxAttempts: MAX_JUDGE_ATTEMPTS, + error: err instanceof Error ? err.message : String(err), + }); + } + } + + if (judgment) { aiReport = { ...aiReport, - title: '内容未分類(改善要望なし)', - isSpam: true, - labels: [], - reason: 'non-actionable', + isSpam: judgment.isSpam, + category: judgment.category as AICategory, + triageLevel: judgment.triageLevel, + component: + judgment.component === 'unknown' + ? null + : (judgment.component as AIComponent), + componentConfidence: judgment.componentConfidence, + confidence: judgment.categoryConfidence, + labels: deriveLabels(judgment), + reason: `${judgment.category} / ${judgment.triageLevel}`, + }; + if (judgment.isSpam) { + aiReport = { ...aiReport, title: NON_ACTIONABLE_TITLE, labels: [] }; + } + } else { + console.error( + 'feedbackTriage: TypeSafe の判定を取得できなかった。分類なしで起票する', + { reportId: report.id } + ); + aiReport = { + ...aiReport, + component: null, + componentConfidence: 0, + confidence: 0, + reason: 'judgment-unavailable', }; } + const spamDecision = applySpamHeuristic(aiReport, report.description, { + triageFailed, + }); + aiReport = spamDecision.report; + const { needsSpamReview } = spamDecision; + if (needsSpamReview) { + console.warn( + 'feedbackTriage: スパム判定がモデルとヒューリスティックで不一致(人手確認に回す)', + { reportId: report.id, confidence: aiReport.confidence } + ); + } + + return { + aiReport, + triageFailed, + // TypeSafe 側の「スパム確定ではないが人手確認に回す」判断も引き継ぐ。 + // resolvePublicIssueRepo はこのフラグで公開リポジトリへの起票を止めるため、 + // 落とすと確認前のフィードバックが公開リポジトリに出る。 + needsSpamReview: + needsSpamReview || judgment === null || judgment.needsSpamReview, + }; +} + +// ---- Discord 通知 ---- + +/** + * Discord へ通知する。GitHub Issue の作成後に呼ばれるため、ここで throw すると + * queue ハンドラが再試行し、同一レポートで Issue が重複作成される。 + * webhook URL 未設定・HTTP エラーに加え、fetch 自体の失敗(ネットワーク断・DNS + * 失敗・不正な URL)も含めて、あらゆる失敗をログに留めて握り潰す。 + * + * 戻り値は「通知を送り終えたか」。false のときは処理済みマーカーを未通知のまま + * 残し、メッセージを再投入したときに通知だけやり直せるようにする。 + */ +async function notifyDiscord( + env: Env, + params: { + report: Report; + aiReport: AIReport; + shouldTagTriage: boolean; + categoryLabel?: string; + triageLabel?: string; + autoModeLabel?: string; + issueUrl: string | null; + publicIssueUrl: string | null; + } +): Promise { + const { + report, + aiReport, + shouldTagTriage, + categoryLabel, + triageLabel, + autoModeLabel, + issueUrl, + publicIssueUrl, + } = params; const { id, createdAt, @@ -479,120 +1199,14 @@ export const processFeedbackMessage = async ( stacktrace, reportType, imageUrl, - appEdition, - appClip, autoModeEnabled, sentryEventId, } = report; - const createdAtText = dayjs(createdAt).format('YYYY/MM/DD HH:mm:ss'); - const osNameLabel = (() => { - if (deviceInfo?.osName === 'iOS') return GITHUB_LABELS.PLATFORM_IOS; - if (deviceInfo?.osName === 'iPadOS') return GITHUB_LABELS.PLATFORM_IPADOS; - if (deviceInfo?.osName === 'Android') return GITHUB_LABELS.PLATFORM_ANDROID; - return GITHUB_LABELS.PLATFORM_OTHER_OS; - })(); - - const autoModeLabel = autoModeEnabled - ? GITHUB_LABELS.AUTOMODE_ENABLED - : undefined; - - // トリアージ生成に失敗したときは誤ったカテゴリ/優先度を付けない。 - const shouldTagTriage = - reportType === 'feedback' && !aiReport.isSpam && !triageFailed; - const categoryLabel = shouldTagTriage - ? CATEGORY_LABELS[aiReport.category] - : undefined; - const triageLabel = shouldTagTriage - ? TRIAGE_LABELS[aiReport.triageLevel] - : undefined; - try { - const res = await fetch( - 'https://api.github.com/repos/TrainLCD/Issues/issues', - { - method: 'post', - headers: { - Accept: 'application/vnd.github+json', - Authorization: `Bearer ${env.OCTOKIT_PAT ?? ''}`, - 'X-GitHub-Api-Version': '2022-11-28', - 'User-Agent': 'trainlcd-worker', - }, - body: JSON.stringify({ - title: aiReport.title ?? '要約未取得', - body: ` -![Image](${imageUrl}) - - -${'```'} -${description} -${'```'} - -## AIによる要約 -${aiReport.summary} - -## 発行日時 -${createdAtText} - -## 端末モデル名 -${deviceInfo?.brand} ${deviceInfo?.modelName}(${deviceInfo?.modelId}) - -## 端末のOS -${deviceInfo?.osName} ${deviceInfo?.osVersion} - -## 端末設定言語 -${deviceInfo?.locale} - -## アプリの設定言語 -${language} - -## アプリのバージョン -${appVersion} - -## オートモード -${autoModeEnabled ? '有効' : '無効'} - -## スタックトレース -${'```'} -${stacktrace} -${'```'} - -## Sentry Event ID -${sentryEventId} - -## レポーターUID -${reporterUid} - `.trim(), - assignees: ['TinyKitten'], - milestone: null, - labels: [ - reportType === 'feedback' && - !aiReport.isSpam && - GITHUB_LABELS.FEEDBACK_TYPE, - reportType === 'crash' && GITHUB_LABELS.CRASH_TYPE, - appEdition === 'production' && GITHUB_LABELS.PRODUCTION_APP, - appEdition === 'canary' && GITHUB_LABELS.CANARY_APP, - appClip && GITHUB_LABELS.PLATFORM_APPCLIP, - aiReport.isSpam && GITHUB_LABELS.SPAM_TYPE, - triageFailed && GITHUB_LABELS.UNKNOWN_TYPE, - osNameLabel, - autoModeLabel, - categoryLabel, - triageLabel, - ].filter(Boolean), - }), - } - ); - - if (res.status !== 201) { - console.error(await res.json()); - throw new Error(`GitHub API failed with status ${res.status}`); - } - - const issuesRes = (await res.json()) as { html_url: string }; - const csWHUrl = env.DISCORD_CS_WEBHOOK_URL; const crashWHUrl = env.DISCORD_CRASH_WEBHOOK_URL; + const issueUrlText = issueUrl ?? '不明'; const embeds: DiscordEmbed[] = deviceInfo ? [ { @@ -627,7 +1241,10 @@ ${reporterUid} autoModeLabel ?? (autoModeEnabled === false ? '無効' : '不明'), }, - { name: 'GitHub Issue', value: issuesRes.html_url }, + { name: 'GitHub Issue', value: issueUrlText }, + ...(publicIssueUrl + ? [{ name: '公開リポジトリ Issue', value: publicIssueUrl }] + : []), { name: 'Sentry Event ID', value: sentryEventId ?? '不明' }, ], }, @@ -656,7 +1273,10 @@ ${reporterUid} autoModeLabel ?? (autoModeEnabled === false ? '無効' : '不明'), }, - { name: 'GitHub Issue', value: issuesRes.html_url }, + { name: 'GitHub Issue', value: issueUrlText }, + ...(publicIssueUrl + ? [{ name: '公開リポジトリ Issue', value: publicIssueUrl }] + : []), ], }, ]; @@ -670,14 +1290,11 @@ ${reporterUid} .slice(0, 10) .join('\n')}\n${stacktraceTooLong ? '...' : ''}\`\`\``; - // 注意: ここから先(GitHub Issue 作成後)の Discord 通知は失敗しても throw しない。 - // throw すると queue ハンドラが retry し、同一レポートで Issue が重複作成されるため、 - // 通知の失敗・URL 未設定はログに留める。 switch (reportType) { case 'feedback': { if (!csWHUrl) { console.error('DISCORD_CS_WEBHOOK_URL is not set; skipping notify'); - break; + return false; } const whRes = await fetch(csWHUrl, { method: 'POST', @@ -693,15 +1310,16 @@ ${reporterUid} if (!whRes.ok) { const msg = await whRes.text().catch(() => ''); console.error('Discord CS webhook failed', whRes.status, msg); + return false; } - break; + return true; } case 'crash': { if (!crashWHUrl) { console.error( 'DISCORD_CRASH_WEBHOOK_URL is not set; skipping notify' ); - break; + return false; } const whRes = await fetch(crashWHUrl, { method: 'POST', @@ -711,15 +1329,256 @@ ${reporterUid} if (!whRes.ok) { const msg = await whRes.text().catch(() => ''); console.error('Discord Crash webhook failed', whRes.status, msg); + return false; } - break; + return true; } default: - break; + // 通知先のない種別。送るものがないので「通知済み」として扱う。 + return true; } } catch (err) { - // 握りつぶすと queue ハンドラが ack してメッセージを失うため、再送出して再試行させる - console.error(err); - throw err; + // fetch 自体の失敗(ネットワークエラー等)。再送出すると Issue が重複するため握り潰す。 + console.error('feedbackTriage: Discord 通知に失敗', { + reportId: id, + error: err instanceof Error ? err.message : String(err), + }); + return false; } +} + +export const processFeedbackMessage = async ( + data: FeedbackQueueMessage, + env: Env +): Promise => { + if (!data?.report) return; + const { report } = data; + + const { + id, + createdAt, + description, + deviceInfo, + language, + appVersion, + reporterUid, + stacktrace, + reportType, + imageUrl, + appEdition, + appClip, + autoModeEnabled, + sentryEventId, + } = report; + + // 再試行や DLQ からの再投入で同じレポートが流れてきたとき、Issue を重複起票しない + // ように、処理済みマーカーを見て済んだ工程を飛ばす。 + const marker = await loadTriageMarker(env, id); + if (marker?.notified) { + console.warn( + 'feedbackTriage: 処理済みのレポートを再受信したためスキップする', + { reportId: id, issueNumber: marker.issueNumber } + ); + return; + } + + // 起票済みなら AI を呼び直さない。呼び直すと結果がぶれ、起票済み Issue と + // Discord 通知の内容がずれるため、マーカーに残したトリアージ結果を使う。 + const { aiReport, triageFailed, needsSpamReview } = marker + ? { + aiReport: marker.aiReport, + triageFailed: marker.triageFailed, + needsSpamReview: marker.needsSpamReview, + } + : await triageFeedback(env, report); + + const createdAtText = dayjs(createdAt).format('YYYY/MM/DD HH:mm:ss'); + const osNameLabel = (() => { + if (deviceInfo?.osName === 'iOS') return GITHUB_LABELS.PLATFORM_IOS; + if (deviceInfo?.osName === 'iPadOS') return GITHUB_LABELS.PLATFORM_IPADOS; + if (deviceInfo?.osName === 'Android') return GITHUB_LABELS.PLATFORM_ANDROID; + return GITHUB_LABELS.PLATFORM_OTHER_OS; + })(); + + const autoModeLabel = autoModeEnabled + ? GITHUB_LABELS.AUTOMODE_ENABLED + : undefined; + + // トリアージ生成に失敗したときは誤ったカテゴリ/優先度を付けない。 + const shouldTagTriage = + reportType === 'feedback' && !aiReport.isSpam && !triageFailed; + const categoryLabel = shouldTagTriage + ? CATEGORY_LABELS[aiReport.category] + : undefined; + const triageLabel = shouldTagTriage + ? TRIAGE_LABELS[aiReport.triageLevel] + : undefined; + + let issueNumber = marker?.issueNumber ?? null; + let issueUrl = marker?.issueUrl ?? null; + let publicIssueUrl = marker?.publicIssueUrl ?? null; + + if (!marker) { + try { + const res = await githubPost(env, `${INTERNAL_REPO}/issues`, { + title: aiReport.title ?? '要約未取得', + body: ` +![Image](${imageUrl}) + + +${'```'} +${description} +${'```'} + +## AIによる要約 +${aiReport.summary} + +## チケットID +${id} + +## 発行日時 +${createdAtText} + +## 端末モデル名 +${deviceInfo?.brand} ${deviceInfo?.modelName}(${deviceInfo?.modelId}) + +## 端末のOS +${deviceInfo?.osName} ${deviceInfo?.osVersion} + +## 端末設定言語 +${deviceInfo?.locale} + +## アプリの設定言語 +${language} + +## アプリのバージョン +${appVersion} + +## オートモード +${autoModeEnabled ? '有効' : '無効'} + +## スタックトレース +${'```'} +${stacktrace} +${'```'} + +## Sentry Event ID +${sentryEventId} + +## レポーターUID +${reporterUid} + `.trim(), + assignees: ['TinyKitten'], + milestone: null, + labels: [ + reportType === 'feedback' && + !aiReport.isSpam && + GITHUB_LABELS.FEEDBACK_TYPE, + reportType === 'crash' && GITHUB_LABELS.CRASH_TYPE, + appEdition === 'production' && GITHUB_LABELS.PRODUCTION_APP, + appEdition === 'canary' && GITHUB_LABELS.CANARY_APP, + appClip && GITHUB_LABELS.PLATFORM_APPCLIP, + aiReport.isSpam && GITHUB_LABELS.SPAM_TYPE, + (triageFailed || needsSpamReview) && GITHUB_LABELS.UNKNOWN_TYPE, + osNameLabel, + autoModeLabel, + categoryLabel, + triageLabel, + ].filter(Boolean), + }); + + if (res.status !== 201) { + console.error(await res.text().catch(() => '')); + throw new Error(`GitHub API failed with status ${res.status}`); + } + + // ここから先は Issue 作成済み。throw して再試行させると重複起票になるため、 + // レスポンスの解析に失敗しても続行し、分かった範囲をマーカーに残す。 + const created = (await res.json().catch((err: unknown) => { + console.error('feedbackTriage: 起票レスポンスの解析に失敗', { + reportId: id, + error: err instanceof Error ? err.message : String(err), + }); + return null; + })) as { html_url?: string; number?: number } | null; + issueNumber = typeof created?.number === 'number' ? created.number : null; + issueUrl = + typeof created?.html_url === 'string' ? created.html_url : null; + } catch (err) { + // Issue 作成前の失敗。握りつぶすと queue ハンドラが ack してメッセージを失うため、 + // 再送出して再試行させる(この時点では Issue は作られていないので重複しない)。 + console.error(err); + throw err; + } + + // 原因コンポーネントが特定できている場合のみ、該当の公開リポジトリにも起票する。 + // 公開側に載せるのは管理 Issue 番号とチケットIDだけで、フィードバックの内容は含めない。 + // 起票後は管理 Issue 側にもコメントでリンクを残し、双方向に追えるようにする。 + const publicRepo = resolvePublicIssueRepo(aiReport, { + reportType, + triageFailed, + needsSpamReview, + }); + if (publicRepo && issueNumber !== null) { + publicIssueUrl = await createPublicIssue(env, { + repo: publicRepo, + internalIssueNumber: issueNumber, + ticketId: id, + }); + if (publicIssueUrl) { + await linkPublicIssue(env, issueNumber, publicIssueUrl); + } + } + + // 起票済みであることを先に永続化する。この後で落ちても、再試行は通知から再開する。 + await saveTriageMarker(env, id, { + issueNumber, + issueUrl, + publicIssueUrl, + aiReport, + triageFailed, + needsSpamReview, + notified: false, + }); + } + + const notified = await notifyDiscord(env, { + report, + aiReport, + shouldTagTriage, + categoryLabel, + triageLabel, + autoModeLabel, + issueUrl, + publicIssueUrl, + }); + + const markerSaved = await saveTriageMarker(env, id, { + issueNumber, + issueUrl, + publicIssueUrl, + aiReport, + triageFailed, + needsSpamReview, + // 通知に失敗したときは未通知のまま残す。再試行では起票を飛ばして通知だけ + // やり直す(成功したことにすると通知が永久に届かない)。 + notified, + }); + + if (notified) return; + + if (!markerSaved) { + // マーカーを残せなかったので、再試行すると Issue を作り直してしまう。 + // 通知を諦めて ack する(フィードバック自体は起票済みで失われない)。 + console.error( + 'feedbackTriage: 通知に失敗したがマーカーも残せなかったため再試行しない', + { reportId: id, issueNumber } + ); + return; + } + + // 起票済みなので、再試行してもマーカーを見て通知から再開する(重複起票しない)。 + // max_retries を使い切ったメッセージは DLQ に残り、Discord 側の障害・設定ミスに + // 気づける。 + throw new FeedbackNotifyError(id); }; diff --git a/src/consumers/typesafeTriage.test.ts b/src/consumers/typesafeTriage.test.ts new file mode 100644 index 0000000..ecec037 --- /dev/null +++ b/src/consumers/typesafeTriage.test.ts @@ -0,0 +1,153 @@ +import { type Answer, compose } from './typesafeTriage'; + +/** 既定は「正当な不具合報告」。各テストで必要な軸だけ上書きする */ +function answers( + over: { + spam?: number; + announcement?: number; + praise?: number; + crash?: number; + stationData?: number; + category?: string; + categoryConf?: number; + component?: string; + componentConf?: number; + severity?: number; + actionability?: number; + } = {} +): Record { + const noul = (v: number): Answer => ({ type: 'noul', noul: v }); + return { + is_spam: noul(over.spam ?? 0.05), + is_announcement_transcript: noul(over.announcement ?? 0.05), + is_praise_only: noul(over.praise ?? 0.05), + is_crash_or_data_loss: noul(over.crash ?? 0.05), + mentions_station_data: noul(over.stationData ?? 0.05), + category: { + type: 'choice', + choice: over.category ?? 'bug', + probabilities: {}, + confidence: over.categoryConf ?? 0.9, + }, + component: { + type: 'choice', + choice: over.component ?? 'mobile_app', + probabilities: {}, + confidence: over.componentConf ?? 0.9, + }, + severity: { + type: 'score', + score: over.severity ?? 1.0, + legend: {}, + probabilities: {}, + confidence: 0.8, + }, + actionability: { + type: 'score', + score: over.actionability ?? 1.0, + legend: {}, + probabilities: {}, + confidence: 0.8, + }, + }; +} + +describe('スパム判定', () => { + it('車内放送の書き起こしは、謝辞を含んでいてもスパムにする', () => { + // 放送文は「ご利用くださいましてありがとうございます」を含むため + // is_praise_only が上がる。感謝ゲートで弾くと放送が素通りする。 + const v = compose( + answers({ announcement: 0.89, praise: 0.65, spam: 0.51 }) + ); + expect(v.isSpam).toBe(true); + }); + + it('スパム信号が明確なら、is_praise_only がちょうど 0.5 でもスパムにする', () => { + // 固定値 praise < 0.5 と比べていたため、この組み合わせが素通りしていた + const v = compose(answers({ spam: 0.96, praise: 0.5 })); + expect(v.isSpam).toBe(true); + }); + + it('感謝の方がスパムらしさより強いときはスパムにしない', () => { + const v = compose(answers({ spam: 0.55, praise: 0.9, category: 'praise' })); + expect(v.isSpam).toBe(false); + }); + + it('正当な報告はスパムにしない', () => { + expect(compose(answers()).isSpam).toBe(false); + }); + + it('スパムのときは category と component を伏せる', () => { + const v = compose( + answers({ spam: 0.9, category: 'bug', component: 'mobile_app' }) + ); + expect(v.category).toBe('question'); + expect(v.component).toBe('unknown'); + expect(v.triageLevel).toBe('low'); + }); +}); + +describe('triageLevel', () => { + it('クラッシュとデータ消失は severity によらず urgent', () => { + const v = compose(answers({ crash: 0.8, severity: 0.1 })); + expect(v.triageLevel).toBe('urgent'); + }); + + it('bug は severity で段階を決める', () => { + expect(compose(answers({ severity: 2.5 })).triageLevel).toBe('urgent'); + expect(compose(answers({ severity: 1.9 })).triageLevel).toBe('high'); + expect(compose(answers({ severity: 1.0 })).triageLevel).toBe('medium'); + expect(compose(answers({ severity: 0.2 })).triageLevel).toBe('low'); + }); + + it('不具合でない要望は、severity が高くても medium を超えない', () => { + // severity は「不具合の重さ」の尺度なので、要望に当てても意味を持たない。 + // 実測では要望に severity 1.94 が付き、優先度が跳ね上がっていた。 + for (const category of ['feature_request', 'improvement']) { + const v = compose(answers({ category, severity: 2.9 })); + expect(v.triageLevel).toBe('medium'); + } + }); + + it('質問と称賛は low', () => { + expect( + compose(answers({ category: 'question', severity: 2.9 })).triageLevel + ).toBe('low'); + expect( + compose(answers({ category: 'praise', severity: 2.9 })).triageLevel + ).toBe('low'); + }); +}); + +describe('component の確信度', () => { + it('unknown のときは確信度を 0 にする', () => { + // 公開リポジトリへの起票判定に使う値なので、絞り込めていないことを 0 で表す + const v = compose(answers({ component: 'unknown', componentConf: 0.95 })); + expect(v.componentConfidence).toBe(0); + }); + + it('component を特定できたときは分布由来の確信度をそのまま渡す', () => { + const v = compose( + answers({ component: 'station_api', componentConf: 0.83 }) + ); + expect(v.componentConfidence).toBeCloseTo(0.83); + }); +}); + +describe('needsSpamReview', () => { + it('スパム確定には届かないがスパムらしさが残るときに立てる', () => { + const v = compose(answers({ spam: 0.4 })); + expect(v.isSpam).toBe(false); + expect(v.needsSpamReview).toBe(true); + }); + + it('スパムらしさが十分低ければ立てない', () => { + expect(compose(answers({ spam: 0.1 })).needsSpamReview).toBe(false); + }); + + it('スパム確定のときは立てない(確認するまでもない)', () => { + const v = compose(answers({ spam: 0.9 })); + expect(v.isSpam).toBe(true); + expect(v.needsSpamReview).toBe(false); + }); +}); diff --git a/src/consumers/typesafeTriage.ts b/src/consumers/typesafeTriage.ts new file mode 100644 index 0000000..6edd8b9 --- /dev/null +++ b/src/consumers/typesafeTriage.ts @@ -0,0 +1,401 @@ +/** + * フィードバックのトリアージ判定を TypeSafe(System One / Jev)で行う。 + * + * TypeSafe が返すのは Choice / Score / Noul の型付き判定だけで、文章生成は行わない + * ()。そのためタイトルと要約は Workers AI が担当し、 + * ここではスパム・カテゴリ・優先度・原因コンポーネントの判定だけを扱う。 + * + * 質問定義と閾値は計測スクリプト(src/cli/typesafe-triage-spike.ts)と共有する。 + * 別々に持つと、スパイクで測ったものと本番で動くものが食い違う。 + */ +import type { Report } from '../models/feedback'; +import type { Env } from '../types'; + +const API_URL = 'https://api.typesafe.ai/v1/systemone'; + +/** 時間を置けば通る可能性があるステータス(レート制限と過負荷) */ +const RETRYABLE_STATUSES = new Set([429, 529]); +const RETRYABLE_ATTEMPTS = 3; +const RETRY_BASE_MS = 250; + +/** + * instructions と criteria は文字列のほか、構造化オブジェクト・配列も取れる + * ()。選択肢が紛らわしいときに what / not_for / + * examples のような欄を持つオブジェクトで書き分けられる。欄の名前は API の + * 予約語ではなく、こちらで決めてよい。 + */ +type Description = string | Record | readonly unknown[]; + +type NoulQuestion = { + type: 'noul'; + instructions: Description; + criteria?: { true: Description; false: Description }; +}; +type ChoiceQuestion = { + type: 'choice'; + instructions: Description; + criteria: Record; +}; +type ScoreQuestion = { + type: 'score'; + instructions: Description; + criteria: readonly Description[]; +}; +type Question = NoulQuestion | ChoiceQuestion | ScoreQuestion; + +export type NoulAnswer = { type: 'noul'; noul: number }; +export type ChoiceAnswer = { + type: 'choice'; + choice: string; + probabilities: Record; + confidence: number; +}; +export type ScoreAnswer = { + type: 'score'; + score: number; + legend: Record; + probabilities: Record; + confidence: number; +}; +export type Answer = NoulAnswer | ChoiceAnswer | ScoreAnswer; + +export type SystemOneResponse = { + model: string; + answers: Record; + usage: { input_tokens: number; output_tokens: number }; +}; + +// ---- 質問定義 ---- + +export type QuestionId = keyof typeof QUESTIONS; + +/** + * 実測(46 件)でフィッティングした閾値。グリッド探索の最良値ではなく、 + * 分離幅の中央に寄せた丸めた値を使う。最良値は 46 点に対して過学習する。 + */ +const T = { + /** + * スパム確定の閾値。実測では正当な報告のスパム信号の最大が 0.27、スパムの最小が + * 0.75 と大きく開いたため、その中間に置いている。正当な報告を握り潰す方が + * スパムを 1 件通すより損失が大きいという方針は現行実装から引き継ぐ。 + */ + SPAM: 0.5, + /** スパム確定には満たないが、人手確認に回す下限 */ + SPAM_REVIEW: 0.3, + /** urgent へのハードルール */ + CRASH: 0.7, + /** bug の severity から triageLevel を決める境界 */ + BUG_URGENT: 2.4, + BUG_HIGH: 1.8, + BUG_MEDIUM: 0.5, + /** 不具合ではない要望を medium に上げる境界 */ + REQUEST_MEDIUM: 1.0, +} as const; + +/** + * 1 リクエストにまとめて投げる。TypeSafe の質問は互いに独立で並列評価されるため、 + * 一部の入力でしか使わない質問(praise 判定など)も投機的に同梱してよい。 + * + * 命令・基準はすべて日本語で書いている。判定対象が日本語のフィードバックで、 + * 「車内放送の書き起こし」のように英語に置き換えると輪郭がぼやける概念を + * 基準に含めるため。一致率が低い場合、次に振るべき変数は命令文の言語。 + */ +export const QUESTIONS = { + is_spam: { + type: 'noul', + instructions: + '`feedback` は、アプリの改善とは無関係な内容か(宣伝、荒らし、無関係な雑談)。', + criteria: { + true: 'アプリの改善に一切つながらない内容。宣伝、荒らし、無関係な雑談、および「テスト」「送信試験」「動作チェック」のように送信を試すためだけの投稿。', + false: + '不具合の報告、要望、質問、感謝など、アプリに向けられた内容。書き方が拙くても、内容がアプリに向いていれば該当する。', + }, + }, + is_announcement_transcript: { + type: 'noul', + instructions: + '`feedback` は、鉄道の車内放送や駅の放送をそのまま書き写したものか。', + criteria: { + true: '「次は」「まもなく」「この電車は〜行きです」のような放送の文言が並び、報告者自身の訴えが無いもの。', + false: + '放送の文言を引用していても、それが誤っている・読み上げられないといった報告者自身の訴えを伴うもの。', + }, + }, + is_praise_only: { + type: 'noul', + instructions: + '`feedback` は、感謝・称賛・応援だけで、対応すべき不具合や要望を含まないか。', + criteria: { + true: '感謝や称賛のみ。直すべきものも、追加してほしいものも書かれていない。', + false: '感謝を述べつつも、不具合の報告や要望が含まれている。', + }, + }, + is_crash_or_data_loss: { + type: 'noul', + instructions: + '`feedback` は、アプリが強制終了する、または保存された設定やデータが失われることを報告しているか。', + criteria: { + true: 'アプリが落ちる、フリーズして操作を受け付けない、設定やデータが消えた。', + false: '表示の誤りや動作の不満であり、強制終了やデータ喪失ではない。', + }, + }, + mentions_station_data: { + type: 'noul', + instructions: + '`feedback` は、駅・路線・列車種別そのもののデータの誤りや欠落を指しているか。', + criteria: { + true: '駅名・路線名・乗換情報・停車駅・列車種別のデータが、実際と違う、または存在しない。', + false: + 'アプリの表示・操作・音声など、データではなくアプリの振る舞いについての内容。', + }, + }, + category: { + type: 'choice', + instructions: '`feedback` を、最もよく当てはまる 1 つに分類せよ。', + criteria: { + bug: '意図どおりに動いていない。誤動作、誤表示、クラッシュ、表示崩れ。', + feature_request: 'まだ存在しない機能を新たに作ってほしいという要望。', + improvement: + '既に存在する機能を、より使いやすくしてほしいという調整の要望。', + question: '使い方の確認や情報の要求。直してほしいものは示されていない。', + praise: '感謝・称賛・応援のみで、対応すべき要望を含まない。', + }, + }, + component: { + // 実測で station_api -> mobile_app の取り違えが 4 件出た(ナンバリング記号・ + // 路線カラー・ロゴ・イメージカラー)。いずれも「表示が誤っている」と読めるため、 + // 値が誤っているのか描画が誤っているのかを focus と not_for で明示する。 + type: 'choice', + instructions: { + question: + '`feedback` が報告している事象の原因は、どこにあると考えられるか。', + focus: + '画面に出ている内容が誤っている場合、その値がデータとして誤っているのか、描画のされ方が誤っているのかで分ける。', + }, + criteria: { + mobile_app: { + what: 'アプリ本体の振る舞い。描画・レイアウト・文字の見切れ・スクロール・テーマ、音声の再生制御、クラッシュ、設定項目、位置情報の追従。路線ロゴの画像そのもの。', + not_for: + '表示されている記号・色・名称・停車駅などの値そのものが実際と異なる場合は station_api。', + examples: [ + '文字が見切れる、乗換案内がスクロールしないと読めない', + '位置情報が更新されず手前の駅を表示し続ける', + 'アプリが強制終了する、オートモードが停止する', + 'アナウンス中に他アプリの音量が下がらない', + '路線のロゴが別の事業者のロゴになっている', + ], + }, + station_api: { + what: '駅・路線・列車種別のデータの値そのもの。駅名、駅ナンバリングの記号と形、路線カラー、停車駅と通過駅、乗換情報、駅名や種別名の多言語表記。', + not_for: + '値は正しく表示の崩れが問題である場合や、路線ロゴの画像が別の事業者のものになっている場合は mobile_app(ロゴ画像はアプリに同梱されている)。', + examples: [ + 'ナンバリングの記号が別の路線のものになっている', + '路線のイメージカラーが実際と異なる', + '特急の停車駅・通過駅の設定が実際と異なる', + '駅名の英語表記・中国語表記が誤っている', + ], + }, + functions: { + what: 'サーバ側の処理。読み上げ音声の合成品質やイントネーション、AI チャットの応答、フィードバックの送信、画像のアップロード。', + not_for: + '音が鳴らない・音量が下がらないといった端末側の再生制御は mobile_app。', + examples: [ + '読み上げのイントネーションが不自然', + 'AI に質問するとエラーしか返らない', + ], + }, + website: { + what: '公式サイト(trainlcd.app)そのもの。', + not_for: + 'アプリ内のエラーやクラッシュはサイトとは無関係なので mobile_app。', + examples: ['公式サイトのリンクが 404 になる'], + }, + unknown: { + what: 'この内容だけでは原因の所在を絞り込めない。', + not_for: + '内容から所在が読み取れるなら、確信が持てなくても該当する選択肢を選ぶ。', + examples: [ + '使い方の質問', + '感謝や称賛のみ', + '症状が漠然としていて対象を特定できない', + ], + }, + }, + }, + severity: { + type: 'score', + instructions: + '`feedback` が報告している事象は、利用者にとってどれだけ重いか。', + criteria: [ + '見た目や文言の体裁の問題で、機能そのものは使える。', + '機能は使えるが、表示される内容が誤っている、または余分な操作が必要になる。', + '特定の機能が使えない、または誤った案内によって利用者が乗車の判断を誤りうる。', + 'アプリが強制終了する、データが失われる、または乗車中にアプリが使い物にならない。', + ], + }, + actionability: { + type: 'score', + instructions: + '`feedback` は、開発者が調査に着手するのに十分な具体性を備えているか。', + criteria: [ + '何が起きたのか特定できず、調査を始められない。', + '症状は分かるが、再現の条件や対象(駅名・路線名・画面名)が不足している。', + '対象と症状が具体的に書かれており、そのまま調査に着手できる。', + ], + }, +} satisfies Record; + +export type Verdict = { + isSpam: boolean; + needsSpamReview: boolean; + category: string; + categoryConfidence: number; + component: string; + componentConfidence: number; + triageLevel: 'urgent' | 'high' | 'medium' | 'low'; + severity: number; + actionability: number; +}; + +function noul(answers: Record, id: QuestionId): number { + const a = answers[id]; + return a?.type === 'noul' ? a.noul : Number.NaN; +} +function choice(answers: Record, id: QuestionId): ChoiceAnswer { + const a = answers[id]; + if (a?.type !== 'choice') throw new Error(`choice 回答が無い: ${id}`); + return a; +} +function score(answers: Record, id: QuestionId): ScoreAnswer { + const a = answers[id]; + if (a?.type !== 'score') throw new Error(`score 回答が無い: ${id}`); + return a; +} + +export function compose(answers: Record): Verdict { + const praise = noul(answers, 'is_praise_only'); + const spamSignal = noul(answers, 'is_spam'); + const announcement = noul(answers, 'is_announcement_transcript'); + + // 車内放送の書き起こしは「ご利用ありがとうございます」を含むため is_praise_only が + // 上がる。放送判定を praise ゲートの外に出さないと、放送がそのまま素通りする。 + // + // 感謝ゲートは固定値と比べない。実測で、明確なスパム(is_spam 0.96)が + // is_praise_only ちょうど 0.50 で弾かれた。守りたいのは「感謝の方がスパムらしさ + // より強いとき」だけなので、両者の大小で判定する。 + const isSpam = + announcement >= T.SPAM || (spamSignal >= T.SPAM && praise < spamSignal); + const needsSpamReview = + !isSpam && + praise < spamSignal && + Math.max(spamSignal, announcement) >= T.SPAM_REVIEW; + + const cat = choice(answers, 'category'); + const comp = choice(answers, 'component'); + const sev = score(answers, 'severity'); + const act = score(answers, 'actionability'); + const category = isSpam ? 'question' : cat.choice; + + return { + isSpam, + needsSpamReview, + category, + categoryConfidence: cat.confidence, + component: isSpam ? 'unknown' : comp.choice, + componentConfidence: comp.choice === 'unknown' ? 0 : comp.confidence, + triageLevel: resolveLevel(answers, isSpam, category, sev.score), + severity: sev.score, + actionability: act.score, + }; +} + +/** + * triageLevel を決める。 + * + * severity は「不具合がどれだけ重いか」の尺度なので、不具合でないものに当てても + * 意味を持たない。実測では要望に高い severity が付いて優先度が跳ね上がっていたため、 + * カテゴリで分岐させ、bug にだけ severity の全域を使う。 + * + * 当初は severity / breadth / actionability の加重和にしていたが、実測で breadth は + * 正解レベルと全く相関しなかった(urgent 0.82 / high 0.81 / medium 1.07 / low 1.04)。 + * 1 通のフィードバックには影響範囲の情報がほとんど含まれていないためで、質問ごと + * 削除した。actionability は着手順の材料として残し、優先度には入れない。 + */ +function resolveLevel( + answers: Record, + isSpam: boolean, + category: string, + severity: number +): Verdict['triageLevel'] { + if (isSpam) return 'low'; + if (category === 'praise' || category === 'question') return 'low'; + // 加重和では表現できないハードルール。単独で urgent に上げる。 + if (noul(answers, 'is_crash_or_data_loss') >= T.CRASH) return 'urgent'; + if (category === 'bug') { + if (severity >= T.BUG_URGENT) return 'urgent'; + if (severity >= T.BUG_HIGH) return 'high'; + if (severity >= T.BUG_MEDIUM) return 'medium'; + return 'low'; + } + // feature_request / improvement は不具合ではないので medium を上限にする + return severity >= T.REQUEST_MEDIUM ? 'medium' : 'low'; +} + +/** TypeSafe に渡す状態。本文だけでなく、判定の手がかりになる文脈も名前付きで渡す */ +function buildState(report: Report): Record { + return { + feedback: report.description, + report_type: report.reportType, + app_version: report.appVersion, + app_edition: report.appEdition, + auto_mode_enabled: report.autoModeEnabled, + os: report.deviceInfo?.osName ?? null, + has_stacktrace: Boolean(report.stacktrace), + }; +} + +/** + * 判定を 1 リクエストで取得する。質問は互いに独立で並列に評価されるため、 + * 一部の入力でしか使わない質問も同じリクエストに含めてよい。 + */ +export async function judgeFeedback( + env: Env, + report: Report +): Promise { + const body = JSON.stringify({ + state: buildState(report), + model: env.TYPESAFE_MODEL, + questions: QUESTIONS, + }); + + let lastError: Error | null = null; + for (let attempt = 0; attempt < RETRYABLE_ATTEMPTS; attempt++) { + const res = await fetch(API_URL, { + method: 'POST', + headers: { + Authorization: `Bearer ${env.TYPESAFE_API_KEY}`, + 'Content-Type': 'application/json', + }, + body, + }); + if (res.ok) { + const json = (await res.json()) as SystemOneResponse; + return compose(json.answers); + } + const text = await res.text().catch(() => ''); + lastError = new Error(`TypeSafe API ${res.status}: ${text.slice(0, 300)}`); + // 429(レート制限)と 529(過負荷)は時間を置けば通る。それ以外は再試行しても + // 同じ結果になるため即座に諦める。 + if (!RETRYABLE_STATUSES.has(res.status)) break; + const retryAfter = Number(res.headers.get('retry-after')); + // ヘッダが無いと get() は null を返し、Number(null) は 0。 + // isFinite(0) は true なので、正値であることまで確かめないと待機しない。 + const waitMs = + Number.isFinite(retryAfter) && retryAfter > 0 + ? retryAfter * 1000 + : RETRY_BASE_MS * 2 ** attempt; + await new Promise((resolve) => setTimeout(resolve, waitMs)); + } + throw lastError ?? new Error('TypeSafe API の呼び出しに失敗した'); +} diff --git a/src/index.ts b/src/index.ts index f43ac5a..e049e00 100644 --- a/src/index.ts +++ b/src/index.ts @@ -3,7 +3,10 @@ * 1 つの Worker に HTTP(fetch) / キュー(queue) / Cron(scheduled) の 3 ハンドラを集約する。 */ import { handleAgentChat, handleAgentChatStream } from './agent/handler'; -import { processFeedbackMessage } from './consumers/feedbackTriage'; +import { + FEEDBACK_RETRY_DELAY_SECONDS, + processFeedbackMessage, +} from './consumers/feedbackTriage'; import { withCallable } from './lib/callable'; import { handleAuthToken } from './routes/auth'; import { handleMaintenanceConfig, handleRemoteConfig } from './routes/config'; @@ -78,7 +81,10 @@ const worker: ExportedHandler = { message.ack(); } catch (e) { console.error(`Queue message failed (${batch.queue}):`, e); - message.retry(); + // 遅延なしで再試行すると、processFeedbackMessage が起票直後に書いた + // 冪等化マーカーを KV のネガティブキャッシュ越しに読めず、Issue を + // 作り直してしまう。キャッシュが切れてから再試行させる。 + message.retry({ delaySeconds: FEEDBACK_RETRY_DELAY_SECONDS }); } } }, diff --git a/src/lib/azure/tts.test.ts b/src/lib/azure/tts.test.ts deleted file mode 100644 index 6f79109..0000000 --- a/src/lib/azure/tts.test.ts +++ /dev/null @@ -1,56 +0,0 @@ -import { buildAzureSsml } from './tts'; - -describe('buildAzureSsml', () => { - it('wraps standard neural voices with prosody/style when provided', () => { - const ssml = buildAzureSsml('東京', 'ja-JP', 'ja-JP-NanamiNeural', { - pitch: '+1st', - style: 'narration-relaxed', - styleDegree: '1.5', - }); - - expect(ssml).toContain(''); - expect(ssml).toContain( - '' - ); - expect(ssml).toContain('name="ja-JP-NanamiNeural"'); - }); - - it('omits prosody and express-as for HD (DragonHD) voices', () => { - const ssml = buildAzureSsml( - '東京', - 'ja-JP', - 'ja-JP-Nanami:DragonHDLatestNeural', - { pitch: '+1st', style: 'narration-relaxed' } - ); - - // HD は / 非対応のため出力しない - expect(ssml).not.toContain('東京'); - }); - - it('omits prosody for HD English voices as well', () => { - const ssml = buildAzureSsml( - 'Tokyo', - 'en-US', - 'en-US-Jenny:DragonHDLatestNeural', - { pitch: '+1st' } - ); - - expect(ssml).not.toContain(' { - const inner = '3 番線'; - const ssml = buildAzureSsml( - inner, - 'ja-JP', - 'ja-JP-Nanami:DragonHDLatestNeural', - {} - ); - - expect(ssml).toContain(inner); - }); -}); diff --git a/src/lib/azure/tts.ts b/src/lib/azure/tts.ts deleted file mode 100644 index fc4bcd8..0000000 --- a/src/lib/azure/tts.ts +++ /dev/null @@ -1,108 +0,0 @@ -/** - * Azure Speech(Cognitive Services TTS)でテキストを音声に変換する。 - * Azure は SSML 必須。クライアントが送る `…` の中身を取り出し、 - * voice/lang/スタイル/プロソディを含む Azure 準拠 SSML に包み直して合成する。出力は MP3。 - */ -import { isAzureHdVoiceName } from '../../utils/ttsVoice'; -import { bytesToBase64 } from '../crypto'; - -// 音質。低ビットレートだと圧縮ノイズで機械っぽく聞こえるため既定を高めにする。 -const DEFAULT_OUTPUT_FORMAT = 'audio-48khz-192kbitrate-mono-mp3'; - -export interface TtsOptions { - /** X-Microsoft-OutputFormat。未指定なら高音質既定 */ - outputFormat?: string; - /** mstts:express-as の style(例: narration-relaxed, customerservice)。未指定なら付けない */ - style?: string; - /** style の強さ(0.01〜2。未指定なら付けない) */ - styleDegree?: string; - /** prosody pitch(例: -2%, +1st)。未指定なら付けない */ - pitch?: string; -} - -/** XML 属性値をエスケープする(", &, <, >, ' を含む値で不正 XML になるのを防ぐ)。 */ -const escapeXmlAttr = (value: string): string => - value - .replace(/&/g, '&') - .replace(/"/g, '"') - .replace(//g, '>') - .replace(/'/g, '''); - -/** クライアント SSML から外側の を剥がして中身だけ返す。 */ -const extractSpeakInner = (ssml: string): string => { - const trimmed = ssml.trim(); - const match = trimmed.match(/^]*>([\s\S]*)<\/speak>$/i); - return (match ? match[1] : trimmed).trim(); -}; - -export const buildAzureSsml = ( - inner: string, - languageCode: string, - voiceName: string, - opts: TtsOptions -): string => { - let content = inner; - - // HD(DragonHD)ボイスは と を非対応のため、 - // これらの装飾を付けると合成エラー・無視の原因になる。HD では出力しない。 - const isHd = isAzureHdVoiceName(voiceName); - - if (!isHd && opts.pitch) { - content = `${content}`; - } - - if (!isHd && opts.style) { - const degree = opts.styleDegree - ? ` styledegree="${escapeXmlAttr(opts.styleDegree)}"` - : ''; - content = `${content}`; - } - - return ( - '${content}` - ); -}; - -export interface SynthesizedAudio { - /** base64 エンコードされた MP3 */ - audioContent: string; - mimeType: 'audio/mpeg'; -} - -export const synthesizeSpeech = async ( - region: string, - subscriptionKey: string, - ssml: string, - languageCode: string, - voiceName: string, - opts: TtsOptions = {} -): Promise => { - const inner = extractSpeakInner(ssml); - const body = buildAzureSsml(inner, languageCode, voiceName, opts); - - const url = `https://${region}.tts.speech.microsoft.com/cognitiveservices/v1`; - const res = await fetch(url, { - method: 'POST', - headers: { - 'Ocp-Apim-Subscription-Key': subscriptionKey, - 'Content-Type': 'application/ssml+xml', - 'X-Microsoft-OutputFormat': opts.outputFormat || DEFAULT_OUTPUT_FORMAT, - 'User-Agent': 'trainlcd-worker', - }, - body, - signal: AbortSignal.timeout(30000), - }); - - if (!res.ok) { - const detail = await res.text().catch(() => ''); - throw new Error( - `Azure TTS returned ${res.status}: ${detail.slice(0, 500)}` - ); - } - - const buf = await res.arrayBuffer(); - return { audioContent: bytesToBase64(buf), mimeType: 'audio/mpeg' }; -}; diff --git a/src/lib/google/accessToken.ts b/src/lib/google/accessToken.ts index 87e9141..d8896fd 100644 Binary files a/src/lib/google/accessToken.ts and b/src/lib/google/accessToken.ts differ diff --git a/src/lib/google/tts.test.ts b/src/lib/google/tts.test.ts new file mode 100644 index 0000000..c56655d --- /dev/null +++ b/src/lib/google/tts.test.ts @@ -0,0 +1,170 @@ +import { + ALLOWED_CLIENT_SPEEDS, + buildSynthesizeRequestBody, + mimeTypeForFormat, + normalizeResponseFormat, + parseClientSpeed, + parsePitch, + parseSpeed, +} from './tts'; + +describe('parseClientSpeed', () => { + it('accepts the announcement speed presets', () => { + for (const speed of ALLOWED_CLIENT_SPEEDS) { + expect(parseClientSpeed(speed)).toBe(speed); + } + }); + + it('accepts a preset sent as a string', () => { + expect(parseClientSpeed('1.15')).toBe(1.15); + }); + + it('ignores values outside the presets so the cache cannot be fanned out', () => { + // 許可リスト方式。任意の値を通すと同じ文が速度違いで際限なくキャッシュされる + for (const speed of [0.9, 1.05, 1.2, 2, 0.25, 4]) { + expect(parseClientSpeed(speed)).toBeUndefined(); + } + }); + + it('treats missing and malformed values as unspecified', () => { + for (const speed of [ + undefined, + null, + '', + 'fast', + {}, + [], + Number.NaN, + Number.POSITIVE_INFINITY, + ]) { + expect(parseClientSpeed(speed)).toBeUndefined(); + } + }); +}); + +describe('buildSynthesizeRequestBody', () => { + it('sends the plain text with the voice and its locale', () => { + expect( + buildSynthesizeRequestBody({ + languageCode: 'ja-JP', + voiceName: 'ja-JP-Standard-B', + text: '次は、オオサキです', + }) + ).toEqual({ + input: { text: '次は、オオサキです' }, + voice: { languageCode: 'ja-JP', name: 'ja-JP-Standard-B' }, + audioConfig: { audioEncoding: 'MP3' }, + }); + }); + + it('omits speakingRate / pitch when they are not configured', () => { + const body = buildSynthesizeRequestBody({ + languageCode: 'en-US', + voiceName: 'en-US-Standard-G', + text: 'test', + opts: {}, + }); + expect(body.audioConfig).toEqual({ audioEncoding: 'MP3' }); + }); + + it('sends speakingRate / pitch as numbers', () => { + // 環境変数は文字列。そのまま送ると API のスキーマ検証に弾かれる + const body = buildSynthesizeRequestBody({ + languageCode: 'ja-JP', + voiceName: 'ja-JP-Standard-B', + text: 'test', + opts: { responseFormat: 'wav', speed: 1.15, pitch: -1.5 }, + }); + expect(body.audioConfig).toEqual({ + audioEncoding: 'LINEAR16', + speakingRate: 1.15, + pitch: -1.5, + }); + }); + + it('omits out-of-range values rather than sending an invalid request', () => { + for (const speed of [0.1, 4.5, Number.NaN, Number.POSITIVE_INFINITY]) { + const body = buildSynthesizeRequestBody({ + languageCode: 'ja-JP', + voiceName: 'ja-JP-Standard-B', + text: 'test', + opts: { speed }, + }); + expect(body.audioConfig).not.toHaveProperty('speakingRate'); + } + for (const pitch of [-20.5, 20.5, Number.NaN]) { + const body = buildSynthesizeRequestBody({ + languageCode: 'ja-JP', + voiceName: 'ja-JP-Standard-B', + text: 'test', + opts: { pitch }, + }); + expect(body.audioConfig).not.toHaveProperty('pitch'); + } + }); +}); + +describe('normalizeResponseFormat', () => { + it('accepts the formats Cloud TTS can return', () => { + for (const format of ['mp3', 'wav', 'opus']) { + expect(normalizeResponseFormat(format)).toBe(format); + } + }); + + it('normalizes case and falls back to mp3 for unknown values', () => { + expect(normalizeResponseFormat('WAV')).toBe('wav'); + expect(normalizeResponseFormat(' Opus ')).toBe('opus'); + // OpenAI 時代の設定値が残っていても 400 にせず mp3 で合成する + expect(normalizeResponseFormat('aac')).toBe('mp3'); + expect(normalizeResponseFormat('flac')).toBe('mp3'); + expect(normalizeResponseFormat('pcm')).toBe('mp3'); + expect(normalizeResponseFormat('')).toBe('mp3'); + expect(normalizeResponseFormat(undefined)).toBe('mp3'); + }); +}); + +describe('mimeTypeForFormat', () => { + it('maps each format to the container Cloud TTS actually returns', () => { + // LINEAR16 は RIFF ヘッダ付き、OGG_OPUS は Ogg コンテナで返る + expect(mimeTypeForFormat('mp3')).toBe('audio/mpeg'); + expect(mimeTypeForFormat('wav')).toBe('audio/wav'); + expect(mimeTypeForFormat('opus')).toBe('audio/ogg'); + expect(mimeTypeForFormat('unknown')).toBe('audio/mpeg'); + }); +}); + +describe('parseSpeed', () => { + it('parses a numeric string from the environment', () => { + expect(parseSpeed('1.1')).toBe(1.1); + expect(parseSpeed(' 0.25 ')).toBe(0.25); + expect(parseSpeed('4')).toBe(4); + }); + + it('accepts numbers as-is', () => { + expect(parseSpeed(1.5)).toBe(1.5); + }); + + it('rejects out-of-range and non-numeric values', () => { + expect(parseSpeed('0.24')).toBeUndefined(); + expect(parseSpeed('4.01')).toBeUndefined(); + expect(parseSpeed('fast')).toBeUndefined(); + expect(parseSpeed('')).toBeUndefined(); + expect(parseSpeed(undefined)).toBeUndefined(); + }); +}); + +describe('parsePitch', () => { + it('accepts the semitone range, including negatives', () => { + expect(parsePitch('-20')).toBe(-20); + expect(parsePitch('0')).toBe(0); + expect(parsePitch('20')).toBe(20); + expect(parsePitch(2.5)).toBe(2.5); + }); + + it('rejects out-of-range and non-numeric values', () => { + expect(parsePitch('-20.1')).toBeUndefined(); + expect(parsePitch('20.1')).toBeUndefined(); + expect(parsePitch('high')).toBeUndefined(); + expect(parsePitch(undefined)).toBeUndefined(); + }); +}); diff --git a/src/lib/google/tts.ts b/src/lib/google/tts.ts new file mode 100644 index 0000000..c0dbcf2 --- /dev/null +++ b/src/lib/google/tts.ts @@ -0,0 +1,187 @@ +/** + * Google Cloud Text-to-Speech API(text:synthesize)でテキストを音声に変換する。 + * + * ボイスは端末内蔵 TTS 相当の系統(Standard / Wavenet / Neural2)を使う。これらは + * 読み方のプロンプト指示を受け付けないため、声の調子は audioConfig の + * speakingRate / pitch で調整する。 + * + * 認証は API キーではなくサービスアカウント(OAuth2)。Cloud TTS は Cloudflare + * AI Gateway の対応プロバイダではないため、Google へ直行する。 + * 応答の audioContent は API 時点で base64 なので、そのまま返して再変換しない。 + */ +import { getGoogleAccessToken } from './accessToken'; + +const SYNTHESIZE_URL = 'https://texttospeech.googleapis.com/v1/text:synthesize'; +const TTS_SCOPE = 'https://www.googleapis.com/auth/cloud-platform'; + +// 合成は数秒で返るが、詰まったときに Worker の CPU/実行時間を食い潰さないよう +// OpenAI / Azure 時代と同じ上限で打ち切る。 +const REQUEST_TIMEOUT_MS = 30_000; + +/** + * 応答フォーマットと audioEncoding / MIME の対応。 + * Cloud TTS の LINEAR16 は RIFF ヘッダ付き(= WAV)、OGG_OPUS は Ogg コンテナで + * 返るため、拡張子判定に使える MIME をこちらで確定させる。 + * aac / flac / 生 PCM は Cloud TTS に無いので mp3 へ倒す。 + */ +const FORMATS = { + mp3: { encoding: 'MP3', mimeType: 'audio/mpeg' }, + wav: { encoding: 'LINEAR16', mimeType: 'audio/wav' }, + opus: { encoding: 'OGG_OPUS', mimeType: 'audio/ogg' }, +} as const; + +export type ResponseFormat = keyof typeof FORMATS; + +export interface TtsOptions { + /** 応答フォーマット。未指定なら mp3 */ + responseFormat?: string; + /** 読み上げ速度(0.25〜4.0)。未指定なら付けない(等速) */ + speed?: number; + /** 声の高さ(-20.0〜20.0 セミトーン)。未指定なら付けない */ + pitch?: number; +} + +export interface SynthesizeSpeechParams { + /** Cloud TTS を呼べるサービスアカウント鍵 JSON */ + saKeyJson: string; + /** ボイスのロケール(例: ja-JP)。ボイス名と食い違うと API が 400 を返す */ + languageCode: string; + voiceName: string; + /** 読み上げるプレーンテキスト */ + text: string; + opts?: TtsOptions; +} + +export interface SynthesizedAudio { + /** base64 エンコードされた音声 */ + audioContent: string; + mimeType: string; +} + +/** + * 応答フォーマットを既知の値へ正規化する。環境変数由来の任意文字列(`MP3` の + * ような大文字や、OpenAI 時代の aac / flac / pcm)をそのまま送ると Cloud TTS が + * 400 を返し、MIME も引けなくなる。 + */ +export const normalizeResponseFormat = (format?: string): ResponseFormat => { + const value = format?.trim().toLowerCase() ?? ''; + return value in FORMATS ? (value as ResponseFormat) : 'mp3'; +}; + +/** 正規化済みフォーマットに対応する MIME。 */ +export const mimeTypeForFormat = (format?: string): string => + FORMATS[normalizeResponseFormat(format)].mimeType; + +/** + * 読み上げ速度を数値へ正規化する。環境変数は文字列なので、そのまま送ると + * API のスキーマ検証(number)に弾かれる。範囲外・非数は未指定として扱う。 + */ +export const parseSpeed = (speed?: string | number): number | undefined => + parseAudioNumber(speed, 0.25, 4.0); + +/** + * アプリから届く読み上げ速度として受け付ける値。アナウンス設定のプリセットと + * 一対一で対応する。任意の値を通すと同じ文が速度違いで R2/KV に無制限へ積み上がり、 + * 合成回数(=文字数課金)もその分だけ増えるため、許可リスト方式にしている。 + * MobileApp の REMOTE_TTS_SPEED_RATES と必ず一致させること。 + */ +export const ALLOWED_CLIENT_SPEEDS = [0.85, 1.0, 1.15] as const; + +// JSON の数値は 1.15 のように二進で表現しきれない値があるため、厳密比較はしない。 +const SPEED_EPSILON = 1e-6; + +/** + * リクエストで指定された読み上げ速度を正規化する。許可リストに無い値・不正な値は + * 「未指定」として扱い、呼び出し側で環境変数の既定値へ倒す。古いアプリは speed を + * 送ってこないため、未指定は正常系であってエラーにはしない。 + */ +export const parseClientSpeed = (value: unknown): number | undefined => { + if (typeof value !== 'number' && typeof value !== 'string') { + return undefined; + } + const parsed = parseSpeed(value); + if (parsed === undefined) { + return undefined; + } + return ALLOWED_CLIENT_SPEEDS.find( + (allowed) => Math.abs(allowed - parsed) < SPEED_EPSILON + ); +}; + +/** 声の高さ(セミトーン)を数値へ正規化する。範囲外・非数は未指定として扱う。 */ +export const parsePitch = (pitch?: string | number): number | undefined => + parseAudioNumber(pitch, -20.0, 20.0); + +const parseAudioNumber = ( + value: string | number | undefined, + min: number, + max: number +): number | undefined => { + if (value === undefined || value === null || value === '') { + return undefined; + } + const parsed = typeof value === 'number' ? value : Number(value.trim()); + if (!Number.isFinite(parsed)) { + return undefined; + } + return parsed >= min && parsed <= max ? parsed : undefined; +}; + +/** Cloud TTS へ送るリクエストボディを組み立てる。 */ +export const buildSynthesizeRequestBody = (params: { + languageCode: string; + voiceName: string; + text: string; + opts?: TtsOptions; +}): Record => { + const { languageCode, voiceName, text, opts = {} } = params; + const speed = parseSpeed(opts.speed); + const pitch = parsePitch(opts.pitch); + return { + input: { text }, + voice: { languageCode, name: voiceName }, + audioConfig: { + audioEncoding: + FORMATS[normalizeResponseFormat(opts.responseFormat)].encoding, + ...(speed !== undefined ? { speakingRate: speed } : {}), + ...(pitch !== undefined ? { pitch } : {}), + }, + }; +}; + +export const synthesizeSpeech = async ( + params: SynthesizeSpeechParams +): Promise => { + const { saKeyJson, languageCode, voiceName, text, opts = {} } = params; + const accessToken = await getGoogleAccessToken(saKeyJson, TTS_SCOPE); + + const res = await fetch(SYNTHESIZE_URL, { + method: 'POST', + headers: { + Authorization: `Bearer ${accessToken}`, + 'Content-Type': 'application/json', + 'User-Agent': 'trainlcd-worker', + }, + body: JSON.stringify( + buildSynthesizeRequestBody({ languageCode, voiceName, text, opts }) + ), + signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS), + }); + + if (!res.ok) { + const detail = await res.text().catch(() => ''); + throw new Error( + `Google TTS returned ${res.status}: ${detail.slice(0, 500)}` + ); + } + + const json = (await res.json()) as { audioContent?: unknown }; + if (typeof json.audioContent !== 'string' || json.audioContent.length === 0) { + throw new Error('Google TTS response missing audioContent'); + } + + return { + audioContent: json.audioContent, + mimeType: mimeTypeForFormat(opts.responseFormat), + }; +}; diff --git a/src/lib/ttsCache.test.ts b/src/lib/ttsCache.test.ts new file mode 100644 index 0000000..ed7d841 --- /dev/null +++ b/src/lib/ttsCache.test.ts @@ -0,0 +1,127 @@ +import type { Env } from '../types'; +import { writeTtsCache } from './ttsCache'; + +const createEnv = () => { + const put = jest.fn().mockResolvedValue(undefined); + const kvPut = jest.fn().mockResolvedValue(undefined); + return { + env: { + TTS_BUCKET: { put }, + TTS_KV: { put: kvPut }, + } as unknown as Env, + put, + kvPut, + }; +}; + +const basePayload = { + id: 'abc123', + model: 'google-cloud-tts', + jaAudioContent: 'QQ==', + enAudioContent: 'QQ==', + jaAudioMimeType: 'audio/mpeg', + enAudioMimeType: 'audio/mpeg', + textJa: '次は、オオサキです', + textEn: 'The next station is Osaki.', + voiceJa: 'nova', + voiceEn: 'nova', +}; + +describe('writeTtsCache', () => { + it('stores both languages and records the metadata', async () => { + const { env, put, kvPut } = createEnv(); + + await writeTtsCache(basePayload, env); + + expect(put).toHaveBeenCalledTimes(2); + expect(put.mock.calls[0][0]).toBe('caches/tts/ja/abc123.mp3'); + expect(put.mock.calls[1][0]).toBe('caches/tts/en/abc123.mp3'); + + const meta = JSON.parse(kvPut.mock.calls[0][1]); + expect(kvPut.mock.calls[0][0]).toBe('voice:abc123'); + expect(meta).toEqual( + expect.objectContaining({ + id: 'abc123', + model: 'google-cloud-tts', + pathJa: 'caches/tts/ja/abc123.mp3', + pathEn: 'caches/tts/en/abc123.mp3', + textJa: '次は、オオサキです', + textEn: 'The next station is Osaki.', + }) + ); + }); + + it('stores only the language that was synthesized', async () => { + // ユーザーが英語を無効にしている場合、英語は合成もキャッシュもしない + const { env, put, kvPut } = createEnv(); + + await writeTtsCache( + { + ...basePayload, + enAudioContent: undefined, + enAudioMimeType: undefined, + textEn: '', + voiceEn: undefined, + }, + env + ); + + expect(put).toHaveBeenCalledTimes(1); + expect(put.mock.calls[0][0]).toBe('caches/tts/ja/abc123.mp3'); + + const meta = JSON.parse(kvPut.mock.calls[0][1]); + expect(meta.pathJa).toBe('caches/tts/ja/abc123.mp3'); + expect(meta).not.toHaveProperty('pathEn'); + expect(meta).not.toHaveProperty('textEn'); + }); + + it('picks the file extension from the mime type', async () => { + const { env, put } = createEnv(); + + await writeTtsCache( + { + ...basePayload, + jaAudioMimeType: 'audio/wav', + enAudioMimeType: 'audio/pcm;rate=24000', + }, + env + ); + + expect(put.mock.calls[0][0]).toBe('caches/tts/ja/abc123.wav'); + expect(put.mock.calls[1][0]).toBe('caches/tts/en/abc123.pcm'); + }); + + it.each([ + ['audio/opus', 'opus'], + ['audio/aac', 'aac'], + ['audio/flac', 'flac'], + ])('stores %s as .%s', async (mimeType, ext) => { + // TTS_RESPONSE_FORMAT は opus/aac/flac も受け付けるため拡張子も合わせる + const { env, put } = createEnv(); + + await writeTtsCache( + { ...basePayload, jaAudioMimeType: mimeType, enAudioMimeType: mimeType }, + env + ); + + expect(put.mock.calls[0][0]).toBe(`caches/tts/ja/abc123.${ext}`); + expect(put.mock.calls[1][0]).toBe(`caches/tts/en/abc123.${ext}`); + }); + + it('writes nothing when no audio was produced', async () => { + const { env, put, kvPut } = createEnv(); + const errorSpy = jest.spyOn(console, 'error').mockImplementation(); + + await writeTtsCache( + { + id: 'abc123', + model: 'google-cloud-tts', + }, + env + ); + + expect(put).not.toHaveBeenCalled(); + expect(kvPut).not.toHaveBeenCalled(); + errorSpy.mockRestore(); + }); +}); diff --git a/src/lib/ttsCache.ts b/src/lib/ttsCache.ts index c1c5b3e..4707601 100644 --- a/src/lib/ttsCache.ts +++ b/src/lib/ttsCache.ts @@ -2,14 +2,24 @@ * 合成済み TTS 音声を R2 に保存し、メタを KV に書き込む。 * Queues のメッセージ上限(128KB)に音声が収まらないため、キューを介さず * /tts ハンドラから ctx.waitUntil で直接呼ぶ。 + * + * 日英どちらか片方だけの合成もありうる(ユーザーが無効にしている言語は + * そもそも合成しない)ため、届いた言語だけを保存する。 */ import type { Env, TtsCachePayload } from '../types'; import { base64ToBytes } from './crypto'; -const getCacheFileExtension = (mimeType: string): 'mp3' | 'wav' | 'pcm' => { +type CacheFileExtension = 'mp3' | 'wav' | 'opus' | 'aac' | 'flac' | 'pcm'; + +// TTS_RESPONSE_FORMAT は opus / aac / flac も受け付けるため、R2 の拡張子も +// 形式に合わせる。判定できない場合のみ生 PCM 扱いにする。 +const getCacheFileExtension = (mimeType: string): CacheFileExtension => { const normalized = mimeType.toLowerCase(); if (normalized.includes('mpeg') || normalized.includes('mp3')) return 'mp3'; if (normalized.includes('wav')) return 'wav'; + if (normalized.includes('opus')) return 'opus'; + if (normalized.includes('aac')) return 'aac'; + if (normalized.includes('flac')) return 'flac'; return 'pcm'; }; @@ -23,13 +33,14 @@ export const writeTtsCache = async ( enAudioContent, jaAudioMimeType, enAudioMimeType, - ssmlJa, - ssmlEn, + textJa, + textEn, + model, voiceJa, voiceEn, } = payload; - if (!id || !jaAudioContent || !enAudioContent) { + if (!id || (!jaAudioContent && !enAudioContent)) { console.error('Invalid payload for tts cache', { hasId: !!id, hasJa: !!jaAudioContent, @@ -38,32 +49,49 @@ export const writeTtsCache = async ( return; } - const jaContentType = jaAudioMimeType || 'audio/pcm'; - const enContentType = enAudioMimeType || 'audio/pcm'; - const jaPath = `caches/tts/ja/${id}.${getCacheFileExtension(jaContentType)}`; - const enPath = `caches/tts/en/${id}.${getCacheFileExtension(enContentType)}`; + const jaContentType = jaAudioMimeType || 'audio/mpeg'; + const enContentType = enAudioMimeType || 'audio/mpeg'; + const jaPath = jaAudioContent + ? `caches/tts/ja/${id}.${getCacheFileExtension(jaContentType)}` + : undefined; + const enPath = enAudioContent + ? `caches/tts/en/${id}.${getCacheFileExtension(enContentType)}` + : undefined; await Promise.all([ - env.TTS_BUCKET.put(jaPath, base64ToBytes(jaAudioContent), { - httpMetadata: { contentType: jaContentType }, - }), - env.TTS_BUCKET.put(enPath, base64ToBytes(enAudioContent), { - httpMetadata: { contentType: enContentType }, - }), + jaAudioContent && jaPath + ? env.TTS_BUCKET.put(jaPath, base64ToBytes(jaAudioContent), { + httpMetadata: { contentType: jaContentType }, + }) + : null, + enAudioContent && enPath + ? env.TTS_BUCKET.put(enPath, base64ToBytes(enAudioContent), { + httpMetadata: { contentType: enContentType }, + }) + : null, ]); await env.TTS_KV.put( `voice:${id}`, JSON.stringify({ id, - ssmlJa, - pathJa: jaPath, - jaAudioMimeType: jaContentType, - voiceJa, - ssmlEn, - pathEn: enPath, - enAudioMimeType: enContentType, - voiceEn, + model, + ...(jaPath + ? { + textJa, + pathJa: jaPath, + jaAudioMimeType: jaContentType, + voiceJa, + } + : {}), + ...(enPath + ? { + textEn, + pathEn: enPath, + enAudioMimeType: enContentType, + voiceEn, + } + : {}), createdAt: new Date().toISOString(), }) ); diff --git a/src/models/ai.ts b/src/models/ai.ts index 404ddbf..4e46f10 100644 --- a/src/models/ai.ts +++ b/src/models/ai.ts @@ -6,12 +6,26 @@ export const AI_CATEGORIES = [ 'feature_request', 'improvement', 'question', + /** 感謝・称賛・応援。対応は不要だがスパムではない */ + 'praise', ] as const; export type AICategory = (typeof AI_CATEGORIES)[number]; export const AI_TRIAGE_LEVELS = ['urgent', 'high', 'medium', 'low'] as const; export type AITriageLevel = (typeof AI_TRIAGE_LEVELS)[number]; +/** + * 原因があると推定されるコンポーネント。公開リポジトリへの起票先の決定に使う。 + * 特定できない場合は AIReport.component が null になる。 + */ +export const AI_COMPONENTS = [ + 'mobile_app', + 'station_api', + 'functions', + 'website', +] as const; +export type AIComponent = (typeof AI_COMPONENTS)[number]; + export type AIReport = { /** レポートのタイトル */ title: string; @@ -29,6 +43,10 @@ export type AIReport = { category: AICategory; /** トリアージ(優先度)レベル */ triageLevel: AITriageLevel; + /** 原因があると推定されるコンポーネント(特定できなければ null) */ + component: AIComponent | null; + /** component の推定信頼度 (0.0 - 1.0)。component が null のときは 0 */ + componentConfidence: number; }; export type FewShotItem = { diff --git a/src/routes/feedback.test.ts b/src/routes/feedback.test.ts new file mode 100644 index 0000000..6dbf8e6 --- /dev/null +++ b/src/routes/feedback.test.ts @@ -0,0 +1,77 @@ +import { CallableError } from '../lib/callable'; +import type { Env } from '../types'; +import { handleFeedback } from './feedback'; + +jest.mock('../lib/auth/session', () => ({ + verifySessionToken: jest.fn(async () => 'verified-uid'), +})); + +const send = jest.fn(async (_message: unknown) => undefined); +const env = { FEEDBACK_QUEUE: { send } } as unknown as Env; + +const buildRequest = (report: unknown): Request => + new Request('https://example.com/postFeedback', { + method: 'POST', + headers: { + 'content-type': 'application/json; charset=UTF-8', + Authorization: 'Bearer dummy', + }, + body: JSON.stringify({ data: { report } }), + }); + +const baseReport = { + id: 'feedback-id', + reportType: 'feedback', + description: '遅い', + reporterUid: 'client-claimed-uid', +}; + +describe('handleFeedback', () => { + afterEach(() => { + jest.clearAllMocks(); + }); + + // 文字数の下限は撤廃したため、短い本文もそのままキューへ流す + it('queues a short report', async () => { + const res = await handleFeedback(buildRequest(baseReport), env); + + expect(res.status).toBe(200); + expect(send).toHaveBeenCalledTimes(1); + expect(send.mock.calls[0][0]).toMatchObject({ + id: 'feedback-id', + report: { description: '遅い', reporterUid: 'verified-uid' }, + }); + }); + + // code は HTTP ステータスへ直結する(invalid-argument なら 400)。CallableError で + // あることしか見ていないと、internal へ変わって 500 を返すようになっても気づけない。 + // アプリは 4xx と 5xx で扱いを変えられるため、コードまで固定する。 + it.each([ + ['empty', ''], + ['whitespace only', ' \n\t '], + ['number', 1], + ['null', null], + ])( + 'rejects a %s description without queueing', + async (_label, description) => { + const rejected = handleFeedback( + buildRequest({ ...baseReport, description }), + env + ); + + await expect(rejected).rejects.toThrow(CallableError); + await expect(rejected).rejects.toMatchObject({ + code: 'invalid-argument', + }); + expect(send).not.toHaveBeenCalled(); + } + ); + + it('rejects a missing description without queueing', async () => { + const rejected = handleFeedback(buildRequest({ id: 'feedback-id' }), env); + + await expect(rejected).rejects.toThrow(/report.description required/); + await expect(rejected).rejects.toMatchObject({ code: 'invalid-argument' }); + expect(send).not.toHaveBeenCalled(); + }); +}); diff --git a/src/routes/feedback.ts b/src/routes/feedback.ts index 9971299..5a9cccc 100644 --- a/src/routes/feedback.ts +++ b/src/routes/feedback.ts @@ -19,6 +19,11 @@ export const handleFeedback = async ( if (!report?.id) { throw new CallableError('invalid-argument', 'report.id required'); } + // 本文が空白のみのフィードバックはトリアージへ流さない。文字数の下限は設けず、 + // 短い本文やクラッシュレポートの短いエラーメッセージはそのまま受け付ける。 + if (typeof report.description !== 'string' || !report.description.trim()) { + throw new CallableError('invalid-argument', 'report.description required'); + } // reporterUid はクライアント申告を信用せず、検証済みトークンの sub で上書きする。 // (他ユーザーの UID を名乗って Issue/Discord に載せるなりすましを防ぐ) diff --git a/src/routes/tts.test.ts b/src/routes/tts.test.ts new file mode 100644 index 0000000..3ba15de --- /dev/null +++ b/src/routes/tts.test.ts @@ -0,0 +1,86 @@ +import { CallableError } from '../lib/callable'; +import { computeId, parseTtsText } from './tts'; + +describe('parseTtsText', () => { + it('returns the trimmed text', () => { + expect(parseTtsText(' 次は、オオサキです ', 'textJa')).toBe( + '次は、オオサキです' + ); + }); + + it('treats undefined/null/empty as "language not requested"', () => { + // 合成は文字数課金のため、アプリは無効な言語を送ってこない + expect(parseTtsText(undefined, 'textJa')).toBe(''); + expect(parseTtsText(null, 'textJa')).toBe(''); + expect(parseTtsText(' ', 'textJa')).toBe(''); + }); + + it('rejects non-string values', () => { + expect(() => parseTtsText(42, 'textJa')).toThrow(CallableError); + expect(() => parseTtsText({}, 'textEn')).toThrow(/must be a string/); + }); + + it('strips tags so stray SSML is never read aloud', () => { + // Cloud TTS の input.text は SSML を解釈せずタグをそのまま読み上げてしまう + expect( + parseTtsText('次は大崎です', 'textJa') + ).toBe('次はオオサキです'); + }); + + it('leaves plain text untouched', () => { + expect(parseTtsText('The next station is Osaki, J-Y 24.', 'textEn')).toBe( + 'The next station is Osaki, J-Y 24.' + ); + }); + + it('rejects text beyond the byte limit', () => { + // 日本語は 1 文字 3 バイトなので 4000 バイト超はすぐ作れる + const long = 'あ'.repeat(1400); + expect(() => parseTtsText(long, 'textJa')).toThrow(/byte limit/); + }); + + it('measures the limit in bytes, not characters', () => { + // 1300 文字 = 3900 バイトなので通る + expect(parseTtsText('あ'.repeat(1300), 'textJa')).toHaveLength(1300); + }); +}); + +describe('computeId', () => { + const base = { + enVoiceName: 'en-US-Standard-G', + jaVoiceName: 'ja-JP-Standard-B', + pitch: null as number | null, + responseFormat: 'mp3', + speed: null as number | null, + textEn: 'The next station is Osaki.', + textJa: '次は、オオサキです', + }; + + it('is stable for identical input', async () => { + expect(await computeId(base)).toBe(await computeId(base)); + }); + + it.each([ + ['textJa', { textJa: '次は、シンジュクです' }], + ['textEn', { textEn: 'The next station is Shinjuku.' }], + ['jaVoiceName', { jaVoiceName: 'ja-JP-Wavenet-A' }], + ['enVoiceName', { enVoiceName: 'en-US-Wavenet-F' }], + // responseFormat / speed はかつてネストしたオブジェクトに置いていたため、 + // JSON.stringify の配列 replacer に落とされて ID に反映されていなかった + ['responseFormat', { responseFormat: 'wav' }], + ['speed', { speed: 1.25 }], + ['pitch', { pitch: 1.5 }], + ])('changes when %s changes', async (_name, override) => { + expect(await computeId({ ...base, ...override })).not.toBe( + await computeId(base) + ); + }); + + it('distinguishes single-language requests from bilingual ones', async () => { + // 片言語リクエストが両言語のキャッシュへヒットしないこと + const jaOnly = await computeId({ ...base, textEn: '' }); + const enOnly = await computeId({ ...base, textJa: '' }); + const both = await computeId(base); + expect(new Set([jaOnly, enOnly, both]).size).toBe(3); + }); +}); diff --git a/src/routes/tts.ts b/src/routes/tts.ts index be2b71f..78c4194 100644 --- a/src/routes/tts.ts +++ b/src/routes/tts.ts @@ -1,23 +1,40 @@ -/** POST /tts — Azure Speech で音声合成し、KV/R2 キャッシュを介して返す(callable 互換)。 */ +/** POST /tts — Google Cloud TTS で音声合成し、KV/R2 キャッシュを介して返す(callable 互換)。 */ import { verifySessionToken } from '../lib/auth/session'; -import { synthesizeSpeech, type TtsOptions } from '../lib/azure/tts'; import { CallableError, callableSuccess, parseCallableData, } from '../lib/callable'; import { bytesToBase64, sha256Hex } from '../lib/crypto'; +import { + normalizeResponseFormat, + parseClientSpeed, + parsePitch, + parseSpeed, + synthesizeSpeech, + type TtsOptions, +} from '../lib/google/tts'; import { writeTtsCache } from '../lib/ttsCache'; import type { Env } from '../types'; import { normalizeRomanText } from '../utils/normalize'; import { stripSsml, utf8ByteLength } from '../utils/ssml'; -import { resolveAzureVoiceName } from '../utils/ttsVoice'; +import { + languageCodeFromVoiceName, + resolveGoogleVoiceName, +} from '../utils/ttsVoice'; +/** + * model / instructions* は OpenAI(gpt-4o-mini-tts) 時代のフィールド。Cloud TTS の + * Standard 系ボイスにはモデル指定も読み方のプロンプト指示も無いため受け取っても + * 使わないが、旧バージョンのアプリが送ってきても壊れないよう型としては残す。 + */ interface TtsRequest { - ssmlJa?: unknown; - ssmlEn?: unknown; + textJa?: unknown; + textEn?: unknown; jaVoiceName?: unknown; enVoiceName?: unknown; + // アナウンス設定で選んだ読み上げ速度。プリセット外の値は無視される + speed?: unknown; } interface TtsConfig { @@ -25,6 +42,9 @@ interface TtsConfig { enVoiceName?: string; } +/** キャッシュメタに残す合成エンジン名(find-tts-cache の表示用)。 */ +const TTS_ENGINE = 'google-cloud-tts'; + interface VoiceCacheMeta { pathJa?: string; pathEn?: string; @@ -33,8 +53,9 @@ interface VoiceCacheMeta { } const TEXT_BYTE_LIMIT = 4000; -const RAW_SSML_BYTE_LIMIT = 10000; -const HASH_VERSION = 12; +// 合成エンジンの入れ替え(OpenAI → Google Cloud TTS)で同じ入力でも音声が変わるため、 +// 旧キャッシュへヒットしないよう版を上げる +const HASH_VERSION = 14; const TTS_CONFIG_CACHE_TTL_MS = 5 * 60 * 1000; let ttsConfigCache: { data: TtsConfig; fetchedAt: number } | null = null; @@ -57,26 +78,53 @@ const getTtsConfig = async (env: Env): Promise => { } }; -const computeId = async (payload: { +// JSON.stringify の第2引数に配列を渡すと「その名前のキーだけ」を全階層で +// 直列化する。ネストしたオプションは名前がリストに無いと丸ごと落ちるため、 +// キャッシュキーへ含めたい値はすべてトップレベルへ平坦化して渡すこと。 +export const computeId = async (payload: { enVoiceName: string; jaVoiceName: string; - ssmlEn: string; - ssmlJa: string; - ttsOptions: TtsOptions; + pitch: number | null; + responseFormat: string; + speed: number | null; + textEn: string; + textJa: string; }): Promise => { const obj = { ...payload, version: HASH_VERSION } as const; const hashPayload = JSON.stringify(obj, Object.keys(obj).sort()); return sha256Hex(hashPayload); }; -const requireString = (value: unknown, name: string): string => { - if (typeof value !== 'string' || value.length === 0) { +/** + * 読み上げ対象テキストを受け取り、検証済みのプレーンテキストを返す。 + * 未指定・空文字は「その言語を要求しない」を意味する(合成は文字数課金のため、 + * アプリはユーザーが無効にしている言語を送ってこない)。 + */ +export const parseTtsText = (value: unknown, name: string): string => { + if (value === undefined || value === null) { + return ''; + } + if (typeof value !== 'string') { + throw new CallableError( + 'invalid-argument', + `"${name}" must be a string if provided` + ); + } + // Cloud TTS の input.text は SSML を解釈せずタグをそのまま読み上げるため、万一 + // タグが紛れ込んでも読ませない。プレーンテキストには実質作用しない。 + const stripped = stripSsml(value).trim(); + if (stripped.length === 0) { + return ''; + } + + const bytes = utf8ByteLength(stripped); + if (bytes > TEXT_BYTE_LIMIT) { throw new CallableError( 'invalid-argument', - `The function must be called with one argument "${name}" containing the message to add.` + `${name} exceeds ${TEXT_BYTE_LIMIT} byte limit (${bytes} bytes)` ); } - return value; + return stripped; }; export const handleTts = async ( @@ -88,107 +136,88 @@ export const handleTts = async ( const data = await parseCallableData(req); - const ssmlJa = requireString(data.ssmlJa, 'ssmlJa'); - // 生入力を保持し、バイト数上限は正規化前の値で判定する(正規化での展開/削除で - // 本来通る入力を弾いたり、上限超え入力を通したりしないため)。 - const rawSsmlEn = requireString(data.ssmlEn, 'ssmlEn'); - const ssmlEn = normalizeRomanText(rawSsmlEn); - if (ssmlEn.trim().length === 0) { + const textJa = parseTtsText(data.textJa, 'textJa'); + // 英語は駅名の表記ゆれ(全角記号・略記・長音符・大文字表記)を吸収してから合成する + const textEn = normalizeRomanText(parseTtsText(data.textEn, 'textEn')).trim(); + + const wantsJa = textJa.length > 0; + const wantsEn = textEn.length > 0; + if (!wantsJa && !wantsEn) { throw new CallableError( 'invalid-argument', - 'The function must be called with one argument "ssmlEn" containing the message to add.' + 'The function must be called with at least one of "textJa" or "textEn" containing the text to speak.' + ); + } + + if (!env.GOOGLE_TTS_SA_KEY) { + throw new CallableError( + 'failed-precondition', + 'GOOGLE_TTS_SA_KEY is not configured' ); } const ttsConfig = await getTtsConfig(env); - const jaVoiceName = resolveAzureVoiceName( + const jaVoiceName = resolveGoogleVoiceName( data.jaVoiceName, ttsConfig.jaVoiceName, - env.TTS_JA_VOICE_NAME + env.TTS_JA_VOICE_NAME, + 'ja' ); - const enVoiceName = resolveAzureVoiceName( + const enVoiceName = resolveGoogleVoiceName( data.enVoiceName, ttsConfig.enVoiceName, - env.TTS_EN_VOICE_NAME + env.TTS_EN_VOICE_NAME, + 'en' ); - const strippedJa = stripSsml(ssmlJa); - const strippedEn = stripSsml(ssmlEn); - if (strippedJa.trim().length === 0) { - throw new CallableError( - 'invalid-argument', - 'ssmlJa contains no visible text after stripping SSML tags' - ); - } - if (strippedEn.trim().length === 0) { - throw new CallableError( - 'invalid-argument', - 'ssmlEn contains no visible text after stripping SSML tags' - ); - } - - const jaTextBytes = utf8ByteLength(strippedJa); - const enTextBytes = utf8ByteLength(strippedEn); - if (jaTextBytes > TEXT_BYTE_LIMIT) { - throw new CallableError( - 'invalid-argument', - `ssmlJa text exceeds ${TEXT_BYTE_LIMIT} byte limit (${jaTextBytes} bytes)` - ); - } - if (enTextBytes > TEXT_BYTE_LIMIT) { - throw new CallableError( - 'invalid-argument', - `ssmlEn text exceeds ${TEXT_BYTE_LIMIT} byte limit (${enTextBytes} bytes)` - ); - } - - // 可視テキストだけでなく生 SSML のバイト長にも上限を設け、タグ膨張入力を弾く - if ( - utf8ByteLength(ssmlJa) > RAW_SSML_BYTE_LIMIT || - utf8ByteLength(rawSsmlEn) > RAW_SSML_BYTE_LIMIT - ) { - throw new CallableError( - 'invalid-argument', - `raw SSML exceeds ${RAW_SSML_BYTE_LIMIT} byte limit` - ); - } - - // 合成オプションもキャッシュキーに含める(outputFormat/style/styleDegree/pitch を - // 変えたら別の音声になるため、同じ voice:${id} を再利用させない)。 - const ttsOptions: TtsOptions = { - outputFormat: env.AZURE_TTS_OUTPUT_FORMAT || undefined, - style: env.AZURE_TTS_STYLE || undefined, - styleDegree: env.AZURE_TTS_STYLE_DEGREE || undefined, - pitch: env.AZURE_TTS_PITCH || undefined, - }; + // 環境変数は文字列なので、送信前に正規化した値を作る。この正規化後の値を + // そのままキャッシュキーにも使い、設定変更が確実に別 ID になるようにする。 + const responseFormat = normalizeResponseFormat(env.TTS_RESPONSE_FORMAT); + // 速度はアプリの設定を優先し、未指定・プリセット外なら環境変数の既定値を使う。 + // 速度は computeId に含まれるため、選択が変わればキャッシュも自動的に分かれる。 + const speed = parseClientSpeed(data.speed) ?? parseSpeed(env.TTS_SPEED); + const pitch = parsePitch(env.TTS_PITCH); + const ttsOptions: TtsOptions = { responseFormat, speed, pitch }; const id = await computeId({ enVoiceName, jaVoiceName, - ssmlEn, - ssmlJa, - ttsOptions, + pitch: pitch ?? null, + responseFormat, + speed: speed ?? null, + textEn, + textJa, }); // --- キャッシュ照会 --- + // id は「どの言語を要求したか」まで含めて決まるため、要求した言語のパスが + // 揃っていれば同じ組み合わせの再放送とみなせる。 const meta = await env.TTS_KV.get(`voice:${id}`, 'json'); - if (meta?.pathJa && meta.pathEn) { + if (meta && (!wantsJa || meta.pathJa) && (!wantsEn || meta.pathEn)) { try { const [jaObj, enObj] = await Promise.all([ - env.TTS_BUCKET.get(meta.pathJa), - env.TTS_BUCKET.get(meta.pathEn), + wantsJa && meta.pathJa ? env.TTS_BUCKET.get(meta.pathJa) : null, + wantsEn && meta.pathEn ? env.TTS_BUCKET.get(meta.pathEn) : null, ]); - if (jaObj && enObj) { + if ((!wantsJa || jaObj) && (!wantsEn || enObj)) { const [jaBuf, enBuf] = await Promise.all([ - jaObj.arrayBuffer(), - enObj.arrayBuffer(), + jaObj ? jaObj.arrayBuffer() : null, + enObj ? enObj.arrayBuffer() : null, ]); return callableSuccess({ id, - jaAudioContent: bytesToBase64(jaBuf), - enAudioContent: bytesToBase64(enBuf), - jaAudioMimeType: meta.jaAudioMimeType ?? 'audio/mpeg', - enAudioMimeType: meta.enAudioMimeType ?? 'audio/mpeg', + ...(jaBuf + ? { + jaAudioContent: bytesToBase64(jaBuf), + jaAudioMimeType: meta.jaAudioMimeType ?? 'audio/mpeg', + } + : {}), + ...(enBuf + ? { + enAudioContent: bytesToBase64(enBuf), + enAudioMimeType: meta.enAudioMimeType ?? 'audio/mpeg', + } + : {}), }); } } catch (e) { @@ -199,25 +228,27 @@ export const handleTts = async ( } } - // --- 合成(Azure) --- - // 音質・スタイル・プロソディは env で調整可能(ttsOptions は上で構築済み) + // --- 合成(Google Cloud TTS) --- + // 要求された言語だけ合成する(合成は文字数課金) const [jaAudio, enAudio] = await Promise.all([ - synthesizeSpeech( - env.AZURE_SPEECH_REGION, - env.AZURE_SPEECH_KEY, - ssmlJa, - 'ja-JP', - jaVoiceName, - ttsOptions - ), - synthesizeSpeech( - env.AZURE_SPEECH_REGION, - env.AZURE_SPEECH_KEY, - ssmlEn, - 'en-US', - enVoiceName, - ttsOptions - ), + wantsJa + ? synthesizeSpeech({ + saKeyJson: env.GOOGLE_TTS_SA_KEY, + languageCode: languageCodeFromVoiceName(jaVoiceName), + voiceName: jaVoiceName, + text: textJa, + opts: ttsOptions, + }) + : null, + wantsEn + ? synthesizeSpeech({ + saKeyJson: env.GOOGLE_TTS_SA_KEY, + languageCode: languageCodeFromVoiceName(enVoiceName), + voiceName: enVoiceName, + text: textEn, + opts: ttsOptions, + }) + : null, ]); // キャッシュ書き込みは非同期(失敗してもユーザー応答に影響させない)。 @@ -226,14 +257,15 @@ export const handleTts = async ( writeTtsCache( { id, - jaAudioContent: jaAudio.audioContent, - enAudioContent: enAudio.audioContent, - jaAudioMimeType: jaAudio.mimeType, - enAudioMimeType: enAudio.mimeType, - ssmlJa, - ssmlEn, - voiceJa: jaVoiceName, - voiceEn: enVoiceName, + jaAudioContent: jaAudio?.audioContent, + enAudioContent: enAudio?.audioContent, + jaAudioMimeType: jaAudio?.mimeType, + enAudioMimeType: enAudio?.mimeType, + textJa, + textEn, + model: TTS_ENGINE, + voiceJa: wantsJa ? jaVoiceName : undefined, + voiceEn: wantsEn ? enVoiceName : undefined, }, env ).catch((err) => console.error('Failed to cache tts audio:', err)) @@ -241,9 +273,17 @@ export const handleTts = async ( return callableSuccess({ id, - jaAudioContent: jaAudio.audioContent, - enAudioContent: enAudio.audioContent, - jaAudioMimeType: jaAudio.mimeType, - enAudioMimeType: enAudio.mimeType, + ...(jaAudio + ? { + jaAudioContent: jaAudio.audioContent, + jaAudioMimeType: jaAudio.mimeType, + } + : {}), + ...(enAudio + ? { + enAudioContent: enAudio.audioContent, + enAudioMimeType: enAudio.mimeType, + } + : {}), }); }; diff --git a/src/types.ts b/src/types.ts index ecfa6ec..db3ba13 100644 --- a/src/types.ts +++ b/src/types.ts @@ -11,13 +11,20 @@ export interface Env { TTS_BUCKET: R2Bucket; UPLOAD_BUCKET: R2Bucket; FEEDBACK_QUEUE: Queue; - /** sapi-bff(BFF ルートワーカー)への Service Binding。エージェントの駅検索に使う */ - SAPI_BFF?: Fetcher; + /** stationapi(GraphQL ワーカー)への Service Binding。エージェントの駅検索に使う */ + STATION_API?: Fetcher; // --- Vars(非機密。wrangler.jsonc の vars) --- GOOGLE_PLAY_PACKAGE_NAME: string; - AZURE_SPEECH_REGION: string; AI_TRIAGE_MODEL: string; + /** トリアージ判定に使う TypeSafe の API キー(シークレット) */ + TYPESAFE_API_KEY: string; + /** TypeSafe の System One モデル名。wrangler.jsonc の vars で指定する */ + TYPESAFE_MODEL: string; + /** + * Google Cloud TTS のボイス名(例: ja-JP-Standard-B)。ロケールを含むため + * 日英で別々に指定する。Standard / Wavenet / Neural2 のみ受け付ける + */ TTS_JA_VOICE_NAME: string; TTS_EN_VOICE_NAME: string; SESSION_TOKEN_TTL_SECONDS: string; @@ -25,7 +32,7 @@ export interface Env { FEW_SHOT_KV_KEY: string; FEW_SHOT_LIMIT: string; FEW_SHOT_PER_EX_MAX: string; - /** 対話本体モデル("anthropic:" | "openai:") */ + /** 対話本体モデル("anthropic:" | "openai:" | "google:") */ AGENT_MODEL: string; /** トピックゲート用の Workers AI モデル */ AGENT_GATE_MODEL: string; @@ -35,14 +42,17 @@ export interface Env { AGENT_DAILY_TURN_LIMIT: string; /** Cloudflare AI Gateway のベース URL(空なら各社 API 直行) */ AI_GATEWAY_BASE_URL?: string; - /** Service Binding 不使用時の sapi-bff GraphQL エンドポイント */ - SAPI_BFF_GRAPHQL_URL?: string; + /** Vertex AI のプロジェクト ID(未設定なら GOOGLE_VERTEX_SA_KEY の project_id) */ + GOOGLE_VERTEX_PROJECT?: string; + /** Vertex AI のロケーション(既定 "global"。例: asia-northeast1) */ + GOOGLE_VERTEX_LOCATION?: string; + /** Service Binding 不使用時の stationapi GraphQL エンドポイント */ + STATION_API_GRAPHQL_URL?: string; /** "true" で LangSmith トレーシングを有効化(dev 環境のみ設定すること) */ LANGSMITH_TRACING?: string; // --- Secrets(wrangler secret put で投入) --- SESSION_JWT_SECRET: string; - AZURE_SPEECH_KEY: string; /** Android Publisher 用 Google サービスアカウント鍵 JSON 文字列 */ GOOGLE_PLAY_SA_KEY: string; /** App Store Connect API 鍵 JSON 文字列 ({keyId, issuerId, privateKey}) */ @@ -55,14 +65,24 @@ export interface Env { ANTHROPIC_API_KEY?: string; /** 対話本体(GPT)の API キー。AGENT_MODEL が openai: のとき必須 */ OPENAI_API_KEY?: string; + /** + * 対話本体(Gemini / Vertex AI)用の Google サービスアカウント鍵 JSON 文字列。 + * AGENT_MODEL が google: のとき必須(Vertex AI は API キーではなくサービス + * アカウント認証のため。ロールは Vertex AI User 相当) + */ + GOOGLE_VERTEX_SA_KEY?: string; + /** Cloud TTS(/tts)用の Google サービスアカウント鍵 JSON 文字列 */ + GOOGLE_TTS_SA_KEY?: string; /** LangSmith の API キー(dev 環境のトレーシング用・任意) */ LANGSMITH_API_KEY?: string; - // --- Azure TTS チューニング(任意。未設定なら高音質既定のみ適用) --- - AZURE_TTS_OUTPUT_FORMAT?: string; - AZURE_TTS_STYLE?: string; - AZURE_TTS_STYLE_DEGREE?: string; - AZURE_TTS_PITCH?: string; + // --- TTS チューニング(任意。未設定なら mp3・等速・標準の高さ) --- + /** Cloud TTS の audioEncoding に対応する形式(mp3 / wav / opus) */ + TTS_RESPONSE_FORMAT?: string; + /** 読み上げ速度(speakingRate。0.25〜4.0) */ + TTS_SPEED?: string; + /** 声の高さ(pitch。-20.0〜20.0 セミトーン) */ + TTS_PITCH?: string; // --- 任意のデバッグ変数(未設定可) --- REVIEWS_DEBUG?: string; @@ -71,17 +91,21 @@ export interface Env { APPSTORE_APP_ID?: string; } -/** TTS キャッシュ書き込みのペイロード(R2+KV へ直接保存。キューは介さない) */ +/** + * TTS キャッシュ書き込みのペイロード(R2+KV へ直接保存。キューは介さない)。 + * 片方の言語だけ合成することがあるため、言語ごとのフィールドは任意。 + */ export interface TtsCachePayload { id: string; - jaAudioContent: string; - enAudioContent: string; - jaAudioMimeType: string; - enAudioMimeType: string; - ssmlJa: string; - ssmlEn: string; - voiceJa: string; - voiceEn: string; + model: string; + jaAudioContent?: string; + enAudioContent?: string; + jaAudioMimeType?: string; + enAudioMimeType?: string; + textJa?: string; + textEn?: string; + voiceJa?: string; + voiceEn?: string; } /** feedback-triage キューのメッセージ */ diff --git a/src/utils/normalize.test.ts b/src/utils/normalize.test.ts index a0316aa..40e7ac7 100644 --- a/src/utils/normalize.test.ts +++ b/src/utils/normalize.test.ts @@ -10,6 +10,60 @@ describe('utils/normalize.ts', () => { expect(normalizeRomanText('JR Kobe Line')).toBe('J-R Kobe Line'); }); + it('leaves hyphenated initialisms alone', () => { + // アプリ側が「JR」を J-R へ倒してから送ってくるため、ここで J-r へ + // 崩さないこと(= 二重に適用しても結果が変わらない) + expect(normalizeRomanText('J-R Kobe Line')).toBe('J-R Kobe Line'); + expect(normalizeRomanText(normalizeRomanText('JR Kobe Line'))).toBe( + 'J-R Kobe Line' + ); + expect(normalizeRomanText('Osaki, J-Y 24.')).toBe('Osaki, J-Y 24.'); + }); + + it('keeps hyphenated initialisms next to punctuation', () => { + // 文末やカンマの直前でも J-r に崩さない + expect(normalizeRomanText('Please transfer to the J-R.')).toBe( + 'Please transfer to the J-R.' + ); + expect(normalizeRomanText('Transfer to the J-R, and the subway.')).toBe( + 'Transfer to the J-R, and the subway.' + ); + }); + + it('replaces Keisei with a spelling English TTS reads as けいせい', () => { + // 英語 TTS は "Keisei" を「かいせい」と読むため、辞書語の綴りへ倒す + expect(normalizeRomanText('Change here for the Keisei Main Line.')).toBe( + 'Change here for the Kay-say Main Line.' + ); + expect(normalizeRomanText('The next station is Keisei-Ueno.')).toBe( + 'The next station is Kay-say-ueno.' + ); + expect(normalizeRomanText('KEISEI SKYLINER')).toBe('Kay-say Skyliner'); + // 別語の一部は置換しない + expect(normalizeRomanText('Keiseibus')).toBe('Keiseibus'); + }); + + it('replaces Seibu with a spelling English TTS reads as せいぶ', () => { + expect( + normalizeRomanText('Change here for the Seibu Ikebukuro Line.') + ).toBe('Change here for the Say-boo Ikebukuro Line.'); + expect(normalizeRomanText('The next station is Seibu-Shinjuku.')).toBe( + 'The next station is Say-boo-shinjuku.' + ); + // 西武園 (Seibuen) は 1 語なので語単位の一致では対象外 + expect(normalizeRomanText('Seibuen')).toBe('Seibuen'); + }); + + it('keeps Kay-say stable when normalized twice', () => { + // 二重に適用しても結果が変わらないこと(キャッシュキーの安定性) + expect(normalizeRomanText(normalizeRomanText('Keisei Main Line'))).toBe( + 'Kay-say Main Line' + ); + expect(normalizeRomanText(normalizeRomanText('Seibu Shinjuku Line'))).toBe( + 'Say-boo Shinjuku Line' + ); + }); + it.each(['Tokyo', 'tOkyo'])('text: %s', (text) => { expect(normalizeRomanText(text)).toBe('Tokyo'); }); diff --git a/src/utils/normalize.ts b/src/utils/normalize.ts index 4e14ee5..cfd4deb 100644 --- a/src/utils/normalize.ts +++ b/src/utils/normalize.ts @@ -1,9 +1,19 @@ import { removeMacron } from './removeMacron'; -const capitalizeSegment = (seg: string): string => - /[A-Z]/.test(seg) +// ハイフン区切りの頭字語(J-R / J-Y など)。アプリ側が「JR」を読み間違えられない +// 表記へ倒してから送ってくるため、これを capitalizeSegment に通して "J-r" へ +// 崩されないよう素通しする。文末・カンマ前("J-R." / "J-R,")も対象にするため、 +// セグメント全体一致ではなく後続が英数字でないことを先読みで判定する。 +const HYPHENATED_INITIALISM = /^[A-Z](?:-[A-Z])+(?=$|[^A-Za-z0-9])/; + +const capitalizeSegment = (seg: string): string => { + if (HYPHENATED_INITIALISM.test(seg)) { + return seg; + } + return /[A-Z]/.test(seg) ? seg.charAt(0).toUpperCase() + seg.slice(1).toLowerCase() : seg; +}; // テキストノード(SSML タグの外側)だけに掛ける正規化。タグやその属性値 // ( や 等)を壊さないため、タグ部分には適用しない。 @@ -25,6 +35,14 @@ const normalizeTextNode = (text: string): string => // 明治神宮前駅等の駅名にバッククォートが含まれる場合があるため除去 .replace(/`/g, '') .replace(/JR/gi, 'J-R') + // 「Keisei(京成)」「Seibu(西武)」は英語 TTS が "ei" を /aɪ/ と推定して + // 「かいせい」「さいぶ」と読むため、英単語 "Kay" + "say" / "Say" + "boo" で + // /keɪ.seɪ/(けいせい)/ /seɪ.buː/(せいぶ)を確定させる。読み替え先を + // 未知語の綴りにすると G2P の推定に戻ってエンジンごとに結果がぶれるので、 + // 辞書語のハイフン連結にする。単語境界で一致させ、Keisei-Ueno のような + // ハイフン連結の駅名も語単位で置換する + .replace(/\bKeisei\b/gi, 'Kay-say') + .replace(/\bSeibu\b/gi, 'Say-boo') // 都営バスを想定 .replace(/\bSta\./gi, ' Station') .replace(/\bUniv\./gi, ' University') diff --git a/src/utils/ssml.ts b/src/utils/ssml.ts index de51b04..fa37a95 100644 --- a/src/utils/ssml.ts +++ b/src/utils/ssml.ts @@ -22,3 +22,26 @@ export const stripSsml = (text: string): string => /** UTF-8 バイト長。 */ export const utf8ByteLength = (s: string): number => new TextEncoder().encode(s).length; + +/** + * UTF-8 バイト数の上限に合わせて切り詰める。 + * 文字数で切ると日本語(1 文字 3 バイト)では上限を守れないため、コードポイント + * 単位で積んでバイト数を数える。壊れた文字やサロゲートペアの分割は起きない。 + */ +export const truncateToByteLimit = (text: string, limit: number): string => { + if (limit <= 0 || utf8ByteLength(text) <= limit) { + return text; + } + + let bytes = 0; + let truncated = ''; + for (const char of text) { + const charBytes = utf8ByteLength(char); + if (bytes + charBytes > limit) { + break; + } + bytes += charBytes; + truncated += char; + } + return truncated; +}; diff --git a/src/utils/ttsVoice.test.ts b/src/utils/ttsVoice.test.ts index ccb0f23..a9c19cf 100644 --- a/src/utils/ttsVoice.test.ts +++ b/src/utils/ttsVoice.test.ts @@ -1,66 +1,121 @@ import { - isAzureHdVoiceName, - isAzureVoiceName, - resolveAzureVoiceName, + DEFAULT_TTS_VOICE, + isGoogleVoiceName, + languageCodeFromVoiceName, + resolveGoogleVoiceName, } from './ttsVoice'; -describe('ttsVoice (Azure)', () => { - it('accepts Azure neural voices', () => { - expect(isAzureVoiceName('ja-JP-NanamiNeural')).toBe(true); - expect(isAzureVoiceName('en-US-JennyNeural')).toBe(true); - expect(isAzureVoiceName('en-US-AvaMultilingualNeural')).toBe(true); +describe('isGoogleVoiceName', () => { + it('accepts the voice families we allow', () => { + expect(isGoogleVoiceName('ja-JP-Standard-B', 'ja')).toBe(true); + expect(isGoogleVoiceName('ja-JP-Wavenet-A', 'ja')).toBe(true); + expect(isGoogleVoiceName('ja-JP-Neural2-B', 'ja')).toBe(true); + expect(isGoogleVoiceName(' en-US-Standard-G ', 'en')).toBe(true); }); - it('accepts Azure HD (DragonHD) voices as valid voice names', () => { - expect(isAzureVoiceName('ja-JP-Nanami:DragonHDLatestNeural')).toBe(true); - expect(isAzureVoiceName('en-US-Jenny:DragonHDLatestNeural')).toBe(true); + it('rejects families that are far more expensive per character', () => { + // クライアントに高単価のボイスを名指しさせない + expect(isGoogleVoiceName('ja-JP-Chirp3-HD-Aoede', 'ja')).toBe(false); + expect(isGoogleVoiceName('en-US-Studio-O', 'en')).toBe(false); + expect(isGoogleVoiceName('Kore', 'en')).toBe(false); }); - it('detects HD (DragonHD) voices', () => { - expect(isAzureHdVoiceName('ja-JP-Nanami:DragonHDLatestNeural')).toBe(true); - expect(isAzureHdVoiceName('en-US-Jenny:DragonHDLatestNeural')).toBe(true); - expect(isAzureHdVoiceName('en-US-Ava:DragonHDLatestNeural')).toBe(true); + it('rejects well-formed but non-existent voices', () => { + // 形式だけの検証では通ってしまい、Cloud TTS が + // 400 "Voice ... does not exist" を返して /tts が落ちる + expect(isGoogleVoiceName('ja-US-Standard-A', 'ja')).toBe(false); + expect(isGoogleVoiceName('ja-JP-Standard-Z', 'ja')).toBe(false); + // 系統ごとに欠番がある(ja-JP の Neural2 は A、en-US の Neural2 は B が無い) + expect(isGoogleVoiceName('ja-JP-Neural2-A', 'ja')).toBe(false); + expect(isGoogleVoiceName('en-US-Neural2-B', 'en')).toBe(false); + // 実在するが未対応のロケール。使うなら allowlist へ追加する + expect(isGoogleVoiceName('en-GB-Standard-A', 'en')).toBe(false); }); - it('treats standard neural voices as non-HD', () => { - expect(isAzureHdVoiceName('ja-JP-NanamiNeural')).toBe(false); - expect(isAzureHdVoiceName('en-US-JennyNeural')).toBe(false); - expect(isAzureHdVoiceName('')).toBe(false); + it('rejects a voice whose language does not match the text', () => { + // ja のテキストに en のボイスを渡すと Cloud TTS が 400 を返す + expect(isGoogleVoiceName('en-US-Standard-G', 'ja')).toBe(false); + expect(isGoogleVoiceName('ja-JP-Standard-B', 'en')).toBe(false); }); - it('rejects non-Azure voice ids', () => { - expect(isAzureVoiceName('ja-JP-Standard-B')).toBe(false); - expect(isAzureVoiceName('en-US-Chirp3-HD-Aoede')).toBe(false); - expect(isAzureVoiceName('')).toBe(false); + it('rejects names from the previous engines', () => { + expect(isGoogleVoiceName('shimmer', 'ja')).toBe(false); + expect(isGoogleVoiceName('ja-JP-NanamiNeural', 'ja')).toBe(false); + expect(isGoogleVoiceName('', 'ja')).toBe(false); }); +}); + +describe('languageCodeFromVoiceName', () => { + it('derives the locale from the voice name', () => { + // voice.name と languageCode の食い違いは 400 になるため名前から導出する + expect(languageCodeFromVoiceName('ja-JP-Standard-B')).toBe('ja-JP'); + expect(languageCodeFromVoiceName('en-GB-Wavenet-A')).toBe('en-GB'); + }); +}); +describe('resolveGoogleVoiceName', () => { it('prefers a valid requested voice', () => { expect( - resolveAzureVoiceName( - 'en-US-AriaNeural', - 'en-US-GuyNeural', - 'en-US-JennyNeural' + resolveGoogleVoiceName( + 'ja-JP-Wavenet-A', + 'ja-JP-Standard-A', + 'ja-JP-Standard-B', + 'ja' ) - ).toBe('en-US-AriaNeural'); + ).toBe('ja-JP-Wavenet-A'); }); - it('falls back to a configured voice when the request is invalid', () => { + it('falls back to the KV config, then to the env default', () => { expect( - resolveAzureVoiceName( - 'en-US-Standard-H', - 'en-US-GuyNeural', - 'en-US-JennyNeural' + resolveGoogleVoiceName( + 'ja-JP-Chirp3-HD-Aoede', + 'ja-JP-Standard-A', + 'ja-JP-Standard-B', + 'ja' ) - ).toBe('en-US-GuyNeural'); + ).toBe('ja-JP-Standard-A'); + expect( + resolveGoogleVoiceName(undefined, undefined, 'ja-JP-Standard-B', 'ja') + ).toBe('ja-JP-Standard-B'); + expect(resolveGoogleVoiceName(42, {}, 'en-US-Standard-G', 'en')).toBe( + 'en-US-Standard-G' + ); + }); + + it('validates the env default too, so a stale OpenAI value never reaches Google', () => { + // 環境変数の設定ミスをそのまま送ると Google が 400 を返し /tts が落ちる + expect(resolveGoogleVoiceName(undefined, undefined, 'shimmer', 'ja')).toBe( + DEFAULT_TTS_VOICE.ja + ); + expect(resolveGoogleVoiceName(undefined, undefined, undefined, 'en')).toBe( + DEFAULT_TTS_VOICE.en + ); + }); + + it('falls back for well-formed but non-existent voices', () => { + // 形式が正しいだけの名前を通すと Cloud TTS が 400 を返す + expect( + resolveGoogleVoiceName('ja-US-Standard-A', undefined, undefined, 'ja') + ).toBe(DEFAULT_TTS_VOICE.ja); + expect( + resolveGoogleVoiceName('ja-JP-Standard-Z', undefined, undefined, 'ja') + ).toBe(DEFAULT_TTS_VOICE.ja); + }); + + it('has defaults that are themselves allowed voices', () => { + // 既定値が allowlist から外れると、全リクエストが 400 になる + expect(isGoogleVoiceName(DEFAULT_TTS_VOICE.ja, 'ja')).toBe(true); + expect(isGoogleVoiceName(DEFAULT_TTS_VOICE.en, 'en')).toBe(true); }); - it('falls back to the default voice when both inputs are invalid', () => { + it('never returns a voice from the wrong language', () => { expect( - resolveAzureVoiceName( + resolveGoogleVoiceName( 'ja-JP-Standard-B', + 'ja-JP-Wavenet-A', 'ja-JP-Neural2-B', - 'ja-JP-NanamiNeural' + 'en' ) - ).toBe('ja-JP-NanamiNeural'); + ).toBe(DEFAULT_TTS_VOICE.en); }); }); diff --git a/src/utils/ttsVoice.ts b/src/utils/ttsVoice.ts index 4324cc4..0215440 100644 --- a/src/utils/ttsVoice.ts +++ b/src/utils/ttsVoice.ts @@ -1,42 +1,123 @@ /** - * Azure Speech のニューラルボイス名を扱うユーティリティ。 + * Google Cloud TTS のボイス名を扱うユーティリティ。 * - * Azure のボイス id は `-Neural` 形式(例: `ja-JP-NanamiNeural`, - * `en-US-JennyNeural`, `en-US-AvaMultilingualNeural`)。Google の Standard/WaveNet の - * ような価格差はなく、ニューラルが標準ティアのため「コストガード」は不要だが、 - * クライアントから任意文字列が渡るため最低限の妥当性チェックは行う。 + * ボイス名は `<言語>-<地域>-<系統>-<記号>`(例: ja-JP-Standard-B)で、Azure と同じく + * ロケールを含む。言語ごとに別のボイスを指定する必要があるため、日英で共通の名前は + * 使えない(OpenAI の `shimmer` のような多言語プリセットとは異なる)。 + * + * クライアントから任意文字列が渡るため、未知の名前はそのまま Google へ流さず + * 既定値へ倒す(400 で放送を落とさないため)。 */ -// 標準ニューラル(ja-JP-NanamiNeural)と HD ボイス(ja-JP-Nanami:DragonHDLatestNeural)の両方を許可 -const AZURE_VOICE_PATTERN = /^[a-z]{2,3}-[A-Za-z]+-[A-Za-z0-9:]+Neural$/; -export const isAzureVoiceName = (voiceName: string): boolean => - AZURE_VOICE_PATTERN.test(voiceName); +export type TtsLanguage = 'ja' | 'en'; -// HD(DragonHD)ボイス判定。HD ボイスは id に `:DragonHD...Neural` を含む -// (例: `ja-JP-Nanami:DragonHDLatestNeural`, `en-US-Jenny:DragonHDLatestNeural`)。 -// HD は と を非対応のため、SSML 構築時に -// 未サポート要素を出し分ける用途で使う。 -const AZURE_HD_VOICE_PATTERN = /:DragonHD[A-Za-z0-9]*Neural$/i; +/** + * 使用を許すボイス名(`voices.list` で実在を確認済み)。 + * + * 形式(`<言語>-<地域>-<系統>-<記号>`)だけを検証すると `ja-US-Standard-A` や + * `ja-JP-Standard-Z` のような実在しない名前を通してしまい、Cloud TTS が + * 400("Voice ... does not exist")を返して放送そのものが落ちる。既定値へ倒す + * ためには実在する名前だけを許可する必要がある。 + * + * 系統は Standard / Wavenet / Neural2 に限定する。Android の端末内蔵 TTS と + * 同水準の音質を狙う系統で、Studio / Chirp3-HD / Gemini-TTS は単価が桁違いなので + * 名指しされても受け付けない。 + * + * ロケールは実際に使う ja-JP / en-US のみ。他ロケール(en-GB など)や Google が + * 後から追加したボイスを使うときは、`voices.list` で実在を確認してここへ足す。 + */ +const ALLOWED_VOICES: Record> = { + ja: new Set([ + 'ja-JP-Standard-A', + 'ja-JP-Standard-B', + 'ja-JP-Standard-C', + 'ja-JP-Standard-D', + 'ja-JP-Wavenet-A', + 'ja-JP-Wavenet-B', + 'ja-JP-Wavenet-C', + 'ja-JP-Wavenet-D', + // Neural2 の ja-JP は A が無い + 'ja-JP-Neural2-B', + 'ja-JP-Neural2-C', + 'ja-JP-Neural2-D', + ]), + en: new Set([ + 'en-US-Standard-A', + 'en-US-Standard-B', + 'en-US-Standard-C', + 'en-US-Standard-D', + 'en-US-Standard-E', + 'en-US-Standard-F', + 'en-US-Standard-G', + 'en-US-Standard-H', + 'en-US-Standard-I', + 'en-US-Standard-J', + 'en-US-Wavenet-A', + 'en-US-Wavenet-B', + 'en-US-Wavenet-C', + 'en-US-Wavenet-D', + 'en-US-Wavenet-E', + 'en-US-Wavenet-F', + 'en-US-Wavenet-G', + 'en-US-Wavenet-H', + 'en-US-Wavenet-I', + 'en-US-Wavenet-J', + // Neural2 の en-US は B が無い + 'en-US-Neural2-A', + 'en-US-Neural2-C', + 'en-US-Neural2-D', + 'en-US-Neural2-E', + 'en-US-Neural2-F', + 'en-US-Neural2-G', + 'en-US-Neural2-H', + 'en-US-Neural2-I', + 'en-US-Neural2-J', + ]), +}; + +// 環境変数の設定ミス(OpenAI 時代の "shimmer" の残留など)でも合成を落とさない +// ための最終フォールバック。ここは実在を確認済みの女性ボイス。 +export const DEFAULT_TTS_VOICE: Record = { + ja: 'ja-JP-Standard-B', + en: 'en-US-Standard-G', +}; -export const isAzureHdVoiceName = (voiceName: string): boolean => - AZURE_HD_VOICE_PATTERN.test(voiceName); +/** ボイス名がその言語向けの許可済みボイスか。 */ +export const isGoogleVoiceName = ( + voiceName: string, + language: TtsLanguage +): boolean => ALLOWED_VOICES[language].has(voiceName.trim()); -export const resolveAzureVoiceName = ( +/** + * ボイス名からロケール(languageCode)を取り出す。Cloud TTS は voice.name と + * voice.languageCode の食い違いを 400 で弾くため、必ず名前から導出する。 + */ +export const languageCodeFromVoiceName = (voiceName: string): string => + voiceName.trim().split('-').slice(0, 2).join('-'); + +/** + * 使用するボイス名を決める。 + * 優先順位: リクエスト指定 → KV の設定 → 環境変数の既定値。 + * いずれも「その言語向けの許可済みボイス」のときだけ採用する。 + */ +export const resolveGoogleVoiceName = ( requestedVoiceName: unknown, configuredVoiceName: unknown, - defaultVoiceName: string + defaultVoiceName: string | undefined, + language: TtsLanguage ): string => { - const requested = - typeof requestedVoiceName === 'string' ? requestedVoiceName.trim() : ''; - if (requested && isAzureVoiceName(requested)) { - return requested; - } - - const configured = - typeof configuredVoiceName === 'string' ? configuredVoiceName.trim() : ''; - if (configured && isAzureVoiceName(configured)) { - return configured; + const candidates = [ + requestedVoiceName, + configuredVoiceName, + defaultVoiceName, + ]; + for (const candidate of candidates) { + const value = typeof candidate === 'string' ? candidate.trim() : ''; + // 環境変数由来の既定値も無検証では通さない。不正なら Google が 400 を返し、 + // /tts 全体が失敗してしまうため、既知のボイスへ倒す。 + if (value && isGoogleVoiceName(value, language)) { + return value; + } } - - return defaultVoiceName; + return DEFAULT_TTS_VOICE[language]; }; diff --git a/test/stubs/ai-sdk-provider.ts b/test/stubs/ai-sdk-provider.ts index aa2828a..d0d6d1f 100644 --- a/test/stubs/ai-sdk-provider.ts +++ b/test/stubs/ai-sdk-provider.ts @@ -1,10 +1,33 @@ /** - * Jest 用の @ai-sdk/anthropic / @ai-sdk/openai スタブ(ESM 専用のため差し替え)。 - * モデル ID をそのまま返すだけの形だけ互換。 + * Jest 用の @ai-sdk/anthropic / @ai-sdk/openai / @ai-sdk/google-vertex スタブ + * (ESM 専用のため差し替え)。モデル ID と生成時オプションをそのまま返すだけの形だけ互換。 */ +type ProviderOptions = { + apiKey?: string; + baseURL?: string; + headers?: Record; + project?: string; + location?: string; + googleCredentials?: { + clientEmail?: string; + privateKey?: string; + privateKeyId?: string; + }; +}; + const createProvider = - () => - (modelId: string): { modelId: string } => ({ modelId }); + (provider: string, options?: unknown) => + ( + modelId: string + ): { modelId: string; provider: string; options?: unknown } => ({ + modelId, + provider, + options: options as ProviderOptions | undefined, + }); -export const createAnthropic = (_options?: unknown) => createProvider(); -export const createOpenAI = (_options?: unknown) => createProvider(); +export const createAnthropic = (options?: unknown) => + createProvider('anthropic', options); +export const createOpenAI = (options?: unknown) => + createProvider('openai', options); +export const createVertex = (options?: unknown) => + createProvider('google.vertex.chat', options); diff --git a/wrangler.jsonc b/wrangler.jsonc index 0eda631..c0a9fe4 100644 --- a/wrangler.jsonc +++ b/wrangler.jsonc @@ -26,34 +26,68 @@ { "binding": "UPLOAD_BUCKET", "bucket_name": "trainlcd-uploads-dev" } ], + // dead_letter_queue: max_retries を使い切ったメッセージの退避先。未設定だと + // GitHub 起票が恒久的に失敗する状況(PAT 失効など)でフィードバックが消えるため必須。 + // DLQ 側に consumer は付けない(同じ処理を回すと同じ理由で落ちるだけ)。 + // DLQ にも保持期限があり、期限を過ぎたメッセージは消える。復旧手順と + // 保持期限の延ばし方は README の「Dead letter queue」節を参照。 "queues": { "producers": [ { "binding": "FEEDBACK_QUEUE", "queue": "feedback-triage-dev" } ], "consumers": [ - { "queue": "feedback-triage-dev", "max_batch_size": 5, "max_retries": 3 } + { + "queue": "feedback-triage-dev", + "max_batch_size": 5, + "max_retries": 3, + "dead_letter_queue": "feedback-triage-dev-dlq" + } ] }, - // エージェントの駅検索(stationsByName)は同一アカウントの sapi-bff を Service Binding で呼ぶ - "services": [{ "binding": "SAPI_BFF", "service": "sapi-bff-stg" }], + // エージェントの駅検索(stationsByName)は同一アカウントの stationapi を Service Binding で呼ぶ。 + // BFF 廃止に伴い staging・本番とも stationapi へ移行済み + "services": [{ "binding": "STATION_API", "service": "stationapi-stg" }], "vars": { "GOOGLE_PLAY_PACKAGE_NAME": "me.tinykitten.trainlcd", - "AZURE_SPEECH_REGION": "southeastasia", - "AZURE_TTS_OUTPUT_FORMAT": "audio-48khz-192kbitrate-mono-mp3", - // HD(DragonHD)ボイスは 非対応のため AZURE_TTS_RATE は設定しない - "AI_TRIAGE_MODEL": "@cf/meta/llama-3.1-8b-instruct-fast", - "TTS_JA_VOICE_NAME": "ja-JP-Nanami:DragonHDLatestNeural", - "TTS_EN_VOICE_NAME": "en-US-Jenny:DragonHDOmniLatestNeural", + // トリアージは日本語のタイトル・要約を生成するため、日本語生成品質が要件。 + // 旧 @cf/meta/llama-3.1-8b-instruct-fast は日本語が破綻するうえ、 + // Workers AI のカタログからも消えている(models list / schema に無い)。 + "AI_TRIAGE_MODEL": "@cf/google/gemma-4-26b-a4b-it", + // トリアージの判定(スパム・カテゴリ・優先度・原因コンポーネント)を担う + // TypeSafe の System One モデル。生成は担当せず、タイトルと要約は + // AI_TRIAGE_MODEL のまま。confidence の閾値を特定バージョンに合わせて + // 調整したら、エイリアスではなく "jev-1.13.0" のようにバージョンを固定する + // (エイリアスは新リリースで指す先が動くため)。 + "TYPESAFE_MODEL": "jev-latest", + // --- TTS(/tts)--- + // Google Cloud TTS。ボイス名はロケール込みで、日英それぞれに指定する。 + // 既定は旧 Google TTS 実装(Firebase Functions 時代)と同じ組み合わせ。 + // Standard 系は端末内蔵 TTS と同水準の音質で単価も最安。より高品質にするなら + // 末尾記号はそのまま Wavenet / Neural2 へ差し替える(例: ja-JP-Wavenet-B)。 + // 読み方のプロンプト指示は Standard 系に無いため、速さは TTS_SPEED + // (speakingRate)、高さは TTS_PITCH で調整する。TTS_SPEED はアプリが速度を + // 送ってこなかったときの既定値で、送ってきた場合はそちら(プリセット3値のみ)が優先。 + "TTS_JA_VOICE_NAME": "ja-JP-Standard-B", + "TTS_EN_VOICE_NAME": "en-US-Standard-G", + // アプリのアナウンス速度設定(既定「普通」)と同じ等速。速度を送ってこない + // 旧バージョンのアプリも、設定の既定値と同じ速さで聞こえるようにしている + "TTS_SPEED": "1.0", "SESSION_TOKEN_TTL_SECONDS": "3600", "UPLOAD_PUBLIC_BASE_URL": "https://uploads-dev.trainlcd.app", "FEW_SHOT_KV_KEY": "config:fewshot", - "FEW_SHOT_LIMIT": "12", + "FEW_SHOT_LIMIT": "16", "FEW_SHOT_PER_EX_MAX": "800", // --- AI エージェント(/agent/chat)--- - // モデルは "anthropic:" | "openai:"。比較検証で差し替える - "AGENT_MODEL": "openai:gpt-5.6-luna", + // モデルは "anthropic:" | "openai:" | "google:"。比較検証で差し替える + // ("google:" は Vertex AI 経由。API キーではなく GOOGLE_VERTEX_SA_KEY が必要) + "AGENT_MODEL": "google:gemini-3.8-flash", + // Vertex AI 用(AGENT_MODEL が google: のときのみ使う)。 + // GOOGLE_VERTEX_PROJECT 未設定なら GOOGLE_VERTEX_SA_KEY の project_id を使う。 + // ロケーションは "global" がモデルの提供範囲が最も広い。特定リージョンに寄せるなら + // "asia-northeast1" など(AI Gateway 経由でもどちらも中継される) + "GOOGLE_VERTEX_LOCATION": "global", "AGENT_GATE_MODEL": "@cf/meta/llama-3.1-8b-instruct-fast", "AGENT_FAQ_KV_KEY": "config:agent-faq", // config:remote の agent_daily_turn_limit があればそちらが優先(デプロイなしで調整可) @@ -66,9 +100,10 @@ }, // secrets(`wrangler secret put ` で投入。コミットしない): - // SESSION_JWT_SECRET / AZURE_SPEECH_KEY / GOOGLE_PLAY_SA_KEY / APPSTORE_CONNECT_KEY / + // SESSION_JWT_SECRET / GOOGLE_PLAY_SA_KEY / APPSTORE_CONNECT_KEY / // OCTOKIT_PAT / DISCORD_CS_WEBHOOK_URL / DISCORD_CRASH_WEBHOOK_URL / DISCORD_REVIEW_WEBHOOK_URL / - // ANTHROPIC_API_KEY / OPENAI_API_KEY / LANGSMITH_API_KEY + // ANTHROPIC_API_KEY / OPENAI_API_KEY / GOOGLE_VERTEX_SA_KEY / LANGSMITH_API_KEY / + // GOOGLE_TTS_SA_KEY "env": { "production": { @@ -90,25 +125,35 @@ { "binding": "FEEDBACK_QUEUE", "queue": "feedback-triage" } ], "consumers": [ - { "queue": "feedback-triage", "max_batch_size": 5, "max_retries": 3 } + { + "queue": "feedback-triage", + "max_batch_size": 5, + "max_retries": 3, + "dead_letter_queue": "feedback-triage-dlq" + } ] }, - "services": [{ "binding": "SAPI_BFF", "service": "sapi-bff" }], + "services": [{ "binding": "STATION_API", "service": "stationapi" }], "vars": { "GOOGLE_PLAY_PACKAGE_NAME": "me.tinykitten.trainlcd", - "AZURE_SPEECH_REGION": "southeastasia", - "AZURE_TTS_OUTPUT_FORMAT": "audio-48khz-192kbitrate-mono-mp3", - // HD(DragonHD)ボイスは 非対応のため AZURE_TTS_RATE は設定しない - "AI_TRIAGE_MODEL": "@cf/meta/llama-3.1-8b-instruct-fast", - "TTS_JA_VOICE_NAME": "ja-JP-Nanami:DragonHDLatestNeural", - "TTS_EN_VOICE_NAME": "en-US-Jenny:DragonHDLatestNeural", + // dev と同じ理由で大型モデルを使う(日本語タイトルの破損対策) + "AI_TRIAGE_MODEL": "@cf/google/gemma-4-26b-a4b-it", + // dev と同じ。閾値を調整したらバージョンを固定する + "TYPESAFE_MODEL": "jev-latest", + // --- TTS(/tts)--- + // Google Cloud TTS。速度はアプリのアナウンス速度設定(既定「普通」)に + // 合わせた等速。dev と本番で差を付けると開発時の体感が本番と食い違う + "TTS_JA_VOICE_NAME": "ja-JP-Standard-B", + "TTS_EN_VOICE_NAME": "en-US-Standard-G", + "TTS_SPEED": "1.0", "SESSION_TOKEN_TTL_SECONDS": "3600", "UPLOAD_PUBLIC_BASE_URL": "https://uploads.trainlcd.app", "FEW_SHOT_KV_KEY": "config:fewshot", - "FEW_SHOT_LIMIT": "12", + "FEW_SHOT_LIMIT": "16", "FEW_SHOT_PER_EX_MAX": "800", // --- AI エージェント(/agent/chat)--- - "AGENT_MODEL": "openai:gpt-5.6-luna", + "AGENT_MODEL": "google:gemini-3.8-flash", + "GOOGLE_VERTEX_LOCATION": "global", "AGENT_GATE_MODEL": "@cf/meta/llama-3.1-8b-instruct-fast", "AGENT_FAQ_KV_KEY": "config:agent-faq", // config:remote の agent_daily_turn_limit があればそちらが優先(デプロイなしで調整可)