diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json
index 31a7439..609edf7 100644
--- a/.claude-plugin/marketplace.json
+++ b/.claude-plugin/marketplace.json
@@ -6,19 +6,20 @@
},
"metadata": {
"description": "Skills for OrcaRouter products.",
- "version": "1.5.0"
+ "version": "2.1.0"
},
"plugins": [
{
"name": "orca-code-review",
"source": "./",
- "description": "Set up, reconfigure, troubleshoot, and remove OrcaCode Review — AI pull-request review powered by OrcaRouter — in any GitHub repository.",
+ "description": "Review code changes locally with your own agent as the engine, and set up, reconfigure, troubleshoot, or remove OrcaCode Review — AI pull-request review powered by OrcaRouter — in any GitHub repository.",
"category": "code-review",
"keywords": [
"code-review",
"github-actions",
"ci",
- "pull-request"
+ "pull-request",
+ "local-review"
]
}
]
diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json
index 68671fc..4366054 100644
--- a/.claude-plugin/plugin.json
+++ b/.claude-plugin/plugin.json
@@ -1,7 +1,7 @@
{
"name": "orca-code-review",
- "description": "Set up, reconfigure, troubleshoot, and remove OrcaCode Review — AI pull-request review powered by OrcaRouter — in any GitHub repository.",
- "version": "1.5.0",
+ "description": "Review code changes locally with your own agent as the engine, and set up, reconfigure, troubleshoot, or remove OrcaCode Review — AI pull-request review powered by OrcaRouter — in any GitHub repository.",
+ "version": "2.1.0",
"author": {
"name": "Continuum-AI-Corp",
"url": "https://github.com/Continuum-AI-Corp"
@@ -14,6 +14,7 @@
"github-actions",
"ci",
"pull-request",
- "orcarouter"
+ "orcarouter",
+ "local-review"
]
}
diff --git a/.github/workflows/deprecate.yml b/.github/workflows/deprecate.yml
new file mode 100644
index 0000000..051669e
--- /dev/null
+++ b/.github/workflows/deprecate.yml
@@ -0,0 +1,133 @@
+# Mark versions of `@orcarouter/code-review` as retired on npm.
+#
+# `npm deprecate` is the tool to reach for when a version should stop being
+# installed. It leaves the tarball in place — anyone pinned to it keeps working —
+# and prints a warning on every install, which is the outcome an unpublish only
+# approximates by breaking them instead.
+#
+# It lives in CI because the credential lives in CI: the `NPM_TOKEN` secret is
+# the only one this project has. Unpublishing cannot be done this way (npm
+# refuses a granular token that bypasses 2FA — see RELEASE.md); deprecation is
+# the part of the job that can be automated, so it is.
+#
+# Reversible, unlike everything else that touches the registry: dispatching with
+# an empty `message` clears the flag. That is npm's own convention and the reason
+# this workflow is safe to keep while the unpublish one was not.
+#
+# One guard: the version in `package.json` is refused. Deprecating the release
+# that `dist-tags.latest` points at puts a warning on every single install of
+# the package, which looks like an outage and is one click away.
+#
+# Inputs reach the shell through `env`, never `${{ }}` inside a `run` block —
+# this job holds a publish token.
+
+name: Deprecate
+
+concurrency:
+ group: publish-npm # never race the publish job
+ cancel-in-progress: false
+
+on:
+ workflow_dispatch:
+ inputs:
+ versions:
+ description: "Versions to mark, space-separated (e.g. 1.4.0 1.5.0)"
+ required: true
+ type: string
+ message:
+ description: "Warning shown on install. Empty clears the flag."
+ required: false
+ default: "No longer supported — install @orcarouter/code-review@latest"
+ type: string
+
+permissions:
+ contents: read
+
+jobs:
+ deprecate:
+ runs-on: ubuntu-latest
+ env:
+ VERSIONS: ${{ inputs.versions }}
+ MESSAGE: ${{ inputs.message }}
+ steps:
+ - uses: actions/checkout@v4
+
+ - uses: actions/setup-node@v4
+ with:
+ node-version: "20"
+ registry-url: "https://registry.npmjs.org"
+
+ - name: Guard
+ run: |
+ set -euo pipefail
+ PKG_NAME=$(node -p "require('./package.json').name")
+ PKG_LIVE=$(node -p "require('./package.json').version")
+ echo "PKG_NAME=$PKG_NAME" >> "$GITHUB_ENV"
+ echo "PKG_LIVE=$PKG_LIVE" >> "$GITHUB_ENV"
+
+ for v in $VERSIONS; do
+ if [ "$v" = "$PKG_LIVE" ]; then
+ echo "::error::$v is the version in package.json — deprecating it warns on every install. Refusing."
+ exit 1
+ fi
+ done
+ if [ -z "${MESSAGE:-}" ]; then
+ echo "::notice::CLEARING the deprecation flag on: $VERSIONS"
+ else
+ echo "::notice::marking $VERSIONS — \"$MESSAGE\""
+ fi
+
+ - name: Deprecate
+ run: |
+ set -uo pipefail
+ FAILED=""
+ for v in $VERSIONS; do
+ echo "--- $PKG_NAME@$v"
+ if npm deprecate "$PKG_NAME@$v" "$MESSAGE"; then
+ echo "::notice::marked $PKG_NAME@$v"
+ else
+ # One version's refusal must not strand the rest unmarked.
+ echo "::warning::could not mark $PKG_NAME@$v"
+ FAILED="$FAILED $v"
+ fi
+ done
+ echo "FAILED=${FAILED# }" >> "$GITHUB_ENV"
+ env:
+ NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
+
+ # The packument is what `npm install` reads, and an anonymous read is the
+ # only one that proves what a stranger sees. `npm deprecate` exiting 0 is
+ # not evidence — same reason publish.yml has a gate 6.
+ - name: Verify the registry
+ run: |
+ set -euo pipefail
+ sleep 15
+ SLUG=$(node -p "encodeURIComponent(process.env.PKG_NAME)")
+ curl -fsSL "https://registry.npmjs.org/$SLUG" -o packument.json
+ node -e '
+ const p = require("./packument.json");
+ const live = process.env.PKG_LIVE;
+ const want = (process.env.MESSAGE || "").length > 0;
+ let bad = 0;
+ for (const v of (process.env.VERSIONS || "").split(/\s+/).filter(Boolean)) {
+ const meta = p.versions[v];
+ if (!meta) { console.log(`${v}: not on the registry`); continue; }
+ const got = typeof meta.deprecated === "string" && meta.deprecated.length > 0;
+ console.log(`${v}: ${got ? `deprecated — "${meta.deprecated}"` : "not deprecated"}`);
+ if (got !== want) bad++;
+ }
+ const liveMeta = p.versions[live];
+ if (liveMeta && liveMeta.deprecated) {
+ console.error(`::error::${live} is the live release and it is deprecated`);
+ process.exit(1);
+ }
+ if (bad) {
+ console.error(`::error::${bad} version(s) did not end up in the requested state`);
+ process.exit(1);
+ }
+ '
+
+ if [ -n "${FAILED:-}" ]; then
+ echo "::error::these versions could not be marked: $FAILED"
+ exit 1
+ fi
diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml
index dc878ad..482d135 100644
--- a/.github/workflows/publish.yml
+++ b/.github/workflows/publish.yml
@@ -127,7 +127,20 @@ jobs:
"bin/orcacode-review.mjs",
"bin/platforms.mjs",
"bin/i18n.mjs",
- "skills/setup-orca-code-review/SKILL.md",
+ "bin/harness.mjs",
+ "bin/review.mjs",
+ "bin/selection.mjs",
+ "bin/localconfig.mjs",
+ "scripts/postfilter.mjs",
+ "scripts/severity.mjs",
+ "rules/severity-instruction.md",
+ "rules/output-shape.md",
+ "vendor/open-code-review/LICENSE",
+ "vendor/open-code-review/system_rules.json",
+ "vendor/open-code-review/rule_docs/default.md",
+ "skills/orca-review-action/SKILL.md",
+ "skills/orca-review/SKILL.md",
+ "skills/orca-review/references/contract.md",
];
const missing = required.filter(f => !files.includes(f));
if (missing.length) {
diff --git a/.gitignore b/.gitignore
index 27c179a..56fbb05 100644
--- a/.gitignore
+++ b/.gitignore
@@ -12,3 +12,7 @@ node_modules/
docs/reports/
feedbacks/
rules/*.draft.md
+
+# raw demo recordings — the shipped cuts live in docs/, the raws stay local
+docs/demo-*-raw.mp4
+*.tgz
diff --git a/NOTICE b/NOTICE
index 9d8b991..5dbea33 100644
--- a/NOTICE
+++ b/NOTICE
@@ -1,12 +1,29 @@
OrcaCode Review
Copyright 2026 OrcaRouter
-This product orchestrates Open Code Review ("ocr"), an open-source code review
-tool by Alibaba Group, licensed under the Apache License, Version 2.0:
+This product uses Open Code Review ("ocr"), an open-source code review tool by
+Alibaba Group, licensed under the Apache License, Version 2.0:
https://github.com/alibaba/open-code-review
-Open Code Review is installed and invoked as the LLM review engine. This product
-adds, on top of it: OrcaRouter gateway integration, a cost-tiered model cascade,
-a severity rubric (P0/P1/P2), and merge-gating. No Open Code Review source is
-modified or redistributed here — it is consumed as a published package.
+In two ways:
+
+1. The GitHub Action installs the published Open Code Review package and runs
+ it as the LLM review engine. This product adds, on top of it: OrcaRouter
+ gateway integration, a cost-tiered model cascade, a severity rubric
+ (P0/P1/P2/P3), and merge-gating.
+
+2. The local review (`review plan` / `review submit`) does NOT run Open Code
+ Review — the reviewer is the user's own coding agent. It reuses Open Code
+ Review's file-selection rules and per-language review checklists:
+
+ - vendor/open-code-review/ data copied unmodified from upstream at the
+ tag recorded in vendor/open-code-review/UPSTREAM,
+ with its LICENSE
+ - bin/selection.mjs a JavaScript port of the matching logic, a
+ modified work; the Go sources it derives from
+ are named in its header
+
+Everything else in this repository is Copyright 2026 Continuum-AI-Corp and
+licensed under the MIT License (see LICENSE). The Apache-2.0 material above
+keeps its own license.
diff --git a/README.md b/README.md
index 59c7490..12b5de3 100644
--- a/README.md
+++ b/README.md
@@ -9,7 +9,7 @@ Automatically review every pull request, post findings directly on the affected
---
-
+
---
@@ -48,16 +48,21 @@ One command teaches your AI what OrcaCode Review is. Everything after that, you
npx @orcarouter/code-review
```
-It detects which coding agents you use, installs the skill, and stops. Then:
+It asks how you will use it — local review, the GitHub Action, or both — detects which coding agents you use, installs the matching skills, and stops. Then:
> **you:** set up OrcaCode Review in this repo
Your agent writes the workflow, walks you through the API key, and sets the merge gate — asking only the questions that are actually yours to answer.
+
+
+
+
The same goes for everything else:
| Say | It does |
| --- | --- |
+| *"review my changes"* | Reviews them **locally**, itself — see [below](#review-locally-your-agent-is-the-engine) |
| *"why didn't the review run?"* | Diagnoses the secret, the trigger, the base branch, the gate |
| *"make OrcaCode Review block P0 only"* | Retunes the merge policy |
| *"remove OrcaCode Review from this repo"* | Drops the required check first, then the workflow |
@@ -72,52 +77,57 @@ The same goes for everything else:
### 36 agent platforms
-The same catalog the [OrcaDub MCP server](https://github.com/Continuum-AI-Corp/orcadub-mcp-server) uses, so IDs and paths match across Orca products. Detected agents are pre-ticked; `/` filters the list.
-
-```bash
-npx @orcarouter/code-review skill list # all 36, detected ones marked
-npx @orcarouter/code-review --platform claude,codex --yes # unattended
-```
-
-### Terminal, without an agent
-
-The lifecycle is also available as plain subcommands — the skill is the front door, not the only door:
-
-```bash
-npx @orcarouter/code-review init # write the workflow
-npx @orcarouter/code-review reconfigure # change blocking rules, diff limits, where config lives
-npx @orcarouter/code-review doctor # diagnose reviews that don't run or don't post
-npx @orcarouter/code-review uninstall # remove it (drops the merge gate first)
-```
+The same catalog the [OrcaDub MCP server](https://github.com/Continuum-AI-Corp/orcadub-mcp-server) uses, so IDs and paths match across Orca products. Detected agents are pre-ticked; `/` filters the list. For CI or dotfiles, the same choices are flags (`--mode`, `--scope`, `--platform`, `--yes`) — `--help` lists them.
Prefer to wire it by hand? The manual steps are below.
Claude Code, Cursor, Codex, OpenCode, Windsurf, Cline, RooCode, Continue, GitHub Copilot, Gemini CLI, Amazon Q Developer, Qwen Code, Kilo Code, Auggie, Kimi Code, Kiro, Lingma, Junie, CodeBuddy Code, CoStrict, Crush, Factory Droid, iFlow, Pi, Qoder, Antigravity, Antigravity 2.0, Bob Shell, ForgeCode, Trae, Trae CN, ZCode, MimoCode, Hermes, OpenClaw, Command Code.
-An existing identical skill is left unchanged; an existing **different** one is preserved unless you pass `--force`. Use `--json` for structured output and `NO_COLOR` for plain text.
+Two skills: `orca-review` for reviewing locally and `orca-review-action` for the GitHub Action. The installer asks which you want, or both. An existing identical skill is left unchanged; an existing **different** one is preserved unless you pass `--force`.
### Language
-The CLI speaks **English, Simplified Chinese, Japanese and Korean**, picked from your locale (`LC_ALL` / `LC_MESSAGES` / `LANG`). Override it per run, or pin it for good:
-
-```bash
-npx @orcarouter/code-review --lang en # English
-npx @orcarouter/code-review --lang zh # 简体中文
-npx @orcarouter/code-review --lang ja # 日本語
-npx @orcarouter/code-review --lang ko # 한국어
-export ORCACODE_LANG=ja # pin it
-```
+The CLI speaks **English, Simplified Chinese, Japanese and Korean**, picked from your locale (`LC_ALL` / `LC_MESSAGES` / `LANG`). The guided flow opens with a language screen; `ORCACODE_LANG=zh` pins it for good.
Traditional Chinese locales (`zh-TW`, `zh-HK`) fall back to English on purpose — the vocabulary diverges enough that serving Simplified reads worse than not translating at all.
-Guided flows open with a language screen when `--lang` is not given. Add `--no-banner` to skip the wordmark.
-
Menus are arrow-key driven — `↑↓` to move, `Enter` to pick. Multi-select adds `space` to toggle, `a`/`n` for all/none, and `/` to filter (`ctrl-u` clears it), which is how you find one agent among 36 without scrolling. Terminals without raw mode fall back to typing a number.
Only prose is translated — flags, platform IDs, workflow inputs, and shell commands stay verbatim, because you still have to type them.
---
+## Review locally — your agent is the engine
+
+The Action pays a model in CI to review every PR. You can also run the **same review, right now, in your terminal**, with no Action, no OrcaRouter account, and no API key — because the model is the one your coding agent already has.
+
+> **you:** review my changes
+
+
+
+
+
+Claude Code, Codex, Cursor, or any of the 36 platforms picks up the `orca-review` skill and becomes the reviewer. You say what to review; the skill handles the rest:
+
+| Say | It reviews |
+| --- | --- |
+| *"review my changes"* | Uncommitted work if the tree is dirty, otherwise this branch against its base |
+| *"review this branch"*, *"review that commit"* | The range you named |
+| *"review PR 556"* | That pull request — **without checking it out**. It is fetched into a private ref; your work tree stays exactly where it was. Fork PRs included |
+| *"is this safe to merge?"* | Same, and the gate answers |
+
+Behind the skill are two CLI commands your agent runs for you: `review plan` decides what is in scope — with the reasons for what is not — and hands the agent the per-language checklists, the P0–P3 rubric, and your repo's own `AGENTS.md`/`CLAUDE.md` conventions; `review submit` verifies every finding is filed on the right line, drops duplicates, applies the merge gate, and prints the report your agent relays to you. Nothing that decides what blocks is left to the model.
+
+**It is the same severity contract the Action enforces** — the same `rules/severity-instruction.md`, the same position check, the same result shape. A P1 you find here is a P1 that would block there. That parity is the point: *"it passed locally"* has to mean something.
+
+The file selection is the engine's, without the engine: the exclusion rules and per-language checklists from [Open Code Review](https://github.com/alibaba/open-code-review) (Apache-2.0) ship inside this package, so a local review filters the same files CI would. Nothing extra to install.
+
+Settings that should stick live in a committed `.orcacode-review.json` — which severities block, which language to report in, paths never to review, extra checklists for parts of the tree. You do not write it by hand: after your first review in a repo the agent offers to save the settings it just used, and later *"from now on only block on P0 locally"* or *"never review docs/"* edits the right key.
+
+Scripting your own harness instead of using an agent? `review plan --json` and `review submit --json` are a stable, versioned contract; [`skills/orca-review/references/contract.md`](skills/orca-review/references/contract.md) is the reference.
+
+---
+
## Quick Start
### 1. Enable OrcaCode Review
diff --git a/RELEASE.md b/RELEASE.md
index 44f0e0f..77f046f 100644
--- a/RELEASE.md
+++ b/RELEASE.md
@@ -66,6 +66,36 @@ The repo and the action stay `orca-code-review`.
`--access public` ships it private and the documented install command 404s for
everyone outside the org.
+#### The `1.x` line is retired; `2.0.0` is the supported floor
+
+1.0.2 through 1.5.0 were published on 25–26 Aug 2026 and are all **deprecated**:
+still installable, and every install of one prints
+
+```
+npm warn deprecated @orcarouter/code-review@1.x.y: No longer supported —
+install @orcarouter/code-review@latest
+```
+
+Deprecated rather than unpublished, and not by preference — see "Unpublishing
+cannot be automated" below. Two things left with that line:
+
+- **App mode.** Up to 1.4.0 the skill offered a GitHub App install as an
+ alternative to the Action. Installing an App is a permission grant only a
+ human can approve on a web page, so the agent could only hand over a link and
+ wait; 1.5.0 dropped it. 2.0.0 is the first version published without it, and
+ the major is where that break belongs.
+- **The self-dependency.** 1.1.0 through 1.5.0 declared
+ `@orcarouter/code-review: ^1.0.2` as a dependency *of itself*, so every `npx`
+ fetched a second, older copy of the CLI before running the one it was asked
+ for. Removed in 2.0.0 and pinned by a test — see "the CLI ships with no
+ dependencies" in `scripts/installer.test.mjs`.
+
+Deprecation is the right instrument for this anyway. It reaches exactly the
+people who need telling — anyone who installs one — without breaking anyone
+pinned, which an unpublish does with an E404 and no explanation. Run it with the
+**Deprecate** workflow (`workflow_dispatch`: versions, plus a message; an empty
+message clears the flag).
+
#### The unscoped `orcacode-review` name is retired
1.0.0 and 1.0.1 were published unscoped under a personal account, before the
@@ -77,7 +107,15 @@ numbers permanently — but mark the name dead:
npm deprecate orcacode-review "Moved to @orcarouter/code-review"
```
-Never publish to the old name again.
+**Still outstanding** as of 2.0.0 — both versions are live and unmarked. The
+**Deprecate** workflow cannot do it: `NPM_TOKEN` is scoped to
+`@orcarouter/code-review` alone, which is the property that makes a leak
+survivable. This one runs from the personal account that owns the name.
+
+Never publish to the old name again, and do not unpublish these two: emptying
+the package frees `orcacode-review` for anyone to claim 24 hours later, and the
+docs used to tell people to `npx` it. A dead name we still hold beats a live one
+someone else fills.
**Publishing is automatic.** To cut a release, bump `version` in `package.json`
(and the two `.claude-plugin` files — see below) and merge to `main`. That is
@@ -143,6 +181,22 @@ npm's unpublish window is 72 hours and a withdrawn version number can never be
reused, so anything checkable before publishing is checked there rather than
discovered afterward.
+**Unpublishing cannot be automated, and the `NPM_TOKEN` secret cannot do it.**
+Tried, on the 1.x withdrawal: every version came back
+
+```
+npm error code E403
+Granular access tokens that bypass two-factor authentication may not
+perform this action.
+```
+
+CI can publish and cannot withdraw, by registry policy rather than by
+configuration — so no token, scope, or workflow fixes it, and a workflow that
+offers the button is a guard rail around something that never runs. Withdrawing
+a version takes an interactive `npm login` (web or OTP) from a maintainer's own
+machine, inside the 72 hours. Plan on that when a release needs pulling: the
+person, not the pipeline.
+
Auth is the `NPM_TOKEN` repository secret. Use a **granular access token scoped
to this one package**, not a classic automation token — a leaked classic token
can publish anything in the account. Rotate with `gh secret set NPM_TOKEN`.
@@ -214,7 +268,7 @@ its own plugin marketplace:
The plugin is served from the repo's **default branch**, not from a tag — a
change to `skills/` reaches users as soon as it lands on `main`, with no release
-step. Treat `skills/setup-orca-code-review/` as shipped surface on merge.
+step. Treat `skills/orca-review-action/` as shipped surface on merge.
Three versions have to move together on a feature release: `package.json`,
`.claude-plugin/plugin.json`, and `.claude-plugin/marketplace.json`
diff --git a/action.yml b/action.yml
index 3653451..1b15546 100644
--- a/action.yml
+++ b/action.yml
@@ -95,12 +95,31 @@ inputs:
KB. A skip posts a notice without running the engine; whether the check
then fails or passes is `on-oversized-diff` (default: fail, so the
merge gate cannot be padded around).
+
+
+ The default is a ceiling on WASTE, not on what the engine can handle.
+ It used to be 512, chosen to avoid spending tokens on diffs that review
+ badly — but it also meant a large PR could never be reviewed at all, and
+ "never" is worse than "expensive". The engine reads the diff through its
+ own tooling rather than taking it as one prompt, so size alone does not
+ decide whether it can work.
+
+
+ What DOES bound it is the model's context window, which the action cannot
+ see. Past that point a run reports a wall-clock timeout or an aborted
+ pass instead of a clean skip — louder, and no longer a silent refusal to
+ look. Lower this if you would rather have the clean skip back.
required: false
- default: "512"
+ default: "5000"
max-diff-files:
- description: "Skip the review when the diff touches more than this many files (same notice + on-oversized-diff outcome as max-diff-kb)."
+ description: >-
+ Skip the review when the diff touches more than this many files (same
+ notice + on-oversized-diff outcome as max-diff-kb). Counted separately
+ from size because the two describe different shapes: a mechanical rename
+ across 900 files is small and shallow, one 4 MB generated file is large
+ and shallow, and either can be worth reviewing.
required: false
- default: "300"
+ default: "2000"
on-oversized-diff:
description: >-
What an oversized-diff skip does to the check. "fail" (default) fails
@@ -134,11 +153,17 @@ inputs:
hasn't produced a result within this window it is killed and the run
fails closed with a distinct "wall-clock timeout" error (separate from
"no usable result", so the log makes clear which mode failed). Accepts
- decimals (e.g. "0.5" = 30 seconds) for testing. Default 20 covers most
- PRs on shipped models; bump for very large diffs or slow-per-call
- models where per-file review takes longer.
+ decimals (e.g. "0.5" = 30 seconds) for testing.
+
+
+ PER PASS, which matters more now that the default is 60: `exhaustive`
+ mode makes up to three passes, so the worst case is three times this
+ number. The default rose with the diff limits — a review allowed to
+ accept a much larger diff needs the time to actually read it, and 20
+ minutes would have converted "too big, skipped cleanly" into "killed
+ halfway", which is the same non-review with a worse message.
required: false
- default: "20"
+ default: "60"
precision-filter:
description: >-
Post-processing between the engine and the merge gate. When `"true"`
@@ -147,8 +172,13 @@ inputs:
drops findings whose snippet does not match the claimed path; then an
LLM judge (L2) clusters findings by root cause and drops
low-confidence ones. Set to `"false"` to post the engine's raw
- findings directly. Both layers are soft-fail — errors keep the prior
- stage's findings and never abort the review.
+ findings directly — that skips BOTH layers, and is the supported way to
+ review without a judge.
+
+ L1 is soft-fail: if it errors the engine's findings carry forward. L2 is
+ NOT. L1's output is not publishable on its own, so a judge still
+ unavailable after retries fails the run closed rather than posting
+ unjudged findings.
required: false
default: "true"
judge-model:
@@ -216,10 +246,12 @@ runs:
"$RUNNER_TEMP/cr-facts.json" "$RUNNER_TEMP/proxy.out" "$RUNNER_TEMP/proxy.err" \
"$RUNNER_TEMP/policy-block.json" \
"$RUNNER_TEMP/pr.diff" "$RUNNER_TEMP/diff-guard.json" "$RUNNER_TEMP/prev-summary.md" \
- "$RUNNER_TEMP/wallclock-timeout" "$RUNNER_TEMP/cr-usage.jsonl" \
+ "$RUNNER_TEMP/wallclock-timeout" "$RUNNER_TEMP/judge-unavailable" \
+ "$RUNNER_TEMP/cr-usage.jsonl" \
"$RUNNER_TEMP/cr-clean" \
"$RUNNER_TEMP/result.l1.json" "$RUNNER_TEMP/result.l2.json" \
- "$RUNNER_TEMP/result-extra.l1.json" "$RUNNER_TEMP/result-extra.l2.json"
+ "$RUNNER_TEMP/result-extra.l1.json" "$RUNNER_TEMP/result-extra.l2.json" \
+ "$RUNNER_TEMP/result.l2.log" "$RUNNER_TEMP/result-extra.l2.log"
- name: Resolve PR refs
id: pr
@@ -468,7 +500,11 @@ runs:
# differently from the workspace default, whereas which severities a
# reader SEES is the workspace's call. Defaulting to every severity means
# a gateway that predates the field behaves exactly as before.
- REPORT_ON=$(setting report_on "P0,P1,P2,P3")
+ # Agrees with settings.mjs DEFAULTS on purpose. settings.mjs always writes
+ # a defaulted file, so this fires only if that file is missing or
+ # unreadable — and two different answers to the same question is how one
+ # of them ends up wrong without anyone noticing.
+ REPORT_ON=$(setting report_on "P0,P1")
RUBRIC=$(setting rubric "")
# PRECEDENCE (documented in the README): an explicit `with:` input that
@@ -654,7 +690,7 @@ runs:
`lose signal — ${outcome}\n\n` +
`**What now?** Split the PR (smaller diffs get much higher-signal reviews), raise ` +
`the limits, or make skips advisory:\n\n` +
- '```yaml\nwith:\n max-diff-kb: "1024" # default 512\n max-diff-files: "500" # default 300\n on-oversized-diff: "pass" # default "fail"\n```' +
+ '```yaml\nwith:\n max-diff-kb: "10000" # default 5000\n max-diff-files: "4000" # default 2000\n on-oversized-diff: "pass" # default "fail"\n```' +
footer;
const comments = await github.paginate(github.rest.issues.listComments, {
@@ -904,11 +940,39 @@ runs:
# produced a usable JSON result on this pass — a timeout or crash
# leaves nothing to filter, so we skip straight to CHECK which
# will fail closed on rc anyway. Both layers write to a temp file
- # first and only overwrite $1 on success (fail-safe: any
- # postprocessing error keeps the prior stage's findings, never
- # aborts the review or corrupts the engine output).
+ # first and only overwrite $1 on success, so no postprocessing error
+ # can corrupt the engine output.
+ #
+ # L1 is fail-safe in the full sense: an error there keeps the engine's
+ # findings and the review continues. L2 is not — see the judge branch
+ # below for why an unavailable judge has to fail the pass.
+ #
+ # A PARTIAL RESULT IS ALREADY DOOMED, so do not spend the filter on
+ # it. CHECK fails closed on any non-empty `warnings`, so a pass that
+ # carries them cannot be published however the judge rules. Filtering
+ # anyway buys a judge call for output nobody will read — and if THAT
+ # call is what fails, the summary blames "the L2 judge was
+ # unavailable" for a run whose actual defect was a partial engine
+ # result, the exact mis-attribution the reason= tags were added to
+ # end. Falling through to CHECK lets reason=partial name it, and
+ # name the warnings with it.
+ #
+ # `warnings` IS NORMALIZED THE WAY CHECK NORMALIZES IT — Array.isArray
+ # or nothing (check-result.mjs:44). The two predicates decide the fate
+ # of the same field and must not disagree: a non-array `warnings`, say
+ # the string "oops", has a nonzero .length, so a bare length test reads
+ # it as partial and skips the filter, while CHECK reads it as absent and
+ # publishes. That combination ships the engine's raw findings with no
+ # L1 and no L2 under `precision-filter: true` — the exact thing this
+ # pass fails closed to prevent, arriving as a green review.
+ #
+ # Which fixes the direction this test has to fail in: SKIPPING is the
+ # branch that can publish unjudged, so it may only be taken when the
+ # result is definitely partial. Anything malformed or unrecognized runs
+ # the filter.
+ JUDGE_FAILED=""
if [ "$PRECISION_FILTER" = "true" ] && [ "$rc" = "0" ] && [ -s "$1" ] \
- && node -e "process.exit((require(process.argv[1]).comments||[]).length>0?0:1)" "$1" 2>/dev/null; then
+ && node -e "const r=require(process.argv[1]);const w=Array.isArray(r.warnings)?r.warnings:[];process.exit((r.comments||[]).length>0&&w.length===0?0:1)" "$1" 2>/dev/null; then
echo "::group::Precision filter — $2"
L1_OUT="${1%.json}.l1.json"
if node "$POSTFILTER" "$1" "$GITHUB_WORKSPACE" "$HEAD" --out "$L1_OUT" 2>&1 \
@@ -937,7 +1001,8 @@ runs:
# stamps x-cr-lens: judge, so the recipe's judge rule picks the model.
#
# Same model as before — the shipped recipe routes that rule to
- # deepseek-v4-pro, which is what judge-model used to name. What moves is
+ # a model of its own choosing, which is what judge-model used to name
+ # directly. What moves is
# where the choice lives, so a workspace can price its judge by editing its
# own recipe rather than waiting for a release.
#
@@ -946,8 +1011,17 @@ runs:
# success. That is why the provisioned recipe carries the rule, and why
# setting judge-model explicitly stays the way to pin a model outright.
L2_MODEL="${JUDGE_MODEL:-$ROUTER}"
- if node "$JUDGE" "$1" --model "$L2_MODEL" --threshold "$JUDGE_THRESHOLD" --out "$L2_OUT" 2>&1 \
- && [ -s "$L2_OUT" ]; then
+ # CAPTURED, not just streamed, so the failure branch can name WHICH
+ # failure this was. The judge exits 1 for a 400, a 401, a missing
+ # endpoint and a model that answered with prose, and its exit code
+ # tells those apart from each other not at all — the reason only ever
+ # existed in its stderr. Everything is echoed below either way, so the
+ # log reads exactly as before.
+ L2_LOG="${1%.json}.l2.log"
+ L2_RC=0
+ node "$JUDGE" "$1" --model "$L2_MODEL" --threshold "$JUDGE_THRESHOLD" --out "$L2_OUT" > "$L2_LOG" 2>&1 || L2_RC=$?
+ cat "$L2_LOG"
+ if [ "$L2_RC" = "0" ] && [ -s "$L2_OUT" ]; then
L2_IN_COUNT=$(node -pe "(require('$1').comments||[]).length")
L2_OUT_COUNT=$(node -pe "(require('$L2_OUT').comments||[]).length")
echo "L2 judge ($L2_MODEL, thr=$JUDGE_THRESHOLD): $L2_IN_COUNT -> $L2_OUT_COUNT"
@@ -959,10 +1033,51 @@ runs:
mv -f "$POLICY_BLOCK_SNAPSHOT" "$POLICY_BLOCK"
fi
else
- echo "::warning::L2 judge failed — keeping L1 findings"
- # L2 failed. Restore engine's block if we snapshotted one; else
- # clear any L2-authored block so a sidecar guardrail hit doesn't
- # surface as the engine's own block on the primary review.
+ # L1 AND L2 ARE ONE STAGE, NOT TWO. L1's output is not a shippable
+ # review: it is the engine's raw findings with mismatched snippets
+ # re-homed, still carrying the duplicate and low-value findings that
+ # L2 exists to cluster and drop. Publishing it because the judge was
+ # unreachable ships exactly the noise the filter was turned on to
+ # remove — worse for the reader than no review, and it arrives
+ # looking like a finished one.
+ #
+ # So an unavailable judge fails the pass. `precision-filter: "false"`
+ # remains the supported way to run without a judge, and it skips L1
+ # too, which is the honest version of that choice.
+ # NAME THE FAILURE CLASS, because they are not the same problem and
+ # do not have the same fix. A 502 is the gateway having a moment and
+ # is worth re-running; a 401 is a key the workspace has to correct;
+ # a model answering in prose is a routing or recipe problem. Calling
+ # all three "unavailable after retries" sends the reader to wait out
+ # an outage that is not happening — the same wrong-cause reading
+ # this pass exists to end, one layer up.
+ #
+ # FIRST line, not the last: the judge prints "judge did not return
+ # JSON:" followed by a dump of what the model actually said, so a
+ # tail would quote a fragment of the bad completion in place of the
+ # reason. On a failing exit nothing else has been printed, so the
+ # first line is the message.
+ if [ "$L2_RC" = "0" ]; then
+ L2_REASON="it exited cleanly but wrote no output"
+ else
+ L2_REASON=$(head -n 1 "$L2_LOG")
+ [ -n "$L2_REASON" ] || L2_REASON="it exited $L2_RC without a message"
+ fi
+ # STATED, NOT ADJUDICATED. Whether a failed judge ends the run is
+ # the caller's call, not this function's: on the mandatory pass it
+ # is fatal, on a best-effort exhaustive pass run_review warns and
+ # keeps what it has. An `::error:: … failing closed` here put a red
+ # annotation claiming a failure onto runs that went green — the same
+ # register of wrong-claim this pass exists to remove. Both callers
+ # already print the consequence, so this is a plain log line inside
+ # the group the operator is already reading. Matches the wall-clock
+ # marker above, which reports and lets run_review adjudicate.
+ echo "L2 judge failed — $L2_REASON (L1 output is not publishable on its own, so this pass yields nothing)"
+ printf '%s' "$L2_REASON" > "$RUNNER_TEMP/judge-unavailable"
+ JUDGE_FAILED=1
+ # Restore engine's block if we snapshotted one; else clear any
+ # L2-authored block so a sidecar guardrail hit doesn't surface as
+ # the engine's own block on the primary review.
if [ -f "$POLICY_BLOCK_SNAPSHOT" ]; then
mv -f "$POLICY_BLOCK_SNAPSHOT" "$POLICY_BLOCK"
else
@@ -971,6 +1086,7 @@ runs:
fi
unset POLICY_BLOCK_SNAPSHOT
echo "::endgroup::"
+ [ -z "$JUDGE_FAILED" ] || return 1
fi
node "$CHECK" "$1" "$rc"
}
@@ -988,10 +1104,16 @@ runs:
# $4 = extra passes allowed
run_review() {
if ! run_pass "$RESULT" "$1" "$2" "$3"; then
+ # WHICH GATE FAILED, in the summary and not only inside a collapsed
+ # log group. These all used to print the same "no usable result", so a
+ # reader could not tell an engine crash from a partial run from an
+ # unavailable judge, and picked whichever cause was last in the log.
if [ -f "$RUNNER_TEMP/wallclock-timeout" ]; then
echo "::error::$BRAND: $(cat "$RUNNER_TEMP/wallclock-timeout") — engine killed, failing closed. Bump the 'timeout-minutes' input for very large PRs or slow-per-call models."
+ elif [ -f "$RUNNER_TEMP/judge-unavailable" ]; then
+ echo "::error::$BRAND: the L2 judge failed — $(cat "$RUNNER_TEMP/judge-unavailable") — failing closed. Findings are not published unjudged; set 'precision-filter: false' to review without a judge."
else
- echo "::error::$BRAND: review engine produced no usable result — failing closed."
+ echo "::error::$BRAND: review engine produced no usable result — failing closed. The log carries a reason= tag naming which check failed."
fi
exit 1
fi
@@ -1020,7 +1142,15 @@ runs:
# Otherwise a benign tooling failure (engine crash / partial
# result): extra depth is best-effort, so warn and END exhaustion,
# keeping the findings already in hand.
- echo "::warning::$BRAND: exhaustive pass $extra produced no usable result — keeping the findings so far."
+ # Carry the reason into the downgrade. Without this a judge failure
+ # reads as "no usable result", which is what an engine crash also
+ # reads as — and the reason is the only thing that tells an operator
+ # whether to re-run or fix a key.
+ if [ -f "$RUNNER_TEMP/judge-unavailable" ]; then
+ echo "::warning::$BRAND: exhaustive pass $extra — the L2 judge failed ($(cat "$RUNNER_TEMP/judge-unavailable")) — keeping the findings so far."
+ else
+ echo "::warning::$BRAND: exhaustive pass $extra produced no usable result — keeping the findings so far."
+ fi
break
fi
# Guard the merge + parse like the run_pass failure path above: an
@@ -1704,7 +1834,9 @@ runs:
"$RUNNER_TEMP/cr-facts.json" "$RUNNER_TEMP/proxy.out" "$RUNNER_TEMP/proxy.err" \
"$RUNNER_TEMP/policy-block.json" \
"$RUNNER_TEMP/pr.diff" "$RUNNER_TEMP/diff-guard.json" "$RUNNER_TEMP/prev-summary.md" \
- "$RUNNER_TEMP/wallclock-timeout" "$RUNNER_TEMP/cr-usage.jsonl" \
+ "$RUNNER_TEMP/wallclock-timeout" "$RUNNER_TEMP/judge-unavailable" \
+ "$RUNNER_TEMP/cr-usage.jsonl" \
"$RUNNER_TEMP/cr-clean" \
"$RUNNER_TEMP/result.l1.json" "$RUNNER_TEMP/result.l2.json" \
- "$RUNNER_TEMP/result-extra.l1.json" "$RUNNER_TEMP/result-extra.l2.json"
+ "$RUNNER_TEMP/result-extra.l1.json" "$RUNNER_TEMP/result-extra.l2.json" \
+ "$RUNNER_TEMP/result.l2.log" "$RUNNER_TEMP/result-extra.l2.log"
diff --git a/bin/harness.mjs b/bin/harness.mjs
new file mode 100644
index 0000000..fc02d5f
--- /dev/null
+++ b/bin/harness.mjs
@@ -0,0 +1,952 @@
+// The harness surface: what an outside AI agent calls to run a review itself.
+//
+// The Action drives a review by paying an engine to think (Open Code Review's
+// `ocr review`, against OrcaRouter). This file drives one WITHOUT buying any
+// thinking: the caller — Claude Code, Codex, Cursor, whatever is holding the
+// file — supplies the model out of its own subscription, and we supply
+// everything that must not be left to a language model.
+//
+// Two commands, deliberately split at the point where judgement enters:
+//
+// review plan -> everything BEFORE the model: which files are in scope,
+// which rules apply to each, the severity rubric, the
+// project's own conventions, and the exact result shape.
+// review submit -> everything AFTER the model: shape validation, the L1
+// position check, the severity gate, the report, the exit
+// code.
+//
+// WHY A SPLIT AND NOT ONE COMMAND. We cannot call the agent — the agent is
+// calling us. So the surface has to be two halves it can sandwich itself
+// between. That also makes both halves ordinary CLI commands any harness can
+// script, which is the whole point of shipping a surface rather than a prompt.
+//
+// WHERE THE FILE SELECTION COMES FROM. The same place CI's engine gets it: Open
+// Code Review's exclusion rules and per-language checklists, vendored as data
+// under vendor/open-code-review and read by a port of its matcher in
+// selection.mjs. No binary to install — the engine is 50 MB per platform and
+// the review here needs a few hundred KB of it. What is NOT borrowed is the
+// output contract: its own guidance is High/Medium/Low with "Low — discard
+// silently"; ours is P0-P3 with calibrated boundaries and an explicit "P2 and
+// P3 are all still emitted", because those tags feed a merge gate. So: their
+// file selection, our severity semantics, one result.json shape shared with
+// the Action.
+//
+// NOTHING HERE TALKS TO ORCAROUTER. No API key, no gateway, no control plane.
+// A repo that never installs the Action can still use this.
+
+import fs from "node:fs";
+import path from "node:path";
+import { spawnSync } from "node:child_process";
+import { fileURLToPath } from "node:url";
+
+import { SEVERITIES, severityOf, countSeverities } from "../scripts/severity.mjs";
+import { partition, groupRules } from "./selection.mjs";
+import { loadLocalConfig } from "./localconfig.mjs";
+import { LANGUAGES, makeT } from "./i18n.mjs";
+
+// How the plan names the language findings must be written in. Addressed to a
+// model, so each name is given in the language itself and in English: the
+// model must recognise it whatever language it happens to be thinking in.
+export const LANGUAGE_NAMES = Object.freeze({
+ en: "English",
+ zh: "简体中文 (Simplified Chinese)",
+ ja: "日本語 (Japanese)",
+ ko: "한국어 (Korean)",
+});
+
+const englishT = makeT("en");
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const PKG_ROOT = path.resolve(HERE, "..");
+
+// Bumped only when the shape of `review plan --json` changes incompatibly, so a
+// harness that pinned an older reading can detect the break instead of silently
+// misreading a renamed field. Mirrors `ocr delegate`'s own schema_version.
+export const SCHEMA_VERSION = "1";
+
+// Where the request and the result live. Inside the work tree (every agent
+// sandbox can write there — .git and $TMPDIR are not reliably writable under
+// Codex's workspace-write or a container mount), but excluded from git via
+// .git/info/exclude rather than .gitignore, so reviewing a repo never dirties
+// it with a file the user then has to decide whether to commit.
+export const WORK_DIR = ".orcacode-review";
+
+// Four fields, and every one of them is load-bearing — see the field notes in
+// renderPlan(). Deliberately the SHORT form: `line` rather than the
+// `start_line`/`end_line` pair the pipeline speaks internally, and no
+// `warnings` key. Both are widened at the boundary in validateResult(), and the
+// difference is a third of the bytes an agent has to emit — which is a third
+// less of an unreadable blob scrolling past the person watching it work.
+const RESULT_SHAPE = `{
+ "comments": [
+ {
+ "path": "src/auth.ts",
+ "line": 41,
+ "existing_code": " if (token === expected) {",
+ "content": "[P0] **Token comparison is not constant-time**\\n\\n\`===\` on a secret leaks its prefix through timing. Use \`crypto.timingSafeEqual\`."
+ }
+ ]
+}`;
+
+// The line a finding is anchored to, by the same rule the Action's posting step
+// uses: the end of the range if there is one, else the start. Null when the
+// position check cleared it — a re-homed finding whose line could not be
+// resolved in its new file must render without one rather than with the stale
+// number from the file it was wrongly filed on.
+export function anchorLine(c) {
+ if (Number(c?.end_line) >= 1) return Number(c.end_line);
+ if (Number(c?.start_line) >= 1) return Number(c.start_line);
+ return null;
+}
+
+// ------------------------------------------------------------------- git ---
+
+// `raw: true` skips the trim. Porcelain is a fixed-column format whose first
+// column can be a space (" M foo"), and trimming the whole buffer eats that
+// space off the FIRST line only — which silently truncates one path per run and
+// looks like a git bug rather than ours.
+function git(args, cwd, { raw = false } = {}) {
+ const r = spawnSync("git", args, { cwd, encoding: "utf8", maxBuffer: 1 << 26 });
+ const out = r.stdout || "";
+ return { ok: r.status === 0, out: raw ? out.replace(/\n$/, "") : out.trim(), err: (r.stderr || "").trim() };
+}
+
+export function repoRoot(cwd) {
+ const r = git(["rev-parse", "--show-toplevel"], cwd);
+ return r.ok ? r.out : null;
+}
+
+// Where `info/exclude` lives. In a linked worktree `.git` is a FILE pointing at
+// the real store, so joining root + ".git" lands on a path that is not a
+// directory and the exclude write silently does nothing.
+export function gitCommonDir(cwd) {
+ const r = git(["rev-parse", "--path-format=absolute", "--git-common-dir"], cwd);
+ if (r.ok && r.out) return r.out;
+ // --path-format landed in git 2.31; fall back to the relative form.
+ const rel = git(["rev-parse", "--git-common-dir"], cwd);
+ if (!rel.ok || !rel.out) return null;
+ return path.isAbsolute(rel.out) ? rel.out : path.resolve(cwd, rel.out);
+}
+
+// The branch a PR would target. Tried in the order that gets it right most
+// often: what the remote actually says its HEAD is, then the conventional
+// names. Returns null rather than guessing "main" — a wrong base produces a
+// diff full of other people's work, which is worse than asking the caller.
+export function defaultBranch(cwd) {
+ const sym = git(["symbolic-ref", "--quiet", "refs/remotes/origin/HEAD"], cwd);
+ if (sym.ok && sym.out) return sym.out.replace(/^refs\/remotes\//, "");
+ for (const ref of ["origin/main", "origin/master", "main", "master"]) {
+ if (git(["rev-parse", "--verify", "--quiet", `${ref}^{commit}`], cwd).ok) return ref;
+ }
+ return null;
+}
+
+export function isDirty(cwd) {
+ const r = git(["status", "--porcelain"], cwd);
+ return r.ok && r.out.length > 0;
+}
+
+/**
+ * Decides what "review my changes" means here.
+ *
+ * Explicit flags always win. With none, we pick between the two things a person
+ * standing in a repo could mean, and SAY which we picked and why — an implicit
+ * choice the user cannot see is a bug report waiting to happen:
+ *
+ * dirty work tree -> workspace (they are mid-change; that is the
+ * review they want before committing)
+ * clean, ahead of the base -> range (the branch, i.e. what CI will see)
+ * clean, not ahead -> nothing to review
+ */
+export function resolveMode({ from, to, commit, worktree, pr }, cwd) {
+ if (commit) return { mode: "commit", commit, code: "explicit" };
+ if (worktree) return { mode: "workspace", code: "explicit" };
+ if (from || to) {
+ const base = from || defaultBranch(cwd);
+ if (!base) return { mode: "error", code: "no-base" };
+ // `pr` is only a label here — resolvePr() has already turned the number into
+ // the two refs handed in as from/to. Keeping the resolution out of this
+ // function keeps it synchronous, offline, and testable.
+ const range = { mode: "range", from: base, to: to || "HEAD", code: pr ? "pr" : "explicit" };
+ return pr ? { ...range, pr } : range;
+ }
+
+ if (isDirty(cwd)) return { mode: "workspace", code: "auto-dirty" };
+
+ const base = defaultBranch(cwd);
+ if (!base) return { mode: "error", code: "no-base" };
+ const mb = git(["merge-base", base, "HEAD"], cwd);
+ const head = git(["rev-parse", "HEAD"], cwd);
+ if (!mb.ok || !head.ok) return { mode: "error", code: "no-compare", base };
+ if (mb.out === head.out) return { mode: "empty", code: "not-ahead", base };
+ return { mode: "range", from: base, to: "HEAD", code: "auto-ahead", base };
+}
+
+// The English rendering of `code`, for the plan prompt only. The terminal says
+// the same thing through i18n — the prompt is never translated because it is
+// model input, not user prose.
+export const MODE_REASON = Object.freeze({
+ explicit: "you asked for it",
+ pr: "you named a pull request",
+ "auto-dirty": "the work tree has uncommitted changes",
+ "auto-ahead": "clean work tree, ahead of the base branch",
+ "not-ahead": "HEAD has no commits beyond the base branch",
+ "no-base": "no base branch could be resolved",
+ "no-compare": "HEAD could not be compared with the base branch",
+});
+
+// ------------------------------------------------------------- pull request ---
+
+// Namespaced so the fetch never creates a branch, never shows up in `git
+// branch`, and never collides with anything the user owns. Re-fetching the same
+// PR just moves these two refs.
+const PR_REF_NS = "refs/orcacode/pr";
+
+function gh(args, cwd) {
+ const r = spawnSync("gh", args, { cwd, encoding: "utf8", maxBuffer: 1 << 24 });
+ return {
+ ok: r.status === 0,
+ out: (r.stdout || "").trim(),
+ err: (r.stderr || "").trim(),
+ missing: r.error?.code === "ENOENT",
+ };
+}
+
+const firstLine = (s) => (s || "").split("\n").find((l) => l.trim()) || "";
+
+// `refs/pull/N/head` lives on the BASE repository, which in a fork workflow is
+// not `origin`. Ask gh which repo it resolved, then find the remote pointing at
+// it, so a fork checkout fetches from upstream rather than failing.
+function baseRemote(cwd) {
+ const remotes = git(["remote"], cwd);
+ const names = remotes.ok ? remotes.out.split("\n").filter(Boolean) : [];
+ const slug = gh(["repo", "view", "--json", "nameWithOwner", "-q", ".nameWithOwner"], cwd);
+ if (slug.ok && slug.out) {
+ const want = slug.out.toLowerCase();
+ for (const name of names) {
+ const url = git(["remote", "get-url", name], cwd);
+ if (!url.ok) continue;
+ const got = url.out
+ .toLowerCase()
+ .replace(/\.git$/, "")
+ .replace(/^.*[:/]([^/:]+\/[^/]+)$/, "$1");
+ if (got === want) return name;
+ }
+ }
+ return names.includes("origin") ? "origin" : names[0] || "origin";
+}
+
+const refExists = (ref, cwd) => git(["rev-parse", "--verify", "--quiet", `${ref}^{commit}`], cwd).ok;
+
+/**
+ * Turn a PR number into a reviewable range, without touching the work tree.
+ *
+ * Deliberately does NOT check the branch out. Checking out would stash-or-fail
+ * on a dirty tree, move the user off what they were doing, and leave them
+ * somewhere else when the review ends. Fetching the PR into a private ref
+ * namespace gets the same diff and is invisible.
+ *
+ * `gh` is a soft dependency: absent, this returns `no-gh` and the caller tells
+ * the user to check the branch out by hand. Nothing else in the harness needs
+ * it, and nothing about the review changes when it is missing.
+ *
+ * Returns { ok: true, from, to, background, pr } or { ok: false, code, detail }.
+ */
+export function parsePrNumber(value) {
+ const n = String(value ?? "")
+ .trim()
+ .replace(/^#/, "");
+ return /^\d+$/.test(n) ? n : null;
+}
+
+export function resolvePr(number, cwd) {
+ const n = parsePrNumber(number);
+ if (!n) return { ok: false, code: "bad-number", detail: String(number ?? "") };
+
+ if (gh(["--version"], cwd).missing) return { ok: false, code: "no-gh" };
+
+ const fields = "number,title,body,url,state,baseRefName,headRefName,headRefOid,isCrossRepository";
+ const view = gh(["pr", "view", n, "--json", fields], cwd);
+ if (!view.ok) {
+ const detail = firstLine(view.err);
+ if (/auth|logged in|gh auth login/i.test(detail)) return { ok: false, code: "no-auth", detail };
+ return { ok: false, code: "gh-failed", detail };
+ }
+
+ let meta;
+ try {
+ meta = JSON.parse(view.out);
+ } catch {
+ return { ok: false, code: "gh-failed", detail: "gh returned output that is not JSON" };
+ }
+
+ const remote = baseRemote(cwd);
+ const head = `${PR_REF_NS}/${n}/head`;
+ const base = `${PR_REF_NS}/${n}/base`;
+
+ // Two fetches, not one refspec pair: a base branch deleted after merge must
+ // not take the head fetch down with it.
+ const headFetch = git(["fetch", "--no-tags", "--quiet", remote, `+refs/pull/${n}/head:${head}`], cwd);
+ if (!headFetch.ok && !refExists(meta.headRefOid || head, cwd)) {
+ return { ok: false, code: "fetch-failed", detail: firstLine(headFetch.err) };
+ }
+
+ git(["fetch", "--no-tags", "--quiet", remote, `+refs/heads/${meta.baseRefName}:${base}`], cwd);
+ const from = refExists(base, cwd)
+ ? base
+ : [`${remote}/${meta.baseRefName}`, meta.baseRefName].find((r) => refExists(r, cwd)) || "";
+ if (!from) return { ok: false, code: "no-base-ref", detail: meta.baseRefName };
+
+ return {
+ ok: true,
+ from,
+ to: refExists(head, cwd) ? head : meta.headRefOid,
+ // The PR description is exactly the business context the rubric asks for —
+ // what the change is FOR. Capped because a template-heavy body can dwarf the
+ // rest of the prompt. An explicit --background still wins; see the caller.
+ background: [meta.title, (meta.body || "").trim()].filter(Boolean).join("\n\n").slice(0, 4000),
+ pr: {
+ number: Number(n),
+ title: meta.title || "",
+ url: meta.url || "",
+ state: meta.state || "",
+ base: meta.baseRefName || "",
+ head: meta.headRefName || "",
+ fork: Boolean(meta.isCrossRepository),
+ },
+ };
+}
+
+export function mergeBaseOf(range, cwd) {
+ if (range.mode !== "range") return "";
+ const r = git(["merge-base", range.from, range.to], cwd);
+ return r.ok ? r.out : "";
+}
+
+// The commit whose tree the L1 position check greps. Empty in workspace mode:
+// uncommitted code is in no tree, so there is nothing to grep and the check has
+// to be skipped rather than silently reporting every finding as unlocatable.
+export function groundTruthRef(range, cwd) {
+ if (range.mode === "commit") return range.commit;
+ if (range.mode === "range") {
+ const r = git(["rev-parse", range.to], cwd);
+ return r.ok ? r.out : range.to;
+ }
+ return "";
+}
+
+// The git incantation the agent should run per file to see the change. Handed
+// over as a string rather than run for them: the agent already has shell and a
+// repo checkout, and a diff we paste into the plan would be a second copy that
+// can disagree with the tree they are reading.
+export function diffRecipe(range, mergeBase) {
+ switch (range.mode) {
+ case "range":
+ return `git diff ${mergeBase || range.from}..${range.to} -- `;
+ case "commit":
+ return `git show ${range.commit} -- `;
+ default:
+ return "git diff HEAD -- # and for an untracked file, just read it";
+ }
+}
+
+// --------------------------------------------------------------- selector ---
+
+/**
+ * Which files are in scope, and why the others are not.
+ *
+ * Git says what changed; the vendored Open Code Review rules say what is worth
+ * a reviewer's time — binaries, deleted files, unsupported types, tests and
+ * fixtures and generated code by default pattern, anything under .gitignore or
+ * an always-skipped directory. Each exclusion carries a REASON, which plain git
+ * cannot give and which the plan shows so a reviewer never wonders where a file
+ * went.
+ *
+ * Returns { selector, files, excluded, merge_base }.
+ */
+export function selectFiles(range, cwd, { exclude = [] } = {}) {
+ const { files, excluded } = partition(gitFiles(range, cwd).files, cwd, { exclude });
+ return { selector: "builtin", files, excluded, merge_base: "" };
+}
+
+function gitFiles(range, cwd) {
+ const args =
+ range.mode === "commit"
+ ? ["show", "--name-status", "--format=", range.commit]
+ : range.mode === "range"
+ ? ["diff", "--name-status", `${range.from}...${range.to}`]
+ : // -uall expands an untracked DIRECTORY into its files; without it
+ // porcelain reports "newdir/" and the reviewer gets a path that is
+ // not a file.
+ ["status", "--porcelain", "-uall"];
+
+ const r = git(args, cwd, { raw: range.mode === "workspace" });
+ if (!r.ok || !r.out) return { files: [], excluded: [] };
+
+ const stats = numstat(range, cwd);
+ const files = [];
+ for (const raw of r.out.split("\n")) {
+ // NOT trimmed. Porcelain is a fixed-column format — "XY path" where either
+ // status column may be a SPACE (" M foo" = modified, unstaged). Trimming
+ // the line shifts the path left and slices characters off the front of it.
+ if (!raw.trim()) continue;
+
+ let status;
+ let file;
+ if (range.mode === "workspace") {
+ status = raw.slice(0, 2);
+ file = raw.slice(3);
+ } else {
+ // --name-status is "X\tpath", and a rename is "R100\told\tnew" — the new
+ // path is the one that exists now.
+ const parts = raw.split("\t");
+ status = parts[0];
+ file = parts[parts.length - 1];
+ }
+
+ // A rename in porcelain is "R old -> new" on one line.
+ const arrow = file.indexOf(" -> ");
+ if (arrow >= 0) file = file.slice(arrow + 4);
+ // Porcelain quotes any path with unusual characters and escapes them; the
+ // quotes are not part of the name.
+ file = file.replace(/^"(.*)"$/, "$1");
+ if (!file) continue;
+
+ // A deleted file is kept HERE and excluded by the selector, with the reason
+ // stated — dropping it silently left a reviewer counting files that were
+ // never listed.
+ const st = stats.get(file) || { insertions: 0, deletions: 0, binary: false };
+ files.push({
+ path: file,
+ status: status.trim(),
+ insertions: st.insertions,
+ deletions: st.deletions,
+ binary: st.binary,
+ deleted: status.trim().startsWith("D"),
+ });
+ }
+ return { files, excluded: [] };
+}
+
+// Per-file line counts and the binary flag, from --numstat, keyed by the path
+// that exists now. A binary shows as "-\t-\tpath". A rename shows either as
+// "old => new" or with the changed part braced, "dir/{old => new}/file"; both
+// are reduced to the new name so they join the --name-status row.
+function numstat(range, cwd) {
+ const args =
+ range.mode === "commit"
+ ? ["show", "--numstat", "--format=", range.commit]
+ : range.mode === "range"
+ ? ["diff", "--numstat", `${range.from}...${range.to}`]
+ : ["diff", "--numstat", "HEAD"];
+ const r = git(args, cwd);
+ const out = new Map();
+ if (!r.ok || !r.out) return out;
+ for (const line of r.out.split("\n")) {
+ const m = /^(\d+|-)\t(\d+|-)\t(.+)$/.exec(line);
+ if (!m) continue;
+ let file = m[3].replace(/^"(.*)"$/, "$1");
+ file = file.replace(/\{[^{}]* => ([^{}]*)\}/g, "$1").replace(/\/\//g, "/");
+ const arrow = file.indexOf(" => ");
+ if (arrow >= 0) file = file.slice(arrow + 4);
+ out.set(file, {
+ binary: m[1] === "-",
+ insertions: m[1] === "-" ? 0 : Number(m[1]),
+ deletions: m[2] === "-" ? 0 : Number(m[2]),
+ });
+ }
+ return out;
+}
+
+/**
+ * Review rules per file, grouped so files sharing a checklist appear once.
+ *
+ * The checklist corpus is Open Code Review's, keyed on file type (go.md,
+ * java.md, package_json.md, …) and vendored here; selection.mjs resolves each
+ * path against it. `ref` is the side of the diff to read a file at when the
+ * rule depends on content (the ".m" MATLAB/Objective-C sniff), so a commit that
+ * is not checked out still resolves correctly.
+ */
+export function ruleGroups(files, cwd, range, rules = []) {
+ if (files.length === 0) return [];
+ const ref = range?.mode === "commit" ? range.commit : range?.mode === "range" ? range.to : "";
+ return groupRules(
+ files.map((f) => f.path),
+ { repoDir: cwd, ref, rules },
+ );
+}
+
+// ---------------------------------------------------------------- rubric ---
+
+const readRule = (name) => fs.readFileSync(path.join(PKG_ROOT, "rules", name), "utf8").trim();
+
+// The project's own conventions doc, wrapped in the same untrusted-data framing
+// the Action uses. Read from the work tree here (the local user owns their
+// checkout) rather than from the base revision (where CI reads it, because
+// there a PR author could otherwise rewrite the rules being applied to them).
+export function conventions(cwd) {
+ for (const name of ["AGENTS.md", "CLAUDE.md", "CONTRIBUTING.md"]) {
+ const file = path.join(cwd, name);
+ let doc;
+ try {
+ doc = fs.readFileSync(file, "utf8").trim();
+ } catch {
+ continue;
+ }
+ if (!doc) continue;
+ return {
+ file: name,
+ text: [
+ readRule("conventions-directive.md"),
+ "",
+ `----- BEGIN ${name} (untrusted project conventions; read-only reference) -----`,
+ doc,
+ `----- END ${name} -----`,
+ ].join("\n"),
+ };
+ }
+ return { file: null, text: "" };
+}
+
+export function rubric(cwd) {
+ return {
+ severity: readRule("severity-instruction.md"),
+ output_shape: readRule("output-shape.md"),
+ conventions: conventions(cwd),
+ };
+}
+
+// ------------------------------------------------------------------ plan ---
+
+export function buildPlan(opts, cwd) {
+ const range = resolveMode(opts, cwd);
+ if (range.mode === "error" || range.mode === "empty") return { range, files: [] };
+
+ // The repo's own settings. An invalid file is the caller's to report (see
+ // cmdReviewPlan) — here it counts as absent so buildPlan stays total.
+ const loaded = opts.config ?? loadLocalConfig(cwd);
+ const config = loaded.ok ? loaded.config : {};
+
+ const selection = selectFiles(range, cwd, { exclude: config.exclude || [] });
+ const mergeBase = selection.merge_base || mergeBaseOf(range, cwd);
+
+ return {
+ schema_version: SCHEMA_VERSION,
+ range,
+ mode: range.mode,
+ repository: cwd,
+ from: range.from || "",
+ to: range.to || "",
+ commit: range.commit || "",
+ merge_base: mergeBase,
+ pr: opts.prMeta || null,
+ background: opts.background || "",
+ // The language the findings are to be written in — the user's, as resolved
+ // by the CLI (locale, or --lang). The rubric stays English regardless: it is
+ // a contract shared with the Action, and tags like [P1] are tokens.
+ language: LANGUAGES.includes(opts.language) ? opts.language : "en",
+ config: loaded.ok && loaded.file ? { file: loaded.file, ...config, rules: (config.rules || []).map((r) => ({ path: r.path, replace: r.replace, source: r.source })) } : null,
+ selector: selection.selector,
+ files: selection.files,
+ excluded: selection.excluded,
+ rule_groups: ruleGroups(selection.files, cwd, range, config.rules || []),
+ diff_recipe: diffRecipe(range, mergeBase),
+ rubric: rubric(cwd),
+ result_path: path.join(cwd, WORK_DIR, "result.json"),
+ result_shape: RESULT_SHAPE,
+ };
+}
+
+/**
+ * The plan as the prompt an agent actually follows.
+ *
+ * Written in the imperative, addressed to the reviewer, because that is what it
+ * is: this text is pasted into a model's context. It is deliberately NOT
+ * translated — flags, rule text, and severity tags are verbatim tokens, and the
+ * rubric it embeds is English.
+ */
+export function renderPlan(plan) {
+ const L = [];
+ const p = (s = "") => L.push(s);
+
+ p("# OrcaCode Review — review request");
+ p();
+ p("You are the reviewer. Everything below is fixed; follow it exactly.");
+ p();
+ p("## Scope");
+ p();
+ p(`- mode: ${plan.mode} (${MODE_REASON[plan.range.code] || plan.range.code})`);
+ if (plan.pr) {
+ p(`- pull request: #${plan.pr.number} ${plan.pr.title}`.trimEnd());
+ if (plan.pr.url) p(`- pull request url: ${plan.pr.url}`);
+ // Said out loud because it changes how to read the diff: a fork PR's head is
+ // not in this repo's branches, and the base is where it is going, not where
+ // it came from.
+ if (plan.pr.fork) p(`- from a fork; head branch \`${plan.pr.head}\` is not a branch of this repository`);
+ }
+ if (plan.from) p(`- from: ${plan.from}`);
+ if (plan.to) p(`- to: ${plan.to}`);
+ if (plan.commit) p(`- commit: ${plan.commit}`);
+ if (plan.merge_base) p(`- merge_base: ${plan.merge_base}`);
+ // One line stays a bullet. Anything longer gets its own section — a PR body
+ // pasted after "- background:" swallows the rest of the list into itself and
+ // the reader (a model) loses where the scope ends.
+ if (plan.background && !plan.background.includes("\n")) p(`- background: ${plan.background}`);
+ p("- file selection: git, filtered by the bundled Open Code Review rules (exclusions and per-language checklists)");
+ if (plan.config) p(`- project settings: \`${plan.config.file}\`${plan.config.exclude?.length ? ` (${plan.config.exclude.length} extra exclude${plan.config.exclude.length === 1 ? "" : "s"})` : ""}${plan.config.rules?.length ? ` (${plan.config.rules.length} project rule${plan.config.rules.length === 1 ? "" : "s"})` : ""}`);
+ if (plan.background && plan.background.includes("\n")) {
+ p();
+ p("### Background — what this change is for");
+ p();
+ p("Judge the change against this intent, but do not treat it as true: a");
+ p("description that disagrees with the code is itself a finding.");
+ p();
+ for (const line of plan.background.split("\n")) p(`> ${line}`.trimEnd());
+ }
+ p();
+ p(`### Files to review (${plan.files.length})`);
+ p();
+ for (const f of plan.files) p(`- \`${f.path}\`${f.status ? ` [${f.status}]` : ""}`);
+ if (plan.excluded.length) {
+ p();
+ p(`### Excluded (${plan.excluded.length}) — do not review these`);
+ p();
+ for (const f of plan.excluded) p(`- \`${f.path}\` — ${f.reason}`);
+ }
+ p();
+ p("Review only the changed lines. See each file's change with:");
+ p();
+ p("```bash");
+ p(plan.diff_recipe);
+ p("```");
+ p();
+ p("Read the surrounding file when you need context — you have the repo, and a");
+ p("finding you could not confirm by reading the code is a finding to drop.");
+
+ if (plan.rule_groups.length) {
+ p();
+ p("## Review rules");
+ p();
+ p("Each group is the checklist for the files listed under it.");
+ for (const g of plan.rule_groups) {
+ p();
+ p(`### Group ${g.group_id}${g.pattern ? ` — \`${g.pattern}\`` : ""}`);
+ p();
+ for (const f of g.files || []) p(`- \`${f}\``);
+ p();
+ p(g.rule);
+ }
+ }
+
+ p();
+ p("## Language");
+ p();
+ p(`Write every finding — its title, its explanation, and its **Fix:** paragraph — in ${LANGUAGE_NAMES[plan.language] || plan.language}.`);
+ p("That is the language the user is working in; a finding they have to translate is a");
+ p("finding they will skim. The severity tag stays exactly `[P0]`…`[P3]`, the title stays");
+ p("bold, and the fix stays its own final paragraph — only the prose changes language.");
+ p("The **Fix:** label may be written in that language too. Identifiers, file paths, and");
+ p("code stay verbatim, whatever the language.");
+ p();
+ p("## Severity");
+ p();
+ p(plan.rubric.severity);
+ p();
+ p(plan.rubric.output_shape);
+
+ if (plan.rubric.conventions.text) {
+ p();
+ p("## Project conventions");
+ p();
+ p(plan.rubric.conventions.text);
+ }
+
+ p();
+ p("## How to hand back your findings");
+ p();
+ p(`Write JSON to \`${plan.result_path}\` in exactly this shape:`);
+ p();
+ p("```json");
+ p(plan.result_shape);
+ p("```");
+ p();
+ p("Those four fields are all of it. Write no others — anything extra is ignored,");
+ p("and a longer file is only a longer wall of JSON for the person watching.");
+ p();
+ p("- `path` — repo-relative, and it must be the file that actually contains the code you describe.");
+ p("- `line` — the line in the post-change file. For a finding that genuinely spans");
+ p(" several lines, write `start_line` and `end_line` instead of `line`.");
+ p("- `existing_code` — the source you are quoting, copied VERBATIM from the file.");
+ p(" This is load-bearing: it is grepped against the tree to verify the finding is");
+ p(" filed on the right file, and a finding whose snippet matches nowhere may be dropped.");
+ p(" One line is enough. Do not paste a whole function.");
+ p("- `content` — the severity tag, then the bold title, then the body, per the rubric above.");
+ p();
+ p("Found nothing? Write `{\"comments\": []}`. That is a clean review, not a failure.");
+ p();
+ p("There is one optional key, `warnings`: a list of files you could not review at all.");
+ p("A non-empty `warnings` means the review is PARTIAL and will be REJECTED rather than");
+ p("passed, so omit it entirely unless something genuinely failed.");
+ p();
+ p("Do not print the findings to the terminal; the file is the deliverable.");
+ p();
+ p("Then run:");
+ p();
+ p("```bash");
+ p(`npx @orcarouter/code-review review submit --format md --lang ${plan.language}`);
+ p("```");
+ p();
+ p("It validates the shape, verifies each finding's position, applies the merge gate,");
+ p("and prints the report. Show the user its output. Do not summarise the findings");
+ p("yourself before running it — the gate decides what is blocking, not you.");
+
+ if (!plan.config) {
+ // The onboarding moment, placed where it is acted on: AFTER the report, so
+ // the first review costs the user nothing extra, and they decide with the
+ // actual output in front of them rather than in the abstract. Stateless on
+ // purpose — "no file" is the whole definition of "first run here".
+ p();
+ p("## First review in this repository");
+ p();
+ p("There is no `.orcacode-review.json` here, so this run used the defaults: blocking on");
+ p(`P0 and P1, findings in ${LANGUAGE_NAMES[plan.language] || plan.language}, no extra exclusions.`);
+ p();
+ p("After you have shown the report — not before — offer ONCE, in one sentence, to");
+ p("save these as the repository's settings, and say what would change if they do:");
+ p("the next review here picks them up without flags, for them and for anyone else.");
+ p("If they say yes, run:");
+ p();
+ p("```bash");
+ p(`npx @orcarouter/code-review review config init --lang ${plan.language}`);
+ p("```");
+ p();
+ p("then apply anything they asked for on top (\"only block on P0\" → `--block-on P0` on");
+ p("that command, or edit the key afterwards) and show them the file. If they say no,");
+ p("do not bring it up again in this conversation. Do not create the file unasked.");
+ }
+
+ return L.join("\n");
+}
+
+// ---------------------------------------------------------------- submit ---
+
+/**
+ * Is this a result we can trust? Mirrors check-result.mjs, which is what the
+ * Action asks of the engine — the same answer must not depend on who reviewed.
+ *
+ * Fail-closed on anything partial: converting "some files could not be read"
+ * into a clean pass is how a broken review silently clears a change.
+ */
+export function validateResult(parsed) {
+ if (!parsed || typeof parsed !== "object") return { ok: false, error: "result is not a JSON object" };
+ if (!Array.isArray(parsed.comments)) return { ok: false, error: "result has no `comments` array" };
+
+ const warnings = Array.isArray(parsed.warnings) ? parsed.warnings : [];
+ if (warnings.length > 0) {
+ return { ok: false, error: `review is partial: ${warnings.join("; ")}` };
+ }
+
+ const bad = [];
+ parsed.comments.forEach((c, i) => {
+ if (!c || typeof c !== "object") return bad.push(`#${i} is not an object`);
+ if (typeof c.path !== "string" || !c.path) bad.push(`#${i} has no \`path\``);
+ if (typeof c.content !== "string" || !c.content) bad.push(`#${i} has no \`content\``);
+ });
+ if (bad.length) return { ok: false, error: `malformed findings: ${bad.join(", ")}` };
+
+ // A single `line` is what a model reaches for, but the pipeline — postfilter,
+ // the exhaustive merge, the judge, the Action's poster — is written against
+ // `start_line`/`end_line`. Widen it here, at the boundary, so exactly one
+ // shape flows downstream and a re-home can clear the anchor properly.
+ const comments = parsed.comments.map((c) => {
+ if (c.start_line === undefined && c.end_line === undefined && Number(c.line) >= 1) {
+ const { line, ...rest } = c;
+ return { ...rest, start_line: Number(line), end_line: Number(line) };
+ }
+ return c;
+ });
+
+ return { ok: true, comments };
+}
+
+/**
+ * The L1 position check, run by shelling out to the SAME postfilter.mjs the
+ * Action runs. Not reimplemented: two copies of "is this finding filed on the
+ * right file" would drift, and the whole point is that a local P1 and a CI P1
+ * mean the same thing.
+ *
+ * Skipped in workspace mode — it greps a commit, and uncommitted code is in no
+ * commit. Fail-soft everywhere else: a postfilter that errors leaves the
+ * findings untouched rather than losing them.
+ */
+export function positionCheck(resultFile, cwd, ref) {
+ if (!ref) return { ran: false, code: "no-commit" };
+
+ const script = path.join(PKG_ROOT, "scripts", "postfilter.mjs");
+ const out = `${resultFile.replace(/\.json$/, "")}.l1.json`;
+ const r = spawnSync(process.execPath, [script, resultFile, cwd, ref, "--out", out], {
+ encoding: "utf8",
+ maxBuffer: 1 << 26,
+ });
+ if (r.status !== 0 || !fs.existsSync(out)) {
+ return { ran: false, code: "failed", detail: (r.stderr || "").trim().split("\n").pop() || "" };
+ }
+ return { ran: true, out, log: (r.stderr || "").trim() };
+}
+
+export function gate(comments, blockOn) {
+ const wanted = new Set(
+ String(blockOn || "")
+ .split(",")
+ .map((s) => s.trim().toUpperCase())
+ .filter(Boolean),
+ );
+ const blocking = comments.filter((c) => wanted.has(severityOf(c)));
+ return { counts: countSeverities(comments), blocking, blocked: blocking.length > 0, wanted: [...wanted] };
+}
+
+// -------------------------------------------------------------- rendering ---
+
+// Wraps at a fixed width rather than the terminal's: a report that reflows
+// differently per window cannot be diffed or pasted into an issue intact.
+const WIDTH = 78;
+
+function wrap(text, indent) {
+ const out = [];
+ for (const para of String(text).split("\n")) {
+ if (!para.trim()) {
+ out.push("");
+ continue;
+ }
+ let line = indent;
+ for (const word of para.split(/\s+/)) {
+ if (line.length > indent.length && line.length + 1 + word.length > WIDTH) {
+ out.push(line);
+ line = indent;
+ }
+ line += (line.length > indent.length ? " " : "") + word;
+ }
+ out.push(line);
+ }
+ return out.join("\n");
+}
+
+// The bold title the output shape mandates, split off so the report can lead
+// with it. Falls back to the whole body when a finding ignored the shape —
+// degraded, but never lossy.
+export function splitFinding(content) {
+ const body = String(content).replace(/^\s*\[P[0-3]\]\s*/i, "");
+ const m = body.match(/^\*\*(.+?)\*\*\s*\n?([\s\S]*)$/);
+ if (!m) return { title: body.split("\n")[0].trim(), rest: body.split("\n").slice(1).join("\n").trim() };
+ return { title: m[1].trim(), rest: m[2].trim() };
+}
+
+export function renderReport(comments, result, { color, t = englishT }) {
+ const c = color;
+ const mark = { P0: c.red("●"), P1: c.yellow("●"), P2: c.cyan("○"), P3: c.dim("○") };
+ const L = [];
+
+ // The verdict goes FIRST, before a single finding.
+ //
+ // A blocked review exits 1, and every agent shell renders a non-zero exit as
+ // "Error: Exit code 1" and then TRUNCATES the middle of the output. With the
+ // verdict at the bottom — where it used to be — what survives is an error
+ // banner wrapped around some findings, and the reader cannot tell a working
+ // merge gate from a crashed command. Whatever gets cut, the first line has to
+ // say what happened.
+ const list = result.wanted.join(",") || "—";
+ L.push(
+ result.blocked
+ ? ` ${c.red(c.bold(t("report.verdictBlocked")))} ${t("report.blockedTail", result.blocking.length, comments.length, list)}`
+ : ` ${c.green(c.bold(t("report.verdictPassed")))} ${t("report.passedTail", list, comments.length)}`,
+ );
+ L.push(
+ ` ${SEVERITIES.map((s) => `${s} ${result.counts[s]}`).join(c.dim(" · "))}` +
+ c.dim(` ${c.bold(t("report.notCrash"))}`),
+ );
+
+ for (const sev of SEVERITIES) {
+ for (const finding of comments.filter((x) => severityOf(x) === sev)) {
+ const { title, rest } = splitFinding(finding.content);
+ const at = anchorLine(finding);
+ const where = `${finding.path}${at ? `:${at}` : ""}`;
+ L.push("");
+ L.push(` ${mark[sev]} ${c.bold(sev)} ${c.cyan(where)}`);
+ L.push(` ${c.bold(title)}`);
+ if (rest) {
+ L.push("");
+ L.push(wrap(rest, " "));
+ }
+ }
+ }
+
+ return L.join("\n");
+}
+
+/**
+ * The same report as markdown, for an agent to paste into its conversation.
+ *
+ * The terminal renderer above is ANSI and hard-wrapped at 78 columns — pasted
+ * into a chat it arrives as a wall of pre-formatted text with escape codes.
+ * This one is the same data with markdown structure, so the host renders it as
+ * headings and prose. No colour, no wrapping: the reader reflows it.
+ *
+ * Grouped by file, because a reviewer fixes one file at a time, and the file
+ * with the blocking finding sorts first so the thing that stops the merge is
+ * not below the fold.
+ */
+export function renderReportMarkdown(comments, result, { t = englishT } = {}) {
+ const blocking = new Set(result.blocking);
+ const rank = (c) => SEVERITIES.indexOf(severityOf(c));
+
+ const byFile = new Map();
+ for (const finding of comments) {
+ if (!byFile.has(finding.path)) byFile.set(finding.path, []);
+ byFile.get(finding.path).push(finding);
+ }
+ const files = [...byFile.entries()].sort(
+ (a, b) => Math.min(...a[1].map(rank)) - Math.min(...b[1].map(rank)) || a[0].localeCompare(b[0]),
+ );
+
+ const L = [];
+ const p = (s = "") => L.push(s);
+
+ p("## OrcaCode Review");
+ p();
+ const list = result.wanted.join(", ") || "—";
+ // The full stop is a translated string: "。" in Chinese and Japanese, "." in
+ // English and Korean. Hard-coding "." put a Latin period on a Chinese line.
+ const stop = t("report.stop");
+ if (comments.length === 0) {
+ p(`**${t("report.verdictPassed")}** — ${t("report.clean")}${stop}`);
+ return L.join("\n");
+ }
+ if (result.blocked) {
+ p(`**${t("report.verdictBlocked")}** — ${t("report.blockedTail", result.blocking.length, comments.length, list)}${stop}`);
+ } else {
+ p(`**${t("report.verdictPassed")}** — ${t("report.passedTail", list, comments.length)}${stop}`);
+ }
+ p();
+ p(SEVERITIES.map((s) => `\`${s} ${result.counts[s]}\``).join(" · "));
+
+ for (const [file, found] of files) {
+ p();
+ p(`### \`${file}\``);
+ for (const finding of found.sort((a, b) => rank(a) - rank(b) || (anchorLine(a) || 0) - (anchorLine(b) || 0))) {
+ const { title, rest } = splitFinding(finding.content);
+ const at = anchorLine(finding);
+ // `path:line` unadorned — hosts linkify that shape, and a markdown link
+ // to a relative path does not resolve in most of them.
+ const where = `${finding.path}${at ? `:${at}` : ""}`;
+ p();
+ // Blocking is judged against the gate that actually ran, not the severity
+ // in the abstract — under `--block-on P0` a P1 is reported, not blocking.
+ p(`**${blocking.has(finding) ? "❌" : "💬"} ${severityOf(finding)} · \`${where}\` — ${title}**`);
+ if (rest) {
+ p();
+ p(rest);
+ }
+ }
+ }
+ return L.join("\n");
+}
diff --git a/bin/i18n.mjs b/bin/i18n.mjs
index 6bd9de8..0e3f0d2 100644
--- a/bin/i18n.mjs
+++ b/bin/i18n.mjs
@@ -177,7 +177,7 @@ const EN = {
clean: "No problems found.",
problems: (n) => `${n} problem(s) found — see the fixes above.`,
problemsHint:
- " Full symptom → cause → fix table: skills/setup-orca-code-review/references/troubleshooting.md",
+ " Full symptom → cause → fix table: skills/orca-review-action/references/troubleshooting.md",
},
uninstall: {
@@ -198,6 +198,14 @@ const EN = {
skill: {
missingBundle: "The bundled skill is missing from this package.",
missingBundleHint: "Reinstall with: npx @orcarouter/code-review@latest skill",
+ modeQ: "How will you use OrcaCode Review?",
+ modeBoth: "Both",
+ modeBothDetail: "Local review now, and the GitHub Action on every PR. The two agree on what blocks.",
+ modeLocal: "Local review only",
+ modeLocalDetail: "Your own agent reviews here — no Action, no OrcaRouter account, no API key.",
+ modeAction: "GitHub Action only",
+ modeActionDetail: "Set up, retune, troubleshoot, or remove the CI review on a repository.",
+ unknownMode: (m) => `Unknown mode "${m}".`,
scopeQ: "Where should the skill be installed?",
scopeProject: "This project",
scopeProjectDetail: "Committed with the repo — everyone who clones it gets the skill.",
@@ -207,12 +215,14 @@ const EN = {
unknownScopeHint: "Use --scope project or --scope global.",
unknownPlatform: (id) => `Unknown platform "${id}".`,
unknownPlatformHint: "List them with: npx @orcarouter/code-review skill list",
+ unknownSkill: (n) => `Unknown skill "${n}".`,
noneDetected: "No agent platform detected here, and none was named.",
noneDetectedHint:
"Pass one explicitly, e.g. --platform claude --platform codex (`skill list` shows all 36).",
platformQ: "Which agents should get the skill?",
platformCount: (n) => ` (${n} detected)`,
statusUpdated: "(updated)",
+ statusRetired: (old) => `(replaced the old ${old})`,
statusUnchanged: "(already current)",
statusConflict: "(left alone — a different version is already there)",
forceHint: "Re-run with --force to overwrite the differing copies.",
@@ -220,6 +230,7 @@ const EN = {
pluginHint: " For Claude Code, the plugin keeps itself updated instead:",
handoffTitle: "Now just ask your agent",
handoffPrimary: "set up OrcaCode Review in this repo",
+ handoffReview: "review my changes with OrcaCode Review",
handoffMore: "It can also:",
handoffDoctor: '"why isn\'t OrcaCode Review running?"',
handoffDoctorWhat: "diagnose",
@@ -244,6 +255,65 @@ const EN = {
manualPath: "your repo → Settings → Secrets and variables → Actions → New repository secret",
},
+ report: {
+ verdictBlocked: "❌ BLOCKED",
+ verdictPassed: "✅ PASSED",
+ blockedTail: (n, total, list) => `${n} of ${total} finding${total === 1 ? "" : "s"} at ${list}`,
+ passedTail: (list, total) => `nothing at ${list} — ${total} finding${total === 1 ? "" : "s"} to read`,
+ stop: ".",
+ notCrash: "this is a review result, not a crash",
+ clean: "no findings",
+ },
+ review: {
+ cfgLoaded: (file) => `Using ${file}.`,
+ cfgInvalid: (file, detail) => `${file} is invalid: ${detail}`,
+ cfgNone: (file) => `No ${file} — every setting is at its default.`,
+ cfgWritten: (file) => `Wrote ${file}.`,
+ cfgExists: (file) => `${file} already exists; pass --force to overwrite it.`,
+ cfgEditHint: (file) => `To change a setting, edit ${file} directly — or ask your AI to. A key you leave out keeps its default.`,
+ cfgCreateHint: "To create one: npx @orcarouter/code-review review config init",
+ cfgComment: "OrcaCode Review local settings. block_on: severities that count as blocking (\"P0,P1\", or \"\" for none). language: en | zh | ja | ko. exclude: extra globs never to review. rules: [{ path, rule | rule_file, replace? }] extra checklists per file pattern.",
+ cfgBlockOnFrom: (list, source) => `Blocking on ${list || "nothing"} — from ${source}.`,
+ cfgSrcFlag: "the command line",
+ cfgSrcFile: ".orcacode-review.json",
+ cfgSrcDefault: "the default",
+ cfgSrcLocale: "the system locale",
+ cfgRowExclude: (n) => `${n} extra exclude${n === 1 ? "" : "s"}`,
+ cfgRowRules: (n) => `${n} project rule${n === 1 ? "" : "s"}`,
+ notRepo: "Not a git repository — `review` needs one to diff against.",
+ cannotResolve: "Cannot work out what to review.",
+ noBaseHint: "Name the range yourself: --from main --to HEAD, or use --worktree.",
+ nothingToReview: (base) => `Nothing to review — HEAD has no commits beyond ${base} and the work tree is clean.`,
+ noFiles: (mode) => `No reviewable files in ${mode} mode.`,
+ planned: (n, mode) => `Planned ${n} file(s) — ${mode} mode.`,
+ autoWorktree: "Reviewing uncommitted work, because the work tree is dirty.",
+ autoRange: (base) => `Reviewing this branch against ${base}, the same range CI would use.`,
+ conventions: (f) => `Project conventions from ${f} are part of the request.`,
+ missingResult: (p) => `No result at ${p} — the reviewer has not written one.`,
+ unusable: (m) => `Unusable result, failing closed: ${m}`,
+ positionsChecked: (a, b) => `Positions verified against the tree: ${a} → ${b} finding(s).`,
+ positionsNoCommit: "Position check skipped — uncommitted code is in no commit to grep.",
+ positionsFailed: (d) => `Position check skipped, keeping every finding${d ? ` — ${d}` : ""}.`,
+ prConflict: "--pr already fixes the range; drop --from/--to/--commit/--worktree.",
+ prFetching: (n) => `Resolving pull request #${n}…`,
+ prNoGh: "--pr needs the GitHub CLI (`gh`), which is not installed.",
+ prNoAuth: "gh is installed but not signed in. Run: gh auth login",
+ prBadNumber: (v) => `Not a pull request number: ${v}`,
+ prFetchFailed: (d) => `Could not fetch the pull request${d ? ` — ${d}` : ""}.`,
+ prNoBaseRef: (b) => `The pull request's base branch (${b}) is not available locally.`,
+ prFailed: (d) => `gh could not read that pull request${d ? ` — ${d}` : ""}.`,
+ prHint: "Or check the branch out yourself (gh pr checkout ) and review without --pr.",
+ prResolved: (n, base) => `Reviewing pull request #${n} against ${base}. Your work tree is untouched.`,
+ prState: (s) => `Heads up: this pull request is ${s}, not open.`,
+ prFork: "The pull request comes from a fork; its head was fetched, not checked out.",
+ prBackground: "The pull request's title and description were passed to the reviewer as background.",
+ badFormat: (got, ok) => `Unknown --format ${got}. Use one of: ${ok}`,
+ reportSaved: (p) => `A pasteable markdown report was written to ${p}.`,
+ clean: "Clean — no findings.",
+ blocked: (n, list) => `Blocked: ${n} finding(s) at ${list}.`,
+ passed: (list) => `Passed — nothing at ${list}.`,
+ },
+
usage: {
tagline: "installer for OrcaCode Review (AI PR review by OrcaRouter)",
usage: "Usage",
@@ -256,8 +326,23 @@ const EN = {
cmdReconfigure: "Change the inputs in an existing workflow",
cmdDoctor: "Diagnose an install that is not working",
cmdUninstall: "Remove the workflow",
- cmdSkillInstall: (n) => `Install the agent skill (${n} platforms) — the default`,
+ cmdSkillInstall: (n) => `Install the agent skills (${n} platforms) — the default`,
cmdSkillList: "List the supported platforms and which are detected here",
+ cmdReviewPlan: "Print a review request for an agent to carry out",
+ cmdReviewSubmit: "Check, gate, and report an agent's findings",
+ cmdReviewConfig: "Show or create this repo's local review settings (.orcacode-review.json)",
+ reviewOptions: "Review options",
+ optSkill: "For `skill`: install just one (default: all)",
+ optMode: "What to install: both (default) | local | action",
+ optFrom: "Base ref of the range to review",
+ optTo: "Head ref of the range to review",
+ optPr: "Review a pull request by number (needs gh; does not check it out)",
+ optFormat: "`review submit` output: text (default) | md | json",
+ optCommit: "Review a single commit",
+ optWorktree: "Review uncommitted work instead of a range",
+ optBackground: "Business context to hand the reviewer",
+ optBlockOn: "Severities the report marks ❌ blocking (default: P0,P1)",
+ optFailOnBlock: "Also exit 1 when something blocks — for hooks and scripts; the default is 0",
optYes: "Accept the recommended defaults; never prompt",
optForce: "Overwrite without asking",
optJson: "Machine-readable output (skill install / skill list)",
@@ -401,7 +486,7 @@ const ZH = {
noProtection: (b) => `读不到 ${b} 的分支保护 —— 合并没有门禁。`,
clean: "没有发现问题。",
problems: (n) => `发现 ${n} 个问题 —— 修复方式见上。`,
- problemsHint: " 完整的「现象 → 原因 → 修复」表:skills/setup-orca-code-review/references/troubleshooting.md",
+ problemsHint: " 完整的「现象 → 原因 → 修复」表:skills/orca-review-action/references/troubleshooting.md",
},
uninstall: {
@@ -422,6 +507,14 @@ const ZH = {
skill: {
missingBundle: "这个包里缺少内置的 skill。",
missingBundleHint: "重新安装:npx @orcarouter/code-review@latest skill",
+ modeQ: "你打算怎么用 OrcaCode Review?",
+ modeBoth: "两者都装",
+ modeBothDetail: "本地随时评审,GitHub Action 在每个 PR 上评审。两边对「什么算阻塞」的判断一致。",
+ modeLocal: "只用本地模式",
+ modeLocalDetail: "由你自己的 AI 在本地评审 —— 不需要 Action、不需要 OrcaRouter 账号、不需要 API key。",
+ modeAction: "只用 Action 模式",
+ modeActionDetail: "在仓库里接入、调整、排查或移除 CI 评审。",
+ unknownMode: (m) => `未知的模式「${m}」。`,
scopeQ: "skill 装到哪里?",
scopeProject: "当前项目",
scopeProjectDetail: "随仓库提交 —— 每个 clone 的人都能用。",
@@ -431,11 +524,13 @@ const ZH = {
unknownScopeHint: "用 --scope project 或 --scope global。",
unknownPlatform: (id) => `未知平台「${id}」。`,
unknownPlatformHint: "用 npx @orcarouter/code-review skill list 查看全部。",
+ unknownSkill: (n) => `未知的 skill「${n}」。`,
noneDetected: "这里没检测到任何 Agent 平台,也没有指定平台。",
noneDetectedHint: "请显式指定,例如 --platform claude --platform codex(`skill list` 列出全部 36 个)。",
platformQ: "要把 skill 装到哪些 Agent?",
platformCount: (n) => `(检测到 ${n} 个)`,
statusUpdated: "(已更新)",
+ statusRetired: (old) => `(已替换旧名 ${old})`,
statusUnchanged: "(已是最新)",
statusConflict: "(已跳过 —— 那里存在一份内容不同的副本)",
forceHint: "加 --force 重新运行可覆盖内容不同的副本。",
@@ -443,6 +538,7 @@ const ZH = {
pluginHint: " Claude Code 建议改用插件,它会自动保持更新:",
handoffTitle: "接下来直接对你的 AI 说",
handoffPrimary: "帮我把这个仓库配置上 OrcaCode Review",
+ handoffReview: "用 OrcaCode Review 评审我这次的改动",
handoffMore: "它还能:",
handoffDoctor: "「OrcaCode Review 怎么没跑?」",
handoffDoctorWhat: "排查",
@@ -467,6 +563,65 @@ const ZH = {
manualPath: "你的仓库 → Settings → Secrets and variables → Actions → New repository secret",
},
+ report: {
+ verdictBlocked: "❌ 已拦截",
+ verdictPassed: "✅ 通过",
+ blockedTail: (n, total, list) => `共 ${total} 条发现,其中 ${n} 条达到 ${list} 级别`,
+ passedTail: (list, total) => `没有 ${list} 级别的问题 —— 共 ${total} 条发现可看`,
+ stop: "。",
+ notCrash: "这是评审结果,不是程序崩溃",
+ clean: "没有发现问题",
+ },
+ review: {
+ cfgLoaded: (file) => `已读取 ${file}。`,
+ cfgInvalid: (file, detail) => `${file} 无效:${detail}`,
+ cfgNone: (file) => `没有 ${file} —— 全部使用默认值。`,
+ cfgWritten: (file) => `已写入 ${file}。`,
+ cfgExists: (file) => `${file} 已存在;要覆盖请加 --force。`,
+ cfgEditHint: (file) => `要改设置,直接编辑 ${file} —— 或者让你的 AI 改。没写的键保持默认。`,
+ cfgCreateHint: "创建:npx @orcarouter/code-review review config init",
+ cfgComment: "OrcaCode Review 本地评审设置。block_on:哪些级别算阻塞(\"P0,P1\",\"\" 表示都不阻塞)。language:en | zh | ja | ko。exclude:额外不审的 glob。rules:[{ path, rule 或 rule_file, replace? }] 按文件模式追加的检查清单。",
+ cfgBlockOnFrom: (list, source) => `阻塞级别 ${list || "无"} —— 来自${source}。`,
+ cfgSrcFlag: "命令行参数",
+ cfgSrcFile: ".orcacode-review.json",
+ cfgSrcDefault: "默认值",
+ cfgSrcLocale: "系统 locale",
+ cfgRowExclude: (n) => `${n} 条额外排除`,
+ cfgRowRules: (n) => `${n} 条项目规则`,
+ notRepo: "这里不是 git 仓库 —— `review` 需要仓库才能算出改动。",
+ cannotResolve: "无法判断要评审什么。",
+ noBaseHint: "自己指定范围:--from main --to HEAD,或者用 --worktree。",
+ nothingToReview: (base) => `没有可评审的内容 —— HEAD 相对 ${base} 没有新提交,工作区也是干净的。`,
+ noFiles: (mode) => `${mode} 模式下没有可评审的文件。`,
+ planned: (n, mode) => `已规划 ${n} 个文件 —— ${mode} 模式。`,
+ autoWorktree: "工作区有未提交的改动,所以评审的是未提交的内容。",
+ autoRange: (base) => `按 ${base} 评审当前分支 —— 和 CI 用的是同一个范围。`,
+ conventions: (f) => `已把项目自有约定(${f})放进评审请求。`,
+ missingResult: (p) => `${p} 没有结果 —— 评审方还没写。`,
+ unusable: (m) => `结果不可用,按失败处理:${m}`,
+ positionsChecked: (a, b) => `已对着代码树核验位置:${a} → ${b} 条。`,
+ positionsNoCommit: "跳过位置核验 —— 未提交的代码不在任何 commit 里,无从 grep。",
+ positionsFailed: (d) => `跳过位置核验,保留全部结论${d ? ` —— ${d}` : ""}。`,
+ prConflict: "--pr 已经确定了评审范围,请去掉 --from/--to/--commit/--worktree。",
+ prFetching: (n) => `正在解析 PR #${n}……`,
+ prNoGh: "--pr 需要 GitHub CLI(`gh`),当前未安装。",
+ prNoAuth: "gh 已安装但未登录。请运行:gh auth login",
+ prBadNumber: (v) => `这不是一个 PR 编号:${v}`,
+ prFetchFailed: (d) => `拉取该 PR 失败${d ? ` —— ${d}` : ""}。`,
+ prNoBaseRef: (b) => `本地找不到该 PR 的目标分支(${b})。`,
+ prFailed: (d) => `gh 读取该 PR 失败${d ? ` —— ${d}` : ""}。`,
+ prHint: "或者自己切过去(gh pr checkout <编号>),然后不带 --pr 评审。",
+ prResolved: (n, base) => `评审 PR #${n},对比 ${base}。你的工作区没有被改动。`,
+ prState: (s) => `注意:这个 PR 的状态是 ${s},不是打开状态。`,
+ prFork: "该 PR 来自 fork;它的 head 是拉取下来的,没有切换分支。",
+ prBackground: "该 PR 的标题和描述已作为业务背景交给评审者。",
+ badFormat: (got, ok) => `未知的 --format ${got}。可选:${ok}`,
+ reportSaved: (p) => `可直接粘贴的 markdown 报告已写入 ${p}。`,
+ clean: "干净 —— 没有发现问题。",
+ blocked: (n, list) => `已拦截:${list} 级别共 ${n} 条。`,
+ passed: (list) => `通过 —— 没有 ${list} 级别的问题。`,
+ },
+
usage: {
tagline: "OrcaCode Review 安装器(由 OrcaRouter 驱动的 AI PR 评审)",
usage: "用法",
@@ -481,6 +636,21 @@ const ZH = {
cmdUninstall: "移除 workflow",
cmdSkillInstall: (n) => `安装 Agent Skill(${n} 个平台)—— 默认命令`,
cmdSkillList: "列出支持的平台,并标出本机检测到的",
+ cmdReviewPlan: "输出一份评审请求,交给 AI 去执行",
+ cmdReviewSubmit: "校验 AI 交回的结论,过门禁并出报告",
+ cmdReviewConfig: "查看或创建本仓库的本地评审设置(.orcacode-review.json)",
+ reviewOptions: "评审参数",
+ optSkill: "对 `skill`:只装其中一个(默认全装)",
+ optMode: "装哪部分:both(默认)| local | action",
+ optFrom: "评审范围的起点 ref",
+ optTo: "评审范围的终点 ref",
+ optPr: "按编号评审一个 PR(需要 gh;不会切换分支)",
+ optFormat: "`review submit` 的输出格式:text(默认)| md | json",
+ optCommit: "只评审单个 commit",
+ optWorktree: "评审未提交的改动,而不是某个范围",
+ optBackground: "交给评审方的业务背景",
+ optBlockOn: "报告里标为 ❌ 阻塞的严重度(默认 P0,P1)",
+ optFailOnBlock: "有阻塞时额外以 1 退出 —— 给 hook 和脚本用;默认是 0",
optYes: "使用推荐默认值,不再询问",
optForce: "直接覆盖,不询问",
optJson: "机器可读输出(skill install / skill list)",
@@ -629,7 +799,7 @@ const JA = {
clean: "問題は見つかりませんでした。",
problems: (n) => `${n} 件の問題 — 上の修正方法を参照してください。`,
problemsHint:
- " 症状 → 原因 → 対処の一覧: skills/setup-orca-code-review/references/troubleshooting.md",
+ " 症状 → 原因 → 対処の一覧: skills/orca-review-action/references/troubleshooting.md",
},
uninstall: {
@@ -650,6 +820,14 @@ const JA = {
skill: {
missingBundle: "このパッケージに同梱の skill が見つかりません。",
missingBundleHint: "再インストール: npx @orcarouter/code-review@latest skill",
+ modeQ: "OrcaCode Review をどのように使いますか?",
+ modeBoth: "両方",
+ modeBothDetail: "ローカルでいつでもレビュー、GitHub Action は各 PR でレビュー。何をブロックするかは両者で一致します。",
+ modeLocal: "ローカルレビューのみ",
+ modeLocalDetail: "自分のエージェントがここでレビュー —— Action も OrcaRouter アカウントも API キーも不要。",
+ modeAction: "GitHub Action のみ",
+ modeActionDetail: "リポジトリの CI レビューを導入・調整・診断・削除します。",
+ unknownMode: (m) => `不明なモード「${m}」。`,
scopeQ: "skill をどこにインストールしますか?",
scopeProject: "このプロジェクト",
scopeProjectDetail: "リポジトリと一緒にコミットされ、clone した全員が使えます。",
@@ -659,12 +837,14 @@ const JA = {
unknownScopeHint: "--scope project または --scope global を指定してください。",
unknownPlatform: (id) => `不明なプラットフォーム「${id}」。`,
unknownPlatformHint: "一覧: npx @orcarouter/code-review skill list",
+ unknownSkill: (n) => `不明な skill「${n}」。`,
noneDetected: "エージェントを検出できず、指定もありません。",
noneDetectedHint:
"明示してください。例: --platform claude --platform codex(`skill list` で 36 個すべて表示)。",
platformQ: "どのエージェントに skill を入れますか?",
platformCount: (n) => `(${n} 個検出)`,
statusUpdated: "(更新しました)",
+ statusRetired: (old) => `(旧名 ${old} を置き換えました)`,
statusUnchanged: "(既に最新)",
statusConflict: "(スキップ — 内容の異なる版が既にあります)",
forceHint: "内容が異なる版を上書きするには --force を付けて再実行してください。",
@@ -672,6 +852,7 @@ const JA = {
pluginHint: " Claude Code はプラグインの方が自動で更新されます:",
handoffTitle: "あとはエージェントに頼むだけ",
handoffPrimary: "このリポジトリに OrcaCode Review を設定して",
+ handoffReview: "OrcaCode Review で今回の変更をレビューして",
handoffMore: "他にもできること:",
handoffDoctor: "「OrcaCode Review が動かないのはなぜ?」",
handoffDoctorWhat: "診断",
@@ -696,6 +877,65 @@ const JA = {
manualPath: "リポジトリ → Settings → Secrets and variables → Actions → New repository secret",
},
+ report: {
+ verdictBlocked: "❌ ブロック",
+ verdictPassed: "✅ パス",
+ blockedTail: (n, total, list) => `${total} 件の指摘のうち ${n} 件が ${list} に該当`,
+ passedTail: (list, total) => `${list} に該当する指摘なし —— 指摘は ${total} 件`,
+ stop: "。",
+ notCrash: "これはレビュー結果であり、クラッシュではありません",
+ clean: "指摘はありません",
+ },
+ review: {
+ cfgLoaded: (file) => `${file} を読み込みました。`,
+ cfgInvalid: (file, detail) => `${file} が不正です: ${detail}`,
+ cfgNone: (file) => `${file} がありません —— すべて既定値です。`,
+ cfgWritten: (file) => `${file} を書き出しました。`,
+ cfgExists: (file) => `${file} は既に存在します。上書きするには --force を付けてください。`,
+ cfgEditHint: (file) => `設定を変えるには ${file} を直接編集するか、AI に頼んでください。書かれていないキーは既定値のままです。`,
+ cfgCreateHint: "作成: npx @orcarouter/code-review review config init",
+ cfgComment: "OrcaCode Review ローカルレビュー設定。block_on: ブロック扱いにする深刻度(\"P0,P1\"、\"\" でブロックなし)。language: en | zh | ja | ko。exclude: レビューしない追加 glob。rules: [{ path, rule または rule_file, replace? }] ファイルパターンごとの追加チェックリスト。",
+ cfgBlockOnFrom: (list, source) => `ブロック対象 ${list || "なし"} —— ${source}より。`,
+ cfgSrcFlag: "コマンドライン引数",
+ cfgSrcFile: ".orcacode-review.json",
+ cfgSrcDefault: "既定値",
+ cfgSrcLocale: "システムロケール",
+ cfgRowExclude: (n) => `追加の除外 ${n} 件`,
+ cfgRowRules: (n) => `プロジェクトルール ${n} 件`,
+ notRepo: "git リポジトリではありません —— `review` には差分を取る対象が必要です。",
+ cannotResolve: "何をレビューすべきか判断できません。",
+ noBaseHint: "範囲を明示してください: --from main --to HEAD、または --worktree。",
+ nothingToReview: (base) => `レビュー対象がありません —— HEAD に ${base} を超えるコミットがなく、作業ツリーもクリーンです。`,
+ noFiles: (mode) => `${mode} モードにレビュー可能なファイルがありません。`,
+ planned: (n, mode) => `${n} ファイルを計画しました —— ${mode} モード。`,
+ autoWorktree: "作業ツリーに未コミットの変更があるため、それをレビューします。",
+ autoRange: (base) => `${base} を基準にこのブランチをレビューします —— CI と同じ範囲です。`,
+ conventions: (f) => `プロジェクト独自の規約(${f})をリクエストに含めました。`,
+ missingResult: (p) => `${p} に結果がありません —— レビュー側がまだ書いていません。`,
+ unusable: (m) => `結果が使えないため失敗として扱います: ${m}`,
+ positionsChecked: (a, b) => `ツリーと突き合わせて位置を検証: ${a} → ${b} 件。`,
+ positionsNoCommit: "位置検証をスキップ —— 未コミットのコードはどのコミットにもなく grep できません。",
+ positionsFailed: (d) => `位置検証をスキップし、指摘はすべて保持します${d ? ` —— ${d}` : ""}。`,
+ prConflict: "--pr で範囲は確定します。--from/--to/--commit/--worktree は外してください。",
+ prFetching: (n) => `プルリクエスト #${n} を解決しています…`,
+ prNoGh: "--pr には GitHub CLI(`gh`)が必要ですが、インストールされていません。",
+ prNoAuth: "gh はありますがログインしていません。実行してください:gh auth login",
+ prBadNumber: (v) => `プルリクエスト番号ではありません:${v}`,
+ prFetchFailed: (d) => `プルリクエストを取得できませんでした${d ? ` —— ${d}` : ""}。`,
+ prNoBaseRef: (b) => `プルリクエストのベースブランチ(${b})がローカルにありません。`,
+ prFailed: (d) => `gh がそのプルリクエストを読み取れませんでした${d ? ` —— ${d}` : ""}。`,
+ prHint: "または自分でチェックアウトし(gh pr checkout <番号>)、--pr なしでレビューしてください。",
+ prResolved: (n, base) => `プルリクエスト #${n} を ${base} と比較してレビューします。作業ツリーは変更されません。`,
+ prState: (s) => `注意:このプルリクエストの状態は ${s} で、オープンではありません。`,
+ prFork: "フォークからのプルリクエストです。head は取得しただけで、チェックアウトはしていません。",
+ prBackground: "プルリクエストのタイトルと説明を業務背景としてレビュアーに渡しました。",
+ badFormat: (got, ok) => `不明な --format ${got}。次のいずれかを指定してください:${ok}`,
+ reportSaved: (p) => `貼り付け可能な markdown レポートを ${p} に書き出しました。`,
+ clean: "クリーン —— 指摘はありません。",
+ blocked: (n, list) => `ブロック: ${list} の指摘が ${n} 件。`,
+ passed: (list) => `パス —— ${list} に該当する指摘はありません。`,
+ },
+
usage: {
tagline: "OrcaCode Review インストーラー(OrcaRouter による AI PR レビュー)",
usage: "使い方",
@@ -710,6 +950,21 @@ const JA = {
cmdUninstall: "workflow を削除する",
cmdSkillInstall: (n) => `エージェント skill を導入(${n} プラットフォーム)— 既定`,
cmdSkillList: "対応プラットフォームと検出状況を一覧表示",
+ cmdReviewPlan: "エージェントが実行するレビュー依頼を出力",
+ cmdReviewSubmit: "エージェントの指摘を検証し、ゲートを適用して報告",
+ cmdReviewConfig: "このリポジトリのローカルレビュー設定を表示または作成(.orcacode-review.json)",
+ reviewOptions: "レビューのオプション",
+ optSkill: "`skill` 用: 1 つだけ入れる(既定は全部)",
+ optMode: "何をインストールするか: both(既定)| local | action",
+ optFrom: "レビュー範囲の起点 ref",
+ optTo: "レビュー範囲の終点 ref",
+ optPr: "番号でプルリクエストをレビュー(gh が必要。チェックアウトはしない)",
+ optFormat: "`review submit` の出力形式:text(既定)| md | json",
+ optCommit: "単一コミットをレビュー",
+ optWorktree: "範囲ではなく未コミットの変更をレビュー",
+ optBackground: "レビュー側に渡す業務背景",
+ optBlockOn: "レポートで ❌ ブロックと扱う深刻度(既定: P0,P1)",
+ optFailOnBlock: "ブロックがあれば 1 で終了 —— フックやスクリプト向け。既定は 0",
optYes: "推奨値を使い、一切確認しない",
optForce: "確認せず上書きする",
optJson: "機械可読な出力(skill install / skill list)",
@@ -857,7 +1112,7 @@ const KO = {
clean: "문제를 찾지 못했습니다.",
problems: (n) => `문제 ${n}건 — 위의 해결 방법을 보세요.`,
problemsHint:
- " 증상 → 원인 → 해결 표: skills/setup-orca-code-review/references/troubleshooting.md",
+ " 증상 → 원인 → 해결 표: skills/orca-review-action/references/troubleshooting.md",
},
uninstall: {
@@ -878,6 +1133,14 @@ const KO = {
skill: {
missingBundle: "이 패키지에 포함된 skill을 찾을 수 없습니다.",
missingBundleHint: "다시 설치: npx @orcarouter/code-review@latest skill",
+ modeQ: "OrcaCode Review 를 어떻게 사용하시겠습니까?",
+ modeBoth: "둘 다",
+ modeBothDetail: "로컬에서 바로 리뷰하고, GitHub Action 이 모든 PR 을 리뷰합니다. 무엇을 차단할지는 둘이 일치합니다.",
+ modeLocal: "로컬 리뷰만",
+ modeLocalDetail: "내 에이전트가 여기서 리뷰 —— Action 도 OrcaRouter 계정도 API 키도 필요 없음.",
+ modeAction: "GitHub Action 만",
+ modeActionDetail: "저장소의 CI 리뷰를 설정·조정·진단·제거합니다.",
+ unknownMode: (m) => `알 수 없는 모드 "${m}".`,
scopeQ: "skill을 어디에 설치할까요?",
scopeProject: "이 프로젝트",
scopeProjectDetail: "저장소와 함께 커밋되어, clone한 모두가 쓸 수 있습니다.",
@@ -887,12 +1150,14 @@ const KO = {
unknownScopeHint: "--scope project 또는 --scope global을 쓰세요.",
unknownPlatform: (id) => `알 수 없는 플랫폼 "${id}".`,
unknownPlatformHint: "목록: npx @orcarouter/code-review skill list",
+ unknownSkill: (n) => `알 수 없는 skill "${n}".`,
noneDetected: "에이전트를 감지하지 못했고, 지정된 것도 없습니다.",
noneDetectedHint:
"직접 지정하세요. 예: --platform claude --platform codex (`skill list`로 36개 전체 확인).",
platformQ: "어떤 에이전트에 skill을 설치할까요?",
platformCount: (n) => ` (${n}개 감지)`,
statusUpdated: "(업데이트됨)",
+ statusRetired: (old) => `(이전 이름 ${old} 을 대체함)`,
statusUnchanged: "(이미 최신)",
statusConflict: "(건너뜀 — 내용이 다른 버전이 이미 있음)",
forceHint: "내용이 다른 복사본을 덮어쓰려면 --force로 다시 실행하세요.",
@@ -900,6 +1165,7 @@ const KO = {
pluginHint: " Claude Code는 플러그인 쪽이 자동으로 최신을 유지합니다:",
handoffTitle: "이제 에이전트에게 말하기만 하면 됩니다",
handoffPrimary: "이 저장소에 OrcaCode Review를 설정해줘",
+ handoffReview: "OrcaCode Review로 이번 변경을 리뷰해줘",
handoffMore: "이런 것도 됩니다:",
handoffDoctor: '"OrcaCode Review가 왜 안 돌지?"',
handoffDoctorWhat: "진단",
@@ -924,6 +1190,65 @@ const KO = {
manualPath: "저장소 → Settings → Secrets and variables → Actions → New repository secret",
},
+ report: {
+ verdictBlocked: "❌ 차단",
+ verdictPassed: "✅ 통과",
+ blockedTail: (n, total, list) => `지적 ${total}건 중 ${n}건이 ${list} 등급`,
+ passedTail: (list, total) => `${list} 등급 지적 없음 —— 지적 ${total}건`,
+ stop: ".",
+ notCrash: "이것은 리뷰 결과이며 크래시가 아닙니다",
+ clean: "지적 없음",
+ },
+ review: {
+ cfgLoaded: (file) => `${file} 을 읽었습니다.`,
+ cfgInvalid: (file, detail) => `${file} 이 잘못되었습니다: ${detail}`,
+ cfgNone: (file) => `${file} 이 없습니다 —— 모두 기본값입니다.`,
+ cfgWritten: (file) => `${file} 을 작성했습니다.`,
+ cfgExists: (file) => `${file} 이 이미 있습니다. 덮어쓰려면 --force 를 붙이세요.`,
+ cfgEditHint: (file) => `설정을 바꾸려면 ${file} 을 직접 편집하거나 AI 에게 요청하세요. 쓰지 않은 키는 기본값을 유지합니다.`,
+ cfgCreateHint: "생성: npx @orcarouter/code-review review config init",
+ cfgComment: "OrcaCode Review 로컬 리뷰 설정. block_on: 차단으로 볼 심각도(\"P0,P1\", \"\" 는 차단 없음). language: en | zh | ja | ko. exclude: 리뷰하지 않을 추가 glob. rules: [{ path, rule 또는 rule_file, replace? }] 파일 패턴별 추가 체크리스트.",
+ cfgBlockOnFrom: (list, source) => `차단 등급 ${list || "없음"} —— ${source}에서.`,
+ cfgSrcFlag: "명령줄 인자",
+ cfgSrcFile: ".orcacode-review.json",
+ cfgSrcDefault: "기본값",
+ cfgSrcLocale: "시스템 로케일",
+ cfgRowExclude: (n) => `추가 제외 ${n}건`,
+ cfgRowRules: (n) => `프로젝트 규칙 ${n}건`,
+ notRepo: "git 저장소가 아닙니다 —— `review`는 비교할 저장소가 필요합니다.",
+ cannotResolve: "무엇을 리뷰할지 판단할 수 없습니다.",
+ noBaseHint: "범위를 직접 지정하세요: --from main --to HEAD 또는 --worktree.",
+ nothingToReview: (base) => `리뷰할 것이 없습니다 —— HEAD에 ${base} 이후 커밋이 없고 작업 트리도 깨끗합니다.`,
+ noFiles: (mode) => `${mode} 모드에 리뷰 가능한 파일이 없습니다.`,
+ planned: (n, mode) => `${n}개 파일을 계획했습니다 —— ${mode} 모드.`,
+ autoWorktree: "작업 트리에 커밋되지 않은 변경이 있어 그것을 리뷰합니다.",
+ autoRange: (base) => `${base} 기준으로 이 브랜치를 리뷰합니다 —— CI와 같은 범위입니다.`,
+ conventions: (f) => `프로젝트 자체 규약(${f})을 요청에 포함했습니다.`,
+ missingResult: (p) => `${p}에 결과가 없습니다 —— 리뷰어가 아직 작성하지 않았습니다.`,
+ unusable: (m) => `결과를 쓸 수 없어 실패로 처리합니다: ${m}`,
+ positionsChecked: (a, b) => `트리와 대조해 위치를 검증: ${a} → ${b}건.`,
+ positionsNoCommit: "위치 검증 생략 —— 커밋되지 않은 코드는 어떤 커밋에도 없어 grep할 수 없습니다.",
+ positionsFailed: (d) => `위치 검증을 생략하고 모든 지적을 유지합니다${d ? ` —— ${d}` : ""}.`,
+ prConflict: "--pr 가 이미 범위를 정합니다. --from/--to/--commit/--worktree 는 빼주세요.",
+ prFetching: (n) => `풀 리퀘스트 #${n} 을(를) 확인하는 중…`,
+ prNoGh: "--pr 에는 GitHub CLI(`gh`)가 필요하지만 설치되어 있지 않습니다.",
+ prNoAuth: "gh 는 설치되어 있으나 로그인되어 있지 않습니다. 실행하세요: gh auth login",
+ prBadNumber: (v) => `풀 리퀘스트 번호가 아닙니다: ${v}`,
+ prFetchFailed: (d) => `풀 리퀘스트를 가져오지 못했습니다${d ? ` —— ${d}` : ""}.`,
+ prNoBaseRef: (b) => `풀 리퀘스트의 대상 브랜치(${b})를 로컬에서 찾을 수 없습니다.`,
+ prFailed: (d) => `gh 가 해당 풀 리퀘스트를 읽지 못했습니다${d ? ` —— ${d}` : ""}.`,
+ prHint: "또는 직접 체크아웃한 뒤(gh pr checkout <번호>) --pr 없이 리뷰하세요.",
+ prResolved: (n, base) => `풀 리퀘스트 #${n} 을(를) ${base} 와(과) 비교해 리뷰합니다. 작업 트리는 그대로입니다.`,
+ prState: (s) => `참고: 이 풀 리퀘스트의 상태는 ${s} 이며 열린 상태가 아닙니다.`,
+ prFork: "포크에서 온 풀 리퀘스트입니다. head 는 가져오기만 했고 체크아웃하지 않았습니다.",
+ prBackground: "풀 리퀘스트의 제목과 설명을 업무 배경으로 리뷰어에게 전달했습니다.",
+ badFormat: (got, ok) => `알 수 없는 --format ${got}. 다음 중 하나를 쓰세요: ${ok}`,
+ reportSaved: (p) => `붙여넣을 수 있는 markdown 리포트를 ${p} 에 저장했습니다.`,
+ clean: "깨끗함 —— 지적 사항 없음.",
+ blocked: (n, list) => `차단: ${list} 등급 지적 ${n}건.`,
+ passed: (list) => `통과 —— ${list} 등급 지적 없음.`,
+ },
+
usage: {
tagline: "OrcaCode Review 설치 도구 (OrcaRouter 기반 AI PR 리뷰)",
usage: "사용법",
@@ -938,6 +1263,21 @@ const KO = {
cmdUninstall: "workflow 제거",
cmdSkillInstall: (n) => `에이전트 skill 설치 (${n}개 플랫폼) — 기본`,
cmdSkillList: "지원 플랫폼과 감지 여부 목록",
+ cmdReviewPlan: "에이전트가 수행할 리뷰 요청을 출력",
+ cmdReviewSubmit: "에이전트의 지적을 검증하고 게이트를 적용해 보고",
+ cmdReviewConfig: "이 저장소의 로컬 리뷰 설정을 표시하거나 생성 (.orcacode-review.json)",
+ reviewOptions: "리뷰 옵션",
+ optSkill: "`skill`용: 하나만 설치 (기본: 전부)",
+ optMode: "무엇을 설치할지: both(기본) | local | action",
+ optFrom: "리뷰 범위의 시작 ref",
+ optTo: "리뷰 범위의 끝 ref",
+ optPr: "번호로 풀 리퀘스트를 리뷰(gh 필요, 체크아웃하지 않음)",
+ optFormat: "`review submit` 출력 형식: text(기본) | md | json",
+ optCommit: "커밋 하나만 리뷰",
+ optWorktree: "범위 대신 커밋되지 않은 변경을 리뷰",
+ optBackground: "리뷰어에게 넘길 업무 배경",
+ optBlockOn: "리포트에서 ❌ 차단으로 표시할 심각도 (기본: P0,P1)",
+ optFailOnBlock: "차단이 있으면 1 로 종료 —— 훅과 스크립트용. 기본은 0",
optYes: "권장값을 쓰고 묻지 않음",
optForce: "묻지 않고 덮어쓰기",
optJson: "기계 판독용 출력 (skill install / skill list)",
diff --git a/bin/localconfig.mjs b/bin/localconfig.mjs
new file mode 100644
index 0000000..62c7dbb
--- /dev/null
+++ b/bin/localconfig.mjs
@@ -0,0 +1,129 @@
+// `.orcacode-review.json` — the one place a repository configures the LOCAL
+// review. Committed, tiny, four keys.
+//
+// WHY A FILE AT ALL. Flags are per run; a repo that wants "block on P0 only" or
+// "never review docs/" wants it every run, for every person and every agent
+// that reviews here. The Action has its workflow file for that; the local
+// review had nothing.
+//
+// WHY NOT MORE KEYS. Each key answers one question a user actually asked:
+// what blocks, what language, what to skip, what else to check. Anything the
+// rubric itself decides (severity boundaries, output shape) is deliberately
+// not configurable — those are the contract shared with CI.
+//
+// VALIDATION IS STRICT AND LOUD. A typo'd key that was silently ignored would
+// make "block_on" quietly fall back to the default — the file would say P0 and
+// the review would block on P1. So unknown keys and malformed values are
+// errors, named precisely, and the commands refuse to run until fixed.
+
+import fs from "node:fs";
+import path from "node:path";
+
+import { LANGUAGES } from "./i18n.mjs";
+import { SEVERITIES } from "../scripts/severity.mjs";
+
+export const CONFIG_FILE = ".orcacode-review.json";
+export const CONFIG_KEYS = Object.freeze(["block_on", "language", "exclude", "rules"]);
+
+// Keys tolerated and ignored, so a file can carry a note to its reader.
+const NOTE_KEYS = new Set(["$comment", "//"]);
+
+const fail = (error) => ({ ok: false, file: CONFIG_FILE, error });
+
+/**
+ * Read and validate the repo's config. Returns
+ * { ok: true, file: null, config: {} } when there is no file
+ * { ok: true, file: CONFIG_FILE, config: {…} } when it parsed and validated
+ * { ok: false, file: CONFIG_FILE, error: "…" } otherwise
+ *
+ * `config` is normalised: `block_on` a comma string, `exclude` an array of
+ * lowercase globs, `rules` an array of { path, text, replace, source } with
+ * rule files already read.
+ */
+export function loadLocalConfig(root) {
+ const file = path.join(root, CONFIG_FILE);
+ if (!fs.existsSync(file)) return { ok: true, file: null, config: {} };
+
+ let raw;
+ try {
+ raw = JSON.parse(fs.readFileSync(file, "utf8"));
+ } catch (e) {
+ return fail(`not valid JSON — ${e.message}`);
+ }
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return fail("must be a JSON object");
+
+ const unknown = Object.keys(raw).filter((k) => !CONFIG_KEYS.includes(k) && !NOTE_KEYS.has(k));
+ if (unknown.length) return fail(`unknown key${unknown.length === 1 ? "" : "s"} ${unknown.map((k) => `"${k}"`).join(", ")} — allowed: ${CONFIG_KEYS.join(", ")}`);
+
+ const config = {};
+
+ if ("block_on" in raw) {
+ const v = raw.block_on;
+ const list = Array.isArray(v) ? v : typeof v === "string" ? v.split(",") : null;
+ if (list === null) return fail(`"block_on" must be a string like "P0,P1" or an array of severities`);
+ const cleaned = list.map((s) => String(s).trim().toUpperCase()).filter(Boolean);
+ const bad = cleaned.filter((s) => !SEVERITIES.includes(s));
+ if (bad.length) return fail(`"block_on" has ${bad.map((b) => `"${b}"`).join(", ")} — each entry must be one of ${SEVERITIES.join(", ")}`);
+ config.block_on = cleaned.join(",");
+ }
+
+ if ("language" in raw) {
+ if (!LANGUAGES.includes(raw.language)) return fail(`"language" must be one of ${LANGUAGES.join(", ")}`);
+ config.language = raw.language;
+ }
+
+ if ("exclude" in raw) {
+ if (!Array.isArray(raw.exclude) || raw.exclude.some((g) => typeof g !== "string" || !g.trim())) {
+ return fail(`"exclude" must be an array of glob strings, e.g. ["docs/**", "**/*.generated.ts"]`);
+ }
+ config.exclude = raw.exclude.map((g) => g.trim());
+ }
+
+ if ("rules" in raw) {
+ if (!Array.isArray(raw.rules)) return fail(`"rules" must be an array of { "path": glob, "rule": text } or { "path": glob, "rule_file": path }`);
+ const rules = [];
+ for (const [i, r] of raw.rules.entries()) {
+ const at = `"rules"[${i}]`;
+ if (!r || typeof r !== "object") return fail(`${at} must be an object`);
+ if (typeof r.path !== "string" || !r.path.trim()) return fail(`${at} needs a "path" glob`);
+ const hasText = typeof r.rule === "string";
+ const hasFile = typeof r.rule_file === "string";
+ if (hasText === hasFile) return fail(`${at} needs exactly one of "rule" (inline text) or "rule_file" (a repo-relative path)`);
+ const extra = Object.keys(r).filter((k) => !["path", "rule", "rule_file", "replace"].includes(k));
+ if (extra.length) return fail(`${at} has unknown key${extra.length === 1 ? "" : "s"} ${extra.map((k) => `"${k}"`).join(", ")}`);
+ if ("replace" in r && typeof r.replace !== "boolean") return fail(`${at} "replace" must be true or false`);
+
+ let text = r.rule;
+ if (hasFile) {
+ // Confined to the repo: a rule file is project content, and a config
+ // that points outside the checkout is a config that reads someone
+ // else's files into a prompt.
+ const abs = path.resolve(root, r.rule_file);
+ if (!abs.startsWith(`${path.resolve(root)}${path.sep}`)) return fail(`${at} "rule_file" must stay inside the repository`);
+ try {
+ text = fs.readFileSync(abs, "utf8");
+ } catch {
+ return fail(`${at} "rule_file" ${r.rule_file} could not be read`);
+ }
+ }
+ text = text.trim();
+ if (!text) return fail(`${at} rule text is empty`);
+ rules.push({ path: r.path.trim(), text, replace: r.replace === true, source: hasFile ? r.rule_file : "inline" });
+ }
+ config.rules = rules;
+ }
+
+ return { ok: true, file: CONFIG_FILE, config };
+}
+
+/** The file `review config init` writes. `language` is the caller's, so the template already says what they would have said. */
+export function configTemplate({ language = "en", blockOn, comment = "" } = {}) {
+ const doc = {
+ ...(comment ? { $comment: comment } : {}),
+ block_on: blockOn === undefined ? "P0,P1" : String(blockOn).split(",").map((s) => s.trim().toUpperCase()).filter(Boolean).join(","),
+ language,
+ exclude: [],
+ rules: [],
+ };
+ return `${JSON.stringify(doc, null, 2)}\n`;
+}
diff --git a/bin/orcacode-review.mjs b/bin/orcacode-review.mjs
index dfd13ae..605d5b1 100755
--- a/bin/orcacode-review.mjs
+++ b/bin/orcacode-review.mjs
@@ -34,10 +34,13 @@ import { fileURLToPath } from "node:url";
import { spawnSync } from "node:child_process";
import { SKILL_PLATFORMS, POPULAR_PLATFORM_IDS, findPlatform, detectPlatforms, resolveTargets } from "./platforms.mjs";
-import { installTree, STATUS } from "./skill-tree.mjs";
+import { installTree, retireLegacy, STATUS } from "./skill-tree.mjs";
import { LANGUAGES, makeT, detectLanguage, parseLanguage } from "./i18n.mjs";
import { renderBanner } from "./banner.mjs";
import * as tui from "./prompt.mjs";
+import { cmdReviewPlan, cmdReviewSubmit, cmdReviewConfig } from "./review.mjs";
+import { repoRoot } from "./harness.mjs";
+import { loadLocalConfig } from "./localconfig.mjs";
const HERE = path.dirname(fileURLToPath(import.meta.url));
const PKG_ROOT = path.resolve(HERE, "..");
@@ -46,7 +49,28 @@ const SECRET = "ORCAROUTER_API_KEY";
const ACTION_REF = "Continuum-AI-Corp/orca-code-review@v1";
const CONSOLE_TOKENS = "https://www.orcarouter.ai/console/token";
const CONSOLE_APPS = "https://www.orcarouter.ai/";
-const SKILL_NAME = "setup-orca-code-review";
+// Two skills ship in this package and the bare command installs both.
+//
+// They teach opposite halves of the same product: `setup-` tells an agent how
+// to wire up the GitHub Action (OrcaRouter reviews every PR in CI), `run-` tells
+// it how to BE the reviewer itself, right here, against the harness surface in
+// harness.mjs — no Action, no gateway, no API key, on whatever model the agent
+// already has. Installing only the first would leave the second undiscoverable,
+// and a capability an agent is never told about is one that does not ship.
+const SKILLS = Object.freeze(["orca-review-action", "orca-review"]);
+// Names these skills shipped under before. Installing the new name removes the
+// old directory beside it — see retireLegacy for the guard on whose it is.
+// The three ways to use the product, as the installer asks about them. A mode is
+// a set of skills; `--skill` still addresses one skill by name for scripts.
+const MODES = Object.freeze({
+ both: [...SKILLS],
+ local: ["orca-review"],
+ action: ["orca-review-action"],
+});
+const LEGACY_SKILL_NAMES = Object.freeze({
+ "orca-review": ["run-orca-code-review"],
+ "orca-review-action": ["setup-orca-code-review"],
+});
// Resolved from the locale up front so every path — including an early `die()`
// during argument parsing — has a working translator.
@@ -75,6 +99,27 @@ const fail = (s) => {
process.exitCode = 1;
};
+// Progress for a human to read along with, on stderr. `review plan` writes a
+// prompt to stdout and NOTHING else, so anything a pipe would swallow — or
+// worse, feed to a model as part of the prompt — has to go here instead.
+const note = (s) => console.error(`${dim("›")} ${s}`);
+
+// The output kit review.mjs renders through. Passing it in keeps this file the
+// single owner of colour, i18n, and exit behaviour, so the harness commands
+// cannot drift into a second house style.
+const reviewUi = () => ({
+ t,
+ language: LANG,
+ say,
+ ok,
+ warn,
+ info,
+ fail,
+ note,
+ die,
+ color: { bold, dim, red, green, yellow, cyan },
+});
+
function die(msg, hint) {
console.error(`${red("✖")} ${msg}`);
if (hint) console.error(` ${dim(hint)}`);
@@ -644,8 +689,32 @@ const homeDir = () => process.env.HOME || process.env.USERPROFILE || "";
const lookPath = (exe) => sh(process.platform === "win32" ? "where" : "which", [exe]).ok;
async function cmdSkill(argv) {
- const source = path.join(PKG_ROOT, "skills", SKILL_NAME);
- if (!fs.existsSync(source)) die(t("skill.missingBundle"), t("skill.missingBundleHint"));
+ // Which half of the product — asked FIRST, because the answer decides what
+ // the rest of the flow is installing. `--skill ` addresses one skill by
+ // name for scripts; `--mode` is the same choice in the words the question
+ // uses; unattended with neither installs both, because the two halves are
+ // one product and a user who only has "action" has no way to discover that
+ // local review exists.
+ if (argv.mode !== undefined && !MODES[argv.mode]) die(t("skill.unknownMode", argv.mode), Object.keys(MODES).join(" | "));
+ let wanted;
+ if (argv.skill) wanted = [argv.skill];
+ else if (argv.mode) wanted = MODES[argv.mode];
+ else if (ASSUME_YES || !process.stdin.isTTY) wanted = MODES.both;
+ else {
+ wanted = MODES[
+ await select(t("skill.modeQ"), [
+ { label: t("skill.modeBoth"), value: "both", recommended: true, detail: t("skill.modeBothDetail") },
+ { label: t("skill.modeLocal"), value: "local", detail: t("skill.modeLocalDetail") },
+ { label: t("skill.modeAction"), value: "action", detail: t("skill.modeActionDetail") },
+ ])
+ ];
+ }
+ for (const name of wanted) {
+ if (!SKILLS.includes(name)) die(t("skill.unknownSkill", name), SKILLS.join(" | "));
+ if (!fs.existsSync(path.join(PKG_ROOT, "skills", name))) {
+ die(t("skill.missingBundle"), t("skill.missingBundleHint"));
+ }
+ }
const home = homeDir();
// Detection always looks at the working directory, even for a global install:
@@ -698,35 +767,50 @@ async function cmdSkill(argv) {
// keep a scratch directory with an AGENTS setup and no git.
const gitRoot = sh("git", ["rev-parse", "--show-toplevel"], { cwd });
const projectDir = scope === "project" && gitRoot.ok ? gitRoot.out : cwd;
- const targets = resolveTargets({ platformIds, scope, projectDir, homeDir: home, skillName: SKILL_NAME });
- const results = targets.map((target) => {
- try {
- return { ...target, status: installTree(source, target.path, { force: argv.force }) };
- } catch (e) {
- return { ...target, status: STATUS.error, error: e.message };
+ // One pass per skill. resolveTargets keys the destination on the skill name,
+ // so the platforms that share a root (Codex and both Antigravities all use
+ // `.agents`) still collapse to one write per skill without the two skills
+ // ever colliding with each other.
+ const results = [];
+ for (const skillName of wanted) {
+ const source = path.join(PKG_ROOT, "skills", skillName);
+ for (const target of resolveTargets({ platformIds, scope, projectDir, homeDir: home, skillName })) {
+ try {
+ const retired = (LEGACY_SKILL_NAMES[skillName] || []).filter((old) => retireLegacy(target.path, old));
+ results.push({ ...target, skill: skillName, retired, status: installTree(source, target.path, { force: argv.force }) });
+ } catch (e) {
+ results.push({ ...target, skill: skillName, status: STATUS.error, error: e.message });
+ }
}
- });
+ }
closeUi();
if (argv.json) {
- say(JSON.stringify({ scope, source, results }, null, 2));
+ say(JSON.stringify({ scope, skills: wanted, results }, null, 2));
if (results.some((r) => r.status === STATUS.error)) process.exitCode = 1;
return;
}
say();
- for (const r of results) {
- const names = r.platformNames.join(", ");
- const where = dim(path.relative(scope === "project" ? projectDir : home, r.path) || r.path);
- switch (r.status) {
- case STATUS.installed: ok(`${names} ${where}`); break;
- case STATUS.updated: ok(`${names} ${where} ${dim(t("skill.statusUpdated"))}`); break;
- case STATUS.unchanged: say(`${dim("·")} ${names} ${where} ${dim(t("skill.statusUnchanged"))}`); break;
- case STATUS.conflict: warn(`${names} ${where} ${dim(t("skill.statusConflict"))}`); break;
- default: fail(`${names} ${where} — ${r.error}`);
+ for (const skillName of wanted) {
+ const forSkill = results.filter((r) => r.skill === skillName);
+ if (forSkill.length === 0) continue;
+ say(` ${bold(skillName)}`);
+ for (const r of forSkill) {
+ const names = r.platformNames.join(", ");
+ const where = dim(path.relative(scope === "project" ? projectDir : home, r.path) || r.path);
+ const retired = r.retired?.length ? ` ${dim(t("skill.statusRetired", r.retired.join(", ")))}` : "";
+ switch (r.status) {
+ case STATUS.installed: ok(`${names} ${where}${retired}`); break;
+ case STATUS.updated: ok(`${names} ${where} ${dim(t("skill.statusUpdated"))}${retired}`); break;
+ case STATUS.unchanged: say(`${dim("·")} ${names} ${where} ${dim(t("skill.statusUnchanged"))}`); break;
+ case STATUS.conflict: warn(`${names} ${where} ${dim(t("skill.statusConflict"))}`); break;
+ default: fail(`${names} ${where} — ${r.error}`);
+ }
}
+ say();
}
if (results.some((r) => r.status === STATUS.conflict)) {
@@ -745,6 +829,7 @@ function handoff(results) {
say(bold(t("skill.handoffTitle")));
say();
say(` ${cyan(t("skill.handoffPrimary"))}`);
+ say(` ${cyan(t("skill.handoffReview"))}`);
say();
say(dim(t("skill.handoffMore")));
for (const [phrase, what] of [
@@ -836,6 +921,9 @@ ${bold(t("usage.commands"))}
uninstall ${t("usage.cmdUninstall")}
skill install ${t("usage.cmdSkillInstall", n)}
skill list ${t("usage.cmdSkillList")}
+ review plan ${t("usage.cmdReviewPlan")}
+ review submit ${t("usage.cmdReviewSubmit")}
+ review config ${t("usage.cmdReviewConfig")}
${bold(t("usage.options"))}
--yes, -y ${t("usage.optYes")}
@@ -845,12 +933,29 @@ ${bold(t("usage.options"))}
--no-banner ${t("usage.optNoBanner")}
--scope ${t("usage.optScope")}
--platform ${t("usage.optPlatform")}
+ --mode ${t("usage.optMode")}
+ --skill ${t("usage.optSkill")}
--help, -h ${t("usage.optHelp")}
--version, -v ${t("usage.optVersion")}
+${bold(t("usage.reviewOptions"))}
+ --pr ${t("usage.optPr")}
+ --from [ ${t("usage.optFrom")}
+ --to ][ ${t("usage.optTo")}
+ --commit, -c ${t("usage.optCommit")}
+ --worktree ${t("usage.optWorktree")}
+ --background, -b ${t("usage.optBackground")}
+ --block-on ] ${t("usage.optBlockOn")}
+ --format ${t("usage.optFormat")}
+ --fail-on-block ${t("usage.optFailOnBlock")}
+
${bold(t("usage.examples"))}
npx @orcarouter/code-review --platform claude --platform codex --yes
+ npx @orcarouter/code-review --mode local --scope global --platform claude --yes
npx @orcarouter/code-review skill list
+ npx @orcarouter/code-review review plan --pr 556
+ npx @orcarouter/code-review review plan --from main --to HEAD
+ npx @orcarouter/code-review review submit .orcacode-review/result.json
npx @orcarouter/code-review doctor
${bold(t("usage.docs"))} https://github.com/Continuum-AI-Corp/orca-code-review`);
@@ -871,6 +976,37 @@ function parse(args) {
else if (a.startsWith("--lang=")) out.lang = a.slice(7);
else if (a === "--scope") out.scope = args[++i];
else if (a.startsWith("--scope=")) out.scope = a.slice(8);
+ else if (a === "--mode") out.mode = args[++i];
+ else if (a.startsWith("--mode=")) out.mode = a.slice(7);
+ else if (a === "--skill") out.skill = args[++i];
+ else if (a.startsWith("--skill=")) out.skill = a.slice(8);
+ // `review` range selection, named to match `ocr delegate` so anyone who
+ // knows one already knows the other.
+ else if (a === "--from") out.from = args[++i];
+ else if (a.startsWith("--from=")) out.from = a.slice(7);
+ else if (a === "--to") out.to = args[++i];
+ else if (a.startsWith("--to=")) out.to = a.slice(5);
+ else if (a === "--format") out.format = args[++i];
+ else if (a.startsWith("--format=")) out.format = a.slice(9);
+ else if (a === "--md" || a === "--markdown") out.format = "md";
+ else if (a === "--pr") out.pr = args[++i];
+ else if (a.startsWith("--pr=")) out.pr = a.slice(5);
+ else if (a === "--commit" || a === "-c") out.commit = args[++i];
+ else if (a.startsWith("--commit=")) out.commit = a.slice(9);
+ else if (a === "--worktree" || a === "--workspace") out.worktree = true;
+ else if (a === "--background" || a === "-b") out.background = args[++i];
+ else if (a.startsWith("--background=")) out.background = a.slice(13);
+ else if (a === "--block-on") out.blockOn = args[++i];
+ else if (a.startsWith("--block-on=")) out.blockOn = a.slice(11);
+ // Opt-in for a script that wants the verdict as a process status — a hook, a
+ // CI step, a `submit && push`. Off by default: the local harness exists to
+ // tell an agent what the bugs are, and an agent reads the report, not the
+ // exit code. Its shell tool labels any non-zero exit "Error", which turned a
+ // working review into an apparent crash, twice, in front of a real user.
+ else if (a === "--fail-on-block") out.failOnBlock = true;
+ // Escape hatch for a harness that has already verified its own positions,
+ // or is submitting a result it built without a plan.
+ else if (a === "--no-postfilter") out.postfilter = false;
// Repeatable, matching orcadub: --platform claude --platform codex.
// A comma-separated list is also accepted because people type it anyway.
else if (a === "--platform" || a === "--platforms") pushPlatforms(out, args[++i]);
@@ -901,6 +1037,13 @@ async function main() {
// user's locale, so their own language is still the right one to use.
die(t("common.unknownLanguage", argv.lang), t("common.unknownLanguageHint"));
}
+ } else if (argv._[0] === "review") {
+ // The repo may pin a language in .orcacode-review.json; it outranks the
+ // locale and yields to --lang. An invalid file is left for the command to
+ // report properly — here it simply does not get a vote.
+ const root = repoRoot(process.cwd());
+ const cfg = root ? loadLocalConfig(root) : null;
+ if (cfg?.ok && cfg.config.language) setLanguage(cfg.config.language);
}
if (argv.help) return usage();
@@ -938,6 +1081,17 @@ async function main() {
case "check": return cmdDoctor();
case "uninstall":
case "remove": return cmdUninstall();
+ // The harness surface. `review` on its own is ambiguous between the two
+ // halves — planning a review and handing one back are different acts with
+ // different exit codes — so the sub-verb is required rather than guessed.
+ case "review": {
+ const sub = argv._[1];
+ if (sub === "plan") return cmdReviewPlan(argv, reviewUi());
+ if (sub === "submit") return cmdReviewSubmit(argv, reviewUi());
+ if (sub === "config") return cmdReviewConfig(argv, reviewUi());
+ die(t("common.unknownCommand", sub ? `review ${sub}` : "review"), "review plan | review submit | review config");
+ return;
+ }
case "skill": {
// `skill`, `skill install`, `skill list` — the sub-verb is optional so the
// bare form keeps working for anyone who learned it before `list` existed.
diff --git a/bin/review.mjs b/bin/review.mjs
new file mode 100644
index 0000000..db5c85e
--- /dev/null
+++ b/bin/review.mjs
@@ -0,0 +1,366 @@
+// `review plan` and `review submit` — the two command entry points of the
+// harness surface. The reviewing logic lives in harness.mjs; this file is the
+// part that touches the filesystem, the terminal, and the exit code.
+//
+// STDOUT vs STDERR MATTERS HERE. `review plan` writes the review request — a
+// prompt — to stdout and nothing else, so piping it straight into an agent
+// works and the captured text is not laced with progress lines. Everything
+// meant for a human reading along goes to stderr. `review submit` is the
+// reverse: its report is for the human, so that goes to stdout.
+//
+// Exit codes:
+// 0 reviewed — the report carries the verdict, blocked or not
+// 1 reviewed and blocked, ONLY under --fail-on-block (hooks, CI steps)
+// 2 no usable result (unparseable, malformed, or partial) — never a pass
+
+import fs from "node:fs";
+import path from "node:path";
+
+import {
+ WORK_DIR,
+ buildPlan,
+ renderPlan,
+ groundTruthRef,
+ validateResult,
+ positionCheck,
+ gate,
+ renderReport,
+ renderReportMarkdown,
+ parsePrNumber,
+ repoRoot,
+ resolvePr,
+ gitCommonDir,
+} from "./harness.mjs";
+import { CONFIG_FILE, loadLocalConfig, configTemplate } from "./localconfig.mjs";
+
+const DEFAULT_BLOCK_ON = "P0,P1";
+const FORMATS = ["text", "md", "json"];
+
+// Keep the scratch directory out of `git status` without editing .gitignore —
+// that file belongs to the project, and a review must not leave a diff behind.
+// .git/info/exclude is the per-clone equivalent and is never committed.
+function excludeFromGit(root) {
+ const gitDir = gitCommonDir(root);
+ if (!gitDir) return;
+ const file = path.join(gitDir, "info", "exclude");
+ try {
+ const current = fs.existsSync(file) ? fs.readFileSync(file, "utf8") : "";
+ if (current.split("\n").some((l) => l.trim() === `/${WORK_DIR}/`)) return;
+ fs.mkdirSync(path.dirname(file), { recursive: true });
+ fs.appendFileSync(file, `${current && !current.endsWith("\n") ? "\n" : ""}/${WORK_DIR}/\n`);
+ } catch {
+ // A read-only or otherwise unwritable git store — the review still works,
+ // the directory just shows up as untracked. Not worth failing over.
+ }
+}
+
+// `--pr` is the one part of the harness that talks to a forge. It stays here,
+// at the edge, behind a soft dependency on `gh` — the review itself never
+// learns what a pull request is, so a host without `gh` (or without GitHub)
+// loses this shortcut and nothing else.
+function applyPr(argv, root, ui) {
+ if (argv.pr === undefined) return {};
+ if (argv.from || argv.to || argv.commit || argv.worktree) ui.die(ui.t("review.prConflict"));
+
+ // Validate before announcing. "Resolving pull request #abc…" followed by
+ // "that is not a number" reads as though we tried.
+ const n = parsePrNumber(argv.pr);
+ if (!n) ui.die(ui.t("review.prBadNumber", String(argv.pr ?? "")), ui.t("review.prHint"));
+
+ ui.note(ui.t("review.prFetching", n));
+ const r = resolvePr(n, root);
+ if (!r.ok) {
+ const message = {
+ "no-gh": () => ui.t("review.prNoGh"),
+ "no-auth": () => ui.t("review.prNoAuth"),
+ "bad-number": () => ui.t("review.prBadNumber", r.detail),
+ "fetch-failed": () => ui.t("review.prFetchFailed", r.detail),
+ "no-base-ref": () => ui.t("review.prNoBaseRef", r.detail),
+ }[r.code];
+ ui.die(message ? message() : ui.t("review.prFailed", r.detail), ui.t("review.prHint"));
+ }
+
+ ui.note(ui.t("review.prResolved", r.pr.number, r.pr.base));
+ // A merged or closed PR still reviews fine, but reviewing one by accident —
+ // because the number was a typo — should not be silent.
+ if (r.pr.state && r.pr.state !== "OPEN") ui.note(ui.t("review.prState", r.pr.state));
+ if (r.pr.fork) ui.note(ui.t("review.prFork"));
+
+ // An explicit --background is the user speaking; the PR body is a default.
+ const background = argv.background || r.background;
+ if (!argv.background && r.background) ui.note(ui.t("review.prBackground"));
+
+ return { from: r.from, to: r.to, pr: r.pr.number, prMeta: r.pr, background };
+}
+
+// The repo's settings file, or a loud refusal. Both commands call this first:
+// a config that fails to parse must stop the review, not be shrugged off into
+// defaults — the file says P0 and the run would block on P1.
+function requireConfig(root, ui) {
+ const loaded = loadLocalConfig(root);
+ if (!loaded.ok) {
+ ui.fail(ui.t("review.cfgInvalid", loaded.file, loaded.error));
+ process.exit(2);
+ }
+ return loaded;
+}
+
+export async function cmdReviewPlan(argv, ui) {
+ const root = repoRoot(process.cwd());
+ if (!root) ui.die(ui.t("review.notRepo"));
+
+ const config = requireConfig(root, ui);
+ const pr = applyPr(argv, root, ui);
+
+ const plan = buildPlan(
+ {
+ from: argv.from,
+ to: argv.to,
+ commit: argv.commit,
+ worktree: argv.worktree,
+ background: argv.background,
+ language: ui.language,
+ config,
+ ...pr,
+ },
+ root,
+ );
+
+ if (plan.range.mode === "error") ui.die(ui.t("review.cannotResolve"), ui.t("review.noBaseHint"));
+ if (plan.range.mode === "empty") {
+ ui.warn(ui.t("review.nothingToReview", plan.range.base || ""));
+ return;
+ }
+ if (plan.files.length === 0) {
+ ui.warn(ui.t("review.noFiles", plan.range.mode));
+ return;
+ }
+
+ const workDir = path.join(root, WORK_DIR);
+ fs.mkdirSync(workDir, { recursive: true });
+ excludeFromGit(root);
+
+ const request = renderPlan(plan);
+ // plan.json is what `review submit` reads back for the range — submit must
+ // grep the same commit the review was planned against, and re-deriving it
+ // later could pick a different one if the user committed in between.
+ fs.writeFileSync(
+ path.join(workDir, "plan.json"),
+ `${JSON.stringify({ ...plan, ground_truth_ref: groundTruthRef(plan.range, root) }, null, 2)}\n`,
+ );
+ fs.writeFileSync(path.join(workDir, "request.md"), `${request}\n`);
+
+ if (argv.json) {
+ process.stdout.write(`${JSON.stringify(plan, null, 2)}\n`);
+ } else {
+ process.stdout.write(`${request}\n`);
+ }
+
+ // Human notes on stderr so they never contaminate a piped prompt.
+ ui.note(ui.t("review.planned", plan.files.length, plan.range.mode));
+ if (config.file) ui.note(ui.t("review.cfgLoaded", config.file));
+ // Only explain a mode nobody asked for. When the user typed the flag, saying
+ // it back to them is noise.
+ if (plan.range.code === "auto-dirty") ui.note(ui.t("review.autoWorktree"));
+ if (plan.range.code === "auto-ahead") ui.note(ui.t("review.autoRange", plan.range.base));
+ if (plan.rubric.conventions.file) ui.note(ui.t("review.conventions", plan.rubric.conventions.file));
+}
+
+export async function cmdReviewSubmit(argv, ui) {
+ // Before anything else: a mistyped flag is a usage error, and reporting it as
+ // "no result file" sends the caller looking in the wrong place.
+ const format = argv.json ? "json" : argv.format || "text";
+ if (!FORMATS.includes(format)) ui.die(ui.t("review.badFormat", format, FORMATS.join(", ")));
+
+ const root = repoRoot(process.cwd());
+ if (!root) ui.die(ui.t("review.notRepo"));
+
+ const config = requireConfig(root, ui);
+
+ const resultFile = path.resolve(root, argv._[2] || path.join(WORK_DIR, "result.json"));
+ if (!fs.existsSync(resultFile)) {
+ ui.fail(ui.t("review.missingResult", path.relative(root, resultFile)));
+ process.exitCode = 2;
+ return;
+ }
+
+ let parsed;
+ try {
+ parsed = JSON.parse(fs.readFileSync(resultFile, "utf8"));
+ } catch (e) {
+ ui.fail(ui.t("review.unusable", e.message));
+ process.exitCode = 2;
+ return;
+ }
+
+ const check = validateResult(parsed);
+ if (!check.ok) {
+ ui.fail(ui.t("review.unusable", check.error));
+ process.exitCode = 2;
+ return;
+ }
+
+ // The commit the position check greps, recorded when the plan was built.
+ let ref = "";
+ try {
+ ref = JSON.parse(fs.readFileSync(path.join(root, WORK_DIR, "plan.json"), "utf8")).ground_truth_ref || "";
+ } catch {
+ // Submitting a result nobody planned is allowed — a harness may have built
+ // the request itself. We just cannot verify positions without a commit.
+ }
+
+ let comments = check.comments;
+ if (comments.length > 0 && argv.postfilter !== false) {
+ // postfilter.mjs reads a FILE, not our parsed copy, so it must be handed the
+ // widened shape — it is written against `start_line`/`end_line`, and pointing
+ // it at a result that used a bare `line` silently returns every finding with
+ // no anchor at all. Normalising here is also what keeps this caller and the
+ // Action's caller feeding that script byte-identical input.
+ const normalized = path.join(root, WORK_DIR, "result.normalized.json");
+ fs.mkdirSync(path.dirname(normalized), { recursive: true });
+ fs.writeFileSync(normalized, `${JSON.stringify({ comments, warnings: [] })}\n`);
+
+ const l1 = positionCheck(normalized, root, ref);
+ fs.rmSync(normalized, { force: true });
+ if (l1.ran) {
+ const filtered = JSON.parse(fs.readFileSync(l1.out, "utf8"));
+ ui.note(ui.t("review.positionsChecked", comments.length, filtered.comments.length));
+ comments = filtered.comments;
+ fs.rmSync(l1.out, { force: true });
+ } else if (l1.code === "no-commit") {
+ ui.note(ui.t("review.positionsNoCommit"));
+ } else {
+ ui.note(ui.t("review.positionsFailed", l1.detail));
+ }
+ }
+
+ // Flag beats file beats default. Said out loud only when the file decided —
+ // a flag is the user's own typing, and the default needs no announcement.
+ const blockOn = argv.blockOn ?? config.config.block_on ?? DEFAULT_BLOCK_ON;
+ if (argv.blockOn === undefined && config.config.block_on !== undefined) {
+ ui.note(ui.t("review.cfgBlockOnFrom", blockOn, ui.t("review.cfgSrcFile")));
+ }
+ const result = gate(comments, blockOn);
+
+ // Always written, whatever the chosen format. An agent that ran the default
+ // terminal report still has somewhere to read a pasteable version from, and a
+ // human has something to attach to a ticket.
+ const markdown = renderReportMarkdown(comments, result, { t: ui.t });
+ const reportPath = path.join(WORK_DIR, "report.md");
+ try {
+ const workDir = path.join(root, WORK_DIR);
+ fs.mkdirSync(workDir, { recursive: true });
+ fs.writeFileSync(path.join(root, reportPath), `${markdown}\n`);
+ // Only worth saying when the report was NOT what we just printed.
+ if (format === "text" && comments.length > 0) ui.note(ui.t("review.reportSaved", reportPath));
+ } catch {
+ // A report we could not save is not a reason to fail a review.
+ }
+
+ // The process status carries the verdict only when asked. The local harness
+ // is there to tell an agent what the bugs are; the agent reads the report,
+ // and its shell tool would stamp a 1 with "Error" and bury that report under
+ // a crash banner. A hook or CI step that wants `&&` to mean something passes
+ // --fail-on-block. A `2` upstream is different — the result was unusable —
+ // and nothing here touches it: a review that did not happen must never come
+ // back looking like one that passed.
+ const verdictExit = () => {
+ if (result.blocked && argv.failOnBlock) process.exitCode = 1;
+ };
+
+ if (format === "json") {
+ process.stdout.write(
+ `${JSON.stringify({ comments, counts: result.counts, block_on: result.wanted, blocked: result.blocked }, null, 2)}\n`,
+ );
+ verdictExit();
+ return;
+ }
+
+ if (format === "md") {
+ // Markdown is the whole output — the verdict is already its second line, so
+ // repeating it through ui.ok/ui.fail would render as a stray ANSI line in
+ // the middle of a chat message.
+ process.stdout.write(`${markdown}\n`);
+ verdictExit();
+ return;
+ }
+
+ ui.say();
+ if (comments.length === 0) {
+ ui.ok(ui.t("review.clean"));
+ return;
+ }
+
+ ui.say(renderReport(comments, result, { color: ui.color, t: ui.t }));
+ ui.say();
+ if (result.blocked) {
+ // ui.fail() sets exitCode 1 as a side effect, so it is only the right
+ // channel when that IS the request. Otherwise the same line goes out as a
+ // note: the verdict is information here, not a failure.
+ if (argv.failOnBlock) ui.fail(ui.t("review.blocked", result.blocking.length, result.wanted.join(",")));
+ else ui.note(ui.t("review.blocked", result.blocking.length, result.wanted.join(",")));
+ verdictExit();
+ } else {
+ ui.ok(ui.t("review.passed", result.wanted.join(",") || "—"));
+ }
+}
+
+// `review config` — show the settings that will apply and where each came
+// from, or `review config init` to write the file. The show form exists so a
+// user never has to reason about precedence in their head: three sources
+// (flag, file, default) collapse to one column that says which won.
+export async function cmdReviewConfig(argv, ui) {
+ const root = repoRoot(process.cwd());
+ if (!root) ui.die(ui.t("review.notRepo"));
+
+ if (argv._[2] === "init") {
+ const file = path.join(root, CONFIG_FILE);
+ if (fs.existsSync(file) && !argv.force) {
+ ui.warn(ui.t("review.cfgExists", CONFIG_FILE));
+ process.exitCode = 1;
+ return;
+ }
+ // Pre-filled with what this user was already using — the language they
+ // spoke, the gate they asked for — so the template is a record of the
+ // present, not a form to fill in.
+ fs.writeFileSync(
+ file,
+ configTemplate({ language: ui.language, blockOn: argv.blockOn, comment: ui.t("review.cfgComment") }),
+ );
+ ui.ok(ui.t("review.cfgWritten", CONFIG_FILE));
+ ui.say(` ${ui.t("review.cfgEditHint", CONFIG_FILE)}`);
+ return;
+ }
+
+ const loaded = requireConfig(root, ui);
+ const c = loaded.config;
+ const src = {
+ flag: ui.t("review.cfgSrcFlag"),
+ file: ui.t("review.cfgSrcFile"),
+ dflt: ui.t("review.cfgSrcDefault"),
+ locale: ui.t("review.cfgSrcLocale"),
+ };
+ const effective = {
+ file: loaded.file,
+ block_on: { value: argv.blockOn ?? c.block_on ?? DEFAULT_BLOCK_ON, source: argv.blockOn !== undefined ? "flag" : c.block_on !== undefined ? "file" : "default" },
+ language: { value: ui.language, source: argv.lang ? "flag" : c.language ? "file" : "locale" },
+ exclude: c.exclude || [],
+ rules: (c.rules || []).map((r) => ({ path: r.path, replace: r.replace, source: r.source })),
+ };
+
+ if (argv.json) {
+ process.stdout.write(`${JSON.stringify(effective, null, 2)}\n`);
+ return;
+ }
+
+ const name = (k) => ({ flag: src.flag, file: src.file, default: src.dflt, locale: src.locale })[k];
+ if (loaded.file) ui.note(ui.t("review.cfgLoaded", loaded.file));
+ else ui.note(ui.t("review.cfgNone", CONFIG_FILE));
+ ui.say();
+ ui.say(` ${ui.color.bold("block_on".padEnd(10))} ${String(effective.block_on.value || '""').padEnd(10)} ${ui.color.dim(`← ${name(effective.block_on.source)}`)}`);
+ ui.say(` ${ui.color.bold("language".padEnd(10))} ${effective.language.value.padEnd(10)} ${ui.color.dim(`← ${name(effective.language.source)}`)}`);
+ ui.say(` ${ui.color.bold("exclude".padEnd(10))} ${ui.t("review.cfgRowExclude", effective.exclude.length)}${effective.exclude.length ? ui.color.dim(` ${effective.exclude.join(", ")}`) : ""}`);
+ ui.say(` ${ui.color.bold("rules".padEnd(10))} ${ui.t("review.cfgRowRules", effective.rules.length)}${effective.rules.length ? ui.color.dim(` ${effective.rules.map((r) => `${r.path} → ${r.source}${r.replace ? " (replace)" : ""}`).join(", ")}`) : ""}`);
+ ui.say();
+ ui.say(` ${ui.color.dim(loaded.file ? ui.t("review.cfgEditHint", loaded.file) : ui.t("review.cfgCreateHint"))}`);
+}
diff --git a/bin/selection.mjs b/bin/selection.mjs
new file mode 100644
index 0000000..d65106a
--- /dev/null
+++ b/bin/selection.mjs
@@ -0,0 +1,404 @@
+// SPDX-License-Identifier: Apache-2.0
+//
+// File selection and per-file review rules, ported from Open Code Review.
+//
+// Derived from alibaba/open-code-review (Apache-2.0), at the tag recorded in
+// vendor/open-code-review/UPSTREAM. This is a JavaScript port — a modified
+// work under Apache §4(b) — of these Go sources:
+//
+// internal/config/rules/system_rules.go ordered path -> rule resolution,
+// brace expansion
+// internal/config/rules/sniffer.go the ".m" MATLAB / Objective-C sniff
+// internal/config/allowlist/allowed_ext.go extension allowlist, default excludes
+// internal/diff/git.go always-skipped dirs, .gitignore matching
+// internal/agent/preview.go the exclusion order and reason codes
+// internal/delegate/rulegroup.go grouping files that share a rule
+//
+// The DATA it reads (checklists, path map, allowlist, exclude globs) is copied
+// unmodified into vendor/open-code-review/.
+//
+// WHY A PORT AND NOT THE BINARY. `ocr` is a 50 MB Go executable per platform;
+// what the local review needs from it is a few hundred KB of markdown and a
+// glob matcher. Shipping the data and porting the matcher gives a repo that
+// never installed anything the same file selection CI's engine applies — and
+// the reviewer here is the user's own agent, so nothing that thinks is lost.
+//
+// WHAT IS NOT PORTED, deliberately: Open Code Review's user rule layers
+// (.opencodereview/rule.json, the global rule file). Those are its own config
+// format; a repo that uses it is a repo that has `ocr` set up, and this path is
+// for repos that do not.
+
+import fs from "node:fs";
+import path from "node:path";
+import { spawnSync } from "node:child_process";
+import { fileURLToPath } from "node:url";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+export const VENDOR_DIR = path.resolve(HERE, "..", "vendor", "open-code-review");
+
+// ------------------------------------------------------------------- glob ---
+
+/**
+ * Expand every `{a,b,c}` group in a pattern into the list of plain patterns
+ * it stands for. `system_rules.go` expands one group; the allowlist hands its
+ * braces to doublestar, which expands all of them. Expanding all here serves
+ * both — a pattern with one group is a pattern with all of its groups expanded.
+ */
+export function expandBraces(pattern) {
+ const open = pattern.indexOf("{");
+ if (open < 0) return [pattern];
+ const close = pattern.indexOf("}", open);
+ if (close < 0) return [pattern];
+ const head = pattern.slice(0, open);
+ const tail = pattern.slice(close + 1);
+ const out = [];
+ for (const alt of pattern.slice(open + 1, close).split(",")) {
+ for (const rest of expandBraces(head + alt + tail)) out.push(rest);
+ }
+ return out;
+}
+
+const cache = new Map();
+
+/**
+ * doublestar's Match semantics as a RegExp, for one brace-free pattern:
+ * `**` as a whole segment spans zero or more directories
+ * `*` any run of non-`/` characters
+ * `?` one non-`/` character
+ * `[…]` a character class; `[!…]` negates
+ * Anchored at both ends — doublestar.Match is a whole-string match.
+ */
+export function globToRegExp(pattern) {
+ let hit = cache.get(pattern);
+ if (hit) return hit;
+
+ let re = "";
+ const segs = pattern.split("/");
+ for (let i = 0; i < segs.length; i++) {
+ const seg = segs[i];
+ const first = i === 0;
+ const last = i === segs.length - 1;
+ if (seg === "**") {
+ // A whole-segment globstar spans zero or more directories. Where it sits
+ // decides which separator it owns: a leading one swallows the slash after
+ // it, a trailing one the slash before it, a middle one repeats "/seg".
+ if (first && last) re += ".*";
+ else if (first) re += "(?:.*/)?";
+ else if (last) re += "(?:/.*)?";
+ else re += "(?:/[^/]*)*";
+ continue;
+ }
+ // A leading globstar already emitted the separator that follows it.
+ if (!first && !(i === 1 && segs[0] === "**")) re += "/";
+ for (let j = 0; j < seg.length; j++) {
+ const ch = seg[j];
+ if (ch === "*") re += "[^/]*";
+ else if (ch === "?") re += "[^/]";
+ else if (ch === "[") {
+ const close = seg.indexOf("]", j + 1);
+ if (close < 0) {
+ re += "\\[";
+ continue;
+ }
+ let cls = seg.slice(j + 1, close);
+ if (cls.startsWith("!")) cls = `^${cls.slice(1)}`;
+ re += `[${cls.replace(/\\/g, "\\\\")}]`;
+ j = close;
+ } else re += ch.replace(/[.+^${}()|\\]/g, "\\$&");
+ }
+ }
+ hit = new RegExp(`^${re}$`);
+ cache.set(pattern, hit);
+ return hit;
+}
+
+/** Whole-string glob match, braces expanded. */
+export function globMatch(pattern, str) {
+ for (const p of expandBraces(pattern)) if (globToRegExp(p).test(str)) return true;
+ return false;
+}
+
+// ------------------------------------------------------------------- data ---
+
+const read = (name) => fs.readFileSync(path.join(VENDOR_DIR, name), "utf8");
+let data;
+function load() {
+ if (data) return data;
+ const sys = JSON.parse(read("system_rules.json"));
+ // Object key order is insertion order for string keys, which is what the Go
+ // side goes to some length to preserve: first match wins.
+ const pathRules = Object.entries(sys.path_rule_map || {}).map(([pattern, doc]) => ({
+ pattern,
+ // Lowercased once, like resolveDetail does per call; the path is lowercased
+ // to match, so `*.R` and `*.r` are the same rule.
+ expanded: expandBraces(pattern.toLowerCase()),
+ doc,
+ }));
+ const docs = new Map();
+ const doc = (name) => {
+ if (!docs.has(name)) docs.set(name, read(path.join("rule_docs", name)).replace(/\n+$/, ""));
+ return docs.get(name);
+ };
+ data = {
+ pathRules,
+ defaultDoc: sys.default_rule,
+ doc,
+ allowedExt: new Set(JSON.parse(read("supported_file_types.json")).map((e) => e.toLowerCase())),
+ excludeGlobs: JSON.parse(read("default_exclude_patterns.json")).map((p) => p.toLowerCase()),
+ };
+ return data;
+}
+
+// -------------------------------------------------------------- exclusion ---
+
+/** Directory prefixes always skipped; a .gitignore negation cannot re-admit them. */
+export const IGNORED_DIRS = [
+ ".idea/",
+ ".vscode/",
+ ".svn/",
+ ".git/",
+ "vendor/",
+ "node_modules/",
+ "target/",
+ ".happypack/",
+ ".cachefile/",
+ "_packages/",
+ "rpm/",
+ "pkgs/",
+];
+
+/** The root .gitignore only, as upstream reads it. Comments and blanks dropped. */
+export function loadGitignorePatterns(repoDir) {
+ let text;
+ try {
+ text = fs.readFileSync(path.join(repoDir, ".gitignore"), "utf8");
+ } catch {
+ return [];
+ }
+ return text
+ .split("\n")
+ .map((l) => l.trim())
+ .filter((l) => l && !l.startsWith("#"));
+}
+
+// One gitignore pattern body (leading "!" already removed) against a path.
+function matchGitignoreBody(relPath, body) {
+ if (body.endsWith("/")) return matchGitignoreDirectory(relPath, body.slice(0, -1));
+
+ let anchored = false;
+ if (body.startsWith("/")) {
+ body = body.slice(1);
+ anchored = true;
+ }
+ if (body.includes("**")) return globMatch(body, relPath);
+ if (!body.includes("/")) {
+ const target = anchored ? relPath : relPath.slice(relPath.lastIndexOf("/") + 1);
+ return globMatch(body, target);
+ }
+ if (globMatch(body, relPath)) return true;
+ // "src/main.go" must not match "othersrc/main.go": the suffix has to start
+ // on a component boundary, and an anchored pattern names one path only.
+ return !anchored && relPath.endsWith(`/${body}`);
+}
+
+function matchGitignoreDirectory(relPath, pattern) {
+ let anchored = false;
+ if (pattern.startsWith("/")) {
+ pattern = pattern.slice(1);
+ anchored = true;
+ }
+ if (!pattern) return false;
+ const lastSlash = relPath.lastIndexOf("/");
+ if (lastSlash < 0) return false;
+ const components = relPath.slice(0, lastSlash).split("/");
+ const fullPath = anchored || pattern.includes("/");
+ for (let i = 0; i < components.length; i++) {
+ const candidate = fullPath ? components.slice(0, i + 1).join("/") : components[i];
+ if (globMatch(pattern, candidate)) return true;
+ }
+ return false;
+}
+
+/**
+ * Skipped outright: an always-ignored directory, or ignored by the root
+ * .gitignore resolved the way git resolves it — in order, last match wins, a
+ * leading "!" inverts. Upstream drops these before anything else sees them.
+ */
+export function isIgnored(relPath, patterns) {
+ for (const prefix of IGNORED_DIRS) {
+ if (relPath === prefix.slice(0, -1) || relPath.startsWith(prefix)) return true;
+ }
+ let excluded = false;
+ for (const pat of patterns) {
+ const negated = pat.startsWith("!");
+ const body = negated ? pat.slice(1) : pat;
+ if (!body) continue;
+ // A negated directory-only pattern (`!*/`) tells git to keep descending; it
+ // does not re-admit the files below.
+ if (negated && body.endsWith("/")) continue;
+ if (matchGitignoreBody(relPath, body)) excluded = !negated;
+ }
+ return excluded;
+}
+
+/** Lowercased extension with the dot; "" for none. A leading dot (".env") is a name, not an extension. */
+export function extOf(relPath) {
+ const base = relPath.slice(relPath.lastIndexOf("/") + 1);
+ const dot = base.lastIndexOf(".");
+ return dot <= 0 ? "" : base.slice(dot).toLowerCase();
+}
+
+export const isAllowedExt = (ext) => load().allowedExt.has(ext.toLowerCase());
+
+export function isDefaultExcludedPath(relPath) {
+ const lower = relPath.toLowerCase();
+ return load().excludeGlobs.some((g) => globMatch(g, lower));
+}
+
+/** Reason codes, upstream's vocabulary plus `ignored` for what it drops silently. */
+export const REASON = Object.freeze({
+ ignored: "in an always-skipped directory, or ignored by .gitignore",
+ binary: "binary",
+ project_exclude: "excluded by .orcacode-review.json",
+ unsupported_ext: "file type is not reviewed",
+ default_path: "matches a default exclude pattern (tests, fixtures, snapshots, generated code)",
+ deleted: "deleted — nothing left to review",
+});
+
+/**
+ * Why a changed file is out of scope, or "" if it is in. Same order as
+ * upstream's whyExcluded: ignore list, binary, the project's own excludes,
+ * extension, default path, deleted. `file` is { path, binary?, deleted? };
+ * `projectExclude` is the repo's `.orcacode-review.json` globs.
+ */
+export function excludeReason(file, gitignore, projectExclude = []) {
+ if (isIgnored(file.path, gitignore)) return "ignored";
+ if (file.binary) return "binary";
+ const lower = file.path.toLowerCase();
+ if (projectExclude.some((g) => globMatch(g.toLowerCase(), lower))) return "project_exclude";
+ const ext = extOf(file.path);
+ if (ext && !isAllowedExt(ext)) return "unsupported_ext";
+ if (isDefaultExcludedPath(file.path)) return "default_path";
+ if (file.deleted) return "deleted";
+ return "";
+}
+
+/**
+ * Split changed files into the reviewable and the excluded, each excluded
+ * entry carrying the reason. Extensionless files (Makefile, Dockerfile) pass
+ * the extension gate, as upstream: no extension is not an unsupported one.
+ */
+export function partition(files, repoDir, { exclude = [] } = {}) {
+ const gitignore = loadGitignorePatterns(repoDir);
+ const reviewable = [];
+ const excluded = [];
+ for (const f of files) {
+ const code = excludeReason(f, gitignore, exclude);
+ if (code) excluded.push({ path: f.path, code, reason: REASON[code] });
+ else reviewable.push(f);
+ }
+ return { files: reviewable, excluded };
+}
+
+// ------------------------------------------------------------------ rules ---
+
+// First-line signals for Objective-C in a ".m" file. MATLAB comments start
+// with "%" and a MATLAB file cannot begin with "/", so a C comment opener is
+// itself a signal. Not widened to a bare "#": Octave also uses ".m" and treats
+// "#" as a comment.
+const OBJC_PREFIXES = [
+ "#import", "#include", "#pragma", "#if", "#define",
+ "@import", "@interface", "@implementation", "@class", "@protocol",
+ "//", "/*",
+];
+
+function firstNonBlankLine(text) {
+ for (const line of text.split("\n")) {
+ const t = line.trim();
+ if (t) return t;
+ }
+ return "";
+}
+
+// Peek a ".m" file's first line at `ref` (so a commit that is not checked out
+// still resolves) or from the work tree. Any failure is "", which leaves the
+// path-based rule — MATLAB — in place.
+function peekFirstLine(relPath, repoDir, ref) {
+ try {
+ if (ref) {
+ const r = spawnSync("git", ["-c", "core.quotepath=false", "show", "--end-of-options", `${ref}:${relPath}`], {
+ cwd: repoDir,
+ encoding: "utf8",
+ timeout: 5000,
+ maxBuffer: 1 << 24,
+ });
+ return r.status === 0 ? firstNonBlankLine(r.stdout) : "";
+ }
+ return firstNonBlankLine(fs.readFileSync(path.join(repoDir, relPath), "utf8"));
+ } catch {
+ return "";
+ }
+}
+
+function sniffsAsObjC(relPath, repoDir, ref) {
+ if (!relPath.toLowerCase().endsWith(".m")) return false;
+ const line = peekFirstLine(relPath, repoDir, ref);
+ return !!line && OBJC_PREFIXES.some((p) => line.startsWith(p));
+}
+
+/**
+ * The review checklist for one path: { pattern, rule }. First matching
+ * pattern in system_rules.json wins, case-insensitively; none matching falls
+ * back to the default checklist with pattern "default". A ".m" file whose
+ * content sniffs as Objective-C gets objc.md while keeping the glob it matched
+ * as its pattern, exactly as upstream reports it.
+ */
+export function resolveRule(relPath, { repoDir = process.cwd(), ref = "", rules = [] } = {}) {
+ const { pathRules, defaultDoc, doc } = load();
+ const lower = relPath.toLowerCase();
+ let hit = { pattern: "default", docName: defaultDoc };
+ for (const pr of pathRules) {
+ if (pr.expanded.some((p) => globToRegExp(p).test(lower))) {
+ hit = { pattern: pr.pattern, docName: pr.doc };
+ break;
+ }
+ }
+ if (sniffsAsObjC(relPath, repoDir, ref)) hit.docName = "objc.md";
+ const system = doc(hit.docName);
+
+ // The project's own rule, from .orcacode-review.json, first match wins. By
+ // default it is ADDED to the bundled checklist — "also check X for API files"
+ // is what people mean — and only `replace: true` drops the bundled one.
+ const own = rules.find((r) => globMatch(r.path.toLowerCase(), lower));
+ if (own) {
+ return {
+ pattern: own.path,
+ source: "project",
+ rule: own.replace ? own.text : `${own.text}\n\n${system}`,
+ };
+ }
+ return { pattern: hit.pattern, source: "system", rule: system };
+}
+
+/**
+ * Files grouped by the checklist they resolve to, in the shape `ocr delegate
+ * rule --format json` emits: { group_id, source, pattern, files, rule }. Two
+ * files share a group only when pattern AND rule text coincide, so a group's
+ * pattern is true of every file in it.
+ */
+export function groupRules(paths, opts) {
+ const index = new Map();
+ const groups = [];
+ for (const p of paths) {
+ const { pattern, source, rule } = resolveRule(p, opts);
+ const key = `${source}\0${pattern}\0${rule}`;
+ let g = index.get(key);
+ if (!g) {
+ g = { group_id: groups.length + 1, source, pattern, files: [], rule };
+ index.set(key, g);
+ groups.push(g);
+ }
+ g.files.push(p);
+ }
+ return groups;
+}
diff --git a/bin/skill-tree.mjs b/bin/skill-tree.mjs
index 4ef63cf..fefee56 100644
--- a/bin/skill-tree.mjs
+++ b/bin/skill-tree.mjs
@@ -128,3 +128,29 @@ function pruneEmptyDirs(root) {
if (fs.readdirSync(abs).length === 0) fs.rmdirSync(abs);
}
}
+
+/**
+ * Remove a skill installed under a name this package USED to ship, next to
+ * where the renamed one is about to land. Returns true if something was removed.
+ *
+ * Renaming a skill without this leaves two copies on every machine that ever
+ * installed the old one, both matching the same user phrases — the agent then
+ * picks one at random, and the stale one wins half the time.
+ *
+ * Only OUR old skill is touched: the directory must hold a SKILL.md whose
+ * frontmatter `name:` is exactly the legacy name. A same-named directory that
+ * belongs to someone else is left where it is.
+ */
+export function retireLegacy(dest, legacyName) {
+ const legacy = path.join(path.dirname(dest), legacyName);
+ let head;
+ try {
+ head = fs.readFileSync(path.join(legacy, "SKILL.md"), "utf8").slice(0, 2048);
+ } catch {
+ return false;
+ }
+ const m = /^---\n(?:[^\n]*\n)*?name:\s*([^\n]+)\n/.exec(head);
+ if (!m || m[1].trim() !== legacyName) return false;
+ fs.rmSync(legacy, { recursive: true, force: true });
+ return true;
+}
diff --git a/docs/demo-install.gif b/docs/demo-install.gif
new file mode 100644
index 0000000..3fbf4bc
Binary files /dev/null and b/docs/demo-install.gif differ
diff --git a/docs/demo-review.gif b/docs/demo-review.gif
new file mode 100644
index 0000000..0f9e320
Binary files /dev/null and b/docs/demo-review.gif differ
diff --git a/docs/demo-setup.gif b/docs/demo-setup.gif
new file mode 100644
index 0000000..e2074eb
Binary files /dev/null and b/docs/demo-setup.gif differ
diff --git a/docs/demo.gif b/docs/demo.gif
deleted file mode 100644
index 69f6e14..0000000
Binary files a/docs/demo.gif and /dev/null differ
diff --git a/docs/demo.mp4 b/docs/demo.mp4
deleted file mode 100644
index 77a8bc8..0000000
Binary files a/docs/demo.mp4 and /dev/null differ
diff --git a/package-lock.json b/package-lock.json
index ba7b5a6..277f116 100644
--- a/package-lock.json
+++ b/package-lock.json
@@ -1,27 +1,12 @@
{
"name": "@orcarouter/code-review",
- "version": "1.0.2",
+ "version": "2.1.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "@orcarouter/code-review",
- "version": "1.0.2",
- "license": "MIT",
- "dependencies": {
- "@orcarouter/code-review": "^1.0.2"
- },
- "bin": {
- "orcacode-review": "bin/orcacode-review.mjs"
- },
- "engines": {
- "node": ">=18.17"
- }
- },
- "node_modules/@orcarouter/code-review": {
- "version": "1.0.2",
- "resolved": "https://registry.npmjs.org/@orcarouter/code-review/-/code-review-1.0.2.tgz",
- "integrity": "sha512-PsM9S+qAN6bDLzRU600fGNrSzjy+DfGMcel49iFFe8FsbZ4JRtoWbfMfEq5yemRD3KbR3FbkBW/xB1GEHhs+FQ==",
+ "version": "2.1.0",
"license": "MIT",
"bin": {
"orcacode-review": "bin/orcacode-review.mjs"
diff --git a/package.json b/package.json
index 263741d..a6d3b07 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,6 @@
{
"name": "@orcarouter/code-review",
- "version": "1.5.0",
+ "version": "2.1.0",
"description": "One-command installer for OrcaCode Review — AI pull-request review powered by OrcaRouter.",
"bin": {
"orcacode-review": "bin/orcacode-review.mjs"
@@ -12,6 +12,10 @@
"files": [
"bin/",
"skills/",
+ "rules/",
+ "vendor/",
+ "scripts/severity.mjs",
+ "scripts/postfilter.mjs",
"NOTICE"
],
"scripts": {
@@ -38,8 +42,5 @@
"license": "MIT",
"publishConfig": {
"access": "public"
- },
- "dependencies": {
- "@orcarouter/code-review": "^1.0.2"
}
}
diff --git a/recipes/orcacode-review.dsl.yaml b/recipes/orcacode-review.dsl.yaml
index dc0b6cc..79727a5 100644
--- a/recipes/orcacode-review.dsl.yaml
+++ b/recipes/orcacode-review.dsl.yaml
@@ -39,16 +39,31 @@
# one pass — is an edit to this file in your own workspace rather than a wait for
# an Action release.
#
-# THE MODEL IS FREELY CHOOSABLE, and this is the line to edit. Swap in anything your
-# OrcaRouter workspace can reach (openai/gpt-5.5, anthropic/claude-opus-4-8,
-# z-ai/glm-5.1, deepseek/deepseek-v4-pro).
+# TWO CALLS, TWO LINES TO EDIT. The Action asks this router for two things, both
+# with the alias as the `model` rather than a model name:
#
-# deepseek-v4-flash below is what setup provisions, so this file and a freshly
-# created router agree. A stronger model here costs more per review and is the single
-# highest-leverage change you can make to review quality — nothing else in this
-# recipe affects what the reviewer can see or say.
+# the review — carries no angle, so it takes `default:`
+# the L2 judge — stamps `x-cr-lens: judge`, so the rule below claims it
+#
+# Swap in anything your OrcaRouter workspace can reach (openai/gpt-5.5,
+# anthropic/claude-opus-4-8, z-ai/glm-5.1, z-ai/glm-5.3). The default
+# is the single highest-leverage change you can make to review quality — nothing
+# else in this recipe affects what the reviewer can see or say.
+#
+# THE JUDGE MUST NOT NAME THE DEFAULT'S MODEL. The judge scores what the review
+# found and drops what it cannot support; on the reviewer's own model it agrees
+# with itself, so the pass goes inert while still reporting success. Deleting the
+# rule does not disable the judge — the Action runs it either way — it only sends
+# it to `default:`, which is the one place it must not go. If you change the
+# default, change the judge too.
+#
+# These values are what setup provisions, so this file and a freshly created
+# router agree.
version: 1
-
+rules:
+ - id: judge
+ when: 'headers["x-cr-lens"] == "judge"'
+ use: { model: "z-ai/glm-5.3" }
default:
- model: "deepseek/deepseek-v4-flash"
+ model: "deepseek/deepseek-v4-flash-0731"
diff --git a/rules/output-shape.md b/rules/output-shape.md
index daea7d1..af56494 100644
--- a/rules/output-shape.md
+++ b/rules/output-shape.md
@@ -1,7 +1,27 @@
-MANDATORY OUTPUT SHAPE: after the severity tag, open every comment with a SHORT TITLE in **bold** — at most about ten words naming the defect — then a blank line, then the explanation. The title says what is WRONG, not what to do, and takes no full stop.
+MANDATORY OUTPUT SHAPE. Every comment has three parts, in this order, separated by blank lines:
+
+1. the severity tag, then a SHORT TITLE in **bold**
+2. the explanation
+3. a final paragraph opening with **Fix:**
[P1] **fetchAll drops the last item of every page**
- The loop bound `i < items.length - 1` never pushes the final element, so each page contributes one row fewer than it holds. Use `i < items.length`.
+ The loop bound `i < items.length - 1` never pushes the final element, so each page contributes one row fewer than it holds. Callers paginating a 50-row table receive 49.
-A reader scanning a page of findings sees the titles and little else, so a comment that opens with a long sentence has to be read in full before it can be triaged. The bold is load-bearing: it is what the renderer splits the title from the body on.
+ **Fix:** use `i < items.length`.
+
+THE TITLE NAMES THE DEFECT. It is a statement of what is WRONG — not an instruction, not a summary of the remedy. At most about ten words. No full stop. No semicolon, and no second clause: if you need "and" or ";" to fit it, you are writing two findings or writing the fix into the title.
+
+ WRONG **Serialize the budget-cap check per key; the atomicity claim only holds on SQLite**
+ RIGHT **Budget cap is not atomic on Postgres**
+
+ WRONG **Add a null check before dereferencing the session**
+ RIGHT **Session is dereferenced before the null check**
+
+The first of each pair starts with a verb telling the reader what to do, which is what the **Fix:** paragraph is for. The second names the defect, which is what a reader triaging thirty findings needs to see.
+
+THE EXPLANATION IS PROSE, NOT A WALL. Lead with the consequence — what actually breaks, for whom. Then the evidence: the specific code path, condition, or value that causes it. Break the paragraph when the subject changes. Four short paragraphs are read; one twelve-line paragraph is skipped.
+
+THE FIX IS ITS OWN PARAGRAPH, and it is the last thing in the comment. Concrete: the expression, the call, the ordering. If there is genuinely no single fix, say what the options trade off — but still in its own **Fix:** paragraph, so a reader can always find it in the same place.
+
+Every one of these serves the same reader: someone scanning a page of findings who must decide, per finding, whether to act now. They see titles and little else. A comment that opens with a long sentence has to be read in full before it can be triaged, and a fix buried in the last clause of a dense paragraph will be missed. The bold is also load-bearing mechanically: it is what the renderer splits the title from the body on.
diff --git a/scripts/check-result.mjs b/scripts/check-result.mjs
index 1fae05c..b7030bf 100644
--- a/scripts/check-result.mjs
+++ b/scripts/check-result.mjs
@@ -10,6 +10,13 @@
// all mean we do NOT have a trustworthy result. Converting those into a clean
// "no issues found" pass would let a bad key, gateway outage, or CLI crash
// silently clear a PR — so we surface them as an unavailable review instead.
+//
+// EACH REASON PRINTS A DISTINCT `reason=` TAG, because the caller collapses all
+// of them into one "no usable result" error and an operator reading only that
+// cannot tell an engine crash from a partial run. That ambiguity cost a real
+// support round-trip: a run whose judge logged a 502 just before this failed was
+// read as "the judge threw the review away", when the judge is soft and the
+// engine was what failed.
import fs from "node:fs";
@@ -17,7 +24,7 @@ const [file, rcRaw] = process.argv.slice(2);
const rc = Number(rcRaw || "0");
if (rc !== 0) {
- console.error(`review unavailable: engine exited ${rc}`);
+ console.error(`reason=engine-exit review unavailable: engine exited ${rc}`);
process.exit(1);
}
@@ -25,18 +32,18 @@ let parsed;
try {
parsed = JSON.parse(fs.readFileSync(file, "utf8"));
} catch (e) {
- console.error(`review unavailable: no parseable result in ${file} (${e.message})`);
+ console.error(`reason=unparseable review unavailable: no parseable result in ${file} (${e.message})`);
process.exit(1);
}
if (!Array.isArray(parsed.comments)) {
- console.error("review unavailable: result has no `comments` array");
+ console.error("reason=no-comments review unavailable: result has no `comments` array");
process.exit(1);
}
const warnings = Array.isArray(parsed.warnings) ? parsed.warnings : [];
if (warnings.length > 0) {
- console.error(`review partial: ${warnings.length} warning(s) — ${warnings.join("; ")}`);
+ console.error(`reason=partial review partial: ${warnings.length} warning(s) — ${warnings.join("; ")}`);
process.exit(1);
}
diff --git a/scripts/check-result.test.mjs b/scripts/check-result.test.mjs
index 8f9436b..5b2b44e 100644
--- a/scripts/check-result.test.mjs
+++ b/scripts/check-result.test.mjs
@@ -16,7 +16,7 @@
// it.
import { spawnSync } from "node:child_process";
-import { mkdtempSync, writeFileSync, rmSync } from "node:fs";
+import { mkdtempSync, writeFileSync, rmSync, readFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join, dirname } from "node:path";
import { fileURLToPath } from "node:url";
@@ -175,3 +175,74 @@ describe("output contract", () => {
assert.equal(run(writeResult("{bad"), "0").stdout, "");
});
});
+
+// THE PRECISION GATE AND THIS SCRIPT JUDGE THE SAME FIELD, so they must never
+// disagree about it. action.yml skips L1+L2 when the engine result looks
+// partial, on the reasoning that CHECK will reject it anyway — which holds only
+// while both read `warnings` the same way.
+//
+// They did not. The gate tested `(r.warnings||[]).length`, so a non-array value
+// like the string "oops" measured 4 and read as partial; CHECK normalizes any
+// non-array to [] and publishes. Skip plus publish is the combination that
+// ships the engine's raw findings with no L1 and no L2 under
+// `precision-filter: true` — unjudged output arriving as a green review, i.e.
+// exactly what the fail-closed judge branch exists to stop.
+//
+// Which sets the direction this predicate has to fail in: SKIPPING is the
+// branch that can publish unjudged, so it may only be taken when the result is
+// DEFINITELY partial. The real predicate is extracted from action.yml rather
+// than restated here, so editing one without the other fails this test.
+describe("the action's skip-the-filter predicate agrees with this script", () => {
+ const actionYml = readFileSync(join(SCRIPTS, "..", "action.yml"), "utf8");
+
+ // The `node -e "…"` program inside the precision gate's `if`.
+ const predicate = (() => {
+ const m = actionYml.match(/&& node -e "(const r=require\(process\.argv\[1\]\)[^"]*warnings[^"]*)"/);
+ assert.ok(m, "the precision gate's warnings predicate must exist in action.yml");
+ return m[1];
+ })();
+
+ const filtersRun = (file) => spawnSync("node", ["-e", predicate, file], { encoding: "utf8" }).status === 0;
+
+ // Every shape an engine (or a corrupted write) can put in `warnings`.
+ const shapes = {
+ "a real non-empty array": ["conventions lens failed"],
+ "an empty array": [],
+ "the field absent": undefined,
+ "null": null,
+ "a string": "oops",
+ "a number": 5,
+ "an object": { a: 1 },
+ };
+
+ for (const [name, warnings] of Object.entries(shapes)) {
+ test(`${name}: the filter is never skipped on a result CHECK will publish`, () => {
+ const body = { comments: [{ content: "[P2] a finding" }] };
+ if (warnings !== undefined) body.warnings = warnings;
+ const file = writeResult(body);
+
+ const skipped = !filtersRun(file);
+ const published = run(file, "0").status === 0;
+ assert.ok(
+ !(skipped && published),
+ `unjudged publish: filter skipped AND CHECK published for warnings=${JSON.stringify(warnings)}`,
+ );
+ });
+ }
+
+ test("a genuinely partial result is the one case that skips — and CHECK stops it", () => {
+ const file = writeResult({ comments: [{ content: "[P2] a" }], warnings: ["lens failed"] });
+ assert.equal(filtersRun(file), false, "a partial result must not spend a judge call");
+ const r = run(file, "0");
+ assert.equal(r.status, 1);
+ assert.match(r.stderr, /reason=partial/);
+ });
+
+ test("a clean result with findings still enters the filter", () => {
+ assert.equal(filtersRun(writeResult({ comments: [{ content: "[P2] a" }], warnings: [] })), true);
+ });
+
+ test("no findings means nothing to filter", () => {
+ assert.equal(filtersRun(writeResult({ comments: [], warnings: [] })), false);
+ });
+});
diff --git a/scripts/diff-guard.mjs b/scripts/diff-guard.mjs
index 0d43838..49c1cff 100644
--- a/scripts/diff-guard.mjs
+++ b/scripts/diff-guard.mjs
@@ -13,8 +13,10 @@
// not an error:
// {"decision":"review"|"skip","reason":"…","size_kb":,"files":}
//
-// Defaults: 512 KB / 300 files (surfaced as the action's `max-diff-kb` /
-// `max-diff-files` inputs). The size check is decided from stat() alone — an
+// Defaults: 5000 KB / 2000 files (surfaced as the action's `max-diff-kb` /
+// `max-diff-files` inputs, and they must agree — a caller that passes an
+// empty flag lands here, so a stale number in this file would quietly give a
+// different limit than the documented one). The size check is decided from stat() alone — an
// oversized diff is never read into memory, so a size-skip reports files: 0
// (not counted). Only an under-cap diff is read to count files. `files`
// counts `diff --git` file headers — in a unified diff every content line is
@@ -39,8 +41,8 @@ for (let i = 0; i < argv.length; i += 1) {
else if (argv[i] === "--max-kb") maxKbRaw = argv[++i];
else if (argv[i] === "--max-files") maxFilesRaw = argv[++i];
}
-const maxKb = positive(maxKbRaw, 512);
-const maxFiles = positive(maxFilesRaw, 300);
+const maxKb = positive(maxKbRaw, 5000);
+const maxFiles = positive(maxFilesRaw, 2000);
function decide() {
if (!diffPath) {
diff --git a/scripts/diff-guard.test.mjs b/scripts/diff-guard.test.mjs
index b238047..64eefb8 100644
--- a/scripts/diff-guard.test.mjs
+++ b/scripts/diff-guard.test.mjs
@@ -6,7 +6,7 @@
// throwing anywhere below would fail the test), and anything unreadable fails
// OPEN to "review" so a guard glitch can never silently disable the review.
//
-// Limits: --max-kb (default 512) on byte size, --max-files (default 300) on
+// Limits: --max-kb (default 5000) on byte size, --max-files (default 2000) on
// `diff --git` headers. AT a limit still reviews; only strictly-over skips.
import { execFileSync } from "node:child_process";
@@ -107,20 +107,22 @@ describe("file-count threshold (--max-files)", () => {
});
});
-describe("defaults (512 KB / 300 files)", () => {
- test("a >512 KB diff skips with no flags given", () => {
- const out = run(["--diff", writeDiff(`diff --git a/a b/a\n+${"x".repeat(513 * 1024)}\n`)]);
+describe("defaults (5000 KB / 2000 files)", () => {
+ test("a >5000 KB diff skips with no flags given", () => {
+ const out = run(["--diff", writeDiff(`diff --git a/a b/a\n+${"x".repeat(5001 * 1024)}\n`)]);
assert.equal(out.decision, "skip");
- assert.match(out.reason, /over the 512 KB limit/);
+ assert.match(out.reason, /over the 5000 KB limit/);
});
- test("301 files skips, 300 reviews, with no flags given", () => {
+ test("2001 files skips, 2000 reviews, with no flags given", () => {
+ // ~90 bytes per block, so 2001 of them is ~180 KB — far under the size
+ // limit, which is what makes this a file-COUNT test and not a size one.
let many = "";
- for (let i = 0; i < 301; i += 1) many += fileBlock(i);
+ for (let i = 0; i < 2001; i += 1) many += fileBlock(i);
assert.equal(run(["--diff", writeDiff(many)]).decision, "skip");
let exactly = "";
- for (let i = 0; i < 300; i += 1) exactly += fileBlock(i);
+ for (let i = 0; i < 2000; i += 1) exactly += fileBlock(i);
assert.equal(run(["--diff", writeDiff(exactly)]).decision, "review");
});
@@ -130,6 +132,60 @@ describe("defaults (512 KB / 300 files)", () => {
});
});
+// THE FALLBACK AND THE DOCUMENTED DEFAULT ARE ONE NUMBER IN TWO PLACES, and
+// this pins them together. action.yml passes `--max-kb "$MAX_KB"` QUOTED, so a
+// workspace that sets `max-diff-kb: ""` reaches this script as an empty
+// argument and lands on the fallback — which means a stale number here would
+// silently enforce a limit nobody documented, and no test would notice.
+//
+// Read out of action.yml rather than restated, so editing one side without the
+// other fails here instead of in production.
+describe("the script's fallbacks match action.yml's documented defaults", () => {
+ // NEWLINES NORMALIZED, because this repo has no `.gitattributes` and
+ // `core.autocrlf` is the Windows default: every Windows checkout gets a CRLF
+ // action.yml, and the anchors below are written with bare "\\n". Without this the
+ // test passes on Linux CI and fails on every Windows machine — a shape worth
+ // avoiding on purpose, since CI green would be read as "works".
+ const actionYml = () =>
+ readFileSync(join(dirname(fileURLToPath(import.meta.url)), "..", "action.yml"), "utf8").replace(
+ /\r\n/g,
+ "\n",
+ );
+
+ // The block for ONE input: from its key to the next key at the same indent.
+ // Bounded rather than a non-greedy scan of the whole file, so an input with
+ // no numeric default fails the assert instead of quietly borrowing a later
+ // input's.
+ function inputDefault(name) {
+ const yml = actionYml();
+ const start = yml.indexOf(`\n ${name}:\n`);
+ assert.notEqual(start, -1, `${name} must exist in action.yml`);
+ const rest = yml.slice(start + 1);
+ const next = rest.search(/\n [a-z][a-z0-9-]*:\n/);
+ const block = next === -1 ? rest : rest.slice(0, next);
+ const m = block.match(/\n default: "(\d+)"/);
+ assert.ok(m, `${name} must have a numeric default in action.yml`);
+ return Number(m[1]);
+ }
+
+ test("an empty --max-kb enforces action.yml's max-diff-kb", () => {
+ const documented = inputDefault("max-diff-kb");
+ const oversized = `diff --git a/a b/a\n+${"x".repeat((documented + 1) * 1024)}\n`;
+ const out = run(["--diff", writeDiff(oversized), "--max-kb", ""]);
+ assert.equal(out.decision, "skip");
+ assert.match(out.reason, new RegExp(`over the ${documented} KB limit`));
+ });
+
+ test("an empty --max-files enforces action.yml's max-diff-files", () => {
+ const documented = inputDefault("max-diff-files");
+ let many = "";
+ for (let i = 0; i < documented + 1; i += 1) many += fileBlock(i);
+ const out = run(["--diff", writeDiff(many), "--max-files", ""]);
+ assert.equal(out.decision, "skip");
+ assert.match(out.reason, new RegExp(`over the ${documented}-file limit`));
+ });
+});
+
describe("action.yml wiring (oversized-diff outcome)", () => {
const actionYml = () =>
readFileSync(join(dirname(fileURLToPath(import.meta.url)), "..", "action.yml"), "utf8");
diff --git a/scripts/harness.test.mjs b/scripts/harness.test.mjs
new file mode 100644
index 0000000..7999c78
--- /dev/null
+++ b/scripts/harness.test.mjs
@@ -0,0 +1,759 @@
+// The harness surface — the half of OrcaCode Review an outside agent drives.
+//
+// What is worth testing here is the CONTRACT, not the prose: the result shape
+// the pipeline shares with the Action, the gate's fail-safe, the porcelain
+// parsing that decides which files are in scope, and the promises the plan makes
+// to a model that cannot ask a follow-up question.
+
+import test from "node:test";
+import assert from "node:assert/strict";
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { execFileSync, spawnSync } from "node:child_process";
+import { fileURLToPath } from "node:url";
+
+import { makeT } from "../bin/i18n.mjs";
+import {
+ LANGUAGE_NAMES,
+ MODE_REASON,
+ SCHEMA_VERSION,
+ WORK_DIR,
+ anchorLine,
+ buildPlan,
+ gate,
+ renderPlan,
+ renderReport,
+ renderReportMarkdown,
+ resolveMode,
+ resolvePr,
+ selectFiles,
+ splitFinding,
+ validateResult,
+} from "../bin/harness.mjs";
+
+// ------------------------------------------------------------- validation ---
+
+test("a bare `line` is widened to the start/end pair the pipeline speaks", () => {
+ // postfilter.mjs, judge.mjs, exhaustive-merge.mjs and the Action's poster all
+ // read start_line/end_line. A finding carrying only `line` would keep a stale
+ // anchor through a re-home, so the widening has to happen at the boundary.
+ const r = validateResult({ comments: [{ path: "a.ts", line: 7, content: "[P1] x" }] });
+ assert.equal(r.ok, true);
+ assert.deepEqual(r.comments[0], { path: "a.ts", start_line: 7, end_line: 7, content: "[P1] x" });
+ assert.equal("line" in r.comments[0], false, "the alias must not survive alongside the pair");
+});
+
+test("the short hand-back form — four fields, no warnings key — is accepted", () => {
+ // What the plan actually asks an agent to write. If this ever stops being
+ // valid, every review starts failing at the last step.
+ const short = {
+ comments: [{ path: "a.ts", line: 7, existing_code: " x();", content: "[P1] **T**\n\nbody" }],
+ };
+ const r = validateResult(short);
+ assert.equal(r.ok, true);
+ assert.deepEqual(
+ { s: r.comments[0].start_line, e: r.comments[0].end_line, line: r.comments[0].line },
+ { s: 7, e: 7, line: undefined },
+ "the bare line must be widened and consumed, not left alongside the pair",
+ );
+});
+
+test("an explicit start/end range is left exactly as written", () => {
+ const c = { path: "a.ts", start_line: 4, end_line: 9, content: "[P2] x" };
+ assert.deepEqual(validateResult({ comments: [c] }).comments[0], c);
+});
+
+test("a partial review is rejected rather than passed", () => {
+ // The whole point of failing closed: "some files could not be read" must never
+ // render as a clean review, or a broken run silently clears a change.
+ const r = validateResult({ comments: [], warnings: ["could not read src/b.ts"] });
+ assert.equal(r.ok, false);
+ assert.match(r.error, /partial/);
+});
+
+test("a finding without a path or content is malformed, not silently dropped", () => {
+ assert.match(validateResult({ comments: [{ content: "[P1] x" }] }).error, /`path`/);
+ assert.match(validateResult({ comments: [{ path: "a.ts" }] }).error, /`content`/);
+});
+
+test("a result that is not an object, or has no comments array, is unusable", () => {
+ assert.equal(validateResult(null).ok, false);
+ assert.equal(validateResult({}).ok, false);
+ assert.equal(validateResult({ comments: "none" }).ok, false);
+});
+
+// ------------------------------------------------------------------- gate ---
+
+const finding = (content) => ({ path: "a.ts", content });
+
+test("the gate blocks on the configured severities and nothing else", () => {
+ const r = gate([finding("[P0] a"), finding("[P2] b"), finding("[P3] c")], "P0,P1");
+ assert.equal(r.blocked, true);
+ assert.equal(r.blocking.length, 1);
+ assert.deepEqual(r.counts, { P0: 1, P1: 0, P2: 1, P3: 1 });
+});
+
+test("an untagged finding counts as P1, so a missing tag escalates", () => {
+ // severity.mjs owns this rule; the test is here because the harness is a new
+ // caller of it and the fail-safe is the reason an agent can be trusted to tag.
+ const r = gate([finding("no tag at all")], "P0,P1");
+ assert.equal(r.blocked, true);
+ assert.equal(r.counts.P1, 1);
+});
+
+test("an empty --block-on never blocks", () => {
+ const r = gate([finding("[P0] a")], "");
+ assert.equal(r.blocked, false);
+ assert.deepEqual(r.wanted, []);
+});
+
+test("block-on is case- and space-insensitive", () => {
+ assert.equal(gate([finding("[P0] a")], " p0 , p1 ").blocked, true);
+});
+
+// ---------------------------------------------------------------- anchors ---
+
+test("the anchor prefers end_line, falls back to start_line, and may be absent", () => {
+ assert.equal(anchorLine({ start_line: 3, end_line: 9 }), 9);
+ assert.equal(anchorLine({ start_line: 3 }), 3);
+ // Cleared by a re-home whose new line could not be resolved: rendering the
+ // stale number would point at unrelated code in the new file.
+ assert.equal(anchorLine({ start_line: null, end_line: null }), null);
+ assert.equal(anchorLine({}), null);
+});
+
+test("the bold title is split off the body, and a shapeless finding still renders", () => {
+ const ok = splitFinding("[P1] **Loop drops the last item**\n\nThe bound is off by one.");
+ assert.deepEqual(ok, { title: "Loop drops the last item", rest: "The bound is off by one." });
+
+ const shapeless = splitFinding("[P3] just a sentence, no bold");
+ assert.equal(shapeless.title, "just a sentence, no bold");
+});
+
+// ------------------------------------------------------------ output shape ---
+
+// rules/output-shape.md is injected into BOTH prompts — action.yml's OUTPUT_SHAPE
+// and this harness's rubric — so the Action's PR comments and a local review are
+// shaped by the same file. Nothing else guarded it.
+const SHAPE = fs.readFileSync(new URL("../rules/output-shape.md", import.meta.url), "utf8");
+
+test("the output shape mandates a title, an explanation, and a separate fix", () => {
+ assert.match(SHAPE, /\*\*Fix:\*\*/);
+ assert.match(SHAPE, /THE FIX IS ITS OWN PARAGRAPH/);
+ assert.match(SHAPE, /last thing in the comment/);
+});
+
+test("the output shape shows a wrong/right pair for the title, not just a rule", () => {
+ // Observed on a real PR: a P1 titled "Serialize the budget-cap check per key;
+ // the atomicity claim only holds on SQLite" — an instruction, sixteen words,
+ // with a semicolon. The prose rule was already there; the counter-example is
+ // what makes it land.
+ assert.match(SHAPE, /WRONG\s+\*\*/);
+ assert.match(SHAPE, /RIGHT\s+\*\*/);
+ assert.match(SHAPE, /No semicolon/);
+ assert.match(SHAPE, /statement of what is WRONG/);
+});
+
+test("the plan hands the reviewer that exact file, not a paraphrase of it", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "a.ts"), "export const a = 1;\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const text = renderPlan(buildPlan({ from: "main~1", to: "HEAD" }, dir));
+ assert.ok(text.includes("THE FIX IS ITS OWN PARAGRAPH"), "the local review must be shaped by the shared rule");
+});
+
+// --------------------------------------------------------- submit, for real ---
+
+const CLI = fileURLToPath(new URL("../bin/orcacode-review.mjs", import.meta.url));
+
+// The bug this guards: postfilter.mjs reads a FILE, so handing it the result as
+// the agent wrote it — with a bare `line` — returned every finding anchorless.
+// Only an end-to-end run catches that; the widening in validateResult looks
+// correct in isolation and is simply bypassed.
+test("a finding written in the short form keeps its line through the position check", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "x.js"), "export function f(a) {\n return a.b;\n}\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "x");
+
+ const work = path.join(dir, WORK_DIR);
+ fs.mkdirSync(work, { recursive: true });
+ fs.writeFileSync(
+ path.join(work, "result.json"),
+ JSON.stringify({
+ comments: [{ path: "x.js", line: 2, existing_code: " return a.b;", content: "[P0] **Null deref**\n\nbody" }],
+ }),
+ );
+ fs.writeFileSync(
+ path.join(work, "plan.json"),
+ JSON.stringify({ ground_truth_ref: execFileSync("git", ["rev-parse", "HEAD"], { cwd: dir, encoding: "utf8" }).trim() }),
+ );
+
+ let out = "";
+ let code = 0;
+ try {
+ out = execFileSync(process.execPath, [CLI, "review", "submit", "--format", "md", "--lang", "en"], {
+ cwd: dir,
+ encoding: "utf8",
+ stdio: ["ignore", "pipe", "ignore"],
+ });
+ } catch (e) {
+ out = e.stdout || "";
+ code = e.status;
+ }
+
+ assert.equal(code, 0, "a review that ran exits 0; the verdict is in the report");
+ assert.match(out, /❌ BLOCKED/, "a P0 must block");
+ assert.match(out, /`x\.js:2`/, "the anchor must survive postfilter, not be dropped to a bare path");
+ assert.ok(!fs.existsSync(path.join(work, "result.normalized.json")), "the scratch file must not be left behind");
+ assert.ok(fs.existsSync(path.join(work, "report.md")), "the markdown report is always saved");
+});
+
+// Runs the real CLI and reports what a caller actually observes.
+function submit(dir, args) {
+ const r = spawnSync(process.execPath, [CLI, "review", "submit", "--lang", "en", ...args], {
+ cwd: dir,
+ encoding: "utf8",
+ });
+ return { code: r.status, out: r.stdout || "", err: r.stderr || "" };
+}
+
+// A blocking P0 seeded into a repo, ready to submit.
+function repoWithABlockingFinding(t, result) {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "x.js"), "export function f(a) {\n return a.b;\n}\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "x");
+
+ const work = path.join(dir, WORK_DIR);
+ fs.mkdirSync(work, { recursive: true });
+ fs.writeFileSync(
+ path.join(work, "result.json"),
+ JSON.stringify(
+ result ?? {
+ comments: [{ path: "x.js", line: 2, existing_code: " return a.b;", content: "[P0] **Null deref**\n\nbody" }],
+ },
+ ),
+ );
+ fs.writeFileSync(
+ path.join(work, "plan.json"),
+ JSON.stringify({ ground_truth_ref: execFileSync("git", ["rev-parse", "HEAD"], { cwd: dir, encoding: "utf8" }).trim() }),
+ );
+ return { dir, work };
+}
+
+// Why the default is 0: the local harness exists to tell an agent what the
+// bugs are, and the agent reads the report. Its shell tool stamps any non-zero
+// exit "Error" — which, twice, turned a working review into an apparent crash
+// in front of a real user.
+test("a blocked review exits 0 by default and carries the verdict in the report", (t) => {
+ const { dir, work } = repoWithABlockingFinding(t);
+ const r = submit(dir, ["--format", "md"]);
+ assert.equal(r.code, 0, "the process must not fail on a verdict");
+ assert.match(r.out, /❌ BLOCKED/, "the verdict lives in the report, not the exit code");
+ assert.match(fs.readFileSync(path.join(work, "report.md"), "utf8"), /❌ BLOCKED/);
+});
+
+test("--fail-on-block is the opt-in that turns the verdict into a status", (t) => {
+ const { dir } = repoWithABlockingFinding(t);
+ const r = submit(dir, ["--format", "md", "--fail-on-block"]);
+ assert.equal(r.code, 1, "a hook or CI step that asked for it gets the 1");
+ assert.match(r.out, /❌ BLOCKED/, "and still gets the report");
+});
+
+test("the machine-readable verdict does not follow the exit code", (t) => {
+ const { dir } = repoWithABlockingFinding(t);
+ const r = submit(dir, ["--format", "json"]);
+ assert.equal(r.code, 0);
+ assert.equal(JSON.parse(r.out).blocked, true);
+});
+
+test("the default terminal report says BLOCKED without failing the process", (t) => {
+ const { dir } = repoWithABlockingFinding(t);
+ const r = submit(dir, []);
+ assert.equal(r.code, 0);
+ assert.match(r.out, /❌ BLOCKED/, "the terminal renderer shouts it; the markdown one does not");
+});
+
+// The one thing the 0 default must never cover. A masked 2 turns "the review
+// did not happen" into "the review passed", which is the failure this pipeline
+// exists to prevent.
+test("an unusable result still exits 2", (t) => {
+ const { dir } = repoWithABlockingFinding(t, { comments: [], warnings: ["could not read src/a.ts"] });
+ const r = submit(dir, ["--format", "md"]);
+ assert.equal(r.code, 2, "a partial review is a failure, not a verdict");
+ assert.ok(!/✅ PASSED/.test(r.out), "and it must never render as a pass");
+});
+
+test("a clean review exits 0 with or without --fail-on-block", (t) => {
+ const { dir } = repoWithABlockingFinding(t, { comments: [] });
+ assert.equal(submit(dir, ["--format", "md"]).code, 0);
+ assert.equal(submit(dir, ["--format", "md", "--fail-on-block"]).code, 0);
+});
+
+// ------------------------------------------------------------- language ---
+
+test("the plan tells the reviewer which language to write findings in", (t) => {
+ // The rubric is English and shared with the Action; the findings are for a
+ // person, and that person asked in their own language.
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "a.ts"), "1\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const zh = renderPlan(buildPlan({ from: "main~1", to: "HEAD", language: "zh" }, dir));
+ assert.match(zh, /## Language/);
+ assert.match(zh, /in 简体中文 \(Simplified Chinese\)\./);
+ assert.match(zh, /review submit --format md --lang zh/, "the hand-back command carries the same language");
+ // The structure rule survives the language change.
+ assert.match(zh, /severity tag stays exactly `\[P0\]`…`\[P3\]`/);
+
+ const dflt = buildPlan({ from: "main~1", to: "HEAD" }, dir);
+ assert.equal(dflt.language, "en");
+ const bogus = buildPlan({ from: "main~1", to: "HEAD", language: "tlh" }, dir);
+ assert.equal(bogus.language, "en", "an unknown language falls back rather than being echoed into a prompt");
+ for (const code of ["en", "zh", "ja", "ko"]) assert.ok(LANGUAGE_NAMES[code], `a name for ${code}`);
+});
+
+test("the report skeleton follows the language, the findings are left alone", () => {
+ const comments = [mdFinding("a.ts", 1, "[P1] **空指针解引用**\n\n调用方传 undefined 会崩。\n\n**修复:**先判空。")];
+ const result = gate(comments, "P0,P1");
+ const zh = renderReportMarkdown(comments, result, { t: makeT("zh") });
+ assert.match(zh, /\*\*❌ 已拦截\*\* — 共 1 条发现,其中 1 条达到 P0, P1 级别。/, "and a Chinese full stop, not a Latin one");
+ assert.match(zh, /空指针解引用/, "the finding text is the agent's and is not touched");
+ const en = renderReportMarkdown(comments, result);
+ assert.match(en, /\*\*❌ BLOCKED\*\* — 1 of 1 finding at P0, P1\./, "no translator means English");
+
+ const plain = { bold: (s) => s, dim: (s) => s, red: (s) => s, green: (s) => s, yellow: (s) => s, cyan: (s) => s };
+ const term = renderReport(comments, result, { color: plain, t: makeT("zh") }).split("\n");
+ assert.match(term[0], /❌ 已拦截\s+共 1 条发现/);
+ assert.match(term[1], /这是评审结果,不是程序崩溃/);
+});
+
+test("--lang reaches the submitted report end to end", (t) => {
+ const { dir } = repoWithABlockingFinding(t);
+ const r = spawnSync(process.execPath, [CLI, "review", "submit", "--format", "md", "--lang", "zh"], {
+ cwd: dir,
+ encoding: "utf8",
+ });
+ assert.equal(r.status, 0);
+ assert.match(r.stdout, /\*\*❌ 已拦截\*\*/);
+});
+
+// ------------------------------------------------------ markdown reporting ---
+
+const mdFinding = (path, line, content) => ({ path, start_line: line, end_line: line, content });
+
+test("the terminal report leads with the verdict, because the tail gets truncated", () => {
+ // Agent shells cut the middle out of long output. With the verdict at the
+ // bottom, all the reader is left with is the top of a findings list and no
+ // idea whether any of it blocks.
+ const plain = { bold: (s) => s, dim: (s) => s, red: (s) => s, green: (s) => s, yellow: (s) => s, cyan: (s) => s };
+ const comments = [mdFinding("a.ts", 1, "[P1] **High**\n\nbody"), mdFinding("a.ts", 2, "[P3] **Nit**\n\nbody")];
+
+ const blocked = renderReport(comments, gate(comments, "P0,P1"), { color: plain }).split("\n");
+ assert.match(blocked[0], /❌ BLOCKED\s+1 of 2 findings at P0,P1/);
+ assert.match(blocked[1], /this is a review result, not a crash/);
+
+ const passed = renderReport(comments, gate(comments, "P0"), { color: plain }).split("\n");
+ assert.match(passed[0], /✅ PASSED\s+nothing at P0/);
+
+ // The counts belong with the verdict, not repeated at the far end.
+ assert.equal(blocked.filter((l) => /P0 \d+ /.test(l)).length, 1);
+});
+
+test("the markdown report leads with the verdict, not with the first finding", () => {
+ const comments = [mdFinding("a.ts", 1, "[P2] **Nit**\n\nbody")];
+ const clean = renderReportMarkdown([], gate([], "P0,P1"));
+ assert.match(clean, /✅ PASSED\*\* — no findings\./);
+
+ const passed = renderReportMarkdown(comments, gate(comments, "P0,P1"));
+ assert.match(passed.split("\n")[2], /✅ PASSED/);
+ assert.match(passed, /1 finding to read\./);
+
+ const p0 = [mdFinding("a.ts", 1, "[P0] **Boom**\n\nbody")];
+ assert.match(renderReportMarkdown(p0, gate(p0, "P0,P1")).split("\n")[2], /❌ BLOCKED/);
+});
+
+test("the markdown report groups by file and puts the blocking file first", () => {
+ const comments = [
+ mdFinding("z.ts", 1, "[P3] **Nit in z**\n\nbody"),
+ mdFinding("a.ts", 9, "[P0] **Boom in a**\n\nbody"),
+ mdFinding("a.ts", 2, "[P2] **Advisory in a**\n\nbody"),
+ ];
+ const md = renderReportMarkdown(comments, gate(comments, "P0,P1"));
+ assert.ok(md.indexOf("### `a.ts`") < md.indexOf("### `z.ts`"), "the file that blocks must sort first");
+ // Within a file: worst first, and only one heading per file.
+ assert.ok(md.indexOf("Boom in a") < md.indexOf("Advisory in a"));
+ assert.equal(md.match(/### `a\.ts`/g).length, 1);
+});
+
+test("the ❌ mark follows the gate that ran, not the severity in the abstract", () => {
+ const comments = [mdFinding("a.ts", 1, "[P1] **High**\n\nbody")];
+ assert.match(renderReportMarkdown(comments, gate(comments, "P0,P1")), /❌ P1/);
+ // Same finding, narrower gate: reported, but this run does not block on it.
+ const narrow = renderReportMarkdown(comments, gate(comments, "P0"));
+ assert.match(narrow, /💬 P1/);
+ assert.ok(!narrow.includes("❌ P1"));
+});
+
+test("the markdown report keeps the body as prose rather than pre-formatted text", () => {
+ const long =
+ "[P0] **Title**\n\n" +
+ "A body long enough that the terminal renderer would hard-wrap it at seventy-eight columns and destroy it.";
+ const md = renderReportMarkdown([mdFinding("a.ts", 1, long)], gate([], "P0,P1"));
+ // One unbroken line — the reader reflows, we do not.
+ assert.match(md, /^A body long enough that the terminal renderer would hard-wrap it at seventy-eight columns and destroy it\.$/m);
+ assert.ok(!md.includes("["), "no ANSI escapes may reach a chat transcript");
+});
+
+// --------------------------------------------------------------- git modes ---
+
+// A throwaway repo, because every mode decision is a question about real git
+// state and mocking git would only prove the mock agrees with itself.
+function scratchRepo(t) {
+ // realpath because macOS resolves /var -> /private/var, and git reports the
+ // resolved form — an unresolved base would make every path comparison fail.
+ const dir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ocr-harness-"));
+ t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
+ const run = (...args) => execFileSync("git", args, { cwd: dir, encoding: "utf8" });
+ run("init", "-q", "-b", "main");
+ run("config", "user.email", "t@example.com");
+ run("config", "user.name", "t");
+ run("commit", "-q", "--allow-empty", "-m", "root");
+ return { dir, run };
+}
+
+test("an explicit flag always wins over the auto-detection", (t) => {
+ const { dir } = scratchRepo(t);
+ assert.equal(resolveMode({ commit: "abc123" }, dir).mode, "commit");
+ assert.equal(resolveMode({ worktree: true }, dir).mode, "workspace");
+ assert.equal(resolveMode({ from: "main", to: "HEAD" }, dir).mode, "range");
+});
+
+// ------------------------------------------------------------ pull request ---
+
+test("a pull request range is labelled as one, so the plan can explain itself", (t) => {
+ const { dir } = scratchRepo(t);
+ // resolvePr() has already done the network half; resolveMode only labels.
+ const range = resolveMode({ from: "refs/orcacode/pr/7/base", to: "refs/orcacode/pr/7/head", pr: 7 }, dir);
+ assert.equal(range.mode, "range");
+ assert.equal(range.code, "pr");
+ assert.equal(range.pr, 7);
+ assert.ok(MODE_REASON.pr, "the prompt must be able to render the reason");
+});
+
+test("a range the user typed is not mistaken for a pull request", (t) => {
+ const { dir } = scratchRepo(t);
+ const range = resolveMode({ from: "main", to: "HEAD" }, dir);
+ assert.equal(range.code, "explicit");
+ assert.equal(range.pr, undefined);
+});
+
+test("a pull request number that is not one fails before anything is fetched", (t) => {
+ const { dir } = scratchRepo(t);
+ for (const bad of ["", "abc", "12x", undefined, "-3"]) {
+ const r = resolvePr(bad, dir);
+ assert.equal(r.ok, false);
+ assert.equal(r.code, "bad-number", `${bad} should be rejected as a number`);
+ }
+ // A leading # is how people write it, and is not an error.
+ assert.notEqual(resolvePr("#556", dir).code, "bad-number");
+});
+
+test("the plan carries pull request metadata only when there is a pull request", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "a.ts"), "export const a = 1;\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const plan = buildPlan({ from: "main~1", to: "HEAD" }, dir);
+ assert.equal(plan.pr, null, "no --pr means no pr block, not a partial one");
+ assert.ok(!renderPlan(plan).includes("pull request:"));
+});
+
+test("the prompt names the pull request and warns when its head is a fork", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "a.ts"), "export const a = 1;\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const plan = buildPlan(
+ {
+ from: "main~1",
+ to: "HEAD",
+ pr: 556,
+ prMeta: { number: 556, title: "Fix the thing", url: "https://x/556", fork: true, head: "feat/x", base: "main" },
+ },
+ dir,
+ );
+ const text = renderPlan(plan);
+ assert.match(text, /pull request: #556 Fix the thing/);
+ assert.match(text, /https:\/\/x\/556/);
+ assert.match(text, /from a fork/);
+});
+
+test("a multi-line background becomes its own section instead of eating the scope list", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "a.ts"), "export const a = 1;\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+
+ const oneLine = renderPlan(buildPlan({ from: "main~1", to: "HEAD", background: "make it fast" }, dir));
+ assert.match(oneLine, /- background: make it fast/);
+ assert.ok(!oneLine.includes("### Background"));
+
+ // A PR body is always multi-line; inlined after "- background:" it swallows
+ // the rest of the bullet list.
+ const body = renderPlan(buildPlan({ from: "main~1", to: "HEAD", background: "Title\n\nSecond para" }, dir));
+ assert.ok(!body.includes("- background:"));
+ assert.match(body, /### Background/);
+ assert.match(body, /^> Title$/m);
+ assert.match(body, /^> Second para$/m);
+ // The scope list must still be intact below it.
+ assert.match(body, /### Files to review \(1\)/);
+ // And the reviewer must be told the description is a claim, not a fact.
+ assert.match(body, /disagrees with the code is itself a finding/);
+});
+
+test("a dirty tree auto-selects workspace, and says so", (t) => {
+ const { dir } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "a.txt"), "hello\n");
+ const r = resolveMode({}, dir);
+ assert.equal(r.mode, "workspace");
+ assert.equal(r.code, "auto-dirty");
+});
+
+test("a clean tree with nothing ahead of the base has nothing to review", (t) => {
+ const { dir } = scratchRepo(t);
+ const r = resolveMode({}, dir);
+ assert.equal(r.mode, "empty");
+ assert.equal(r.code, "not-ahead");
+});
+
+test("a clean tree ahead of the base auto-selects the range CI would use", (t) => {
+ const { dir, run } = scratchRepo(t);
+ run("checkout", "-q", "-b", "feature");
+ fs.writeFileSync(path.join(dir, "a.txt"), "hello\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "add a");
+ const r = resolveMode({}, dir);
+ assert.equal(r.mode, "range");
+ assert.equal(r.code, "auto-ahead");
+ assert.equal(r.to, "HEAD");
+});
+
+test("porcelain's leading status space does not eat the first path", (t) => {
+ // `git status --porcelain` is fixed-column: " M foo" for an unstaged edit.
+ // Trimming the buffer strips that space off the FIRST line only, which
+ // silently truncates one path per run and looks like a git bug, not ours.
+ // (.ts, not .txt: the selector now drops unsupported types, and this test is
+ // about porcelain parsing, not about what is reviewable.)
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "tracked.ts"), "one\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "tracked");
+ fs.writeFileSync(path.join(dir, "tracked.ts"), "two\n");
+ fs.writeFileSync(path.join(dir, "untracked.ts"), "new\n");
+
+ const { files } = selectFiles({ mode: "workspace" }, dir);
+ const paths = files.map((f) => f.path).sort();
+ assert.deepEqual(paths, ["tracked.ts", "untracked.ts"]);
+});
+
+test("an untracked directory is expanded into files, not reported as a directory", (t) => {
+ const { dir } = scratchRepo(t);
+ fs.mkdirSync(path.join(dir, "nested", "deep"), { recursive: true });
+ fs.writeFileSync(path.join(dir, "nested", "deep", "x.ts"), "x\n");
+ const { files } = selectFiles({ mode: "workspace" }, dir);
+ assert.deepEqual(files.map((f) => f.path), ["nested/deep/x.ts"]);
+});
+
+test("a deleted file is not offered for review", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "gone.txt"), "bye\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "add");
+ fs.rmSync(path.join(dir, "gone.txt"));
+ const { files } = selectFiles({ mode: "workspace" }, dir);
+ assert.deepEqual(files, [], "there is no code left to file a finding against");
+});
+
+// ------------------------------------------------------------------- plan ---
+
+test("the plan carries everything a reviewer needs and nothing it must ask for", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "AGENTS.md"), "# House rules\n\nTabs, not spaces.\n");
+ fs.writeFileSync(path.join(dir, "a.ts"), "export const a = 1;\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "c");
+ fs.writeFileSync(path.join(dir, "a.ts"), "export const a = 2;\n");
+
+ const plan = buildPlan({}, dir);
+ assert.equal(plan.schema_version, SCHEMA_VERSION);
+ assert.equal(plan.mode, "workspace");
+ assert.equal(plan.result_path, path.join(dir, WORK_DIR, "result.json"));
+
+ const text = renderPlan(plan);
+ // The severity contract, in full — a reviewer that has to go find the rubric
+ // will use one it already knows, and that one is not P0-P3.
+ assert.match(text, /\[P0\].*\[P1\].*\[P2\].*\[P3\]/s);
+ assert.match(text, /calibration, not suppression/, "the P2/P3 emit rule must survive");
+ assert.match(text, /start_line/);
+ assert.match(text, /existing_code/);
+ assert.match(text, /review submit/, "the reviewer must be told how to hand back");
+ // Project conventions, wrapped in the untrusted-data framing the Action uses.
+ assert.match(text, /House rules/);
+ assert.match(text, /untrusted project conventions/);
+});
+
+test("the plan filters files with the bundled rules and says why each one is out", (t) => {
+ // Before: `ocr delegate preview` if the binary happened to be installed,
+ // plain git otherwise, and the plan had to confess which. Now there is one
+ // path, it needs nothing installed, and every exclusion names its reason.
+ const { dir, run } = scratchRepo(t);
+ fs.mkdirSync(path.join(dir, "src"));
+ fs.writeFileSync(path.join(dir, "src/a.ts"), "export const a = 1;\n");
+ fs.writeFileSync(path.join(dir, "src/a.test.ts"), "test();\n");
+ fs.writeFileSync(path.join(dir, "logo.png"), Buffer.from([0x89, 0x50, 0x4e, 0x47, 0, 0, 0, 0]));
+ fs.writeFileSync(path.join(dir, "Makefile"), "all:\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "files");
+ const plan = buildPlan({ from: "main~1", to: "HEAD" }, dir);
+
+ assert.equal(plan.selector, "builtin");
+ assert.deepEqual(plan.files.map((f) => f.path).sort(), ["Makefile", "src/a.ts"]);
+ assert.deepEqual(
+ plan.excluded.map((f) => [f.path, f.code]).sort(),
+ [["logo.png", "binary"], ["src/a.test.ts", "default_path"]],
+ );
+ // The .ts file resolves to a real checklist, not the default one.
+ const ts = plan.rule_groups.find((g) => g.files.includes("src/a.ts"));
+ assert.ok(ts, "a rule group for the .ts file");
+ assert.notEqual(ts.pattern, "default");
+
+ const text = renderPlan(plan);
+ assert.match(text, /bundled Open Code Review rules/);
+ assert.match(text, /### Excluded \(2\)/);
+ assert.match(text, /`logo\.png` — binary/);
+ assert.ok(!/ocr not installed/.test(text), "the old fallback confession is gone");
+});
+
+test("a deleted file is listed as excluded, not silently dropped", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.writeFileSync(path.join(dir, "gone.ts"), "1\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "add");
+ run("rm", "-q", "gone.ts");
+ run("commit", "-q", "-m", "rm");
+ const plan = buildPlan({ from: "main~1", to: "HEAD" }, dir);
+ assert.deepEqual(plan.files, []);
+ assert.deepEqual(plan.excluded.map((f) => [f.path, f.code]), [["gone.ts", "deleted"]]);
+});
+
+test("numstat gives real line counts and survives a rename", (t) => {
+ const { dir, run } = scratchRepo(t);
+ fs.mkdirSync(path.join(dir, "src"));
+ fs.writeFileSync(path.join(dir, "src/old.ts"), "a\nb\nc\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "one");
+ run("mv", "src/old.ts", "src/new.ts");
+ fs.appendFileSync(path.join(dir, "src/new.ts"), "d\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "two");
+ const plan = buildPlan({ from: "main~1", to: "HEAD" }, dir);
+ const f = plan.files.find((x) => x.path === "src/new.ts");
+ assert.ok(f, "the renamed file is listed under its new name");
+ assert.equal(f.insertions, 1);
+ assert.equal(f.deletions, 0);
+});
+
+test("a repo with nothing to review yields no plan rather than an empty one", (t) => {
+ const { dir } = scratchRepo(t);
+ const plan = buildPlan({}, dir);
+ assert.equal(plan.range.mode, "empty");
+ assert.deepEqual(plan.files, []);
+});
+
+// ------------------------------------------------------------------ skill ---
+
+// The skill is a prompt, and the rules below are the ones an editor is most
+// likely to soften without noticing what they were load-bearing for.
+const SKILL = fs.readFileSync(new URL("../skills/orca-review/SKILL.md", import.meta.url), "utf8");
+
+test("the skill routes Action setup to the OTHER skill instead of absorbing it", () => {
+ // Two skills install together. If this one starts explaining CI setup, an
+ // agent will half-configure a workflow from a file that never mentions the
+ // secret, and the user gets a workflow that fails on auth.
+ assert.match(SKILL, /orca-review-action/);
+ assert.match(SKILL, /\*\*Not this skill\.\*\*/);
+});
+
+test("the skill forbids substituting a summary for the gate", () => {
+ // The position check re-homes and deduplicates, so the findings the agent is
+ // holding are NOT the findings that survive. Summarising instead of running
+ // `submit` reports a set that the merge gate never agreed to.
+ assert.match(SKILL, /Relay that markdown to the user verbatim/);
+ assert.match(SKILL, /Do not re-summarise it/i);
+ assert.match(SKILL, /the gate decides what blocks, not you/i);
+});
+
+test("the skill reads the verdict from the report, not from the exit code", () => {
+ // Observed in the wild, twice: the host rendered exit 1 as "Error: Exit code
+ // 1", and both the agent and the user read a working review as a crash. Exit
+ // 0 is now the default, so the skill must be explicit that 0 ≠ passed.
+ assert.match(SKILL, /The exit code is not the verdict/);
+ // The verdict word is localized now, so the skill teaches the mark, not the word.
+ assert.match(SKILL, /❌ \(`Blocked`,\s+`已拦截`/);
+ assert.match(SKILL, /✅ means nothing/);
+});
+
+test("the skill keeps exit 2 a failure", () => {
+ // The one thing the 0 default must not teach: an unusable result is not a pass.
+ assert.match(SKILL, /Exit `2` is the only failure, and it is never a pass/);
+});
+
+test("the skill asks for the markdown report, not the terminal one", () => {
+ // The default is ANSI hard-wrapped at 78 columns; pasted into a chat it is a
+ // wall of pre-formatted text, which is the whole reason --format md exists.
+ // The bare form: no extra flag an agent could forget.
+ assert.match(SKILL, /review submit --format md --lang en/);
+ // Observed: an agent copied the example's `--lang zh` verbatim for an English
+ // request. The example must not carry a language that looks like an answer.
+ assert.match(SKILL, /Do not copy the example's value; read the conversation/);
+});
+
+test("the skill states the credential posture, because that is the reason it exists", () => {
+ assert.match(SKILL, /no OrcaRouter account, no API key/i);
+ assert.match(SKILL, /Nothing here talks to OrcaRouter/);
+});
+
+test("the skill keeps P2/P3 reportable and warns off a foreign severity scheme", () => {
+ assert.match(SKILL, /Do not drop P2 and P3/);
+ assert.match(SKILL, /High\/Medium\/Low is a different tool's\s+vocabulary/);
+ // Untagged defaults to P1 and blocks; an agent that does not know this will
+ // ship untagged findings and be surprised by a red gate.
+ assert.match(SKILL, /Untagged findings default to P1/);
+});
+
+test("the skill names the verification key and why a paraphrase breaks it", () => {
+ assert.match(SKILL, /existing_code/);
+ assert.match(SKILL, /copied verbatim/i);
+ assert.match(SKILL, /start_line/);
+});
+
+test("the skill forbids checking a branch out to review a pull request", () => {
+ // The whole point of --pr is that it is non-destructive. An agent that
+ // "helpfully" checks out first can destroy uncommitted work.
+ assert.match(SKILL, /--pr 556/);
+ assert.match(SKILL, /do \*\*not\*\* check the branch out first/i);
+ assert.match(SKILL, /Do not switch branches to review a PR/i);
+});
+
+test("the skill tells the agent to hand gh problems back rather than fix them", () => {
+ assert.match(SKILL, /gh pr checkout/);
+ assert.match(SKILL, /Do not run it for them|do not try to install or\s*\n?authenticate it/i);
+});
+
+test("the skill and the plan agree on where the result goes", () => {
+ assert.ok(SKILL.includes(`${WORK_DIR}/result.json`), `SKILL.md must name ${WORK_DIR}/result.json`);
+});
diff --git a/scripts/installer.test.mjs b/scripts/installer.test.mjs
index eb6253b..b0b7efa 100644
--- a/scripts/installer.test.mjs
+++ b/scripts/installer.test.mjs
@@ -19,11 +19,17 @@ import {
renderWorkflow,
parseOverrides,
} from "../bin/orcacode-review.mjs";
+import { retireLegacy } from "../bin/skill-tree.mjs";
const CLI = fileURLToPath(new URL("../bin/orcacode-review.mjs", import.meta.url));
// --------------------------------------------------------------- entry point ---
+// A prerelease is a real version here: test builds ship as 2.1.0-rc.N, and these
+// assertions are about the entry-point guard printing SOMETHING, not about which
+// release channel it came from.
+const SEMVER = /^\d+\.\d+\.\d+(-[0-9A-Za-z.-]+)?$/;
+
test("the CLI runs when executed through a symlink", () => {
// npm installs a `bin` as a SYMLINK at node_modules/.bin/, so argv[1]
// is the link and import.meta.url is its target. An entry-point guard that
@@ -42,13 +48,13 @@ test("the CLI runs when executed through a symlink", () => {
const r = spawnSync(process.execPath, [link, "--version"], { encoding: "utf8" });
assert.equal(r.status, 0, r.stderr);
- assert.match(r.stdout.trim(), /^\d+\.\d+\.\d+$/, "no version printed — the guard rejected the symlink");
+ assert.match(r.stdout.trim(), SEMVER, "no version printed — the guard rejected the symlink");
});
test("the CLI runs when executed by its real path", () => {
const r = spawnSync(process.execPath, [CLI, "--version"], { encoding: "utf8" });
assert.equal(r.status, 0, r.stderr);
- assert.match(r.stdout.trim(), /^\d+\.\d+\.\d+$/);
+ assert.match(r.stdout.trim(), SEMVER);
});
test("importing the CLI runs nothing", () => {
@@ -162,6 +168,18 @@ test("a scoped package declares public access", () => {
assert.equal(PKG.publishConfig?.access, "public");
});
+test("the CLI ships with no dependencies", () => {
+ // `npx` downloads the whole tree before running anything, so a dependency is
+ // latency on every install. 1.1.0 through 1.5.0 shipped with the package
+ // depending on ITSELF at ^1.0.2 — every npx fetched a second, older copy of
+ // the CLI into node_modules, and nothing failed, so five releases carried it.
+ // A self-reference is also a registry dependent, which is why this is a test
+ // and not a note in RELEASE.md.
+ for (const field of ["dependencies", "peerDependencies", "optionalDependencies"]) {
+ assert.deepEqual(Object.keys(PKG[field] ?? {}), [], `${field} must stay empty`);
+ }
+});
+
test("the installed command is short even though the package name is scoped", () => {
// `npm i -g` creates a command named by the bin KEY, not the package name.
assert.deepEqual(Object.keys(PKG.bin), ["orcacode-review"]);
@@ -192,7 +210,7 @@ test("every npx invocation in the README names the real package", () => {
// ------------------------------------------------------- skill safety rules ---
const SKILL = fs.readFileSync(
- new URL("../skills/setup-orca-code-review/SKILL.md", import.meta.url),
+ new URL("../skills/orca-review-action/SKILL.md", import.meta.url),
"utf8",
);
@@ -238,7 +256,7 @@ test("the double-review symptom is still documented", () => {
// The App exists whether or not this skill installs it, so someone can still
// end up with two reviewers.
const trouble = fs.readFileSync(
- new URL("../skills/setup-orca-code-review/references/troubleshooting.md", import.meta.url),
+ new URL("../skills/orca-review-action/references/troubleshooting.md", import.meta.url),
"utf8",
);
assert.match(trouble, /two sets of comments/i);
@@ -257,7 +275,7 @@ test("the double-review symptom is still documented", () => {
const readYaml = (rel) => fs.readFileSync(new URL(rel, import.meta.url), "utf8");
const EXAMPLE = readYaml("../workflows/orca-code-review.yml");
-const TEMPLATE = readYaml("../skills/setup-orca-code-review/assets/workflow.yml");
+const TEMPLATE = readYaml("../skills/orca-review-action/assets/workflow.yml");
// Strips comments and blank lines, leaving only what GitHub actually reads.
const effective = (yaml) =>
@@ -341,3 +359,84 @@ test("the help shows the scoped package name users actually type", () => {
assert.equal(invocation, `npx ${PKG.name}`, `stale invocation in --help: ${invocation}`);
}
});
+
+// ------------------------------------------------------------ renamed skills ---
+
+// run-orca-code-review became orca-review; setup-orca-code-review became
+// orca-review-action. Every machine that installed the old names has them
+// still, and two skills matching the same phrases is a coin toss per review.
+
+function skillsRoot(t) {
+ const dir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ocr-legacy-"));
+ t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
+ return dir;
+}
+
+test("installing the new name removes our old one beside it", (t) => {
+ const root = skillsRoot(t);
+ fs.mkdirSync(path.join(root, "run-orca-code-review", "references"), { recursive: true });
+ fs.writeFileSync(path.join(root, "run-orca-code-review", "SKILL.md"), "---\nname: run-orca-code-review\ndescription: x\n---\nbody\n");
+ fs.writeFileSync(path.join(root, "run-orca-code-review", "references", "a.md"), "a");
+ assert.equal(retireLegacy(path.join(root, "orca-review"), "run-orca-code-review"), true);
+ assert.equal(fs.existsSync(path.join(root, "run-orca-code-review")), false);
+});
+
+test("a same-named directory that is not ours is left alone", (t) => {
+ const root = skillsRoot(t);
+ fs.mkdirSync(path.join(root, "run-orca-code-review"));
+ fs.writeFileSync(path.join(root, "run-orca-code-review", "SKILL.md"), "---\nname: someone-elses-skill\n---\n");
+ assert.equal(retireLegacy(path.join(root, "orca-review"), "run-orca-code-review"), false);
+ assert.equal(fs.existsSync(path.join(root, "run-orca-code-review")), true);
+
+ // No SKILL.md at all — also not ours to remove.
+ fs.mkdirSync(path.join(root, "setup-orca-code-review"));
+ fs.writeFileSync(path.join(root, "setup-orca-code-review", "notes.txt"), "keep");
+ assert.equal(retireLegacy(path.join(root, "orca-review-action"), "setup-orca-code-review"), false);
+ assert.equal(fs.existsSync(path.join(root, "setup-orca-code-review", "notes.txt")), true);
+});
+
+test("nothing to retire is a quiet false", (t) => {
+ assert.equal(retireLegacy(path.join(skillsRoot(t), "orca-review"), "run-orca-code-review"), false);
+});
+
+test("the shipped skills carry the new names and never the old", () => {
+ for (const [dir, name] of [["orca-review", "orca-review"], ["orca-review-action", "orca-review-action"]]) {
+ const text = fs.readFileSync(new URL(`../skills/${dir}/SKILL.md`, import.meta.url), "utf8");
+ assert.match(text, new RegExp(`^---\\nname: ${name}\\n`));
+ assert.ok(!/run-orca-code-review|setup-orca-code-review/.test(text), `${dir} still mentions an old name`);
+ }
+});
+
+// ------------------------------------------------------------------- --mode ---
+
+function unattended(t, ...args) {
+ const dir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ocr-mode-"));
+ t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
+ fs.mkdirSync(path.join(dir, ".claude"));
+ const r = spawnSync(process.execPath, [CLI, ...args, "--scope", "project", "--platform", "claude", "--yes", "--json", "--lang", "en"], {
+ cwd: dir,
+ encoding: "utf8",
+ env: { ...process.env, NO_COLOR: "1" },
+ });
+ return { dir, r };
+}
+
+test("--mode picks the skill set: local, action, or both", (t) => {
+ assert.deepEqual(JSON.parse(unattended(t, "--mode", "local").r.stdout).skills, ["orca-review"]);
+ assert.deepEqual(JSON.parse(unattended(t, "--mode", "action").r.stdout).skills, ["orca-review-action"]);
+ assert.deepEqual(JSON.parse(unattended(t, "--mode=both").r.stdout).skills.sort(), ["orca-review", "orca-review-action"]);
+});
+
+test("unattended with no --mode installs both — one product, two halves", (t) => {
+ const { dir, r } = unattended(t);
+ assert.deepEqual(JSON.parse(r.stdout).skills.sort(), ["orca-review", "orca-review-action"]);
+ assert.ok(fs.existsSync(path.join(dir, ".claude", "skills", "orca-review", "SKILL.md")));
+ assert.ok(fs.existsSync(path.join(dir, ".claude", "skills", "orca-review-action", "SKILL.md")));
+});
+
+test("an unknown --mode is refused by name, with the choices", (t) => {
+ const { r } = unattended(t, "--mode", "remote");
+ assert.notEqual(r.status, 0);
+ assert.match(r.stderr, /Unknown mode "remote"/);
+ assert.match(r.stderr, /both \| local \| action/);
+});
diff --git a/scripts/judge.mjs b/scripts/judge.mjs
index 25fc4b2..22f1c92 100644
--- a/scripts/judge.mjs
+++ b/scripts/judge.mjs
@@ -21,6 +21,24 @@
import fs from "node:fs";
import os from "node:os";
+// A STABLE FIRST LINE, WHATEVER HAPPENS. action.yml lifts the first line of
+// this script's stderr into the job's user-facing L2 failure reason, so an
+// unforeseen throw reports a source-file path as "why your review failed" —
+// which names nothing and implies no fix. The guards further down cover the
+// malformed response shapes we know about; enumerating shapes is always one
+// shape short, so this covers the rest.
+//
+// The stack still follows on the next lines: the action echoes the whole log,
+// and only the FIRST line is promoted, so nothing needed for debugging is
+// lost by putting a summary in front of it.
+const crash = (label) => (e) => {
+ console.error(`judge crashed (${label}): ${e?.message || e}`);
+ if (e?.stack) console.error(e.stack);
+ process.exit(1);
+};
+process.on("uncaughtException", crash("uncaught"));
+process.on("unhandledRejection", crash("unhandled rejection"));
+
// Parses a strict plain decimal (e.g. "0.7", "-1", "0.5"). Returns the number
// or NaN. Deliberately does NOT use parseFloat — parseFloat stops at the
// first non-numeric character so "0.8oops" would silently become 0.8 and
@@ -51,7 +69,7 @@ for (let i = 0; i < rest.length; i += 1) {
}
else if (rest[i] === "--model") modelOverride = rest[++i];
}
-if (!file) { console.error("usage: node judge.mjs [--out f] [--threshold 0.7] [--model deepseek/deepseek-v4-pro]"); process.exit(2); }
+if (!file) { console.error("usage: node judge.mjs [--out f] [--threshold 0.7] [--model orcarouter/]"); process.exit(2); }
// Env vars first (production); config.json only as a local-harness fallback.
let llmUrl = process.env.OCR_LLM_URL || null;
@@ -119,40 +137,116 @@ const body = JSON.stringify({
messages: [{ role: "system", content: system }, { role: "user", content: user }],
});
-const res = await fetch(llmUrl, {
- method: "POST",
- headers: {
- "content-type": "application/json",
- [llmAuthHeader]: "Bearer " + llmToken,
- // THE ANGLE, so a router recipe can put this call on its own model.
- //
- // --model is normally the router ALIAS (action.yml passes it when judge-model
- // is unset), and an alias resolves through the workspace's DSL, which has no
- // other way to tell a judge call from a review call: the reviewer stamps no
- // angle. Without this header the judge takes the recipe's default — the
- // reviewer's own model — and a judge scoring work its own model produced
- // agrees with it, so the pass goes inert while still reporting success.
- //
- // Harmless when --model names a concrete model: nothing resolves an alias, so
- // nothing reads the header. Sent unconditionally rather than only for aliases
- // because "is this an alias" is the gateway's judgement, not this script's.
- "x-cr-lens": "judge",
- },
- body,
-});
-const raw = await res.text();
+// ONE REQUEST, AND DELIBERATELY NO RETRY HERE. In production OCR_LLM_URL is the
+// local fact proxy the action starts (action.yml), and fact-proxy.mjs already
+// retries 429/502/503/504 four times with backoff. A retry loop in this script
+// would multiply against that — four here times four there is sixteen upstream
+// requests for one judge call.
+//
+// It would also be a WIDER set than the proxy's on purpose-built ground: the
+// proxy excludes 500 because a 500 can mean the completion was produced and
+// billed, and replaying it buys the same bill twice. Anything added here would
+// bypass that policy without knowing it existed.
+//
+// Retry belongs at the proxy, which is the layer that knows whether the request
+// was idempotent. If this ever needs its own, it has to be the statuses the
+// proxy passes through, not a second copy of its list.
+// WRAPPED, because a transport failure is a failure class too and it has to
+// arrive named. Without this the process dies on an unhandled rejection and
+// the first line of stderr is a path inside undici — and action.yml quotes
+// that first line as the reason in its summary annotation, so a dead gateway
+// would be reported to the operator as a stack frame.
+//
+// THE BODY READ IS INSIDE THE TRY, not just the fetch. `fetch` resolves as
+// soon as the response headers arrive, so a connection that dies mid-body
+// rejects at `res.text()` instead — with "terminated", from a different
+// undici frame. That is not a hypothetical here: fact-proxy relays headers
+// first and then destroys the connection on a mid-stream upstream failure
+// (fact-proxy.mjs, "the headers are out, so destroy the connection"), and
+// the proxy is what serves this call in production.
+//
+// The message alone is not enough either: fetch collapses every transport
+// failure into the string "fetch failed" and puts the part worth reading —
+// ECONNREFUSED, ECONNRESET, a DNS failure, a headers timeout — in `cause`.
+let res;
+let raw;
+try {
+ res = await fetch(llmUrl, {
+ method: "POST",
+ headers: {
+ "content-type": "application/json",
+ [llmAuthHeader]: "Bearer " + llmToken,
+ // THE ANGLE, so a router recipe can put this call on its own model.
+ //
+ // --model is normally the router ALIAS (action.yml passes it when judge-model
+ // is unset), and an alias resolves through the workspace's DSL, which has no
+ // other way to tell a judge call from a review call: the reviewer stamps no
+ // angle. Without this header the judge takes the recipe's default — the
+ // reviewer's own model — and a judge scoring work its own model produced
+ // agrees with it, so the pass goes inert while still reporting success.
+ //
+ // Harmless when --model names a concrete model: nothing resolves an alias, so
+ // nothing reads the header. Sent unconditionally rather than only for aliases
+ // because "is this an alias" is the gateway's judgement, not this script's.
+ "x-cr-lens": "judge",
+ },
+ body,
+ });
+ raw = await res.text();
+} catch (e) {
+ // "complete", not "reach": by the time a mid-body drop lands here the
+ // request did reach the gateway. The cause code is what separates the two —
+ // ECONNREFUSED never arrived, ECONNRESET/terminated arrived and was cut off.
+ const cause = e?.cause?.code || e?.cause?.message || "";
+ console.error(`could not complete the LLM request: ${e?.message || e}${cause ? ` (${cause})` : ""}`);
+ process.exit(1);
+}
+// The upstream body, not just the status: that text is how an operator tells the
+// gateway's own 5xx from one it passed through — and by the time it reaches here
+// the proxy has already spent its retries on it.
if (!res.ok) { console.error(`HTTP ${res.status}: ${raw.slice(0, 400)}`); process.exit(1); }
let content;
try { content = JSON.parse(raw).choices[0].message.content; }
catch (e) { console.error("bad completion envelope: " + raw.slice(0, 400)); process.exit(1); }
+// A 200 CAN STILL CARRY NOTHING TO READ. The envelope parses, `content` is
+// present, and it is null or a number — so the extraction above succeeds and
+// the `.replace` below throws instead. Type it here, where the raw response
+// is still in hand to quote.
+if (typeof content !== "string") {
+ console.error(`judge response had no text content (${content === null ? "null" : typeof content}): ${raw.slice(0, 400)}`);
+ process.exit(1);
+}
+
const jsonText = content.replace(/^```(?:json)?/m, "").replace(/```$/m, "").trim();
let parsed;
try { parsed = JSON.parse(jsonText); }
catch (e) { console.error("judge did not return JSON:\n" + content.slice(0, 600)); process.exit(1); }
-const groups = parsed.groups || [];
+// `groups` MUST BE AN ARRAY, and the test names what is ALLOWED rather than
+// what is rejected. Two exemptions, both meaning "the judge classified
+// nothing", which the fail-open pass below handles deliberately: the field is
+// absent, or it is null. Everything else is a schema violation and fails.
+//
+// Not a truthiness test, which is what this was and what was wrong with it.
+// `false`, `0` and `""` are falsy AND non-arrays, so they slipped through to
+// the `|| []` below, every finding came out unclassified, the fail-open pass
+// kept all of them, and the judge exited 0 — publishing every finding as
+// JUDGED off a response that violated the schema. That is the precise outcome
+// this pass fails closed to prevent, and a reject-list guard reintroduced it
+// while looking like it closed it.
+//
+// An accept-list cannot have that shape of hole: a value that is neither
+// exemption nor an array has nowhere to go but the failure branch, whatever
+// type someone invents next.
+const groupsRaw = parsed.groups;
+const groupsOmitted = groupsRaw === undefined || groupsRaw === null;
+if (!groupsOmitted && !Array.isArray(groupsRaw)) {
+ console.error(`judge returned a non-array groups (${typeof groupsRaw}): ${jsonText.slice(0, 600)}`);
+ process.exit(1);
+}
+const groups = groupsOmitted ? [] : groupsRaw;
const covered = new Set();
for (const g of groups) for (const id of g.member_ids || []) covered.add(id);
// Fail-open for findings the judge did not classify into any group: mark
diff --git a/scripts/judge.test.mjs b/scripts/judge.test.mjs
index c6640e1..83991a0 100644
--- a/scripts/judge.test.mjs
+++ b/scripts/judge.test.mjs
@@ -150,6 +150,86 @@ const oneGroup = (over) =>
],
});
+// LLM double that fails the first `failFor` requests with `status`, then
+// succeeds. Counts every request so a test can assert how many attempts the
+// judge actually made.
+async function startFlakyLlm({ failFor, status, reply }) {
+ let calls = 0;
+ const server = http.createServer((req, res) => {
+ req.on("data", () => {});
+ req.on("end", () => {
+ calls += 1;
+ if (calls <= failFor) {
+ res.writeHead(status, { "content-type": "application/json" });
+ res.end(JSON.stringify({ error: { message: "upstream is having a moment" } }));
+ return;
+ }
+ res.writeHead(200, { "content-type": "application/json" });
+ res.end(JSON.stringify({ choices: [{ message: { content: reply } }] }));
+ });
+ });
+ const port = await listen(server);
+ return { port, calls: () => calls, close: () => new Promise((r) => server.close(r)) };
+}
+
+async function judgeAgainstFlaky({ comments, failFor, status, reply, threshold }) {
+ const llm = await startFlakyLlm({ failFor, status, reply });
+ try {
+ const input = writeInput(comments);
+ const out = join(dir, `${(seq += 1)}-out.json`);
+ const args = [input, "--out", out, "--model", "test/judge-model"];
+ if (threshold !== undefined) args.push("--threshold", String(threshold));
+ const r = await spawnJudge(args, {
+ OCR_LLM_URL: `http://127.0.0.1:${llm.port}/v1/chat/completions`,
+ OCR_LLM_TOKEN: "test-token",
+ });
+ return { ...r, out, calls: llm.calls(), read: () => JSON.parse(readFileSync(out, "utf8")) };
+ } finally {
+ await llm.close();
+ }
+}
+
+// RETRY BELONGS AT THE PROXY, NOT HERE. In production OCR_LLM_URL is the local
+// fact proxy, which already retries 429/502/503/504 four times — a loop in
+// judge.mjs multiplies against that (four times four upstream requests for one
+// judge call) and would replay statuses the proxy excludes on purpose, like a
+// 500 whose completion may already have been produced and billed.
+//
+// This pins the request count so that second layer cannot be reintroduced
+// without a test saying so out loud.
+describe("upstream failures are not retried in-process", () => {
+ const keepAll = JSON.stringify({
+ groups: [{ member_ids: [0], representative_id: 0, confidence: 0.9, keep: true }],
+ });
+
+ for (const status of [502, 503, 429, 500]) {
+ test(`a ${status} makes exactly one request and reports the body`, async () => {
+ const r = await judgeAgainstFlaky({
+ comments: [finding("a real finding")],
+ failFor: 99,
+ status,
+ reply: keepAll,
+ });
+ assert.notEqual(r.status, 0);
+ assert.equal(r.calls, 1, "the proxy owns retries — this must not add a second layer");
+ assert.match(r.stderr, new RegExp(`HTTP ${status}`));
+ assert.match(r.stderr, /upstream is having a moment/);
+ });
+ }
+
+ test("a recovered upstream still needs only the one request", async () => {
+ const r = await judgeAgainstFlaky({
+ comments: [finding("a real finding")],
+ failFor: 0,
+ status: 502,
+ reply: keepAll,
+ });
+ assert.equal(r.status, 0, r.stderr);
+ assert.equal(r.calls, 1);
+ assert.equal(r.read().comments.length, 1);
+ });
+});
+
describe("argument validation (no LLM call is made)", () => {
test("a missing input file exits 2", async () => {
const r = await spawnJudge([], {});
@@ -418,6 +498,86 @@ describe("transport and envelope failures", () => {
assert.match(r.stderr, /judge did not return JSON/);
});
+ // A TRANSPORT FAILURE IS A FAILURE CLASS TOO, and it has to arrive named.
+ // With no catch the process died on an unhandled rejection and the first line
+ // of stderr was a path inside undici. action.yml quotes that first line as the
+ // reason in its summary annotation, so a dead gateway reached the operator as
+ // a stack frame — no class, no fix implied.
+ //
+ // TWO SHAPES, not one, and the first fix only covered the first. `fetch`
+ // rejects when it cannot reach the endpoint at all, but RESOLVES as soon as
+ // response headers arrive — so a connection that dies mid-body rejects at
+ // `res.text()`, from a different undici frame, and stayed uncaught. Not
+ // hypothetical: fact-proxy relays headers first and then destroys the
+ // connection on a mid-stream upstream failure, and the proxy is what serves
+ // this call in production.
+ //
+ // The no-stack-trace assertions are the load-bearing ones: a `catch` that
+ // merely re-printed the error would still satisfy the message match.
+ test("an unreachable endpoint is named, not a stack trace", async () => {
+ // A port that was bound and released is reliably closed, unlike a guess.
+ const probe = http.createServer();
+ const closedPort = await listen(probe);
+ await new Promise((r) => probe.close(r));
+
+ const out = join(dir, `${(seq += 1)}-unreachable.json`);
+ const r = await spawnJudge([writeInput([finding("[P1] x")]), "--model", "m", "--out", out], {
+ OCR_LLM_URL: `http://127.0.0.1:${closedPort}/v1/chat/completions`,
+ OCR_LLM_TOKEN: "t",
+ });
+
+ assert.equal(r.status, 1);
+ assert.match(r.stderr, /could not complete the LLM request/);
+ assert.match(r.stderr, /ECONNREFUSED/, "the cause carries the part worth reading");
+ assert.doesNotMatch(r.stderr, /node:internal/, "an unhandled rejection must not be the diagnostic");
+ assert.equal(existsSync(out), false);
+ });
+
+ test("a connection dropped mid-body is named too, not a stack trace", async () => {
+ // Headers out, then the socket dies — what fact-proxy does once it has
+ // already relayed headers. `fetch` has resolved by then, so this rejects at
+ // the body read, which the first version of the catch did not cover.
+ const server = http.createServer((req, res) => {
+ req.resume();
+ req.on("end", () => {
+ res.writeHead(200, { "content-type": "application/json", "content-length": "9999" });
+ res.write('{"choices"');
+ setTimeout(() => res.socket.destroy(), 20);
+ });
+ });
+ const port = await listen(server);
+
+ const out = join(dir, `${(seq += 1)}-dropped.json`);
+ const r = await spawnJudge([writeInput([finding("[P1] x")]), "--model", "m", "--out", out], {
+ OCR_LLM_URL: `http://127.0.0.1:${port}/v1/chat/completions`,
+ OCR_LLM_TOKEN: "t",
+ });
+ await new Promise((res2) => server.close(res2));
+
+ assert.equal(r.status, 1);
+ assert.match(r.stderr, /could not complete the LLM request/);
+ assert.doesNotMatch(r.stderr, /node:internal/, "the body read must be inside the catch");
+ assert.equal(existsSync(out), false);
+ });
+
+ // Every terminal failure names its class on the FIRST line, because that is
+ // the line action.yml lifts into the summary. The judge's bad-reply message is
+ // followed by a dump of what the model actually said, so a `tail` would quote
+ // a fragment of the bad completion in place of the reason.
+ test("the first stderr line names the class, for every terminal failure", async () => {
+ const cases = [
+ [{ status: 400, reply: "{}" }, /^HTTP 400: /],
+ [{ status: 401, reply: "{}" }, /^HTTP 401: /],
+ [{ envelope: "not json at all" }, /^bad completion envelope: /],
+ [{ reply: "Sure! Here are my thoughts\nabout the finding." }, /^judge did not return JSON:$/],
+ ];
+ for (const [opts, expected] of cases) {
+ const r = await judge({ comments: [finding("[P1] x")], reply: "{}", ...opts });
+ assert.equal(r.status, 1);
+ assert.match(r.stderr.split("\n")[0], expected);
+ }
+ });
+
test("a fenced ```json reply is unwrapped", async () => {
const r = await judge({
comments: [finding("[P1] x")],
@@ -429,6 +589,80 @@ describe("transport and envelope failures", () => {
});
});
+// A 200 WITH THE WRONG SHAPE MUST STILL NAME ITSELF. action.yml promotes the
+// FIRST line of this script's stderr into the job's user-facing L2 failure
+// reason, so a throw from a malformed-but-parseable response reported
+// `judge.mjs:195` as why the review failed — a source location, naming nothing
+// and implying no fix.
+//
+// Every assertion here checks the first line specifically, because that is the
+// only line the operator sees. A fix that printed a good message *after* a stack
+// trace would pass a plain `match` and change nothing.
+describe("malformed-but-parseable responses name themselves on the first line", () => {
+ const firstLine = (r) => r.stderr.split("\n")[0];
+ const notASourceLocation = (r) => {
+ assert.doesNotMatch(firstLine(r), /judge\.mjs:\d+/, "a source location is not a diagnosis");
+ assert.doesNotMatch(firstLine(r), /^node:internal/, "a node internal is not a diagnosis");
+ };
+
+ // The envelope parses and `content` is present but unreadable, so extraction
+ // succeeds and the `.replace` after it used to throw.
+ for (const [name, content] of [["null", null], ["a number", 42], ["an object", { a: 1 }]]) {
+ test(`content is ${name}`, async () => {
+ const r = await judge({
+ comments: [finding("[P1] x")],
+ envelope: JSON.stringify({ choices: [{ message: { content } }] }),
+ });
+ assert.equal(r.status, 1);
+ assert.match(firstLine(r), /judge response had no text content/);
+ notASourceLocation(r);
+ });
+ }
+
+ // A string iterates by character and an object is not iterable at all, so
+ // these threw a few lines further down. Failing is right rather than falling
+ // open: falling open keeps every finding unjudged, the one outcome the judge
+ // exists to prevent.
+ for (const [name, groups] of [["a string", '"nope"'], ["an object", '{"a":1}'], ["a number", "7"]]) {
+ test(`groups is ${name}`, async () => {
+ const r = await judge({ comments: [finding("[P1] x")], reply: `{"groups":${groups}}` });
+ assert.equal(r.status, 1);
+ assert.match(firstLine(r), /judge returned a non-array groups/);
+ notASourceLocation(r);
+ });
+ }
+
+ // FALSY IS NOT THE SAME QUESTION AS ABSENT, and conflating them is what the
+ // first version of this guard did. `false`, `0`, `""` and `NaN` are falsy AND
+ // non-arrays, so a truthiness test let them through to `|| []`: every finding
+ // came out unclassified, the fail-open pass kept all of them, and the judge
+ // exited 0 — publishing every finding as JUDGED off a response that violated
+ // the schema, under `precision-filter: true`. A reject-list guard reintroduced
+ // the exact hole it looked like it was closing.
+ //
+ // Asserting the exit code is not enough on its own here: the failure mode was
+ // a SUCCESSFUL exit, so these also assert that no output was written.
+ for (const [name, groups] of [["false", "false"], ["0", "0"], ['""', '""']]) {
+ test(`groups is ${name} — falsy but still a schema violation`, async () => {
+ const r = await judge({ comments: [finding("[P1] x")], reply: `{"groups":${groups}}` });
+ assert.notEqual(r.status, 0, "a falsy non-array must not publish findings as judged");
+ assert.match(r.stderr.split("\n")[0], /judge returned a non-array groups/);
+ assert.equal(existsSync(r.out), false, "nothing may be written on a schema violation");
+ });
+ }
+
+ // The two exemptions are named on purpose and mean "the judge classified
+ // nothing", which the fail-open pass handles deliberately. Tightening the
+ // guard must not sweep them up.
+ for (const [name, groups] of [["absent", "{}"], ["null", '{"groups":null}']]) {
+ test(`groups ${name} still falls open, not closed`, async () => {
+ const r = await judge({ comments: [finding("[P1] x")], reply: groups, threshold: 0.7 });
+ assert.equal(r.status, 0, r.stderr);
+ assert.equal(r.read().comments.length, 1, "an unclassified finding is kept");
+ });
+ }
+});
+
describe("request shape", () => {
test("the model, a zero temperature and bearer auth are sent", async () => {
const r = await judge({
diff --git a/scripts/localconfig.test.mjs b/scripts/localconfig.test.mjs
new file mode 100644
index 0000000..f005a7c
--- /dev/null
+++ b/scripts/localconfig.test.mjs
@@ -0,0 +1,275 @@
+// .orcacode-review.json — the local review's one settings file.
+//
+// The contract worth guarding is that the file can never be HALF-read: every
+// malformed shape is refused by name, and a valid file changes exactly what it
+// says and nothing else.
+
+import test from "node:test";
+import assert from "node:assert/strict";
+import fs from "node:fs";
+import os from "node:os";
+import path from "node:path";
+import { execFileSync, spawnSync } from "node:child_process";
+import { fileURLToPath } from "node:url";
+
+import { CONFIG_FILE, CONFIG_KEYS, loadLocalConfig, configTemplate } from "../bin/localconfig.mjs";
+import { buildPlan, renderPlan } from "../bin/harness.mjs";
+
+const CLI = fileURLToPath(new URL("../bin/orcacode-review.mjs", import.meta.url));
+
+function scratch(t, config) {
+ const dir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ocr-cfg-"));
+ t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
+ if (config !== undefined) {
+ fs.writeFileSync(path.join(dir, CONFIG_FILE), typeof config === "string" ? config : JSON.stringify(config));
+ }
+ return dir;
+}
+
+function repo(t, config) {
+ const dir = scratch(t, config);
+ const run = (...a) => execFileSync("git", a, { cwd: dir, encoding: "utf8" });
+ run("init", "-q", "-b", "main");
+ run("config", "user.email", "t@t");
+ run("config", "user.name", "t");
+ // A base commit, so `main~1..HEAD` is a real range in every test below.
+ run("commit", "-q", "--allow-empty", "-m", "init");
+ return { dir, run };
+}
+
+// ---------------------------------------------------------------- loading ---
+
+test("no file means defaults, not an error", (t) => {
+ const r = loadLocalConfig(scratch(t));
+ assert.deepEqual(r, { ok: true, file: null, config: {} });
+});
+
+test("a valid file is normalised: block_on to a string, rules with their text read in", (t) => {
+ const dir = scratch(t);
+ fs.mkdirSync(path.join(dir, "docs"));
+ fs.writeFileSync(path.join(dir, "docs/api.md"), "Check tenant scoping.\n");
+ fs.writeFileSync(
+ path.join(dir, CONFIG_FILE),
+ JSON.stringify({
+ $comment: "ignored",
+ block_on: ["p0"],
+ language: "zh",
+ exclude: [" docs/** "],
+ rules: [
+ { path: "src/api/**/*.ts", rule_file: "docs/api.md" },
+ { path: "**/*.sql", rule: "Reversible?", replace: true },
+ ],
+ }),
+ );
+ const r = loadLocalConfig(dir);
+ assert.equal(r.ok, true);
+ assert.equal(r.file, CONFIG_FILE);
+ assert.equal(r.config.block_on, "P0");
+ assert.equal(r.config.language, "zh");
+ assert.deepEqual(r.config.exclude, ["docs/**"]);
+ assert.deepEqual(r.config.rules, [
+ { path: "src/api/**/*.ts", text: "Check tenant scoping.", replace: false, source: "docs/api.md" },
+ { path: "**/*.sql", text: "Reversible?", replace: true, source: "inline" },
+ ]);
+});
+
+test('block_on "" is a valid "never block", and is kept distinct from absent', (t) => {
+ assert.equal(loadLocalConfig(scratch(t, { block_on: "" })).config.block_on, "");
+ assert.equal("block_on" in loadLocalConfig(scratch(t, {})).config, false);
+});
+
+// Each of these would, if tolerated, make the file lie about what the review
+// does. So each is refused, and the message names the key.
+for (const [label, body, expect] of [
+ ["not JSON", "{ block_on: P0 }", /not valid JSON/],
+ ["an array", "[]", /must be a JSON object/],
+ ["a typo'd key", { blockOn: "P0" }, /unknown key "blockOn" — allowed: block_on, language, exclude, rules/],
+ ["a bad severity", { block_on: "P0,P9" }, /"block_on" has "P9"/],
+ ["block_on as a number", { block_on: 1 }, /"block_on" must be a string/],
+ ["an unknown language", { language: "de" }, /"language" must be one of en, zh, ja, ko/],
+ ["exclude as a string", { exclude: "docs/**" }, /"exclude" must be an array/],
+ ["an empty exclude glob", { exclude: [""] }, /"exclude" must be an array/],
+ ["rules as an object", { rules: {} }, /"rules" must be an array/],
+ ["a rule without a path", { rules: [{ rule: "x" }] }, /"rules"\[0\] needs a "path"/],
+ ["a rule with neither text nor file", { rules: [{ path: "*.ts" }] }, /exactly one of "rule"/],
+ ["a rule with both text and file", { rules: [{ path: "*.ts", rule: "x", rule_file: "y" }] }, /exactly one of "rule"/],
+ ["a rule with a stray key", { rules: [{ path: "*.ts", rule: "x", when: "always" }] }, /"rules"\[0\] has unknown key "when"/],
+ ["a non-boolean replace", { rules: [{ path: "*.ts", rule: "x", replace: "yes" }] }, /"replace" must be true or false/],
+ ["an empty rule", { rules: [{ path: "*.ts", rule: " " }] }, /rule text is empty/],
+ ["a missing rule_file", { rules: [{ path: "*.ts", rule_file: "nope.md" }] }, /could not be read/],
+ ["a rule_file outside the repo", { rules: [{ path: "*.ts", rule_file: "../../etc/hostname" }] }, /must stay inside the repository/],
+]) {
+ test(`refuses ${label}, naming the problem`, (t) => {
+ const r = loadLocalConfig(scratch(t, body));
+ assert.equal(r.ok, false);
+ assert.equal(r.file, CONFIG_FILE);
+ assert.match(r.error, expect);
+ });
+}
+
+test("the template is valid, carries the caller's language, and round-trips", (t) => {
+ const text = configTemplate({ language: "ja", comment: "note" });
+ const dir = scratch(t, text);
+ const r = loadLocalConfig(dir);
+ assert.equal(r.ok, true);
+ assert.equal(r.config.language, "ja");
+ assert.equal(r.config.block_on, "P0,P1");
+ assert.deepEqual(Object.keys(JSON.parse(text)), ["$comment", ...CONFIG_KEYS]);
+});
+
+// ----------------------------------------------------------------- effect ---
+
+test("exclude and rules from the file reach the plan, with the reason and the merged checklist", (t) => {
+ const { dir, run } = repo(t);
+ fs.mkdirSync(path.join(dir, "docs"));
+ fs.mkdirSync(path.join(dir, "src"));
+ fs.writeFileSync(path.join(dir, "docs/x.ts"), "1\n");
+ fs.writeFileSync(path.join(dir, "src/a.ts"), "1\n");
+ fs.writeFileSync(path.join(dir, "src/b.sql"), "select 1;\n");
+ fs.writeFileSync(
+ path.join(dir, CONFIG_FILE),
+ JSON.stringify({
+ exclude: ["docs/**"],
+ rules: [
+ { path: "src/**/*.ts", rule: "PROJECT-TS-RULE" },
+ { path: "**/*.sql", rule: "PROJECT-SQL-RULE", replace: true },
+ ],
+ }),
+ );
+ run("add", "-A");
+ run("commit", "-q", "-m", "x");
+ const plan = buildPlan({ from: "main~1", to: "HEAD" }, dir);
+
+ assert.deepEqual(plan.excluded.map((f) => [f.path, f.code]), [["docs/x.ts", "project_exclude"]]);
+ assert.equal(plan.config.file, CONFIG_FILE);
+
+ const ts = plan.rule_groups.find((g) => g.files.includes("src/a.ts"));
+ assert.equal(ts.source, "project");
+ assert.equal(ts.pattern, "src/**/*.ts");
+ assert.match(ts.rule, /^PROJECT-TS-RULE\n\n/, "project text first");
+ assert.match(ts.rule, /Dead Code/, "the bundled TS checklist is kept underneath");
+
+ const sql = plan.rule_groups.find((g) => g.files.includes("src/b.sql"));
+ assert.equal(sql.rule, "PROJECT-SQL-RULE", "replace drops the bundled checklist");
+
+ assert.match(renderPlan(plan), /project settings: `\.orcacode-review\.json` \(1 extra exclude\) \(2 project rules\)/);
+});
+
+test("without a file, plan.config is null and the plan carries the first-run offer", (t) => {
+ const { dir, run } = repo(t);
+ fs.writeFileSync(path.join(dir, "a.ts"), "1\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const plan = buildPlan({ from: "main~1", to: "HEAD", language: "zh" }, dir);
+ assert.equal(plan.config, null);
+ const text = renderPlan(plan);
+ assert.ok(!/project settings/.test(text));
+ // The onboarding is in the plan — the one document the agent reads every
+ // time — and it is explicit about when and how often.
+ assert.match(text, /## First review in this repository/);
+ assert.match(text, /After you have shown the report — not before — offer ONCE/);
+ assert.match(text, /review config init --lang zh/);
+ assert.match(text, /Do not create the file unasked/);
+ assert.match(text, /findings in 简体中文/, "it names the language this run used, so 'save these' is concrete");
+});
+
+test("with a file, the first-run offer is gone", (t) => {
+ const { dir, run } = repo(t, { block_on: "P0" });
+ fs.writeFileSync(path.join(dir, "a.ts"), "1\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const text = renderPlan(buildPlan({ from: "main~1", to: "HEAD" }, dir));
+ assert.ok(!/First review in this repository/.test(text));
+});
+
+test("init pre-fills block_on from --block-on, so 'save what we used' is one command", (t) => {
+ const { dir } = repo(t);
+ const r = cli(dir, "review", "config", "init", "--lang", "zh", "--block-on", "p0");
+ assert.equal(r.status, 0);
+ const written = JSON.parse(fs.readFileSync(path.join(dir, CONFIG_FILE), "utf8"));
+ assert.equal(written.block_on, "P0");
+ assert.equal(written.language, "zh");
+ assert.equal(loadLocalConfig(dir).ok, true, "and what it wrote is valid by its own rules");
+});
+
+// -------------------------------------------------------------------- CLI ---
+
+const cli = (dir, ...args) =>
+ spawnSync(process.execPath, [CLI, ...args], { cwd: dir, encoding: "utf8", env: { ...process.env, NO_COLOR: "1" } });
+
+test("plan and submit refuse an invalid file with exit 2 — never defaults", (t) => {
+ const { dir, run } = repo(t, { block_on: "P7" });
+ fs.writeFileSync(path.join(dir, "a.ts"), "1\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const plan = cli(dir, "review", "plan", "--lang", "en", "--from", "main~1", "--to", "HEAD");
+ assert.equal(plan.status, 2);
+ assert.match(plan.stderr, /\.orcacode-review\.json is invalid: "block_on" has "P7"/);
+ const submit = cli(dir, "review", "submit", "--lang", "en");
+ assert.equal(submit.status, 2);
+ assert.match(submit.stderr, /is invalid/);
+});
+
+test("block_on: flag beats file beats default, and the file's win is announced", (t) => {
+ const { dir, run } = repo(t, { block_on: "P0" });
+ fs.writeFileSync(path.join(dir, "x.js"), "export const f = (a) => a.b;\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "x");
+ fs.mkdirSync(path.join(dir, ".orcacode-review"));
+ fs.writeFileSync(
+ path.join(dir, ".orcacode-review/result.json"),
+ JSON.stringify({ comments: [{ path: "x.js", line: 1, existing_code: "export const f", content: "[P1] **T**\n\nb" }] }),
+ );
+ fs.writeFileSync(path.join(dir, ".orcacode-review/plan.json"), JSON.stringify({ ground_truth_ref: run("rev-parse", "HEAD").trim() }));
+
+ const fromFile = cli(dir, "review", "submit", "--format", "json", "--lang", "en");
+ assert.equal(JSON.parse(fromFile.stdout).blocked, false, "the file says P0 only, so a P1 does not block");
+ assert.match(fromFile.stderr, /Blocking on P0 — from \.orcacode-review\.json/);
+
+ const fromFlag = cli(dir, "review", "submit", "--format", "json", "--lang", "en", "--block-on", "P0,P1");
+ assert.equal(JSON.parse(fromFlag.stdout).blocked, true, "the flag outranks the file");
+ assert.ok(!/from \.orcacode-review\.json/.test(fromFlag.stderr), "and the file's setting is not announced when it lost");
+});
+
+test("language: the file outranks the locale and yields to --lang", (t) => {
+ const { dir, run } = repo(t, { language: "zh" });
+ fs.writeFileSync(path.join(dir, "a.ts"), "1\n");
+ run("add", "-A");
+ run("commit", "-q", "-m", "a");
+ const env = { ...process.env, NO_COLOR: "1", LANG: "en_US.UTF-8", LC_ALL: "en_US.UTF-8", ORCACODE_LANG: "" };
+ const byFile = spawnSync(process.execPath, [CLI, "review", "plan", "--from", "main~1", "--to", "HEAD"], { cwd: dir, encoding: "utf8", env });
+ assert.match(byFile.stdout, /in 简体中文/, "the plan asks for Chinese findings because the file said so");
+ assert.match(byFile.stderr, /已读取/, "and the CLI speaks Chinese too");
+ const byFlag = spawnSync(process.execPath, [CLI, "review", "plan", "--from", "main~1", "--to", "HEAD", "--lang", "ja"], { cwd: dir, encoding: "utf8", env });
+ assert.match(byFlag.stdout, /in 日本語/);
+});
+
+test("review config shows each value and its source; init writes the template once", (t) => {
+ const { dir } = repo(t);
+ const none = cli(dir, "review", "config", "--lang", "en");
+ assert.equal(none.status, 0);
+ assert.match(none.stdout, /block_on\s+P0,P1\s+← the default/);
+ assert.match(none.stdout, /language\s+en\s+← the command line/);
+ assert.match(none.stdout, /review config init/, "tells the user how to create one");
+
+ const init = cli(dir, "review", "config", "init", "--lang", "zh");
+ assert.equal(init.status, 0);
+ const written = JSON.parse(fs.readFileSync(path.join(dir, CONFIG_FILE), "utf8"));
+ assert.equal(written.language, "zh", "the template is pre-filled with the language the user was using");
+ assert.match(written.$comment, /block_on/);
+
+ const again = cli(dir, "review", "config", "init", "--lang", "en");
+ assert.equal(again.status, 1, "refuses to overwrite");
+ assert.match(again.stdout + again.stderr, /already exists; pass --force/);
+
+ fs.writeFileSync(path.join(dir, CONFIG_FILE), JSON.stringify({ block_on: "P0", exclude: ["docs/**"] }));
+ const show = cli(dir, "review", "config", "--lang", "en");
+ assert.match(show.stdout, /block_on\s+P0\s+← \.orcacode-review\.json/);
+ assert.match(show.stdout, /exclude\s+1 extra exclude\s+docs\/\*\*/);
+ assert.match(show.stdout, /edit \.orcacode-review\.json directly/);
+
+ const json = cli(dir, "review", "config", "--json", "--lang", "en");
+ const parsed = JSON.parse(json.stdout);
+ assert.deepEqual(parsed.block_on, { value: "P0", source: "file" });
+ assert.deepEqual(parsed.exclude, ["docs/**"]);
+});
diff --git a/scripts/platforms.test.mjs b/scripts/platforms.test.mjs
index 40c1106..2e82683 100644
--- a/scripts/platforms.test.mjs
+++ b/scripts/platforms.test.mjs
@@ -259,14 +259,14 @@ test("no temp files survive an install", () => {
test("the bundled skill is a valid Agent Skill for every host", () => {
// Codex, OpenCode, and Claude Code all require exactly `name` + `description`
// in the frontmatter, and `name` must equal the directory name.
- const dir = new URL("../skills/setup-orca-code-review/", import.meta.url);
+ const dir = new URL("../skills/orca-review-action/", import.meta.url);
const text = fs.readFileSync(new URL("SKILL.md", dir), "utf8");
const frontmatter = text.match(/^---\n([\s\S]*?)\n---\n/);
assert.ok(frontmatter, "SKILL.md has no YAML frontmatter");
const name = frontmatter[1].match(/^name:\s*(\S+)/m);
const description = frontmatter[1].match(/^description:\s*(.+)/m);
- assert.equal(name?.[1], "setup-orca-code-review");
+ assert.equal(name?.[1], "orca-review-action");
assert.ok(description, "SKILL.md has no description");
assert.ok(description[1].length <= 1024, "description exceeds the 1024-char limit hosts enforce");
assert.match(name[1], /^[a-z0-9]+(-[a-z0-9]+)*$/, "name must be lowercase with single hyphens");
diff --git a/scripts/recipe.test.mjs b/scripts/recipe.test.mjs
new file mode 100644
index 0000000..5589334
--- /dev/null
+++ b/scripts/recipe.test.mjs
@@ -0,0 +1,134 @@
+// Contract tests for the shipped routing recipes.
+//
+// These files are documentation with consequences: an operator pastes one into
+// their router and it becomes the live model policy. They have drifted twice —
+// once on the fact contract (`x-cr-prev-tier` documented as none|cheap|strong
+// after only one value was ever sent), and once by losing the judge rule while
+// the copy inside the control plane kept it. Both were caught by a person
+// reading the file, which is the wrong last line of defence.
+//
+// Cross-repo agreement CANNOT be asserted here — the authority for what setup
+// provisions is a Go constant in another repository, and nothing in this one can
+// see it. What these tests do instead is pin the properties that make a recipe
+// self-consistent, so it cannot silently decay into something that parses and
+// routes wrongly.
+
+import { readFileSync, readdirSync } from "node:fs";
+import { join, dirname } from "node:path";
+import { fileURLToPath } from "node:url";
+import { describe, test } from "node:test";
+import assert from "node:assert/strict";
+
+const RECIPES = join(dirname(fileURLToPath(import.meta.url)), "..", "recipes");
+const ACTION_RECIPE = "orcacode-review.dsl.yaml";
+
+const read = (name) => readFileSync(join(RECIPES, name), "utf8");
+
+// The rule ids and models, without a YAML parser: these files are line-oriented
+// by convention and the shape is what is being asserted.
+function rules(src) {
+ const out = [];
+ let id = null;
+ for (const line of src.split("\n")) {
+ const idMatch = line.match(/^\s*-\s*id:\s*(\S+)/);
+ if (idMatch) {
+ id = idMatch[1];
+ continue;
+ }
+ const useMatch = line.match(/^\s*use:\s*\{\s*model:\s*"([^"]+)"/);
+ if (useMatch && id) {
+ out.push({ id, model: useMatch[1] });
+ id = null;
+ }
+ }
+ return out;
+}
+
+function defaultModel(src) {
+ const i = src.lastIndexOf("default:");
+ if (i < 0) return "";
+ const m = src.slice(i).match(/model:\s*"([^"]+)"/);
+ return m ? m[1] : "";
+}
+
+describe("every shipped recipe", () => {
+ const names = readdirSync(RECIPES).filter((f) => f.endsWith(".dsl.yaml"));
+
+ test("there is at least one, and this suite sees the Action's", () => {
+ assert.ok(names.length > 0, "no recipes found — has the directory moved?");
+ assert.ok(names.includes(ACTION_RECIPE), `${ACTION_RECIPE} is missing`);
+ });
+
+ for (const name of names) {
+ test(`${name}: routes everything somewhere`, () => {
+ const src = read(name);
+ assert.match(src, /^version:\s*1\s*$/m, "a recipe needs a version");
+ assert.ok(defaultModel(src), "a recipe with no default can drop a request");
+ for (const r of rules(src)) {
+ assert.ok(r.model, `rule ${r.id} names no model`);
+ }
+ });
+
+ test(`${name}: every rule can actually be reached`, () => {
+ // A rule whose condition nothing sends is worse than no rule: the router
+ // reads as a policy the workspace is not running. The Action stamps
+ // x-cr-lens only on the judge call, and x-cr-prev-tier/p0p1 on every call.
+ const src = read(name);
+ const conditions = src.split("\n").filter((l) => /^\s*when:/.test(l));
+ for (const c of conditions) {
+ assert.match(
+ c,
+ /headers\["x-cr-(lens|prev-tier|prev-p0p1)"\]/,
+ `condition keys on a fact nothing sends: ${c.trim()}`,
+ );
+ }
+ });
+ }
+});
+
+describe("the Action's recipe", () => {
+ test("keeps the judge rule", () => {
+ // Deleting it does not disable the judge — the Action runs the L2 pass either
+ // way — it sends the judge to `default:`, i.e. the reviewer's own model. This
+ // file lost the rule once while the control plane's copy kept it, so a
+ // workspace pasting it got a judge that grades its own work.
+ const found = rules(read(ACTION_RECIPE)).find((r) => r.id === "judge");
+ assert.ok(found, "no judge rule — a pasted copy would send the judge to the default");
+ });
+
+ test("does not point the judge at the default's model", () => {
+ // The whole property, and it fails silently: a judge sharing the reviewer's
+ // model agrees with it and still reports the pass as successful.
+ const src = read(ACTION_RECIPE);
+ const judge = rules(src).find((r) => r.id === "judge");
+ assert.ok(judge, "no judge rule to check");
+ assert.notEqual(
+ judge.model,
+ defaultModel(src),
+ "the judge names the default's model — that is not an independent second opinion",
+ );
+ });
+
+ test("carries no per-angle rule, which no Action call can reach", () => {
+ // The review call stamps no angle. Those rules belong to the multi-angle
+ // reviewer, and shipping them here described a policy that never applied.
+ const ids = rules(read(ACTION_RECIPE)).map((r) => r.id);
+ for (const lens of ["ripple", "parity", "ordering", "failure", "assumption", "conventions"]) {
+ assert.ok(!ids.includes(lens), `carries the ${lens} rule, which the Action never triggers`);
+ }
+ });
+
+ test("names the router by the alias the action defaults to", () => {
+ // The recipe tells the reader which router to paste it into. When the router
+ // was renamed, this line was the one that had to move with it — and the
+ // failure mode of getting it wrong is a not-found on every review.
+ const src = read(ACTION_RECIPE);
+ const actionYml = readFileSync(join(RECIPES, "..", "action.yml"), "utf8");
+ const m = actionYml.match(/default:\s*"orcarouter\/([a-z0-9-]+)"/);
+ assert.ok(m, "could not read the router input's default from action.yml");
+ assert.ok(
+ src.includes(`orcarouter/${m[1]}`),
+ `the recipe does not mention orcarouter/${m[1]}, which is what the action asks for`,
+ );
+ });
+});
diff --git a/scripts/selection.test.mjs b/scripts/selection.test.mjs
new file mode 100644
index 0000000..7f4ba5c
--- /dev/null
+++ b/scripts/selection.test.mjs
@@ -0,0 +1,230 @@
+// The vendored Open Code Review rules and the port of its matcher.
+//
+// Two things are worth guarding: that the matcher agrees with doublestar and
+// with gitignore where upstream's Go does (so local file selection stays the
+// selection CI applies), and that the vendored corpus is complete — every
+// checklist the path map names must exist, or a plan quietly loses a rule.
+
+import test from "node:test";
+import assert from "node:assert/strict";
+import fs from "node:fs";
+import path from "node:path";
+
+import {
+ VENDOR_DIR,
+ IGNORED_DIRS,
+ REASON,
+ expandBraces,
+ globMatch,
+ isIgnored,
+ extOf,
+ isAllowedExt,
+ isDefaultExcludedPath,
+ excludeReason,
+ partition,
+ resolveRule,
+ groupRules,
+} from "../bin/selection.mjs";
+
+// ------------------------------------------------------------------- glob ---
+
+test("globstar spans zero or more directories, wherever it sits", () => {
+ for (const [p, s] of [
+ ["**/*.go", "a.go"],
+ ["**/*.go", "x/y/a.go"],
+ ["**/__tests__/**", "__tests__/x.js"],
+ ["**/__tests__/**", "src/__tests__/x.js"],
+ ["a/**/b", "a/b"],
+ ["a/**/b", "a/x/y/b"],
+ ["lib/**/*.sol", "lib/forge-std/x.sol"],
+ ]) assert.equal(globMatch(p, s), true, `${p} should match ${s}`);
+ for (const [p, s] of [
+ ["**/*.go", "a.py"],
+ ["a/**/b", "a/xb"],
+ ["lib/**/*.sol", "src/x.sol"],
+ ["**/__tests__/**", "src/tests/x.js"],
+ ]) assert.equal(globMatch(p, s), false, `${p} should not match ${s}`);
+});
+
+test("a single star never crosses a slash — the root-only idiom depends on it", () => {
+ assert.equal(globMatch("*_test.go", "foo_test.go"), true);
+ assert.equal(globMatch("*_test.go", "pkg/foo_test.go"), false);
+});
+
+test("braces expand at every depth, and classes and ? are honoured", () => {
+ assert.deepEqual(expandBraces("**/*.{a,b}.{x,y}"), ["**/*.a.x", "**/*.a.y", "**/*.b.x", "**/*.b.y"]);
+ assert.equal(globMatch("**/*.test.{js,jsx,ts,tsx}", "src/app.test.tsx"), true);
+ assert.equal(globMatch("**/*{mapper,dao}*.xml", "src/usermapper.xml"), true);
+ assert.equal(globMatch("**/*.[ch]", "x.c"), true);
+ assert.equal(globMatch("**/*.[ch]", "x.o"), false);
+ assert.equal(globMatch("a?c", "abc"), true);
+ assert.equal(globMatch("a?c", "a/c"), false);
+});
+
+// --------------------------------------------------------------- gitignore ---
+
+const GI = ["*.log", "build/", "!keep.log", "/root.txt", "docs/*.md", "**/gen/*.js"];
+
+test("gitignore is resolved in order with last match winning and ! inverting", () => {
+ assert.equal(isIgnored("app.log", GI), true);
+ assert.equal(isIgnored("keep.log", GI), false, "a later negation re-admits");
+});
+
+test("a slashless pattern matches a basename at any depth; a slash anchors it", () => {
+ assert.equal(isIgnored("deep/app.log", GI), true);
+ assert.equal(isIgnored("root.txt", GI), true);
+ assert.equal(isIgnored("sub/root.txt", GI), false, "a leading / names the root file only");
+ assert.equal(isIgnored("docs/a.md", GI), true);
+ // Same as git: "docs/*.md" is relative to the .gitignore, not "any docs dir".
+ assert.equal(isIgnored("x/docs/a.md", GI), false);
+});
+
+test("a directory pattern excludes everything below a directory of that name", () => {
+ assert.equal(isIgnored("build/a.js", GI), true);
+ assert.equal(isIgnored("src/build/a.js", GI), true);
+ assert.equal(isIgnored("build", GI), false, "a file named like the directory is not the directory");
+});
+
+test("the always-skipped directories cannot be re-admitted by a negation", () => {
+ assert.equal(isIgnored("node_modules/x.js", ["!node_modules/"]), true);
+ assert.equal(isIgnored("vendor/y.go", []), true);
+ assert.equal(isIgnored("vendored/y.go", []), false, "prefix means a path component, not a string prefix");
+ assert.ok(IGNORED_DIRS.includes(".git/"));
+});
+
+// --------------------------------------------------------------- exclusion ---
+
+test("extension is taken from the basename, lowercased, and a dotfile has none", () => {
+ assert.equal(extOf("a/B.GO"), ".go");
+ assert.equal(extOf(".env"), "");
+ assert.equal(extOf("Makefile"), "");
+ assert.equal(extOf("dir.v1/file"), "");
+});
+
+test("exclusion reasons follow upstream's order and vocabulary", () => {
+ const r = (f) => excludeReason(f, []);
+ assert.equal(r({ path: "src/a.go" }), "");
+ assert.equal(r({ path: "src/a_test.go" }), "default_path");
+ assert.equal(r({ path: "img.png" }), "unsupported_ext");
+ assert.equal(r({ path: "a.bin", binary: true }), "binary");
+ assert.equal(r({ path: "old.go", deleted: true }), "deleted");
+ // Extensionless files pass the extension gate: no extension is not an
+ // unsupported one.
+ assert.equal(r({ path: "Makefile" }), "");
+ // Ignore list beats everything, including "it is binary".
+ assert.equal(r({ path: "node_modules/a.bin", binary: true }), "ignored");
+ // A deleted test file is reported as a test file — deleted is the last check.
+ assert.equal(r({ path: "a_test.go", deleted: true }), "default_path");
+ for (const code of Object.keys(REASON)) assert.ok(REASON[code], `reason text for ${code}`);
+});
+
+test("partition keeps order and attaches a human reason to each exclusion", (t) => {
+ const dir = fs.mkdtempSync(path.join(fs.realpathSync(process.env.TMPDIR || "/tmp"), "sel-"));
+ t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
+ fs.writeFileSync(path.join(dir, ".gitignore"), "*.log\n");
+ const { files, excluded } = partition(
+ [{ path: "src/a.ts" }, { path: "debug.log" }, { path: "src/a.test.ts" }, { path: "b.py" }],
+ dir,
+ );
+ assert.deepEqual(files.map((f) => f.path), ["src/a.ts", "b.py"]);
+ assert.deepEqual(
+ excluded.map((e) => [e.path, e.code]),
+ [["debug.log", "ignored"], ["src/a.test.ts", "default_path"]],
+ );
+ assert.equal(excluded[0].reason, REASON.ignored);
+});
+
+test("the allowlist and the default excludes are case-insensitive", () => {
+ assert.equal(isAllowedExt(".GO"), true);
+ assert.equal(isDefaultExcludedPath("SRC/FOO_TEST.GO"), true);
+});
+
+// ------------------------------------------------------------------- rules ---
+
+test("first matching pattern wins, in file order, so the specific beats the general", () => {
+ // Three patterns match a workflow file; the one declared first is the answer.
+ assert.equal(resolveRule(".github/workflows/ci.yml").pattern, ".github/workflows/**/*.{yaml,yml}");
+ assert.equal(resolveRule(".github/dependabot.yml").pattern, ".github/**/*.{yaml,yml}");
+ assert.equal(resolveRule("conf/app.yaml").pattern, "**/*.{yaml,yml}");
+ assert.equal(resolveRule("x/pom.xml").pattern, "**/pom.xml");
+});
+
+test("an unmatched path gets the default checklist, named as such", () => {
+ const r = resolveRule("README.md");
+ assert.equal(r.pattern, "default");
+ assert.match(r.rule, /#### Correctness/);
+});
+
+test("matching is case-insensitive on both sides", () => {
+ assert.equal(resolveRule("a/b.R").pattern, "**/*.R");
+ assert.equal(resolveRule("a/b.r").pattern, "**/*.R");
+ assert.equal(resolveRule("SRC/MAIN.GO").pattern, "**/*.go");
+});
+
+test("a .m file is MATLAB by path and Objective-C by content", (t) => {
+ const dir = fs.mkdtempSync(path.join(fs.realpathSync(process.env.TMPDIR || "/tmp"), "sniff-"));
+ t.after(() => fs.rmSync(dir, { recursive: true, force: true }));
+ fs.writeFileSync(path.join(dir, "plot.m"), "% MATLAB comment\nx = 1;\n");
+ fs.writeFileSync(path.join(dir, "View.m"), "\n#import \n@implementation View\n");
+ fs.writeFileSync(path.join(dir, "hdr.m"), "// license header\n#import \n");
+ const matlab = resolveRule("plot.m", { repoDir: dir });
+ const objc = resolveRule("View.m", { repoDir: dir });
+ const commented = resolveRule("hdr.m", { repoDir: dir });
+ assert.match(matlab.rule, /MATLAB/i);
+ assert.match(objc.rule, /Objective-C/i);
+ assert.match(commented.rule, /Objective-C/i, "a C comment opener is itself an ObjC signal");
+ // The glob that matched is reported unchanged; the sniff only swaps the text.
+ assert.equal(objc.pattern, "**/*.m");
+ // Unreadable: the path-based answer stands.
+ assert.match(resolveRule("missing.m", { repoDir: dir }).rule, /MATLAB/i);
+});
+
+test("groups join files only when pattern and checklist both coincide", () => {
+ const g = groupRules(["a.go", "b/c.go", "x.ts", "y.tsx", "README.md"]);
+ assert.deepEqual(
+ g.map((x) => [x.group_id, x.pattern, x.files]),
+ [
+ [1, "**/*.go", ["a.go", "b/c.go"]],
+ [2, "**/*.{ts,js,tsx,jsx,mjs,cjs}", ["x.ts", "y.tsx"]],
+ [3, "default", ["README.md"]],
+ ],
+ );
+ for (const x of g) {
+ assert.equal(x.source, "system");
+ assert.ok(x.rule.length > 0);
+ }
+});
+
+// ------------------------------------------------------------------ vendor ---
+
+test("every checklist the path map names is present in the vendored corpus", () => {
+ const sys = JSON.parse(fs.readFileSync(path.join(VENDOR_DIR, "system_rules.json"), "utf8"));
+ const docs = new Set(fs.readdirSync(path.join(VENDOR_DIR, "rule_docs")));
+ const named = new Set([sys.default_rule, "objc.md", ...Object.values(sys.path_rule_map)]);
+ for (const d of named) assert.ok(docs.has(d), `missing rule doc ${d}`);
+ // And nothing vendored that nothing names — a stray file is a sign the copy
+ // came from a different tag than UPSTREAM says.
+ for (const d of docs) assert.ok(named.has(d), `unreferenced rule doc ${d}`);
+});
+
+test("the vendored directory carries its licence and provenance", () => {
+ const license = fs.readFileSync(path.join(VENDOR_DIR, "LICENSE"), "utf8");
+ assert.match(license, /Apache License/);
+ assert.match(license, /Version 2\.0/);
+ const upstream = fs.readFileSync(path.join(VENDOR_DIR, "UPSTREAM"), "utf8");
+ assert.match(upstream, /alibaba\/open-code-review/);
+ assert.match(upstream, /tag\s+v\d+\.\d+\.\d+/);
+ assert.match(upstream, /commit\s+[0-9a-f]{40}/);
+});
+
+test("the package ships the vendored corpus", () => {
+ const pkg = JSON.parse(fs.readFileSync(new URL("../package.json", import.meta.url), "utf8"));
+ assert.ok(pkg.files.includes("vendor/"), "vendor/ must be in package.json files or the plan has no rules");
+});
+
+test("nothing in the harness spawns ocr any more", () => {
+ for (const f of ["harness.mjs", "review.mjs", "selection.mjs"]) {
+ const src = fs.readFileSync(new URL(`../bin/${f}`, import.meta.url), "utf8");
+ assert.ok(!/spawnSync\(\s*"ocr"/.test(src), `${f} still shells out to ocr`);
+ }
+});
diff --git a/scripts/settings.mjs b/scripts/settings.mjs
index f6fd613..82612eb 100644
--- a/scripts/settings.mjs
+++ b/scripts/settings.mjs
@@ -47,16 +47,21 @@ const DEFAULTS = Object.freeze({
quiet: false,
fix_first: "P0,P1",
block_on: "P0,P1",
- // EVERY severity, and this is the FAILURE value rather than the product
- // default. A workspace created after report_on shipped gets P0,P1 written
- // explicitly by the gateway, while one that predates the setting still
- // resolves to every severity — so no single value here can match "what this
- // workspace normally sees".
+ // THE VALUE THE GATEWAY WOULD HAVE GIVEN, which is the whole job of a failure
+ // default: reproduce the answer we could not fetch.
//
- // So it fails OPEN. An install that normally shows P2/P3 must not lose them
- // because a settings call timed out; the opposite mistake only shows a reader
- // more than they asked for, once, in a run that already logged a fetch failure.
- report_on: "P0,P1,P2,P3",
+ // This used to be every severity, on the reasoning that a workspace predating
+ // report_on resolved to all four, so no single value could match "what this
+ // workspace normally sees" — and losing P2/P3 to a timeout was judged worse
+ // than showing them once. That premise expired. The gateway's own default is
+ // now P0,P1 for every workspace that has not set the field, so all-severities
+ // matches nothing any more: on a settings outage it published P2/P3 to
+ // installs that would never have been shown them.
+ //
+ // Which is not a cosmetic mistake here. These severities are the ones the
+ // judge exists to filter, so an outage that opens this up is also the outage
+ // most likely to have degraded the filtering.
+ report_on: "P0,P1",
rubric: "",
});
@@ -94,8 +99,20 @@ function validateSettings(data) {
// report_on rides in this loop because it is the same shape as the other two —
// a comma-joined severity subset where "" is a deliberate "none" — but it
// answers a different question: fix_first and block_on decide what the review
- // ENFORCES, report_on only what it SHOWS. A gateway that predates the field
- // omits it, which lands on the default above and behaves as before.
+ // ENFORCES, report_on only what it SHOWS.
+ //
+ // A RESPONSE THAT OMITS THE FIELD LANDS ON THE DEFAULT ABOVE, and that is no
+ // longer the same thing as "behaves as before" — this comment used to say so
+ // and was left standing when the default moved from all four severities to
+ // P0,P1. An omitted field now resolves to P0,P1, and since the report filter
+ // shows report_on UNION block_on (block_on also defaulting to P0,P1), P2/P3
+ // are not recovered anywhere else.
+ //
+ // Which is deliberate, and rests on a fact rather than on compatibility: the
+ // gateway sets P0,P1 itself for every workspace that has not chosen a value,
+ // so an omitted field means "this workspace never set one", not "this server
+ // is too old to have the field". Reproducing the answer we could not fetch is
+ // the whole job of a failure default, and the answer is P0,P1.
for (const field of ["fix_first", "block_on", "report_on"]) {
const normalized = normalizeSeverityList(data[field]);
if (normalized !== null) out[field] = normalized;
diff --git a/scripts/settings.test.mjs b/scripts/settings.test.mjs
index 7bf3d67..5efbcb8 100644
--- a/scripts/settings.test.mjs
+++ b/scripts/settings.test.mjs
@@ -45,10 +45,11 @@ const DEFAULTS = {
quiet: false,
fix_first: "P0,P1",
block_on: "P0,P1",
- // Fail-open, NOT the product default: this is what the action falls back to
- // when the dashboard is unreachable, and an outage must never silently
- // withhold findings. The configured default is narrower and lives server-side.
- report_on: "P0,P1,P2,P3",
+ // What the action falls back to when the dashboard is unreachable. It mirrors
+ // the gateway's own default for a workspace that never set the field, so an
+ // outage reproduces the answer it could not fetch. Kept in sync with
+ // settings.mjs DEFAULTS — see the reasoning there.
+ report_on: "P0,P1",
rubric: "",
};
@@ -247,7 +248,7 @@ describe("settings: field-wise fallback on invalid values", () => {
quiet: false,
fix_first: "P0,P1",
block_on: "P0,P1",
- report_on: "P0,P1,P2,P3",
+ report_on: "P0,P1",
rubric: "",
});
assert.match(r.stderr, /trigger/, "field fallbacks must be noted on stderr");
diff --git a/scripts/summary-comment.mjs b/scripts/summary-comment.mjs
index 5c0e4ba..f8faff6 100644
--- a/scripts/summary-comment.mjs
+++ b/scripts/summary-comment.mjs
@@ -3,7 +3,7 @@
//
// node summary-comment.mjs --tier cheap|strong --push
// --gate pass|blocked [--prev ]
-// [--passes ] [--quiet] [--block-on P0,P1] [--held] [--fix-first P0,P1]
+// [--passes ] [--quiet] [--block-on P0,P1]
//
// Prints the summary MARKDOWN to stdout. The driver (action.yml) writes it into
// a marker-delimited region of the PR DESCRIPTION body (scripts/inject-summary.mjs),
@@ -41,9 +41,15 @@
// a cheap pass withholding the strong review over a fix-first finding — counted
// over the fix-first set instead, because block-on need not contain those
// severities and "❌ 0 findings block merge" beside a "held" tier line
-// contradicts itself. There is no held run and no tier line now, so the count
+// contradicts itself. The cascade went, and the tier line with it, so the count
// and the gate read the same set, which is the property that matters: the
// summary cannot claim something the merge gate does not enforce.
+//
+// --held and --fix-first went with it here, having been parsed and then read by
+// nothing for as long as the exception has been gone. fix-first REMAINS an action
+// input — it stops the exhaustive loop early, see action.yml — and must not be
+// wired back into this count on the strength of that name: the two answer
+// different questions, which is what the paragraph above is about.
import fs from "node:fs";
import { SEVERITIES, countSeverities } from "./severity.mjs";
@@ -54,14 +60,13 @@ const STATE_RE = //;
const usage = () => {
console.error(
"usage: node summary-comment.mjs --tier cheap|strong --push " +
- "--gate pass|blocked [--prev ] [--passes ] [--quiet] [--block-on P0,P1] " +
- "[--held] [--fix-first P0,P1]",
+ "--gate pass|blocked [--prev ] [--passes ] [--quiet] [--block-on P0,P1]",
);
process.exit(2);
};
const [file, ...rest] = process.argv.slice(2);
-const opts = { passes: "1", blockOn: "P0,P1", fixFirst: "P0,P1" };
+const opts = { passes: "1", blockOn: "P0,P1" };
for (let i = 0; i < rest.length; i += 1) {
if (rest[i] === "--tier") opts.tier = rest[++i];
else if (rest[i] === "--push") opts.push = rest[++i];
@@ -70,20 +75,21 @@ for (let i = 0; i < rest.length; i += 1) {
else if (rest[i] === "--passes") opts.passes = rest[++i];
else if (rest[i] === "--quiet") opts.quiet = true;
else if (rest[i] === "--block-on") opts.blockOn = rest[++i];
- else if (rest[i] === "--held") opts.held = true;
- else if (rest[i] === "--fix-first") opts.fixFirst = rest[++i];
+ // NO else: an unrecognised flag, and the value that follows it, are skipped
+ // one token at a time. That is what makes removing --fix-first safe to ship
+ // ahead of the caller that still passes it — the pair falls through without
+ // shifting the flags after it.
}
const push = Number(opts.push);
const passes = Number(opts.passes);
-// Severity-set normalization shared by --block-on and --fix-first: trim +
-// uppercase, empty = the empty set (valid — "block on nothing" / no fix-first).
+// Severity-set normalization for --block-on: trim + uppercase, empty = the
+// empty set (valid — a deliberate "block on nothing").
const parseSet = (v) =>
String(v ?? "")
.split(",")
.map((s) => s.trim().toUpperCase())
.filter(Boolean);
const blockOn = parseSet(opts.blockOn);
-const fixFirst = parseSet(opts.fixFirst);
if (
!file ||
!["pass", "blocked"].includes(opts.gate) ||
@@ -91,8 +97,7 @@ if (
push < 1 ||
!Number.isInteger(passes) ||
passes < 1 ||
- blockOn.some((s) => !SEVERITIES.includes(s)) ||
- fixFirst.some((s) => !SEVERITIES.includes(s))
+ blockOn.some((s) => !SEVERITIES.includes(s))
) {
usage();
}
@@ -151,7 +156,7 @@ lines.push("");
// Counted over block-on, which is what the gate enforces. The --held variant
// (count over fix-first instead) went with the cascade: it existed so a withheld
// escalation would not render "❌ 0 findings block merge" beside a held tier line,
-// and there is neither withholding nor a tier line any more.
+// and there is neither withholding nor a tier line any more. See the header.
const blockingSet = blockOn;
const blocking = blockingSet.reduce((n, s) => n + counts[s], 0);
lines.push(
diff --git a/scripts/summary-comment.test.mjs b/scripts/summary-comment.test.mjs
index 7cac1d6..305940e 100644
--- a/scripts/summary-comment.test.mjs
+++ b/scripts/summary-comment.test.mjs
@@ -173,12 +173,15 @@ describe("the ❌ count follows block-on", () => {
// the summary read "❌ 0 findings block merge" beside a held tier line. There is
// no withholding any more, so what survives is the case that was never about
// held runs at all.
- test("the count follows --block-on, not --fix-first", () => {
+ test("the count follows --block-on, and a removed flag cannot steer it", () => {
// --fix-first is deliberately a DIFFERENT set here: it used to steer this
- // count on a held run, and now it must not steer it at all.
+ // count on a held run, and now it must not steer it at all. It is no longer
+ // parsed, and it sits in the MIDDLE of the argv on purpose — the flag AND
+ // its value have to be skipped without shifting the --block-on that follows,
+ // which is what lets this script ship ahead of a caller still passing it.
const out = run(
["[P0] a", "[P2] b"],
- ["--push", "1", "--gate", "blocked", "--block-on", "P2", "--fix-first", "P0"],
+ ["--push", "1", "--fix-first", "P0", "--gate", "blocked", "--block-on", "P2"],
);
assert.ok(out.includes("❌ 1 finding blocks merge"), `got:
${out}`);
diff --git a/skills/setup-orca-code-review/SKILL.md b/skills/orca-review-action/SKILL.md
similarity index 99%
rename from skills/setup-orca-code-review/SKILL.md
rename to skills/orca-review-action/SKILL.md
index b764005..d02f671 100644
--- a/skills/setup-orca-code-review/SKILL.md
+++ b/skills/orca-review-action/SKILL.md
@@ -1,5 +1,5 @@
---
-name: setup-orca-code-review
+name: orca-review-action
description: Set up, reconfigure, troubleshoot, or remove OrcaCode Review — AI pull-request review powered by OrcaRouter — in a GitHub repository. Handles the whole lifecycle end to end without asking the user to run a CLI. Use whenever the user mentions OrcaCode Review, OrcaRouter code review, "@orcarouter code review", or /orcacode-review, and whenever they ask to set up AI code review on a repo, add the orca-code-review action, change which severities block merges, find out why a review did not run or did not post findings, or take the review workflow back out.
---
diff --git a/skills/setup-orca-code-review/assets/workflow.yml b/skills/orca-review-action/assets/workflow.yml
similarity index 96%
rename from skills/setup-orca-code-review/assets/workflow.yml
rename to skills/orca-review-action/assets/workflow.yml
index a020b0a..2483e51 100644
--- a/skills/setup-orca-code-review/assets/workflow.yml
+++ b/skills/orca-review-action/assets/workflow.yml
@@ -55,7 +55,7 @@ jobs:
# fix-first: "P0,P1" # exhaustive only: stop adding passes once found
# on-oversized-diff: "fail" # "pass" makes an oversized skip advisory
# auto-review-authors: "" # public repos: e.g. "OWNER,MEMBER,COLLABORATOR,CONTRIBUTOR"
- # max-diff-kb: "512"
- # max-diff-files: "300"
- # timeout-minutes: "20"
+ # max-diff-kb: "5000"
+ # max-diff-files: "2000"
+ # timeout-minutes: "60"
# report: "true" # severity counts only — never code
diff --git a/skills/setup-orca-code-review/references/inputs.md b/skills/orca-review-action/references/inputs.md
similarity index 85%
rename from skills/setup-orca-code-review/references/inputs.md
rename to skills/orca-review-action/references/inputs.md
index 7ab3d75..fb52b59 100644
--- a/skills/setup-orca-code-review/references/inputs.md
+++ b/skills/orca-review-action/references/inputs.md
@@ -3,6 +3,12 @@
Every input of `Continuum-AI-Corp/orca-code-review@v1`. Read the row before
answering a question about behavior — do not guess a default.
+**These rows describe this action.** The hosted OrcaCode Review GitHub App runs a
+different review engine, so a setting of the same name can behave differently
+there — `fix-first` and `exhaustive` in particular. For the App, the console at
+**OrcaRouter → Apps → OrcaCode Review** is the authority; do not answer a question
+about the App from this file.
+
## Required
| Input | Default | What it does |
@@ -44,8 +50,8 @@ allowlist and put a budget + alert on the key.
| Input | Default | What it does |
| --- | --- | --- |
-| `max-diff-kb` | `512` | Skip the review when the merge-base diff exceeds this size. |
-| `max-diff-files` | `300` | Skip when the diff touches more files than this. |
+| `max-diff-kb` | `5000` | Skip the review when the merge-base diff exceeds this size. Raised from 512 so large PRs are reviewed rather than refused; the real ceiling is the model context window, which this cannot see. |
+| `max-diff-files` | `2000` | Skip when the diff touches more files than this. Raised from 300. |
| `on-oversized-diff` | `fail` | What a skip does to the check. `fail` means a diff padded past the limits cannot bypass a required gate. `pass` restores advisory behavior. |
The guard runs **before** the engine, so an oversized PR costs nothing.
@@ -67,7 +73,7 @@ an error keeps the prior stage's findings and never aborts the review.
| --- | --- | --- |
| `concurrency` | `24` | Max concurrent file reviews. Raise to shorten wall clock; lower it under a tight per-minute request quota. |
| `max-tools` | `""` (engine default) | Max tool-call rounds per file. Lowering cuts cost and time but can cost review depth. |
-| `timeout-minutes` | `20` | Wall-clock ceiling for **one** engine pass. Exceeding it fails closed with a distinct "wall-clock timeout" error. Accepts decimals. Bump for very large diffs or slow models. |
+| `timeout-minutes` | `60` | Wall-clock ceiling for **one** engine pass — `exhaustive` makes up to three, so the worst case is 3x this. Exceeding it fails closed with a distinct "wall-clock timeout" error. Accepts decimals. Raised from 20 alongside the diff limits: a larger diff needs the time to actually be read. |
## Control plane
diff --git a/skills/setup-orca-code-review/references/troubleshooting.md b/skills/orca-review-action/references/troubleshooting.md
similarity index 98%
rename from skills/setup-orca-code-review/references/troubleshooting.md
rename to skills/orca-review-action/references/troubleshooting.md
index ae84778..e915dc8 100644
--- a/skills/setup-orca-code-review/references/troubleshooting.md
+++ b/skills/orca-review-action/references/troubleshooting.md
@@ -108,7 +108,7 @@ the console rather than in the file.
## The run times out
-`timeout-minutes` (default 20) is a ceiling on **one** engine pass. Exceeding it
+`timeout-minutes` (default 60) is a ceiling on **one** engine pass. Exceeding it
fails closed with a distinct "wall-clock timeout" error, separate from
"no usable result" — the log says which. Raise it for very large diffs or
slow-per-call models, or lower `concurrency` if the model is rate-limiting and the
diff --git a/skills/orca-review/SKILL.md b/skills/orca-review/SKILL.md
new file mode 100644
index 0000000..315c82a
--- /dev/null
+++ b/skills/orca-review/SKILL.md
@@ -0,0 +1,281 @@
+---
+name: orca-review
+description: Review code changes yourself, locally, with OrcaCode Review's severity contract and merge gate — no GitHub Action, no OrcaRouter account, no API key. You are the reviewer; the CLI supplies file selection, the P0-P3 rubric, position verification, and the gate. Use whenever the user asks you to review their changes, review a diff, branch, commit, or a pull request by number ("review PR 556"), check work before committing or pushing, run OrcaCode Review here, or asks "is this safe to merge?" — and whenever they want a local review that agrees with what CI will say.
+---
+
+# OrcaCode Review — local review
+
+You are the reviewer. Your own model does the thinking; the CLI does everything
+that must not be left to a model — which files are in scope, which rules apply,
+whether a finding is filed on the right line, and what blocks a merge.
+
+This is the same severity contract the GitHub Action enforces, so a P1 you find
+here is a P1 that would block there. That parity is the point: "it passed
+locally" has to mean something.
+
+**Severity contract:** `P0` critical / `P1` high → ❌ block. `P2` conditional bug
+/ `P3` nit → 💬 report, never block.
+
+## When this skill applies
+
+| The user says something like… | You do |
+| --- | --- |
+| "review my changes", "看一下我这次的改动", "check this before I push" | The [workflow](#workflow) below |
+| "review this branch / this commit / these staged files" | Same, with the matching [range flags](#choosing-the-range) |
+| "review PR 556", "帮我审一下 #556" | Same, with `--pr 556` — do **not** check the branch out first |
+| "is this safe to merge?" | Same — the gate answers it |
+| "set up OrcaCode Review on this repo", "why didn't CI review my PR?" | **Not this skill.** That is `orca-review-action`, which wires up the Action |
+
+Nothing here talks to OrcaRouter. If the user wants automatic review on every
+PR, that is the other skill.
+
+## Workflow
+
+### 1. Plan
+
+```bash
+npx @orcarouter/code-review review plan --lang en # en | zh | ja | ko
+```
+
+`--lang` is the language the user is speaking to you in — `en`, `zh`, `ja`, or
+`ko`. **Do not copy the example's value; read the conversation.** An English
+request gets `en`, a Chinese one `zh`. Pass it on every command. The plan will tell you to write your findings in
+that language, and `submit` will render its report in it. Without the flag the
+CLI falls back to the machine's locale, which is usually right and sometimes is
+not; you know which language the conversation is in, so say so.
+
+**Everything you say to the user is in that language too** — the one-line
+acknowledgement before you start, the report you relay, the next step you offer,
+the settings question. A Chinese request answered with "I'll review PR 92" and
+then a Chinese report reads as two different people.
+
+It prints a complete review request to stdout: the files in scope, the files it
+excluded and why, the git command that shows each change, per-language review
+checklists, the full P0-P3 rubric, the project's own conventions, and the exact
+result shape. Progress notes go to stderr, so the stdout text is the whole
+prompt and nothing else.
+
+Read all of it. Everything you need is in there — do not go looking for a rubric
+elsewhere, and do not substitute a severity scheme you know from somewhere else.
+
+The file list has already been filtered: binaries, deleted files, unsupported
+types, tests and fixtures and generated code, and anything under `.gitignore`
+are listed under **Excluded** with the reason. Do not review those, and do not
+second-guess the list — it is the same selection CI applies.
+
+### 2. Review
+
+Work file by file through the list in the request:
+
+1. Get the diff with the command the request gives you.
+2. **Read the actual file**, not just the diff. A finding you cannot confirm by
+ reading the surrounding code is a finding to drop.
+3. Apply that file's rule group, then the severity rubric.
+
+Comment only on changed lines. The rubric's precision section is binding, in
+particular: one comment per distinct issue, never the same root cause restated
+per file, and never a finding pinned to a file you did not open.
+
+### 3. Hand back
+
+Write `.orcacode-review/result.json` in exactly the shape the request specifies.
+**Four fields per finding, and nothing else** — the file is an internal handoff,
+not a report, and every extra byte is a line of raw JSON scrolling past the
+person watching you work:
+
+```json
+{"comments": [
+ {"path": "src/auth.ts", "line": 41,
+ "existing_code": " if (token === expected) {",
+ "content": "[P0] **Title**\n\nBody."}
+]}
+```
+
+- `path` — repo-relative, and it must be the file that actually contains the code.
+- `line` — in the post-change file. Only for a finding that genuinely spans
+ several lines, write `start_line`/`end_line` instead.
+- `existing_code` — the source you are quoting, **copied verbatim**. This gets
+ grepped against the tree to check you filed the finding on the right file. A
+ paraphrase here reads as a wrong location and the finding may be dropped. One
+ line is enough; do not paste a whole function.
+- `content` — `[P0]`/`[P1]`/`[P2]`/`[P3]`, then a bold title, then the body,
+ **written in the user's language**. The tag, the bold, and the separate
+ **Fix:** paragraph are structure and never change; the words inside them do.
+ Identifiers, paths, and code stay verbatim.
+
+Found nothing? Write `{"comments": []}`. That is a clean review, not a failure.
+
+`warnings` is optional and you will almost never want it: it lists files you
+could **not** review, and a non-empty `warnings` makes the review partial, which
+`submit` rejects rather than passes. Omit the key unless something genuinely
+failed.
+
+### 4. Submit
+
+```bash
+npx @orcarouter/code-review review submit --format md --lang en # same --lang as the plan
+```
+
+Same `--lang` as the plan, so the report's verdict line and headings come out in
+the user's language alongside the findings you wrote in it.
+
+This verifies every finding's position against the tree, drops duplicates,
+tags what would block a merge, and prints the report as markdown.
+
+Use `--format md` rather than the default. The default is an ANSI terminal
+report hard-wrapped at 78 columns, which arrives in a conversation as a wall of
+pre-formatted text.
+
+**Relay that markdown to the user verbatim.** It is written to be read in a
+chat: grouped by file, blocking file first, ❌ on what stops the merge and 💬
+on what does not. Do not re-summarise it, do not re-order it, and do not
+substitute your own severity call — the gate decides what blocks, not you, and
+it will have re-homed or dropped findings you were about to report.
+
+**The exit code is not the verdict.** `submit` exits `0` whenever the review
+ran, blocked or not — your job is to tell the user what the bugs are, and that
+is the report. The line under its heading says which it was: ❌ (`Blocked`,
+`已拦截`, …) means something at P0/P1 would stop a merge, ✅ means nothing
+would. Relay whichever you got.
+
+| Exit | Means | You say |
+| --- | --- | --- |
+| `0` | The review ran. The report has the findings and the verdict | Relay the report as-is |
+| `2` | The result was unusable | Fix the JSON and submit again. **Never** a pass |
+
+Exit `2` is the only failure, and it is never a pass. (A `1` only exists under
+`--fail-on-block`, which is for hooks and CI scripts — you have no reason to
+pass it.)
+
+The same markdown is always saved to `.orcacode-review/report.md`.
+
+Then say, in one line, what you would do next. Offer to fix; do not start
+fixing unless the user asked for it in the first place.
+
+## The first review in a repository
+
+If the plan ends with a section titled **First review in this repository**,
+there is no `.orcacode-review.json` yet. Do the review exactly as normal. Then,
+after the report is on screen, offer once — one sentence — to save the settings
+this run used, and say what saving buys them: the next review here needs no
+flags, for them or for anyone else who clones the repo. Something like:
+
+> 这个仓库还没有本地评审配置。要我把这次的设置(中文、P0/P1 阻塞)存成
+> `.orcacode-review.json` 吗?以后在这里评审就不用再指定了。
+
+Yes → run the `review config init` command the plan gives you, apply anything
+they asked for on top, show them the file. No → drop it for the rest of the
+conversation. Never create the file without being asked, and never ask before
+the report: they should decide with the output in front of them, not in the
+abstract.
+
+## Changing how the local review behaves
+
+The repo's local review settings live in one committed file,
+`.orcacode-review.json`. When the user asks for a lasting change — not "this
+once" but "from now on" — **edit that file**; do not reach for a flag, and do
+not put it in `CLAUDE.md`.
+
+| The user says something like… | You change |
+| --- | --- |
+| "本地只挡 P0", "only block on critical locally" | `"block_on": "P0"` |
+| "别挡了,都只提醒", "never block, just report" | `"block_on": ""` |
+| "以后用中文报", "report in Japanese from now on" | `"language": "zh"` / `"ja"` |
+| "不要审 docs/", "skip generated files" | append to `"exclude"`: `"docs/**"`, `"**/*.generated.ts"` |
+| "API 文件多查一下鉴权", "add a checklist for migrations" | append to `"rules"`: `{ "path": "src/api/**/*.ts", "rule": "…" }` |
+
+Before editing, run this to see what applies now and where each value came from:
+
+```bash
+npx @orcarouter/code-review review config --lang en # the user's language
+```
+
+If the file does not exist yet, create it with the template rather than by hand:
+
+```bash
+npx @orcarouter/code-review review config init --lang en # the user's language
+```
+
+Then edit only the key the user asked about. Show them the resulting file. Keys
+you do not write keep their defaults; unknown keys are an error, not a typo the
+tool overlooks — if `plan` or `submit` refuses with "`.orcacode-review.json` is
+invalid", read the message, it names the key and the allowed values.
+
+The four keys, and nothing else:
+
+- `block_on` — `"P0,P1"` (default), `"P0"`, or `""`. What the report marks ❌.
+- `language` — `en` | `zh` | `ja` | `ko`. Outranks the machine locale; `--lang`
+ still outranks it.
+- `exclude` — extra globs never to review, on top of the bundled rules.
+ `docs/**`, `**/*.snap`, `legacy/**`.
+- `rules` — extra checklists. Each `{ "path": glob, "rule": "text" }` or
+ `{ "path": glob, "rule_file": "docs/review/api.md" }`. By default the text is
+ **added** to the bundled checklist for those files; `"replace": true` swaps it
+ in instead.
+
+For a one-off ("just this time only block on P0") use the flag on that one
+command instead, and change nothing on disk.
+
+## Choosing the range
+
+With no flags it picks for you and says which it picked: uncommitted work if the
+tree is dirty, otherwise this branch against its base — the range CI would use.
+Override when the user was specific:
+
+| The user means | Flags |
+| --- | --- |
+| "what I'm working on right now" | `--worktree` |
+| "this branch" / "the PR I'm on" | `--from main --to HEAD` |
+| "PR 556" — a number, not the current branch | `--pr 556` |
+| "that commit" | `--commit ` |
+| "only block on critical" — this once | `--block-on P0` on `submit` |
+| "only block on critical" — from now on | `"block_on": "P0"` in `.orcacode-review.json` (see [above](#changing-how-the-local-review-behaves)) |
+
+Pass `--background "…"` when the user has told you what the change is *for*.
+It goes to the reviewer — you — as business context, and it is what lets you
+judge whether the change does what it was supposed to. With `--pr` the PR's
+title and description become the background automatically; only pass
+`--background` on top of it if the user told you something the PR does not say.
+
+### Reviewing a pull request by number
+
+`--pr` needs the GitHub CLI (`gh`) and **does not check the branch out**. It
+fetches the PR into a private ref and leaves the work tree exactly as it was —
+so it is safe to run mid-change, and you must not `git checkout` or `git stash`
+around it. Fork PRs work; the head is fetched from the base repository.
+
+If it reports that `gh` is missing or not signed in, do not try to install or
+authenticate it. Say so, and offer the fallback the CLI prints:
+
+```bash
+gh pr checkout 556 # the user runs this, then you review with no --pr
+```
+
+## Getting it wrong
+
+- **Do not invent severities.** High/Medium/Low is a different tool's
+ vocabulary. Untagged findings default to P1, which blocks — so tag everything.
+- **Do not drop P2 and P3 to look decisive.** The rubric is explicit that they
+ are still emitted. They do not block; hiding them is not calibration.
+- **Do not skip `submit` because you already know what you found.** Position
+ verification and deduplication happen there, and they change the answer.
+- **Do not report a clean review when `submit` exited 2.** That is an unusable
+ result, not a pass. Fix the JSON and submit again.
+- **Do not switch branches to review a PR.** `--pr` exists so you never have to.
+ Checking out loses the user's uncommitted work or fails outright, and leaves
+ them somewhere they did not ask to be.
+
+## If something fails
+
+| Symptom | Cause |
+| --- | --- |
+| "Not a git repository" | Run from inside the repo |
+| "Nothing to review" | Clean tree with no commits past the base — use `--worktree` or name a range |
+| "Unusable result" | The JSON is malformed, or `warnings` is non-empty. Read the message, fix, resubmit |
+| "Position check skipped" | Expected on uncommitted work — nothing to grep. Findings are all kept |
+| "No result at …" | You did not write the file, or wrote it somewhere else |
+| "`--pr` needs the GitHub CLI" | No `gh`. Tell the user; offer `gh pr checkout ` as the fallback |
+| "gh is installed but not signed in" | Theirs to fix: `gh auth login`. Do not run it for them |
+
+`references/contract.md` has the full command reference, the JSON schema, and
+the exit codes — read it before guessing at a flag.
diff --git a/skills/orca-review/references/contract.md b/skills/orca-review/references/contract.md
new file mode 100644
index 0000000..695d75c
--- /dev/null
+++ b/skills/orca-review/references/contract.md
@@ -0,0 +1,321 @@
+# Harness contract
+
+The machine-readable surface behind the local review. Two commands, split at the
+point where judgement enters: `plan` is everything before the model, `submit` is
+everything after it.
+
+Nothing here contacts OrcaRouter, and no API key exists in this flow.
+
+## `review plan`
+
+```
+npx @orcarouter/code-review review plan [range] [--background ] [--lang en|zh|ja|ko] [--json]
+```
+
+Writes the review request to **stdout** and progress notes to **stderr**, so
+`review plan > request.md` captures the prompt clean. Also writes both
+`.orcacode-review/request.md` and `.orcacode-review/plan.json`, and adds
+`/.orcacode-review/` to `.git/info/exclude` so reviewing never dirties the repo.
+
+### Range flags
+
+| Flag | Meaning |
+| --- | --- |
+| *(none)* | Auto: uncommitted work if the tree is dirty, else this branch vs its base |
+| `--worktree` | Uncommitted work (staged + unstaged + untracked) |
+| `--from [ --to ][` | A range. `--to` defaults to `HEAD` |
+| `--commit `, `-c` | One commit |
+| `--pr ` | A GitHub pull request. Mutually exclusive with the four above |
+| `--background `, `-b` | Business context passed to the reviewer |
+
+Named to match Open Code Review's `ocr delegate` flags, so anyone who knows one
+already knows the other.
+
+### `--pr`
+
+The only flag that talks to a forge, and the only one with a dependency:
+`gh`, resolved at call time. Without it the command fails with `no-gh` and
+tells the caller to `gh pr checkout` instead. Nothing else in the harness knows
+what a pull request is.
+
+It does **not** check the branch out. Two fetches put the PR into a private ref
+namespace and the range is built from those:
+
+```
+refs/orcacode/pr//head <- +refs/pull//head (exists for fork PRs too)
+refs/orcacode/pr//base <- +refs/heads/
+```
+
+The work tree is never touched, so `--pr` is safe to run mid-change. Re-running
+moves the same two refs; nothing accumulates and no branch is created. If the
+base branch has been deleted (a merged PR), it falls back to
+`/` and then the bare name, and fails with `no-base-ref`
+if neither resolves.
+
+The remote is whichever one points at the repository `gh` resolved — not
+blindly `origin` — so a fork checkout fetches from upstream.
+
+The PR title and body become `background` unless `--background` was given
+explicitly, capped at 4000 characters. `plan.pr` carries the metadata:
+
+```json
+{ "number": 556, "title": "…", "url": "…", "state": "OPEN",
+ "base": "main", "head": "feat/x", "fork": false }
+```
+
+It is `null` for every other mode. `range.code` is `"pr"`.
+
+### `--json`
+
+Emits the plan as structured data instead of the prompt. `schema_version` is
+`"1"`; it changes only when a field changes incompatibly.
+
+```json
+{
+ "schema_version": "1",
+ "mode": "range | commit | workspace",
+ "repository": "/abs/path",
+ "from": "main", "to": "HEAD", "commit": "", "merge_base": "",
+ "pr": null,
+ "background": "",
+ "language": "en",
+ "config": null,
+ "selector": "builtin",
+ "files": [{ "path": "src/a.ts", "status": "M", "insertions": 12, "deletions": 3 }],
+ "excluded": [{ "path": "pnpm-lock.yaml", "reason": "lockfile" }],
+ "rule_groups": [{ "group_id": 1, "source": "…", "pattern": "*.ts", "files": ["src/a.ts"], "rule": "…" }],
+ "diff_recipe": "git diff ..HEAD -- ",
+ "rubric": { "severity": "…", "output_shape": "…", "conventions": { "file": "AGENTS.md", "text": "…" } },
+ "result_path": "/abs/path/.orcacode-review/result.json"
+}
+```
+
+### `language`
+
+The language the reviewer is told to write findings in: `--lang` if given, else
+the CLI's locale detection, else `en`. The plan carries a `## Language` section
+naming it. Only the findings' prose and the report's own strings follow it; the
+severity rubric, the output-shape rule, and the tags are English tokens shared
+with the Action and do not translate.
+
+### `selector`
+
+Always `builtin`. Changed files come from git; which of them are reviewable,
+and which checklist applies to each, comes from Open Code Review's rule corpus
+vendored under `vendor/open-code-review/` and a JavaScript port of its matcher
+(`bin/selection.mjs`). Nothing is installed and nothing is spawned. The plan's
+`excluded` list carries a reason per file:
+
+| `code` | Meaning |
+| --- | --- |
+| `ignored` | Under an always-skipped directory (`node_modules/`, `vendor/`, `.git/`, …) or matched by the root `.gitignore` |
+| `binary` | Git reports no line counts |
+| `unsupported_ext` | Extension not in the reviewable allowlist. Extensionless files pass |
+| `default_path` | Matches a default exclude glob: tests, fixtures, snapshots, generated code |
+| `project_exclude` | Matches an `exclude` glob in `.orcacode-review.json` |
+| `deleted` | Nothing left to review |
+
+`rule_groups` is the same corpus's per-language checklists, one group per
+distinct (pattern, checklist) pair, in the shape `ocr delegate rule` emits.
+
+Not ported: Open Code Review's own project/global rule layers
+(`.opencodereview/rule.json`). A repo that uses those has `ocr` set up already;
+this path is for repos that do not.
+
+## `review config`
+
+```
+npx @orcarouter/code-review review config [--json]
+npx @orcarouter/code-review review config init [--force]
+```
+
+Shows the settings that will apply to the next `plan`/`submit` and the source
+of each — `flag`, `file`, `default`, `locale` — or writes the template. `--json`
+emits `{ file, block_on: {value, source}, language: {value, source}, exclude,
+rules }`.
+
+### `.orcacode-review.json`
+
+The repository's local review settings. Committed. Read by `plan`, `submit`,
+and `config`; an invalid file makes all three exit `2` with the offending key
+named, because a setting that was silently ignored would let the file say one
+thing and the review do another.
+
+```json
+{
+ "$comment": "optional note; ignored",
+ "block_on": "P0,P1",
+ "language": "zh",
+ "exclude": ["docs/**", "**/*.generated.ts"],
+ "rules": [
+ { "path": "src/api/**/*.ts", "rule": "Every handler checks the caller's tenant before touching a row." },
+ { "path": "migrations/**/*.sql", "rule_file": "docs/review/migrations.md", "replace": false }
+ ]
+}
+```
+
+| Key | Type | Effect | Precedence |
+| --- | --- | --- | --- |
+| `block_on` | `"P0,P1"` string or array; `""` = none | Which findings the report marks ❌ | `--block-on` > file > `P0,P1` |
+| `language` | `en` `zh` `ja` `ko` | Findings and report language | `--lang` > file > locale > `en` |
+| `exclude` | array of globs (doublestar syntax, case-insensitive) | Files never reviewed; listed under Excluded with reason `project_exclude` | Adds to the bundled excludes |
+| `rules` | array of `{ path, rule \| rule_file, replace? }` | Extra checklist for files matching `path`; first match wins | Added to the bundled checklist unless `replace: true` |
+
+`rule_file` is repo-relative and must resolve inside the repository. Unknown
+top-level keys, unknown keys inside a rule, a `rule` and `rule_file` on the same
+entry, or an unreadable `rule_file` are all errors. `plan --json` reports the
+loaded file under `config` (or `null`).
+
+When the file is absent, the plan text ends with a **First review in this
+repository** section instructing the agent to offer, once and after the report,
+to create it via `review config init`. `init` pre-fills `language` from the
+run's language and `block_on` from `--block-on` if given. Nothing is ever
+written unasked.
+
+Not the same thing as `.orcacode-review/` — that directory is per-run scratch
+(`plan.json`, `result.json`, `report.md`) and is kept out of git.
+
+## `review submit`
+
+```
+npx @orcarouter/code-review review submit [file] [--block-on P0,P1]
+ [--format text|md|json] [--fail-on-block]
+ [--no-postfilter]
+```
+
+`file` defaults to `.orcacode-review/result.json`.
+
+### `--format`
+
+| Value | Output on stdout | For |
+| --- | --- | --- |
+| `text` *(default)* | ANSI report, hard-wrapped at 78 columns | A human at a terminal |
+| `md` | Markdown report | An agent relaying it into a conversation |
+| `json` | The structured result below | A scripted harness |
+
+`--json` is an alias for `--format json`; `--md`/`--markdown` for `--format md`.
+An unknown value is an error, not a fallback to `text` — a caller expecting
+machine output must never silently receive ANSI.
+
+The markdown report is **always** written to `.orcacode-review/report.md`,
+whatever format was requested, so it can be read back or attached to a ticket.
+It groups findings by file, sorts the file holding the worst finding first, and
+marks each finding ❌ or 💬 according to the gate that actually ran — under
+`--block-on P0` a P1 is marked 💬, because that run does not block on it.
+
+Pipeline, in order:
+
+1. **Shape check** — `comments` must be an array; every finding needs `path` and
+ `content`. A bare `line` is widened to `start_line`/`end_line` here. A
+ non-empty `warnings` means a partial review and is rejected. This mirrors the
+ Action's `check-result.mjs`: a partial review must never become a clean pass.
+2. **Position check** — each finding's `existing_code` is grepped against the
+ reviewed commit. A snippet that matches a *different* file re-homes the
+ finding to that file; one that matches nowhere and is pinned to a non-code
+ file is dropped; then duplicates by normalized content are collapsed. This is
+ the Action's `postfilter.mjs`, run unmodified.
+ Skipped in workspace mode — uncommitted code is in no commit to grep — and
+ fail-soft everywhere else: an error keeps every finding rather than losing any.
+3. **Gate** — findings are tagged by leading `[P0]`…`[P3]`; an untagged finding
+ counts as **P1** (fail-safe, so a missing tag escalates rather than passing).
+ `--block-on` defaults to `P0,P1`; `--block-on ""` never blocks.
+
+### Exit codes
+
+| Code | Meaning |
+| --- | --- |
+| `0` | Reviewed. The report carries the verdict, blocked or not |
+| `1` | Reviewed and blocked — **only under `--fail-on-block`** |
+| `2` | No usable result — unparseable, malformed, or partial. Never a pass |
+
+The default is `0` for a review that ran, whatever it found. The local harness
+exists to tell an agent what the bugs are, and an agent reads the report — its
+shell tool labels any non-zero exit `Error`, so a verdict carried in the status
+arrives looking like a crashed command. The verdict is in the report instead:
+`**❌ Blocked**` / `**✅ Passed**` in markdown, `❌ BLOCKED` / `✅ PASSED` in the
+terminal report, `"blocked": true|false` under `--format json`.
+
+### `--fail-on-block`
+
+Makes a blocked review also exit `1`, for a caller that consumes the status
+rather than the report: a pre-push hook, a CI step, a `review submit && git
+push`. Nothing else changes.
+
+`2` is unaffected by either setting. An unusable result must never come back as
+`0` — that would turn a review that did not happen into a review that passed,
+which is the one failure mode this whole pipeline exists to prevent.
+
+### `--json`
+
+```json
+{
+ "comments": [ … ],
+ "counts": { "P0": 0, "P1": 2, "P2": 1, "P3": 4 },
+ "block_on": ["P0", "P1"],
+ "blocked": true
+}
+```
+
+`comments` here is the post-verification set — re-homed and deduplicated — which
+is why it can differ from what was submitted.
+
+## Result shape
+
+What an agent writes — the short form, four fields:
+
+```json
+{
+ "comments": [
+ {
+ "path": "src/auth.ts",
+ "line": 41,
+ "existing_code": " if (token === expected) {",
+ "content": "[P0] **Token comparison is not constant-time**\n\n`===` on a secret leaks its prefix through timing. Use `crypto.timingSafeEqual`."
+ }
+ ]
+}
+```
+
+What the pipeline speaks internally — the wide form, which is byte-for-byte what
+the GitHub Action's engine emits, and what lets the same gate, the same filters,
+and the same severity tally run over both:
+
+```json
+{ "path": "src/auth.ts", "start_line": 41, "end_line": 41, "…": "…" }
+```
+
+`validateResult` widens the short form into the wide one at the boundary, and
+`submit` normalises the whole result to a scratch file before handing it to
+`postfilter.mjs` — that script reads a *file*, not the parsed object, so it must
+be given the wide form or every finding comes back with no anchor.
+
+- `line` is the short form and is what you should write. `start_line`/`end_line`
+ is for a finding that genuinely spans lines. Do not write both.
+- The Action's poster collapses the pair to a single line as
+ `end_line || start_line`, which is the same rule `anchorLine()` uses here.
+- `warnings` is optional. Omit it; a non-empty one means a partial review and is
+ rejected. An empty one is accepted and means the same as omitting it.
+- The position check can **clear** the anchor when it re-homes a finding whose
+ line it cannot resolve in the new file. That is deliberate — a stale line from
+ the wrong file would post the comment on unrelated code.
+- `existing_code` is the verification key. Copy it verbatim from the file; a
+ paraphrase reads as a wrong location.
+- `content` must open with the severity tag, then a bold title of about ten words
+ naming what is **wrong** (not what to do, no full stop), then a blank line,
+ then the body. The renderer splits on that bold.
+
+## Relationship to the Action
+
+| | Local (`review plan`/`submit`) | GitHub Action |
+| --- | --- | --- |
+| Who thinks | Your coding agent's model | The engine, via OrcaRouter |
+| Credentials | None | `ORCAROUTER_API_KEY` |
+| File selection | Vendored Open Code Review rules, ported matcher | The engine, same rules |
+| Severity rubric | `rules/severity-instruction.md` | The same file |
+| Position check | `postfilter.mjs` | The same script |
+| L2 LLM judge | Not run — it needs a second, independent model | Optional |
+| Result shape | Identical | Identical |
+| Findings posted | Terminal | Inline PR comments |
+
+The differences are about who pays for the thinking and where the output lands.
+The severity judgement is the same on both sides on purpose.
diff --git a/vendor/open-code-review/LICENSE b/vendor/open-code-review/LICENSE
new file mode 100644
index 0000000..5db0382
--- /dev/null
+++ b/vendor/open-code-review/LICENSE
@@ -0,0 +1,201 @@
+ Apache License
+ Version 2.0, January 2004
+ http://www.apache.org/licenses/
+
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
+
+ 1. Definitions.
+
+ "License" shall mean the terms and conditions for use, reproduction,
+ and distribution as defined by Sections 1 through 9 of this document.
+
+ "Licensor" shall mean the copyright owner or entity authorized by
+ the copyright owner that is granting the License.
+
+ "Legal Entity" shall mean the union of the acting entity and all
+ other entities that control, are controlled by, or are under common
+ control with that entity. For the purposes of this definition,
+ "control" means (i) the power, direct or indirect, to cause the
+ direction or management of such entity, whether by contract or
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
+ outstanding shares, or (iii) beneficial ownership of such entity.
+
+ "You" (or "Your") shall mean an individual or Legal Entity
+ exercising permissions granted by this License.
+
+ "Source" form shall mean the preferred form for making modifications,
+ including but not limited to software source code, documentation
+ source, and configuration files.
+
+ "Object" form shall mean any form resulting from mechanical
+ transformation or translation of a Source form, including but
+ not limited to compiled object code, generated documentation,
+ and conversions to other media types.
+
+ "Work" shall mean the work of authorship, whether in Source or
+ Object form, made available under the License, as indicated by a
+ copyright notice that is included in or attached to the work
+ (an example is provided in the Appendix below).
+
+ "Derivative Works" shall mean any work, whether in Source or Object
+ form, that is based on (or derived from) the Work and for which the
+ editorial revisions, annotations, elaborations, or other modifications
+ represent, as a whole, an original work of authorship. For the purposes
+ of this definition, Derivative Works shall not include works that remain
+ separable from, or merely link (or bind by name) to the interfaces of,
+ the Work and Derivative Works thereof.
+
+ "Contribution" shall mean any work of authorship, including
+ the original version of the Work and any modifications or additions
+ to that Work or Derivative Works thereof, that is intentionally
+ submitted to the Licensor for inclusion in the Work by the copyright owner
+ or by an individual or Legal Entity authorized to submit on behalf of
+ the copyright owner. For the purposes of this definition, "submitted"
+ means any form of electronic, verbal, or written communication sent
+ to the Licensor or its representatives, including but not limited to
+ communication on electronic mailing lists, source code control systems,
+ and issue tracking systems that are managed by, or on behalf of, the
+ Licensor for the purpose of discussing and improving the Work, but
+ excluding communication that is conspicuously marked or otherwise
+ designated in writing by the copyright owner as "Not a Contribution."
+
+ "Contributor" shall mean Licensor and any individual or Legal Entity
+ on behalf of whom a Contribution has been received by Licensor and
+ subsequently incorporated within the Work.
+
+ 2. Grant of Copyright License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ copyright license to reproduce, prepare Derivative Works of,
+ publicly display, publicly perform, sublicense, and distribute the
+ Work and such Derivative Works in Source or Object form.
+
+ 3. Grant of Patent License. Subject to the terms and conditions of
+ this License, each Contributor hereby grants to You a perpetual,
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
+ (except as stated in this section) patent license to make, have made,
+ use, offer to sell, sell, import, and otherwise transfer the Work,
+ where such license applies only to those patent claims licensable
+ by such Contributor that are necessarily infringed by their
+ Contribution(s) alone or by combination of their Contribution(s)
+ with the Work to which such Contribution(s) was submitted. If You
+ institute patent litigation against any entity (including a
+ cross-claim or counterclaim in a complaint) alleging that the Work
+ or a Contribution incorporated within the Work constitutes direct
+ or contributory patent infringement, then any patent licenses
+ granted to You under this License for that Work shall terminate
+ as of the date such litigation is filed.
+
+ 4. Redistribution. You may reproduce and distribute copies of the
+ Work or Derivative Works thereof in any medium, with or without
+ modifications, and in Source or Object form, provided that You
+ meet the following conditions:
+
+ (a) You must give any other recipients of the Work or
+ Derivative Works a copy of this License; and
+
+ (b) You must cause any modified files to carry prominent notices
+ stating that You changed the files; and
+
+ (c) You must retain, in the Source form of any Derivative Works
+ that You distribute, all copyright, patent, trademark, and
+ attribution notices from the Source form of the Work,
+ excluding those notices that do not pertain to any part of
+ the Derivative Works; and
+
+ (d) If the Work includes a "NOTICE" text file as part of its
+ distribution, then any Derivative Works that You distribute must
+ include a readable copy of the attribution notices contained
+ within such NOTICE file, excluding those notices that do not
+ pertain to any part of the Derivative Works, in at least one
+ of the following places: within a NOTICE text file distributed
+ as part of the Derivative Works; within the Source form or
+ documentation, if provided along with the Derivative Works; or,
+ within a display generated by the Derivative Works, if and
+ wherever such third-party notices normally appear. The contents
+ of the NOTICE file are for informational purposes only and
+ do not modify the License. You may add Your own attribution
+ notices within Derivative Works that you distribute, alongside
+ or as an addendum to the NOTICE text from the Work, provided
+ that such additional attribution notices cannot be construed
+ as modifying the License.
+
+ You may add Your own copyright statement to Your modifications and
+ may provide additional or different license terms and conditions
+ for use, reproduction, or distribution of Your modifications, or
+ for any such Derivative Works as a whole, provided Your use,
+ reproduction, and distribution of the Work otherwise complies with
+ the conditions stated in this License.
+
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
+ any Contribution intentionally submitted for inclusion in the Work
+ by You to the Licensor shall be under the terms and conditions of
+ this License, without any additional terms or conditions.
+ Notwithstanding the above, nothing herein shall supersede or modify
+ the terms of any separate license agreement you may have executed
+ regarding such Contributions.
+
+ 6. Trademarks. This License does not grant permission to use the trade
+ names, trademarks, service marks, or product names of the Licensor,
+ except as required for reasonable and customary use in describing the
+ origin of the Work and reproducing the content of the NOTICE file.
+
+ 7. Disclaimer of Warranty. Unless required by applicable law or
+ agreed to in writing, Licensor provides the Work (and each
+ Contributor provides its Contributions) on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
+ implied, including, without limitation, any warranties or conditions
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
+ PARTICULAR PURPOSE. You are solely responsible for determining the
+ appropriateness of using or redistributing the Work and assume any
+ risks associated with Your exercise of permissions under this License.
+
+ 8. Limitation of Liability. In no event and under no legal theory,
+ whether in tort (including negligence), contract, or otherwise,
+ unless required by applicable law (such as deliberate and grossly
+ negligent acts) or agreed to in writing, shall any Contributor be
+ liable to You for damages, including any direct, indirect, special,
+ incidental, or consequential damages of any character arising as a
+ result of this License or out of the use or inability to use the
+ Work (including but not limited to damages for loss of goodwill,
+ work stoppage, computer failure or malfunction, or any and all
+ other commercial damages or losses), even if such Contributor
+ has been advised of the possibility of such damages.
+
+ 9. Accepting Warranty or Additional Liability. While redistributing
+ the Work or Derivative Works thereof, You may choose to offer,
+ and charge a fee for, acceptance of support, warranty, indemnity,
+ or other liability obligations and/or rights consistent with this
+ License. However, in accepting such obligations, You may act on
+ Your own behalf and on Your sole responsibility, not on behalf
+ of any other Contributor, and only if You agree to indemnify,
+ defend, and hold each Contributor harmless for any liability
+ incurred by, or claims asserted against, such Contributor by reason
+ of your accepting any such warranty or additional liability.
+
+ END OF TERMS AND CONDITIONS
+
+ APPENDIX: How to apply the Apache License to your work.
+
+ To apply the Apache License to your work, attach the following
+ boilerplate notice, with the fields enclosed by brackets "{}"
+ replaced with your own identifying information. (Don't include
+ the brackets!) The text should be enclosed in the appropriate
+ comment syntax for the file format. We also recommend that a
+ file or class name and description of purpose be included on the same
+ "printed page" as the copyright notice for easier identification within
+ third-party archives.
+
+ Copyright 2026 alibaba/open-code-review Contributors
+
+ Licensed under the Apache License, Version 2.0 (the "License");
+ you may not use this file except in compliance with the License.
+ You may obtain a copy of the License at
+
+ http://www.apache.org/licenses/LICENSE-2.0
+
+ Unless required by applicable law or agreed to in writing, software
+ distributed under the License is distributed on an "AS IS" BASIS,
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ See the License for the specific language governing permissions and
+ limitations under the License.
diff --git a/vendor/open-code-review/UPSTREAM b/vendor/open-code-review/UPSTREAM
new file mode 100644
index 0000000..d95334d
--- /dev/null
+++ b/vendor/open-code-review/UPSTREAM
@@ -0,0 +1,24 @@
+Vendored from alibaba/open-code-review, unmodified.
+
+ repository https://github.com/alibaba/open-code-review
+ tag v1.11.2
+ commit 96306531fae4d29fc3cc84895493dc28dd8a7773
+ license Apache-2.0 (LICENSE in this directory)
+ taken 2026-09-02
+
+Files, and where they live upstream:
+
+ system_rules.json internal/config/rules/system_rules.json
+ rule_docs/*.md internal/config/rules/rule_docs/
+ supported_file_types.json internal/config/allowlist/supported_file_types.json
+ default_exclude_patterns.json internal/config/allowlist/default_exclude_patterns.json
+
+These are DATA: the per-language review checklists, the path -> checklist map,
+the reviewable-extension allowlist, and the default exclude globs. The logic
+that reads them is a port, not a copy — see bin/selection.mjs, which names the
+Go files it was derived from.
+
+Do not edit the files in this directory. To update, re-copy from a newer tag
+and change the tag/commit above. Apache-2.0 requires that a modified file be
+marked as modified; keeping them byte-identical to upstream is what lets this
+notice stay true.
diff --git a/vendor/open-code-review/default_exclude_patterns.json b/vendor/open-code-review/default_exclude_patterns.json
new file mode 100644
index 0000000..98961e5
--- /dev/null
+++ b/vendor/open-code-review/default_exclude_patterns.json
@@ -0,0 +1,56 @@
+[
+ "**/*_test.go",
+ "**/src/test/java/**/*.java",
+ "**/src/test/**/*.kt",
+ "**/*.test.{js,jsx,ts,tsx}",
+ "**/*.spec.{js,jsx,ts,tsx}",
+ "**/__tests__/**",
+ "**/test/**/*_test.py",
+ "**/tests/**/*_test.py",
+ "**/*_test.py",
+ "**/*_spec.rb",
+ "**/spec/**/*_spec.rb",
+ "**/*Test.java",
+ "**/*Tests.java",
+ "**/*_test.rs",
+ "**/oh_modules/**",
+ "**/*.test.ets",
+ "**/test/**/*.jl",
+ "**/test/**/*.hs",
+ "**/*Spec.hs",
+ "**/test/**/*.lhs",
+ "**/*Spec.lhs",
+ "**/tests/**/*.nim",
+ "**/tests/**/*.R",
+ "**/__snapshots__/**",
+ "**/*.snap",
+ "**/testdata/**",
+ "**/fixtures/**",
+ "**/.ipynb_checkpoints/**",
+ "**/*.generated.*",
+ "**/*.gen.go",
+ "**/*.pb.go",
+ "**/*.pb.cc",
+ "**/*.pb.h",
+ "**/*Test.swift",
+ "**/*Tests.swift",
+ "**/Tests/**/*.swift",
+ "**/tests/**/*.elm",
+ "**/vendor/**/*.{jsonnet,libsonnet}",
+ "**/test/**/*.zig",
+ "**/*_test.zig",
+ "**/kitex_gen/**/*.go",
+ "**/*.capnp.h",
+ "**/*.capnp.go",
+ "**/*.capnp.ts",
+ "**/*_capnp.rs",
+ "**/*_capnp.py",
+ "**/tb_*.{v,sv,vhd,vhdl}",
+ "**/*_tb.{v,sv,vhd,vhdl}",
+ "lib/**/*.sol",
+ "**/*.t.sol",
+ "**/test/**/*.sol",
+ "**/tests/**/*.sol",
+ "**/test/**/*.vy",
+ "**/tests/**/*.vy"
+]
diff --git a/vendor/open-code-review/rule_docs/arkts.md b/vendor/open-code-review/rule_docs/arkts.md
new file mode 100644
index 0000000..563b58f
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/arkts.md
@@ -0,0 +1,55 @@
+#### Obvious Typos or Spelling Errors
+- Spelling errors in component names, variable names, or function names
+- Spelling errors in log or error messages that affect readability
+
+#### Dead Code
+- Code blocks that will never be executed (e.g., branches where the condition is always false, code after a return statement)
+- Variables that are declared but never read or referenced
+- Large blocks of commented-out code (with no apparent intent to retain)
+
+#### State Decorator Usage
+- Arrays/objects decorated with `@State` will not trigger UI refresh when modified via push/property changes; references must be replaced
+- Verify correct usage of `@Prop` (one-way) vs `@Link` (two-way) for the given scenario
+- Nested object state updates must use `@Observed` + `@ObjectLink`
+- Props drilling beyond 3 levels should use `@Provide/@Consume` instead
+- `@StorageLink/@StorageProp` should only be used for truly global state; avoid overuse
+
+#### Component Lifecycle
+- Timers and listeners created in `aboutToAppear` must be released in `aboutToDisappear`
+- Page-level logic should be placed in `onPageShow/onPageHide` rather than component lifecycle hooks
+- Avoid executing time-consuming synchronous operations in lifecycle hooks that block the UI thread
+
+#### ArkUI Declarative Syntax
+- Side effects (network requests, timers, logging) are prohibited in the `build` method
+- `ForEach` / `LazyForEach` must provide a unique and stable key generator function
+- Use `if/else` for conditional rendering, not `switch`
+- Direct manipulation of component instances outside the `build` method is prohibited
+
+#### Performance Optimization
+- Large lists (>20 items) must use `LazyForEach` instead of `ForEach`
+- Creating new objects, closures, or calling functions that return styles in the `build` method is prohibited, as it causes unnecessary child component rebuilds
+- Complex computation results should be cached via `@Watch` to avoid redundant calculations on each render
+- Image resources should have proper caching strategies to avoid repeated loading
+
+#### Resource Access Standards
+- String hardcoding is prohibited; use `$r('app.string.key')` to support internationalization
+- Images must use `$r('app.media.icon')` or `$rawfile('path')`; hardcoded paths are prohibited
+- Colors/dimensions should use resource references like `$r('app.color.primary')` to support theme switching
+
+#### Component Communication
+- Parent→Child: use `@Prop`/`@Link`; Child→Parent: use callback function `onEvent` pattern
+- Cross-component communication: use `@Provide/@Consume`; global state: use `AppStorage`
+- Avoid passing local component state through `AppStorage`
+
+#### General TypeScript Standards
+- Using `any` type is prohibited; if unavoidable, a comment explaining the reason is required
+- Using `var` is prohibited; use `let` or `const`
+- Using `==` and `!=` is prohibited; use `===` and `!==`
+- Async functions must include try-catch error handling with user-friendly error messages
+- Prefer async/await; callback hell is prohibited; use `Promise.all` for independent async operations
+- Null checks: perform null checks when accessing values or destructuring to avoid null pointer exceptions
+
+#### Code Security Checks
+- User input must be validated (length, format, range); direct concatenation into SQL or command strings is prohibited
+- Sensitive information (keys, passwords, tokens) must not be logged or uploaded
+- Network requests must use HTTPS with certificate verification
diff --git a/vendor/open-code-review/rule_docs/astro.md b/vendor/open-code-review/rule_docs/astro.md
new file mode 100644
index 0000000..5bdb8d9
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/astro.md
@@ -0,0 +1,45 @@
+#### Obvious Typos or Spelling Errors
+- Spelling errors in component names, props, slots, or user-facing strings that affect readability
+
+#### Dead Code
+- Unused islands, framework components, scripts, or template branches that add client cost without affecting rendered behavior
+
+#### Astro Component Boundaries
+- When frontmatter data reaches client HTML, inline scripts, or hydrated islands, verify whether it was computed at build time or request time and whether exposing non-`PUBLIC_` env values, cookies, headers, sessions, `Astro.locals`, secrets, request-only data, or server-only APIs is intentional
+- Flag `.astro` templates that appear to assume frontmatter values are reactive in the browser
+- Flag framework components used only to render static markup when plain Astro markup would avoid unnecessary client JavaScript
+
+#### Hydration and Islands
+- `client:*` applies only to directly imported UI framework components, not `.astro` components or dynamic tags
+- Flag `client:load` on non-critical UI, missed `client:idle` or `client:visible` opportunities, `client:media` where the media query does not actually gate the interaction need, and over-hydration from large or overly numerous islands
+- Flag `client:only` without the framework string or without fallback content when the result is blank or confusing pre-hydration UI
+
+#### Server-to-Client Data Transfer
+- Flag hydrated framework component props or server-fetched data passed client-side without reducing to the minimal interaction payload; props crossing hydrated boundaries must use Astro-supported serializable types, so flag functions, class instances, circular objects, secrets, and unnecessarily large payloads.
+- `` (and relevant line-separator characters for the supported JavaScript target), allowing data to end the script element or break parsing
+- Secrets, session data, authorization material, internal-only fields, or unnecessarily large objects serialized into client-visible markup or scripts
+- Client-compiled templates (`compileClient` / `compileFileClient`) that can render attacker-controlled raw output before the caller inserts the result with `innerHTML` or an equivalent HTML sink
+- Server-only assumptions embedded in client-compiled Pug, including access to unavailable globals, modules, filesystem state, or locals the browser never receives
+
+#### Template Trust, JavaScript, Missing Data, and Side Effects
+- Attacker-controlled template source passed to `compile` / `render`, or attacker-writable files passed to `compileFile` / `renderFile`; Pug templates can execute embedded JavaScript, so treating untrusted templates as presentation data creates a server-side template injection and code-execution boundary
+- Request-controlled template file selection without a strict name allowlist and template-root boundary, allowing unintended templates or paths to be rendered even though `include` and `extends` directives inside a trusted template are static
+- Request, query, or body objects spread wholesale into Pug render/compiler options, allowing untrusted keys to alter options such as `filename`, `basedir`, `filters`, `globals`, `cache`, or `pretty` as well as locals; keep compiler options fixed, expose an explicit locals shape, and verify the project's Pug version before claiming exploitability
+- Unbuffered code (`-` or an unbuffered code block) that performs database, network, filesystem, shared-state mutation, or other externally observable side effects during rendering
+- Required locals that can become empty output, omitted attributes, invalid identifiers, or broken URLs without an explicit default or branch; do not require defaults for optional display-only values
+- Exceptions from property access or function calls on absent locals that can abort the whole render, especially inside shared layouts or error pages
+- Business logic, authorization decisions, or non-trivial data shaping implemented in the template instead of the controller/view-model layer
+
+#### Includes, Inheritance, Mixins, and Filters
+- Relative `include` or `extends` paths used without the `filename` context needed for correct resolution, or absolute paths that unintentionally depend on a different `basedir`
+- Non-Pug includes that insert raw HTML, JavaScript, CSS, or secrets when the author appears to expect Pug parsing or escaping
+- Child templates that replace, append, or prepend the wrong named block and thereby drop required metadata, scripts, security controls, fallback content, or accessibility structure
+- Content or unbuffered code placed at the top level of an extending template even though child templates may only define named blocks and mixins there
+- Mixins invoked with arguments in the wrong order, implicit caller context, or an overly broad attribute object, producing missing data or leaking fields across component boundaries
+- Filters treated as runtime sanitizers or as processors for dynamic locals even though Pug filters run at compile time and cannot support dynamic content or options
+
+#### Indentation, Control Flow, and Whitespace
+- Indentation changes that move an element into the wrong parent, loop, conditional, or block, changing form ownership, DOM structure, visibility, or execution scope
+- `while` loops with a reachable non-terminating path, or expensive expressions and function calls repeated inside large `each` loops
+- `each ... else` or conditional branches that omit required empty/error states, duplicate IDs, or render controls without the data needed by their handlers
+- Inline tags and text whose Pug whitespace rules concatenate words or tokens, or reliance on trailing spaces that formatters and editors can remove
+- Literal HTML or plain-text blocks whose indentation looks like Pug nesting but is emitted mostly unprocessed, producing a different structure than intended
+
+#### Accessibility
+- Interactive non-button elements without equivalent keyboard activation, focusability, and semantics when a native element would provide them
+- Inputs generated without an associated label, repeated form-control IDs, or labels/ARIA references that point to a different loop item
+- Images with missing or misleading alternatives, and conditional content that changes visible state without updating accessible name, role, or `aria-*` state
+- Inheritance or mixin overrides that silently remove landmarks, heading structure, fallback text, focus management, or required attributes supplied by the parent
+
+#### Performance and Review Scope
+- Templates compiled on every request or repeatedly inside a hot loop when they are static and the host can precompile or cache them with a stable `filename` key
+- Expensive helpers, mixins, serialization, or data reshaping repeated per item when the result can be prepared once by the host application
+- Report performance findings only when request frequency, collection size, or repeated compilation is evident; do not turn indentation preferences or equivalent shorthand forms into blocking findings
diff --git a/vendor/open-code-review/rule_docs/python.md b/vendor/open-code-review/rule_docs/python.md
new file mode 100644
index 0000000..644d2a0
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/python.md
@@ -0,0 +1,76 @@
+> Favor precision over recall: only raise an issue when you are confident it is a real defect, and stay silent when the surrounding context is unclear — a false alarm costs more reviewer trust than a missed minor issue. Treat security and correctness findings as blocking, and style or idiom suggestions as non-blocking.
+
+#### Obvious Typos or Spelling Errors
+- Spelling errors in variable, function, class, or module names at their declaration sites; do not report spelling errors at reference sites, as these are determined by the declaration
+- Strings in log messages or exception messages containing spelling errors that affect readability
+
+#### Dead Code
+- Code blocks that can never be reached (e.g., branches where the condition is always false, code after a `return`, `raise`, `break`, or `continue`)
+- Variables, imports, or function parameters that are declared but never read or referenced
+- Large blocks of commented-out code with no apparent intent to preserve
+
+#### Mutable Default Arguments and Shared State
+- Mutable default arguments such as `def f(x=[])` or `def f(x={})`; the default is created once and shared across every call. Default to `None` and build the value inside the body
+- Class-level mutable attributes shared unintentionally across instances when a per-instance value was intended
+- Module-level mutable globals (lists, dicts, caches) mutated across requests or threads, retaining state in ways that surprise the caller
+- Closures that capture a loop variable by reference and all end up seeing its final value
+- Do not report when the function never mutates the argument, or when the shared default is a deliberate, documented cache or sentinel
+
+#### Boundary and Edge-Case Handling
+- Empty inputs assumed to be non-empty: indexing `xs[0]`, `max()`/`min()`, or slicing without first handling the empty `list`, `str`, `dict`, or iterator
+- Off-by-one and out-of-range access on indices, ranges, or slices, especially at the first/last element
+- `None` reaching code that assumes a value, when an upstream call or default can legitimately return `None` (confirm the data source with `file_read` before flagging)
+- Comparing floats for exact equality with `==`; use `math.isclose` or an explicit tolerance, since floating-point results are not exact
+- Integer/float and division assumptions: unintended truncation with `//`, or `ZeroDivisionError` when a divisor can be zero
+- Heterogeneous or unexpected element types in a collection that the code assumes are uniform (e.g., mixing `None`, numbers, and strings)
+- Dictionary access by key without handling the missing-key case (`d[k]` vs `d.get(k)`), or set/dict operations that assume a key is present
+- Do not report edge cases that a caller or type contract has already ruled out, or inputs that cannot occur given validated boundaries upstream
+
+#### Error Handling and Exceptions
+- Bare `except:` swallows everything, including `KeyboardInterrupt` and `SystemExit`; catch `except Exception` at minimum, and prefer the specific exception types you expect
+- `except Exception` that is still broader than the failure being handled; narrow it to the exceptions actually raised by the guarded call
+- Exceptions caught and silently discarded (`pass`) without logging or re-raising
+- Original traceback lost when re-raising; prefer `raise NewError(...) from err` to preserve the cause
+- Broad `try` blocks that wrap far more than the line that can actually fail, hiding where the error originates
+- `assert` used for runtime validation of external input — assertions are stripped under `python -O`
+
+#### Identity and Equality Comparisons
+- Using `is`/`is not` to compare against literals such as strings, numbers, or tuples; this relies on implementation-specific interning rather than value equality — use `==` (a real correctness risk)
+- Comparing against `True`/`False` with `==`, where a truthy-but-not-`True` value (e.g. `1`, a non-empty container) would compare unequal; prefer a plain truthiness check
+- Reserve `is` for identity checks against singletons and sentinels
+- Comparing against `None` with `==`/`!=` rather than `is`/`is not` is a style preference; report as minor, not blocking
+
+#### Resource Management
+- Files, sockets, locks, or database connections opened without a `with` statement, risking leaks on early return or exception
+- Context managers available but bypassed in favor of manual `open()`/`close()` pairs
+- Resources acquired in a `try` whose `finally` cleanup is missing or incomplete on the error path
+- Iterators or generators holding resources open longer than necessary
+- Do not report short-lived scripts, or handles already managed by an enclosing `with` or framework-managed lifecycle (confirm the surrounding scope with `file_read` before flagging)
+
+#### Performance
+Confirm data scale and that the code is on a hot path before flagging:
+- Building strings with `+=` in a loop instead of accumulating in a list and `"".join(...)`, or using an f-string
+- Repeated membership tests against a `list` where a `set` or `dict` would turn O(n) lookups into O(1)
+- Building a full list when a generator would avoid holding everything in memory
+- Recomputing inside a loop a value that is invariant across iterations (e.g., compiling a regex, attribute lookups in hot paths)
+- Passing an eagerly formatted f-string to `logging` (e.g., `logging.info(f"...")`) instead of `logging.info("%s", value)`, which defeats lazy formatting when the level is disabled
+
+#### Concurrency and Async
+Only flag concurrency issues when there is evidence of multi-threaded, multi-process, or async invocation (confirm the call context before reporting):
+- CPU-bound work parallelized with `threading` under the GIL where `multiprocessing` or a process pool is the right tool (traditional CPython; free-threaded builds excepted); I/O-bound work is the case threads actually help
+- Check-then-act races on shared state without a `Lock`, or non-atomic compound updates assumed to be atomic
+- Blocking calls (synchronous I/O, `time.sleep`, `requests`, CPU-heavy work) inside `async def`, stalling the event loop; use the async equivalent or run them in an executor
+- `asyncio` tasks created and never awaited, so exceptions are swallowed and the work may be garbage-collected before it finishes
+- Shared mutable state across threads or tasks without synchronization or a thread-safe structure
+
+Do not report local variables (each thread has its own), read-only access to shared data, or code with no evidence of concurrent use.
+
+#### Security-Sensitive Code
+Validate the data source before flagging; confirm the input is actually attacker-controlled rather than a trusted constant:
+- `eval`, `exec`, or `compile` on untrusted input; this is arbitrary code execution
+- `subprocess` with `shell=True` built from unsanitized input; pass an argument list and avoid the shell
+- `pickle`, `marshal`, or `yaml.load` (without `SafeLoader`) on untrusted data; deserialization can execute arbitrary code
+- SQL built by string concatenation or f-strings instead of parameterized queries
+- Secrets, tokens, passwords, or PII written to logs or committed in source
+- Weak or misused cryptography (`hashlib.md5`/`sha1` for passwords, `random` for security tokens); use `secrets` and vetted libraries
+- Untrusted file paths joined without validation, allowing path traversal
diff --git a/vendor/open-code-review/rule_docs/r.md b/vendor/open-code-review/rule_docs/r.md
new file mode 100644
index 0000000..8d3fef2
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/r.md
@@ -0,0 +1,53 @@
+#### R Code Review Principles
+
+> Favor correctness, numerical precision, memory efficiency, and vectorization over style: report only defects likely real in the changed code, statistical models, pipelines, and package structure. Treat data corruption, subtle scope bugs, vectorized logic failures, unsafe evaluation, and package build breaks as blocking; style-only suggestions are non-blocking. Do not duplicate errors that `R CMD check`, `lintr`, `styler`, or standard R parser tools identify mechanically unless the diff reveals a concrete production or runtime execution risk.
+
+Before reporting a non-local claim, inspect function definitions, namespace loads (`library()`, `require()`, `pkg::func`), lazy evaluation boundaries, NSE (non-standard evaluation) contexts, formula interfaces, environment chains, and package dependencies. Do not assume a vectorized function, S3/S4 method dispatch, or memory operation is unsafe without evidence of caller context, input data structures, or performance limits.
+
+#### Vectorization, Data Types, and Type Safety
+
+* Unintentional scalar logical operations (`&&`, `||`) used on vectors where element-wise operations (`&`, `|`) are required, or element-wise logicals used inside scalar conditionals like `if (...)`.
+* Implicit type coercion caused by mixing types in vectors, matrices, or `c()` calls (e.g., mixing `character` and `numeric`), or relying on implicit `factors`-to-character/numeric conversions without explicit `as.character()` or `as.numeric(as.character())`.
+* Missing or incorrect edge-case handling for zero-length inputs (`numeric(0)`, `character(0)`), empty data frames, single-row/column matrices dropping dimensions (`drop = FALSE`), or unexpected `NA`/`NULL`/`NaN`/`Inf` propagation.
+* Unsafe recycling of vectors in arithmetic, comparisons, or data frame assignments where vector lengths are not equal or exact multiples, leading to silent standard R recycling or subtle calculation bugs.
+* Relying on base equality checks (`==`) with floating-point numbers instead of `isTRUE(all.equal(...))` or setting threshold tolerances (`abs(x - y) < eps`).
+
+#### Non-Standard Evaluation (NSE) and Tidyverse/Data.Table Syntax
+
+* Unquoted column references, dynamic variable names, or programmatic evaluation using NSE (e.g., `dplyr::select()`, `ggplot2::aes()`, `data.table` expressions) without proper quasiquotation (`!!`, `{{{ }}}`, `sym()`, `all_of()`, `any_of()`) when passed as function parameters.
+* Ambiguity between data frame column names and environment variables inside `dplyr`, `data.table`, or `subset()` expressions, missing explicit `.data$` or `.env$` pronoun usage in package code.
+* Side effects in `data.table` in-place modification (`:=`) leaking into caller environments or modifying passed arguments without explicit deep copying (`copy()`).
+* Misuse of standard base evaluation inside tidyverse pipeline functions or vice-versa, causing delayed execution failures or unexpected binding contexts.
+
+#### Scope, Lazy Evaluation, and Environment Boundaries
+
+* Scoping bugs where functions implicitly rely on global environment variables (`.GlobalEnv`) rather than explicitly passed arguments or package options (`getOption()`).
+* Unintended variable capture in delayed evaluation contexts, lazy promises, `lapply()` / `purrr::map()` loops, or standard `for` loops where iteration variables are referenced lazily inside closures/lambdas.
+* Modifying caller environments using `assign()`, `<<-`, or `parent.frame()` without explicit architectural justification, clear lock boundaries, or documentation of side effects.
+* Mismanaging S3, S4, or R6 method dispatch, wrong class inheritance order, or failing to call `UseMethod()` / `callNextMethod()` correctly.
+
+#### Memory Management, Performance, and I/O
+
+* Repeated memory re-allocation inside loops (e.g., appending rows to data frames with `rbind()` or extending vectors dynamically) instead of pre-allocating output vectors or using vector/list accumulation.
+* Deep copying of large data objects in memory when passing to functions or executing multi-step transformations where memory-efficient tools (`data.table`, `arrow`, `dbplyr`, or `vroom`) should be used.
+* Missing explicit connection closures or resource cleanup (`close()`, `on.exit()`) when opening file handles, database connections, graphics devices (`dev.off()`), or temporary directories.
+* Unfiltered large dataset imports using `read.csv()` or generic base I/O instead of chunked, memory-mapped, or fast parallel alternatives (`data.table::fread()`, `arrow::read_parquet()`, `vroom::vroom()`).
+
+#### Statistical Precision and Numerical Stability
+
+* Numerical instability or overflow/underflow in custom mathematical functions, likelihoods, or matrix operations where log-scale computations (`log1p()`, `expm1()`, `log-sum-exp`), specialized solvers, or QR decomposition should be used instead of direct inversion (`solve()`).
+* Improper handling of missing data (`NA`) in statistical summaries, aggregates, or model estimation (`na.rm = TRUE`, `na.action` settings), leading to unhandled `NA` results or unexpected row dropped patterns.
+* Random number generation (RNG) calls (`rnorm()`, `runif()`, etc.) lacking reproducible `set.seed()` calls in tests or stochastic workflows, or unsafe seed state handling in parallel execution (`L'Ecuyer-CMRG` workers).
+
+#### Package Structure, Dependencies, and Namespace
+
+* Direct use of `library()` or `require()` inside package functions instead of properly declaring imports in `DESCRIPTION` (`Imports`, `Suggests`) and namespace imports via `NAMESPACE` (`importFrom`).
+* Unqualified calls to non-base package functions inside package code that depend on global search path order, rather than using `package::function()` prefixing.
+* Polluting the global search path or masking core methods through overly broad `import(pkg)` directives in package development.
+* Non-portable file paths using hardcoded path separators (`/` or `\`), absolute local paths, or user-specific home directories instead of `file.path()`, `here::here()`, or standard R temporary directory utilities (`tempdir()`, `tempfile()`).
+
+#### Review Scope
+
+* Focus on logical correctness, vectorization bugs, memory safety, data frame integrity, statistical precision, and package export safety.
+* Do not report pure code styling choices (e.g., `=` vs `<-` assignment, indentation width, snake_case vs camelCase naming) or documentation missingness unless it breaks package vignettes or `R CMD check`.
+* When the code change is intentionally part of a major library overhaul or migration (e.g., converting base R code to `dtplyr` or `rlang`), review the full execution context and dynamic inputs before flagging a compatibility issue.
\ No newline at end of file
diff --git a/vendor/open-code-review/rule_docs/rust.md b/vendor/open-code-review/rule_docs/rust.md
new file mode 100644
index 0000000..022b9b6
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/rust.md
@@ -0,0 +1,61 @@
+#### Obvious Typos or Spelling Errors
+- Spelling errors in type names, function names, variable names, enum variants, trait names, or module names at their declaration sites; do not report spelling errors at call sites
+- Strings in log messages, panic messages, error messages, or public diagnostics containing spelling errors that affect readability
+
+#### Ownership and Lifetime Correctness
+- Incorrectly returned references, borrowed values escaping their valid scope, or lifetime relationships that make an API unsound or unusable
+- Excessive or unnecessary `clone()` calls introduced to satisfy borrowing when a borrow, iterator, `Cow`, or ownership transfer would be clearer and cheaper
+- Interior mutability (`RefCell`, `Cell`, `Mutex`) used to work around ownership without a real shared-mutability requirement
+- Reference cycles with `Rc>` or `Arc>` where `Weak` should be used to break ownership cycles
+
+#### Error Handling and Panics
+- `unwrap()`, `expect()`, `panic!`, `todo!`, or `unimplemented!` in production/library paths where the failure is recoverable or can be propagated with `Result`
+- Errors converted to strings too early or discarded without context; prefer preserving the original error and adding actionable context at boundaries
+- `Result` or `Option` values ignored, swallowed, or mapped to misleading defaults
+- Public APIs that panic on ordinary invalid input instead of returning a typed error, unless the panic documents a clear programming invariant
+
+#### Unsafe Code Boundaries
+- `unsafe` blocks that are broader than necessary or hide multiple unrelated invariants
+- Missing or stale safety rationale for `unsafe` blocks, `unsafe fn`, `unsafe impl Send`, or `unsafe impl Sync`
+- Raw pointer dereferences without clear validity, alignment, initialization, aliasing, and lifetime guarantees
+- FFI boundaries that do not validate null pointers, buffer lengths, ownership transfer, string encoding, or allocator compatibility
+- `static mut`, unchecked `transmute`, `MaybeUninit`, `mem::zeroed`, or manual drop logic used without a documented invariant that makes the operation sound
+
+#### Concurrency and Shared State
+- Holding `Mutex`, `RwLock`, or `RefCell` guards longer than necessary, especially across calls into user code or potentially blocking operations
+- Holding synchronous locks across `.await`, or using blocking I/O, sleeps, or CPU-heavy work directly inside async tasks
+- Check-then-act races around shared state, cache initialization, file creation, or atomics
+- Atomic operations with ordering that is too weak for the data being protected, or overly strong orderings that hide the intended synchronization contract
+- Unsafe `Send` or `Sync` implementations that do not prove all contained state is thread-safe under the documented invariants
+
+#### Async and Cancellation Safety
+- Spawned tasks whose `JoinHandle` is dropped when failures, cancellation, or shutdown still need to be observed
+- Futures that are not cancellation-safe around partial writes, lock acquisition, transactions, or resource cleanup
+- Async functions that use synchronous filesystem, network, or process APIs in request/worker paths where the runtime can be blocked
+- Retry loops without backoff, timeout, cancellation propagation, or bounded attempts
+
+#### Collections, Iterators, and Performance
+- Avoid unnecessary allocations in hot paths, such as repeated `String` construction, `format!`, `collect()`, or `to_vec()` where borrowing or streaming is sufficient
+- Prefer iterator adapters and standard library collection APIs when they make ownership and complexity clearer; avoid dense iterator chains that obscure error handling or side effects
+- Ensure hash maps, vectors, and strings are preallocated when the expected size is known and growth cost is material
+- Avoid O(n^2) lookups from nested loops when a `HashMap`, `HashSet`, sorting, or indexing strategy would clearly reduce complexity
+
+#### Type and API Design
+- Model domain states with enums, newtypes, and typed IDs instead of booleans, strings, or primitive integers when invalid states would otherwise be representable
+- Prefer standard conversion and borrowing traits (`From`, `TryFrom`, `AsRef`, `Borrow`, `IntoIterator`) when designing reusable APIs
+- Public structs, enums, traits, and errors should have useful names, visibility, trait derives, and documentation appropriate to the crate boundary
+- Avoid exposing concrete collection or synchronization types in public APIs when a slice, iterator, trait, or narrower abstraction would preserve flexibility
+
+#### Macros and Metaprogramming
+Only flag when the diff actually defines a `macro_rules!` or procedural macro; do not report on ordinary macro invocations.
+- An `$x:expr` fragment interpolated more than once in the expansion, so the caller's expression — and any side effects — runs multiple times; bind it to a `let` once inside the expansion
+- Exported or publicly used macros that reference items without `$crate::`, so name resolution breaks or binds the wrong item when the macro is invoked from another crate
+- Token-tree (`$t:tt`) fragments re-emitted without parentheses, where operator precedence can silently change the intended meaning; this applies only to token-level (`tt`) interpolation, since an `:expr` fragment and a whole expansion are each parsed as one complete expression
+- Procedural macros that `unwrap()`, `expect()`, or `panic!` on malformed input instead of emitting a `syn::Error` / `compile_error!` with a useful span
+- Macro-hygiene assumptions that break: generated identifiers relying on names from the call-site scope, or items that collide when the macro is invoked more than once in the same module
+
+#### Security-Sensitive Code
+- Validate path, URL, command, SQL, and serialized input before use; do not build shell commands or SQL with unchecked string concatenation
+- Do not log secrets, tokens, credentials, private keys, or personally identifiable information
+- Check integer conversions, byte slicing, and length arithmetic for overflow, truncation, and UTF-8 boundary errors
+- Cryptographic, random, authentication, and authorization code must use well-reviewed crates and explicit error handling; flag ad hoc implementations
diff --git a/vendor/open-code-review/rule_docs/solidity.md b/vendor/open-code-review/rule_docs/solidity.md
new file mode 100644
index 0000000..e5159c9
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/solidity.md
@@ -0,0 +1,95 @@
+#### Solidity Review Principles
+> Favor precision over recall: report only issues likely to cause loss of funds, incorrect accounting, permanent lockup, or a security vulnerability on a reachable path. Read the `pragma` before raising any arithmetic finding — Solidity `>=0.8` reverts on overflow and underflow by default, so reporting unchecked wraparound outside an `unchecked` block is a false positive. The `.sol` extension is also used for Gerber PCB solder-mask layers; if the file is not Solidity source, report nothing.
+
+Do not duplicate what `solc`, Slither, or solhint already flag unless the diff creates concrete correctness impact. Before reporting non-local behavior, verify the caller, the modifier chain, the inherited implementation, and the storage layout rather than inferring them from names.
+
+#### Obvious Typos or Spelling Errors
+- Misspellings in declaration sites that become part of the contract's surface: function, event, error, state variable, parameter, struct field, or `constant` names
+- Do not report typos inside comments, NatSpec prose, or revert strings unless the string is matched on elsewhere
+
+#### External Calls and Reentrancy
+- State written after an external call rather than before it, where the callee can re-enter and observe stale state (Checks-Effects-Interactions)
+- Cross-function reentrancy: the entry point carries `nonReentrant` but a sibling function sharing the same storage does not
+- Cross-contract reentrancy where two contracts share accounting state and only one guards the path
+- Read-only reentrancy: a `view` getter consulted by a third party while a callback is in flight returns a value derived from half-updated state
+- ERC777/ERC1155/ERC721 transfer hooks and `receive`/`fallback` treated as inert; any token transfer to an arbitrary address is an external call
+- A guard added to the wrapper while the `internal` function it delegates to is also reachable from an unguarded path
+
+#### Unchecked Call Results
+- Return value of `.call`, `.send`, or `.delegatecall` discarded, so a failed call proceeds as success
+- A low-level call to an address with no deployed code returns `true`; missing an `extcodesize` or equivalent existence check before trusting it
+- ERC20 `transfer`, `transferFrom`, or `approve` used directly on tokens that return nothing or return `false` instead of reverting, where `SafeERC20` is required
+- Success flag captured into a variable that is never asserted on
+
+#### Delegatecall and Proxy Upgradeability
+- `delegatecall` to a target that a caller can influence, or to an address read from mutable storage without an allowlist
+- Storage-slot collision between proxy and implementation, or between an old and a new implementation: variables reordered, inserted, or removed, and consumed `__gap` slots not reduced by the same count
+- A variable that became `constant` or `immutable`, or stopped being one, across an upgrade — every subsequent slot shifts
+- Implementation contract left initializable: `initializer` modifier missing, `_disableInitializers()` absent from the constructor, an inherited `__X_init` never called, or a `reinitializer` version reused
+- `selfdestruct` or `delegatecall` reachable in an implementation, which can brick every proxy pointing at it
+- Function-selector collision or shadowing between the proxy and the implementation, silently routing a call to the wrong body
+- Upgrade authorization (`_authorizeUpgrade`, `onlyProxy`) missing or guarded by a role that is not the intended one
+
+#### Access Control
+- Privileged function missing its `onlyOwner`/`onlyRole` modifier, or carrying a modifier that does not gate the value it is meant to gate
+- `tx.origin` used to authenticate a caller
+- Unprotected initializer, setter, or minting entry point
+- Single-step ownership transfer to an address that is never verified, where a two-step handshake is required
+- Role admin left as the role itself, letting members grant the role to anyone
+- A `public`/`external` visibility change on a function that was previously `internal`
+
+#### Arithmetic and Conversions
+- An `unchecked` block whose bound is not locally provable — a loop counter compared against `.length` is fine, a subtraction on caller-supplied input is not
+- Downcast (`uint256` to a narrower type, or signed/unsigned conversion) that truncates or flips sign on a reachable value
+- Division before multiplication, compounding precision loss in a fee, share, or interest calculation
+- Token decimals assumed to be 18, or two tokens' decimals assumed equal
+- Rounding that favors the caller over the protocol on both deposit and withdrawal paths
+
+#### Storage and Initialization
+- Uninitialized local `storage` pointer, which writes to slot 0
+- A `memory` copy of a struct or array mutated where `storage` was intended, so the write is discarded
+- `delete` applied to a struct or array containing a mapping, which leaves the mapping populated
+- `immutable` or `constant` expected but the value is set in a function that can run more than once
+
+#### Ether and Token Handling
+- `transfer` or `send` relying on the 2300-gas stipend against a recipient that may be a contract
+- Push payments in a loop where one reverting or gas-hungry recipient blocks every other recipient; a pull pattern is required
+- Fee-on-transfer or rebasing tokens credited with the amount argument rather than the measured balance delta
+- `msg.value` read inside a loop or a `payable` multicall, letting one deposit be counted many times
+- Contract that can receive Ether with no withdrawal path, or a `receive`/`fallback` that reverts on a path the protocol depends on
+
+#### Randomness, Time, and Oracles
+- `block.timestamp`, `block.number`, `blockhash`, or `block.prevrandao` used as a randomness source for anything of value
+- A spot pool reserve, `getReserves`, or a single-block TWAP read as a price, which is flash-loan manipulable
+- Chainlink `latestRoundData` consumed without checking staleness (`updatedAt`), a positive answer, or round completeness
+- Timestamp comparisons that assume a fixed block interval
+
+#### Front-Running and MEV
+- Swap, mint, or redeem without both a deadline and a minimum-output (or maximum-input) bound
+- The ERC20 approve race, where a non-zero allowance is overwritten with another non-zero value
+- A value derived from state an attacker can move within the same block or transaction bundle
+- Commit-reveal or auction logic whose commitment can be front-run because it omits the sender or a salt
+
+#### Gas and Denial of Service
+- A loop over an array or mapping-backed list that any caller can grow without bound
+- An external call, a `SLOAD`-heavy read, or a token transfer executed inside a loop
+- Unbounded array growth in storage with no removal or pagination path
+
+#### Events and Signatures
+- A state change with no event emitted, or an event whose `indexed` fields do not identify the affected party
+- `abi.encodePacked` over two or more dynamic arguments feeding a hash, allowing collisions; use `abi.encode`
+- A signature accepted without a nonce, a chain id, or an EIP-712 domain separator, permitting replay across accounts, chains, or contracts
+- `ecrecover` used without rejecting the zero address and without rejecting the malleable high-`s` form
+- Signature verification that does not bind the signed payload to the caller or the specific action
+
+#### Inline Assembly
+- Assembly that writes past the free memory pointer, or fails to update it after allocating
+- `mload`/`mstore`/`calldatacopy` with an offset or length derived from calldata without a bounds check
+- A block annotated `memory-safe` that does not satisfy the memory-safety rules
+- Storage slots computed by hand that overlap a compiler-assigned slot
+
+#### Testing Correctness
+- A fuzz test whose `vm.assume` or bound filters out exactly the domain the property is about
+- A test asserting nothing about the behavior named in the test, or asserting only that a call did not revert
+- A fork test pinned to no block number, making it non-reproducible
+- Mocks that always return success, hiding the failure path the change introduced
diff --git a/vendor/open-code-review/rule_docs/swift.md b/vendor/open-code-review/rule_docs/swift.md
new file mode 100644
index 0000000..528e99f
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/swift.md
@@ -0,0 +1,105 @@
+#### Swift Review Principles
+> Favor precision over recall: report only defects likely real in changed code and reachable execution paths. Prioritize crashes, data corruption, security issues, privacy issues, and concurrency bugs. Do not report style preferences.
+
+Before reporting non-local behavior, use `file_read` and `code_search` to verify ownership, callers, synchronization, lifecycle, and input sources. Do not infer threading, retain cycles, or error contracts only from names or types. Do not duplicate compiler, SwiftLint, or Xcode analyzer findings unless the diff creates concrete correctness impact.
+
+#### Optionals and Runtime Failures
+- Force unwrap, force cast, or `try!` on runtime-derived values (user input, network responses, persistence, decoding, external state) where failure is reachable and not handled.
+- Implicitly unwrapped optionals outside controlled framework lifecycle patterns where access can occur before initialization or after invalidation.
+- Optional handling that converts required failure into silent incorrect behavior, missing data, or invalid state.
+
+#### Memory Ownership and ARC
+- Escaping closures stored by an object that strongly capture that same object, creating a retain cycle.
+- `[unowned]` captures in escaping closures where object lifetime is not guaranteed until execution.
+- Delegate, observer, callback, timer, or task relationships that create ownership cycles or continue work after owner destruction.
+- Combine subscriptions capturing `self` strongly inside an owner of the cancellable when it prevents expected deallocation.
+- Async streams, notifications, timers, or subscriptions started without lifecycle cleanup when they continue after dismissal/deallocation.
+
+#### Error Handling
+- Throwing operations or `Result` failures ignored, replaced with success values, or hidden when failure changes behavior or data correctness.
+- `try?` removing required failure information where callers need failure distinction.
+- Empty error handling blocks suppressing failures affecting integrity, security, or user-visible behavior.
+- Error wrapping removing typed error information required by callers.
+- `fatalError`/`preconditionFailure` used for recoverable runtime failures instead of typed propagation.
+
+#### Swift Concurrency and Isolation
+- Mutable state accessed across actor boundaries without isolation or synchronization where concurrent access is possible.
+- Non-`Sendable` values crossing isolation boundaries where races or unsafe assumptions are possible.
+- `@unchecked Sendable` or `nonisolated(unsafe)` introduced without a proven thread-safety invariant.
+- Actor-isolated state accessed from callbacks, delegates, or closures without preserving isolation.
+- Fire-and-forget tasks that outlive owners, cannot be cancelled, or continue side effects after lifecycle ends.
+- Detached tasks used where inherited actor context, priority, cancellation, or isolation is required.
+- Async work ignoring cancellation and continuing expensive computation or side effects.
+- Continuation wrappers that can resume multiple times, never resume, or resume after ownership/lifecycle invalidation.
+- Locks or synchronous waits used across `await` boundaries.
+- Independent async operations introduced sequentially causing measurable user-visible latency regression.
+
+#### SwiftUI State and Lifecycle
+- View-owned reference state recreated across renders because ownership/lifetime is incorrect.
+- Dynamic collections using unstable identity causing incorrect row reuse or state association.
+- Side effects executed from `body` or computed properties causing repeated execution.
+- Lifecycle async work continuing after disappearance when cancellation ownership is required.
+- `.task(id:)` missing where replaced inputs can allow stale results to overwrite newer state.
+- UI state mutated outside required main actor isolation when concurrent updates are possible.
+- Lifecycle effects duplicated or misattributed across remount, presentation, or dismissal paths.
+- User-visible strings added or changed without localization coverage.
+
+#### Persistence and Data Integrity (SwiftData / Core Data)
+- Persistence writes leaving stored state partially updated or inconsistent after failure.
+- Schema or relationship changes without compatible migration handling for existing data.
+- Relationship configuration changes causing orphaned objects, invalid references, or incorrect delete behavior.
+- `@Query`/fetch predicates or sort descriptors matching incorrect data or causing avoidable expensive fetches.
+- Cached or persisted values treated as authoritative when they can become stale and affect correctness.
+
+#### Health and Privacy Data
+- HealthKit access performed without required authorization handling or safe fallback behavior.
+- Health or sensitive data written to logs, analytics, insecure storage, or plaintext persistence.
+- Health claims introduced without required supporting source or compliance basis.
+- Health queries, observer queries, or background delivery registrations missing lifecycle handling, causing missed updates or unnecessary resource use.
+
+#### Purchases and Entitlements
+- Purchase, restore, or entitlement state failing to handle pending, offline, or verification outcomes.
+- Transaction listeners missing, incorrectly scoped, or failing to consume verified transactions.
+- Paywall or entitlement UI using stale state instead of canonical entitlement state.
+- Trial, restore, or purchase error paths granting or revoking entitlement incorrectly.
+
+#### Combine and Reactive Streams
+- Combine subscriptions causing ownership cycles or continuing after intended lifecycle.
+- UI updates delivered without required scheduler guarantees (`receive(on:)`/equivalent), causing incorrect thread execution.
+- Expensive upstream work executed on inappropriate schedulers where it blocks UI or causes latency.
+- Streams without cancellation/backpressure handling where unbounded work or memory growth is possible.
+
+#### Networking
+- Authentication tokens, credentials, or sensitive data exposed through logs, storage, or requests.
+- Signed/authenticated URLs with bypassed expiry, validation, or authorization checks.
+- Retry logic causing request storms or missing backoff for transient failures.
+- Cache handling serving stale or unauthorized responses.
+- Disabled transport protections or weakened certificate validation where an existing security boundary depends on it.
+- Client-controlled identity, authorization, or payment values trusted without server validation.
+
+#### Web Views, Deep Links, and External Input
+- WKWebView JavaScript bridges accepting unvalidated messages or exposing privileged actions.
+- Navigation handlers allowing untrusted URLs or schemes without validation.
+- Deep-link inputs changing authenticated state or sensitive actions without validation.
+
+#### Performance and Resource Usage
+- Expensive synchronous work on the main actor/thread blocking interaction.
+- Repeated expensive work on frequently executed paths causing measurable regressions.
+- Unbounded memory growth from collections, caches, tasks, streams, or retained objects.
+- Inefficient algorithms on demonstrably large collections causing user-visible slowdown.
+
+#### Security
+- Secrets, credentials, tokens, private keys, or sensitive user data added to source, logs, fixtures, or insecure storage.
+- User-controlled input passed into executable contexts, unsafe URLs, queries, or commands without validation.
+
+#### Unsafe Interoperability
+- Unsafe pointer, buffer, or memory APIs used without guaranteed lifetime or bounds.
+- Objective-C/C bridging violating ownership, nullability, or lifetime assumptions.
+
+#### Testing Correctness
+- Tests relying on arbitrary sleeps or timing delays instead of async expectations or direct awaiting.
+- Tests not exercising changed behavior paths where regressions are likely.
+- Tests sharing mutable global state causing isolation failures.
+- Async tests leaving tasks running after completion.
+- Assertions that cannot fail for the regression they intend to detect.
+- Tests depending on uncontrolled environment state (network, time, locale, global persistence) where isolation is required.
diff --git a/vendor/open-code-review/rule_docs/terraform.md b/vendor/open-code-review/rule_docs/terraform.md
new file mode 100644
index 0000000..e17595f
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/terraform.md
@@ -0,0 +1,29 @@
+> Favor precision over recall: only raise an issue when you are confident it is a real defect, and stay silent when the surrounding context is unclear — a false alarm costs more reviewer trust than a missed minor issue. Treat security and correctness findings as blocking, and style or idiom suggestions as non-blocking. Review only what is observable in the HCL under review; do not infer runtime provider behavior, cloud account configuration, or state stored outside this file.
+
+#### Obvious Typos or Spelling Errors
+- Spelling errors in resource/module/variable/output names at their declaration sites; do not report spelling errors at reference sites
+- Typos in `description` fields that affect readability of the module's public interface
+
+#### Hardcoded Secrets and Credentials
+- A literal password, API key, access key/secret pair, private key, or connection string assigned directly to a resource argument or a `variable`/`locals` default instead of coming from a secret manager, `sensitive` input, or environment-backed data source
+- A `.tfvars` file (this file type is the conventional home for real input values, and is frequently committed by accident with production secrets in it) assigning a real-looking secret value rather than a placeholder
+- A `variable` block that clearly holds a credential (name/description implies password, token, key, or secret) missing `sensitive = true`
+
+#### Overly Permissive Access
+- A security group / firewall / network ACL rule with an unrestricted source (`0.0.0.0/0`, `::/0`, or `"*"`) on a sensitive port (SSH/22, RDP/3389, database ports) or on all ports
+- An IAM policy, role, or resource policy granting a wildcard action (`"Action": "*"`) or wildcard resource (`"Resource": "*"`) instead of a scoped permission set
+- Public read/write ACLs or public access settings enabled on a storage resource (bucket, blob container) that has no clear public-content purpose stated in the diff
+
+#### State and Lifecycle
+- A `terraform.tfstate` or `*.tfstate.backup` file included in the diff — state files can contain resource attributes and secrets in plaintext and should never be committed
+- Removing or weakening a `lifecycle { prevent_destroy = true }` block on a resource that looks stateful/critical (database, persistent volume, KMS key) without an explanation in the diff
+- A stateful resource (database, storage bucket, KMS key) newly created without any `lifecycle` protection, when sibling resources of the same kind in the diff do have one — an inconsistency worth flagging, not an absolute rule
+
+#### Versioning and Reproducibility
+- A `required_providers`/module `source` version constraint left fully unbounded (e.g. no version argument at all, or `>= 0.0.0`) where sibling entries in the same file pin a version — inconsistent, not universally wrong, since some root modules intentionally float
+- Do not flag a deliberately wide constraint (e.g. `~>`, a documented range) that is clearly intentional from the surrounding code
+
+#### Style and Structure
+- Duplicate resource/data-source labels within the same module (would fail `terraform validate`, if not already caught by other tooling)
+- Variables declared but never referenced anywhere in the diff's module, or referenced variables never declared in the diff's scope
+- Do not flag formatting/whitespace that `terraform fmt` would silently fix — focus on structural and semantic issues
diff --git a/vendor/open-code-review/rule_docs/thrift.md b/vendor/open-code-review/rule_docs/thrift.md
new file mode 100644
index 0000000..c144573
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/thrift.md
@@ -0,0 +1,35 @@
+> Favor precision over recall: only raise an issue when you are confident it is a real defect, and stay silent when the surrounding context is unclear — a false alarm costs more reviewer trust than a missed minor issue. Treat wire-compatibility breaks as blocking, and naming or layout preferences as non-blocking.
+
+#### Field IDs and Wire Compatibility
+- Reusing the id of a deleted field; Thrift has no `reserved` keyword, so a retired id must be held open by a placeholder field carrying a "do not reuse this id" comment
+- Renumbering an existing field, or inserting a new field by shifting the ids of everything after it, instead of appending the next unused id
+- Changing the declared type of an existing id, including `i32` to `i64` and swapping an enum for the integer that backs it; the type byte travels in the field header
+- Deleting a field that peers still send without leaving its id held open for the same reason
+- Do not report purely additive fields that take a fresh unused id, comment-only edits, or `namespace` and `include` changes
+
+#### Requiredness and Defaults
+- Adding a `required` field to an existing struct: `required` is permanent and unskippable, so every existing peer fails to deserialize in both directions the moment one side adopts it
+- Flipping an existing field between `required` and `optional`, which changes what a peer is allowed to omit
+- Changing the default value of an existing optional field; an unset field and a field holding the default are indistinguishable to the peer, so the change lands silently
+- Fields left with default requiredness where absence must be distinguishable from the zero value
+- Do not report the choice of default requiredness itself when the file is internally consistent
+
+#### Services and Methods
+- Renaming a service method: method names travel on the wire in `TMessageBegin`, unlike field names, so a rename breaks every existing caller
+- Changing the ids of an existing method's parameters, or adding a parameter declared `required`
+- Adding an exception to an existing `throws` clause that older clients have no branch to decode
+- Changing a method to or from `oneway`, which changes whether the caller waits for a reply at all
+- Do not report new methods appended to an existing service; those are backward compatible
+
+#### Enums and Constants
+- Enum members declared without explicit numeric values, which makes every value positional and shifts them all on the first insertion
+- Inserting a new enum member into the middle of an existing numeric range instead of appending
+- Code that treats an unknown enum value as unreachable; peers on a newer schema will send values this build has never seen
+- Do not report enum members appended with new explicit values
+
+#### Security and Resource Limits
+- Unbounded `list`, `set`, `map`, `string`, or `binary` fields carried over an untrusted transport with no application-level size limit
+- Recursive struct definitions with no documented depth bound on untrusted input
+- Secrets, tokens, or credentials embedded in constants, default values, or comments
+- `string` used to carry non-UTF-8 bytes where `binary` is meant, at a boundary that validates neither
+- Do not report when limits are enforced by transport or server configuration and that boundary is clearly documented
diff --git a/vendor/open-code-review/rule_docs/ts_js_tsx_jsx.md b/vendor/open-code-review/rule_docs/ts_js_tsx_jsx.md
new file mode 100644
index 0000000..5fca87d
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/ts_js_tsx_jsx.md
@@ -0,0 +1,40 @@
+#### Obvious Typos or Spelling Errors
+- Spelling errors in variable names, function names, component names, or Props property names
+- Strings in log or error messages containing spelling errors that affect readability
+
+#### Dead Code
+- Code blocks that will never be executed (e.g., branches where the condition is always false, code after a return statement)
+- Variables that are declared but never read or referenced
+- Large blocks of commented-out code (with no apparent intent to retain)
+
+#### Code Quality Checks
+- **Duplicate Code**: Check for common logic that can be extracted
+- **Code Comments**: Complex business logic should have clear explanatory comments (avoid commenting obvious code)
+- **Hardcoding**: Business-related hardcoded strings are prohibited, especially URL paths and business numbers; simple UI text may be relaxed
+- **Variable Declarations**: Using `var` is strictly prohibited; use `let` or `const`
+- **Equality Comparisons**: Using `==` and `!=` is prohibited; use strict equality `===` and `!==`
+- **TypeScript Types**: Avoid using `any` type; if necessary, provide a comment explaining the reason
+- **Null Checks**: Perform null checks when accessing values or destructuring to avoid null pointer exceptions
+- **Ternary Expressions**: Nested ternary expressions are not allowed
+
+#### React Best Practices
+- **Hooks Usage**: Verify compliance with Hooks rules (only call at the top level, only call in React functions)
+- **State Management**: Ensure state is placed at the appropriate level; avoid unnecessary state lifting
+- **Side Effect Handling**: Verify useEffect correctly handles dependencies and cleanup functions
+- **Performance Optimization**: Verify proper use of React.memo, useMemo, useCallback (based on performance analysis; avoid over-optimization)
+- **Render Side Effects**: Side effects in React component render methods are strictly prohibited (e.g., API calls, DOM manipulation)
+- **Inline Styles**: Avoid using inline `style` attributes, except for dynamic styles
+- **Inner Components**: Declaring new components inside a component is prohibited; use render methods instead (e.g., `renderItem`, not `] `)
+
+#### Async Handling Standards
+- **Error Handling**: Async functions must include proper error handling with user-friendly error messages
+- **Prefer async/await**: Prefer async/await over Promises; callback hell is prohibited
+- **Async in Loops**: Distinguish between independent async operations (use `Promise.all` for parallelism) and dependent async operations (use sequential execution); prefer `Promise.all` for performance
+
+#### Code Security Checks
+- **XSS Protection**: Verify that user input is properly escaped
+- **innerHTML Safety**: Using innerHTML to directly insert user input is prohibited; use textContent or apply XSS protection
+- **Code Injection Protection**: Using eval(), Function() constructor, and string argument forms of setTimeout/setInterval is strictly prohibited
+- **Dangerous Methods**: Using document.write() is prohibited as it causes page reflow and security issues
+- **Sensitive Information**: Check whether API keys or sensitive data are exposed
+- **Prototype Chain Safety**: Modifying native object prototypes (e.g., Array.prototype, Object.prototype) is prohibited
diff --git a/vendor/open-code-review/rule_docs/verilog.md b/vendor/open-code-review/rule_docs/verilog.md
new file mode 100644
index 0000000..88ac1f4
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/verilog.md
@@ -0,0 +1,40 @@
+#### Verilog and SystemVerilog Review Principles
+> Favor precision over recall: report only defects likely to change synthesized hardware behavior, cause simulation/synthesis mismatch, or introduce timing hazards in the changed RTL. Account for whether the file is synthesizable design code or a simulation-only model before raising findings, and do not report style that a linter or formatter already handles. This rule covers Verilog (`.v`), Verilog headers (`.vh`), and SystemVerilog (`.sv`). The `.v` extension is also used by Coq and by the V language; if the file is not Verilog or SystemVerilog, report nothing HDL-specific and fall back to general review principles.
+
+#### Blocking and Non-Blocking Assignments
+- Non-blocking assignments (`<=`) used for combinational logic, or blocking assignments (`=`) used for sequential logic inside a clocked `always`/`always_ff` block, where the mixed style changes simulated ordering or infers unintended hardware
+- Mixed blocking and non-blocking assignments to the same variable when later statements observe a value different from the intended value because non-blocking updates are deferred
+- Do not claim that statement ordering within one procedural block is inherently nondeterministic
+- A register or variable intended to have a single procedural driver being written by multiple processes, particularly violations of `always_ff`, `always_comb`, or `always_latch` single-writer rules
+- Do not flag intentional multiple drivers on an appropriate net type unless an actual conflict is demonstrated
+- Misuse of `always_ff`, `always_comb`, or `always_latch` that violates their event-control, assignment, or single-writer semantics
+- Do not report a correctly formed `always @(*)` solely as a style issue
+
+#### Inferred Latches
+- A combinational `always`/`always_comb` block where a signal is not assigned on every path (missing `else`, an incomplete `case`, or a `case` without `default`), inferring an unintended transparent latch
+- Missing default assignments at the top of a combinational block, so a newly added branch silently reintroduces a latch
+- Outputs left unassigned for some input combinations of a decoder, mux, or FSM next-state logic
+
+#### Signal Width and Signedness
+- Assignments or comparisons between operands of different bit widths that truncate or zero-extend silently, dropping significant bits or changing a comparison result
+- Arithmetic that overflows the declared width of the result, or an intermediate expression narrowed before it is widened
+- Mixed signed/unsigned operands where Verilog's context-determined signedness makes a comparison or shift behave unexpectedly; be explicit with `$signed`/`$unsigned`
+- Part-selects, concatenations, or replication counts whose width does not match the target, and reliance on implicit `reg`/`wire` width from an undeclared net
+
+#### Clock and Reset Handling
+- Reset that is not correctly synchronous or asynchronous as intended, a reset polarity mismatch, or a reset released asynchronously without a synchronizing deassertion (reset-recovery hazard)
+- Sequential logic missing the reset in the sensitivity list for an asynchronous reset, or a value that must survive reset being placed on the reset branch
+- Gated, derived, or combinationally generated clocks used where a clock enable is intended, and multiple clocks driving the same register
+- Registers with no reset where the design assumes a known power-on state
+
+#### Clock-Domain Crossings and Races
+- A signal sampled in one clock domain that is driven from another without a synchronizer (two-flop for single-bit control, handshake or asynchronous FIFO for buses), risking metastability
+- Multi-bit buses synchronized bit-by-bit, so bits arrive skewed and produce transient invalid values; use gray coding or a handshake
+- Combinational feedback loops, or read-during-write races on inferred memory without a defined collision policy
+
+#### Simulation vs. Synthesis and Unsafe Constructs
+- Simulation-oriented constructs such as `#delay`, `fork`/`join`, `force`/`release`, or an `initial` block whose behavior is required by the design but is unsupported by the declared target synthesis flow. Do not flag initialization solely by syntax when the target FPGA or synthesis tool documents support for it
+- Incomplete or overlapping sensitivity lists in a bare `always @(...)` that make simulation differ from the synthesized combinational function; prefer `@(*)` or `always_comb`
+- `casex`/`casez` whose don't-care matching hides priority bugs, or a `case` relying on `x`/`z` matching; prefer `case` with `unique`/`priority` where the intent is exclusive or prioritized
+- `$display`/`$finish`/assertions guarding real behavior, and non-synthesizable system tasks left in the design path
+- Full-case/parallel-case pragmas that assert properties the logic does not actually guarantee
diff --git a/vendor/open-code-review/rule_docs/vhdl.md b/vendor/open-code-review/rule_docs/vhdl.md
new file mode 100644
index 0000000..5e67421
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/vhdl.md
@@ -0,0 +1,36 @@
+#### VHDL Review Principles
+> Favor precision over recall: report only defects likely to change synthesized hardware behavior, cause simulation/synthesis mismatch, or introduce timing hazards in the changed RTL. Account for whether the file is synthesizable design code or a simulation-only model before raising findings, and do not report style that a linter or formatter already handles.
+
+#### numeric_std and Type Usage
+- Arithmetic implemented with non-standard vendor packages such as `std_logic_arith` or `std_logic_unsigned` instead of `numeric_std`, where package-specific overloads make behavior nonportable or ambiguous
+- Incorrect explicit conversions between `integer`, `signed`, `unsigned`, and `std_logic_vector`, including incorrect target widths supplied to `to_unsigned`, `to_signed`, or `resize`
+- Arithmetic or comparisons whose selected overload or result width does not preserve the intended value
+- Width or overload mistakes as distinct from direct incompatible `signed`/`unsigned` mixing, which normally requires an explicit conversion and may be rejected by analysis
+- Metavalue-dependent arithmetic or control flow involving `'U'`, `'X'`, or `'Z'` that synthesized hardware cannot reproduce
+
+#### Ranges, Indexing, and Resize
+- Vector indexed or sliced outside its declared range, or with the wrong direction (`to` vs. `downto`), causing a runtime error in simulation and wrong bits in hardware
+- A `resize` to a smaller width that discards significant information; for `SIGNED`, account for the defined behavior of retaining the original sign bit together with the rightmost part, and for `UNSIGNED`, dropping the leftmost bits
+- Off-by-one range arithmetic in `(N-1 downto 0)` declarations and loop bounds, and `others => '0'` aggregates applied to a target of unexpected width
+
+#### Inferred Latches and Process Sensitivity
+- A combinational process whose sensitivity list omits signals it reads, so simulation and synthesis disagree; prefer `process(all)` (VHDL-2008) where available
+- A combinational process that does not assign every output on every path (missing `else`, incomplete `case`/`when`, no default assignment), inferring an unintended latch
+- A branch within a legal, complete `case` statement that fails to assign an output also assigned by the other branches, causing that output to retain its previous value and infer a latch. Do not require `when others` when every value of the selector subtype is explicitly covered; an incomplete VHDL `case` statement is a compile-time error
+
+#### Clock and Reset Handling
+- Clocked logic that tests only the clock level (for example, `if clk = '1'`) instead of an edge, or places data-path assignments outside the intended edge condition. Accept both `rising_edge(clk)`/`falling_edge(clk)` and the conventional `clk'event and clk = '1'`/`clk'event and clk = '0'` forms when their semantics are intentional
+- Reset that is not synchronous or asynchronous as intended, a reset polarity mismatch, or an asynchronous reset released without synchronized deassertion (recovery/removal hazard)
+- Signals that must retain state across reset placed on the reset branch, or registers with no defined reset where a known startup state is assumed
+- Gated or derived clocks where a clock enable is intended, and more than one clock driving the same register
+
+#### Clock-Domain Crossings and Resolved Signals
+- A signal generated in one clock domain and sampled in another without a synchronizer (two-flop for single-bit, handshake or asynchronous FIFO for buses), risking metastability
+- Multi-bit buses crossing domains without gray coding or a handshake, so bits arrive skewed
+- Multiple drivers on a resolved signal (`std_logic`) relied on for wired logic, where the resolution function hides an unintended multi-driver conflict; unintended shared drivers on what should be a point-to-point signal
+
+#### Simulation vs. Synthesis and Unsafe Constructs
+- Simulation-oriented constructs such as `wait for`, `after` delays, file operations, or reporting side effects whose behavior is required by the design but is unsupported by the declared target synthesis flow. Do not flag initial signal values or assertions solely by syntax when the target tool documents support or safely ignores verification-only statements
+- Variables (`:=`) inside a clocked process used where their immediate-update semantics differ from signal (`<=`) semantics and change the inferred register or its ordering
+- Reads of a signal in the same process cycle expecting the updated value, ignoring VHDL's postponed signal update (delta-cycle) semantics
+- Nonsynthesizable constructs (`access` types, files, unbounded loops, dynamic memory) left in the design path, and metavalue-dependent branches that behave differently in gate-level simulation
diff --git a/vendor/open-code-review/rule_docs/vyper.md b/vendor/open-code-review/rule_docs/vyper.md
new file mode 100644
index 0000000..dca0659
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/vyper.md
@@ -0,0 +1,64 @@
+#### Vyper Review Principles
+> Favor precision over recall: report only issues likely to cause loss of funds, incorrect accounting, or a security vulnerability on a reachable path. Vyper syntax and semantics changed hard at 0.4 — 0.3.x writes `@nonreentrant("key")` and calls externals with plain method syntax, 0.4.x writes bare `@nonreentrant` and requires the `extcall`/`staticcall` keywords — so read the `# pragma version` before applying any version-specific rule. Do not import Solidity idioms: Vyper has no inheritance, no inline assembly, no recursion, no unbounded loops, and arithmetic is checked by default.
+
+#### Obvious Typos or Spelling Errors
+- Misspellings in declaration sites that become part of the contract's surface: function, event, struct field, interface method, storage variable, parameter, or `constant` names
+- Do not report typos inside comments, docstrings, or `assert` messages unless the message is matched on elsewhere
+
+#### Reentrancy and `@nonreentrant`
+- A function that makes an external call before writing state and carries no `@nonreentrant` decorator
+- A decorator removed to satisfy the compiler's "cannot call `X` since it is `@nonreentrant` and reachable from `Y`" error: that error means the lock is genuinely nested, and dropping the decorator removes real protection instead of fixing the call graph
+- `@nonreentrant` counts as state access, so adding it to a function used from a `@view` context or a pure module changes what the module is allowed to do
+- A pinned compiler in the 0.2.15–0.3.0 range, where the reentrancy lock was miscompiled; the version pin is itself the finding, not a style note
+- `raw_call` or a token transfer to an arbitrary address treated as inert; any such call can re-enter
+
+#### Language Restrictions
+- `for i in range(N)` whose bound `N` is smaller than the collection being iterated: the loop silently truncates rather than reverting, so the tail is never processed
+- `raw_call` used as a hand-rolled dispatch table to reconstruct inheritance, losing the compiler's type and mutability checks
+- `create_from_blueprint` or `create_copy_of` factories where the blueprint address is mutable or unverified
+- Logic that assumes recursion or a dynamically sized loop is available and works around its absence incorrectly
+
+#### External Calls and `raw_call`
+- `raw_call` with `max_outsize=0` where the callee's success or return data actually matters
+- `revert_on_failure=False` whose returned success flag is dropped instead of asserted
+- `delegate_call=True` to a target that is not a compile-time constant
+- A state-mutating call made where `is_static_call=True` was intended, or vice versa
+- Return data decoded to a fixed `Bytes[N]` smaller than what the callee can return
+
+#### Access Control and Module Initialization
+- Vyper has no modifiers, so the `assert msg.sender == self.owner` line is written per function and is easy to omit on one of several privileged entry points
+- A `payable` `__default__` that accepts Ether the contract has no way to withdraw
+- `@external` where `@internal` was meant, exposing an internal helper
+- 0.4 `initializes:` / `exports:` re-exporting more of a module's surface than intended, or an `initializes:` module whose `__init__` is never called
+- `__init__` logic reachable a second time through a factory deployment path
+
+#### Arithmetic and Bounds
+- Arithmetic is checked by default — do not report overflow as though this were Solidity's `unchecked`. The real findings are the explicit escapes: `unsafe_add`, `unsafe_sub`, `unsafe_mul`, `unsafe_div` used where the bound is not locally provable
+- `DynArray[T, N]` appended to past `N`, or a length assumed rather than checked
+- `slice`, `concat`, or `extract32` silently bounded by a `Bytes[N]` / `String[N]` capacity smaller than the real input
+- A narrowing `convert()` that truncates or flips sign on a reachable value
+- Division before multiplication in a fee, share, or interest calculation, and rounding that favors the caller on both deposit and withdrawal
+
+#### Storage Layout and Deployment
+- Vyper has no proxy standard, so an upgrade path implies a blueprint or copy deployment; state migration logic that assumes slot compatibility is wrong
+- `--storage-layout-file` overrides that no longer match the declared variables after a reorder or removal
+- Transient storage reused across calls where the value is expected to be cleared
+
+#### Ether, Tokens, and Oracles
+- `send()` relying on the 2300-gas stipend against a recipient that may be a contract; `raw_call` with an explicit gas budget is required
+- `self.balance` used as accounting truth, which any forced transfer can move
+- Non-bool-returning ERC20s called through an interface without `default_return_value=True`, or through `raw_call` with the result unchecked
+- A spot pool reserve or single-block price read as an oracle
+- `block.timestamp`, `block.number`, or `blockhash` used as a randomness source
+
+#### Front-Running and MEV
+- Swap, mint, or redeem without both a deadline and a minimum-output (or maximum-input) bound
+- A value derived from state an attacker can move within the same block
+
+#### Events
+- A state change with no corresponding `log`, or an event whose `indexed` fields do not identify the affected party
+
+#### Testing Correctness
+- A `boa.env.prank` or `boa.reverts` block that asserts nothing about the behavior it names
+- A fuzz test whose bounds exclude the boundary value the property is about
+- A test asserting only that a call did not revert
diff --git a/vendor/open-code-review/rule_docs/yaml.md b/vendor/open-code-review/rule_docs/yaml.md
new file mode 100644
index 0000000..d53fc53
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/yaml.md
@@ -0,0 +1 @@
+Check for spelling errors in yaml-keys within YAML files; ignore the content of yaml-values.
diff --git a/vendor/open-code-review/rule_docs/zig.md b/vendor/open-code-review/rule_docs/zig.md
new file mode 100644
index 0000000..795db4f
--- /dev/null
+++ b/vendor/open-code-review/rule_docs/zig.md
@@ -0,0 +1,35 @@
+> Favor precision over recall: report only issues that are likely to cause incorrect behavior, memory unsafety, security vulnerabilities, or material performance problems. Do not report formatting handled by `zig fmt`, and account for the project's Zig version, build mode (`Debug`, `ReleaseSafe`, `ReleaseFast`, `ReleaseSmall`), and active `comptime` configuration before raising compatibility findings.
+
+#### Memory Safety and Illegal Behavior
+- Slices, pointers, or `[]const u8` views that outlive the storage they refer to, especially addresses derived from stack locals, temporaries, or a buffer that is reused or freed
+- Detectable illegal behavior — out-of-bounds indexing, integer overflow, `@intCast`/`@truncate` narrowing that loses value, null-unwrap of an optional, or invalid `@ptrCast`/`@alignCast` — reachable in `ReleaseFast` or `ReleaseSmall`, where safety checks are disabled and the same code becomes silent undefined behavior
+- Reads of `undefined` memory, or use of a value before it is fully initialized
+- `@ptrCast`, `@alignCast`, `@bitCast`, or pointer arithmetic without a locally established type, alignment, provenance, and lifetime invariant
+- Do not report ordinary value copies or bounds-checked access in `Debug`/`ReleaseSafe` without evidence of a real lifetime or aliasing defect
+
+#### Allocators and Resource Cleanup
+- Memory obtained from an `Allocator` without a matching `free`/`destroy`, or freed with a different allocator than the one that allocated it
+- Resources acquired without a corresponding `defer` or `errdefer`, so an early `return` or error path leaks them or leaves partial state
+- `errdefer` missing on a value that is cleaned up only on the success path, causing a leak when a later step in the same function fails
+- Double-free or use-after-free from a `deinit` that runs on an already-released or aliased object
+- Ignoring the result of an allocation or a fallible call at a boundary where failure changes correctness
+
+#### Errors, Optionals, and Control Flow
+- Error unions discarded with `catch unreachable`, `catch undefined`, or `_ =` where the error is actually reachable at runtime
+- `orelse unreachable` or `.?` on an optional that can legitimately be null for untrusted or runtime input
+- `unreachable` or `@panic` used for ordinary invalid input in reusable library or server code, especially where it becomes illegal behavior in release-unsafe modes
+- `switch` on an error set or tagged union that silently handles unrelated cases with `else` and hides a newly added variant
+- Assertions (`std.debug.assert`) used to validate untrusted input, since they are compiled out in release-unsafe builds
+
+#### Comptime, Generics, and Build Code
+- `comptime` code or `@This()`-based generics that read mutable external state and make builds non-reproducible without an explicit requirement
+- Type-parameter functions that assume capabilities (fields, methods, layout) a caller's type may not provide, producing confusing compile errors instead of a checked constraint
+- `build.zig` steps that fetch, execute, or trust untrusted input, or that hardcode absolute paths and platform assumptions
+- Do not report ordinary `comptime` use when the generated behavior is clear and inputs are validated
+
+#### Concurrency and C Interop
+- Shared mutable state accessed from multiple threads without a `std.Thread.Mutex`, atomic, or established single-owner design
+- Locks held across blocking operations or callbacks, creating deadlock or starvation risk
+- `extern`/`export` declarations or `callconv` annotations with incompatible types, struct layout, nullability, or ownership relative to the C side
+- C strings or buffers consumed without validating length, null termination, encoding, and lifetime
+- User-controlled data passed to process spawning, path access, SQL construction, or deserialization without validation, and secrets embedded in source, logs, or error messages
diff --git a/vendor/open-code-review/supported_file_types.json b/vendor/open-code-review/supported_file_types.json
new file mode 100644
index 0000000..57ca49c
--- /dev/null
+++ b/vendor/open-code-review/supported_file_types.json
@@ -0,0 +1,108 @@
+[
+ ".java",
+ ".kt",
+ ".kts",
+ ".scala",
+ ".groovy",
+ ".py",
+ ".pyi",
+ ".js",
+ ".jsx",
+ ".ts",
+ ".tsx",
+ ".mjs",
+ ".cjs",
+ ".c",
+ ".h",
+ ".cpp",
+ ".cc",
+ ".cxx",
+ ".hpp",
+ ".hxx",
+ ".cs",
+ ".vb",
+ ".fs",
+ ".go",
+ ".rs",
+ ".rb",
+ ".rake",
+ ".gemspec",
+ ".php",
+ ".phtml",
+ ".swift",
+ ".m",
+ ".mm",
+ ".sh",
+ ".bash",
+ ".zsh",
+ ".fish",
+ ".ps1",
+ ".sql",
+ ".css",
+ ".scss",
+ ".sass",
+ ".less",
+ ".html",
+ ".htm",
+ ".ftl",
+ ".ftlh",
+ ".ftlx",
+ ".hbs",
+ ".mustache",
+ ".pug",
+ ".astro",
+ ".vue",
+ ".ipynb",
+ ".svelte",
+ ".xml",
+ ".yaml",
+ ".yml",
+ ".json",
+ ".toml",
+ ".ini",
+ ".env",
+ ".gradle",
+ ".cmake",
+ ".r",
+ ".lua",
+ ".pl",
+ ".pm",
+ ".ex",
+ ".exs",
+ ".erl",
+ ".hrl",
+ ".ets",
+ ".json5",
+ ".dart",
+ ".tf",
+ ".graphql",
+ ".gql",
+ ".prisma",
+ ".jl",
+ ".hcl",
+ ".tfvars",
+ ".bicep",
+ ".proto",
+ ".nix",
+ ".hs",
+ ".lhs",
+ ".nim",
+ ".nims",
+ ".nimble",
+ ".elm",
+ ".properties",
+ ".po",
+ ".pot",
+ ".jsonnet",
+ ".libsonnet",
+ ".zig",
+ ".thrift",
+ ".capnp",
+ ".v",
+ ".sv",
+ ".vh",
+ ".vhd",
+ ".vhdl",
+ ".sol",
+ ".vy"
+]
diff --git a/vendor/open-code-review/system_rules.json b/vendor/open-code-review/system_rules.json
new file mode 100644
index 0000000..d350614
--- /dev/null
+++ b/vendor/open-code-review/system_rules.json
@@ -0,0 +1,53 @@
+{
+ "default_rule": "default.md",
+ "path_rule_map": {
+ "**/*.properties": "properties.md",
+ "**/*{mapper,dao}*.xml": "mapper_dao_xml.md",
+ "**/pom.xml": "pom_xml.md",
+ "**/build.gradle": "build_gradle.md",
+ "**/package.json": "package_json.md",
+ "**/Cargo.toml": "cargo_toml.md",
+ "**/composer.json": "composer_json.md",
+ "**/*.{json,json5}": "json.md",
+ ".github/workflows/**/*.{yaml,yml}": "github_workflows.md",
+ ".github/**/*.{yaml,yml}": "github_config.md",
+ "**/*.{yaml,yml}": "yaml.md",
+ "**/*.java": "java.md",
+ "**/*.go": "go.md",
+ "**/*.{ftl,ftlh,ftlx}": "freemarker.md",
+ "**/*.{hbs,mustache}": "handlebars_mustache.md",
+ "**/*.pug": "pug.md",
+ "**/*.ets": "arkts.md",
+ "**/*.astro": "astro.md",
+ "**/*.{ts,js,tsx,jsx,mjs,cjs}": "ts_js_tsx_jsx.md",
+ "**/*.{kt}": "kotlin.md",
+ "**/*.rs": "rust.md",
+ "**/*.{cpp,cc,cxx,hpp,hxx}": "cpp.md",
+ "**/*.c": "c.md",
+ "**/*.{py,ipynb}": "python.md",
+ "**/*.{php,phtml}": "php.md",
+ "**/*.proto": "protobuf.md",
+ "**/*.po": "po.md",
+ "**/*.pot": "pot.md",
+ "**/*.{graphql,gql}": "graphql.md",
+ "**/*.prisma": "prisma.md",
+ "**/*.jl": "julia.md",
+ "**/*.R": "r.md",
+ "**/*.{tf,hcl,tfvars}": "terraform.md",
+ "**/*.bicep": "bicep.md",
+ "**/*.nix": "nix.md",
+ "**/*.{hs,lhs}": "haskell.md",
+ "**/*.{nim,nims,nimble}": "nim.md",
+ "**/*.swift": "swift.md",
+ "**/*.elm": "elm.md",
+ "**/*.{jsonnet,libsonnet}": "jsonnet.md",
+ "**/*.zig": "zig.md",
+ "**/*.thrift": "thrift.md",
+ "**/*.capnp": "capnp.md",
+ "**/*.{v,sv,vh}": "verilog.md",
+ "**/*.{vhd,vhdl}": "vhdl.md",
+ "**/*.m": "matlab.md",
+ "**/*.sol": "solidity.md",
+ "**/*.vy": "vyper.md"
+ }
+}
diff --git a/workflows/orca-code-review.yml b/workflows/orca-code-review.yml
index 9938035..6055c30 100644
--- a/workflows/orca-code-review.yml
+++ b/workflows/orca-code-review.yml
@@ -60,9 +60,9 @@ jobs:
# router: "orcarouter/orcacode-review" # router whose DSL picks the model
# fix-first: "P0,P1" # exhaustive only: stop adding passes once one is found
# block-on: "P0" # fail the check (block merge) on these
- # max-diff-kb: "512" # bigger diffs skip the review (outcome: on-oversized-diff)
- # max-diff-files: "300" # same skip when the diff touches more files than this
+ # max-diff-kb: "5000" # bigger diffs skip the review (outcome: on-oversized-diff)
+ # max-diff-files: "2000" # same skip when the diff touches more files than this
# on-oversized-diff: "fail" # oversized skip fails the check; "pass" makes it advisory
# settings: "true" # "false" skips the dashboard fetch — this file is authoritative
# report: "true" # per-run severity counts (never code) to your control plane
- # timeout-minutes: "20" # wall-clock ceiling per engine pass; bump for very large PRs
+ # timeout-minutes: "60" # wall-clock ceiling per engine pass; bump for very large PRs