Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -397,6 +397,7 @@ Every community PR that lands in main is credited here — that's a project rule
- [@davekopecek](https://github.com/davekopecek) (Dave Kopecek) — committed the design-reference fixture so the design-token guard runs on every machine (#30)
- [@snapsynapse](https://github.com/snapsynapse) (Sam Rogers) — graceful shutdown on SIGINT/SIGTERM with worker-tree cleanup and finished state, plus the 14-test end-to-end CLI regression suite (#4)
- [@mlava](https://github.com/mlava) (Mark Lavercombe) — named setup failures across every diagnostic surface (#37), `run --baseline`, the no-workers check preflight (#38), guidance on check-writing failure modes (#57), early warnings for missing worker commands (#59), and preserving fix-swarm patches across retries (#56)
- [@hcoronel1](https://github.com/hcoronel1) (Hernán Coronel) — added Claude Code CLI as a ringer engine, with a `claude-code-json` result parser that keeps API/infra failures out of the scoreboard without hiding real agent failures

Contributions are welcome — see [CONTRIBUTING.md](CONTRIBUTING.md) for the philosophy and what gets a PR merged fast. The short version: small and scoped, rebased on current main, every claim backed by an executed test. Authorship is always preserved — where a maintainer pushes a mechanical fix to your branch, you remain the commit author.

Expand Down
13 changes: 13 additions & 0 deletions config.sample.toml
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,19 @@ token_regex = "tokens\\s+used\\s*:?\\s*([0-9][0-9,]*)"
# default reads the `model:` header; override it here only if that format changes.
model_report_regex = "(?m)^model:[ \\t]*([^ \\t\\r\\n]+)[ \\t]*\\r?$"

# Claude Code CLI (opt in). Claude Code permissions are not an OS filesystem
# sandbox; use an OS-level wrapper when filesystem containment is required.
# [engines.claude]
# bin = "claude"
# model_default = "claude-sonnet-5"
# args_template = [
# "-p", "--model", "{model}", "--output-format", "json",
# "{access_args}", "{engine_args}", "{spec}",
# ]
# sandbox_args = ["--permission-mode", "acceptEdits"]
# full_access_args = ["--dangerously-skip-permissions"]
# result_parser = "claude-code-json"

# Grok Build CLI (xAI). Install: curl -fsSL https://x.ai/cli/install.sh | bash
# (or: npm install -g @xai-official/grok), then `grok login` — OAuth on a
# SuperGrok or X Premium Plus plan. Headless mode (-p) runs a full agentic
Expand Down
62 changes: 62 additions & 0 deletions registry/model-identity.toml
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,68 @@ access = "OpenRouter API"
# names. Unlisted slugs are marked unregistered and derive a display name;
# their complete raw value appears only in scoreboard diagnostics until verified.

[engines.claude]
harness = "Claude Code CLI"
access = "Claude subscription"
default_model_key = "claude-sonnet-5"

[engines.claude.models."claude-sonnet-5"]
display = "Claude Sonnet 5"
lab = "Anthropic"
confidence = "verified"
source = "https://docs.anthropic.com/en/docs/about-claude/models/overview"
last_verified = 2026-09-16

[engines.claude.models."sonnet"]
display = "Claude Sonnet 5 (alias)"
lab = "Anthropic"
alias = true
confidence = "verified"
source = "https://docs.anthropic.com/en/docs/about-claude/models/overview"
last_verified = 2026-09-16

[engines.claude.models."claude-opus-5"]
display = "Claude Opus 5"
lab = "Anthropic"
confidence = "verified"
source = "https://docs.anthropic.com/en/docs/about-claude/models/overview"
last_verified = 2026-09-16

[engines.claude.models."claude-fable-5-1"]
display = "Claude Fable 5.1"
lab = "Anthropic"
confidence = "verified"
source = "https://docs.anthropic.com/en/docs/about-claude/models/overview"
last_verified = 2026-09-16

[engines.claude.models."claude-haiku-4-5"]
display = "Claude Haiku 4.5"
lab = "Anthropic"
confidence = "verified"
source = "https://docs.anthropic.com/en/docs/about-claude/models/overview"
last_verified = 2026-09-16

[engines.claude-sonnet]
# Legacy engine name retained for historical rows; the harness is unchanged.
harness = "Claude Code CLI"
access = "Claude subscription"
default_model_key = "claude-sonnet-5"

[engines.claude-sonnet.models."claude-sonnet-5"]
display = "Claude Sonnet 5"
lab = "Anthropic"
confidence = "verified"
source = "https://docs.anthropic.com/en/docs/about-claude/models/overview"
last_verified = 2026-09-16

[engines.claude-sonnet.models."sonnet"]
display = "Claude Sonnet 5 (alias)"
lab = "Anthropic"
alias = true
confidence = "verified"
source = "https://docs.anthropic.com/en/docs/about-claude/models/overview"
last_verified = 2026-09-16

[engines.opencode.models."openrouter/z-ai/glm-5.2"]
display = "GLM 5.2"
lab = "Z.ai (Zhipu AI)"
Expand Down
Loading