Skip to content

feat(browser): server-side traced, /server and Next.js entries #2817

feat(browser): server-side traced, /server and Next.js entries

feat(browser): server-side traced, /server and Next.js entries #2817

Workflow file for this run

name: MCP Evals
# LLM tool-selection + full-execution evals for the MCP server.
#
# These drive a REAL model through OpenRouter — every run costs money and is
# nondeterministic — so the job is OPT-IN. It only runs when the PR carries the
# `run-evals` label (the "badge"), or via manual dispatch:
#
# • Add the `run-evals` label to a PR → evals run on that event and on every
# subsequent push while the label stays on.
# • Remove the label → later pushes skip the job (free).
# • Actions ▸ MCP Evals ▸ Run workflow → run on demand against any branch.
#
# Only the model-driven `*.eval.ts` files are gated here. The deterministic
# renderer regression tests (`apps/api/src/mcp/__evals__/*.test.ts`) need no
# model and run for free in the normal `test` CI job. Without OPENROUTER_API_KEY
# the eval suite self-skips (green), so this job is safe before the secret is set.
on:
workflow_dispatch:
# No `paths` filter: the `run-evals` label is the gate (see the job `if:`),
# so the workflow can be triggered on any PR by adding the badge.
pull_request:
branches: [main]
types: [opened, synchronize, reopened, labeled]
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}
cancel-in-progress: true
permissions:
contents: read
jobs:
eval:
name: MCP Evals
# Opt-in: only spend OpenRouter budget when the `run-evals` badge is on
# the PR (or on manual dispatch). Otherwise the job is skipped, not failed.
if: >-
github.event_name == 'workflow_dispatch' ||
contains(github.event.pull_request.labels.*.name, 'run-evals')
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v6
- uses: jdx/mise-action@c2a87611a18de5b3828c5652fe268e992400cb5c # v4.3.0
- uses: ./.github/actions/bun-install
with:
filters: "@maple/ai"
- name: Run MCP evals
run: bun run --filter @maple/ai eval
env:
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
# Override to eval a different model (defaults to the prod kimi-k2.5).
MCP_EVAL_MODEL: ${{ vars.MCP_EVAL_MODEL }}