{
  "version": 3,
  "run_id": "jev-20260920-50x3-metered",
  "repeats": 3,
  "concurrency": 4,
  "seed": 20260919,
  "stop_after_charged_cents": 1200,
  "sample": "50 synthetic prompts, 10 per bucket; exactly the same prompts in both arms and all repeats",
  "control": "One Claude Opus 4.6 call; no router and no retry",
  "treatment": "One Jev request containing five typed questions, deterministic policy, one answer call and at most one higher-lane fallback",
  "lanes": {
    "fast": "google/gemini-2.5-flash-lite",
    "balanced": "openai/gpt-4.1-mini",
    "frontier": "anthropic/claude-opus-4.6"
  },
  "router": "typesafe/jev-1.13 via Vaaya openrouter/decisions; provider may return the versioned model identifier",
  "confidence_threshold": 0.8,
  "boolean_probability_threshold": 0.5,
  "confidence_rule": "Minimum of route and complexity choice confidence; use selected-choice probability only when confidence is absent; missing confidence is low confidence",
  "policy": "Apply eligibility floors: deep requires frontier; multi-step OR needs_tools OR needs_verification requires at least balanced. High stakes OR low confidence then raises the eligible lane by one tier, capped at frontier. Invalid router response fails closed to frontier.",
  "fallback": "Only failed provider/transport responses, truncated output, or unparseable/non-object JSON. One higher tier only, no fallback above frontier. Never use grader feedback to trigger a fallback.",
  "grader": "Frozen deterministic checks in prompts.json: exact expected fields for 40 prompts, required strings, forbidden strings, and word limits for 10 rewriting prompts. All checks must pass. Formatting fences are stripped. No LLM judge.",
  "quality_tolerance_percentage_points": 5,
  "success_criterion": "Treatment must have lower total recorded Vaaya customer usage price AND its pooled frozen-check pass rate must be no more than 5 percentage points below control. Descriptive acceptance rule, not statistical non-inferiority.",
  "spend": "Sum actual per-call customer prices returned by Vaaya: Jev data.vaaya_billing.price_microusd; non-streaming LLM API usage.cost converted to micro-USD (already includes Vaaya fee). Count each call once, including fallbacks and charged failures. Do not add a markup again or count aggregate settlement twice. No reconstructed prices.",
  "cost_per_passing_answer": "Total recorded Vaaya customer usage price for an arm divided by passing final answers; undefined if none pass.",
  "latency": "Client wall time from the start of an arm through its final answer, including router and fallback. Nearest-rank p95. Local queue wait and grading excluded. Four concurrent paired-prompt workers; arms within a pair run sequentially in seeded random order.",
  "escalations": "Report deterministic policy promotion above Jev recommendation separately from a second answer-model call. Three repeats reuse 50 prompts; 150 responses per arm are not 150 independent tasks.",
  "scope": "Closed-context text tasks and tool plans. No tools, money movement, access changes or legal commitments execute. Passing deterministic rewriting checks does not establish prose quality, and classification does not replace tool permissions or human approval.",
  "comparison_scope": "Both answer arms call the same Vaaya non-streaming /api/llm/v1/chat/completions endpoint, with identical model IDs, messages, max_tokens and temperature. The endpoint returns customer usage.cost, already including 3%. Jev calls /api/run/openrouter/decisions.",
  "ledger_verification": "Jev: authenticated transactions API after each repeat. Answer calls: read-only LLM usage export matched by a unique User-Agent call tag, model, token counts and returned customer price. An operator export verifies the billing rows without editing them. No account identifier or credentials are published.",
  "interruption": "A prior catalog-route attempt was interrupted by a payment-rail outage. Its artifacts are archived separately and excluded. This new run uses a new run ID and freezes the endpoint change before calls."
}
