{
  "_provenance": "All measured numbers are recorded Qwen3-8B B4 composition-corpus runs from docs/results.jsonl (2026-07-12..15). B4 used 432 composition worlds. Within-row base/tuned comparisons use the same stated harness; cross-row absolutes are not treated as interchangeable. Qwen3-14B remains null because no 14B training/evaluation run is recorded. The current customer-created /sandbox world has executable validation but has not itself been isolated as the causal training corpus for these results.",
  "model": "Qwen3-8B B4 composition adapter · Qwen3-14B not yet run",
  "worlds_evaluated": 432,
  "benchmarks": [
    {
      "name": "τ²-bench retail (Qwen3-8B B4 · paired n=78)",
      "baseline": 38.14,
      "post_trained": 53.42,
      "delta": 15.28,
      "note": "Recorded paired B4 result: 95% CI [+6.8,+23.8]pp, p=0.0004. The broader n=100x4 run is ledgered; the paired join contains 78 tasks."
    },
    {
      "name": "τ²-bench telecom (Qwen3-8B B4 · n=100)",
      "baseline": 17.0,
      "post_trained": 47.0,
      "delta": 30.0,
      "note": "Recorded never-trained-domain result with deterministic ENV_ASSERTION reward, p<1e-8. The tuned arm completed 181/200 planned episodes."
    },
    {
      "name": "τ²-bench airline (Qwen3-8B B4 · n=50)",
      "baseline": 19.3,
      "post_trained": 23.5,
      "delta": 4.2,
      "note": "Recorded paired unsteered result; sign-test p=0.327. This is an honest non-significant boundary, not evidence of airline lift."
    },
    {
      "name": "τ²-bench three-domain average (Qwen3-8B B4)",
      "baseline": 25.8,
      "post_trained": 41.3,
      "delta": 15.5,
      "note": "Derived from the recorded retail, telecom, and airline domain results. The base nearly matches the paper's 26.2, but 41.3 remains well below the paper's 61.8 target."
    },
    {
      "name": "BFCL V4 multi-turn (Qwen3-8B B4 · all 800)",
      "baseline": 10.6,
      "post_trained": 30.5,
      "delta": 19.9,
      "note": "Recorded full multi-turn slice using the official data and checker; tuned-only 192 vs base-only 33, McNemar p<0.0001. This is a slice result, not the paper's full BFCL-V4 aggregate."
    },
    {
      "name": "BFCL V4 AST + abstention (Qwen3-8B B4 · n=696)",
      "baseline": 82.6,
      "post_trained": 85.9,
      "delta": 3.3,
      "note": "Recorded same-session comparison with the corrected decoder and identical vLLM harness for both arms."
    },
    {
      "name": "MCP-Mark filesystem + postgres (Qwen3-8B · n=204)",
      "baseline": 3.9,
      "post_trained": 2.9,
      "delta": -1.0,
      "note": "Recorded diagnostic adapter result: postgres was exactly flat (6.0%→6.0%); the combined one-point decrease is floor-level noise. No MCP-Mark improvement was demonstrated."
    },
    {
      "name": "Qwen3-14B composite-world benchmark battery",
      "baseline": null,
      "post_trained": null,
      "delta": null,
      "note": "Not run. Configuration and benchmark locks exist, but no recorded 14B checkpoint or evaluation result exists in this repository."
    }
  ]
}
