{
  "id": "luna6-compaction-20-v1",
  "model": "gpt-6-luna",
  "reasoning": "high",
  "piVersion": "0.80.10",
  "harborRevision": "0c972beca87b4a70aa4c6d2465f10b19ef8fedc7",
  "dataset": "Terminal-Bench 2.1",
  "datasetRepository": "https://github.com/harbor-framework/terminal-bench-2-1",
  "datasetRevision": "5c8eadf1f393183288fa08b8f73ca9a469cc5e00",
  "threshold": 50000,
  "arms": ["TextualFixed50kPi", "Native50kPi"],
  "repeats": 1,
  "concurrency": 4,
  "scheduling": "Two task pairs at a time in listed task order; reverse arm configuration order on alternating tasks. Each pair is a separate Harbor job.",
  "runtime": "Bun 1.4.2 with Pi 0.80.10; original 50k compaction extensions and adapter patches. Native disables textual fallback; textual keeps Pi's default 20k recent tokens. Pi retries disabled. Model context 272k, max output 128k. Task and summary requests stream through the same metered gateway.",
  "uncertainty": "95% paired task bootstrap, 20000 resamples, fixed seed 0x614a. No resampling across individual requests. Report at completion.",
  "modelPin": "Provider exposes gpt-6-luna without a dated snapshot; record actual model ID on every response.",
  "budgetUsd": 40,
  "apiBudgetUsd": 38,
  "pricesPerMillion": {"input": 0.1, "cacheRead": 0.01, "cacheWrite": 0.125, "output": 0.5},
  "pricingSource": "https://developers.openai.com/api/docs/models/gpt-6-luna",
  "pricingNotes": "Standard rates. Above 272k input, double input/cache rates and multiply output by 1.5. Existing trusted VM, no added rental cost.",
  "selection": "Keep the original seven tasks; add thirteen multi-stage implementation, debugging, scientific-computing and build tasks, selected from instructions and resource metadata before GPT-6 scores. No screening or replacement based on outcomes.",
  "qualification": "Compaction must be followed by a later agent provider request. Report observed and qualified counts, including tasks that never compact.",
  "comparison": "Task-paired pass/fail, costs, requests and input+output tokens (cached input counted once). Resource means and topline paired scores use only tasks with both arms scored. Errors remain separate. One attempt per arm/task; no target-score-dependent retries.",
  "limits": "Use original dataset timeouts and resource limits; 12h overall deadline. Reserve worst-case request cost before dispatch; stop new API calls at $38. Transport errors are retained, not silently rerun.",
  "tasks": [
    "sanitize-git-repo", "make-mips-interpreter", "fix-ocaml-gc", "make-doom-for-mips", "caffe-cifar-10", "path-tracing-reverse", "compile-compcert",
    "circuit-fibsqrt", "regex-chess", "schemelike-metacircular-eval", "write-compressor", "gpt2-codegolf", "path-tracing", "custom-memory-heap-crash", "build-pov-ray", "bn-fit-modify", "mcmc-sampling-stan", "rstan-to-pystan", "torch-pipeline-parallelism", "torch-tensor-parallelism"
  ]
}
