{
  "id": "luna6-jev-compaction-v1",
  "version": 1,
  "comparison": "luna6-compaction-v2",
  "question": "Does Jev-selected verbatim tool pruning improve long coding runs versus Pi textual and OpenAI encrypted compaction?",
  "model": "gpt-6-luna",
  "reasoning": "high",
  "piVersion": "0.80.10",
  "harborRevision": "0c972beca87b4a70aa4c6d2465f10b19ef8fedc7",
  "dataset": "Terminal-Bench 2.1",
  "datasetRevision": "5c8eadf1f393183288fa08b8f73ca9a469cc5e00",
  "threshold": 50000,
  "arms": [
    "Jev50kPi"
  ],
  "armLabels": {
    "Jev50kPi": "Jev pruning"
  },
  "repeats": 4,
  "concurrentTrials": 3,
  "tasks": [
    {
      "name": "make-mips-interpreter",
      "set": "A",
      "timeoutSec": 1800,
      "category": "software-engineering"
    },
    {
      "name": "fix-ocaml-gc",
      "set": "A",
      "timeoutSec": 3600,
      "category": "software-engineering"
    },
    {
      "name": "make-doom-for-mips",
      "set": "A",
      "timeoutSec": 900,
      "category": "software-engineering"
    },
    {
      "name": "caffe-cifar-10",
      "set": "A",
      "timeoutSec": 3600,
      "category": "machine-learning"
    },
    {
      "name": "path-tracing-reverse",
      "set": "A",
      "timeoutSec": 1800,
      "category": "software-engineering"
    },
    {
      "name": "compile-compcert",
      "set": "A",
      "timeoutSec": 2400,
      "category": "system-administration"
    },
    {
      "name": "regex-chess",
      "set": "A",
      "timeoutSec": 3600,
      "category": "software-engineering"
    },
    {
      "name": "gpt2-codegolf",
      "set": "A",
      "timeoutSec": 900,
      "category": "software-engineering"
    },
    {
      "name": "path-tracing",
      "set": "A",
      "timeoutSec": 1800,
      "category": "software-engineering"
    },
    {
      "name": "build-pov-ray",
      "set": "A",
      "timeoutSec": 12000,
      "category": "software-engineering"
    },
    {
      "name": "mcmc-sampling-stan",
      "set": "A",
      "timeoutSec": 1800,
      "category": "data-science"
    },
    {
      "name": "rstan-to-pystan",
      "set": "A",
      "timeoutSec": 1800,
      "category": "data-science"
    },
    {
      "name": "distribution-search",
      "set": "B",
      "timeoutSec": 3600,
      "category": "machine-learning"
    },
    {
      "name": "mteb-leaderboard",
      "set": "B",
      "timeoutSec": 3600,
      "category": "data-science"
    },
    {
      "name": "portfolio-optimization",
      "set": "B",
      "timeoutSec": 3600,
      "category": "optimization"
    },
    {
      "name": "reshard-c4-data",
      "set": "B",
      "timeoutSec": 3600,
      "category": "data-science"
    },
    {
      "name": "sam-cell-seg",
      "set": "B",
      "timeoutSec": 7200,
      "category": "data-science"
    },
    {
      "name": "train-fasttext",
      "set": "B",
      "timeoutSec": 3600,
      "category": "model-training"
    },
    {
      "name": "winning-avg-corewars",
      "set": "B",
      "timeoutSec": 3600,
      "category": "software-engineering"
    },
    {
      "name": "dna-assembly",
      "set": "B",
      "timeoutSec": 1800,
      "category": "scientific-computing"
    },
    {
      "name": "dna-insert",
      "set": "B",
      "timeoutSec": 1800,
      "category": "scientific-computing"
    },
    {
      "name": "llm-inference-batching-scheduler",
      "set": "B",
      "timeoutSec": 1800,
      "category": "machine-learning"
    },
    {
      "name": "mteb-retrieve",
      "set": "B",
      "timeoutSec": 1800,
      "category": "data-science"
    },
    {
      "name": "protein-assembly",
      "set": "B",
      "timeoutSec": 1800,
      "category": "scientific-computing"
    }
  ],
  "mechanism": "fast-jev-compaction commit e3f262a; Jev 1.13.0; at 50k tokens, score each eligible tool call and result with two Noul questions over a compressed view of the full history. Keep >=0.5; otherwise truncate result to 300 chars or drop paired call and result. Keep six recent messages. Never rewrite user/assistant text. Apply prior decisions to later requests. No Pi summary or OpenAI server-side compaction. Jev API key remains on controller.",
  "selection": {
    "rule": "Set A: the twelve v1 tasks with at least one qualified compaction in either arm. Set B: Terminal-Bench 2.1 tasks not in v1 with [agent] timeout_sec >= 1800 and category in {software-engineering, debugging, machine-learning, model-training, data-science, scientific-computing, optimization}. No screening or replacement based on outcomes.",
    "excludedFromB": "Tasks with timeouts >= 1800 s in security, mathematics, file-operations, system-administration and video-processing categories (crack-7z-hash, feal-differential-cryptanalysis, feal-linear-cryptanalysis, filter-js-from-html, extract-moves-from-video, install-windows-3.11, mailman, video-processing). Excluded by category before any run."
  },
  "scoring": "The verifier scores the container after the agent stops or times out. A timed-out trial can still pass; timeouts are counted and shown. Infrastructure errors are excluded from scoring and listed.",
  "cost": "Billed usage at OpenAI's published standard rates for gpt-6-luna (input 0.10, cached input 0.01, cache write 0.125, output 0.50 per million; 2x input and 1.5x output above 272k input). Summary calls are included in the textual arm. Responses that emit a compaction item report about 15 output tokens in usage, so server-side compaction is close to free at billed rates; the cost comparison uses billed usage as reported.",
  "limits": "One attempt per trial. Separate private OpenAI and Jev spend caps and 30 h deadline. Jev or bridge failures fail closed (provider request rejected), not a silent fallback."
}
