{
  "id": "luna6-compaction-v2",
  "version": 2,
  "previous": "https://benchmarks.rubric.sh/archive/luna-compaction-v1/",
  "question": "Does OpenAI's encrypted server-side compaction beat Pi's textual summary compaction on long coding tasks, holding model, agent, threshold and tasks fixed?",
  "model": "gpt-6-luna",
  "reasoning": "high",
  "piVersion": "0.80.10",
  "harborRevision": "0c972beca87b4a70aa4c6d2465f10b19ef8fedc7",
  "dataset": "Terminal-Bench 2.1",
  "datasetRepository": "https://github.com/harbor-framework/terminal-bench-2-1",
  "datasetRevision": "5c8eadf1f393183288fa08b8f73ca9a469cc5e00",
  "threshold": 50000,
  "arms": ["TextualFixed50kPi", "Native50kPi"],
  "armLabels": {"TextualFixed50kPi": "Pi textual", "Native50kPi": "OpenAI encrypted"},
  "repeats": 4,
  "concurrentPairs": 3,
  "scheduling": "Round-major: every task runs once per arm (one pair) before any task gets its second repeat. Both arms of a pair start together in the same Harbor job so they share machine load. Task order reverses on odd rounds so a budget or deadline stop leaves repeats spread evenly. Three pairs run concurrently.",
  "runtime": "Bun 1.4.2 with Pi 0.80.10 in fresh containers. Textual arm: Pi summary compaction patched to trigger above 50k context tokens, keepRecentTokens 20000, reserveTokens 16384, resume message 'Continue.'. Encrypted arm: Pi compaction disabled, context_management compact_threshold 50000, items before the latest compaction item dropped except the leading developer message. Pi retries disabled. Model context 272k, max output 128k. Task and summary requests stream through the same metered gateway.",
  "changesFromV1": [
    "Four repeats per task per arm instead of one, so within-task variance is estimated instead of assumed away.",
    "Tasks selected for long context: v1 tasks that compacted in at least one arm, plus unused Terminal-Bench 2.1 tasks with agent timeouts of 30 minutes or more in implementation, debugging, ML, data-science, scientific-computing and optimization categories. Selection uses v1 compaction counts and dataset metadata only, never pass/fail.",
    "Primary analysis is restricted to tasks where compaction actually happens (task compaction rate at least 0.5 across both arms). Tasks that never compact are reported as a placebo set, where the arms are identical configurations and any difference is noise.",
    "Encrypted arm keeps Pi's developer message after compaction. In v1 it was dropped with the other pre-compaction items, so that arm lost its system prompt.",
    "Textual arm resumes after compaction with 'Continue.' instead of an instruction to finish and verify the task.",
    "Mechanistic measures per arm: compactions per trial, context tokens at compaction, residual input tokens on the first request after compaction, requests, duration, timeouts, and file re-reads around compaction.",
    "Reporting: discordant pair counts and the interval lead the summary. Timed-out trials are still verified and labelled. Static HTML carries the current numbers for readers without JavaScript."
  ],
  "selection": {
    "rule": "Set A: the twelve v1 tasks with at least one qualified compaction in either arm. Set B: Terminal-Bench 2.1 tasks not in v1 with [agent] timeout_sec >= 1800 and category in {software-engineering, debugging, machine-learning, model-training, data-science, scientific-computing, optimization}. No screening or replacement based on outcomes.",
    "excludedFromB": "Tasks with timeouts >= 1800 s in security, mathematics, file-operations, system-administration and video-processing categories (crack-7z-hash, feal-differential-cryptanalysis, feal-linear-cryptanalysis, filter-js-from-html, extract-moves-from-video, install-windows-3.11, mailman, video-processing). Excluded by category before any run."
  },
  "tasks": [
    {"name": "make-mips-interpreter", "set": "A", "timeoutSec": 1800, "category": "software-engineering"},
    {"name": "fix-ocaml-gc", "set": "A", "timeoutSec": 3600, "category": "software-engineering"},
    {"name": "make-doom-for-mips", "set": "A", "timeoutSec": 900, "category": "software-engineering"},
    {"name": "caffe-cifar-10", "set": "A", "timeoutSec": 3600, "category": "machine-learning"},
    {"name": "path-tracing-reverse", "set": "A", "timeoutSec": 1800, "category": "software-engineering"},
    {"name": "compile-compcert", "set": "A", "timeoutSec": 2400, "category": "system-administration"},
    {"name": "regex-chess", "set": "A", "timeoutSec": 3600, "category": "software-engineering"},
    {"name": "gpt2-codegolf", "set": "A", "timeoutSec": 900, "category": "software-engineering"},
    {"name": "path-tracing", "set": "A", "timeoutSec": 1800, "category": "software-engineering"},
    {"name": "build-pov-ray", "set": "A", "timeoutSec": 12000, "category": "software-engineering"},
    {"name": "mcmc-sampling-stan", "set": "A", "timeoutSec": 1800, "category": "data-science"},
    {"name": "rstan-to-pystan", "set": "A", "timeoutSec": 1800, "category": "data-science"},
    {"name": "distribution-search", "set": "B", "timeoutSec": 3600, "category": "machine-learning"},
    {"name": "mteb-leaderboard", "set": "B", "timeoutSec": 3600, "category": "data-science"},
    {"name": "portfolio-optimization", "set": "B", "timeoutSec": 3600, "category": "optimization"},
    {"name": "reshard-c4-data", "set": "B", "timeoutSec": 3600, "category": "data-science"},
    {"name": "sam-cell-seg", "set": "B", "timeoutSec": 7200, "category": "data-science"},
    {"name": "train-fasttext", "set": "B", "timeoutSec": 3600, "category": "model-training"},
    {"name": "winning-avg-corewars", "set": "B", "timeoutSec": 3600, "category": "software-engineering"},
    {"name": "dna-assembly", "set": "B", "timeoutSec": 1800, "category": "scientific-computing"},
    {"name": "dna-insert", "set": "B", "timeoutSec": 1800, "category": "scientific-computing"},
    {"name": "llm-inference-batching-scheduler", "set": "B", "timeoutSec": 1800, "category": "machine-learning"},
    {"name": "mteb-retrieve", "set": "B", "timeoutSec": 1800, "category": "data-science"},
    {"name": "protein-assembly", "set": "B", "timeoutSec": 1800, "category": "scientific-computing"}
  ],
  "qualification": "A compaction counts only if a later agent request follows it. Task compaction rate = scored trials (both arms pooled) with at least one qualified compaction, divided by scored trials.",
  "analysis": {
    "unit": "Task. Each task contributes one pass rate per arm (passes over scored trials). Effect = mean over tasks of encrypted minus textual pass rate.",
    "primary": "Tasks with compaction rate >= 0.5.",
    "secondary": "All tasks.",
    "placebo": "Tasks with compaction rate 0. Arms are identical configurations there; the estimate should be near zero.",
    "uncertainty": "95% percentile interval from a task-cluster bootstrap: resample tasks with replacement keeping all their trials, 20000 resamples, LCG seed 0x614a. Trials of one task are never treated as independent tasks.",
    "discordance": "Trial pairs (same task, same round, started together) are cross-tabulated: both pass, both fail, encrypted only, textual only. A two-sided exact sign test on discordant pairs is reported as descriptive; pairs within a task are not independent.",
    "mechanics": "Per arm: mean compactions per trial, median context tokens at compaction, median residual input tokens on the first request after a compaction, mean requests, mean cost, median duration, timed-out trials, and mean file-read tool calls in the five agent requests before versus after each compaction (exploratory).",
    "scoring": "The verifier scores the container after the agent stops or times out. A timed-out trial can still pass; timeouts are counted and shown. Infrastructure errors are excluded from scoring and listed."
  },
  "cost": "Billed usage at OpenAI's published standard rates for gpt-6-luna (input 0.10, cached input 0.01, cache write 0.125, output 0.50 per million; 2x input and 1.5x output above 272k input). Summary calls are included in the textual arm. Responses that emit a compaction item report about 15 output tokens in usage, so server-side compaction is close to free at billed rates; the cost comparison uses billed usage as reported.",
  "pricingSource": "https://developers.openai.com/api/docs/models/gpt-6-luna",
  "modelPin": "Provider exposes gpt-6-luna without a dated snapshot; the actual model ID is recorded on every response and audited.",
  "limits": "Original dataset timeouts and resource limits. A fixed private API spend cap stops new requests; the runner stops starting pairs when the cap or a 30 hour deadline is near. Transport errors are retained, never silently rerun. One attempt per trial; no target-score-dependent retries."
}
