{
  "version": "aime25-step-search-v1",
  "created": "2026-09-21",
  "model": "Qwen/Qwen3-1.7B",
  "jevModel": "jev-1.13.0",
  "dataset": "math-ai/aime25",
  "problems": 30,
  "seeds": [
    2025,
    2026,
    2027
  ],
  "methods": [
    "nonreasoning",
    "reasoning",
    "self",
    "jev"
  ],
  "concurrency": 4,
  "n": 8,
  "maxSteps": 32,
  "stepTokens": 384,
  "finalTokens": 2048,
  "baselineTokens": 38912,
  "reasoningSampling": {
    "temperature": 0.6,
    "top_p": 0.95,
    "top_k": 20
  },
  "nonreasoningSampling": {
    "temperature": 0.7,
    "top_p": 0.8,
    "top_k": 20
  },
  "candidateSampling": {
    "temperature": 1,
    "top_p": 0.95,
    "top_k": 20
  },
  "judgeSampling": {
    "temperature": 0,
    "top_p": 1,
    "top_k": -1
  },
  "jevUsdPerMillionInputTokens": 0.042,
  "gpu": "RTX 4090 24 GB",
  "gpuUsdPerHour": 0.4111111111111111,
  "instanceId": 51930498,
  "inference": "vLLM 0.10.2, bfloat16, max model length 40960, prefix caching disabled",
  "methodNotes": "Both search methods sample eight non-thinking next-step continuations, select exactly one, and append it to the retained path. No lookahead, answer voting, oracle grading, or execution tools. Same generator, rubric, ordering, budgets, and seeds. Qwen self-judge uses non-thinking constrained categorical output. Jev uses a single Choice over the same candidates. Paths diverge after judge choices. A selected final boxed integer ends search. At step limit, one non-thinking finalization call is used, without judging or resampling.",
  "measurement": "Three fixed seeds over all 30 problems (90 trials per method). Process one method and seed at a time with four concurrent problems. Rotate method order by seed. Time each trajectory end-to-end and each method batch separately. Cost = occupied GPU batch wall time at rental rate plus Jev input-token charges, allocated across completed trajectories. Includes GPU waiting during API calls, excludes installation, warmup, and analysis. Throughput cost and loaded latency are not isolated-request latency or token-list-price estimates.",
  "grading": "Last boxed integer in the final answer, normalized to 0..999; missing/invalid answers and output-budget exhaustion without a final answer count wrong. In thinking mode only the content after </think> is graded. Gold answers are loaded only by summarize.ts, not inference or judges.",
  "uncertainty": "Accuracy is mean of 90 binary trial outcomes. 95% percentile intervals and paired differences use 10000 bootstrap resamples of the 30 problem clusters, retaining all three seeds per problem. Report paired differences rather than a claimed win from small point differences. Only four measured settings, not an exhaustive frontier.",
  "limitations": "Small public benchmark, possible training contamination, three stochastic repeats, one GPU and serving configuration. Search and baseline compute budgets differ; both judges' search budgets match. N=8 is a practical power-of-two search width, not a universal research standard. No accuracy-driven prompt tuning on AIME 2025. Synthetic arithmetic smoke tests only. Never claim expected Jev improvement unless measured.",
  "safety": "Six-hour maximum rental lifetime. Stop and preserve partial data on repeated API errors. Destroy only instance 51930498 after data is synced. No Brex purchase needed unless an account funding problem blocks execution.",
  "modelRevision": "70d244cc86ccca08cf5af4e1e306ecf908b1ad5e",
  "datasetRevision": "563bb8404243c5f09de6ec262f2db674fe5bce9b",
  "infrastructureNote": "The first non-thinking batch aborted after 29 completions when an HTTP client timed out at 300 seconds on a long response. That entire unscored incomplete batch is archived separately, excluded from primary metrics, and rerun with the same seeds and prompts after changing transport to SSE streaming. No AIME accuracy was computed before this transport-only fix. Old rental 51928761 was destroyed. Aborted/setup spend is reported separately."
}
