{
  "version": "aime25-jev-reroll-v1",
  "created": "2026-09-21",
  "model": "Qwen/Qwen3-1.7B",
  "jevModel": "jev-1.13.0",
  "dataset": "math-ai/aime25",
  "problems": 30,
  "seeds": [
    2025,
    2026,
    2027
  ],
  "methods": [
    "jev-reroll"
  ],
  "concurrency": 4,
  "n": 8,
  "maxSteps": 32,
  "stepTokens": 384,
  "finalTokens": 2048,
  "baselineTokens": 38912,
  "reasoningSampling": {
    "temperature": 0.6,
    "top_p": 0.95,
    "top_k": 20
  },
  "nonreasoningSampling": {
    "temperature": 0.7,
    "top_p": 0.8,
    "top_k": 20
  },
  "candidateSampling": {
    "temperature": 1,
    "top_p": 0.95,
    "top_k": 20
  },
  "judgeSampling": {
    "temperature": 0,
    "top_p": 1,
    "top_k": -1
  },
  "jevUsdPerMillionInputTokens": 0.042,
  "gpu": "RTX 4090 24 GB",
  "gpuUsdPerHour": 0.49851417478456805,
  "instanceId": 51967955,
  "inference": "vLLM 0.10.2, bfloat16, max model length 40960, prefix caching disabled",
  "methodNotes": "Per-candidate Noul acceptance estimates replace the original relative Choice ranking. Accept the highest-scored candidate at >=0.8; if all fail, sample a fresh batch at the same prefix, up to two rerolls. If all three batches fail, choose the highest-scored candidate across all batches. Stable tie-break: earliest round, then shuffled candidate order. No judge feedback or discarded candidates are passed to Qwen. First-batch generation/shuffle seeds match the original; rerolls use distinct deterministic seed purposes. Accepted-step cap remains 32. Rerolls do not advance that cap. Grading/finalization/generation settings unchanged.",
  "measurement": "Three fixed seeds over all 30 problems (90 trials per method). Process one method and seed at a time with four concurrent problems. Rotate method order by seed. Time each trajectory end-to-end and each method batch separately. Cost = occupied GPU batch wall time at rental rate plus Jev input-token charges, allocated across completed trajectories. Includes GPU waiting during API calls, excludes installation, warmup, and analysis. Throughput cost and loaded latency are not isolated-request latency or token-list-price estimates.",
  "grading": "Last boxed integer in the final answer, normalized to 0..999; missing/invalid answers and output-budget exhaustion without a final answer count wrong. In thinking mode only the content after </think> is graded. Gold answers are loaded only by summarize.ts, not inference or judges.",
  "uncertainty": "Accuracy is mean of 90 binary trial outcomes. 95% percentile intervals and paired differences use 10000 bootstrap resamples of the 30 problem clusters, retaining all three seeds per problem. Report paired differences rather than a claimed win from small point differences. Only four measured settings, not an exhaustive frontier.",
  "limitations": "Follow-up chosen after inspecting the original four-method result. Changes both judging primitive/ranking and reroll policy, so not a pure reroll ablation. Noul values estimate step acceptability, not calibrated eventual-solve probability. More compute allowed. Same GPU model and serving configuration, separate rental/time; physical host and load may differ. Report paired descriptive intervals and actual plus common-rate cost accounting. Original 360 trials remain immutable.",
  "safety": "Six-hour inference limit, seven-hour rental watchdog, $10 estimated inference spend cap (excludes taxes/network/setup). Explicit instance-scoped cleanup and verification, durable results on trusted VM. No gold available to the inference process.",
  "modelRevision": "70d244cc86ccca08cf5af4e1e306ecf908b1ad5e",
  "datasetRevision": "563bb8404243c5f09de6ec262f2db674fe5bce9b",
  "infrastructureNote": "Initial rental 51966897 was destroyed and verified absent after a roughly 12-minute image-pull stall; no AIME inference ran on it. Its setup expense is kept separately. The active follow-up uses a separate Swedish RTX 4090 host from the original study, with otherwise matching serving settings. Source hashes are frozen before AIME evaluation.",
  "threshold": 0.8,
  "maxRerolls": 2,
  "maxInferenceHours": 6,
  "maxEstimatedInferenceUsd": 10,
  "comparisonGpuUsdPerHour": 0.4111111111111111,
  "followup": true,
  "preregisteredAt": "2026-09-21T21:44:34.625Z",
  "machineId": 48093,
  "gpuDriver": "590.48.01",
  "costComparison": "Use original $0.4111111111111111/hour GPU rate for the five-method plot; retain the new actual $0.49851417478456805/hour rental rate in expenses. Both include measured Jev token charges."
}
