{
  "complete": true,
  "live": false,
  "matchedTrials": 90,
  "generatedAt": "2026-09-21T20:38:16.321Z",
  "expected": 360,
  "recorded": 360,
  "problems": 30,
  "seeds": [
    2025,
    2026,
    2027
  ],
  "config": {
    "version": "aime25-step-search-v1",
    "created": "2026-09-21",
    "model": "Qwen/Qwen3-1.7B",
    "jevModel": "jev-1.13.0",
    "dataset": "math-ai/aime25",
    "problems": 30,
    "seeds": [
      2025,
      2026,
      2027
    ],
    "methods": [
      "nonreasoning",
      "reasoning",
      "self",
      "jev"
    ],
    "concurrency": 4,
    "n": 8,
    "maxSteps": 32,
    "stepTokens": 384,
    "finalTokens": 2048,
    "baselineTokens": 38912,
    "reasoningSampling": {
      "temperature": 0.6,
      "top_p": 0.95,
      "top_k": 20
    },
    "nonreasoningSampling": {
      "temperature": 0.7,
      "top_p": 0.8,
      "top_k": 20
    },
    "candidateSampling": {
      "temperature": 1,
      "top_p": 0.95,
      "top_k": 20
    },
    "judgeSampling": {
      "temperature": 0,
      "top_p": 1,
      "top_k": -1
    },
    "jevUsdPerMillionInputTokens": 0.042,
    "gpu": "RTX 4090 24 GB",
    "gpuUsdPerHour": 0.4111111111111111,
    "instanceId": 51930498,
    "inference": "vLLM 0.10.2, bfloat16, max model length 40960, prefix caching disabled",
    "methodNotes": "Both search methods sample eight non-thinking next-step continuations, select exactly one, and append it to the retained path. No lookahead, answer voting, oracle grading, or execution tools. Same generator, rubric, ordering, budgets, and seeds. Qwen self-judge uses non-thinking constrained categorical output. Jev uses a single Choice over the same candidates. Paths diverge after judge choices. A selected final boxed integer ends search. At step limit, one non-thinking finalization call is used, without judging or resampling.",
    "measurement": "Three fixed seeds over all 30 problems (90 trials per method). Process one method and seed at a time with four concurrent problems. Rotate method order by seed. Time each trajectory end-to-end and each method batch separately. Cost = occupied GPU batch wall time at rental rate plus Jev input-token charges, allocated across completed trajectories. Includes GPU waiting during API calls, excludes installation, warmup, and analysis. Throughput cost and loaded latency are not isolated-request latency or token-list-price estimates.",
    "grading": "Last boxed integer in the final answer, normalized to 0..999; missing/invalid answers and output-budget exhaustion without a final answer count wrong. In thinking mode only the content after </think> is graded. Gold answers are loaded only by summarize.ts, not inference or judges.",
    "uncertainty": "Accuracy is mean of 90 binary trial outcomes. 95% percentile intervals and paired differences use 10000 bootstrap resamples of the 30 problem clusters, retaining all three seeds per problem. Report paired differences rather than a claimed win from small point differences. Only four measured settings, not an exhaustive frontier.",
    "limitations": "Small public benchmark, possible training contamination, three stochastic repeats, one GPU and serving configuration. Search and baseline compute budgets differ; both judges' search budgets match. N=8 is a practical power-of-two search width, not a universal research standard. No accuracy-driven prompt tuning on AIME 2025. Synthetic arithmetic smoke tests only. Never claim expected Jev improvement unless measured.",
    "safety": "Six-hour maximum rental lifetime. Stop and preserve partial data on repeated API errors. Destroy only instance 51930498 after data is synced. No Brex purchase needed unless an account funding problem blocks execution.",
    "modelRevision": "70d244cc86ccca08cf5af4e1e306ecf908b1ad5e",
    "datasetRevision": "563bb8404243c5f09de6ec262f2db674fe5bce9b",
    "infrastructureNote": "The first non-thinking batch aborted after 29 completions when an HTTP client timed out at 300 seconds on a long response. That entire unscored incomplete batch is archived separately, excluded from primary metrics, and rerun with the same seeds and prompts after changing transport to SSE streaming. No AIME accuracy was computed before this transport-only fix. Old rental 51928761 was destroyed. Aborted/setup spend is reported separately."
  },
  "methods": [
    {
      "method": "reasoning",
      "label": "Reasoning",
      "count": 90,
      "correct": 33,
      "accuracy": 0.36666666666666664,
      "ci": [
        0.23333333333333334,
        0.5111111111111111
      ],
      "perSeed": [
        {
          "seed": 2025,
          "count": 30,
          "correct": 10
        },
        {
          "seed": 2026,
          "count": 30,
          "correct": 11
        },
        {
          "seed": 2027,
          "count": 30,
          "correct": 12
        }
      ],
      "meanLatencySeconds": 211.63247825644441,
      "medianLatencySeconds": 193.52464301850029,
      "p95LatencySeconds": 383.7443843956504,
      "gpuSeconds": 5007.3365149149995,
      "activeGpuSeconds": 0,
      "costProvisional": false,
      "gpuCost": 0.5718254662094289,
      "jevCost": 0,
      "totalCost": 0.5718254662094289,
      "costPerProblem": 0.006353616291215876,
      "qwenInputTokens": 17796,
      "qwenOutputTokens": 1613802,
      "jevInputTokens": 0,
      "jevOutputTokens": 0,
      "meanSteps": 0,
      "meanJevSeconds": 0,
      "missingAnswers": 8,
      "outputLimitRuns": 1,
      "stepLimitRuns": 0,
      "truncatedCalls": 1,
      "retries": 0
    },
    {
      "method": "nonreasoning",
      "label": "Non-reasoning",
      "count": 90,
      "correct": 9,
      "accuracy": 0.1,
      "ci": [
        0.03333333333333333,
        0.18888888888888888
      ],
      "perSeed": [
        {
          "seed": 2025,
          "count": 30,
          "correct": 3
        },
        {
          "seed": 2026,
          "count": 30,
          "correct": 3
        },
        {
          "seed": 2027,
          "count": 30,
          "correct": 3
        }
      ],
      "meanLatencySeconds": 24.29410478697774,
      "medianLatencySeconds": 14.61786413099896,
      "p95LatencySeconds": 44.737310422550976,
      "gpuSeconds": 927.6211834500027,
      "activeGpuSeconds": 0,
      "costProvisional": false,
      "gpuCost": 0.10593204872731511,
      "jevCost": 0,
      "totalCost": 0.10593204872731511,
      "costPerProblem": 0.0011770227636368345,
      "qwenInputTokens": 18156,
      "qwenOutputTokens": 285690,
      "jevInputTokens": 0,
      "jevOutputTokens": 0,
      "meanSteps": 0,
      "meanJevSeconds": 0,
      "missingAnswers": 17,
      "outputLimitRuns": 2,
      "stepLimitRuns": 0,
      "truncatedCalls": 2,
      "retries": 0
    },
    {
      "method": "self",
      "label": "Best-of-8 · Qwen judge",
      "count": 90,
      "correct": 6,
      "accuracy": 0.06666666666666667,
      "ci": [
        0,
        0.15555555555555556
      ],
      "perSeed": [
        {
          "seed": 2025,
          "count": 30,
          "correct": 1
        },
        {
          "seed": 2026,
          "count": 30,
          "correct": 2
        },
        {
          "seed": 2027,
          "count": 30,
          "correct": 3
        }
      ],
      "meanLatencySeconds": 156.84938332304446,
      "medianLatencySeconds": 62.90670913749956,
      "p95LatencySeconds": 573.849406564,
      "gpuSeconds": 3854.200034789999,
      "activeGpuSeconds": 0,
      "costProvisional": false,
      "gpuCost": 0.4401401274297221,
      "jevCost": 0,
      "totalCost": 0.4401401274297221,
      "costPerProblem": 0.004890445860330246,
      "qwenInputTokens": 11542314,
      "qwenOutputTokens": 2748176,
      "jevInputTokens": 0,
      "jevOutputTokens": 0,
      "meanSteps": 10.877777777777778,
      "meanJevSeconds": 0,
      "missingAnswers": 12,
      "outputLimitRuns": 1,
      "stepLimitRuns": 17,
      "truncatedCalls": 831,
      "retries": 0
    },
    {
      "method": "jev",
      "label": "Best-of-8 · Jev judge",
      "count": 90,
      "correct": 10,
      "accuracy": 0.1111111111111111,
      "ci": [
        0.03333333333333333,
        0.2
      ],
      "perSeed": [
        {
          "seed": 2025,
          "count": 30,
          "correct": 2
        },
        {
          "seed": 2026,
          "count": 30,
          "correct": 5
        },
        {
          "seed": 2027,
          "count": 30,
          "correct": 3
        }
      ],
      "meanLatencySeconds": 247.52522047419995,
      "medianLatencySeconds": 79.23904134349945,
      "p95LatencySeconds": 811.6240026325506,
      "gpuSeconds": 5929.522450233,
      "activeGpuSeconds": 0,
      "costProvisional": false,
      "gpuCost": 0.6771368230204352,
      "jevCost": 0.46097931600000003,
      "totalCost": 1.1381161390204353,
      "costPerProblem": 0.012645734878004836,
      "qwenInputTokens": 6632620,
      "qwenOutputTokens": 3763246,
      "jevInputTokens": 10975698,
      "jevOutputTokens": 96287,
      "meanSteps": 14.655555555555555,
      "meanJevSeconds": 6.065001201155515,
      "missingAnswers": 24,
      "outputLimitRuns": 3,
      "stepLimitRuns": 26,
      "truncatedCalls": 1189,
      "retries": 0
    }
  ],
  "pairs": [
    {
      "other": "reasoning",
      "delta": -0.25555555555555554,
      "ci": [
        -0.35555555555555557,
        -0.15555555555555556
      ],
      "conclusion": "Jev lower in this experiment"
    },
    {
      "other": "self",
      "delta": 0.044444444444444446,
      "ci": [
        -0.02249999999999975,
        0.13333333333333333
      ],
      "conclusion": "Difference unresolved"
    },
    {
      "other": "nonreasoning",
      "delta": 0.011111111111111112,
      "ci": [
        -0.06666666666666667,
        0.08888888888888889
      ],
      "conclusion": "Difference unresolved"
    }
  ],
  "costFrontier": [
    "nonreasoning",
    "reasoning"
  ],
  "runtimeFrontier": [
    "nonreasoning",
    "reasoning"
  ],
  "abstract": "We compared four methods: an LLM, a reasoning LLM, LLM-as-a-judge, and Jev-as-a-judge. All generation used Qwen3-1.7B. The first two methods disabled or enabled thinking; the latter two used non-thinking Qwen to generate eight candidate next steps and repeatedly selected one using Qwen or Jev 1.13.0, rather than choosing among complete solutions. Across all 30 AIME 2025 problems and three fixed seeds (90 trials per method), accuracy was 10.0%, 36.7%, 6.7%, and 11.1%, respectively. Jev underperformed the reasoning LLM in this setup. Its difference from Qwen self-judging remained unresolved. Jev's paired differences were -25.6 percentage points versus reasoning (95% problem-cluster interval -35.6 to -15.6) and +4.4 versus self-judging (-2.2 to +13.3). On one RTX 4090 at concurrency four, Jev averaged $0.0126 and 247.5 seconds per problem, versus $0.0064 and 211.6 seconds for reasoning. These results describe one small, public-benchmark experiment with unequal compute budgets, not an exhaustive comparison of search policies.",
  "totalMeasuredCost": 2.2560137813869012
}