{
  "schema_version": 2,
  "title": "Can models learn to search better?",
  "date": "2026-09-30",
  "status": "Exploratory study \u00b7 separate Claude and Jev judgments of the same saved answers",
  "learner": "Qwen3-4B-Instruct-2507",
  "training": {
    "questions": 240,
    "attempts_per_question": 8,
    "updates_per_arm": 30,
    "conditions": [
      "Exa",
      "Perplexity Search",
      "Serper",
      "No search"
    ],
    "method": "Dr. GRPO with rank-32 LoRA adapters",
    "seeds_per_condition": 1,
    "questions_per_batch": 8,
    "learning_rate": 1e-05,
    "reward": "Correct = 1, incorrect or unanswered = 0; unresolved judgments excluded from group advantage.",
    "zero_gradient_updates": {
      "none": 12,
      "exa": 6,
      "serper": 1,
      "perplexity_search": 9
    }
  },
  "exam": {
    "questions": 150,
    "datasets": [
      "HotpotQA",
      "MuSiQue"
    ],
    "questions_per_dataset": 75,
    "answers": 2250,
    "conditions": [
      {
        "id": "keenable",
        "label": "Keenable"
      },
      {
        "id": "valyu",
        "label": "Valyu"
      },
      {
        "id": "none",
        "label": "No search"
      }
    ],
    "sampling_unit": "150 questions evaluated in 15 checkpoint/tool conditions; answers are not independent observations."
  },
  "provenance": {
    "pre_run_freeze_at": "2026-09-28T23:00:38Z",
    "publication_status": "Post-run disclosure of a locally hash-frozen pre-run plan; public preregistration is not established.",
    "grading_status": "Aggregate results from both completed exam analyses are included separately. Claude supplied training rewards; Jev regraded only saved exam answers. Unresolved grades are retained within their original judge.",
    "plan_sha256": "2dedef9631b4ee562ded3da5afdf92386513835f22f255c7f09eafdba958466a",
    "answer_judge_prompt_sha256": "8f8c3d80ce37712cf6d42ad0d967e260c5478f4e0e0256baa3644bb3755792fa",
    "question_screen_prompt_sha256": "01744a5b978eb116dee8bfb0b1ae547475af9a89d16d1214b5c6aea746678293",
    "operational_changes": [
      "Case identifier length limit increased during development evaluation.",
      "Transient search retries and an unavailable-search fallback added during the exam; Valyu empty responses treated as unavailable."
    ],
    "data_release_scope": "Aggregate scores, counts, settings and hashes. Question text, reference answers, per-case judgments, raw traces, credentials and infrastructure details are excluded."
  },
  "metric_notes": {
    "accuracy_resolved": "Correct / resolved judgments. Dropping unresolved judgments can bias the displayed point score.",
    "accuracy_bounds": "Lower = correct / 150; upper = (correct + pending) / 150. These bound unresolved grades only.",
    "search_unavailable": "Number of final exam attempts with a recorded search failure. The model answered without results from the failed search; earlier successful searches, if any, were still available."
  },
  "judges": {
    "claude": {
      "label": "Claude",
      "model": "claude-fable-5-1",
      "unresolved_answers": 32,
      "results": [
        {
          "id": "original",
          "label": "Qwen3-4B-Instruct-2507",
          "cells": {
            "keenable": {
              "correct": 64,
              "incorrect": 82,
              "pending": 4,
              "total": 150,
              "accuracy_resolved": 0.4383561643835616,
              "accuracy_lower": 0.4266666666666667,
              "accuracy_upper": 0.4533333333333333,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 53,
              "incorrect": 92,
              "pending": 5,
              "total": 150,
              "accuracy_resolved": 0.36551724137931035,
              "accuracy_lower": 0.35333333333333333,
              "accuracy_upper": 0.38666666666666666,
              "search_unavailable": 5
            },
            "none": {
              "correct": 27,
              "incorrect": 121,
              "pending": 2,
              "total": 150,
              "accuracy_resolved": 0.18243243243243243,
              "accuracy_lower": 0.18,
              "accuracy_upper": 0.19333333333333333,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "none",
          "label": "Trained without search",
          "cells": {
            "keenable": {
              "correct": 66,
              "incorrect": 83,
              "pending": 1,
              "total": 150,
              "accuracy_resolved": 0.4429530201342282,
              "accuracy_lower": 0.44,
              "accuracy_upper": 0.44666666666666666,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 55,
              "incorrect": 92,
              "pending": 3,
              "total": 150,
              "accuracy_resolved": 0.3741496598639456,
              "accuracy_lower": 0.36666666666666664,
              "accuracy_upper": 0.38666666666666666,
              "search_unavailable": 6
            },
            "none": {
              "correct": 27,
              "incorrect": 121,
              "pending": 2,
              "total": 150,
              "accuracy_resolved": 0.18243243243243243,
              "accuracy_lower": 0.18,
              "accuracy_upper": 0.19333333333333333,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "exa",
          "label": "Trained with Exa",
          "cells": {
            "keenable": {
              "correct": 73,
              "incorrect": 74,
              "pending": 3,
              "total": 150,
              "accuracy_resolved": 0.4965986394557823,
              "accuracy_lower": 0.4866666666666667,
              "accuracy_upper": 0.5066666666666667,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 58,
              "incorrect": 88,
              "pending": 4,
              "total": 150,
              "accuracy_resolved": 0.3972602739726027,
              "accuracy_lower": 0.38666666666666666,
              "accuracy_upper": 0.41333333333333333,
              "search_unavailable": 5
            },
            "none": {
              "correct": 28,
              "incorrect": 122,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.18666666666666668,
              "accuracy_lower": 0.18666666666666668,
              "accuracy_upper": 0.18666666666666668,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "serper",
          "label": "Trained with Serper",
          "cells": {
            "keenable": {
              "correct": 73,
              "incorrect": 75,
              "pending": 2,
              "total": 150,
              "accuracy_resolved": 0.49324324324324326,
              "accuracy_lower": 0.4866666666666667,
              "accuracy_upper": 0.5,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 58,
              "incorrect": 89,
              "pending": 3,
              "total": 150,
              "accuracy_resolved": 0.3945578231292517,
              "accuracy_lower": 0.38666666666666666,
              "accuracy_upper": 0.4066666666666667,
              "search_unavailable": 6
            },
            "none": {
              "correct": 27,
              "incorrect": 122,
              "pending": 1,
              "total": 150,
              "accuracy_resolved": 0.18120805369127516,
              "accuracy_lower": 0.18,
              "accuracy_upper": 0.18666666666666668,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "perplexity_search",
          "label": "Trained with Perplexity Search",
          "cells": {
            "keenable": {
              "correct": 85,
              "incorrect": 64,
              "pending": 1,
              "total": 150,
              "accuracy_resolved": 0.5704697986577181,
              "accuracy_lower": 0.5666666666666667,
              "accuracy_upper": 0.5733333333333334,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 60,
              "incorrect": 90,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.4,
              "accuracy_lower": 0.4,
              "accuracy_upper": 0.4,
              "search_unavailable": 4
            },
            "none": {
              "correct": 27,
              "incorrect": 122,
              "pending": 1,
              "total": 150,
              "accuracy_resolved": 0.18120805369127516,
              "accuracy_lower": 0.18,
              "accuracy_upper": 0.18666666666666668,
              "search_unavailable": 0
            }
          }
        }
      ],
      "primary": {
        "definition": "Mean of the three search-trained checkpoints minus the trained-without-search checkpoint, averaged over Keenable and Valyu only. Unresolved-grade bounds and their bootstrap intervals use equal dataset weights.",
        "resolved_only_definition": "Descriptive average of the displayed checkpoint/tool resolved-only accuracies: mean of the three search-trained versions minus trained-without-search, over Keenable and Valyu. Within each checkpoint/tool score, resolved cases are pooled across the datasets.",
        "resolved_only_gain_pp": 5.013695641067945,
        "unresolved_grade_bounds_pp": [
          3.555555555555551,
          6.333333333333336
        ],
        "lower_endpoint_interval_pp": [
          -0.5278950216450252,
          7.701478687739463
        ],
        "upper_endpoint_interval_pp": [
          2.768062798783132,
          10.19158320240901
        ],
        "bootstrap": "10,000 paired question-family bootstrap draws, stratified by dataset; 95% pointwise intervals. Not multiplicity-adjusted or training-seed uncertainty.",
        "uncertainty_note": "Unresolved-grade bounds are not confidence intervals. Separate bootstrap intervals describe sampling uncertainty in each bound endpoint; the lower-endpoint interval includes zero.",
        "kind": "bounds"
      },
      "analysis_sha256": "fb5b78d296b7dc023d48e3ea45a1a55bdcc5b54eef4039a5fedad8fa5f8f6d53"
    },
    "jev": {
      "label": "Jev",
      "model": "jev-1.13.0",
      "unresolved_answers": 0,
      "results": [
        {
          "id": "original",
          "label": "Qwen3-4B-Instruct-2507",
          "cells": {
            "keenable": {
              "correct": 62,
              "incorrect": 88,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.41333333333333333,
              "accuracy_lower": 0.41333333333333333,
              "accuracy_upper": 0.41333333333333333,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 52,
              "incorrect": 98,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.3466666666666667,
              "accuracy_lower": 0.3466666666666667,
              "accuracy_upper": 0.3466666666666667,
              "search_unavailable": 5
            },
            "none": {
              "correct": 28,
              "incorrect": 122,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.18666666666666668,
              "accuracy_lower": 0.18666666666666668,
              "accuracy_upper": 0.18666666666666668,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "none",
          "label": "Trained without search",
          "cells": {
            "keenable": {
              "correct": 63,
              "incorrect": 87,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.42,
              "accuracy_lower": 0.42,
              "accuracy_upper": 0.42,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 53,
              "incorrect": 97,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.35333333333333333,
              "accuracy_lower": 0.35333333333333333,
              "accuracy_upper": 0.35333333333333333,
              "search_unavailable": 6
            },
            "none": {
              "correct": 28,
              "incorrect": 122,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.18666666666666668,
              "accuracy_lower": 0.18666666666666668,
              "accuracy_upper": 0.18666666666666668,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "exa",
          "label": "Trained with Exa",
          "cells": {
            "keenable": {
              "correct": 71,
              "incorrect": 79,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.47333333333333333,
              "accuracy_lower": 0.47333333333333333,
              "accuracy_upper": 0.47333333333333333,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 54,
              "incorrect": 96,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.36,
              "accuracy_lower": 0.36,
              "accuracy_upper": 0.36,
              "search_unavailable": 5
            },
            "none": {
              "correct": 27,
              "incorrect": 123,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.18,
              "accuracy_lower": 0.18,
              "accuracy_upper": 0.18,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "serper",
          "label": "Trained with Serper",
          "cells": {
            "keenable": {
              "correct": 71,
              "incorrect": 79,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.47333333333333333,
              "accuracy_lower": 0.47333333333333333,
              "accuracy_upper": 0.47333333333333333,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 54,
              "incorrect": 96,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.36,
              "accuracy_lower": 0.36,
              "accuracy_upper": 0.36,
              "search_unavailable": 6
            },
            "none": {
              "correct": 27,
              "incorrect": 123,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.18,
              "accuracy_lower": 0.18,
              "accuracy_upper": 0.18,
              "search_unavailable": 0
            }
          }
        },
        {
          "id": "perplexity_search",
          "label": "Trained with Perplexity Search",
          "cells": {
            "keenable": {
              "correct": 80,
              "incorrect": 70,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.5333333333333333,
              "accuracy_lower": 0.5333333333333333,
              "accuracy_upper": 0.5333333333333333,
              "search_unavailable": 0
            },
            "valyu": {
              "correct": 57,
              "incorrect": 93,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.38,
              "accuracy_lower": 0.38,
              "accuracy_upper": 0.38,
              "search_unavailable": 4
            },
            "none": {
              "correct": 28,
              "incorrect": 122,
              "pending": 0,
              "total": 150,
              "accuracy_resolved": 0.18666666666666668,
              "accuracy_lower": 0.18666666666666668,
              "accuracy_upper": 0.18666666666666668,
              "search_unavailable": 0
            }
          }
        }
      ],
      "primary": {
        "kind": "point",
        "definition": "Mean of the three search-trained checkpoints minus the trained-without-search checkpoint, averaged over Keenable and Valyu only, with equal dataset weights.",
        "estimate_pp": 4.333333333333337,
        "interval_pp": [
          0.966440896629572,
          7.823353909465018
        ],
        "bootstrap": "10,000 paired question-family bootstrap draws, stratified by dataset; 95% pointwise intervals. Not multiplicity-adjusted or training-seed uncertainty.",
        "uncertainty_note": "All grades are resolved. The interval describes sampling uncertainty over questions, conditional on one trained instance per condition."
      },
      "analysis_sha256": "f4c6e5b6d875aa702feb586a0debab4d37402e88856dde61967c29edd69d5a6c"
    }
  },
  "judge_agreement": {
    "agreed": 2147,
    "total": 2250,
    "fraction": 0.9542222222222222,
    "binary_agreed": 2147,
    "binary_total": 2218,
    "binary_fraction": 0.9679891794409378,
    "note": "Separate judgments of the same saved exam answers; agreement is not ground truth or an independent experimental replication.",
    "by_exam_tool": [
      {
        "id": "keenable",
        "total": 750,
        "agreed": 713,
        "claude": {
          "correct": 361,
          "resolved": 739,
          "accuracy": 0.4884979702300406
        },
        "jev": {
          "correct": 347,
          "resolved": 750,
          "accuracy": 0.46266666666666667
        },
        "difference_pp": -2.583130356337393,
        "agreement": 0.9506666666666667,
        "label": "Keenable"
      },
      {
        "id": "valyu",
        "total": 750,
        "agreed": 707,
        "claude": {
          "correct": 284,
          "resolved": 735,
          "accuracy": 0.38639455782312926
        },
        "jev": {
          "correct": 270,
          "resolved": 750,
          "accuracy": 0.36
        },
        "difference_pp": -2.639455782312927,
        "agreement": 0.9426666666666667,
        "label": "Valyu"
      },
      {
        "id": "none",
        "total": 750,
        "agreed": 727,
        "claude": {
          "correct": 136,
          "resolved": 744,
          "accuracy": 0.1827956989247312
        },
        "jev": {
          "correct": 138,
          "resolved": 750,
          "accuracy": 0.184
        },
        "difference_pp": 0.12043010752688099,
        "agreement": 0.9693333333333334,
        "label": "No search"
      }
    ],
    "by_exam_tool_definition": "Pooled over all five model versions. Accuracy is summed correct / summed resolved for each judge. Agreement is matching verdicts / all 750 saved answers per exam tool. These reuse 150 questions and are descriptive, not independent question samples.",
    "binary_disagreements": 71
  },
  "judge_operations": {
    "scope": "Successful grading of the same 2250 saved exam answers, under the two observed judge configurations.",
    "exam_answers": 2250,
    "cost_is_estimate": true,
    "cost_exclusions": [
      "diagnostics",
      "failed calls",
      "earlier attempts",
      "training",
      "answer generation"
    ],
    "latency_comparable": false,
    "claude": {
      "successful_requests": 45,
      "batch_sizes": [
        22,
        64
      ],
      "estimated_usd": 13.72969475,
      "cost_basis": "CLI-reported API-equivalent list-price estimate; subscription cost not allocated; not verified billing.",
      "total_request_seconds": 1886.0606205388904,
      "median_request_seconds": 47.590141439810395,
      "timing_basis": "Local elapsed time for each completed CLI request, including transport and CLI overhead. Each request graded 22 or 64 answers.",
      "response_receipts_sha256": "b10c647cc29d4df14d270d50350ffc30b5ea2eb667dd46940207c2930d7fc51b"
    },
    "jev": {
      "successful_requests": 2250,
      "batch_sizes": [
        1
      ],
      "input_tokens": 1335879,
      "output_tokens": 92250,
      "estimated_usd": 0.056106918000000006,
      "input_usd_per_million": 0.042,
      "output_usd_per_million": 0,
      "price_source_url": "https://docs.typesafe.ai/models",
      "price_checked_at": "2026-09-30",
      "cost_basis": "Successful exam request input-token usage multiplied by published price; not verified billing.",
      "median_request_seconds": null,
      "total_request_seconds": null,
      "exam_workflow_seconds": 2369.413026,
      "timing_basis": "From first final-journal exam claim through final complete summary. Includes deliberate minimum one-second call-start spacing, five validation failures and their retries, restart delays, and local bookkeeping; excludes initial diagnostic and previous journals. Not directly comparable to Claude sum-of-request elapsed time, and not pure API latency.",
      "minimum_request_start_interval_seconds": 1,
      "failed_exam_requests": 5,
      "source_manifest_sha256": "28e20a6bf91a231b2b6a47eb2905c3ee05736ee169d63c86141ab9efb42b1366",
      "final_run_summary_sha256": "56b7f5ee776987068301aa5d9991ba67c505dd64ad3130010c5ad012216b8c6e"
    }
  }
}
