{
  "schema_version": 1,
  "scope": "Exploratory synthetic task evaluation; training seed range. Not a production or generalization claim.",
  "runs": [
    {
      "label": "qwen35-base-100-199",
      "benchmark_version": "0.3.1",
      "timestamp_utc": "2026-10-10T01:47:55.357401+00:00",
      "split": "train",
      "seeds": [
        100,
        101,
        102,
        103,
        104,
        105,
        106,
        107,
        108,
        109,
        110,
        111,
        112,
        113,
        114,
        115,
        116,
        117,
        118,
        119,
        120,
        121,
        122,
        123,
        124,
        125,
        126,
        127,
        128,
        129,
        130,
        131,
        132,
        133,
        134,
        135,
        136,
        137,
        138,
        139,
        140,
        141,
        142,
        143,
        144,
        145,
        146,
        147,
        148,
        149,
        150,
        151,
        152,
        153,
        154,
        155,
        156,
        157,
        158,
        159,
        160,
        161,
        162,
        163,
        164,
        165,
        166,
        167,
        168,
        169,
        170,
        171,
        172,
        173,
        174,
        175,
        176,
        177,
        178,
        179,
        180,
        181,
        182,
        183,
        184,
        185,
        186,
        187,
        188,
        189,
        190,
        191,
        192,
        193,
        194,
        195,
        196,
        197,
        198,
        199
      ],
      "configuration": {
        "difficulty_override": null,
        "max_steps_by_difficulty": {
          "easy": 10,
          "medium": 15,
          "hard": 20
        },
        "seed_selection": "explicit",
        "provider_timeout_seconds": 180.0,
        "reasoning_effort": null,
        "max_completion_tokens": 256,
        "completion_token_parameter": "max_completion_tokens"
      },
      "metrics": {
        "episodes": 100,
        "successes": 41,
        "success_rate": 0.41,
        "average_steps": 11.56,
        "invalid_actions": 0,
        "policy_violations": 76,
        "provider_errors": 0,
        "infrastructure_errors": 0
      },
      "source_sha256": "6913cb88da1c1a2bfd1863d5ef259343bfd1f3fc28225d77be2a0dbd6d5e230e"
    },
    {
      "label": "qwen35-sft-v2-val",
      "benchmark_version": "0.3.1",
      "timestamp_utc": "2026-10-10T00:28:16.860170+00:00",
      "split": "train",
      "seeds": [
        100,
        101,
        102,
        103,
        104,
        105,
        106,
        107,
        108,
        109,
        110,
        111,
        112,
        113,
        114,
        115,
        116,
        117,
        118,
        119,
        120,
        121,
        122,
        123,
        124,
        125,
        126,
        127,
        128,
        129,
        130,
        131,
        132,
        133,
        134,
        135,
        136,
        137,
        138,
        139,
        140,
        141,
        142,
        143,
        144,
        145,
        146,
        147,
        148,
        149,
        150,
        151,
        152,
        153,
        154,
        155,
        156,
        157,
        158,
        159,
        160,
        161,
        162,
        163,
        164,
        165,
        166,
        167,
        168,
        169,
        170,
        171,
        172,
        173,
        174,
        175,
        176,
        177,
        178,
        179,
        180,
        181,
        182,
        183,
        184,
        185,
        186,
        187,
        188,
        189,
        190,
        191,
        192,
        193,
        194,
        195,
        196,
        197,
        198,
        199
      ],
      "configuration": {
        "difficulty_override": null,
        "max_steps_by_difficulty": {
          "easy": 10,
          "medium": 15,
          "hard": 20
        },
        "seed_selection": "explicit",
        "provider_timeout_seconds": 180.0,
        "reasoning_effort": null,
        "max_completion_tokens": 256,
        "completion_token_parameter": "max_completion_tokens"
      },
      "metrics": {
        "episodes": 100,
        "successes": 63,
        "success_rate": 0.63,
        "average_steps": 11.35,
        "invalid_actions": 83,
        "policy_violations": 0,
        "provider_errors": 0,
        "infrastructure_errors": 0
      },
      "source_sha256": "553e15af99c16693f93ccc902ee6c44df3b5cb96f65b7ff48a2d73ee3269b7be"
    }
  ]
}