{
  "schema": 1,
  "checked_utc": "2026-09-30T06:55:00Z",
  "overall": null,
  "overall_reason": "No common set of comparable measured benchmarks for every profile. Missing results are not zero and are not silently omitted from an average.",
  "conflict_disclosure": "JevBench and ImageJevBench are run by Florian Standhartinger, the same founder as System1 Models. Workflow Evals is published by TypeSafe AI, the vendor of the reference model.",
  "models": [
    {
      "key": "fast",
      "profile": "s1-fast",
      "name": "Plumb-4B",
      "revision": "55de037801a8a9b9de3db5c0e16cef86210c2186",
      "repository": "https://huggingface.co/crh225/plumb-4b"
    },
    {
      "key": "pro",
      "profile": "s1-pro",
      "name": "Winnow-12B Q8",
      "revision": "b2b14213dfa252e6d6b543c8b334762e51772d29",
      "repository": "https://huggingface.co/EldanRing/Winnow-12B"
    },
    {
      "key": "reference",
      "profile": "TypeSafe reference",
      "name": "Jev 1.13.0",
      "revision": null,
      "repository": "https://docs.typesafe.ai"
    }
  ],
  "benchmarks": [
    {
      "key": "decision-index",
      "name": "Decision Index",
      "version": "0.2.1",
      "metric": "Chance-corrected index / 100",
      "metric_de": "Zufallsbereinigter Index / 100",
      "scale": 100,
      "digits": 2,
      "source": "https://multimodalart-jev-decision-index.static.hf.space/index.html",
      "source_data": "https://multimodalart-jev-decision-index.static.hf.space/data/index.json",
      "published": "2026-09-28T00:39:36+00:00",
      "ownership": "third-party",
      "snapshot": "decision-index.json.gz",
      "sha256": "a5a4aa0a2cce152056964a5c732c919f55fd8e0d1115f9cb7f849c35322eb9df",
      "values": {
        "fast": null,
        "pro": 50.02,
        "reference": 57.91
      },
      "note": "38 index benchmarks across five areas. Winnow Q8_0 is identified, but the tested checkpoint revision is not published. Model implementation results, not a System1 service run. The board uses its own label-lift runtime configuration; identical System1 engine configuration is not established."
    },
    {
      "key": "workflow-evals",
      "name": "Workflow Evals",
      "version": "datasets 2026-09-28 · code 0ac3b8a",
      "metric": "Model-reference agreement",
      "metric_de": "Übereinstimmung mit Modellreferenzen",
      "scale": 100,
      "digits": 1,
      "source": "https://evals.typesafe.ai/",
      "published": "2026-09-28",
      "ownership": "vendor",
      "values": {
        "fast": null,
        "pro": null,
        "reference": 67.76161455573221
      },
      "note": "Four public vendor workflow datasets, 705 cases; unweighted mean of the headline consensus metric reported in each manifest for Jev 1.13.0: invoice and customer service use exact action sets (accuracy), security incidents use exact action sets (agreement), and agent traces use primary-action disposition. Reference mode: all questions, reasoning off. Reference labels blend large-model answers, with some fallback contributors; agreement is not human-labelled accuracy. Dataset manifests publish no Plumb/Winnow runs. A small hosted-profile pilot is being prepared and would not replace this full published cohort.",
      "datasets": [
        {
          "workflow": "agent_trace_observability",
          "snapshot": "evalsafe-agent-trace-observability-dataset.json.gz",
          "sha256": "5a3657ff2940ce2b550fa2dee5ee12ba5810b9747c98843792fb7fd6d3007b33",
          "reference_value": 0.7162162162162162,
          "metric": "disposition",
          "source": "https://huggingface.co/datasets/typesafe/evalsafe-agent-trace-observability/blob/8635540973910a92465fe2bc53e195375aa6e1a8/dataset.json",
          "revision": "8635540973910a92465fe2bc53e195375aa6e1a8"
        },
        {
          "workflow": "customer_service",
          "snapshot": "evalsafe-customer-service-dataset.json.gz",
          "sha256": "2179468e73465c384cf0c6a807d895f857f2d8ac13ad43342e06a5c87a49cb54",
          "reference_value": 0.7598039215686274,
          "metric": "accuracy",
          "source": "https://huggingface.co/datasets/typesafe/evalsafe-customer-service/blob/b1342f5a704587dbc465867c38c2694348ff86e4/dataset.json",
          "revision": "b1342f5a704587dbc465867c38c2694348ff86e4"
        },
        {
          "workflow": "invoice_processing",
          "snapshot": "evalsafe-invoice-processing-dataset.json.gz",
          "sha256": "72e0cea51735ae3d24dba0dad53536aca72b5698262d57d2506db1b900730e37",
          "reference_value": 0.6177777777777778,
          "metric": "accuracy",
          "source": "https://huggingface.co/datasets/typesafe/evalsafe-invoice-processing/blob/6beeb2d2acd65c086c835022f5f4d7434114cafc/dataset.json",
          "revision": "6beeb2d2acd65c086c835022f5f4d7434114cafc"
        },
        {
          "workflow": "security_incidents",
          "snapshot": "evalsafe-security-incidents-dataset.json.gz",
          "sha256": "8bbf0d6837ea933924c1a66111cbee5b61b6920703894dc7a69bb3ed591b676a",
          "reference_value": 0.6166666666666667,
          "metric": "agreement",
          "source": "https://huggingface.co/datasets/typesafe/evalsafe-security-incidents/blob/fbe1ea5c69cf494157fd23f2002a0d9d9a418443/dataset.json",
          "revision": "fbe1ea5c69cf494157fd23f2002a0d9d9a418443"
        }
      ]
    },
    {
      "key": "rerank",
      "name": "Jev Rerank Bench",
      "version": "2026-09-25",
      "metric": "Dataset-macro nDCG@10 / 1",
      "metric_de": "Datensatz-Mittel nDCG@10 / 1",
      "scale": 1,
      "digits": 3,
      "source": "https://github.com/anessbelbati/jev-rerank-bench/blob/fecba75a443c580a8979ce2587e2d452d4d8e511/README.md",
      "published": "2026-09-25",
      "ownership": "third-party",
      "snapshot": "rerank-README.md.gz",
      "sha256": "41f46b56657e47fc7982514651a33056d4a96be10dafdac4e061f2c950ac5ef6",
      "values": {
        "fast": null,
        "pro": 0.634,
        "reference": 0.67
      },
      "note": "Eight English retrieval datasets; 1,617 scored questions. Same yes/no-per-pair method for both models. Winnow Q8 checkpoint b2b1421 is explicitly identified. Its A5000 run is not System1 service performance. Public dataset training overlap cannot be excluded. nDCG is ranking utility, not classification accuracy. The reference also has a stronger multi-level, batched configuration; Winnow was not tested with that method, so the like-for-like row is not the reference model’s ceiling.",
      "reference_best_other_method": {
        "score": 0.692,
        "method": "4-level rubric, 30 in one call",
        "not_comparable_to_candidate": true
      }
    },
    {
      "key": "jevbench",
      "name": "JevBench",
      "version": "1.5.1",
      "metric": "Composite score / 100",
      "metric_de": "Gesamtwertung / 100",
      "scale": 100,
      "digits": 2,
      "source": "https://benchmarkheaven.com/jev-models/v1.5.1",
      "source_data": "https://benchmarkheaven.com/api/jevbench/v1.5.1",
      "published": "2026-09-29",
      "ownership": "same-founder",
      "snapshot": "jevbench-v1.5.1.json.gz",
      "sha256": "6f2fa547454b1108fad701ef302f48450742562393d532d45eccd048f736a9e2",
      "values": {
        "fast": 71.56069240662383,
        "pro": 73.23310265651646,
        "reference": 72.13292897895896
      },
      "note": "Official released edition shown above. Its composite weights intelligence, calibration, speed and cost equally. This comparison uses only published aggregate results; no sealed item text was used to build it. The Plumb and Winnow rows do not record tested checkpoint revisions. Their local-GPU speed is adjusted with an assumed multiplier and gateway allowance; their hosted costs are estimates, while the TypeSafe reference uses observed hosted API latency and list price. These axes do not establish System1 service latency or cost. Plumb intelligence is well below the reference and its composite benefits from estimated speed and cost. Winnow has a slightly higher composite than the reference, with overlapping confidence intervals.",
      "comparison_caveats": {
        "candidate_speed_basis": "local GPU latency ×2 + 0.15 s; assumed, not measured end-to-end",
        "candidate_cost_basis": "estimated hosted-model cost; Plumb uses base-model estimate",
        "reference_speed_basis": "observed hosted API latency",
        "reference_cost_basis": "published list price",
        "fast_intelligence": 55.84843638193568,
        "reference_intelligence": 71.99974990450261,
        "pro_score_ci95": [
          72.02170416109966,
          73.99468092883008
        ],
        "reference_score_ci95": [
          71.01427970926099,
          72.61481293732982
        ],
        "revision_identified": false
      }
    },
    {
      "key": "imagejevbench",
      "name": "ImageJevBench",
      "version": "0.1.4",
      "metric": "Composite score / 100",
      "metric_de": "Gesamtwertung / 100",
      "scale": 100,
      "digits": 2,
      "source": "https://benchmarkheaven.com/image-jev-bench",
      "published": "2026-09-29",
      "ownership": "same-founder",
      "snapshot": "imagejev-live.html.gz",
      "sha256": "958fc3373f254e4258c2f3b535d18c2fe53a51aec0df0ae2af955920e982b968",
      "values": {
        "fast": null,
        "pro": 43.55597082862413,
        "reference": null
      },
      "note": "Public edition shown above. Winnow Q8_0 uses an F16 vision projector in a separate image implementation. The row does not record the tested checkpoint revision or projector hash. System1 customer image routing remains gated. TypeSafe Jev 1.13.0 and Plumb have no comparable image rows; other image-model results are not substituted. The held next release is not used."
    }
  ],
  "method": "All five benchmarks receive equal column width and the same visual treatment. Source metrics and units stay separate. No overall score is published. A future overall must use a declared common benchmark set and the unweighted arithmetic mean of normalized scores: mean(100 * score / scale), with chance correction retained where the source already applies it. Never count submetrics or model-card copies of the same benchmark as extra votes; never transfer backbone scores to a fine-tune or host service.",
  "coverage_search": "Primary-source discovery checked Decision Index, Workflow Evals, Jev Rerank Bench, JevBench, ImageJevBench, Hugging Face model cards, Artificial Analysis, Epoch AI and Vals. No verified exact Plumb/Winnow results from the last three were found in the sources checked. Backbone evaluations are linked as background only; they are not System1 profile results. Author-reported measurements are not counted again as independent benchmark evidence."
}
