{
  "id": "jev-first-look-2026-09-20",
  "status": "external_author_run_site_recalculated",
  "source": {
    "author": "Colin McNamara",
    "repository": "https://github.com/colinmcnamara/jev-first-look",
    "commit": "e32717b1c592c3fa40e4892e9608638afb76c5fd",
    "commitUrl": "https://github.com/colinmcnamara/jev-first-look/tree/e32717b1c592c3fa40e4892e9608638afb76c5fd",
    "runDate": "2026-09-20",
    "location": "Austin, TX",
    "model": "jev-1.13.0",
    "sdk": "typesafe-sdk 0.7.0",
    "license": "MIT for upstream code; dataset text remains with its owners",
    "calibrationRawUrl": "https://raw.githubusercontent.com/colinmcnamara/jev-first-look/e32717b1c592c3fa40e4892e9608638afb76c5fd/results/t3_raw.json",
    "calibrationRawSha256": "5a46d61b1074d350ed5c915ff36d0ceb0692fe1cf1f904494cc342f0af83ea1f",
    "stabilityRawUrl": "https://raw.githubusercontent.com/colinmcnamara/jev-first-look/e32717b1c592c3fa40e4892e9608638afb76c5fd/results/t1_results.json",
    "stabilityRawSha256": "cac6a52f33c74dfd1c7c6f8bb21cba5cacb057b4bd029b13fedb43206af46a40"
  },
  "method": {
    "sampleSizePerTask": 500,
    "seed": 20260920,
    "sampling": "random.Random(seed).sample over the full split",
    "concurrency": 4,
    "bins": 10,
    "binning": "equal-width; left inclusive, right exclusive; 1.0 included in the final bin",
    "siteRecalculatedAt": "2026-09-30",
    "siteRecalculation": "Fetched pinned raw outputs, verified SHA-256, and recalculated aggregates without an API key or new model calls"
  },
  "tasks": [
    {
      "id": "sst2_noul",
      "dataset": "SST-2 validation",
      "primitive": "Noul",
      "n": 500,
      "positiveLabels": 266,
      "accuracy": 0.94,
      "ece": 0.0934,
      "brier": 0.0491,
      "brierDefinition": "binary Brier score for positive-class probability",
      "bins": [
        {
          "bin": "0.0-0.1",
          "n": 164,
          "meanPredicted": 0.04,
          "observed": 0.012
        },
        {
          "bin": "0.1-0.2",
          "n": 46,
          "meanPredicted": 0.146,
          "observed": 0.087
        },
        {
          "bin": "0.2-0.3",
          "n": 19,
          "meanPredicted": 0.243,
          "observed": 0.105
        },
        {
          "bin": "0.3-0.4",
          "n": 13,
          "meanPredicted": 0.348,
          "observed": 0.538
        },
        {
          "bin": "0.4-0.5",
          "n": 14,
          "meanPredicted": 0.459,
          "observed": 0.786
        },
        {
          "bin": "0.5-0.6",
          "n": 4,
          "meanPredicted": 0.547,
          "observed": 1
        },
        {
          "bin": "0.6-0.7",
          "n": 25,
          "meanPredicted": 0.644,
          "observed": 0.92
        },
        {
          "bin": "0.7-0.8",
          "n": 34,
          "meanPredicted": 0.755,
          "observed": 0.941
        },
        {
          "bin": "0.8-0.9",
          "n": 54,
          "meanPredicted": 0.849,
          "observed": 1
        },
        {
          "bin": "0.9-1.0",
          "n": 127,
          "meanPredicted": 0.949,
          "observed": 1
        }
      ]
    },
    {
      "id": "agnews_choice",
      "dataset": "AG News test",
      "primitive": "Choice",
      "n": 500,
      "accuracy": 0.882,
      "ece": 0.0866,
      "brier": 0.1013,
      "brierDefinition": "binary top-label Brier score, not multiclass Brier",
      "bins": [
        {
          "bin": "0.0-0.1",
          "n": 0
        },
        {
          "bin": "0.1-0.2",
          "n": 0
        },
        {
          "bin": "0.2-0.3",
          "n": 0
        },
        {
          "bin": "0.3-0.4",
          "n": 0
        },
        {
          "bin": "0.4-0.5",
          "n": 1,
          "meanPredicted": 0.47,
          "observed": 0
        },
        {
          "bin": "0.5-0.6",
          "n": 15,
          "meanPredicted": 0.543,
          "observed": 0.733
        },
        {
          "bin": "0.6-0.7",
          "n": 15,
          "meanPredicted": 0.648,
          "observed": 0.667
        },
        {
          "bin": "0.7-0.8",
          "n": 17,
          "meanPredicted": 0.741,
          "observed": 0.765
        },
        {
          "bin": "0.8-0.9",
          "n": 25,
          "meanPredicted": 0.857,
          "observed": 0.48
        },
        {
          "bin": "0.9-1.0",
          "n": 427,
          "meanPredicted": 0.995,
          "observed": 0.925
        }
      ]
    }
  ],
  "stability": {
    "scope": "three repeated calls per fixed example; descriptive only",
    "ticketA": {
      "n": 3,
      "noul": [
        0.21,
        0.22,
        0.21
      ],
      "choiceYes": [
        0,
        0.01,
        0
      ]
    },
    "ticketB": {
      "n": 3,
      "refund": [
        0.72,
        0.73,
        0.72
      ],
      "notRefund": [
        0.44,
        0.47,
        0.41
      ]
    }
  },
  "limitations": [
    "The site recalculated published outputs; it did not reproduce the model calls.",
    "One author-written question wording was used per task.",
    "SST-2 and AG News are famous benchmarks and may appear in training data.",
    "Several middle reliability bins contain fewer than 20 items.",
    "The three repeated calls per example do not establish a stability rate or guarantee.",
    "Results do not determine a production threshold for another workflow or data distribution."
  ]
}
