{
  "version": "bufo-score-v1",
  "head_sha256": "800bc15b9d8259927fa7e94557c84664c82f0ca5f74e99435dbe6c95c21de2ba",
  "training_n": 192,
  "label": "Bufo Score",
  "scale": "0\u2013100, mean over all ten reference-conditioned tasks",
  "description": "We use the same image classifier for every drawing. It measures Bufo likeness, not the odds that a person will like the result. The score is experimental; it does not judge humor or check every detail of the prompt.",
  "training_exclusion": "Benchmark images and related groups were excluded from head fitting; the encoder is pretrained.",
  "validation_note": "Earlier generator-held-out experiments showed useful but imperfect ranking signal with different heads. They do not establish accuracy for this fixed scoring release."
}