{
  "title": "Ackren recorded evaluation: publication extract",
  "published": "2026-09-25",
  "basis": "counted from retained report; no new engine run",
  "build_ids": [
    "700cb343ae839710"
  ],
  "adapter_version": "3.0.0",
  "qualified_episodes": 94,
  "scored_episodes": 120,
  "groups": {
    "ambiguity_repair": {
      "passed": 20,
      "total": 20
    },
    "contrast": {
      "passed": 14,
      "total": 40
    },
    "paraphrase": {
      "passed": 40,
      "total": 40
    },
    "teaching_correction": {
      "passed": 20,
      "total": 20
    }
  },
  "replay": {
    "basis": "measured",
    "passed": 716,
    "rate": 1.0,
    "total": 716
  },
  "explanation_audit": {
    "qualification": "Structural/source-link checks, not human validation of explanation prose. Record resolution is limited to public records included in the response/state; static pack references absent from those records remain unresolved. Hash shape is checked, not an undocumented hash algorithm.",
    "record_references": {
      "basis": "measured",
      "passed": 4741,
      "rate": 1.0,
      "total": 4741
    },
    "schema_valid": {
      "basis": "measured",
      "passed": 716,
      "rate": 1.0,
      "total": 716
    },
    "source_references": {
      "basis": "measured",
      "passed": 5197,
      "rate": 1.0,
      "total": 5197
    },
    "trace_hash_shape_valid": {
      "basis": "measured",
      "passed": 716,
      "rate": 1.0,
      "total": 716
    }
  },
  "human_participants": 0,
  "scored_checkpoints": 476,
  "diagnostic_turns": 240,
  "source_sha256": "30a3392bc3b3b186528248033c746287781b0b12a0a58d21bbebd8b366e95930",
  "source_name": "candidate-four-conversation-report.json",
  "limits": [
    "Exposed agent-authored regression; no independent human study or sealed generalisation sample.",
    "Overall and meaning-contrast qualification thresholds failed.",
    "Evaluator and runtime both changed during development; historical scores are not a controlled comparison.",
    "Structural records do not establish complete explanation faithfulness.",
    "These results concern the earlier frozen candidate, not the newer compositional prototype."
  ],
  "thresholds": {
    "basis": "decided in evaluation protocol",
    "overall_episodes": 108,
    "contrast_episodes": 36
  },
  "source_availability": "Curated result extract. Full source snapshot and replay package are not deposited here."
}
