{
  "schemaVersion": "1.0.0",
  "evaluatedThrough": "2026-08-31T01:29:51.711899+00:00",
  "status": "incident_baseline_established_no_live_model_comparison",
  "currentModel": "z-ai/glm-5.3",
  "modelDisplayName": "GLM-5.3",
  "router": "OpenRouter",
  "purpose": "Turn preserved research failures into public regression cases and define the evidence required before claiming the agent desk improved or changing its default model.",
  "honestClaim": "This is a three-incident control baseline, not a leaderboard, accuracy estimate, or proof of general model quality.",
  "sampleLimitations": [
    "Only three preserved online runs exist, all from one research series and one day.",
    "All online roles used the same model family, so role separation did not create independent judgments.",
    "The incidents were observed in production-style research runs, not a blinded challenger benchmark.",
    "No challenger model has been called, scored, approved, or rejected under this protocol yet."
  ],
  "summary": {
    "onlineRuns": 3,
    "configuredRoleOpportunities": 15,
    "completedRoleOutputs": 13,
    "knownFailureCases": 3,
    "failureCasesWithImplementedControls": 3,
    "implementedControls": 9,
    "liveChallengerEvaluations": 0,
    "modelSwapsApproved": 0
  },
  "roleScorecard": [
    {
      "role": "researcher",
      "model": "z-ai/glm-5.3",
      "configuredRuns": 3,
      "completedOutputs": 2,
      "decisions": {
        "pass": 2,
        "revise": 0,
        "block": 0,
        "notInvoked": 1
      },
      "disclosure": "Counts describe preserved artifacts, not independent reviewers or general model accuracy."
    },
    {
      "role": "statistician",
      "model": "z-ai/glm-5.3",
      "configuredRuns": 3,
      "completedOutputs": 3,
      "decisions": {
        "pass": 1,
        "revise": 1,
        "block": 1,
        "notInvoked": 0
      },
      "disclosure": "Counts describe preserved artifacts, not independent reviewers or general model accuracy."
    },
    {
      "role": "adversarial_reviewer",
      "model": "z-ai/glm-5.3",
      "configuredRuns": 3,
      "completedOutputs": 3,
      "decisions": {
        "pass": 2,
        "revise": 1,
        "block": 0,
        "notInvoked": 0
      },
      "disclosure": "Counts describe preserved artifacts, not independent reviewers or general model accuracy."
    },
    {
      "role": "citation_checker",
      "model": "z-ai/glm-5.3",
      "configuredRuns": 3,
      "completedOutputs": 3,
      "decisions": {
        "pass": 2,
        "revise": 0,
        "block": 1,
        "notInvoked": 0
      },
      "disclosure": "Counts describe preserved artifacts, not independent reviewers or general model accuracy."
    },
    {
      "role": "editor",
      "model": "z-ai/glm-5.3",
      "configuredRuns": 3,
      "completedOutputs": 2,
      "decisions": {
        "pass": 1,
        "revise": 1,
        "block": 0,
        "notInvoked": 1
      },
      "disclosure": "Counts describe preserved artifacts, not independent reviewers or general model accuracy."
    }
  ],
  "failureCases": [
    {
      "id": "FAIL-001",
      "runId": "2026-08-31-touchdown-regression-v1",
      "title": "The workflow let the model rewrite the contract",
      "layerThatStoppedRelease": "research_gate_and_post_run_audit",
      "finalDisposition": "rejected_by_research_gate",
      "failureCount": 5,
      "failures": [
        "The online researcher was incorrectly allowed to replace the locked preregistration.",
        "Reviewers received the memo and analysis summary but not the full provenance, article, or artifact inventory.",
        "The test-season range was encoded ambiguously as [2018, 2024].",
        "The pipeline did not generate a row-level attrition reconciliation.",
        "The editor role was configured but was not invoked."
      ],
      "remediationCount": 6,
      "remediations": [
        "Load the preregistration from the versioned series registry and make it immutable during a run.",
        "Use the researcher as a conformance reviewer rather than an author of the locked memo.",
        "Supply provenance, article text, analysis, mechanical checks, and a SHA-256 artifact inventory to every specialist.",
        "Make role, model, and timestamp transport-authoritative.",
        "Publish an attrition-by-season artifact and a full list of walk-forward test seasons.",
        "Invoke the editor after specialist review and block on any non-pass decision."
      ],
      "controls": [
        "CTRL-001",
        "CTRL-002",
        "CTRL-003",
        "CTRL-004",
        "CTRL-005"
      ],
      "evidence": {
        "processNote": "/research-artifacts/2026-08-31-touchdown-regression-v1/process-note.json",
        "gateDecision": "/research-artifacts/2026-08-31-touchdown-regression-v1/publication-decision.json"
      }
    },
    {
      "id": "FAIL-002",
      "runId": "2026-08-31T010623Z-touchdown-regression-v1",
      "title": "Valid JSON still contained invalid judgment",
      "layerThatStoppedRelease": "specialist_nonpass_and_research_gate",
      "finalDisposition": "rejected_by_research_gate",
      "failureCount": 4,
      "failures": [
        "The statistician issued a revise decision for concrete article framing changes around bootstrap interpretation, selection effects, and individual-level application.",
        "The adversarial reviewer returned a pass decision containing a literal placeholder blocking finding.",
        "The editor returned literal placeholder summary and finding text.",
        "The JSON schema enforced structure but did not yet enforce minimum semantic content or decision/finding coherence."
      ],
      "remediationCount": 6,
      "remediations": [
        "Require minimum lengths for review summaries and finding fields.",
        "Reject a pass decision containing a blocking finding and a block decision without one.",
        "Retry invalid structured outputs rather than writing placeholder artifacts.",
        "Remove the bootstrap win-rate framing and explain that resamples are not independent replications.",
        "Bound the tiebreaker advice to a population-level heuristic and disclose the uncertain direction of survivor conditioning.",
        "Trace the email subject to a generated statistic and cite both nflverse source files."
      ],
      "controls": [
        "CTRL-006",
        "CTRL-007",
        "CTRL-008"
      ],
      "evidence": {
        "processNote": "/research-artifacts/2026-08-31T010623Z-touchdown-regression-v1/process-note.json",
        "gateDecision": "/research-artifacts/2026-08-31T010623Z-touchdown-regression-v1/publication-decision.json"
      }
    },
    {
      "id": "FAIL-003",
      "runId": "2026-08-31T011519Z-touchdown-regression-v1",
      "title": "Every model passed; the pixels were still wrong",
      "layerThatStoppedRelease": "human_visual_qa",
      "finalDisposition": "withheld_after_visual_qa",
      "failureCount": 2,
      "failures": [
        "figures/forecast-error.png: visible 8.9%; analysis 11.7%",
        "figures/regression-quintiles.png: visible 2.8 touchdowns; analysis 3.0 touchdowns"
      ],
      "remediationCount": 1,
      "remediations": [
        "Every future figure is rendered from a JSON figure contract containing its visible title, source metrics, and final PNG hash. The mechanical gate and promotion command independently validate that contract against analysis.json."
      ],
      "controls": [
        "CTRL-009"
      ],
      "evidence": {
        "processNote": "/research-artifacts/2026-08-31T011519Z-touchdown-regression-v1/process-note.json",
        "gateDecision": "/research-artifacts/2026-08-31T011519Z-touchdown-regression-v1/publication-decision.json"
      }
    }
  ],
  "controls": [
    {
      "id": "CTRL-001",
      "name": "Immutable registration",
      "protectsAgainst": "A model replacing the preregistered estimand or design after results exist.",
      "implementationFiles": [
        "research/src/fourth_down_labs/registry.py",
        "research/src/fourth_down_labs/pipeline.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_registry.py",
          "name": "test_registration_memo_is_loaded_from_versioned_registry"
        }
      ],
      "publicContracts": [
        "/methods/prompts/researcher.md"
      ],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-002",
      "name": "Complete reviewer evidence bundle",
      "protectsAgainst": "Specialists reviewing a summary without provenance, article text, inventory, or mechanical checks.",
      "implementationFiles": [
        "research/src/fourth_down_labs/pipeline.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_agent_controls.py",
          "name": "test_every_specialist_receives_the_complete_locked_evidence_bundle"
        }
      ],
      "publicContracts": [
        "/methods/prompts/statistician.md",
        "/methods/prompts/adversarial-reviewer.md"
      ],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-003",
      "name": "Transport-authoritative identity",
      "protectsAgainst": "A model self-reporting a different role, model name, or timestamp inside its answer.",
      "implementationFiles": [
        "research/src/fourth_down_labs/openrouter.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_openrouter.py",
          "name": "test_transport_metadata_overrides_model_self_report"
        }
      ],
      "publicContracts": [
        "/methods/schemas/OpenRouterCallRecord.schema.json"
      ],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-004",
      "name": "Five-role completion and all-pass gate",
      "protectsAgainst": "A configured specialist never running, or a revise/block decision being ignored.",
      "implementationFiles": [
        "research/src/fourth_down_labs/pipeline.py",
        "research/src/fourth_down_labs/promotion.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_promotion.py",
          "name": "test_promotion_rejects_nonpass_specialist_decision"
        }
      ],
      "publicContracts": [
        "/methods/schemas/ReviewArtifact.schema.json"
      ],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-005",
      "name": "Row-level attrition reconciliation",
      "protectsAgainst": "Forecast metrics silently conditioning on an unexplained surviving subset.",
      "implementationFiles": [
        "research/src/fourth_down_labs/pipeline.py",
        "research/src/fourth_down_labs/analysis/touchdown_regression.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_analysis.py",
          "name": "test_attrition_reconciles_every_prediction"
        }
      ],
      "publicContracts": [],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-006",
      "name": "Semantic review validation",
      "protectsAgainst": "Placeholder findings and incoherent pass/block combinations satisfying a merely structural schema.",
      "implementationFiles": [
        "research/src/fourth_down_labs/models.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_openrouter.py",
          "name": "test_placeholder_review_text_is_rejected"
        },
        {
          "file": "research/tests/test_openrouter.py",
          "name": "test_passing_review_cannot_hide_a_blocking_finding"
        }
      ],
      "publicContracts": [
        "/methods/schemas/ReviewArtifact.schema.json"
      ],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-007",
      "name": "Per-attempt call receipts and bounded retry",
      "protectsAgainst": "A malformed attempt disappearing after retry, along with its cost, identity, or failure status.",
      "implementationFiles": [
        "research/src/fourth_down_labs/openrouter.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_openrouter.py",
          "name": "test_rejected_attempt_keeps_cost_receipt_without_payload"
        }
      ],
      "publicContracts": [
        "/methods/schemas/OpenRouterCallRecord.schema.json"
      ],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-008",
      "name": "Generated claim framing",
      "protectsAgainst": "Bootstrap resamples being described as independent replication or a population heuristic becoming a player-level rule.",
      "implementationFiles": [
        "research/src/fourth_down_labs/pipeline.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_analysis.py",
          "name": "test_pairing_and_bootstrap_are_deterministic"
        }
      ],
      "publicContracts": [
        "/methods/prompts/editor-publisher.md"
      ],
      "status": "implemented_and_tested"
    },
    {
      "id": "CTRL-009",
      "name": "Figure claim contract and atomic promotion",
      "protectsAgainst": "A current chart containing a stale visible title or being copied separately from its reviewed run.",
      "implementationFiles": [
        "research/src/fourth_down_labs/analysis/touchdown_regression.py",
        "research/src/fourth_down_labs/promotion.py"
      ],
      "testCases": [
        {
          "file": "research/tests/test_promotion.py",
          "name": "test_promotion_rejects_semantically_tampered_figure_contract"
        },
        {
          "file": "research/tests/test_rookie_promotion.py",
          "name": "test_rookie_promotion_rejects_semantically_tampered_figure_contract"
        }
      ],
      "publicContracts": [],
      "status": "implemented_and_tested"
    }
  ],
  "modelChangeProtocol": {
    "status": "ready_for_an_approved_public_or_disclosure_authorized_corpus",
    "incumbent": "z-ai/glm-5.3",
    "challenger": null,
    "corpusPolicy": [
      "Use already-public artifacts or obtain explicit disclosure approval before any prepublication bundle leaves the repository.",
      "Freeze the same evidence, role prompt, output schema, temperature, output cap, retry budget, and scoring rules for incumbent and challenger.",
      "Keep the failure-case labels outside the model-visible evidence and score every role output from preserved artifacts."
    ],
    "requiredMeasures": [
      "Schema-valid output rate and retry rate",
      "Known blocking-failure recall",
      "False-pass count on known failure cases",
      "Evidence-specificity and placeholder rejection",
      "Role-decision coherence",
      "Exact provider-reported tokens, billed cost, and transport latency"
    ],
    "approvalRules": [
      "Zero publication-eligible false passes across the known failure corpus after deterministic gating.",
      "No regression on any incumbent control case without a documented, public exception.",
      "Complete per-attempt receipts for both models; missing usage remains null and cannot be estimated.",
      "A human release decision that names quality, cost, latency, shared-model risk, and the limits of the sample."
    ],
    "currentDecision": "No model change is justified. GLM-5.3 remains the named default and no challenger has been evaluated."
  },
  "publicEvidence": {
    "aiOperations": "/ai-operations",
    "aiOperationsJson": "/ai-operations-ledger.json",
    "rawRunIndex": "/research-artifacts/index.json",
    "receiptSchema": "/methods/schemas/OpenRouterCallRecord.schema.json",
    "reviewSchema": "/methods/schemas/ReviewArtifact.schema.json"
  }
}
