{
  "method": "Exclude only 33 explicitly reviewed invalid expectations/judgments; partial, blocked, content gaps, mixed and unreviewed problems retained. No verdict relabeling or new calls. Same exclusions applied to before and latest.",
  "models": {
    "openai": {
      "originalUat": {
        "rawLatest": {
          "cases": 95,
          "verdicts": {
            "partial": 33,
            "pass": 47,
            "blocked": 3,
            "fail": 12
          },
          "strictPassPercent": 49.5
        },
        "adjustedBefore": {
          "cases": 94,
          "verdicts": {
            "fail": 18,
            "pass": 44,
            "partial": 29,
            "blocked": 3
          },
          "strictPassPercent": 46.8
        },
        "adjustedLatest": {
          "cases": 94,
          "verdicts": {
            "partial": 33,
            "pass": 47,
            "blocked": 3,
            "fail": 11
          },
          "strictPassPercent": 50.0
        },
        "excludedOccurrences": 1
      },
      "overall": {
        "rawLatest": {
          "cases": 279,
          "verdicts": {
            "pass": 163,
            "fail": 52,
            "blocked": 6,
            "partial": 58
          },
          "strictPassPercent": 58.4
        },
        "adjustedBefore": {
          "cases": 266,
          "verdicts": {
            "pass": 151,
            "fail": 57,
            "blocked": 6,
            "partial": 52
          },
          "strictPassPercent": 56.8
        },
        "adjustedLatest": {
          "cases": 266,
          "verdicts": {
            "pass": 163,
            "fail": 41,
            "blocked": 6,
            "partial": 56
          },
          "strictPassPercent": 61.3
        },
        "excludedOccurrences": 13
      }
    },
    "claude": {
      "originalUat": {
        "rawLatest": {
          "cases": 95,
          "verdicts": {
            "partial": 27,
            "fail": 13,
            "pass": 50,
            "blocked": 5
          },
          "strictPassPercent": 52.6
        },
        "adjustedBefore": {
          "cases": 90,
          "verdicts": {
            "fail": 13,
            "partial": 24,
            "blocked": 5,
            "pass": 48
          },
          "strictPassPercent": 53.3
        },
        "adjustedLatest": {
          "cases": 90,
          "verdicts": {
            "partial": 26,
            "fail": 9,
            "pass": 50,
            "blocked": 5
          },
          "strictPassPercent": 55.6
        },
        "excludedOccurrences": 5
      },
      "overall": {
        "rawLatest": {
          "cases": 279,
          "verdicts": {
            "pass": 177,
            "fail": 50,
            "partial": 46,
            "blocked": 6
          },
          "strictPassPercent": 63.4
        },
        "adjustedBefore": {
          "cases": 259,
          "verdicts": {
            "pass": 168,
            "fail": 48,
            "partial": 37,
            "blocked": 6
          },
          "strictPassPercent": 64.9
        },
        "adjustedLatest": {
          "cases": 259,
          "verdicts": {
            "pass": 177,
            "fail": 35,
            "partial": 41,
            "blocked": 6
          },
          "strictPassPercent": 68.3
        },
        "excludedOccurrences": 20
      }
    }
  },
  "exclusions": [
    {
      "model": "openai",
      "block": "feedback",
      "case": "F07",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/feedback/F07.md"
    },
    {
      "model": "claude",
      "block": "feedback",
      "case": "F07",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/feedback/F07.md"
    },
    {
      "model": "openai",
      "block": "feedback",
      "case": "F09",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/feedback/F09.md"
    },
    {
      "model": "claude",
      "block": "feedback",
      "case": "F09",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/feedback/F09.md"
    },
    {
      "model": "openai",
      "block": "ticket-precision",
      "case": "F07",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/ticket-precision/F07.md"
    },
    {
      "model": "claude",
      "block": "ticket-precision",
      "case": "F07",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/ticket-precision/F07.md"
    },
    {
      "model": "openai",
      "block": "ticket-precision",
      "case": "F09",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/ticket-precision/F09.md"
    },
    {
      "model": "claude",
      "block": "ticket-precision",
      "case": "F09",
      "category": "outdated prices",
      "reason": "Only the old fixed 2,995/4,995 prices cause failure; the recorded structured badge source returns 3,195/5,495. Other evaluated flow and cancellation criteria were satisfied.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/ticket-precision/F09.md"
    },
    {
      "model": "openai",
      "block": "ticket-precision",
      "case": "T01",
      "category": "outdated prices",
      "reason": "Diamond answer matches the recorded structured price, rather than the fixed historical 4,995 expected by this test.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/ticket-precision/T01.md"
    },
    {
      "model": "claude",
      "block": "ticket-precision",
      "case": "T01",
      "category": "outdated prices",
      "reason": "Diamond answer matches the recorded structured price, rather than the fixed historical 4,995 expected by this test.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/ticket-precision/T01.md"
    },
    {
      "model": "openai",
      "block": "uat-additions",
      "case": "C98",
      "category": "conflicting wording",
      "reason": "The blanket no-British-spelling criterion forbids the current approved app-owned personalised CTA. Rewrite the style expectation before scoring this case.",
      "sourceEvidence": "../../../apps/web/src/lib/agendaPlan.ts",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-ai-defect-retest/openai/transcripts/uat-additions/C98.md"
    },
    {
      "model": "claude",
      "block": "uat-additions",
      "case": "C98",
      "category": "conflicting wording",
      "reason": "The blanket no-British-spelling criterion forbids the current approved app-owned personalised CTA. Rewrite the style expectation before scoring this case.",
      "sourceEvidence": "../../../apps/web/src/lib/agendaPlan.ts",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/uat-additions/C98.md"
    },
    {
      "model": "openai",
      "block": "uat-additions",
      "case": "C104",
      "category": "outdated roster",
      "reason": "The frozen August six-speaker note contradicts the seven keynote-format sessions in the current live source. Latest answers match that source.",
      "sourceEvidence": "round3/source-refresh-current.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-ai-defect-retest/round3/openai/transcripts/uat-additions/C104.md"
    },
    {
      "model": "claude",
      "block": "uat-additions",
      "case": "C104",
      "category": "outdated roster",
      "reason": "The frozen August six-speaker note contradicts the seven keynote-format sessions in the current live source. Latest answers match that source.",
      "sourceEvidence": "round3/source-refresh-current.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-ai-defect-retest/round3/claude/transcripts/uat-additions/C104.md"
    },
    {
      "model": "openai",
      "block": "knowledge",
      "case": "K09",
      "category": "outdated contact policy",
      "reason": "Grading penalises support@unleash.ai instead of the former Customer Success/events contact route. Current policy uses Support for these non-SPEX requests.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "partial",
      "rawLatest": "partial",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/knowledge/K09.md"
    },
    {
      "model": "claude",
      "block": "knowledge",
      "case": "K09",
      "category": "outdated contact policy",
      "reason": "Grading penalises support@unleash.ai instead of the former Customer Success/events contact route. Current policy uses Support for these non-SPEX requests.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "partial",
      "rawLatest": "partial",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/knowledge/K09.md"
    },
    {
      "model": "openai",
      "block": "knowledge",
      "case": "K18",
      "category": "outdated contact policy",
      "reason": "Grading penalises support@unleash.ai instead of the former Customer Success/events contact route. Current policy uses Support for these non-SPEX requests.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/knowledge/K18.md"
    },
    {
      "model": "claude",
      "block": "knowledge",
      "case": "K18",
      "category": "outdated contact policy",
      "reason": "Grading penalises support@unleash.ai instead of the former Customer Success/events contact route. Current policy uses Support for these non-SPEX requests.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/knowledge/K18.md"
    },
    {
      "model": "openai",
      "block": "knowledge",
      "case": "K21",
      "category": "outdated contact policy",
      "reason": "Grading penalises support@unleash.ai instead of the former Customer Success/events contact route. Current policy uses Support for these non-SPEX requests.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "partial",
      "rawLatest": "partial",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/knowledge/K21.md"
    },
    {
      "model": "claude",
      "block": "knowledge",
      "case": "K21",
      "category": "outdated contact policy",
      "reason": "Grading penalises support@unleash.ai instead of the former Customer Success/events contact route. Current policy uses Support for these non-SPEX requests.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "partial",
      "rawLatest": "partial",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/knowledge/K21.md"
    },
    {
      "model": "openai",
      "block": "knowledge",
      "case": "K20",
      "category": "outdated privacy policy",
      "reason": "Old criteria require a privacy-page redirect and deny access; the current explicit policy says the bot cannot share stored information and directs the visitor to support@unleash.ai.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/knowledge/K20.md"
    },
    {
      "model": "claude",
      "block": "knowledge",
      "case": "K20",
      "category": "outdated privacy policy",
      "reason": "Old criteria require a privacy-page redirect and deny access; the current explicit policy says the bot cannot share stored information and directs the visitor to support@unleash.ai.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/knowledge/K20.md"
    },
    {
      "model": "claude",
      "block": "knowledge",
      "case": "K17",
      "category": "outdated contact policy",
      "reason": "The failure is solely the superseded complaints email expectation. OpenAI is retained because its sign-off also hits the duplicate-message warning.",
      "sourceEvidence": "../../../apps/web/src/lib/aiResponsePolicy.ts",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/knowledge/K17.md"
    },
    {
      "model": "openai",
      "block": "uat-additions",
      "case": "C97",
      "category": "outdated consent policy",
      "reason": "The raw failure penalises obtaining explicit final consent after the name arrives; that confirmation is required. Claude is retained for its separate missing agenda/dispatch problems.",
      "sourceEvidence": "review-notes.md",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/uat-additions/C97.md"
    },
    {
      "model": "openai",
      "block": "original-uat",
      "case": "C51",
      "category": "judge source-evidence gap",
      "reason": "The judge cannot verify prices/benefits from its context; the recorded structured badge source contains the quoted prices and tier summaries. Missing judge evidence is not demonstrated invention.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/openai/transcripts/original-uat/C51.md"
    },
    {
      "model": "claude",
      "block": "original-uat",
      "case": "C51",
      "category": "judge source-evidence gap",
      "reason": "The judge cannot verify prices/benefits from its context; the recorded structured badge source contains the quoted prices and tier summaries. Missing judge evidence is not demonstrated invention.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/badge-source-evidence.json",
      "rawBefore": "partial",
      "rawLatest": "partial",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/original-uat/C51.md"
    },
    {
      "model": "claude",
      "block": "original-uat",
      "case": "C78",
      "category": "judge source-evidence gap",
      "reason": "Recorded badgesLookup and FAQ observations substantiate the prices and conditional onsite policy penalised as invented.",
      "sourceEvidence": "registration-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-ai-defect-retest/round2/claude/transcripts/original-uat/C78.md"
    },
    {
      "model": "claude",
      "block": "original-uat",
      "case": "C54",
      "category": "judge source-evidence gap",
      "reason": "The fabrication allegation is contradicted by the case-specific retrieved package, startup story or transfer-policy evidence. This does not certify ongoing source freshness.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/additional-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/original-uat/C54.md"
    },
    {
      "model": "claude",
      "block": "original-uat",
      "case": "C77",
      "category": "judge source-evidence gap",
      "reason": "The fabrication allegation is contradicted by the case-specific retrieved package, startup story or transfer-policy evidence. This does not certify ongoing source freshness.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/additional-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/original-uat/C77.md"
    },
    {
      "model": "claude",
      "block": "original-uat",
      "case": "C80",
      "category": "judge source-evidence gap",
      "reason": "The fabrication allegation is contradicted by the case-specific retrieved package, startup story or transfer-policy evidence. This does not certify ongoing source freshness.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/additional-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/original-uat/C80.md"
    },
    {
      "model": "claude",
      "block": "knowledge",
      "case": "K04",
      "category": "judge source-evidence gap",
      "reason": "The retrieved Terms & Conditions contain the refund tiers and service charge; current Support routing supersedes the old contact criterion. OpenAI remains for its independent incomplete terms explanation.",
      "sourceEvidence": "../2026-10-05-full-model-comparison/refund-volunteer-source-evidence.json",
      "rawBefore": "fail",
      "rawLatest": "fail",
      "transcript": "../2026-10-05-full-model-comparison/claude/transcripts/knowledge/K04.md"
    },
    {
      "model": "claude",
      "block": "workflow-router",
      "case": "W06",
      "category": "judge summary-evidence gap",
      "reason": "The visible seeded agenda, current recipient summary, explicit confirmation and single accepted agenda-email capture satisfy the current summary requirement; reprinting the entire agenda was not required.",
      "sourceEvidence": "review-notes.md",
      "rawBefore": "fail",
      "rawLatest": "partial",
      "transcript": "../2026-10-05-ai-defect-retest/round2/claude/transcripts/workflow-router/W06.md"
    },
    {
      "model": "claude",
      "block": "flow-selector",
      "case": "W06",
      "category": "judge summary-evidence gap",
      "reason": "The visible seeded agenda, current recipient summary, explicit confirmation and single accepted agenda-email capture satisfy the current summary requirement; reprinting the entire agenda was not required.",
      "sourceEvidence": "review-notes.md",
      "rawBefore": "fail",
      "rawLatest": "partial",
      "transcript": "../2026-10-05-ai-defect-retest/round2/claude/transcripts/flow-selector/W06.md"
    }
  ]
}
