{
  "run_id": "20260809-release2-theme-qa-v2",
  "captured_at": "2026-08-09T18:33:13Z",
  "classifier_version": "review-themes-v1.0.1",
  "model": "gpt-4o-mini-2024-07-18",
  "sample_method": "deterministic stratified sample of predicted positives, predicted negatives, and rating strata",
  "sample_size": 483,
  "corpus_checksum": "11633c6469e3874be4271bccc443b169655e65d3e50ad20a480d3037cd91f715",
  "sample_checksum": "041733ee5cb690d64bd7d5ebaf88c86889cf09a13439109df2cf14a4d1b20750",
  "privacy": "Reviewer identity omitted; URLs, emails, and phone numbers redacted; text truncated to 1,500 characters; no review text retained in this result.",
  "thresholds": {
    "minimum_precision": 0.8,
    "minimum_reviewed_predicted_positives": 10
  },
  "theme_metrics": {
    "communication": {
      "label": "Communication",
      "true_positive": 53,
      "false_positive": 2,
      "false_negative": 174,
      "true_negative": 254,
      "predicted_positive_reviewed": 55,
      "precision": 0.9636,
      "recall": 0.2335,
      "f1": 0.3759,
      "publication_qa_pass": true
    },
    "responsiveness_delays": {
      "label": "Responsiveness and delays",
      "true_positive": 36,
      "false_positive": 2,
      "false_negative": 102,
      "true_negative": 343,
      "predicted_positive_reviewed": 38,
      "precision": 0.9474,
      "recall": 0.2609,
      "f1": 0.4091,
      "publication_qa_pass": true
    },
    "staff_accessibility": {
      "label": "Staff accessibility",
      "true_positive": 12,
      "false_positive": 15,
      "false_negative": 28,
      "true_negative": 428,
      "predicted_positive_reviewed": 27,
      "precision": 0.4444,
      "recall": 0.3,
      "f1": 0.3582,
      "publication_qa_pass": false
    },
    "case_handoffs": {
      "label": "Case handoffs",
      "true_positive": 15,
      "false_positive": 9,
      "false_negative": 12,
      "true_negative": 447,
      "predicted_positive_reviewed": 24,
      "precision": 0.625,
      "recall": 0.5556,
      "f1": 0.5882,
      "publication_qa_pass": false
    },
    "intake_experience": {
      "label": "Intake experience",
      "true_positive": 4,
      "false_positive": 23,
      "false_negative": 6,
      "true_negative": 450,
      "predicted_positive_reviewed": 27,
      "precision": 0.1481,
      "recall": 0.4,
      "f1": 0.2162,
      "publication_qa_pass": false
    },
    "fees": {
      "label": "Fees",
      "true_positive": 6,
      "false_positive": 18,
      "false_negative": 10,
      "true_negative": 449,
      "predicted_positive_reviewed": 24,
      "precision": 0.25,
      "recall": 0.375,
      "f1": 0.3,
      "publication_qa_pass": false
    },
    "process_clarity": {
      "label": "Process clarity",
      "true_positive": 8,
      "false_positive": 19,
      "false_negative": 30,
      "true_negative": 426,
      "predicted_positive_reviewed": 27,
      "precision": 0.2963,
      "recall": 0.2105,
      "f1": 0.2462,
      "publication_qa_pass": false
    },
    "follow_up": {
      "label": "Follow-up",
      "true_positive": 17,
      "false_positive": 14,
      "false_negative": 27,
      "true_negative": 425,
      "predicted_positive_reviewed": 31,
      "precision": 0.5484,
      "recall": 0.3864,
      "f1": 0.4533,
      "publication_qa_pass": false
    },
    "timeliness": {
      "label": "Timeliness",
      "true_positive": 11,
      "false_positive": 15,
      "false_negative": 20,
      "true_negative": 437,
      "predicted_positive_reviewed": 26,
      "precision": 0.4231,
      "recall": 0.3548,
      "f1": 0.386,
      "publication_qa_pass": false
    },
    "compassion": {
      "label": "Compassion",
      "true_positive": 31,
      "false_positive": 8,
      "false_negative": 75,
      "true_negative": 369,
      "predicted_positive_reviewed": 39,
      "precision": 0.7949,
      "recall": 0.2925,
      "f1": 0.4276,
      "publication_qa_pass": false
    },
    "outcome_language": {
      "label": "Outcome language (unverified)",
      "true_positive": 36,
      "false_positive": 8,
      "false_negative": 95,
      "true_negative": 344,
      "predicted_positive_reviewed": 44,
      "precision": 0.8182,
      "recall": 0.2748,
      "f1": 0.4114,
      "publication_qa_pass": true
    }
  },
  "limitations": [
    "OpenAI labels are an independent QA reference, not perfect ground truth.",
    "The stratified sample estimates precision for explicit rule matches and is not a prevalence estimate.",
    "Themes that fail the QA gate are retained privately for improvement and suppressed from public office records."
  ]
}
