{
  "example_kind": "original synthetic manual planning case",
  "slug": "forecast-base-rate-reference",
  "title": "Compare a task classifier with a simple base-rate rule",
  "scenario": "Synthetic planning case: A ten-case evaluation contains eight not-ready outcomes and two ready outcomes. Rule A always predicts not ready and advertises 80% accuracy as evidence that it detects readiness.",
  "layout": "evaluation",
  "dataset": {
    "heading": "Inspect the invented case records",
    "headers": [
      "Observed class",
      "Cases",
      "A predicts",
      "Correct",
      "Ready cases detected"
    ],
    "rows": [
      [
        "Not ready",
        "8",
        "Not ready",
        "8",
        "Not applicable"
      ],
      [
        "Ready",
        "2",
        "Not ready",
        "0",
        "0 of 2"
      ]
    ]
  },
  "derivation": "Overall accuracy is 8/10=80%. Ready-case detection is 0/2=0%. The high accuracy follows the observed class mix and the constant rule, not evidence that A recognizes a ready case.",
  "result": "Rule A matches eight of ten labels but detects zero of two ready cases. Report the majority-outcome baseline and the required ready-case result before treating overall accuracy as useful.",
  "boundary": "The artifact tests an apparently strong accuracy claim against an explicit majority-class reference.",
  "limitations": "This tiny synthetic cohort does not validate a classifier, class distribution or live task readiness model.",
  "sources": [
    "original"
  ],
  "tasks": [
    [
      "Preserve class counts",
      "Evaluation owner",
      "The eight/two split is retained with the ten-case cohort."
    ],
    [
      "Write the constant reference rule",
      "Reviewer",
      "Always-not-ready is an explicit comparator rather than hidden in an accuracy claim."
    ],
    [
      "Check the decision-relevant class",
      "Planning lead",
      "The two required ready detections remain visible before selecting the rule."
    ]
  ],
  "faqs": [
    [
      "Is 80% accuracy necessarily bad?",
      "Its usefulness depends on the actual decision and error consequences; it does not alone demonstrate readiness detection."
    ],
    [
      "Does the eight/two split apply to future tasks?",
      "No. It describes only these invented evaluation cases."
    ]
  ],
  "method_references": [
    {
      "id": "original",
      "title": "Original worked-case definitions",
      "scope": "Definitions, policy choices, records and calculations are authored for this worksheet. No external standard, statistical validation or live product measurement is claimed."
    }
  ]
}
