{
  "example_kind": "original synthetic manual planning case",
  "slug": "forecast-group-aggregate-calibration",
  "title": "Inspect forecast probabilities by relevant work group",
  "scenario": "Synthetic planning case: Twelve binary forecasts each assign 0.8 to success. Group A has ten cases and eight successes; group B has two cases and zero successes. A report hides groups inside a single result.",
  "layout": "evaluation",
  "dataset": {
    "heading": "Inspect the invented case records",
    "headers": [
      "Group",
      "Cases",
      "Predicted success probability per case",
      "Observed successes",
      "Observed fraction"
    ],
    "rows": [
      [
        "A",
        "10",
        "0.8",
        "8",
        "80%"
      ],
      [
        "B",
        "2",
        "0.8",
        "0",
        "0%"
      ],
      [
        "Combined",
        "12",
        "0.8",
        "8",
        "66.7% rounded"
      ]
    ]
  },
  "derivation": "Aggregate observed fraction is 8/12≈66.7%, versus a mean supplied prediction of 80%. Group A’s observed fraction happens to match 0.8; group B’s does not. Twelve outcomes, especially a two-case subgroup, cannot establish stable calibration.",
  "result": "Show group A at 8/10 and group B at 0/2 beside the aggregate 8/12. Neither the small group results nor the aggregate establish calibrated future forecasts.",
  "boundary": "The artifact exposes unequal-size group behavior concealed by an aggregate probability comparison.",
  "limitations": "All cases and probabilities are invented; no subgroup competence or product model claim is supported.",
  "sources": [
    "original"
  ],
  "tasks": [
    [
      "Retain the declared grouping",
      "Evaluation owner",
      "A and B membership is recorded independently of outcomes."
    ],
    [
      "Compute denominators visibly",
      "Reviewer",
      "Group results use ten and two; aggregate uses twelve without averaging group percentages equally."
    ],
    [
      "Bound the calibration interpretation",
      "Planning lead",
      "Mismatch is a descriptive finding needing more justified evaluation, not proof about people or future tasks."
    ]
  ],
  "faqs": [
    [
      "May I average 80% and 0% into 40%?",
      "Not as the case-weighted overall fraction: the groups have different sizes."
    ],
    [
      "Is A proven calibrated?",
      "No. A single small observed match at one supplied probability is insufficient."
    ]
  ],
  "method_references": [
    {
      "id": "original",
      "title": "Original worked-case definitions",
      "scope": "Definitions, policy choices, records and calculations are authored for this worksheet. No external standard, statistical validation or live product measurement is claimed."
    }
  ]
}
