{
  "example_kind": "original synthetic manual planning case",
  "slug": "forecast-error-unit-comparison",
  "title": "Compare forecast methods on matched cases",
  "scenario": "Synthetic planning case: Two methods predict the same four case durations in minutes. A’s absolute errors are 2,4,8,10. B has errors 1,3 for the first two cases, but its last two predictions are missing. Their reported MAEs are six and two.",
  "layout": "evaluation",
  "dataset": {
    "heading": "Inspect the invented case records",
    "headers": [
      "Case",
      "A absolute error in minutes",
      "B absolute error in minutes",
      "Matched comparison"
    ],
    "rows": [
      [
        "1",
        "2",
        "1",
        "Eligible pair"
      ],
      [
        "2",
        "4",
        "3",
        "Eligible pair"
      ],
      [
        "3",
        "8",
        "Missing prediction",
        "No pair"
      ],
      [
        "4",
        "10",
        "Missing prediction",
        "No pair"
      ]
    ]
  },
  "derivation": "A all-case MAE=(2+4+8+10)/4=6. B observed-case MAE=(1+3)/2=2. Matched-case A MAE=(2+4)/2=3, versus B’s 2 on exactly those same cases. Missing predictions prevent a four-case B MAE; no zero error is imputed.",
  "result": "The six-versus-two comparison uses different case sets. On the two matched cases A’s MAE is three and B’s two; retain that narrower observed comparison while leaving the four-case B result unavailable.",
  "boundary": "The artifact exercises matched-case comparison after a misleading same-metric aggregate ranking, rather than converting hours to minutes.",
  "limitations": "All errors are invented and already use the same unit. The matched subset gives a descriptive comparison only, with no imputation or model validation.",
  "sources": [
    "forecast"
  ],
  "tasks": [
    [
      "Freeze target, unit and case identities",
      "Evaluation owner",
      "Both methods target the same duration in minutes on cases 1–4."
    ],
    [
      "Build the matched evidence set",
      "Reviewer",
      "Only cases 1 and 2 have errors for both methods; missing predictions remain visible."
    ],
    [
      "Bound the comparative claim",
      "Planning lead",
      "B’s smaller observed MAE is stated for the two matched cases, not all four or future cases."
    ]
  ],
  "faqs": [
    [
      "Can I hide cases 3 and 4 after pairing?",
      "No. Preserve the excluded identities and the missing-prediction reason alongside the narrower comparison."
    ],
    [
      "Does matching two cases remove selection bias?",
      "No. Missing predictions may be informative; the small paired subset cannot validate general superiority."
    ]
  ],
  "method_references": [
    {
      "id": "forecast",
      "title": "Forecasting: Principles and Practice — accuracy",
      "url": "https://otexts.com/fpp3/accuracy.html",
      "scope": "Genuine held-out forecast and point-error evaluation context. Binary scoring examples use their own explicit toy definitions; no real task model is validated."
    }
  ]
}
