{
  "example_kind": "original synthetic manual planning case",
  "slug": "forecast-outcome-missing-policy",
  "title": "Keep unresolved outcomes outside scored verdicts",
  "scenario": "Synthetic planning case: A four-case binary evaluation has one correct prediction, one incorrect prediction and two unresolved outcome labels. A draft reports one success out of four.",
  "layout": "protocol",
  "dataset": {
    "heading": "Inspect the invented case records",
    "headers": [
      "Case",
      "Outcome evidence",
      "Score treatment"
    ],
    "rows": [
      [
        "A",
        "Known: prediction correct",
        "Correct"
      ],
      [
        "B",
        "Known: prediction incorrect",
        "Incorrect"
      ],
      [
        "C",
        "Unknown",
        "Unresolved, not scored"
      ],
      [
        "D",
        "Unknown",
        "Unresolved, not scored"
      ]
    ]
  },
  "derivation": "Known scored cases=1+1=2. Known-label accuracy=1/2=50%; label availability=2/4=50%. One-of-four would assert failures for both unknown labels.",
  "result": "Known-label accuracy is one of two, or 50%, with two of four labels unresolved. Do not silently count missing outcomes as failures or generalize the known subset to all cases.",
  "boundary": "The artifact distinguishes forecast-label availability from accuracy among observed labels.",
  "limitations": "This exercise supplies no missing-data model or population accuracy estimate. Unknown is an evidence state, not a negative outcome.",
  "sources": [
    "forecast"
  ],
  "tasks": [
    [
      "Define the evaluation-label rule",
      "Evaluation owner",
      "A scored outcome must be observed under the agreed binary event definition."
    ],
    [
      "Keep unresolved cases in the ledger",
      "Recorder",
      "C and D are not removed from visibility or relabelled failed."
    ],
    [
      "Publish denominator and coverage together",
      "Reviewer",
      "The one-of-two result and two missing labels accompany the interpretation."
    ]
  ],
  "faqs": [
    [
      "Can missingness be systematic?",
      "Yes. Harder or slower cases may be unresolved more often; the scored subset may not describe all forecasts."
    ],
    [
      "Should late labels replace the frozen score silently?",
      "No. Issue an explicit evaluation update with its label cutoff and changed cases."
    ]
  ],
  "method_references": [
    {
      "id": "forecast",
      "title": "Forecasting: Principles and Practice — accuracy",
      "url": "https://otexts.com/fpp3/accuracy.html",
      "scope": "Genuine held-out forecast and point-error evaluation context. Binary scoring examples use their own explicit toy definitions; no real task model is validated."
    }
  ]
}
