{
  "example_kind": "original synthetic manual planning case",
  "slug": "forecast-probability-score-example",
  "title": "Score a toy binary task forecast",
  "scenario": "Synthetic planning case: Two toy forecasts each assign 0.8 to an accepted proof being ready by its declared checkpoint. Observed labels are one for ready and zero for not ready. A comparison assigns 0.5 to both.",
  "layout": "evaluation",
  "dataset": {
    "heading": "Inspect the invented case records",
    "headers": [
      "Case",
      "Supplied probability",
      "Binary outcome",
      "Squared error at 0.8",
      "Squared error at 0.5"
    ],
    "rows": [
      [
        "A",
        "0.8",
        "1",
        "0.04",
        "0.25"
      ],
      [
        "B",
        "0.8",
        "0",
        "0.64",
        "0.25"
      ]
    ]
  },
  "derivation": "For each case use (p−y)². Mean at 0.8=(0.04+0.64)/2=0.34; at 0.5=(0.25+0.25)/2=0.25. Lower means less error under this explicit toy score only.",
  "result": "Under the stated mean squared probability-error rule, the 0.8 pair scores 0.34 and the 0.5 pair 0.25. This two-case comparison does not establish calibrated probabilities or a preferred real forecasting model.",
  "boundary": "This calculates a probability-error score, rather than comparing duration errors or promising task completion confidence.",
  "limitations": "No real probability model, sampling design or calibration evidence is supplied. Binary scoring uses the page’s own explicit mathematical rule.",
  "sources": [
    "forecast"
  ],
  "tasks": [
    [
      "Define the binary event",
      "Evaluation owner",
      "Ready means accepted by the specific checkpoint; labels are genuinely known."
    ],
    [
      "Keep probabilities and outcomes separate",
      "Recorder",
      "The 0.8 inputs were supplied before the outcomes, not fitted afterward."
    ],
    [
      "Interpret the sample score narrowly",
      "Reviewer",
      "Two cases compare the stated rule without a calibration or future-confidence claim."
    ]
  ],
  "faqs": [
    [
      "Does 0.8 mean this example proves an 80% success chance?",
      "No. It is a supplied toy forecast value; this worksheet does not validate it."
    ],
    [
      "Can unknown outcomes be scored as zero?",
      "No. Zero is an observed not-ready label under this definition; unknown remains unresolved."
    ]
  ],
  "method_references": [
    {
      "id": "forecast",
      "title": "Forecasting: Principles and Practice — accuracy",
      "url": "https://otexts.com/fpp3/accuracy.html",
      "scope": "Genuine held-out forecast and point-error evaluation context. Binary scoring examples use their own explicit toy definitions; no real task model is validated."
    }
  ]
}
