{
  "story_id": "68cecff6cdd42559dbe96d52063c1ad9",
  "desk": "drm3",
  "revision": 2,
  "published_at": "2026-09-05T00:12:32.000Z",
  "content_hash": "17c6a6623aa2a97a22b9b08a9656148c3d00e13a03386ce262cc6d35bafa74a1",
  "hash_basis": "sha256 over `headline\\ndek\\nprose`, plus `\\n` + the canonical citations JSON when any source is placed, plus `\\n#blog` for blogs",
  "basis": {
    "headline": "OpenAI Retractions and Metric Shifts Spark Benchmaxxing Concerns for GPT-6 Astra",
    "dek": "OpenAI updated GPT-6 Astra benchmarks post-launch, raising questions about metric manipulation amid industry competition.",
    "prose": "OpenAI updated several evaluation benchmarks for its GPT-6 Astra model after publishing a blog post on September 3 that experienced deployment issues and a brief retraction. [^1]\n\nMengqi Yuan from the XLANG Lab at the University of Hong Kong presented OSWorld 2.0, a benchmark consisting of 108 long-horizon, real-world computer-use workflows spanning 31 self-hosted websites and professional desktop applications. [^2] An OpenAI spokesperson stated that fixes were made to the launch blog to ensure numbers represented the best estimate of available model performance for meaningful user comparisons. [^3] Researchers from the Stanford Intelligent Systems Laboratory and Stanford Trustworthy AI Lab suggested that the rapid changes in metrics indicate 'benchmaxxing,' a practice of maximizing scores by re-running evaluations with different conditions. [^4]\n\nThe reported hallucination rate for GPT-6 Astra changed from 4.2% in early snapshots to 2% in a later version before reverting back to 4.2%. [^5] Claude Fable 5.1, released less than 24 hours before the presentation, pushed OSWorld 2.0 partial-credit scores above 60% and binary completion above 45%. [^6] The average task in OSWorld 2.0 requires more than 300 agent steps, and 69.6% of tasks take a skilled human over an hour to complete. [^7]\n\nThe best system evaluated in the paper completes only 20.6% of OSWorld 2.0 tasks outright, with a partial-credit score of 54.8%. [^8]",
    "cited": "[{\"statement\":\"OpenAI updated several evaluation benchmarks for its GPT-6 Astra model after publishing a blog post on September 3 that experienced deployment issues and a brief retraction.\",\"source\":\"fortune.com\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-05T00:12:32.000Z\",\"publisher_count\":1,\"sources\":[\"fortune.com\"]},{\"statement\":\"Mengqi Yuan from the XLANG Lab at the University of Hong Kong presented OSWorld 2.0, a benchmark consisting of 108 long-horizon, real-world computer-use workflows spanning 31 self-hosted websites and professional desktop applications.\",\"source\":\"Snorkel AI\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T15:08:04.000Z\",\"publisher_count\":1,\"sources\":[\"Snorkel AI\"]},{\"statement\":\"An OpenAI spokesperson stated that fixes were made to the launch blog to ensure numbers represented the best estimate of available model performance for meaningful user comparisons.\",\"source\":\"fortune.com\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-05T00:12:32.000Z\",\"publisher_count\":1,\"sources\":[\"fortune.com\"]},{\"statement\":\"Researchers from the Stanford Intelligent Systems Laboratory and Stanford Trustworthy AI Lab suggested that the rapid changes in metrics indicate 'benchmaxxing,' a practice of maximizing scores by re-running evaluations with different conditions.\",\"source\":\"fortune.com\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-05T00:12:32.000Z\",\"publisher_count\":1,\"sources\":[\"fortune.com\"]},{\"statement\":\"The reported hallucination rate for GPT-6 Astra changed from 4.2% in early snapshots to 2% in a later version before reverting back to 4.2%.\",\"source\":\"fortune.com\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-05T00:12:32.000Z\",\"publisher_count\":1,\"sources\":[\"fortune.com\"]},{\"statement\":\"Claude Fable 5.1, released less than 24 hours before the presentation, pushed OSWorld 2.0 partial-credit scores above 60% and binary completion above 45%.\",\"source\":\"Snorkel AI\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T15:08:04.000Z\",\"publisher_count\":1,\"sources\":[\"Snorkel AI\"]},{\"statement\":\"The average task in OSWorld 2.0 requires more than 300 agent steps, and 69.6% of tasks take a skilled human over an hour to complete.\",\"source\":\"Snorkel AI\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T15:08:04.000Z\",\"publisher_count\":1,\"sources\":[\"Snorkel AI\"]},{\"statement\":\"The best system evaluated in the paper completes only 20.6% of OSWorld 2.0 tasks outright, with a partial-credit score of 54.8%.\",\"source\":\"Snorkel AI\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T15:08:04.000Z\",\"publisher_count\":1,\"sources\":[\"Snorkel AI\"]}]",
    "kind": "news"
  },
  "receipt_verify": "Ed25519 over the dot-joined string `slice_hash.cursor_from.cursor_to.view.view_version.row_count`; public_key and sig are base64url of the raw 32-byte key / 64-byte signature",
  "receipt": null,
  "receipt_note": "this revision predates receipt-keeping (before v0.37.0); the filed row lives in the record",
  "generation_chain": {
    "wire": {
      "stream": "fountain_news",
      "story_id": "4a67d9289cd26bedb2ecd5ff71eb0236",
      "thread_id": "fc587b7ab1a77fd93486f6b01a847e80",
      "thread_label": "Snorkel AI",
      "novelty": "UPDATE",
      "content_hash": "13a6810a232ba6dc263e557315e28f6cf9662930039bd0193a22a20d66ad712a",
      "last_published_at": "2026-09-05T00:12:32.000Z",
      "read_receipt": {
        "slice_hash": "1c32200f02da81d2f4fae1e63fdeff66d61d30930dc085ec304ca43c2002f496",
        "cursor_from": "eyJ0cyI6IjIwMjYtMDktMDRUMjM6NTY6MjQuMDAwMDAwWiIsImlkIjoiZmJlYjk2OGVkMGRiMThiYjAxOWNhMGUyYTliY2VmZGYiLCJ2IjoiMSJ9",
        "cursor_to": "eyJ0cyI6IjIwMjYtMDktMDVUMDA6NDE6MDkuMDAwMDAwWiIsImlkIjoiZmQzNDkyZGMwZWEyYWMwOTg0NDE3Njc1NTMxMDY3ODgiLCJ2IjoiMSJ9",
        "view": "v_fountain_news",
        "view_version": "1",
        "row_count": 100,
        "window_days": 3,
        "bytes_scanned": 11524527,
        "credits": 8,
        "price_per_100_rows": 8,
        "sig": "rAN0L5850xQrp_pYcXUYWw7VZbMPJs03p-RxxiNrUBt85iiIHmjAvD7xDyNEvs0SjGazzkPEyXgMxf10KxfNAQ",
        "public_key": "bMUigy8O0jOnBxQ4Sc-5lwhIZ8LQVAhxMbR7qESVuUE",
        "signer_path": "lakehouse/data-extract/v1",
        "alg": "Ed25519",
        "signed": true
      },
      "compose": 3,
      "recomposed_at": "2026-09-05T18:55:54.540Z"
    },
    "written_at": "2026-09-05T06:24:15.501Z"
  },
  "cited_facts": [
    {
      "statement": "OpenAI updated several evaluation benchmarks for its GPT-6 Astra model after publishing a blog post on September 3 that experienced deployment issues and a brief retraction.",
      "source": "fortune.com",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-05T00:12:32.000Z",
      "publisher_count": 1,
      "sources": [
        "fortune.com"
      ]
    },
    {
      "statement": "Mengqi Yuan from the XLANG Lab at the University of Hong Kong presented OSWorld 2.0, a benchmark consisting of 108 long-horizon, real-world computer-use workflows spanning 31 self-hosted websites and professional desktop applications.",
      "source": "Snorkel AI",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T15:08:04.000Z",
      "publisher_count": 1,
      "sources": [
        "Snorkel AI"
      ]
    },
    {
      "statement": "An OpenAI spokesperson stated that fixes were made to the launch blog to ensure numbers represented the best estimate of available model performance for meaningful user comparisons.",
      "source": "fortune.com",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-05T00:12:32.000Z",
      "publisher_count": 1,
      "sources": [
        "fortune.com"
      ]
    },
    {
      "statement": "Researchers from the Stanford Intelligent Systems Laboratory and Stanford Trustworthy AI Lab suggested that the rapid changes in metrics indicate 'benchmaxxing,' a practice of maximizing scores by re-running evaluations with different conditions.",
      "source": "fortune.com",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-05T00:12:32.000Z",
      "publisher_count": 1,
      "sources": [
        "fortune.com"
      ]
    },
    {
      "statement": "The reported hallucination rate for GPT-6 Astra changed from 4.2% in early snapshots to 2% in a later version before reverting back to 4.2%.",
      "source": "fortune.com",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-05T00:12:32.000Z",
      "publisher_count": 1,
      "sources": [
        "fortune.com"
      ]
    },
    {
      "statement": "Claude Fable 5.1, released less than 24 hours before the presentation, pushed OSWorld 2.0 partial-credit scores above 60% and binary completion above 45%.",
      "source": "Snorkel AI",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T15:08:04.000Z",
      "publisher_count": 1,
      "sources": [
        "Snorkel AI"
      ]
    },
    {
      "statement": "The average task in OSWorld 2.0 requires more than 300 agent steps, and 69.6% of tasks take a skilled human over an hour to complete.",
      "source": "Snorkel AI",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T15:08:04.000Z",
      "publisher_count": 1,
      "sources": [
        "Snorkel AI"
      ]
    },
    {
      "statement": "The best system evaluated in the paper completes only 20.6% of OSWorld 2.0 tasks outright, with a partial-credit score of 54.8%.",
      "source": "Snorkel AI",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T15:08:04.000Z",
      "publisher_count": 1,
      "sources": [
        "Snorkel AI"
      ]
    }
  ],
  "note": "A signature proves who filed this and that it has not changed since. It never makes a claim true."
}