{
  "story_id": "a2004f11c96fc504ee66fce84a708501",
  "desk": "drm3",
  "revision": 1,
  "published_at": "2026-09-03T04:00:00.000Z",
  "content_hash": "b3af8b517dc8a1c1bc11cd0fd3f3dda35ad3b3c8e7b8ae106c465f8beabceeba",
  "hash_basis": "sha256 over `headline\\ndek\\nprose`, plus `\\n` + the canonical citations JSON when any source is placed, plus `\\n#blog` for blogs",
  "basis": {
    "headline": "ExBind Benchmark Diagnoses Visual-to-Executable Mapping Errors",
    "dek": "ExBind benchmark isolates visual-to-executable correspondence errors in multimodal coding models.",
    "prose": "ExBind is a controlled diagnostic benchmark designed to isolate the visual-to-executable correspondence layer between semantic localization and action execution in multimodal coding and editing systems. [^1]\n\nResearchers demonstrated that Vision-Language Models (VLMs) often rewrite imperfect text into more plausible forms instead of transcribing it faithfully. [^2]\n\nQwen2.5-VL-3B achieved 98.4% candidate validity but only 76.4% exact accuracy on the benchmark. [^3]\n\nEvaluation of 15 systems spanning general-purpose VLMs, OCR-specialized VLMs, and traditional OCR pipelines showed that general-purpose VLMs degrade by up to 6.9 points in Word Error Rate under perturbation. [^4]\n\nProbing the Qwen3-VL-4B model layer-by-layer identified that rewriting fires only when a perturbed word's final layer FFN representation stays close to the original encoding. [^5]\n\nThe authors introduced FaithC4, a multilingual perturbation benchmark consisting of 1,455 single-page documents in English, Chinese, and Korean. [^6]",
    "cited": "[{\"statement\":\"ExBind is a controlled diagnostic benchmark designed to isolate the visual-to-executable correspondence layer between semantic localization and action execution in multimodal coding and editing systems.\",\"source\":\"takara.ai\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-01T14:52:57.000Z\",\"publisher_count\":1,\"sources\":[\"takara.ai\"]},{\"statement\":\"Researchers demonstrated that Vision-Language Models (VLMs) often rewrite imperfect text into more plausible forms instead of transcribing it faithfully.\",\"source\":\"arXiv.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T04:00:00.000Z\",\"publisher_count\":1,\"sources\":[\"arXiv.org\"]},{\"statement\":\"Qwen2.5-VL-3B achieved 98.4% candidate validity but only 76.4% exact accuracy on the benchmark.\",\"source\":\"takara.ai\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-01T14:52:57.000Z\",\"publisher_count\":1,\"sources\":[\"takara.ai\"]},{\"statement\":\"Evaluation of 15 systems spanning general-purpose VLMs, OCR-specialized VLMs, and traditional OCR pipelines showed that general-purpose VLMs degrade by up to 6.9 points in Word Error Rate under perturbation.\",\"source\":\"arXiv.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T04:00:00.000Z\",\"publisher_count\":1,\"sources\":[\"arXiv.org\"]},{\"statement\":\"Probing the Qwen3-VL-4B model layer-by-layer identified that rewriting fires only when a perturbed word's final layer FFN representation stays close to the original encoding.\",\"source\":\"arXiv.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T04:00:00.000Z\",\"publisher_count\":1,\"sources\":[\"arXiv.org\"]},{\"statement\":\"The authors introduced FaithC4, a multilingual perturbation benchmark consisting of 1,455 single-page documents in English, Chinese, and Korean.\",\"source\":\"arXiv.org\",\"instrument\":\"News\",\"claim_key\":null,\"published_at\":\"2026-09-03T04:00:00.000Z\",\"publisher_count\":1,\"sources\":[\"arXiv.org\"]}]",
    "kind": "news"
  },
  "receipt_verify": "Ed25519 over the dot-joined string `slice_hash.cursor_from.cursor_to.view.view_version.row_count`; public_key and sig are base64url of the raw 32-byte key / 64-byte signature",
  "receipt": null,
  "receipt_note": "this revision predates receipt-keeping (before v0.37.0); the filed row lives in the record",
  "generation_chain": {
    "wire": {
      "stream": "fountain_news",
      "story_id": "cbde9d5de56472a22c4aa2cbdcbcfbb8",
      "thread_id": "e962f251e45fcd24a7847500e84f3f57",
      "thread_label": "Qwen3-VL-4B",
      "novelty": "UPDATE",
      "content_hash": "6b92f7cc926fee51d23dddca29118b3646556f08cfa9a5c2880fb8a008140949",
      "last_published_at": "2026-09-03T04:00:00.000Z",
      "read_receipt": {
        "slice_hash": "a6974819026a155e1c79c99aba73ab29d5f9d0aaa9cffb05e88a08023354a690",
        "cursor_from": "eyJ0cyI6IjIwMjYtMDktMDNUMDM6MzI6MTkuMDAwMDAwWiIsImlkIjoiNzMzZDYyYTFiNGQwMmJmNjYzNTk3YjhmN2JhZDBiZTIiLCJ2IjoiMSJ9",
        "cursor_to": "eyJ0cyI6IjIwMjYtMDktMDNUMDQ6MDk6MDAuMDAwMDAwWiIsImlkIjoiNjQ5MjI3ZDNmNWQyMGQ3NmI1ZTE1NGNhODNlMTI0ZTMiLCJ2IjoiMSJ9",
        "view": "v_fountain_news",
        "view_version": "1",
        "row_count": 100,
        "window_days": 3,
        "bytes_scanned": 12568115,
        "credits": 8,
        "price_per_100_rows": 8,
        "sig": "k4fkL1dHRNIJwSawY8K6wWBrTSdTIRM-buyZr56gcruiCoH5_IdJUeMzPH7d51ZTXAE0Qe6xSFoDys5UGTzsAw",
        "public_key": "bMUigy8O0jOnBxQ4Sc-5lwhIZ8LQVAhxMbR7qESVuUE",
        "signer_path": "lakehouse/data-extract/v1",
        "alg": "Ed25519",
        "signed": true
      }
    },
    "written_at": "2026-09-03T06:46:30.370Z"
  },
  "cited_facts": [
    {
      "statement": "ExBind is a controlled diagnostic benchmark designed to isolate the visual-to-executable correspondence layer between semantic localization and action execution in multimodal coding and editing systems.",
      "source": "takara.ai",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-01T14:52:57.000Z",
      "publisher_count": 1,
      "sources": [
        "takara.ai"
      ]
    },
    {
      "statement": "Researchers demonstrated that Vision-Language Models (VLMs) often rewrite imperfect text into more plausible forms instead of transcribing it faithfully.",
      "source": "arXiv.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T04:00:00.000Z",
      "publisher_count": 1,
      "sources": [
        "arXiv.org"
      ]
    },
    {
      "statement": "Qwen2.5-VL-3B achieved 98.4% candidate validity but only 76.4% exact accuracy on the benchmark.",
      "source": "takara.ai",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-01T14:52:57.000Z",
      "publisher_count": 1,
      "sources": [
        "takara.ai"
      ]
    },
    {
      "statement": "Evaluation of 15 systems spanning general-purpose VLMs, OCR-specialized VLMs, and traditional OCR pipelines showed that general-purpose VLMs degrade by up to 6.9 points in Word Error Rate under perturbation.",
      "source": "arXiv.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T04:00:00.000Z",
      "publisher_count": 1,
      "sources": [
        "arXiv.org"
      ]
    },
    {
      "statement": "Probing the Qwen3-VL-4B model layer-by-layer identified that rewriting fires only when a perturbed word's final layer FFN representation stays close to the original encoding.",
      "source": "arXiv.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T04:00:00.000Z",
      "publisher_count": 1,
      "sources": [
        "arXiv.org"
      ]
    },
    {
      "statement": "The authors introduced FaithC4, a multilingual perturbation benchmark consisting of 1,455 single-page documents in English, Chinese, and Korean.",
      "source": "arXiv.org",
      "instrument": "News",
      "claim_key": null,
      "published_at": "2026-09-03T04:00:00.000Z",
      "publisher_count": 1,
      "sources": [
        "arXiv.org"
      ]
    }
  ],
  "note": "A signature proves who filed this and that it has not changed since. It never makes a claim true."
}