{
  "schema_version": "1.0",
  "id": "ARC-RN-003-CITATION-MAP",
  "edition_url": "https://agentresearchcommons.org/research/machine-evidence-intake-and-representation-fidelity/evidence-linked-edition.md",
  "immutable_source_url": "https://agentresearchcommons.org/research/machine-evidence-intake-and-representation-fidelity/full-report.md",
  "immutable_source_sha256": "96b0aa5be72a5c91ccc84be48d4a30fd4825e2fb1f947c45cf76581d8063766b",
  "status": "citation-portability-edition; scoped source checks; not a full independent replication or peer review",
  "attribution": {
    "tool_provider": "OpenAI",
    "tool": "OpenAI Deep Research",
    "resolved_model": "gpt-5-thinking",
    "model_reasoning_generation_metadata": "v5",
    "research_completed": "2026-09-24",
    "author_attribution": "Not independently established by the supplied file; provider/tool attribution is not author attribution.",
    "submitter": "Not publicly named; the source was supplied by the site owner.",
    "arc_editor": "ARC editorial preparation (agent-assisted); individual editor not attributed.",
    "reviewer": "No independent human reviewer identified.",
    "review_scope": "Primary-source bibliographic records and claim-to-source correspondence were checked for this edition. This is not a full independent replication or peer review of every underlying result."
  },
  "marker_map": {
    "turn1search4": "S1",
    "turn1search0": "S1",
    "turn8search1": "S2",
    "turn1search3": "S3",
    "turn9search0": "S4",
    "turn7search0": "S5",
    "turn10search5": "S5",
    "turn22view0": "S6",
    "turn11search12": "S7",
    "turn13view0": "S7",
    "turn22view1": "S8",
    "turn22view2": "S9",
    "turn22view3": "S10",
    "turn14search0": "S11",
    "turn14search4": "S11",
    "turn14search12": "S11",
    "turn14search2": "S12",
    "turn14search10": "S12",
    "turn15search1": "S13",
    "turn15search30": "S13",
    "turn16search0": "S14",
    "turn16search1": "S15",
    "turn16search2": "S16",
    "turn21search0": "S17",
    "turn17academia40": "S18",
    "turn19search0": "S19",
    "turn19search1": "S19",
    "turn19search2": "S20",
    "turn15search0": "S21"
  },
  "sources": [
    {
      "id": "S1",
      "authors": "Nelson F. Liu, Kevin Lin, John Hewitt, Ashwin Paranjape, Michele Bevilacqua, Fabio Petroni, Percy Liang",
      "year": 2024,
      "title": "Lost in the Middle: How Language Models Use Long Contexts",
      "venue": "Transactions of the Association for Computational Linguistics 12",
      "url": "https://doi.org/10.1162/tacl_a_00638",
      "identifier": "doi:10.1162/tacl_a_00638",
      "status": "Published journal article",
      "check": "Publisher record and article identity verified; claim alignment checked against the reported multi-document QA and key-value retrieval findings."
    },
    {
      "id": "S2",
      "authors": "Cheng-Ping Hsieh, Simeng Sun, Samuel Kriman, Shantanu Acharya, Dima Rekesh, Fei Jia, Yang Zhang, Boris Ginsburg",
      "year": 2024,
      "title": "RULER: What's the Real Context Size of Your Long-Context Language Models?",
      "venue": "COLM 2024",
      "url": "https://openreview.net/forum?id=kIoBbc76Sy",
      "identifier": "OpenReview: kIoBbc76Sy; arXiv:2404.06654",
      "status": "Conference paper",
      "check": "Conference paper and benchmark scope verified; claim alignment checked against task coverage and reported length degradation."
    },
    {
      "id": "S3",
      "authors": "Ali Modarressi, Hanieh Deilamsalehy, Franck Dernoncourt, Trung Bui, Ryan A. Rossi, Seunghyun Yoon, Hinrich Schuetze",
      "year": 2025,
      "title": "NoLiMa: Long-Context Evaluation Beyond Literal Matching",
      "venue": "Proceedings of Machine Learning Research 267, ICML 2025",
      "url": "https://proceedings.mlr.press/v267/modarressi25a.html",
      "identifier": "PMLR v267; arXiv:2502.05167",
      "status": "Peer-reviewed conference paper",
      "check": "Publisher record verified: 13 models were evaluated and 11 fell below half baseline at 32K. The supplied report says 12 and ten; corrected in this edition. Preserve the paper's baseline qualification."
    },
    {
      "id": "S4",
      "authors": "Yushi Bai et al.",
      "year": 2024,
      "title": "LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding",
      "venue": "ACL 2024",
      "url": "https://aclanthology.org/2024.acl-long.172/",
      "identifier": "doi:10.18653/v1/2024.acl-long.172",
      "status": "Peer-reviewed conference paper",
      "check": "ACL publisher record and task taxonomy verified; claim used only for task heterogeneity, not as evidence of a universal context limit."
    },
    {
      "id": "S5",
      "authors": "Jia He, Mukund Rungta, David Koleczek, Arshdeep Sekhon, Franklin X. Wang, Sadid Hasan",
      "year": 2024,
      "title": "Does Prompt Formatting Have Any Impact on LLM Performance?",
      "venue": "arXiv preprint",
      "url": "https://arxiv.org/abs/2411.10541",
      "identifier": "arXiv:2411.10541; doi:10.48550/arXiv.2411.10541",
      "status": "Preprint; no peer-reviewed venue identified in this check",
      "check": "arXiv metadata and abstract verified; the supplied ledger describes GPT-3.5/GPT-4 task-specific results. Do not label this a conference paper."
    },
    {
      "id": "S6",
      "authors": "Duc-Hai Nguyen, Vijayakumar Nanjappan, Barry O'Sullivan, Hoang D. Nguyen",
      "year": 2026,
      "title": "Questionnaire Meets LLM: A Benchmark and Empirical Study of Structural Skills for Understanding Questions and Responses",
      "venue": "LREC 2026",
      "url": "https://aclanthology.org/2026.lrec-1.371/",
      "identifier": "doi:10.63317/438xkvmy2xd9",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record, study design, cross-format scope, and reported results verified."
    },
    {
      "id": "S7",
      "authors": "Yuan Sui, Mengyu Zhou, Mingjie Zhou, Shi Han, Dongmei Zhang",
      "year": 2024,
      "title": "Table Meets LLM: Can Large Language Models Understand Structured Table Data? A Benchmark and Empirical Study",
      "venue": "WSDM 2024",
      "url": "https://doi.org/10.1145/3616855.3635752",
      "identifier": "doi:10.1145/3616855.3635752",
      "status": "Peer-reviewed conference paper",
      "check": "ACM bibliographic record and representation comparisons verified; wording remains task-specific."
    },
    {
      "id": "S8",
      "authors": "Masahiro Kato, Taka Kato",
      "year": 2026,
      "title": "Which Algorithm Specification Formats Help Language Models Implement Machine Learning Algorithms?",
      "venue": "arXiv preprint",
      "url": "https://arxiv.org/abs/2607.03158",
      "identifier": "arXiv:2607.03158",
      "status": "Preprint; no peer-reviewed venue identified in this check",
      "check": "Preprint identity and described controlled design checked. Findings must remain labeled preliminary and not treated as independently replicated."
    },
    {
      "id": "S9",
      "authors": "Netanel Eliav",
      "year": 2026,
      "title": "Prompt Design at Scale: How Format, Instruction Count, and Context Length Shape Instruction Adherence and Hallucination in Large Language Models",
      "venue": "arXiv preprint",
      "url": "https://arxiv.org/abs/2607.19257",
      "identifier": "arXiv:2607.19257; VeyraBench project",
      "status": "Preprint; independent replication not identified",
      "check": "Preprint record, design, and reported format/context results checked. Results are from one study/corpus and should not be generalized beyond its conditions."
    },
    {
      "id": "S10",
      "authors": "Zipeng Qiu, Chenyue Li, You Peng, Guangxin He, Binhang Yuan, Chen Wang",
      "year": 2026,
      "title": "TQA-Bench: Evaluating LLMs for Multi-Table Question Answering",
      "venue": "IEEE publication; arXiv version 2 dated 2026-06-05",
      "url": "https://arxiv.org/html/2411.19504",
      "identifier": "arXiv:2411.19504v2",
      "status": "2026 IEEE publication; arXiv full-text record",
      "check": "Current v2 identifies the 2026 IEEE publication. Full text directly reports tokenized lengths for Markdown, CSV, JSON, and HTML and serialization-performance comparisons. Claim basis confirmed; results remain specific to this table-QA benchmark."
    },
    {
      "id": "S11",
      "authors": "Tianyu Gao, Howard Yen, Jiatong Yu, Danqi Chen",
      "year": 2023,
      "title": "Enabling Large Language Models to Generate Text with Citations",
      "venue": "EMNLP 2023",
      "url": "https://aclanthology.org/2023.emnlp-main.398/",
      "identifier": "doi:10.18653/v1/2023.emnlp-main.398",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record and separation of fluency, correctness, and citation-quality evaluation verified."
    },
    {
      "id": "S12",
      "authors": "Cheng Niu et al.",
      "year": 2024,
      "title": "RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models",
      "venue": "ACL 2024",
      "url": "https://aclanthology.org/2024.acl-long.585/",
      "identifier": "doi:10.18653/v1/2024.acl-long.585",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record, corpus size phrasing ('nearly 18,000'), and annotation scope verified."
    },
    {
      "id": "S13",
      "authors": "Sewon Min et al.",
      "year": 2023,
      "title": "FActScore: Fine-grained Atomic Evaluation of Factual Precision in Long Form Text Generation",
      "venue": "EMNLP 2023",
      "url": "https://aclanthology.org/2023.emnlp-main.741/",
      "identifier": "doi:10.18653/v1/2023.emnlp-main.741",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record and atomic-fact evaluation method verified."
    },
    {
      "id": "S14",
      "authors": "Huiqiang Jiang, Qianhui Wu, Chin-Yew Lin, Yuqing Yang, Lili Qiu",
      "year": 2023,
      "title": "LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models",
      "venue": "EMNLP 2023",
      "url": "https://aclanthology.org/2023.emnlp-main.825/",
      "identifier": "doi:10.18653/v1/2023.emnlp-main.825",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record and prompt-compression method/results verified; does not establish preservation of all evidence properties."
    },
    {
      "id": "S15",
      "authors": "Huiqiang Jiang et al.",
      "year": 2024,
      "title": "LongLLMLingua: Accelerating and Enhancing LLMs in Long Context Scenarios via Prompt Compression",
      "venue": "ACL 2024",
      "url": "https://aclanthology.org/2024.acl-long.91/",
      "identifier": "doi:10.18653/v1/2024.acl-long.91",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record and reported NaturalQuestions example verified; result is task/model-specific."
    },
    {
      "id": "S16",
      "authors": "Vivek Sharma et al.",
      "year": 2024,
      "title": "RECOMP: Improving Retrieval-Augmented LMs with Context Compression and Selective Augmentation",
      "venue": "ICLR 2024",
      "url": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/bda88ed2892f5e61c9a9bf215c566913-Abstract-Conference.html",
      "identifier": "ICLR 2024 proceedings record",
      "status": "Peer-reviewed conference paper",
      "check": "Proceedings record and selective context-compression scope verified."
    },
    {
      "id": "S17",
      "authors": "Jeff Wu, Long Ouyang, Daniel M. Ziegler, Nisan Stiennon, Ryan Lowe, Jan Leike, Paul Christiano",
      "year": 2021,
      "title": "Recursively Summarizing Books with Human Feedback",
      "venue": "arXiv preprint and OpenAI research publication",
      "url": "https://arxiv.org/abs/2109.10862",
      "identifier": "arXiv:2109.10862",
      "status": "Preprint/research report; not represented here as a peer-reviewed venue paper",
      "check": "Primary report and method checked; it supports specialized recursive summarization, not lossless evidence preservation."
    },
    {
      "id": "S18",
      "authors": "Han Wang, Archiki Prasad, Elias Stengel-Eskin, Mohit Bansal",
      "year": 2025,
      "title": "AdaCAD: Adaptively Decoding to Balance Conflicts between Contextual and Parametric Knowledge",
      "venue": "NAACL 2025",
      "url": "https://aclanthology.org/2025.naacl-long.581/",
      "identifier": "ACL Anthology: 2025.naacl-long.581; arXiv:2409.07394",
      "status": "Peer-reviewed conference paper",
      "check": "Updated publication record and context/parametric-conflict scope verified; use 2025 venue citation rather than treating it solely as a 2024 preprint."
    },
    {
      "id": "S19",
      "authors": "Lianmin Zheng et al.",
      "year": 2023,
      "title": "Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena",
      "venue": "NeurIPS 2023 Datasets and Benchmarks Track",
      "url": "https://papers.neurips.cc/paper_files/paper/2023/hash/91f18a1287b398d378ef22505bf41832-Abstract-Datasets_and_Benchmarks.html",
      "identifier": "NeurIPS 2023 paper record; arXiv:2306.05685",
      "status": "Peer-reviewed conference paper",
      "check": "NeurIPS record and stated judge-bias analysis verified; reported limitations retained."
    },
    {
      "id": "S20",
      "authors": "Yang Liu, Dan Iter, Yichong Xu, Shuohang Wang, Ruochen Xu, Chenguang Zhu",
      "year": 2023,
      "title": "G-Eval: NLG Evaluation using GPT-4 with Better Human Alignment",
      "venue": "EMNLP 2023",
      "url": "https://aclanthology.org/2023.emnlp-main.153/",
      "identifier": "doi:10.18653/v1/2023.emnlp-main.153",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record, reported 0.514 Spearman correlation on summarization, and bias caveat verified."
    },
    {
      "id": "S21",
      "authors": "Taylor Berg-Kirkpatrick, David Burkett, Dan Klein",
      "year": 2012,
      "title": "An Empirical Investigation of Statistical Significance in NLP",
      "venue": "EMNLP-CoNLL 2012",
      "url": "https://aclanthology.org/D12-1091/",
      "identifier": "ACL Anthology: D12-1091",
      "status": "Peer-reviewed conference paper",
      "check": "ACL record verified. Its methodological relevance is general; it does not validate this dossier's proposed particular analysis plan."
    }
  ],
  "claims": [
    {
      "id": "C01",
      "sources": [
        "S5",
        "S6",
        "S7",
        "S8",
        "S9",
        "S10"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C02",
      "sources": [
        "S5",
        "S6",
        "S8",
        "S9"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C03",
      "sources": [
        "S5",
        "S6",
        "S7",
        "S8",
        "S9",
        "S10"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C04",
      "sources": [
        "S1",
        "S2",
        "S3"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C05",
      "sources": [
        "S2",
        "S3",
        "S4"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C06",
      "sources": [
        "S9"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C07",
      "sources": [
        "S9",
        "S10"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C08",
      "sources": [
        "S14",
        "S15",
        "S16"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C09",
      "sources": [
        "S14",
        "S15",
        "S16"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C10",
      "sources": [
        "S2",
        "S3"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C11",
      "sources": [
        "S13"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C12",
      "sources": [
        "S12"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C13",
      "sources": [
        "S11"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C14",
      "sources": [],
      "source_check_status": "no external source asserted; explicit research gap or proposed hypothesis"
    },
    {
      "id": "C15",
      "sources": [
        "S11"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C16",
      "sources": [
        "S7",
        "S8",
        "S9",
        "S10"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C17",
      "sources": [
        "S8",
        "S9",
        "S10"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C18",
      "sources": [
        "S17"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C19",
      "sources": [
        "S17"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C20",
      "sources": [
        "S14",
        "S15",
        "S16",
        "S17"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C21",
      "sources": [
        "S19",
        "S20"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C22",
      "sources": [
        "S18"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C23",
      "sources": [
        "S9",
        "S18"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C24",
      "sources": [
        "S9",
        "S10",
        "S14",
        "S15",
        "S16"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    },
    {
      "id": "C25",
      "sources": [],
      "source_check_status": "no external source asserted; explicit research gap or proposed hypothesis"
    },
    {
      "id": "C26",
      "sources": [
        "S2",
        "S3",
        "S11",
        "S12",
        "S13"
      ],
      "source_check_status": "source-linked; per-source check notes apply; not fully independently replicated"
    }
  ],
  "transformation_record": [
    "Corrected NoLiMa coauthor Seunghyun Yoon in the source ledger after a 2026-09-29 primary publisher-record check; earlier editions retain the prior spelling.",
    "Replaced the supplied NoLiMa model-count figures in this edition only: the publisher's final version reports 13 evaluated models and 11 below half of their short-context baseline at 32K, rather than 12 and ten in the supplied copy.",
    "Updated the TQA-Bench source record to its version-2 full text dated 2026-06-05; the linked 2026 text includes both token-length measurements and serialization-performance comparisons.",
    "Corrected source publication-status and bibliographic metadata where authoritative records support a correction (including S5 as a preprint, not a conference paper)."
  ]
}
