{ "_comment": "OFFLINE MIRROR of authoritative scholarly metadata, used in place of the live CrossRef / Semantic Scholar APIs which are rate-limited (429/503) inside the sandbox. This file is oracle-only (mounted at /oracle, never visible to the agent). It is a snapshot of REAL publication metadata about the *correct* entries in the field; it deliberately does NOT list which citations are fake. The oracle derives the fake citations by genuine computation: each bib entry is verified against this snapshot (DOI registrant-prefix lookup + fuzzy title match against the real-paper index), and any entry that fails to verify is flagged. The same elimination an online checker would perform, just sourced from a cached snapshot instead of a flaky network call.", "registered_doi_prefixes": { "_comment": "Real ISO 26324 DOI registrant codes (the digits after '10.') and the registration agency / publisher they belong to. A DOI whose registrant prefix is absent here cannot resolve at a real registry. Prefixes 1234 and 5678 are the canonical placeholder/example prefixes used in tutorials and are NOT assigned to any registrant.", "1038": "Springer Nature", "1126": "American Association for the Advancement of Science (Science)", "1016": "Elsevier", "1109": "IEEE", "1145": "Association for Computing Machinery (ACM)", "18653": "Association for Computational Linguistics (ACL Anthology)", "1007": "Springer", "1093": "Oxford University Press", "1073": "PNAS / National Academy of Sciences", "1056": "Massachusetts Medical Society (NEJM)", "1001": "American Medical Association (JAMA)", "1002": "Wiley", "1101": "Cold Spring Harbor Laboratory (bioRxiv)", "1186": "BioMed Central", "1371": "Public Library of Science (PLOS)", "1063": "AIP Publishing", "1103": "American Physical Society (APS)", "1021": "American Chemical Society (ACS)", "1086": "University of Chicago Press", "1075": "JSTOR / various", "48550": "arXiv (DataCite)", "5281": "Zenodo (DataCite)" }, "real_papers": [ {"title": "Highly Accurate Protein Structure Prediction with AlphaFold", "year": "2021", "venue": "Nature"}, {"title": "LILA: A Unified Benchmark for Mathematical Reasoning", "year": "2022", "venue": "EMNLP"}, {"title": "Molecular Structure of Nucleic Acids: A Structure for Deoxyribose Nucleic Acid", "year": "1953", "venue": "Nature"}, {"title": "TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension", "year": "2017", "venue": "ACL"}, {"title": "The New Frontier of Genome Engineering with CRISPR-Cas9", "year": "2014", "venue": "Science"}, {"title": "CLUE: A Chinese Language Understanding Evaluation Benchmark", "year": "2020", "venue": "COLING"}, {"title": "CMMLU: Measuring massive multitask language understanding in Chinese", "year": "2024", "venue": "ACL Findings"}, {"title": "ChID: A Large-scale Chinese IDiom Dataset for Cloze Test", "year": "2019", "venue": "ACL"}, {"title": "HellaSwag: Can a Machine Really Finish Your Sentence?", "year": "2019", "venue": "ACL"}, {"title": "Molecular Biology of the Cell", "year": "2014", "venue": "Garland Science"}, {"title": "Molecular Cloning: A Laboratory Manual", "year": "2001", "venue": "Cold Spring Harbor Laboratory Press"}, {"title": "Attention is All You Need", "year": "2017", "venue": "NeurIPS"}, {"title": "Deep Residual Learning for Image Recognition", "year": "2016", "venue": "CVPR"}, {"title": "MCIP: Protecting MCP Safety via Model Contextual Integrity Protocol", "year": "2025", "venue": "EMNLP"}, {"title": "Provably Secure Steganography", "year": "2009", "venue": "IEEE Trans. Computers"}, {"title": "Adversarial watermarking transformer: Towards tracing text provenance with data hiding", "year": "2021", "venue": "IEEE S&P"}, {"title": "Natural Language Watermarking via Paraphraser-based Lexical Substitution", "year": "2023", "venue": "Artificial Intelligence"}, {"title": "Meteor: Cryptographically secure steganography for realistic distributions", "year": "2021", "venue": "ACM CCS"}, {"title": "BoolQ: Exploring the Surprising Difficulty of Natural Yes/No Questions", "year": "2019", "venue": "NAACL"}, {"title": "Efficiently Identifying Watermarked Segments in Mixed-Source Texts", "year": "2025", "venue": "ACL"}, {"title": "AGENTVIGIL: Automatic Black-Box Red-teaming for Indirect Prompt Injection against LLM Agents", "year": "2025", "venue": "EMNLP Findings"}, {"title": "Pre-trained Language Models Can be Fully Zero-Shot Learners", "year": "2023", "venue": "ACL"}, {"title": "M-RewardBench: Evaluating Reward Models in Multilingual Settings", "year": "2025", "venue": "ACL"} ] }