{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NPX6FUB62VISOHSEP3EEZOCWM7","short_pith_number":"pith:NPX6FUB6","schema_version":"1.0","canonical_sha256":"6befe2d03ed551271e447ec84cb85667dabfc1ef51500a69a71aeef111f0b237","source":{"kind":"arxiv","id":"2406.13805","version":1},"attestation_state":"computed","paper":{"title":"WikiContradict: A Benchmark for Evaluating LLMs on Real-World Knowledge Conflicts from Wikipedia","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alessandra Pascale, Elizabeth Daly, Inkit Padhi, Javier Carnerero-Cano, Prasanna Sattigeri, Radu Marinescu, Tigran Tchrakian, Yufang Hou","submitted_at":"2024-06-19T20:13:42Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has emerged as a promising solution to mitigate the limitations of large language models (LLMs), such as hallucinations and outdated information. However, it remains unclear how LLMs handle knowledge conflicts arising from different augmented retrieved passages, especially when these passages originate from the same source and have equal trustworthiness. In this work, we conduct a comprehensive evaluation of LLM-generated answers to questions that have varying answers based on contradictory passages from Wikipedia, a dataset widely regarded as a high-qualit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.13805","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-19T20:13:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a6dd5824694be97fe9c8f9d7a1d9ddcada8ace58474e431ece565a88f3a30e09","abstract_canon_sha256":"c0ed3dfaed2110cfa8520eaffaa0f99b7809df60b0fa6e9282d8cf8b5cb46303"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:34.641772Z","signature_b64":"mNglAKtyMu6sCb40N8vix5yccIjSbACB7yom/hN0zBRjynxTkHYaBmbPMPaTA1/1FAo8DzHZrp4vGzhc0003Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6befe2d03ed551271e447ec84cb85667dabfc1ef51500a69a71aeef111f0b237","last_reissued_at":"2026-07-05T08:34:34.641363Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:34.641363Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WikiContradict: A Benchmark for Evaluating LLMs on Real-World Knowledge Conflicts from Wikipedia","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alessandra Pascale, Elizabeth Daly, Inkit Padhi, Javier Carnerero-Cano, Prasanna Sattigeri, Radu Marinescu, Tigran Tchrakian, Yufang Hou","submitted_at":"2024-06-19T20:13:42Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has emerged as a promising solution to mitigate the limitations of large language models (LLMs), such as hallucinations and outdated information. However, it remains unclear how LLMs handle knowledge conflicts arising from different augmented retrieved passages, especially when these passages originate from the same source and have equal trustworthiness. In this work, we conduct a comprehensive evaluation of LLM-generated answers to questions that have varying answers based on contradictory passages from Wikipedia, a dataset widely regarded as a high-qualit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.13805","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.13805/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.13805","created_at":"2026-07-05T08:34:34.641421+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.13805v1","created_at":"2026-07-05T08:34:34.641421+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.13805","created_at":"2026-07-05T08:34:34.641421+00:00"},{"alias_kind":"pith_short_12","alias_value":"NPX6FUB62VIS","created_at":"2026-07-05T08:34:34.641421+00:00"},{"alias_kind":"pith_short_16","alias_value":"NPX6FUB62VISOHSE","created_at":"2026-07-05T08:34:34.641421+00:00"},{"alias_kind":"pith_short_8","alias_value":"NPX6FUB6","created_at":"2026-07-05T08:34:34.641421+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.27157","citing_title":"Detecting Is Not Resolving: The Monitoring Control Gap in Retrieval Augmented LLMs","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7","json":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7.json","graph_json":"https://pith.science/api/pith-number/NPX6FUB62VISOHSEP3EEZOCWM7/graph.json","events_json":"https://pith.science/api/pith-number/NPX6FUB62VISOHSEP3EEZOCWM7/events.json","paper":"https://pith.science/paper/NPX6FUB6"},"agent_actions":{"view_html":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7","download_json":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7.json","view_paper":"https://pith.science/paper/NPX6FUB6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.13805&json=true","fetch_graph":"https://pith.science/api/pith-number/NPX6FUB62VISOHSEP3EEZOCWM7/graph.json","fetch_events":"https://pith.science/api/pith-number/NPX6FUB62VISOHSEP3EEZOCWM7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7/action/storage_attestation","attest_author":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7/action/author_attestation","sign_citation":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7/action/citation_signature","submit_replication":"https://pith.science/pith/NPX6FUB62VISOHSEP3EEZOCWM7/action/replication_record"}},"created_at":"2026-07-05T08:34:34.641421+00:00","updated_at":"2026-07-05T08:34:34.641421+00:00"}