{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:3XLH4DXJ6KAHG4JFLYMICAUOTH","short_pith_number":"pith:3XLH4DXJ","schema_version":"1.0","canonical_sha256":"ddd67e0ee9f2807371255e1881028e99e14533432e5fcc18d6c7fc5716aa8b1c","source":{"kind":"arxiv","id":"2607.20090","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning for Large Language Model Selective Evidence Adoption from Contaminated Retrieval Results","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dongsheng Shi, Lichang Dai, Yanyu Chen, Yongyi Cui, Yue Li","submitted_at":"2026-07-22T12:45:04Z","abstract_excerpt":"Retrieval-augmented large language models frequently face contexts that interleave useful evidence with misleading statements or instruction-like content. Blanket refusal discards valid evidence, whereas uncritical adoption yields incorrect or unsafe answers. The ability to selectively adopt relevant information while rejecting deceptive or harmful content is therefore critical for reliable deployment in real-world retrieval settings. We introduce SelectBench, a controlled benchmark and training set for selective evidence adoption, and post-train Qwen3.5-4B directly with DAPO using either dete"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.20090","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-07-22T12:45:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4b2e4b4dbfc3495262069b2c8e63267e931a6eac0f7518483456b48fd35a561f","abstract_canon_sha256":"a37b576a41b2697b4136ac7e43c4268a78665d023e31e74136ea301cf566d02b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-23T01:25:00.866817Z","signature_b64":"Y51cOojic5GQW6plRJcv6yIWbpNWiRb5iXNRtk6Tx0UquT2FRAeN3G0OaORXsXWEBNUpP59EakHXGEIP8KntDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddd67e0ee9f2807371255e1881028e99e14533432e5fcc18d6c7fc5716aa8b1c","last_reissued_at":"2026-07-23T01:25:00.865936Z","signature_status":"signed_v1","first_computed_at":"2026-07-23T01:25:00.865936Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning for Large Language Model Selective Evidence Adoption from Contaminated Retrieval Results","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dongsheng Shi, Lichang Dai, Yanyu Chen, Yongyi Cui, Yue Li","submitted_at":"2026-07-22T12:45:04Z","abstract_excerpt":"Retrieval-augmented large language models frequently face contexts that interleave useful evidence with misleading statements or instruction-like content. Blanket refusal discards valid evidence, whereas uncritical adoption yields incorrect or unsafe answers. The ability to selectively adopt relevant information while rejecting deceptive or harmful content is therefore critical for reliable deployment in real-world retrieval settings. We introduce SelectBench, a controlled benchmark and training set for selective evidence adoption, and post-train Qwen3.5-4B directly with DAPO using either dete"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.20090","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.20090/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.20090","created_at":"2026-07-23T01:25:00.866405+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.20090v1","created_at":"2026-07-23T01:25:00.866405+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.20090","created_at":"2026-07-23T01:25:00.866405+00:00"},{"alias_kind":"pith_short_12","alias_value":"3XLH4DXJ6KAH","created_at":"2026-07-23T01:25:00.866405+00:00"},{"alias_kind":"pith_short_16","alias_value":"3XLH4DXJ6KAHG4JF","created_at":"2026-07-23T01:25:00.866405+00:00"},{"alias_kind":"pith_short_8","alias_value":"3XLH4DXJ","created_at":"2026-07-23T01:25:00.866405+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.03794","citing_title":"Evaluating LLMs in Database Scenarios: A Lifecycle Benchmark for Assessing Their Potential in Core Database Tasks","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH","json":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH.json","graph_json":"https://pith.science/api/pith-number/3XLH4DXJ6KAHG4JFLYMICAUOTH/graph.json","events_json":"https://pith.science/api/pith-number/3XLH4DXJ6KAHG4JFLYMICAUOTH/events.json","paper":"https://pith.science/paper/3XLH4DXJ"},"agent_actions":{"view_html":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH","download_json":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH.json","view_paper":"https://pith.science/paper/3XLH4DXJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.20090&json=true","fetch_graph":"https://pith.science/api/pith-number/3XLH4DXJ6KAHG4JFLYMICAUOTH/graph.json","fetch_events":"https://pith.science/api/pith-number/3XLH4DXJ6KAHG4JFLYMICAUOTH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH/action/storage_attestation","attest_author":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH/action/author_attestation","sign_citation":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH/action/citation_signature","submit_replication":"https://pith.science/pith/3XLH4DXJ6KAHG4JFLYMICAUOTH/action/replication_record"}},"created_at":"2026-07-23T01:25:00.866405+00:00","updated_at":"2026-07-23T01:25:00.866405+00:00"}