{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3T24765ZESZO2CGAOW3PAAB7TG","short_pith_number":"pith:3T24765Z","schema_version":"1.0","canonical_sha256":"dcf5cffbb924b2ed08c075b6f0003f99a219b8708d4746319b6fe0a6cb88bdac","source":{"kind":"arxiv","id":"2406.15968","version":2},"attestation_state":"computed","paper":{"title":"ReCaLL: Membership Inference via Relative Conditional Log-Likelihoods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bhuwan Dhingra, Jian Pei, Junlin Wang, Minxing Zhang, Neil Zhenqiang Gong, Rong Ge, Roy Xie, Ruomin Huang","submitted_at":"2024-06-23T00:23:13Z","abstract_excerpt":"The rapid scaling of large language models (LLMs) has raised concerns about the transparency and fair use of the data used in their pretraining. Detecting such content is challenging due to the scale of the data and limited exposure of each instance during training. We propose ReCaLL (Relative Conditional Log-Likelihood), a novel membership inference attack (MIA) to detect LLMs' pretraining data by leveraging their conditional language modeling capabilities. ReCaLL examines the relative change in conditional log-likelihoods when prefixing target data points with non-member context. Our empiric"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15968","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-23T00:23:13Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7127610c61a8b1e5ab7419c7e63d9e519177b7adc26a072ffeedac3ad1968fc9","abstract_canon_sha256":"0b3034df83a8d7cdead1136f77016e07f9016a1ec84639c870dbfac05e2f9b25"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:58.245632Z","signature_b64":"EgVEfvJN1N9KxfQRbAeJOcWh7hwG72p0/NwY3m4xNbbiwE83nSglereI7d+4F9SYe1+tUGqJOmH2ZxDJxtMLBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dcf5cffbb924b2ed08c075b6f0003f99a219b8708d4746319b6fe0a6cb88bdac","last_reissued_at":"2026-07-05T11:07:58.244910Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:58.244910Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReCaLL: Membership Inference via Relative Conditional Log-Likelihoods","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bhuwan Dhingra, Jian Pei, Junlin Wang, Minxing Zhang, Neil Zhenqiang Gong, Rong Ge, Roy Xie, Ruomin Huang","submitted_at":"2024-06-23T00:23:13Z","abstract_excerpt":"The rapid scaling of large language models (LLMs) has raised concerns about the transparency and fair use of the data used in their pretraining. Detecting such content is challenging due to the scale of the data and limited exposure of each instance during training. We propose ReCaLL (Relative Conditional Log-Likelihood), a novel membership inference attack (MIA) to detect LLMs' pretraining data by leveraging their conditional language modeling capabilities. ReCaLL examines the relative change in conditional log-likelihoods when prefixing target data points with non-member context. Our empiric"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15968","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15968/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15968","created_at":"2026-07-05T11:07:58.244982+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15968v2","created_at":"2026-07-05T11:07:58.244982+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15968","created_at":"2026-07-05T11:07:58.244982+00:00"},{"alias_kind":"pith_short_12","alias_value":"3T24765ZESZO","created_at":"2026-07-05T11:07:58.244982+00:00"},{"alias_kind":"pith_short_16","alias_value":"3T24765ZESZO2CGA","created_at":"2026-07-05T11:07:58.244982+00:00"},{"alias_kind":"pith_short_8","alias_value":"3T24765Z","created_at":"2026-07-05T11:07:58.244982+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17464","citing_title":"CheckMIABench: Firm Foundations For Membership Inference Attacks on Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03121","citing_title":"Lost in Modality: Evaluating the Effectiveness of Text-Based Membership Inference Attacks on Large Multimodal Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14045","citing_title":"Auditing Data Membership in Reinforcement Learning With Verifiable Rewards","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG","json":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG.json","graph_json":"https://pith.science/api/pith-number/3T24765ZESZO2CGAOW3PAAB7TG/graph.json","events_json":"https://pith.science/api/pith-number/3T24765ZESZO2CGAOW3PAAB7TG/events.json","paper":"https://pith.science/paper/3T24765Z"},"agent_actions":{"view_html":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG","download_json":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG.json","view_paper":"https://pith.science/paper/3T24765Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15968&json=true","fetch_graph":"https://pith.science/api/pith-number/3T24765ZESZO2CGAOW3PAAB7TG/graph.json","fetch_events":"https://pith.science/api/pith-number/3T24765ZESZO2CGAOW3PAAB7TG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG/action/storage_attestation","attest_author":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG/action/author_attestation","sign_citation":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG/action/citation_signature","submit_replication":"https://pith.science/pith/3T24765ZESZO2CGAOW3PAAB7TG/action/replication_record"}},"created_at":"2026-07-05T11:07:58.244982+00:00","updated_at":"2026-07-05T11:07:58.244982+00:00"}