{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GCYLMTTF2KVMBR6CHH7YNLNDMH","short_pith_number":"pith:GCYLMTTF","schema_version":"1.0","canonical_sha256":"30b0b64e65d2aac0c7c239ff86ada361ebc95409937e89d450539094a4a049aa","source":{"kind":"arxiv","id":"2410.20247","version":2},"attestation_state":"computed","paper":{"title":"Model Equality Testing: Which Model Is This API Serving?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Carlos Guestrin, Irena Gao, Percy Liang","submitted_at":"2024-10-26T18:34:53Z","abstract_excerpt":"Users often interact with large language models through black-box inference APIs, both for closed- and open-weight models (e.g., Llama models are popularly accessed via Amazon Bedrock and Azure AI Studio). In order to cut costs or add functionality, API providers may quantize, watermark, or finetune the underlying model, changing the output distribution -- possibly without notifying users. We formalize detecting such distortions as Model Equality Testing, a two-sample testing problem, where the user collects samples from the API and a reference distribution and conducts a statistical test to s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.20247","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-26T18:34:53Z","cross_cats_sorted":[],"title_canon_sha256":"33742cb8b516a202c9c725709daeaaccbd4612896670b315a1c69315d00d2084","abstract_canon_sha256":"30a431cc25efbef961b809f690ba7a42e1401be7a344fc2be062bf0fed9c3689"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:21.127489Z","signature_b64":"+yZuVr4OXVtvV5ARlvLKrXRIhe+LeunedQKTbNOc2S3FJn3hMGD5ZjJJPaSg5EnkGkQ5J1tpv2Zoa5X5VtryBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30b0b64e65d2aac0c7c239ff86ada361ebc95409937e89d450539094a4a049aa","last_reissued_at":"2026-07-05T10:46:21.127025Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:21.127025Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Equality Testing: Which Model Is This API Serving?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Carlos Guestrin, Irena Gao, Percy Liang","submitted_at":"2024-10-26T18:34:53Z","abstract_excerpt":"Users often interact with large language models through black-box inference APIs, both for closed- and open-weight models (e.g., Llama models are popularly accessed via Amazon Bedrock and Azure AI Studio). In order to cut costs or add functionality, API providers may quantize, watermark, or finetune the underlying model, changing the output distribution -- possibly without notifying users. We formalize detecting such distortions as Model Equality Testing, a two-sample testing problem, where the user collects samples from the API and a reference distribution and conducts a statistical test to s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.20247","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.20247/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.20247","created_at":"2026-07-05T10:46:21.127081+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.20247v2","created_at":"2026-07-05T10:46:21.127081+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.20247","created_at":"2026-07-05T10:46:21.127081+00:00"},{"alias_kind":"pith_short_12","alias_value":"GCYLMTTF2KVM","created_at":"2026-07-05T10:46:21.127081+00:00"},{"alias_kind":"pith_short_16","alias_value":"GCYLMTTF2KVMBR6C","created_at":"2026-07-05T10:46:21.127081+00:00"},{"alias_kind":"pith_short_8","alias_value":"GCYLMTTF","created_at":"2026-07-05T10:46:21.127081+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03330","citing_title":"FLIPS: Instance-Fingerprinting for LLMs via Pseudo-random Sequences","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29524","citing_title":"KBF: Knowledge Boundary as Fingerprint for Language Model and Black-Box API Auditing","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02821","citing_title":"When Is the Same Model Not the Same Service? A Measurement Study of Hosted Open-Weight LLM APIs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02821","citing_title":"When Is the Same Model Not the Same Service? A Measurement Study of Hosted Open-Weight LLM APIs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02187","citing_title":"Rewriting the Response Path: Silent Tampering and Provider-Signed Defense in BYOK LLM Agents","ref_index":70,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH","json":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH.json","graph_json":"https://pith.science/api/pith-number/GCYLMTTF2KVMBR6CHH7YNLNDMH/graph.json","events_json":"https://pith.science/api/pith-number/GCYLMTTF2KVMBR6CHH7YNLNDMH/events.json","paper":"https://pith.science/paper/GCYLMTTF"},"agent_actions":{"view_html":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH","download_json":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH.json","view_paper":"https://pith.science/paper/GCYLMTTF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.20247&json=true","fetch_graph":"https://pith.science/api/pith-number/GCYLMTTF2KVMBR6CHH7YNLNDMH/graph.json","fetch_events":"https://pith.science/api/pith-number/GCYLMTTF2KVMBR6CHH7YNLNDMH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH/action/storage_attestation","attest_author":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH/action/author_attestation","sign_citation":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH/action/citation_signature","submit_replication":"https://pith.science/pith/GCYLMTTF2KVMBR6CHH7YNLNDMH/action/replication_record"}},"created_at":"2026-07-05T10:46:21.127081+00:00","updated_at":"2026-07-05T10:46:21.127081+00:00"}