{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:X3VHHN36B6NBOOG2CS7JAVOCXK","short_pith_number":"pith:X3VHHN36","schema_version":"1.0","canonical_sha256":"beea73b77e0f9a1738da14be9055c2bab9ab48c13e87eb753edb8adf4e980e8e","source":{"kind":"arxiv","id":"2304.04736","version":3},"attestation_state":"computed","paper":{"title":"On the Possibilities of AI-Generated Text Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amrit Singh Bedi, Bang An, Dinesh Manocha, Furong Huang, Sicheng Zhu, Souradip Chakraborty","submitted_at":"2023-04-10T17:47:39Z","abstract_excerpt":"Our work addresses the critical issue of distinguishing text generated by Large Language Models (LLMs) from human-produced text, a task essential for numerous applications. Despite ongoing debate about the feasibility of such differentiation, we present evidence supporting its consistent achievability, except when human and machine text distributions are indistinguishable across their entire support. Drawing from information theory, we argue that as machine-generated text approximates human-like quality, the sample size needed for detection increases. We establish precise sample complexity bou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.04736","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-10T17:47:39Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e733862a0b7400ef4a8755b652c2e71fbcecfe91f877fb257522c6cd96233e68","abstract_canon_sha256":"b3e5c6738b58b3e672f58fae18b1cc02812a87df53465afbe4accc0d22e79836"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:56:32.378068Z","signature_b64":"MJTUDCFEJ/FD54UGW/3g8xxb10jr4eHtycCBaE4awUHG6OyA2luJN7/VCO4BiUYgd0dxP9TUeNdII0wdOdFUDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"beea73b77e0f9a1738da14be9055c2bab9ab48c13e87eb753edb8adf4e980e8e","last_reissued_at":"2026-07-05T06:56:32.377579Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:56:32.377579Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On the Possibilities of AI-Generated Text Detection","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amrit Singh Bedi, Bang An, Dinesh Manocha, Furong Huang, Sicheng Zhu, Souradip Chakraborty","submitted_at":"2023-04-10T17:47:39Z","abstract_excerpt":"Our work addresses the critical issue of distinguishing text generated by Large Language Models (LLMs) from human-produced text, a task essential for numerous applications. Despite ongoing debate about the feasibility of such differentiation, we present evidence supporting its consistent achievability, except when human and machine text distributions are indistinguishable across their entire support. Drawing from information theory, we argue that as machine-generated text approximates human-like quality, the sample size needed for detection increases. We establish precise sample complexity bou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.04736","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.04736/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.04736","created_at":"2026-07-05T06:56:32.377636+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.04736v3","created_at":"2026-07-05T06:56:32.377636+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.04736","created_at":"2026-07-05T06:56:32.377636+00:00"},{"alias_kind":"pith_short_12","alias_value":"X3VHHN36B6NB","created_at":"2026-07-05T06:56:32.377636+00:00"},{"alias_kind":"pith_short_16","alias_value":"X3VHHN36B6NBOOG2","created_at":"2026-07-05T06:56:32.377636+00:00"},{"alias_kind":"pith_short_8","alias_value":"X3VHHN36","created_at":"2026-07-05T06:56:32.377636+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01867","citing_title":"An Exploratory Study on LLM-Generated Code and Comments in Code Repositories","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25358","citing_title":"AI-Associated Lexical Shifts Across 34 Languages: Cross-Lingual Convergence and Diachronic Uptake in News Writing","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07183","citing_title":"Monitoring AI-Modified Content at Scale: A Case Study on the Impact of ChatGPT on AI Conference Peer Reviews","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2410.23728","citing_title":"GigaCheck: Detecting LLM-generated Content via Object-Centric Span Localization","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19516","citing_title":"Base Models Look Human To AI Detectors","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18333","citing_title":"Position: LLM Watermarking Should Align Stakeholders' Incentives for Practical Adoption","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2406.20094","citing_title":"Scaling Synthetic Data Creation with 1,000,000,000 Personas","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14240","citing_title":"Paraphrasing Attack Resilience of Various AI-Generated Text Detection Methods","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK","json":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK.json","graph_json":"https://pith.science/api/pith-number/X3VHHN36B6NBOOG2CS7JAVOCXK/graph.json","events_json":"https://pith.science/api/pith-number/X3VHHN36B6NBOOG2CS7JAVOCXK/events.json","paper":"https://pith.science/paper/X3VHHN36"},"agent_actions":{"view_html":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK","download_json":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK.json","view_paper":"https://pith.science/paper/X3VHHN36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.04736&json=true","fetch_graph":"https://pith.science/api/pith-number/X3VHHN36B6NBOOG2CS7JAVOCXK/graph.json","fetch_events":"https://pith.science/api/pith-number/X3VHHN36B6NBOOG2CS7JAVOCXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK/action/storage_attestation","attest_author":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK/action/author_attestation","sign_citation":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK/action/citation_signature","submit_replication":"https://pith.science/pith/X3VHHN36B6NBOOG2CS7JAVOCXK/action/replication_record"}},"created_at":"2026-07-05T06:56:32.377636+00:00","updated_at":"2026-07-05T06:56:32.377636+00:00"}