{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ES7FUWPL5PJPNF7ZSGIQ6KOH7Y","short_pith_number":"pith:ES7FUWPL","schema_version":"1.0","canonical_sha256":"24be5a59ebebd2f697f991910f29c7fe174e24c5b839bb11ab8628ee2536330c","source":{"kind":"arxiv","id":"2507.00838","version":2},"attestation_state":"computed","paper":{"title":"Stylometry recognizes human and LLM-generated texts in short samples","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Iwona Grabska-Gradzi\\'nska, Jan K. Argasi\\'nski, Jeremi K. Ochab, Karol Przystalski","submitted_at":"2025-07-01T15:08:53Z","abstract_excerpt":"The paper explores stylometry as a method to distinguish between texts created by Large Language Models (LLMs) and humans, addressing issues of model attribution, intellectual property, and ethical AI use. Stylometry has been used extensively to characterise the style and attribute authorship of texts. By applying it to LLM-generated texts, we identify their emergent writing patterns. The paper involves creating a benchmark dataset based on Wikipedia, with (a) human-written term summaries, (b) texts generated purely by LLMs (GPT-3.5/4, LLaMa 2/3, Orca, and Falcon), (c) processed through multip"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.00838","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-01T15:08:53Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"15672399995f8140b50e5d5cb7686e19fdcc3e8dfdd9cb764a06c978a71b8ed4","abstract_canon_sha256":"477194810c326365bc178fe1c1752131e9f9c69a56621b5ac6f5091c84b0e9e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:31.427588Z","signature_b64":"rnktIAGAl3/pXV1RjrQZijZvfzle9MHP7xReOVRvYICj8ePKMQx+mR7cRHUHT41kHxioZEkGdNiKnWDHq7J1Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"24be5a59ebebd2f697f991910f29c7fe174e24c5b839bb11ab8628ee2536330c","last_reissued_at":"2026-07-05T11:42:31.427044Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:31.427044Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Stylometry recognizes human and LLM-generated texts in short samples","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Iwona Grabska-Gradzi\\'nska, Jan K. Argasi\\'nski, Jeremi K. Ochab, Karol Przystalski","submitted_at":"2025-07-01T15:08:53Z","abstract_excerpt":"The paper explores stylometry as a method to distinguish between texts created by Large Language Models (LLMs) and humans, addressing issues of model attribution, intellectual property, and ethical AI use. Stylometry has been used extensively to characterise the style and attribute authorship of texts. By applying it to LLM-generated texts, we identify their emergent writing patterns. The paper involves creating a benchmark dataset based on Wikipedia, with (a) human-written term summaries, (b) texts generated purely by LLMs (GPT-3.5/4, LLaMa 2/3, Orca, and Falcon), (c) processed through multip"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00838","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.00838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.00838","created_at":"2026-07-05T11:42:31.427105+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.00838v2","created_at":"2026-07-05T11:42:31.427105+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00838","created_at":"2026-07-05T11:42:31.427105+00:00"},{"alias_kind":"pith_short_12","alias_value":"ES7FUWPL5PJP","created_at":"2026-07-05T11:42:31.427105+00:00"},{"alias_kind":"pith_short_16","alias_value":"ES7FUWPL5PJPNF7Z","created_at":"2026-07-05T11:42:31.427105+00:00"},{"alias_kind":"pith_short_8","alias_value":"ES7FUWPL","created_at":"2026-07-05T11:42:31.427105+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09854","citing_title":"Can Multi-Agent LLMs Identify Their Peers? Stylometric Fingerprinting in Role-Constrained Political Analysis","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08465","citing_title":"From Safety Risk to Design Principle: Peer-Preservation in Multi-Agent LLM Systems and Its Implications for Orchestrated Democratic Discourse Analysis","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y","json":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y.json","graph_json":"https://pith.science/api/pith-number/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/graph.json","events_json":"https://pith.science/api/pith-number/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/events.json","paper":"https://pith.science/paper/ES7FUWPL"},"agent_actions":{"view_html":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y","download_json":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y.json","view_paper":"https://pith.science/paper/ES7FUWPL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.00838&json=true","fetch_graph":"https://pith.science/api/pith-number/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/graph.json","fetch_events":"https://pith.science/api/pith-number/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/action/storage_attestation","attest_author":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/action/author_attestation","sign_citation":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/action/citation_signature","submit_replication":"https://pith.science/pith/ES7FUWPL5PJPNF7ZSGIQ6KOH7Y/action/replication_record"}},"created_at":"2026-07-05T11:42:31.427105+00:00","updated_at":"2026-07-05T11:42:31.427105+00:00"}