{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JWNG7EKDMZEFHKS6QQ5Z7BPD5J","short_pith_number":"pith:JWNG7EKD","schema_version":"1.0","canonical_sha256":"4d9a6f9143664853aa5e843b9f85e3ea7a11e17bb764f1b9c298b44e1485732b","source":{"kind":"arxiv","id":"2507.01844","version":1},"attestation_state":"computed","paper":{"title":"Low-Perplexity LLM-Generated Sequences and Where To Find Them","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anastasiia Kucherenko, Andrei Kucharavy, Arthur Wuhrmann","submitted_at":"2025-07-02T15:58:51Z","abstract_excerpt":"As Large Language Models (LLMs) become increasingly widespread, understanding how specific training data shapes their outputs is crucial for transparency, accountability, privacy, and fairness. To explore how LLMs leverage and replicate their training data, we introduce a systematic approach centered on analyzing low-perplexity sequences - high-probability text spans generated by the model. Our pipeline reliably extracts such long sequences across diverse topics while avoiding degeneration, then traces them back to their sources in the training data. Surprisingly, we find that a substantial po"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.01844","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-02T15:58:51Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f0410a282f43c0ebfe1649ce0b22a866335641742d81f53cc5b21c751fd2dd43","abstract_canon_sha256":"1db6911225f9b48894475370690781789ea20bd5b1a7231584d784c8b8b065ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:56.364582Z","signature_b64":"zA456qS4+yIWkhnxIXi6YavH6Bs0GnHSjUABmb+6x6/fbOlG0xIKTS//Vgm6QK6Oyw5x1srUlRbkdgofGKnOCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d9a6f9143664853aa5e843b9f85e3ea7a11e17bb764f1b9c298b44e1485732b","last_reissued_at":"2026-07-05T11:30:56.364122Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:56.364122Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Low-Perplexity LLM-Generated Sequences and Where To Find Them","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anastasiia Kucherenko, Andrei Kucharavy, Arthur Wuhrmann","submitted_at":"2025-07-02T15:58:51Z","abstract_excerpt":"As Large Language Models (LLMs) become increasingly widespread, understanding how specific training data shapes their outputs is crucial for transparency, accountability, privacy, and fairness. To explore how LLMs leverage and replicate their training data, we introduce a systematic approach centered on analyzing low-perplexity sequences - high-probability text spans generated by the model. Our pipeline reliably extracts such long sequences across diverse topics while avoiding degeneration, then traces them back to their sources in the training data. Surprisingly, we find that a substantial po"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.01844","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.01844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.01844","created_at":"2026-07-05T11:30:56.364180+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.01844v1","created_at":"2026-07-05T11:30:56.364180+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.01844","created_at":"2026-07-05T11:30:56.364180+00:00"},{"alias_kind":"pith_short_12","alias_value":"JWNG7EKDMZEF","created_at":"2026-07-05T11:30:56.364180+00:00"},{"alias_kind":"pith_short_16","alias_value":"JWNG7EKDMZEFHKS6","created_at":"2026-07-05T11:30:56.364180+00:00"},{"alias_kind":"pith_short_8","alias_value":"JWNG7EKD","created_at":"2026-07-05T11:30:56.364180+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23591","citing_title":"Quantifying the Agreement Between Data-Influence and Data-Similarity to Understand LLM Behavior","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J","json":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J.json","graph_json":"https://pith.science/api/pith-number/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/graph.json","events_json":"https://pith.science/api/pith-number/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/events.json","paper":"https://pith.science/paper/JWNG7EKD"},"agent_actions":{"view_html":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J","download_json":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J.json","view_paper":"https://pith.science/paper/JWNG7EKD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.01844&json=true","fetch_graph":"https://pith.science/api/pith-number/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/graph.json","fetch_events":"https://pith.science/api/pith-number/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/action/storage_attestation","attest_author":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/action/author_attestation","sign_citation":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/action/citation_signature","submit_replication":"https://pith.science/pith/JWNG7EKDMZEFHKS6QQ5Z7BPD5J/action/replication_record"}},"created_at":"2026-07-05T11:30:56.364180+00:00","updated_at":"2026-07-05T11:30:56.364180+00:00"}