{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JGLQSOTL7HBXGR66ZTKJG2IL6E","short_pith_number":"pith:JGLQSOTL","schema_version":"1.0","canonical_sha256":"4997093a6bf9c37347deccd493690bf10a2ae0239c7578875a8feacafba26469","source":{"kind":"arxiv","id":"2310.14840","version":1},"attestation_state":"computed","paper":{"title":"Transparency at the Source: Evaluating and Interpreting Language Models With Access to the True Distribution","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jaap Jumelet, Willem Zuidema","submitted_at":"2023-10-23T12:03:01Z","abstract_excerpt":"We present a setup for training, evaluating and interpreting neural language models, that uses artificial, language-like data. The data is generated using a massive probabilistic grammar (based on state-split PCFGs), that is itself derived from a large natural language corpus, but also provides us complete control over the generative process. We describe and release both grammar and corpus, and test for the naturalness of our generated data. This approach allows us to define closed-form expressions to efficiently compute exact lower bounds on obtainable perplexity using both causal and masked "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.14840","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-23T12:03:01Z","cross_cats_sorted":[],"title_canon_sha256":"1c009fbb4a3c5c259b6f0ce0ebaf2e3c5c6b7965f9de1a013d019d90311ac0e9","abstract_canon_sha256":"fb9ae770c049643bc0d4fa91f05cfa658e70d7a5cafef9bc48541ae615abefc2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:04:02.823035Z","signature_b64":"JtYIeyGGsN8Jy+4I+9zlb+zk0VsoZENfWxQkhq6vB4bjMxwU7DV5qLMRTvie2uQ3C18+fR0OCUfN7bnrNSEPDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4997093a6bf9c37347deccd493690bf10a2ae0239c7578875a8feacafba26469","last_reissued_at":"2026-07-05T07:04:02.822591Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:04:02.822591Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transparency at the Source: Evaluating and Interpreting Language Models With Access to the True Distribution","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jaap Jumelet, Willem Zuidema","submitted_at":"2023-10-23T12:03:01Z","abstract_excerpt":"We present a setup for training, evaluating and interpreting neural language models, that uses artificial, language-like data. The data is generated using a massive probabilistic grammar (based on state-split PCFGs), that is itself derived from a large natural language corpus, but also provides us complete control over the generative process. We describe and release both grammar and corpus, and test for the naturalness of our generated data. This approach allows us to define closed-form expressions to efficiently compute exact lower bounds on obtainable perplexity using both causal and masked "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.14840","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.14840/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.14840","created_at":"2026-07-05T07:04:02.822672+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.14840v1","created_at":"2026-07-05T07:04:02.822672+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.14840","created_at":"2026-07-05T07:04:02.822672+00:00"},{"alias_kind":"pith_short_12","alias_value":"JGLQSOTL7HBX","created_at":"2026-07-05T07:04:02.822672+00:00"},{"alias_kind":"pith_short_16","alias_value":"JGLQSOTL7HBXGR66","created_at":"2026-07-05T07:04:02.822672+00:00"},{"alias_kind":"pith_short_8","alias_value":"JGLQSOTL","created_at":"2026-07-05T07:04:02.822672+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.14777","citing_title":"Rethinking Memorization Measures and their Implications in Large Language Models","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E","json":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E.json","graph_json":"https://pith.science/api/pith-number/JGLQSOTL7HBXGR66ZTKJG2IL6E/graph.json","events_json":"https://pith.science/api/pith-number/JGLQSOTL7HBXGR66ZTKJG2IL6E/events.json","paper":"https://pith.science/paper/JGLQSOTL"},"agent_actions":{"view_html":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E","download_json":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E.json","view_paper":"https://pith.science/paper/JGLQSOTL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.14840&json=true","fetch_graph":"https://pith.science/api/pith-number/JGLQSOTL7HBXGR66ZTKJG2IL6E/graph.json","fetch_events":"https://pith.science/api/pith-number/JGLQSOTL7HBXGR66ZTKJG2IL6E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E/action/storage_attestation","attest_author":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E/action/author_attestation","sign_citation":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E/action/citation_signature","submit_replication":"https://pith.science/pith/JGLQSOTL7HBXGR66ZTKJG2IL6E/action/replication_record"}},"created_at":"2026-07-05T07:04:02.822672+00:00","updated_at":"2026-07-05T07:04:02.822672+00:00"}