{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:MELSDOUWMMUAMLN4MKVYGJVJ3O","short_pith_number":"pith:MELSDOUW","schema_version":"1.0","canonical_sha256":"611721ba966328062dbc62ab8326a9dbb9be1c06cbe5345bbd499d243d7d3fef","source":{"kind":"arxiv","id":"1904.09086","version":2},"attestation_state":"computed","paper":{"title":"Learning Programmatic Idioms for Scalable Semantic Parsing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alvin Cheung, Luke Zettlemoyer, Srinivasan Iyer","submitted_at":"2019-04-19T05:56:45Z","abstract_excerpt":"Programmers typically organize executable source code using high-level coding patterns or idiomatic structures such as nested loops, exception handlers and recursive blocks, rather than as individual code tokens. In contrast, state of the art (SOTA) semantic parsers still map natural language instructions to source code by building the code syntax tree one node at a time. In this paper, we introduce an iterative method to extract code idioms from large source code corpora by repeatedly collapsing most-frequent depth-2 subtrees of their syntax trees, and train semantic parsers to apply these id"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.09086","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-04-19T05:56:45Z","cross_cats_sorted":[],"title_canon_sha256":"8b00dc578ba3ce89f67ec4421faac4c284d3300935995bf41d57177def521fad","abstract_canon_sha256":"a70c2a81105df0c1e0c9a8053475609bfb9e6f4078f514893666131b41f2f860"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:02:38.577374Z","signature_b64":"EAKfNUvIiB1jq5bCvMUHwa8PTCDpj5ZiuPGMqSxDhRODZAL+wnceAIrjoTBqgKxEZr3KXD3XIhVMbBNCSLiADg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"611721ba966328062dbc62ab8326a9dbb9be1c06cbe5345bbd499d243d7d3fef","last_reissued_at":"2026-07-05T00:02:38.576814Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:02:38.576814Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Programmatic Idioms for Scalable Semantic Parsing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alvin Cheung, Luke Zettlemoyer, Srinivasan Iyer","submitted_at":"2019-04-19T05:56:45Z","abstract_excerpt":"Programmers typically organize executable source code using high-level coding patterns or idiomatic structures such as nested loops, exception handlers and recursive blocks, rather than as individual code tokens. In contrast, state of the art (SOTA) semantic parsers still map natural language instructions to source code by building the code syntax tree one node at a time. In this paper, we introduce an iterative method to extract code idioms from large source code corpora by repeatedly collapsing most-frequent depth-2 subtrees of their syntax trees, and train semantic parsers to apply these id"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.09086","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.09086/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.09086","created_at":"2026-07-05T00:02:38.576878+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.09086v2","created_at":"2026-07-05T00:02:38.576878+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.09086","created_at":"2026-07-05T00:02:38.576878+00:00"},{"alias_kind":"pith_short_12","alias_value":"MELSDOUWMMUA","created_at":"2026-07-05T00:02:38.576878+00:00"},{"alias_kind":"pith_short_16","alias_value":"MELSDOUWMMUAMLN4","created_at":"2026-07-05T00:02:38.576878+00:00"},{"alias_kind":"pith_short_8","alias_value":"MELSDOUW","created_at":"2026-07-05T00:02:38.576878+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2102.04664","citing_title":"CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O","json":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O.json","graph_json":"https://pith.science/api/pith-number/MELSDOUWMMUAMLN4MKVYGJVJ3O/graph.json","events_json":"https://pith.science/api/pith-number/MELSDOUWMMUAMLN4MKVYGJVJ3O/events.json","paper":"https://pith.science/paper/MELSDOUW"},"agent_actions":{"view_html":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O","download_json":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O.json","view_paper":"https://pith.science/paper/MELSDOUW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.09086&json=true","fetch_graph":"https://pith.science/api/pith-number/MELSDOUWMMUAMLN4MKVYGJVJ3O/graph.json","fetch_events":"https://pith.science/api/pith-number/MELSDOUWMMUAMLN4MKVYGJVJ3O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O/action/storage_attestation","attest_author":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O/action/author_attestation","sign_citation":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O/action/citation_signature","submit_replication":"https://pith.science/pith/MELSDOUWMMUAMLN4MKVYGJVJ3O/action/replication_record"}},"created_at":"2026-07-05T00:02:38.576878+00:00","updated_at":"2026-07-05T00:02:38.576878+00:00"}