{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:OYGGAZ2ZGSQNOU565GGSRXCLFN","short_pith_number":"pith:OYGGAZ2Z","schema_version":"1.0","canonical_sha256":"760c60675934a0d753bee98d28dc4b2b79c8adfaa9bce964768c055a1028123f","source":{"kind":"arxiv","id":"2103.13136","version":1},"attestation_state":"computed","paper":{"title":"Representing Numbers in NLP: a Survey and a Vision","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Avijit Thawani, Filip Ilievski, Jay Pujara, Pedro A. Szekely","submitted_at":"2021-03-24T12:28:22Z","abstract_excerpt":"NLP systems rarely give special consideration to numbers found in text. This starkly contrasts with the consensus in neuroscience that, in the brain, numbers are represented differently from words. We arrange recent NLP work on numeracy into a comprehensive taxonomy of tasks and methods. We break down the subjective notion of numeracy into 7 subtasks, arranged along two dimensions: granularity (exact vs approximate) and units (abstract vs grounded). We analyze the myriad representational choices made by 18 previously published number encoders and decoders. We synthesize best practices for repr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.13136","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2021-03-24T12:28:22Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"94b6b90068ef7579c902c6081253ad9d38bfe5d864c1f4392c3ad230831a4c81","abstract_canon_sha256":"1e618aa81c7d16e34f119749dcee32904dc88f06a2c0a2bb1f3bfb8f859b6dd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:26:06.178599Z","signature_b64":"mU/LdoIZcOaYwaoMl23ru7C44fCmPsdyv4XTFGx6lPD7OBvU7cq+aiYGl7v5czRSyjUkDBkAVFrQXpH3ymWhBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"760c60675934a0d753bee98d28dc4b2b79c8adfaa9bce964768c055a1028123f","last_reissued_at":"2026-07-05T02:26:06.178135Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:26:06.178135Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Representing Numbers in NLP: a Survey and a Vision","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Avijit Thawani, Filip Ilievski, Jay Pujara, Pedro A. Szekely","submitted_at":"2021-03-24T12:28:22Z","abstract_excerpt":"NLP systems rarely give special consideration to numbers found in text. This starkly contrasts with the consensus in neuroscience that, in the brain, numbers are represented differently from words. We arrange recent NLP work on numeracy into a comprehensive taxonomy of tasks and methods. We break down the subjective notion of numeracy into 7 subtasks, arranged along two dimensions: granularity (exact vs approximate) and units (abstract vs grounded). We analyze the myriad representational choices made by 18 previously published number encoders and decoders. We synthesize best practices for repr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.13136","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.13136/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.13136","created_at":"2026-07-05T02:26:06.178191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.13136v1","created_at":"2026-07-05T02:26:06.178191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.13136","created_at":"2026-07-05T02:26:06.178191+00:00"},{"alias_kind":"pith_short_12","alias_value":"OYGGAZ2ZGSQN","created_at":"2026-07-05T02:26:06.178191+00:00"},{"alias_kind":"pith_short_16","alias_value":"OYGGAZ2ZGSQNOU56","created_at":"2026-07-05T02:26:06.178191+00:00"},{"alias_kind":"pith_short_8","alias_value":"OYGGAZ2Z","created_at":"2026-07-05T02:26:06.178191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.09741","citing_title":"FoNE: Precise Single-Token Number Embeddings via Fourier Features","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14547","citing_title":"Predicting Post-Traumatic Epilepsy from Clinical Records using Large Language Model Embeddings","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN","json":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN.json","graph_json":"https://pith.science/api/pith-number/OYGGAZ2ZGSQNOU565GGSRXCLFN/graph.json","events_json":"https://pith.science/api/pith-number/OYGGAZ2ZGSQNOU565GGSRXCLFN/events.json","paper":"https://pith.science/paper/OYGGAZ2Z"},"agent_actions":{"view_html":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN","download_json":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN.json","view_paper":"https://pith.science/paper/OYGGAZ2Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.13136&json=true","fetch_graph":"https://pith.science/api/pith-number/OYGGAZ2ZGSQNOU565GGSRXCLFN/graph.json","fetch_events":"https://pith.science/api/pith-number/OYGGAZ2ZGSQNOU565GGSRXCLFN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN/action/storage_attestation","attest_author":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN/action/author_attestation","sign_citation":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN/action/citation_signature","submit_replication":"https://pith.science/pith/OYGGAZ2ZGSQNOU565GGSRXCLFN/action/replication_record"}},"created_at":"2026-07-05T02:26:06.178191+00:00","updated_at":"2026-07-05T02:26:06.178191+00:00"}