{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:D3GSNSVTZVJY63XS6G2BDAM7YR","short_pith_number":"pith:D3GSNSVT","schema_version":"1.0","canonical_sha256":"1ecd26cab3cd538f6ef2f1b411819fc47668d439c99096c08d36a36a95640f47","source":{"kind":"arxiv","id":"2307.13244","version":1},"attestation_state":"computed","paper":{"title":"Multi-Granularity Prediction with Learnable Fusion for Scene Text Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Da, Cong Yao, Peng Wang","submitted_at":"2023-07-25T04:12:50Z","abstract_excerpt":"Due to the enormous technical challenges and wide range of applications, scene text recognition (STR) has been an active research topic in computer vision for years. To tackle this tough problem, numerous innovative methods have been successively proposed, and incorporating linguistic knowledge into STR models has recently become a prominent trend. In this work, we first draw inspiration from the recent progress in Vision Transformer (ViT) to construct a conceptually simple yet functionally powerful vision STR model, which is built upon ViT and a tailored Adaptive Addressing and Aggregation (A"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.13244","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-07-25T04:12:50Z","cross_cats_sorted":[],"title_canon_sha256":"4a594968c7b13184555f0dd7289d79c3bdf0547870f281fad57b52f6e266c495","abstract_canon_sha256":"1ba40481ecc9f454e56019588921899e676c064b618326bca04e074a532f7375"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:34:11.284607Z","signature_b64":"bCd4h3cD3c0496quuCQTDpPv3s5NSDmbUp0BLpncAloTR5corlkog0ZO8e7t3RLwBtmDtF08INWtiYaw6R3wBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1ecd26cab3cd538f6ef2f1b411819fc47668d439c99096c08d36a36a95640f47","last_reissued_at":"2026-07-05T06:34:11.284206Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:34:11.284206Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-Granularity Prediction with Learnable Fusion for Scene Text Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Da, Cong Yao, Peng Wang","submitted_at":"2023-07-25T04:12:50Z","abstract_excerpt":"Due to the enormous technical challenges and wide range of applications, scene text recognition (STR) has been an active research topic in computer vision for years. To tackle this tough problem, numerous innovative methods have been successively proposed, and incorporating linguistic knowledge into STR models has recently become a prominent trend. In this work, we first draw inspiration from the recent progress in Vision Transformer (ViT) to construct a conceptually simple yet functionally powerful vision STR model, which is built upon ViT and a tailored Adaptive Addressing and Aggregation (A"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.13244","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.13244/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.13244","created_at":"2026-07-05T06:34:11.284261+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.13244v1","created_at":"2026-07-05T06:34:11.284261+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.13244","created_at":"2026-07-05T06:34:11.284261+00:00"},{"alias_kind":"pith_short_12","alias_value":"D3GSNSVTZVJY","created_at":"2026-07-05T06:34:11.284261+00:00"},{"alias_kind":"pith_short_16","alias_value":"D3GSNSVTZVJY63XS","created_at":"2026-07-05T06:34:11.284261+00:00"},{"alias_kind":"pith_short_8","alias_value":"D3GSNSVT","created_at":"2026-07-05T06:34:11.284261+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.04088","citing_title":"Multimodal Tabular Reasoning with Privileged Structured Information","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR","json":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR.json","graph_json":"https://pith.science/api/pith-number/D3GSNSVTZVJY63XS6G2BDAM7YR/graph.json","events_json":"https://pith.science/api/pith-number/D3GSNSVTZVJY63XS6G2BDAM7YR/events.json","paper":"https://pith.science/paper/D3GSNSVT"},"agent_actions":{"view_html":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR","download_json":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR.json","view_paper":"https://pith.science/paper/D3GSNSVT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.13244&json=true","fetch_graph":"https://pith.science/api/pith-number/D3GSNSVTZVJY63XS6G2BDAM7YR/graph.json","fetch_events":"https://pith.science/api/pith-number/D3GSNSVTZVJY63XS6G2BDAM7YR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR/action/storage_attestation","attest_author":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR/action/author_attestation","sign_citation":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR/action/citation_signature","submit_replication":"https://pith.science/pith/D3GSNSVTZVJY63XS6G2BDAM7YR/action/replication_record"}},"created_at":"2026-07-05T06:34:11.284261+00:00","updated_at":"2026-07-05T06:34:11.284261+00:00"}