{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:PLGTJF7TLFJL4UPD7IDIFPQPGY","short_pith_number":"pith:PLGTJF7T","schema_version":"1.0","canonical_sha256":"7acd3497f35952be51e3fa0682be0f361c119f35eb23d2a521cd534032fe24ed","source":{"kind":"arxiv","id":"1912.06365","version":1},"attestation_state":"computed","paper":{"title":"Fast Image Caption Generation with Position Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Zheng-cong Fei","submitted_at":"2019-12-13T09:06:46Z","abstract_excerpt":"Recent neural network models for image captioning usually employ an encoder-decoder architecture, where the decoder adopts a recursive sequence decoding way. However, such autoregressive decoding may result in sequential error accumulation and slow generation which limit the applications in practice. Non-autoregressive (NA) decoding has been proposed to cover these issues but suffers from language quality problem due to the indirect modeling of the target distribution. Towards that end, we propose an improved NA prediction framework to accelerate image captioning. Our decoding part consists of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.06365","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2019-12-13T09:06:46Z","cross_cats_sorted":[],"title_canon_sha256":"982801142782b61e89f8d38c1fdc0f053069d97e020c915854dcc6a92098c4c5","abstract_canon_sha256":"09eabd893f057771a65718e80a731560e6572af5911c6a3def560983e4a230aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:25:59.747874Z","signature_b64":"ysVFHi1A+7peSEVcAjpxwa1kUCuDiaEx60JsJAU95gNdXNtfEsbj5ZPVlRxqZ9UzEYpWv7G/oy3AKaiPlBHlAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7acd3497f35952be51e3fa0682be0f361c119f35eb23d2a521cd534032fe24ed","last_reissued_at":"2026-07-05T00:25:59.747403Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:25:59.747403Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast Image Caption Generation with Position Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Zheng-cong Fei","submitted_at":"2019-12-13T09:06:46Z","abstract_excerpt":"Recent neural network models for image captioning usually employ an encoder-decoder architecture, where the decoder adopts a recursive sequence decoding way. However, such autoregressive decoding may result in sequential error accumulation and slow generation which limit the applications in practice. Non-autoregressive (NA) decoding has been proposed to cover these issues but suffers from language quality problem due to the indirect modeling of the target distribution. Towards that end, we propose an improved NA prediction framework to accelerate image captioning. Our decoding part consists of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.06365","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.06365/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.06365","created_at":"2026-07-05T00:25:59.747460+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.06365v1","created_at":"2026-07-05T00:25:59.747460+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.06365","created_at":"2026-07-05T00:25:59.747460+00:00"},{"alias_kind":"pith_short_12","alias_value":"PLGTJF7TLFJL","created_at":"2026-07-05T00:25:59.747460+00:00"},{"alias_kind":"pith_short_16","alias_value":"PLGTJF7TLFJL4UPD","created_at":"2026-07-05T00:25:59.747460+00:00"},{"alias_kind":"pith_short_8","alias_value":"PLGTJF7T","created_at":"2026-07-05T00:25:59.747460+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.01790","citing_title":"Ingredients: Blending Custom Photos with Video Diffusion Transformers","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY","json":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY.json","graph_json":"https://pith.science/api/pith-number/PLGTJF7TLFJL4UPD7IDIFPQPGY/graph.json","events_json":"https://pith.science/api/pith-number/PLGTJF7TLFJL4UPD7IDIFPQPGY/events.json","paper":"https://pith.science/paper/PLGTJF7T"},"agent_actions":{"view_html":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY","download_json":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY.json","view_paper":"https://pith.science/paper/PLGTJF7T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.06365&json=true","fetch_graph":"https://pith.science/api/pith-number/PLGTJF7TLFJL4UPD7IDIFPQPGY/graph.json","fetch_events":"https://pith.science/api/pith-number/PLGTJF7TLFJL4UPD7IDIFPQPGY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY/action/storage_attestation","attest_author":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY/action/author_attestation","sign_citation":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY/action/citation_signature","submit_replication":"https://pith.science/pith/PLGTJF7TLFJL4UPD7IDIFPQPGY/action/replication_record"}},"created_at":"2026-07-05T00:25:59.747460+00:00","updated_at":"2026-07-05T00:25:59.747460+00:00"}