{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:AOVQY24TNAYUHFXLUCO2BWSEGC","short_pith_number":"pith:AOVQY24T","schema_version":"1.0","canonical_sha256":"03ab0c6b9368314396eba09da0da4430906f42f93552ff50995be8e5ad7adb18","source":{"kind":"arxiv","id":"2211.02321","version":1},"attestation_state":"computed","paper":{"title":"OSIC: A New One-Stage Image Captioner Coined","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Wang, Meng Wang, Mingbo Zhao, Mingliang Xu, Xiaojie Jin, Zhao Zhang","submitted_at":"2022-11-04T08:50:09Z","abstract_excerpt":"Mainstream image caption models are usually two-stage captioners, i.e., calculating object features by pre-trained detector, and feeding them into a language model to generate text descriptions. However, such an operation will cause a task-based information gap to decrease the performance, since the object features in detection task are suboptimal representation and cannot provide all necessary information for subsequent text generation. Besides, object features are usually represented by the last layer features that lose the local details of input images. In this paper, we propose a novel One"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.02321","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-11-04T08:50:09Z","cross_cats_sorted":[],"title_canon_sha256":"c67377008eb4d1a151de2da4585f37b47db7c80b50f123e9c54e40538c21598e","abstract_canon_sha256":"925d1d3f5d775f5ba987507516a5b8accc657c425ac1b82f9f05362efb69496c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:13:14.136589Z","signature_b64":"v+cNLFpyDyIR+7+aGvidewkz8YzZFY6/W5vhf+p2Bx5SG2J/qPhm/sMwCe+7Cf6T9LMN11PGCUf0IcZvHW17Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03ab0c6b9368314396eba09da0da4430906f42f93552ff50995be8e5ad7adb18","last_reissued_at":"2026-07-05T05:13:14.136026Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:13:14.136026Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OSIC: A New One-Stage Image Captioner Coined","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Wang, Meng Wang, Mingbo Zhao, Mingliang Xu, Xiaojie Jin, Zhao Zhang","submitted_at":"2022-11-04T08:50:09Z","abstract_excerpt":"Mainstream image caption models are usually two-stage captioners, i.e., calculating object features by pre-trained detector, and feeding them into a language model to generate text descriptions. However, such an operation will cause a task-based information gap to decrease the performance, since the object features in detection task are suboptimal representation and cannot provide all necessary information for subsequent text generation. Besides, object features are usually represented by the last layer features that lose the local details of input images. In this paper, we propose a novel One"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.02321","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.02321/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.02321","created_at":"2026-07-05T05:13:14.136092+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.02321v1","created_at":"2026-07-05T05:13:14.136092+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.02321","created_at":"2026-07-05T05:13:14.136092+00:00"},{"alias_kind":"pith_short_12","alias_value":"AOVQY24TNAYU","created_at":"2026-07-05T05:13:14.136092+00:00"},{"alias_kind":"pith_short_16","alias_value":"AOVQY24TNAYUHFXL","created_at":"2026-07-05T05:13:14.136092+00:00"},{"alias_kind":"pith_short_8","alias_value":"AOVQY24T","created_at":"2026-07-05T05:13:14.136092+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC","json":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC.json","graph_json":"https://pith.science/api/pith-number/AOVQY24TNAYUHFXLUCO2BWSEGC/graph.json","events_json":"https://pith.science/api/pith-number/AOVQY24TNAYUHFXLUCO2BWSEGC/events.json","paper":"https://pith.science/paper/AOVQY24T"},"agent_actions":{"view_html":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC","download_json":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC.json","view_paper":"https://pith.science/paper/AOVQY24T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.02321&json=true","fetch_graph":"https://pith.science/api/pith-number/AOVQY24TNAYUHFXLUCO2BWSEGC/graph.json","fetch_events":"https://pith.science/api/pith-number/AOVQY24TNAYUHFXLUCO2BWSEGC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC/action/storage_attestation","attest_author":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC/action/author_attestation","sign_citation":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC/action/citation_signature","submit_replication":"https://pith.science/pith/AOVQY24TNAYUHFXLUCO2BWSEGC/action/replication_record"}},"created_at":"2026-07-05T05:13:14.136092+00:00","updated_at":"2026-07-05T05:13:14.136092+00:00"}