{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JOGTLBMK6MGHAJAQUYUUNWMFHD","short_pith_number":"pith:JOGTLBMK","schema_version":"1.0","canonical_sha256":"4b8d35858af30c702410a62946d98538fd466e909cabbf9c5c244ebe7c1163e5","source":{"kind":"arxiv","id":"2306.13460","version":3},"attestation_state":"computed","paper":{"title":"Learning Descriptive Image Captioning via Semipermeable Maximum Likelihood Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anwen Hu, Liang Zhang, Qin Jin, Zihao Yue","submitted_at":"2023-06-23T12:03:07Z","abstract_excerpt":"Image captioning aims to describe visual content in natural language. As 'a picture is worth a thousand words', there could be various correct descriptions for an image. However, with maximum likelihood estimation as the training objective, the captioning model is penalized whenever its prediction mismatches with the label. For instance, when the model predicts a word expressing richer semantics than the label, it will be penalized and optimized to prefer more concise expressions, referred to as conciseness optimization. In contrast, predictions that are more concise than labels lead to richne"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.13460","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-23T12:03:07Z","cross_cats_sorted":[],"title_canon_sha256":"cba7683d03e8036824aee281c7d84f0a47c84b4f08202c4abfdcbfa7a38e2a7c","abstract_canon_sha256":"867726d5421a6d187e3b82e63d4f6e6ca3074a950ed56cbd33fb7fd5612c92a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:30.575512Z","signature_b64":"55mDLYjYfMhD/EV8svCVWRu0JPBglt9kpfHLzEhig5cLDA1U7JAzmS10wpyg3ZjTrGvq2k5yAv02s0O7csM4BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4b8d35858af30c702410a62946d98538fd466e909cabbf9c5c244ebe7c1163e5","last_reissued_at":"2026-07-05T07:06:30.574930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:30.574930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Descriptive Image Captioning via Semipermeable Maximum Likelihood Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Anwen Hu, Liang Zhang, Qin Jin, Zihao Yue","submitted_at":"2023-06-23T12:03:07Z","abstract_excerpt":"Image captioning aims to describe visual content in natural language. As 'a picture is worth a thousand words', there could be various correct descriptions for an image. However, with maximum likelihood estimation as the training objective, the captioning model is penalized whenever its prediction mismatches with the label. For instance, when the model predicts a word expressing richer semantics than the label, it will be penalized and optimized to prefer more concise expressions, referred to as conciseness optimization. In contrast, predictions that are more concise than labels lead to richne"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.13460","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.13460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.13460","created_at":"2026-07-05T07:06:30.574995+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.13460v3","created_at":"2026-07-05T07:06:30.574995+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.13460","created_at":"2026-07-05T07:06:30.574995+00:00"},{"alias_kind":"pith_short_12","alias_value":"JOGTLBMK6MGH","created_at":"2026-07-05T07:06:30.574995+00:00"},{"alias_kind":"pith_short_16","alias_value":"JOGTLBMK6MGHAJAQ","created_at":"2026-07-05T07:06:30.574995+00:00"},{"alias_kind":"pith_short_8","alias_value":"JOGTLBMK","created_at":"2026-07-05T07:06:30.574995+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.20077","citing_title":"The Devil is in the EOS: Sequence Training for Detailed Image Captioning","ref_index":31,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD","json":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD.json","graph_json":"https://pith.science/api/pith-number/JOGTLBMK6MGHAJAQUYUUNWMFHD/graph.json","events_json":"https://pith.science/api/pith-number/JOGTLBMK6MGHAJAQUYUUNWMFHD/events.json","paper":"https://pith.science/paper/JOGTLBMK"},"agent_actions":{"view_html":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD","download_json":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD.json","view_paper":"https://pith.science/paper/JOGTLBMK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.13460&json=true","fetch_graph":"https://pith.science/api/pith-number/JOGTLBMK6MGHAJAQUYUUNWMFHD/graph.json","fetch_events":"https://pith.science/api/pith-number/JOGTLBMK6MGHAJAQUYUUNWMFHD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD/action/storage_attestation","attest_author":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD/action/author_attestation","sign_citation":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD/action/citation_signature","submit_replication":"https://pith.science/pith/JOGTLBMK6MGHAJAQUYUUNWMFHD/action/replication_record"}},"created_at":"2026-07-05T07:06:30.574995+00:00","updated_at":"2026-07-05T07:06:30.574995+00:00"}