{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3PXAOLCFQBFTRZ4HQFXDYWJ7R3","short_pith_number":"pith:3PXAOLCF","schema_version":"1.0","canonical_sha256":"dbee072c45804b38e787816e3c593f8ec71fcbae73f29cdaa83dd5499085819c","source":{"kind":"arxiv","id":"2501.14828","version":1},"attestation_state":"computed","paper":{"title":"An Ensemble Model with Attention Based Mechanism for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bassam Hammo, Israa Al Badarneh, Omar Al-Kadi","submitted_at":"2025-01-22T12:28:37Z","abstract_excerpt":"Image captioning creates informative text from an input image by creating a relationship between the words and the actual content of an image. Recently, deep learning models that utilize transformers have been the most successful in automatically generating image captions. The capabilities of transformer networks have led to notable progress in several activities related to vision. In this paper, we thoroughly examine transformer models, emphasizing the critical role that attention mechanisms play. The proposed model uses a transformer encoder-decoder architecture to create textual captions an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14828","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-22T12:28:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"afd78a1044c6a97d9c0d832faa24000c38836817d9a04da439434d6a23d57ab3","abstract_canon_sha256":"ca316426b46191554d8a95ed11673f792739e0b17fa56e84e7ec3c8bf5e4db02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:05:19.552765Z","signature_b64":"t7EC2JaUlh1doQAWDGgFuHESE0oXxO1Xw4PnY6QvEdU2myNwC7Bf+bzXgHv8YtLGRVYG6lbvY4tU936lBKFtAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dbee072c45804b38e787816e3c593f8ec71fcbae73f29cdaa83dd5499085819c","last_reissued_at":"2026-07-05T10:05:19.552269Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:05:19.552269Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Ensemble Model with Attention Based Mechanism for Image Captioning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bassam Hammo, Israa Al Badarneh, Omar Al-Kadi","submitted_at":"2025-01-22T12:28:37Z","abstract_excerpt":"Image captioning creates informative text from an input image by creating a relationship between the words and the actual content of an image. Recently, deep learning models that utilize transformers have been the most successful in automatically generating image captions. The capabilities of transformer networks have led to notable progress in several activities related to vision. In this paper, we thoroughly examine transformer models, emphasizing the critical role that attention mechanisms play. The proposed model uses a transformer encoder-decoder architecture to create textual captions an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14828","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14828/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14828","created_at":"2026-07-05T10:05:19.552337+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14828v1","created_at":"2026-07-05T10:05:19.552337+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14828","created_at":"2026-07-05T10:05:19.552337+00:00"},{"alias_kind":"pith_short_12","alias_value":"3PXAOLCFQBFT","created_at":"2026-07-05T10:05:19.552337+00:00"},{"alias_kind":"pith_short_16","alias_value":"3PXAOLCFQBFTRZ4H","created_at":"2026-07-05T10:05:19.552337+00:00"},{"alias_kind":"pith_short_8","alias_value":"3PXAOLCF","created_at":"2026-07-05T10:05:19.552337+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3","json":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3.json","graph_json":"https://pith.science/api/pith-number/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/graph.json","events_json":"https://pith.science/api/pith-number/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/events.json","paper":"https://pith.science/paper/3PXAOLCF"},"agent_actions":{"view_html":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3","download_json":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3.json","view_paper":"https://pith.science/paper/3PXAOLCF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14828&json=true","fetch_graph":"https://pith.science/api/pith-number/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/graph.json","fetch_events":"https://pith.science/api/pith-number/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/action/storage_attestation","attest_author":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/action/author_attestation","sign_citation":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/action/citation_signature","submit_replication":"https://pith.science/pith/3PXAOLCFQBFTRZ4HQFXDYWJ7R3/action/replication_record"}},"created_at":"2026-07-05T10:05:19.552337+00:00","updated_at":"2026-07-05T10:05:19.552337+00:00"}