{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:MGHAONRIFA5JUNSO3UBYKMKTOG","short_pith_number":"pith:MGHAONRI","schema_version":"1.0","canonical_sha256":"618e073628283a9a364edd0385315371ab88dcc854ed98152f7e183cf35ca950","source":{"kind":"arxiv","id":"2107.11707","version":3},"attestation_state":"computed","paper":{"title":"Boosting Video Captioning with Dynamic Loss Network","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Nasib Ullah, Partha Pratim Mohanta","submitted_at":"2021-07-25T01:32:02Z","abstract_excerpt":"Video captioning is one of the challenging problems at the intersection of vision and language, having many real-life applications in video retrieval, video surveillance, assisting visually challenged people, Human-machine interface, and many more. Recent deep learning based methods have shown promising results but are still on the lower side than other vision tasks (such as image classification, object detection). A significant drawback with existing video captioning methods is that they are optimized over cross-entropy loss function, which is uncorrelated to the de facto evaluation metrics ("},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.11707","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-07-25T01:32:02Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e88a07325a1e68856aef26b68b3c9d59c15a9317956a690a6dee66a7340c15b6","abstract_canon_sha256":"aca030bfece8ff9b7f597508308f6b9f0cd347ca73147eace0357b3963484e6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:53:25.480254Z","signature_b64":"UyVvpJ7RCFu9Rck+6beHOUoxisPJd8Z5FRT9jKd92vKGBztvq4U7v4LUtndRTlaDd7mMR6a65RnfXuQHVPyBBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"618e073628283a9a364edd0385315371ab88dcc854ed98152f7e183cf35ca950","last_reissued_at":"2026-07-05T03:53:25.479729Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:53:25.479729Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Boosting Video Captioning with Dynamic Loss Network","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Nasib Ullah, Partha Pratim Mohanta","submitted_at":"2021-07-25T01:32:02Z","abstract_excerpt":"Video captioning is one of the challenging problems at the intersection of vision and language, having many real-life applications in video retrieval, video surveillance, assisting visually challenged people, Human-machine interface, and many more. Recent deep learning based methods have shown promising results but are still on the lower side than other vision tasks (such as image classification, object detection). A significant drawback with existing video captioning methods is that they are optimized over cross-entropy loss function, which is uncorrelated to the de facto evaluation metrics ("},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.11707","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.11707/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.11707","created_at":"2026-07-05T03:53:25.479793+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.11707v3","created_at":"2026-07-05T03:53:25.479793+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.11707","created_at":"2026-07-05T03:53:25.479793+00:00"},{"alias_kind":"pith_short_12","alias_value":"MGHAONRIFA5J","created_at":"2026-07-05T03:53:25.479793+00:00"},{"alias_kind":"pith_short_16","alias_value":"MGHAONRIFA5JUNSO","created_at":"2026-07-05T03:53:25.479793+00:00"},{"alias_kind":"pith_short_8","alias_value":"MGHAONRI","created_at":"2026-07-05T03:53:25.479793+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.16594","citing_title":"Temporal Object Captioning for Street Scene Videos from LiDAR Tracks","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG","json":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG.json","graph_json":"https://pith.science/api/pith-number/MGHAONRIFA5JUNSO3UBYKMKTOG/graph.json","events_json":"https://pith.science/api/pith-number/MGHAONRIFA5JUNSO3UBYKMKTOG/events.json","paper":"https://pith.science/paper/MGHAONRI"},"agent_actions":{"view_html":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG","download_json":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG.json","view_paper":"https://pith.science/paper/MGHAONRI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.11707&json=true","fetch_graph":"https://pith.science/api/pith-number/MGHAONRIFA5JUNSO3UBYKMKTOG/graph.json","fetch_events":"https://pith.science/api/pith-number/MGHAONRIFA5JUNSO3UBYKMKTOG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG/action/storage_attestation","attest_author":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG/action/author_attestation","sign_citation":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG/action/citation_signature","submit_replication":"https://pith.science/pith/MGHAONRIFA5JUNSO3UBYKMKTOG/action/replication_record"}},"created_at":"2026-07-05T03:53:25.479793+00:00","updated_at":"2026-07-05T03:53:25.479793+00:00"}