{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DNZX72A22V3PGEHSIEDWDSAIN2","short_pith_number":"pith:DNZX72A2","schema_version":"1.0","canonical_sha256":"1b737fe81ad576f310f2410761c8086e9bdda6436cdadb4ea78e41dd2874aec3","source":{"kind":"arxiv","id":"2404.14066","version":3},"attestation_state":"computed","paper":{"title":"SHE-Net: Syntax-Hierarchy-Enhanced Text-Video Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CV","authors_text":"Chen Jiang, Ming Yang, Qingpei Guo, Tian Gan, Xingning Dong, Xuzheng Yu","submitted_at":"2024-04-22T10:23:59Z","abstract_excerpt":"The user base of short video apps has experienced unprecedented growth in recent years, resulting in a significant demand for video content analysis. In particular, text-video retrieval, which aims to find the top matching videos given text descriptions from a vast video corpus, is an essential function, the primary challenge of which is to bridge the modality gap. Nevertheless, most existing approaches treat texts merely as discrete tokens and neglect their syntax structures. Moreover, the abundant spatial and temporal clues in videos are often underutilized due to the lack of interaction wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14066","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-04-22T10:23:59Z","cross_cats_sorted":["cs.IR"],"title_canon_sha256":"662f76b3dce19da585689edc17e8d6f876106545b5bdf883d365fb7b1788d2a7","abstract_canon_sha256":"a0f7b45a68040b416809afd867e6b98bf524fa8955e0e490455535a1078ab615"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:36.412770Z","signature_b64":"Nt93YOnSnDRHoUQw62B5X+yq3qFKGxwlSM3WdOkE7n9ly4r/bX8zCPNhhJ9LtBarKkxiaDlWaIbnNb4H+wX7Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b737fe81ad576f310f2410761c8086e9bdda6436cdadb4ea78e41dd2874aec3","last_reissued_at":"2026-07-05T10:18:36.412250Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:36.412250Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SHE-Net: Syntax-Hierarchy-Enhanced Text-Video Retrieval","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IR"],"primary_cat":"cs.CV","authors_text":"Chen Jiang, Ming Yang, Qingpei Guo, Tian Gan, Xingning Dong, Xuzheng Yu","submitted_at":"2024-04-22T10:23:59Z","abstract_excerpt":"The user base of short video apps has experienced unprecedented growth in recent years, resulting in a significant demand for video content analysis. In particular, text-video retrieval, which aims to find the top matching videos given text descriptions from a vast video corpus, is an essential function, the primary challenge of which is to bridge the modality gap. Nevertheless, most existing approaches treat texts merely as discrete tokens and neglect their syntax structures. Moreover, the abundant spatial and temporal clues in videos are often underutilized due to the lack of interaction wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14066","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14066/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14066","created_at":"2026-07-05T10:18:36.412310+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14066v3","created_at":"2026-07-05T10:18:36.412310+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14066","created_at":"2026-07-05T10:18:36.412310+00:00"},{"alias_kind":"pith_short_12","alias_value":"DNZX72A22V3P","created_at":"2026-07-05T10:18:36.412310+00:00"},{"alias_kind":"pith_short_16","alias_value":"DNZX72A22V3PGEHS","created_at":"2026-07-05T10:18:36.412310+00:00"},{"alias_kind":"pith_short_8","alias_value":"DNZX72A2","created_at":"2026-07-05T10:18:36.412310+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.23952","citing_title":"Leveraging Auxiliary Information in Text-to-Video Retrieval: A Review","ref_index":104,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2","json":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2.json","graph_json":"https://pith.science/api/pith-number/DNZX72A22V3PGEHSIEDWDSAIN2/graph.json","events_json":"https://pith.science/api/pith-number/DNZX72A22V3PGEHSIEDWDSAIN2/events.json","paper":"https://pith.science/paper/DNZX72A2"},"agent_actions":{"view_html":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2","download_json":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2.json","view_paper":"https://pith.science/paper/DNZX72A2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14066&json=true","fetch_graph":"https://pith.science/api/pith-number/DNZX72A22V3PGEHSIEDWDSAIN2/graph.json","fetch_events":"https://pith.science/api/pith-number/DNZX72A22V3PGEHSIEDWDSAIN2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2/action/storage_attestation","attest_author":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2/action/author_attestation","sign_citation":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2/action/citation_signature","submit_replication":"https://pith.science/pith/DNZX72A22V3PGEHSIEDWDSAIN2/action/replication_record"}},"created_at":"2026-07-05T10:18:36.412310+00:00","updated_at":"2026-07-05T10:18:36.412310+00:00"}