{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PEHVVNOOKVN45EZGM4CW2ITFM2","short_pith_number":"pith:PEHVVNOO","schema_version":"1.0","canonical_sha256":"790f5ab5ce555bce932667056d2265669267c40605c4d4b6476fd2ff451fcef3","source":{"kind":"arxiv","id":"2301.03832","version":1},"attestation_state":"computed","paper":{"title":"Video Semantic Segmentation with Inter-Frame Feature Fusion and Inner-Frame Feature Refinement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiafan Zhuang, Junjie Li, Zilei Wang","submitted_at":"2023-01-10T07:57:05Z","abstract_excerpt":"Video semantic segmentation aims to generate accurate semantic maps for each video frame. To this end, many works dedicate to integrate diverse information from consecutive frames to enhance the features for prediction, where a feature alignment procedure via estimated optical flow is usually required. However, the optical flow would inevitably suffer from inaccuracy, and then introduce noises in feature fusion and further result in unsatisfactory segmentation results. In this paper, to tackle the misalignment issue, we propose a spatial-temporal fusion (STF) module to model dense pairwise rel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.03832","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-01-10T07:57:05Z","cross_cats_sorted":[],"title_canon_sha256":"84b1fa929a247bf6940006cff6cd73a9824379ffcf8722d5991e37fd6ee872f5","abstract_canon_sha256":"ed992f29ac6c775bf23e75d11cd1e8e9cbfca6fb31f08f99e97bb1d3f888ae1f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:32:06.343299Z","signature_b64":"On7oquollNX1sWgHz3kqnrln19zxzHr/SYMWYnhBsRnPdgx6honhMzhqPr0vFcqy73+gPtG6M9YSbptSJvk5AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"790f5ab5ce555bce932667056d2265669267c40605c4d4b6476fd2ff451fcef3","last_reissued_at":"2026-07-05T05:32:06.342877Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:32:06.342877Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video Semantic Segmentation with Inter-Frame Feature Fusion and Inner-Frame Feature Refinement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiafan Zhuang, Junjie Li, Zilei Wang","submitted_at":"2023-01-10T07:57:05Z","abstract_excerpt":"Video semantic segmentation aims to generate accurate semantic maps for each video frame. To this end, many works dedicate to integrate diverse information from consecutive frames to enhance the features for prediction, where a feature alignment procedure via estimated optical flow is usually required. However, the optical flow would inevitably suffer from inaccuracy, and then introduce noises in feature fusion and further result in unsatisfactory segmentation results. In this paper, to tackle the misalignment issue, we propose a spatial-temporal fusion (STF) module to model dense pairwise rel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.03832","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.03832/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.03832","created_at":"2026-07-05T05:32:06.342930+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.03832v1","created_at":"2026-07-05T05:32:06.342930+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.03832","created_at":"2026-07-05T05:32:06.342930+00:00"},{"alias_kind":"pith_short_12","alias_value":"PEHVVNOOKVN4","created_at":"2026-07-05T05:32:06.342930+00:00"},{"alias_kind":"pith_short_16","alias_value":"PEHVVNOOKVN45EZG","created_at":"2026-07-05T05:32:06.342930+00:00"},{"alias_kind":"pith_short_8","alias_value":"PEHVVNOO","created_at":"2026-07-05T05:32:06.342930+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.27499","citing_title":"Towards All-Day Perception for Off-Road Driving: A Large-Scale Multispectral Dataset and Comprehensive Benchmark","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2","json":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2.json","graph_json":"https://pith.science/api/pith-number/PEHVVNOOKVN45EZGM4CW2ITFM2/graph.json","events_json":"https://pith.science/api/pith-number/PEHVVNOOKVN45EZGM4CW2ITFM2/events.json","paper":"https://pith.science/paper/PEHVVNOO"},"agent_actions":{"view_html":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2","download_json":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2.json","view_paper":"https://pith.science/paper/PEHVVNOO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.03832&json=true","fetch_graph":"https://pith.science/api/pith-number/PEHVVNOOKVN45EZGM4CW2ITFM2/graph.json","fetch_events":"https://pith.science/api/pith-number/PEHVVNOOKVN45EZGM4CW2ITFM2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2/action/storage_attestation","attest_author":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2/action/author_attestation","sign_citation":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2/action/citation_signature","submit_replication":"https://pith.science/pith/PEHVVNOOKVN45EZGM4CW2ITFM2/action/replication_record"}},"created_at":"2026-07-05T05:32:06.342930+00:00","updated_at":"2026-07-05T05:32:06.342930+00:00"}