{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JI3SDLS4N2A3XSIT45H7NXKMN7","short_pith_number":"pith:JI3SDLS4","schema_version":"1.0","canonical_sha256":"4a3721ae5c6e81bbc913e74ff6dd4c6fc9f984f10bb73addcf0bb3d0c3926db5","source":{"kind":"arxiv","id":"2303.08319","version":2},"attestation_state":"computed","paper":{"title":"FAQ: Feature Aggregated Queries for Transformer-based Video Object Detectors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Linjie Yang, Yiming Cui","submitted_at":"2023-03-15T02:14:56Z","abstract_excerpt":"Video object detection needs to solve feature degradation situations that rarely happen in the image domain. One solution is to use the temporal information and fuse the features from the neighboring frames. With Transformerbased object detectors getting a better performance on the image domain tasks, recent works began to extend those methods to video object detection. However, those existing Transformer-based video object detectors still follow the same pipeline as those used for classical object detectors, like enhancing the object feature representations by aggregation. In this work, we ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.08319","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-03-15T02:14:56Z","cross_cats_sorted":[],"title_canon_sha256":"0aed11707dfd60bb90c1d694e010cbf033235e967d1611b914e0bdb6b08d5d52","abstract_canon_sha256":"9c7a16ff87f8b712068dace4c110f9002bc19be5826ec164a6ae85b34cf33ea6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:52:46.940058Z","signature_b64":"7SA4Nvu4xGHlPYyv/QjzcSPIpL7dSjgX0aKOyELQIFZdHL8LVR4owwGipHhm4mx3RUaZpMzd3ScILeJ6V/mbDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a3721ae5c6e81bbc913e74ff6dd4c6fc9f984f10bb73addcf0bb3d0c3926db5","last_reissued_at":"2026-07-05T05:52:46.939669Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:52:46.939669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FAQ: Feature Aggregated Queries for Transformer-based Video Object Detectors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Linjie Yang, Yiming Cui","submitted_at":"2023-03-15T02:14:56Z","abstract_excerpt":"Video object detection needs to solve feature degradation situations that rarely happen in the image domain. One solution is to use the temporal information and fuse the features from the neighboring frames. With Transformerbased object detectors getting a better performance on the image domain tasks, recent works began to extend those methods to video object detection. However, those existing Transformer-based video object detectors still follow the same pipeline as those used for classical object detectors, like enhancing the object feature representations by aggregation. In this work, we ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.08319","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.08319/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.08319","created_at":"2026-07-05T05:52:46.939723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.08319v2","created_at":"2026-07-05T05:52:46.939723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.08319","created_at":"2026-07-05T05:52:46.939723+00:00"},{"alias_kind":"pith_short_12","alias_value":"JI3SDLS4N2A3","created_at":"2026-07-05T05:52:46.939723+00:00"},{"alias_kind":"pith_short_16","alias_value":"JI3SDLS4N2A3XSIT","created_at":"2026-07-05T05:52:46.939723+00:00"},{"alias_kind":"pith_short_8","alias_value":"JI3SDLS4","created_at":"2026-07-05T05:52:46.939723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.17079","citing_title":"Few-Shot Learning in Video and 3D Object Detection: A Survey","ref_index":89,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7","json":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7.json","graph_json":"https://pith.science/api/pith-number/JI3SDLS4N2A3XSIT45H7NXKMN7/graph.json","events_json":"https://pith.science/api/pith-number/JI3SDLS4N2A3XSIT45H7NXKMN7/events.json","paper":"https://pith.science/paper/JI3SDLS4"},"agent_actions":{"view_html":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7","download_json":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7.json","view_paper":"https://pith.science/paper/JI3SDLS4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.08319&json=true","fetch_graph":"https://pith.science/api/pith-number/JI3SDLS4N2A3XSIT45H7NXKMN7/graph.json","fetch_events":"https://pith.science/api/pith-number/JI3SDLS4N2A3XSIT45H7NXKMN7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7/action/storage_attestation","attest_author":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7/action/author_attestation","sign_citation":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7/action/citation_signature","submit_replication":"https://pith.science/pith/JI3SDLS4N2A3XSIT45H7NXKMN7/action/replication_record"}},"created_at":"2026-07-05T05:52:46.939723+00:00","updated_at":"2026-07-05T05:52:46.939723+00:00"}