{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6XCDNUM3MILRE2GB4VXFBXXL4B","short_pith_number":"pith:6XCDNUM3","schema_version":"1.0","canonical_sha256":"f5c436d19b62171268c1e56e50deebe0549eb5dd5d0b85811a258297181fa100","source":{"kind":"arxiv","id":"2505.14139","version":1},"attestation_state":"computed","paper":{"title":"FlowQ: Energy-Guided Flow Policies for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Botond Cseke, Marvin Alles, Nutan Chen, Patrick van der Smagt","submitted_at":"2025-05-20T09:43:05Z","abstract_excerpt":"The use of guidance to steer sampling toward desired outcomes has been widely explored within diffusion models, especially in applications such as image and trajectory generation. However, incorporating guidance during training remains relatively underexplored. In this work, we introduce energy-guided flow matching, a novel approach that enhances the training of flow models and eliminates the need for guidance at inference time. We learn a conditional velocity field corresponding to the flow policy by approximating an energy-guided probability path as a Gaussian path. Learning guided trajector"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.14139","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-20T09:43:05Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"094c29d1810ca9f5d4347fcdd4dc5cd5362420c7144d9380a080298bdcadd2c4","abstract_canon_sha256":"2c929739b7398f292433b0d7426a8acfc470f22b749f70432a42c043aefa2be9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:56.711721Z","signature_b64":"ddaIh087BhaRiYg2gXqm0aazpmWdJoFOZDFBrjCeED/9aLecaTuCP7INL9qHXM6FeLXeM8ARX/hX+sLe7MgiCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5c436d19b62171268c1e56e50deebe0549eb5dd5d0b85811a258297181fa100","last_reissued_at":"2026-07-05T11:05:56.711263Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:56.711263Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FlowQ: Energy-Guided Flow Policies for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Botond Cseke, Marvin Alles, Nutan Chen, Patrick van der Smagt","submitted_at":"2025-05-20T09:43:05Z","abstract_excerpt":"The use of guidance to steer sampling toward desired outcomes has been widely explored within diffusion models, especially in applications such as image and trajectory generation. However, incorporating guidance during training remains relatively underexplored. In this work, we introduce energy-guided flow matching, a novel approach that enhances the training of flow models and eliminates the need for guidance at inference time. We learn a conditional velocity field corresponding to the flow policy by approximating an energy-guided probability path as a Gaussian path. Learning guided trajector"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.14139","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.14139/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.14139","created_at":"2026-07-05T11:05:56.711329+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.14139v1","created_at":"2026-07-05T11:05:56.711329+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.14139","created_at":"2026-07-05T11:05:56.711329+00:00"},{"alias_kind":"pith_short_12","alias_value":"6XCDNUM3MILR","created_at":"2026-07-05T11:05:56.711329+00:00"},{"alias_kind":"pith_short_16","alias_value":"6XCDNUM3MILRE2GB","created_at":"2026-07-05T11:05:56.711329+00:00"},{"alias_kind":"pith_short_8","alias_value":"6XCDNUM3","created_at":"2026-07-05T11:05:56.711329+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09115","citing_title":"Counterfactual Transport Flows for Offline Conservative Trajectory Refinement","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08602","citing_title":"Reinforcement Learning for Flow-Matching Policies with Density Transport","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12622","citing_title":"Action Emergence from Streaming Intent","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12625","citing_title":"Driving Intents Amplify Planning-Oriented Reinforcement Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12622","citing_title":"Action Emergence from Streaming Intent","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12625","citing_title":"Driving Intents Amplify Planning-Oriented Reinforcement Learning","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B","json":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B.json","graph_json":"https://pith.science/api/pith-number/6XCDNUM3MILRE2GB4VXFBXXL4B/graph.json","events_json":"https://pith.science/api/pith-number/6XCDNUM3MILRE2GB4VXFBXXL4B/events.json","paper":"https://pith.science/paper/6XCDNUM3"},"agent_actions":{"view_html":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B","download_json":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B.json","view_paper":"https://pith.science/paper/6XCDNUM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.14139&json=true","fetch_graph":"https://pith.science/api/pith-number/6XCDNUM3MILRE2GB4VXFBXXL4B/graph.json","fetch_events":"https://pith.science/api/pith-number/6XCDNUM3MILRE2GB4VXFBXXL4B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B/action/storage_attestation","attest_author":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B/action/author_attestation","sign_citation":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B/action/citation_signature","submit_replication":"https://pith.science/pith/6XCDNUM3MILRE2GB4VXFBXXL4B/action/replication_record"}},"created_at":"2026-07-05T11:05:56.711329+00:00","updated_at":"2026-07-05T11:05:56.711329+00:00"}