- B-06: DagExecutor 接收 execution_id 下沉到 NodeContext,消除 dummy-execution-id 硬编码 - B-07: NodeContext.node_status 共享 self.state_machine.clone()(is_cancelled 可工作) - B-03a: HumanNode execute 重写为 subscribe→send→select! 循环(响应/超时/取消 + execution_id+node_id 双键 + Lagged 容忍),删假返回"同意" - eventbus.rs 删 try_recv_human_approval 死代码 - 新增 human_node 7 单测(正常/双键过滤/超时/非法决策/自由文本) 照 B-03-人工审批响应机制.md 设计。cargo check 0 error,df-nodes 14 + df-workflow 11 test 全过
243 lines
8.6 KiB
Rust
243 lines
8.6 KiB
Rust
//! DAG 执行器 — 按拓扑层级调度执行
|
||
|
||
use std::collections::HashMap;
|
||
|
||
use df_core::events::WorkflowEvent;
|
||
use df_core::types::NodeId;
|
||
|
||
use crate::dag::Dag;
|
||
use crate::eventbus::EventBus;
|
||
use crate::node::{NodeContext, NodeOutput};
|
||
use crate::state::StateMachine;
|
||
|
||
/// DAG 执行器
|
||
pub struct DagExecutor {
|
||
/// 事件总线
|
||
event_bus: EventBus,
|
||
/// 节点状态机
|
||
state_machine: StateMachine,
|
||
/// 工作流执行 ID(由调用方传入,下沉到每个 NodeContext)
|
||
execution_id: String,
|
||
}
|
||
|
||
impl DagExecutor {
|
||
/// 创建执行器
|
||
///
|
||
/// `execution_id` 为本次工作流执行的唯一标识,会下沉到每个节点的 NodeContext,
|
||
/// 用于节点内的事件关联、审计追踪等。
|
||
pub fn new(event_bus: EventBus, execution_id: String) -> Self {
|
||
Self {
|
||
event_bus,
|
||
state_machine: StateMachine::new(),
|
||
execution_id,
|
||
}
|
||
}
|
||
|
||
/// 执行 DAG 工作流
|
||
///
|
||
/// 同一拓扑层内的节点并发执行,层与层之间串行;
|
||
/// 本层全部节点完成后统一更新状态,任一失败则中止后续层。
|
||
pub async fn run(
|
||
&mut self,
|
||
dag: &Dag,
|
||
initial_config: serde_json::Value,
|
||
) -> anyhow::Result<HashMap<NodeId, NodeOutput>> {
|
||
let layers = dag.topological_layers()?;
|
||
let mut outputs: HashMap<NodeId, NodeOutput> = HashMap::new();
|
||
let start = std::time::Instant::now();
|
||
|
||
tracing::info!("DAG 执行开始,共 {} 层", layers.len());
|
||
|
||
for (layer_idx, layer) in layers.iter().enumerate() {
|
||
tracing::info!("执行第 {} 层,共 {} 个节点", layer_idx, layer.len());
|
||
|
||
// 阶段一:逐节点发 NodeStarted、置为运行中,并构建执行 future
|
||
let mut node_futures = Vec::with_capacity(layer.len());
|
||
for node_id in layer {
|
||
let node = dag.nodes.get(node_id).ok_or_else(|| {
|
||
anyhow::anyhow!("节点 {} 不存在于 DAG 中", node_id)
|
||
})?;
|
||
|
||
// 发送 NodeStarted 事件
|
||
self.event_bus
|
||
.send(WorkflowEvent::NodeStarted {
|
||
node_id: node_id.clone(),
|
||
})
|
||
.await;
|
||
|
||
self.state_machine.set_running(node_id.clone())?;
|
||
|
||
// 构建节点上下文
|
||
let mut inputs = HashMap::new();
|
||
for pred_id in dag.predecessors(node_id) {
|
||
if let Some(out) = outputs.get(&pred_id) {
|
||
inputs.insert(pred_id, out.clone());
|
||
}
|
||
}
|
||
|
||
let ctx = NodeContext {
|
||
node_id: node_id.clone(),
|
||
inputs,
|
||
config: initial_config.clone(),
|
||
execution_id: self.execution_id.clone(),
|
||
event_bus: self.event_bus.clone(),
|
||
// 共享执行器状态机,使节点(如 HumanNode)能读取真实状态而非空状态机
|
||
node_status: self.state_machine.clone(),
|
||
};
|
||
|
||
// 仅捕获节点共享引用与所有权上下文,避免与 self 借用冲突
|
||
let id = node_id.clone();
|
||
node_futures.push(async move {
|
||
let node_start = std::time::Instant::now();
|
||
let result = node.execute(ctx).await;
|
||
(id, result, node_start.elapsed().as_millis() as u64)
|
||
});
|
||
}
|
||
|
||
// 阶段二:同层节点并发执行,等待全部完成
|
||
let results = futures::future::join_all(node_futures).await;
|
||
|
||
// 阶段三:统一更新状态并发送事件,任一失败则整体返回 Err
|
||
let mut first_err: Option<anyhow::Error> = None;
|
||
for (node_id, result, duration) in results {
|
||
match result {
|
||
Ok(output) => {
|
||
self.state_machine.set_completed(node_id.clone())?;
|
||
self.event_bus
|
||
.send(WorkflowEvent::NodeCompleted {
|
||
node_id: node_id.clone(),
|
||
duration_ms: duration,
|
||
})
|
||
.await;
|
||
outputs.insert(node_id, output);
|
||
}
|
||
Err(e) => {
|
||
self.state_machine.set_failed(node_id.clone())?;
|
||
self.event_bus
|
||
.send(WorkflowEvent::NodeFailed {
|
||
node_id: node_id.clone(),
|
||
error: e.to_string(),
|
||
})
|
||
.await;
|
||
// 同层多个失败时只报告第一个
|
||
if first_err.is_none() {
|
||
first_err = Some(e.context(format!("节点 {} 执行失败", node_id)));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
if let Some(e) = first_err {
|
||
return Err(e);
|
||
}
|
||
}
|
||
|
||
let total = start.elapsed().as_millis() as u64;
|
||
self.event_bus
|
||
.send(WorkflowEvent::WorkflowCompleted {
|
||
total_duration_ms: total,
|
||
})
|
||
.await;
|
||
|
||
tracing::info!("DAG 执行完成,耗时 {}ms", total);
|
||
Ok(outputs)
|
||
}
|
||
}
|
||
|
||
#[cfg(test)]
|
||
mod tests {
|
||
use super::*;
|
||
use crate::node::{Node, NodeResult, NodeSchema};
|
||
use async_trait::async_trait;
|
||
use std::time::Duration;
|
||
|
||
/// 测试节点:sleep 指定毫秒后返回空输出
|
||
struct SleepNode {
|
||
sleep_ms: u64,
|
||
}
|
||
|
||
#[async_trait]
|
||
impl Node for SleepNode {
|
||
async fn execute(&self, _ctx: NodeContext) -> NodeResult {
|
||
tokio::time::sleep(Duration::from_millis(self.sleep_ms)).await;
|
||
Ok(NodeOutput::empty())
|
||
}
|
||
|
||
fn schema(&self) -> NodeSchema {
|
||
NodeSchema {
|
||
params: serde_json::Value::Null,
|
||
output: serde_json::Value::Null,
|
||
}
|
||
}
|
||
|
||
fn node_type(&self) -> &str {
|
||
"sleep"
|
||
}
|
||
}
|
||
|
||
/// 测试节点:直接返回错误
|
||
struct FailNode;
|
||
|
||
#[async_trait]
|
||
impl Node for FailNode {
|
||
async fn execute(&self, _ctx: NodeContext) -> NodeResult {
|
||
Err(anyhow::anyhow!("故意失败"))
|
||
}
|
||
|
||
fn schema(&self) -> NodeSchema {
|
||
NodeSchema {
|
||
params: serde_json::Value::Null,
|
||
output: serde_json::Value::Null,
|
||
}
|
||
}
|
||
|
||
fn node_type(&self) -> &str {
|
||
"fail"
|
||
}
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_same_layer_runs_in_parallel() {
|
||
// 两个无依赖节点位于同一层,各 sleep 100ms
|
||
let mut dag = Dag::new();
|
||
dag.add_node("a".to_string(), Box::new(SleepNode { sleep_ms: 100 }));
|
||
dag.add_node("b".to_string(), Box::new(SleepNode { sleep_ms: 100 }));
|
||
|
||
let mut executor = DagExecutor::new(EventBus::new(), "test-exec".to_string());
|
||
let start = std::time::Instant::now();
|
||
let outputs = executor.run(&dag, serde_json::Value::Null).await.unwrap();
|
||
let elapsed = start.elapsed();
|
||
|
||
assert_eq!(outputs.len(), 2);
|
||
// 串行需要约 200ms,并行应明显小于 180ms
|
||
assert!(
|
||
elapsed < Duration::from_millis(180),
|
||
"同层节点应并行执行,实际耗时 {:?}",
|
||
elapsed
|
||
);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_layer_failure_aborts_following_layers() {
|
||
// a(失败) 与 b(成功) 同层,c 依赖 a,失败后 c 不应执行
|
||
let mut dag = Dag::new();
|
||
dag.add_node("a".to_string(), Box::new(FailNode));
|
||
dag.add_node("b".to_string(), Box::new(SleepNode { sleep_ms: 10 }));
|
||
dag.add_node("c".to_string(), Box::new(SleepNode { sleep_ms: 10 }));
|
||
dag.add_edge("a".to_string(), "c".to_string());
|
||
|
||
let mut executor = DagExecutor::new(EventBus::new(), "test-exec".to_string());
|
||
let err = executor
|
||
.run(&dag, serde_json::Value::Null)
|
||
.await
|
||
.unwrap_err();
|
||
assert!(err.to_string().contains("节点 a 执行失败"));
|
||
|
||
// 同层成功节点状态正常更新,下游节点保持 Pending
|
||
use df_core::types::NodeStatus;
|
||
assert_eq!(executor.state_machine.get(&"a".to_string()), NodeStatus::Failed);
|
||
assert_eq!(executor.state_machine.get(&"b".to_string()), NodeStatus::Completed);
|
||
assert_eq!(executor.state_machine.get(&"c".to_string()), NodeStatus::Pending);
|
||
}
|
||
}
|