First Commit
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s
ci / go (push) Waiting to run
ci / go-db (agent) (push) Waiting to run
ci / go-db (config) (push) Waiting to run
ci / go-db (db) (push) Waiting to run
ci / go-db (evidence) (push) Waiting to run
ci / go-db (llmrec) (push) Waiting to run
ci / go-db (server) (push) Waiting to run
web / web (push) Waiting to run
docs / links (push) Canceled after 0s
detections / detections (push) Canceled after 0s
This commit is contained in:
@@ -0,0 +1,272 @@
|
||||
package server
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"strconv"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/Autumn-27/artex/agent"
|
||||
"github.com/Autumn-27/artex/db"
|
||||
)
|
||||
|
||||
// 任务级超时协调器(见 docs/任务级超时与收尾设计.md §4/§8.5)。
|
||||
// 绝对墙钟:每个带 timeout 的任务一个定时 goroutine,到点驱动有序收尾时序:
|
||||
// ① settling → ② worker 停领新意图 / ③ planner 丢弃普通 notify
|
||||
// ④ 等在跑 worker drain(受 grace) → ⑤ 终局一轮 planner 判定 → ⑥ 定终态(带守卫)
|
||||
|
||||
const (
|
||||
settleDrainGrace = 90 * time.Second // 等在跑 worker 优雅收尾的上限;超过则硬 cancel
|
||||
deadlinePollInterval = 2 * time.Second // deadline 未盖章/LLM 未就绪时的轮询间隔
|
||||
deadlineMaxSleep = 30 * time.Second // 单次最长睡眠(便于周期复查终态)
|
||||
)
|
||||
|
||||
// ---------- settling 状态 ----------
|
||||
|
||||
func (e *Engine) isSettling(taskID string) bool {
|
||||
v, _ := e.settling.Load(taskID)
|
||||
b, _ := v.(bool)
|
||||
return b
|
||||
}
|
||||
|
||||
// markSettling flips settling on; returns true only for the first caller.
|
||||
func (e *Engine) markSettling(taskID string) bool {
|
||||
_, loaded := e.settling.LoadOrStore(taskID, true)
|
||||
return !loaded
|
||||
}
|
||||
|
||||
// ---------- 在跑计数(worker.Execute + planner.Plan),用于 drain ----------
|
||||
|
||||
func (e *Engine) inflightCounter(taskID string) *int64 {
|
||||
v, _ := e.inflight.LoadOrStore(taskID, new(int64))
|
||||
return v.(*int64)
|
||||
}
|
||||
|
||||
// beginTaskOperation atomically registers a task-owned operation unless deletion
|
||||
// has already installed its barrier. The delete handler can therefore wait for
|
||||
// inflight==0 without a check-then-start race recreating files after cleanup.
|
||||
func (e *Engine) beginTaskOperation(taskID string) bool {
|
||||
e.deleteMu.RLock()
|
||||
defer e.deleteMu.RUnlock()
|
||||
if e.IsDeleting(taskID) {
|
||||
return false
|
||||
}
|
||||
atomic.AddInt64(e.inflightCounter(taskID), 1)
|
||||
return true
|
||||
}
|
||||
|
||||
func (e *Engine) decInflight(taskID string) { atomic.AddInt64(e.inflightCounter(taskID), -1) }
|
||||
func (e *Engine) inflightCount(taskID string) int64 {
|
||||
return atomic.LoadInt64(e.inflightCounter(taskID))
|
||||
}
|
||||
|
||||
// ---------- deadline ----------
|
||||
|
||||
// taskDeadline returns the task's absolute deadline (unix). Prefers the in-process
|
||||
// map (stamped this session); falls back to the DB-loaded value (restart), seeding
|
||||
// the map. 0 = no timeout / not yet stamped.
|
||||
func (e *Engine) taskDeadline(t *Task) int64 {
|
||||
if v, ok := e.deadline.Load(t.ID); ok {
|
||||
return v.(int64)
|
||||
}
|
||||
deadlineAt := t.lifecycleSnapshot().DeadlineAt
|
||||
if deadlineAt > 0 {
|
||||
e.deadline.Store(t.ID, deadlineAt)
|
||||
return deadlineAt
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// resetTimeoutRevival clears only the per-run timeout state after PostgreSQL has
|
||||
// atomically committed timeout -> running and reset first_run_at/deadline_at. The
|
||||
// configured TimeoutSeconds remains on Task, so the next real Planner/Worker run
|
||||
// stamps a fresh full budget. coordStarted is reset because the coordinator that
|
||||
// produced the timeout has already completed (or is in its final return path).
|
||||
func (e *Engine) resetTimeoutRevival(taskID string) {
|
||||
e.settling.Delete(taskID)
|
||||
e.deadline.Delete(taskID)
|
||||
e.stamped.Delete(taskID)
|
||||
e.coordStarted.Delete(taskID)
|
||||
}
|
||||
|
||||
// stampFirstRun records first_run_at + deadline_at on the FIRST real run (LLM ready)
|
||||
// of a timeout task, once per process. No-op when the task has no timeout.
|
||||
func (e *Engine) stampFirstRun(t *Task) {
|
||||
if t.TimeoutSeconds <= 0 {
|
||||
return
|
||||
}
|
||||
if _, loaded := e.stamped.LoadOrStore(t.ID, true); loaded {
|
||||
return
|
||||
}
|
||||
dl, err := e.m.StampTaskFirstRun(t.ID)
|
||||
if err != nil {
|
||||
log.Printf("[deadline] task %s 盖章 first_run 失败: %v", t.ID, err)
|
||||
e.stamped.Delete(t.ID) // 允许下次重试
|
||||
return
|
||||
}
|
||||
if dl > 0 {
|
||||
e.deadline.Store(t.ID, dl)
|
||||
log.Printf("[deadline] task %s 首次运行,截止于 %s", t.ID, time.Unix(dl, 0).Format("2006-01-02 15:04:05"))
|
||||
}
|
||||
}
|
||||
|
||||
// clockCtx layers the task's TaskClock (absolute deadline) onto a run's context so
|
||||
// worker/planner can clamp their wall-clock budget and pick per-run vs task-timeout
|
||||
// wrap-up words. final marks the coordinator-driven terminal planner round.
|
||||
func (e *Engine) clockCtx(base context.Context, t *Task, final bool) context.Context {
|
||||
dl := e.taskDeadline(t)
|
||||
if dl <= 0 && !final {
|
||||
return base // no timeout → unchanged behavior
|
||||
}
|
||||
return agent.WithTaskClock(base, agent.TaskClock{DeadlineUnix: dl, Final: final})
|
||||
}
|
||||
|
||||
// ---------- 协调器 ----------
|
||||
|
||||
// startDeadlineCoordinator launches the per-task deadline timer once (idempotent).
|
||||
// Called from Run() and from the restart reload path, so non-active timeout tasks
|
||||
// still get settled after their deadline even without live planner/worker loops.
|
||||
func (e *Engine) startDeadlineCoordinator(ctx context.Context, t *Task) {
|
||||
if t == nil || t.TimeoutSeconds <= 0 {
|
||||
return
|
||||
}
|
||||
e.deleteMu.RLock()
|
||||
if e.IsDeleting(t.ID) {
|
||||
e.deleteMu.RUnlock()
|
||||
return
|
||||
}
|
||||
if _, loaded := e.coordStarted.LoadOrStore(t.ID, true); loaded {
|
||||
e.deleteMu.RUnlock()
|
||||
return
|
||||
}
|
||||
rt := e.registerTaskRoutines(ctx, t.ID, 1)
|
||||
e.deleteMu.RUnlock()
|
||||
runTaskRoutine(rt, func(loopCtx context.Context) { e.deadlineCoordinator(loopCtx, t) })
|
||||
}
|
||||
|
||||
// deadlineCoordinator waits until the task's absolute deadline, then runs the settle
|
||||
// sequence. Absolute wall-clock: it keeps counting through pauses.
|
||||
func (e *Engine) deadlineCoordinator(ctx context.Context, t *Task) {
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
default:
|
||||
}
|
||||
if isTerminalStatus(e.m.TaskStatus(t.ID)) {
|
||||
return // already finished (goals met / failed) — nothing to time out
|
||||
}
|
||||
dl := e.taskDeadline(t)
|
||||
if dl <= 0 {
|
||||
if sleepCtx(ctx, deadlinePollInterval) { // not yet stamped (task hasn't really run)
|
||||
return
|
||||
}
|
||||
continue
|
||||
}
|
||||
if remaining := time.Until(time.Unix(dl, 0)); remaining > 0 {
|
||||
nap := remaining
|
||||
if nap > deadlineMaxSleep {
|
||||
nap = deadlineMaxSleep
|
||||
}
|
||||
if sleepCtx(ctx, nap) {
|
||||
return
|
||||
}
|
||||
continue
|
||||
}
|
||||
e.settleTask(ctx, t)
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// settleTask runs the ordered settle sequence once (§4 steps ①–⑥).
|
||||
func (e *Engine) settleTask(ctx context.Context, t *Task) {
|
||||
if !e.markSettling(t.ID) {
|
||||
return
|
||||
}
|
||||
log.Printf("[deadline] task %s 到达超时上限,进入收尾时序", t.ID)
|
||||
|
||||
// ④ 等在跑 worker/planner drain(在跑 run 因夹逼的 MaxDuration 自行进收尾);
|
||||
// 超过 grace 仍未清空 → 硬 cancel 该任务 exec ctx(settling-aware 分支正确归类)。
|
||||
hardStop := time.Now().Add(settleDrainGrace)
|
||||
for e.inflightCount(t.ID) > 0 {
|
||||
if time.Now().After(hardStop) {
|
||||
log.Printf("[deadline] task %s drain 超时(%s),硬取消在跑 run", t.ID, settleDrainGrace)
|
||||
e.cancelExec(t.ID, agent.AbortSettleDrainTimeout)
|
||||
_ = sleepCtx(ctx, 3*time.Second) // 给 worker 分支一点时间落库/归类
|
||||
break
|
||||
}
|
||||
if sleepCtx(ctx, 500*time.Millisecond) {
|
||||
return // 引擎整体关停
|
||||
}
|
||||
}
|
||||
|
||||
// ⑤ 终局一轮 planner(任务超时词,最后目标判定,不产新意图)。
|
||||
met := e.runFinalPlannerRound(ctx, t)
|
||||
if !e.beginTaskOperation(t.ID) {
|
||||
return
|
||||
}
|
||||
defer e.decInflight(t.ID)
|
||||
|
||||
// ⑥ 定终态(带守卫):met → done(completed);否则 timeout。若常规路径已先落 done,
|
||||
// 守卫(SetTaskStatusGuarded)会拒绝覆盖,保留 completed 语义。
|
||||
status := "timeout"
|
||||
if met {
|
||||
status = "done"
|
||||
}
|
||||
won, err := e.m.SetTaskStatusGuarded(t.ID, status)
|
||||
switch {
|
||||
case err != nil:
|
||||
log.Printf("[deadline] task %s 落终态失败: %v", t.ID, err)
|
||||
case won:
|
||||
log.Printf("[deadline] task %s 收尾完成,终态=%s", t.ID, status)
|
||||
default:
|
||||
log.Printf("[deadline] task %s 收尾时已是终态,保留原状态", t.ID)
|
||||
}
|
||||
}
|
||||
|
||||
// runFinalPlannerRound drives exactly ONE terminal planner round with the
|
||||
// task-timeout planner words (final goal judgment; no new intents). Waits for the
|
||||
// LLM to be ready (bounded by ctx) so a completable task isn't mis-judged timeout.
|
||||
// timeoutFinalRoundSummaryFmt 는 작업 시간 초과 시 마지막 계획 라운드의 표시 전용
|
||||
// 활동 요약이다(node_id 없음·전사에만 노출·되먹임 경로 미접촉). [[G132]]
|
||||
const timeoutFinalRoundSummaryFmt = "작업 시간 초과 마무리·최종 판정(%d차)"
|
||||
|
||||
func (e *Engine) runFinalPlannerRound(ctx context.Context, t *Task) (met bool) {
|
||||
if e.IsDeleting(t.ID) {
|
||||
return false
|
||||
}
|
||||
planner, _ := e.snapshotFor(t)
|
||||
for planner == nil {
|
||||
if sleepCtx(ctx, deadlinePollInterval) {
|
||||
return false
|
||||
}
|
||||
if isTerminalStatus(e.m.TaskStatus(t.ID)) {
|
||||
return false
|
||||
}
|
||||
if e.IsDeleting(t.ID) {
|
||||
return false
|
||||
}
|
||||
planner, _ = e.snapshotFor(t)
|
||||
}
|
||||
// 独立 ctx(不挂 execCancel,避免 pause/硬 cancel 打断这最后一轮),带 Final 注入任务超时词。
|
||||
fctx := e.clockCtx(ctx, t, true)
|
||||
if !e.beginTaskOperation(t.ID) {
|
||||
return false
|
||||
}
|
||||
defer e.decInflight(t.ID)
|
||||
emit := func(r db.Activity) { e.emitActivity(t, r) }
|
||||
e.emitActivity(t, db.Activity{Worker: "planner", Kind: "round",
|
||||
Summary: fmt.Sprintf(timeoutFinalRoundSummaryFmt, e.nextPlannerRound(t.ID))})
|
||||
tTaskID, _ := strconv.ParseInt(t.ID, 10, 64)
|
||||
e.BeginLLMCall(t.ID)
|
||||
met, reason, err := planner.Plan(fctx, tTaskID, e.m.assets, t.Store, t.Goal, t.drainTriggers(), emit)
|
||||
e.EndLLMCall(t.ID)
|
||||
if err != nil {
|
||||
log.Printf("[deadline] task %s 终局规划出错: %v", t.ID, err)
|
||||
} else if met {
|
||||
log.Printf("[deadline] task %s 终局判定目标达成: %s", t.ID, reason)
|
||||
}
|
||||
return met
|
||||
}
|
||||
Reference in New Issue
Block a user