Compare commits

..

5 Commits

Author SHA1 Message Date
Codex 6fa95d796d docs: refresh planner priorities for 4416
Harness (E2E) / Harnesses (mock LLM) (push) Waiting to run
Harness (E2E) / Provider harnesses (live LLM conformance) (push) Waiting to run
Lint / golangci-lint (push) Waiting to run
Run Tests / Unit Tests (push) Waiting to run
Run Tests / Etcd Integration Tests (push) Waiting to run
2026-07-09 00:43:36 +00:00
Asim Aslam ad8ff2abde Harden model call timeout enforcement (#4411)
Co-authored-by: Codex <codex@openai.com>
2026-07-09 01:05:25 +01:00
Asim Aslam 5e6d51d261 docs: refresh planner priorities for 4407 (#4409)
Co-authored-by: Codex <codex@openai.com>
2026-07-09 00:42:18 +01:00
Asim Aslam 687a33faee Preserve checkpointed tool calls on agent resume (#4406)
goreleaser / goreleaser (push) Waiting to run
Co-authored-by: Codex <codex@openai.com>
2026-07-09 00:01:13 +01:00
Asim Aslam 48e752bbce docs: refresh planner priorities for 4403 (#4404)
Co-authored-by: Codex <codex@openai.com>
2026-07-08 23:34:57 +01:00
7 changed files with 165 additions and 3 deletions
+1 -1
View File
@@ -21,7 +21,7 @@ changes, architectural rewrites. Those go to the human.
## Work queue (ranked)
1. **Resume agent runs from checkpoints** ([#4368](https://github.com/micro/go-micro/issues/4368)) — #4391 closed the OpenTelemetry RunInfo gap, #4397 documented resume checkpoint limits, and #4399/#4402 moved the getting-started contract into CI, so the highest-value remaining depth seam is a focused, non-breaking durable-agent resume slice that preserves completed tool calls and avoids duplicate side effects before broader durability or API design work.
1. **Fix race in GenerateWithRetry timeout test** ([#4415](https://github.com/micro/go-micro/issues/4415)) — #4411 closed the provider-timeout hardening slice, but the follow-up CI signal shows the new per-attempt timeout coverage has an unsafe test-local counter under `go test -race`. Restore the green evaluator first so the loop can safely continue shipping adoption and harness work.
2. **Broaden provider streaming conformance** ([#4386](https://github.com/micro/go-micro/issues/4386)) — The blog says Anthropic streaming shipped, but the roadmap still calls for provider-backed streaming across chat and A2A. Add a focused, provider-gated conformance slice so streaming stays end-to-end rather than becoming a one-provider success story.
_Seeded by Claude Code from the roadmap + open issues; thereafter maintained by the
+5 -1
View File
@@ -432,9 +432,13 @@ func (a *agentImpl) askLocked(ctx context.Context, runID, message, parentRunID s
reply += resp.Answer
}
completedToolCalls := checkpointToolCalls(run.Steps)
if a.currentRun != nil {
completedToolCalls = checkpointToolCalls(a.currentRun.Steps)
}
res := &Response{
Reply: reply,
ToolCalls: resp.ToolCalls,
ToolCalls: mergeCheckpointToolCalls(completedToolCalls, resp.ToolCalls),
Agent: a.opts.Name,
RunID: a.runID,
ParentID: parentRunID,
+54
View File
@@ -4,6 +4,7 @@ import (
"context"
"encoding/json"
"fmt"
"strings"
"time"
"go-micro.dev/v6/ai"
@@ -288,3 +289,56 @@ func upsertStep(steps *[]flow.StepRecord, rec flow.StepRecord) int {
*steps = append(*steps, rec)
return len(*steps) - 1
}
func checkpointToolCalls(steps []flow.StepRecord) []ai.ToolCall {
calls := make([]ai.ToolCall, 0, len(steps))
for _, step := range steps {
call, ok := checkpointToolCall(step)
if !ok {
continue
}
calls = append(calls, call)
}
return calls
}
func checkpointToolCall(step flow.StepRecord) (ai.ToolCall, bool) {
if step.Status != "done" || !strings.HasPrefix(step.Name, "tool:") {
return ai.ToolCall{}, false
}
parts := strings.SplitN(strings.TrimPrefix(step.Name, "tool:"), ":", 2)
if len(parts) != 2 || parts[0] == "" {
return ai.ToolCall{}, false
}
input := map[string]any{}
if parts[1] != "null" && parts[1] != "" {
if err := json.Unmarshal([]byte(parts[1]), &input); err != nil {
return ai.ToolCall{}, false
}
}
return ai.ToolCall{Name: parts[0], Input: input, Result: step.Result}, true
}
func mergeCheckpointToolCalls(checkpointed, current []ai.ToolCall) []ai.ToolCall {
if len(checkpointed) == 0 {
return current
}
seen := make(map[string]struct{}, len(current))
for _, call := range current {
seen[toolCallKey(call.Name, call.Input)] = struct{}{}
}
merged := make([]ai.ToolCall, 0, len(checkpointed)+len(current))
for _, call := range checkpointed {
if _, ok := seen[toolCallKey(call.Name, call.Input)]; ok {
continue
}
merged = append(merged, call)
}
merged = append(merged, current...)
return merged
}
func toolCallKey(name string, input map[string]any) string {
b, _ := json.Marshal(input)
return name + ":" + string(b)
}
+6
View File
@@ -104,6 +104,12 @@ func TestResumeFailedCheckpointDoesNotReplayCompletedTool(t *testing.T) {
if resp.Reply != "finished from checkpoint" {
t.Fatalf("Resume reply = %q", resp.Reply)
}
if len(resp.ToolCalls) != 1 || resp.ToolCalls[0].Name != "external.charge" || resp.ToolCalls[0].Result != "charged" {
t.Fatalf("resumed tool calls = %#v, want preserved completed charge call", resp.ToolCalls)
}
if got := resp.ToolCalls[0].Input["order"]; got != "42" {
t.Fatalf("resumed tool input order = %#v, want 42", got)
}
if toolRuns != 1 {
t.Fatalf("tool executions after Resume = %d, want completed tool was not replayed", toolRuns)
}
+49
View File
@@ -240,6 +240,55 @@ func TestAskCancellationDuringToolCallFailsRun(t *testing.T) {
}
}
func TestSlowProviderTimeoutPreventsLateToolSideEffects(t *testing.T) {
started := make(chan struct{})
release := make(chan struct{})
done := make(chan struct{})
fakeGen = func(ctx context.Context, opts ai.Options, req *ai.Request) (*ai.Response, error) {
close(started)
<-release
defer close(done)
if opts.ToolHandler == nil {
t.Fatal("missing tool handler")
}
res := opts.ToolHandler(ctx, ai.ToolCall{ID: "late-1", Name: "external.create", Input: map[string]any{"title": "too late"}})
if !strings.Contains(res.Content, context.DeadlineExceeded.Error()) {
t.Errorf("late tool result = %q, want deadline exceeded", res.Content)
}
return &ai.Response{Reply: "late", ToolCalls: []ai.ToolCall{{ID: "late-1", Name: "external.create", Input: map[string]any{"title": "too late"}, Result: res.Content}}}, nil
}
defer func() { fakeGen = nil }()
toolRuns := 0
a := newTestAgent(
Name("slow-provider-late-tool"),
ModelCallTimeout(10*time.Millisecond),
WithTool("external.create", "create once", nil, func(context.Context, map[string]any) (string, error) {
toolRuns++
return "created", nil
}),
)
_, err := a.Ask(context.Background(), "provider times out before tool")
if !errors.Is(err, context.DeadlineExceeded) {
t.Fatalf("Ask error = %v, want deadline exceeded", err)
}
select {
case <-started:
default:
t.Fatal("provider was not called")
}
close(release)
select {
case <-done:
case <-time.After(time.Second):
t.Fatal("late provider call did not finish")
}
if toolRuns != 0 {
t.Fatalf("late tool executions = %d, want 0", toolRuns)
}
}
func TestAskCheckpointRecordsTerminalOperationalFailureStatus(t *testing.T) {
tests := []struct {
name string
+22 -1
View File
@@ -157,7 +157,7 @@ func GenerateWithRetry(ctx context.Context, m Model, req *Request, policy Genera
info.MaxAttempts = policy.MaxAttempts
callCtx = WithRunInfo(callCtx, info)
}
resp, err := m.Generate(callCtx, req, opts...)
resp, err := generateAttempt(callCtx, m, req, opts...)
cancel()
// Caller cancellation/deadline always wins and is not retried, even if
@@ -197,6 +197,27 @@ func GenerateWithRetry(ctx context.Context, m Model, req *Request, policy Genera
return nil, &RetryError{Attempts: policy.MaxAttempts, Kind: ClassifyError(last), Err: last}
}
func generateAttempt(ctx context.Context, m Model, req *Request, opts ...GenerateOption) (*Response, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
type result struct {
resp *Response
err error
}
done := make(chan result, 1)
go func() {
resp, err := m.Generate(ctx, req, opts...)
done <- result{resp: resp, err: err}
}()
select {
case res := <-done:
return res.resp, res.err
case <-ctx.Done():
return nil, ctx.Err()
}
}
func retryBackoff(err error, attempt int, base time.Duration) time.Duration {
backoff := base
if backoff <= 0 {
+28
View File
@@ -135,6 +135,34 @@ func TestGenerateWithRetryAddsAttemptMetadataToRunInfo(t *testing.T) {
}
}
func TestGenerateWithRetryReturnsWhenProviderIgnoresTimeout(t *testing.T) {
started := make(chan struct{})
release := make(chan struct{})
model := retryModel{generate: func(ctx context.Context, req *Request, opts ...GenerateOption) (*Response, error) {
close(started)
<-release
return &Response{Reply: "late"}, nil
}}
defer close(release)
start := time.Now()
_, err := GenerateWithRetry(context.Background(), model, &Request{Prompt: "hi"}, GeneratePolicy{
Timeout: 10 * time.Millisecond,
MaxAttempts: 1,
})
if !errors.Is(err, context.DeadlineExceeded) {
t.Fatalf("GenerateWithRetry error = %v, want deadline exceeded", err)
}
if elapsed := time.Since(start); elapsed > 200*time.Millisecond {
t.Fatalf("GenerateWithRetry took %s after deadline, want prompt return", elapsed)
}
select {
case <-started:
default:
t.Fatal("provider was not called")
}
}
type statusErr int
func (e statusErr) Error() string { return "provider status" }