Agent runtime: majordomo in-process, Ollama Cloud config, chat endpoint (#56) (#70)
Build image / build-and-push (push) Successful in 6s
Build image / build-and-push (push) Successful in 6s
Co-authored-by: Steve Dudenhoeffer <[email protected]>
This commit was merged in pull request #70.
This commit is contained in:
+32
-14
@@ -1,18 +1,36 @@
|
||||
// Package agent adapts pansy's service layer to majordomo tools so an agent can
|
||||
// drive a garden in natural language ("fill the NE corner with garlic, the NW
|
||||
// with basil, the south half with beans"). Each tool runs as a fixed actor and
|
||||
// therefore inherits pansy's ACL checks for free — a viewer's fill_region returns
|
||||
// ErrForbidden, exactly as the REST API would.
|
||||
// Package agent is pansy's garden assistant: a majordomo toolbox over the
|
||||
// service layer, plus the run loop that drives a model with it.
|
||||
//
|
||||
// Two deliberate separations keep the core server lean:
|
||||
// Every tool runs as a fixed actor, so the agent inherits pansy's permission
|
||||
// checks for free — a viewer's fill_region returns ErrForbidden, exactly as the
|
||||
// REST API would. That is the whole reason the tools are thin adapters over
|
||||
// internal/service and never reach past it; the moment one does, the ACL story
|
||||
// stops being automatic and becomes something to remember.
|
||||
//
|
||||
// - cmd/pansy does NOT import this package, so the server binary carries no
|
||||
// agent/LLM dependencies.
|
||||
// - the tool wiring (tools.go) sits behind the `majordomo` build tag, so the
|
||||
// default `go build ./...` / `go test ./...` compiles without the majordomo
|
||||
// module. Build the agent tools with `-tags majordomo` once majordomo is a
|
||||
// dependency (`go get gitea.stevedudenhoeffer.com/steve/majordomo`).
|
||||
// # A turn is one change set
|
||||
//
|
||||
// The agent harness itself (model loop, chat surface) lives in the
|
||||
// majordomo/executus stack, outside this repo; this package is only the toolbox.
|
||||
// Runs wrap their whole turn in a single change set (source "agent"), so
|
||||
// "empty the garlic bed and plant cucumbers" — one object edit and a dozen
|
||||
// planting inserts — undoes as one action rather than thirteen. That is what
|
||||
// makes acting without a confirmation prompt defensible: the answer to "it did
|
||||
// the wrong thing" is one click, not an archaeology exercise.
|
||||
//
|
||||
// # No build tag
|
||||
//
|
||||
// This package used to sit behind a `majordomo` build tag, with cmd/pansy
|
||||
// deliberately not importing it, so the default build carried no LLM
|
||||
// dependency. Both separations are gone, on purpose: a tag that keeps the agent
|
||||
// out of the binary only earns its keep if you would ever ship a build without
|
||||
// the agent, and the agent is the point. Keeping it would have meant an
|
||||
// untagged CI that never compiled the code that matters.
|
||||
//
|
||||
// majordomo is stdlib-first and pure Go, so CGO_ENABLED=0 and the single static
|
||||
// binary survive it.
|
||||
//
|
||||
// # Unconfigured instances
|
||||
//
|
||||
// With no API key the assistant is simply not offered: the chat route isn't
|
||||
// registered and the capability isn't advertised — the same shape as OIDC
|
||||
// 404ing when unconfigured. An instance without a key starts and serves the app
|
||||
// exactly as it did before.
|
||||
package agent
|
||||
|
||||
@@ -0,0 +1,240 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log/slog"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.stevedudenhoeffer.com/steve/majordomo"
|
||||
"gitea.stevedudenhoeffer.com/steve/majordomo/agent"
|
||||
"gitea.stevedudenhoeffer.com/steve/majordomo/llm"
|
||||
"gitea.stevedudenhoeffer.com/steve/majordomo/provider/ollama"
|
||||
|
||||
"gitea.stevedudenhoeffer.com/steve/pansy/internal/config"
|
||||
"gitea.stevedudenhoeffer.com/steve/pansy/internal/domain"
|
||||
"gitea.stevedudenhoeffer.com/steve/pansy/internal/service"
|
||||
)
|
||||
|
||||
// Loop bounds. These exist so a stuck run terminates, NOT to control spend —
|
||||
// pansy is a personal tool and multi-tenant cost control is explicitly not a
|
||||
// concern here. A model that gets wedged calling describe_garden forever should
|
||||
// stop on its own, and a hung upstream shouldn't hold a connection open all day.
|
||||
const (
|
||||
maxSteps = 24
|
||||
// A turn that legitimately fills several beds does a lot of round trips, so
|
||||
// this is generous; it is a backstop, not a budget.
|
||||
runTimeout = 4 * time.Minute
|
||||
// Successive all-error steps, and identical repeated calls, that end a run.
|
||||
maxConsecutiveToolErrors = 4
|
||||
maxSameCallRepeats = 3
|
||||
)
|
||||
|
||||
// Runner drives a model over pansy's toolbox. One per process; Run is safe to
|
||||
// call concurrently.
|
||||
type Runner struct {
|
||||
svc *service.Service
|
||||
model llm.Model
|
||||
}
|
||||
|
||||
// NewRunner resolves the configured model and returns a Runner, or an error if
|
||||
// the assistant can't be offered. Callers should treat an error as "no
|
||||
// assistant" rather than a startup failure — an instance with no key must still
|
||||
// serve the app.
|
||||
func NewRunner(svc *service.Service, cfg *config.Config) (*Runner, error) {
|
||||
if !cfg.Agent.Ready() {
|
||||
return nil, errors.New("agent: not configured")
|
||||
}
|
||||
|
||||
// A private registry, not the package-level default: pansy passes the key it
|
||||
// was configured with rather than depending on ambient environment, and
|
||||
// majordomo's own ollama-cloud preset reads OLLAMA_API_KEY while pansy (like
|
||||
// gadfly) is configured with OLLAMA_CLOUD_API_KEY. Registering the provider
|
||||
// explicitly makes that bridge visible instead of a mysterious empty token.
|
||||
reg := majordomo.New()
|
||||
reg.RegisterProvider(ollama.Cloud(ollama.WithToken(cfg.Agent.OllamaCloudAPIKey)))
|
||||
|
||||
// The spec goes to Parse VERBATIM. The grammar — including comma-separated
|
||||
// failover chains — is majordomo's, and re-implementing any of it here would
|
||||
// only mean two places to update when it grows.
|
||||
model, err := reg.Parse(cfg.Agent.Model)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("agent: resolve model %q: %w", cfg.Agent.Model, err)
|
||||
}
|
||||
return &Runner{svc: svc, model: model}, nil
|
||||
}
|
||||
|
||||
// Turn is the outcome of one exchange.
|
||||
type Turn struct {
|
||||
// Reply is what to show the user.
|
||||
Reply string `json:"reply"`
|
||||
// ChangeSetID is the change set this turn produced, if it changed anything —
|
||||
// the handle the UI needs to offer Undo.
|
||||
ChangeSetID *int64 `json:"changeSetId,omitempty"`
|
||||
// Steps is how many model round trips it took.
|
||||
Steps int `json:"steps"`
|
||||
// Truncated is set when the run hit its step cap rather than finishing.
|
||||
Truncated bool `json:"truncated,omitempty"`
|
||||
}
|
||||
|
||||
// Run executes one turn against a garden, as actorID.
|
||||
//
|
||||
// The whole turn runs inside ONE change set, so everything the model did undoes
|
||||
// together. That is what makes acting without a confirmation prompt defensible.
|
||||
// The scope is opened even for a turn that turns out to be a question — a change
|
||||
// set with no revisions is never written, so asking costs nothing.
|
||||
func (r *Runner) Run(ctx context.Context, actorID, gardenID int64, message string, history []llm.Message, onStep func(agent.Step)) (*Turn, error) {
|
||||
message = strings.TrimSpace(message)
|
||||
if message == "" {
|
||||
return nil, domain.ErrInvalidInput
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(ctx, runTimeout)
|
||||
defer cancel()
|
||||
|
||||
garden, err := r.svc.GetGarden(ctx, actorID, gardenID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// An id for this run, stamped on the change set so a row in the history list
|
||||
// can be matched to the log lines that produced it. Without it, "the agent
|
||||
// did something odd on Tuesday" has no thread back to what it was thinking.
|
||||
runID := newRunID()
|
||||
slog.Info("agent: run start", "run", runID, "garden", gardenID, "actor", actorID)
|
||||
|
||||
var (
|
||||
result *agent.Result
|
||||
runErr error
|
||||
truncErr bool
|
||||
)
|
||||
changeSet, err := r.svc.WithChangeSet(ctx, actorID, gardenID, service.ChangeSetOptions{
|
||||
Source: domain.SourceAgent,
|
||||
Summary: turnSummary(message),
|
||||
AgentRunID: &runID,
|
||||
}, func(ctx context.Context) error {
|
||||
box := NewToolbox(r.svc, actorID)
|
||||
a := agent.New(r.model, systemPrompt(garden),
|
||||
agent.WithMaxSteps(maxSteps),
|
||||
agent.WithToolErrorLimits(maxConsecutiveToolErrors, maxSameCallRepeats),
|
||||
)
|
||||
a.AddToolbox(box)
|
||||
|
||||
opts := []agent.RunOption{agent.WithHistory(history)}
|
||||
if onStep != nil {
|
||||
opts = append(opts, agent.OnStep(onStep))
|
||||
}
|
||||
result, runErr = a.Run(ctx, message, opts...)
|
||||
// A run that ran out of steps still DID things, and those things must be
|
||||
// recorded and undoable. So the loop-guard errors don't fail the scope —
|
||||
// they're reported to the user instead.
|
||||
if runErr != nil && isLoopLimit(runErr) {
|
||||
truncErr = true
|
||||
return nil
|
||||
}
|
||||
return runErr
|
||||
})
|
||||
if err != nil {
|
||||
// WithChangeSet records whatever committed before failing, so the partial
|
||||
// work is undoable even though the turn errored.
|
||||
return nil, err
|
||||
}
|
||||
|
||||
turn := &Turn{Truncated: truncErr}
|
||||
if changeSet != nil {
|
||||
turn.ChangeSetID = &changeSet.ID
|
||||
}
|
||||
if result != nil {
|
||||
turn.Reply = result.Output
|
||||
turn.Steps = len(result.Steps)
|
||||
}
|
||||
if turn.Reply == "" {
|
||||
turn.Reply = fallbackReply(turn)
|
||||
}
|
||||
return turn, nil
|
||||
}
|
||||
|
||||
// isLoopLimit reports whether an error is one of majordomo's loop guards firing
|
||||
// rather than a genuine failure. Those runs have a partial result worth keeping.
|
||||
func isLoopLimit(err error) bool {
|
||||
return errors.Is(err, agent.ErrMaxSteps) || errors.Is(err, agent.ErrToolLoop)
|
||||
}
|
||||
|
||||
// fallbackReply covers a run that finished with no text — a model that made its
|
||||
// last tool call and then stopped. Silence reads as a failure, so say what
|
||||
// happened.
|
||||
func fallbackReply(t *Turn) string {
|
||||
switch {
|
||||
case t.Truncated:
|
||||
return "I stopped partway through — that turned into more steps than I should take in one go. " +
|
||||
"Have a look at what changed, and tell me what to do next."
|
||||
case t.ChangeSetID != nil:
|
||||
return "Done — have a look at the canvas."
|
||||
default:
|
||||
return "I didn't change anything."
|
||||
}
|
||||
}
|
||||
|
||||
// newRunID returns a short random identifier for one run.
|
||||
func newRunID() string {
|
||||
var b [8]byte
|
||||
if _, err := rand.Read(b[:]); err != nil {
|
||||
// The id is for correlating logs, not for security. A clock-based
|
||||
// fallback is worse than random and better than an empty string.
|
||||
return fmt.Sprintf("t%d", time.Now().UnixNano())
|
||||
}
|
||||
return hex.EncodeToString(b[:])
|
||||
}
|
||||
|
||||
// turnSummary is what the history list shows for this turn. The user's own words
|
||||
// are the most useful label available, trimmed to fit a list row.
|
||||
//
|
||||
// Trimmed by RUNES, not bytes: slicing a byte offset would cut a multibyte
|
||||
// character in half and store invalid UTF-8 in the summary — which is not a
|
||||
// hypothetical for text people type.
|
||||
func turnSummary(message string) string {
|
||||
const max = 120
|
||||
s := strings.Join(strings.Fields(message), " ")
|
||||
runes := []rune(s)
|
||||
if len(runes) > max {
|
||||
s = strings.TrimSpace(string(runes[:max])) + "…"
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// systemPrompt gives the model the conventions it cannot infer.
|
||||
//
|
||||
// The compass convention in particular is not guessable: -y is north because
|
||||
// screen y grows downward, and a model that assumes otherwise plants the south
|
||||
// half when asked for the north one.
|
||||
func systemPrompt(g *domain.Garden) string {
|
||||
units := "metric — all measurements are centimeters"
|
||||
if g.UnitPref == domain.UnitImperial {
|
||||
units = "imperial for display, but every measurement you send or receive is in CENTIMETERS"
|
||||
}
|
||||
return fmt.Sprintf(`You are pansy's garden assistant. You help plan and edit a real garden by calling tools.
|
||||
|
||||
The garden you are working on is %q (id %d), %.0f x %.0f cm. The user's units are %s.
|
||||
|
||||
Conventions you cannot guess and must not assume:
|
||||
- Positions in a garden are centimeters from its top-left corner: x grows east, y grows SOUTH.
|
||||
- Inside an object (a bed), positions are relative to that object's CENTER, and -y is NORTH.
|
||||
So the north half of a bed is negative y. Getting this backwards plants the wrong end.
|
||||
- Objects and plantings are version-guarded. Use the version from describe_garden when editing.
|
||||
|
||||
How to work:
|
||||
- Start from describe_garden to see what is actually there. Do not guess ids.
|
||||
- Use find_plant to turn a plant name into an id. If it returns several candidates,
|
||||
pick the one that matches what the user said, or ask them which they meant.
|
||||
- To replant a bed with something else: clear_object, then fill_region with region "all".
|
||||
- When a tool refuses (for example, the user only has view access to this garden),
|
||||
explain what happened in plain words. Do not retry it.
|
||||
|
||||
When you are done, say briefly what you changed — the user is watching the canvas
|
||||
and wants to know what to look at. If you changed nothing, say that too.`,
|
||||
g.Name, g.ID, g.WidthCM, g.HeightCM, units)
|
||||
}
|
||||
@@ -0,0 +1,359 @@
|
||||
package agent
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
"gitea.stevedudenhoeffer.com/steve/majordomo/llm"
|
||||
"gitea.stevedudenhoeffer.com/steve/majordomo/provider/fake"
|
||||
|
||||
"gitea.stevedudenhoeffer.com/steve/pansy/internal/config"
|
||||
"gitea.stevedudenhoeffer.com/steve/pansy/internal/domain"
|
||||
"gitea.stevedudenhoeffer.com/steve/pansy/internal/service"
|
||||
)
|
||||
|
||||
// scriptedRunner wires a Runner to a fake provider so the whole loop — tools,
|
||||
// change-set scoping, loop guards — is exercised with no live model.
|
||||
func scriptedRunner(t *testing.T, svc *service.Service, steps ...fake.Step) *Runner {
|
||||
t.Helper()
|
||||
p := fake.New("fake")
|
||||
for _, s := range steps {
|
||||
p.Enqueue("m", s)
|
||||
}
|
||||
model, err := p.Model("m")
|
||||
if err != nil {
|
||||
t.Fatalf("fake model: %v", err)
|
||||
}
|
||||
return &Runner{svc: svc, model: model}
|
||||
}
|
||||
|
||||
// toolCall scripts one model turn that calls a tool.
|
||||
func toolCall(name string, args any) fake.Step {
|
||||
raw, _ := json.Marshal(args)
|
||||
return fake.ReplyWith(llm.Response{
|
||||
FinishReason: llm.FinishToolCalls,
|
||||
ToolCalls: []llm.ToolCall{{ID: "c1", Name: name, Arguments: raw}},
|
||||
})
|
||||
}
|
||||
|
||||
// TestTurnIsOneChangeSet is the acceptance criterion the whole "act freely"
|
||||
// posture rests on: a turn that clears a bed and replants it — one object edit
|
||||
// and many planting inserts — has to undo as ONE action, not thirteen.
|
||||
func TestTurnIsOneChangeSet(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
svc, owner := newAgentTestService(t)
|
||||
|
||||
g, err := svc.CreateGarden(ctx, owner, service.GardenInput{Name: "Plot", WidthCM: 2000, HeightCM: 2000})
|
||||
if err != nil {
|
||||
t.Fatalf("garden: %v", err)
|
||||
}
|
||||
garlic := mustPlant(t, svc, owner, "Garlic", 15, "🧄")
|
||||
cucumber := mustPlant(t, svc, owner, "Cucumber", 45, "🥒")
|
||||
bed, err := svc.CreateObject(ctx, owner, g.ID, service.ObjectInput{
|
||||
Kind: domain.KindBed, Name: "Garlic bed", XCM: 1000, YCM: 1000, WidthCM: 400, HeightCM: 400,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("bed: %v", err)
|
||||
}
|
||||
if _, err := svc.FillNamedRegion(ctx, owner, bed.ID, "all", garlic.ID, nil); err != nil {
|
||||
t.Fatalf("seed garlic: %v", err)
|
||||
}
|
||||
|
||||
before, _, err := svc.GardenHistory(ctx, owner, g.ID, 0, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("history: %v", err)
|
||||
}
|
||||
|
||||
r := scriptedRunner(t, svc,
|
||||
toolCall("clear_object", map[string]any{"objectId": bed.ID}),
|
||||
toolCall("fill_region", map[string]any{"objectId": bed.ID, "region": "all", "plantId": cucumber.ID}),
|
||||
fake.Reply("Cleared the garlic and replanted the bed with cucumbers."),
|
||||
)
|
||||
|
||||
turn, err := r.Run(ctx, owner, g.ID, "change the garlic bed to cucumbers this year", nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Run: %v", err)
|
||||
}
|
||||
if turn.ChangeSetID == nil {
|
||||
t.Fatal("a turn that changed things produced no change set")
|
||||
}
|
||||
if !strings.Contains(turn.Reply, "cucumbers") {
|
||||
t.Errorf("reply = %q", turn.Reply)
|
||||
}
|
||||
|
||||
// Exactly ONE new change set, whatever the model did inside the turn.
|
||||
after, _, err := svc.GardenHistory(ctx, owner, g.ID, 0, 0)
|
||||
if err != nil {
|
||||
t.Fatalf("history: %v", err)
|
||||
}
|
||||
if len(after)-len(before) != 1 {
|
||||
t.Fatalf("turn produced %d change sets, want exactly 1", len(after)-len(before))
|
||||
}
|
||||
cs := after[0]
|
||||
if cs.Source != domain.SourceAgent {
|
||||
t.Errorf("source = %q, want agent", cs.Source)
|
||||
}
|
||||
if cs.Summary != "change the garlic bed to cucumbers this year" {
|
||||
t.Errorf("summary = %q, want the user's own words", cs.Summary)
|
||||
}
|
||||
|
||||
// The bed really is cucumbers now.
|
||||
full, _ := svc.GardenFull(ctx, owner, g.ID, nil)
|
||||
if len(full.Plantings) == 0 {
|
||||
t.Fatal("bed ended up empty")
|
||||
}
|
||||
for _, p := range full.Plantings {
|
||||
if p.PlantID != cucumber.ID {
|
||||
t.Errorf("unexpected plant %d still in the bed", p.PlantID)
|
||||
}
|
||||
}
|
||||
|
||||
// And one undo puts it back.
|
||||
if _, conflicts, err := svc.RevertChangeSet(ctx, owner, *turn.ChangeSetID, domain.SourceUI); err != nil || len(conflicts) != 0 {
|
||||
t.Fatalf("undo: err=%v conflicts=%+v", err, conflicts)
|
||||
}
|
||||
full, _ = svc.GardenFull(ctx, owner, g.ID, nil)
|
||||
for _, p := range full.Plantings {
|
||||
if p.PlantID != garlic.ID {
|
||||
t.Errorf("after undo the bed holds plant %d, want the garlic back", p.PlantID)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestViewerGetsAnExplainableRefusal — the ACL story only works if the model can
|
||||
// narrate the refusal, so a tool denial has to reach it as a tool RESULT it can
|
||||
// read, not as a 500 that ends the run.
|
||||
func TestViewerGetsAnExplainableRefusal(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
svc, owner := newAgentTestService(t)
|
||||
viewer, err := svc.Register(ctx, service.RegisterInput{Email: "[email protected]", DisplayName: "V", Password: "password123"})
|
||||
if err != nil {
|
||||
t.Fatalf("register: %v", err)
|
||||
}
|
||||
g, err := svc.CreateGarden(ctx, owner, service.GardenInput{Name: "Plot", WidthCM: 2000, HeightCM: 2000})
|
||||
if err != nil {
|
||||
t.Fatalf("garden: %v", err)
|
||||
}
|
||||
bed, err := svc.CreateObject(ctx, owner, g.ID, service.ObjectInput{
|
||||
Kind: domain.KindBed, XCM: 1000, YCM: 1000, WidthCM: 400, HeightCM: 400,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("bed: %v", err)
|
||||
}
|
||||
if _, err := svc.AddShare(ctx, owner, g.ID, "[email protected]", domain.RoleViewer); err != nil {
|
||||
t.Fatalf("share: %v", err)
|
||||
}
|
||||
|
||||
// A viewer can't open a change set at all, so the turn is refused up front —
|
||||
// before any model call — and the API turns that into a plain explanation.
|
||||
r := scriptedRunner(t, svc, fake.Reply("unused"))
|
||||
_, err = r.Run(ctx, viewer.ID, g.ID, "plant garlic in that bed", nil, nil)
|
||||
if !errors.Is(err, domain.ErrForbidden) {
|
||||
t.Fatalf("viewer turn err = %v, want ErrForbidden", err)
|
||||
}
|
||||
|
||||
// And at the tool layer, a refusal comes back as a readable tool result
|
||||
// rather than killing the run.
|
||||
box := NewToolbox(svc, viewer.ID)
|
||||
raw, _ := json.Marshal(map[string]any{"objectId": bed.ID})
|
||||
res := box.Execute(ctx, llm.ToolCall{ID: "1", Name: "clear_object", Arguments: raw})
|
||||
if !res.IsError {
|
||||
t.Fatal("a viewer's clear_object succeeded")
|
||||
}
|
||||
if res.Content == "" {
|
||||
t.Error("the refusal carried no text for the model to explain")
|
||||
}
|
||||
}
|
||||
|
||||
// TestRunStopsAtTheStepCap — a model that gets wedged should stop on its own,
|
||||
// and what it managed to do must still be recorded and undoable.
|
||||
func TestRunStopsAtTheStepCap(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
svc, owner := newAgentTestService(t)
|
||||
g, err := svc.CreateGarden(ctx, owner, service.GardenInput{Name: "Plot", WidthCM: 2000, HeightCM: 2000})
|
||||
if err != nil {
|
||||
t.Fatalf("garden: %v", err)
|
||||
}
|
||||
|
||||
// More describe_garden calls than the cap allows, forever.
|
||||
steps := make([]fake.Step, 0, maxSteps+4)
|
||||
for i := 0; i < maxSteps+4; i++ {
|
||||
steps = append(steps, toolCall("describe_garden", map[string]any{"gardenId": g.ID}))
|
||||
}
|
||||
r := scriptedRunner(t, svc, steps...)
|
||||
|
||||
turn, err := r.Run(ctx, owner, g.ID, "look at the garden", nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("a capped run should end cleanly, got %v", err)
|
||||
}
|
||||
if !turn.Truncated {
|
||||
t.Error("turn.Truncated = false; the run hit the cap and should say so")
|
||||
}
|
||||
if turn.Reply == "" {
|
||||
t.Error("a capped run said nothing; silence reads as a hang")
|
||||
}
|
||||
// It only read, so there is nothing to undo — and no empty change set either.
|
||||
if turn.ChangeSetID != nil {
|
||||
t.Errorf("a read-only turn produced change set %d", *turn.ChangeSetID)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReadOnlyTurnWritesNoChangeSet — asking a question must not litter the
|
||||
// history with empty entries.
|
||||
func TestReadOnlyTurnWritesNoChangeSet(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
svc, owner := newAgentTestService(t)
|
||||
g, err := svc.CreateGarden(ctx, owner, service.GardenInput{Name: "Plot", WidthCM: 2000, HeightCM: 2000})
|
||||
if err != nil {
|
||||
t.Fatalf("garden: %v", err)
|
||||
}
|
||||
before, _, _ := svc.GardenHistory(ctx, owner, g.ID, 0, 0)
|
||||
|
||||
r := scriptedRunner(t, svc,
|
||||
toolCall("describe_garden", map[string]any{"gardenId": g.ID}),
|
||||
fake.Reply("It's empty — nothing planted yet."),
|
||||
)
|
||||
turn, err := r.Run(ctx, owner, g.ID, "what's in the garden?", nil, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Run: %v", err)
|
||||
}
|
||||
if turn.ChangeSetID != nil {
|
||||
t.Errorf("a question produced change set %d", *turn.ChangeSetID)
|
||||
}
|
||||
after, _, _ := svc.GardenHistory(ctx, owner, g.ID, 0, 0)
|
||||
if len(after) != len(before) {
|
||||
t.Errorf("history grew by %d for a read-only turn", len(after)-len(before))
|
||||
}
|
||||
}
|
||||
|
||||
// TestNewRunnerNeedsConfiguration — an instance with no key must not get a
|
||||
// half-built runner; the caller treats the error as "no assistant" and carries on.
|
||||
func TestNewRunnerNeedsConfiguration(t *testing.T) {
|
||||
svc, _ := newAgentTestService(t)
|
||||
for _, cfg := range []*config.Config{
|
||||
{Agent: config.AgentConfig{Enabled: false, OllamaCloudAPIKey: "k", Model: "ollama-cloud/x"}},
|
||||
{Agent: config.AgentConfig{Enabled: true, OllamaCloudAPIKey: "", Model: "ollama-cloud/x"}},
|
||||
{Agent: config.AgentConfig{Enabled: true, OllamaCloudAPIKey: "k", Model: ""}},
|
||||
} {
|
||||
if _, err := NewRunner(svc, cfg); err == nil {
|
||||
t.Errorf("NewRunner accepted %+v", cfg.Agent)
|
||||
}
|
||||
}
|
||||
// A model spec naming a provider that doesn't exist is a configuration
|
||||
// error, not a panic at first use.
|
||||
if _, err := NewRunner(svc, &config.Config{Agent: config.AgentConfig{
|
||||
Enabled: true, OllamaCloudAPIKey: "k", Model: "nonesuch/model",
|
||||
}}); err == nil {
|
||||
t.Error("NewRunner accepted an unresolvable model spec")
|
||||
}
|
||||
}
|
||||
|
||||
// TestTurnSummaryFitsAHistoryRow — the user's own words are the most useful
|
||||
// label for a change set, but a paragraph would wreck the list.
|
||||
func TestTurnSummaryFitsAHistoryRow(t *testing.T) {
|
||||
if got := turnSummary(" change the garlic bed\n to cucumbers "); got != "change the garlic bed to cucumbers" {
|
||||
t.Errorf("turnSummary = %q", got)
|
||||
}
|
||||
long := turnSummary(strings.Repeat("plant garlic ", 40))
|
||||
if len(long) > 130 || !strings.HasSuffix(long, "…") {
|
||||
t.Errorf("long summary = %q (%d chars)", long, len(long))
|
||||
}
|
||||
}
|
||||
|
||||
// TestSystemPromptStatesTheCompassConvention — -y being north is not guessable,
|
||||
// and a model that assumes otherwise plants the wrong end of the bed.
|
||||
func TestSystemPromptStatesTheCompassConvention(t *testing.T) {
|
||||
p := systemPrompt(&domain.Garden{ID: 1, Name: "Plot", WidthCM: 500, HeightCM: 400, UnitPref: domain.UnitImperial})
|
||||
for _, want := range []string{"NORTH", "-y", "centimeters", "Plot", "version"} {
|
||||
if !strings.Contains(p, want) {
|
||||
t.Errorf("system prompt is missing %q:\n%s", want, p)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(p, fmt.Sprintf("%.0f", 500.0)) {
|
||||
t.Error("system prompt doesn't state the garden's size")
|
||||
}
|
||||
}
|
||||
|
||||
// TestPartialWorkSurvivesATimeout is the finding that mattered most on this PR.
|
||||
//
|
||||
// The run context carries a timeout. When it fires, WithChangeSet's recovery
|
||||
// path has to record what already committed — and doing that with the SAME
|
||||
// dead context would fail, losing the history for changes that really happened.
|
||||
// The user-facing message says "anything I'd already changed is in History", so
|
||||
// this isn't just a gap, it's a promise the code has to keep.
|
||||
func TestPartialWorkSurvivesATimeout(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
svc, owner := newAgentTestService(t)
|
||||
g, err := svc.CreateGarden(ctx, owner, service.GardenInput{Name: "Plot", WidthCM: 2000, HeightCM: 2000})
|
||||
if err != nil {
|
||||
t.Fatalf("garden: %v", err)
|
||||
}
|
||||
bed, err := svc.CreateObject(ctx, owner, g.ID, service.ObjectInput{
|
||||
Kind: domain.KindBed, Name: "Bed", XCM: 1000, YCM: 1000, WidthCM: 400, HeightCM: 400,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("bed: %v", err)
|
||||
}
|
||||
before, _, _ := svc.GardenHistory(ctx, owner, g.ID, 0, 0)
|
||||
|
||||
// A turn that renames the bed, then dies with the context already cancelled.
|
||||
cancelled, cancel := context.WithCancel(ctx)
|
||||
r := scriptedRunner(t, svc,
|
||||
toolCall("move_object", map[string]any{
|
||||
"objectId": bed.ID, "xCm": 600.0, "yCm": 600.0, "version": bed.Version,
|
||||
}),
|
||||
fake.Step{Err: context.DeadlineExceeded},
|
||||
)
|
||||
// Cancel once the first tool call has landed, so the failure path runs with a
|
||||
// dead context — exactly the timeout case.
|
||||
go func() {
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
cancel()
|
||||
}()
|
||||
_, err = r.Run(cancelled, owner, g.ID, "move the bed", nil, nil)
|
||||
if err == nil {
|
||||
t.Fatal("expected the turn to fail")
|
||||
}
|
||||
|
||||
// The move committed, so it must be in history and undoable.
|
||||
after, _, herr := svc.GardenHistory(ctx, owner, g.ID, 0, 0)
|
||||
if herr != nil {
|
||||
t.Fatalf("history: %v", herr)
|
||||
}
|
||||
if len(after) != len(before)+1 {
|
||||
t.Fatalf("the failed turn recorded %d change sets, want 1 — its work is otherwise un-undoable",
|
||||
len(after)-len(before))
|
||||
}
|
||||
if !strings.Contains(after[0].Summary, "failed partway") {
|
||||
t.Errorf("summary = %q, want it marked as partial", after[0].Summary)
|
||||
}
|
||||
if _, conflicts, rerr := svc.RevertChangeSet(ctx, owner, after[0].ID, domain.SourceUI); rerr != nil || len(conflicts) != 0 {
|
||||
t.Fatalf("the partial turn should be undoable: err=%v conflicts=%+v", rerr, conflicts)
|
||||
}
|
||||
o, _ := svc.DescribeGarden(ctx, owner, g.ID)
|
||||
if len(o.Objects) > 0 && o.Objects[0].XCM != bed.XCM {
|
||||
t.Errorf("undo left the bed at %v, want %v", o.Objects[0].XCM, bed.XCM)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTurnSummaryTrimsByRunes — slicing a byte offset would cut a multibyte
|
||||
// character in half and store invalid UTF-8 in the history summary.
|
||||
func TestTurnSummaryTrimsByRunes(t *testing.T) {
|
||||
// 200 multibyte runes: a byte slice at 120 would land mid-character.
|
||||
got := turnSummary(strings.Repeat("🌱", 200))
|
||||
if !utf8.ValidString(got) {
|
||||
t.Errorf("turnSummary produced invalid UTF-8: %q", got)
|
||||
}
|
||||
if !strings.HasSuffix(got, "…") {
|
||||
t.Errorf("long summary should be elided, got %q", got)
|
||||
}
|
||||
if n := utf8.RuneCountInString(got); n > 121 {
|
||||
t.Errorf("summary is %d runes, want it trimmed", n)
|
||||
}
|
||||
}
|
||||
@@ -1,5 +1,3 @@
|
||||
//go:build majordomo
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
//go:build majordomo
|
||||
|
||||
package agent
|
||||
|
||||
import (
|
||||
|
||||
Reference in New Issue
Block a user