1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
|
// The test that matters here is the one about CAPTURE: Genkit only hands back
// the last message, and the whole point of `Generate` is to hand back the
// intermediate turns too. That is the one thing the `WithMiddleware` → `WithUse`
// migration could have broken silently — the code compiles just as well with a
// hook that never sees anything.
//
// No real model is needed: a fake model registered in Genkit plays one tool turn
// then answers, which is exactly the shape the history must keep a trace of.
//
// The watchdog of this version runs during these tests: since the fake model
// answers instantly it never fires — and if it did, the tests would fail, which
// is the intended behaviour.
package engine
import (
"context"
"testing"
"mm/internal/config"
"github.com/firebase/genkit/go/ai"
"github.com/firebase/genkit/go/genkit"
)
// fakeGenkit registers a "dmr/<Cfg.Model>" model whose every call is served by
// the next answer in `turns`, and a `bash` tool returning a fixed output.
// `calls` counts the calls to the model.
func fakeGenkit(t *testing.T, turns ...*ai.ModelResponse) (*genkit.Genkit, *int) {
t.Helper()
g := genkit.Init(context.Background())
calls := 0
genkit.DefineModel(g, "dmr/"+config.Cfg.Model,
&ai.ModelOptions{Supports: &ai.ModelSupports{Tools: true, Multiturn: true, SystemRole: true}},
func(_ context.Context, _ *ai.ModelRequest, _ ai.ModelStreamCallback) (*ai.ModelResponse, error) {
if calls >= len(turns) {
t.Errorf("the model was called %d times for %d planned turns", calls+1, len(turns))
return turns[len(turns)-1], nil
}
resp := turns[calls]
calls++
return resp, nil
})
genkit.DefineTool(g, "bash", "run a shell command",
func(_ *ai.ToolContext, in struct {
Command string `json:"command"`
}) (string, error) {
return "hello from " + in.Command, nil
})
genkit.DefineTool(g, "read_skill", "load a markdown procedure",
func(_ *ai.ToolContext, in struct {
Name string `json:"name"`
}) (string, error) {
return "# procedure " + in.Name, nil
})
return g, &calls
}
// testEngine wraps a fake Genkit the way New would: the model reference is the
// only thing Generate reads from the Engine, so no provider is needed here.
func testEngine(g *genkit.Genkit) *Engine {
return &Engine{G: g, Model: "dmr/" + config.Cfg.Model}
}
func modelTurn(parts ...*ai.Part) *ai.ModelResponse {
return &ai.ModelResponse{
Message: ai.NewModelMessage(parts...),
FinishReason: ai.FinishReasonStop,
}
}
// TestGenerateReturnsFullConversation is the regression test for the capture: on
// a tool turn, the history handed back must contain the question, the tool
// request, its response AND the final text — not just that last one.
func TestGenerateReturnsFullConversation(t *testing.T) {
toolReq := modelTurn(ai.NewToolRequestPart(&ai.ToolRequest{
Name: "bash",
Input: map[string]any{"command": "echo hi"},
}))
final := modelTurn(ai.NewTextPart("done"))
g, calls := fakeGenkit(t, toolReq, final)
resp, full, err := testEngine(g).Generate(context.Background(),
[]*ai.Message{ai.NewUserTextMessage("run echo hi")},
[]ai.ToolRef{ai.ToolName("bash")})
if err != nil {
t.Fatalf("Generate: %v", err)
}
if *calls != 2 {
t.Errorf("calls to the model = %d, want 2 (tool, then answer)", *calls)
}
if got := resp.Text(); got != "done" {
t.Errorf("resp.Text() = %q, want %q", got, "done")
}
// The full history: without the capture, Genkit only hands back the last
// message and `full` would have length 1.
if len(full) < 4 {
t.Fatalf("history of %d messages, want at least 4 (question, call, tool response, text): %+v", len(full), full)
}
if full[len(full)-1] != final.Message {
t.Errorf("the last message of the history is not the final answer")
}
// The two markers that exist ONLY in the intermediate turns.
var sawRequest, sawResponse bool
for _, m := range full {
for _, p := range m.Content {
if p.IsToolRequest() {
sawRequest = true
}
if p.IsToolResponse() {
sawResponse = true
}
}
}
if !sawRequest {
t.Error("the history contains no tool request")
}
if !sawResponse {
t.Error("the history contains no tool response")
}
// Commands reads the same trace: one command run, exactly one.
if n := Commands(full); n != 1 {
t.Errorf("Commands(full) = %d, want 1", n)
}
// And CommandList hands back its TEXT, which exists only in the call — that
// is what the green recap displays.
if got := CommandList(full); len(got) != 1 || got[0] != "echo hi" {
t.Errorf("CommandList(full) = %q, want [\"echo hi\"]", got)
}
}
// TestCommandsCountsBashOnly guards the bug introduced by adding `read_skill`:
// `Commands` counted EVERY tool response, so loading two skills and running a
// single command displayed "⚙ 3 command(s)". The number then said the opposite
// of what it is for — that the model acted.
func TestCommandsCountsBashOnly(t *testing.T) {
skillReq := modelTurn(ai.NewToolRequestPart(&ai.ToolRequest{
Name: "read_skill",
Input: map[string]any{"name": "go-rename"},
}))
bashReq := modelTurn(ai.NewToolRequestPart(&ai.ToolRequest{
Name: "bash",
Input: map[string]any{"command": "echo hi"},
}))
final := modelTurn(ai.NewTextPart("done"))
g, calls := fakeGenkit(t, skillReq, bashReq, final)
_, full, err := testEngine(g).Generate(context.Background(),
[]*ai.Message{ai.NewUserTextMessage("rename Greet")},
[]ai.ToolRef{ai.ToolName("bash"), ai.ToolName("read_skill")})
if err != nil {
t.Fatalf("Generate: %v", err)
}
if *calls != 3 {
t.Errorf("calls to the model = %d, want 3", *calls)
}
if n := Commands(full); n != 1 {
t.Errorf("Commands = %d, want 1 — only the bash command counts", n)
}
if k := Skills(full); k != 1 {
t.Errorf("Skills = %d, want 1", k)
}
// The list follows the same rule as the count: the skill read is not in it.
if got := CommandList(full); len(got) != 1 || got[0] != "echo hi" {
t.Errorf("CommandList = %q, want [\"echo hi\"] — a skill is not a command", got)
}
}
// TestCommandList covers the shapes the call↔response pairing has to take:
// several commands in order, a Ref that does not follow the order they were
// issued in, a call left without a response (nothing ran: nothing to show)
// and a turn mixing `bash` and `read_skill` with no Ref.
func TestCommandList(t *testing.T) {
req := func(ref, name string, in any) *ai.Part {
return ai.NewToolRequestPart(&ai.ToolRequest{Ref: ref, Name: name, Input: in})
}
resp := func(ref, name string) *ai.Part {
return ai.NewToolResponsePart(&ai.ToolResponse{Ref: ref, Name: name, Output: "ok"})
}
bash := func(cmd string) map[string]any { return map[string]any{"command": cmd} }
cases := []struct {
name string
history []*ai.Message
want []string
}{
{
name: "two commands, in the order they ran",
history: []*ai.Message{
ai.NewModelMessage(req("a", "bash", bash("ls")), req("b", "bash", bash("pwd"))),
ai.NewMessage(ai.RoleTool, nil, resp("a", "bash"), resp("b", "bash")),
},
want: []string{"ls", "pwd"},
},
{
name: "responses out of order: the Ref decides",
history: []*ai.Message{
ai.NewModelMessage(req("a", "bash", bash("ls")), req("b", "bash", bash("pwd"))),
ai.NewMessage(ai.RoleTool, nil, resp("b", "bash"), resp("a", "bash")),
},
want: []string{"pwd", "ls"},
},
{
name: "no Ref, mixed tools: paired against the same name",
history: []*ai.Message{
ai.NewModelMessage(
req("", "read_skill", map[string]any{"name": "go-rename"}),
req("", "bash", bash("ls")),
),
ai.NewMessage(ai.RoleTool, nil, resp("", "read_skill"), resp("", "bash")),
},
want: []string{"ls"},
},
{
name: "call with no response: the command never ran",
history: []*ai.Message{
ai.NewModelMessage(req("a", "bash", bash("ls"))),
},
want: nil,
},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
got := CommandList(tc.history)
if len(got) != len(tc.want) {
t.Fatalf("CommandList = %q, want %q", got, tc.want)
}
for i := range got {
if got[i] != tc.want[i] {
t.Errorf("CommandList[%d] = %q, want %q", i, got[i], tc.want[i])
}
}
})
}
}
// TestGenerateNoToolCall checks the bare case: with no tool called, the history
// stays the question plus the answer, and Commands counts zero — a model that
// tells stories without acting must not be credited with a command.
func TestGenerateNoToolCall(t *testing.T) {
final := modelTurn(ai.NewTextPart("hello"))
g, calls := fakeGenkit(t, final)
_, full, err := testEngine(g).Generate(context.Background(),
[]*ai.Message{ai.NewUserTextMessage("say hello")},
nil)
if err != nil {
t.Fatalf("Generate: %v", err)
}
if *calls != 1 {
t.Errorf("calls to the model = %d, want 1", *calls)
}
if len(full) != 2 {
t.Fatalf("history of %d messages, want 2 (question, answer): %+v", len(full), full)
}
if n := Commands(full); n != 0 {
t.Errorf("Commands(full) = %d, want 0", n)
}
}
// TestGenerateCapturesInputTokens: the compression trigger prefers the server's
// own count of the context to its estimate, and that count only exists inside
// the WrapModel hook. A fake response carrying Usage must surface through
// LastInputTokens; ForgetInputTokens must clear it.
func TestGenerateCapturesInputTokens(t *testing.T) {
withUsage := modelTurn(ai.NewTextPart("ok"))
withUsage.Usage = &ai.GenerationUsage{InputTokens: 1234, OutputTokens: 5}
g, _ := fakeGenkit(t, withUsage)
e := testEngine(g)
if _, _, err := e.Generate(context.Background(),
[]*ai.Message{ai.NewUserTextMessage("hi")}, nil); err != nil {
t.Fatalf("Generate: %v", err)
}
if got := e.LastInputTokens(); got != 1234 {
t.Errorf("LastInputTokens() = %d, want 1234", got)
}
e.ForgetInputTokens()
if got := e.LastInputTokens(); got != 0 {
t.Errorf("after ForgetInputTokens, LastInputTokens() = %d, want 0", got)
}
}
// TestSummarize: the summary request declares no tools (the model must write,
// not act), goes to the Engine's own model reference, and the model's text
// comes back trimmed. An empty answer is an error — replacing the history with
// nothing is worse than keeping it.
func TestSummarize(t *testing.T) {
g := genkit.Init(context.Background())
var sawTools int
answers := []string{" ## Goal\nnotes\n", ""}
call := 0
genkit.DefineModel(g, "dmr/"+config.Cfg.Model,
&ai.ModelOptions{Supports: &ai.ModelSupports{Tools: true, Multiturn: true, SystemRole: true}},
func(_ context.Context, r *ai.ModelRequest, _ ai.ModelStreamCallback) (*ai.ModelResponse, error) {
sawTools = len(r.Tools)
a := answers[call]
call++
return modelTurn(ai.NewTextPart(a)), nil
})
e := testEngine(g)
req := []*ai.Message{ai.NewSystemTextMessage("note-taker"), ai.NewUserTextMessage("summarise")}
text, err := e.Summarize(context.Background(), req, 100)
if err != nil {
t.Fatalf("Summarize: %v", err)
}
if text != "## Goal\nnotes" {
t.Errorf("text = %q, want the trimmed answer", text)
}
if sawTools != 0 {
t.Errorf("the summary request declared %d tool(s), want none", sawTools)
}
if _, err := e.Summarize(context.Background(), req, 100); err == nil {
t.Error("an empty answer must be an error")
}
}
// TestFileOpsKeepsTheOrder: the recap lists file operations in the order they
// ran, across the three tools — a read then an edit must not come out as "all
// reads, then all edits", which a per-tool pass would produce. Commands stay
// bash-only: the file tools have their own column.
func TestFileOpsKeepsTheOrder(t *testing.T) {
req := func(ref, name string, in map[string]any) *ai.Part {
return ai.NewToolRequestPart(&ai.ToolRequest{Ref: ref, Name: name, Input: in})
}
resp := func(ref, name string) *ai.Part {
return ai.NewToolResponsePart(&ai.ToolResponse{Ref: ref, Name: name, Output: "ok"})
}
history := []*ai.Message{
ai.NewModelMessage(req("a", "read_file", map[string]any{"path": "f.go", "start": 1.0, "end": 9.0})),
ai.NewMessage(ai.RoleTool, nil, resp("a", "read_file")),
ai.NewModelMessage(req("b", "bash", map[string]any{"command": "go vet ./..."})),
ai.NewMessage(ai.RoleTool, nil, resp("b", "bash")),
ai.NewModelMessage(req("c", "edit_file", map[string]any{"path": "f.go", "edits": []any{map[string]any{"old": "a", "new": "b"}, map[string]any{"old": "c", "new": "d"}}})),
ai.NewMessage(ai.RoleTool, nil, resp("c", "edit_file")),
ai.NewModelMessage(req("d", "write_file", map[string]any{"path": "notes.md", "content": "x"})),
ai.NewMessage(ai.RoleTool, nil, resp("d", "write_file")),
}
want := []string{"read_file f.go 1-9", "edit_file f.go (2 edit(s))", "write_file notes.md"}
got := FileOps(history)
if len(got) != len(want) {
t.Fatalf("FileOps = %q, want %q", got, want)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("FileOps[%d] = %q, want %q", i, got[i], want[i])
}
}
if n := Commands(history); n != 1 {
t.Errorf("Commands = %d, want 1 — file ops are not commands", n)
}
}
|