第 4 关
何时停止
- user
- assistant
- tool_result
EventBus
问题
第 2 关的循环只有一个正常出口:模型不申请工具就直接回复。但一个找不到目标的模型很少会这么说。它会换一个搜索,再换一种拼写,再换一个。每一圈都要重新发送整段历史记录,所以每一圈都比上一圈更贵。
这就是 Loopboros,永不结束的循环。maxTurns 能拦住它。但再看一眼 World 1-1:时钟走完时,循环并没有失败。它停下来,把最后一条消息交还给你,而那条消息是一个搜索请求。没有答案,也没有错误。
三个出口
-
旗杆
end_turnWorld 1-2
由模型决定。 它不申请任何工具就直接回复。这是正常的出口,大多数运行都应该在这里结束。
返回:一条以文本形式给出答案的消息。
-
传送水管
stopOnToolNamesWorld 1-3
由你的代码决定:当模型调用了你标记过的工具时。 那个工具一旦无错运行完,循环就立刻结束。如果它出错,错误会回到模型那里,循环继续。
返回:一条带有该工具调用的消息。它的输入就是答案,并且已经过 schema 校验。
-
时钟
maxTurnsWorld 1-1
谁都没有决定。预算用完了。 不管模型正在做什么,循环跑满这么多轮就停下。它是一张安全网,不是一个答案。
返回:最后一条消息是什么就返回什么。往往是一个工具调用。
和 end_turn 相比,终止型工具并不能帮你省下一轮。它给你的是数据形式的答案,经过 schema 校验,你的应用拿来就能用:在这里,就是关卡地图上的一个标记。如果没有 stopOnToolNames,mark_star 运行之后,循环还会再问模型一次,只为了让它说一句“完成”。
代码
使用 astorlm:maxTurns 设置时钟,stopOnToolNames 标记传送水管。run() 总是返回最后一条消息,所以要读一读它,才能知道这次运行走的是哪个出口。
从零手写:第 2 关的循环,三个出口都标了出来。它返回的不是一个光秃秃的字符串,而是这次运行如何结束,这样调用方就不会把超时误当成答案。
import { OpenAIProvider, createLocalAgent, tool } from 'astorlm'
import { z } from 'zod'
const searchBlocks = tool({
name: 'search_blocks',
description: 'Search the ? blocks in one area of the level. Returns what each one hides.',
schema: z.object({ area: z.string() }),
execute: async ({ area }) => level.search(area), // your code
})
// The answer as data. Its schema is checked before the loop is allowed to stop.
const markStar = tool({
name: 'mark_star',
description: 'Put a marker on the level map where the star is. Call it once, at the end.',
schema: z.object({ item: z.literal('star'), block: z.number().int().positive() }),
execute: async ({ block }) => map.addMarker(block), // your code
})
const agent = await createLocalAgent({
// Any OpenAI-compatible endpoint: OpenAI, Ollama, LM Studio, vLLM, a proxy…
provider: new OpenAIProvider({
baseURL: 'http://localhost:11434/v1', // e.g. Ollama's default address
model: 'your-model', // e.g. 'llama3.1', 'gpt-4o-mini'
apiKey: 'YOUR_API_KEY', // local servers usually ignore it
}),
tools: [searchBlocks, markStar],
appendSystemPrompt:
'When you know where the star is, deliver it with mark_star. ' +
"If a few searches come back empty, say you couldn't find it.",
maxTurns: 10, // the clock. Without it, the default is 25.
stopOnToolNames: ['mark_star'], // the warp pipe
})
const last = await agent.run('Is there a hidden star in this level?')
// The loop hands back its last message whichever way it ended. Read it to find out which.
const call = last.content.find((block) => block.type === 'tool_use')
if (!call) {
console.log('end_turn:', last.content) // the flag: a plain text answer
} else if (call.name === 'mark_star') {
console.log('answer:', call.input) // the warp pipe: { item, block }, schema-checked
} else {
// The clock: maxTurns ran out while the model was still asking for tools.
throw new Error(`No answer: the run stopped while asking for ${call.name}`)
}
// The three exits of an agent loop, from scratch. Plain fetch, no SDK.
// Any OpenAI-compatible endpoint: OpenAI, Ollama, LM Studio, vLLM, a proxy…
const LLM = {
baseURL: 'http://localhost:11434/v1', // e.g. Ollama's default address
model: 'your-model', // e.g. 'llama3.1', 'gpt-4o-mini'
apiKey: 'YOUR_API_KEY', // local servers usually ignore it
}
type ToolCall = { id: string; function: { name: string; arguments: string } }
type Message =
| { role: 'user'; content: string }
| { role: 'assistant'; content: string | null; tool_calls?: ToolCall[] }
| { role: 'tool'; tool_call_id: string; content: string }
type ToolFn = (args: Record<string, unknown>) => Promise<string>
const tools: Record<string, ToolFn> = { search_blocks: searchBlocks, mark_star: markStar }
const toolSchemas = [/* one JSON Schema per tool */]
// Terminal tools: once one runs without error, the loop is over.
const stopOn = new Set(['mark_star'])
// Say how the run ended, so the caller can tell an answer from a timeout.
type Outcome =
| { stop: 'end_turn'; text: string }
| { stop: 'terminal_tool'; input: Record<string, unknown> }
| { stop: 'max_turns'; turns: number }
export async function runAgent(prompt: string, maxTurns = 10): Promise<Outcome> {
const messages: Message[] = [{ role: 'user', content: prompt }]
for (let turn = 1; turn <= maxTurns; turn++) {
const res = await fetch(`${LLM.baseURL}/chat/completions`, {
method: 'POST',
headers: { 'content-type': 'application/json', authorization: `Bearer ${LLM.apiKey}` },
body: JSON.stringify({ model: LLM.model, messages, tools: toolSchemas }),
})
const [choice] = (await res.json()).choices
const reply: Message = choice.message
messages.push(reply)
// Exit 1, the flag: no tool calls, the model is done.
if (choice.finish_reason !== 'tool_calls') return { stop: 'end_turn', text: reply.content ?? '' }
for (const call of reply.tool_calls ?? []) {
const run = tools[call.function.name]
let input: Record<string, unknown> = {}
let output = `Unknown tool: ${call.function.name}`
let failed = true
if (run) {
try {
input = JSON.parse(call.function.arguments)
output = await run(input)
failed = false
} catch (err) {
output = `Error: ${err instanceof Error ? err.message : err}`
}
}
messages.push({ role: 'tool', tool_call_id: call.id, content: output })
// Exit 2, the warp pipe: a terminal tool ran fine. Its input is the answer.
// If it failed, the error goes back to the model and the loop keeps going.
if (stopOn.has(call.function.name) && !failed) return { stop: 'terminal_tool', input }
}
}
// Exit 3, the clock: out of turns. Say so out loud; don't pass a tool call off as an answer.
return { stop: 'max_turns', turns: maxTurns }
}
# The three exits of an agent loop, from scratch. Standard library only, no SDK.
import json
import urllib.request
# Any OpenAI-compatible endpoint: OpenAI, Ollama, LM Studio, vLLM, a proxy...
LLM = {
"base_url": "http://localhost:11434/v1", # e.g. Ollama's default address
"model": "your-model", # e.g. "llama3.1", "gpt-4o-mini"
"api_key": "YOUR_API_KEY", # local servers usually ignore it
}
TOOLS = {"search_blocks": search_blocks, "mark_star": mark_star}
TOOL_SCHEMAS = [...] # one JSON Schema per tool
# Terminal tools: once one runs without error, the loop is over.
STOP_ON = {"mark_star"}
def chat(messages):
request = urllib.request.Request(
f"{LLM['base_url']}/chat/completions",
data=json.dumps({"model": LLM["model"], "messages": messages, "tools": TOOL_SCHEMAS}).encode(),
headers={"Content-Type": "application/json", "Authorization": f"Bearer {LLM['api_key']}"},
)
with urllib.request.urlopen(request) as response:
return json.load(response)["choices"][0]
def run_agent(prompt, max_turns=10):
"""Returns how the run ended, so the caller can tell an answer from a timeout."""
messages = [{"role": "user", "content": prompt}]
for _ in range(max_turns):
choice = chat(messages)
reply = choice["message"]
messages.append(reply)
# Exit 1, the flag: no tool calls, the model is done.
if choice["finish_reason"] != "tool_calls":
return {"stop": "end_turn", "text": reply.get("content") or ""}
for call in reply.get("tool_calls", []):
name = call["function"]["name"]
args = json.loads(call["function"]["arguments"])
run = TOOLS.get(name)
failed = run is None
try:
output = run(**args) if run else f"Unknown tool: {name}"
except Exception as err:
failed = True
output = f"Error: {err}"
messages.append({"role": "tool", "tool_call_id": call["id"], "content": output})
# Exit 2, the warp pipe: a terminal tool ran fine. Its input is the answer.
# If it failed, the error goes back to the model and the loop keeps going.
if name in STOP_ON and not failed:
return {"stop": "terminal_tool", "input": args}
# Exit 3, the clock: out of turns. Say so out loud; don't pass a tool call off as an answer.
return {"stop": "max_turns", "turns": max_turns}
注意事项
-
一定要设置
maxTurns,并按任务来定大小。一次查询只需要几轮;一次重构可能需要几十轮。astorlm 的默认值是 25。 - 把时钟走完当成失败来处理。如果运行以一个工具调用结束,就告诉用户你没能完成,或者换一个更清晰的提示词重试。不要给他们看一个空答案。
- 给模型一个放弃的办法。在 system prompt 里告诉它:搜索一再落空时,就说“我没找到”。被允许停下的模型,会停得更早。
-
轮数不等于时间。
maxTurns对一轮永远不结束的情况无能为力。那需要超时或者AbortSignal。