昨天我們做出了一個能自主使用工具的 Agent。它看起來很厲害——我們問一句,它自己決定要查什麼、怎麼回答。
但「自主」不是免費的。它同時帶來三個代價:
| 代價 | 具體表現 |
|---|---|
| 不可預測 | 同樣的問題,這次查天氣,下次可能直接回答 |
| 成本不固定 | 這次 2 輪就完成,下次可能跑了 8 輪 |
| 難以除錯 | 出錯時不知道是 prompt 的問題還是模型的問題 |
有些任務不需要這種彈性。如果每天早上的簡報流程都是「查時間 → 查天氣 → 讀待辦 → 產生摘要」,那讓模型每次重新「決定」一遍,是浪費。
今天要介紹的 Workflow,就是這類任務的答案。
用一句話講清楚:
Workflow:路線是你畫的,AI 在路上幫忙。
Agent:目的地是你給的,路線由 AI 自己找。
用程式碼看更清楚:
# Workflow:流程寫死在程式裡
def daily_brief():
now = get_current_datetime() # 我決定第一步
weather = get_weather(city) # 我決定第二步
todos = list_todos() # 我決定第三步
summary = llm.summarize(now, weather, todos) # AI 只負責這一步
return summary
# Agent:流程由模型決定
def daily_brief():
return agent.run("幫我產生今天的簡報")
# 模型自己決定要呼叫哪些工具、順序、幾次
兩種都會用到 AI,差別在於控制權在誰手上。
實務上,大部分系統都在兩者之間。可以這樣分級:
| 層級 | 名稱 | AI 的角色 | 例子 |
|---|---|---|---|
| 0 | 純程式 | 沒有 AI | 查天氣、存檔 |
| 1 | 單次呼叫 | 處理一段文字 | 翻譯、摘要、分類 |
| 2 | 串接(Chain) | 每一步處理一部分 | 抽取 → 驗證 → 產生 |
| 3 | 分支(Routing) | 決定走哪條路 | 意圖分類 → 對應處理 |
| 4 | 平行(Parallel) | 同時處理多件事 | 多角度分析後彙整 |
| 5 | 迴圈(Loop) | 反覆修正直到滿意 | 產生 → 評估 → 改進 |
| 6 | Agent | 自己決定全部 | 開放式任務 |
第 2~5 層都算 Workflow。今天講這四種模式。
重要的原則:能用低層級解決的,就不要用高層級。
理由很簡單——越往下,成本越高、越不可預測、越難除錯。如果一個任務用固定流程就能做好,用 Agent 只是把錢燒掉還換來不穩定。
把一個大任務拆成幾個小步驟,一步的輸出是下一步的輸入。
使用者輸入 → [抽取] → [驗證] → [執行] → [產生回覆]
實作:把自然語言變成待辦事項
"""workflows/add_todo_flow.py"""
import logging
from datetime import datetime, timedelta
from typing import Literal, Optional
from pydantic import BaseModel, Field
logger = logging.getLogger(__name__)
class ExtractedTodo(BaseModel):
"""步驟 1 的輸出。"""
is_todo: bool = Field(description="這句話是否真的在描述一件待辦事項")
title: Optional[str] = Field(default=None, description="待辦的簡短標題")
priority: Literal["高", "中", "低"] = Field(default="中")
due_offset_days: Optional[int] = Field(
default=None,
description="距今幾天後到期。今天=0,明天=1。沒提到則為 null。",
)
def add_todo_workflow(client, user_input, today=None):
"""把一句話變成待辦事項的完整流程。"""
today = today or datetime.now()
# === 步驟 1:抽取 ===
logger.info("步驟 1:抽取待辦資訊")
response = client.messages.parse(
model="claude-opus-5",
max_tokens=1024,
system=(
f"今天是 {today.strftime('%Y-%m-%d')}。"
"從使用者的話中抽取待辦事項。"
"如果使用者只是在閒聊、詢問資訊,或沒有明確的待辦意圖,"
"把 is_todo 設為 false。"
),
messages=[{"role": "user", "content": user_input}],
output_format=ExtractedTodo,
)
extracted = response.parsed_output
# === 步驟 2:驗證(純程式,不用 AI)===
logger.info("步驟 2:驗證")
if not extracted.is_todo:
return {"ok": False, "message": "這句話看起來不是要新增待辦事項。"}
if not extracted.title or not extracted.title.strip():
return {"ok": False, "message": "沒辦法從這句話判斷要記什麼事。"}
due = None
if extracted.due_offset_days is not None:
if not 0 <= extracted.due_offset_days <= 365:
logger.warning("不合理的期限:%s 天", extracted.due_offset_days)
else:
due = (today + timedelta(days=extracted.due_offset_days)).date().isoformat()
# === 步驟 3:執行(純程式)===
logger.info("步驟 3:寫入")
todo = save_todo(extracted.title.strip(), extracted.priority, due)
# === 步驟 4:產生回覆 ===
due_text = f",期限 {due}" if due else ""
return {
"ok": True,
"todo": todo,
"message": f"已記下「{todo['title']}」({todo['priority']}優先{due_text})。",
}
注意這個流程的特色:
比起讓 Agent 自己處理,這個流程:成本固定、速度快、結果可預測。
先分類,再走不同的路。這就是 Day 20 的意圖分類器的用途。
"""workflows/router.py"""
import logging
logger = logging.getLogger(__name__)
class Router:
"""依據意圖把請求分派給不同的處理流程。"""
def __init__(self, client, agent):
self.client = client
self.agent = agent
self.handlers = {
"query_weather": self.handle_weather,
"add_todo": self.handle_add_todo,
"list_todos": self.handle_list_todos,
"daily_brief": self.handle_daily_brief,
"chat": self.handle_chat,
"unknown": self.handle_unknown,
}
def route(self, user_input, history=None):
intent = classify_intent(self.client, user_input) # Day 20 寫的
logger.info("路由:%s(信心 %.2f)", intent.action, intent.confidence)
# 信心不足時,不要硬猜
if intent.confidence < 0.6:
logger.info("信心不足,改用 Agent 處理")
return self.agent.run(user_input, history=history)[0]
handler = self.handlers.get(intent.action, self.handle_unknown)
return handler(user_input, intent)
# --- 各種處理流程 ---
def handle_weather(self, user_input, intent):
"""固定流程:查天氣 → 分析 → 回答。"""
city = intent.city or config.DEFAULT_CITY
data = get_weather(city, config.CWA_API_KEY)
insight = analyze_weather(data) # Day 16 的分析
return make_advice(insight) # 純程式,不用 AI!
def handle_add_todo(self, user_input, intent):
result = add_todo_workflow(self.client, user_input)
return result["message"]
def handle_list_todos(self, user_input, intent):
todos = load_todos()
return make_todo_summary(analyze_todos(todos)) # 純程式
def handle_daily_brief(self, user_input, intent):
return daily_brief_workflow(self.client)
def handle_chat(self, user_input, intent):
# 純聊天不需要工具,直接呼叫模型比較快也比較便宜
response = self.client.messages.create(
model="claude-opus-5",
max_tokens=1024,
system=CHAT_SYSTEM,
messages=[{"role": "user", "content": user_input}],
)
return extract_text(response)
def handle_unknown(self, user_input, intent):
# 看不懂的就交給 Agent 自由發揮
return self.agent.run(user_input)[0]
這個設計的精髓在最後兩個 handler:
把 Agent 當成「預設分支」。
這是我覺得最實用的混合架構。80% 的請求走固定流程,20% 的長尾交給 Agent。
注意 handle_weather 和 handle_list_todos 完全沒有呼叫 AI——資料查詢加上 Day 16 的分析函式,就能產生好的回答。能省的就省。
多件事同時做,最後彙整。
"""workflows/parallel.py"""
import logging
from concurrent.futures import ThreadPoolExecutor, as_completed
logger = logging.getLogger(__name__)
def gather_context(tasks, timeout=15):
"""平行執行多個取資料的任務。
Args:
tasks: {名稱: 無參數的函式}
Returns:
{名稱: 結果或錯誤訊息}
"""
results = {}
with ThreadPoolExecutor(max_workers=len(tasks)) as executor:
futures = {executor.submit(fn): name for name, fn in tasks.items()}
for future in as_completed(futures, timeout=timeout):
name = futures[future]
try:
results[name] = future.result()
except Exception as e:
logger.warning("取得 %s 失敗:%s", name, e)
results[name] = f"(無法取得{name}:{e})"
return results
# 用法
context = gather_context({
"天氣": lambda: summarize(get_weather("臺北市", config.CWA_API_KEY)),
"待辦": lambda: make_todo_summary(analyze_todos(load_todos())),
"時間": lambda: DateTimeTool().run(),
})
原本要 0.8 + 0.1 + 0.001 秒依序跑完,現在只要最慢的那個(0.8 秒)。
而且注意每個任務的失敗是獨立的——天氣查不到不影響待辦清單。結果裡會放一段說明文字,AI 看到就知道這項缺了。
💡 這裡用
ThreadPoolExecutor而不是asyncio,是因為我們的工具都是同步的(requests)。對 I/O 密集的任務,執行緒就夠用了,而且改動最小。
PERSPECTIVES = {
"天氣影響": "從天氣的角度分析,這個行程安排有什麼要注意的?",
"時間可行性": "從時間安排的角度分析,這個行程排得合理嗎?",
"優先順序": "從事情的輕重緩急分析,順序應該調整嗎?",
}
def multi_perspective_review(client, plan_text):
def make_task(question):
def task():
response = client.messages.create(
model="claude-opus-5",
max_tokens=1024,
messages=[{
"role": "user",
"content": f"<plan>\n{plan_text}\n</plan>\n\n{question}\n用三句話回答。",
}],
)
return extract_text(response)
return task
reviews = gather_context({
name: make_task(q) for name, q in PERSPECTIVES.items()
})
# 彙整
combined = "\n\n".join(f"【{k}】\n{v}" for k, v in reviews.items())
response = client.messages.create(
model="claude-opus-5",
max_tokens=2048,
messages=[{
"role": "user",
"content": (
f"以下是三個角度的分析:\n\n{combined}\n\n"
"請彙整成一段給使用者的建議,不超過 150 字,"
"指出最重要的一個調整建議。"
),
}],
)
return extract_text(response)
這個模式叫 fan-out / fan-in:先分散(每個角度各問一次),再收斂(彙整成一個答案)。
注意 make_task(q) 這個包一層的寫法——如果直接在迴圈裡寫 lambda: ...q...,所有的 lambda 會共用同一個 q(閉包陷阱),全部變成最後一個問題。這是 Python 常見的坑。
產生 → 評估 → 改進,直到夠好或達到次數上限。
"""workflows/refine.py"""
from pydantic import BaseModel, Field
class Evaluation(BaseModel):
reasoning: str = Field(description="逐項檢查每個標準的結果")
score: int = Field(ge=1, le=10, description="整體評分")
issues: list[str] = Field(description="需要改進的具體問題,最多三個")
passed: bool = Field(description="是否達到可以交付的標準")
def generate_with_refinement(client, task, criteria, max_rounds=3, target=8):
"""反覆產生與改進,直到通過評估。"""
draft = None
history = []
for round_num in range(1, max_rounds + 1):
# --- 產生 ---
if draft is None:
prompt = task
else:
issues = "\n".join(f"- {i}" for i in evaluation.issues)
prompt = (
f"{task}\n\n"
f"<previous_draft>\n{draft}\n</previous_draft>\n\n"
f"上一版有以下問題,請針對這些問題改進:\n{issues}\n\n"
f"直接輸出改進後的完整版本。"
)
response = client.messages.create(
model="claude-opus-5",
max_tokens=2048,
messages=[{"role": "user", "content": prompt}],
)
draft = extract_text(response)
# --- 評估 ---
evaluation = client.messages.parse(
model="claude-opus-5",
max_tokens=2048,
system="你是一個嚴格但公正的評審。逐項檢查後再給分。",
messages=[{
"role": "user",
"content": (
f"<criteria>\n{criteria}\n</criteria>\n\n"
f"<draft>\n{draft}\n</draft>\n\n"
"依照標準評估這份草稿。"
),
}],
output_format=Evaluation,
)
history.append({"round": round_num, "score": evaluation.score,
"issues": evaluation.issues})
logger.info("第 %d 輪:%d/10 分", round_num, evaluation.score)
if evaluation.passed and evaluation.score >= target:
break
return {"result": draft, "rounds": history, "final_score": evaluation.score}
用法:
result = generate_with_refinement(
client,
task="幫我寫一封給房東的信,說明冷氣壞了需要維修,希望本週內處理。",
criteria="""1. 語氣禮貌但明確
2. 具體說明問題(什麼壞了、什麼時候開始)
3. 明確提出期望的處理時間
4. 不超過 150 字
5. 使用繁體中文""",
)
print(result["result"])
print(f"\n跑了 {len(result['rounds'])} 輪,最終 {result['final_score']}/10")
⚠️ 這個模式很貴——每一輪要呼叫兩次模型。只在品質真的很重要時使用,而且一定要設 max_rounds。
另外,Evaluation 的 reasoning 欄位放在 score 前面,這是 Day 20 提過的技巧:讓模型先分析再給分,比直接給分準確。
我自己的決策流程:
這個任務的步驟固定嗎?
├─ 是 → 步驟間有依賴嗎?
│ ├─ 有 → Chain
│ └─ 沒有 → Parallel
└─ 否 → 是有限的幾種情況嗎?
├─ 是 → Routing(每個分支各自是 Chain 或 Parallel)
└─ 否 → 品質要求很高嗎?
├─ 是 → Evaluator-Optimizer
└─ 否 → Agent
再加一條:先用最簡單的做,不夠再升級。
我看過很多專案一開始就上 Agent,結果花了大錢、跑得很慢、行為不穩定,最後發現 80% 的請求其實用三行 if-else 就能處理。
有趣的一點:Workflow 本身也可以是 Agent 的工具。
class DailyBriefTool(Tool):
name = "generate_daily_brief"
tags = ("daily",)
description = (
"產生今日簡報,包含天氣重點、優先待辦事項與貼心提醒。"
"當使用者問「今天怎麼樣」「幫我看今天」「早安」時使用。"
"這個工具會自動取得所有需要的資訊,不需要先呼叫其他工具。"
)
input_schema = {"type": "object", "properties": {}}
def __init__(self, client):
self.client = client
def run(self):
return daily_brief_workflow(self.client)
這形成了一個很好的架構:
複雜度被封裝在工具裡,Agent 的決策空間變小,反而更準確更便宜。
明天深入 Workflow 的控制流——條件、迴圈、錯誤處理、狀態傳遞。