相較於只會「走一步看一步」的基礎 ReAct Agent,Planning Agent(規劃型 Agent) 具備前瞻性的架構設計。它在真正呼叫工具(Tool Call)之前,會先建構一個動態的任務圖譜或步驟清單,並在執行過程中根據環境反饋隨時調整計畫。
下面將說明 Planning Agent 的核心設計 pattern,並以 Python 實作一個支援 前置規劃(Planner)$\rightarrow$ 逐步執行(Executor)$\rightarrow$ 動態重規劃(Replanner) 的完整 Agent 引擎。
┌───────────────────────┐
│ 1. User Goal │
└───────────┬───────────┘
│
▼
┌───────────────────────┐
│ 2. Planner (高階推理) │ ───生成初始 Plan [1, 2, 3...]
└───────────┬───────────┘
│
▼
┌──────────────────────────────────────────────────────────┐
│ 3. Execution & Replanning Loop (執行與重規劃迴圈) │
│ │
│ ┌───────────────────────────┐ │
│ │ Executor (工具執行器) │ ◄─── 取出 Plan 的 Step 1 │
│ └─────────────┬─────────────┘ │
│ │ (執行並回傳 Result / Error) │
│ ▼ │
│ ┌───────────────────────────┐ │
│ │ Replanner (評估與校正) │ │
│ └─────────────┬─────────────┘ │
│ │ │
│ ┌───────┴───────┐ │
│ [成功續行] [遭遇異常/目標變更] │
│ │ │ │
│ ▼ ▼ │
│ 執行 Step 2 更新 Plan 內容 │
└─────────────────────────┬────────────────────────────────┘
│
▼
┌───────────────────────┐
│ 4. Final Result │
└───────────────────────┘
這個範例展示一個具備自我校正與動態重規劃能力的 Planning Agent,只需使用 Python 內建庫與 OpenAI SDK。
import json
from typing import List, Dict, Any, Optional
from pydantic import BaseModel, Field
from openai import OpenAI
client = OpenAI()
# ==========================================
# 1. 定義 Schema (Plan / Tool / Replanning)
# ==========================================
class Plan(BaseModel):
steps: List[str] = Field(description="解決使用者目標的詳細順序步驟清單")
class PlanResponse(BaseModel):
thought: str = Field(description="對目標的分析與規劃思考過程")
plan: Plan
class ReplanResponse(BaseModel):
thought: str = Field(description="分析當前執行結果與異常情況")
is_completed: bool = Field(description="目標是否已完全達成")
final_answer: Optional[str] = Field(None, description="若目標達成,填入給使用者的最終回覆")
updated_plan: Optional[Plan] = Field(None, description="若目標未達成,提供剩餘/調整後的步驟清單")
# ==========================================
# 2. 模擬外部工具庫 (Tools)
# ==========================================
def mock_search_product(query: str) -> str:
"""模擬商品搜尋 API"""
if "無線耳機" in query:
return json.dumps({"product": "SoundPro X", "price": 2990, "stock": True})
return json.dumps({"error": "找不到符合的商品"})
def mock_get_user_coupon(user_id: str) -> str:
"""模擬優惠券查詢 API"""
return json.dumps({"coupons": [{"code": "VIP88", "discount_rate": 0.88}]})
def mock_calculate_final_price(price: float, discount_rate: float) -> str:
"""計算折後價格"""
final_price = price * discount_rate
return json.dumps({"original_price": price, "final_price": round(final_price, 2)})
TOOLS_MAP = {
"search_product": mock_search_product,
"get_user_coupon": mock_get_user_coupon,
"calculate_final_price": mock_calculate_final_price
}
# ==========================================
# 3. Planning Agent 引擎實作
# ==========================================
class PlanningAgent:
def __init__(self, model: str = "gpt-4o-mini"):
self.model = model
def _generate_initial_plan(self, goal: str) -> PlanResponse:
"""階段一:根據 Goal 生成初始 Plan"""
prompt = (
f"你是一個頂級的 Planning Agent。\n"
f"使用者的最終目標為:【{goal}】\n\n"
f"可用的工具列表:\n"
f"- search_product(query): 搜尋商品與原價\n"
f"- get_user_coupon(user_id): 取得使用者優惠券\n"
f"- calculate_final_price(price, discount_rate): 計算折後價\n\n"
f"請分析目標並將其拆解為具體、可執行的步驟清單。"
)
completion = client.beta.chat.completions.parse(
model=self.model,
messages=[{"role": "user", "content": prompt}],
response_format=PlanResponse,
)
return completion.choices[0].message.parsed
def _execute_step(self, step: str, context_history: List[str]) -> str:
"""階段二:Executor 執行當前單一步驟(呼叫對應工具)"""
prompt = (
f"你是一個執行器 (Executor)。請根據歷史執行結果,執行當前步驟。\n\n"
f"歷史執行紀錄:\n" + "\n".join(context_history) + f"\n\n"
f"當前需執行的步驟:【{step}】\n"
f"請決定要呼叫的工具名稱與參數 (JSON 格式,如: {{\\\"tool\\\": \\\"search_product\\\", \\\"args\\\": {{\\\"query\\\": \\\"無線耳機\\\"}}}})"
)
response = client.chat.completions.create(
model=self.model,
messages=[{"role": "user", "content": prompt}],
response_format={"type": "json_object"}
)
action = json.loads(response.choices[0].message.content)
tool_name = action.get("tool")
args = action.get("args", {})
if tool_name in TOOLS_MAP:
result = TOOLS_MAP[tool_name](**args)
return f"執行工具 [{tool_name}] 參數 {args} -> 結果: {result}"
return f"無對應工具可執行步驟,模擬執行完成。"
def _replan(self, goal: str, completed_steps: List[str], current_plan: List[str]) -> ReplanResponse:
"""階段三:Replanner 評估執行狀況,進行動態校正或結束"""
prompt = (
f"使用者目標:【{goal}】\n\n"
f"已完成的步驟與結果:\n" + "\n".join(completed_steps) + f"\n\n"
f"原定未執行的步驟:\n" + "\n".join(current_plan) + f"\n\n"
f"請評估目標是否已達成:\n"
f"1. 若已完全達成,設定 is_completed=True 並填寫 final_answer。\n"
f"2. 若未達成,請重新調整並回傳剩餘的 updated_plan。"
)
completion = client.beta.chat.completions.parse(
model=self.model,
messages=[{"role": "user", "content": prompt}],
response_format=ReplanResponse,
)
return completion.choices[0].message.parsed
def run(self, goal: str) -> str:
print(f"🎯 [Goal]: {goal}\n")
# 1. 生成初始計畫
initial_plan_obj = self._generate_initial_plan(goal)
current_plan = initial_plan_obj.plan.steps
print(f"🧠 [Planner Thought]: {initial_plan_obj.thought}")
print("📋 [Initial Plan]:")
for idx, s in enumerate(current_plan, 1):
print(f" {idx}. {s}")
print("-" * 50)
completed_history: List[str] = []
step_count = 0
max_steps = 10 # 安全防護
# 2. 執行與動態重規劃迴圈
while current_plan and step_count < max_steps:
step_count += 1
current_step = current_plan.pop(0)
print(f"\n⚙️ [Step {step_count}]: {current_step}")
# 執行步驟
exec_result = self._execute_step(current_step, completed_history)
print(f" └─ {exec_result}")
completed_history.append(f"步驟: {current_step} | 結果: {exec_result}")
# 觸發 Replan 檢查
replan_res = self._replan(goal, completed_history, current_plan)
print(f"🔍 [Replanner Thought]: {replan_res.thought}")
if replan_res.is_completed:
print("\n✅ [Goal Achieved!]")
return replan_res.final_answer
if replan_res.updated_plan:
current_plan = replan_res.updated_plan.steps
print("🔄 [Plan Updated]:")
for idx, s in enumerate(current_plan, 1):
print(f" {idx}. {s}")
return "任務執行超時或未能順利完成。"
# ==========================================
# 4. 測試執行
# ==========================================
if __name__ == "__main__":
agent = PlanningAgent()
final_output = agent.run("幫我查詢使用者 'user_123' 購買 '無線耳機' 使用專屬優惠券折抵後的最終價格。")
print(f"\n🏁 [Final Answer]:\n{final_output}")
執行上述程式碼時,Console 會印出清晰的目標拆解 $\rightarrow$ 逐步執行 $\rightarrow$ 動態校正軌跡:
🎯 [Goal]: 幫我查詢使用者 'user_123' 購買 '無線耳機' 使用專屬優惠券折抵後的最終價格。
🧠 [Planner Thought]: 需要先查詢無線耳機的原價,再取得 user_123 的優惠券折扣,最後計算折後價格。
📋 [Initial Plan]:
1. 搜尋 '無線耳機' 商品並取得原價
2. 查詢使用者 'user_123' 的優惠券與折扣率
3. 計算折後價格並回傳結果
--------------------------------------------------
⚙️ [Step 1]: 搜尋 '無線耳機' 商品並取得原價
└─ 執行工具 [search_product] 參數 {'query': '無線耳機'} -> 結果: {"product": "SoundPro X", "price": 2990, "stock": true}
🔍 [Replanner Thought]: 已順利取得 SoundPro X 的原價 2990 元,接下來需查詢優惠券。
⚙️ [Step 2]: 查詢使用者 'user_123' 的優惠券與折扣率
└─ 執行工具 [get_user_coupon] 參數 {'user_id': 'user_123'} -> 結果: {"coupons": [{"code": "VIP88", "discount_rate": 0.88}]}
🔍 [Replanner Thought]: 已取得折扣率 0.88,接下來執行最終的價格計算。
⚙️ [Step 3]: 計算折後價格並回傳結果
└─ 執行工具 [calculate_final_price] 參數 {'price': 2990, 'discount_rate': 0.88} -> 結果: {"original_price": 2990, "final_price": 2631.2}
🔍 [Replanner Thought]: 已獲得折後最終價格,目標完全達成。
✅ [Goal Achieved!]
🏁 [Final Answer]:
使用者 'user_123' 購買 '無線耳機' (SoundPro X,原價 2,990 元) 套用 VIP88 優惠券 (88 折) 後,最終折扣價格為 2,631.2 元。
| 核心機制 | 作用說明 | 程式碼對應實作 |
|---|---|---|
| Separation of Concerns(職責分離) | 將「全域規劃」與「單步工具調用」交給不同 Prompt / 模型角色,避免任務發散。 | _generate_initial_plan 與 _execute_step 分離 |
| Dynamic Replanning(動態重規劃) | 每一步執行後,根據外部環境真實反饋重新評估,若發現工具出錯可立即修改後續 Plan。 | _replan 檢查點迴圈 |
| Guaranteed Output Schema | 使用 beta.chat.completions.parse 結合 Pydantic,確保 Plan 永遠是可程式化解析的結構陣列。 |
PlanResponse / ReplanResponse |