一、为什么需要三层架构单一决策方式都有短板方式优势劣势硬规则确定性强、零延迟、零成本无法处理模糊/多变场景Jev语义理解、低成本、低延迟不支持生成、复杂推理弱LLM强大生成能力、复杂推理高延迟、高成本、有幻觉三层架构把每个决策分配给最适合的层。二、三层决策流水线用户输入 │ ▼ ┌────────────────────────────────────────────────────────────┐ │ 第一层硬规则 │ │ ┌──────────────────────────────────────────────────────┐ │ │ │ • 黑名单/白名单匹配 │ │ │ │ • 正则表达式检测 │ │ │ │ • 频率限制 │ │ │ │ • 权限检查 │ │ │ │ 特点O(1)时间零成本不可绕过 │ │ │ └──────────────────────────────────────────────────────┘ │ │ │ │ │ ▼ │ │ ┌──────────────────────────────────────────────────────┐ │ │ │ 第二层Jev 决策 │ │ │ │ ┌──────────────────────────────────────────────────┐ │ │ │ │ │ • 意图分类 │ │ │ │ │ │ • 工具选择 │ │ │ │ │ │ • 风险打分 │ │ │ │ │ │ • 内容安全检测 │ │ │ │ │ │ 特点~100ms极低成本语义理解 │ │ │ │ │ └──────────────────────────────────────────────────┘ │ │ │ └──────────────────────────────────────────────────────┘ │ │ │ │ │ ▼ │ │ ┌──────────────────────────────────────────────────────┐ │ │ │ 第三层LLM 生成 │ │ │ │ ┌──────────────────────────────────────────────────┐ │ │ │ │ │ • 复杂推理 │ │ │ │ │ │ • 多步规划 │ │ │ │ │ │ • 回复生成 │ │ │ │ │ │ • 代码生成 │ │ │ │ │ │ 特点1-10s高成本最强能力 │ │ │ │ │ └──────────────────────────────────────────────────┘ │ │ │ └──────────────────────────────────────────────────────┘ │ │ │ │ │ ▼ │ │ 输出给用户 │ └────────────────────────────────────────────────────────────┘三、Go 代码实现package main import ( encoding/json fmt regexp strings sync sync/atomic time ) // // 1. 三层决策引擎核心接口 // type DecisionInput struct { UserID string json:user_id Role string json:role Message string json:message SessionID string json:session_id Extra map[string]interface{} json:extra } type DecisionResult struct { Allowed bool json:allowed Route string json:route // 路由目标 RiskLevel int json:risk_level // 1-5 Intent string json:intent ToolName string json:tool_name,omitempty SkipLLM bool json:skip_llm // 是否跳过 LLM HardReply string json:hard_reply,omitempty // 硬规则直接回复 DecisionBy string json:decision_by // rule, jev, llm LatencyMs int64 json:latency_ms Steps []DecisionStep json:steps } type DecisionStep struct { Layer string json:layer Action string json:action Result string json:result LatencyMs int64 json:latency_ms } // // 2. 第一层硬规则引擎 // type RuleEngine struct { rules []Rule } type Rule struct { Name string Priority int // 越小优先级越高 Match func(input DecisionInput) bool Action func(input DecisionInput) *DecisionResult } func NewRuleEngine() *RuleEngine { return RuleEngine{} } func (re *RuleEngine) AddRule(r Rule) { re.rules append(re.rules, r) // 按优先级排序 for i : 0; i len(re.rules); i { for j : i 1; j len(re.rules); j { if re.rules[j].Priority re.rules[i].Priority { re.rules[i], re.rules[j] re.rules[j], re.rules[i] } } } } func (re *RuleEngine) Evaluate(input DecisionInput) (*DecisionResult, bool) { for _, rule : range re.rules { if rule.Match(input) { result : rule.Action(input) result.DecisionBy rule return result, true } } return nil, false } // 预置规则工厂 func DefaultRules() *RuleEngine { re : NewRuleEngine() // 规则1黑名单用户直接拦截 blacklist : sync.Map{} blacklist.Store(spammer_001, true) blacklist.Store(hacker_007, true) re.AddRule(Rule{ Name: 黑名单拦截, Priority: 1, Match: func(input DecisionInput) bool { _, blocked : blacklist.Load(input.UserID) return blocked }, Action: func(input DecisionInput) *DecisionResult { return DecisionResult{ Allowed: false, Route: blocked, RiskLevel: 5, HardReply: 您的账号已被限制访问请联系客服。, SkipLLM: true, } }, }) // 规则2敏感词直接拦截 sensitiveWords : []string{炸弹, 毒品, 枪支} re.AddRule(Rule{ Name: 敏感词拦截, Priority: 2, Match: func(input DecisionInput) bool { msg : strings.ToLower(input.Message) for _, w : range sensitiveWords { if strings.Contains(msg, w) { return true } } return false }, Action: func(input DecisionInput) *DecisionResult { return DecisionResult{ Allowed: false, Route: blocked, RiskLevel: 5, HardReply: 您发送的内容包含敏感信息已拦截。, SkipLLM: true, } }, }) // 规则3正则匹配订单号直接路由 orderPattern : regexp.MustCompile(订单[号]?[:]?\s*([A-Za-z0-9]{6,20})) re.AddRule(Rule{ Name: 订单查询路由, Priority: 3, Match: func(input DecisionInput) bool { return orderPattern.MatchString(input.Message) }, Action: func(input DecisionInput) *DecisionResult { matches : orderPattern.FindStringSubmatch(input.Message) orderID : if len(matches) 1 { orderID matches[1] } return DecisionResult{ Allowed: true, Route: order_service, Intent: 查询订单, RiskLevel: 1, SkipLLM: false, Extra: map[string]interface{}{order_id: orderID}, } }, }) // 规则4问候语直接回复 greetings : regexp.MustCompile(^(你好|您好|hi|hello|嗨|hey)\s*[!。.]*$) re.AddRule(Rule{ Name: 问候语快捷回复, Priority: 4, Match: func(input DecisionInput) bool { return greetings.MatchString(strings.TrimSpace(input.Message)) }, Action: func(input DecisionInput) *DecisionResult { return DecisionResult{ Allowed: true, Route: direct_reply, RiskLevel: 1, SkipLLM: true, HardReply: 您好我是 AI 助手有什么可以帮您的吗, } }, }) return re } // // 3. 第二层Jev 决策引擎 // type JevEngine struct { client *JevClient } func NewJevEngine(apiKey string) *JevEngine { return JevEngine{client: JevClient{}} } func (je *JevEngine) Evaluate(input DecisionInput) *DecisionResult { start : time.Now() // 模拟 Jev 批量决策 // 实际应调用 Jev API result : DecisionResult{ Allowed: true, Route: general, Intent: general_query, RiskLevel: 2, DecisionBy: jev, } // 模拟决策逻辑 msg : strings.ToLower(input.Message) switch { case strings.Contains(msg, 退款): result.Intent 退款申请 result.Route refund_service result.RiskLevel 3 case strings.Contains(msg, 投诉): result.Intent 投诉反馈 result.Route complaint result.RiskLevel 4 case strings.Contains(msg, 密码) || strings.Contains(msg, 账号): result.Intent 账号问题 result.Route account_service result.RiskLevel 3 default: result.Intent general_query result.Route general result.RiskLevel 1 } result.LatencyMs time.Since(start).Milliseconds() return result } // // 4. 第三层LLM 引擎简化 // type LLMEngine struct { model string } func NewLLMEngine(model string) *LLMEngine { return LLMEngine{model: model} } func (le *LLMEngine) Generate(input DecisionInput, context map[string]interface{}) (string, error) { // 模拟 LLM 调用 time.Sleep(1500 * time.Millisecond) // 模拟 1.5s 延迟 return fmt.Sprintf(针对您的问题「%s」我们的建议是请联系客服热线 400-xxx-xxxx 获取进一步帮助。, input.Message), nil } // // 5. 三层编排器 // type ThreeLayerOrchestrator struct { ruleEngine *RuleEngine jevEngine *JevEngine llmEngine *LLMEngine stats *Statistics } type Statistics struct { RuleCount atomic.Int64 JevCount atomic.Int64 LLMCount atomic.Int64 Blocked atomic.Int64 } func NewThreeLayerOrchestrator(apiKey string) *ThreeLayerOrchestrator { return ThreeLayerOrchestrator{ ruleEngine: DefaultRules(), jevEngine: NewJevEngine(apiKey), llmEngine: NewLLMEngine(gpt-4o-mini), stats: Statistics{}, } } func (tlo *ThreeLayerOrchestrator) Process(input DecisionInput) *DecisionResult { overallStart : time.Now() var steps []DecisionStep // 第一层硬规则 stepStart : time.Now() if result, matched : tlo.ruleEngine.Evaluate(input); matched { tlo.stats.RuleCount.Add(1) if !result.Allowed { tlo.stats.Blocked.Add(1) } result.LatencyMs time.Since(overallStart).Milliseconds() result.Steps append(steps, DecisionStep{ Layer: rule, Action: result.Route, Result: fmt.Sprintf(命中规则: %s, result.HardReply), LatencyMs: time.Since(stepStart).Milliseconds(), }) return result } steps append(steps, DecisionStep{ Layer: rule, Action: no_match, Result: 未命中任何规则, LatencyMs: time.Since(stepStart).Milliseconds(), }) // 第二层Jev 决策 stepStart time.Now() jevResult : tlo.jevEngine.Evaluate(input) tlo.stats.JevCount.Add(1) steps append(steps, DecisionStep{ Layer: jev, Action: jevResult.Intent, Result: fmt.Sprintf(路由: %s, 风险: %d/5, jevResult.Route, jevResult.RiskLevel), LatencyMs: time.Since(stepStart).Milliseconds(), }) // Jev 高风险直接拦截 if jevResult.RiskLevel 5 { tlo.stats.Blocked.Add(1) return DecisionResult{ Allowed: false, Route: blocked_by_jev, RiskLevel: 5, Intent: jevResult.Intent, SkipLLM: true, HardReply: 系统检测到高风险操作已自动拦截。如有疑问请联系客服。, DecisionBy: jev, LatencyMs: time.Since(overallStart).Milliseconds(), Steps: steps, } } // Jev 低风险可直接回复的场景 if jevResult.RiskLevel 1 jevResult.Intent general_query { tlo.stats.LLMCount.Add(1) // 仍然需要 LLM 生成回复 stepStart time.Now() reply, _ : tlo.llmEngine.Generate(input, map[string]interface{}{ route: jevResult.Route, intent: jevResult.Intent, }) steps append(steps, DecisionStep{ Layer: llm, Action: generate_reply, Result: reply[:min(50, len(reply))] ..., LatencyMs: time.Since(stepStart).Milliseconds(), }) return DecisionResult{ Allowed: true, Route: jevResult.Route, RiskLevel: jevResult.RiskLevel, Intent: jevResult.Intent, SkipLLM: false, HardReply: reply, DecisionBy: llm, LatencyMs: time.Since(overallStart).Milliseconds(), Steps: steps, } } // 第三层LLM 生成 stepStart time.Now() tlo.stats.LLMCount.Add(1) reply, _ : tlo.llmEngine.Generate(input, map[string]interface{}{ route: jevResult.Route, intent: jevResult.Intent, risk_level: jevResult.RiskLevel, }) steps append(steps, DecisionStep{ Layer: llm, Action: complex_generation, Result: reply[:min(50, len(reply))] ..., LatencyMs: time.Since(stepStart).Milliseconds(), }) return DecisionResult{ Allowed: true, Route: jevResult.Route, RiskLevel: jevResult.RiskLevel, Intent: jevResult.Intent, SkipLLM: false, HardReply: reply, DecisionBy: llm, LatencyMs: time.Since(overallStart).Milliseconds(), Steps: steps, } } func (tlo *ThreeLayerOrchestrator) PrintStats() { fmt.Println(\n--- 三层引擎统计 ---) fmt.Printf( 硬规则处理: %d 次\n, tlo.stats.RuleCount.Load()) fmt.Printf( Jev 决策: %d 次\n, tlo.stats.JevCount.Load()) fmt.Printf( LLM 生成: %d 次\n, tlo.stats.LLMCount.Load()) fmt.Printf( 拦截总数: %d 次\n, tlo.stats.Blocked.Load()) total : tlo.stats.RuleCount.Load() tlo.stats.JevCount.Load() if total 0 { fmt.Printf( 硬规则拦截占比: %.1f%%\n, float64(tlo.stats.Blocked.Load())/float64(total)*100) } } // // 6. 回退策略 // type FallbackStrategy struct { maxRetries int timeoutMs int fallbackMsg string } func NewFallbackStrategy() *FallbackStrategy { return FallbackStrategy{ maxRetries: 2, timeoutMs: 8000, fallbackMsg: 系统繁忙请稍后再试。, } } func (fs *FallbackStrategy) ExecuteWithFallback( input DecisionInput, fn func(DecisionInput) (*DecisionResult, error), ) *DecisionResult { var lastErr error for i : 0; i fs.maxRetries; i { done : make(chan *DecisionResult, 1) errCh : make(chan error, 1) go func() { result, err : fn(input) if err ! nil { errCh - err return } done - result }() select { case result : -done: return result case err : -errCh: lastErr err time.Sleep(time.Duration(100*(1i)) * time.Millisecond) case -time.After(time.Duration(fs.timeoutMs) * time.Millisecond): lastErr fmt.Errorf(超时) } } // 全部失败返回兜底回复 return DecisionResult{ Allowed: true, Route: fallback, RiskLevel: 1, SkipLLM: true, HardReply: fs.fallbackMsg, DecisionBy: fallback, Extra: map[string]interface{}{error: lastErr.Error()}, } } // // 7. 成本监控 // type CostTracker struct { mu sync.Mutex costs map[string]*CostItem } type CostItem struct { Calls int64 json:calls TotalMs int64 json:total_ms CostUSD float64 json:cost_usd } func NewCostTracker() *CostTracker { return CostTracker{ costs: make(map[string]*CostItem), } } func (ct *CostTracker) Record(layer string, calls int64, latencyMs int64) { ct.mu.Lock() defer ct.mu.Unlock() item, ok : ct.costs[layer] if !ok { item CostItem{} ct.costs[layer] item } item.Calls calls item.TotalMs latencyMs // 成本估算 switch layer { case rule: item.CostUSD 0 // 硬规则免费 case jev: item.CostUSD float64(calls) * 0.000042 // $0.042/千次 case llm: item.CostUSD float64(calls) * 0.003 // $0.003/次 (GPT-4o-mini) } } func (ct *CostTracker) Report() { ct.mu.Lock() defer ct.mu.Unlock() fmt.Println(\n--- 成本报告 ---) var totalCost float64 for layer, item : range ct.costs { fmt.Printf( %s: %d 次调用, 总延迟 %dms, 费用 $%.4f\n, layer, item.Calls, item.TotalMs, item.CostUSD) totalCost item.CostUSD } fmt.Printf( 总费用: $%.4f\n, totalCost) } // // 8. 主程序演示 // func main() { fmt.Println( 第4讲混合架构LLM Jev 硬规则 \n) orchestrator : NewThreeLayerOrchestrator(your-api-key) costTracker : NewCostTracker() fallback : NewFallbackStrategy() testCases : []DecisionInput{ { UserID: user_normal, Role: user, Message: 你好, }, { UserID: user_normal, Role: user, Message: 帮我查一下订单号 ORD20260923001, }, { UserID: spammer_001, Role: user, Message: 帮我退款, }, { UserID: user_normal, Role: user, Message: 我想买一把玩具枪, }, { UserID: user_normal, Role: user, Message: 我要投诉你们的服务质量等了半小时没人理, }, { UserID: user_normal, Role: user, Message: 我忘记密码了怎么重置, }, } for i, tc : range testCases { fmt.Printf(\n 场景 %d: %s \n, i1, tc.Message) // 使用回退策略执行 result : fallback.ExecuteWithFallback(tc, func(input DecisionInput) (*DecisionResult, error) { return orchestrator.Process(input), nil }) // 记录成本 costTracker.Record(result.DecisionBy, 1, result.LatencyMs) // 输出结果 fmt.Printf( 决策层: %s\n, result.DecisionBy) fmt.Printf( 允许: %v | 路由: %s | 风险: %d/5\n, result.Allowed, result.Route, result.RiskLevel) fmt.Printf( 意图: %s\n, result.Intent) fmt.Printf( 总耗时: %dms\n, result.LatencyMs) fmt.Printf( 回复: %s\n, result.HardReply) fmt.Printf( 决策链路:\n) for _, step : range result.Steps { fmt.Printf( [%dms] %s → %s\n, step.LatencyMs, step.Layer, step.Result) } } // 输出统计 orchestrator.PrintStats() costTracker.Report() // 成本对比总结 fmt.Println(\n--- 成本对比假设日均 10 万次请求 ---) fmt.Println( 纯 LLM 方案: 100,000 × $0.003 $300/天) fmt.Println( 三层架构方案:) fmt.Println( 硬规则拦截 ~30%: 30,000 × $0 $0) fmt.Println( Jev 决策 ~60%: 60,000 × $0.000042 $2.52) fmt.Println( LLM 生成 ~10%: 10,000 × $0.003 $30) fmt.Println( 合计: $32.52/天) fmt.Println( 节省: 89.2%) } func min(a, b int) int { if a b { return a } return b }四、决策流程图用户输入 │ ▼ ┌─────────────────────┐ │ 第一层硬规则 │ │ │ │ 黑名单─────是─────► 直接拦截 回复 │ │ │ 敏感词─────是─────► 直接拦截 回复 │ │ │ 问候语─────是─────► 快捷回复 (跳过后续) │ │ │ 订单号─────是─────► 路由到订单服务 │ │ │ 均不匹配 │ └──────────┬──────────┘ │ ▼ ┌─────────────────────┐ │ 第二层Jev 决策 │ │ │ │ 意图分类 │ │ 风险打分 │ │ 工具选择 │ │ │ │ 高风险─────是─────► 拦截 告警 │ │ │ 低风险─────是─────► 直接 LLM 生成回复 │ │ │ 中等风险 │ └──────────┬──────────┘ │ ▼ ┌─────────────────────┐ │ 第三层LLM 生成 │ │ │ │ 复杂推理 │ │ 多步规划 │ │ 回复生成 │ └──────────┬──────────┘ │ ▼ 输出五、各层职责矩阵场景硬规则JevLLM黑名单用户✅ 拦截——敏感词✅ 拦截——问候语✅ 快捷回复——订单号匹配✅ 路由——意图分类—✅—风险打分—✅—工具选择—✅—内容安全检测✅ 基础✅ 语义—复杂推理——✅回复生成——✅代码生成——✅情感分析—✅✅(精细)六、生产部署配置# config.yaml three_layer: rule: enabled: true cache_size: 10000 reload_interval: 60s jev: model: jev-1 timeout: 2000ms retry: max_attempts: 2 backoff: 100ms llm: model: gpt-4o-mini timeout: 10000ms max_tokens: 1024 temperature: 0.3 fallback: max_retries: 2 timeout_ms: 8000 message: 系统繁忙请稍后再试。 cost_control: daily_budget_usd: 100 alert_threshold: 0.8 # 达到预算 80% 告警 metrics: prometheus: true log_interval: 60s七、关键要点硬规则是第一道防线 — 零成本、零延迟、不可绕过Jev 承担 60-70% 的决策 — 大幅降低 LLM 调用量LLM 只做最后 10-20% — 复杂推理和生成回退策略保证可用性 — 任何一层失败都有兜底成本可降低 80-90% — 相比纯 LLM 方案延迟可降低 50-70% — 大部分请求止步于前两层 开发之余的小工具推荐调试三层架构时经常需要查看 JSON 格式的配置文件和决策日志。zz365.top 的 JSON 格式化工具可以快速整理配置文件结构。还有 Base64 编解码器调试 API 鉴权的 token 时很方便。所有工具纯前端本地处理生产配置数据不会上传到服务器。下一讲预告 第5讲「生产案例客服路由 / 内容审核 / 工具调用守卫」—— 三个完整的端到端生产案例含部署配置和监控面板设计。