feat: 完整实现 NLP 三元组提取系统 + token budget 上下文分配

- 重写 extractor.go: 分句、17条 POS 模板、依存模板 + COO 链、ATT合并
- parser.go: 分句循环 + TransE 向量验证(h+r≈t)
- fallback.go: jieba POS 降级解析器
- bridge.go: nlp.Triple ↔ memory.Triple 转换
- pipeline.go: extractKeyTriples 改用 NLP 提取器, 删除5条旧前缀规则
- distill.go: docToTriples 改用 NLP 提取器
- reorgGraph: 语义相似度增强检测, 保持纯 LLM 决断
- Provider 接口加 MaxContextTokens() + 模型窗口映射表
- tokenbudget.go: 中文 token 估算器 + budget 分配(80%利用率)
- process.go/buildSystemPrompt: 按 token 预算截断 memory+timeline
This commit is contained in:
root
2026-07-27 15:26:23 +08:00
parent d19b7bd13e
commit 1cb3e87dde
30 changed files with 1508 additions and 773 deletions

View File

@ -4,6 +4,7 @@ import (
"log"
"os"
"path/filepath"
"strings"
"sync"
"github.com/yanyiwu/gojieba"
@ -92,13 +93,34 @@ var stopWords = map[string]bool{
"when": true, "who": true, "whom": true,
}
// TokenizeWords 使用 jieba 精确模式分词,返回去重后的所有词 token(不过滤停用词)
func TokenizeWords(text string) []string {
text = CleanText(text)
x := GetJieba()
if x == nil {
return nil
}
words := x.Cut(text, false)
var result []string
seen := make(map[string]bool)
for _, w := range words {
w = strings.TrimSpace(w)
if w == "" || seen[w] {
continue
}
seen[w] = true
result = append(result, w)
}
return result
}
func ExtractKeywords(text string) []string {
text = CleanText(text)
x := GetJieba()
if x == nil {
return nil
}
words := x.Cut(text, true)
words := x.Cut(text, false)
var keywords []string
seen := make(map[string]bool)
for _, w := range words {