feat: 完整实现 NLP 三元组提取系统 + token budget 上下文分配

- 重写 extractor.go: 分句、17条 POS 模板、依存模板 + COO 链、ATT合并
- parser.go: 分句循环 + TransE 向量验证(h+r≈t)
- fallback.go: jieba POS 降级解析器
- bridge.go: nlp.Triple ↔ memory.Triple 转换
- pipeline.go: extractKeyTriples 改用 NLP 提取器, 删除5条旧前缀规则
- distill.go: docToTriples 改用 NLP 提取器
- reorgGraph: 语义相似度增强检测, 保持纯 LLM 决断
- Provider 接口加 MaxContextTokens() + 模型窗口映射表
- tokenbudget.go: 中文 token 估算器 + budget 分配(80%利用率)
- process.go/buildSystemPrompt: 按 token 预算截断 memory+timeline
This commit is contained in:
root
2026-07-27 15:26:23 +08:00
parent d19b7bd13e
commit 1cb3e87dde
30 changed files with 1508 additions and 773 deletions

View File

@ -40,11 +40,8 @@ func TestDocToTriplesConversation(t *testing.T) {
}
triples := docToTriples(doc)
minLen := 2
hasJieba := needJieba()
if hasJieba && len(triples) <= minLen {
t.Errorf("expected more than %d triples with jieba, got %d", minLen, len(triples))
if len(triples) < 2 {
t.Errorf("expected at least 2 triples (主题+来源), got %d", len(triples))
}
for i, tr := range triples {
@ -55,19 +52,6 @@ func TestDocToTriplesConversation(t *testing.T) {
t.Errorf("triple[%d] has non-positive confidence: %+v", i, tr)
}
}
relCount := 0
for _, tr := range triples {
if tr.Relation == "关联" {
relCount++
if tr.Subject == tr.Object {
t.Errorf("关联 triple has same subject and object: %+v", tr)
}
}
}
if hasJieba && relCount == 0 {
t.Errorf("expected 关联 triples with jieba enabled, got 0 in %+v", triples)
}
}
func TestDocToTriplesMultiLine(t *testing.T) {