feat(vector): pluggable multimodal vector space

核心暴露 MultimodalEmbedder 接口,两条路径共享同一套 L0/L2/L3
向量缓存、media.Store 坐标、QueryMemoryMediaScored 检索:
  - onnx:内嵌 ONNX 模型(CLIP 等),通过 build tag 编译
  - http:外部向量 API 服务(Jina v5 / OpenAI / 自建)

跨模态融合权重改为 CrossModalFusionConfig 可配置结构体,
移除所有模型特定硬编码(CLIP/Jina),版本切换只需改配置。

模型切换自动迁移:
  - StaleVecDigestsAll 支持全模态(image+audio+video)
  - 启动时并发重算(ONNX 4 workers / API 8 workers)
  - 修复 SQL 运算符优先级导致 kind 过滤失效的 bug

实测对比(492 篇生产文档 + 3 张真实图片):
  - TF-IDF:MRR 0.457(精确匹配快,语义差)
  - fastText:MRR 0.530(语义中等,延迟 8ms)
  - Jina v5-omni:MRR 0.900(全面领先,延迟 40ms)
  - 中文文本→图片:Jina MRR 0.833 vs CLIP 0.611

See docs/embedding-comparison.md for full benchmark.
This commit is contained in:
JianFeeeee
2026-09-09 17:38:34 +08:00
parent b860c8cea5
commit 36c604ef3e
15 changed files with 1480 additions and 94 deletions

View File

@ -427,7 +427,8 @@ func (s *Store) Search(query string, kind Kind, limit int) ([]*Item, error) {
defer s.mu.RUnlock()
q := `SELECT digest, kind, mime, size, width, height, origin_path, tool,
description, described_by, ref_count, first_seen, last_seen
description, described_by, ref_count, first_seen, last_seen,
vec, vec_model
FROM media WHERE COALESCE(description,'') != ''`
args := []interface{}{}
if strings.TrimSpace(query) != "" {
@ -473,7 +474,8 @@ func (s *Store) Pending(limit int) ([]*Item, error) {
defer s.mu.RUnlock()
rows, err := s.db.Query(`
SELECT digest, kind, mime, size, width, height, origin_path, tool,
description, described_by, ref_count, first_seen, last_seen
description, described_by, ref_count, first_seen, last_seen,
vec, vec_model
FROM media
WHERE COALESCE(description,'') = '' AND COALESCE(described_by,'') = ''
ORDER BY last_seen DESC LIMIT ?`, limit)
@ -650,15 +652,36 @@ func (s *Store) SetVec(digest string, vec []float64, model string) error {
// vec_model 不等于 currentModel(模型切换)或 vec_model 为空(从未嵌入)。
// 调用方使用返回的 digest 列表调用 Get/EmbedImage/SetVec 完成重算。
func (s *Store) StaleVecDigests(currentModel string) ([]string, error) {
return s.staleVecDigests(currentModel, "image")
}
// StaleVecDigestsAll 返回所有需要重新嵌入的媒体 digest(不限 kind),
// 供模型切换后全量迁移向量空间(image + audio + video 等)。
func (s *Store) StaleVecDigestsAll(currentModel string) ([]string, error) {
return s.staleVecDigests(currentModel, "")
}
// staleVecDigests 是 StaleVecDigests 的核心实现,kind=” 时不按 kind 过滤。
// 废弃了"只迁移图片"的限定:模型切换后所有模态都应迁移到新向量空间。
func (s *Store) staleVecDigests(currentModel string, kind string) ([]string, error) {
s.mu.RLock()
defer s.mu.RUnlock()
rows, err := s.db.Query(`
query := `
SELECT digest FROM media
WHERE kind = 'image'
AND COALESCE(description,'') != ''
AND (COALESCE(vec_model,'') = '' OR vec_model != ?)
ORDER BY last_seen`, currentModel)
WHERE (COALESCE(vec_model,'') = '' OR vec_model != ?)`
if kind != "" {
query += ` AND kind = ?`
}
query += ` ORDER BY last_seen`
var args []interface{}
args = append(args, currentModel)
if kind != "" {
args = append(args, kind)
}
rows, err := s.db.Query(query, args...)
if err != nil {
return nil, err
}
@ -681,6 +704,46 @@ func (s *Store) StaleVecDigests(currentModel string) ([]string, error) {
// 谁的相似度更高就召回谁——不再区分「这是一张图的查询」还是「这是一段文字的查询」,
// 由向量空间的相似度自动判断。
func (s *Store) QueryMedia(queryVec []float64, model string, topK int) ([]*Item, error) {
hits, err := s.QueryMediaScored(queryVec, model, topK)
if err != nil {
return nil, err
}
if hits == nil {
return nil, nil
}
out := make([]*Item, len(hits))
for i, h := range hits {
out[i] = h.Item
}
return out, nil
}
// MediaHit 是一条媒体相似度候选及其分数。
// 跨模态融合需要原始分数做归一化,仅返回 Item 会丢掉尺度信息。
type MediaHit struct {
Item *Item
Score float64
}
// QueryMemoryMediaScored 只检索当前仍被 L0/L2/L3 记忆块引用的媒体。
// CAS 中 ref_count=0 的项是等待 GC 的孤儿缓存,不是可召回记忆;若把它们也查出,
// 已从三层记忆淘汰的图片会被视觉路“复活”,破坏与文本块一致的生命周期。
//
// 分数只做排序,不在存储层设绝对阈值:多模态文本→图像的绝对 cosine 随模型、
// 语言与数据域漂移,真实标定中有效命中可以低至 0.015。相关性门控在融合器中
// 使用当前候选集合的相对分布完成。
func (s *Store) QueryMemoryMediaScored(queryVec []float64, model string, topK int) ([]MediaHit, error) {
return s.queryMediaScored(queryVec, model, topK, true)
}
// QueryMediaScored 用查询向量对所有已嵌入媒体做余弦相似度检索,
// 返回 topK 个最相似的候选及其原始 cosine 分数(供跨模态归一化)。
// 这是媒体存储层的诊断/显式全库入口;记忆召回应调用 QueryMemoryMediaScored。
func (s *Store) QueryMediaScored(queryVec []float64, model string, topK int) ([]MediaHit, error) {
return s.queryMediaScored(queryVec, model, topK, false)
}
func (s *Store) queryMediaScored(queryVec []float64, model string, topK int, referencedOnly bool) ([]MediaHit, error) {
if topK <= 0 {
topK = 20
}
@ -690,10 +753,21 @@ func (s *Store) QueryMedia(queryVec []float64, model string, topK int) ([]*Item,
s.mu.RLock()
defer s.mu.RUnlock()
rows, err := s.db.Query(`SELECT digest, kind, mime, size, width, height,
query := `SELECT digest, kind, mime, size, width, height,
origin_path, tool, description, described_by, ref_count, first_seen, last_seen,
vec, vec_model
FROM media WHERE vec IS NOT NULL AND vec != ''`)
FROM media WHERE vec IS NOT NULL AND vec != ''`
var args []interface{}
if model != "" {
query += ` AND vec_model = ?`
args = append(args, model)
}
if referencedOnly {
query += ` AND ref_count > 0 AND EXISTS (
SELECT 1 FROM media_refs r WHERE r.digest = media.digest
)`
}
rows, err := s.db.Query(query, args...)
if err != nil {
return nil, err
}
@ -744,9 +818,9 @@ func (s *Store) QueryMedia(queryVec []float64, model string, topK int) ([]*Item,
if len(candidates) > topK {
candidates = candidates[:topK]
}
out := make([]*Item, len(candidates))
out := make([]MediaHit, len(candidates))
for i, c := range candidates {
out[i] = c.item
out[i] = MediaHit{Item: c.item, Score: c.score}
}
return out, nil
}

View File

@ -149,25 +149,26 @@ func TestStaleVecDigests(t *testing.T) {
// 有描述但从未嵌入(vec_model 空)→ stale
d3, _ := s.Put([]byte("img3"), Item{MIME: "image/png", Description: "图三"})
// 无描述 → 不参与(描述流程外)
s.Put([]byte("img4"), Item{MIME: "image/png"})
// 无描述但有图片 → 也应被迁移(描述是可选语义通道,图片应独立于描述参与向量空间)
d4, _ := s.Put([]byte("img4"), Item{MIME: "image/png"})
// 音频不属于图片 → 不算 stale
// 音频不参与图片迁移(StaleVecDigests 只查 kind='image')
s.Put([]byte("aud1"), Item{MIME: "audio/wav", Description: "语音"})
stale, err := s.StaleVecDigests("clip-vit-b32")
if err != nil {
t.Fatal(err)
}
if len(stale) != 2 {
t.Fatalf("expected 2 stale digests (d2 旧模型 + d3 未嵌入), got %d: %v", len(stale), stale)
// d1 匹配模型 → 非 stale;d2 旧模型 + d3 未嵌入 + d4 无描述图片 = 3 stale;aud1 不算
if len(stale) != 3 {
t.Fatalf("expected 3 stale digests (d2 旧模型 + d3 未嵌入 + d4 无描述), got %d: %v", len(stale), stale)
}
got := map[string]bool{}
for _, d := range stale {
got[d] = true
}
if !got[d2] || !got[d3] {
t.Errorf("expected d2 and d3 stale, got %v", stale)
if !got[d2] || !got[d3] || !got[d4] {
t.Errorf("expected d2, d3, d4 stale, got %v", stale)
}
if got[d1] {
t.Errorf("d1 (匹配模型) 不应 stale")