mirror of
https://gitcode.com/JianFeeeee/HomeAgent.git
synced 2026-10-03 15:53:56 +00:00
feat(clip): 多模态向量器(CLIP ONNX)——文本/图像 512 维共享空间 + 媒体向量写入与重算
- internal/memory/clip:CLIP ONNX 向量器(onnxruntime 构建标签控制,默认构建不链接 ONNX) - clip.New(modelDir) 加载 text.onnx/vision.onnx(输出 text_embed/image_embed [batch,512]) - 实现 vector.Vectorizer + vector.MultimodalEmbedder(Vectorize/EmbedImage + Dense 变体) - 词级 BPE tokenizer:merges 合并后词末片段带 </w> 查 vocab,与官方 encode 逐 id 对齐 - EmbedImage:解码→resize 224→NCHW→normalize→vision session - Fingerprint(text+vision 文件 sha256)供模型切换检测 - stub 版(无 onnxruntime 标签)保持默认构建行为不变 - vector/store.go:新增 MultimodalEmbedder 接口 - media.Store:新增 StaleVecDigests(currentModel)——查 vec_model 不匹配/缺失的图片 - agent core:AgentConfig.ClipEmbedder + Agent.clipEmb 接线; describePendingMedia 描述成功后 EmbedImageDense→SetVec; 新增 reembedStaleMedia 启动补算历史无向量图片 - config:core.memory.media.clip_model_dir(未配置退化为现有 fastText/TF-IDF 行为) - cmd/homed:读 clip_model_dir 加载 CLIP,失败仅记日志不阻塞启动 测试:TestSmokeLoadAndEncode(文本语义 cat>dog 0.914>physics 0.740)、 TestCrossModalAlignment(red-image vs red-text 0.063>blue -0.009,与 Python 一致)、 TestTokEnd(与官方 encode 逐 id 对齐)、TestStaleVecDigests,含 -race 全绿
This commit is contained in:
@ -24,9 +24,9 @@ import (
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
"sort"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
@ -646,6 +646,34 @@ func (s *Store) SetVec(digest string, vec []float64, model string) error {
|
||||
return err
|
||||
}
|
||||
|
||||
// StaleVecDigests 返回所有需要重新嵌入的图片 digest:
|
||||
// vec_model 不等于 currentModel(模型切换)或 vec_model 为空(从未嵌入)。
|
||||
// 调用方使用返回的 digest 列表调用 Get/EmbedImage/SetVec 完成重算。
|
||||
func (s *Store) StaleVecDigests(currentModel string) ([]string, error) {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
|
||||
rows, err := s.db.Query(`
|
||||
SELECT digest FROM media
|
||||
WHERE kind = 'image'
|
||||
AND COALESCE(description,'') != ''
|
||||
AND (COALESCE(vec_model,'') = '' OR vec_model != ?)
|
||||
ORDER BY last_seen`, currentModel)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
|
||||
var digests []string
|
||||
for rows.Next() {
|
||||
var d string
|
||||
if err := rows.Scan(&d); err == nil {
|
||||
digests = append(digests, d)
|
||||
}
|
||||
}
|
||||
return digests, rows.Err()
|
||||
}
|
||||
|
||||
// QueryMedia 用查询向量对所有已嵌入媒体做余弦相似度检索,返回 topK 个最相似的 Item。
|
||||
//
|
||||
// 这是跨模态检索的关键:查询可以是图片也可以是文本(经文本向量化后调用此方法),
|
||||
|
||||
@ -133,3 +133,43 @@ func TestSetVec_PersistsCorrectly(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStaleVecDigests(t *testing.T) {
|
||||
s := newTestStore(t, 0)
|
||||
defer s.Close()
|
||||
|
||||
// 有描述且 vec_model 匹配 → 非 stale
|
||||
d1, _ := s.Put([]byte("img1"), Item{MIME: "image/png", Description: "图一"})
|
||||
s.SetVec(d1, []float64{0.1}, "clip-vit-b32")
|
||||
|
||||
// 有描述但 vec_model 旧 → stale
|
||||
d2, _ := s.Put([]byte("img2"), Item{MIME: "image/png", Description: "图二"})
|
||||
s.SetVec(d2, []float64{0.2}, "clip-vit-b14")
|
||||
|
||||
// 有描述但从未嵌入(vec_model 空)→ stale
|
||||
d3, _ := s.Put([]byte("img3"), Item{MIME: "image/png", Description: "图三"})
|
||||
|
||||
// 无描述 → 不参与(描述流程外)
|
||||
s.Put([]byte("img4"), Item{MIME: "image/png"})
|
||||
|
||||
// 音频不属于图片 → 不算 stale
|
||||
s.Put([]byte("aud1"), Item{MIME: "audio/wav", Description: "语音"})
|
||||
|
||||
stale, err := s.StaleVecDigests("clip-vit-b32")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(stale) != 2 {
|
||||
t.Fatalf("expected 2 stale digests (d2 旧模型 + d3 未嵌入), got %d: %v", len(stale), stale)
|
||||
}
|
||||
got := map[string]bool{}
|
||||
for _, d := range stale {
|
||||
got[d] = true
|
||||
}
|
||||
if !got[d2] || !got[d3] {
|
||||
t.Errorf("expected d2 and d3 stale, got %v", stale)
|
||||
}
|
||||
if got[d1] {
|
||||
t.Errorf("d1 (匹配模型) 不应 stale")
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user