# Image-level similarity:
# Input: Image I, Text T ("black dog sitting on the left")
# Output: scalar score s in [0, 1] or dot product
similarity_score = image_text_model(image, text)
# Referring Expression Comprehension (Visual Grounding):
# Input: Image I, Text T ("black dog sitting on the left")
# Output: Bounding box coordinates [x_min, y_min, x_max, y_max]
bbox = grounding_model.predict_box(image, text)
import torch
import torch.nn.functional as F
# Closed-set matching: full pairwise similarity matrix over fixed evaluation set
image_embeds = F.normalize(torch.randn(1000, 512), dim=-1)
text_embeds = F.normalize(torch.randn(1000, 512), dim=-1)
sim_matrix = torch.matmul(text_embeds, image_embeds.T) # (1000, 1000)
# Open-world retrieval: query vector queried against an indexed corpus using ANN
# index = faiss.IndexHNSWFlat(512, 32)
# distances, indices = index.search(query_embed, k=10)
import torch
import torch.nn.functional as F
# Image 0 and Image 1 both depict 'a red sports car'
# Text 0: 'a red sports car', Text 1: 'a fast red car'
# In standard InfoNCE, Text 1 is treated as a negative for Image 0.
sim = torch.tensor([[0.95, 0.90], # Image 0 similarities
[0.88, 0.94]]) # Image 1 similarities
loss_img0 = -F.log_softmax(sim / 0.07, dim=-1)[0, 0]
print(f'InfoNCE loss for Image 0: {loss_img0.item():.4f}')
# High similarity to false negative Text 1 (0.90) heavily penalizes the model
マルチモーダル検索システムを大規模に更新する際には、表現のドリフト(representation drift)という問題が生じます。新しく生成された埋め込みベクトルは異なる幾何学的潜在空間に位置するため、既存のギャラリー埋め込みと直接比較すると再現率(recall)の大幅な低下を招きます。したがって、ダウンタイムなしでの更新には、二重インデックスによるブルー/グリーンデプロイ、互換性学習(後方互換性を持つ学習や変換アダプターなど)、または段階的な再インデックスパイプラインのいずれかが必要になります。標準的なブルー/グリーンの再インデックスワークフローでは、オフラインバッチジョブがギャラリー全体の新しい埋め込みを計算し、新しいベクトルインデックス(HNSWやIVFなど)を構築します。再構築が実行されている間も、本番の検索トラフィックは稼働中の旧インデックスに対してクエリを送信し続けます。新しく追加されるアイテムは、デュアルライト(二重書き込み)またはCDC(Change Data Capture)のストリーミングログを介して処理され、両方のインデックスが最新状態に保たれます。新しいインデックスがシャドウトラフィックによって検証され、メモリ上にウォームアップされた後、トラフィックは新しいエンコーダとインデックスへアトミックに切り替えられ、その後に旧インデックスが廃止されます。別のアプローチとして、後方互換性学習を用いて新しいクエリエンコーダを旧ギャラリー空間に揃えるよう制約することで、ギャラリーを非同期に時間をかけてバックフィルしつつ、クエリモデルを即座にアップグレードすることも可能です。
class MultimodalRetrievalRouter:
def __init__(self, v1_service, v2_service, dual_write=True):
self.v1 = v1_service # Model v1 + Index v1
self.v2 = v2_service # Model v2 + Index v2
self.active_version = "v1"
self.dual_write = dual_write
def search(self, query_multimodal):
# Route query strictly to the encoder matching the active index
if self.active_version == "v1":
return self.v1.search(query_multimodal)
return self.v2.search(query_multimodal)
def ingest_new_item(self, item):
# Dual-write during migration window so the new index is up-to-date at cutover
self.v1.index_item(item)
if self.dual_write and self.v2.is_ready_for_ingest:
self.v2.index_item(item)