NEWVectors or files. Pick a path.Start →
    SimilarCross-Media

    Multimodal Search with MVS

    Build multimodal search by embedding different content types (text, images, video frames) with your own models and searching across them in a single MVS namespace. Use CLIP or any multimodal embedding model for cross-modal retrieval.

    text
    image
    video
    Multi-Tier

    "dog playing outdoors"

    Why This Matters

    Multimodal search without a managed pipeline. Use your preferred CLIP, SigLIP, or any embedding model to embed text, images, and video into a shared vector space, then search across all of them with MVS.

    from mixpeek import Mixpeek
    client = Mixpeek(api_key="YOUR_API_KEY")
    NAMESPACE = "multimodal-search"
    # --- local embedding model (runs on your machine; any CLIP-compatible model works) ---
    import open_clip
    import torch
    from PIL import Image
    model, _, preprocess = open_clip.create_model_and_transforms("ViT-B-32", pretrained="laion2b_s34b_b79k")
    tokenizer = open_clip.get_tokenizer("ViT-B-32")
    def embed_text(text):
    with torch.no_grad():
    features = model.encode_text(tokenizer([text]))
    return (features / features.norm(dim=-1, keepdim=True))[0].tolist()
    def embed_image(path):
    with torch.no_grad():
    features = model.encode_image(preprocess(Image.open(path)).unsqueeze(0))
    return (features / features.norm(dim=-1, keepdim=True))[0].tolist()
    # --- end local embedding model ---
    # 1. A standalone namespace for 512-number CLIP vectors
    client.namespaces.create(
    namespace_id=NAMESPACE,
    mode="standalone",
    vector_configs=[{"name": "clip", "dimension": 512, "metric": "cosine"}],
    )
    # 2. Text and images share the space, so both go in as the same vector name
    texts = ["A golden retriever playing fetch in the park", "Sunset over the ocean with sailboats"]
    images = ["dog_park.jpg", "sunset.jpg"]
    client.namespaces.documents.upsert(
    namespace_id=NAMESPACE,
    documents=[
    {"document_id": f"text-{i}", "vectors": {"clip": embed_text(t)}, "payload": {"modality": "text", "content": t}}
    for i, t in enumerate(texts)
    ]
    + [
    {"document_id": f"image-{i}", "vectors": {"clip": embed_image(p)}, "payload": {"modality": "image", "file": p}}
    for i, p in enumerate(images)
    ],
    )
    # 3. A retriever over the CLIP vector
    retriever = client.retrievers.create(
    retriever_name="clip-search",
    input_schema={"qv": {"type": "array", "required": True}},
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "clip",
    "query": {"input_mode": "vector", "value": "{{INPUT.qv}}"},
    "top_k": 10,
    },
    ],
    "final_top_k": 10,
    },
    },
    ],
    )
    # A text query finds images and text alike
    query = embed_text("dog playing outdoors")
    results = client.retrievers.execute(retriever["retriever_id"], inputs={"qv": query})
    for doc in results["documents"]:
    print(round(doc["score"], 3), doc.get("modality"), doc.get("content") or doc.get("file"))
    # The same query, filtered to images
    images_only = client.retrievers.execute(
    retriever["retriever_id"],
    inputs={"qv": query},
    filters={"AND": [{"field": "modality", "operator": "eq", "value": "image"}]},
    )
    print(len(images_only["documents"]), "image results")

    Feature Extractors

    Retriever Stages

    feature search

    Search and filter documents by vector similarity using feature embeddings

    filter