NEWVectors or files. Pick a path.Start →
    Similar

    Video Semantic Search Pipeline

    Build a production-ready video search engine that lets users find specific moments across thousands of hours of video using natural language queries.

    video
    text
    Multi-Tier
    from mixpeek import Mixpeek
    client = Mixpeek(api_key="YOUR_API_KEY", namespace="video-search")
    # 1. A bucket for the videos, and one collection that splits them into scenes with a visual and a transcript embedding each
    bucket = client.buckets.create(
    bucket_name="videos",
    bucket_schema={
    "properties": {
    "video": {
    "type": "video",
    },
    },
    },
    )
    collection = client.collections.create(
    collection_name="videos",
    source={"type": "bucket", "bucket_ids": [bucket["bucket_id"]]},
    feature_extractor={
    "feature_extractor_name": "multimodal_extractor",
    "version": "v1",
    "parameters": {
    "split_method": "scene",
    "run_transcription": True,
    "run_transcription_embedding": True,
    },
    },
    )
    # 2. Upload and process
    client.buckets.upload(
    bucket["bucket_id"],
    blobs=[{"property": "video", "type": "video", "data": "s3://your-bucket/videos/keynote.mp4"}],
    )
    client.collections.trigger(collection["collection_id"])
    # 3. Visual and transcript search fused with RRF, then a cross-encoder rerank
    retriever = client.retrievers.create(
    retriever_name="video-search",
    collection_identifiers=["videos"],
    input_schema={
    "query": {
    "type": "text",
    "required": True,
    },
    },
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/vertex_multimodal_embedding",
    "query": {
    "input_mode": "text",
    "value": "{{INPUT.query}}",
    },
    "top_k": 50,
    },
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/multilingual_e5_large_instruct_v1",
    "query": {
    "input_mode": "text",
    "value": "{{INPUT.query}}",
    },
    "top_k": 50,
    },
    ],
    "fusion": "rrf",
    "final_top_k": 50,
    },
    },
    {
    "stage_name": "rerank",
    "stage_id": "rerank",
    "parameters": {
    "inference_name": "BAAI__bge_reranker_v2_m3",
    "query": "{{INPUT.query}}",
    "document_field": "transcription",
    "top_k": 10,
    },
    },
    ],
    )
    # 4. Search
    results = client.retrievers.execute(
    retriever["retriever_id"],
    inputs={
    "query": "person explaining machine learning concepts",
    },
    )
    for doc in results["documents"]:
    print(doc["start_time"], doc["end_time"], doc["score"])

    Feature Extractors

    Multimodal Extractor

    Unified embeddings for video, audio, image, and text: scene/silence chunking, Whisper transcription, thumbnails, and Gemini vision.

    Retriever Stages

    feature search

    Search and filter documents by vector similarity using feature embeddings

    filter

    rerank

    Rerank documents using cross-encoder models for accurate relevance

    sort