NEWVectors or files. Pick a path.Start →

    Searchable Video Library

    Turn an unstructured video archive into a fully searchable library. Each video is decomposed into scenes with transcriptions, visual embeddings, and metadata. Users search by natural language and jump directly to the relevant moment in any video.

    video
    text
    audio
    Multi-Tier
    from mixpeek import Mixpeek
    client = Mixpeek(api_key="YOUR_API_KEY", namespace="video-library")
    # 1. A bucket for the archive, and a collection that splits each video into scenes with transcripts and embeddings
    bucket = client.buckets.create(
    bucket_name="video-archive",
    bucket_schema={
    "properties": {
    "video": {
    "type": "video",
    },
    },
    },
    )
    collection = client.collections.create(
    collection_name="video_library",
    source={"type": "bucket", "bucket_ids": [bucket["bucket_id"]]},
    feature_extractor={
    "feature_extractor_name": "multimodal_extractor",
    "version": "v1",
    "parameters": {
    "split_method": "scene",
    "run_transcription": True,
    "run_transcription_embedding": True,
    },
    },
    )
    # 2. Upload and process
    client.buckets.upload(
    bucket["bucket_id"],
    blobs=[{"property": "video", "type": "video", "data": "s3://your-bucket/video-archive/all-hands-2026-06.mp4"}],
    )
    client.collections.trigger(collection["collection_id"])
    # 3. Visual and transcript search fused with RRF, then a rerank against each scene transcript
    retriever = client.retrievers.create(
    retriever_name="video-library-search",
    collection_identifiers=["video_library"],
    input_schema={
    "query": {
    "type": "text",
    "required": True,
    },
    },
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/vertex_multimodal_embedding",
    "query": {
    "input_mode": "text",
    "value": "{{INPUT.query}}",
    },
    "top_k": 50,
    },
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/multilingual_e5_large_instruct_v1",
    "query": {
    "input_mode": "text",
    "value": "{{INPUT.query}}",
    },
    "top_k": 50,
    },
    ],
    "fusion": "rrf",
    "final_top_k": 50,
    },
    },
    {
    "stage_name": "rerank",
    "stage_id": "rerank",
    "parameters": {
    "inference_name": "BAAI__bge_reranker_v2_m3",
    "query": "{{INPUT.query}}",
    "document_field": "transcription",
    "top_k": 10,
    },
    },
    ],
    )
    # 4. Search
    results = client.retrievers.execute(
    retriever["retriever_id"],
    inputs={
    "query": "product roadmap presentation Q3 goals",
    },
    )
    for doc in results["documents"]:
    print(doc["source_object_id"], doc["start_time"], doc["end_time"], doc["score"])

    Feature Extractors

    Multimodal Extractor

    Unified embeddings for video, audio, image, and text: scene/silence chunking, Whisper transcription, thumbnails, and Gemini vision.

    Retriever Stages

    feature search

    Search and filter documents by vector similarity using feature embeddings

    filter

    rerank

    Rerank documents using cross-encoder models for accurate relevance

    sort