NEWVectors or files. Pick a path.Start →
    Training

    Video Keyframe Extraction Pipeline

    Extract representative keyframes from videos with scene detection. Generate thumbnails, visual summaries, and frame-level embeddings.

    video
    image
    Single Tier
    from mixpeek import Mixpeek
    client = Mixpeek(api_key="YOUR_API_KEY", namespace="keyframes")
    # 1. A bucket for the videos, and a collection that cuts at scene changes and
    # writes a thumbnail and a multimodal embedding for every scene
    bucket = client.buckets.create(
    bucket_name="product-videos",
    bucket_schema={"properties": {"video": {"type": "video"}}},
    )
    collection = client.collections.create(
    collection_name="product-videos",
    source={"type": "bucket", "bucket_ids": [bucket["bucket_id"]]},
    feature_extractor={
    "feature_extractor_name": "multimodal_extractor",
    "version": "v1",
    "parameters": {
    "split_method": "scene",
    "scene_detection_threshold": 0.5,
    "enable_thumbnails": True,
    },
    },
    )
    # 2. Upload and process
    client.buckets.upload(
    bucket["bucket_id"],
    blobs=[{"property": "video", "type": "video", "data": "s3://your-bucket/product-videos/unboxing.mp4"}],
    )
    client.collections.trigger(collection["collection_id"])
    # 3. Search the scenes visually
    retriever = client.retrievers.create(
    retriever_name="keyframe-search",
    collection_identifiers=["product-videos"],
    input_schema={"query": {"type": "text", "required": True}},
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/vertex_multimodal_embedding",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 20,
    },
    ],
    "final_top_k": 20,
    },
    },
    ],
    )
    results = client.retrievers.execute(retriever["retriever_id"], inputs={"query": "product being unboxed"})
    for doc in results["documents"]:
    print(doc["thumbnail_url"], doc["start_time"], doc["score"])

    Feature Extractors

    Multimodal Extractor

    Unified embeddings for video, audio, image, and text: scene/silence chunking, Whisper transcription, thumbnails, and Gemini vision.

    Retriever Stages

    feature search

    Search and filter documents by vector similarity using feature embeddings

    filter