NEWVectors or files. Pick a path.Start →
    QualitySimilar

    Sports Highlights Pipeline

    Automatically identify highlight-worthy moments in sports broadcasts using multimodal analysis, visual action detection, audio spike recognition (crowd noise, commentator excitement), and on-screen graphic parsing. Returns timestamped event manifests ready for clip assembly.

    video
    audio
    text
    Production

    Why This Matters

    Reduces highlight turnaround from 4-8 hours of manual editing to 15-20 minutes of automated processing. Captures 95%+ of key moments vs ~65% with human editors working under time pressure.

    import requests
    from mixpeek import Mixpeek
    client = Mixpeek(api_key="YOUR_API_KEY", namespace="sports")
    API = "https://api.mixpeek.com/v1"
    HEADERS = {"Authorization": "Bearer YOUR_API_KEY", "X-Namespace": "sports"}
    # 1. Game footage, cut at scene changes, with commentary transcripts and embeddings
    bucket = client.buckets.create(
    bucket_name="sports-footage",
    bucket_schema={"properties": {"broadcast": {"type": "video"}}},
    )
    collection = client.collections.create(
    collection_name="game-scenes",
    source={"type": "bucket", "bucket_ids": [bucket["bucket_id"]]},
    feature_extractor={
    "feature_extractor_name": "multimodal_extractor",
    "version": "v1",
    "parameters": {
    "split_method": "scene",
    "run_transcription": True,
    "run_transcription_embedding": True,
    },
    },
    )
    client.buckets.upload(
    bucket["bucket_id"],
    blobs=[{"property": "broadcast", "type": "video", "data": "s3://your-bucket/games/cl-final.mp4"}],
    )
    client.collections.trigger(collection["collection_id"])
    # 2. Event exemplars (goal, save, foul, celebration) and the taxonomy over them
    example_bucket = client.buckets.create(
    bucket_name="event-examples",
    bucket_schema={"properties": {"clip": {"type": "video"}}},
    )
    examples = client.collections.create(
    collection_name="event-examples",
    source={"type": "bucket", "bucket_ids": [example_bucket["bucket_id"]]},
    feature_extractor={
    "feature_extractor_name": "multimodal_extractor",
    "version": "v1",
    },
    )
    client.buckets.upload(
    example_bucket["bucket_id"],
    blobs=[{"property": "clip", "type": "video", "data": "s3://your-bucket/events/goal-01.mp4"}],
    )
    client.collections.trigger(examples["collection_id"])
    matcher = client.retrievers.create(
    retriever_name="event-matcher",
    collection_identifiers=["event-examples"],
    input_schema={"query": {"type": "text", "required": True}},
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/vertex_multimodal_embedding",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 3,
    },
    ],
    "final_top_k": 3,
    },
    },
    ],
    )
    # A flat taxonomy matches each document, through a retriever, against a
    # collection of labeled examples. The SDK has no taxonomies resource, so this is REST.
    taxonomy = requests.post(API + "/taxonomies", headers=HEADERS, json={
    "taxonomy_name": "soccer_events",
    "config": {
    "taxonomy_type": "flat",
    "retriever_id": matcher["retriever_id"],
    "input_mappings": [{"input_key": "query", "source_type": "payload", "path": "transcription"}],
    "source_collection": {"collection_id": examples["collection_id"]},
    },
    }).json()
    # 3. The highlights retriever: action and commentary search, event labels, top 20
    highlights = client.retrievers.create(
    retriever_name="soccer-highlights",
    collection_identifiers=["game-scenes"],
    input_schema={"query": {"type": "text", "required": True}},
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/vertex_multimodal_embedding",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 100,
    },
    {
    "feature_uri": "mixpeek://multimodal_extractor@v1/multilingual_e5_large_instruct_v1",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 100,
    },
    ],
    "fusion": "rrf",
    "final_top_k": 100,
    },
    },
    {"stage_name": "label", "stage_id": "taxonomy_enrich", "parameters": {"taxonomy_id": taxonomy["taxonomy_id"], "top_k": 1}},
    {"stage_name": "top", "stage_id": "limit", "parameters": {"limit": 20}},
    ],
    )
    results = client.retrievers.execute(highlights["retriever_id"], inputs={"query": "goal celebration crowd roars"})
    for i, doc in enumerate(results["documents"], 1):
    print(i, doc["start_time"], doc["end_time"], doc["score"], doc.get("thumbnail_url"))

    Feature Extractors

    Multimodal Extractor

    Unified embeddings for video, audio, image, and text: scene/silence chunking, Whisper transcription, thumbnails, and Gemini vision.

    Retriever Stages

    feature search

    Search and filter documents by vector similarity using feature embeddings

    filter

    taxonomy enrich

    Classify documents against taxonomy nodes via vector similarity

    apply

    limit

    Truncate results to a maximum count with optional offset for pagination

    reduce

    Use Cases Using This Recipe

    Advanced
    7 min

    Sports Highlights

    Auto-generate highlight reels from full-length sports footage

    24x faster

    Highlight generation time

    Who It's For

    Sports broadcasters, media companies, and content teams processing 100+ hours of live footage weekly