NEWVectors or files. Pick a path.Start →
    Enhanced

    Legal Document RAG Pipeline

    Purpose-built RAG pipeline for legal documents. High-precision retrieval with strong keyword matching for legal terminology, citations, and clause references.

    text
    Production
    from mixpeek import Mixpeek
    client = Mixpeek(api_key="YOUR_API_KEY", namespace="legal-docs")
    # 1. A bucket for contracts, and a collection that splits pages into layout blocks
    bucket = client.buckets.create(
    bucket_name="contracts",
    bucket_schema={"properties": {"contract": {"type": "pdf"}}},
    )
    collection = client.collections.create(
    collection_name="contracts",
    source={"type": "bucket", "bucket_ids": [bucket["bucket_id"]]},
    feature_extractor={
    "feature_extractor_name": "document_graph_extractor",
    "version": "v1",
    },
    )
    # 2. Upload and process
    client.buckets.upload(
    bucket["bucket_id"],
    blobs=[{"property": "contract", "type": "pdf", "data": "s3://your-bucket/contracts/msa-acme.pdf"}],
    )
    client.collections.trigger(collection["collection_id"])
    # 3. Dense and BM25 search fused with RRF (BM25 reads the namespace text payload
    # indexes), then a cross-encoder for precision
    retriever = client.retrievers.create(
    retriever_name="legal-search",
    collection_identifiers=["contracts"],
    input_schema={"query": {"type": "text", "required": True}},
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "mixpeek://document_graph_extractor@v1/intfloat__multilingual_e5_large_instruct",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 50,
    },
    {
    "feature_uri": "mixpeek://document_graph_extractor@v1/intfloat__multilingual_e5_large_instruct",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 50,
    "lexical": True,
    },
    ],
    "fusion": "rrf",
    "final_top_k": 50,
    },
    },
    {
    "stage_name": "rerank",
    "stage_id": "rerank",
    "parameters": {
    "inference_name": "BAAI__bge_reranker_v2_m3",
    "query": "{{INPUT.query}}",
    "document_field": "text_raw",
    "top_k": 10,
    },
    },
    ],
    )
    results = client.retrievers.execute(retriever["retriever_id"], inputs={"query": "Section 14.2 indemnification obligations"})
    for doc in results["documents"]:
    print(doc["page_number"], doc["text_raw"][:120], doc["score"])

    Feature Extractors

    Document Graph Extractor

    Decompose PDFs into spatial blocks (paragraphs, tables, forms, headers) with layout classification and E5 text embeddings.

    Retriever Stages

    feature search

    Search and filter documents by vector similarity using feature embeddings

    filter

    rerank

    Rerank documents using cross-encoder models for accurate relevance

    sort