NEWVectors or files. Pick a path.Start →

    Document RAG Pipeline

    Retrieval-augmented generation for document collections. Extracts text, tables, and figures from PDFs using OCR and layout analysis, then retrieves relevant page sections to answer natural language questions with precise page and section citations.

    text
    image
    Multi-Stage
    from mixpeek import Mixpeek
    client = Mixpeek(api_key="YOUR_API_KEY", namespace="policy-docs")
    # 1. A bucket for the PDFs, and a collection that splits every page into layout blocks
    bucket = client.buckets.create(
    bucket_name="policies",
    bucket_schema={"properties": {"policy": {"type": "pdf"}}},
    )
    collection = client.collections.create(
    collection_name="policies",
    source={"type": "bucket", "bucket_ids": [bucket["bucket_id"]]},
    feature_extractor={
    "feature_extractor_name": "document_graph_extractor",
    "version": "v1",
    },
    )
    client.buckets.upload(
    bucket["bucket_id"],
    blobs=[{"property": "policy", "type": "pdf", "data": "s3://your-bucket/policies/enterprise-terms.pdf"}],
    )
    client.collections.trigger(collection["collection_id"])
    # 2. The retriever finds blocks, reranks them and writes the answer. BM25 reads the
    # namespace's text payload indexes, so declare one on text_raw before relying on it.
    retriever = client.retrievers.create(
    retriever_name="policy-rag",
    collection_identifiers=["policies"],
    input_schema={"query": {"type": "text", "required": True}},
    stages=[
    {
    "stage_name": "search",
    "stage_id": "feature_search",
    "parameters": {
    "searches": [
    {
    "feature_uri": "mixpeek://document_graph_extractor@v1/intfloat__multilingual_e5_large_instruct",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 50,
    },
    {
    "feature_uri": "mixpeek://document_graph_extractor@v1/intfloat__multilingual_e5_large_instruct",
    "query": {"input_mode": "text", "value": "{{INPUT.query}}"},
    "top_k": 50,
    "lexical": True,
    },
    ],
    "fusion": "rrf",
    "final_top_k": 50,
    },
    },
    {
    "stage_name": "rerank",
    "stage_id": "rerank",
    "parameters": {
    "inference_name": "BAAI__bge_reranker_v2_m3",
    "query": "{{INPUT.query}}",
    "document_field": "text_raw",
    "top_k": 8,
    },
    },
    {
    "stage_name": "answer",
    "stage_id": "summarize",
    "parameters": {
    "prompt": "Answer the question {{INPUT.query}} using only these numbered passages, and cite the passage numbers. {{DOCUMENTS}}",
    "provider": "google",
    "model_name": "gemini-2.5-flash-lite",
    "content_field": "text_raw",
    "output_field": "answer",
    "include_sources": True,
    },
    },
    ],
    )
    results = client.retrievers.execute(retriever["retriever_id"], inputs={"query": "What is the refund policy for enterprise customers?"})
    answer = results["documents"][0]
    print(answer["answer"])
    # The summary document lists the documents it read; fetch them for citations
    for i, document_id in enumerate(answer.get("source_document_ids") or [], 1):
    source = client.documents.get(collection["collection_id"], document_id)
    print(f"[{i}]", source.get("root_object_id"), source.get("page_number"))

    Feature Extractors

    Document Graph Extractor

    Decompose PDFs into spatial blocks (paragraphs, tables, forms, headers) with layout classification and E5 text embeddings.

    Retriever Stages

    feature search

    Search and filter documents by vector similarity using feature embeddings

    filter

    rerank

    Rerank documents using cross-encoder models for accurate relevance

    sort

    summarize

    Condense multiple documents into a summary using an LLM

    reduce