Docling Tools Run

Entry point that runs the basic conversion and OCR Docling examples in sequence.

run.py
from basic_examples import run_basic_examples
from ocr_example import run_ocr_example

if __name__ == "__main__":
    run_basic_examples()
    run_ocr_example()

The example imports these helper modules from the same directory:

basic_examples.py
from agno.agent import Agent
from agno.tools.docling import DoclingTools
from paths import (
    audio_video_path,
    docx_path,
    html_path,
    image_path,
    md_path,
    pdf_path,
    pptx_path,
    xlsx_path,
    xml_path,
)


def run_basic_examples() -> None:
    agent = Agent(
        tools=[DoclingTools(all=True)],
        description="You are an agent that converts documents from all Docling parsers and exports to all supported output formats.",
    )

    agent.print_response(
        "List supported Docling input parsers and active allowed parsers.",
        markdown=True,
    )

    agent.print_response(
        f"Convert to Markdown: {pdf_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to JSON and return the full JSON without summarizing: {pdf_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to YAML: {pdf_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to DocTags: {pdf_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to VTT: {pdf_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to HTML split page: {pdf_path}",
        markdown=True,
    )

    # Additional parser examples based on static resources.
    agent.print_response(
        f"Convert to Markdown: {docx_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to Markdown: {md_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to Markdown: {html_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to Markdown: {xml_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to Markdown: {xlsx_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to Markdown: {pptx_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to Markdown: {image_path}",
        markdown=True,
    )
    agent.print_response(
        f"Convert to VTT: {audio_video_path}",
        markdown=True,
    )

    # convert_string is limited by Docling to Markdown and HTML source content.
    agent.print_response(
        "Use convert_string_content to convert this markdown string to JSON: # Inline Markdown\n\nThis is a parser test.",
        markdown=True,
    )
    agent.print_response(
        "Use convert_string_content to convert this html string to Markdown: <h1>Inline HTML</h1><p>This is a parser test.</p>",
        markdown=True,
    )
ocr_example.py
from agno.agent import Agent
from agno.tools.docling import DoclingTools
from paths import pdf_path


def run_ocr_example() -> None:
    # pdf_ocr_engine accepts: auto | easyocr | tesseract | tesseract_cli | ocrmac | rapidocr
    # Some engines may require extra runtime dependencies in your environment.
    ocr_tools = DoclingTools(
        pdf_enable_ocr=True,
        pdf_ocr_engine="easyocr",
        pdf_ocr_lang=["pt", "en"],
        pdf_force_full_page_ocr=True,
        pdf_enable_table_structure=True,
        pdf_enable_picture_description=False,
        pdf_document_timeout=120.0,
    )

    ocr_agent = Agent(
        tools=[ocr_tools],
        description="You are an agent that converts PDFs using advanced OCR.",
    )

    ocr_agent.print_response(
        f"Convert to Markdown: {pdf_path}",
        markdown=True,
    )
paths.py
from pathlib import Path

repo_root = Path(__file__).resolve().parents[3]
testing_resources_path = repo_root / "cookbook/07_knowledge/testing_resources"


def get_test_resource_path(filename: str) -> str:
    return str(testing_resources_path / filename)


pdf_path = get_test_resource_path("cv_1.pdf")
docx_path = get_test_resource_path("project_proposal.docx")
md_path = get_test_resource_path("coffee.md")
html_path = get_test_resource_path("company_info.html")
xml_path = get_test_resource_path("patent_sample.xml")
xlsx_path = get_test_resource_path("sample_products.xlsx")
pptx_path = get_test_resource_path("ai_presentation.pptx")
image_path = get_test_resource_path("restaurant_invoice.png")
audio_video_path = get_test_resource_path("agno_description.mp4")

Conversion environment

The shared runner executes every basic conversion and then OCR, so its setup includes both PDF/OCR and video dependencies. The first conversion may download document, OCR, or speech model weights. See Docling's audio/video setup.

The tools return exports as text to the agent; a prompt asking for full JSON does not enforce the final model response. For an exact export, call the relevant DoclingTools method directly. VTT export requires an installed Docling document type with export_to_vtt; otherwise the tool returns an unsupported-export error.

Run the Example

Set up your virtual environment

uv venv --python 3.12
source .venv/bin/activate

Install dependencies

uv pip install -U agno "docling[asr]" "docling-slim[format-video]" easyocr openai

Export your OpenAI API key

export OPENAI_API_KEY="your_openai_api_key_here"

Clone Agno

Clone the pinned Agno source and run the remaining commands from its root:

git clone https://github.com/agno-agi/agno.git
cd agno
git checkout d703c34f3abf3c41275d3fb2da6e0518a8881f24

Install FFmpeg

Install the FFmpeg system package required for the MP4-to-VTT example and verify it is available:

ffmpeg -version

Run the example

Run the example from the repository root:

python cookbook/91_tools/docling_tools/run.py

Full source: cookbook/91_tools/docling_tools/run.py