Skip to content

teff.tool.builtin.pdf

teff.tool.builtin.pdf

PDF read tool — extract text from a PDF file.

Classes:

Name Description
PDFReadTool

Extract text from a PDF file.

PDFReadTool

Bases: Tool

Extract text from a PDF file.

Requires pypdf (from teff[tools]). Returns the text of each page, optionally limited to max_chars characters.

Parameters:

Name Type Description Default
config dict | None

Optional dict. Currently unused, kept for config parity.

None
Source code in teff/tool/builtin/pdf.py
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
class PDFReadTool(Tool):
    """Extract text from a PDF file.

    Requires ``pypdf`` (from ``teff[tools]``). Returns the text of each
    page, optionally limited to *max_chars* characters.

    Args:
        config: Optional dict. Currently unused, kept for config parity.
    """

    name = "read_pdf"
    description = "Extract text from a PDF file"

    def __init__(self, config: dict | None = None):
        pass

    def run(self, path: str = "", max_chars: int = 50000) -> str:  # type: ignore[override]
        if not path:
            raise ValueError("path is required")
        try:
            from pypdf import PdfReader
        except ImportError as e:
            msg = "read_pdf requires 'pypdf' (pip install teff[tools])"
            raise ImportError(msg) from e

        reader = PdfReader(path)
        parts: list[str] = []
        for i, page in enumerate(reader.pages, 1):
            text = page.extract_text() or ""
            parts.append(f"--- page {i} ---\n{text}")
        result = "\n".join(parts).strip()
        if not result:
            return "no text found in pdf"
        return result[:max_chars]