Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 18 additions & 1 deletion Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -76,8 +76,25 @@ ENV PYTHONUNBUFFERED=1 \
UV_PROJECT_ENVIRONMENT=/app/.venv \
PATH="/app/.venv/bin:${PATH}"

# System libs required by the binary normalizer's image adapters:
# * ``libheif1`` -- pillow-heif (HEIC / HEIF / AVIF)
# * ``libcairo2`` / ``libpango*`` / ``libgdk-pixbuf-2.0-0`` -- cairosvg (SVG)
#
# Office conversion (DOCX/XLSX/PPTX/RTF/HTML) goes through the Gotenberg
# sidecar by default (``FLYDESK_IDP_OFFICE_CONVERTER=gotenberg``), so
# ``soffice`` is intentionally NOT installed here -- it would bloat the
# image by ~700MB and is not needed when running against the canonical
# compose stack. Operators who want the in-container subprocess path
# (``FLYDESK_IDP_OFFICE_CONVERTER=libreoffice``) extend this Dockerfile
# with ``libreoffice-core`` + ``fonts-noto-cjk`` etc. on their side.
RUN apt-get update \
&& apt-get install -y --no-install-recommends curl \
&& apt-get install -y --no-install-recommends \
curl \
libheif1 \
libcairo2 \
libpango-1.0-0 \
libpangocairo-1.0-0 \
libgdk-pixbuf-2.0-0 \
&& rm -rf /var/lib/apt/lists/* \
&& useradd --uid 10001 --shell /usr/sbin/nologin --no-create-home idp

Expand Down
29 changes: 29 additions & 0 deletions docker-compose.yml
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,27 @@ services:
timeout: 3s
retries: 30

# Office-doc conversion sidecar. Owns headless LibreOffice + Chromium
# internally so the API + worker containers stay distroless-friendly
# (no ``soffice`` binary, no writable filesystem, no shell). The
# ``flydesk-idp`` containers POST DOCX/XLSX/PPTX/RTF/HTML bytes here
# and get back the rendered PDF -- selected by
# ``FLYDESK_IDP_OFFICE_CONVERTER=gotenberg`` (the default).
gotenberg:
image: gotenberg/gotenberg:8
container_name: flydesk-idp-gotenberg
ports:
- "${GOTENBERG_PORT:-3000}:3000"
healthcheck:
test: ["CMD", "curl", "--fail", "http://localhost:3000/health"]
interval: 5s
timeout: 3s
retries: 30
command:
- "gotenberg"
- "--api-timeout=60s"
- "--libreoffice-restart-after=10"

api:
build: *service-build
image: flydesk-idp:latest
Expand All @@ -63,6 +84,8 @@ services:
FLYDESK_IDP_REDIS_URL: redis://redis:6379/0
FLYDESK_IDP_EDA_ADAPTER: postgres
FLYDESK_IDP_MODEL: ${FLYDESK_IDP_MODEL:-anthropic:claude-sonnet-4-6}
FLYDESK_IDP_OFFICE_CONVERTER: ${FLYDESK_IDP_OFFICE_CONVERTER:-gotenberg}
FLYDESK_IDP_GOTENBERG_URL: http://gotenberg:3000
ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY:-}
OPENAI_API_KEY: ${OPENAI_API_KEY:-}
RUN_MIGRATIONS: "true"
Expand All @@ -71,6 +94,8 @@ services:
condition: service_healthy
redis:
condition: service_healthy
gotenberg:
condition: service_healthy
healthcheck:
test: ["CMD", "curl", "--fail", "http://localhost:8400/actuator/health/readiness"]
interval: 5s
Expand All @@ -90,12 +115,16 @@ services:
FLYDESK_IDP_REDIS_URL: redis://redis:6379/0
FLYDESK_IDP_EDA_ADAPTER: postgres
FLYDESK_IDP_MODEL: ${FLYDESK_IDP_MODEL:-anthropic:claude-sonnet-4-6}
FLYDESK_IDP_OFFICE_CONVERTER: ${FLYDESK_IDP_OFFICE_CONVERTER:-gotenberg}
FLYDESK_IDP_GOTENBERG_URL: http://gotenberg:3000
ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY:-}
OPENAI_API_KEY: ${OPENAI_API_KEY:-}
RUN_MIGRATIONS: "false"
depends_on:
api:
condition: service_healthy
gotenberg:
condition: service_healthy

volumes:
postgres_data:
10 changes: 8 additions & 2 deletions docs/overview.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,8 +13,14 @@ flydesk-idp turns documents into structured, validated, audit-ready
decisions. A single HTTP call does five things that operations teams
normally have to glue together themselves:

1. **Read** the document — any layout, any of the formats a multimodal
LLM accepts (PDF, PNG, JPEG, WebP, TIFF, …).
1. **Read** the document — any layout, any binary the **binary
normalizer** can resolve to LLM-renderable bytes: PDFs and provider-
native rasters (PNG, JPEG, GIF, WebP) pass straight through; Office
docs (DOCX/XLSX/PPTX/RTF/ODT/HTML) go to PDF via a Gotenberg sidecar
(or in-container LibreOffice as fallback); images the providers
don't read (HEIC/HEIF/AVIF, multi-frame TIFF, SVG, BMP) convert via
Pillow + cairosvg; archives + email bundles (ZIP/7z/TAR/EML/MSG)
fan out into multiple per-attachment requests.
2. **Extract** the fields you asked for — each one with a value, a
page number, a normalised bounding box, and a confidence score.
3. **Validate** every field with deterministic checkers (IBAN
Expand Down
27 changes: 23 additions & 4 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -19,11 +19,29 @@ dependencies = [
# over pydantic-ai. Pulls in the OpenAI / Anthropic / Bedrock providers via pydantic-ai-slim.
"fireflyframework-agentic[rest,security,redis]>=26.5.11",

# Document handling -- pypdf only for cheap page counting; the document
# bytes themselves are shipped straight to the multimodal LLM (no local
# rasterisation, no Pillow), so we stay format-agnostic (PDF, PNG, JPEG,
# WebP, TIFF, DOCX, ...).
# Document handling.
#
# ``pypdf`` -- cheap page counting + encrypted-PDF detection in the
# binary normalizer (raw bytes still ship straight to the multimodal
# LLM, no local rasterisation for the extraction path).
#
# ``Pillow`` + ``pillow-heif`` -- the binary normalizer uses Pillow
# to canonicalise raster inputs the LLM provider can't natively read
# (HEIC/HEIF iPhone photos, multi-frame TIFF fax scans, animated GIFs,
# SVG via cairosvg) into PNG or multi-page PDF.
#
# ``cairosvg`` -- vector SVG rasterisation; the providers don't accept
# SVG so we render to PNG before shipping.
#
# ``py7zr`` -- 7-Zip archive expansion (.zip and .tar are stdlib).
#
# ``extract-msg`` -- Outlook .msg parsing; .eml is stdlib ``email``.
"pypdf>=4.3.0",
"Pillow>=11.0",
"pillow-heif>=0.18",
"cairosvg>=2.7",
"py7zr>=0.22",
"extract-msg>=0.51",

# Pydantic AI models for multimodal/vision LLMs.
"pydantic-ai-slim[anthropic,openai,bedrock]>=1.56.0",
Expand Down Expand Up @@ -65,6 +83,7 @@ dev = [
"respx>=0.21",
"httpx>=0.28",
"reportlab>=4.2", # synthesise sample PDFs in fixtures
"python-docx>=1.1", # synthesise sample DOCX in fixtures
]

[build-system]
Expand Down
60 changes: 60 additions & 0 deletions src/flydesk_idp/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -76,6 +76,66 @@ class IDPSettings(BaseSettings):
webhook_max_attempts: int = 5
webhook_hmac_secret: str | None = None

# -- Binary normalization ------------------------------------------
# The binary normalizer (``core/services/binary``) turns any caller-
# supplied binary into one or more LLM-renderable inputs (PDF or
# PNG/JPG/GIF/WebP). It expands archives + email attachments,
# converts Office docs via headless LibreOffice, rasterises HEIC /
# multi-frame TIFF / SVG via Pillow + cairosvg, and rejects
# encrypted / corrupt PDFs with a typed error.
binary_normalize_enabled: bool = Field(
default=True,
description=(
"Master kill-switch for the binary normalizer. When False the "
"loader passes raw bytes through as before — useful for debugging "
"or for deployments that pre-normalise upstream."
),
)
binary_max_recursion_depth: int = Field(
default=3,
ge=0,
description=(
"Max nesting depth for archives / emails. A ZIP containing a ZIP "
"containing a PDF is depth 3. Prevents zip-bomb style recursion."
),
)
binary_max_expanded_files: int = Field(
default=50,
ge=1,
description="Hard cap on expanded files per inbound binary.",
)
office_converter: str = Field(
default="gotenberg",
description=(
"Adapter used by the binary normalizer for Office → PDF "
"conversion. ``gotenberg`` (HTTP sidecar, distroless-friendly, "
"default) or ``libreoffice`` (in-container subprocess; requires "
"``soffice`` + multilingual font packs in the runtime image)."
),
)
gotenberg_url: str = Field(
default="http://gotenberg:3000",
description=(
"Base URL of the Gotenberg sidecar. Used only when ``office_converter == 'gotenberg'``."
),
)
gotenberg_timeout_s: int = Field(
default=60,
ge=1,
description="Per-call HTTP timeout against the Gotenberg sidecar.",
)
binary_libreoffice_path: str = Field(
default="soffice",
description=(
"Path to the headless LibreOffice binary. Used only when ``office_converter == 'libreoffice'``."
),
)
binary_libreoffice_timeout_s: int = Field(
default=60,
ge=1,
description=("Per-call subprocess timeout when ``office_converter == 'libreoffice'``."),
)

# -- Security -------------------------------------------------------
api_keys: str | None = Field(
default=None,
Expand Down
58 changes: 58 additions & 0 deletions src/flydesk_idp/core/configuration.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,16 @@
VisualAuthenticityChecker,
)
from flydesk_idp.core.services.bbox import BboxValidator
from flydesk_idp.core.services.binary import (
ArchiveUnpacker,
BinaryNormalizer,
EmailUnpacker,
GotenbergConverter,
ImageNormalizer,
LibreOfficeConverter,
OfficeConverter,
PdfGuard,
)
from flydesk_idp.core.services.classification import DocumentClassifier
from flydesk_idp.core.services.escalation import JudgeEscalator
from flydesk_idp.core.services.extraction.extractor import MultimodalExtractor
Expand Down Expand Up @@ -132,6 +142,52 @@ def request_validator(self) -> RequestValidator:
"""Pre-flight semantic checker on the public ExtractionRequest."""
return RequestValidator()

# ------------------------------------------------------------------
# Binary normalization
#
# Office conversion is pluggable behind the OfficeConverter
# protocol. The default ``gotenberg`` adapter keeps the runtime
# container distroless-friendly by delegating to a Gotenberg
# sidecar; ``libreoffice`` falls back to an in-container subprocess
# for slim/dev images that bundle ``soffice``.
# ------------------------------------------------------------------

@bean
def office_converter(self, settings: IDPSettings) -> OfficeConverter:
kind = (settings.office_converter or "gotenberg").lower()
if kind == "libreoffice":
return LibreOfficeConverter(settings=settings)
if kind == "gotenberg":
return GotenbergConverter(settings=settings)
raise ValueError(
f"unknown FLYDESK_IDP_OFFICE_CONVERTER={settings.office_converter!r}; "
"expected 'gotenberg' or 'libreoffice'"
)

@bean
def binary_normalizer(
self,
settings: IDPSettings,
pdf_guard: PdfGuard,
image_normalizer: ImageNormalizer,
office_converter: OfficeConverter,
archive_unpacker: ArchiveUnpacker,
email_unpacker: EmailUnpacker,
) -> BinaryNormalizer:
return BinaryNormalizer(
settings=settings,
pdf_guard=pdf_guard,
image=image_normalizer,
office=office_converter,
archive=archive_unpacker,
email_=email_unpacker,
)

# PdfGuard / ImageNormalizer / ArchiveUnpacker / EmailUnpacker carry
# ``@service`` decorators -- pyfly autoscan picks them up via the
# ``flydesk_idp.core`` scan_packages entry. They are listed here only
# so the dependency graph is auditable from this single file.

@bean
def visual_checker(self, settings: IDPSettings, prompts: PromptCatalog) -> VisualAuthenticityChecker:
return VisualAuthenticityChecker(template=prompts.visual_authenticity, model=settings.model)
Expand Down Expand Up @@ -175,6 +231,7 @@ def orchestrator(
classifier: DocumentClassifier,
field_validator: FieldValidator,
bbox_validator: BboxValidator,
binary_normalizer: BinaryNormalizer,
visual_checker: VisualAuthenticityChecker,
content_checker: ContentAuthenticityChecker,
judge: Judge,
Expand All @@ -188,6 +245,7 @@ def orchestrator(
classifier=classifier,
field_validator=field_validator,
bbox_validator=bbox_validator,
binary_normalizer=binary_normalizer,
visual_checker=visual_checker,
content_checker=content_checker,
judge=judge,
Expand Down
61 changes: 61 additions & 0 deletions src/flydesk_idp/core/services/binary/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
# Copyright 2026 Firefly Software Solutions Inc
"""Binary normalization -- turn any caller-supplied binary into LLM-renderable bytes.

The multimodal LLM providers we ship against (Anthropic, OpenAI, Bedrock)
only natively read PDF + a small set of raster image formats. Real
callers send everything: DOCX, XLSX, PPTX, RTF, ODT, HTML, EML/MSG email
with attachments, ZIP / 7z / TAR bundles, HEIC iPhone photos, multi-frame
TIFF fax scans, SVG, encrypted PDFs.

This package normalises every inbound binary into one or more
:class:`NormalisedBinary` rows -- each carrying ready-to-ship bytes plus
the resolved media type. A single inbound ZIP can fan out to many rows;
a born-digital PDF or a clean PNG is a one-row passthrough.

The normalizer is wired through pyfly DI -- :class:`BinaryNormalizer`
is the entry point; the per-format adapters (:class:`LibreOfficeConverter`,
:class:`EmailUnpacker`, :class:`ArchiveUnpacker`, :class:`ImageNormalizer`,
:class:`PdfGuard`) are autoscanned ``@service`` beans injected into it.

Errors raise typed :class:`BinaryNormalizationError` subclasses so
:class:`ExceptionAdvice` can map them to RFC 7807 problem-details.
"""

from __future__ import annotations

from flydesk_idp.core.services.binary.archive import ArchiveUnpacker
from flydesk_idp.core.services.binary.email import EmailUnpacker
from flydesk_idp.core.services.binary.errors import (
ArchiveExtractionError,
BinaryNormalizationError,
EncryptedPdfError,
ImageConversionError,
OfficeConversionError,
UnsupportedBinaryError,
)
from flydesk_idp.core.services.binary.gotenberg import GotenbergConverter
from flydesk_idp.core.services.binary.image import ImageNormalizer
from flydesk_idp.core.services.binary.libreoffice import LibreOfficeConverter
from flydesk_idp.core.services.binary.normalizer import BinaryNormalizer, NormalisedBinary
from flydesk_idp.core.services.binary.office_converter import OfficeConverter
from flydesk_idp.core.services.binary.pdf_guard import PdfGuard
from flydesk_idp.core.services.binary.sniffer import sniff_media_type

__all__ = [
"ArchiveExtractionError",
"ArchiveUnpacker",
"BinaryNormalizationError",
"BinaryNormalizer",
"EmailUnpacker",
"EncryptedPdfError",
"GotenbergConverter",
"ImageConversionError",
"ImageNormalizer",
"LibreOfficeConverter",
"NormalisedBinary",
"OfficeConversionError",
"OfficeConverter",
"PdfGuard",
"UnsupportedBinaryError",
"sniff_media_type",
]
Loading
Loading