"""Document renderer — block composition → HTML → PDF (Phase L1-L3).
Owns the generic rendering for all block types. Modules contribute data and
metadata (placeholders, block descriptors) but never markup — the renderer
turns every block into HTML itself, which keeps the PDF surface sandboxed
(WeasyPrint URL fetcher allows data: URIs only).
Pipeline:
blocks + letterhead config + data
→ ``collect_placeholder_defaults`` fills missing data keys with the
module's example values (editor preview without entity)
→ ``render_blocks_html`` renders each block (Jinja2 for text content,
escaped; shapes/dividers as styled divs; images as data:-URI img)
→ ``render_document_html`` wraps content in the letterhead page frame
(@page geometry + running header/footer elements)
→ ``generate_pdf`` (pdf_generator) → bytes
"""
from __future__ import annotations
import base64
import html as _html
import re
import uuid
from typing import Any
from jinja2.sandbox import SandboxedEnvironment
# Page sizes in mm (CSS @page)
PAGE_SIZES = {
"A4": "210mm 297mm",
"A5": "148mm 210mm",
"letter": "8.5in 11in",
}
_PLACEHOLDER_RE = re.compile(r"\{\{\s*([a-zA-Z_][a-zA-Z0-9_.]*)\s*\}\}")
def _jinja_env() -> SandboxedEnvironment:
env = SandboxedEnvironment(autoescape=True, trim_blocks=True, lstrip_blocks=True)
return env
# ─── Placeholder defaults ───────────────────────────────────────────────────
def collect_placeholder_defaults(entity_type: str | None) -> dict[str, Any]:
"""Collect placeholder example values for an entity type.
Aggregates ``document_placeholders(entity_type)`` contributions from
all plugin contracts. Returns ``{key: example}`` for the editor preview
(rendering without live entity data must not raise).
"""
if not entity_type:
return {}
defaults: dict[str, Any] = {}
from app.plugins.builtins.contracts import get_contract_registry
from app.plugins.registry import get_registry
for plugin_name in get_registry().list_discovered():
contract = get_contract_registry().get_contract(plugin_name)
fn = getattr(contract, "document_placeholders", None)
if fn is None:
continue
try:
placeholders = fn(entity_type) or []
except Exception: # noqa: BLE001
continue
for p in placeholders:
if isinstance(p, dict) and p.get("key"):
defaults[p["key"]] = p.get("example", "")
return defaults
def merge_placeholder_defaults(data: dict | None, entity_type: str | None) -> dict[str, Any]:
"""Overlay missing keys with placeholder examples (preview-safe data)."""
merged: dict[str, Any] = dict(data or {})
for key, example in collect_placeholder_defaults(entity_type).items():
if key not in merged or merged[key] in (None, ""):
merged[key] = example
return merged
# ─── Block → HTML ───────────────────────────────────────────────────────────
def _style_attr(style: dict | None) -> str:
"""Convert a small style dict into an inline style attribute."""
if not isinstance(style, dict):
return ""
allowed = {
"fontSize": "font-size",
"font-size": "font-size",
"color": "color",
"textAlign": "text-align",
"text-align": "text-align",
}
parts = []
if style.get("bold"):
parts.append("font-weight: bold")
if style.get("italic"):
parts.append("font-style: italic")
for k, v in style.items():
css = allowed.get(k)
if css and isinstance(v, (str, int, float)):
parts.append(f"{css}: {_html.escape(str(v))}")
return f' style="{"; ".join(parts)}"' if parts else ""
def _render_text(content: str, data: dict[str, Any]) -> str:
"""Render Jinja2 placeholders inside a text block (autoescaped)."""
try:
template = _jinja_env().from_string(content)
return template.render(**data)
except Exception: # noqa: BLE001 — a broken expression renders literally
return _html.escape(content)
def _img_url(config: dict, assets_map: dict[str, str]) -> str | None:
"""Resolve an image block to a data:-URI (sandbox policy for WeasyPrint)."""
url = config.get("url")
if isinstance(url, str) and url.startswith("data:"):
return url
asset_id = config.get("asset_id")
if asset_id:
data_url = assets_map.get(str(asset_id))
if data_url:
return data_url
return None
def render_block_html(block: dict, data: dict[str, Any], assets_map: dict[str, str] | None = None) -> str:
"""Render one block dict to HTML. Unknown types render nothing."""
assets_map = assets_map or {}
btype = block.get("type")
config = block.get("config") or {}
if btype == "text":
rendered = _render_text(str(config.get("content", "")), data)
return f'
{rendered}
'
if btype == "image":
url = _img_url(config, assets_map)
if not url:
return ''
dims = ""
if isinstance(config.get("width"), (int, float)):
dims += f' width="{int(config["width"])}"'
if isinstance(config.get("height"), (int, float)):
dims += f' height="{int(config["height"])}"'
alt = _html.escape(str(config.get("alt", "")))
align = config.get("align", "left")
return f''
if btype == "shape":
shape = config.get("shape")
color = _html.escape(str(config.get("color", "#111827")))
background = _html.escape(str(config.get("background", "#e5e7eb")))
width = config.get("width", "100%")
height = int(config.get("height") or 2)
radius = int(config.get("radius") or 50)
if shape == "line":
return (f'
')
if shape == "rect":
return (f'')
if shape == "circle":
size = height if height > 4 else 40
return (f'')
return ""
if btype == "divider":
color = _html.escape(str(config.get("color", "#d1d5db")))
thickness = int(config.get("thickness") or 1)
return f'
'
if btype == "spacer":
height = int(config.get("height") or 24)
return f''
if btype == "table":
columns = config.get("columns") or []
rows = config.get("rows") or []
striped = " doc-table-striped" if config.get("striped") else ""
head = ""
if columns:
head = "" + "".join(f"| {_html.escape(str(c))} | " for c in columns) + "
"
body_rows = []
for row in rows:
if not isinstance(row, (list, tuple)):
row = [row]
cells = "".join(f"{_render_text(str(c), data) if isinstance(c, str) else _html.escape(str(c))} | " for c in row)
body_rows.append(f"{cells}
")
body = "" + "".join(body_rows) + "" if body_rows else ""
width_style = f' style="width: {_html.escape(str(config["width"]))}"' if config.get("width") else ""
return f''
if btype == "placeholder":
key = str(config.get("key", ""))
label = config.get("label") or key
value = data.get(key, "")
return (f''
f'{_html.escape(str(label))}: '
f'{_html.escape(str(value if value is not None else ""))}
')
if btype == "pagebreak":
return ''
# Module-contributed block: generic key-value table over declared fields
from app.plugins.builtins.report_generator.document_blocks import contribution_fields
fields = contribution_fields(btype) if btype else None
if fields:
rows = "".join(
f"| {_html.escape(str(f))} | {_html.escape(str(data.get(f, '')))} |
"
for f in fields
)
return f''
return ""
def render_blocks_html(blocks: list[dict], data: dict[str, Any], assets_map: dict[str, str] | None = None) -> str:
"""Render a block list to a HTML fragment."""
return "\n".join(render_block_html(b, data, assets_map) for b in blocks if isinstance(b, dict))
# ─── Letterhead frame ───────────────────────────────────────────────────────
def _esc(value: Any) -> str:
return _html.escape(str(value))
def render_document_html(
blocks: list[dict],
data: dict[str, Any],
letterhead_config: dict | None = None,
assets_map: dict[str, str] | None = None,
) -> str:
"""Wrap rendered blocks in the letterhead page frame (full HTML doc)."""
config = letterhead_config or {}
page = config.get("page") or {}
size = page.get("size", "A4")
orientation = page.get("orientation", "portrait")
margins = page.get("margins") or {}
m_top = margins.get("top", 25)
m_right = margins.get("right", 20)
m_bottom = margins.get("bottom", 25)
m_left = margins.get("left", 20)
page_css = PAGE_SIZES.get(size, PAGE_SIZES["A4"])
if orientation == "landscape":
# swap width/height for landscape
w, h = page_css.split()
page_css = f"{h} {w}"
header = config.get("header") or {}
footer = config.get("footer") or {}
header_html = ""
footer_html = ""
extra_top = 0
extra_bottom = 0
if header.get("enabled"):
header_html = render_blocks_html(header.get("blocks") or [], data, assets_map)
extra_top = 20 # reserve space for the running header
if footer.get("enabled"):
footer_html = render_blocks_html(footer.get("blocks") or [], data, assets_map)
extra_bottom = 18
watermark = config.get("watermark") or {}
watermark_html = ""
if watermark.get("enabled"):
text = _esc(watermark.get("text", ""))
watermark_html = (
f'{text}
'
)
content = render_blocks_html(blocks, data, assets_map)
header_css = ""
if header_html:
header_css = (
"#doc-header { position: running(header); }\n"
"@page { @top-center { content: element(header); } }\n"
)
footer_css = ""
if footer_html:
footer_css = (
"#doc-footer { position: running(footer); }\n"
"@page { @bottom-center { content: element(footer); } }\n"
)
return f"""
{f'' if header_html else ''}
{f'' if footer_html else ''}
{watermark_html}
{content}
"""
# ─── Assets ────────────────────────────────────────────────────────────────
async def load_assets_data_urls(
db,
tenant_id: uuid.UUID,
asset_ids: list[str] | None = None,
letterhead_id: str | None = None,
) -> dict[str, str]:
"""Load DocumentAssets and return ``{asset_id: data_url}``.
Images are embedded as data:-URIs because the WeasyPrint URL fetcher
blocks external resources (SSRF policy). Missing assets are skipped.
"""
from sqlalchemy import select
from app.plugins.builtins.report_generator.models import DocumentAsset
if not asset_ids and not letterhead_id:
return {}
q = select(DocumentAsset).where(
DocumentAsset.tenant_id == tenant_id,
DocumentAsset.deleted_at.is_(None),
)
if asset_ids:
try:
ids = [uuid.UUID(a) for a in asset_ids if a]
except (ValueError, TypeError):
ids = []
if not ids:
return {}
q = q.where(DocumentAsset.id.in_(ids))
elif letterhead_id:
try:
lh = uuid.UUID(letterhead_id)
except (ValueError, TypeError):
return {}
q = q.where(DocumentAsset.letterhead_id == lh)
from app.core.storage import get_storage_backend
assets = (await db.execute(q)).scalars().all()
storage = get_storage_backend()
result: dict[str, str] = {}
for asset in assets:
try:
content = await storage.read(asset.storage_path)
except Exception: # noqa: BLE001 — missing blob renders as empty
continue
b64 = base64.b64encode(content).decode("ascii")
result[str(asset.id)] = f"data:{asset.mime_type};base64,{b64}"
return result
def collect_block_asset_ids(blocks: list[dict]) -> list[str]:
"""Extract asset_id references from image blocks."""
ids: list[str] = []
for b in blocks or []:
if not isinstance(b, dict) or b.get("type") != "image":
continue
asset_id = (b.get("config") or {}).get("asset_id")
if isinstance(asset_id, str) and asset_id:
ids.append(asset_id)
return ids