feat(T03): OCR-Erfassung via OpenRouter Qwen2.5-VL + OCR UI with drag-and-drop

This commit is contained in:
2026-07-17 00:45:14 +02:00
parent b128ea6b39
commit 149ac04dc1
20 changed files with 2348 additions and 182 deletions
+187
View File
@@ -0,0 +1,187 @@
"""OpenRouter API client for Qwen2.5-VL vision model OCR processing.
Sends image to the vision model with a structured prompt and parses the
returned JSON containing brand, model, vin, first_registration, mileage,
power_kw, fuel_type plus a confidence score.
"""
from __future__ import annotations
import base64
import json
import logging
from typing import Any
import httpx
from app.config import settings
logger = logging.getLogger(__name__)
OCR_SYSTEM_PROMPT = (
"You are an expert OCR system specialized in reading German vehicle "
"registration documents (Zulassungsbescheinigung Teil I and II). "
"Extract the following fields from the provided image and return them "
"as a JSON object. If a field is not readable or not present, use null. "
"Fields to extract: brand, model, vin, first_registration (DD.MM.YYYY), "
"mileage (integer km), power_kw (integer), fuel_type. "
"Also provide a confidence_score between 0.0 and 1.0 reflecting how "
"confident you are in the extracted data. Return ONLY valid JSON, "
"no markdown, no explanation."
)
EXPECTED_FIELDS = {
"brand",
"model",
"vin",
"first_registration",
"mileage",
"power_kw",
"fuel_type",
}
def _encode_image(image_bytes: bytes, mime_type: str = "image/png") -> str:
"""Encode image bytes to a base64 data URI."""
b64 = base64.b64encode(image_bytes).decode("utf-8")
return f"data:{mime_type};base64,{b64}"
def _build_messages(image_data_uri: str) -> list[dict[str, Any]]:
"""Build the chat messages for the OpenRouter vision API."""
return [
{
"role": "system",
"content": OCR_SYSTEM_PROMPT,
},
{
"role": "user",
"content": [
{
"type": "text",
"text": (
"Please extract the vehicle data from this "
"registration document image and return as JSON."
),
},
{
"type": "image_url",
"image_url": {"url": image_data_uri},
},
],
},
]
def _parse_response(raw_content: str) -> dict[str, Any]:
"""Parse the model response into structured data + confidence.
Handles markdown code fences and extracts the JSON object.
"""
text = raw_content.strip()
# Strip markdown code fences if present
if text.startswith("```"):
lines = text.split("\n")
# Remove first line (```json or ```) and last line (```)
lines = [l for l in lines if not l.strip().startswith("```")]
text = "\n".join(lines).strip()
try:
data = json.loads(text)
except json.JSONDecodeError:
# Try to find JSON object within the text
start = text.find("{")
end = text.rfind("}")
if start != -1 and end != -1:
try:
data = json.loads(text[start : end + 1])
except json.JSONDecodeError:
logger.error("Failed to parse OpenRouter response: %s", text[:200])
return {"structured_data": {}, "confidence_score": 0.0, "raw_text": raw_content}
else:
logger.error("No JSON found in OpenRouter response: %s", text[:200])
return {"structured_data": {}, "confidence_score": 0.0, "raw_text": raw_content}
# Extract confidence score (may be inside or outside the data)
confidence = data.pop("confidence_score", None)
if confidence is None:
confidence = data.pop("confidence", 0.5)
try:
confidence_float = float(confidence)
except (TypeError, ValueError):
confidence_float = 0.5
# Clamp to 0.0-1.0
confidence_float = max(0.0, min(1.0, confidence_float))
# Ensure all expected fields exist (default None)
structured: dict[str, Any] = {}
for field in EXPECTED_FIELDS:
structured[field] = data.get(field)
# Convert mileage and power_kw to int if present
if structured.get("mileage") is not None:
try:
structured["mileage"] = int(structured["mileage"])
except (TypeError, ValueError):
pass
if structured.get("power_kw") is not None:
try:
structured["power_kw"] = int(structured["power_kw"])
except (TypeError, ValueError):
pass
return {
"structured_data": structured,
"confidence_score": confidence_float,
"raw_text": raw_content,
}
async def perform_ocr(
image_bytes: bytes,
mime_type: str = "image/png",
api_key: str | None = None,
model: str | None = None,
) -> dict[str, Any]:
"""Send image to OpenRouter Qwen2.5-VL and return parsed OCR result.
Returns dict with keys:
- structured_data: dict with brand, model, vin, etc.
- confidence_score: float 0.0-1.0
- raw_text: str (raw model response)
Raises httpx.HTTPStatusError on API failure.
"""
key = api_key or settings.OPENROUTER_API_KEY
if not key:
raise ValueError("OPENROUTER_API_KEY is not configured")
model_name = model or settings.OPENROUTER_OCR_MODEL
image_data_uri = _encode_image(image_bytes, mime_type)
messages = _build_messages(image_data_uri)
headers = {
"Authorization": f"Bearer {key}",
"Content-Type": "application/json",
}
payload: dict[str, Any] = {
"model": model_name,
"messages": messages,
"temperature": 0.1,
"max_tokens": 1024,
}
base_url = settings.OPENROUTER_BASE_URL.rstrip("/")
url = f"{base_url}/chat/completions"
async with httpx.AsyncClient(timeout=httpx.Timeout(60.0)) as client:
response = await client.post(url, headers=headers, json=payload)
response.raise_for_status()
body = response.json()
content = body.get("choices", [{}])[0].get("message", {}).get("content", "")
return _parse_response(content)