Implement Datalab-compatible API with async endpoints and deepseek OCR backend
- New endpoints: /api/v1/convert, /api/v1/extract, /api/v1/segment, /api/v1/ocr, /api/v1/table_rec, /api/v1/marker (deprecated), /api/v1/create-document, /api/v1/files/*, /api/v1/thumbnails - Async submit-and-poll pattern matching Datalab API spec - X-API-Key authentication via API_KEY env var - Filesystem-based request store for multi-worker support - OCR_BACKEND=deepseek mode: uses deepseek-ocr via ollama for OCR (prompt: <|grounding|>Free OCR.) - build_options() now propagates output_format to parsed_opts - Fixed docs page CSS curly brace conflict with .format() - Renamed marker_endpoint_referece.md -> marker_endpoint_reference.md - Containerfile: added poppler-utils, pandoc; 3 gunicorn workers - gunicorn.conf: post_fork hook for background worker threads
This commit is contained in:
+106
@@ -0,0 +1,106 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import io
|
||||
import os
|
||||
import tempfile
|
||||
import subprocess
|
||||
import requests
|
||||
from PIL import Image
|
||||
|
||||
OLLAMA_HOST = os.environ.get("DEEPSEEK_OLLAMA_HOST", "http://localhost:11434")
|
||||
DEEPSEEK_MODEL = os.environ.get("DEEPSEEK_OCR_MODEL", "deepseek-ocr:latest")
|
||||
DEEPSEEK_PROMPT = os.environ.get("DEEPSEEK_OCR_PROMPT", "<|grounding|>Free OCR.")
|
||||
|
||||
|
||||
def _page_to_image(raw_bytes: bytes, page_num: int = 0) -> bytes | None:
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
result = subprocess.run(
|
||||
[
|
||||
"pdftoppm", "-png",
|
||||
"-f", str(page_num + 1),
|
||||
"-l", str(page_num + 1),
|
||||
"-",
|
||||
os.path.join(tmpdir, "page"),
|
||||
],
|
||||
input=raw_bytes,
|
||||
capture_output=True,
|
||||
timeout=120,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
return None
|
||||
out_files = os.listdir(tmpdir)
|
||||
if not out_files:
|
||||
return None
|
||||
img_path = os.path.join(tmpdir, out_files[0])
|
||||
img = Image.open(img_path)
|
||||
buf = io.BytesIO()
|
||||
img.save(buf, format="PNG")
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def _image_to_text(image_bytes: bytes) -> str:
|
||||
b64 = base64.b64encode(image_bytes).decode()
|
||||
resp = requests.post(
|
||||
f"{OLLAMA_HOST}/api/generate",
|
||||
json={
|
||||
"model": DEEPSEEK_MODEL,
|
||||
"prompt": DEEPSEEK_PROMPT,
|
||||
"images": [b64],
|
||||
"stream": False,
|
||||
"options": {"temperature": 0},
|
||||
},
|
||||
timeout=120,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
return resp.json().get("response", "")
|
||||
|
||||
|
||||
def ocr_pdf(raw_bytes: bytes, max_pages: int | None = None, page_range: str | None = None) -> dict:
|
||||
result = {
|
||||
"success": True,
|
||||
"text": "",
|
||||
"pages": [],
|
||||
"page_count": 0,
|
||||
}
|
||||
pages_to_process = _parse_page_range(page_range, max_pages, raw_bytes)
|
||||
for page_num in pages_to_process:
|
||||
img_bytes = _page_to_image(raw_bytes, page_num)
|
||||
if img_bytes is None:
|
||||
continue
|
||||
text = _image_to_text(img_bytes)
|
||||
result["pages"].append({"page": page_num, "text": text})
|
||||
result["text"] += f"\n\n--- PAGE {page_num + 1} ---\n\n{text}"
|
||||
result["page_count"] = len(result["pages"])
|
||||
return result
|
||||
|
||||
|
||||
def _parse_page_range(page_range: str | None, max_pages: int | None, raw_bytes: bytes) -> list[int]:
|
||||
import tempfile, os, subprocess
|
||||
total = 0
|
||||
with tempfile.NamedTemporaryFile(delete=False, suffix=".pdf") as f:
|
||||
f.write(raw_bytes)
|
||||
f.flush()
|
||||
r = subprocess.run(
|
||||
["pdfinfo", f.name],
|
||||
capture_output=True, text=True, timeout=30,
|
||||
)
|
||||
os.unlink(f.name)
|
||||
for line in r.stdout.splitlines():
|
||||
if line.startswith("Pages:"):
|
||||
total = int(line.split(":")[1].strip())
|
||||
break
|
||||
if page_range:
|
||||
pages = set()
|
||||
for part in page_range.split(","):
|
||||
part = part.strip()
|
||||
if "-" in part:
|
||||
a, b = part.split("-", 1)
|
||||
for p in range(int(a.strip()), int(b.strip()) + 1):
|
||||
pages.add(p)
|
||||
else:
|
||||
pages.add(int(part))
|
||||
return sorted(pages)
|
||||
if max_pages is not None:
|
||||
return list(range(min(max_pages, total)))
|
||||
return list(range(total))
|
||||
Reference in New Issue
Block a user