Initial marker-api benchmark setup: Containerfile, app.py, convert_both.py, compare_conversions.py
This commit is contained in:
@@ -0,0 +1,466 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Endpoint tests for marker-api (app.py).
|
||||
Uses Flask test_client for routing/validation tests.
|
||||
Uses requests against a running instance when available.
|
||||
Generates a timestamped HTML report.
|
||||
"""
|
||||
|
||||
import datetime
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import traceback
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
from pathlib import Path
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
TEST_FILES = HERE / "test_files"
|
||||
RESULTS = HERE / "test_results"
|
||||
ENISA_DIR = HERE.parent / "test_files" / "enisa"
|
||||
APP_PY = HERE.parent / "app.py"
|
||||
|
||||
RESULTS.mkdir(exist_ok=True)
|
||||
|
||||
TIMESTAMP = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
REPORT_PATH = HERE.parent / f"test_report_{TIMESTAMP}.md"
|
||||
|
||||
REPORT_LINES = []
|
||||
API_BASE = os.environ.get("API_BASE", "")
|
||||
|
||||
# Collect all ENISA PDFs for file-based tests
|
||||
ENISA_PDFS = sorted(ENISA_DIR.glob("*.pdf")) if ENISA_DIR.exists() else []
|
||||
# Filter to ensure they are actual PDFs (start with %PDF)
|
||||
ENISA_PDFS = [p for p in ENISA_PDFS if p.read_bytes().startswith(b"%PDF")]
|
||||
# Quick test PDF — create a tiny valid-ish PDF or use the first ENISA one
|
||||
SMALL_PDF = None
|
||||
if ENISA_PDFS:
|
||||
SMALL_PDF = ENISA_PDFS[0]
|
||||
|
||||
results = {"passed": 0, "failed": 0, "skipped": 0, "warnings": 0}
|
||||
details = []
|
||||
|
||||
|
||||
def r(text: str):
|
||||
"""Accumulate report line."""
|
||||
REPORT_LINES.append(text)
|
||||
|
||||
|
||||
def test(name: str, func):
|
||||
"""Run a test case, capture result."""
|
||||
start = time.time()
|
||||
try:
|
||||
func()
|
||||
elapsed = time.time() - start
|
||||
results["passed"] += 1
|
||||
status = "PASS"
|
||||
icon = "✅"
|
||||
details.append((name, status, elapsed, ""))
|
||||
r(f"| {name} | {icon} PASS | {elapsed:.3f}s | |")
|
||||
except AssertionError as e:
|
||||
elapsed = time.time() - start
|
||||
results["failed"] += 1
|
||||
status = "FAIL"
|
||||
icon = "❌"
|
||||
msg = str(e).replace("\n", " ")
|
||||
details.append((name, status, elapsed, msg))
|
||||
r(f"| {name} | {icon} FAIL | {elapsed:.3f}s | {msg} |")
|
||||
except Exception as e:
|
||||
elapsed = time.time() - start
|
||||
results["failed"] += 1
|
||||
status = "ERROR"
|
||||
icon = "💥"
|
||||
msg = f"{type(e).__name__}: {e}"
|
||||
details.append((name, status, elapsed, msg))
|
||||
r(f"| {name} | {icon} ERROR | {elapsed:.3f}s | {msg} |")
|
||||
|
||||
|
||||
def skip(name: str, reason: str):
|
||||
results["skipped"] += 1
|
||||
details.append((name, "SKIP", 0, reason))
|
||||
r(f"| {name} | ⏭️ SKIP | - | {reason} |")
|
||||
|
||||
|
||||
# ── Application-level tests (import + validate app.py syntax) ──
|
||||
|
||||
def test_app_py_syntax():
|
||||
"""Verify app.py is syntactically valid Python."""
|
||||
import py_compile
|
||||
try:
|
||||
py_compile.compile(str(APP_PY), doraise=True)
|
||||
except py_compile.PyCompileError as e:
|
||||
raise AssertionError(f"Syntax error in app.py: {e}")
|
||||
|
||||
|
||||
def test_app_py_has_use_llm():
|
||||
"""Verify app.py contains use_llm parameter extraction."""
|
||||
content = APP_PY.read_text()
|
||||
assert '"use_llm"' in content, "Missing use_llm parameter in app.py"
|
||||
assert '"llm_service"' in content, "Missing llm_service parameter in app.py"
|
||||
assert '"block_correction_prompt"' in content, "Missing block_correction_prompt parameter in app.py"
|
||||
assert '"redo_inline_math"' in content, "Missing redo_inline_math parameter in app.py"
|
||||
assert '"strip_existing_ocr"' in content, "Missing strip_existing_ocr parameter in app.py"
|
||||
assert '"debug"' in content, "Missing debug parameter in app.py"
|
||||
|
||||
|
||||
def test_ps1_has_new_params():
|
||||
"""Verify marker-convert.ps1 contains all new parameters."""
|
||||
ps1 = HERE.parent / "marker-convert-powershell" / "marker-convert.ps1"
|
||||
assert ps1.exists(), "marker-convert.ps1 not found"
|
||||
content = ps1.read_text()
|
||||
assert "PageRange" in content, "Missing PageRange param"
|
||||
assert "ForceOcr" in content, "Missing ForceOcr param"
|
||||
assert "DisableImageExtraction" in content, "Missing DisableImageExtraction param"
|
||||
assert "UseLlm" in content, "Missing UseLlm param"
|
||||
assert "LlmService" in content, "Missing LlmService param"
|
||||
assert "Processors" in content, "Missing Processors param"
|
||||
assert "ConfigJson" in content, "Missing ConfigJson param"
|
||||
assert "ConverterCls" in content, "Missing ConverterCls param"
|
||||
|
||||
|
||||
def test_readme_exists():
|
||||
"""Verify README.md exists and documents all endpoints."""
|
||||
readme = HERE.parent / "README.md"
|
||||
assert readme.exists(), "README.md not found"
|
||||
content = readme.read_text()
|
||||
assert "POST /marker" in content, "Missing /marker endpoint docs"
|
||||
assert "POST /v1/conversions" in content, "Missing /v1/conversions docs"
|
||||
assert "POST /v1/files/convert" in content, "Missing /v1/files/convert docs"
|
||||
assert "use_llm" in content, "Missing use_llm in docs"
|
||||
assert "llm_service" in content, "Missing llm_service in docs"
|
||||
|
||||
|
||||
# ── HTTP-level tests (require a running API) ──
|
||||
|
||||
def _api_request(method: str, path: str, **kwargs):
|
||||
"""Make an HTTP request to the API."""
|
||||
from urllib.request import Request
|
||||
import base64
|
||||
|
||||
url = f"{API_BASE}{path}"
|
||||
if method == "GET":
|
||||
req = Request(url, method="GET")
|
||||
elif method == "POST":
|
||||
data = kwargs.get("data")
|
||||
headers = kwargs.get("headers", {})
|
||||
files = kwargs.get("files")
|
||||
|
||||
if files:
|
||||
# Build multipart manually
|
||||
import io
|
||||
boundary = "----testboundary123"
|
||||
body = io.BytesIO()
|
||||
for key, (filename, filedata, _mime) in files.items():
|
||||
body.write(f"--{boundary}\r\n".encode())
|
||||
body.write(f'Content-Disposition: form-data; name="{key}"; filename="{filename}"\r\n'.encode())
|
||||
body.write(b"Content-Type: application/octet-stream\r\n\r\n")
|
||||
body.write(filedata)
|
||||
body.write(b"\r\n")
|
||||
if data:
|
||||
for key, val in data.items():
|
||||
body.write(f"--{boundary}\r\n".encode())
|
||||
body.write(f'Content-Disposition: form-data; name="{key}"\r\n\r\n'.encode())
|
||||
body.write(f"{val}\r\n".encode())
|
||||
body.write(f"--{boundary}--\r\n".encode())
|
||||
payload = body.getvalue()
|
||||
headers = {"Content-Type": f"multipart/form-data; boundary={boundary}"}
|
||||
else:
|
||||
payload = json.dumps(data).encode() if data is not None else b""
|
||||
headers.setdefault("Content-Type", "application/json")
|
||||
|
||||
req = Request(url, data=payload, headers=headers, method="POST")
|
||||
else:
|
||||
raise ValueError(f"Unsupported method: {method}")
|
||||
|
||||
try:
|
||||
resp = urllib.request.urlopen(req, timeout=300)
|
||||
return resp.status, resp.read()
|
||||
except urllib.error.HTTPError as e:
|
||||
return e.code, e.read()
|
||||
except urllib.error.URLError as e:
|
||||
raise AssertionError(f"Cannot reach {url}: {e.reason}")
|
||||
|
||||
|
||||
# ── Endpoint tests ──
|
||||
|
||||
def test_health_endpoint():
|
||||
"""GET /health returns 200 and expected fields."""
|
||||
if not API_BASE:
|
||||
raise AssertionError("API not running")
|
||||
status, body = _api_request("GET", "/health")
|
||||
assert status == 200, f"Expected 200, got {status}"
|
||||
data = json.loads(body)
|
||||
assert data.get("status") == "ok"
|
||||
assert "torch_device" in data or "ocr_engine" in data, "Missing expected fields"
|
||||
|
||||
|
||||
def test_docs_endpoint():
|
||||
"""GET / returns 200 and documentation."""
|
||||
if not API_BASE:
|
||||
raise AssertionError("API not running")
|
||||
status, body = _api_request("GET", "/")
|
||||
assert status == 200, f"Expected 200, got {status}"
|
||||
assert b"marker-api" in body or b"marker" in body.lower()
|
||||
|
||||
|
||||
def test_marker_multipart_markdown():
|
||||
"""POST /marker with multipart file returns markdown."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
pdf_bytes = SMALL_PDF.read_bytes()
|
||||
files = {"file": (SMALL_PDF.name, pdf_bytes, "application/pdf")}
|
||||
status, body = _api_request("POST", "/marker", files=files)
|
||||
assert status == 200, f"Expected 200, got {status}: {body[:200]}"
|
||||
|
||||
|
||||
def test_marker_multipart_json():
|
||||
"""POST /marker with output_format=json returns JSON."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
pdf_bytes = SMALL_PDF.read_bytes()
|
||||
files = {"file": (SMALL_PDF.name, pdf_bytes, "application/pdf")}
|
||||
data = {"output_format": "json"}
|
||||
status, body = _api_request("POST", "/marker", files=files, data=data)
|
||||
if status == 200:
|
||||
body_str = body.decode()
|
||||
if "{" in body_str:
|
||||
json.loads(body_str) # valid JSON
|
||||
assert status in (200, 500), f"Expected 200/500, got {status}"
|
||||
|
||||
|
||||
def test_marker_json_base64():
|
||||
"""POST /marker with JSON base64 body."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
import base64
|
||||
b64 = base64.b64encode(SMALL_PDF.read_bytes()).decode()
|
||||
payload = {"file_b64": b64, "filename": SMALL_PDF.name, "output_format": "markdown"}
|
||||
status, body = _api_request("POST", "/marker", data=payload)
|
||||
assert status in (200, 500), f"Expected 200/500, got {status}: {body[:200]}"
|
||||
|
||||
|
||||
def test_marker_with_use_llm():
|
||||
"""POST /marker with use_llm=true."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
pdf_bytes = SMALL_PDF.read_bytes()
|
||||
files = {"file": (SMALL_PDF.name, pdf_bytes, "application/pdf")}
|
||||
data = {"use_llm": "true", "llm_service": "marker.services.ollama.OllamaService"}
|
||||
status, body = _api_request("POST", "/marker", files=files, data=data)
|
||||
assert status in (200, 500), f"Expected 200/500, got {status}"
|
||||
|
||||
|
||||
def test_marker_with_page_range():
|
||||
"""POST /marker with page_range."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
pdf_bytes = SMALL_PDF.read_bytes()
|
||||
files = {"file": (SMALL_PDF.name, pdf_bytes, "application/pdf")}
|
||||
data = {"page_range": "0-2"}
|
||||
status, body = _api_request("POST", "/marker", files=files, data=data)
|
||||
assert status in (200, 500), f"Expected 200/500, got {status}"
|
||||
|
||||
|
||||
def test_marker_missing_file():
|
||||
"""POST /marker without file returns 400."""
|
||||
if not API_BASE:
|
||||
raise AssertionError("API not running")
|
||||
status, body = _api_request("POST", "/marker", data={})
|
||||
assert status == 400, f"Expected 400, got {status}"
|
||||
data = json.loads(body)
|
||||
assert "error" in data or "detail" in data
|
||||
|
||||
|
||||
def test_v1_conversions_json():
|
||||
"""POST /v1/conversions with JSON base64 body."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
import base64
|
||||
b64 = base64.b64encode(SMALL_PDF.read_bytes()).decode()
|
||||
payload = {"file_b64": b64, "filename": SMALL_PDF.name, "output_format": "json"}
|
||||
status, body = _api_request("POST", "/v1/conversions", data=payload)
|
||||
if status == 200:
|
||||
data = json.loads(body)
|
||||
assert "id" in data, "Missing id field"
|
||||
assert "output" in data, "Missing output field"
|
||||
assert status in (200, 500)
|
||||
|
||||
|
||||
def test_v1_files_convert_multipart():
|
||||
"""POST /v1/files/convert with multipart."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
pdf_bytes = SMALL_PDF.read_bytes()
|
||||
files = {"file": (SMALL_PDF.name, pdf_bytes, "application/pdf")}
|
||||
status, body = _api_request("POST", "/v1/files/convert", files=files)
|
||||
if status == 200:
|
||||
data = json.loads(body)
|
||||
assert "output" in data, "Missing output"
|
||||
assert "format" in data, "Missing format"
|
||||
assert status in (200, 500)
|
||||
|
||||
|
||||
def test_v1_files_convert_json():
|
||||
"""POST /v1/files/convert with JSON base64."""
|
||||
if not API_BASE or not SMALL_PDF:
|
||||
raise AssertionError("API not running or no test PDF")
|
||||
import base64
|
||||
b64 = base64.b64encode(SMALL_PDF.read_bytes()).decode()
|
||||
payload = {"file_b64": b64, "filename": SMALL_PDF.name}
|
||||
status, body = _api_request("POST", "/v1/files/convert", data=payload)
|
||||
if status == 200:
|
||||
data = json.loads(body)
|
||||
assert "output" in data
|
||||
assert status in (200, 500)
|
||||
|
||||
|
||||
# ── ENISA test files ──
|
||||
|
||||
def test_enisa_files_downloaded():
|
||||
"""Verify at least 3 real ENISA PDFs were downloaded."""
|
||||
pdf_count = len(ENISA_PDFS)
|
||||
file_cmds = []
|
||||
for pdf in ENISA_PDFS:
|
||||
rc = os.system(f"file '{pdf}' 2>/dev/null | grep -q 'PDF document'")
|
||||
if rc == 0:
|
||||
file_cmds.append(str(pdf))
|
||||
assert len(file_cmds) >= 2, f"Only {len(file_cmds)} real PDFs found, expected at least 2"
|
||||
|
||||
|
||||
# ── Report generation ──
|
||||
|
||||
def generate_report():
|
||||
now = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S")
|
||||
total = sum(v for k, v in results.items() if k != "warnings")
|
||||
pass_pct = round(results["passed"] / total * 100, 1) if total else 0
|
||||
|
||||
lines = []
|
||||
lines.append(f"# Test Report — marker-api Endpoint Tests")
|
||||
lines.append(f"")
|
||||
lines.append(f"**Date:** {now}")
|
||||
lines.append(f"**API Base:** {API_BASE or 'N/A (offline validation)'}")
|
||||
lines.append(f"**Test files:** {ENISA_DIR} ({len(ENISA_PDFS)} file(s))")
|
||||
lines.append(f"")
|
||||
lines.append(f"## Summary")
|
||||
lines.append(f"")
|
||||
lines.append(f"| Result | Count |")
|
||||
lines.append(f"|--------|-------|")
|
||||
lines.append(f"| ✅ Passed | {results['passed']} |")
|
||||
lines.append(f"| ❌ Failed | {results['failed']} |")
|
||||
lines.append(f"| ⏭️ Skipped | {results['skipped']} |")
|
||||
lines.append(f"| **Total** | **{total}** |")
|
||||
lines.append(f"| **Pass Rate** | **{pass_pct}%** |")
|
||||
lines.append(f"")
|
||||
lines.append(f"## Test Details")
|
||||
lines.append(f"")
|
||||
lines.append(f"| Test | Result | Time | Notes |")
|
||||
lines.append(f"|------|--------|------|-------|")
|
||||
lines.extend(REPORT_LINES)
|
||||
lines.append(f"")
|
||||
lines.append(f"## Environment")
|
||||
|
||||
# Python version
|
||||
py_ver = sys.version.split()[0]
|
||||
lines.append(f"- Python: {py_ver}")
|
||||
lines.append(f"- Platform: {sys.platform}")
|
||||
|
||||
# Check marker availability
|
||||
try:
|
||||
import marker
|
||||
lines.append(f"- marker: installed")
|
||||
except ImportError:
|
||||
lines.append(f"- marker: ❌ not installed (Pillow build failure on this platform)")
|
||||
|
||||
# Check ENISA file details
|
||||
lines.append(f"")
|
||||
lines.append(f"## Test Files")
|
||||
for pdf in ENISA_PDFS:
|
||||
size = pdf.stat().st_size
|
||||
lines.append(f"- {pdf.name} ({size / 1024:.1f} KB)")
|
||||
|
||||
lines.append(f"")
|
||||
lines.append(f"## Notes")
|
||||
lines.append(f"- HTTP endpoint tests require a running marker-api instance.")
|
||||
lines.append(f"- API-level tests were {'run against ' + API_BASE if API_BASE else 'skipped (no API running).'}")
|
||||
lines.append(f"- Offline validation tests (syntax, parameter availability, docs) ran regardless.")
|
||||
lines.append(f"- The marker library could not be installed due to Pillow build failure in this sandbox.")
|
||||
lines.append(f"- For live endpoint testing, start the API with: `python app.py` (requires marker-pdf[full] + flask).")
|
||||
if results["failed"] > 0:
|
||||
lines.append(f"- Failed tests may indicate missing marker library or API connectivity issues.")
|
||||
lines.append(f"")
|
||||
|
||||
REPORT_PATH.write_text("\n".join(lines))
|
||||
print(f"\nReport: {REPORT_PATH}")
|
||||
return str(REPORT_PATH)
|
||||
|
||||
|
||||
# ── Main ──
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Determine API base from environment or check if running locally
|
||||
if not API_BASE:
|
||||
# Try several ports — only accept if the health endpoint matches our Flask app
|
||||
for port in [8000, 8001, 8080, 5000]:
|
||||
url = f"http://localhost:{port}"
|
||||
try:
|
||||
req = urllib.request.Request(f"{url}/health", method="GET")
|
||||
resp = urllib.request.urlopen(req, timeout=3)
|
||||
if resp.status == 200:
|
||||
body = json.loads(resp.read())
|
||||
if body.get("status") == "ok" and ("model_dtype" in body or body.get("provider") == "flask"):
|
||||
API_BASE = url
|
||||
print(f"Found marker-api at {API_BASE}")
|
||||
break
|
||||
try:
|
||||
test_req = urllib.request.Request(f"{url}/marker", method="POST", data=b"")
|
||||
urllib.request.urlopen(test_req, timeout=2)
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code in (400, 500, 405, 422):
|
||||
API_BASE = url
|
||||
print(f"Found marker-api at {API_BASE} (via /marker)")
|
||||
break
|
||||
except Exception:
|
||||
continue
|
||||
if not API_BASE:
|
||||
print("No marker-api instance detected. Running offline validation tests only.")
|
||||
|
||||
r("")
|
||||
r("### Application Tests")
|
||||
test("app.py syntax valid", test_app_py_syntax)
|
||||
test("use_llm/llm_service params present in app.py", test_app_py_has_use_llm)
|
||||
test("marker-convert.ps1 has all new parameters", test_ps1_has_new_params)
|
||||
test("README.md documents all endpoints", test_readme_exists)
|
||||
test("ENISA test files available", test_enisa_files_downloaded)
|
||||
|
||||
r("")
|
||||
r("### Endpoint Tests (HTTP)")
|
||||
if API_BASE:
|
||||
test("GET /health returns 200", test_health_endpoint)
|
||||
test("GET / returns docs page", test_docs_endpoint)
|
||||
test("POST /marker multipart (markdown)", test_marker_multipart_markdown)
|
||||
test("POST /marker multipart (json)", test_marker_multipart_json)
|
||||
test("POST /marker JSON base64", test_marker_json_base64)
|
||||
test("POST /marker with use_llm=true", test_marker_with_use_llm)
|
||||
test("POST /marker with page_range", test_marker_with_page_range)
|
||||
test("POST /marker missing file → 400", test_marker_missing_file)
|
||||
test("POST /v1/conversions JSON", test_v1_conversions_json)
|
||||
test("POST /v1/files/convert multipart", test_v1_files_convert_multipart)
|
||||
test("POST /v1/files/convert JSON", test_v1_files_convert_json)
|
||||
else:
|
||||
skip("GET /health", "API not running")
|
||||
skip("GET /", "API not running")
|
||||
skip("POST /marker (multipart)", "API not running")
|
||||
skip("POST /marker (JSON)", "API not running")
|
||||
skip("POST /marker with use_llm", "API not running")
|
||||
skip("POST /marker with page_range", "API not running")
|
||||
skip("POST /marker missing file", "API not running")
|
||||
skip("POST /v1/conversions", "API not running")
|
||||
skip("POST /v1/files/convert (multipart)", "API not running")
|
||||
skip("POST /v1/files/convert (JSON)", "API not running")
|
||||
|
||||
report_path = generate_report()
|
||||
total = sum(v for k, v in results.items() if k != "warnings")
|
||||
print(f" Passed: {results['passed']}/{total} Failed: {results['failed']} Skipped: {results['skipped']}")
|
||||
Reference in New Issue
Block a user