55bca92691
Core features: - Class 1 T.30 protocol: full send/receive implementation - HDLC: DLE-stuffing, FCS strip, USR5637 bit-reversal handling - T.4 MH encoder/decoder (1728px A4 standard) - Document pipeline: PDF (Ghostscript), PNG, TIFF input - Width clamping: US Letter 1734px → 1728px fax standard - Cover page: CJK rasterization (TW/CN/JP/EN), TIFF + HTML output - OCR verification: Tesseract 5 with eng+chi_tra, CJK space-tolerant - API server (axum): health, send, jobs, cover, retry, cancel - Background worker: auto-poll queue, speed fallback, retry policy - Modem detection, pool management Real-world test results (2026-07-23): - V90 → 25153038: 4 pages, V.17 12000 bps, 2:33 ✅ - USR5637 → 25153038: 4 pages, V.17 12000 bps, 2:26 ✅ - Both faxes confirmed received on remote machine Tested: loopback (100% pixel match), multi-page, all input formats, cover pages, OCR verify, API endpoints, worker processing. 13 unit tests pass, 0 new clippy warnings.
68 lines
1.6 KiB
Python
68 lines
1.6 KiB
Python
#!/usr/bin/env python3
|
|
|
|
print("=== OCR Test Results ===\n")
|
|
|
|
import subprocess
|
|
import os
|
|
|
|
test_dir = "/tmp"
|
|
os.chdir(test_dir)
|
|
|
|
print("1. Testing English OCR:")
|
|
print("-" * 50)
|
|
result = subprocess.run(
|
|
["tesseract", "test_ocr_english2.tif", "stdout"],
|
|
capture_output=True,
|
|
text=True
|
|
)
|
|
print(result.stdout[:300] + "...")
|
|
print(f"\nWord count: ~{len(result.stdout.split())}")
|
|
print()
|
|
|
|
print("2. Testing Chinese Traditional OCR:")
|
|
print("-" * 50)
|
|
result = subprocess.run(
|
|
["tesseract", "test_ocr_chinese.png", "stdout", "-l", "chi_tra"],
|
|
capture_output=True,
|
|
text=True
|
|
)
|
|
print(result.stdout[:200] + "...")
|
|
print(f"Character count: ~{len(result.stdout)}")
|
|
print()
|
|
|
|
print("3. Testing Chinese Simplified OCR:")
|
|
print("-" * 50)
|
|
result = subprocess.run(
|
|
["tesseract", "test_ocr_chinese.png", "stdout", "-l", "chi_sim"],
|
|
capture_output=True,
|
|
text=True
|
|
)
|
|
print(result.stdout[:200] + "...")
|
|
print()
|
|
|
|
print("4. Testing Mixed Language OCR (English + Chinese):")
|
|
print("-" * 50)
|
|
result = subprocess.run(
|
|
["tesseract", "test_ocr_mixed.png", "stdout", "-l", "chi_tra+eng"],
|
|
capture_output=True,
|
|
text=True
|
|
)
|
|
print(result.stdout[:300] + "...")
|
|
print()
|
|
|
|
print("5. Available Languages:")
|
|
print("-" * 50)
|
|
result = subprocess.run(
|
|
["tesseract", "--list-langs"],
|
|
capture_output=True,
|
|
text=True
|
|
)
|
|
print(result.stdout)
|
|
|
|
print("=== OCR Test Complete ===")
|
|
print("\n✅ OCR functionality working!")
|
|
print("✅ Multi-language support installed:")
|
|
print(" - English (eng)")
|
|
print(" - Chinese Traditional (chi_tra)")
|
|
print(" - Chinese Simplified (chi_sim)")
|
|
print(" - Japanese (jpn)") |