Use macOS Vision OCR for image extraction

This commit is contained in:
732642856
2026-07-02 18:46:22 +08:00
parent 8174fdcfc5
commit 11440b539a
11 changed files with 666 additions and 395 deletions
@@ -1,7 +1,7 @@
{
"scope": "external",
"status": "pass",
"generated_at": "2026-07-02T09:43:31.236878+00:00",
"generated_at": "2026-07-02T10:44:46.651144+00:00",
"mode": {
"heavy_ocr": false,
"whole_machine_scan": false,
@@ -2,7 +2,7 @@
- status: pass
- scope: external
- generated_at: 2026-07-02T09:43:31.236878+00:00
- generated_at: 2026-07-02T10:44:46.651144+00:00
- total_files: 883
- unhashed_files: 0
- unclassified_files: 0
@@ -1,7 +1,7 @@
{
"scope": "extraction-queue",
"status": "pass",
"generated_at": "2026-07-02T09:43:43.085623+00:00",
"generated_at": "2026-07-02T10:45:03.271522+00:00",
"mode": {
"heavy_ocr": false,
"whole_machine_scan": false,
@@ -2,7 +2,7 @@
- status: pass
- scope: extraction-queue
- generated_at: 2026-07-02T09:43:43.085623+00:00
- generated_at: 2026-07-02T10:45:03.271522+00:00
- queued_files: 66
- unhashed_files: 0
- Heavy OCR: disabled
File diff suppressed because it is too large Load Diff
@@ -2,28 +2,30 @@
- status: pass
- scope: extraction-results
- generated_at: 2026-07-02T09:44:22.267617+00:00
- generated_at: 2026-07-02T10:45:41.296864+00:00
- total_files: 66
- unhashed_files: 0
- stored_text_payload_fields: 0
- Heavy OCR: disabled
- macos_vision_available: True
- ocr_backend_policy: prefer_tesseract_else_macos_vision_else_blocked
## Result Counts
- `ocr_blocked_missing_engine`: 53
- `text_extracted`: 13
- `text_empty`: 17
- `text_extracted`: 49
## Method Counts
- `docx`: 11
- `macos_vision`: 53
- `pdfplumber`: 2
- `pytesseract`: 53
## Post-Extraction Classification Counts
- `extracted_candidate_for_review`: 10
- `extracted_candidate_for_review`: 44
- `extracted_private_reference_only`: 3
- `extracted_reference_only`: 53
- `extracted_reference_only`: 19
## Boundary
@@ -1,7 +1,7 @@
{
"scope": "project",
"status": "pass",
"generated_at": "2026-07-02T09:43:08.434092+00:00",
"generated_at": "2026-07-02T10:44:17.758196+00:00",
"mode": {
"heavy_ocr": false,
"whole_machine_scan": false,
@@ -47,7 +47,7 @@
},
"docs/research": {
"files": 538,
"bytes": 2965145
"bytes": 2969221
},
"references": {
"files": 236,
@@ -5173,7 +5173,7 @@
"path": "docs/research/character_level_external_manifest_latest.json",
"suffix": ".json",
"byte_count": 666472,
"sha256": "9fceb19121870ae252b77645b4e49515b8c1c6b7bd465de227dca8de47437b35",
"sha256": "e7a98a44b5ad6c2582bddb9482bf7d980a93784d74a5a448b6aee51c358269fd",
"extraction_status": "text_indexed",
"character_count": 661666,
"line_count": 12446,
@@ -5187,7 +5187,7 @@
"path": "docs/research/character_level_external_manifest_latest.md",
"suffix": ".md",
"byte_count": 1280,
"sha256": "c030624888b4bc361635468d268653358d51f359d57a597b30620a5ccf24182e",
"sha256": "dd2c84a86dc7e691d5bd9088829ba2088ff32ad305f8eb8e7d492f6eb40ab9f4",
"extraction_status": "text_indexed",
"character_count": 1272,
"line_count": 47,
@@ -5201,7 +5201,7 @@
"path": "docs/research/character_level_extraction_queue_latest.json",
"suffix": ".json",
"byte_count": 31649,
"sha256": "7aba199609805e911f07d0443e7b759ca46cbffa0527af932581c712808358e6",
"sha256": "4946bd810c6b312bae1fc9bb7a78bc34f886a4a5c7532d74732399fe7f612cdb",
"extraction_status": "text_indexed",
"character_count": 30151,
"line_count": 758,
@@ -5215,7 +5215,7 @@
"path": "docs/research/character_level_extraction_queue_latest.md",
"suffix": ".md",
"byte_count": 633,
"sha256": "8e5277feb3e35c124f65869c39e9a6038a403f284ed7dea786fea50b76365ca2",
"sha256": "8baea40e157fca6adc9228988561253f0803eb719b7392bc74ef4e64145394b6",
"extraction_status": "text_indexed",
"character_count": 633,
"line_count": 27,
@@ -5228,11 +5228,11 @@
"docs/research/character_level_extraction_results_latest.json": {
"path": "docs/research/character_level_extraction_results_latest.json",
"suffix": ".json",
"byte_count": 61620,
"sha256": "02413ba1ec2877ef67794bffde2f4a0ed35ea60677ee94c9bcbe517fac5bd778",
"byte_count": 65711,
"sha256": "a06874a9c54148891373069cd3933b3be51e3b045e51bd8b121c751f9dcf83ac",
"extraction_status": "text_indexed",
"character_count": 60122,
"line_count": 1544,
"character_count": 64213,
"line_count": 1650,
"decode_replacement_count": 0,
"classification": "research_governance",
"priority": "priority_3",
@@ -5242,10 +5242,10 @@
"docs/research/character_level_extraction_results_latest.md": {
"path": "docs/research/character_level_extraction_results_latest.md",
"suffix": ".md",
"byte_count": 764,
"sha256": "fd16ff882e9b290a4cb4aa6f613b41e54eb1e9065577e9af322f817455b0a383",
"byte_count": 749,
"sha256": "6f7d1e04b3bbb0a977e44b2cd6ddc6a71fe7b97b016a76b6e7d3dc0da37b32af",
"extraction_status": "text_indexed",
"character_count": 764,
"character_count": 749,
"line_count": 33,
"decode_replacement_count": 0,
"classification": "research_governance",
@@ -5257,7 +5257,7 @@
"path": "docs/research/character_level_inventory_manifest_latest.json",
"suffix": ".json",
"byte_count": 722955,
"sha256": "197892b1f3b259f7aa868593670cbb693dd8717bd5f304407b37bdaaf7a97463",
"sha256": "4fcf6ce8815022fbe308fca4d725c50a481ce2f1383397c3617f08bb138cc3fc",
"extraction_status": "text_indexed",
"character_count": 722405,
"line_count": 15118,
@@ -5271,7 +5271,7 @@
"path": "docs/research/character_level_inventory_manifest_latest.md",
"suffix": ".md",
"byte_count": 1090,
"sha256": "18ffd1520954f0a98b79a56f9d3919fdfb5a9e447c5802bc4a3c011204d2934e",
"sha256": "547a1e4ec8187636e39dbb09551541a0d2296cb798ee1895804feaca2612e1c1",
"extraction_status": "text_indexed",
"character_count": 1090,
"line_count": 44,
@@ -2,7 +2,7 @@
- status: pass
- scope: project
- generated_at: 2026-07-02T09:43:08.434092+00:00
- generated_at: 2026-07-02T10:44:17.758196+00:00
- total_files: 1075
- unhashed_files: 0
- unclassified_files: 0
@@ -16,7 +16,7 @@
| --- | ---: | ---: |
| `AGENTS.md` | 1 | 3932 |
| `SKILL.md` | 1 | 43187 |
| `docs/research` | 538 | 2965145 |
| `docs/research` | 538 | 2969221 |
| `references` | 236 | 4008411 |
| `references/open_source_sources` | 299 | 9365015 |
+7
View File
@@ -836,3 +836,10 @@
- 增强 `scripts/character_level_inventory_manifest.py --scope extraction-results`:图片 OCR 记录 `ocr_engine_available``ocr_requested_languages``ocr_available_languages`,便于安装 OCR 后复跑。
- 为异常 DOCX 增加 `word/document.xml` 备用提取路径;`/Users/wuyongnaren/文件仓库/印度占星文章/4印度占星.docx` 已从 `extraction_failed` 转为 `text_extracted`,字符数 `19643`
- 当前提取结果更新为:`text_extracted=13``ocr_blocked_missing_engine=53``extraction_failed=0`;仍未保存正文,`stored_text_payload_fields=0`
## 2026-07-02T18:43:00+08:00 - Intel macOS 12 Vision OCR 完成
- 机器确认:macOS `12.7.6`、Intel `x86_64`、Core i7-4980HQHomebrew 当前 tesseract 主程序没有可用 bottle,只能取源码 tarball,继续安装会拖入大规模依赖/源码构建。
- 采用更适合本机的 macOS Vision OCR:脚本自动编译并缓存 `~/.cache/jyotish-ocr/vision_ocr`,不依赖 Homebrew tesseract。
- `scripts/character_level_inventory_manifest.py --scope extraction-results` 新增 OCR cache;缓存只保存 OCR 文本 hash、字符数、行数、后端、状态,不保存 OCR 正文。
- 53 张图片已用 `macos_vision` 复跑完成;当前结果:`text_extracted=49``text_empty=17``ocr_blocked_missing_engine=0``extraction_failed=0``stored_text_payload_fields=0`
+155 -4
View File
@@ -12,6 +12,7 @@ import argparse
import hashlib
import json
import shutil
import subprocess
import sys
import zipfile
from datetime import datetime, timezone
@@ -29,6 +30,47 @@ EXTRACTION_QUEUE_JSON = ROOT / "docs" / "research" / "character_level_extraction
EXTRACTION_QUEUE_MD = ROOT / "docs" / "research" / "character_level_extraction_queue_latest.md"
EXTRACTION_RESULTS_JSON = ROOT / "docs" / "research" / "character_level_extraction_results_latest.json"
EXTRACTION_RESULTS_MD = ROOT / "docs" / "research" / "character_level_extraction_results_latest.md"
VISION_OCR_SOURCE = r'''
import Foundation
import Vision
import AppKit
let args = CommandLine.arguments
if args.count < 2 {
fputs("usage: vision_ocr image\n", stderr)
exit(2)
}
let url = URL(fileURLWithPath: args[1])
guard let image = NSImage(contentsOf: url), let tiff = image.tiffRepresentation,
let bitmap = NSBitmapImageRep(data: tiff), let cgImage = bitmap.cgImage else {
fputs("cannot load image\n", stderr)
exit(3)
}
let request = VNRecognizeTextRequest { request, error in
if let error = error {
fputs("vision error: \(error)\n", stderr)
exit(4)
}
let observations = request.results as? [VNRecognizedTextObservation] ?? []
for obs in observations {
if let top = obs.topCandidates(1).first {
print(top.string)
}
}
}
request.recognitionLevel = .accurate
request.usesLanguageCorrection = true
if #available(macOS 11.0, *) {
request.recognitionLanguages = ["zh-Hans", "en-US"]
}
let handler = VNImageRequestHandler(cgImage: cgImage, options: [:])
do {
try handler.perform([request])
} catch {
fputs("perform error: \(error)\n", stderr)
exit(5)
}
'''
PROJECT_SCAN_ROOTS = [
"references",
@@ -533,7 +575,7 @@ def _extract_pdf_text(path: Path) -> tuple[str, str, str | None]:
def _extract_image_text(path: Path) -> tuple[str, str, str | None]:
if not shutil.which("tesseract"):
return "", "pytesseract", "tesseract executable not found"
return _extract_image_text_with_vision(path)
try:
from PIL import Image
import pytesseract
@@ -544,6 +586,99 @@ def _extract_image_text(path: Path) -> tuple[str, str, str | None]:
return "", "pytesseract", f"{type(exc).__name__}: {exc}"
def _vision_ocr_binary() -> Path | None:
if sys.platform != "darwin" or not shutil.which("swiftc"):
return None
cache_dir = Path.home() / ".cache" / "jyotish-ocr"
cache_dir.mkdir(parents=True, exist_ok=True)
source = cache_dir / "vision_ocr.swift"
binary = cache_dir / "vision_ocr"
if binary.exists() and binary.stat().st_mtime >= source.stat().st_mtime if source.exists() else False:
return binary
source.write_text(VISION_OCR_SOURCE, encoding="utf-8")
completed = subprocess.run(
["swiftc", str(source), "-o", str(binary)],
text=True,
capture_output=True,
timeout=60,
check=False,
)
if completed.returncode != 0:
return None
return binary
def _extract_image_text_with_vision(path: Path) -> tuple[str, str, str | None]:
binary = _vision_ocr_binary()
if not binary:
return "", "none", "no OCR backend available: tesseract missing and macOS Vision unavailable"
completed = subprocess.run(
[str(binary), str(path)],
text=True,
capture_output=True,
timeout=90,
check=False,
)
if completed.returncode != 0:
return "", "macos_vision", completed.stderr.strip() or f"vision_ocr exit {completed.returncode}"
return completed.stdout, "macos_vision", None
def _ocr_cache_path(item: dict[str, Any], method: str) -> Path:
cache_dir = Path.home() / ".cache" / "jyotish-ocr" / "results"
cache_dir.mkdir(parents=True, exist_ok=True)
return cache_dir / f"{item['sha256']}.{method}.json"
def _load_ocr_cached_item(item: dict[str, Any]) -> dict[str, Any] | None:
for method in ["macos_vision", "pytesseract"]:
path = _ocr_cache_path(item, method)
if not path.exists():
continue
try:
cached = json.loads(path.read_text(encoding="utf-8"))
except (OSError, json.JSONDecodeError):
continue
if cached.get("source_sha256") != item.get("sha256"):
continue
return {
**item,
"extraction_result": cached["extraction_result"],
"extraction_method": cached["extraction_method"],
"blocked_reason": cached.get("blocked_reason"),
"extracted_character_count": cached["extracted_character_count"],
"extracted_line_count": cached["extracted_line_count"],
"text_sha256": cached.get("text_sha256"),
"post_extraction_classification": cached["post_extraction_classification"],
"ocr_engine_available": cached["ocr_engine_available"],
"ocr_requested_languages": cached["ocr_requested_languages"],
"ocr_available_languages": cached["ocr_available_languages"],
"ocr_backend": cached["ocr_backend"],
"ocr_cache_status": "hit",
}
return None
def _write_ocr_cached_item(item: dict[str, Any]) -> None:
method = str(item.get("ocr_backend") or item.get("extraction_method") or "unknown")
path = _ocr_cache_path(item, method)
cache_payload = {
"source_sha256": item["sha256"],
"extraction_result": item["extraction_result"],
"extraction_method": item["extraction_method"],
"blocked_reason": item.get("blocked_reason"),
"extracted_character_count": item["extracted_character_count"],
"extracted_line_count": item["extracted_line_count"],
"text_sha256": item.get("text_sha256"),
"post_extraction_classification": item["post_extraction_classification"],
"ocr_engine_available": item.get("ocr_engine_available"),
"ocr_requested_languages": item.get("ocr_requested_languages", ["chi_sim", "eng"]),
"ocr_available_languages": item.get("ocr_available_languages", []),
"ocr_backend": item.get("ocr_backend", method),
}
path.write_text(json.dumps(cache_payload, ensure_ascii=False, indent=2), encoding="utf-8")
def _ocr_available_languages() -> list[str]:
if not shutil.which("tesseract"):
return []
@@ -580,9 +715,12 @@ def _extract_queued_item(item: dict[str, Any]) -> dict[str, Any]:
elif status == "pdf_text_extraction_queued":
text, method, error = _extract_pdf_text(path)
elif status == "image_ocr_queued":
cached = _load_ocr_cached_item(item)
if cached:
return cached
text, method, error = _extract_image_text(path)
if error and "tesseract executable not found" in error:
return {
if error and "no OCR backend available" in error:
blocked_item = {
**item,
"extraction_result": "ocr_blocked_missing_engine",
"extraction_method": method,
@@ -594,7 +732,11 @@ def _extract_queued_item(item: dict[str, Any]) -> dict[str, Any]:
"ocr_engine_available": False,
"ocr_requested_languages": ["chi_sim", "eng"],
"ocr_available_languages": [],
"ocr_backend": "none",
"ocr_cache_status": "miss",
}
_write_ocr_cached_item(blocked_item)
return blocked_item
else:
text = ""
method = "unsupported"
@@ -607,7 +749,7 @@ def _extract_queued_item(item: dict[str, Any]) -> dict[str, Any]:
result = "text_extracted"
else:
result = "text_empty"
return {
result_item = {
**item,
"extraction_result": result,
"extraction_method": method,
@@ -621,11 +763,16 @@ def _extract_queued_item(item: dict[str, Any]) -> dict[str, Any]:
"ocr_engine_available": shutil.which("tesseract") is not None,
"ocr_requested_languages": ["chi_sim", "eng"],
"ocr_available_languages": _ocr_available_languages(),
"ocr_backend": "tesseract" if method == "pytesseract" else method,
"ocr_cache_status": "miss",
}
if status == "image_ocr_queued"
else {}
),
}
if status == "image_ocr_queued":
_write_ocr_cached_item(result_item)
return result_item
def build_extraction_results(*, write: bool = True) -> dict[str, Any]:
@@ -647,6 +794,8 @@ def build_extraction_results(*, write: bool = True) -> dict[str, Any]:
"generated_at": datetime.now(timezone.utc).isoformat(),
"mode": {
"heavy_ocr": shutil.which("tesseract") is not None,
"macos_vision_available": _vision_ocr_binary() is not None,
"ocr_backend_policy": "prefer_tesseract_else_macos_vision_else_blocked",
"whole_machine_scan": False,
"external_high_relevance_scan": True,
"semantic_promotion": False,
@@ -791,6 +940,8 @@ def _render_extraction_results_markdown(report: dict[str, Any]) -> str:
f"- unhashed_files: {report['summary']['unhashed_files']}",
f"- stored_text_payload_fields: {report['summary']['stored_text_payload_fields']}",
f"- Heavy OCR: {'enabled' if report['mode']['heavy_ocr'] else 'disabled'}",
f"- macos_vision_available: {report['mode'].get('macos_vision_available', False)}",
f"- ocr_backend_policy: {report['mode'].get('ocr_backend_policy', 'unknown')}",
"",
"## Result Counts",
"",
@@ -190,6 +190,7 @@ def test_extraction_results_extract_pdf_and_docx_without_storing_text() -> None:
assert report["result_counts"].get("extraction_failed", 0) == 0
assert report["method_counts"]["docx"] >= 10
assert report["method_counts"]["pdfplumber"] + report["method_counts"].get("pypdf", 0) >= 2
assert report["mode"]["macos_vision_available"] in {True, False}
for item in report["results"]:
assert item["sha256"]
@@ -210,10 +211,12 @@ def test_extraction_results_extract_pdf_and_docx_without_storing_text() -> None:
"extracted_private_reference_only",
"extracted_candidate_for_review",
}
if item["extraction_result"] == "ocr_blocked_missing_engine":
assert item["ocr_engine_available"] is False
if item["extraction_status"] == "image_ocr_queued":
assert item["ocr_requested_languages"] == ["chi_sim", "eng"]
assert item["ocr_available_languages"] == []
assert item["ocr_backend"] in {"tesseract", "macos_vision", "none"}
if item["ocr_backend"] == "none":
assert item["extraction_result"] == "ocr_blocked_missing_engine"
assert item["ocr_engine_available"] is False
json_path = ROOT / report["artifacts"]["json_report"]
md_path = ROOT / report["artifacts"]["markdown_report"]