Files
VRCWT-OSC/src/ocr/ocr_actions.py
2026-06-09 02:37:06 +09:00

264 lines
7.2 KiB
Python

import os
import time
from datetime import datetime
os.environ["FLAGS_enable_pir_api"] = "0"
os.environ["FLAGS_use_onednn"] = "0"
os.environ["FLAGS_use_mkldnn"] = "0"
os.environ["FLAGS_use_onednn_bfloat16"] = "0"
os.environ["MKLDNN_DISABLE_WORKSPACE"] = "1"
os.environ["ONEDNN_VERBOSE"] = "0"
import pyperclip
from paddleocr import PaddleOCR
from PIL import Image
from PIL import ImageFilter
from PIL import ImageOps
from common.config_loader import loadOcrConfig
from common.runtime_log import appendRuntimeLog
from screen.screen_actions import captureOcrRegion
_ocr_engine = None
def log(level, message):
now = datetime.now().strftime("%H:%M:%S")
print(f"[{level}] {now} {message}", flush=True)
def getOcrEngine():
global _ocr_engine
if _ocr_engine is None:
log("INFO", "PaddleOCR initializing")
config = loadOcrConfig()
options = {
"lang": config["lang"],
"use_angle_cls": config["use_angle_cls"],
"show_log": config["show_log"],
}
try:
_ocr_engine = PaddleOCR(**options)
except (TypeError, ValueError):
options.pop("show_log", None)
_ocr_engine = PaddleOCR(**options)
return _ocr_engine
_gemini_client = None
def getGeminiClient(api_key):
global _gemini_client
if _gemini_client is not None:
return _gemini_client
if not api_key:
raise RuntimeError("Gemini API key is not set in secrets.toml [ai]")
try:
from google import genai
except Exception as e:
raise RuntimeError(f"Gemini SDK is missing detail={e}")
_gemini_client = genai.Client(api_key=api_key)
return _gemini_client
def saveOcrText(image_path, text):
return appendRuntimeLog(
"OCR TEXT",
f"image={image_path}\n{text}",
)
def extractTextFromAny(obj):
lines = []
def walk(x):
if x is None:
return
if isinstance(x, dict):
for key in ("rec_texts", "texts"):
value = x.get(key)
if isinstance(value, list):
for item in value:
text = str(item).strip()
if text:
lines.append(text)
for key in ("text", "rec_text"):
value = x.get(key)
if isinstance(value, str) and value.strip():
lines.append(value.strip())
for value in x.values():
walk(value)
return
if isinstance(x, (list, tuple)):
if len(x) >= 2 and isinstance(x[1], (list, tuple)) and len(x[1]) >= 1:
candidate = x[1][0]
if isinstance(candidate, str) and candidate.strip():
lines.append(candidate.strip())
for item in x:
walk(item)
walk(obj)
unique_lines = []
seen = set()
for line in lines:
if line not in seen:
unique_lines.append(line)
seen.add(line)
return "\n".join(unique_lines)
def printOcrResult(text):
cleaned = text.strip()
if not cleaned:
log("INFO", "OCR結果は空です")
return
log("CHECK", "OCR結果 full")
for line in cleaned.splitlines():
print(line, flush=True)
def preprocessOcrImage(image_path):
config = loadOcrConfig()
enabled = config["preprocess"]
if not enabled:
return image_path
try:
image = Image.open(image_path)
image = image.convert("L")
image = ImageOps.autocontrast(image)
image = image.filter(ImageFilter.SHARPEN)
scale = max(1.0, min(2.0, config["scale"]))
if scale > 1.0:
width, height = image.size
resampling = getattr(
getattr(Image, "Resampling", Image),
"LANCZOS",
)
image = image.resize(
(int(width * scale), int(height * scale)),
resampling,
)
image.save(image_path)
log("INFO", f"OCR image preprocessed={image_path}")
except Exception as e:
log("ERROR", f"OCR preprocess failed detail={e}")
return image_path
def runPaddleOcr(engine, image_path):
if hasattr(engine, "predict"):
return engine.predict(str(image_path))
return engine.ocr(str(image_path))
def runGeminiOcr(image_path):
config = loadOcrConfig()
model_name = config["model_name"] or "gemini-2.5-flash"
prompt = (
"Extract every readable text fragment from this image with maximum accuracy. "
"Do not summarize, explain, translate, or guess. "
"Return only the text exactly as it appears, preserving line breaks and order."
)
try:
from google.genai import types
except Exception as e:
raise RuntimeError(f"Gemini types import failed detail={e}")
client = getGeminiClient(config["api_key"])
mime_type = "image/png"
suffix = str(image_path).lower()
if suffix.endswith(".jpg") or suffix.endswith(".jpeg"):
mime_type = "image/jpeg"
elif suffix.endswith(".webp"):
mime_type = "image/webp"
image_bytes = image_path.read_bytes()
image_part = types.Part.from_bytes(data=image_bytes, mime_type=mime_type)
response = client.models.generate_content(
model=model_name,
contents=[prompt, image_part],
)
text = (response.text or "").strip()
return text
def runOcrFromImage(image_path, show_result=True):
try:
image_path = preprocessOcrImage(image_path)
config = loadOcrConfig()
provider = config["provider"]
log("ACTION", f"OCR start provider={provider} image={image_path}")
if provider == "gemini":
text = runGeminiOcr(image_path)
else:
engine = getOcrEngine()
result = runPaddleOcr(engine, image_path)
text = extractTextFromAny(result)
if not text:
raise RuntimeError("OCR text is empty")
except Exception as e:
log("ERROR", f"OCR failed detail={e}")
if config.get("provider") == "gemini":
log("INFO", "Falling back to PaddleOCR")
try:
engine = getOcrEngine()
result = runPaddleOcr(engine, image_path)
text = extractTextFromAny(result)
except Exception as fallback_error:
log("ERROR", f"PaddleOCR fallback failed detail={fallback_error}")
return None
else:
return None
text_path = saveOcrText(image_path, text)
try:
pyperclip.copy(text)
log("INFO", "OCR結果をクリップボードへコピーしました")
except Exception as e:
log("ERROR", f"clipboard copy failed detail={e}")
log("INFO", f"OCR text appended={text_path}")
if show_result:
printOcrResult(text)
return {
"image_path": image_path,
"text_path": text_path,
"text": text,
}
def runOcrFromScreen():
log("ACTION", "runOcrFromScreen called")
image_path = captureOcrRegion()
delay_seconds = max(0.0, loadOcrConfig().get("delay_seconds", 1.0))
if delay_seconds:
log("INFO", f"OCR delayed seconds={delay_seconds}")
time.sleep(delay_seconds)
return runOcrFromImage(image_path, show_result=True)