Files
daily-robots/daily/localize.py
2026-07-02 11:31:16 +08:00

237 lines
7.9 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""将英文描述批量改写为简短中文(大模型 + 本地缓存)。"""
from __future__ import annotations
import hashlib
import json
import logging
from dataclasses import dataclass
from daily.config import CACHE_DIR, env, env_int
from daily.cursor_editor import is_enabled as cursor_editor_enabled
from daily.llm_client import extract_json_object, llm_chat
from daily.text_utils import clip_text, trim_brief
logger = logging.getLogger(__name__)
_CACHE_FILE = CACHE_DIR / "zh-desc-cache.json"
@dataclass(frozen=True)
class LocalizeJob:
key: str
text: str
limit: int
def _enabled(*, archive: bool = False) -> bool:
if not archive and cursor_editor_enabled():
return False
raw = (env("DAILY_ZH_DESC") or "1").strip().lower()
return raw not in {"0", "false", "no", "off"}
def _is_mostly_chinese(text: str) -> bool:
text = text.strip()
if not text:
return True
cjk = sum(1 for c in text if "\u4e00" <= c <= "\u9fff")
latin = sum(1 for c in text if c.isascii() and c.isalpha())
return cjk >= max(latin, 1)
def needs_chinese(text: str) -> bool:
"""文本非空且尚未以中文为主。"""
text = (text or "").strip()
if not text:
return False
return not _is_mostly_chinese(text)
def _cache_key(text: str) -> str:
return hashlib.sha1(text.encode("utf-8")).hexdigest()[:16]
def _legacy_cache_key(text: str, limit: int) -> str:
return hashlib.sha1(f"{limit}:{text}".encode("utf-8")).hexdigest()[:16]
def _cache_is_truncated(text: str) -> bool:
t = (text or "").rstrip()
return t.endswith("") or t.endswith("...")
def _lookup_cached(cache: dict[str, str], text: str, limit: int) -> str | None:
hit = cache.get(_cache_key(text))
if hit and not (limit <= 0 and _cache_is_truncated(hit)):
return hit
for legacy_limit in (72, 120, 200):
legacy = cache.get(_legacy_cache_key(text, legacy_limit))
if legacy and not (limit <= 0 and _cache_is_truncated(legacy)):
cache[_cache_key(text)] = legacy
return legacy
return None
def _load_cache() -> dict[str, str]:
if not _CACHE_FILE.exists():
return {}
try:
data = json.loads(_CACHE_FILE.read_text(encoding="utf-8"))
return {str(k): str(v) for k, v in (data.get("entries") or {}).items()}
except (OSError, json.JSONDecodeError, TypeError):
return {}
def _save_cache(entries: dict[str, str]) -> None:
CACHE_DIR.mkdir(parents=True, exist_ok=True)
_CACHE_FILE.write_text(
json.dumps({"entries": entries}, ensure_ascii=False, indent=2),
encoding="utf-8",
)
def _build_prompt(jobs: list[LocalizeJob], *, brief: bool = False) -> tuple[str, str]:
if brief:
system = (
"你是技术早报编辑。把输入 JSON 中每条描述改写为**一句**中文简要介绍。"
"要求:输出必须是中文;只保留核心能力与典型场景;不要逐字翻译;不要加引号或编号;"
"每条必须语义完整、可独立阅读limit 为建议最大字数,请控制在 limit 以内且不要用省略号截断;"
"已是中文且足够简短时可适度精简;"
"只输出 JSON 对象key 与输入一致value 为中文简介字符串。"
)
else:
system = (
"你是技术早报编辑。把输入 JSON 中每条英文描述改写为**完整**中文介绍。"
"要求:输出必须是中文;保留关键能力与使用场景;不要逐字翻译;不要加引号或编号;"
"不要以省略号截断;已是中文则原样或适度精简;"
"只输出 JSON 对象key 与输入一致value 为中文简介字符串。"
)
payload = {job.key: {"text": job.text, "limit": job.limit} for job in jobs}
user = json.dumps(payload, ensure_ascii=False, indent=2)
return system, user
def _brief_cache_key(text: str, limit: int) -> str:
return hashlib.sha1(f"brief:{limit}:{text}".encode("utf-8")).hexdigest()[:16]
def _lookup_brief_cached(cache: dict[str, str], text: str, limit: int) -> str | None:
hit = cache.get(_brief_cache_key(text, limit))
if hit and not _cache_is_truncated(hit):
return hit
return None
def _translate_batch(jobs: list[LocalizeJob], *, brief: bool = False) -> dict[str, str]:
if not jobs:
return {}
system, user = _build_prompt(jobs, brief=brief)
raw = llm_chat(system, user)
if not raw:
return {}
parsed = extract_json_object(raw)
out: dict[str, str] = {}
for job in jobs:
value = parsed.get(job.key)
if isinstance(value, str) and value.strip():
trimmed = value.strip()
if brief and job.limit > 0:
out[job.key] = trim_brief(trimmed, job.limit)
else:
out[job.key] = clip_text(trimmed, job.limit) if job.limit > 0 else trimmed
return out
def localize_brief_descriptions(jobs: list[LocalizeJob], *, archive: bool = False) -> dict[str, str]:
"""企微用:将描述改写为一句简要中文。"""
if not _enabled(archive=archive) or not jobs:
return {}
cache = _load_cache()
result: dict[str, str] = {}
pending: list[LocalizeJob] = []
for job in jobs:
if not job.text.strip():
continue
limit = job.limit if job.limit > 0 else 48
text = job.text.strip()
if _is_mostly_chinese(text) and (limit <= 0 or len(text) <= limit):
result[job.key] = text
continue
cached = _lookup_brief_cached(cache, text, limit)
if cached:
result[job.key] = trim_brief(cached, limit) if limit > 0 else cached
else:
pending.append(LocalizeJob(job.key, text, limit))
if not pending:
return result
batch_size = max(3, min(10, env_int("DAILY_ZH_DESC_BATCH", 20)))
for i in range(0, len(pending), batch_size):
chunk = pending[i : i + batch_size]
try:
translated = _translate_batch(chunk, brief=True)
except Exception as exc:
logger.warning("企微简要摘要批次失败,保留原文:%s", exc)
continue
for job in chunk:
zh = translated.get(job.key)
if not zh:
continue
ck = _brief_cache_key(job.text, job.limit if job.limit > 0 else 48)
cache[ck] = zh
result[job.key] = zh
if cache:
_save_cache(cache)
return result
def localize_descriptions(jobs: list[LocalizeJob], *, archive: bool = False) -> dict[str, str]:
"""返回 job.key -> 中文简介。archive=True 时用于完整版 .md不受 Cursor 编辑层开关影响。"""
if not _enabled(archive=archive) or not jobs:
return {}
cache = _load_cache()
result: dict[str, str] = {}
pending: list[LocalizeJob] = []
for job in jobs:
if not job.text.strip():
continue
if _is_mostly_chinese(job.text):
result[job.key] = clip_text(job.text, job.limit)
continue
ck = _cache_key(job.text)
cached = _lookup_cached(cache, job.text, job.limit)
if cached:
result[job.key] = clip_text(cached, job.limit)
else:
pending.append(job)
if not pending:
return result
batch_size = max(5, env_int("DAILY_ZH_DESC_BATCH", 20))
for i in range(0, len(pending), batch_size):
chunk = pending[i : i + batch_size]
try:
translated = _translate_batch(chunk, brief=False)
except Exception as exc:
logger.warning("中文摘要批次失败,保留英文:%s", exc)
continue
for job in chunk:
zh = translated.get(job.key)
if not zh:
continue
ck = _cache_key(job.text)
cache[ck] = zh
result[job.key] = clip_text(zh, job.limit)
if cache:
_save_cache(cache)
return result