项目初始化
This commit is contained in:
94
daily/wecom_split.py
Normal file
94
daily/wecom_split.py
Normal file
@@ -0,0 +1,94 @@
|
||||
"""企微 markdown 按字节上限拆分为多条消息(按区块,不截断正文)。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
_SECTION_START = re.compile(r"^[📰💡🎯🌍📈🔥🐙🌱🤖📦📄]")
|
||||
|
||||
|
||||
def _utf8_len(text: str) -> int:
|
||||
return len(text.encode("utf-8"))
|
||||
|
||||
|
||||
def _split_lines_by_budget(text: str, limit: int) -> list[str]:
|
||||
lines = text.splitlines()
|
||||
chunks: list[str] = []
|
||||
buf: list[str] = []
|
||||
for line in lines:
|
||||
candidate = "\n".join(buf + [line]) if buf else line
|
||||
if _utf8_len(candidate) <= limit:
|
||||
buf.append(line)
|
||||
continue
|
||||
if buf:
|
||||
chunks.append("\n".join(buf))
|
||||
buf = []
|
||||
if _utf8_len(line) <= limit:
|
||||
buf = [line]
|
||||
else:
|
||||
encoded = line.encode("utf-8")
|
||||
start = 0
|
||||
while start < len(encoded):
|
||||
piece = encoded[start : start + limit].decode("utf-8", errors="ignore")
|
||||
chunks.append(piece)
|
||||
start += len(piece.encode("utf-8"))
|
||||
if buf:
|
||||
chunks.append("\n".join(buf))
|
||||
return chunks
|
||||
|
||||
|
||||
def _split_sections(text: str) -> list[str]:
|
||||
sections: list[str] = []
|
||||
current: list[str] = []
|
||||
for line in text.splitlines():
|
||||
if _SECTION_START.match(line) and current:
|
||||
sections.append("\n".join(current))
|
||||
current = [line]
|
||||
else:
|
||||
current.append(line)
|
||||
if current:
|
||||
sections.append("\n".join(current))
|
||||
return sections
|
||||
|
||||
|
||||
def split_wecom_messages(text: str, limit: int = 4096) -> list[str]:
|
||||
"""超长时拆成多条;每条不超过 limit 字节,按区块边界优先。"""
|
||||
text = text.strip()
|
||||
if not text or _utf8_len(text) <= limit:
|
||||
return [text] if text else []
|
||||
|
||||
footer_reserve = 40
|
||||
pack_limit = max(512, limit - footer_reserve)
|
||||
|
||||
sections: list[str] = []
|
||||
for sec in _split_sections(text):
|
||||
if _utf8_len(sec) <= pack_limit:
|
||||
sections.append(sec)
|
||||
else:
|
||||
sections.extend(_split_lines_by_budget(sec, pack_limit))
|
||||
|
||||
packed: list[str] = []
|
||||
buf: list[str] = []
|
||||
for sec in sections:
|
||||
candidate = "\n\n".join(buf + [sec]) if buf else sec
|
||||
if _utf8_len(candidate) <= pack_limit:
|
||||
buf.append(sec)
|
||||
else:
|
||||
if buf:
|
||||
packed.append("\n\n".join(buf))
|
||||
buf = [sec]
|
||||
if buf:
|
||||
packed.append("\n\n".join(buf))
|
||||
|
||||
total = len(packed)
|
||||
if total <= 1:
|
||||
return packed
|
||||
|
||||
result: list[str] = []
|
||||
for i, chunk in enumerate(packed, 1):
|
||||
suffix = f"\n\n> 📄 {i}/{total}"
|
||||
body = chunk
|
||||
while body and _utf8_len(body + suffix) > limit:
|
||||
body = body.rsplit("\n", 1)[0] if "\n" in body else body[:-1]
|
||||
result.append(body + suffix)
|
||||
return result
|
||||
Reference in New Issue
Block a user