feat(qqpush): 浏览器长图渲染公告

This commit is contained in:
2026-07-30 21:21:49 +08:00
parent f8e9544fd6
commit d2187e5545
3 changed files with 165 additions and 12 deletions

View File

@@ -0,0 +1,89 @@
"""阴阳师等长公告的安全 HTML 截图渲染。"""
import base64
import html
import re
from typing import Optional
from pyppeteer import launch
try:
import bleach
except ImportError:
bleach = None
ARTICLE_WIDTH = 900
ARTICLE_PADDING = 48
_ALLOWED_TAGS = [
"p", "br", "strong", "b", "em", "i", "u", "span", "div",
"h1", "h2", "h3", "ul", "ol", "li", "blockquote", "hr",
]
def _sanitize_article_html(content_html: str) -> str:
"""净化外部公告 HTML缺少 bleach 时退化为安全的纯文本段落。"""
if bleach is not None:
return bleach.clean(content_html, tags=_ALLOWED_TAGS, attributes={}, strip=True)
# 生产环境缺少可选依赖时,宁可丢失样式也不把不可信标签交给浏览器执行。
text = re.sub(r"(?is)<(script|style).*?>.*?</\1>", "", content_html)
text = re.sub(r"(?i)<br\s*/?>", "\n", text)
text = re.sub(r"(?i)</(p|div|h[1-6]|li|blockquote)>", "\n", text)
text = re.sub(r"<[^>]+>", "", text)
return "<p>" + html.escape(text).replace("\n", "<br>") + "</p>"
def build_article_document(content_html: str, title: str, published_at: Optional[str]) -> str:
"""净化上游公告 HTML并包装为固定版式的截图页面。"""
safe_content = _sanitize_article_html(content_html)
safe_title = html.escape(title)
safe_time = html.escape(published_at or "")
return f"""
<!doctype html>
<html lang="zh-CN">
<head><meta charset="utf-8"><style>
* {{ box-sizing: border-box; }}
html, body {{ margin: 0; background: #f5f6f8; color: #1f2329; }}
body {{ width: {ARTICLE_WIDTH}px; font-family: "Noto Sans CJK SC", "Microsoft YaHei", sans-serif; }}
article {{ margin: 24px; padding: {ARTICLE_PADDING}px; background: #fff; border-radius: 16px; }}
.label {{ color: #1677ff; font-size: 22px; font-weight: 700; }}
h1 {{ margin: 16px 0 10px; font-size: 34px; line-height: 1.35; }}
.time {{ color: #86909c; font-size: 18px; padding-bottom: 24px; border-bottom: 1px solid #e5e6eb; }}
.content {{ padding-top: 26px; font-size: 24px; line-height: 1.8; overflow-wrap: anywhere; }}
.content p {{ margin: 0 0 20px; }}
.content h1, .content h2, .content h3 {{ font-size: 28px; margin: 28px 0 14px; }}
.content ul, .content ol {{ padding-left: 1.4em; margin: 0 0 20px; }}
.content blockquote {{ margin: 0 0 20px; padding-left: 18px; border-left: 4px solid #91caff; color: #4e5969; }}
</style></head>
<body><article>
<div class="label">阴阳师维护公告</div>
<h1>{safe_title}</h1>
<div class="time">发布时间:{safe_time}</div>
<section class="content">{safe_content}</section>
</article></body></html>
"""
async def render_article_to_base64(content_html: str, title: str, published_at: Optional[str]) -> str:
"""使用无头 Chromium 渲染完整公告为一张长图。"""
browser = None
page = None
try:
browser = await launch(
headless=True,
args=["--no-sandbox", "--disable-setuid-sandbox", "--disable-gpu"],
)
page = await browser.newPage()
await page.setViewport({"width": ARTICLE_WIDTH, "height": 1200, "deviceScaleFactor": 1})
await page.setContent(build_article_document(content_html, title, published_at))
height = await page.evaluate("document.documentElement.scrollHeight")
if height <= 0:
raise ValueError("公告截图高度无效")
image = await page.screenshot({"fullPage": True, "type": "png"})
return f"base64://{base64.b64encode(image).decode('ascii')}"
finally:
if page is not None:
await page.close()
if browser is not None:
await browser.close()