readout_program/doc_builder/build.py

207 lines
6.4 KiB
Python
Raw Normal View History

2026-07-24 20:24:39 +08:00
#!/usr/bin/env python3
"""
Docs as Code 构建入口
读取 project.yaml 拼接 chapters/*.md 渲染为自包含 HTML 报告
Python 实现依赖 requirements.txt 中的 PyYAML, markdown, Jinja2
用法:
python doc_builder/build.py
输出:
output/读出子系统编程控制模型.html 自包含 HTML可离线分发
"""
import os
import sys
import base64
import re
from pathlib import Path
import yaml
import markdown
from jinja2 import Environment, FileSystemLoader
# ---------- 配置 ----------
ROOT = Path(__file__).resolve().parent.parent
CHAPTERS_DIR = ROOT / "chapters"
ASSETS_DIR = ROOT / "assets"
OUTPUT_DIR = ROOT / "output"
OUTPUT_NAME = "读出子系统编程控制模型.html"
# ---------- 工具函数 ----------
def load_project_config() -> dict:
"""读取 project.yaml 并校验。"""
config_path = ROOT / "project.yaml"
if not config_path.exists():
sys.exit(f"错误: 找不到 {config_path}")
with open(config_path, "r", encoding="utf-8") as f:
config = yaml.safe_load(f)
return config
def resolve_image_path(md_content: str, assets_dir: Path, inline_images: bool = True) -> tuple[str, dict]:
"""
处理 Markdown 中的图片引用:
- 如果 inline_images=True将图片内嵌为 base64 data URI
- 否则转换为相对路径引用
返回: (处理后的内容, {原始路径: data_uri 字典})
"""
image_map = {}
def replace_img(match):
alt_text = match.group(1)
img_path = match.group(2)
# 解析路径: ./assets/xxx.png 或 assets/xxx.png
if img_path.startswith("./"):
img_path = img_path[2:]
if img_path.startswith("assets/"):
img_path = img_path[7:]
full_path = assets_dir / img_path
if not full_path.exists():
print(f" 警告: 找不到图片 {full_path},保留原始引用")
return match.group(0)
if inline_images:
# 内嵌为 base64
with open(full_path, "rb") as img_file:
img_data = base64.b64encode(img_file.read()).decode("ascii")
ext = full_path.suffix.lower()
mime_map = {".png": "image/png", ".svg": "image/svg+xml",
".jpg": "image/jpeg", ".jpeg": "image/jpeg",
".gif": "image/gif"}
mime = mime_map.get(ext, "image/png")
data_uri = f"data:{mime};base64,{img_data}"
image_map[img_path] = data_uri
return f"![{alt_text}]({data_uri})"
else:
return f"![{alt_text}](assets/{img_path})"
return re.sub(r'!\[([^\]]*)\]\(\./assets/([^)]+)\)', replace_img, md_content), image_map
def read_and_assemble(config: dict) -> str:
"""按 project.yaml 的章节列表拼接所有 chapters/*.md 文件。"""
chapters = config.get("chapters", [])
if not chapters:
sys.exit("错误: project.yaml 中未定义 chapters 列表")
parts = []
for ch_file in chapters:
ch_path = CHAPTERS_DIR / ch_file
if not ch_path.exists():
print(f" 警告: 章节文件不存在,跳过: {ch_file}")
continue
with open(ch_path, "r", encoding="utf-8") as f:
content = f.read()
parts.append(content)
return "\n\n".join(parts)
def extract_toc(md_content: str) -> list[dict]:
"""从 Markdown 内容中提取目录结构 (H1, H2)。"""
toc = []
for line in md_content.split("\n"):
m = re.match(r'^(#{1,3})\s+(.+)$', line)
if m:
level = len(m.group(1))
title = m.group(2).strip()
# 生成锚点:去除特殊字符,空格转连字符
anchor = re.sub(r'[^\w\s一-鿿-]', '', title)
anchor = anchor.strip().replace(' ', '-').lower()
toc.append({"level": level, "title": title, "anchor": anchor})
return toc
# ---------- HTML 生成 ----------
def generate_html(md_content: str, config: dict) -> str:
"""将 Markdown 内容转换为完整的 HTML 页面。"""
# 处理图片内嵌
processed_md, _ = resolve_image_path(md_content, ASSETS_DIR, inline_images=True)
# Markdown → HTML
md_extensions = [
"markdown.extensions.tables",
"markdown.extensions.fenced_code",
"markdown.extensions.codehilite",
"markdown.extensions.toc",
"markdown.extensions.nl2br",
]
html_body = markdown.markdown(processed_md, extensions=md_extensions)
# 提取目录
toc = extract_toc(md_content)
# 加载 Jinja2 模板
templates_dir = ROOT / "doc_builder" / "templates"
env = Environment(loader=FileSystemLoader(str(templates_dir)))
template = env.get_template("report.html")
# 读取主题 CSS
themes_dir = ROOT / "doc_builder" / "themes"
css_screen = ""
css_print = ""
screen_css_path = themes_dir / "report.css"
print_css_path = themes_dir / "print.css"
if screen_css_path.exists():
css_screen = screen_css_path.read_text(encoding="utf-8")
if print_css_path.exists():
css_print = print_css_path.read_text(encoding="utf-8")
return template.render(
title=config.get("title", "文档"),
subtitle=config.get("subtitle", ""),
author=config.get("author", ""),
version=config.get("version", ""),
toc=toc,
body=html_body,
css_screen=css_screen,
css_print=css_print,
)
# ---------- 主入口 ----------
def main():
print("=== ez-Q 2.5 读出子系统编程控制模型 构建 ===")
print()
# 1. 加载配置
print("[1/3] 读取 project.yaml ...")
config = load_project_config()
print(f" 项目: {config.get('title', '未命名')}")
print(f" 版本: {config.get('version', 'N/A')}")
print(f" 章节数: {len(config.get('chapters', []))}")
# 2. 拼接章节
print("[2/3] 拼接章节 ...")
assembled = read_and_assemble(config)
print(f" 总字符数: {len(assembled)}")
# 3. 生成 HTML
print("[3/3] 生成 HTML ...")
html = generate_html(assembled, config)
# 4. 输出
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
output_path = OUTPUT_DIR / OUTPUT_NAME
output_path.write_text(html, encoding="utf-8")
print(f" 输出: {output_path}")
print(f" 文件大小: {output_path.stat().st_size:,} 字节")
print()
print("Build completed successfully!")
if __name__ == "__main__":
main()