界面(去 AI 味): - 大屏页清除 114 处生成器残留属性 data-page-node-id - 视觉系统改回工程控制台风格:去 radial/linear-gradient、去辉光、 去标题前彩色装饰条,改为中性灰阶 + 单一蓝色强调色;KPI 色条改状态点 - 精简各页说教式长提示;修掉 profile.html 泄漏到页面上的 Markdown 星号 - 删除登录页过时的「默认账号 admin / admin123」提示(1.4.0 起已无默认口令) 安全与隐私(按「将会被公网访问」收口): - 内部异常只回 8 位事件号,完整堆栈进服务端日志(web/api.py::_internal) - 导出文件名收敛:防响应头注入与路径穿越;manage.py passwd 补用户名校验 - 登录对不存在的账号也走一次哑哈希,抹平用户名枚举的时序差异 - /api/* 读接口限速 240 次 / 60 秒 / 账号(挡住循环调 /api/bundle) - 进程 umask 0077 + 目录 0700 / 文件 0600:对话正文与主密钥的落盘权限 - 表名与库文件路径只对管理员下发;大屏页所有数据插值转义 - --debug 只允许绑定回环地址;新增 Permissions-Policy 与 413 处理器 文档: - DEPLOYMENT 新增第十三节「安全与隐私基线」;迁移表补 1.4.0 → 1.5.0 行 - SECURITY 更新支持范围、新增「信息泄漏收敛」小节与上线检查项 - .codebuddy/ 加入 .gitignore(助手工作记忆不进仓库) 版本:1.4.0 → 1.5.0(无库结构变更,user_version 仍为 4) 验证:python tools/smoke.py → ok=264 fail=0;python tools/check_docs.py → 0 处问题
339 行
14 KiB
Python
339 行
14 KiB
Python
#!/usr/bin/env python3
|
||
# SPDX-License-Identifier: MIT
|
||
"""文档自检:内部链接 / 跨文件锚点 / 图片引用 / 绝对路径泄漏 / 版本一致性 / 产品名硬编码。
|
||
|
||
文档一旦互相引用(README → docs/DEPLOYMENT.md#某节),章节重排就会让锚点**静默失效** ——
|
||
Markdown 不会报错,页面只是不跳转、图片只是显示裂图。这个脚本把这类问题变成可执行的断言。
|
||
|
||
用法:
|
||
python tools/check_docs.py # 有问题则退出码 1
|
||
python tools/check_docs.py --no-fail # 只看报告,不因问题而失败
|
||
|
||
退出码:0 通过,1 发现问题。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import os
|
||
import re
|
||
import sys
|
||
|
||
# ---------------------------------------------------------------- 常量
|
||
|
||
# 递归扫描时跳过的目录。**自动发现所有 .md**,而不是写死文件名列表 ——
|
||
# 写死列表的版本曾漏掉 THIRD-PARTY-NOTICES.md 与 CODE_OF_CONDUCT.md
|
||
# (新增文档时必然漏,而且没人会发现)。
|
||
SKIP_DIRS = {
|
||
".git", ".hg", ".svn", "node_modules", "vendor", "dist", "build",
|
||
".venv", "venv", "__pycache__", ".mypy_cache", ".pytest_cache", ".idea", ".vscode",
|
||
# AI 助手的工作记忆与技能缓存:是开发过程产物,不是项目文档。
|
||
# 它已在 .gitignore 里,这里一并跳过,免得自检报告被一堆无关 .md 淹没
|
||
# (而且那些笔记里的链接是给助手看的,不该按项目文档的规则去校验)。
|
||
".codebuddy",
|
||
}
|
||
|
||
# markdown 链接:[文本](目标)
|
||
LINK_RE = re.compile(r"\[([^\]]*)\]\(([^)\s]+)(?:\s+\"[^\"]*\")?\)")
|
||
# markdown 图片:
|
||
IMG_RE = re.compile(r"!\[[^\]]*\]\(([^)\s]+)\)")
|
||
# markdown 标题:# / ## / ... (用于生成锚点)
|
||
HEADING_RE = re.compile(r"^(#{1,6})\s+(.*?)\s*$")
|
||
# 显式 HTML 锚点:<a name="x"></a> 或 <a id="x"></a>
|
||
HTML_ANCHOR_RE = re.compile(r"<a\s+(?:name|id)=[\"']([^\"']+)[\"']")
|
||
# 代码块围栏(三反引号或三波浪线)
|
||
FENCE_RE = re.compile(r"^\s*(```|~~~)")
|
||
|
||
# 绝对路径泄漏:正向白名单写不出来,反着匹配已知模式足够有效。
|
||
# 每个模式的**第 1 个捕获组**是「用户名那一段」,用来判占位符。
|
||
LEAK_PATTERNS = [
|
||
(re.compile(r"[A-Za-z]:[\\/](?:Users|users)[\\/]([^\\/\s\"'`)]+)"),
|
||
"用户目录绝对路径"),
|
||
(re.compile(r"[A-Za-z]:[\\/]Documents and Settings[\\/]([^\\/\s\"'`)]+)"),
|
||
"用户目录绝对路径"),
|
||
(re.compile(r"/(?:home|Users)/([A-Za-z0-9._-]+)/"), "用户目录绝对路径"),
|
||
]
|
||
# 占位符豁免:`C:\Users\<用户名>\…` 是**良好实践**,不是泄漏。
|
||
PLACEHOLDER_RE = re.compile(
|
||
r"[<>%*{}]|^\.{2,}$|^[-_]+$"
|
||
r"|^(?:user|users|username|user-?name|your-?name|youruser|"
|
||
r"用户名|你的用户名|xxx+|yyy+|zzz+|aaa+|example|placeholder|"
|
||
r"me|someone|nobody)$",
|
||
re.IGNORECASE,
|
||
)
|
||
|
||
# 版本一致性:这四处必须互相一致
|
||
VERSION_INIT = os.path.join("workbuddy_portal", "__init__.py")
|
||
INIT_VER_RE = re.compile(r'^__version__\s*=\s*["\']([^"\']+)["\']', re.M)
|
||
DOCKERFILE = "Dockerfile"
|
||
OCI_VER_RE = re.compile(r'org\.opencontainers\.image\.version\s*=\s*"([^"]+)"')
|
||
README_VER_RE = re.compile(r"^\|\s*版本\s*\|\s*v?([0-9][^\s|]*)\s*\|", re.M)
|
||
CHANGELOG_VER_RE = re.compile(r"^##\s*\[?v?([0-9][^\s\]—-]*)", re.M)
|
||
|
||
# 产品名硬编码:应走 config.PROJECT_NAME 等上下文变量,不写进模板/JS
|
||
HARDCODE_NEEDLES = ["WorkBuddy Portal", "WorkBuddy 用量"]
|
||
HARDCODE_DIRS = [
|
||
os.path.join("workbuddy_portal", "web", "templates"),
|
||
os.path.join("workbuddy_portal", "web", "static", "js"),
|
||
]
|
||
|
||
|
||
# ---------------------------------------------------------------- 基础工具
|
||
|
||
def read(path: str) -> str:
|
||
with open(path, "r", encoding="utf-8", errors="replace") as fh:
|
||
return fh.read()
|
||
|
||
|
||
def strip_code_blocks(text: str) -> str:
|
||
"""去掉围栏代码块内容 —— 里面的 `#` 不是标题,里面的链接不该被当链接。
|
||
|
||
用空串占位(保留换行),这样**行号不会错位**,报错才能定位到真实位置。
|
||
"""
|
||
out, in_fence = [], False
|
||
for line in text.splitlines():
|
||
if FENCE_RE.match(line):
|
||
in_fence = not in_fence
|
||
out.append("")
|
||
continue
|
||
out.append("" if in_fence else line)
|
||
return "\n".join(out)
|
||
|
||
|
||
def slugify(text: str) -> str:
|
||
"""把标题/锚点文本转成可比对的 key。
|
||
|
||
**刻意不逐字复刻 GitHub/Gitea 的 slug 算法。** 各家的标点处理规则并不一致
|
||
(GitHub 会删 `+`/`:` 却保留 `、`/`:`,而「连续空格是折叠成一个连字符
|
||
还是每个空格一个连字符」也随实现而变)。照着某个实现写死,换个托管平台
|
||
就会批量误报,反而让真问题淹掉。
|
||
|
||
这里只保留「有效字符」:小写字母、数字、汉字。标点、空格、连字符**全部丢弃**。
|
||
于是 `八、配置系统:写时校验 + 读时兜底` 与链接里的
|
||
`八配置系统写时校验--读时兜底` 归一后相等 —— 章节被重命名时,
|
||
有效字符会变,锚点仍然照抓不误。
|
||
|
||
代价:仅标点不同的两个标题会被视为同一锚点(极端罕见,可接受)。
|
||
"""
|
||
return re.sub(r"[^0-9a-z\u4e00-\u9fff]", "", text.lower())
|
||
|
||
|
||
def is_external(target: str) -> bool:
|
||
"""外部链接(http: / mailto: 等)不检查。"""
|
||
return bool(re.match(r"^[a-zA-Z][a-zA-Z0-9+.-]*:", target))
|
||
|
||
|
||
def looks_like_placeholder(segment: str) -> bool:
|
||
return bool(PLACEHOLDER_RE.search(segment.strip()))
|
||
|
||
|
||
def discover(root: str) -> list[str]:
|
||
"""递归找出所有 .md,返回相对 root 的 posix 路径。"""
|
||
found: list[str] = []
|
||
for dirpath, dirnames, filenames in os.walk(root):
|
||
dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS]
|
||
for fn in filenames:
|
||
if fn.lower().endswith((".md", ".markdown")):
|
||
rel = os.path.relpath(os.path.join(dirpath, fn), root)
|
||
found.append(rel.replace("\\", "/"))
|
||
return sorted(found)
|
||
|
||
|
||
_anchor_cache: dict[str, set[str]] = {}
|
||
|
||
|
||
def anchors_of(abs_path: str) -> set[str]:
|
||
"""一个 md 文件里所有可跳转的锚点(标题 + 显式 HTML 锚点)。"""
|
||
if abs_path not in _anchor_cache:
|
||
text = strip_code_blocks(read(abs_path))
|
||
got: set[str] = set()
|
||
for line in text.splitlines():
|
||
m = HEADING_RE.match(line)
|
||
if m:
|
||
got.add(slugify(m.group(2)))
|
||
for m in HTML_ANCHOR_RE.finditer(text):
|
||
got.add(slugify(m.group(1)))
|
||
_anchor_cache[abs_path] = got
|
||
return _anchor_cache[abs_path]
|
||
|
||
|
||
# ---------------------------------------------------------------- 各检查项
|
||
|
||
def check_links(root: str, files: list[str]) -> list[str]:
|
||
problems: list[str] = []
|
||
for rel in files:
|
||
abs_path = os.path.join(root, rel.replace("/", os.sep))
|
||
base_dir = os.path.dirname(abs_path)
|
||
text = strip_code_blocks(read(abs_path))
|
||
for lineno, line in enumerate(text.splitlines(), 1):
|
||
for m in LINK_RE.finditer(line):
|
||
target = m.group(2)
|
||
if is_external(target):
|
||
continue
|
||
|
||
# 纯页内锚点:指回本文件
|
||
if target.startswith("#"):
|
||
frag = target[1:]
|
||
if frag and slugify(frag) not in anchors_of(abs_path):
|
||
problems.append(f"{rel}:{lineno} 页内锚点失效 {target}")
|
||
continue
|
||
|
||
path_part, _, frag = target.partition("#")
|
||
if not path_part:
|
||
continue
|
||
resolved = os.path.normpath(os.path.join(base_dir, path_part))
|
||
if not os.path.exists(resolved):
|
||
problems.append(f"{rel}:{lineno} 链接目标不存在 {target}")
|
||
continue
|
||
if frag and resolved.lower().endswith((".md", ".markdown")):
|
||
if slugify(frag) not in anchors_of(resolved):
|
||
problems.append(f"{rel}:{lineno} 跨文件锚点失效 {target}")
|
||
return problems
|
||
|
||
|
||
def check_images(root: str, files: list[str]) -> list[str]:
|
||
problems: list[str] = []
|
||
for rel in files:
|
||
abs_path = os.path.join(root, rel.replace("/", os.sep))
|
||
base_dir = os.path.dirname(abs_path)
|
||
text = strip_code_blocks(read(abs_path))
|
||
for lineno, line in enumerate(text.splitlines(), 1):
|
||
for m in IMG_RE.finditer(line):
|
||
src = m.group(1)
|
||
if is_external(src):
|
||
continue
|
||
resolved = os.path.normpath(os.path.join(base_dir, src))
|
||
if not os.path.exists(resolved):
|
||
problems.append(f"{rel}:{lineno} 图片不存在 {src}")
|
||
return problems
|
||
|
||
|
||
def check_path_leaks(root: str, files: list[str]) -> list[str]:
|
||
"""扫绝对路径 —— 文档里出现多半是从本机命令里抄进来的(会连带泄漏用户名)。
|
||
|
||
命中后请**逐条人工判断**:占位符(`C:\\Users\\<用户名>`)已豁免,
|
||
但 `os.path.join(home, "AppData", ...)` 这类合法的路径发现代码不会命中
|
||
(它不含盘符或 `/home/` 前缀)。这里只报告,不自动修改。
|
||
"""
|
||
problems: list[str] = []
|
||
for rel in files:
|
||
abs_path = os.path.join(root, rel.replace("/", os.sep))
|
||
for lineno, line in enumerate(read(abs_path).splitlines(), 1):
|
||
for pat, label in LEAK_PATTERNS:
|
||
for m in pat.finditer(line):
|
||
seg = m.group(1) if m.groups() else ""
|
||
if seg and looks_like_placeholder(seg):
|
||
continue
|
||
problems.append(f"{rel}:{lineno} {label} {m.group(0)}")
|
||
return problems
|
||
|
||
|
||
def check_versions(root: str) -> list[str]:
|
||
"""版本号四处(__init__ / Dockerfile / README / CHANGELOG)必须一致。"""
|
||
found: dict[str, str] = {}
|
||
|
||
sources = [
|
||
(VERSION_INIT, INIT_VER_RE),
|
||
(DOCKERFILE, OCI_VER_RE),
|
||
("README.md", README_VER_RE),
|
||
(os.path.join("docs", "CHANGELOG.md"), CHANGELOG_VER_RE),
|
||
]
|
||
for rel, pat in sources:
|
||
p = os.path.join(root, rel)
|
||
if os.path.isfile(p):
|
||
m = pat.search(read(p))
|
||
if m:
|
||
found[rel.replace("\\", "/")] = m.group(1)
|
||
|
||
if not found:
|
||
return ["版本一致性: 一个版本号都没找到,检查脚本本身"]
|
||
|
||
if len(set(found.values())) > 1:
|
||
detail = "、".join("%s=%s" % (k, v) for k, v in sorted(found.items()))
|
||
return ["版本不一致: %s" % detail]
|
||
return []
|
||
|
||
|
||
def check_name_hardcode(root: str) -> list[str]:
|
||
"""产品名应走上下文变量(config.PROJECT_NAME),不该硬编码进模板/JS。"""
|
||
problems: list[str] = []
|
||
for d in HARDCODE_DIRS:
|
||
full = os.path.join(root, d)
|
||
if not os.path.isdir(full):
|
||
continue
|
||
for dirpath, _dirnames, filenames in os.walk(full):
|
||
for fn in filenames:
|
||
if not fn.lower().endswith((".html", ".js")):
|
||
continue
|
||
fp = os.path.join(dirpath, fn)
|
||
rel = os.path.relpath(fp, root).replace("\\", "/")
|
||
for i, line in enumerate(read(fp).splitlines(), 1):
|
||
for n in HARDCODE_NEEDLES:
|
||
if n in line:
|
||
problems.append(
|
||
f"{rel}:{i} 疑似硬编码产品名「{n}」(应走上下文变量)")
|
||
return problems
|
||
|
||
|
||
# ---------------------------------------------------------------- main
|
||
|
||
def main() -> int:
|
||
ap = argparse.ArgumentParser(description="文档自检")
|
||
ap.add_argument("--root", default=".", help="仓库根(默认当前目录)")
|
||
ap.add_argument("--no-fail", action="store_true",
|
||
help="即使发现问题也返回 0(仅用于人工查看报告)")
|
||
ap.add_argument("--strict", action="store_true", help=argparse.SUPPRESS)
|
||
ap.add_argument("--quiet", action="store_true", help="只打印问题")
|
||
args = ap.parse_args()
|
||
|
||
root = os.path.abspath(args.root)
|
||
if not os.path.isdir(root):
|
||
print("--root 不是目录: %s" % root)
|
||
return 1
|
||
|
||
files = discover(root)
|
||
if not files:
|
||
print("没找到任何 .md 文件。--root 是否正确?"
|
||
"(注意:Git Bash 的 /tmp/x 交给原生 Python 会变成 C:\\tmp\\x)")
|
||
return 1
|
||
|
||
sections = [
|
||
("内部链接与锚点", check_links(root, files)),
|
||
("图片引用", check_images(root, files)),
|
||
("绝对路径泄漏", check_path_leaks(root, files)),
|
||
("版本一致性", check_versions(root)),
|
||
("产品名硬编码", check_name_hardcode(root)),
|
||
]
|
||
|
||
total = sum(len(p) for _, p in sections)
|
||
|
||
if not args.quiet:
|
||
print("扫描 %d 个 Markdown 文件:" % len(files))
|
||
for rel in files:
|
||
print(" - %s" % rel)
|
||
print()
|
||
|
||
for title, problems in sections:
|
||
if args.quiet and not problems:
|
||
continue
|
||
print("=== %s ===" % title)
|
||
if problems:
|
||
for p in problems:
|
||
print(" [!!] %s" % p)
|
||
else:
|
||
print(" OK")
|
||
print()
|
||
|
||
print("RESULT: %d 处问题" % total)
|
||
if total == 0:
|
||
print("文档自检全部通过。")
|
||
# 默认「有问题就失败」——「报出 7 处问题却返回 0」是个静默无用的陷阱,
|
||
# 想只看报告请显式加 --no-fail。
|
||
if total and not args.no_fail:
|
||
return 1
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|