chore: 项目定名为 workbuddy-portal,容器化并补齐文档体系

## 项目定名
- 目录 wb_usage_portal → workbuddy-portal
- Python 包 wb_usage → workbuddy_portal(含 session cookie 名)
- 界面品牌统一为 WorkBuddy Portal;项目标识收敛到 config 单一来源

## 容器化
- Dockerfile:多阶段构建,依赖层与源码解耦;非 root(uid 1000);内置健康检查
- docker-compose.yml:单服务 + 绑定挂载 data/logs + 日志轮转 + TZ
- docker/entrypoint.sh:幂等初始化 → exec serve(LF 行尾,已由 .gitattributes 锁定)
- docker/healthcheck.py:纯标准库探活 /login(slim 镜像无 curl)
- .dockerignore / .env.example;数据目录可用 WB_DATA_DIR 等环境变量覆盖

## 文档
- docs/USER-GUIDE.md    用户使用手册(含 9 张真实界面截图)
- docs/DEPLOYMENT.md    部署运维(Docker / 裸机 / 反代 / 备份 / 推 Gitea 注册表)
- docs/ARCHITECTURE.md  架构与设计说明(含已知坑与红线、验证体系)
- docs/API.md           接口参考(路径 / 参数 / 返回结构 / 错误码)
- docs/FAQ.md           常见问题;docs/CHANGELOG.md 变更日志

## 修复缺陷(8)
1. /records/export 必然 500:生成器在请求上下文销毁后才迭代,改用自建连接
2. 大屏页图表全白:相对路径把 echarts.min.js 解析成 /vendor/... → 404
3. /users 500:路由已注册但模板缺失
4. 明细页日期筛选失效:视图传 f.frm、模板读 f.from
5. 配置页维护按钮全死:调用了不存在的 WBU.bindMaint()
6. 审计只能看最近 40 条:LIMIT 写死
7. 明细页多跑一条无用 SELECT:day_list() 取了没人用
8. 登录页锁定阈值未从配置注入

## 安全加固
- 新增 safe_next():拒绝 //evil.com 等协议相对 URL 的开放重定向
- 缺 CSRF 的写请求统一 400
- 默认开启云端 HTTPS 证书校验(ssl_verify=1);Cookie 是账号凭证
- 登录失败计数表加上限与 TTL
- /logout 拆分为 POST(执行) + GET(仅提示),防 <img src=/logout> 静默退出
- settings 内部簿记键 slot:* 读写两侧过滤,不再从 /api/settings 泄漏

## 内部质量与工具
- 设置项写时校验 + 读时兜底,杜绝「一个手滑的数字让采集整个跑不起来」
- 全局 ValueError → 400:手写 query string 不再暴露 500 页面
- CSV 导出改 csv.writer 流式写入(原手工拼串,字段含逗号会串列)
- bundle 明细加 20000 上限并回传 recordsTotal/recordsTruncated,不静默丢数据
- tools/smoke.py 离线回归 99 项;tools/check_live.py 真实 HTTP 56 项
- tools/shots.py Playwright 逐页截图 + JS 报错收集

## 验证
- compileall 通过;smoke 99/99;对容器实例 check_live 56/56;截图 0 JS 报错
- 容器内采集实测成功(trigger=startup 补跑:新增 11 条)
这个提交包含在:
2026-09-14 14:55:50 +08:00
当前提交 86631ae7ab
共修改 58 个文件,包含 9409 行新增和 0 行删除
+318
查看文件
@@ -0,0 +1,318 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""离线回归:用 Flask test_client 对**真实库**做全页面只读渲染 + 缺陷防回归断言。
与 tools/check_live.py 的分工:
* check_live.py 对运行中的服务发真实 HTTP,验「起没起来、登录/CSRF/API 通不通」
* smoke.py(本脚本)不发网络请求,直接把请求灌进 WSGI 应用,
因此能覆盖到「页面模板渲染是否正确」,且不需要先起服务、不需要密码。
覆盖内容:
1. 全页面渲染(含 /users,需管理员身份)——模板报错会直接暴露成 500
2. 模板未渲染残留(HTML 里出现 {{ / {% 说明有变量名写错)
3. 历史缺陷防回归(见下 REGRESSIONS)
4. CSV 导出可被标准 csv 解析、列数一致
5. 页面 HTML 里的 class 与 app.css 的选择器做差集(抓类名拼写错误)
用法:
cd workbuddy-portal
python tools/smoke.py
退出码:0 全通过;1 有失败项。
"""
from __future__ import annotations
import csv
import io
import json
import os
import re
import sys
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, BASE)
OK = 0
FAIL = 0
FAILS: list[str] = []
NOTES: list[str] = []
def chk(name: str, cond: bool, extra: str = "") -> bool:
global OK, FAIL
if cond:
OK += 1
print(" [OK] %s %s" % (name, extra))
else:
FAIL += 1
FAILS.append(name)
print(" [FAIL] %s %s" % (name, extra))
return bool(cond)
def note(msg: str) -> None:
NOTES.append(msg)
print(" [note] %s" % msg)
def login(cli, admin=True):
"""注入会话绕过登录:GET 不触发 CSRF,因此可直接测页面渲染。"""
with cli.session_transaction() as s:
s["uid"] = 1
s["uname"] = "admin" if admin else "viewer"
s["dname"] = "管理员" if admin else "只读账号"
s["adm"] = 1 if admin else 0
s["_csrf"] = "smoke-csrf-token"
def page(cli, path, method="GET", **kw):
r = getattr(cli, method.lower())(path, **kw)
return r.status_code, r.get_data(as_text=True)
def run() -> None:
from workbuddy_portal import create_app, db, query
print("== 0. 构建应用 ==")
app = create_app(start_scheduler=False, do_init_db=False)
app.config["WTF_CSRF_ENABLED"] = False
n_routes = len([r for r in app.url_map.iter_rules()])
chk("create_app 成功", app is not None)
chk("路由数量 >= 35", n_routes >= 35, "routes=%d" % n_routes)
# ---------------- 1. 未登录 ----------------
print("== 1. 未登录:受保护页应跳登录、API 应 401 ==")
with app.test_client() as cli:
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users"):
st, _ = page(cli, p)
chk("GET %-10s 未登录=302" % p, st == 302, "status=%s" % st)
for p in ("/api/summary", "/api/users", "/api/settings", "/api/audit"):
st, _ = page(cli, p)
chk("GET %-14s 未登录=401" % p, st == 401, "status=%s" % st)
st, html = page(cli, "/login")
chk("登录页含 CSRF 隐藏域", 'name="_csrf"' in html)
# ---------------- 2. 管理员:全页面渲染 ----------------
print("== 2. 管理员:全页面渲染 ==")
with app.test_client() as cli:
login(cli, admin=True)
pages = [
("/", "概览"), ("/records", "数据明细"), ("/tasks", "任务管理"),
("/config", "配置管理"), ("/logs", "日志管理"), ("/users", "用户管理"),
("/dashboard", "<html"),
]
for p, kw in pages:
st, html = page(cli, p)
ok = chk("GET %-10s 200" % p, st == 200, "status=%s len=%d" % (st, len(html)))
if ok:
chk(" └ 含关键字 %s" % kw, kw in html)
chk(" └ 无模板残留 {{ / {%%", "{{" not in html and "{%" not in html)
chk(" └ 含导航栏", "topbar" in html or p == "/dashboard")
st, html = page(cli, "/users")
chk("用户管理页列出账号", 'data-uid=' in html, "含行内编辑按钮")
chk("用户管理页含新建表单", 'id="formNewUser"' in html)
chk("用户管理页含审计表", "用户操作审计" in html)
# 配置页的维护按钮 + 大屏回后台入口
st, cfg = page(cli, "/config")
chk("配置页含维护按钮组", cfg.count("data-maint=") >= 3, "n=%d" % cfg.count("data-maint="))
chk("配置页含 TLS 校验下拉", 'name="ssl_verify"' in cfg)
st, rec = page(cli, "/records")
chk("明细页含快捷区间", 'data-range="today"' in rec and 'data-range="30d"' in rec)
chk("明细页表格包在 .tablewrap", "tablewrap" in rec)
# ---------------- 2b. 静态资源引用可解析 ----------------
print("== 2b. 页面引用的静态资源全部可达 ==")
asset_re = re.compile(r"\.(?:js|css|svg|png|jpe?g|gif|webp|ico|woff2?)(?:\?|$)", re.I)
with app.test_client() as cli:
login(cli, admin=True)
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users", "/dashboard"):
_, html = page(cli, p)
# 先剥掉 HTML 注释:注释里常写示例路径(src="vendor/x.js"),
# 不剥会把示例当真实引用误报。
html = re.sub(r"<!--.*?-->", "", html, flags=re.S)
refs = set(re.findall(r'src="([^"]+)"', html))
refs |= {h for h in re.findall(r'href="([^"]+)"', html) if asset_re.search(h)}
bad, n = [], 0
for r in sorted(refs):
if r.startswith(("data:", "http:", "https:", "//", "#")):
continue
target = r
if not r.startswith("/"): # 相对路径按该页 URL 解析(曾因此 404)
target = (p if p.endswith("/") else p.rsplit("/", 1)[0] + "/") + r
n += 1
ast, _ = page(cli, target)
if ast != 200:
bad.append("%s -> %s(%s)" % (r, target, ast))
chk("%-11s 资源引用全部 200" % p, not bad,
("坏引用=%s" % bad) if bad else "%d 个引用" % n)
# ---------------- 3. 非管理员:权限边界 ----------------
print("== 3. 非管理员:/users 必须 403,导航不出现该入口 ==")
with app.test_client() as cli:
login(cli, admin=False)
st, html = page(cli, "/users")
chk("GET /users 非管理员=403", st == 403, "status=%s" % st)
st, _ = page(cli, "/api/users")
chk("GET /api/users 非管理员=403", st == 403, "status=%s" % st)
st, _ = page(cli, "/api/users")
st, html = page(cli, "/")
chk("概览导航不含「用户管理」", "用户管理" not in html)
for p in ("/", "/records", "/tasks", "/logs"):
st, _ = page(cli, p)
chk("GET %-10s 非管理员=200" % p, st == 200, "status=%s" % st)
# ---------------- 4. 历史缺陷防回归 ----------------
print("== 4. 历史缺陷防回归 ==")
with app.test_client() as cli:
login(cli, admin=True)
# ① 非法日期曾 500
st, body = page(cli, "/api/summary?from=abc&to=def")
chk("① /api/summary 非法日期=400", st == 400, "status=%s" % st)
chk(" └ 返回 JSON 错误体", '"ok": false' in body.replace('":', '": '))
# ② /tasks 非法页码曾 500
st, _ = page(cli, "/tasks?page=abc")
chk("② /tasks?page=abc=200", st == 200, "status=%s" % st)
# ③ 日志尾部非法行数曾 500
st, _ = page(cli, "/logs/tail?lines=abc")
chk("③ /logs/tail?lines=abc=200", st == 200, "status=%s" % st)
# ④ 内部簿记键 slot:* 曾泄漏到 /api/settings
st, body = page(cli, "/api/settings")
chk("④ /api/settings 无 slot:* 键", "slot:" not in body, "status=%s" % st)
# ⑤ 概览「云端」列曾因 SQL 少选列而恒为空
st, ov = page(cli, "/")
chk("⑤ 概览含「云端」列", "云端" in ov)
# ⑥ 明细页日期回填:模板曾读 f.from,导致输入框永远为空
st, rec = page(cli, "/records?from=2026-09-08&to=2026-09-10")
chk("⑥ 明细页回填 from", 'value="2026-09-08"' in rec)
chk("⑥ 明细页回填 to", 'value="2026-09-10"' in rec)
# ⑦ 导出链接必须带规范参数名 from(模板内部用 frm,拼 URL 时要换回来)
m = re.search(r'href="(/records/export[^"]*)"', rec)
chk("⑦ 导出链接存在", bool(m))
if m:
chk("⑦ 导出链接带 from=", "from=2026-09-08" in m.group(1), m.group(1))
chk("⑦ 导出链接带 to=", "to=2026-09-10" in m.group(1))
# ⑧ 导出接口本身也要认 from/to
st, csv_body = page(cli, "/records/export?from=2026-09-08&to=2026-09-10")
chk("⑧ GET /records/export=200", st == 200, "status=%s" % st)
rows = list(csv.reader(io.StringIO(csv_body.lstrip("\ufeff"))))
chk("⑧ CSV 至少含表头", len(rows) >= 1, "rows=%d" % len(rows))
chk("⑧ CSV 列数一致",
len({len(r) for r in rows if r}) == 1,
"列数集合=%s" % sorted({len(r) for r in rows if r}))
chk("⑧ CSV 表头为官方同构列",
rows and rows[0] == ["RequestID", "积分消耗", "User Prompt", "模型", "客户端", "时间"],
"header=%s" % (rows[0] if rows else None))
# ⑨ 非法设置必须在写入时被拒(曾让采集崩掉)
st, body = page(cli, "/api/settings", method="POST", json={"page_size": "abc"},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑨ 非法设置写入=400", st == 400, "status=%s" % st)
st, body = page(cli, "/api/settings", method="POST", json={"slot:09:00": "x"},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑨ 内部键被忽略而非写入",
st == 200 and "slot:09:00" in (json.loads(body).get("ignored") or []),
"status=%s body=%s" % (st, body[:140]))
# ⑩ 维护动作
st, _ = page(cli, "/api/maintenance/nope", method="POST", json={},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑩ 未知维护动作=404", st == 404, "status=%s" % st)
st, body = page(cli, "/api/maintenance/recount", method="POST", json={},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑩ recount=200", st == 200, "status=%s body=%s" % (st, body[:90]))
# ⑪ 非管理员调用户管理 API
st, _ = page(cli, "/api/users/1/delete", method="POST", json={},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑪ 删除自己=400(不允许)", st == 400, "status=%s" % st)
# ⑫ CSRF 缺失必须 400
st, _ = page(cli, "/api/settings", method="POST", json={"max_prompt": "100"})
chk("⑫ 缺 CSRF=400", st == 400, "status=%s" % st)
# ⑬ 分页/筛选参数非法不能 500
for p in ("/logs?page=abc", "/logs?apage=abc", "/records?page=abc&size=abc",
"/records?from=2026-13-99", "/logs?status=%27%20OR%201=1--"):
st, _ = page(cli, p)
chk("⑬ GET %-30s =200" % p, st == 200, "status=%s" % st)
# ⑭ 审计筛选:只返回指定动作,且计数与筛选一致
st, lg = page(cli, "/logs")
chk("⑭ 日志页含操作审计筛选", "操作审计" in lg and "seg" in lg)
m = re.search(r'href="/logs\?act=([^"&]+)', lg)
if chk("⑭ 审计筛选有可选动作", bool(m), "act=%s" % (m.group(1) if m else "无")):
act = m.group(1)
st, lg2 = page(cli, "/logs?act=" + act)
chk("⑭ 带 act=%s 仍 200" % act, st == 200, "status=%s" % st)
# 从审计表的 id 处切片:前面采集表里的 trigger 也用了 tag mute,不能混入
i = lg2.find('id="auditTable"')
tail = lg2[i:] if i >= 0 else ""
chk("⑭ 审计表存在", i >= 0)
tags = re.findall(r'<td><span class="tag mute">([^<]+)</span></td>', tail)
others = sorted({t for t in tags if t != act})
chk("⑭ 审计筛选结果不含其他动作", not others and bool(tags),
"命中=%d 混入=%s" % (len(tags), others))
# ---------------- 5. 数据自洽 ----------------
print("== 5. 数据自洽(只读) ==")
with app.test_client() as cli:
login(cli, admin=True)
mf = json.loads(page(cli, "/api/manifest")[1])
src = (mf.get("sources") or [{}])[0]
chk("manifest 存档条数 == 数据源条数",
mf["totals"]["records"] == src.get("count"),
"%s vs %s" % (mf["totals"]["records"], src.get("count")))
sm = json.loads(page(cli, "/api/summary")[1])
chk("summary 全量 calls == 存档条数",
sm.get("calls") == mf["totals"]["records"],
"calls=%s records=%s" % (sm.get("calls"), mf["totals"]["records"]))
chk("summary 全量 credits 自洽",
abs(float(sm.get("credits", 0)) - float(mf["totals"]["credits"])) < 0.005,
"%s vs %s" % (sm.get("credits"), mf["totals"]["credits"]))
d = query.daily(db.get_db())
chk("daily 逐日积分求和 == 存档总额",
abs(round(sum(float(x["c"]) for x in d), 2)
- round(float(mf["totals"]["credits"]), 2)) < 0.005)
chk("daily 逐日 h[24] 求和 == 当日积分",
all(abs(round(sum(x["h"]), 2) - round(x["c"], 2)) < 0.005 for x in d))
note("存档 %s 条 / %s 积分 / %d 天" % (mf["totals"]["records"],
mf["totals"]["credits"], len(d)))
# ---------------- 6. class 名与 CSS 选择器对账 ----------------
print("== 6. 页面 class 与 app.css 选择器对账 ==")
css = open(os.path.join(BASE, "workbuddy_portal", "web", "static", "css", "app.css"),
encoding="utf-8").read()
css_classes = set(re.findall(r"\.([A-Za-z][\w-]*)", css))
with app.test_client() as cli:
login(cli, admin=True)
used: set[str] = set()
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users", "/login"):
if p == "/login":
with app.test_client() as c2:
html = page(c2, p)[1]
else:
html = page(cli, p)[1]
for m in re.findall(r'class="([^"]*)"', html):
used.update(t for t in m.split() if t)
# 允许的无样式类:JS 钩子、第三方/语义标记
allow = {"no-js", "on", "cur", "gap", "meta", "unit", "field-err"}
missing = sorted(c for c in used - css_classes - allow)
chk("无「用了但 CSS 里不存在」的类名", not missing, "缺失=%s" % missing if missing else "")
def main() -> int:
print("工程目录:%s\n" % BASE)
try:
run()
except Exception as e: # noqa: BLE001
import traceback
traceback.print_exc()
print("\n[FATAL] 脚本本身异常:%s" % e)
return 1
print("\nRESULT: ok=%d fail=%d" % (OK, FAIL))
if NOTES:
print("备注:")
for n in NOTES:
print(" - " + n)
if FAILS:
print("失败项:\n - " + "\n - ".join(FAILS))
return 1 if FAIL else 0
if __name__ == "__main__":
sys.exit(main())