## 项目定名 - 目录 wb_usage_portal → workbuddy-portal - Python 包 wb_usage → workbuddy_portal(含 session cookie 名) - 界面品牌统一为 WorkBuddy Portal;项目标识收敛到 config 单一来源 ## 容器化 - Dockerfile:多阶段构建,依赖层与源码解耦;非 root(uid 1000);内置健康检查 - docker-compose.yml:单服务 + 绑定挂载 data/logs + 日志轮转 + TZ - docker/entrypoint.sh:幂等初始化 → exec serve(LF 行尾,已由 .gitattributes 锁定) - docker/healthcheck.py:纯标准库探活 /login(slim 镜像无 curl) - .dockerignore / .env.example;数据目录可用 WB_DATA_DIR 等环境变量覆盖 ## 文档 - docs/USER-GUIDE.md 用户使用手册(含 9 张真实界面截图) - docs/DEPLOYMENT.md 部署运维(Docker / 裸机 / 反代 / 备份 / 推 Gitea 注册表) - docs/ARCHITECTURE.md 架构与设计说明(含已知坑与红线、验证体系) - docs/API.md 接口参考(路径 / 参数 / 返回结构 / 错误码) - docs/FAQ.md 常见问题;docs/CHANGELOG.md 变更日志 ## 修复缺陷(8) 1. /records/export 必然 500:生成器在请求上下文销毁后才迭代,改用自建连接 2. 大屏页图表全白:相对路径把 echarts.min.js 解析成 /vendor/... → 404 3. /users 500:路由已注册但模板缺失 4. 明细页日期筛选失效:视图传 f.frm、模板读 f.from 5. 配置页维护按钮全死:调用了不存在的 WBU.bindMaint() 6. 审计只能看最近 40 条:LIMIT 写死 7. 明细页多跑一条无用 SELECT:day_list() 取了没人用 8. 登录页锁定阈值未从配置注入 ## 安全加固 - 新增 safe_next():拒绝 //evil.com 等协议相对 URL 的开放重定向 - 缺 CSRF 的写请求统一 400 - 默认开启云端 HTTPS 证书校验(ssl_verify=1);Cookie 是账号凭证 - 登录失败计数表加上限与 TTL - /logout 拆分为 POST(执行) + GET(仅提示),防 <img src=/logout> 静默退出 - settings 内部簿记键 slot:* 读写两侧过滤,不再从 /api/settings 泄漏 ## 内部质量与工具 - 设置项写时校验 + 读时兜底,杜绝「一个手滑的数字让采集整个跑不起来」 - 全局 ValueError → 400:手写 query string 不再暴露 500 页面 - CSV 导出改 csv.writer 流式写入(原手工拼串,字段含逗号会串列) - bundle 明细加 20000 上限并回传 recordsTotal/recordsTruncated,不静默丢数据 - tools/smoke.py 离线回归 99 项;tools/check_live.py 真实 HTTP 56 项 - tools/shots.py Playwright 逐页截图 + JS 报错收集 ## 验证 - compileall 通过;smoke 99/99;对容器实例 check_live 56/56;截图 0 JS 报错 - 容器内采集实测成功(trigger=startup 补跑:新增 11 条)
319 行
15 KiB
Python
319 行
15 KiB
Python
#!/usr/bin/env python
|
||
# -*- coding: utf-8 -*-
|
||
"""离线回归:用 Flask test_client 对**真实库**做全页面只读渲染 + 缺陷防回归断言。
|
||
|
||
与 tools/check_live.py 的分工:
|
||
* check_live.py 对运行中的服务发真实 HTTP,验「起没起来、登录/CSRF/API 通不通」
|
||
* smoke.py(本脚本)不发网络请求,直接把请求灌进 WSGI 应用,
|
||
因此能覆盖到「页面模板渲染是否正确」,且不需要先起服务、不需要密码。
|
||
|
||
覆盖内容:
|
||
1. 全页面渲染(含 /users,需管理员身份)——模板报错会直接暴露成 500
|
||
2. 模板未渲染残留(HTML 里出现 {{ / {% 说明有变量名写错)
|
||
3. 历史缺陷防回归(见下 REGRESSIONS)
|
||
4. CSV 导出可被标准 csv 解析、列数一致
|
||
5. 页面 HTML 里的 class 与 app.css 的选择器做差集(抓类名拼写错误)
|
||
|
||
用法:
|
||
cd workbuddy-portal
|
||
python tools/smoke.py
|
||
退出码:0 全通过;1 有失败项。
|
||
"""
|
||
from __future__ import annotations
|
||
|
||
import csv
|
||
import io
|
||
import json
|
||
import os
|
||
import re
|
||
import sys
|
||
|
||
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||
sys.path.insert(0, BASE)
|
||
|
||
OK = 0
|
||
FAIL = 0
|
||
FAILS: list[str] = []
|
||
NOTES: list[str] = []
|
||
|
||
|
||
def chk(name: str, cond: bool, extra: str = "") -> bool:
|
||
global OK, FAIL
|
||
if cond:
|
||
OK += 1
|
||
print(" [OK] %s %s" % (name, extra))
|
||
else:
|
||
FAIL += 1
|
||
FAILS.append(name)
|
||
print(" [FAIL] %s %s" % (name, extra))
|
||
return bool(cond)
|
||
|
||
|
||
def note(msg: str) -> None:
|
||
NOTES.append(msg)
|
||
print(" [note] %s" % msg)
|
||
|
||
|
||
def login(cli, admin=True):
|
||
"""注入会话绕过登录:GET 不触发 CSRF,因此可直接测页面渲染。"""
|
||
with cli.session_transaction() as s:
|
||
s["uid"] = 1
|
||
s["uname"] = "admin" if admin else "viewer"
|
||
s["dname"] = "管理员" if admin else "只读账号"
|
||
s["adm"] = 1 if admin else 0
|
||
s["_csrf"] = "smoke-csrf-token"
|
||
|
||
|
||
def page(cli, path, method="GET", **kw):
|
||
r = getattr(cli, method.lower())(path, **kw)
|
||
return r.status_code, r.get_data(as_text=True)
|
||
|
||
|
||
def run() -> None:
|
||
from workbuddy_portal import create_app, db, query
|
||
|
||
print("== 0. 构建应用 ==")
|
||
app = create_app(start_scheduler=False, do_init_db=False)
|
||
app.config["WTF_CSRF_ENABLED"] = False
|
||
n_routes = len([r for r in app.url_map.iter_rules()])
|
||
chk("create_app 成功", app is not None)
|
||
chk("路由数量 >= 35", n_routes >= 35, "routes=%d" % n_routes)
|
||
|
||
# ---------------- 1. 未登录 ----------------
|
||
print("== 1. 未登录:受保护页应跳登录、API 应 401 ==")
|
||
with app.test_client() as cli:
|
||
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users"):
|
||
st, _ = page(cli, p)
|
||
chk("GET %-10s 未登录=302" % p, st == 302, "status=%s" % st)
|
||
for p in ("/api/summary", "/api/users", "/api/settings", "/api/audit"):
|
||
st, _ = page(cli, p)
|
||
chk("GET %-14s 未登录=401" % p, st == 401, "status=%s" % st)
|
||
st, html = page(cli, "/login")
|
||
chk("登录页含 CSRF 隐藏域", 'name="_csrf"' in html)
|
||
|
||
# ---------------- 2. 管理员:全页面渲染 ----------------
|
||
print("== 2. 管理员:全页面渲染 ==")
|
||
with app.test_client() as cli:
|
||
login(cli, admin=True)
|
||
pages = [
|
||
("/", "概览"), ("/records", "数据明细"), ("/tasks", "任务管理"),
|
||
("/config", "配置管理"), ("/logs", "日志管理"), ("/users", "用户管理"),
|
||
("/dashboard", "<html"),
|
||
]
|
||
for p, kw in pages:
|
||
st, html = page(cli, p)
|
||
ok = chk("GET %-10s 200" % p, st == 200, "status=%s len=%d" % (st, len(html)))
|
||
if ok:
|
||
chk(" └ 含关键字 %s" % kw, kw in html)
|
||
chk(" └ 无模板残留 {{ / {%%", "{{" not in html and "{%" not in html)
|
||
chk(" └ 含导航栏", "topbar" in html or p == "/dashboard")
|
||
|
||
st, html = page(cli, "/users")
|
||
chk("用户管理页列出账号", 'data-uid=' in html, "含行内编辑按钮")
|
||
chk("用户管理页含新建表单", 'id="formNewUser"' in html)
|
||
chk("用户管理页含审计表", "用户操作审计" in html)
|
||
|
||
# 配置页的维护按钮 + 大屏回后台入口
|
||
st, cfg = page(cli, "/config")
|
||
chk("配置页含维护按钮组", cfg.count("data-maint=") >= 3, "n=%d" % cfg.count("data-maint="))
|
||
chk("配置页含 TLS 校验下拉", 'name="ssl_verify"' in cfg)
|
||
st, rec = page(cli, "/records")
|
||
chk("明细页含快捷区间", 'data-range="today"' in rec and 'data-range="30d"' in rec)
|
||
chk("明细页表格包在 .tablewrap", "tablewrap" in rec)
|
||
|
||
# ---------------- 2b. 静态资源引用可解析 ----------------
|
||
print("== 2b. 页面引用的静态资源全部可达 ==")
|
||
asset_re = re.compile(r"\.(?:js|css|svg|png|jpe?g|gif|webp|ico|woff2?)(?:\?|$)", re.I)
|
||
with app.test_client() as cli:
|
||
login(cli, admin=True)
|
||
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users", "/dashboard"):
|
||
_, html = page(cli, p)
|
||
# 先剥掉 HTML 注释:注释里常写示例路径(src="vendor/x.js"),
|
||
# 不剥会把示例当真实引用误报。
|
||
html = re.sub(r"<!--.*?-->", "", html, flags=re.S)
|
||
refs = set(re.findall(r'src="([^"]+)"', html))
|
||
refs |= {h for h in re.findall(r'href="([^"]+)"', html) if asset_re.search(h)}
|
||
bad, n = [], 0
|
||
for r in sorted(refs):
|
||
if r.startswith(("data:", "http:", "https:", "//", "#")):
|
||
continue
|
||
target = r
|
||
if not r.startswith("/"): # 相对路径按该页 URL 解析(曾因此 404)
|
||
target = (p if p.endswith("/") else p.rsplit("/", 1)[0] + "/") + r
|
||
n += 1
|
||
ast, _ = page(cli, target)
|
||
if ast != 200:
|
||
bad.append("%s -> %s(%s)" % (r, target, ast))
|
||
chk("%-11s 资源引用全部 200" % p, not bad,
|
||
("坏引用=%s" % bad) if bad else "%d 个引用" % n)
|
||
|
||
# ---------------- 3. 非管理员:权限边界 ----------------
|
||
print("== 3. 非管理员:/users 必须 403,导航不出现该入口 ==")
|
||
with app.test_client() as cli:
|
||
login(cli, admin=False)
|
||
st, html = page(cli, "/users")
|
||
chk("GET /users 非管理员=403", st == 403, "status=%s" % st)
|
||
st, _ = page(cli, "/api/users")
|
||
chk("GET /api/users 非管理员=403", st == 403, "status=%s" % st)
|
||
st, _ = page(cli, "/api/users")
|
||
st, html = page(cli, "/")
|
||
chk("概览导航不含「用户管理」", "用户管理" not in html)
|
||
for p in ("/", "/records", "/tasks", "/logs"):
|
||
st, _ = page(cli, p)
|
||
chk("GET %-10s 非管理员=200" % p, st == 200, "status=%s" % st)
|
||
|
||
# ---------------- 4. 历史缺陷防回归 ----------------
|
||
print("== 4. 历史缺陷防回归 ==")
|
||
with app.test_client() as cli:
|
||
login(cli, admin=True)
|
||
# ① 非法日期曾 500
|
||
st, body = page(cli, "/api/summary?from=abc&to=def")
|
||
chk("① /api/summary 非法日期=400", st == 400, "status=%s" % st)
|
||
chk(" └ 返回 JSON 错误体", '"ok": false' in body.replace('":', '": '))
|
||
# ② /tasks 非法页码曾 500
|
||
st, _ = page(cli, "/tasks?page=abc")
|
||
chk("② /tasks?page=abc=200", st == 200, "status=%s" % st)
|
||
# ③ 日志尾部非法行数曾 500
|
||
st, _ = page(cli, "/logs/tail?lines=abc")
|
||
chk("③ /logs/tail?lines=abc=200", st == 200, "status=%s" % st)
|
||
# ④ 内部簿记键 slot:* 曾泄漏到 /api/settings
|
||
st, body = page(cli, "/api/settings")
|
||
chk("④ /api/settings 无 slot:* 键", "slot:" not in body, "status=%s" % st)
|
||
# ⑤ 概览「云端」列曾因 SQL 少选列而恒为空
|
||
st, ov = page(cli, "/")
|
||
chk("⑤ 概览含「云端」列", "云端" in ov)
|
||
# ⑥ 明细页日期回填:模板曾读 f.from,导致输入框永远为空
|
||
st, rec = page(cli, "/records?from=2026-09-08&to=2026-09-10")
|
||
chk("⑥ 明细页回填 from", 'value="2026-09-08"' in rec)
|
||
chk("⑥ 明细页回填 to", 'value="2026-09-10"' in rec)
|
||
# ⑦ 导出链接必须带规范参数名 from(模板内部用 frm,拼 URL 时要换回来)
|
||
m = re.search(r'href="(/records/export[^"]*)"', rec)
|
||
chk("⑦ 导出链接存在", bool(m))
|
||
if m:
|
||
chk("⑦ 导出链接带 from=", "from=2026-09-08" in m.group(1), m.group(1))
|
||
chk("⑦ 导出链接带 to=", "to=2026-09-10" in m.group(1))
|
||
# ⑧ 导出接口本身也要认 from/to
|
||
st, csv_body = page(cli, "/records/export?from=2026-09-08&to=2026-09-10")
|
||
chk("⑧ GET /records/export=200", st == 200, "status=%s" % st)
|
||
rows = list(csv.reader(io.StringIO(csv_body.lstrip("\ufeff"))))
|
||
chk("⑧ CSV 至少含表头", len(rows) >= 1, "rows=%d" % len(rows))
|
||
chk("⑧ CSV 列数一致",
|
||
len({len(r) for r in rows if r}) == 1,
|
||
"列数集合=%s" % sorted({len(r) for r in rows if r}))
|
||
chk("⑧ CSV 表头为官方同构列",
|
||
rows and rows[0] == ["RequestID", "积分消耗", "User Prompt", "模型", "客户端", "时间"],
|
||
"header=%s" % (rows[0] if rows else None))
|
||
# ⑨ 非法设置必须在写入时被拒(曾让采集崩掉)
|
||
st, body = page(cli, "/api/settings", method="POST", json={"page_size": "abc"},
|
||
headers={"X-CSRF-Token": "smoke-csrf-token"})
|
||
chk("⑨ 非法设置写入=400", st == 400, "status=%s" % st)
|
||
st, body = page(cli, "/api/settings", method="POST", json={"slot:09:00": "x"},
|
||
headers={"X-CSRF-Token": "smoke-csrf-token"})
|
||
chk("⑨ 内部键被忽略而非写入",
|
||
st == 200 and "slot:09:00" in (json.loads(body).get("ignored") or []),
|
||
"status=%s body=%s" % (st, body[:140]))
|
||
# ⑩ 维护动作
|
||
st, _ = page(cli, "/api/maintenance/nope", method="POST", json={},
|
||
headers={"X-CSRF-Token": "smoke-csrf-token"})
|
||
chk("⑩ 未知维护动作=404", st == 404, "status=%s" % st)
|
||
st, body = page(cli, "/api/maintenance/recount", method="POST", json={},
|
||
headers={"X-CSRF-Token": "smoke-csrf-token"})
|
||
chk("⑩ recount=200", st == 200, "status=%s body=%s" % (st, body[:90]))
|
||
# ⑪ 非管理员调用户管理 API
|
||
st, _ = page(cli, "/api/users/1/delete", method="POST", json={},
|
||
headers={"X-CSRF-Token": "smoke-csrf-token"})
|
||
chk("⑪ 删除自己=400(不允许)", st == 400, "status=%s" % st)
|
||
# ⑫ CSRF 缺失必须 400
|
||
st, _ = page(cli, "/api/settings", method="POST", json={"max_prompt": "100"})
|
||
chk("⑫ 缺 CSRF=400", st == 400, "status=%s" % st)
|
||
# ⑬ 分页/筛选参数非法不能 500
|
||
for p in ("/logs?page=abc", "/logs?apage=abc", "/records?page=abc&size=abc",
|
||
"/records?from=2026-13-99", "/logs?status=%27%20OR%201=1--"):
|
||
st, _ = page(cli, p)
|
||
chk("⑬ GET %-30s =200" % p, st == 200, "status=%s" % st)
|
||
# ⑭ 审计筛选:只返回指定动作,且计数与筛选一致
|
||
st, lg = page(cli, "/logs")
|
||
chk("⑭ 日志页含操作审计筛选", "操作审计" in lg and "seg" in lg)
|
||
m = re.search(r'href="/logs\?act=([^"&]+)', lg)
|
||
if chk("⑭ 审计筛选有可选动作", bool(m), "act=%s" % (m.group(1) if m else "无")):
|
||
act = m.group(1)
|
||
st, lg2 = page(cli, "/logs?act=" + act)
|
||
chk("⑭ 带 act=%s 仍 200" % act, st == 200, "status=%s" % st)
|
||
# 从审计表的 id 处切片:前面采集表里的 trigger 也用了 tag mute,不能混入
|
||
i = lg2.find('id="auditTable"')
|
||
tail = lg2[i:] if i >= 0 else ""
|
||
chk("⑭ 审计表存在", i >= 0)
|
||
tags = re.findall(r'<td><span class="tag mute">([^<]+)</span></td>', tail)
|
||
others = sorted({t for t in tags if t != act})
|
||
chk("⑭ 审计筛选结果不含其他动作", not others and bool(tags),
|
||
"命中=%d 混入=%s" % (len(tags), others))
|
||
|
||
# ---------------- 5. 数据自洽 ----------------
|
||
print("== 5. 数据自洽(只读) ==")
|
||
with app.test_client() as cli:
|
||
login(cli, admin=True)
|
||
mf = json.loads(page(cli, "/api/manifest")[1])
|
||
src = (mf.get("sources") or [{}])[0]
|
||
chk("manifest 存档条数 == 数据源条数",
|
||
mf["totals"]["records"] == src.get("count"),
|
||
"%s vs %s" % (mf["totals"]["records"], src.get("count")))
|
||
sm = json.loads(page(cli, "/api/summary")[1])
|
||
chk("summary 全量 calls == 存档条数",
|
||
sm.get("calls") == mf["totals"]["records"],
|
||
"calls=%s records=%s" % (sm.get("calls"), mf["totals"]["records"]))
|
||
chk("summary 全量 credits 自洽",
|
||
abs(float(sm.get("credits", 0)) - float(mf["totals"]["credits"])) < 0.005,
|
||
"%s vs %s" % (sm.get("credits"), mf["totals"]["credits"]))
|
||
d = query.daily(db.get_db())
|
||
chk("daily 逐日积分求和 == 存档总额",
|
||
abs(round(sum(float(x["c"]) for x in d), 2)
|
||
- round(float(mf["totals"]["credits"]), 2)) < 0.005)
|
||
chk("daily 逐日 h[24] 求和 == 当日积分",
|
||
all(abs(round(sum(x["h"]), 2) - round(x["c"], 2)) < 0.005 for x in d))
|
||
note("存档 %s 条 / %s 积分 / %d 天" % (mf["totals"]["records"],
|
||
mf["totals"]["credits"], len(d)))
|
||
|
||
# ---------------- 6. class 名与 CSS 选择器对账 ----------------
|
||
print("== 6. 页面 class 与 app.css 选择器对账 ==")
|
||
css = open(os.path.join(BASE, "workbuddy_portal", "web", "static", "css", "app.css"),
|
||
encoding="utf-8").read()
|
||
css_classes = set(re.findall(r"\.([A-Za-z][\w-]*)", css))
|
||
with app.test_client() as cli:
|
||
login(cli, admin=True)
|
||
used: set[str] = set()
|
||
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users", "/login"):
|
||
if p == "/login":
|
||
with app.test_client() as c2:
|
||
html = page(c2, p)[1]
|
||
else:
|
||
html = page(cli, p)[1]
|
||
for m in re.findall(r'class="([^"]*)"', html):
|
||
used.update(t for t in m.split() if t)
|
||
# 允许的无样式类:JS 钩子、第三方/语义标记
|
||
allow = {"no-js", "on", "cur", "gap", "meta", "unit", "field-err"}
|
||
missing = sorted(c for c in used - css_classes - allow)
|
||
chk("无「用了但 CSS 里不存在」的类名", not missing, "缺失=%s" % missing if missing else "")
|
||
|
||
|
||
def main() -> int:
|
||
print("工程目录:%s\n" % BASE)
|
||
try:
|
||
run()
|
||
except Exception as e: # noqa: BLE001
|
||
import traceback
|
||
traceback.print_exc()
|
||
print("\n[FATAL] 脚本本身异常:%s" % e)
|
||
return 1
|
||
print("\nRESULT: ok=%d fail=%d" % (OK, FAIL))
|
||
if NOTES:
|
||
print("备注:")
|
||
for n in NOTES:
|
||
print(" - " + n)
|
||
if FAILS:
|
||
print("失败项:\n - " + "\n - ".join(FAILS))
|
||
return 1 if FAIL else 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
sys.exit(main())
|