chore: 项目定名为 workbuddy-portal,容器化并补齐文档体系

## 项目定名
- 目录 wb_usage_portal → workbuddy-portal
- Python 包 wb_usage → workbuddy_portal(含 session cookie 名)
- 界面品牌统一为 WorkBuddy Portal;项目标识收敛到 config 单一来源

## 容器化
- Dockerfile:多阶段构建,依赖层与源码解耦;非 root(uid 1000);内置健康检查
- docker-compose.yml:单服务 + 绑定挂载 data/logs + 日志轮转 + TZ
- docker/entrypoint.sh:幂等初始化 → exec serve(LF 行尾,已由 .gitattributes 锁定)
- docker/healthcheck.py:纯标准库探活 /login(slim 镜像无 curl)
- .dockerignore / .env.example;数据目录可用 WB_DATA_DIR 等环境变量覆盖

## 文档
- docs/USER-GUIDE.md    用户使用手册(含 9 张真实界面截图)
- docs/DEPLOYMENT.md    部署运维(Docker / 裸机 / 反代 / 备份 / 推 Gitea 注册表)
- docs/ARCHITECTURE.md  架构与设计说明(含已知坑与红线、验证体系)
- docs/API.md           接口参考(路径 / 参数 / 返回结构 / 错误码)
- docs/FAQ.md           常见问题;docs/CHANGELOG.md 变更日志

## 修复缺陷(8)
1. /records/export 必然 500:生成器在请求上下文销毁后才迭代,改用自建连接
2. 大屏页图表全白:相对路径把 echarts.min.js 解析成 /vendor/... → 404
3. /users 500:路由已注册但模板缺失
4. 明细页日期筛选失效:视图传 f.frm、模板读 f.from
5. 配置页维护按钮全死:调用了不存在的 WBU.bindMaint()
6. 审计只能看最近 40 条:LIMIT 写死
7. 明细页多跑一条无用 SELECT:day_list() 取了没人用
8. 登录页锁定阈值未从配置注入

## 安全加固
- 新增 safe_next():拒绝 //evil.com 等协议相对 URL 的开放重定向
- 缺 CSRF 的写请求统一 400
- 默认开启云端 HTTPS 证书校验(ssl_verify=1);Cookie 是账号凭证
- 登录失败计数表加上限与 TTL
- /logout 拆分为 POST(执行) + GET(仅提示),防 <img src=/logout> 静默退出
- settings 内部簿记键 slot:* 读写两侧过滤,不再从 /api/settings 泄漏

## 内部质量与工具
- 设置项写时校验 + 读时兜底,杜绝「一个手滑的数字让采集整个跑不起来」
- 全局 ValueError → 400:手写 query string 不再暴露 500 页面
- CSV 导出改 csv.writer 流式写入(原手工拼串,字段含逗号会串列)
- bundle 明细加 20000 上限并回传 recordsTotal/recordsTruncated,不静默丢数据
- tools/smoke.py 离线回归 99 项;tools/check_live.py 真实 HTTP 56 项
- tools/shots.py Playwright 逐页截图 + JS 报错收集

## 验证
- compileall 通过;smoke 99/99;对容器实例 check_live 56/56;截图 0 JS 报错
- 容器内采集实测成功(trigger=startup 补跑:新增 11 条)
这个提交包含在:
2026-09-14 14:55:50 +08:00
当前提交 86631ae7ab
共修改 58 个文件,包含 9409 行新增和 0 行删除
+297
查看文件
@@ -0,0 +1,297 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""端到端验收:对**运行中的**服务发真实 HTTP 请求,走完整登录/CSRF/API 链路。
与 tests 里用 Flask test_client 的冒烟测试互补——这里验证的是「真的起起来了、
真的能登录、真的能取到数」,适合部署到局域网后随手跑一遍。
用法:
python tools/check_live.py # 默认 http://127.0.0.1:8848
python tools/check_live.py --base http://10.0.0.5:8848
python tools/check_live.py -u admin -p 你的密码
python tools/check_live.py --from 2026-09-08 --to 2026-09-14
退出码:0 全通过;1 有失败项(会打印失败清单)。
注意:脚本会读取窗口数据但**不写库**(不触发采集、不改配置),可安全反复运行。
"""
from __future__ import annotations
import argparse
import http.cookiejar
import json
import re
import sys
import urllib.error
import urllib.parse
import urllib.request
from datetime import datetime
OK = 0
FAIL = 0
FAILS: list[str] = []
def _d(s: str):
"""把 YYYY-MM-DD 解析成本地 datetime(不用 date.fromisoformat 之外的时区处理)。"""
return datetime.strptime(s, "%Y-%m-%d")
def chk(name: str, cond: bool, extra: str = "") -> None:
global OK, FAIL
if cond:
OK += 1
print(" [OK] %s %s" % (name, extra))
else:
FAIL += 1
FAILS.append(name)
print(" [FAIL] %s %s" % (name, extra))
class _NoRedirect(urllib.request.HTTPRedirectHandler):
"""不要自动跟随 302 —— 检查跳转目标本身是否安全时必须看到原始 Location。"""
def redirect_request(self, req, fp, code, msg, headers, newurl):
return None
class Live:
def __init__(self, base: str, timeout: int = 20):
self.base = base.rstrip("/")
self.timeout = timeout
# 关键:显式清空代理,否则本机代理会把 127.0.0.1 也拦成 502
self.cj = http.cookiejar.CookieJar()
self.op = urllib.request.build_opener(
urllib.request.ProxyHandler({}),
urllib.request.HTTPCookieProcessor(self.cj),
)
self.op.addheaders = [("User-Agent", "workbuddy-portal-check/1.1")]
# 不跟随跳转的 opener:共用同一个 cookie jar,保证是同一会话
self.op_nr = urllib.request.build_opener(
urllib.request.ProxyHandler({}),
urllib.request.HTTPCookieProcessor(self.cj),
_NoRedirect,
)
self.op_nr.addheaders = [("User-Agent", "workbuddy-portal-check/1.1")]
def get(self, path: str):
try:
r = self.op.open(urllib.request.Request(self.base + path), timeout=self.timeout)
return r.status, r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
return e.code, e.read().decode("utf-8", "replace")
def post(self, path: str, data: dict, csrf: str | None = None, as_json: bool = False):
if as_json:
body, ct = json.dumps(data).encode(), "application/json"
else:
body, ct = urllib.parse.urlencode(data).encode(), "application/x-www-form-urlencoded"
req = urllib.request.Request(self.base + path, data=body, method="POST")
req.add_header("Content-Type", ct)
if csrf:
req.add_header("X-CSRF-Token", csrf)
try:
r = self.op.open(req, timeout=self.timeout)
return r.status, r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
return e.code, e.read().decode("utf-8", "replace")
def post_raw(self, path: str, data: dict):
"""表单 POST 且**不跟随**跳转,返回 (status, Location)。"""
body = urllib.parse.urlencode(data).encode()
req = urllib.request.Request(self.base + path, data=body, method="POST")
req.add_header("Content-Type", "application/x-www-form-urlencoded")
try:
r = self.op_nr.open(req, timeout=self.timeout)
return r.status, r.headers.get("Location")
except urllib.error.HTTPError as e:
return e.code, e.headers.get("Location")
def jget(self, path: str) -> dict:
st, body = self.get(path)
return json.loads(body) if st == 200 else {}
def run(L: Live, user: str, pwd: str, frm: str, to: str) -> None:
print("== 1. 未登录访问受保护资源 ==")
st, body = L.get("/")
chk("GET / 未登录落登录页", st == 200 and "登录" in body, "status=%s" % st)
for p in ("/api/summary", "/api/bundle", "/api/manifest"):
st, _ = L.get(p)
chk("GET %-14s 未登录=401" % p, st == 401, "status=%s" % st)
print("== 2. 登录(含 CSRF) ==")
st, html = L.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
chk("登录页含 CSRF 隐藏域", bool(m))
st, _ = L.post("/login", {"username": user, "password": pwd,
"_csrf": m.group(1) if m else ""})
chk("登录成功", st in (200, 302), "status=%s" % st)
st, html = L.get("/")
chk("登录后 GET / 到概览", st == 200 and "概览" in html, "len=%d" % len(html))
print("== 3. 后台页面均可达 ==")
for p, kw in [("/", "概览"), ("/tasks", "任务"), ("/config", "配置"),
("/logs", "日志"), ("/records", "记录")]:
st, html = L.get(p)
chk("GET %-10s" % p, st == 200 and kw in html, "status=%s len=%d" % (st, len(html)))
print("== 4. 登录后写操作仍需 CSRF ==")
st, _ = L.post("/api/collect", {}, csrf=None, as_json=True)
chk("POST /api/collect 缺 CSRF=400", st == 400, "status=%s" % st)
print("== 5. 结构与数值 ==")
mf = L.jget("/api/manifest")
chk("manifest 含 health/archive/totals/sources",
all(k in mf for k in ("health", "archive", "totals", "sources")))
src = (mf.get("sources") or [{}])[0]
chk("manifest 存档条数与数据源一致",
mf["totals"]["records"] == src.get("count"),
"records=%s src=%s" % (mf["totals"]["records"], src.get("count")))
sm = L.jget("/api/summary?from=%s&to=%s" % (frm, to))
want_days = (_d(to) - _d(frm)).days + 1
chk("summary 窗口天数正确", sm.get("window", {}).get("days") == want_days,
"window=%s 期望 %d 天" % (sm.get("window"), want_days))
chk("summary 窗口内有记录", (sm.get("records") or 0) > 0, "records=%s" % sm.get("records"))
chk("summary 环比 prev 存在", bool(sm.get("prev")),
"prev=%s~%s" % ((sm.get("prev") or {}).get("firstDay"), (sm.get("prev") or {}).get("lastDay")))
chk("summary avgPerCall 自洽",
not sm.get("calls") or abs(sm["avgPerCall"] - round(sm["credits"] / sm["calls"], 4)) < 1e-6)
bd = L.jget("/api/bundle?from=%s&to=%s" % (frm, to))
chk("bundle 顶层键齐全",
{"manifest", "daily", "dims", "top", "records", "totals", "window"} <= set(bd),
"keys=%s" % list(bd.keys()))
recs, daily = bd.get("records", []), bd.get("daily", [])
rsum = round(sum(float(x["c"]) for x in recs), 2)
tsum = round(float(bd.get("totals", {}).get("credits", 0)), 2)
print(" 窗口 %d 条,records 求和 %.2f ;totals.credits %.2f" % (len(recs), rsum, tsum))
chk("records 求和 == totals.credits", abs(rsum - tsum) < 0.005, "diff=%.4f" % (rsum - tsum))
chk("records 求和 == summary.credits",
abs(rsum - float(sm.get("credits", 0))) < 0.005,
"diff=%.4f" % (rsum - float(sm.get("credits", 0))))
chk("daily 为全量(多于窗口天数,供日历/日期轴)", len(daily) > want_days,
"daily=%d 天 > 窗口 %d 天" % (len(daily), want_days))
chk("daily 全量求和 == 存档总额",
abs(round(sum(float(x["c"]) for x in daily), 2)
- round(float(mf["totals"]["credits"]), 2)) < 0.005)
chk("逐日 h[24] 求和 == 当日 c",
all(abs(round(sum(d["h"]), 2) - round(d["c"], 2)) < 0.005 for d in daily))
chk("dims.hour 补齐 24 槽", len(bd.get("dims", {}).get("hour", [])) == 24)
chk("dims.model 非空", len(bd.get("dims", {}).get("model", [])) > 0)
top = bd.get("top", [])
chk("top 榜按积分降序",
all(top[i]["c"] >= top[i + 1]["c"] for i in range(len(top) - 1)), "n=%d" % len(top))
print("== 6. 明细分页/筛选/排序 ==")
rj = L.jget("/api/records?page=1&size=5")
chk("分页返回 5 条", len(rj.get("items", [])) == 5,
"total=%s pages=%s" % (rj.get("total"), rj.get("pages")))
chk("分页 total 与存档一致", rj.get("total") == mf["totals"]["records"])
chk("分页字段为可读全名", "request_id" in (rj.get("items") or [{}])[0])
chk("分页页码自洽",
rj.get("pages") == max(1, (rj.get("total", 0) + rj.get("size", 1) - 1) // rj.get("size", 1)))
r2 = L.jget("/api/records?page=2&size=5")
chk("第 2 页与第 1 页不重叠",
set(x["request_id"] for x in r2.get("items", [])).isdisjoint(
set(x["request_id"] for x in rj.get("items", []))))
models = bd.get("dims", {}).get("model", [])
if models:
mn = models[0]["name"]
rf = L.jget("/api/records?page=1&size=5&model=" + urllib.parse.quote(mn))
chk("按模型筛选生效", all(x["model"] == mn for x in rf.get("items", [])),
"model=%s total=%s" % (mn, rf.get("total")))
ro = L.jget("/api/records?page=1&size=10&order=credits_desc")
chk("按积分降序生效",
all(ro["items"][i]["credits"] >= ro["items"][i + 1]["credits"]
for i in range(len(ro.get("items", [])) - 1)))
print("== 7. 凭据不外泄 ==")
stj = L.jget("/api/settings")
chk("settings 无 cookie 明文字段", "cookie" not in stj, "keys=%s" % list(stj.keys()))
chk("settings 仅回 cookie_hint 掩码",
bool(stj.get("cookie_hint")) and len(str(stj.get("cookie_hint"))) < 200,
"hint=%s" % stj.get("cookie_hint"))
chk("配置页 HTML 不含 cookie 明文", "eyJ" not in L.get("/config")[1])
print("== 8. 错误处理 ==")
for p in ("/api/nope", "/nope"):
st, _ = L.get(p)
chk("GET %-12s =404" % p, st == 404, "status=%s" % st)
for p in ("/api/summary?from=abc&to=def", "/api/daily?from=2026-13-99"):
st, _ = L.get(p)
chk("GET %-32s 非法日期=400" % p, st == 400, "status=%s" % st)
print("== 9. 新增能力:用户管理 / 审计 / 流式导出 ==")
st, html = L.get("/users")
chk("GET /users 管理员可达", st == 200 and "用户管理" in html, "status=%s" % st)
chk("用户管理页不回传口令散列", "pbkdf2:" not in html)
au = L.jget("/api/audit?size=5")
chk("GET /api/audit 结构完整",
all(k in au for k in ("total", "page", "size", "pages", "actions", "items")),
"keys=%s" % list(au.keys()))
chk("审计条目带 actor/action/at",
not au.get("items") or {"actor", "action", "at"} <= set(au["items"][0]),
"n=%d" % len(au.get("items", [])))
st, csv_body = L.get("/records/export?from=%s&to=%s" % (frm, to))
chk("GET /records/export=200", st == 200, "status=%s" % st)
chk("导出带 UTF-8 BOM(Excel 不乱码)", csv_body.startswith("\ufeff"))
lines = [x for x in csv_body.lstrip("\ufeff").split("\r\n") if x]
chk("导出表头为官网同构列",
lines and lines[0] == "RequestID,积分消耗,User Prompt,模型,客户端,时间",
"header=%s" % (lines[0] if lines else None))
chk("导出行数 == 窗口记录数 + 表头", len(lines) == (sm.get("records") or 0) + 1,
"csv=%d 记录=%s" % (len(lines), sm.get("records")))
r1 = L.jget("/api/records?page=1&size=1&from=%s&to=%s" % (frm, to))
first_id = ((r1.get("items") or [{}])[0]).get("request_id")
chk("导出与明细同源同序(首行 == 明细首条)",
len(lines) > 1 and bool(first_id) and first_id in lines[1],
"api=%s csv=%s" % (first_id, (lines[1][:40] if len(lines) > 1 else None)))
print("== 10. 安全:开放重定向与凭证外泄 ==")
L2 = Live(L.base) # 全新会话,避免已登录被直跳
st, html = L2.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
csrf = m.group(1) if m else ""
st, loc = L2.post_raw("/login", {"username": user, "password": pwd,
"_csrf": csrf, "next": "//evil.com"})
chk("next=//evil.com 被拒(不出现协议相对跳转)",
st == 302 and "evil.com" not in (loc or "") and not (loc or "").startswith("//"),
"status=%s Location=%s" % (st, loc))
L3 = Live(L.base)
st, html = L3.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
st, loc = L3.post_raw("/login", {"username": user, "password": pwd,
"_csrf": m.group(1) if m else "", "next": "/records"})
chk("next=/records 站内路径正常放行", st == 302 and loc == "/records",
"status=%s Location=%s" % (st, loc))
st, loc = L3.post_raw("/login", {"username": user, "password": pwd, "_csrf": "wrong"})
chk("错误 CSRF 的登录 POST=400", st == 400, "status=%s" % st)
st, body = L.get("/logout")
chk("GET /logout 不执行退出(仅提示)", st == 200 and "退出" in body, "status=%s" % st)
st, html = L.get("/")
chk("GET /logout 后仍处于登录态", st == 200 and "概览" in html, "status=%s" % st)
def main() -> int:
ap = argparse.ArgumentParser(description="对运行中的用量门户做端到端验收")
ap.add_argument("--base", default="http://127.0.0.1:8848", help="服务地址")
ap.add_argument("-u", "--user", default="admin", help="登录用户名")
ap.add_argument("-p", "--password", default="admin123", help="登录密码")
ap.add_argument("--from", dest="frm", default="2026-09-08", help="验收窗口起")
ap.add_argument("--to", dest="to", default="2026-09-14", help="验收窗口止")
ap.add_argument("--timeout", type=int, default=20)
a = ap.parse_args()
print("目标:%s 窗口:%s ~ %s\n" % (a.base, a.frm, a.to))
run(Live(a.base, a.timeout), a.user, a.password, a.frm, a.to)
print("\nRESULT: ok=%d fail=%d" % (OK, FAIL))
if FAILS:
print("失败项:%s" % "、".join(FAILS))
return 1 if FAIL else 0
if __name__ == "__main__":
sys.exit(main())
+152
查看文件
@@ -0,0 +1,152 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""界面实检:登录后逐页截图,用来目视确认「统一美化」是否真的落地。
用法:
python manage.py serve --port 8849 --no-scheduler # 另开一个终端
python tools/shots.py --base http://127.0.0.1:8849
python tools/shots.py --full # 整页长图(默认只截首屏)
产物:data/shots/*.png(已被 .gitignore 之外的目录,可直接删)。
为什么不用 headless chrome 直出:本项目的页面都要登录态,
`--screenshot` 无法注入会话 Cookie,所以必须用 Playwright 走一次真实登录。
"""
from __future__ import annotations
import argparse
import os
import sys
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, BASE)
PAGES = [
("login", "/login", "登录页"),
("overview", "/", "概览"),
("records", "/records", "数据明细"),
("tasks", "/tasks", "任务管理"),
("config", "/config", "配置管理"),
("logs", "/logs", "日志管理"),
("users", "/users", "用户管理"),
("dashboard", "/dashboard", "用量大屏"),
]
def _find_browser() -> str | None:
"""找一个可用的 Chromium 可执行文件。
Playwright 的驱动版本与本地已下载的浏览器版本常常错位(例如驱动要
chromium_headless_shell-1234 而本机只有 -1228),此时 launch() 会直接报
「Executable doesn't exist」。这里按优先级探测,命中就显式传 executable_path。
"""
import glob
home = os.environ.get("LOCALAPPDATA") or os.path.expanduser("~")
pats = [
os.path.join(home, "ms-playwright", "chromium-*", "chrome-win64", "chrome.exe"),
os.path.join(home, "ms-playwright", "chromium_headless_shell-*",
"chrome-headless-shell-win64", "chrome-headless-shell.exe"),
r"C:\Program Files\Google\Chrome\Application\chrome.exe",
r"C:\Program Files (x86)\Google\Chrome\Application\chrome.exe",
r"C:\Program Files (x86)\Microsoft\Edge\Application\msedge.exe",
r"C:\Program Files\Microsoft\Edge\Application\msedge.exe",
]
for p in pats:
hits = sorted(glob.glob(p))
if hits:
return hits[-1]
return None
def main() -> int:
ap = argparse.ArgumentParser()
ap.add_argument("--base", default="http://127.0.0.1:8849")
ap.add_argument("-u", "--user", default="admin")
ap.add_argument("-p", "--password", default="admin123")
ap.add_argument("--out", default=os.path.join(BASE, "data", "shots"))
ap.add_argument("--full", action="store_true", help="截整页长图")
ap.add_argument("--browser", default="", help="显式指定 chrome/msedge 可执行文件")
ap.add_argument("--width", type=int, default=1440)
ap.add_argument("--height", type=int, default=900)
a = ap.parse_args()
from playwright.sync_api import sync_playwright
exe = a.browser or _find_browser()
os.makedirs(a.out, exist_ok=True)
problems = []
with sync_playwright() as pw:
try:
br = pw.chromium.launch(executable_path=exe) if exe else pw.chromium.launch()
except Exception as e: # noqa: BLE001
print("[FAIL] 启动浏览器失败:%s" % e)
print(" 可用 --browser 显式指定,或用 `playwright install chromium` 装齐。")
return 1
print("浏览器:%s" % (exe or "playwright 默认"))
ctx = br.new_context(viewport={"width": a.width, "height": a.height},
device_scale_factor=2, locale="zh-CN")
page = ctx.new_page()
errors = []
page.on("console", lambda m: errors.append(m.text) if m.type == "error" else None)
page.on("pageerror", lambda e: errors.append(str(e)))
# 1) 先截未登录的登录页
page.goto(a.base + "/login", wait_until="networkidle")
page.screenshot(path=os.path.join(a.out, "00-login.png"), full_page=a.full)
print("[ok] 00-login.png")
# 2) 登录
page.fill('input[name=username]', a.user)
page.fill('input[name=password]', a.password)
page.click('button[type=submit]')
page.wait_for_load_state("networkidle")
if "/login" in page.url:
print("[FAIL] 登录失败,后续截图无意义")
br.close()
return 1
# 3) 逐页截图
for i, (slug, path, label) in enumerate(PAGES[1:], start=1):
errors.clear()
page.goto(a.base + path, wait_until="networkidle")
page.wait_for_timeout(900) # 等 ECharts / 表格渲染稳下来
page.screenshot(path=os.path.join(a.out, "%02d-%s.png" % (i, slug)),
full_page=a.full)
js_err = [e for e in errors if "favicon" not in e.lower()]
flag = "" if not js_err else " [JS错误] " + " | ".join(js_err[:3])
if js_err:
problems.append("%s: %s" % (label, js_err[:3]))
print("[ok] %02d-%s.png %s%s" % (i, slug, label, flag))
# 4) 大屏页再点几个交互,确认控件联动不炸
page.goto(a.base + "/dashboard", wait_until="networkidle")
page.wait_for_timeout(1200)
for sel in ["#segRange button", ".seg button"]:
btns = page.query_selector_all(sel)
if len(btns) > 1:
errors.clear()
btns[1].click()
page.wait_for_timeout(900)
page.screenshot(path=os.path.join(a.out, "08-dashboard-interact.png"),
full_page=a.full)
js_err = [e for e in errors if "favicon" not in e.lower()]
print("[ok] 08-dashboard-interact.png 点击 %s 第 2 项%s"
% (sel, "" if not js_err else " [JS错误] %s" % js_err[:2]))
if js_err:
problems.append("大屏交互: %s" % js_err[:2])
break
br.close()
print("\n截图目录:%s" % a.out)
if problems:
print("发现问题:")
for p in problems:
print(" - " + p)
return 1
print("RESULT: 全页面截图完成,无 JS 报错")
return 0
if __name__ == "__main__":
sys.exit(main())
+318
查看文件
@@ -0,0 +1,318 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""离线回归:用 Flask test_client 对**真实库**做全页面只读渲染 + 缺陷防回归断言。
与 tools/check_live.py 的分工:
* check_live.py 对运行中的服务发真实 HTTP,验「起没起来、登录/CSRF/API 通不通」
* smoke.py(本脚本)不发网络请求,直接把请求灌进 WSGI 应用,
因此能覆盖到「页面模板渲染是否正确」,且不需要先起服务、不需要密码。
覆盖内容:
1. 全页面渲染(含 /users,需管理员身份)——模板报错会直接暴露成 500
2. 模板未渲染残留(HTML 里出现 {{ / {% 说明有变量名写错)
3. 历史缺陷防回归(见下 REGRESSIONS)
4. CSV 导出可被标准 csv 解析、列数一致
5. 页面 HTML 里的 class 与 app.css 的选择器做差集(抓类名拼写错误)
用法:
cd workbuddy-portal
python tools/smoke.py
退出码:0 全通过;1 有失败项。
"""
from __future__ import annotations
import csv
import io
import json
import os
import re
import sys
BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, BASE)
OK = 0
FAIL = 0
FAILS: list[str] = []
NOTES: list[str] = []
def chk(name: str, cond: bool, extra: str = "") -> bool:
global OK, FAIL
if cond:
OK += 1
print(" [OK] %s %s" % (name, extra))
else:
FAIL += 1
FAILS.append(name)
print(" [FAIL] %s %s" % (name, extra))
return bool(cond)
def note(msg: str) -> None:
NOTES.append(msg)
print(" [note] %s" % msg)
def login(cli, admin=True):
"""注入会话绕过登录:GET 不触发 CSRF,因此可直接测页面渲染。"""
with cli.session_transaction() as s:
s["uid"] = 1
s["uname"] = "admin" if admin else "viewer"
s["dname"] = "管理员" if admin else "只读账号"
s["adm"] = 1 if admin else 0
s["_csrf"] = "smoke-csrf-token"
def page(cli, path, method="GET", **kw):
r = getattr(cli, method.lower())(path, **kw)
return r.status_code, r.get_data(as_text=True)
def run() -> None:
from workbuddy_portal import create_app, db, query
print("== 0. 构建应用 ==")
app = create_app(start_scheduler=False, do_init_db=False)
app.config["WTF_CSRF_ENABLED"] = False
n_routes = len([r for r in app.url_map.iter_rules()])
chk("create_app 成功", app is not None)
chk("路由数量 >= 35", n_routes >= 35, "routes=%d" % n_routes)
# ---------------- 1. 未登录 ----------------
print("== 1. 未登录:受保护页应跳登录、API 应 401 ==")
with app.test_client() as cli:
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users"):
st, _ = page(cli, p)
chk("GET %-10s 未登录=302" % p, st == 302, "status=%s" % st)
for p in ("/api/summary", "/api/users", "/api/settings", "/api/audit"):
st, _ = page(cli, p)
chk("GET %-14s 未登录=401" % p, st == 401, "status=%s" % st)
st, html = page(cli, "/login")
chk("登录页含 CSRF 隐藏域", 'name="_csrf"' in html)
# ---------------- 2. 管理员:全页面渲染 ----------------
print("== 2. 管理员:全页面渲染 ==")
with app.test_client() as cli:
login(cli, admin=True)
pages = [
("/", "概览"), ("/records", "数据明细"), ("/tasks", "任务管理"),
("/config", "配置管理"), ("/logs", "日志管理"), ("/users", "用户管理"),
("/dashboard", "<html"),
]
for p, kw in pages:
st, html = page(cli, p)
ok = chk("GET %-10s 200" % p, st == 200, "status=%s len=%d" % (st, len(html)))
if ok:
chk(" └ 含关键字 %s" % kw, kw in html)
chk(" └ 无模板残留 {{ / {%%", "{{" not in html and "{%" not in html)
chk(" └ 含导航栏", "topbar" in html or p == "/dashboard")
st, html = page(cli, "/users")
chk("用户管理页列出账号", 'data-uid=' in html, "含行内编辑按钮")
chk("用户管理页含新建表单", 'id="formNewUser"' in html)
chk("用户管理页含审计表", "用户操作审计" in html)
# 配置页的维护按钮 + 大屏回后台入口
st, cfg = page(cli, "/config")
chk("配置页含维护按钮组", cfg.count("data-maint=") >= 3, "n=%d" % cfg.count("data-maint="))
chk("配置页含 TLS 校验下拉", 'name="ssl_verify"' in cfg)
st, rec = page(cli, "/records")
chk("明细页含快捷区间", 'data-range="today"' in rec and 'data-range="30d"' in rec)
chk("明细页表格包在 .tablewrap", "tablewrap" in rec)
# ---------------- 2b. 静态资源引用可解析 ----------------
print("== 2b. 页面引用的静态资源全部可达 ==")
asset_re = re.compile(r"\.(?:js|css|svg|png|jpe?g|gif|webp|ico|woff2?)(?:\?|$)", re.I)
with app.test_client() as cli:
login(cli, admin=True)
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users", "/dashboard"):
_, html = page(cli, p)
# 先剥掉 HTML 注释:注释里常写示例路径(src="vendor/x.js"),
# 不剥会把示例当真实引用误报。
html = re.sub(r"<!--.*?-->", "", html, flags=re.S)
refs = set(re.findall(r'src="([^"]+)"', html))
refs |= {h for h in re.findall(r'href="([^"]+)"', html) if asset_re.search(h)}
bad, n = [], 0
for r in sorted(refs):
if r.startswith(("data:", "http:", "https:", "//", "#")):
continue
target = r
if not r.startswith("/"): # 相对路径按该页 URL 解析(曾因此 404)
target = (p if p.endswith("/") else p.rsplit("/", 1)[0] + "/") + r
n += 1
ast, _ = page(cli, target)
if ast != 200:
bad.append("%s -> %s(%s)" % (r, target, ast))
chk("%-11s 资源引用全部 200" % p, not bad,
("坏引用=%s" % bad) if bad else "%d 个引用" % n)
# ---------------- 3. 非管理员:权限边界 ----------------
print("== 3. 非管理员:/users 必须 403,导航不出现该入口 ==")
with app.test_client() as cli:
login(cli, admin=False)
st, html = page(cli, "/users")
chk("GET /users 非管理员=403", st == 403, "status=%s" % st)
st, _ = page(cli, "/api/users")
chk("GET /api/users 非管理员=403", st == 403, "status=%s" % st)
st, _ = page(cli, "/api/users")
st, html = page(cli, "/")
chk("概览导航不含「用户管理」", "用户管理" not in html)
for p in ("/", "/records", "/tasks", "/logs"):
st, _ = page(cli, p)
chk("GET %-10s 非管理员=200" % p, st == 200, "status=%s" % st)
# ---------------- 4. 历史缺陷防回归 ----------------
print("== 4. 历史缺陷防回归 ==")
with app.test_client() as cli:
login(cli, admin=True)
# ① 非法日期曾 500
st, body = page(cli, "/api/summary?from=abc&to=def")
chk("① /api/summary 非法日期=400", st == 400, "status=%s" % st)
chk(" └ 返回 JSON 错误体", '"ok": false' in body.replace('":', '": '))
# ② /tasks 非法页码曾 500
st, _ = page(cli, "/tasks?page=abc")
chk("② /tasks?page=abc=200", st == 200, "status=%s" % st)
# ③ 日志尾部非法行数曾 500
st, _ = page(cli, "/logs/tail?lines=abc")
chk("③ /logs/tail?lines=abc=200", st == 200, "status=%s" % st)
# ④ 内部簿记键 slot:* 曾泄漏到 /api/settings
st, body = page(cli, "/api/settings")
chk("④ /api/settings 无 slot:* 键", "slot:" not in body, "status=%s" % st)
# ⑤ 概览「云端」列曾因 SQL 少选列而恒为空
st, ov = page(cli, "/")
chk("⑤ 概览含「云端」列", "云端" in ov)
# ⑥ 明细页日期回填:模板曾读 f.from,导致输入框永远为空
st, rec = page(cli, "/records?from=2026-09-08&to=2026-09-10")
chk("⑥ 明细页回填 from", 'value="2026-09-08"' in rec)
chk("⑥ 明细页回填 to", 'value="2026-09-10"' in rec)
# ⑦ 导出链接必须带规范参数名 from(模板内部用 frm,拼 URL 时要换回来)
m = re.search(r'href="(/records/export[^"]*)"', rec)
chk("⑦ 导出链接存在", bool(m))
if m:
chk("⑦ 导出链接带 from=", "from=2026-09-08" in m.group(1), m.group(1))
chk("⑦ 导出链接带 to=", "to=2026-09-10" in m.group(1))
# ⑧ 导出接口本身也要认 from/to
st, csv_body = page(cli, "/records/export?from=2026-09-08&to=2026-09-10")
chk("⑧ GET /records/export=200", st == 200, "status=%s" % st)
rows = list(csv.reader(io.StringIO(csv_body.lstrip("\ufeff"))))
chk("⑧ CSV 至少含表头", len(rows) >= 1, "rows=%d" % len(rows))
chk("⑧ CSV 列数一致",
len({len(r) for r in rows if r}) == 1,
"列数集合=%s" % sorted({len(r) for r in rows if r}))
chk("⑧ CSV 表头为官方同构列",
rows and rows[0] == ["RequestID", "积分消耗", "User Prompt", "模型", "客户端", "时间"],
"header=%s" % (rows[0] if rows else None))
# ⑨ 非法设置必须在写入时被拒(曾让采集崩掉)
st, body = page(cli, "/api/settings", method="POST", json={"page_size": "abc"},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑨ 非法设置写入=400", st == 400, "status=%s" % st)
st, body = page(cli, "/api/settings", method="POST", json={"slot:09:00": "x"},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑨ 内部键被忽略而非写入",
st == 200 and "slot:09:00" in (json.loads(body).get("ignored") or []),
"status=%s body=%s" % (st, body[:140]))
# ⑩ 维护动作
st, _ = page(cli, "/api/maintenance/nope", method="POST", json={},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑩ 未知维护动作=404", st == 404, "status=%s" % st)
st, body = page(cli, "/api/maintenance/recount", method="POST", json={},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑩ recount=200", st == 200, "status=%s body=%s" % (st, body[:90]))
# ⑪ 非管理员调用户管理 API
st, _ = page(cli, "/api/users/1/delete", method="POST", json={},
headers={"X-CSRF-Token": "smoke-csrf-token"})
chk("⑪ 删除自己=400(不允许)", st == 400, "status=%s" % st)
# ⑫ CSRF 缺失必须 400
st, _ = page(cli, "/api/settings", method="POST", json={"max_prompt": "100"})
chk("⑫ 缺 CSRF=400", st == 400, "status=%s" % st)
# ⑬ 分页/筛选参数非法不能 500
for p in ("/logs?page=abc", "/logs?apage=abc", "/records?page=abc&size=abc",
"/records?from=2026-13-99", "/logs?status=%27%20OR%201=1--"):
st, _ = page(cli, p)
chk("⑬ GET %-30s =200" % p, st == 200, "status=%s" % st)
# ⑭ 审计筛选:只返回指定动作,且计数与筛选一致
st, lg = page(cli, "/logs")
chk("⑭ 日志页含操作审计筛选", "操作审计" in lg and "seg" in lg)
m = re.search(r'href="/logs\?act=([^"&]+)', lg)
if chk("⑭ 审计筛选有可选动作", bool(m), "act=%s" % (m.group(1) if m else "无")):
act = m.group(1)
st, lg2 = page(cli, "/logs?act=" + act)
chk("⑭ 带 act=%s 仍 200" % act, st == 200, "status=%s" % st)
# 从审计表的 id 处切片:前面采集表里的 trigger 也用了 tag mute,不能混入
i = lg2.find('id="auditTable"')
tail = lg2[i:] if i >= 0 else ""
chk("⑭ 审计表存在", i >= 0)
tags = re.findall(r'<td><span class="tag mute">([^<]+)</span></td>', tail)
others = sorted({t for t in tags if t != act})
chk("⑭ 审计筛选结果不含其他动作", not others and bool(tags),
"命中=%d 混入=%s" % (len(tags), others))
# ---------------- 5. 数据自洽 ----------------
print("== 5. 数据自洽(只读) ==")
with app.test_client() as cli:
login(cli, admin=True)
mf = json.loads(page(cli, "/api/manifest")[1])
src = (mf.get("sources") or [{}])[0]
chk("manifest 存档条数 == 数据源条数",
mf["totals"]["records"] == src.get("count"),
"%s vs %s" % (mf["totals"]["records"], src.get("count")))
sm = json.loads(page(cli, "/api/summary")[1])
chk("summary 全量 calls == 存档条数",
sm.get("calls") == mf["totals"]["records"],
"calls=%s records=%s" % (sm.get("calls"), mf["totals"]["records"]))
chk("summary 全量 credits 自洽",
abs(float(sm.get("credits", 0)) - float(mf["totals"]["credits"])) < 0.005,
"%s vs %s" % (sm.get("credits"), mf["totals"]["credits"]))
d = query.daily(db.get_db())
chk("daily 逐日积分求和 == 存档总额",
abs(round(sum(float(x["c"]) for x in d), 2)
- round(float(mf["totals"]["credits"]), 2)) < 0.005)
chk("daily 逐日 h[24] 求和 == 当日积分",
all(abs(round(sum(x["h"]), 2) - round(x["c"], 2)) < 0.005 for x in d))
note("存档 %s 条 / %s 积分 / %d 天" % (mf["totals"]["records"],
mf["totals"]["credits"], len(d)))
# ---------------- 6. class 名与 CSS 选择器对账 ----------------
print("== 6. 页面 class 与 app.css 选择器对账 ==")
css = open(os.path.join(BASE, "workbuddy_portal", "web", "static", "css", "app.css"),
encoding="utf-8").read()
css_classes = set(re.findall(r"\.([A-Za-z][\w-]*)", css))
with app.test_client() as cli:
login(cli, admin=True)
used: set[str] = set()
for p in ("/", "/records", "/tasks", "/config", "/logs", "/users", "/login"):
if p == "/login":
with app.test_client() as c2:
html = page(c2, p)[1]
else:
html = page(cli, p)[1]
for m in re.findall(r'class="([^"]*)"', html):
used.update(t for t in m.split() if t)
# 允许的无样式类:JS 钩子、第三方/语义标记
allow = {"no-js", "on", "cur", "gap", "meta", "unit", "field-err"}
missing = sorted(c for c in used - css_classes - allow)
chk("无「用了但 CSS 里不存在」的类名", not missing, "缺失=%s" % missing if missing else "")
def main() -> int:
print("工程目录:%s\n" % BASE)
try:
run()
except Exception as e: # noqa: BLE001
import traceback
traceback.print_exc()
print("\n[FATAL] 脚本本身异常:%s" % e)
return 1
print("\nRESULT: ok=%d fail=%d" % (OK, FAIL))
if NOTES:
print("备注:")
for n in NOTES:
print(" - " + n)
if FAILS:
print("失败项:\n - " + "\n - ".join(FAILS))
return 1 if FAIL else 0
if __name__ == "__main__":
sys.exit(main())