chore: 项目定名为 workbuddy-portal,容器化并补齐文档体系

## 项目定名
- 目录 wb_usage_portal → workbuddy-portal
- Python 包 wb_usage → workbuddy_portal(含 session cookie 名)
- 界面品牌统一为 WorkBuddy Portal;项目标识收敛到 config 单一来源

## 容器化
- Dockerfile:多阶段构建,依赖层与源码解耦;非 root(uid 1000);内置健康检查
- docker-compose.yml:单服务 + 绑定挂载 data/logs + 日志轮转 + TZ
- docker/entrypoint.sh:幂等初始化 → exec serve(LF 行尾,已由 .gitattributes 锁定)
- docker/healthcheck.py:纯标准库探活 /login(slim 镜像无 curl)
- .dockerignore / .env.example;数据目录可用 WB_DATA_DIR 等环境变量覆盖

## 文档
- docs/USER-GUIDE.md    用户使用手册(含 9 张真实界面截图)
- docs/DEPLOYMENT.md    部署运维(Docker / 裸机 / 反代 / 备份 / 推 Gitea 注册表)
- docs/ARCHITECTURE.md  架构与设计说明(含已知坑与红线、验证体系)
- docs/API.md           接口参考(路径 / 参数 / 返回结构 / 错误码)
- docs/FAQ.md           常见问题;docs/CHANGELOG.md 变更日志

## 修复缺陷(8)
1. /records/export 必然 500:生成器在请求上下文销毁后才迭代,改用自建连接
2. 大屏页图表全白:相对路径把 echarts.min.js 解析成 /vendor/... → 404
3. /users 500:路由已注册但模板缺失
4. 明细页日期筛选失效:视图传 f.frm、模板读 f.from
5. 配置页维护按钮全死:调用了不存在的 WBU.bindMaint()
6. 审计只能看最近 40 条:LIMIT 写死
7. 明细页多跑一条无用 SELECT:day_list() 取了没人用
8. 登录页锁定阈值未从配置注入

## 安全加固
- 新增 safe_next():拒绝 //evil.com 等协议相对 URL 的开放重定向
- 缺 CSRF 的写请求统一 400
- 默认开启云端 HTTPS 证书校验(ssl_verify=1);Cookie 是账号凭证
- 登录失败计数表加上限与 TTL
- /logout 拆分为 POST(执行) + GET(仅提示),防 <img src=/logout> 静默退出
- settings 内部簿记键 slot:* 读写两侧过滤,不再从 /api/settings 泄漏

## 内部质量与工具
- 设置项写时校验 + 读时兜底,杜绝「一个手滑的数字让采集整个跑不起来」
- 全局 ValueError → 400:手写 query string 不再暴露 500 页面
- CSV 导出改 csv.writer 流式写入(原手工拼串,字段含逗号会串列)
- bundle 明细加 20000 上限并回传 recordsTotal/recordsTruncated,不静默丢数据
- tools/smoke.py 离线回归 99 项;tools/check_live.py 真实 HTTP 56 项
- tools/shots.py Playwright 逐页截图 + JS 报错收集

## 验证
- compileall 通过;smoke 99/99;对容器实例 check_live 56/56;截图 0 JS 报错
- 容器内采集实测成功(trigger=startup 补跑:新增 11 条)
这个提交包含在:
2026-09-14 14:55:50 +08:00
当前提交 86631ae7ab
共修改 58 个文件,包含 9409 行新增和 0 行删除
+297
查看文件
@@ -0,0 +1,297 @@
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""端到端验收:对**运行中的**服务发真实 HTTP 请求,走完整登录/CSRF/API 链路。
与 tests 里用 Flask test_client 的冒烟测试互补——这里验证的是「真的起起来了、
真的能登录、真的能取到数」,适合部署到局域网后随手跑一遍。
用法:
python tools/check_live.py # 默认 http://127.0.0.1:8848
python tools/check_live.py --base http://10.0.0.5:8848
python tools/check_live.py -u admin -p 你的密码
python tools/check_live.py --from 2026-09-08 --to 2026-09-14
退出码:0 全通过;1 有失败项(会打印失败清单)。
注意:脚本会读取窗口数据但**不写库**(不触发采集、不改配置),可安全反复运行。
"""
from __future__ import annotations
import argparse
import http.cookiejar
import json
import re
import sys
import urllib.error
import urllib.parse
import urllib.request
from datetime import datetime
OK = 0
FAIL = 0
FAILS: list[str] = []
def _d(s: str):
"""把 YYYY-MM-DD 解析成本地 datetime(不用 date.fromisoformat 之外的时区处理)。"""
return datetime.strptime(s, "%Y-%m-%d")
def chk(name: str, cond: bool, extra: str = "") -> None:
global OK, FAIL
if cond:
OK += 1
print(" [OK] %s %s" % (name, extra))
else:
FAIL += 1
FAILS.append(name)
print(" [FAIL] %s %s" % (name, extra))
class _NoRedirect(urllib.request.HTTPRedirectHandler):
"""不要自动跟随 302 —— 检查跳转目标本身是否安全时必须看到原始 Location。"""
def redirect_request(self, req, fp, code, msg, headers, newurl):
return None
class Live:
def __init__(self, base: str, timeout: int = 20):
self.base = base.rstrip("/")
self.timeout = timeout
# 关键:显式清空代理,否则本机代理会把 127.0.0.1 也拦成 502
self.cj = http.cookiejar.CookieJar()
self.op = urllib.request.build_opener(
urllib.request.ProxyHandler({}),
urllib.request.HTTPCookieProcessor(self.cj),
)
self.op.addheaders = [("User-Agent", "workbuddy-portal-check/1.1")]
# 不跟随跳转的 opener:共用同一个 cookie jar,保证是同一会话
self.op_nr = urllib.request.build_opener(
urllib.request.ProxyHandler({}),
urllib.request.HTTPCookieProcessor(self.cj),
_NoRedirect,
)
self.op_nr.addheaders = [("User-Agent", "workbuddy-portal-check/1.1")]
def get(self, path: str):
try:
r = self.op.open(urllib.request.Request(self.base + path), timeout=self.timeout)
return r.status, r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
return e.code, e.read().decode("utf-8", "replace")
def post(self, path: str, data: dict, csrf: str | None = None, as_json: bool = False):
if as_json:
body, ct = json.dumps(data).encode(), "application/json"
else:
body, ct = urllib.parse.urlencode(data).encode(), "application/x-www-form-urlencoded"
req = urllib.request.Request(self.base + path, data=body, method="POST")
req.add_header("Content-Type", ct)
if csrf:
req.add_header("X-CSRF-Token", csrf)
try:
r = self.op.open(req, timeout=self.timeout)
return r.status, r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
return e.code, e.read().decode("utf-8", "replace")
def post_raw(self, path: str, data: dict):
"""表单 POST 且**不跟随**跳转,返回 (status, Location)。"""
body = urllib.parse.urlencode(data).encode()
req = urllib.request.Request(self.base + path, data=body, method="POST")
req.add_header("Content-Type", "application/x-www-form-urlencoded")
try:
r = self.op_nr.open(req, timeout=self.timeout)
return r.status, r.headers.get("Location")
except urllib.error.HTTPError as e:
return e.code, e.headers.get("Location")
def jget(self, path: str) -> dict:
st, body = self.get(path)
return json.loads(body) if st == 200 else {}
def run(L: Live, user: str, pwd: str, frm: str, to: str) -> None:
print("== 1. 未登录访问受保护资源 ==")
st, body = L.get("/")
chk("GET / 未登录落登录页", st == 200 and "登录" in body, "status=%s" % st)
for p in ("/api/summary", "/api/bundle", "/api/manifest"):
st, _ = L.get(p)
chk("GET %-14s 未登录=401" % p, st == 401, "status=%s" % st)
print("== 2. 登录(含 CSRF) ==")
st, html = L.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
chk("登录页含 CSRF 隐藏域", bool(m))
st, _ = L.post("/login", {"username": user, "password": pwd,
"_csrf": m.group(1) if m else ""})
chk("登录成功", st in (200, 302), "status=%s" % st)
st, html = L.get("/")
chk("登录后 GET / 到概览", st == 200 and "概览" in html, "len=%d" % len(html))
print("== 3. 后台页面均可达 ==")
for p, kw in [("/", "概览"), ("/tasks", "任务"), ("/config", "配置"),
("/logs", "日志"), ("/records", "记录")]:
st, html = L.get(p)
chk("GET %-10s" % p, st == 200 and kw in html, "status=%s len=%d" % (st, len(html)))
print("== 4. 登录后写操作仍需 CSRF ==")
st, _ = L.post("/api/collect", {}, csrf=None, as_json=True)
chk("POST /api/collect 缺 CSRF=400", st == 400, "status=%s" % st)
print("== 5. 结构与数值 ==")
mf = L.jget("/api/manifest")
chk("manifest 含 health/archive/totals/sources",
all(k in mf for k in ("health", "archive", "totals", "sources")))
src = (mf.get("sources") or [{}])[0]
chk("manifest 存档条数与数据源一致",
mf["totals"]["records"] == src.get("count"),
"records=%s src=%s" % (mf["totals"]["records"], src.get("count")))
sm = L.jget("/api/summary?from=%s&to=%s" % (frm, to))
want_days = (_d(to) - _d(frm)).days + 1
chk("summary 窗口天数正确", sm.get("window", {}).get("days") == want_days,
"window=%s 期望 %d 天" % (sm.get("window"), want_days))
chk("summary 窗口内有记录", (sm.get("records") or 0) > 0, "records=%s" % sm.get("records"))
chk("summary 环比 prev 存在", bool(sm.get("prev")),
"prev=%s~%s" % ((sm.get("prev") or {}).get("firstDay"), (sm.get("prev") or {}).get("lastDay")))
chk("summary avgPerCall 自洽",
not sm.get("calls") or abs(sm["avgPerCall"] - round(sm["credits"] / sm["calls"], 4)) < 1e-6)
bd = L.jget("/api/bundle?from=%s&to=%s" % (frm, to))
chk("bundle 顶层键齐全",
{"manifest", "daily", "dims", "top", "records", "totals", "window"} <= set(bd),
"keys=%s" % list(bd.keys()))
recs, daily = bd.get("records", []), bd.get("daily", [])
rsum = round(sum(float(x["c"]) for x in recs), 2)
tsum = round(float(bd.get("totals", {}).get("credits", 0)), 2)
print(" 窗口 %d 条,records 求和 %.2f ;totals.credits %.2f" % (len(recs), rsum, tsum))
chk("records 求和 == totals.credits", abs(rsum - tsum) < 0.005, "diff=%.4f" % (rsum - tsum))
chk("records 求和 == summary.credits",
abs(rsum - float(sm.get("credits", 0))) < 0.005,
"diff=%.4f" % (rsum - float(sm.get("credits", 0))))
chk("daily 为全量(多于窗口天数,供日历/日期轴)", len(daily) > want_days,
"daily=%d 天 > 窗口 %d 天" % (len(daily), want_days))
chk("daily 全量求和 == 存档总额",
abs(round(sum(float(x["c"]) for x in daily), 2)
- round(float(mf["totals"]["credits"]), 2)) < 0.005)
chk("逐日 h[24] 求和 == 当日 c",
all(abs(round(sum(d["h"]), 2) - round(d["c"], 2)) < 0.005 for d in daily))
chk("dims.hour 补齐 24 槽", len(bd.get("dims", {}).get("hour", [])) == 24)
chk("dims.model 非空", len(bd.get("dims", {}).get("model", [])) > 0)
top = bd.get("top", [])
chk("top 榜按积分降序",
all(top[i]["c"] >= top[i + 1]["c"] for i in range(len(top) - 1)), "n=%d" % len(top))
print("== 6. 明细分页/筛选/排序 ==")
rj = L.jget("/api/records?page=1&size=5")
chk("分页返回 5 条", len(rj.get("items", [])) == 5,
"total=%s pages=%s" % (rj.get("total"), rj.get("pages")))
chk("分页 total 与存档一致", rj.get("total") == mf["totals"]["records"])
chk("分页字段为可读全名", "request_id" in (rj.get("items") or [{}])[0])
chk("分页页码自洽",
rj.get("pages") == max(1, (rj.get("total", 0) + rj.get("size", 1) - 1) // rj.get("size", 1)))
r2 = L.jget("/api/records?page=2&size=5")
chk("第 2 页与第 1 页不重叠",
set(x["request_id"] for x in r2.get("items", [])).isdisjoint(
set(x["request_id"] for x in rj.get("items", []))))
models = bd.get("dims", {}).get("model", [])
if models:
mn = models[0]["name"]
rf = L.jget("/api/records?page=1&size=5&model=" + urllib.parse.quote(mn))
chk("按模型筛选生效", all(x["model"] == mn for x in rf.get("items", [])),
"model=%s total=%s" % (mn, rf.get("total")))
ro = L.jget("/api/records?page=1&size=10&order=credits_desc")
chk("按积分降序生效",
all(ro["items"][i]["credits"] >= ro["items"][i + 1]["credits"]
for i in range(len(ro.get("items", [])) - 1)))
print("== 7. 凭据不外泄 ==")
stj = L.jget("/api/settings")
chk("settings 无 cookie 明文字段", "cookie" not in stj, "keys=%s" % list(stj.keys()))
chk("settings 仅回 cookie_hint 掩码",
bool(stj.get("cookie_hint")) and len(str(stj.get("cookie_hint"))) < 200,
"hint=%s" % stj.get("cookie_hint"))
chk("配置页 HTML 不含 cookie 明文", "eyJ" not in L.get("/config")[1])
print("== 8. 错误处理 ==")
for p in ("/api/nope", "/nope"):
st, _ = L.get(p)
chk("GET %-12s =404" % p, st == 404, "status=%s" % st)
for p in ("/api/summary?from=abc&to=def", "/api/daily?from=2026-13-99"):
st, _ = L.get(p)
chk("GET %-32s 非法日期=400" % p, st == 400, "status=%s" % st)
print("== 9. 新增能力:用户管理 / 审计 / 流式导出 ==")
st, html = L.get("/users")
chk("GET /users 管理员可达", st == 200 and "用户管理" in html, "status=%s" % st)
chk("用户管理页不回传口令散列", "pbkdf2:" not in html)
au = L.jget("/api/audit?size=5")
chk("GET /api/audit 结构完整",
all(k in au for k in ("total", "page", "size", "pages", "actions", "items")),
"keys=%s" % list(au.keys()))
chk("审计条目带 actor/action/at",
not au.get("items") or {"actor", "action", "at"} <= set(au["items"][0]),
"n=%d" % len(au.get("items", [])))
st, csv_body = L.get("/records/export?from=%s&to=%s" % (frm, to))
chk("GET /records/export=200", st == 200, "status=%s" % st)
chk("导出带 UTF-8 BOM(Excel 不乱码)", csv_body.startswith("\ufeff"))
lines = [x for x in csv_body.lstrip("\ufeff").split("\r\n") if x]
chk("导出表头为官网同构列",
lines and lines[0] == "RequestID,积分消耗,User Prompt,模型,客户端,时间",
"header=%s" % (lines[0] if lines else None))
chk("导出行数 == 窗口记录数 + 表头", len(lines) == (sm.get("records") or 0) + 1,
"csv=%d 记录=%s" % (len(lines), sm.get("records")))
r1 = L.jget("/api/records?page=1&size=1&from=%s&to=%s" % (frm, to))
first_id = ((r1.get("items") or [{}])[0]).get("request_id")
chk("导出与明细同源同序(首行 == 明细首条)",
len(lines) > 1 and bool(first_id) and first_id in lines[1],
"api=%s csv=%s" % (first_id, (lines[1][:40] if len(lines) > 1 else None)))
print("== 10. 安全:开放重定向与凭证外泄 ==")
L2 = Live(L.base) # 全新会话,避免已登录被直跳
st, html = L2.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
csrf = m.group(1) if m else ""
st, loc = L2.post_raw("/login", {"username": user, "password": pwd,
"_csrf": csrf, "next": "//evil.com"})
chk("next=//evil.com 被拒(不出现协议相对跳转)",
st == 302 and "evil.com" not in (loc or "") and not (loc or "").startswith("//"),
"status=%s Location=%s" % (st, loc))
L3 = Live(L.base)
st, html = L3.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
st, loc = L3.post_raw("/login", {"username": user, "password": pwd,
"_csrf": m.group(1) if m else "", "next": "/records"})
chk("next=/records 站内路径正常放行", st == 302 and loc == "/records",
"status=%s Location=%s" % (st, loc))
st, loc = L3.post_raw("/login", {"username": user, "password": pwd, "_csrf": "wrong"})
chk("错误 CSRF 的登录 POST=400", st == 400, "status=%s" % st)
st, body = L.get("/logout")
chk("GET /logout 不执行退出(仅提示)", st == 200 and "退出" in body, "status=%s" % st)
st, html = L.get("/")
chk("GET /logout 后仍处于登录态", st == 200 and "概览" in html, "status=%s" % st)
def main() -> int:
ap = argparse.ArgumentParser(description="对运行中的用量门户做端到端验收")
ap.add_argument("--base", default="http://127.0.0.1:8848", help="服务地址")
ap.add_argument("-u", "--user", default="admin", help="登录用户名")
ap.add_argument("-p", "--password", default="admin123", help="登录密码")
ap.add_argument("--from", dest="frm", default="2026-09-08", help="验收窗口起")
ap.add_argument("--to", dest="to", default="2026-09-14", help="验收窗口止")
ap.add_argument("--timeout", type=int, default=20)
a = ap.parse_args()
print("目标:%s 窗口:%s ~ %s\n" % (a.base, a.frm, a.to))
run(Live(a.base, a.timeout), a.user, a.password, a.frm, a.to)
print("\nRESULT: ok=%d fail=%d" % (OK, FAIL))
if FAILS:
print("失败项:%s" % "、".join(FAILS))
return 1 if FAIL else 0
if __name__ == "__main__":
sys.exit(main())