chore: 项目定名为 workbuddy-portal,容器化并补齐文档体系

## 项目定名
- 目录 wb_usage_portal → workbuddy-portal
- Python 包 wb_usage → workbuddy_portal(含 session cookie 名)
- 界面品牌统一为 WorkBuddy Portal;项目标识收敛到 config 单一来源

## 容器化
- Dockerfile:多阶段构建,依赖层与源码解耦;非 root(uid 1000);内置健康检查
- docker-compose.yml:单服务 + 绑定挂载 data/logs + 日志轮转 + TZ
- docker/entrypoint.sh:幂等初始化 → exec serve(LF 行尾,已由 .gitattributes 锁定)
- docker/healthcheck.py:纯标准库探活 /login(slim 镜像无 curl)
- .dockerignore / .env.example;数据目录可用 WB_DATA_DIR 等环境变量覆盖

## 文档
- docs/USER-GUIDE.md    用户使用手册(含 9 张真实界面截图)
- docs/DEPLOYMENT.md    部署运维(Docker / 裸机 / 反代 / 备份 / 推 Gitea 注册表)
- docs/ARCHITECTURE.md  架构与设计说明(含已知坑与红线、验证体系)
- docs/API.md           接口参考(路径 / 参数 / 返回结构 / 错误码)
- docs/FAQ.md           常见问题;docs/CHANGELOG.md 变更日志

## 修复缺陷(8)
1. /records/export 必然 500:生成器在请求上下文销毁后才迭代,改用自建连接
2. 大屏页图表全白:相对路径把 echarts.min.js 解析成 /vendor/... → 404
3. /users 500:路由已注册但模板缺失
4. 明细页日期筛选失效:视图传 f.frm、模板读 f.from
5. 配置页维护按钮全死:调用了不存在的 WBU.bindMaint()
6. 审计只能看最近 40 条:LIMIT 写死
7. 明细页多跑一条无用 SELECT:day_list() 取了没人用
8. 登录页锁定阈值未从配置注入

## 安全加固
- 新增 safe_next():拒绝 //evil.com 等协议相对 URL 的开放重定向
- 缺 CSRF 的写请求统一 400
- 默认开启云端 HTTPS 证书校验(ssl_verify=1);Cookie 是账号凭证
- 登录失败计数表加上限与 TTL
- /logout 拆分为 POST(执行) + GET(仅提示),防 <img src=/logout> 静默退出
- settings 内部簿记键 slot:* 读写两侧过滤,不再从 /api/settings 泄漏

## 内部质量与工具
- 设置项写时校验 + 读时兜底,杜绝「一个手滑的数字让采集整个跑不起来」
- 全局 ValueError → 400:手写 query string 不再暴露 500 页面
- CSV 导出改 csv.writer 流式写入(原手工拼串,字段含逗号会串列)
- bundle 明细加 20000 上限并回传 recordsTotal/recordsTruncated,不静默丢数据
- tools/smoke.py 离线回归 99 项;tools/check_live.py 真实 HTTP 56 项
- tools/shots.py Playwright 逐页截图 + JS 报错收集

## 验证
- compileall 通过;smoke 99/99;对容器实例 check_live 56/56;截图 0 JS 报错
- 容器内采集实测成功(trigger=startup 补跑:新增 11 条)
这个提交包含在:
2026-09-14 14:55:50 +08:00
当前提交 86631ae7ab
共修改 58 个文件,包含 9409 行新增和 0 行删除
+353
查看文件
@@ -0,0 +1,353 @@
# -*- coding: utf-8 -*-
"""SQL 聚合层:所有统计都在 SQLite 里算完再出去,页面不再搬运全量明细。
返回结构刻意与旧版 dashboard/data/*.json 的字段保持一致(d/c/k/fc/bc/m/h、
id/c/m/cl/t/px …),这样 ECharts 大屏的渲染代码一行都不用改,只换数据来源。
"""
from datetime import datetime, timedelta
from . import config, db
SCHEMA_VERSION = 4
TOP_EXCERPT_LEN = 400
DEFAULT_TOP_N = 200
# 大屏页一次最多下发多少条窗口明细(页面要拿它在浏览器里算窗口 TOP / 散点)。
# 存档长大后不能让首屏体量线性膨胀,所以设上限并在响应里标注是否被截断。
BUNDLE_RECORDS_CAP = 20000
# 分页接口单页上限(导出走独立的流式游标,不受此限)
MAX_PAGE_SIZE = 500
def norm_day(s):
"""把各种写法归一成 'YYYY-MM-DD';无法识别返回 None。
接受 '2026-09-08'、'2026-09-08 12:00:00'、'2026/09/08'。
"""
if s is None:
return None
t = str(s).strip().replace("/", "-")
if not t:
return None
t = t.split(" ")[0].split("T")[0]
try:
return datetime.strptime(t, "%Y-%m-%d").strftime("%Y-%m-%d")
except ValueError:
return None
def norm_window(frm=None, to=None):
"""归一化并保证 from <= to。返回 (from, to),任一无法识别则为 None。"""
f, t = norm_day(frm), norm_day(to)
if f and t and f > t:
f, t = t, f
return f, t
def _where(frm=None, to=None, model=None, client=None, q=None):
w, p = [], []
if frm:
w.append("day >= ?")
p.append(frm)
if to:
w.append("day <= ?")
p.append(to)
if model:
w.append("model = ?")
p.append(model)
if client:
w.append("client = ?")
p.append(client)
if q:
w.append("(prompt LIKE ? OR request_id LIKE ?)")
p += ["%" + q + "%", "%" + q + "%"]
return ("WHERE " + " AND ".join(w)) if w else "", p
def _excerpt(s, n):
s = " ".join(str(s or "").split())
if not n or len(s) <= n:
return s
return s[:n].rstrip() + "…"
# ---------------- 逐日聚合 ----------------
def daily(conn, frm=None, to=None, with_maps=True):
w, p = _where(frm, to)
days = {}
for r in conn.execute(
"SELECT day d, COUNT(*) k, ROUND(SUM(credits),2) c,"
" SUM(CASE WHEN credits<=0 THEN 1 ELSE 0 END) fc,"
" MIN(ts) first, MAX(ts) last"
" FROM usage_records %s GROUP BY day ORDER BY day" % w, p):
k = r["k"] or 0
fc = r["fc"] or 0
days[r["d"]] = {"d": r["d"], "c": r["c"] or 0.0, "k": k, "fc": fc, "bc": k - fc,
"m": {}, "h": [0.0] * 24, "first": r["first"] or "", "last": r["last"] or ""}
if with_maps and days:
for r in conn.execute(
"SELECT day d, model, ROUND(SUM(credits),2) c FROM usage_records %s"
" GROUP BY day, model" % w, p):
if r["d"] in days:
days[r["d"]]["m"][r["model"]] = r["c"] or 0.0
for r in conn.execute(
"SELECT day d, hour h, ROUND(SUM(credits),2) c FROM usage_records %s"
" GROUP BY day, hour" % w, p):
if r["d"] in days:
days[r["d"]]["h"][r["h"]] = r["c"] or 0.0
return [days[d] for d in sorted(days)]
# ---------------- 维度汇总 ----------------
def _dim(conn, col, frm=None, to=None):
w, p = _where(frm, to)
rows = conn.execute(
"SELECT %s name, COUNT(*) calls, ROUND(SUM(credits),2) credits,"
" SUM(CASE WHEN credits<=0 THEN 1 ELSE 0 END) freeCalls,"
" COUNT(DISTINCT day) activeDays, MIN(day) firstDay, MAX(day) lastDay"
" FROM usage_records %s GROUP BY %s"
" ORDER BY credits DESC, calls DESC" % (col, w, col), p)
out = []
for r in rows:
calls = r["calls"] or 0
fc = r["freeCalls"] or 0
cr = r["credits"] or 0.0
out.append({
"name": r["name"], "calls": calls, "credits": cr,
"freeCalls": fc, "billableCalls": calls - fc,
"activeDays": r["activeDays"] or 0,
"firstDay": r["firstDay"] or "", "lastDay": r["lastDay"] or "",
"avgPerCall": round(cr / calls, 4) if calls else 0.0,
"freeRate": round(fc / calls, 4) if calls else 0.0,
})
return out
def dims(conn, frm=None, to=None):
hours = {int(r["name"]): r for r in _dim(conn, "printf('%02d',hour)", frm, to)}
hlist = []
for i in range(24):
h = "%02d" % i
hlist.append(hours.get(i, {"name": h, "calls": 0, "credits": 0.0, "freeCalls": 0,
"billableCalls": 0, "activeDays": 0, "firstDay": "",
"lastDay": "", "avgPerCall": 0.0, "freeRate": 0.0}))
for i, o in enumerate(hlist):
o["name"] = "%02d" % i
return {"model": _dim(conn, "model", frm, to),
"client": _dim(conn, "client", frm, to),
"hour": hlist}
# ---------------- 单笔榜 ----------------
def top(conn, frm=None, to=None, n=DEFAULT_TOP_N):
w, p = _where(frm, to)
items = []
for i, r in enumerate(conn.execute(
"SELECT request_id, credits, model, client, ts, prompt FROM usage_records %s"
" ORDER BY credits DESC, ts LIMIT ?" % w, p + [n])):
items.append({"rank": i + 1, "id": r["request_id"], "c": r["credits"] or 0.0,
"m": r["model"], "cl": r["client"], "t": r["ts"],
"px": _excerpt(r["prompt"], TOP_EXCERPT_LEN)})
return {"n": len(items), "items": items}
# ---------------- 明细(窗口内精简记录,不带 prompt 全文)----------------
def records(conn, frm=None, to=None, excerpt=96, limit=0, offset=0, newest_first=False):
w, p = _where(frm, to)
order = "ORDER BY ts DESC, request_id DESC" if newest_first else "ORDER BY ts, request_id"
sql = ("SELECT request_id, credits, model, client, ts,"
" substr(replace(replace(COALESCE(prompt,''),char(10),' '),char(13),' '),1,?) px"
" FROM usage_records %s %s" % (w, order))
args = [excerpt] + p
if limit:
sql += " LIMIT ? OFFSET ?"
args += [limit, offset]
return [{"id": r["request_id"], "c": r["credits"] or 0.0, "m": r["model"],
"cl": r["client"], "t": r["ts"], "px": (r["px"] or "")} for r in conn.execute(sql, args)]
def records_page(conn, frm=None, to=None, model=None, client=None, q=None,
page=1, size=50, order="ts_desc", with_prompt=True):
frm, to = norm_window(frm, to)
size = max(1, min(int(size or 50), MAX_PAGE_SIZE))
page = max(1, int(page or 1))
w, p = _where(frm, to, model, client, q)
total = conn.execute("SELECT COUNT(*) FROM usage_records %s" % w, p).fetchone()[0]
agg = conn.execute("SELECT ROUND(COALESCE(SUM(credits),0),2) c FROM usage_records %s" % w, p).fetchone()
orders = {"ts_desc": "ts DESC, request_id", "ts": "ts, request_id",
"credits_desc": "credits DESC, ts DESC", "credits": "credits, ts"}
ob = orders.get(order, orders["ts_desc"])
cols = "request_id,credits,model,client,ts,first_seen,last_seen,day,hour"
cols += ",prompt" if with_prompt else ""
rows = conn.execute("SELECT %s FROM usage_records %s ORDER BY %s LIMIT ? OFFSET ?" % (cols, w, ob),
p + [size, (page - 1) * size])
items = []
for r in rows:
o = {"request_id": r["request_id"], "credits": r["credits"] or 0.0,
"model": r["model"], "client": r["client"], "ts": r["ts"], "day": r["day"],
"hour": r["hour"], "first_seen": r["first_seen"], "last_seen": r["last_seen"]}
if with_prompt:
o["prompt"] = r["prompt"] or ""
items.append(o)
return {"total": total, "credits": agg["c"] or 0.0, "page": page, "size": size,
"pages": max(1, (total + size - 1) // size), "items": items}
def iter_records(conn, frm=None, to=None, model=None, client=None, q=None,
order="ts_desc", with_prompt=True, batch=1000):
"""流式产出明细(给导出用):不把整个结果集读进内存。"""
frm, to = norm_window(frm, to)
w, p = _where(frm, to, model, client, q)
orders = {"ts_desc": "ts DESC, request_id", "ts": "ts, request_id",
"credits_desc": "credits DESC, ts DESC", "credits": "credits, ts"}
ob = orders.get(order, orders["ts_desc"])
cols = "request_id,credits,model,client,ts"
cols += ",prompt" if with_prompt else ""
cur = conn.execute("SELECT %s FROM usage_records %s ORDER BY %s" % (cols, w, ob), p)
while True:
chunk = cur.fetchmany(batch)
if not chunk:
return
for r in chunk:
o = {"request_id": r["request_id"], "credits": r["credits"] or 0.0,
"model": r["model"], "client": r["client"], "ts": r["ts"]}
if with_prompt:
o["prompt"] = r["prompt"] or ""
yield o
# ---------------- 全局元信息 ----------------
def months(conn):
return [r[0] for r in conn.execute(
"SELECT DISTINCT substr(day,1,7) m FROM usage_records ORDER BY m")]
def totals(conn, frm=None, to=None):
w, p = _where(frm, to)
r = conn.execute(
"SELECT COUNT(*) n, ROUND(COALESCE(SUM(credits),0),2) c,"
" SUM(CASE WHEN credits<=0 THEN 1 ELSE 0 END) fc,"
" MIN(day) d0, MAX(day) d1, COUNT(DISTINCT day) nd,"
" COUNT(DISTINCT model) nm, COUNT(DISTINCT client) nc,"
" MIN(ts) t0, MAX(ts) t1"
" FROM usage_records %s" % w, p).fetchone()
n = r["n"] or 0
fc = r["fc"] or 0
return {"records": n, "credits": r["c"] or 0.0, "calls": n,
"freeCalls": fc, "billableCalls": n - fc,
"firstDay": r["d0"] or "", "lastDay": r["d1"] or "", "days": r["nd"] or 0,
"models": r["nm"] or 0, "clients": r["nc"] or 0,
"first": r["t0"] or "", "last": r["t1"] or ""}
def day_list(conn):
return [r[0] for r in conn.execute("SELECT DISTINCT day FROM usage_records ORDER BY day")]
def manifest(conn):
t = totals(conn)
db_bytes = conn.execute("PRAGMA page_count").fetchone()[0] * \
conn.execute("PRAGMA page_size").fetchone()[0]
runs = conn.execute("SELECT COUNT(*) FROM collect_runs").fetchone()[0]
last_run = conn.execute("SELECT * FROM collect_runs ORDER BY id DESC LIMIT 1").fetchone()
health = db.get_setting(conn, "cookie", "")
mons = months(conn) # 只算一次(原来在返回体里调了两遍)
return {
"schema": SCHEMA_VERSION,
"generated": db.now_str(),
"archive": "data/usage.sqlite",
"producer": "workbuddy-portal(Flask + SQLite)",
"note": "数据正本为 SQLite 表 usage_records;daily/dims/top 均为 SQL 实时聚合结果。",
"totals": {"records": t["records"], "credits": t["credits"], "calls": t["calls"],
"freeCalls": t["freeCalls"], "billableCalls": t["billableCalls"],
"days": day_list(conn), "months": mons,
"models": t["models"], "clients": t["clients"],
"first": t["first"], "last": t["last"],
"topCredits": (conn.execute("SELECT COALESCE(MAX(credits),0) FROM usage_records")
.fetchone()[0] or 0.0)},
"months": mons,
"sources": [
{"path": "usage_records", "role": "明细正本(SQLite 表)", "count": t["records"],
"bytes": db_bytes},
{"path": "daily 聚合视图", "role": "逐日聚合(SQL GROUP BY day)", "count": t["days"], "bytes": 0},
{"path": "dims 聚合视图", "role": "模型/客户端/时段汇总(SQL GROUP BY)",
"count": t["models"] + t["clients"] + 24, "bytes": 0},
{"path": "top 查询", "role": "单笔消耗榜(ORDER BY credits DESC)", "count": DEFAULT_TOP_N, "bytes": 0},
{"path": "collect_runs", "role": "采集运行历史", "count": runs, "bytes": 0},
],
"focusDay": (last_run["win_to"] or "")[:10] if last_run else "",
"health": {"cookie": bool(health and health.strip()),
"lastRunAt": last_run["started_at"] if last_run else "",
"lastRunStatus": last_run["status"] if last_run else ""},
}
def bundle(conn, frm=None, to=None, top_n=DEFAULT_TOP_N, excerpt=140,
records_cap=BUNDLE_RECORDS_CAP):
"""大屏页一次请求拿齐所需数据。
窗口裁剪:records(明细)、dims(维度)、totals(KPI)随 frm/to 变化。
刻意不裁剪:daily(全量逐日,供日历与日期轴,体量小)、top(全局 TOP 榜)。
records 有上限(records_cap)并在响应里标注 recordsTruncated,
避免存档长大后「全部」区间把整包明细都压到浏览器。
"""
frm, to = norm_window(frm, to)
tot = totals(conn, frm, to)
# 只有真的会超限时才改成「取最近 N 条」,避免改变现有正常路径的行为
truncated = tot["records"] > records_cap
recs = records(conn, frm, to, excerpt=excerpt,
limit=records_cap if truncated else 0, newest_first=truncated)
return {
"manifest": manifest(conn),
"daily": daily(conn), # 全量逐日(体量小,供日历与日期轴)
"dims": dims(conn, frm, to), # 窗口内维度
"top": top(conn, None, None, top_n)["items"], # 全局 TOP 榜(对应「全局 TOP200」视图)
"records": recs,
"recordsTotal": tot["records"],
"recordsCap": records_cap,
"recordsTruncated": truncated,
"totals": tot,
"window": {"from": frm or "", "to": to or ""},
}
# ---------------- 环比 ----------------
def summary(conn, frm, to):
"""KPI + 环比。前一段必须完整落在存档范围内,否则不给假数字。
frm/to 会先归一化(容错 '2026-09-08 12:00:00'、'2026/09/08' 等写法),
并在 from > to 时自动交换——否则环比区间会算到未来去。
"""
frm, to = norm_window(frm, to)
if not frm or not to:
# 无法识别的日期:退化成全量口径,不抛异常(API 层会先校验并返回 400)
t = totals(conn)
frm, to = t["firstDay"], t["lastDay"]
if not frm or not to:
frm = to = datetime.now().strftime("%Y-%m-%d")
cur = totals(conn, frm, to)
days = (datetime.strptime(to, "%Y-%m-%d") - datetime.strptime(frm, "%Y-%m-%d")).days + 1
p_to = (datetime.strptime(frm, "%Y-%m-%d") - timedelta(days=1)).strftime("%Y-%m-%d")
p_frm = (datetime.strptime(p_to, "%Y-%m-%d") - timedelta(days=days - 1)).strftime("%Y-%m-%d")
first_day = conn.execute("SELECT MIN(day) FROM usage_records").fetchone()[0]
prev = None
if first_day and p_frm >= first_day:
prev = totals(conn, p_frm, p_to)
out = dict(cur)
out["window"] = {"from": frm, "to": to, "days": days}
out["avgPerCall"] = round(cur["credits"] / cur["calls"], 4) if cur["calls"] else 0.0
out["prev"] = prev
if prev:
out["delta"] = {
"credits": (cur["credits"] - prev["credits"]) / prev["credits"] * 100 if prev["credits"] else None,
"calls": (cur["calls"] - prev["calls"]) / prev["calls"] * 100 if prev["calls"] else None,
"window": {"from": p_frm, "to": p_to},
}
else:
out["delta"] = None
# 残日:最后一天不是完整的一天
if to == datetime.now().strftime("%Y-%m-%d"):
row = conn.execute("SELECT MAX(ts) FROM usage_records WHERE day=?", (to,)).fetchone()
if row and row[0]:
out["partial"] = {"date": to, "hhmm": row[0][11:16]}
return out