feat(multi-user): 多用户化 + 凭证加密 + 自助注册与图形验证码

数据隔离
- settings / usage_records 主键改为 (user_id, key) / (user_id, request_id),
  索引一律以 user_id 打头;collect_runs / audit_log 增加 user_id
- query / collect / scheduler 全链路把 uid 作为 conn 之后的第一个位置参数且无默认值
  (漏传直接 TypeError,不会退化成「返回全量」)
- 配置三级回落 个人→实例→DEFAULTS;NO_FALLBACK_KEYS={cookie,user_agent} 不回落

凭证保密
- 新增 workbuddy_portal/crypto.py:手写 ChaCha20(RFC8439 §2.3) + HMAC-SHA256
  encrypt-then-MAC,零第三方依赖;主密钥 cookie_key 与 SECRET_KEY 分键位存放
- get_secret() 是取明文的唯一通道;get_settings() 把加密键置空;
  secret_state() 只回 {set,chars,tail,broken};升级时自动加密历史明文

注册与验证码
- 新增 /register 与 workbuddy_portal/captcha.py(手写 PNG + 点阵字模 + 干扰线)
- 验证码答案只存服务端表、不进 session,一次性、5 分钟过期、按 purpose 隔离
- allow_register / register_max_per_ip / captcha_policy / captcha_length 四个实例级开关
- 失败限速改为 IP + 用户名双维度;停用账号每请求回查、立即失效

页面
- 新增 /profile(个人中心)与注册页;登录页加验证码与自助注册入口
- /config 增加凭证状态、cookie_broken 告警、实例级设置区;/users 增加邮箱/状态与启停

修复
- base.html 顶层 {% set me %} 覆盖子模板同名变量,导致个人中心「注册于」渲染为空
- WB_COOKIE_SECURE 未写进 compose 的 environment,在 .env 里设了不生效
- 「修改登录密码」提示写「至少 6 位」,与实际策略(≥8 位 + 两类字符)不符
- 「用户管理」删除说明写「可勾选保留」,与页面实际行为不符
- 注册页与 flash 文案里的 **强调** Markdown 字面量

验证与文档
- smoke.py 99 → 165 项断言(多用户隔离 / 凭证保密 / 注册与验证码 / 3 条防回归)
- check_live.py 56 → 83 项断言(新增注册 / 验证码 / 安全响应头一节)
- demo_data.py 造两个账号;shots.py 自动过验证码、重出 11 张截图
- README / SECURITY / ARCHITECTURE / API / DEPLOYMENT / USER-GUIDE / FAQ / CHANGELOG / CONTRIBUTING 同步
这个提交包含在:
2026-09-15 17:32:35 +08:00
父节点 23799b4ea5
当前提交 df7db3582e
共修改 46 个文件,包含 4348 行新增和 953 行删除
+165 -20
查看文件
@@ -21,13 +21,17 @@
from __future__ import annotations
import argparse
import base64
import http.cookiejar
import json
import os
import re
import sqlite3
import sys
import urllib.error
import urllib.parse
import urllib.request
import zlib
from datetime import datetime
OK = 0
@@ -40,6 +44,31 @@ def _d(s: str):
return datetime.strptime(s, "%Y-%m-%d")
def decode_session(cj) -> dict:
"""从 Flask 会话 cookie 里解出那份**未加密**的载荷。
Flask 的会话是「签名 + base64,**不加密**」的 —— 也就是说持有 cookie 的人
就能读到里面的内容。本项目因此把验证码答案放在服务端 captchas 表里,
会话里只留一个随机 id;本函数存在的意义就是取出那个 id,
好让自动化验收能跨过验证码这一关(顺便也验证了「答案不在会话里」)。
"""
for c in cj:
if not c.name.startswith("workbuddy_portal_sid"):
continue
seg = urllib.parse.unquote(c.value).split(".")[0]
seg += "=" * (-len(seg) % 4)
try:
raw = base64.urlsafe_b64decode(seg)
try:
raw = zlib.decompress(raw) # 某些版本的 itsdangerous 会压
except zlib.error:
pass
return json.loads(raw.decode("utf-8"))
except Exception: # noqa: BLE001
return {}
return {}
def chk(name: str, cond: bool, extra: str = "") -> None:
global OK, FAIL
if cond:
@@ -59,9 +88,10 @@ class _NoRedirect(urllib.request.HTTPRedirectHandler):
class Live:
def __init__(self, base: str, timeout: int = 20):
def __init__(self, base: str, timeout: int = 20, db_path: str | None = None):
self.base = base.rstrip("/")
self.timeout = timeout
self.db_path = db_path
# 关键:显式清空代理,否则本机代理会把 127.0.0.1 也拦成 502
self.cj = http.cookiejar.CookieJar()
self.op = urllib.request.build_opener(
@@ -114,6 +144,62 @@ class Live:
st, body = self.get(path)
return json.loads(body) if st == 200 else {}
def raw(self, path: str):
"""返回 (status, headers, bytes)——验证码/响应头这类要原始字节的场景用。"""
try:
r = self.op.open(urllib.request.Request(self.base + path), timeout=self.timeout)
return r.status, r.headers, r.read()
except urllib.error.HTTPError as e:
return e.code, e.headers, e.read()
def form_csrf(self, path: str) -> str:
"""取某个页面里的 CSRF 隐藏域(该页面必须与当前会话同源)。"""
_, html = self.get(path)
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
return m.group(1) if m else ""
# ---- 验证码辅助(仅验收脚本用)----
def solve_captcha(self, purpose: str):
"""取一张图 -> 从会话里读 id -> 从本地库里取答案。返回 (答案, 会话载荷)。"""
self.raw("/captcha.png?purpose=" + purpose)
sess = decode_session(self.cj)
cid = sess.get("cap_" + purpose)
if not cid or not self.db_path or not os.path.exists(self.db_path):
return None, sess
try:
con = sqlite3.connect(self.db_path)
try:
row = con.execute("SELECT answer FROM captchas WHERE id=?", (cid,)).fetchone()
finally:
con.close()
except sqlite3.Error:
return None, sess
return (row[0] if row else None), sess
def login(self, user: str, pwd: str, nxt: str = "", follow: bool = True):
"""完整登录(验证码策略为 always 时自动解)。
follow=False 时返回原始 (status, Location),用于验证跳转目标是否安全。
返回 (status, location, need_captcha, session_payload)。
"""
html = self.get("/login")[1]
need_cap = 'name="captcha"' in html
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
data = {"username": user, "password": pwd, "_csrf": m.group(1) if m else ""}
if nxt:
data["next"] = nxt
sess = {}
if need_cap:
ans, sess = self.solve_captcha("login")
if ans is None:
return None, None, True, sess
data["captcha"] = ans
if follow:
st, _ = self.post("/login", data)
return st, None, need_cap, sess
st, loc = self.post_raw("/login", data)
return st, loc, need_cap, sess
def run(L: Live, user: str, pwd: str, frm: str, to: str) -> None:
print("== 1. 未登录访问受保护资源 ==")
@@ -123,13 +209,15 @@ def run(L: Live, user: str, pwd: str, frm: str, to: str) -> None:
st, _ = L.get(p)
chk("GET %-14s 未登录=401" % p, st == 401, "status=%s" % st)
print("== 2. 登录(含 CSRF) ==")
print("== 2. 登录(含 CSRF;验证码策略为 always 时自动解) ==")
st, html = L.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
chk("登录页含 CSRF 隐藏域", bool(m))
st, _ = L.post("/login", {"username": user, "password": pwd,
"_csrf": m.group(1) if m else ""})
chk("登录页含 CSRF 隐藏域", bool(re.search(r'name="_csrf"\s+value="([^"]+)"', html)))
st, _, need_cap, sess = L.login(user, pwd)
chk("登录成功", st in (200, 302), "status=%s" % st)
if need_cap:
# 会话里只应有 id,不该有答案本身
chk("会话里只存验证码 id(不是答案)", bool(sess.get("cap_login")),
"cap_login=%s" % (sess.get("cap_login") or "无"))
st, html = L.get("/")
chk("登录后 GET / 到概览", st == 200 and "概览" in html, "len=%d" % len(html))
@@ -212,11 +300,25 @@ def run(L: Live, user: str, pwd: str, frm: str, to: str) -> None:
print("== 7. 凭据不外泄 ==")
stj = L.jget("/api/settings")
chk("settings 无 cookie 明文字段", "cookie" not in stj, "keys=%s" % list(stj.keys()))
# 契约:settings 里 cookie 这个键**必须为空**(db.get_settings 统一置空),
# 真正的状态只通过 cookie_hint / cookie_broken 这两个派生字段暴露。
chk("settings 里 cookie 字段为空串",
"cookie" in stj and not str(stj.get("cookie") or "").strip(),
"cookie=%r" % stj.get("cookie"))
chk("settings 用 cookie_hint/cookie_broken 代替明文",
"cookie_hint" in stj and "cookie_broken" in stj)
chk("settings 仅回 cookie_hint 掩码",
bool(stj.get("cookie_hint")) and len(str(stj.get("cookie_hint"))) < 200,
"hint=%s" % stj.get("cookie_hint"))
chk("settings 回传实例级键清单", isinstance(stj.get("_globalKeys"), list)
and bool(stj.get("_globalKeys")), "%s" % stj.get("_globalKeys"))
chk("settings 标明能否改实例级配置", stj.get("_canEditGlobal") is True)
chk("配置页 HTML 不含 cookie 明文", "eyJ" not in L.get("/config")[1])
# 密文形态:v1.<b64salt>.<b64nonce>.<b64ct>.<b64tag>,恰好用正则判定,
# 免得把版本号 "v1.2.0" 当成泄漏(这两者前缀撞车)
cipher_re = re.compile(r"v1\.[A-Za-z0-9+/=]{8,}\.[A-Za-z0-9+/=]{8,}\.")
for p in ("/config", "/profile", "/"):
chk("%-9s HTML 里没有 Cookie 密文" % p, not cipher_re.search(L.get(p)[1]))
print("== 8. 错误处理 ==")
for p in ("/api/nope", "/nope"):
@@ -254,20 +356,13 @@ def run(L: Live, user: str, pwd: str, frm: str, to: str) -> None:
"api=%s csv=%s" % (first_id, (lines[1][:40] if len(lines) > 1 else None)))
print("== 10. 安全:开放重定向与凭证外泄 ==")
L2 = Live(L.base) # 全新会话,避免已登录被直跳
st, html = L2.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
csrf = m.group(1) if m else ""
st, loc = L2.post_raw("/login", {"username": user, "password": pwd,
"_csrf": csrf, "next": "//evil.com"})
L2 = Live(L.base, L.timeout, L.db_path) # 全新会话,避免已登录被直跳
st, loc, _, _ = L2.login(user, pwd, nxt="//evil.com", follow=False)
chk("next=//evil.com 被拒(不出现协议相对跳转)",
st == 302 and "evil.com" not in (loc or "") and not (loc or "").startswith("//"),
"status=%s Location=%s" % (st, loc))
L3 = Live(L.base)
st, html = L3.get("/login")
m = re.search(r'name="_csrf"\s+value="([^"]+)"', html)
st, loc = L3.post_raw("/login", {"username": user, "password": pwd,
"_csrf": m.group(1) if m else "", "next": "/records"})
L3 = Live(L.base, L.timeout, L.db_path)
st, loc, _, _ = L3.login(user, pwd, nxt="/records", follow=False)
chk("next=/records 站内路径正常放行", st == 302 and loc == "/records",
"status=%s Location=%s" % (st, loc))
st, loc = L3.post_raw("/login", {"username": user, "password": pwd, "_csrf": "wrong"})
@@ -277,6 +372,50 @@ def run(L: Live, user: str, pwd: str, frm: str, to: str) -> None:
st, html = L.get("/")
chk("GET /logout 后仍处于登录态", st == 200 and "概览" in html, "status=%s" % st)
print("== 11. 多用户:注册入口 / 验证码 / 安全响应头 ==")
L4 = Live(L.base, L.timeout, L.db_path) # 全新未登录会话
st, html = L4.get("/register")
chk("GET /register 可达", st == 200 and "注册" in html, "status=%s" % st)
chk("注册页带验证码图", "capimg" in html and "/captcha.png" in html)
chk("注册页带 CSRF 隐藏域", bool(L4.form_csrf("/register")))
st, html = L4.get("/login")
chk("登录页带验证码图", "capimg" in html and 'name="captcha"' in html)
chk("登录页带自助注册链接", "/register" in html)
# 出图:真实字节 + 禁缓存 + 确实每次都不一样
shots = {}
for purpose in ("login", "register"):
code, hdr, data = L4.raw("/captcha.png?purpose=" + purpose)
chk("GET /captcha.png?purpose=%-8s 出 PNG" % purpose,
code == 200 and data[:4] == b"\x89PNG" and len(data) > 200,
"status=%s len=%d" % (code, len(data)))
chk(" └ 禁缓存 no-store", "no-store" in (hdr.get("Cache-Control") or ""))
chk(" └ 类型 image/png", (hdr.get("Content-Type") or "").startswith("image/png"))
shots[purpose] = data
_, _, again = L4.raw("/captcha.png?purpose=login")
chk("两次取图内容不同(不是一张静态图)", again != shots["login"])
chk("login 与 register 的图互不相同", shots["login"] != shots["register"])
# 图必须由服务端单独下发,不能把答案内联进页面
st, html = L4.get("/login")
chk("登录页没有内联 data: 图片(答案不走页面源码)",
"data:image" not in html and "base64," not in html)
# 安全响应头
code, hdr, _ = L4.raw("/login")
for name, want in (("X-Content-Type-Options", "nosniff"),
("X-Frame-Options", "DENY"),
("Referrer-Policy", "same-origin")):
chk("响应头 %-24s" % name, (hdr.get(name) or "") == want, "=%s" % hdr.get(name))
chk("响应头含 CSP 且 frame-ancestors 'none'",
"frame-ancestors 'none'" in (hdr.get("Content-Security-Policy") or ""))
code, hdr, _ = L4.raw("/captcha.png?purpose=login")
chk("/captcha 路径带 no-store", "no-store" in (hdr.get("Cache-Control") or ""))
# 路径穿越式 purpose 必须被收敛到已知用途,而不是 500
code, _, data = L4.raw("/captcha.png?purpose=../../etc/passwd")
chk("非法 purpose 不报 500", code == 200 and data[:4] == b"\x89PNG",
"status=%s" % code)
def main() -> int:
ap = argparse.ArgumentParser(description="对运行中的用量门户做端到端验收")
@@ -285,11 +424,17 @@ def main() -> int:
ap.add_argument("-p", "--password", default="admin123", help="登录密码")
ap.add_argument("--from", dest="frm", default="2026-09-08", help="验收窗口起")
ap.add_argument("--to", dest="to", default="2026-09-14", help="验收窗口止")
ap.add_argument("--db", default=None,
help="SQLite 路径(默认 <repo>/data/usage.sqlite)。"
"验证码策略为 always 时用它取答案以完成自动登录;"
"指向不存在的文件则跳过需要验证码的登录")
ap.add_argument("--timeout", type=int, default=20)
a = ap.parse_args()
print("目标:%s 窗口:%s ~ %s\n" % (a.base, a.frm, a.to))
run(Live(a.base, a.timeout), a.user, a.password, a.frm, a.to)
db_path = a.db or os.path.join(
os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "data", "usage.sqlite")
print("目标:%s 窗口:%s ~ %s\n验证码答案源:%s\n" % (a.base, a.frm, a.to, db_path))
run(Live(a.base, a.timeout, db_path), a.user, a.password, a.frm, a.to)
print("\nRESULT: ok=%d fail=%d" % (OK, FAIL))
if FAILS:
print("失败项:%s" % "、".join(FAILS))