diff --git a/.gitignore b/.gitignore index 729dec3..4040dfb 100644 --- a/.gitignore +++ b/.gitignore @@ -5,18 +5,27 @@ # ============================================================================= # ---- 数据与运行产物:正本不进版本库(体积大、含凭证衍生物)---- +# 同时写 data/* 与 data/**/* 两种:只写前者会漏掉子目录 +# (曾因此让 data/demo/usage.sqlite 逃过忽略规则) data/*.sqlite data/*.sqlite-wal data/*.sqlite-shm +data/**/*.sqlite +data/**/*.sqlite-wal +data/**/*.sqlite-shm # 含 secret_key,泄露等于会话签名密钥外泄,绝不可提交 data/instance.json +data/**/instance.json data/exports/*.csv # 界面截图(tools/shots.py 生成的临时产物;手册配图在 docs/images/) data/shots/ +# 示例数据(tools/demo_data.py 生成,随时可重建,不必入库) +data/demo/ + # ---- 日志 ---- logs/* diff --git a/docs/API.md b/docs/API.md index bdb897a..9ff73a7 100644 --- a/docs/API.md +++ b/docs/API.md @@ -47,16 +47,16 @@ "producer": "workbuddy-portal(Flask + SQLite)", "note": "...", "totals": { - "records": 1665, "credits": 8513.36, "calls": 1086, - "freeCalls": 579, "billableCalls": 507, - "models": 12, "clients": 3, - "first": "2026-08-10 00:00:00", "last": "2026-09-14 14:51:00", - "topCredits": 319.5 + "records": 944, "credits": 4961.63, "calls": 944, + "freeCalls": 328, "billableCalls": 616, + "models": 7, "clients": 3, + "first": "2026-08-16 09:12:00", "last": "2026-09-14 15:52:00", + "topCredits": 412.8 }, "months": ["2026-08", "2026-09"], - "sources": [{ "path": "usage.sqlite", "role": "primary", "count": 1665, "bytes": 1234567 }], + "sources": [{ "path": "usage.sqlite", "role": "primary", "count": 944, "bytes": 434176 }], "focusDay": "2026-09-14", - "health": { "cookie": true, "lastRunAt": "2026-09-14 14:51:28", "lastRunStatus": "ok" } + "health": { "cookie": true, "lastRunAt": "2026-09-14 15:52:10", "lastRunStatus": "ok" } } ``` @@ -73,14 +73,14 @@ { "manifest": { ... 同上 ... }, "daily": [ - { "d": "2026-09-08", "c": 1423.5, "k": 88, "fc": 40, "bc": 48, - "m": { "deepseek-v4-flash": 800.2, "glm-5.3-flash": 623.3 }, + { "d": "2026-09-08", "c": 168.4, "k": 31, "fc": 12, "bc": 19, + "m": { "demo-flash": 62.1, "demo-pro": 41.7 }, "h": [0,0,0,0,0,0,0,0,0,12.5, ...] } ], "dims": { "model": [...], "client": [...], "hour": [...] }, - "top": [ { "id": "...", "c": 319.5, "m": "kimi-k3-1", "cl": "VSCode", "t": "2026-09-12 15:04:00", "px": "摘要…" } ], + "top": [ { "id": "...", "c": 412.8, "m": "demo-reason", "cl": "vscode", "t": "2026-09-12 15:04:00", "px": "摘要…" } ], "records": [ { "id": "...", "c": 5.78, "m": "...", "cl": "...", "t": "...", "px": "..." } ], - "recordsTotal": 1665, + "recordsTotal": 944, "recordsCap": 20000, "recordsTruncated": false, "totals": { ... }, @@ -106,13 +106,13 @@ KPI + 环比。`from`/`to` 缺省时自动取全量区间。 ```json { - "records": 428, "credits": 2145.6, "calls": 300, - "freeCalls": 120, "billableCalls": 180, + "records": 226, "credits": 1180.4, "calls": 226, + "freeCalls": 78, "billableCalls": 148, "firstDay": "2026-09-08", "lastDay": "2026-09-14", "days": 7, - "models": 9, "clients": 2, + "models": 7, "clients": 3, "first": "...", "last": "...", "window": { "from": "2026-09-08", "to": "2026-09-14", "days": 7 }, - "avgPerCall": 7.15, + "avgPerCall": 5.22, "prev": { ... 上一段等长窗口的同样结构 ... }, "delta": { "credits": 12.3, "calls": -4.1, "window": { "from": "...", "to": "..." } }, "partial": { "date": "2026-09-14", "hhmm": "14:52" } @@ -132,7 +132,7 @@ KPI + 环比。`from`/`to` 缺省时自动取全量区间。 | `from` / `to` | 筛选窗口 | ```json -{ "days": [ { "d": "2026-09-08", "c": 1423.5, "k": 88, "fc": 40, "bc": 48, +{ "days": [ { "d": "2026-09-08", "c": 168.4, "k": 31, "fc": 12, "bc": 19, "m": {...}, "h": [24 个元素] } ] } ``` @@ -146,8 +146,8 @@ KPI + 环比。`from`/`to` 缺省时自动取全量区间。 ```json { - "model": [ { "name": "deepseek-v4-flash", "credits": 3784.86, "calls": 234, "avg": 16.17, "free": 0 } ], - "client": [ { "name": "VSCode", ... } ], + "model": [ { "name": "demo-flash", "credits": 1286.4, "calls": 252, "avg": 5.1, "free": 0 } ], + "client": [ { "name": "vscode", ... } ], "hour": [ { "name": "14", ... } ] } ``` @@ -162,7 +162,7 @@ KPI + 环比。`from`/`to` 缺省时自动取全量区间。 | `n` | 50 | 返回条数(1~1000) | ```json -[ { "id": "…", "c": 319.5, "m": "kimi-k3-1", "cl": "VSCode", +[ { "id": "…", "c": 412.8, "m": "demo-reason", "cl": "vscode", "t": "2026-09-12 15:04:00", "px": "截断后的 Prompt 摘要" } ] ``` @@ -184,7 +184,7 @@ KPI + 环比。`from`/`to` 缺省时自动取全量区间。 { "items": [ { "request_id": "…", "credits": 5.78, "model": "…", "client": "…", "ts": "2026-09-14 14:20:00", "prompt": "…" } ], - "total": 1665, "page": 1, "size": 50, "pages": 34, + "total": 944, "page": 1, "size": 50, "pages": 19, "window": { "from": "...", "to": "..." } } ``` @@ -292,7 +292,7 @@ KPI + 环比。`from`/`to` 缺省时自动取全量区间。 | 场景 | 响应 | |---|---| -| 成功 | `200 {"ok": true, "result": {"message": "新增 11 条,重复 6 条,存档共 1665 条", ...}}` | +| 成功 | `200 {"ok": true, "result": {"message": "新增 11 条,重复 6 条,存档共 944 条", ...}}` | | 已有采集在跑 | `409 {"ok": false, "error": "busy", "message": "..."}` | | Cookie 失效 | `401 {"ok": false, "error": "cookie_expired", "message": "..."}` | | 云端异常 | `502 {"ok": false, "error": "api", "message": "..."}` | diff --git a/docs/DEPLOYMENT.md b/docs/DEPLOYMENT.md index 643c339..beb6820 100644 --- a/docs/DEPLOYMENT.md +++ b/docs/DEPLOYMENT.md @@ -544,10 +544,10 @@ sqlite3.OperationalError: unable to open database file docker compose -f docker-compose.yml -f docker-compose.hostdir.yml up -d docker compose exec portal python -c \ "import sqlite3;print(sqlite3.connect('/app/data/usage.sqlite').execute('select count(*) from usage_records').fetchone())" -# -> (1665,) 容器侧正常 +# -> (944,) 容器侧正常 python manage.py stats # 宿主侧随手跑一次「纯读」的 CLI -# -> 存档:1665 条 … +# -> 存档:944 条 … docker compose exec portal python -c \ "import sqlite3;sqlite3.connect('/app/data/usage.sqlite')" diff --git a/docs/USER-GUIDE.md b/docs/USER-GUIDE.md index 0f0108a..f057672 100644 --- a/docs/USER-GUIDE.md +++ b/docs/USER-GUIDE.md @@ -3,6 +3,11 @@ > 面向**使用者**(不是开发者)。读完这份就能独立完成日常操作: > 登录 → 看用量 → 配置采集 → 查明细 → 导数据 → 处理常见异常。 +> **关于配图**:本文所有截图都用 `tools/demo_data.py` 生成的**合成示例数据**渲染 +> ——模型名统一是 `demo-*`,客户端为 `vscode`/`webconsole`/`sdk`,Prompt 是通用示例文本。 +> 所以你可以照着重现出几乎一样的界面,也不必担心文档里夹带真实账号信息。 +> 想自己搭一份这样的环境:`python tools/demo_data.py` 然后按输出的提示起服务即可。 + **目录** - [一、这个系统是做什么的](#一这个系统是做什么的) diff --git a/docs/images/01-overview.png b/docs/images/01-overview.png index e9248b4..8e4c109 100644 Binary files a/docs/images/01-overview.png and b/docs/images/01-overview.png differ diff --git a/docs/images/02-records.png b/docs/images/02-records.png index 67e82b3..b407d27 100644 Binary files a/docs/images/02-records.png and b/docs/images/02-records.png differ diff --git a/docs/images/03-tasks.png b/docs/images/03-tasks.png index 9eb7a82..91013c4 100644 Binary files a/docs/images/03-tasks.png and b/docs/images/03-tasks.png differ diff --git a/docs/images/04-config.png b/docs/images/04-config.png index a5790c4..e789add 100644 Binary files a/docs/images/04-config.png and b/docs/images/04-config.png differ diff --git a/docs/images/05-logs.png b/docs/images/05-logs.png index b6f2216..f088aa4 100644 Binary files a/docs/images/05-logs.png and b/docs/images/05-logs.png differ diff --git a/docs/images/06-users.png b/docs/images/06-users.png index 4d85034..0fb72d2 100644 Binary files a/docs/images/06-users.png and b/docs/images/06-users.png differ diff --git a/docs/images/07-dashboard.png b/docs/images/07-dashboard.png index f013d03..f28a9c3 100644 Binary files a/docs/images/07-dashboard.png and b/docs/images/07-dashboard.png differ diff --git a/docs/images/08-dashboard-interact.png b/docs/images/08-dashboard-interact.png index 387d5a1..6f3aefb 100644 Binary files a/docs/images/08-dashboard-interact.png and b/docs/images/08-dashboard-interact.png differ diff --git a/tools/demo_data.py b/tools/demo_data.py new file mode 100644 index 0000000..0f4019e --- /dev/null +++ b/tools/demo_data.py @@ -0,0 +1,287 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# SPDX-License-Identifier: MIT +# Copyright (c) 2026 Wang Chuanli +"""生成**脱敏示例数据**,用于本地体验、界面截图与文档配图。 + +为什么需要它 +------------ +`docs/images/` 里的界面截图必须是可公开的,但真实库里的 Prompt 全文、 +请求 ID、用量分布与本机路径都属于私有信息。与其手工打码,不如用一份 +**完全合成**的数据集重新截图——顺便也让后来者能一键把界面跑起来看。 + +生成内容 +-------- +| 表 | 说明 | +|---|---| +| `usage_records` | 约 900 条合成记录,跨 30 天,含假模型名 / 假 Prompt / 偏斜的积分分布 | +| `collect_runs` | 14 条采集历史,含 ok / warn / error 三种状态 | +| `settings` | 走项目默认值(`config.DEFAULTS`),**不写入任何凭据** | +| `audit_log` | 30 条操作审计 | +| `users` | 由 `db.init_db()` 建一个管理员 | + +用法 +---- + # 默认写到 data/demo/(该目录在 .gitignore 内,不会误提交) + python tools/demo_data.py + + # 指定目录与管理员口令,然后起服务看效果 + python tools/demo_data.py --out data/demo --admin-password demo123 + WB_DATA_DIR=$PWD/data/demo python manage.py serve --port 8849 --no-scheduler + +注意 +---- +本脚本**只写 `--out` 指定的目录**,不会读取也不会修改 `data/usage.sqlite`。 +已存在的目标库会被拒绝覆盖,除非显式加 `--force`。 +""" +from __future__ import annotations + +import argparse +import os +import random +import sys +from datetime import datetime, timedelta + +BASE = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, BASE) + +# ---- 合成素材:刻意保持通用,不含任何真实的产品名、Prompt 或业务信息 ---- +MODELS = [ + # (模型名, 权重, 积分中位数) + ("demo-flash", 26, 0.35), + ("demo-lite", 20, 0.60), + ("demo-pro", 18, 2.20), + ("demo-reason", 14, 4.80), + ("demo-mini", 12, 0.22), + ("demo-vision", 7, 6.50), + ("demo-nano", 3, 0.12), +] + +CLIENTS = [("vscode", 74), ("webconsole", 18), ("sdk", 8)] + +# 写进 settings 的假 Cookie。刻意用重复串,一眼就能看出不是真凭据; +# 作用只是让概览页的健康指示灯是绿的(首装状态是「缺 Cookie」告警)。 +DEMO_COOKIE = "wb_demo_session=" + "deadbeef" * 15 + +PROMPTS_SHORT = [ + "帮我解释一下这段代码的作用", + "把这段 SQL 优化一下,避免全表扫描", + "写一个 Python 脚本,把目录里的 CSV 批量转成 JSON", + "这个报错是什么意思:connection refused", + "帮我 review 一下这个接口设计,有什么问题", + "解释一下 JWT 和 Session 的区别", + "生成一份周报模板", + "把下面的需求整理成技术方案", + "这个正则怎么写:匹配 11 位手机号", + "Explain the difference between processes and threads", + "Refactor this function to be more readable", + "What is the time complexity of this algorithm?", + "Write unit tests for this module", + "How do I make this loop faster?", + "Summarize the key points of this document", +] + +PROMPTS_LONG = [ + "你是资深后端工程师。请审查下面的接口实现,重点看:\n" + "1) 并发写入是否安全;\n2) 异常分支是否都有兜底;\n3) 有没有可以合并的重复查询。\n" + "请按「问题 / 影响 / 建议」三栏输出。", + + "下面的表结构要支持按天和按模型两个维度聚合,数据量大约千万级。\n" + "请给出索引设计,并说明每个索引命中的查询模式。", + + "把这段代码从回调风格改成 async/await,保持对外行为不变," + "并补充必要的错误处理。改完给出前后对比。", + + "Review the provided module and suggest improvements.\n" + "Focus on readability, error handling, and testability.\n" + "Reply as a short bullet list.", +] + + +def _weighted(rng: random.Random, pairs): + """按权重取一项(pairs 为 (值, 权重) 列表)。""" + total = sum(w for _, w in pairs) + pick = rng.uniform(0, total) + acc = 0.0 + for val, w in pairs: + acc += w + if pick <= acc: + return val + return pairs[-1][0] + + +def _credits(rng: random.Random, median: float) -> float: + """对数正态分布的积分值,偶尔出现大额。""" + val = rng.lognormvariate(0.0, 0.85) * median + if rng.random() < 0.02: # 2% 的大额长任务 + val *= rng.uniform(12, 60) + return round(min(val, 900.0), 2) + + +def _prompt(rng: random.Random) -> str | None: + r = rng.random() + if r < 0.12: # 一部分请求不带 Prompt + return None + if r < 0.55: + return rng.choice(PROMPTS_SHORT) + if r < 0.75: + return rng.choice(PROMPTS_LONG) + # 其余用短句拼接,避免重复得过于整齐 + return rng.choice(PROMPTS_SHORT) + ":" + rng.choice(PROMPTS_SHORT) + + +def build(out_dir: str, days: int, seed: int, admin_password: str, + admin_user: str) -> str: + """建库并写入合成数据,返回数据库文件路径。""" + out_dir = os.path.abspath(out_dir) + os.makedirs(out_dir, exist_ok=True) + db_file = os.path.join(out_dir, "usage.sqlite") + + # 必须在 import 项目模块之前设好环境变量:config 在导入时读取它们。 + # 日志目录用 setdefault —— 容器里 WB_LOG_DIR 已由镜像 ENV 指定, + # 不该在数据卷下再凭空建一个 logs/。 + os.environ["WB_DATA_DIR"] = out_dir + os.environ.setdefault("WB_LOG_DIR", os.path.join(out_dir, "logs")) + os.environ["WB_DB"] = db_file + + from workbuddy_portal import db # noqa: E402 + + conn = db.connect() + try: + db.init_db(conn, create_admin=True, admin_user=admin_user, + admin_password=admin_password) + + # 写入一个**明显是假值**的 Cookie:让概览页的健康状态显示为「正常」 + # 而不是首装的「缺 Cookie」告警——演示与截图应当呈现「配置完成」后的样子。 + # 这个值不会被任何真实服务接受,也不含任何真实凭据。 + db.set_setting(conn, "cookie", DEMO_COOKIE) + + rng = random.Random(seed) + now = datetime.now().replace(second=0, microsecond=0) + today0 = now.replace(hour=0, minute=0, second=0) + model_pairs = [(m, w) for m, w, _ in MODELS] + medians = {m: md for m, _, md in MODELS} + client_pairs = list(CLIENTS) + + # ---------- usage_records ---------- + rows = [] + for d in range(days - 1, -1, -1): + day0 = today0 - timedelta(days=d) + for _ in range(rng.randint(14, 46)): + # 工作时间加权:9-19 点更密 + hour = _weighted(rng, [(h, 6 if 9 <= h <= 19 else 1) for h in range(24)]) + minute = rng.randint(0, 59) + sec = rng.randint(0, 59) + ts = day0 + timedelta(hours=hour, minutes=minute, seconds=sec) + if ts > now: + continue + model = _weighted(rng, model_pairs) + client = _weighted(rng, client_pairs) + rid = "req-%s" % "".join(rng.choice("0123456789abcdef") for _ in range(16)) + stamp = ts.strftime("%Y-%m-%d %H:%M:%S") + rows.append(( + rid, stamp, ts.strftime("%Y-%m-%d"), hour, model, client, + _credits(rng, medians[model]), _prompt(rng), + stamp, stamp, stamp, + )) + conn.execute("BEGIN") + conn.executemany( + "INSERT OR REPLACE INTO usage_records" + "(request_id,ts,day,hour,model,client,credits,prompt," + " first_seen,last_seen,cloud_ts) VALUES(?,?,?,?,?,?,?,?,?,?,?)", rows) + conn.execute("COMMIT") + + # ---------- collect_runs ---------- + runs = [] + total = 0 + for i in range(14, 0, -1): + started = now - timedelta(minutes=i * 37 + rng.randint(0, 9)) + fetched = rng.randint(3, 22) + dup = rng.randint(0, max(1, fetched - 2)) + added = max(0, fetched - dup) + total += added + status = "ok" + msg = "新增 %d 条,重复 %d 条,存档共 %d 条" % (added, dup, total) + exit_code = 0 + if i == 9: + status, exit_code = "warn", 1 + msg = "参数错误:from 不是合法日期;abc(正确写法 2026-09-01)" + if i == 5: + status, exit_code = "error", 2 + msg = "云端返回 401 Unauthorized:Cookie 可能已过期,请重新粘贴" + fetched = dup = added = 0 + trigger = "schedule" if i % 3 else "manual" + runs.append(( + trigger, status, started.strftime("%Y-%m-%d %H:%M:%S"), + (started + timedelta(milliseconds=rng.randint(180, 1400)) + ).strftime("%Y-%m-%d %H:%M:%S"), + rng.randint(180, 1400), + (started - timedelta(days=1)).strftime("%Y-%m-%d %H:%M:%S"), + started.strftime("%Y-%m-%d %H:%M:%S"), + fetched, added, dup, total, 0, exit_code, msg, + "[%s] %s" % (status, msg), + )) + conn.execute("BEGIN") + conn.executemany( + "INSERT INTO collect_runs(trigger,status,started_at,finished_at,duration_ms," + "win_from,win_to,fetched,added,dup,total,conflicts,exit_code,message,detail)" + " VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)", runs) + conn.execute("COMMIT") + + # ---------- audit_log ---------- + audits = [] + # IP 用 RFC 5737 的文档专用网段(TEST-NET-1), + # 保证示例里出现的地址永远不可能是真实主机 + actions = [ + ("login", "登录成功", "192.0.2.10"), + ("login", "登录成功", "192.0.2.10"), + ("settings", "修改:调度时刻 09:00,17:00", "192.0.2.10"), + ("settings", "修改:分页大小 200", "192.0.2.10"), + ("maintenance.count", "存档当前 %d 条记录" % total, "192.0.2.10"), + ("settings_rejected", "参数错误:page_size 必须是数字(条/页)", "192.0.2.10"), + ] + for i in range(30): + act, detail, ip = actions[i % len(actions)] + at = now - timedelta(hours=i * 3 + rng.randint(0, 2)) + audits.append((at.strftime("%Y-%m-%d %H:%M:%S"), admin_user, act, detail, ip)) + conn.execute("BEGIN") + conn.executemany( + "INSERT INTO audit_log(at,actor,action,detail,ip) VALUES(?,?,?,?,?)", audits) + conn.execute("COMMIT") + + # 统计一下,便于打印 + n = conn.execute("SELECT COUNT(*) FROM usage_records").fetchone()[0] + c = conn.execute("SELECT ROUND(SUM(credits),2) FROM usage_records").fetchone()[0] + d = conn.execute("SELECT COUNT(DISTINCT day) FROM usage_records").fetchone()[0] + print("示例库:%s" % db_file) + print(" 记录 %d 条 / 积分 %s / 覆盖 %d 天" % (n, c, d)) + print(" 管理员 %s,口令 %s" % (admin_user, admin_password)) + print(" 该目录在 .gitignore 内,不会被提交") + finally: + conn.close() + return db_file + + +def main() -> int: + ap = argparse.ArgumentParser(description="生成脱敏示例数据(完全合成,不碰真实库)") + ap.add_argument("--out", default=os.path.join(BASE, "data", "demo"), + help="输出目录,默认 data/demo") + ap.add_argument("--days", type=int, default=30, help="覆盖天数,默认 30") + ap.add_argument("--seed", type=int, default=20260914, help="随机种子,保证可复现") + ap.add_argument("--admin-user", default="admin") + ap.add_argument("--admin-password", default="admin123", + help="示例管理员口令,默认 admin123(仅供本地演示)") + ap.add_argument("--force", action="store_true", help="目标库已存在时覆盖") + a = ap.parse_args() + + db_file = os.path.join(os.path.abspath(a.out), "usage.sqlite") + if os.path.exists(db_file) and not a.force: + print("目标库已存在:%s" % db_file) + print("如需重建请加 --force") + return 1 + build(a.out, a.days, a.seed, a.admin_password, a.admin_user) + return 0 + + +if __name__ == "__main__": + sys.exit(main())