diff --git a/Agent.md b/Agent.md index b53df1aa..ea6ef940 100644 --- a/Agent.md +++ b/Agent.md @@ -119,7 +119,7 @@ Community needs voiced in HN agent-UI discussions map directly to EMRG's design: pkill -f "emrg.server"; rm -f ~/.emrg/emrgd.token; python -m emrg ``` -Python: `uv run pytest tests/ -v` (1128) — import check: `uv run python -c "from emrg.client.app import run_client"` +Python: `uv run pytest tests/ -v` (1133) — import check: `uv run python -c "from emrg.client.app import run_client"` GUI: `cd emrg/gui && npm test` (95: 45 daemon_client + 20 conn-manager + 8 integration + 6 build-config + 7 gui-state + 3 preload-api + 4 boot-contract + 2 theme-guard) — syntax: `node --check main.js preload.js daemon_client.js` Renderer: `cd emrg/gui/renderer && npm run typecheck && npm test` (448: 5 snapshot-store + 9 utils + 3 ErrorBoundary + 2 App smoke + 11 commands + 4 copywriting + 11 i18n + 11 markdown + 15 transcript + 7 TranscriptView + 15 history + 22 composer + 14 Composer + 12 sidebar + 17 Sidebar + 9 fileTree + 9 FileTree + 16 resultPanel + 8 ResultPanel + 27 workspaceView + 10 WorkspaceView + 10 dialog + 6 Dialog + 9 ConfirmDialog + 9 RenameDialog + 10 dialogLists + 3 HelpDialog + 9 MemoryDialog + 6 SkillsDialog + 9 openSession + 6 WelcomeDialog + 8 OpenSessionDialog + 7 NewSessionDialog + 7 rewind + 8 RewindDialog + 7 GithubDeviceDialog + 15 daemonBridge + 7 DaemonBridgeProvider + 25 Shell + 15 DialogHost + 20 SettingsPanel + 6 TaskFormDialog + 5 RantDialog + 4 vendorMarkdown) + `npm run build` → `renderer/dist/` CI: `uv run pytest` (ubuntu + **windows-2025 matrix** — Windows pytest 回归在 PR CI 即失败,v0.2.29 教训 #725) + GUI tests + **actionlint workflow lint** (`rhysd/actionlint@v1.7.12` gate, #444 — workflow 解析错误在 PR CI 即失败,如 `if:` secrets 上下文) diff --git a/scripts/llm-cost-report.py b/scripts/llm-cost-report.py new file mode 100644 index 00000000..d38fadde --- /dev/null +++ b/scripts/llm-cost-report.py @@ -0,0 +1,292 @@ +#!/usr/bin/env python3 +"""Estimate LLM API spend from session llm.jsonl usage records. + +Comparable-tool inspiration: Claude Code v2.1.247 added `/claude-api +cost-optimize` (API cost profiling). EMRG already persists real token usage +per request — emrg/session.py ``append_llm`` writes every exchange to +``llm.jsonl`` with ``type: request`` (carrying ``model``) followed by +``type: response`` (carrying the ``usage`` dict: prompt_tokens / +completion_tokens / reasoning_tokens / cache_hit_tokens). The missing piece +was only an aggregator + pricing table, which this script provides. + +Behavior: + + * Scans session directories (default: every session in + ``~/.emrg/sessions_index.json``; ``--root`` for one ``.emrg/sessions`` + dir; ``--session-id`` for a single indexed session). + * Reads ``llm.jsonl`` plus rotated ``.N`` backups in chronological order + (``.3`` oldest → ``llm.jsonl`` newest) so a response in an old backup is + still paired with its request. + * Pairs each response's usage with the model of the nearest preceding + request record; a response with no preceding request (truncated log) is + reported under ``(unknown)``. + * Applies a per-model pricing table ($ per 1M tokens, prompt/completion). + Prices are public-list approximations — output is an ESTIMATE, not an + invoice. + * Prompt-cache semantics: if ``prompt_tokens_details.cached_tokens`` is + present (OpenAI style, prompt_tokens excludes cache), it is billed at 10% + of the prompt price. Otherwise ``cache_hit_tokens`` is assumed to be + included in prompt_tokens (DeepSeek style) and billed at 10% of the + prompt price. + * Unknown models are listed separately at $0 with a warning; add or + override prices with ``--pricing model:prompt_ppm:completion_ppm``. + +Exit code is always 0 for a successful report (broken lines are skipped with +a warning on stderr — a report tool must not fail on one corrupt record). + +Usage examples:: + + scripts/llm-cost-report.py # all indexed sessions + scripts/llm-cost-report.py --session-id emrg-evolution-emrg-task + scripts/llm-cost-report.py --root ~/.emrg/.emrg/sessions --json + scripts/llm-cost-report.py --pricing my-model:0.5:1.5 +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from collections import OrderedDict +from pathlib import Path +from typing import Iterator + +# ($ per 1,000,000 tokens: prompt, completion) — public-list approximations. +PPM_DEFAULTS: dict[str, tuple[float, float]] = OrderedDict( + [ + ("deepseek-v4-flash", (0.20, 1.20)), + ("deepseek-v4", (1.00, 3.00)), + ("deepseek-reasoner", (0.55, 2.19)), + ("deepseek-chat", (0.27, 1.10)), + ("claude-sonnet-4-5", (3.00, 15.00)), + ("claude-opus-4-5", (5.00, 25.00)), + ("claude-3-5-sonnet", (3.00, 15.00)), + ("gpt-5", (1.25, 10.00)), + ("gpt-5-mini", (0.25, 2.00)), + ("gpt-4o", (2.50, 10.00)), + ("qwen-max", (1.60, 6.40)), + ("qwen-plus", (0.40, 1.20)), + ] +) + +CACHE_PRICE_FRACTION = 0.10 # prompt-cache hits are typically ~10% of prompt price + +UNKNOWN = "(unknown)" + + +def default_index_path() -> Path: + return Path.home() / ".emrg" / "sessions_index.json" + + +def session_dirs( + root: str | None = None, + session_id: str | None = None, + index_path: Path | None = None, +) -> list[Path]: + """Resolve the session directories to scan.""" + if root: + base = Path(root).expanduser() + if not base.is_dir(): + sys.stderr.write(f"warning: --root {root} is not a directory\n") + return [] + return sorted(p for p in base.iterdir() if p.is_dir()) + index = index_path or default_index_path() + try: + mapping = json.loads(index.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + sys.stderr.write(f"warning: cannot read {index}: {exc}\n") + return [] + if session_id: + path = mapping.get(session_id) + if not path: + sys.stderr.write(f"warning: session {session_id!r} not in {index}\n") + return [] + return [Path(path)] + return [Path(p) for p in mapping.values() if Path(p).is_dir()] + + +def _llm_log_files(session_dir: Path) -> list[Path]: + """llm.jsonl + rotated backups, oldest first (chronological).""" + + def key(f: Path) -> int: + suffix = f.name[len("llm.jsonl"):] # "" or ".1", ".2", ... + # Rotation shifts main -> .1 -> .2 -> .3, so .3 is the OLDEST backup + # and the main file is the newest. Read chronologically: .3 first. + return -int(suffix[1:]) if suffix else 10**9 + + files = sorted(session_dir.glob("llm.jsonl*"), key=key) + return files + + +def iter_llm_records(session_dir: Path) -> Iterator[tuple[str, dict]]: + """Yield (model, usage) for every response record that carries usage. + + ``model`` is taken from the nearest preceding request record in + chronological order (across rotation files); ``(unknown)`` when none. + Corrupt lines are skipped with a warning. + """ + last_model: str | None = None + for path in _llm_log_files(session_dir): + try: + fh = path.open(encoding="utf-8", errors="replace") + except OSError as exc: + sys.stderr.write(f"warning: cannot open {path}: {exc}\n") + continue + with fh: + for lineno, line in enumerate(fh, 1): + line = line.strip() + if not line: + continue + try: + rec = json.loads(line) + except ValueError: + sys.stderr.write( + f"warning: {path.name}:{lineno}: corrupt line skipped\n" + ) + continue + if not isinstance(rec, dict): + continue + if rec.get("type") == "request": + if isinstance(rec.get("model"), str): + last_model = rec["model"] + continue + if rec.get("type") == "response" and isinstance(rec.get("usage"), dict): + yield (last_model or UNKNOWN, rec["usage"]) + + +def estimate_cost(usage: dict, prompt_ppm: float, completion_ppm: float) -> float: + """Dollar estimate for one response, honoring prompt-cache pricing.""" + prompt = usage.get("prompt_tokens") or 0 + completion = usage.get("completion_tokens") or 0 + details = usage.get("prompt_tokens_details") + cache = 0 + if isinstance(details, dict) and isinstance(details.get("cached_tokens"), int): + # OpenAI style: prompt_tokens excludes the cache; bill cache separately. + cache = details["cached_tokens"] + billed_prompt = prompt + else: + # DeepSeek style: cache_hit_tokens ⊂ prompt_tokens. + cache = usage.get("cache_hit_tokens") or 0 + billed_prompt = max(prompt - cache, 0) + return ( + prompt_ppm * billed_prompt / 1e6 + + completion_ppm * completion / 1e6 + + CACHE_PRICE_FRACTION * prompt_ppm * cache / 1e6 + ) + + +def report(session_dirs_: list[Path], pricing: dict[str, tuple[float, float]]) -> dict: + """Aggregate usage+cost across sessions. Returns per-model rows + totals.""" + agg: dict[str, dict] = {} + unknown_models: set[str] = set() + for sd in session_dirs_: + for model, usage in iter_llm_records(sd): + row = agg.setdefault( + model, + { + "requests": 0, + "prompt_tokens": 0, + "cache_hit_tokens": 0, + "completion_tokens": 0, + "cost": 0.0, + "priced": False, + }, + ) + row["requests"] += 1 + row["prompt_tokens"] += usage.get("prompt_tokens") or 0 + row["cache_hit_tokens"] += usage.get("cache_hit_tokens") or 0 + row["completion_tokens"] += usage.get("completion_tokens") or 0 + pp, cp = pricing.get(model, (0.0, 0.0)) + row["cost"] += estimate_cost(usage, pp, cp) + if model in pricing: + row["priced"] = True + else: + unknown_models.add(model) + totals = { + "requests": sum(r["requests"] for r in agg.values()), + "prompt_tokens": sum(r["prompt_tokens"] for r in agg.values()), + "cache_hit_tokens": sum(r["cache_hit_tokens"] for r in agg.values()), + "completion_tokens": sum(r["completion_tokens"] for r in agg.values()), + "cost": sum(r["cost"] for r in agg.values()), + } + return {"models": agg, "totals": totals, "unknown_models": sorted(unknown_models)} + + +def render_human(result: dict) -> str: + models = result["models"] + if not models: + return "No usage records found in the given sessions.\n" + rows = sorted(models.items(), key=lambda kv: -kv[1]["cost"]) + lines = [ + f"{'Model':<22}{'Requests':>9}{'Prompt tok':>13}{'Cache tok':>11}" + f"{'Compl tok':>12}{'Est cost':>14}" + ] + lines.append("-" * len(lines[0])) + for name, r in rows: + priced = " *" if not r["priced"] else "" + cost = f"${r['cost']:,.4f}" + lines.append( + f"{name:<22}{r['requests']:>9,}{r['prompt_tokens']:>13,}" + f"{r['cache_hit_tokens']:>11,}{r['completion_tokens']:>12,}" + f"{cost:>14}{priced}" + ) + t = result["totals"] + lines.append("-" * len(lines[0])) + cost = f"${t['cost']:,.4f}" + lines.append( + f"{'TOTAL':<22}{t['requests']:>9,}{t['prompt_tokens']:>13,}" + f"{t['cache_hit_tokens']:>11,}{t['completion_tokens']:>12,}" + f"{cost:>14}" + ) + lines.append("") + if result["unknown_models"]: + lines.append( + "(*) no price — add with --pricing model:prompt_ppm:completion_ppm: " + + ", ".join(result["unknown_models"]) + ) + else: + lines.append("Estimated cost — public-list approximations, not an invoice.") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Estimate LLM API spend from session llm.jsonl usage records." + ) + parser.add_argument("--root", help="scan this .emrg/sessions directory") + parser.add_argument("--session-id", help="scan one indexed session by id") + parser.add_argument( + "--json", action="store_true", help="emit machine-readable JSON instead of a table" + ) + parser.add_argument( + "--pricing", + action="append", + default=[], + metavar="MODEL:PROMPT_PPM:COMPLETION_PPM", + help="add/override a model price ($ per 1M tokens); repeatable", + ) + args = parser.parse_args(argv) + + pricing = dict(PPM_DEFAULTS) + for spec in args.pricing: + parts = spec.split(":") + if len(parts) != 3: + parser.error(f"--pricing expects MODEL:PROMPT_PPM:COMPLETION_PPM, got {spec!r}") + model, prompt_ppm, completion_ppm = parts + pricing[model] = (float(prompt_ppm), float(completion_ppm)) + + dirs = session_dirs(root=args.root, session_id=args.session_id) + if not dirs: + sys.stderr.write("No session directories to scan.\n") + return 1 + result = report(dirs, pricing) + if args.json: + print(json.dumps(result, ensure_ascii=False, indent=2)) + else: + print(render_human(result)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_llm_cost_report.py b/tests/test_llm_cost_report.py new file mode 100644 index 00000000..406a9726 --- /dev/null +++ b/tests/test_llm_cost_report.py @@ -0,0 +1,149 @@ +"""Tests for scripts/llm-cost-report.py (LLM API cost profiler). + +Covers: token/cost aggregation math, model pairing across rotated llm.jsonl +backups, the unknown-model fallback, the OpenAI-style cache-details branch, +and the CLI end-to-end. +""" + +import json +import subprocess +import sys +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parent.parent +SCRIPT = REPO_ROOT / "scripts" / "llm-cost-report.py" + + +def _write_llm_log(session_dir: Path, records: list[dict], name: str = "llm.jsonl") -> None: + (session_dir / name).write_text( + "".join(json.dumps(r, ensure_ascii=False) + "\n" for r in records), + encoding="utf-8", + ) + + +def _request(model: str) -> dict: + return {"type": "request", "model": model, "messages": [], "tools": None, "payload": None} + + +def _response(prompt: int, completion: int, cache: int = 0) -> dict: + return { + "type": "response", + "content": "", + "finish_reason": "stop", + "usage": { + "prompt_tokens": prompt, + "completion_tokens": completion, + "reasoning_tokens": 0, + "cache_hit_tokens": cache, + }, + } + + +def _run_report(root: Path, *extra: str) -> dict: + out = subprocess.run( + [sys.executable, str(SCRIPT), "--root", str(root), "--json", *extra], + capture_output=True, + text=True, + check=True, + ) + return json.loads(out.stdout) + + +def test_aggregates_tokens_and_cost_per_model(tmp_path: Path) -> None: + sd = tmp_path / "s1" + sd.mkdir() + # deepseek-v4-flash @ $0.20/$1.20 per 1M: cache billed at 10% of prompt price. + _write_llm_log(sd, [_request("deepseek-v4-flash"), _response(1000, 200, cache=900)]) + result = _run_report(tmp_path, "--pricing", "deepseek-v4-flash:0.2:1.2") + + model = result["models"]["deepseek-v4-flash"] + assert model["requests"] == 1 + assert model["prompt_tokens"] == 1000 + assert model["cache_hit_tokens"] == 900 + assert model["completion_tokens"] == 200 + # billed prompt = 1000 - 900 = 100 -> 0.20*100/1e6 + # completion = 200 -> 1.20*200/1e6 ; cache = 0.1*0.20*900/1e6 + expected = (0.20 * 100 + 1.20 * 200 + 0.1 * 0.20 * 900) / 1e6 + assert model["cost"] == pytest.approx(expected) + assert result["totals"] == { + "requests": 1, + "prompt_tokens": 1000, + "cache_hit_tokens": 900, + "completion_tokens": 200, + "cost": pytest.approx(expected), + } + assert result["unknown_models"] == [] + + +def test_model_pairing_across_rotation_and_unknown(tmp_path: Path) -> None: + sd = tmp_path / "s2" + sd.mkdir() + # Old backup (.1): request only. Newer main file: response pairs with it. + _write_llm_log(sd, [_request("deepseek-v4")], name="llm.jsonl.1") + _write_llm_log(sd, [_response(500, 50)]) + # A response with no preceding request anywhere -> (unknown). + sd2 = tmp_path / "s3" + sd2.mkdir() + _write_llm_log(sd2, [_response(10, 5)]) + + result = _run_report(tmp_path, "--pricing", "deepseek-v4:1.0:3.0") + assert result["models"]["deepseek-v4"]["requests"] == 1 + assert result["models"]["(unknown)"]["requests"] == 1 + assert result["models"]["(unknown)"]["cost"] == 0.0 + assert result["unknown_models"] == ["(unknown)"] + + +def test_openai_style_prompt_tokens_details(tmp_path: Path) -> None: + sd = tmp_path / "s4" + sd.mkdir() + resp = { + "type": "response", + "content": "", + "finish_reason": "stop", + "usage": { + "prompt_tokens": 400, # excludes cache + "completion_tokens": 100, + "prompt_tokens_details": {"cached_tokens": 300}, + }, + } + _write_llm_log(sd, [_request("gpt-5"), resp]) + result = _run_report(tmp_path, "--pricing", "gpt-5:1.25:10.0") + model = result["models"]["gpt-5"] + # prompt 400 billed in full; cache 300 at 10% of prompt price + expected = (1.25 * 400 + 10.0 * 100 + 0.1 * 1.25 * 300) / 1e6 + assert model["cost"] == pytest.approx(expected) + + +def test_cli_human_output_smoke(tmp_path: Path) -> None: + sd = tmp_path / "s5" + sd.mkdir() + _write_llm_log(sd, [_request("deepseek-v4-flash"), _response(100, 10)]) + out = subprocess.run( + [sys.executable, str(SCRIPT), "--root", str(tmp_path)], + capture_output=True, + text=True, + check=True, + ) + assert "deepseek-v4-flash" in out.stdout + assert "TOTAL" in out.stdout + assert "Estimated cost" in out.stdout + + +def test_cross_backup_pairing_reads_oldest_first(tmp_path: Path) -> None: + """A response in a mid backup pairs with the request in the OLDEST backup. + + Rotation shifts main -> .1 -> .2 -> .3, so .3 is the oldest backup and + must be read before .1. Reading .1 first would pair this response with + a request that chronologically came AFTER it. + """ + sd = tmp_path / "s6" + sd.mkdir() + _write_llm_log(sd, [_request("deepseek-v4")], name="llm.jsonl.3") + _write_llm_log(sd, [_request("other-model")], name="llm.jsonl.1") + _write_llm_log(sd, [_response(500, 50)], name="llm.jsonl.2") + + result = _run_report(tmp_path, "--pricing", "deepseek-v4:1.0:3.0") + assert result["models"]["deepseek-v4"]["requests"] == 1 + assert "other-model" not in result["models"]