diff --git a/Agent.md b/Agent.md index ea6ef940..a394fc86 100644 --- a/Agent.md +++ b/Agent.md @@ -119,7 +119,7 @@ Community needs voiced in HN agent-UI discussions map directly to EMRG's design: pkill -f "emrg.server"; rm -f ~/.emrg/emrgd.token; python -m emrg ``` -Python: `uv run pytest tests/ -v` (1133) — import check: `uv run python -c "from emrg.client.app import run_client"` +Python: `uv run pytest tests/ -v` (1136) — import check: `uv run python -c "from emrg.client.app import run_client"` GUI: `cd emrg/gui && npm test` (95: 45 daemon_client + 20 conn-manager + 8 integration + 6 build-config + 7 gui-state + 3 preload-api + 4 boot-contract + 2 theme-guard) — syntax: `node --check main.js preload.js daemon_client.js` Renderer: `cd emrg/gui/renderer && npm run typecheck && npm test` (448: 5 snapshot-store + 9 utils + 3 ErrorBoundary + 2 App smoke + 11 commands + 4 copywriting + 11 i18n + 11 markdown + 15 transcript + 7 TranscriptView + 15 history + 22 composer + 14 Composer + 12 sidebar + 17 Sidebar + 9 fileTree + 9 FileTree + 16 resultPanel + 8 ResultPanel + 27 workspaceView + 10 WorkspaceView + 10 dialog + 6 Dialog + 9 ConfirmDialog + 9 RenameDialog + 10 dialogLists + 3 HelpDialog + 9 MemoryDialog + 6 SkillsDialog + 9 openSession + 6 WelcomeDialog + 8 OpenSessionDialog + 7 NewSessionDialog + 7 rewind + 8 RewindDialog + 7 GithubDeviceDialog + 15 daemonBridge + 7 DaemonBridgeProvider + 25 Shell + 15 DialogHost + 20 SettingsPanel + 6 TaskFormDialog + 5 RantDialog + 4 vendorMarkdown) + `npm run build` → `renderer/dist/` CI: `uv run pytest` (ubuntu + **windows-2025 matrix** — Windows pytest 回归在 PR CI 即失败,v0.2.29 教训 #725) + GUI tests + **actionlint workflow lint** (`rhysd/actionlint@v1.7.12` gate, #444 — workflow 解析错误在 PR CI 即失败,如 `if:` secrets 上下文) diff --git a/emrg/server/daemon.py b/emrg/server/daemon.py index 24319c08..e034804a 100644 --- a/emrg/server/daemon.py +++ b/emrg/server/daemon.py @@ -3143,11 +3143,27 @@ def _detect_silent_anchor_drift(self, session, real_pt: int, estimate: int) -> N session.session_id, shift, real_pt, old_real, estimate, old_est, ) + # Conservative per-image token allowance for vision content (Codex + # #41003 class bug): a pasted image in an image_url block costs the API + # hundreds-to-thousands of tokens, but the char-based estimator counted + # it as ~0 (content was a list, not a str). Auto-compact then never fired + # on image-heavy sessions and the context could overflow. The exact bill + # depends on resolution/detail (OpenAI high-detail ≈ 85 + 170/tile, max + # ~765); 1000 covers typical pasted screenshots. The usage anchor + # (self._usage_anchors) self-corrects the residual on the next real + # API usage, so this only needs to be in the right ballpark. + _TOKENS_PER_IMAGE = 1000 + def _estimate_tokens(self, messages: list[dict]) -> int: """Rough token estimation from OpenAI-format messages. Character-aware: CJK ≈ 2 chars/token, ASCII ≈ 4 chars/token. Adds +3 tokens per message for role/content metadata overhead. + + Vision-aware: content may be a list of parts (OpenAI format: + {"type": "text" | "image_url", ...}) — text parts are char-counted, + each image_url block gets a fixed _TOKENS_PER_IMAGE allowance, and + unknown dict parts fall back to JSON char-counting. """ total = 0 for m in messages: @@ -3155,11 +3171,32 @@ def _estimate_tokens(self, messages: list[dict]) -> int: content = m.get("content") or "" if isinstance(content, str): total += self._count_chars_for_tokens(content) + elif isinstance(content, list): + total += self._estimate_content_parts(content) for tc in (m.get("tool_calls") or []): tc_str = json.dumps(tc, ensure_ascii=False) total += self._count_chars_for_tokens(tc_str) return total + @staticmethod + def _estimate_content_parts(parts: list) -> int: + """Estimate tokens for OpenAI vision-format content parts.""" + total = 0 + for part in parts: + if isinstance(part, str): + total += EmrgServer._count_chars_for_tokens(part) + elif isinstance(part, dict): + ptype = part.get("type") + if ptype == "text" and isinstance(part.get("text"), str): + total += EmrgServer._count_chars_for_tokens(part["text"]) + elif ptype == "image_url": + total += EmrgServer._TOKENS_PER_IMAGE + else: + total += EmrgServer._count_chars_for_tokens( + json.dumps(part, ensure_ascii=False) + ) + return total + @staticmethod def _estimate_text(text: str) -> int: """Rough token count for a plain text string.""" diff --git a/tests/test_daemon.py b/tests/test_daemon.py index da6c2ee2..372dc1a5 100644 --- a/tests/test_daemon.py +++ b/tests/test_daemon.py @@ -973,6 +973,49 @@ def test_estimate_tokens_multiple_messages(): assert result == 7 +def test_estimate_tokens_vision_image_block(): + """Vision content: text parts char-counted, image_url blocks get allowance. + + Codex #41003-class fix: before, list-form content was skipped entirely + (only +3 overhead), so pasted images estimated at ~0 tokens and + auto-compact never fired on image-heavy sessions. + """ + server = _make_server() + msgs = [{ + "role": "user", + "content": [ + {"type": "text", "text": "what is in this screenshot"}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}}, + ], + }] + result = server._estimate_tokens(msgs) + text_tokens = server._count_chars_for_tokens("what is in this screenshot") + # 3 overhead + text tokens + 1 image allowance + assert result == 3 + text_tokens + server._TOKENS_PER_IMAGE + + +def test_estimate_tokens_vision_multiple_images(): + """Three image blocks → three allowances (plus overhead).""" + server = _make_server() + msgs = [{ + "role": "user", + "content": [ + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,BBBB"}}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,CCCC"}}, + ], + }] + result = server._estimate_tokens(msgs) + assert result == 3 + 3 * server._TOKENS_PER_IMAGE + + +def test_estimate_tokens_vision_empty_list(): + """Empty content list → just overhead (no crash).""" + server = _make_server() + msgs = [{"role": "user", "content": []}] + assert server._estimate_tokens(msgs) == 3 + + # ── _estimate_single ──────────────────────────────────────────────