feat: virtual company M2 control + tier-3/4 capabilities + per-dept session fix
- M2-a/b/c: stop/pause/resume/skip dept, plan/rework approval gates, budget & timeout circuit breakers, per-dept cancel and on-demand rework - tier-3: batched event persistence (_EventBuffer), project/dept concurrency semaphores, company_ws event streaming - tier-4: cross-dept deliverable handoff, CEO memory feedback loop, per-dept model override, one-click Markdown report export (RFC5987) - fix: per-department independent SQLAlchemy session in parallel waves (shared Session race across asyncio tasks in company_orchestrator) - fix: rename process_open_fds gauge to app_process_open_fds (collides with prometheus_client ProcessCollector on Linux) - fix: dept role resolution ladder, real token usage & cost tracking, deliverable file capture (mtime scan of dept workspace) - tests: test_company_orchestrator.py 22 cases with mocked LLM - frontend: CompanyControlRoom, companyExecution store, EChart, dept model config dialog, ws proxy, build/typecheck decoupling - docs: virtual company design/usage docs, M2 control plan, deepseek4pro handover docs Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -254,11 +254,44 @@ class AgentRuntime:
|
||||
# 返回 True 表示预算充足;返回 False 或抛出异常表示超限
|
||||
self.on_llm_invocation: Optional[Callable[[], Any]] = None
|
||||
|
||||
# 真实 token 累计(与 token_budget 开关无关,永远统计本次运行的真实用量)
|
||||
self._real_usage: Dict[str, int] = {
|
||||
"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0, "llm_calls": 0,
|
||||
}
|
||||
|
||||
def _record_real_usage(self, metrics: Dict[str, Any]) -> None:
|
||||
"""无条件累计每次 LLM 调用的真实用量(来自 API usage)。"""
|
||||
p = int(metrics.get("prompt_tokens") or 0)
|
||||
c = int(metrics.get("completion_tokens") or 0)
|
||||
t = int(metrics.get("total_tokens") or 0) or (p + c)
|
||||
self._real_usage["prompt_tokens"] += p
|
||||
self._real_usage["completion_tokens"] += c
|
||||
self._real_usage["total_tokens"] += t
|
||||
self._real_usage["llm_calls"] += 1
|
||||
|
||||
def _real_token_usage_payload(self) -> Dict[str, Any]:
|
||||
"""构造随终态事件下发的真实用量 dict(形状含 total_tokens,供下游/前端与计费消费)。"""
|
||||
return {
|
||||
"prompt_tokens": self._real_usage["prompt_tokens"],
|
||||
"completion_tokens": self._real_usage["completion_tokens"],
|
||||
"total_tokens": self._real_usage["total_tokens"],
|
||||
"llm_calls": self._real_usage["llm_calls"],
|
||||
"model": self.config.llm.model,
|
||||
}
|
||||
|
||||
def _attach_token_usage(self, result: AgentResult) -> AgentResult:
|
||||
"""将 TokenBudget 摘要附加到 AgentResult(若启用)。"""
|
||||
"""将 token 用量摘要附加到 AgentResult。预算启用则用其摘要;否则用真实累计用量。"""
|
||||
from app.agent_runtime.schemas import TokenUsageInfo
|
||||
if self._token_budget:
|
||||
from app.agent_runtime.schemas import TokenUsageInfo
|
||||
result.token_usage = TokenUsageInfo(**self._token_budget.summary())
|
||||
else:
|
||||
ru = self._real_usage
|
||||
result.token_usage = TokenUsageInfo(
|
||||
cumulative_total=ru["total_tokens"],
|
||||
cumulative_prompt=ru["prompt_tokens"],
|
||||
cumulative_completion=ru["completion_tokens"],
|
||||
llm_call_count=ru["llm_calls"],
|
||||
)
|
||||
return result
|
||||
|
||||
def _build_execution_log_kwargs(self, user_input: str, result: AgentResult, latency_ms: int) -> dict:
|
||||
@@ -391,6 +424,7 @@ class AgentRuntime:
|
||||
llm_callback_ctx = {"step_type": "think", "tool_name": None}
|
||||
|
||||
def _llm_callback(metrics: Dict[str, Any]):
|
||||
self._record_real_usage(metrics) # 无条件累计真实 token(不受预算开关影响)
|
||||
# Token 预算追踪 (P2)
|
||||
if self._token_budget:
|
||||
prompt_tok = metrics.get("prompt_tokens", 0)
|
||||
@@ -869,6 +903,7 @@ class AgentRuntime:
|
||||
llm_callback_ctx = {"step_type": "think", "tool_name": None}
|
||||
|
||||
def _llm_callback(metrics: Dict[str, Any]):
|
||||
self._record_real_usage(metrics) # 无条件累计真实 token(不受预算开关影响)
|
||||
# Token 预算追踪 (P2)
|
||||
if self._token_budget:
|
||||
prompt_tok = metrics.get("prompt_tokens", 0)
|
||||
@@ -1016,7 +1051,9 @@ class AgentRuntime:
|
||||
self.context.add_user_message(fix_prompt)
|
||||
continue # 回到 ReAct 循环,让 LLM 修正
|
||||
|
||||
token_usage_final = self._token_budget.summary() if self._token_budget else None
|
||||
token_usage_final = self._real_token_usage_payload()
|
||||
if self._token_budget:
|
||||
token_usage_final = {**self._token_budget.summary(), **token_usage_final}
|
||||
yield {
|
||||
"type": "final",
|
||||
"content": final_text,
|
||||
@@ -1278,7 +1315,9 @@ class AgentRuntime:
|
||||
# 提取知识到全局知识池(即便截断,工具调用序列仍有参考价值)
|
||||
if last_content:
|
||||
await self._extract_global_knowledge(user_input, last_content, steps)
|
||||
token_usage_truncated = self._token_budget.summary() if self._token_budget else None
|
||||
token_usage_truncated = self._real_token_usage_payload()
|
||||
if self._token_budget:
|
||||
token_usage_truncated = {**self._token_budget.summary(), **token_usage_truncated}
|
||||
yield {
|
||||
"type": "final",
|
||||
"content": last_content or "已达最大迭代次数,但模型未返回最终回答。",
|
||||
|
||||
Reference in New Issue
Block a user