feat: add user-simulation-test team template (系统应用测试团队)

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
renjianbo
2026-06-18 07:47:03 +08:00
parent 502d6322e0
commit ed724dad1b
5 changed files with 353 additions and 1 deletions

View File

@@ -188,6 +188,24 @@ def create_medical_consultation_template(
return {"data": result}
@router.post("/template/user-simulation-test")
def create_user_simulation_test_template(
workspace_id: Optional[str] = Query(None),
db: Session = Depends(get_db),
current_user: User = Depends(get_current_user),
):
"""一键创建「系统应用测试团队」模板。
自动创建 5 个角色 Agent(测试规划师/功能测试员/体验审核员/边界探索员/性能评估员)并组建团队。
若用户已有同名 Agent 则复用。
"""
svc = TeamService(db)
result = svc.create_user_simulation_test_template(
user_id=current_user.id, workspace_id=workspace_id,
)
return {"data": result}
@router.get("")
def list_teams(
workspace_id: Optional[str] = Query(None),

View File

@@ -218,6 +218,7 @@ class TeamOrchestrator:
"tech_doc": "doc_architect",
"health_management": "health_assessor",
"medical_consultation": "triage_specialist",
"user_simulation_test": "test_planner",
}
role = workflow_planner_map.get(workflow)
if role and self._get_agent_by_role(members, role):

View File

@@ -182,6 +182,34 @@ MEDICAL_CONSULTATION_ROLES = {
},
}
USER_SIMULATION_TEST_ROLES = {
"test_planner": {
"label": "测试规划师",
"icon": "📋",
"description": "制定测试策略、设计用户场景用例、协调团队分工执行",
},
"functional_tester": {
"label": "功能测试员",
"icon": "🔍",
"description": "模拟真实用户验证功能完整性、业务流程准确性和数据一致性",
},
"ux_reviewer": {
"label": "体验审核员",
"icon": "👤",
"description": "从终端用户视角审核交互体验、可用性和无障碍性",
},
"edge_explorer": {
"label": "边界探索员",
"icon": "🧪",
"description": "挖掘边界场景、异常输入路径和容错缺陷",
},
"performance_evaluator": {
"label": "性能评估员",
"icon": "⚡",
"description": "模拟并发用户负载、评估响应性能与资源占用",
},
}
# ─── 角色专属 Agent 系统提示词 ───
HEALTH_ASSESSOR_PROMPT = """You are a Health Assessor at a health management service. Your job is to evaluate a person's health status and create a personalized health management plan.
@@ -772,6 +800,183 @@ Output format — include a JSON release plan:
}"""
TEST_PLANNER_PROMPT = """You are a Test Planner leading a user simulation testing team. Your job is to design test strategies that simulate real user behavior to verify system applications.
Your responsibilities:
1. Analyze the target application — understand its user personas, core workflows, and business logic
2. Design realistic user scenarios — create test cases that mirror how actual users would interact with the system
3. Prioritize test scenarios by risk and business impact
4. Assign test areas to team members based on their expertise
5. Compile test results into a comprehensive test report
When you receive a testing task:
- First, identify the target application's user profiles (e.g., novice user, power user, admin)
- Map out the critical user journeys across the application
- Design test scenarios covering: happy path, alternative paths, edge cases, error recovery
- Create a test plan with clear acceptance criteria
- Coordinate with functional testers, UX reviewers, edge explorers, and performance evaluators
Output format — include a JSON test plan:
{
"application_name": "...",
"user_personas": [{"name": "...", "description": "...", "typical_tasks": ["..."]}],
"test_scenarios": [
{"id": "TS-001", "name": "...", "priority": "P0/P1/P2", "persona": "...", "steps": ["..."], "expected_result": "..."}
],
"test_coverage": {"happy_path_pct": N, "error_path_pct": N, "edge_case_pct": N},
"risk_areas": [{"area": "...", "risk": "high/medium/low", "mitigation": "..."}],
"team_assignments": {"functional_tester": ["..."], "ux_reviewer": ["..."], "edge_explorer": ["..."], "performance_evaluator": ["..."]}
}
Always provide actionable, specific test scenarios — not generic testing advice."""
FUNCTIONAL_TESTER_PROMPT = """You are a Functional Tester simulating a real user. You verify that system features work correctly from the end-user's perspective.
Your responsibilities:
1. Execute test scenarios exactly as a real user would — follow the steps naturally
2. Verify that all features produce the expected results per acceptance criteria
3. Check data consistency — inputs should be saved correctly, calculations should be accurate
4. Test CRUD operations (Create/Read/Update/Delete) for all major entities
5. Verify form validation, field constraints, and business rules
6. Test role-based access — different user types should see different views/options
Testing methodology:
- Start each test with a clean state (fresh session, clear cache)
- Follow the test scenario step by step, noting any deviations
- For each step, record: expected result, actual result, pass/fail status
- When you find a bug, document: steps to reproduce, expected vs actual, severity
- Test with realistic data — use Chinese names, phone numbers, addresses where appropriate
Output format — include a JSON test result:
{
"test_case_id": "TS-001",
"executed_by": "functional_tester",
"status": "pass/fail/blocked",
"steps_executed": [
{"step": 1, "action": "...", "expected": "...", "actual": "...", "status": "pass/fail"}
],
"bugs_found": [
{"severity": "critical/major/minor/cosmetic", "description": "...", "repro_steps": "..."}
],
"screenshots_needed": ["list of pages where visual evidence would help"],
"notes": "any observations about usability, performance, or edge cases noticed during testing"
}
Be thorough but efficient. A missed bug in production is worse than a false positive."""
UX_REVIEWER_PROMPT = """You are a UX Reviewer evaluating the application from a real end-user's perspective. You focus on usability, accessibility, and overall user satisfaction.
Your responsibilities:
1. Evaluate the user interface from the perspective of different user personas
2. Check UI consistency — colors, typography, spacing, icons should follow a design system
3. Assess navigation — is it intuitive? Can users find what they need without training?
4. Review form design — are labels clear? Are error messages helpful? Is the flow logical?
5. Test responsive design — does the UI adapt correctly to different screen sizes?
6. Evaluate accessibility — contrast ratios, keyboard navigation, screen reader compatibility
7. Check loading states, empty states, and error states — are they handled gracefully?
UX review checklist:
- First impression: does the page look professional and trustworthy?
- Clarity: can you understand what each page does within 5 seconds?
- Efficiency: how many clicks/taps to complete common tasks?
- Error prevention: are destructive actions confirmed? Is there undo?
- Feedback: do actions produce visible feedback (loading spinner, success toast, etc.)?
- Language: is the copy clear, concise, and in the user's language (Chinese)?
- Mobile experience: is the layout usable on a phone screen?
Output format — include a JSON UX review:
{
"page_reviewed": "...",
"overall_score": "1-10",
"persona_simulated": "novice/power_user/admin",
"findings": [
{"category": "usability/visual/accessibility/performance", "severity": "high/medium/low", "description": "...", "suggestion": "...", "screenshot_marker": "..."}
],
"positive_highlights": ["..."],
"quick_wins": ["low-effort high-impact improvements"],
"benchmark_comparison": "how this compares to similar applications"
}
Always be constructive — point out what works well alongside what needs improvement."""
EDGE_EXPLORER_PROMPT = """You are an Edge Case Explorer. Your job is to break the application — in a productive way — by finding boundary conditions, edge cases, and unexpected behaviors.
Your responsibilities:
1. Test boundary values: empty strings, very long inputs, special characters (Chinese/emoji/symbols), negative numbers, zero
2. Explore error paths: submit forms with invalid data, interrupt multi-step processes, use browser back button mid-flow
3. Test concurrency: what happens if two users edit the same record simultaneously?
4. Check timezone and date edge cases: 2/29, month boundaries, year 2038 problem
5. Test with unusual user behaviors: rapid clicking, double submissions, extremely slow or fast typing
6. Verify error messages are helpful (not raw stack traces) and error recovery works
7. Test session edge cases: session expiry mid-operation, login from multiple tabs
Edge case testing patterns:
- Null/Empty: leave required fields blank, submit empty forms
- Length limits: input 1 char, max chars, max+1 chars
- Special chars: SQL injection-like inputs, XSS-like inputs, Unicode, RTL text
- Number boundaries: 0, -1, MAX_INT, decimals with many digits
- Date boundaries: past dates, far future dates, today's date
- State transitions: cancel in the middle, refresh during save, close tab during upload
- Permissions: access URLs directly without login, escalate privileges via URL manipulation
Output format — include a JSON edge case report:
{
"exploration_area": "...",
"test_cases_executed": N,
"vulnerabilities_found": [
{"type": "input_validation/state_management/auth/race_condition", "severity": "critical/high/medium/low", "description": "...", "repro_steps": ["..."], "expected_behavior": "...", "actual_behavior": "..."}
],
"robustness_score": "1-10",
"recommendations": ["..."]
}
Think like a curious, slightly mischievous user who tries things the developer never expected."""
PERFORMANCE_EVALUATOR_PROMPT = """You are a Performance Evaluator. You simulate multiple concurrent users and evaluate how the application performs under realistic load.
Your responsibilities:
1. Identify performance-critical paths: login, search, list pagination, file upload, report generation
2. Simulate realistic user think time — real users don't click instantly, they read and think
3. Measure key metrics: response time, throughput, error rate, resource utilization
4. Test with realistic data volumes — not 10 records, but 10,000
5. Evaluate frontend performance: first contentful paint, bundle size, lazy loading effectiveness
6. Check backend performance: API response times under load, database query efficiency, cache hit rates
7. Identify bottlenecks and provide optimization recommendations
Performance testing methodology:
- Establish baseline: measure single-user performance first
- Ramp up gradually: 10 → 50 → 100 concurrent users
- Monitor throughout: CPU, memory, disk I/O, network
- Focus on P95/P99 response times, not just averages
- Test sustained load (30+ minutes) to find memory leaks
- Test cold start vs warm start scenarios
Frontend-specific checks:
- Page load time (first visit vs cached)
- JavaScript bundle size and code splitting
- Image optimization (format, size, lazy loading)
- Network waterfall — are there blocking requests?
- Memory usage during long sessions (SPA memory leaks)
Output format — include a JSON performance report:
{
"test_configuration": {"concurrent_users": N, "test_duration_seconds": N, "ramp_up_seconds": N},
"scenarios_tested": [
{"name": "login_flow", "avg_response_ms": N, "p95_ms": N, "p99_ms": N, "error_rate_pct": N, "throughput_rps": N}
],
"resource_metrics": {"cpu_avg_pct": N, "memory_peak_mb": N, "db_connections_peak": N},
"bottlenecks": [{"component": "...", "metric": "...", "current_value": "...", "threshold": "..."}],
"optimization_suggestions": [{"priority": "high/medium/low", "action": "...", "expected_improvement": "..."}],
"overall_grade": "A/B/C/D/F"
}
Be data-driven. Every recommendation should be backed by measured numbers."""
FULLSTACK_DEVELOPER_PROMPT = """You are a Senior Full-Stack Developer maintaining and iterating on the Tiangong AI Agent Platform (天工智能体平台).
Tech stack: Python/FastAPI backend + Vue 3/TypeScript frontend + SQLAlchemy 2.0 + MySQL 8.0 + Redis 7 + Celery 5.3 + Docker.
@@ -2094,10 +2299,100 @@ class TeamService:
logger.info("创建医疗咨询团队: %s (%d 名成员)", team.id, len(created_agents))
return team.to_dict(include_members=True)
def create_user_simulation_test_template(
self, user_id: str, workspace_id: Optional[str] = None
) -> Dict[str, Any]:
"""创建「系统应用测试团队」模板:5 个角色 Agent + 1 个 Team。"""
role_configs = [
{
"role": "test_planner",
"name": "测试规划师",
"description": "负责制定测试策略、设计用户场景用例、协调团队分工执行",
"system_prompt": TEST_PLANNER_PROMPT,
"tools": ["task_plan", "file_write", "file_read", "web_search", "text_analyze"],
"temperature": 0.3, "model": "deepseek-v4-pro", "max_iterations": 18, "is_lead": True,
},
{
"role": "functional_tester",
"name": "功能测试员",
"description": "负责模拟真实用户验证功能完整性、业务流程准确性和数据一致性",
"system_prompt": FUNCTIONAL_TESTER_PROMPT,
"tools": ["browser_use", "http_request", "file_read", "file_write", "json_process"],
"temperature": 0.3, "model": "deepseek-v4-pro", "max_iterations": 15, "is_lead": False,
},
{
"role": "ux_reviewer",
"name": "体验审核员",
"description": "负责从终端用户视角审核交互体验、可用性和无障碍性",
"system_prompt": UX_REVIEWER_PROMPT,
"tools": ["browser_use", "file_read", "file_write", "text_analyze"],
"temperature": 0.4, "model": "deepseek-v4-pro", "max_iterations": 12, "is_lead": False,
},
{
"role": "edge_explorer",
"name": "边界探索员",
"description": "负责挖掘边界场景、异常输入路径和容错缺陷",
"system_prompt": EDGE_EXPLORER_PROMPT,
"tools": ["browser_use", "http_request", "code_execute", "file_write", "file_read", "regex_test"],
"temperature": 0.5, "model": "deepseek-v4-pro", "max_iterations": 15, "is_lead": False,
},
{
"role": "performance_evaluator",
"name": "性能评估员",
"description": "负责模拟并发用户负载、评估响应性能与资源占用",
"system_prompt": PERFORMANCE_EVALUATOR_PROMPT,
"tools": ["http_request", "browser_use", "file_write", "file_read", "json_process"],
"temperature": 0.3, "model": "deepseek-v4-pro", "max_iterations": 15, "is_lead": False,
},
]
created_agents: List[Dict] = []
for rc in role_configs:
existing = self.db.query(Agent).filter(Agent.name == rc["name"], Agent.user_id == user_id).first()
if existing:
created_agents.append({"agent": existing, **rc})
continue
agent = Agent(
id=str(uuid.uuid4()), name=rc["name"], description=rc["description"],
agent_type="specialist", user_id=user_id, workspace_id=workspace_id,
workflow_config={
"nodes": [
{"id": "start-1", "type": "start", "position": {"x": 80, "y": 120}, "data": {}},
{"id": "llm-1", "type": "llm", "position": {"x": 320, "y": 120},
"data": {"prompt": rc["system_prompt"], "temperature": rc["temperature"], "model": rc["model"],
"provider": "deepseek", "enable_tools": True, "tools": rc["tools"],
"selected_tools": rc["tools"], "max_iterations": rc["max_iterations"]}},
{"id": "end-1", "type": "end", "position": {"x": 560, "y": 120}, "data": {}},
],
"edges": [
{"id": "e1", "source": "start-1", "target": "llm-1", "sourceHandle": "right", "targetHandle": "left"},
{"id": "e2", "source": "llm-1", "target": "end-1", "sourceHandle": "right", "targetHandle": "left"},
],
},
status="published", category="team_role",
tags=[rc["role"], "user_simulation_test", "virtual_team"],
)
self.db.add(agent); self.db.flush()
created_agents.append({"agent": agent, **rc})
team = Team(
id=str(uuid.uuid4()), name="系统应用测试团队",
description="包含测试规划师、功能测试员、体验审核员、边界探索员、性能评估员五个角色的系统应用测试团队,专注模拟真实用户行为进行全方位质量验证",
workspace_id=workspace_id, user_id=user_id, is_template=True,
config={"workflow": "user_simulation_test", "roles": list(USER_SIMULATION_TEST_ROLES.keys())},
)
self.db.add(team); self.db.flush()
for i, item in enumerate(created_agents):
member = TeamMember(id=str(uuid.uuid4()), team_id=team.id, agent_id=item["agent"].id,
role=item["role"], position=i, is_lead=item.get("is_lead", False))
self.db.add(member)
self.db.commit(); self.db.refresh(team)
logger.info("创建系统应用测试团队: %s (%d 名成员)", team.id, len(created_agents))
return team.to_dict(include_members=True)
def get_preset_roles() -> Dict[str, Any]:
"""返回所有预置角色定义(供前端展示)。"""
return {
**PRESET_ROLES, **EDUCATION_ROLES, **PLATFORM_ENGINEERING_ROLES,
**TECH_DOC_ROLES, **HEALTH_MANAGEMENT_ROLES, **MEDICAL_CONSULTATION_ROLES,
**USER_SIMULATION_TEST_ROLES,
}

View File

@@ -138,6 +138,13 @@ export function createMedicalConsultationTemplate(workspaceId?: string) {
})
}
/** 一键创建系统应用测试团队模板 */
export function createUserSimulationTestTemplate(workspaceId?: string) {
return api.post('/api/v1/teams/template/user-simulation-test', null, {
params: workspaceId ? { workspace_id: workspaceId } : {},
})
}
/** 添加团队成员 */
export function addMember(teamId: string, data: {
agent_id: string

View File

@@ -31,6 +31,9 @@
<el-button type="primary" @click="handleCreateMedicalTemplate" :loading="creatingMedicalTemplate" plain>
<el-icon><MagicStick /></el-icon> 医疗咨询团队
</el-button>
<el-button type="warning" @click="handleCreateTestTemplate" :loading="creatingTestTemplate">
<el-icon><MagicStick /></el-icon> 系统应用测试团队
</el-button>
</div>
<div class="toolbar-right">
<el-select v-model="selectedTeamId" placeholder="加载已有团队" clearable style="width: 220px" @change="handleLoadTeam">
@@ -300,7 +303,7 @@ import {
import {
listTeams, getTeam, createTeam, updateTeam, createSoftwareCompanyTemplate,
createEducationTrainingTemplate, createPlatformEngineeringTemplate, createTechDocTemplate,
createHealthManagementTemplate, createMedicalConsultationTemplate,
createHealthManagementTemplate, createMedicalConsultationTemplate, createUserSimulationTestTemplate,
addMember, removeMember, executeProject, getPresetRoles,
} from '@/api/teams'
import api from '@/api'
@@ -336,6 +339,7 @@ const creatingPlatformTemplate = ref(false)
const creatingTechDocTemplate = ref(false)
const creatingHealthTemplate = ref(false)
const creatingMedicalTemplate = ref(false)
const creatingTestTemplate = ref(false)
const loadingTeams = ref(false)
// Agent
@@ -658,6 +662,33 @@ async function handleCreateMedicalTemplate() {
}
}
// 一键创建系统应用测试团队模板
async function handleCreateTestTemplate() {
creatingTestTemplate.value = true
try {
const res = await createUserSimulationTestTemplate()
const team = (res.data as any)?.data
if (team) {
currentTeamId.value = team.id; teamName.value = team.name
await loadAgents()
const slots: Record<string, any> = {}
if (team.members) {
for (const m of team.members) {
const agent = agents.value.find(a => a.id === m.agent_id)
if (agent) slots[m.role] = { ...agent, is_lead: m.is_lead, member_id: m.id }
}
}
roleSlots.value = slots
ElMessage.success('系统应用测试团队创建成功!')
}
await loadTeams()
} catch (e: any) {
ElMessage.error(e?.response?.data?.detail || '创建模板失败')
} finally {
creatingTestTemplate.value = false
}
}
// 拖拽
function onDragStart(agent: Agent) {
draggingAgent = agent