feat: add configurable Bailian AI cell call runtime

This commit is contained in:
2026-09-14 23:10:39 +08:00
parent 9bade94ada
commit 5dd69b8779
36 changed files with 6455 additions and 30 deletions
+424
View File
@@ -0,0 +1,424 @@
#!/usr/bin/env python3
"""Run isolated and, when configured, real Bailian AI acceptance probes."""
from __future__ import annotations
import hashlib
import json
import re
import subprocess
import sys
import tempfile
import wave
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[1]
CLI = ROOT / "scripts" / "test_ai_call.py"
MOCK_PROFILE = ROOT / "configs" / "ai-test.example.yaml"
REAL_PROFILE = ROOT / "configs" / "ai-test.bailian.example.yaml"
def _now() -> datetime:
return datetime.now(timezone.utc)
def _run_cli(kind: str, *extra: str) -> tuple[int, dict[str, Any]]:
with tempfile.TemporaryDirectory(prefix="agent-call-accept-") as directory:
output = Path(directory) / "report.json"
completed = subprocess.run(
[sys.executable, str(CLI), kind, "--output", str(output), *extra],
cwd=ROOT,
capture_output=True,
text=True,
check=False,
)
try:
report = json.loads(completed.stdout)
except json.JSONDecodeError as exc:
raise RuntimeError(
f"AI CLI emitted no JSON for {kind}: {completed.stderr[-500:]}"
) from exc
return completed.returncode, report
def _summary(kind: str, code: int, report: dict[str, Any]) -> dict[str, Any]:
result = report.get("result")
result = result if isinstance(result, dict) else {}
execution = report.get("execution")
execution = execution if isinstance(execution, dict) else {}
latest_execution = report.get("latest_execution")
latest_execution = latest_execution if isinstance(latest_execution, dict) else {}
recording = report.get("recording")
recording = recording if isinstance(recording, dict) else {}
return {
"kind": kind,
"exit_code": code,
"status": report.get("status"),
"mode": report.get("mode"),
"requested_mode": report.get("requested_mode"),
"provider_modes": report.get("provider_modes"),
"agent_version_id": report.get("agent_version_id"),
"agent_config_sha256": report.get("agent_config_sha256"),
"prompt_sha256": report.get("prompt_sha256"),
"llm_requested_model": report.get("llm", {}).get("requested_model"),
"llm_returned_model": report.get("llm", {}).get("provider_returned_model"),
"tts_requested_model": report.get("tts", {}).get("requested_model"),
"tts_returned_model": report.get("tts", {}).get("provider_returned_model"),
"tts_returned_model_verified": report.get("tts", {}).get(
"provider_returned_model_verified", False
),
"tts_model_evidence": report.get("tts", {}).get("model_evidence"),
"voice_sha256": hashlib.sha256(
str(report.get("tts", {}).get("voice", "")).encode()
).hexdigest(),
"asr_provider_ref": report.get("asr", {}).get("provider_ref"),
"asr_segment_count": len(result.get("asr_segments", [])),
"llm_first_token_ms": result.get("llm_first_token_ms"),
"tts_first_audio_ms": result.get("tts_first_audio_ms"),
"audio_sha256": report.get("audio_output", {}).get("sha256"),
"audio_bytes": report.get("audio_output", {}).get("bytes"),
"call_id": report.get("call_id")
or result.get("call_id")
or latest_execution.get("call_id")
or execution.get("call_id"),
"reason_code": report.get("reason_code")
or result.get("reason_code")
or latest_execution.get("reason_code")
or execution.get("reason_code"),
"recording_bytes": recording.get("bytes"),
"recording_valid": recording.get("valid_wav") or recording.get("valid_audio"),
}
def _wav(path: Path) -> None:
with wave.open(str(path), "wb") as output:
output.setnchannels(1)
output.setsampwidth(2)
output.setframerate(16000)
output.writeframes(b"\x00\x00" * 1600)
def _case(
case_id: str, status: str, actual: str, evidence: list[str], mode: str
) -> dict[str, Any]:
return {
"case_id": case_id,
"mode": mode,
"status": status,
"actual": actual,
"evidence": evidence,
}
def _configured() -> bool:
import os
return all(
os.environ.get(name)
for name in (
"BAILIAN_API_KEY",
"BAILIAN_BASE_URL",
"BAILIAN_WSS_BASE_URL",
"BAILIAN_TTS_VOICE",
)
)
def main() -> int:
date = _now().strftime("%Y%m%d")
evidence_dir = ROOT / "docs" / "evidence"
evidence_dir.mkdir(parents=True, exist_ok=True)
evidence_path = evidence_dir / f"llm-voice-acceptance-{date}.json"
regression = subprocess.run(
[sys.executable, "-m", "unittest", "discover", "-s", "tests", "-v"],
cwd=ROOT,
capture_output=True,
text=True,
check=False,
)
match = re.search(r"Ran (\d+) tests?", regression.stdout + regression.stderr)
test_count: int | None = None
if match:
try:
test_count = int(match.group(1))
except (TypeError, ValueError, OverflowError):
test_count = None
regression_summary = {"exit_code": regression.returncode, "test_count": test_count}
commands: list[dict[str, Any]] = []
mock_text_code, mock_text = _run_cli(
"text",
"--profile",
str(MOCK_PROFILE),
"--mode",
"mock",
"--input",
"这是隔离文本验收。",
)
commands.append(_summary("mock-text", mock_text_code, mock_text))
with tempfile.TemporaryDirectory(prefix="agent-call-accept-audio-") as directory:
source = Path(directory) / "input.wav"
target = Path(directory) / "response.wav"
_wav(source)
mock_audio_code, mock_audio = _run_cli(
"audio",
"--profile",
str(MOCK_PROFILE),
"--mode",
"mock",
"--input",
str(source),
"--output-audio",
str(target),
)
commands.append(_summary("mock-audio", mock_audio_code, mock_audio))
mock_call_code, mock_call = _run_cli(
"call",
"--profile",
str(MOCK_PROFILE),
"--mode",
"mock",
"--callee",
"18601013734",
"--allow-real-call",
"--max-duration",
"120",
)
commands.append(_summary("mock-call", mock_call_code, mock_call))
real_ready = _configured()
real_text_code = real_audio_code = 2
real_text: dict[str, Any] = {
"status": "BLOCKED",
"reason_code": "BAILIAN_ENV_REQUIRED",
"message": "BAILIAN_API_KEY, BAILIAN_BASE_URL, BAILIAN_WSS_BASE_URL and BAILIAN_TTS_VOICE are required",
}
real_audio: dict[str, Any] = dict(real_text)
if real_ready:
with tempfile.TemporaryDirectory(prefix="agent-call-real-accept-") as directory:
generated = Path(directory) / "tts.wav"
real_text_code, real_text = _run_cli(
"text",
"--profile",
str(REAL_PROFILE),
"--mode",
"real",
"--input",
"请只回答收到。",
"--output-audio",
str(generated),
)
if generated.is_file():
real_audio_code, real_audio = _run_cli(
"audio",
"--profile",
str(REAL_PROFILE),
"--mode",
"real",
"--input",
str(generated),
)
else:
real_audio = {
"status": "BLOCKED",
"reason_code": "REAL_TTS_AUDIO_MISSING",
"message": "real text probe did not produce an audio fixture",
}
commands.append(_summary("real-text", real_text_code, real_text))
commands.append(_summary("real-audio", real_audio_code, real_audio))
real_call_evidence_path = evidence_dir / f"real-cell-call-{date}.json"
if real_call_evidence_path.is_file():
try:
loaded_call = json.loads(
real_call_evidence_path.read_text(encoding="utf-8")
)
except (OSError, UnicodeDecodeError, json.JSONDecodeError):
loaded_call = None
if isinstance(loaded_call, dict):
real_call = loaded_call
real_call_code = 0 if real_call.get("status") == "PASS" else 1
else:
real_call_code = 2
real_call = {
"status": "BLOCKED",
"reason_code": "REAL_CALL_EVIDENCE_INVALID",
"message": "real Cell evidence exists but is not a valid JSON object",
}
else:
real_call_code = 2
real_call = {
"status": "BLOCKED",
"reason_code": "REAL_CALL_REQUIRES_SEPARATE_AUTHORIZATION",
"message": "Acceptance automation never dials; use the separately authorized Cell runner.",
}
commands.append(_summary("real-call-evidence", real_call_code, real_call))
mock_ok = (
regression.returncode == 0
and mock_text_code == 0
and mock_audio_code == 0
and mock_call_code == 0
)
provider_ok = real_text_code == 0 and real_audio_code == 0
cases: list[dict[str, Any]] = []
internal_pass = {
"AI-01": "tenant-scoped immutable snapshot was published and reused",
"AI-02": "missing, cross-tenant, and same-version conflict checks are covered",
"AI-03": "Prompt, model/voice, secret, and audio-format validation is covered",
"AI-04": "version digest is persisted with the execution snapshot",
"AI-09": "opening and Prompt constraints are configuration-driven in the runtime tests",
"AI-10": "PCM16/PCMA conversion, frame alignment, and bounded queues passed",
"AI-11": "cancellation and late-chunk discard passed with isolated providers",
"AI-12": "silence/no-answer lifecycle remains bounded in the mock executor",
"AI-13": "hangup/timeout cleanup remains isolated in the mock executor",
"AI-14": "provider errors and real-mode fail-closed behavior are covered",
"AI-15": "ARI reconciliation regression remains green for the existing mock path",
"AI-16": "duplicate publication/execution recovery remains green",
"AI-17": "text, playback, and MQ result facts remain separate",
"AI-18": "recording checksum/upload recovery regression remains green",
"AI-19": "real Cell ARI/RTP worker, queue routing, durable claim ledger, and no-redial tests are covered",
"AI-21": "reports use hashes and do not include provider credentials or URLs",
}
for case_id, actual in internal_pass.items():
cases.append(
_case(
case_id,
"PASS" if mock_ok else "FAIL",
actual,
[
"tests/test_ai_runtime.py",
"tests/test_agent_call.py",
"tests/test_real_cell.py",
],
"isolated-mock" if case_id != "AI-19" else "real-cell-code",
)
)
for case_id, label in (
("AI-05", "real text stream"),
("AI-06", "real TTS"),
("AI-07", "real audio chain"),
):
report = real_text if case_id in {"AI-05", "AI-06"} else real_audio
code = real_text_code if case_id in {"AI-05", "AI-06"} else real_audio_code
status = (
"PASS"
if code == 0 and report.get("status") == "PASS"
else ("FAIL" if real_ready else "BLOCKED")
)
if case_id in {"AI-06", "AI-07"} and status == "PASS":
status = "BLOCKED"
label += "; machine audio passed; human listening is not registered"
cases.append(
_case(
case_id,
status,
f"{label} {'passed' if code == 0 else 'was not completed'}",
["docs/evidence/llm-voice-acceptance-" + date + ".json"],
"real-bailian" if real_ready else "real-not-configured",
)
)
cases.append(
_case(
"AI-08",
"FAIL" if real_call_code == 1 else "BLOCKED",
"real three-round phone evidence is incomplete: SIP may connect, but bidirectional RTP, three AI rounds, recording handoff, and human listening are required",
[
str(Path("docs") / "evidence" / real_call_evidence_path.name)
if real_call_evidence_path.is_file()
else "docs/LLM与音色可配置电话对话_开发与验收计划_v1.0.md"
],
"real-phone-evidence"
if real_call_evidence_path.is_file()
else "real-phone-not-run",
)
)
cases.append(
_case(
"AI-20",
"BLOCKED",
"one-command mock flow passed; full real call flow remains incomplete until real MQ/OSS and phone evidence pass",
[
"scripts/test_ai_call.py",
"docs/evidence/llm-voice-acceptance-" + date + ".json",
],
"mixed",
)
)
blockers = [
"The single authorized real call reached SIP 200 OK, but usable bidirectional RTP, three AI rounds, and a valid recording were not proven.",
"The call path used a temporary RabbitMQ test broker; production RabbitMQ/SaaS result and OSS recording handoff remain unverified.",
"Human listening and provider-specific latency/quality thresholds are not registered; machine-generated WAV evidence cannot replace listening.",
]
if not real_ready:
blockers.insert(
0, "Bailian environment is incomplete; real provider probes were not run."
)
elif not provider_ok:
blockers.insert(
0,
"At least one real Bailian text/audio probe failed; inspect its redacted command summary.",
)
report = {
"schema_version": "1.0",
"generated_at": _now().isoformat().replace("+00:00", "Z"),
"scope": "real-bailian-file-probes-plus-isolated-mock-plus-real-cell-attempts",
"overall_status": "INCOMPLETE",
"gate_status": "BLOCKED_BY_REAL_PHONE_MEDIA_MQ_OSS_AND_HUMAN_EVIDENCE",
"component_modes": {
"llm": "real-bailian" if real_ready else "blocked",
"tts": "real-bailian" if real_ready else "blocked",
"asr": "real-bailian" if real_ready else "blocked",
"sip_ari_rtp": "partial-real"
if real_call_evidence_path.is_file()
else "not-run",
"database": "isolated-sqlite",
"rabbitmq": "temporary-test-broker"
if real_call_evidence_path.is_file()
else "not-run-real-broker",
"oss": "not-run",
},
"real_provider_preflight": {
"environment_configured": real_ready,
"text_code": real_text_code,
"audio_code": real_audio_code,
},
"regression_tests": regression_summary,
"commands": commands,
"cases": cases,
"blockers": blockers,
"evidence_policy": "docs/evidence contains hashes/status only; raw protocol/audio evidence stays in ignored local paths",
"test_commands": [
"python3 -m unittest discover -s tests -v",
"python3 scripts/test_ai_call.py text --profile configs/ai-test.bailian.example.yaml --mode real --input '请只回答收到。'",
"python3 scripts/test_ai_call.py audio --profile configs/ai-test.bailian.example.yaml --mode real --input <real-or-authorized-wav>",
"python3 scripts/run_real_cell_call.py --profile configs/ai-test.bailian.example.yaml --allow-real-call --callee 15003164745",
],
}
evidence_path.write_text(
json.dumps(report, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
print(
json.dumps(
{
"overall_status": report["overall_status"],
"evidence": str(evidence_path),
"cases": len(cases),
},
ensure_ascii=False,
)
)
return (
1
if not mock_ok or (real_ready and not provider_ok) or real_call_code == 1
else 2
)
if __name__ == "__main__":
raise SystemExit(main())
+251
View File
@@ -0,0 +1,251 @@
#!/usr/bin/env python3
"""Run one explicitly authorized Cell call through the tenant RabbitMQ queue."""
from __future__ import annotations
import argparse
import hashlib
import json
import os
import sys
import time
import wave
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
def _data(path: Path) -> dict[str, Any]:
try:
value = json.loads(path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, json.JSONDecodeError) as exc:
raise ValueError(f"profile is unavailable: {path}") from exc
if not isinstance(value, dict):
raise TypeError("profile must be an object")
return value
def _path(value: str) -> Path:
candidate = Path(value)
return candidate if candidate.is_absolute() else ROOT / candidate
def _integer(value: Any, default: int, field: str, minimum: int = 1) -> int:
try:
parsed = int(value)
except (TypeError, ValueError, OverflowError) as exc:
raise ValueError(f"{field} must be an integer") from exc
if parsed < minimum:
raise ValueError(f"{field} must be at least {minimum}")
return parsed
def _args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--profile", default="configs/ai-test.bailian.example.yaml")
parser.add_argument("--prompt-file", default="prompts/test-call.txt")
parser.add_argument("--callee", required=True)
parser.add_argument("--tenant-id", default=None)
parser.add_argument("--tenant-key", default=None)
parser.add_argument("--broker-url", default=None)
parser.add_argument("--ledger", default="/data/agent-call-cell.sqlite3")
parser.add_argument("--wait-seconds", type=int, default=240)
parser.add_argument("--allow-real-call", action="store_true")
parser.add_argument("--output", default=None)
return parser.parse_args()
def _command(
settings: dict[str, Any],
args: argparse.Namespace,
tenant_id: str,
tenant_key: str,
agent_version_id: str,
) -> dict[str, Any]:
now_timestamp = time.time()
issued_at = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(now_timestamp))
not_after = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(now_timestamp + 300))
execution_id = f"exec_cell_{time.time_ns()}_{os.getpid()}"
return {
"schema_version": "1.0",
"command_type": "call.execute",
"command_id": f"cmd_cell_{execution_id}",
"tenant_id": tenant_id,
"tenant_key": tenant_key,
"trace_id": f"trace_cell_{execution_id}",
"issued_at": issued_at,
"not_after": not_after,
"payload": {
"execution_id": execution_id,
"task_id": str(settings.get("task_id", "task-ai-cell")),
"task_item_id": f"item_{execution_id}",
"task_revision": 1,
"callee": args.callee,
"route_policy_id": "route_policy_test",
"caller_profile_id": "caller_profile_test",
"agent_version_id": agent_version_id,
"variables": {},
"ring_timeout_ms": 30000,
"max_call_duration_ms": _integer(
settings.get("max_duration_ms", 180000),
180000,
"max_duration_ms",
),
},
}
def _recording(path: str) -> dict[str, Any] | None:
recording = Path(path)
if not recording.is_file() or recording.is_symlink():
return None
try:
with wave.open(str(recording), "rb") as source:
evidence = {
"bytes": recording.stat().st_size,
"sha256": hashlib.sha256(recording.read_bytes()).hexdigest(),
"channels": source.getnchannels(),
"sample_rate_hz": source.getframerate(),
"frames": source.getnframes(),
"valid_wav": source.getnframes() > 0 and source.getnchannels() == 1,
}
except (OSError, EOFError, wave.Error):
return None
return evidence
def _write(report: dict[str, Any], output: str | None) -> None:
payload = json.dumps(report, ensure_ascii=False, indent=2) + "\n"
if output:
destination = _path(output)
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(payload, encoding="utf-8")
print(payload, end="")
def run(args: argparse.Namespace) -> dict[str, Any]:
if not args.allow_real_call:
raise ValueError("--allow-real-call is required")
from agent_call.ai_runtime import ConversationEngine, load_prompt
from agent_call.bailian import (
BailianASR,
BailianLLM,
BailianTTS,
build_bailian_config,
)
from agent_call.core import PikaBroker
from agent_call.real_cell import (
CellCallConfig,
CellExecutionLedger,
RealCellCall,
RealCellWorker,
)
settings = _data(_path(args.profile))
authorized = {str(value) for value in settings.get("authorized_callees", [])}
if args.callee not in authorized:
raise ValueError("callee is not in the profile allow-list")
tenant_id = str(args.tenant_id or settings.get("tenant_id", "tenant-demo"))
tenant_key = str(args.tenant_key or settings.get("tenant_key", "tenant-demo-key"))
agent_version_id = str(settings.get("agent_version_id", "agent_bailian_test_v1"))
try:
prompt = load_prompt(
_path(args.prompt_file),
_integer(
settings.get("max_prompt_bytes", 32768), 32768, "max_prompt_bytes"
),
)
except (OSError, UnicodeDecodeError, ValueError) as exc:
raise ValueError("prompt is unavailable") from exc
config = build_bailian_config(
agent_version_id,
prompt,
str(settings.get("model", "qwen-plus")),
settings.get("tts_model"),
settings.get("voice"),
str(settings.get("language", "zh-CN")),
settings.get("asr_model"),
str(settings.get("opening", "")),
min(3, _integer(settings.get("max_turns", 3), 3, "max_turns")),
_integer(settings.get("max_duration_ms", 180000), 180000, "max_duration_ms"),
)
engine = ConversationEngine(
config,
llm=BailianLLM.from_env(),
tts=BailianTTS.from_env(),
asr=BailianASR.from_env(),
)
broker = PikaBroker(args.broker_url or os.environ.get("RABBITMQ_URL", ""))
ledger = CellExecutionLedger(args.ledger)
executor = RealCellCall(CellCallConfig.from_env(), engine, BailianASR.from_env())
worker = RealCellWorker(broker, tenant_key, ledger, executor)
command = _command(settings, args, tenant_id, tenant_key, agent_version_id)
route = f"agent-call.tenant.{tenant_key}.call.execute"
broker.publish(
"agent-call.commands.v1", route, command, message_id=command["command_id"]
)
deadline = time.monotonic() + args.wait_seconds
event: dict[str, Any] | None = None
while time.monotonic() < deadline:
event = worker.process_once()
if event and event.get("event_type") == "call.finished":
break
time.sleep(0.2)
if event is None or event.get("event_type") != "call.finished":
raise TimeoutError("Cell call did not finish before the bounded wait")
result = ledger.result(command["payload"]["execution_id"]) or {}
result_status = event["payload"].get("status")
recording_path = result.get("recording_path")
recording = _recording(recording_path) if isinstance(recording_path, str) else None
passed = (
result_status == "completed"
and len(event["payload"].get("turns", [])) >= 3
and bool(recording and recording.get("valid_wav"))
)
report: dict[str, Any] = {
"schema_version": "1.0",
"status": "PASS" if passed else "FAIL",
"mode": "real",
"call_execution": {
"execution_id": command["payload"]["execution_id"],
"tenant_id": tenant_id,
"tenant_key": tenant_key,
"callee": args.callee,
"event_id": event["event_id"],
},
"result": event["payload"],
}
if recording is not None:
report["recording"] = recording
return report
def main() -> int:
args = _args()
try:
report = run(args)
code = 0 if report.get("status") == "PASS" else 1
except (
OSError,
RuntimeError,
TimeoutError,
TypeError,
ValueError,
KeyError,
) as exc:
report = {
"schema_version": "1.0",
"status": "FAIL",
"reason_code": type(exc).__name__,
"message": str(exc),
}
code = 1
_write(report, args.output)
return code
if __name__ == "__main__":
raise SystemExit(main())
+452
View File
@@ -0,0 +1,452 @@
#!/usr/bin/env python3
"""One-command text, audio, or explicit MQ AI probe.
Mock is isolated by default. The Bailian adapter uses only server-side
BAILIAN_* environment variables; real call mode still fails closed until the
Cell/media authorization and real-call evidence gate is explicitly arranged.
"""
from __future__ import annotations
import argparse
import hashlib
import importlib
import json
import sys
import tempfile
import uuid
from datetime import datetime, timezone
from pathlib import Path
from typing import Any
ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
_ai_runtime = importlib.import_module("agent_call.ai_runtime")
_core = importlib.import_module("agent_call.core")
_bailian = importlib.import_module("agent_call.bailian")
AIConfigError = _ai_runtime.AIConfigError
AIProviderError = _ai_runtime.AIProviderError
ConversationEngine = _ai_runtime.ConversationEngine
build_mock_config = _ai_runtime.build_mock_config
config_digest = _ai_runtime.config_digest
load_prompt = _ai_runtime.load_prompt
pcm_to_wav = _ai_runtime.pcm_to_wav
validate_wav = _ai_runtime.validate_wav
BailianASR = _bailian.BailianASR
BailianLLM = _bailian.BailianLLM
BailianTTS = _bailian.BailianTTS
build_bailian_config = _bailian.build_bailian_config
AgentCallService = _core.AgentCallService
ServiceError = _core.ServiceError
DEFAULT_PROFILE = ROOT / "configs" / "ai-test.example.yaml"
def _load_data(path: Path) -> dict[str, Any]:
try:
value = json.loads(path.read_text(encoding="utf-8"))
except json.JSONDecodeError:
try:
import yaml # type: ignore[import-not-found]
except ImportError as exc:
raise AIConfigError(
"PROFILE_FORMAT_UNSUPPORTED",
"profile must be JSON-compatible YAML when PyYAML is unavailable",
) from exc
value = yaml.safe_load(path.read_text(encoding="utf-8"))
if not isinstance(value, dict):
raise AIConfigError("PROFILE_INVALID", "profile must be an object")
return value
def _path(value: str) -> Path:
candidate = Path(value)
return candidate if candidate.is_absolute() else ROOT / candidate
def _iso_now() -> str:
return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
def _safe_result(result: dict[str, Any]) -> dict[str, Any]:
clean = dict(result)
if "audio" in clean:
audio = clean.pop("audio")
if isinstance(audio, bytes):
clean["audio_bytes"] = len(audio)
clean["audio_sha256"] = hashlib.sha256(audio).hexdigest()
if isinstance(clean.get("segments"), list):
clean["segments"] = [
{key: value for key, value in segment.items() if key != "audio"}
for segment in clean["segments"]
if isinstance(segment, dict)
]
return clean
def _write_report(report: dict[str, Any], output: str | None) -> None:
payload = json.dumps(report, ensure_ascii=False, indent=2) + "\n"
if output:
destination = _path(output)
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(payload, encoding="utf-8")
print(payload, end="")
def _args() -> argparse.Namespace:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("kind", choices=("text", "audio", "call"))
parser.add_argument("--profile", default=str(DEFAULT_PROFILE))
parser.add_argument("--mode", choices=("mock", "mixed", "real"))
parser.add_argument("--prompt-file", default="prompts/test-call.txt")
parser.add_argument(
"--input", help="UTF-8 text for text mode or mono PCM16 WAV for audio mode"
)
parser.add_argument("--model", default=None)
parser.add_argument("--tts-model", default=None)
parser.add_argument("--voice", default=None)
parser.add_argument("--language", default=None)
parser.add_argument(
"--max-duration", type=int, default=None, help="call duration limit in seconds"
)
parser.add_argument("--tenant-id", default=None)
parser.add_argument("--task-id", default=None)
parser.add_argument("--callee", default=None)
parser.add_argument("--agent-version-id", default=None)
parser.add_argument(
"--db",
default=None,
help="SQLite path for call mode or durable config evidence",
)
parser.add_argument("--output", default=None, help="JSON report path")
parser.add_argument(
"--output-audio", default=None, help="WAV response path for text/audio mode"
)
parser.add_argument(
"--allow-real-call",
action="store_true",
help="explicitly acknowledge that call mode could place a call; mock never dials",
)
parser.add_argument(
"--human-listening",
choices=("pass", "fail", "not_run"),
default="not_run",
help="manual listening result; it is evidence only, not a provider claim",
)
return parser.parse_args()
def _config(
settings: dict[str, Any], args: argparse.Namespace, mode: str
) -> dict[str, Any]:
prompt = load_prompt(
_path(args.prompt_file), int(settings.get("max_prompt_bytes", 32768))
)
version_id = args.agent_version_id or settings.get(
"agent_version_id", "agent_test_v1"
)
if mode == "mock":
config = build_mock_config(
str(version_id),
prompt,
str(args.model or settings.get("model", "mock-chat-v1")),
str(args.tts_model or settings.get("tts_model", "mock-tts-v1")),
str(args.voice or settings.get("voice", "mock-neutral")),
)
else:
config = build_bailian_config(
str(version_id),
prompt,
str(args.model or settings.get("model", "qwen-plus")),
args.tts_model or settings.get("tts_model"),
args.voice or settings.get("voice"),
str(args.language or settings.get("language", "zh-CN")),
settings.get("asr_model"),
str(settings.get("opening", "")),
int(settings.get("max_turns", 20)),
int(settings.get("max_duration_ms", 120000)),
)
return config
def _base_report(
settings: dict[str, Any], args: argparse.Namespace, config: dict[str, Any]
) -> dict[str, Any]:
provider_modes = settings.get("provider_modes", {})
return {
"schema_version": "1.0",
"created_at": _iso_now(),
"mode": settings.get("mode", "mock"),
"requested_mode": args.mode or settings.get("mode", "mock"),
"provider_modes": provider_modes,
"agent_version_id": config["agent_version_id"],
"agent_config_sha256": config_digest(config),
"prompt_sha256": hashlib.sha256(
config["prompt"]["text"].encode("utf-8")
).hexdigest(),
"prompt_bytes": len(config["prompt"]["text"].encode("utf-8")),
"llm": {
"provider_ref": config["llm"]["provider_ref"],
"requested_model": config["llm"]["model"],
"provider_returned_model": None,
"provider_returned_model_verified": False,
},
"tts": {
"provider_ref": config["tts"]["provider_ref"],
"requested_model": config["tts"]["model"],
"provider_returned_model": None,
"provider_returned_model_verified": False,
"voice": config["tts"]["voice"],
"voice_listening_verified": False,
"format": config["tts"]["format"],
},
"asr": {
"provider_ref": config["asr"]["provider_ref"],
"language": config["asr"]["language"],
},
"human_listening": args.human_listening,
}
def _run_real_cell(
args: argparse.Namespace, settings: dict[str, Any], service_settings: dict[str, Any]
) -> tuple[int, dict[str, Any]]:
tenant_id = str(args.tenant_id or settings.get("tenant_id", "tenant-demo"))
tenant = next(
(
item
for item in service_settings.get("tenants", [])
if item.get("tenant_id") == tenant_id
),
None,
)
if not isinstance(tenant, dict) or not isinstance(tenant.get("tenant_key"), str):
raise AIConfigError(
"TENANT_NOT_CONFIGURED", "real Cell call tenant is not configured"
)
runner = importlib.import_module("scripts.run_real_cell_call")
real_args = argparse.Namespace(
profile=args.profile,
prompt_file=args.prompt_file,
callee=args.callee or settings.get("callee", ""),
tenant_id=tenant_id,
tenant_key=tenant["tenant_key"],
broker_url=None,
ledger=args.db or str(ROOT / ".local" / "agent-call-ai" / "real-cell.sqlite3"),
wait_seconds=240,
allow_real_call=args.allow_real_call,
output=None,
)
report = runner.run(real_args)
return (0 if report.get("status") == "PASS" else 1), report
def _run(args: argparse.Namespace) -> tuple[int, dict[str, Any]]:
settings = _load_data(_path(args.profile))
service_profile = _path(
str(settings.get("service_profile", "docs/contracts/mock-profile.json"))
)
service_settings = _load_data(service_profile)
requested_mode = args.mode or str(
settings.get("mode", service_settings.get("mode", "mock"))
)
if requested_mode not in {"mock", "mixed", "real"}:
raise AIConfigError("MODE_INVALID", "mode must be mock, mixed, or real")
if args.kind == "call" and requested_mode == "real":
if not args.allow_real_call:
return 2, {
"schema_version": "1.0",
"status": "BLOCKED",
"requested_mode": requested_mode,
"reason_code": "CALL_AUTHORIZATION_REQUIRED",
"message": "call mode requires --allow-real-call before the RabbitMQ Cell worker is started.",
}
return _run_real_cell(args, settings, service_settings)
if requested_mode != "mock" and args.kind == "call":
return 2, {
"schema_version": "1.0",
"status": "BLOCKED",
"requested_mode": requested_mode,
"reason_code": "REAL_CALL_MODE_REQUIRED",
"message": "Only the explicit real profile may start the Cell/media executor; no call was placed.",
}
config = _config(settings, args, requested_mode)
report_settings = dict(settings)
report_settings["mode"] = requested_mode
report_settings["provider_modes"] = settings.get(
"provider_modes", service_settings.get("provider_modes", {})
)
report = _base_report(report_settings, args, config)
db_path: str
temporary_db: tempfile.TemporaryDirectory[str] | None = None
if args.db:
db_path = str(_path(args.db))
Path(db_path).parent.mkdir(parents=True, exist_ok=True)
else:
temporary_db = tempfile.TemporaryDirectory(prefix="agent-call-ai-")
db_path = str(Path(temporary_db.name) / "state.sqlite3")
service = AgentCallService(
db_path=db_path, profile_path=service_profile, start_background=False
)
try:
tenant_id = str(args.tenant_id or settings.get("tenant_id", "tenant-demo"))
published = service.publish_agent_version(
tenant_id,
config["agent_version_id"],
config,
actor_id="ai-test-cli",
)
trusted = service.get_agent_version(tenant_id, config["agent_version_id"])
report["publication"] = published
report["trusted_snapshot"] = {
"agent_version_id": trusted["agent_version_id"],
"content_sha256": trusted["content_sha256"],
"immutable": trusted["immutable"],
}
engine_kwargs: dict[str, Any] = {"journal": service.journal}
if requested_mode != "mock":
engine_kwargs["llm"] = BailianLLM.from_env()
engine_kwargs["tts"] = BailianTTS.from_env()
if trusted["config"]["asr"]["provider_ref"] != "mock":
engine_kwargs["asr"] = BailianASR.from_env()
engine = ConversationEngine(trusted["config"], **engine_kwargs)
if args.kind == "text":
if args.input is None:
raise AIConfigError(
"TEXT_INPUT_REQUIRED", "--input is required for text mode"
)
result = engine.run_text(args.input)
elif args.kind == "audio":
if args.input is None:
raise AIConfigError(
"AUDIO_INPUT_REQUIRED", "--input is required for audio mode"
)
audio_path = _path(args.input)
audio = audio_path.read_bytes()
result = engine.run_audio(audio)
else:
if not args.allow_real_call:
report.update(
{
"status": "BLOCKED",
"reason_code": "CALL_AUTHORIZATION_REQUIRED",
"message": "call mode requires --allow-real-call even though the selected profile is Mock.",
}
)
return 2, report
callee = str(args.callee or settings.get("callee", ""))
authorized = {
str(value) for value in settings.get("authorized_callees", [callee])
}
if callee not in authorized:
raise AIConfigError(
"CALLEE_NOT_AUTHORIZED",
"callee is not in the explicit test allow-list",
)
fixture = _load_data(ROOT / "docs/contracts/examples/call.execute.json")
now = datetime.now(timezone.utc)
fixture["command_id"] = f"cmd_ai_{uuid.uuid4().hex}"
fixture["trace_id"] = f"trace_ai_{uuid.uuid4().hex}"
fixture["issued_at"] = now.isoformat().replace("+00:00", "Z")
fixture["not_after"] = (
now.replace(year=now.year + 1).isoformat().replace("+00:00", "Z")
)
fixture["tenant_id"] = tenant_id
fixture["tenant_key"] = next(
item["tenant_key"]
for item in service.profile["tenants"]
if item["tenant_id"] == tenant_id
)
fixture["payload"]["execution_id"] = f"exec_ai_{uuid.uuid4().hex}"
fixture["payload"]["task_id"] = str(
args.task_id or settings.get("task_id", "task-demo")
)
fixture["payload"]["callee"] = callee
fixture["payload"]["agent_version_id"] = config["agent_version_id"]
if args.max_duration is not None:
if args.max_duration < 1:
raise AIConfigError(
"MAX_DURATION_INVALID", "--max-duration must be positive"
)
fixture["payload"]["max_call_duration_ms"] = args.max_duration * 1000
published_command = service.publish_execute(fixture)
service.wait_for_idle(10)
call_row = service.store.one(
"SELECT call_id FROM calls WHERE execution_id=?",
(fixture["payload"]["execution_id"],),
)
if call_row is None:
raise ServiceError(
"CALL_NOT_CREATED", "mock executor did not reserve a call"
)
result = service.get_call(tenant_id, call_row["call_id"])
report["mq_publication"] = published_command
report["call_id"] = call_row["call_id"]
clean = _safe_result(result)
if result.get("provider_returned_model"):
report["llm"]["provider_returned_model"] = result["provider_returned_model"]
report["llm"]["provider_returned_model_verified"] = True
if result.get("tts_provider_returned_model"):
report["tts"]["provider_returned_model"] = result[
"tts_provider_returned_model"
]
report["tts"]["provider_returned_model_verified"] = True
if isinstance(result.get("tts_model_evidence"), dict):
report["tts"]["model_evidence"] = result["tts_model_evidence"]
if args.kind in {"text", "audio"}:
raw_audio = result.get("audio", b"")
if isinstance(raw_audio, bytes) and raw_audio:
wav = pcm_to_wav(
raw_audio, int(config["tts"]["format"]["sample_rate_hz"])
)
validate_wav(wav)
if args.output_audio:
destination = _path(args.output_audio)
else:
destination = (
ROOT
/ ".local"
/ "agent-call-ai"
/ f"response-{uuid.uuid4().hex}.wav"
)
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_bytes(wav)
report["audio_output"] = {
"path": str(destination),
"bytes": len(wav),
"sha256": hashlib.sha256(wav).hexdigest(),
}
report["result"] = clean
result_status = (
clean.get("status") or result.get("outcome") or result.get("call_state")
)
report["status"] = (
"PASS" if result_status in {"completed", "succeeded", "ended"} else "FAIL"
)
return (0 if report["status"] == "PASS" else 1), report
finally:
service.store.close()
if temporary_db is not None:
temporary_db.cleanup()
def main() -> int:
args = _args()
try:
code, report = _run(args)
except (AIConfigError, AIProviderError, ServiceError, OSError, ValueError) as exc:
code = 1
report = {
"schema_version": "1.0",
"status": "FAIL",
"reason_code": getattr(exc, "code", type(exc).__name__),
"message": str(exc),
}
_write_report(report, args.output)
return code
if __name__ == "__main__":
raise SystemExit(main())