Skip to content

ShadowEvals

ShadowEvals replays a past conversation against the current version of an agent, streaming the caller's original recorded audio turn by turn and falling back to TTS only when the new agent needs information no recording contains. See the Shadow Evals guide for the YAML format, prerequisites and workflows.

Key concepts:

  • ShadowTestCase: one case, which is a past conversation_id plus natural-language expectations, optional seeded session_parameters / use_tool_fakes, and a replay_mode (hybrid or exact).
  • ShadowEvals.preflight(): checks a case before running it (trace reachable, caller turns recorded, target app recording, seeding hints) and returns a ShadowPreflightResult.
  • ShadowEvals.run_shadow_evals(): runs many cases × runs, optionally in parallel, with preflight on by default. Pass artifacts_dir to keep audio and a side-by-side HTML report per run.
  • ShadowUserConversation: the simulated caller; its turn_logs record, for every turn, whether recorded audio or TTS was used and why.

Quick Example

from cxas_scrapi.evals import ShadowEvals

shadow = ShadowEvals(app_name="projects/my-project/locations/us/apps/my-app")
case = {
    "conversation_id": "7f3c9a52-...",
    "gcs_bucket": "gs://my-recordings-bucket",
    "session_parameters": {"customer_tier": "gold"},
    "use_tool_fakes": True,
    "expectations": ["The agent explains the delivery status of the order."],
}

print(shadow.preflight(case).issues)
conv = shadow.run_shadow_conversation(case, artifacts_dir="shadow_artifacts/")
print(conv.generate_report())
print("Side-by-side report:", conv.report_path)

Reference

ShadowEvals

ShadowEvals(app_name, rate_limiter=None, expectations_only=True, deployment_id=None, vertex_location='global', naturalness=None, gcs_bucket=None, **kwargs)

Bases: Apps

Replays historical conversations (with past GCS audio + hybrid TTS simUser) on a target CXAS Agent and evaluates natural-language expectations.

Source code in src/cxas_scrapi/evals/shadow_evals.py
def __init__(
    self,
    app_name: str,
    rate_limiter: RateLimiter | None = None,
    expectations_only: bool = True,
    deployment_id: str | None = None,
    vertex_location: str = "global",
    naturalness: bool | dict[str, Any] | None = None,
    gcs_bucket: str | None = None,
    **kwargs: typing.Any,
) -> None:
    self.expectations_only = expectations_only
    self.vertex_location = vertex_location
    self.naturalness = naturalness
    self.default_gcs_bucket = gcs_bucket
    project_id = app_name.split("/")[1]
    location = app_name.split("/")[3]
    super().__init__(project_id=project_id, location=location, **kwargs)
    self.app_name = app_name
    self.sessions_client = Sessions(
        app_name,
        deployment_id=deployment_id,
        rate_limiter=rate_limiter,
        **kwargs,
    )
    self.tools_map = Tools(app_name=app_name, **kwargs).get_tools_map()
    self.genai_client = GeminiGenerate(
        project_id=self.project_id,
        location=self.vertex_location,
        credentials=self.creds,
    )

preflight

preflight(test_case, modality='audio')

Checks that a ShadowEval can faithfully replay its past conversation before any session is opened.

Errors (the case should not run): - the past conversation trace cannot be loaded; - it has no caller turns; - in audio modality, none of the caller turns has a recording (the replay would be pure TTS).

Warnings / info: - some caller turns have no recording; - the target app does not record audio (new calls will not be in GCS; use artifacts_dir to keep local copies); - no session_parameters / use_tool_fakes are set (replays often diverge when the original session relied on seeded state).

Source code in src/cxas_scrapi/evals/shadow_evals.py
def preflight(
    self,
    test_case: ShadowTestCase | dict[str, Any],
    modality: str = "audio",
) -> ShadowPreflightResult:
    """Checks that a ShadowEval can faithfully replay its past conversation
    before any session is opened.

    Errors (the case should not run):
      - the past conversation trace cannot be loaded;
      - it has no caller turns;
      - in audio modality, none of the caller turns has a recording (the
        replay would be pure TTS).

    Warnings / info:
      - some caller turns have no recording;
      - the target app does not record audio (new calls will not be in
        GCS; use `artifacts_dir` to keep local copies);
      - no `session_parameters` / `use_tool_fakes` are set (replays often
        diverge when the original session relied on seeded state).
    """
    tc = (
        ShadowTestCase(**test_case)
        if isinstance(test_case, dict)
        else test_case
    )
    result = ShadowPreflightResult(
        name=tc.name, conversation_id=tc.conversation_id
    )
    source_app_name = self._resolve_source_app_name(tc)
    try:
        traces_client, normalized, bucket = self._open_past_conversation(tc)
    except Exception as exc:
        result.add(
            "error",
            f"Could not load past conversation '{tc.conversation_id}' "
            f"from {source_app_name}: {exc}. Check `project_id`, "
            "`location`, `app_id` and the conversation ID.",
        )
        return result

    spoken = self._spoken_user_turn_numbers(normalized)
    replayable = self._all_replayable_user_turn_numbers(normalized)
    result.past_user_turns = len(spoken) or len(replayable)
    if not replayable:
        result.add(
            "error",
            f"Past conversation '{tc.conversation_id}' has no caller turns "
            "to replay.",
        )
        return result

    if modality == "audio" and spoken:
        uris = self._list_turn_audio_uris(
            traces_client, tc.conversation_id, normalized, "user"
        )
        result.past_user_turns_with_audio = sum(
            1 for n in spoken if n in uris
        )
        bucket_label = bucket or "resolved from the source app"
        recording_hint = (
            "Set `gcs_bucket` to the bucket the original call was "
            f"recorded to (currently {bucket_label}) "
            "and make sure the source app had audio recording "
            "(loggingSettings.audioRecordingConfig) enabled at the time."
        )
        if result.past_user_turns_with_audio == 0:
            result.add(
                "error",
                "No recorded caller audio was found for any of the "
                f"{len(spoken)} caller turns, so the replay would be pure "
                f"TTS. {recording_hint}",
            )
        elif result.past_user_turns_with_audio < len(spoken):
            result.add(
                "warning",
                f"Only {result.past_user_turns_with_audio}/{len(spoken)} "
                "caller turns have recorded audio; the rest will be "
                f"synthesized via TTS. {recording_hint}",
            )
    if modality == "audio" and not self._target_recording_bucket():
        result.add(
            "info",
            f"Audio recording is not enabled on {self.app_name}, so "
            "new calls will not be recorded to GCS. Pass "
            "`artifacts_dir` to keep local copies of the replayed "
            "audio and a side-by-side HTML report.",
        )

    if not tc.session_parameters and tc.use_tool_fakes is None:
        result.add(
            "info",
            "No `session_parameters` or `use_tool_fakes` set. If the "
            "original session relied on seeded session variables or tool "
            "fakes (e.g. a mocked caller lookup), set them in the case or "
            "the `config:` block, or the replay may diverge.",
        )
    return result

fetch_past_conversation_data

fetch_past_conversation_data(test_case, session_dir=None)

Downloads past conversation transcript and per-turn user audio from GCS for a ShadowTestCase.

Parameters:

Name Type Description Default
test_case ShadowTestCase | dict[str, Any]

ShadowTestCase or dict specifying conversation_id (and optional project_id, location, app_id, gcs_bucket).

required
session_dir str | None

Optional local directory to store downloaded WAV files.

None

Returns:

Type Description
tuple[list[ShadowPastTurn], str, dict[str, Any]]

Tuple of (past_turns, full_past_history_text, normalized_trace).

Source code in src/cxas_scrapi/evals/shadow_evals.py
def fetch_past_conversation_data(
    self,
    test_case: ShadowTestCase | dict[str, Any],
    session_dir: str | None = None,
) -> tuple[list[ShadowPastTurn], str, dict[str, Any]]:
    """Downloads past conversation transcript and per-turn user audio from
    GCS for a `ShadowTestCase`.

    Args:
        test_case: `ShadowTestCase` or dict specifying `conversation_id`
            (and optional `project_id`, `location`, `app_id`, `gcs_bucket`).
        session_dir: Optional local directory to store downloaded WAV files.

    Returns:
        Tuple of `(past_turns, full_past_history_text, normalized_trace)`.
    """
    tc = (
        ShadowTestCase(**test_case)
        if isinstance(test_case, dict)
        else test_case
    )
    traces_client, normalized, bucket_override = (
        self._open_past_conversation(tc)
    )
    user_audio_uris = self._list_turn_audio_uris(
        traces_client, tc.conversation_id, normalized, "user"
    )
    agent_audio_uris = self._list_turn_audio_uris(
        traces_client, tc.conversation_id, normalized, "agent"
    )

    gcs_client = GCSUtils(creds=self.creds) if user_audio_uris else None

    # Extract turns and full history from normalized conversation
    raw_conv = normalized.get("raw") or {}
    raw_turns = raw_conv.get("turns") or []

    past_turns: list[ShadowPastTurn] = []
    history_lines: list[str] = []
    last_agent_texts: list[str] = []
    user_turn_counter = 0

    if raw_turns:
        for turn_idx, p_turn in enumerate(raw_turns, start=1):
            turn_user_texts: list[str] = []
            turn_agent_texts: list[str] = []
            # Agent text spoken after the user in this turn (the reply).
            turn_agent_reply: list[str] = []

            for msg in p_turn.get("messages", []) or []:
                role = (msg.get("role") or "").strip()
                chunks = msg.get("chunks") or []
                for chunk in chunks:
                    text_val = (
                        chunk.get("text") or chunk.get("transcript") or ""
                    ).strip()
                    if text_val:
                        if role.lower() == "user":
                            turn_user_texts.append(text_val)
                            history_lines.append(
                                f"[Turn {turn_idx}] User: {text_val}"
                            )
                        else:
                            turn_agent_texts.append(text_val)
                            if turn_user_texts:
                                turn_agent_reply.append(text_val)
                            history_lines.append(
                                f"[Turn {turn_idx}] Agent ({role}): "
                                f"{text_val}"
                            )
                    if "tool_call" in chunk:
                        tc_chunk = chunk["tool_call"]
                        t_name = (
                            tc_chunk.get("display_name")
                            or tc_chunk.get("name")
                            or tc_chunk.get("tool")
                            or ""
                        )
                        t_args = tc_chunk.get("args", {})
                        history_lines.append(
                            f"[Turn {turn_idx}] Tool Call: {t_name} "
                            f"args={t_args}"
                        )
                    if "tool_response" in chunk:
                        tr_chunk = chunk["tool_response"]
                        t_name = (
                            tr_chunk.get("display_name")
                            or tr_chunk.get("name")
                            or tr_chunk.get("tool")
                            or ""
                        )
                        t_resp = tr_chunk.get("response", {})
                        history_lines.append(
                            f"[Turn {turn_idx}] Tool Response: {t_name} "
                            f"response={t_resp}"
                        )

            cleaned_user_text = _extract_user_turn_transcript(
                turn_user_texts
            )
            if cleaned_user_text:
                user_turn_counter += 1
                is_non_spoken = _is_non_spoken_turn_text(cleaned_user_text)
                # Platform recordings are numbered per 1-based
                # conversation turn (`user-turn-N` == raw turn N), so
                # prefer `turn_idx`; fall back to the user-turn counter.
                audio_uri = (
                    None
                    if is_non_spoken
                    else (
                        user_audio_uris.get(turn_idx)
                        or user_audio_uris.get(user_turn_counter)
                    )
                )
                audio_path = None
                pcm_bytes = None

                if audio_uri and gcs_client is not None:
                    try:
                        raw_wav = gcs_client.download_blob(audio_uri)
                        pcm_bytes = extract_pcm_bytes_from_wav(raw_wav)
                        if pcm_bytes and session_dir:
                            os.makedirs(session_dir, exist_ok=True)
                            audio_path = os.path.join(
                                session_dir,
                                f"past_user_turn_{user_turn_counter}.wav",
                            )
                            with open(audio_path, "wb") as f:
                                f.write(raw_wav)
                    except Exception as exc:
                        logger.warning(
                            "Failed to download user audio %s: %s",
                            audio_uri,
                            exc,
                        )

                past_turns.append(
                    ShadowPastTurn(
                        turn_index=user_turn_counter,
                        user_transcript=cleaned_user_text,
                        preceding_agent_response=" ".join(last_agent_texts),
                        audio_uri=audio_uri,
                        audio_path=audio_path,
                        has_audio=bool(pcm_bytes),
                        used=False,
                        audio_bytes=pcm_bytes or None,
                        agent_response=" ".join(turn_agent_reply),
                        agent_audio_uri=agent_audio_uris.get(turn_idx),
                    )
                )
                last_agent_texts = []

            if turn_agent_texts:
                last_agent_texts.extend(turn_agent_texts)
    else:
        # Fallback to flat `entries` if `raw.turns` is absent
        for entry in self._coalesce_user_entries(
            normalized.get("entries", []) or []
        ):
            kind = entry.get("kind")
            turn_num = int(entry.get("turn", 0)) + 1
            if kind == "user":
                u_text = (entry.get("text") or "").strip()
                history_lines.append(f"[Turn {turn_num}] User: {u_text}")
                cleaned_user_text = _extract_user_turn_transcript(u_text)
                if not cleaned_user_text:
                    continue
                user_turn_counter += 1
                is_non_spoken = _is_non_spoken_turn_text(cleaned_user_text)
                audio_uri = (
                    None
                    if is_non_spoken
                    else (
                        user_audio_uris.get(turn_num)
                        or user_audio_uris.get(user_turn_counter)
                    )
                )
                audio_path = None
                pcm_bytes = None
                if audio_uri and gcs_client is not None:
                    try:
                        raw_wav = gcs_client.download_blob(audio_uri)
                        pcm_bytes = extract_pcm_bytes_from_wav(raw_wav)
                    except Exception as exc:
                        logger.warning(
                            "Failed to download user audio %s: %s",
                            audio_uri,
                            exc,
                        )
                past_turns.append(
                    ShadowPastTurn(
                        turn_index=user_turn_counter,
                        user_transcript=cleaned_user_text,
                        preceding_agent_response=" ".join(last_agent_texts),
                        audio_uri=audio_uri,
                        audio_path=audio_path,
                        has_audio=bool(pcm_bytes),
                        used=False,
                        audio_bytes=pcm_bytes or None,
                        agent_audio_uri=agent_audio_uris.get(turn_num),
                    )
                )
                last_agent_texts = []
            elif kind == "agent":
                a_text = (entry.get("text") or "").strip()
                last_agent_texts.append(a_text)
                if past_turns and a_text:
                    past_turns[-1].agent_response = " ".join(
                        filter(
                            None, [past_turns[-1].agent_response, a_text]
                        )
                    )
                history_lines.append(f"[Turn {turn_num}] Agent: {a_text}")
            elif kind == "tool_call":
                history_lines.append(
                    f"[Turn {turn_num}] Tool Call: {entry.get('tool')} "
                    f"args={entry.get('args')}"
                )

    # Missing recordings silently degrade the replay to TTS, which defeats
    # the point of a ShadowEval, so surface it loudly.
    spoken = [
        pt
        for pt in past_turns
        if not _is_non_spoken_turn_text(pt.user_transcript)
    ]
    with_audio = sum(1 for pt in spoken if pt.has_audio)
    if spoken and with_audio < len(spoken):
        logger.warning(
            "Only %d/%d past user turns of conversation %s have recorded "
            "audio (bucket=%s); the rest will be synthesized via TTS. "
            "Check `gcs_bucket` and that the app's audio recording "
            "(loggingSettings.audioRecordingConfig) was enabled for the "
            "original call.",
            with_audio,
            len(spoken),
            tc.conversation_id,
            bucket_override or "<default>",
        )

    return past_turns, "\n".join(history_lines), normalized

save_shadow_artifacts

save_shadow_artifacts(shadow_conv, artifacts_dir, test_case, session_id, use_tool_fakes=None, download_past_agent_audio=True)

Persists a replay's audio and a side-by-side HTML report.

Writes into artifacts_dir: - audio/past_user_turn_<N>.wav: recorded caller audio per past turn (exactly what was streamed for use_past_audio turns); - audio/past_agent_turn_<N>.wav: the original agent reply (when the source app recorded it); - audio/new_agent_turn_<N>.wav: the new agent reply (requires capture_agent_audio); - shadow_result.json and shadow_report.html.

Returns:

Type Description
str

Path to the HTML report.

Source code in src/cxas_scrapi/evals/shadow_evals.py
def save_shadow_artifacts(
    self,
    shadow_conv: ShadowUserConversation,
    artifacts_dir: str,
    test_case: ShadowTestCase,
    session_id: str,
    use_tool_fakes: bool | None = None,
    download_past_agent_audio: bool = True,
) -> str:
    """Persists a replay's audio and a side-by-side HTML report.

    Writes into `artifacts_dir`:
      - `audio/past_user_turn_<N>.wav`: recorded caller audio per past
        turn (exactly what was streamed for `use_past_audio` turns);
      - `audio/past_agent_turn_<N>.wav`: the original agent reply (when
        the source app recorded it);
      - `audio/new_agent_turn_<N>.wav`: the new agent reply (requires
        `capture_agent_audio`);
      - `shadow_result.json` and `shadow_report.html`.

    Returns:
        Path to the HTML report.
    """
    audio_dir = os.path.join(artifacts_dir, "audio")
    os.makedirs(audio_dir, exist_ok=True)

    def _rel(path: str) -> str:
        return os.path.relpath(path, artifacts_dir)

    gcs_client = None
    past_payload: dict[int, dict[str, Any]] = {}
    for pt in shadow_conv.past_turns:
        user_audio = None
        if pt.audio_bytes or (
            pt.audio_path and os.path.exists(pt.audio_path)
        ):
            user_audio = os.path.join(
                audio_dir, f"past_user_turn_{pt.turn_index}.wav"
            )
            if pt.audio_path and os.path.exists(pt.audio_path):
                shutil.copyfile(pt.audio_path, user_audio)
            else:
                with open(user_audio, "wb") as f:
                    f.write(pcm_to_wav_bytes(pt.audio_bytes or b""))
        agent_audio = None
        if download_past_agent_audio and pt.agent_audio_uri:
            try:
                gcs_client = gcs_client or GCSUtils(creds=self.creds)
                data = gcs_client.download_blob(pt.agent_audio_uri)
                agent_audio = os.path.join(
                    audio_dir, f"past_agent_turn_{pt.turn_index}.wav"
                )
                with open(agent_audio, "wb") as f:
                    f.write(data)
            except Exception as exc:
                logger.warning(
                    "Could not download past agent audio %s: %s",
                    pt.agent_audio_uri,
                    exc,
                )
                agent_audio = None
        past_payload[pt.turn_index] = {
            "turn_index": pt.turn_index,
            "user_text": pt.user_transcript,
            "user_audio": _rel(user_audio) if user_audio else None,
            "agent_text": pt.agent_response,
            "agent_audio": _rel(agent_audio) if agent_audio else None,
        }

    replayed: set[int] = set()
    turns: list[dict[str, Any]] = []
    for log in shadow_conv.turn_logs:
        past_idx = log.selected_past_turn_index
        if log.decision == "event" and shadow_conv.past_turns:
            first = shadow_conv.past_turns[0]
            if (first.user_transcript or "").lstrip().startswith("<event"):
                past_idx = first.turn_index
        past = past_payload.get(past_idx) if past_idx is not None else None
        if past_idx is not None:
            replayed.add(past_idx)
        new_agent_audio = None
        if log.agent_audio_path and os.path.exists(log.agent_audio_path):
            new_agent_audio = os.path.join(
                audio_dir, f"new_agent_turn_{log.sim_turn}.wav"
            )
            shutil.copyfile(log.agent_audio_path, new_agent_audio)
        is_past_audio = (
            log.decision == ShadowDecisionType.USE_PAST_AUDIO.value
        )
        turns.append(
            {
                "sim_turn": log.sim_turn,
                "decision": log.decision,
                "justification": log.decision_justification,
                "user_text": log.user_utterance,
                "user_audio": (
                    past.get("user_audio")
                    if past and is_past_audio
                    else None
                ),
                "agent_text": log.agent_response,
                "agent_audio": (
                    _rel(new_agent_audio) if new_agent_audio else None
                ),
                "past": past,
            }
        )

    results = shadow_conv.expectation_results or []
    met = sum(1 for r in results if r.status == ExpectationStatus.MET)
    data = {
        "name": test_case.name,
        "conversation_id": test_case.conversation_id,
        "source_app": self._resolve_source_app_name(test_case),
        "target_app": self.app_name,
        "session_id": session_id,
        "replay_mode": test_case.replay_mode,
        "session_parameters": test_case.session_parameters,
        "use_tool_fakes": use_tool_fakes,
        "summary": {
            "passed": bool(results) and met == len(results),
            "expectations": f"{met}/{len(results)}",
            "past_audio_turns": sum(
                1
                for t in shadow_conv.turn_logs
                if t.decision == ShadowDecisionType.USE_PAST_AUDIO.value
            ),
            "tts_turns": sum(
                1
                for t in shadow_conv.turn_logs
                if t.decision == ShadowDecisionType.GENERATE_TTS.value
            ),
        },
        "expectations": [
            {
                "expectation": r.expectation,
                "status": r.status.value,
                "justification": r.justification,
            }
            for r in results
        ],
        "turns": turns,
        "unreplayed_past_turns": [
            p
            for idx, p in past_payload.items()
            if idx not in replayed
            and _extract_user_turn_transcript(p["user_text"])
            and not _extract_user_turn_transcript(
                p["user_text"]
            ).startswith(("<event", "<must_reply"))
        ],
    }
    with open(
        os.path.join(artifacts_dir, "shadow_result.json"),
        "w",
        encoding="utf-8",
    ) as f:
        json.dump(data, f, indent=2, default=str)
    report_path = os.path.join(artifacts_dir, "shadow_report.html")
    with open(report_path, "w", encoding="utf-8") as f:
        f.write(render_shadow_html_report(data))
    return report_path

run_shadow_conversation

run_shadow_conversation(test_case, sim_user_model=_DEFAULT_GEMINI_MODEL, eval_model=_DEFAULT_GEMINI_MODEL, session_id=None, console_logging=True, modality='audio', capture_agent_audio=False, background_noise_file=None, burst_noise_files=None, use_tool_fakes=False, voice_config=None, initial_utterance=None, skip_playback_wait=False, single_bidi_stream=False, max_turns=None, naturalness=None, artifacts_dir=None, **kwargs)

Replays a past conversation on the target agent over Bidi audio (or text) using ShadowUserConversation to arbitrate between past audio files and TTS.

Downloaded and captured audio lives in a temporary session directory that is deleted when this returns. Pass artifacts_dir to keep it, together with a side-by-side HTML report (see save_shadow_artifacts); the report path is then available as conversation.report_path.

Source code in src/cxas_scrapi/evals/shadow_evals.py
@cleanup_session_dir
def run_shadow_conversation(
    self,
    test_case: ShadowTestCase | dict[str, Any],
    sim_user_model: str | None = _DEFAULT_GEMINI_MODEL,
    eval_model: str | None = _DEFAULT_GEMINI_MODEL,
    session_id: str | None = None,
    console_logging: bool = True,
    modality: str = "audio",
    capture_agent_audio: bool = False,
    background_noise_file: str | None = None,
    burst_noise_files: list[str] | None = None,
    use_tool_fakes: bool = False,
    voice_config: dict[str, Any] | None = None,
    initial_utterance: str | None = None,
    skip_playback_wait: bool = False,
    single_bidi_stream: bool = False,
    max_turns: int | None = None,
    naturalness: bool | dict[str, Any] | None = None,
    artifacts_dir: str | None = None,
    **kwargs: Any,
) -> ShadowUserConversation:
    """Replays a past conversation on the target agent over Bidi audio (or
    text) using `ShadowUserConversation` to arbitrate between past audio
    files and TTS.

    Downloaded and captured audio lives in a temporary session directory
    that is deleted when this returns. Pass `artifacts_dir` to keep it,
    together with a side-by-side HTML report (see
    `save_shadow_artifacts`); the report path is then available as
    `conversation.report_path`.
    """
    sim_user_model = sim_user_model or _DEFAULT_GEMINI_MODEL
    eval_model = eval_model or _DEFAULT_GEMINI_MODEL
    if session_id is None:
        session_id = str(uuid.uuid4())

    tc_model = (
        ShadowTestCase(**test_case)
        if isinstance(test_case, dict)
        else test_case
    )
    tc_dict = tc_model.model_dump()
    naturalness_config = parse_naturalness_config(
        tc_dict,
        naturalness if naturalness is not None else self.naturalness,
    )
    voice_config = voice_config or tc_model.voice_config
    if tc_model.use_tool_fakes is not None:
        use_tool_fakes = tc_model.use_tool_fakes
    # The report needs the new agent audio, so capture it when keeping
    # artifacts.
    capture_agent_audio = capture_agent_audio or bool(artifacts_dir)

    session_dir = f"/tmp/scrapi_evals/{session_id}"
    past_turns, full_past_history, _ = self.fetch_past_conversation_data(
        tc_model, session_dir=session_dir
    )

    shadow_conv = ShadowUserConversation(
        genai_client=self.genai_client,
        genai_model=sim_user_model,
        test_case=tc_model,
        past_turns=past_turns,
        full_past_history=full_past_history,
        max_turns=max_turns,
        initial_utterance=initial_utterance,
    )

    current_sim_turn = 0
    interactive_session = None
    if modality == "audio" and single_bidi_stream:
        client = self.sessions_client
        interactive_session = client.create_interactive_session(
            session_id=session_id,
            capture_agent_audio=capture_agent_audio,
            background_noise_file=background_noise_file,
            use_tool_fakes=use_tool_fakes,
            skip_playback_wait=skip_playback_wait,
            voice_config=voice_config,
        )
        interactive_session.start()

    try:
        if console_logging:
            print(
                f"Starting shadow conversation replay for "
                f"conversation_id={tc_model.conversation_id} "
                f"(session_id={session_id})"
            )

        user_utterance, audio_bytes, variables, turn_log = (
            shadow_conv.next_user_turn()
        )
        accumulated_variables: dict[str, Any] = {}
        if variables:
            accumulated_variables.update(variables)

        detailed_trace: list[str] = []
        if user_utterance:
            past_idx = (
                turn_log.selected_past_turn_index if turn_log else None
            )
            mode_tag = (
                f"[Past Audio Turn #{past_idx}]"
                if turn_log
                and turn_log.decision
                == ShadowDecisionType.USE_PAST_AUDIO.value
                else (
                    "[Generated TTS]"
                    if turn_log
                    and turn_log.decision
                    == ShadowDecisionType.GENERATE_TTS.value
                    else "[Event]"
                )
            )
            detailed_trace.append(f"User {mode_tag}: {user_utterance}")

        while user_utterance:
            if modality == "audio" and interactive_session:
                response = interactive_session.send_turn(
                    user_utterance,
                    accumulated_variables,
                    audio_bytes=audio_bytes,
                )
                if isinstance(response, dict) and response.get(
                    "session_ended"
                ):
                    if response.get("connection_error"):
                        raise BidiSessionError(
                            f"Interactive session WebSocket error: "
                            f"{response['connection_error']}"
                        )
                    break
            else:
                response = self._send_shadow_request_with_retry(
                    session_id=session_id,
                    user_utterance=user_utterance,
                    audio_bytes=audio_bytes,
                    variables=accumulated_variables,
                    modality=modality,
                    console_logging=console_logging,
                    turn_num=current_sim_turn,
                    capture_agent_audio=capture_agent_audio,
                    background_noise_file=background_noise_file,
                    burst_noise_files=burst_noise_files,
                    use_tool_fakes=use_tool_fakes,
                    voice_config=voice_config,
                )
            if not response:
                break

            if response and getattr(response, "agent_audio_paths", None):
                audio_path = response.agent_audio_paths.get(0)
                if audio_path:
                    shadow_conv.agent_audio_paths[current_sim_turn] = (
                        audio_path
                    )
                    if shadow_conv.turn_logs and isinstance(
                        audio_path, str
                    ):
                        shadow_conv.turn_logs[
                            -1
                        ].agent_audio_path = audio_path

            if console_logging:
                self.sessions_client.parse_result(response)

            agent_text, trace_chunks, session_ended, tool_calls = (
                self._parse_agent_response(response)
            )
            detailed_trace.append("\n".join(trace_chunks))

            if session_ended:
                if agent_text:
                    shadow_conv._add_agent_response(agent_text)
                    if shadow_conv.turn_logs:
                        shadow_conv.turn_logs[
                            -1
                        ].agent_response = agent_text
                shadow_conv._add_agent_tool_calls(tool_calls)
                for prog in shadow_conv.steps_progress:
                    if prog.status != StepStatus.COMPLETED:
                        prog.status = StepStatus.COMPLETED
                        prog.justification = (
                            "Session ended by agent; marking step complete."
                        )
                break

            shadow_conv._add_agent_tool_calls(tool_calls)
            user_utterance, audio_bytes, variables, turn_log = (
                shadow_conv.next_user_turn(agent_text)
            )
            if variables:
                accumulated_variables.update(variables)
            if user_utterance:
                past_idx = (
                    turn_log.selected_past_turn_index if turn_log else None
                )
                mode_tag = (
                    f"[Past Audio Turn #{past_idx}]"
                    if turn_log
                    and turn_log.decision
                    == ShadowDecisionType.USE_PAST_AUDIO.value
                    else "[Generated TTS]"
                )
                detailed_trace.append(f"User {mode_tag}: {user_utterance}")

            current_sim_turn += 1

        self._evaluate_expectations(
            shadow_conv,
            detailed_trace,
            eval_model,
            console_logging,
            capture_agent_audio=capture_agent_audio,
        )
        self._evaluate_naturalness(
            shadow_conv,
            detailed_trace,
            eval_model,
            console_logging,
            naturalness_config,
            modality=modality,
        )
        shadow_conv._session_id = session_id
        shadow_conv.session_id = session_id
        shadow_conv._detailed_trace = detailed_trace
        shadow_conv.detailed_trace = detailed_trace
        shadow_conv.report_path = None
        if artifacts_dir:
            try:
                shadow_conv.report_path = self.save_shadow_artifacts(
                    shadow_conv,
                    artifacts_dir,
                    tc_model,
                    session_id,
                    use_tool_fakes=use_tool_fakes,
                )
                if console_logging:
                    print(f"ShadowEval report: {shadow_conv.report_path}")
            except Exception as exc:
                logger.warning(
                    "Failed to save ShadowEval artifacts to %s: %s",
                    artifacts_dir,
                    exc,
                )
        return shadow_conv
    finally:
        if interactive_session:
            interactive_session.close()

run_shadow_evals

run_shadow_evals(test_cases, runs=1, parallel=1, sim_user_model=_DEFAULT_GEMINI_MODEL, eval_model=_DEFAULT_GEMINI_MODEL, modality='audio', verbose=False, capture_agent_audio=False, background_noise_file=None, burst_noise_files=None, use_tool_fakes=False, expectations_only=None, skip_playback_wait=False, single_bidi_stream=False, progress_callback=None, naturalness=None, artifacts_dir=None, preflight=True)

Runs a batch of shadow evaluation test cases across runs and parallel workers.

Parameters:

Name Type Description Default
artifacts_dir str | None

If set, each run keeps its audio and writes a side-by-side HTML report under <artifacts_dir>/<case>_run<N>/ (see save_shadow_artifacts).

None
preflight bool

Check every case with preflight first. Cases with preflight errors are reported as failed without opening a session; warnings are printed and attached to the results as preflight_issues.

True
Source code in src/cxas_scrapi/evals/shadow_evals.py
def run_shadow_evals(
    self,
    test_cases: list[ShadowTestCase | dict[str, Any]],
    runs: int = 1,
    parallel: int = 1,
    sim_user_model: str | None = _DEFAULT_GEMINI_MODEL,
    eval_model: str | None = _DEFAULT_GEMINI_MODEL,
    modality: str = "audio",
    verbose: bool = False,
    capture_agent_audio: bool = False,
    background_noise_file: str | None = None,
    burst_noise_files: list[str] | None = None,
    use_tool_fakes: bool = False,
    expectations_only: bool | None = None,
    skip_playback_wait: bool = False,
    single_bidi_stream: bool = False,
    progress_callback: Callable[[int, int], None] | None = None,
    naturalness: bool | dict[str, Any] | None = None,
    artifacts_dir: str | None = None,
    preflight: bool = True,
) -> list[dict[str, Any]]:
    """Runs a batch of shadow evaluation test cases across `runs` and
    `parallel` workers.

    Args:
        artifacts_dir: If set, each run keeps its audio and writes a
            side-by-side HTML report under
            `<artifacts_dir>/<case>_run<N>/` (see `save_shadow_artifacts`).
        preflight: Check every case with `preflight` first. Cases with
            preflight errors are reported as failed without opening a
            session; warnings are printed and attached to the results as
            `preflight_issues`.
    """
    if expectations_only is not None:
        self.expectations_only = expectations_only
    if naturalness is not None:
        self.naturalness = naturalness

    sim_user_model = sim_user_model or _DEFAULT_GEMINI_MODEL
    eval_model = eval_model or _DEFAULT_GEMINI_MODEL

    validated_cases = [
        ShadowTestCase(**tc) if isinstance(tc, dict) else tc
        for tc in test_cases
    ]

    results: list[dict[str, Any]] = []
    blocked: list[dict[str, Any]] = []
    preflight_issues: dict[int, list[dict[str, str]]] = {}
    if preflight:
        runnable: list[ShadowTestCase] = []
        for tc in validated_cases:
            check = self.preflight(tc, modality=modality)
            for issue in check.issues:
                print(
                    f"  [{issue.level.upper()}] {tc.name}: {issue.message}"
                )
            issues = [i.model_dump() for i in check.issues]
            if check.ok:
                runnable.append(tc)
                if issues:
                    preflight_issues[id(tc)] = issues
                continue
            errors = "; ".join(
                i.message for i in check.issues if i.level == "error"
            )
            blocked.append(
                {
                    "name": tc.name,
                    "conversation_id": tc.conversation_id,
                    "run": 1,
                    "passed": False,
                    "error": f"Preflight failed: {errors}",
                    "preflight_issues": issues,
                }
            )
        validated_cases = runnable

    jobs = [
        (tc, run_idx) for tc in validated_cases for run_idx in range(runs)
    ]
    issues_by_name = {
        tc.name: preflight_issues[id(tc)]
        for tc in validated_cases
        if id(tc) in preflight_issues
    }

    with Progress() as progress:
        task_id = progress.add_task("Running Shadow Evals", total=len(jobs))
        if parallel <= 1:
            for tc, run_idx in jobs:
                results.append(
                    self._run_single_shadow_job(
                        tc,
                        run_idx,
                        runs,
                        sim_user_model,
                        eval_model,
                        modality,
                        verbose,
                        parallel,
                        capture_agent_audio=capture_agent_audio,
                        background_noise_file=background_noise_file,
                        burst_noise_files=burst_noise_files,
                        use_tool_fakes=use_tool_fakes,
                        skip_playback_wait=skip_playback_wait,
                        single_bidi_stream=single_bidi_stream,
                        artifacts_dir=artifacts_dir,
                    )
                )
                progress.update(task_id, advance=1)
                if progress_callback:
                    progress_callback(len(results), len(jobs))
        else:
            max_workers = min(parallel, 25)
            with ThreadPoolExecutor(max_workers=max_workers) as executor:
                futures = {
                    executor.submit(
                        self._run_single_shadow_job,
                        tc,
                        run_idx,
                        runs,
                        sim_user_model,
                        eval_model,
                        modality,
                        verbose,
                        parallel,
                        capture_agent_audio=capture_agent_audio,
                        background_noise_file=background_noise_file,
                        burst_noise_files=burst_noise_files,
                        use_tool_fakes=use_tool_fakes,
                        skip_playback_wait=skip_playback_wait,
                        single_bidi_stream=single_bidi_stream,
                        artifacts_dir=artifacts_dir,
                    ): (tc.name, run_idx)
                    for tc, run_idx in jobs
                }
                for future in as_completed(futures):
                    results.append(future.result())
                    progress.update(task_id, advance=1)
                    if progress_callback:
                        progress_callback(len(results), len(jobs))

    for res in results:
        if res.get("name") in issues_by_name:
            res.setdefault("preflight_issues", issues_by_name[res["name"]])
    return blocked + results

load_shadow_test_cases_from_yaml staticmethod

load_shadow_test_cases_from_yaml(yaml_data)

Loads and validates ShadowTestCase items from a YAML string.

Source code in src/cxas_scrapi/evals/shadow_evals.py
@staticmethod
def load_shadow_test_cases_from_yaml(
    yaml_data: str,
) -> list[ShadowTestCase]:
    """Loads and validates `ShadowTestCase` items from a YAML string."""
    raw_data = yaml.safe_load(yaml_data)
    if not raw_data:
        return []

    global_config: dict[str, Any] = {}
    if isinstance(raw_data, list):
        raw_evals = raw_data
    elif isinstance(raw_data, dict):
        global_config = raw_data.get("config", {}) or {}
        raw_evals = (
            raw_data.get("shadow_evals")
            or raw_data.get("evals")
            or raw_data.get("conversations")
            or []
        )
    else:
        return []

    cases: list[ShadowTestCase] = []
    inherited_keys = (
        "project_id",
        "location",
        "app_id",
        "gcs_bucket",
        "max_turns",
        "voice_config",
        "replay_mode",
        "use_tool_fakes",
    )
    # Unknown keys are silently dropped by pydantic, so a typo such as
    # `session_params` would quietly run the replay without seeding.
    known_case_keys = set(ShadowTestCase.model_fields)
    ShadowEvals._warn_unknown_keys(
        global_config,
        set(inherited_keys) | {"session_parameters"},
        "config",
    )
    for item in raw_evals:
        if not isinstance(item, dict):
            continue
        label = item.get("name") or item.get("conversation_id")
        ShadowEvals._warn_unknown_keys(
            item, known_case_keys, f"shadow eval '{label}'"
        )
        merged = dict(item)
        for key in inherited_keys:
            if merged.get(key) is None and key in global_config:
                merged[key] = global_config[key]
        if "session_parameters" in global_config:
            merged_params = dict(global_config["session_parameters"])
            merged_params.update(merged.get("session_parameters") or {})
            merged["session_parameters"] = merged_params
        cases.append(ShadowTestCase(**merged))
    return cases

load_shadow_test_cases_from_file classmethod

load_shadow_test_cases_from_file(file_path)

Loads ShadowTestCase items from a YAML file path.

Source code in src/cxas_scrapi/evals/shadow_evals.py
@classmethod
def load_shadow_test_cases_from_file(
    cls, file_path: str
) -> list[ShadowTestCase]:
    """Loads `ShadowTestCase` items from a YAML file path."""
    with open(file_path, encoding="utf-8") as f:
        return cls.load_shadow_test_cases_from_yaml(f.read())

load_shadow_tests_from_dir classmethod

load_shadow_tests_from_dir(directory_path='evals/shadows')

Recursively loads all YAML shadow test cases from a directory.

Source code in src/cxas_scrapi/evals/shadow_evals.py
@classmethod
def load_shadow_tests_from_dir(
    cls, directory_path: str = "evals/shadows"
) -> list[ShadowTestCase]:
    """Recursively loads all YAML shadow test cases from a directory."""
    all_tests: list[ShadowTestCase] = []
    if not os.path.exists(directory_path):
        return all_tests

    for root, _, files in os.walk(directory_path):
        for file in sorted(files):
            if file.endswith((".yaml", ".yml")):
                file_path = os.path.join(root, file)
                all_tests.extend(
                    cls.load_shadow_test_cases_from_file(file_path)
                )
    return all_tests

ShadowTestCase

Bases: BaseModel

Configuration schema for a single ShadowEval test case.

Requires conversation_id and non-empty natural-language expectations that define the pass criteria for replaying the conversation.

ShadowPreflightResult

Bases: BaseModel

Outcome of ShadowEvals.preflight for one test case.

ok property

ok

True when no error-level issue was found.

ShadowPreflightIssue

Bases: BaseModel

A single problem found while checking a ShadowEval before running.

ShadowUserConversation

ShadowUserConversation(genai_client, genai_model, test_case, past_turns, full_past_history='', max_turns=None, initial_utterance=None)

Bases: Conversation

Simulated user for ShadowEvals that arbitrates between replaying past recorded audio turns and generating new responses via TTS.

Source code in src/cxas_scrapi/evals/shadow_evals.py
def __init__(
    self,
    genai_client: GeminiGenerate,
    genai_model: str,
    test_case: ShadowTestCase | dict[str, Any],
    past_turns: list[ShadowPastTurn],
    full_past_history: str = "",
    max_turns: int | None = None,
    initial_utterance: str | None = None,
) -> None:
    super().__init__()
    self.genai_client = genai_client
    self.genai_model = genai_model
    if isinstance(test_case, dict):
        self.test_case_model = ShadowTestCase(**test_case)
        self.test_case = test_case
    else:
        self.test_case_model = test_case
        self.test_case = test_case.model_dump()

    self.past_turns = [t.model_copy(deep=True) for t in past_turns]
    self.full_past_history = full_past_history or self._build_past_history()
    self.initial_utterance = (
        initial_utterance
        if initial_utterance is not None
        else self.test_case_model.initial_utterance
    )
    self.replay_mode = getattr(
        self.test_case_model, "replay_mode", "hybrid"
    )

    if max_turns is not None:
        self.max_turns = max_turns
    elif self.test_case_model.max_turns is not None:
        self.max_turns = self.test_case_model.max_turns
    else:
        default_limit = max(len(self.past_turns) * 2 + 4, 15)
        self.max_turns = min(default_limit, _MAX_TURNS)

    # Derive goal & response_guide if not explicitly provided
    self.goal = self.test_case_model.goal or self._infer_default_goal()
    self.response_guide = (
        self.test_case_model.response_guide
        or self._infer_default_response_guide()
    )

    # Setup step progress tracking
    self.steps_progress: list[StepProgress] = []
    if self.test_case_model.steps:
        for step in self.test_case_model.steps:
            self.steps_progress.append(
                StepProgress(
                    step=step,
                    status=StepStatus.NOT_STARTED,
                    justification="",
                )
            )
    else:
        exp_summary = "; ".join(
            str(e.get("expectation", "") if isinstance(e, dict) else e)
            for e in self.test_case_model.expectations
        )
        default_step = Step(
            goal=self.goal,
            success_criteria=exp_summary,
            response_guide=self.response_guide,
        )
        self.steps_progress.append(
            StepProgress(
                step=default_step,
                status=StepStatus.NOT_STARTED,
                justification="",
            )
        )

    self.expectations = list(self.test_case_model.expectations)
    self.audio_expectations: list[dict[str, Any]] = []
    for exp in self.test_case_model.audio_expectations:
        if isinstance(exp, dict):
            exp_dict = dict(exp)
            exp_dict["requires_audio_paths"] = True
        else:
            exp_dict = {
                "expectation": str(exp),
                "requires_audio_paths": True,
            }
        self.audio_expectations.append(exp_dict)

    self.expectation_results: list[ExpectationResult] = []
    self.naturalness_result: NaturalnessResult | None = None
    self.turn_logs: list[ShadowTurnLog] = []
    self.agent_audio_paths: dict[int, str] = {}

next_user_turn

next_user_turn(last_agent_response='')

Determines the next user utterance and whether to stream past raw audio bytes or generate new TTS audio.

Returns:

Name Type Description
tuple tuple[str, bytes | None, dict[str, Any], ShadowTurnLog | None]

(user_utterance, audio_bytes, variables_to_inject, turn_log) where user_utterance is the text transcript or event/dtmf string; audio_bytes is raw 16kHz PCM if replaying a past GCS recording, or None if TTS / event should be used; variables_to_inject is the session parameters dict; and turn_log is the ShadowTurnLog entry describing the decision.

Source code in src/cxas_scrapi/evals/shadow_evals.py
def next_user_turn(
    self, last_agent_response: str = ""
) -> tuple[str, bytes | None, dict[str, Any], ShadowTurnLog | None]:
    """Determines the next user utterance and whether to stream past raw
    audio bytes or generate new TTS audio.

    Returns:
        tuple: `(user_utterance, audio_bytes, variables_to_inject,
            turn_log)` where `user_utterance` is the text transcript or
            event/dtmf string; `audio_bytes` is raw 16kHz PCM if
            replaying a past GCS recording, or `None` if TTS / event
            should be used; `variables_to_inject` is the session
            parameters dict; and `turn_log` is the `ShadowTurnLog`
            entry describing the decision.
    """
    if last_agent_response:
        self._add_agent_response(last_agent_response)
        if self.turn_logs:
            self.turn_logs[-1].agent_response = last_agent_response

    if not self._check_conversation_status():
        self._add_user_utterance("")
        self.current_turn += 1
        return "", None, {}, None

    # Turn 0: Initial trigger (e.g. event: welcome) or first past turn
    if self.current_turn == 0:
        session_params = dict(self.test_case_model.session_parameters)
        if self.initial_utterance:
            utterance = _normalize_special_utterance(self.initial_utterance)
            self._add_user_utterance(utterance)
            # If first past turn was also an event or start,
            # mark it used so exact replay skips it
            if self.past_turns and not self.past_turns[0].has_audio:
                first_text = _extract_user_turn_transcript(
                    self.past_turns[0].user_transcript
                )
                if (
                    first_text.startswith("<event")
                    or first_text == utterance
                    or _normalize_special_utterance(first_text) == utterance
                ):
                    self.past_turns[0].used = True
            turn_log = ShadowTurnLog(
                sim_turn=0,
                decision="event",
                selected_past_turn_index=None,
                user_utterance=utterance,
                audio_source="event",
                decision_justification="Initial session event trigger.",
            )
            self.turn_logs.append(turn_log)
            self.current_turn += 1
            return utterance, None, session_params, turn_log

        first_pt = self._find_past_turn(None)
        if first_pt is not None:
            first_pt.used = True
            utterance = _normalize_special_utterance(
                first_pt.user_transcript
            )
            self._add_user_utterance(utterance)
            is_special = utterance.startswith(("dtmf:", "event:"))
            has_raw = bool(
                not is_special
                and first_pt.has_audio
                and first_pt.audio_bytes
            )
            decision_str = (
                ShadowDecisionType.USE_PAST_AUDIO.value
                if has_raw
                else ShadowDecisionType.GENERATE_TTS.value
            )
            turn_log = ShadowTurnLog(
                sim_turn=0,
                decision=decision_str,
                selected_past_turn_index=first_pt.turn_index,
                user_utterance=utterance,
                audio_source=(
                    (
                        first_pt.audio_uri
                        or first_pt.audio_path
                        or "past_audio"
                    )
                    if has_raw
                    else (
                        "DTMF"
                        if utterance.startswith("dtmf:")
                        else (
                            "event"
                            if utterance.startswith("event:")
                            else "TTS (fallback: no past audio)"
                        )
                    )
                ),
                user_audio_path=first_pt.audio_path if has_raw else None,
                decision_justification="Initial turn from past recording.",
            )
            self.turn_logs.append(turn_log)
            self.current_turn += 1
            return (
                utterance,
                first_pt.audio_bytes if has_raw else None,
                session_params,
                turn_log,
            )

    if self.replay_mode == "exact":
        next_pt = self._find_past_turn(None)
        if next_pt is None:
            for prog in self.steps_progress:
                if prog.status == StepStatus.IN_PROGRESS:
                    prog.status = StepStatus.COMPLETED
                    if not prog.justification:
                        prog.justification = (
                            "All historical audio turns replayed."
                        )
            self._add_user_utterance("")
            self.current_turn += 1
            return "", None, {}, None

        next_pt.used = True
        utterance = _normalize_special_utterance(next_pt.user_transcript)
        self._add_user_utterance(utterance)
        is_special = utterance.startswith(("dtmf:", "event:"))
        has_raw = bool(
            not is_special
            and next_pt.has_audio
            and next_pt.audio_bytes is not None
        )
        decision_str = (
            ShadowDecisionType.USE_PAST_AUDIO.value
            if has_raw
            else ShadowDecisionType.GENERATE_TTS.value
        )
        audio_src = (
            (
                next_pt.audio_uri
                or next_pt.audio_path
                or f"user-turn-{next_pt.turn_index}.wav"
            )
            if has_raw
            else (
                "DTMF"
                if utterance.startswith("dtmf:")
                else (
                    "event"
                    if utterance.startswith("event:")
                    else "TTS (no past audio)"
                )
            )
        )
        turn_log = ShadowTurnLog(
            sim_turn=self.current_turn,
            decision=decision_str,
            selected_past_turn_index=next_pt.turn_index,
            user_utterance=utterance,
            audio_source=audio_src,
            user_audio_path=next_pt.audio_path if has_raw else None,
            decision_justification=(
                f"Exact replay of past turn #{next_pt.turn_index} "
                "using recorded caller audio."
                if has_raw
                else f"Exact replay of past turn #{next_pt.turn_index}."
            ),
        )
        self.turn_logs.append(turn_log)
        self.current_turn += 1
        return (
            utterance,
            next_pt.audio_bytes if has_raw else None,
            {},
            turn_log,
        )

    prompt = self._prepare_shadow_llm_prompt()
    output: ShadowUserConversation.Output = self.genai_client.generate(
        prompt=prompt,
        model_name=self.genai_model,
        response_mime_type="application/json",
        response_schema=ShadowUserConversation.Output,
    )

    if not output:
        self._add_user_utterance("")
        self.current_turn += 1
        return "", None, {}, None

    if output.step_progresses:
        self.steps_progress = output.step_progresses

    if (
        output.decision == ShadowDecisionType.END_CONVERSATION
        or not self._check_conversation_status()
    ):
        for prog in self.steps_progress:
            if prog.status == StepStatus.IN_PROGRESS:
                prog.status = StepStatus.COMPLETED
                if not prog.justification:
                    prog.justification = (
                        output.decision_justification
                        or "Shadow conversation completed."
                    )
        self._add_user_utterance("")
        self.current_turn += 1
        return "", None, {}, None

    if output.decision == ShadowDecisionType.USE_PAST_AUDIO:
        matched_pt = self._find_past_turn(output.selected_past_turn_index)
        if matched_pt is not None:
            matched_pt.used = True
            raw_utterance = (
                matched_pt.user_transcript or output.next_user_utterance
            )
            utterance = _normalize_special_utterance(raw_utterance)
            if utterance.startswith(("dtmf:", "event:")):
                self._add_user_utterance(utterance)
                turn_log = ShadowTurnLog(
                    sim_turn=self.current_turn,
                    decision=ShadowDecisionType.GENERATE_TTS.value,
                    selected_past_turn_index=matched_pt.turn_index,
                    user_utterance=utterance,
                    audio_source=(
                        "DTMF" if utterance.startswith("dtmf:") else "event"
                    ),
                    decision_justification=output.decision_justification,
                )
                self.turn_logs.append(turn_log)
                self.current_turn += 1
                return utterance, None, {}, turn_log

            if matched_pt.has_audio and matched_pt.audio_bytes is not None:
                return self._emit_past_audio_turn(
                    matched_pt, output.decision_justification
                )

            # Fallback to TTS if past turn had no audio file in GCS
            utterance = _normalize_special_utterance(
                utterance or output.next_user_utterance
            )
            self._add_user_utterance(utterance)
            turn_log = ShadowTurnLog(
                sim_turn=self.current_turn,
                decision=ShadowDecisionType.GENERATE_TTS.value,
                selected_past_turn_index=matched_pt.turn_index,
                user_utterance=utterance,
                audio_source="TTS (past turn had no audio recording)",
                decision_justification=(
                    f"{output.decision_justification} "
                    "(Fallback to TTS: past audio bytes unavailable)"
                ),
            )
            self.turn_logs.append(turn_log)
            self.current_turn += 1
            return utterance, None, {}, turn_log

    # Guardrail: the LLM sometimes paraphrases a recorded past turn via
    # TTS instead of replaying it. If the generated text restates an
    # unused past turn that has audio, replay the authentic recording.
    utterance = _normalize_special_utterance(output.next_user_utterance)
    special_pt = self._match_unused_special_past_turn(
        utterance, output.selected_past_turn_index
    )
    if special_pt is not None:
        special_pt.used = True
        selected_idx = special_pt.turn_index
    else:
        selected_idx = output.selected_past_turn_index
        if selected_idx is not None:
            explicit_pt = self._find_past_turn(selected_idx)
            if explicit_pt is not None and not explicit_pt.has_audio:
                explicit_pt.used = True

    paraphrased_pt = self._match_unused_past_turn(utterance)
    if paraphrased_pt is not None:
        paraphrased_pt.used = True
        return self._emit_past_audio_turn(
            paraphrased_pt,
            (
                f"{output.decision_justification} (Overridden: generated "
                f"TTS {utterance!r} restates past turn "
                f"#{paraphrased_pt.turn_index}; replaying recorded audio.)"
            ),
        )

    # Deviation mode: GENERATE_TTS (or DTMF / event)
    self._add_user_utterance(utterance)
    audio_source = (
        "DTMF"
        if utterance.startswith("dtmf:")
        else ("event" if utterance.startswith("event:") else "TTS")
    )
    turn_log = ShadowTurnLog(
        sim_turn=self.current_turn,
        decision=ShadowDecisionType.GENERATE_TTS.value,
        selected_past_turn_index=selected_idx,
        user_utterance=utterance,
        audio_source=audio_source,
        decision_justification=output.decision_justification,
    )
    self.turn_logs.append(turn_log)
    self.current_turn += 1
    return utterance, None, {}, turn_log

next_user_utterance

next_user_utterance(last_agent_response='')

Compatibility wrapper returning (utterance, variables).

Source code in src/cxas_scrapi/evals/shadow_evals.py
def next_user_utterance(
    self, last_agent_response: str = ""
) -> tuple[str, dict[str, Any]]:
    """Compatibility wrapper returning `(utterance, variables)`."""
    utterance, _, variables, _ = self.next_user_turn(last_agent_response)
    return utterance, variables

generate_report

generate_report()

Generates a ShadowReport summarizing turn decisions, goals, and expectations.

Source code in src/cxas_scrapi/evals/shadow_evals.py
def generate_report(self) -> ShadowReport:
    """Generates a `ShadowReport` summarizing turn decisions, goals, and
    expectations.
    """
    turn_records = [
        {
            "sim_turn": t.sim_turn,
            "decision": t.decision,
            "past_turn_idx": t.selected_past_turn_index,
            "user_utterance": t.user_utterance,
            "audio_source": t.audio_source,
            "justification": t.decision_justification,
        }
        for t in self.turn_logs
    ]
    turns_df = pd.DataFrame(turn_records)

    goal_records = [
        {
            "goal": prog.step.goal,
            "success_criteria": prog.step.success_criteria,
            "status": prog.status.value,
            "justification": prog.justification,
        }
        for prog in self.steps_progress
    ]
    goals_df = pd.DataFrame(goal_records)

    expectations_df = None
    if self.expectation_results:
        exp_records = [
            {
                "expectation": res.expectation,
                "status": res.status.value,
                "justification": res.justification,
            }
            for res in self.expectation_results
        ]
        expectations_df = pd.DataFrame(exp_records)

    naturalness_df = None
    naturalness_headline = ""
    if self.naturalness_result:
        result = self.naturalness_result
        nat_records = []
        for turn in result.turns:
            record: dict[str, Any] = {
                "turn": turn.turn_index,
                "label": turn.label.value,
                "score": turn.score,
            }
            latency_ms = result.latency_ms_by_turn.get(turn.turn_index)
            if latency_ms is not None:
                record["latency_s"] = round(latency_ms / 1000.0, 2)
            for factor in turn.factors:
                if factor.quality:
                    record[factor.quality] = factor.score
            record["justification"] = turn.justification
            nat_records.append(record)
        naturalness_df = pd.DataFrame(nat_records)
        naturalness_headline = (
            f"Overall: {result.overall_score}/5 "
            f"({result.overall_label.value})"
        )

    return ShadowReport(
        turns_df=turns_df,
        goals_df=goals_df,
        expectations_df=expectations_df,
        naturalness_df=naturalness_df,
        naturalness_headline=naturalness_headline,
    )

ShadowTurnLog

Bases: BaseModel

Logs the Shadow SimUser decision and audio source for a single turn.

ShadowReport

ShadowReport(turns_df, goals_df=None, expectations_df=None, naturalness_df=None, naturalness_headline='')

A report containing Turn Decisions, Goals, and Expectations DataFrames.

Source code in src/cxas_scrapi/evals/shadow_evals.py
def __init__(
    self,
    turns_df: pd.DataFrame,
    goals_df: pd.DataFrame | None = None,
    expectations_df: pd.DataFrame | None = None,
    naturalness_df: pd.DataFrame | None = None,
    naturalness_headline: str = "",
) -> None:
    self.turns_df = turns_df
    self.goals_df = goals_df
    self.expectations_df = expectations_df
    self.naturalness_df = naturalness_df
    self.naturalness_headline = naturalness_headline