Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 12 additions & 1 deletion services/intake/src/nmp/intake/spans/ingest/atif_mapping.py
Original file line number Diff line number Diff line change
Expand Up @@ -981,7 +981,11 @@ def _datetime_from_value(value: Any) -> datetime | None:
def _tool_result_is_error(step: AtifStep, result: AtifObservationResult | None) -> bool:
"""Recognize supported ATIF producer error markers."""
# These markers are emitted by the Claude-Code Harbor trajectories we ingest today.
# ATIF itself does not define a normalized tool-result error field.
# ATIF itself does not define a normalized tool-result error field, so the text marker
# below cannot tell a failure from a success that quotes one.
explicit_result_status = _result_extra_bool(result, "is_error")
if explicit_result_status is not None:
return explicit_result_status
metadata = _matched_tool_result_metadata(step, result)
if metadata.get("is_error") is True:
return True
Expand Down Expand Up @@ -1040,6 +1044,13 @@ def _step_extra_bool(step: AtifStep, key: str) -> bool:
return step.extra is not None and step.extra.get(key) is True


def _result_extra_bool(result: AtifObservationResult | None, key: str) -> bool | None:
"""Read an explicit boolean from observation-result extras."""
extra = result.extra if result is not None else None
value = extra.get(key) if extra is not None else None
return value if isinstance(value, bool) else None


def _step_extra_dict(step: AtifStep, key: str) -> dict[str, Any]:
"""Read a dictionary value from step extras."""
value = step.extra.get(key) if step.extra is not None else None
Expand Down
87 changes: 86 additions & 1 deletion services/intake/tests/test_atif_v17.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@

import pytest
from nmp.intake.spans.api.spans_schemas import Span
from nmp.intake.spans.domain import SpanKind, SpanStatus
from nmp.intake.spans.domain import IntakeSpan, SpanKind, SpanStatus
from nmp.intake.spans.ingest.atif import AtifIngestRequest
from nmp.intake.spans.ingest.atif_domain import (
AtifAgent,
Expand Down Expand Up @@ -704,6 +704,91 @@ def test_atif_mapping_keeps_tool_error_on_tool_span() -> None:
assert tool.status == SpanStatus.ERROR


QUOTED_ERROR_CONTENT = '{"success":true,"snippet":"[ERROR] Failed to download package"}'


def _tool_result_spans(results: list[dict[str, Any]]) -> list[IntakeSpan]:
"""Map a single tool-using step whose observation carries the given results."""
step: dict[str, Any] = {
"step_id": 2,
"source": "agent",
"message": "using a tool",
"tool_calls": [
{"tool_call_id": result["source_call_id"], "function_name": f"tool_{index}"}
for index, result in enumerate(results)
],
"observation": {"results": results},
}
trajectory = AtifTrajectory.model_validate(
{
"schema_version": "ATIF-v1.7",
"session_id": "trace-session-id",
"agent": {"name": "sample-agent", "version": "1.0.0"},
"steps": [{"step_id": 1, "source": "user", "message": "search the incident archive"}, step],
}
)
return trajectory_to_spans(
workspace="default",
trajectory=trajectory,
ingested_at=datetime(2026, 5, 18, tzinfo=timezone.utc),
)


def test_atif_mapping_keeps_successful_tool_result_that_quotes_an_error() -> None:
spans = _tool_result_spans(
[{"source_call_id": "call-1", "content": QUOTED_ERROR_CONTENT, "extra": {"is_error": False}}]
)

tool = next(span for span in spans if span.kind == SpanKind.TOOL)
assert tool.status == SpanStatus.SUCCESS
assert Span.from_domain(tool).error_message is None
assert all(span.status == SpanStatus.SUCCESS for span in spans)


def test_atif_mapping_marks_tool_error_from_explicit_result_extra() -> None:
spans = _tool_result_spans([{"source_call_id": "call-1", "content": "all good", "extra": {"is_error": True}}])

tool = next(span for span in spans if span.kind == SpanKind.TOOL)
assert tool.status == SpanStatus.ERROR
assert Span.from_domain(tool).error_message == "all good"


def test_atif_mapping_resolves_parallel_tool_results_independently() -> None:
spans = _tool_result_spans(
[
{"source_call_id": "call-1", "content": QUOTED_ERROR_CONTENT, "extra": {"is_error": False}},
{"source_call_id": "call-2", "content": "tool crashed", "extra": {"is_error": True}},
]
)

tools = {span.name: span for span in spans if span.kind == SpanKind.TOOL}
assert tools["tool_0"].status == SpanStatus.SUCCESS
assert tools["tool_1"].status == SpanStatus.ERROR


def test_atif_mapping_keeps_legacy_text_marker_without_explicit_status() -> None:
spans = _tool_result_spans([{"source_call_id": "call-1", "content": QUOTED_ERROR_CONTENT}])

tool = next(span for span in spans if span.kind == SpanKind.TOOL)
assert tool.status == SpanStatus.ERROR


def test_atif_mapping_keeps_successful_subagent_result_that_quotes_an_error() -> None:
spans = _tool_result_spans(
[
{
"source_call_id": "call-1",
"content": QUOTED_ERROR_CONTENT,
"extra": {"is_error": False},
"subagent_trajectory_ref": [{"trajectory_id": "subagent-trajectory-1"}],
}
]
)

subagent = next(span for span in spans if span.name.startswith("subagent-"))
assert subagent.status == SpanStatus.SUCCESS


def test_atif_mapping_span_ids_are_trace_native_and_ignore_evaluation_run_id() -> None:
base = {
"schema_version": "ATIF-v1.5",
Expand Down
Loading