kaxil commented on code in PR #71575:
URL: https://github.com/apache/airflow/pull/71575#discussion_r4190202788


##########
providers/common/ai/src/airflow/providers/common/ai/durable/fingerprint.py:
##########
@@ -102,19 +159,201 @@ def _strip_volatile(messages_dump: list[dict[str, Any]]) 
-> list[dict[str, Any]]
     return stripped
 
 
+def _check_guide(guide: Any) -> bool:
+    """
+    Refuse an iterator, and report whether the JSON rendering needs 
``_order_sets``.
+
+    ``guide`` is pydantic's python-mode dump of the payload. It has the shape 
of the
+    json-mode rendering that is hashed -- the same keys, aliases, exclusions,
+    computed fields, extras and serializer output -- but keeps sets as sets 
and dict
+    keys as they are, and wraps an iterator in a ``SerializationIterator`` 
without
+    reading it. So this walk runs no user code, and nothing has been consumed 
yet.
+
+    Raises ``TypeError`` if the payload renders through an iterator anywhere.
+    Rendering it to JSON would consume it, and pydantic validates an 
``Iterable[T]``
+    tool parameter lazily into an iterator; a tool return that holds a 
generator
+    would reach the model empty. Returns ``True`` if the guide holds a set, or 
a dict
+    key that is not a string, since only those can make the JSON rendering 
depend on
+    the hash seed or lose a key.
+    """
+    needs_check = False
+    pending = [guide]
+    seen: set[int] = set()
+    while pending:
+        item = pending.pop()
+        if id(item) in seen:
+            # The dump shares this object, or left a cycle in place: walked 
already.
+            continue
+        seen.add(id(item))
+        children: Iterable[Any]
+        if isinstance(item, dict):
+            needs_check = needs_check or any(not isinstance(key, str) for key 
in item)
+            children = item.values()
+        elif isinstance(item, (list, tuple, deque)):
+            children = item
+        elif isinstance(item, (set, frozenset)):
+            needs_check = True
+            children = item
+        elif isinstance(item, Iterator):
+            raise TypeError(f"cannot fingerprint a value that renders through 
a {type(item).__name__}")
+        elif isinstance(item, enum.Enum):
+            # Python mode keeps an Enum member; JSON renders its value.
+            children = (item.value,)
+        else:
+            continue
+        pending.extend(child for child in children if type(child) not in 
_LEAF_TYPES)
+    return needs_check
+
+
+def _json_order(member: Any) -> str:
+    return json.dumps(member, sort_keys=True)
+
+
+def _template(value: Any) -> Any:
+    """
+    Say where the sets are inside one set member, or ``None`` if that cannot 
be said.
+
+    ``_LEAF`` for a member with no set inside it, ``("set", inner)`` for a set 
whose
+    members all have the template ``inner``, and ``("sequence", parts)`` for a 
tuple
+    with a set somewhere in it.
+    """
+    if isinstance(value, enum.Enum):
+        return _template(value.value)
+    if isinstance(value, (set, frozenset)):
+        inner = _member_template(value)
+        return None if inner is None else ("set", inner)
+    if isinstance(value, (list, tuple, deque)):
+        parts = tuple(_template(item) for item in value)
+        if None in parts:
+            return None
+        return _LEAF if all(part == _LEAF for part in parts) else ("sequence", 
parts)
+    if isinstance(value, dict):
+        # No set member dumps to a dict (python mode refuses a set of models), 
so
+        # nothing is known about one.
+        return None
+    return _LEAF
+
+
+def _member_template(members: Iterable[Any]) -> Any:
+    """Return the template every member of a set shares, or ``None`` if they 
differ."""
+    templates = {_template(member) for member in members}
+    if not templates:
+        return _LEAF
+    return templates.pop() if len(templates) == 1 else None
+
+
+def _apply_template(template: Any, rendered: Any) -> Any:
+    """Sort the lists that ``template`` puts a set at, in the rendering of one 
set member."""
+    if template == _LEAF or not isinstance(rendered, list):
+        return rendered
+    kind, inner = template
+    if kind == "set":
+        return sorted((_apply_template(inner, item) for item in rendered), 
key=_json_order)
+    if len(rendered) != len(inner):
+        return rendered
+    return [_apply_template(part, item) for part, item in zip(inner, rendered)]
+
+
+def _order_sets(guide: Any, rendered: Any) -> Any:
+    """
+    Return ``rendered`` with every list that pydantic rendered from a set 
sorted.
+
+    ``rendered`` is the json-mode rendering that is hashed, and ``guide`` the
+    python-mode dump of the same payload (see ``_check_guide``). They are 
walked
+    side by side: a dict or a sequence pairs up by position, because both dumps
+    keep the same order, and a list is sorted where the guide holds a set. A 
set's
+    members cannot pair up by position -- the dump builds a new set, which may
+    iterate in another order -- so the members' shared template says where any 
set
+    inside one of them sits (see ``_template``). Anything that does not line 
up is
+    kept exactly as rendered, so nothing that did not come from a set is 
reordered.
+
+    Raises ``TypeError`` where distinct dict keys render alike, such as ``1`` 
and
+    ``"1"``, or ``None`` and ``nan`` in a message dump: the guide then holds 
more keys
+    than the rendering, and hashing it would let two different payloads share a
+    digest. A dict whose keys are all strings cannot collide, so one that 
renders
+    fewer keys was reshaped by a serializer that applies only in JSON mode, 
and is
+    kept as rendered.
+    """
+    if isinstance(guide, enum.Enum):
+        # An Enum member renders as its value.
+        return _order_sets(guide.value, rendered)
+    if isinstance(guide, (set, frozenset)):
+        if not isinstance(rendered, list) or len(rendered) != len(guide):
+            return rendered
+        template = _member_template(guide)
+        if template is None:
+            return rendered
+        return sorted((_apply_template(template, item) for item in rendered), 
key=_json_order)
+    if isinstance(guide, dict):
+        if not isinstance(rendered, dict):
+            return rendered
+        if len(rendered) != len(guide):
+            if len(rendered) < len(guide) and any(not isinstance(key, str) for 
key in guide):
+                raise TypeError("dict keys collide once rendered as JSON")

Review Comment:
   Two gaps remain for non-string keys in this block. A JSON-only serializer 
that filters a `dict[int, int]` (dropping zero counts, say) lands here as "dict 
keys collide" even though nothing collides, and since it's in a tool return, 
the rest of the run fingerprints `None` where main hashed it. Raising only when 
rendering the keys on their own, `to_jsonable_python(dict.fromkeys(guide), 
bytes_mode="base64")`, gives fewer keys than the guide would keep the `{1: "a", 
"1": "b"}` refusal. Separately, the bail-out at 296 only checks string keys, so 
a JSON-only serializer that reorders an int-keyed dict pairs a set with its 
neighbour's list. `{2: {"p", "q"}, 1: ["y", "x"]}` and the same with `["x", 
"y"]` then share a digest. Also bailing out for a non-string `guide_key` whose 
rendering `next(iter(to_jsonable_python({guide_key: None})))` differs from 
`key` closes it. That contradicts the docstring's "nothing that did not come 
from a set is reordered" at 268.



##########
providers/common/ai/docs/durable_execution.rst:
##########
@@ -107,6 +107,58 @@ cache:
    never replays responses that belong to a different conversation.
 4. After successful completion, the cached steps are deleted.
 
+Plain JSON arguments and settings are fingerprinted exactly as before. Anything
+else is fingerprinted from pydantic's JSON rendering, so ordinary types that 
are
+not JSON -- a ``datetime`` or ``Decimal`` tool argument, a dataclass in
+``tool_choice``, the bytes in a ``BinaryContent``, a dict keyed by date --
+fingerprint normally. Bytes in tool arguments and settings are rendered as 
base64,
+so binary data that is not valid UTF-8 fingerprints too; bytes inside a 
pydantic
+model follow that model's ``ser_json_bytes`` setting instead, which renders 
them as
+UTF-8 text by default. Because a tool call is fingerprinted from how its 
arguments
+render, a field excluded from serialization and a secret value, which renders
+masked, take no part in it.
+
+A step whose request cannot be fingerprinted is not cached, and on retry it 
runs
+live rather than replaying an unverified entry. That happens when pydantic 
cannot

Review Comment:
   Pydantic does render both of these in JSON mode: `{F(x=1), F(x=2)}` becomes 
a list of dicts, and a model key becomes its str form. The refusal comes from 
the python-mode dump the fingerprint uses to find sets, and it covers frozen 
dataclasses too, so "pydantic cannot render" points readers at the wrong thing. 
If the fallback on `fingerprint.py:376` lands, this only applies to tool 
arguments, where main returned `None` anyway. If it doesn't, it's a second 
exception to "hashes the same way as before" at line 134.



##########
providers/common/ai/src/airflow/providers/common/ai/durable/fingerprint.py:
##########
@@ -123,44 +362,86 @@ def fingerprint_model_request(
     output mode and schema, native tools, ...) so any change to what is sent
     to the model invalidates the cached response.
 
-    Returns ``None`` when the request cannot be serialized; ``None`` compares
-    equal to ``None``, so requests that cannot be fingerprinted degrade to
-    unverified positional replay rather than disabling caching.
+    Returns ``None`` when the request cannot be fingerprinted, which prevents 
the
+    step from being replayed or cached. Model settings, the tool definitions 
in the
+    request parameters and the message history are normally carried into every
+    later request, so such a value in any of them usually degrades every
+    subsequent model step of the run the same way. ``step`` is attached to the
+    warning so the log names where it began.
     """
     try:
+        # Messages and parameters are hashed from pydantic's json-mode dump, 
as they
+        # always were, so stored fingerprints still match. The python-mode 
dumps only
+        # guide ``_order_sets``, which runs where a set or a non-string key 
needs it.
+        messages_guide = ModelMessagesTypeAdapter.dump_python(messages)

Review Comment:
   These python-mode dumps raise where the json-mode dump below succeeds. 
Python mode dumps a model or dataclass to a dict even when it's a set member or 
a dict key, and that fails with `TypeError: unhashable type: 'dict'`. So a tool 
that returns a frozenset of frozen dataclasses or models, or a model with 
`by_region: dict[Region, float]` where `Region` is frozen, makes this request 
fingerprint `None`. Since the return stays in the history, every later model 
step does too, and the rest of the agent re-runs live on retry. Main hashed all 
of these with a digest that was stable across hash seeds, and so did f3ebf75. 
The dict-key case also depends on the pydantic version: 2.12.0 (the floor) 
keeps the key object in python mode and fingerprints it, while 2.13.5 raises, 
so a test would need to cover both. Wrapping each of the two guide dumps in 
`except TypeError` and hashing that json-mode dump unordered gave main's digest 
for every case I tried. `_render` can keep raising, because main ret
 urned `None` for those tool args and settings anyway. The comment at 373-375 
("only guide `_order_sets`") would need correcting either way.



##########
providers/common/ai/src/airflow/providers/common/ai/durable/fingerprint.py:
##########
@@ -102,19 +159,201 @@ def _strip_volatile(messages_dump: list[dict[str, Any]]) 
-> list[dict[str, Any]]
     return stripped
 
 
+def _check_guide(guide: Any) -> bool:
+    """
+    Refuse an iterator, and report whether the JSON rendering needs 
``_order_sets``.
+
+    ``guide`` is pydantic's python-mode dump of the payload. It has the shape 
of the
+    json-mode rendering that is hashed -- the same keys, aliases, exclusions,
+    computed fields, extras and serializer output -- but keeps sets as sets 
and dict
+    keys as they are, and wraps an iterator in a ``SerializationIterator`` 
without
+    reading it. So this walk runs no user code, and nothing has been consumed 
yet.
+
+    Raises ``TypeError`` if the payload renders through an iterator anywhere.
+    Rendering it to JSON would consume it, and pydantic validates an 
``Iterable[T]``
+    tool parameter lazily into an iterator; a tool return that holds a 
generator
+    would reach the model empty. Returns ``True`` if the guide holds a set, or 
a dict
+    key that is not a string, since only those can make the JSON rendering 
depend on
+    the hash seed or lose a key.
+    """
+    needs_check = False
+    pending = [guide]
+    seen: set[int] = set()
+    while pending:
+        item = pending.pop()
+        if id(item) in seen:
+            # The dump shares this object, or left a cycle in place: walked 
already.
+            continue
+        seen.add(id(item))
+        children: Iterable[Any]
+        if isinstance(item, dict):
+            needs_check = needs_check or any(not isinstance(key, str) for key 
in item)
+            children = item.values()
+        elif isinstance(item, (list, tuple, deque)):
+            children = item
+        elif isinstance(item, (set, frozenset)):
+            needs_check = True
+            children = item
+        elif isinstance(item, Iterator):
+            raise TypeError(f"cannot fingerprint a value that renders through 
a {type(item).__name__}")
+        elif isinstance(item, enum.Enum):
+            # Python mode keeps an Enum member; JSON renders its value.
+            children = (item.value,)
+        else:
+            continue
+        pending.extend(child for child in children if type(child) not in 
_LEAF_TYPES)
+    return needs_check
+
+
+def _json_order(member: Any) -> str:
+    return json.dumps(member, sort_keys=True)
+
+
+def _template(value: Any) -> Any:
+    """
+    Say where the sets are inside one set member, or ``None`` if that cannot 
be said.
+
+    ``_LEAF`` for a member with no set inside it, ``("set", inner)`` for a set 
whose
+    members all have the template ``inner``, and ``("sequence", parts)`` for a 
tuple
+    with a set somewhere in it.
+    """
+    if isinstance(value, enum.Enum):
+        return _template(value.value)
+    if isinstance(value, (set, frozenset)):
+        inner = _member_template(value)
+        return None if inner is None else ("set", inner)
+    if isinstance(value, (list, tuple, deque)):
+        parts = tuple(_template(item) for item in value)
+        if None in parts:
+            return None
+        return _LEAF if all(part == _LEAF for part in parts) else ("sequence", 
parts)
+    if isinstance(value, dict):
+        # No set member dumps to a dict (python mode refuses a set of models), 
so
+        # nothing is known about one.
+        return None
+    return _LEAF
+
+
+def _member_template(members: Iterable[Any]) -> Any:
+    """Return the template every member of a set shares, or ``None`` if they 
differ."""
+    templates = {_template(member) for member in members}
+    if not templates:
+        return _LEAF
+    return templates.pop() if len(templates) == 1 else None
+
+
+def _apply_template(template: Any, rendered: Any) -> Any:
+    """Sort the lists that ``template`` puts a set at, in the rendering of one 
set member."""
+    if template == _LEAF or not isinstance(rendered, list):
+        return rendered
+    kind, inner = template
+    if kind == "set":
+        return sorted((_apply_template(inner, item) for item in rendered), 
key=_json_order)
+    if len(rendered) != len(inner):
+        return rendered
+    return [_apply_template(part, item) for part, item in zip(inner, rendered)]
+
+
+def _order_sets(guide: Any, rendered: Any) -> Any:
+    """
+    Return ``rendered`` with every list that pydantic rendered from a set 
sorted.
+
+    ``rendered`` is the json-mode rendering that is hashed, and ``guide`` the
+    python-mode dump of the same payload (see ``_check_guide``). They are 
walked
+    side by side: a dict or a sequence pairs up by position, because both dumps
+    keep the same order, and a list is sorted where the guide holds a set. A 
set's
+    members cannot pair up by position -- the dump builds a new set, which may
+    iterate in another order -- so the members' shared template says where any 
set
+    inside one of them sits (see ``_template``). Anything that does not line 
up is
+    kept exactly as rendered, so nothing that did not come from a set is 
reordered.
+
+    Raises ``TypeError`` where distinct dict keys render alike, such as ``1`` 
and
+    ``"1"``, or ``None`` and ``nan`` in a message dump: the guide then holds 
more keys
+    than the rendering, and hashing it would let two different payloads share a
+    digest. A dict whose keys are all strings cannot collide, so one that 
renders
+    fewer keys was reshaped by a serializer that applies only in JSON mode, 
and is
+    kept as rendered.
+    """
+    if isinstance(guide, enum.Enum):
+        # An Enum member renders as its value.
+        return _order_sets(guide.value, rendered)
+    if isinstance(guide, (set, frozenset)):
+        if not isinstance(rendered, list) or len(rendered) != len(guide):
+            return rendered
+        template = _member_template(guide)
+        if template is None:
+            return rendered
+        return sorted((_apply_template(template, item) for item in rendered), 
key=_json_order)

Review Comment:
   Pairing the dumps by shape doesn't prove the list at a set's position came 
from that set. A `when_used="json"` serializer that keeps keys and lengths but 
moves values gets past it. Take `tags: set[int]` and `rows: list[int]` with a 
serializer returning `{"tags": data["rows"], "rows": data["tags"]}`: `rows=[1, 
2]` and `rows=[2, 1]` get the same digest, both as a tool argument and in a 
`ToolReturnPart` on the model path, where main tells them apart. On the tool 
path a stale replay would also need the same call id, so the real exposure is 
the history. Sorting only where the guide's own JSON rendering, ordered the 
same way, matches the ordered result kept those two apart and kept a plain 
`set[str]` stable across seeds. In the other direction, when pairing gives up 
(members with different shapes such as `{"a", "b", frozenset({"x", "y"})}`, or 
a serializer that renames a set field), the set stays in hash-seed order. The 
step is still cached, and on retry it misses with a "diverged" warn
 ing, so refusing it there would give a more accurate log. Docs line 139 also 
needs a caveat: a JSON-only serializer on a `set` field gets its list re-sorted 
(`["z-first", "a"]` hashes as `["a", "z-first"]`).



##########
providers/common/ai/tests/unit/common/ai/durable/test_fingerprint.py:
##########
@@ -230,3 +267,1267 @@ def test_arg_order_does_not_matter(self):
         assert fingerprint_tool_call("t", {"a": 1, "b": 2}, "id1") == 
fingerprint_tool_call(
             "t", {"b": 2, "a": 1}, "id1"
         )
+
+
+class TestPydanticNativeValues:
+    """Values that are not JSON types but hash the same on every attempt must 
still fingerprint.
+
+    Tool arguments reach ``fingerprint_tool_call`` already coerced by 
pydantic, and
+    ``tool_choice`` accepts a dataclass while genuinely affecting the 
response, so it
+    cannot be stripped as transport-only. Since a step that cannot be 
fingerprinted is
+    no longer cached at all, refusing these values would stop an ordinary 
typed tool
+    from ever being cached.
+    """
+
+    def test_datetime_tool_argument_fingerprints(self):
+        when = datetime.datetime(2026, 1, 1, tzinfo=datetime.timezone.utc)
+
+        assert fingerprint_tool_call("t", {"when": when}, "id1") is not None
+
+    def test_decimal_tool_argument_fingerprints(self):
+        assert fingerprint_tool_call("t", {"amount": Decimal("10.5")}, "id1") 
is not None
+
+    def test_datetime_tool_argument_is_stable_and_distinguishing(self):
+        early = datetime.datetime(2026, 1, 1, tzinfo=datetime.timezone.utc)
+        late = datetime.datetime(2026, 6, 1, tzinfo=datetime.timezone.utc)
+
+        assert fingerprint_tool_call("t", {"when": early}, "id1") == 
fingerprint_tool_call(
+            "t", {"when": early}, "id1"
+        )
+        assert fingerprint_tool_call("t", {"when": early}, "id1") != 
fingerprint_tool_call(
+            "t", {"when": late}, "id1"
+        )
+
+    def test_tool_choice_dataclass_fingerprints(self):
+        fp = fingerprint_model_request(
+            "m",
+            make_messages(),
+            {"tool_choice": ToolOrOutput(function_tools=["my_tool"])},
+            ModelRequestParameters(),
+        )
+
+        assert fp is not None
+
+    def test_tool_choice_dataclass_still_affects_the_fingerprint(self):
+        one = fingerprint_model_request(
+            "m",
+            make_messages(),
+            {"tool_choice": ToolOrOutput(function_tools=["a"])},
+            ModelRequestParameters(),
+        )
+        other = fingerprint_model_request(
+            "m",
+            make_messages(),
+            {"tool_choice": ToolOrOutput(function_tools=["b"])},
+            ModelRequestParameters(),
+        )
+
+        assert one is not None
+        assert one != other
+
+    def test_value_pydantic_cannot_serialize_still_returns_none(self):
+        """Normalization must not turn a genuinely unserializable value into a 
hash."""
+        assert fingerprint_tool_call("t", {"v": object()}, "id1") is None
+
+    def test_plain_payload_digest_is_unchanged_by_normalization(self):
+        """Fingerprints recorded before normalization must still match, so 
cached entries survive."""
+        payload = {"model": "m", "args": {"b": [1, True, None, "x", 2.5]}, 
"settings": None}
+        pre_normalization = hashlib.sha256(json.dumps(payload, 
sort_keys=True).encode()).hexdigest()
+
+        assert _digest(_render(payload)) == pre_normalization
+
+    def test_dict_keys_render_as_a_json_mode_dump_renders_them(self):
+        by_day = {datetime.date(2026, 1, 1): 1.5, datetime.date(2026, 1, 2): 
2.5}
+
+        assert _render({"series": by_day}) == {"series": {"2026-01-01": 1.5, 
"2026-01-02": 2.5}}
+        assert _render({1: "a", "b": 2}) == {"1": "a", "b": 2}
+
+    def test_keys_that_collide_once_rendered_are_refused(self):
+        """``1`` and ``"1"`` render alike, so hashing either payload could 
replay the other."""
+        assert fingerprint_tool_call("t", {"d": {1: "a", "1": "b"}}, "id1") is 
None
+
+    def test_keys_that_collide_only_in_the_message_dump_are_refused(self):
+        """The message dump renders ``inf`` and ``nan`` keys alike, where 
pydantic's defaults do not."""
+        fp = fingerprint_model_request(
+            "m", _with_tool_return({math.inf: 100, math.nan: "unpriced"}), 
None, ModelRequestParameters()
+        )
+
+        assert fp is None
+
+    def test_bytes_that_are_not_utf8_render_as_base64(self):
+        """Decoding bytes as text would raise on an image and stop the tool 
from being cached."""
+        assert _render({"image": _PNG}) == {"image": 
base64.urlsafe_b64encode(_PNG).decode()}
+        assert fingerprint_tool_call("t", {"image": _PNG}, "id1") is not None
+
+    def 
test_bytes_dict_keys_render_as_base64_with_the_sets_under_them_ordered(self):
+        key = base64.urlsafe_b64encode(_PNG).decode()
+
+        assert _render({"d": {_PNG: _MANY}}) == {"d": {key: sorted(_MANY)}}
+
+
+def _json_mode_reference(model_identifier, messages, model_request_parameters, 
settings=None):
+    """The fingerprint main computes: pydantic's json-mode dump, hashed as is, 
with plain JSON settings.
+
+    For anything that is not a set, fingerprints must equal this, or entries 
stored
+    by an earlier version stop matching and the first retry after an upgrade 
re-runs
+    the whole agent.
+    """
+    dumped = ModelMessagesTypeAdapter.dump_python(messages, mode="json")
+    stripped = [
+        {
+            **{k: v for k, v in message.items() if k not in ("timestamp", 
"run_id", "conversation_id")},
+            "parts": [{k: v for k, v in part.items() if k != "timestamp"} for 
part in message["parts"]],
+        }
+        for message in dumped
+    ]
+    params = 
TypeAdapter(ModelRequestParameters).dump_python(model_request_parameters, 
mode="json")
+    payload = {"model": model_identifier, "messages": stripped, "settings": 
settings, "params": params}
+    return hashlib.sha256(json.dumps(payload, 
sort_keys=True).encode()).hexdigest()
+
+
+def _with_tool_return(content):
+    return [
+        ModelRequest(parts=[UserPromptPart(content="go")]),
+        ModelResponse(parts=[ToolCallPart(tool_name="t", args={}, 
tool_call_id="c1")]),
+        ModelRequest(parts=[ToolReturnPart(tool_name="t", content=content, 
tool_call_id="c1")]),
+    ]
+
+
+class TestMessageHistoryMatchesJsonModeDump:
+    """The message history must hash exactly as pydantic's json-mode dump 
renders it.
+
+    Each case here is something a hand-written renderer gets wrong: raw bytes 
(not
+    valid UTF-8), dict keys that are not strings, ``NaN``, tuples, and fields 
whose
+    serializer only applies in JSON mode, such as the set-typed request 
parameters
+    and, on pydantic-ai versions that have it, ``InstructionPart.id``.
+    """
+
+    @pytest.mark.parametrize(
+        "messages",
+        [
+            pytest.param(
+                [
+                    ModelRequest(
+                        parts=[
+                            UserPromptPart(content=["look", 
BinaryContent(data=_PNG, media_type="image/png")])
+                        ]
+                    )
+                ],
+                id="binary-content-in-prompt",
+            ),
+            pytest.param(_with_tool_return(_PNG), id="bytes-tool-return"),
+            pytest.param(
+                _with_tool_return({datetime.date(2026, 1, 1): 1.5, 
datetime.date(2026, 1, 2): 2.5}),
+                id="date-keyed-tool-return",
+            ),
+            pytest.param(_with_tool_return({1: "a", "b": 2}), 
id="mixed-key-tool-return"),
+            pytest.param(_with_tool_return({"x": math.nan}), 
id="nan-tool-return"),
+            pytest.param(
+                _with_tool_return({"when": datetime.datetime(2026, 1, 1)}), 
id="datetime-tool-return"
+            ),
+            
pytest.param([ModelRequest(parts=(UserPromptPart(content="hi"),))], 
id="tuple-parts"),
+            pytest.param(_with_tool_return({"pair": ("a", 1)}), 
id="tuple-tool-return"),
+        ],
+    )
+    def test_fingerprint_equals_the_json_mode_digest(self, messages):
+        fp = fingerprint_model_request("m", messages, None, 
ModelRequestParameters())
+
+        assert fp is not None
+        assert fp == _json_mode_reference("m", messages, 
ModelRequestParameters())
+
+    def test_request_parameters_hash_as_their_json_mode_dump(self):
+        """The request parameters dump differently in python and json mode: 
set-typed fields
+        such as ``revealed_tool_names`` on every supported version, and 
``InstructionPart.id``
+        on the versions that have it."""
+        seen = {}
+
+        class Spy(FunctionModel):
+            async def request(self, messages, model_settings, 
model_request_parameters):
+                seen.setdefault("messages", messages)
+                seen.setdefault("params", model_request_parameters)
+                return await super().request(messages, model_settings, 
model_request_parameters)
+
+        async def respond(messages, info):
+            return ModelResponse(parts=[TextPart(content="ok")])
+
+        Agent(Spy(respond), instructions="Be terse.").run_sync("hi")
+
+        fp = fingerprint_model_request("m", seen["messages"], None, 
seen["params"])
+
+        assert fp == _json_mode_reference("m", seen["messages"], 
seen["params"])
+
+    def test_tuple_parts_still_drop_part_timestamps(self):
+        """Message history passed as objects can hold its parts in a tuple."""
+        t1 = datetime.datetime(2026, 1, 1, tzinfo=datetime.timezone.utc)
+        t2 = datetime.datetime(2026, 1, 2, tzinfo=datetime.timezone.utc)
+
+        def history(timestamp):
+            return [ModelRequest(parts=(UserPromptPart(content="hi", 
timestamp=timestamp),))]
+
+        assert fingerprint_model_request(
+            "m", history(t1), None, ModelRequestParameters()
+        ) == fingerprint_model_request("m", history(t2), None, 
ModelRequestParameters())
+
+    def test_set_in_a_tool_return_hashes_as_its_ordered_members(self):
+        assert fingerprint_model_request(
+            "m", _with_tool_return({"tags": {"b", "c", "a"}}), None, 
ModelRequestParameters()
+        ) == fingerprint_model_request(
+            "m", _with_tool_return({"tags": ["a", "b", "c"]}), None, 
ModelRequestParameters()
+        )
+
+
+class TestSetMemberOrdering:
+    """Sets must hash in a fixed order rather than the interpreter's iteration 
order.
+
+    A ``set[str]`` iterates in an order derived from the process hash seed, 
and every
+    task attempt runs in a fresh process. Hashing that order would produce a 
digest
+    the next attempt never reproduces, so the step would re-run live on every 
retry
+    -- worse than declining to cache it, which at least costs nothing extra.
+    """
+
+    def test_set_hashes_as_its_ordered_members(self):
+        assert _render({"tags": {"beta", "alpha"}}) == {"tags": ["alpha", 
"beta"]}
+
+    def test_set_matches_the_equivalent_list(self):
+        assert _render({"tags": {"alpha", "beta", "gamma"}}) == 
_render({"tags": ["alpha", "beta", "gamma"]})
+
+    def test_frozenset_matches_set(self):
+        assert _render({"tags": frozenset({"a", "b"})}) == _render({"tags": 
{"b", "a"}})
+
+    def test_different_members_still_produce_different_digests(self):
+        assert _render({"tags": {"a", "b"}}) != _render({"tags": {"a", "c"}})
+
+    def test_set_nested_inside_a_list(self):
+        assert _render({"filters": [{"z", "y"}]}) == {"filters": [["y", "z"]]}
+
+    def test_set_inside_a_dataclass_field(self):
+        @dataclasses.dataclass
+        class Filter:
+            tags: set
+
+        assert _render(Filter(tags={"b", "a"})) == {"tags": ["a", "b"]}
+
+    def test_set_inside_a_basemodel_field(self):
+        class Filter(pydantic.BaseModel):
+            tags: set[str]
+
+        assert _render(Filter(tags={"b", "a"})) == {"tags": ["a", "b"]}
+
+    def test_digest_is_stable_across_process_hash_seeds(self):
+        """The real proof: two fresh interpreters must agree, as two attempts 
would.
+
+        In-process comparisons cannot catch a hash-seed dependency, since one 
process
+        has one seed. The subprocess loads the module by path so it does not 
pay for

Review Comment:
   Loading by path still imports Airflow: `exec_module` runs the module's 
`airflow.providers.common.ai.utils` imports, which pull in about 450 modules 
per interpreter. The four seed tests each start four interpreters and took 106 
s of the file's 115 s locally. One snippet that prints all the digests, run 
under two seeds through a shared helper, would test the same thing for a 
fraction of the CI time.
   
   Separately, two of the `_order_sets` guards can be removed while the suite 
still passes, so "reverting any of the 27 safeguards fails the suite" in the 
description doesn't hold for them. Dropping the `return rendered` on a dict 
length mismatch (`fingerprint.py:290-293`) lets a `when_used="json"` model 
serializer that re-adds an excluded field get zipped away: `note="DROP TABLE 
a"` and `note="DROP TABLE b"` next to a set field then share a digest. 
Returning any single template instead of `None` for mixed members at line 242 
also stays green. A test with that serializer asserting distinct digests, and a 
direct `_member_template(...) is None` assertion for mixed members (rather than 
a digest test that depends on the seed), would pin both. The length check in 
`_apply_template` (252-253) is in the same position.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to