From ad447eda3eb25698f05537961322dd3ebfcfd054 Mon Sep 17 00:00:00 2001 From: Pascal Tomecek Date: Mon, 21 Sep 2026 14:37:35 -0400 Subject: [PATCH] [cleanup] Avoid importing pandas at ccflow import time normalize_token registered its only pandas handler by dispatching on pd.Timestamp, which required `import pandas` at module load and so made every `import ccflow` pay for pandas whether or not the caller ever tokenizes one. That handler's body is identical to the datetime one apart from its tag, and pandas was only ever needed as a dispatch key, so identify Timestamp by name inside the existing datetime handler instead. Subclasses are still matched by walking the MRO, and plain datetimes cost one extra identity check. numpy is deliberately left alone: ccflow.exttypes.pydantic_numpy already imports it eagerly, so deferring its handlers would have no effect today. Token values are unchanged, so existing cache keys remain valid. Signed-off-by: Pascal Tomecek --- ccflow/utils/tokenize.py | 22 +++++++--------------- 1 file changed, 7 insertions(+), 15 deletions(-) diff --git a/ccflow/utils/tokenize.py b/ccflow/utils/tokenize.py index 91444488..26e8f700 100644 --- a/ccflow/utils/tokenize.py +++ b/ccflow/utils/tokenize.py @@ -159,6 +159,13 @@ def _normalize_date(obj): @normalize_token.register(datetime) def _normalize_datetime(obj): + cls = type(obj) + # pandas.Timestamp subclasses datetime and must not share a token with an equal plain datetime. + # Matching by name keeps `import ccflow` from paying for pandas just to register one handler. + # DataFrame/Series/Index have no structural handler and fall through to cloudpickle; adding them + # (as dask does) would require real lazy registration rather than this name check. + if cls is not datetime and any(c.__name__ == "Timestamp" and _module_in(c.__module__, ("pandas",)) for c in cls.__mro__): + return ("pd_timestamp", obj.isoformat()) return ("datetime", obj.isoformat()) @@ -371,22 +378,7 @@ def _normalize_np_scalar(obj): return ("np_scalar", str(type(obj).__name__), obj.item()) -def _register_pandas() -> None: - try: - import pandas as pd - except ImportError: # pragma: no cover - return - - # Only Timestamp has a structural handler; DataFrame/Series/Index fall through to the cloudpickle - # fallback. That works today but is fragile across pandas version upgrades — a follow-up could add - # structural handlers (matching dask) for stability and a perf win on large frames. - @normalize_token.register(pd.Timestamp) - def _normalize_pd_timestamp(obj): - return ("pd_timestamp", obj.isoformat()) - - _register_numpy() -_register_pandas() def tokenize(*args: Any, **kwargs: Any) -> str: