From 5945c99c6068cb5f558d304548e2297286b0e159 Mon Sep 17 00:00:00 2001 From: Vishal Bala Date: Thu, 3 Sep 2026 15:07:23 +0200 Subject: [PATCH 1/2] refactor(types): preserve wrapped signatures in deprecation decorators MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit deprecated_argument, deprecated_function and deprecated_class all returned a bare Callable and wrapped the target in an untyped *args/**kwargs closure. That erased the signature of every decorated function, so mypy silently stopped checking their call sites and every override relationship they took part in — 47 @deprecated_argument applications across redisvl/, 36 of them on the vectorizer embed/embed_many surface, plus 10 more between deprecated_function and deprecated_class. Both SearchIndex constructors are among them. Retype the three decorators with ParamSpec so each wrapper keeps the wrapped signature, and give deprecated_class a type-bound TypeVar so it is checked as a class decorator. This commit is the retyping alone; mypy now reports the pre-existing errors it had been hiding, which the next commit fixes. --- redisvl/utils/utils.py | 48 +++++++++++++++++++++++++++--------------- 1 file changed, 31 insertions(+), 17 deletions(-) diff --git a/redisvl/utils/utils.py b/redisvl/utils/utils.py index 85f74397c..804a1052f 100644 --- a/redisvl/utils/utils.py +++ b/redisvl/utils/utils.py @@ -8,7 +8,7 @@ from enum import Enum from functools import wraps from time import time -from typing import Any, Callable, Coroutine, Sequence, TypeVar +from typing import Any, Callable, Coroutine, ParamSpec, Sequence, TypeVar, cast from warnings import warn from pydantic import BaseModel @@ -16,6 +16,9 @@ from ulid import ULID T = TypeVar("T") +R = TypeVar("R") +P = ParamSpec("P") +C = TypeVar("C", bound=type) def create_ulid() -> str: @@ -75,7 +78,9 @@ def deserialize(data: str) -> Any: return json.loads(data) -def deprecated_argument(argument: str, replacement: str | None = None) -> Callable: +def deprecated_argument( + argument: str, replacement: str | None = None +) -> Callable[[Callable[P, R]], Callable[P, R]]: """ Decorator to warn if a deprecated argument is passed. @@ -97,13 +102,16 @@ def test_method(cls, old_arg=None, new_arg=None): if replacement: message += f" Use {replacement} instead." - def decorator(func): - # Check if the function is a classmethod or staticmethod + def decorator(func: Callable[P, R]) -> Callable[P, R]: + # The documented usage puts this decorator inside + # @classmethod/@staticmethod, so this branch only covers the reversed + # order; a classmethod object is not a Callable to the type system, + # hence the casts. if isinstance(func, (classmethod, staticmethod)): - underlying = func.__func__ + underlying = cast(Callable[P, R], func.__func__) @wraps(underlying) - def inner_wrapped(*args, **kwargs): + def inner_wrapped(*args: P.args, **kwargs: P.kwargs) -> R: if argument in kwargs: warn(message, DeprecationWarning, stacklevel=2) else: @@ -114,13 +122,13 @@ def inner_wrapped(*args, **kwargs): return underlying(*args, **kwargs) if isinstance(func, classmethod): - return classmethod(inner_wrapped) + return cast(Callable[P, R], classmethod(inner_wrapped)) else: - return staticmethod(inner_wrapped) + return cast(Callable[P, R], staticmethod(inner_wrapped)) else: @wraps(func) - def inner_normal(*args, **kwargs): + def inner_normal(*args: P.args, **kwargs: P.kwargs) -> R: if argument in kwargs: warn(message, DeprecationWarning, stacklevel=2) else: @@ -145,7 +153,9 @@ def assert_no_warnings(): yield -def deprecated_function(name: str | None = None, replacement: str | None = None): +def deprecated_function( + name: str | None = None, replacement: str | None = None +) -> Callable[[Callable[P, R]], Callable[P, R]]: """ Decorator to mark a function as deprecated. @@ -153,7 +163,7 @@ def deprecated_function(name: str | None = None, replacement: str | None = None) warning. """ - def decorator(func): + def decorator(func: Callable[P, R]) -> Callable[P, R]: fn_name = name or func.__name__ warning_message = ( f"Function {fn_name} is deprecated and will be " @@ -163,7 +173,7 @@ def decorator(func): warning_message += replacement @wraps(func) - def wrapper(*args, **kwargs): + def wrapper(*args: P.args, **kwargs: P.kwargs) -> R: warn(warning_message, category=DeprecationWarning, stacklevel=3) return func(*args, **kwargs) @@ -172,7 +182,9 @@ def wrapper(*args, **kwargs): return decorator -def deprecated_class(name: str | None = None, replacement: str | None = None): +def deprecated_class( + name: str | None = None, replacement: str | None = None +) -> Callable[[C], C]: """ Decorator to mark a class as deprecated. @@ -190,7 +202,7 @@ class OldClass: pass """ - def decorator(cls): + def decorator(cls: C) -> C: class_name = name or cls.__name__ warning_message = ( f"Class {class_name} is deprecated and will be " @@ -199,10 +211,12 @@ def decorator(cls): if replacement: warning_message += replacement - original_init = cls.__init__ + # getattr/setattr rather than attribute access: `cls` is typed as a + # class object, so mypy resolves `cls.__init__` to the metaclass slot. + original_init = getattr(cls, "__init__") @wraps(original_init) - def new_init(self, *args, **kwargs): + def new_init(self: Any, *args: Any, **kwargs: Any) -> None: # Emit only once per instance. When a deprecated subclass wraps a # deprecated parent, both __init__ wrappers run via super().__init__; # the sentinel keeps that to a single warning. @@ -214,7 +228,7 @@ def new_init(self, *args, **kwargs): pass original_init(self, *args, **kwargs) - cls.__init__ = new_init + setattr(cls, "__init__", new_init) return cls return decorator From 323c17b082aed0f7c27694c75ec727359ace557b Mon Sep 17 00:00:00 2001 From: Vishal Bala Date: Thu, 3 Sep 2026 15:14:32 +0200 Subject: [PATCH 2/2] fix(vectorize): repair the override contracts mypy had stopped checking MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit With the deprecation decorators now signature-preserving, mypy checks the vectorizer override hierarchy again and reports 52 pre-existing errors. Two genuine defects were behind them. BaseVectorizer._embed declared `text` as its first parameter and `content` as its second, but embed() dispatches positionally — `self._embed(content, ...)`. Every provider happens to name its first parameter `content`, so this worked; any implementation that followed the base signature instead would have silently received the content in `text`. The four provider hooks (_embed, _embed_many, _aembed, _aembed_many) are internal extension points: the public methods already resolve the deprecated `text`/`texts` alias before dispatching, and nothing in the codebase ever calls them with it. Give them the signature they are actually called with — content first and required, no alias — and drop the now-dead alias parameters and @deprecated_argument decorators from the five providers that carried them. embed_many/aembed_many dispatched the batch hooks by keyword while embed/aembed dispatched positionally; make all four positional so one contract covers the set. One behavioural edge follows from dropping the aliases: `_embed_many([])` used to resolve `[] or None` to None and raise TypeError, and now returns []. The public embed_many/aembed_many short-circuit empty input before dispatching, so this is unreachable from the public API. The four deprecated *TextVectorizer aliases each re-implemented embed and embed_many purely to merge `text` into `content` and warn, which BaseVectorizer already does identically. In doing so they narrowed the contract: they dropped preprocess/batch_size/as_buffer/skip_cache from the positional signature, so `BedrockTextVectorizer().embed("hi", None, fn)` raised TypeError where the parent works, and most of them annotated a `list[float]` return that is bytes whenever as_buffer=True. Delete the redundant overrides; the aliases inherit correct, honestly typed methods and still warn on both the class and the `text` kwarg. One of the 52 errors was outside the vectorizer hierarchy: BaseCache.redis_kwargs was an inferred heterogeneous dict, so BaseCache cast on every read of it and SemanticCache passed its values into SearchIndex unchecked. Typing it as a TypedDict removes the four casts; semantic.py needs no change and now type-checks as it stands. --- redisvl/extensions/cache/base.py | 20 +++++--- redisvl/utils/vectorize/base.py | 35 +++++-------- redisvl/utils/vectorize/text/azureopenai.py | 25 ++-------- redisvl/utils/vectorize/text/bedrock.py | 31 ++---------- redisvl/utils/vectorize/text/cohere.py | 14 ++---- redisvl/utils/vectorize/text/custom.py | 54 ++------------------- redisvl/utils/vectorize/text/huggingface.py | 12 +---- redisvl/utils/vectorize/text/mistral.py | 25 ++-------- redisvl/utils/vectorize/text/openai.py | 25 ++-------- redisvl/utils/vectorize/text/vertexai.py | 32 +++--------- redisvl/utils/vectorize/text/voyageai.py | 54 ++------------------- 11 files changed, 68 insertions(+), 259 deletions(-) diff --git a/redisvl/extensions/cache/base.py b/redisvl/extensions/cache/base.py index 8ff3566db..6c15943cd 100644 --- a/redisvl/extensions/cache/base.py +++ b/redisvl/extensions/cache/base.py @@ -5,7 +5,7 @@ """ from collections.abc import Mapping -from typing import Any, cast +from typing import Any, TypedDict from redis import Redis # For backwards compatibility in type checking from redis.cluster import RedisCluster @@ -14,6 +14,14 @@ from redisvl.types import AsyncRedisClient, SyncRedisClient +class RedisConnectionKwargs(TypedDict): + """The connection details a cache keeps so it can build clients lazily.""" + + redis_client: SyncRedisClient | None + redis_url: str + connection_kwargs: dict[str, Any] + + class BaseCache: """Base abstract cache interface for all RedisVL caches. @@ -50,7 +58,7 @@ def __init__( self._ttl: int | None = None self.set_ttl(ttl) - self.redis_kwargs = { + self.redis_kwargs: RedisConnectionKwargs = { "redis_client": redis_client, "redis_url": redis_url, "connection_kwargs": connection_kwargs, @@ -114,8 +122,8 @@ def _get_redis_client(self) -> SyncRedisClient: """ if self._redis_client is None: # Create new Redis client - url = cast(str | None, self.redis_kwargs["redis_url"]) - kwargs = cast(dict[str, Any], self.redis_kwargs["connection_kwargs"]) + url = self.redis_kwargs["redis_url"] + kwargs = self.redis_kwargs["connection_kwargs"] self._redis_client = RedisConnectionFactory.get_redis_connection( redis_url=url, **kwargs, @@ -136,8 +144,8 @@ async def _get_async_redis_client(self) -> AsyncRedisClient: client ) else: - url = cast(str | None, self.redis_kwargs["redis_url"]) - kwargs = cast(dict[str, Any], self.redis_kwargs["connection_kwargs"]) + url = self.redis_kwargs["redis_url"] + kwargs = self.redis_kwargs["connection_kwargs"] self._async_redis_client = ( RedisConnectionFactory.get_async_redis_connection( redis_url=url, **kwargs diff --git a/redisvl/utils/vectorize/base.py b/redisvl/utils/vectorize/base.py index ca4d72bec..c3652cd6f 100644 --- a/redisvl/utils/vectorize/base.py +++ b/redisvl/utils/vectorize/base.py @@ -184,7 +184,7 @@ def embed_many( if cache_misses: cache_metadata = kwargs.pop("metadata", {}) new_embeddings = self._embed_many( - contents=cache_misses, batch_size=batch_size, **kwargs + cache_misses, batch_size=batch_size, **kwargs ) # Store new embeddings in cache @@ -317,7 +317,7 @@ async def aembed_many( if cache_misses: cache_metadata = kwargs.pop("metadata", {}) new_embeddings = await self._aembed_many( - contents=cache_misses, batch_size=batch_size, **kwargs + cache_misses, batch_size=batch_size, **kwargs ) # Store new embeddings in cache @@ -332,45 +332,36 @@ async def aembed_many( # Process and return results return [self._process_embedding(emb, as_buffer, self.dtype) for emb in results] - @deprecated_argument("text", "content") - def _embed(self, text: Any = "", content: Any = "", **kwargs) -> list[float]: + # The four hooks below are the provider extension points. Every caller — + # the public embed/embed_many/aembed/aembed_many above, and each provider's + # own _set_model_dims probe — passes the content as the first positional + # argument. None passes the deprecated `text`/`texts` alias, because the + # public methods resolve it before dispatching here. + def _embed(self, content: Any, **kwargs) -> list[float]: """Generate a vector embedding for a single item.""" raise NotImplementedError - @deprecated_argument("texts", "contents") def _embed_many( - self, - contents: list[Any] | None = None, - texts: list[Any] | None = None, - batch_size: int = 10, - **kwargs, + self, contents: list[Any], batch_size: int = 10, **kwargs ) -> list[list[float]]: """Generate vector embeddings for a batch of items.""" raise NotImplementedError - @deprecated_argument("text", "content") - async def _aembed(self, content: Any = "", text: Any = "", **kwargs) -> list[float]: + async def _aembed(self, content: Any, **kwargs) -> list[float]: """Asynchronously generate a vector embedding for a single item.""" logger.warning( "This vectorizer has no async embed method. Falling back to sync." ) - return self._embed(content=content or text, **kwargs) + return self._embed(content, **kwargs) - @deprecated_argument("texts", "contents") async def _aembed_many( - self, - contents: list[Any] | None = None, - texts: list[Any] | None = None, - batch_size: int = 10, - **kwargs, + self, contents: list[Any], batch_size: int = 10, **kwargs ) -> list[list[float]]: """Asynchronously generate vector embeddings for a batch of items.""" logger.warning( "This vectorizer has no async embed_many method. Falling back to sync." ) - return self._embed_many( - contents=contents or texts, batch_size=batch_size, **kwargs - ) + return self._embed_many(contents, batch_size=batch_size, **kwargs) def _get_from_cache_batch( self, contents: list[Any], skip_cache: bool diff --git a/redisvl/utils/vectorize/text/azureopenai.py b/redisvl/utils/vectorize/text/azureopenai.py index 9640d1b8a..46ef86927 100644 --- a/redisvl/utils/vectorize/text/azureopenai.py +++ b/redisvl/utils/vectorize/text/azureopenai.py @@ -8,7 +8,6 @@ if TYPE_CHECKING: from redisvl.extensions.cache.embeddings.embeddings import EmbeddingsCache -from redisvl.utils.utils import deprecated_argument from redisvl.utils.vectorize.base import BaseVectorizer # ignore that openai isn't imported @@ -206,30 +205,27 @@ def _set_model_dims(self) -> int: # fall back (TODO get more specific) raise ValueError(f"Error setting embedding model dimensions: {str(e)}") - @deprecated_argument("text", "content") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), retry=retry_if_not_exception_type(TypeError), reraise=True, ) - def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: + def _embed(self, content: str, **kwargs) -> list[float]: """ Generate a vector embedding for a single text using the AzureOpenAI API. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional parameters to pass to the AzureOpenAI API Returns: List[float]: Vector embedding as a list of floats Raises: - TypeError: If text is not a string + TypeError: If content is not a string ValueError: If embedding fails """ - content = content or text if not isinstance(content, str): raise TypeError("Must pass in a str value to embed.") @@ -241,7 +237,6 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: except Exception as e: raise ValueError(f"Embedding text failed: {e}") - @deprecated_argument("texts", "contents") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), @@ -250,8 +245,7 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: ) def _embed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float]]: @@ -260,7 +254,6 @@ def _embed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each API call **kwargs: Additional parameters to pass to the AzureOpenAI API @@ -271,7 +264,6 @@ def _embed_many( TypeError: If contents is not a list of strings ValueError: If embedding fails """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of str values to embed.") if contents and not isinstance(contents[0], str): @@ -288,20 +280,18 @@ def _embed_many( except Exception as e: raise ValueError(f"Embedding texts failed: {e}") - @deprecated_argument("text", "content") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), retry=retry_if_not_exception_type(TypeError), reraise=True, ) - async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[float]: + async def _aembed(self, content: str, **kwargs) -> list[float]: """ Asynchronously generate a vector embedding for a single text using the AzureOpenAI API. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional parameters to pass to the AzureOpenAI API Returns: @@ -311,7 +301,6 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo TypeError: If content is not a string ValueError: If embedding fails """ - content = content or text if not isinstance(content, str): raise TypeError("Must pass in a str value to embed.") @@ -323,7 +312,6 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo except Exception as e: raise ValueError(f"Embedding text failed: {e}") - @deprecated_argument("texts", "contents") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), @@ -332,8 +320,7 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo ) async def _aembed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float]]: @@ -342,7 +329,6 @@ async def _aembed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each API call **kwargs: Additional parameters to pass to the AzureOpenAI API @@ -353,7 +339,6 @@ async def _aembed_many( TypeError: If contents is not a list of strings ValueError: If embedding fails """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of str values to embed.") if contents and not isinstance(contents[0], str): diff --git a/redisvl/utils/vectorize/text/bedrock.py b/redisvl/utils/vectorize/text/bedrock.py index ced3cc50b..51490bbde 100644 --- a/redisvl/utils/vectorize/text/bedrock.py +++ b/redisvl/utils/vectorize/text/bedrock.py @@ -1,6 +1,4 @@ -from typing import Any - -from redisvl.utils.utils import deprecated_argument, deprecated_class +from redisvl.utils.utils import deprecated_class from redisvl.utils.vectorize.bedrock import BedrockVectorizer @@ -8,27 +6,8 @@ name="BedrockTextVectorizer", replacement="Use BedrockVectorizer instead." ) class BedrockTextVectorizer(BedrockVectorizer): - """A backwards-compatible alias for BedrockVectorizer.""" - - @deprecated_argument("text", "content") - def embed(self, content: Any = "", text: Any = "", **kwargs) -> list[float] | bytes: - """Generate a vector embedding for a single input using the AWS Bedrock API. - - Deprecated: Use `BedrockVectorizer.embed` instead. - """ - content = content or text - return super().embed(content=content, **kwargs) - - @deprecated_argument("texts", "contents") - def embed_many( - self, - contents: list[Any] | None = None, - texts: list[Any] | None = None, - **kwargs, - ) -> list[list[float]]: - """Generate vector embeddings for a batch of inputs using the AWS Bedrock API. + """A backwards-compatible alias for BedrockVectorizer. - Deprecated: Use `BedrockVectorizer.embed_many` instead. - """ - contents = contents or texts - return super().embed_many(contents=contents, **kwargs) + The `text`/`texts` keyword arguments still work, and still warn, via + BaseVectorizer. + """ diff --git a/redisvl/utils/vectorize/text/cohere.py b/redisvl/utils/vectorize/text/cohere.py index d5b2270ac..bdedbc74c 100644 --- a/redisvl/utils/vectorize/text/cohere.py +++ b/redisvl/utils/vectorize/text/cohere.py @@ -9,7 +9,6 @@ if TYPE_CHECKING: from redisvl.extensions.cache.embeddings.embeddings import EmbeddingsCache -from redisvl.utils.utils import deprecated_argument from redisvl.utils.vectorize.base import BaseVectorizer # ignore that cohere isn't imported @@ -205,14 +204,12 @@ def _validate_input_type(self, input_type) -> None: "See https://docs.cohere.com/reference/embed." ) - @deprecated_argument("text", "content") - def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float | int]: + def _embed(self, content: str, **kwargs) -> list[float | int]: """ Generate a vector embedding for a single text using the Cohere API. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional parameters to pass to the Cohere API, must include 'input_type' @@ -224,10 +221,9 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float | in - For dtype="int8" or "uint8": Returns a list of integers Raises: - TypeError: If text is not a string or input_type is not provided + TypeError: If content is not a string or input_type is not provided ValueError: If embedding fails """ - content = content or text if not isinstance(content, str): raise TypeError("Must pass in a str value to embed.") @@ -267,7 +263,6 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float | in except Exception as e: raise ValueError(f"Embedding text failed: {e}") - @deprecated_argument("texts", "contents") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), @@ -275,8 +270,7 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float | in ) def _embed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float | int]]: @@ -285,7 +279,6 @@ def _embed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each API call **kwargs: Additional parameters to pass to the Cohere API, must include 'input_type' @@ -297,7 +290,6 @@ def _embed_many( TypeError: If contents is not a list of strings or input_type is not provided ValueError: If embedding fails """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of str values to embed.") if contents and not isinstance(contents[0], str): diff --git a/redisvl/utils/vectorize/text/custom.py b/redisvl/utils/vectorize/text/custom.py index 734a6d7ae..743b21352 100644 --- a/redisvl/utils/vectorize/text/custom.py +++ b/redisvl/utils/vectorize/text/custom.py @@ -1,6 +1,4 @@ -from typing import Any - -from redisvl.utils.utils import deprecated_argument, deprecated_class +from redisvl.utils.utils import deprecated_class from redisvl.utils.vectorize.custom import CustomVectorizer @@ -8,50 +6,8 @@ name="CustomTextVectorizer", replacement="Use CustomVectorizer instead." ) class CustomTextVectorizer(CustomVectorizer): - """A backwards-compatible alias for CustomVectorizer.""" - - @deprecated_argument("text", "content") - def embed(self, content: Any = "", text: Any = "", **kwargs) -> list[float]: - """Generate a vector embedding for a single input using the custom function. - - Deprecated: Use `CustomVectorizer.embed` instead. - """ - content = content or text - return super().embed(content=content, **kwargs) - - @deprecated_argument("texts", "contents") - def embed_many( - self, - contents: list[Any] | None = None, - texts: list[Any] | None = None, - **kwargs, - ) -> list[list[float]]: - """Generate vector embeddings for a batch of inputs using the custom function. - - Deprecated: Use `CustomVectorizer.embed_many` instead. - """ - contents = contents or texts - return super().embed_many(contents=contents, **kwargs) - - @deprecated_argument("text", "content") - async def aembed(self, content: Any = "", text: Any = "", **kwargs) -> list[float]: - """Asynchronously generate a vector embedding for a single input using the custom function. - - Deprecated: Use `CustomVectorizer.aembed` instead. - """ - content = content or text - return await super().aembed(content=content, **kwargs) - - @deprecated_argument("texts", "contents") - async def aembed_many( - self, - contents: list[Any] | None = None, - texts: list[Any] | None = None, - **kwargs, - ) -> list[list[float]]: - """Asynchronously generate vector embeddings for a batch of inputs using the custom function. + """A backwards-compatible alias for CustomVectorizer. - Deprecated: Use `CustomVectorizer.aembed_many` instead. - """ - contents = contents or texts - return await super().aembed_many(contents=contents, **kwargs) + The `text`/`texts` keyword arguments still work, and still warn, via + BaseVectorizer. + """ diff --git a/redisvl/utils/vectorize/text/huggingface.py b/redisvl/utils/vectorize/text/huggingface.py index 5fb6e8bc4..0ff1b6f8c 100644 --- a/redisvl/utils/vectorize/text/huggingface.py +++ b/redisvl/utils/vectorize/text/huggingface.py @@ -5,7 +5,6 @@ if TYPE_CHECKING: from redisvl.extensions.cache.embeddings.embeddings import EmbeddingsCache -from redisvl.utils.utils import deprecated_argument from redisvl.utils.vectorize.base import BaseVectorizer @@ -125,19 +124,16 @@ def _set_model_dims(self): raise ValueError(f"Error setting embedding model dimensions: {str(e)}") return len(embedding) - @deprecated_argument("text", "content") - def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: + def _embed(self, content: str, **kwargs) -> list[float]: """Generate a vector embedding for a single text using the Hugging Face model. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional model-specific parameters Returns: List[float]: Vector embedding as a list of floats """ - content = content or text if "show_progress_bar" not in kwargs: # disable annoying tqdm by default kwargs["show_progress_bar"] = False @@ -145,11 +141,9 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: embedding = self._client.encode([content], **kwargs)[0] return embedding.tolist() - @deprecated_argument("texts", "contents") def _embed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float]]: @@ -157,14 +151,12 @@ def _embed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each batch **kwargs: Additional model-specific parameters Returns: List[List[float]]: List of vector embeddings as lists of floats """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of values to embed.") if "show_progress_bar" not in kwargs: diff --git a/redisvl/utils/vectorize/text/mistral.py b/redisvl/utils/vectorize/text/mistral.py index b1ba36da9..403097b27 100644 --- a/redisvl/utils/vectorize/text/mistral.py +++ b/redisvl/utils/vectorize/text/mistral.py @@ -8,7 +8,6 @@ if TYPE_CHECKING: from redisvl.extensions.cache.embeddings.embeddings import EmbeddingsCache -from redisvl.utils.utils import deprecated_argument from redisvl.utils.vectorize.base import BaseVectorizer # ignore that mistralai isn't imported @@ -163,19 +162,17 @@ def _set_model_dims(self) -> int: # fall back (TODO get more specific) raise ValueError(f"Error setting embedding model dimensions: {str(e)}") - @deprecated_argument("text", "content") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), retry=retry_if_not_exception_type(TypeError), ) - def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: + def _embed(self, content: str, **kwargs) -> list[float]: """ Generate a vector embedding for a single text using the MistralAI API. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional parameters to pass to the MistralAI API Returns: @@ -185,7 +182,6 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: TypeError: If content is not a string ValueError: If embedding fails """ - content = content or text if not isinstance(content, str): raise TypeError("Must pass in a str value to embed.") @@ -197,7 +193,6 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: except Exception as e: raise ValueError(f"Embedding text failed: {e}") - @deprecated_argument("texts", "contents") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), @@ -205,8 +200,7 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: ) def _embed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float]]: @@ -215,7 +209,6 @@ def _embed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each API call **kwargs: Additional parameters to pass to the MistralAI API @@ -226,7 +219,6 @@ def _embed_many( TypeError: If contents is not a list of strings ValueError: If embedding fails """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of str values to embed.") if contents and not isinstance(contents[0], str): @@ -243,19 +235,17 @@ def _embed_many( except Exception as e: raise ValueError(f"Embedding texts failed: {e}") - @deprecated_argument("text", "content") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), retry=retry_if_not_exception_type(TypeError), ) - async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[float]: + async def _aembed(self, content: str, **kwargs) -> list[float]: """ Asynchronously generate a vector embedding for a single text using the MistralAI API. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional parameters to pass to the MistralAI API Returns: @@ -265,7 +255,6 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo TypeError: If `content` is not a string ValueError: If embedding fails """ - content = content or text if not isinstance(content, str): raise TypeError("Must pass in a str value to embed.") @@ -277,7 +266,6 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo except Exception as e: raise ValueError(f"Embedding content failed: {e}") - @deprecated_argument("texts", "contents") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), @@ -285,8 +273,7 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo ) async def _aembed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float]]: @@ -295,7 +282,6 @@ async def _aembed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each API call **kwargs: Additional parameters to pass to the MistralAI API @@ -303,10 +289,9 @@ async def _aembed_many( List[List[float]]: List of vector embeddings as lists of floats Raises: - TypeError: If texts is not a list of strings + TypeError: If contents is not a list of strings ValueError: If embedding fails """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of str values to embed.") if contents and not isinstance(contents[0], str): diff --git a/redisvl/utils/vectorize/text/openai.py b/redisvl/utils/vectorize/text/openai.py index 5102a8f82..cecb65cbe 100644 --- a/redisvl/utils/vectorize/text/openai.py +++ b/redisvl/utils/vectorize/text/openai.py @@ -8,7 +8,6 @@ if TYPE_CHECKING: from redisvl.extensions.cache.embeddings.embeddings import EmbeddingsCache -from redisvl.utils.utils import deprecated_argument from redisvl.utils.vectorize.base import BaseVectorizer # ignore that openai isn't imported @@ -162,28 +161,25 @@ def _set_model_dims(self) -> int: # fall back (TODO get more specific) raise ValueError(f"Error setting embedding model dimensions: {str(e)}") - @deprecated_argument("text", "content") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), retry=retry_if_not_exception_type(TypeError), ) - def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: + def _embed(self, content: str, **kwargs) -> list[float]: """Generate a vector embedding for a single text using the OpenAI API. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional parameters to pass to the OpenAI API Returns: List[float]: Vector embedding as a list of floats Raises: - TypeError: If text is not a string + TypeError: If content is not a string ValueError: If embedding fails """ - content = content or text if not isinstance(content, str): raise TypeError("Must pass in a str value to embed.") @@ -195,7 +191,6 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: except Exception as e: raise ValueError(f"Embedding text failed: {e}") - @deprecated_argument("texts", "contents") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), @@ -203,8 +198,7 @@ def _embed(self, content: str = "", text: str = "", **kwargs) -> list[float]: ) def _embed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float]]: @@ -212,7 +206,6 @@ def _embed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each API call **kwargs: Additional parameters to pass to the OpenAI API @@ -223,7 +216,6 @@ def _embed_many( TypeError: If contents is not a list of strings ValueError: If embedding fails """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of str values to embed.") if contents and not isinstance(contents[0], str): @@ -240,18 +232,16 @@ def _embed_many( raise ValueError(f"Embedding texts failed: {e}") return embeddings - @deprecated_argument("text", "content") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), retry=retry_if_not_exception_type(TypeError), ) - async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[float]: + async def _aembed(self, content: str, **kwargs) -> list[float]: """Asynchronously generate a vector embedding for a single text using the OpenAI API. Args: content: Text to embed - text: Text to embed (deprecated - use `content` instead) **kwargs: Additional parameters to pass to the OpenAI API Returns: @@ -261,7 +251,6 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo TypeError: If content is not a string ValueError: If embedding fails """ - content = content or text if not isinstance(content, str): raise TypeError("Must pass in a str value to embed.") @@ -273,7 +262,6 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo except Exception as e: raise ValueError(f"Embedding text failed: {e}") - @deprecated_argument("texts", "contents") @retry( wait=wait_random_exponential(min=1, max=60), stop=stop_after_attempt(6), @@ -281,8 +269,7 @@ async def _aembed(self, content: str = "", text: str = "", **kwargs) -> list[flo ) async def _aembed_many( self, - contents: list[str] | None = None, - texts: list[str] | None = None, + contents: list[str], batch_size: int = 10, **kwargs, ) -> list[list[float]]: @@ -290,7 +277,6 @@ async def _aembed_many( Args: contents: List of texts to embed - texts: List of texts to embed (deprecated - use `contents` instead) batch_size: Number of texts to process in each API call **kwargs: Additional parameters to pass to the OpenAI API @@ -301,7 +287,6 @@ async def _aembed_many( TypeError: If contents is not a list of strings ValueError: If embedding fails """ - contents = contents or texts if not isinstance(contents, list): raise TypeError("Must pass in a list of str values to embed.") if contents and not isinstance(contents[0], str): diff --git a/redisvl/utils/vectorize/text/vertexai.py b/redisvl/utils/vectorize/text/vertexai.py index 8e1a77f50..3d2620301 100644 --- a/redisvl/utils/vectorize/text/vertexai.py +++ b/redisvl/utils/vectorize/text/vertexai.py @@ -1,6 +1,4 @@ -from typing import Any - -from redisvl.utils.utils import deprecated_argument, deprecated_class +from redisvl.utils.utils import deprecated_class from redisvl.utils.vectorize.vertexai import VertexAIVectorizer @@ -9,27 +7,9 @@ replacement="Use GoogleGenAIVectorizer instead.", ) class VertexAITextVectorizer(VertexAIVectorizer): - """A backwards-compatible alias for VertexAIVectorizer.""" - - @deprecated_argument("text", "content") - def embed(self, content: str = "", text: Any = "", **kwargs) -> list[float]: - """Generate a vector embedding for a single input using the VertexAI API. - - Deprecated: Use `VertexAIVectorizer.embed` instead. - """ - content = content or text - return super().embed(content=content, **kwargs) - - @deprecated_argument("texts", "contents") - def embed_many( - self, - contents: list[str] | None = None, - texts: list[Any] | None = None, - **kwargs, - ) -> list[list[float]]: - """Generate vector embeddings for a batch of inputs using the VertexAI API. + """A backwards-compatible alias for VertexAIVectorizer, which is itself + deprecated: use GoogleGenAIVectorizer instead. - Deprecated: Use `VertexAIVectorizer.embed_many` instead. - """ - contents = contents or texts - return super().embed_many(contents=contents, **kwargs) + The `text`/`texts` keyword arguments still work, and still warn, via + BaseVectorizer. + """ diff --git a/redisvl/utils/vectorize/text/voyageai.py b/redisvl/utils/vectorize/text/voyageai.py index 45db4c6d0..c7f067a63 100644 --- a/redisvl/utils/vectorize/text/voyageai.py +++ b/redisvl/utils/vectorize/text/voyageai.py @@ -1,6 +1,4 @@ -from typing import Any - -from redisvl.utils.utils import deprecated_argument, deprecated_class +from redisvl.utils.utils import deprecated_class from redisvl.utils.vectorize.voyageai import VoyageAIVectorizer @@ -8,50 +6,8 @@ name="VoyageAITextVectorizer", replacement="Use VoyageAIVectorizer instead." ) class VoyageAITextVectorizer(VoyageAIVectorizer): - """A backwards-compatible alias for VoyageAIVectorizer.""" - - @deprecated_argument("text", "content") - def embed(self, content: Any = "", text: Any = "", **kwargs) -> list[float]: - """Generate a vector embedding for a single text using the VoyageAI API. - - Deprecated: Use `VoyageAIVectorizer.embed` instead. - """ - content = content or text - return super().embed(content=content, **kwargs) - - @deprecated_argument("texts", "contents") - def embed_many( - self, - contents: list[Any] | None = None, - texts: list[Any] | None = None, - **kwargs, - ) -> list[list[float]]: - """Generate vector embeddings for a batch of texts using the VoyageAI API. - - Deprecated: Use `VoyageAIVectorizer.embed_many` instead. - """ - contents = contents or texts - return super().embed_many(contents=contents, **kwargs) - - @deprecated_argument("text", "content") - async def aembed(self, content: Any = "", text: Any = "", **kwargs) -> list[float]: - """Asynchronously generate a vector embedding for a single text using the VoyageAI API. - - Deprecated: Use `VoyageAIVectorizer.aembed` instead. - """ - content = content or text - return await super().aembed(content=content, **kwargs) - - @deprecated_argument("texts", "contents") - async def aembed_many( - self, - contents: list[Any] | None = None, - texts: list[Any] | None = None, - **kwargs, - ) -> list[list[float]]: - """Asynchronously generate vector embeddings for a batch of texts using the VoyageAI API. + """A backwards-compatible alias for VoyageAIVectorizer. - Deprecated: Use `VoyageAIVectorizer.aembed_many` instead. - """ - contents = contents or texts - return await super().aembed_many(contents=contents, **kwargs) + The `text`/`texts` keyword arguments still work, and still warn, via + BaseVectorizer. + """