Skip to content

ontocast.tool.vector_store.embedding

Embedding provider abstraction for vector store workflows.

Attributes

logger = logging.getLogger(__name__) module-attribute

Classes

EmbeddingTool

Bases: Tool

Base embedding tool with provider-specific implementations.

Source code in ontocast/tool/vector_store/embedding.py
class EmbeddingTool(Tool):
    """Base embedding tool with provider-specific implementations."""

    config: EmbeddingConfig = Field(default_factory=EmbeddingConfig)

    @abc.abstractmethod
    def _embed_raw(self, texts: list[str]) -> list[list[float]]:
        """Return vectors for all given texts, prefixes already applied."""

    def embed(self, texts: list[str]) -> list[list[float]]:
        """Return vectors for all given texts as *documents*.

        Serialisation, where it is needed, belongs to whatever owns the model —
        the shared encoder for local checkpoints, nothing for remote providers.
        """
        if not texts:
            return []
        return self._embed_raw(self._apply(self.config.document_prefix, texts))

    def embed_query(self, texts: list[str]) -> list[list[float]]:
        """Return vectors for all given texts as *queries*.

        Asymmetric retrieval models are trained with distinct query and document
        instructions and lose accuracy when both sides are encoded identically. With
        empty prefixes — the default, suiting a symmetric paraphrase model — this is
        exactly :meth:`embed`.
        """
        if not texts:
            return []
        return self._embed_raw(self._apply(self.config.query_prefix, texts))

    @staticmethod
    def _apply(prefix: str, texts: list[str]) -> list[str]:
        return texts if not prefix else [f"{prefix}{text}" for text in texts]

    @property
    def sequence_limit(self) -> int | None:
        """Tokens this provider accepts before it silently truncates, if known.

        Truncation is the failure mode with no symptom: the provider returns a
        vector of the right shape for a prefix of the text, and the caller cannot
        tell that the tail was dropped. Exposing the limit is what lets a caller
        report it instead of discovering it as unexplained recall loss.

        Returns:
            int | None: The limit, or None where the provider does not state one.
        """
        return None

    def token_lengths(self, texts: list[str]) -> list[int] | None:
        """Word pieces each text costs this encoder, or None if unknowable.

        Lengths rather than a count of overflows, because the two answer different
        questions: a count says how many queries were cut, while the distribution
        says whether a budget is nearly right or wildly wrong -- and only the
        latter can be used to size one.

        Returns:
            list[int] | None: One length per text, or None where the provider
            exposes no tokenizer. A caller must read None as "cannot tell", never
            as zero.
        """
        return None

    def count_over_limit(self, texts: list[str]) -> int | None:
        """How many of ``texts`` exceed :attr:`sequence_limit`.

        Returns:
            int | None: The count, or None when the limit or the tokenizer is
            unknown.
        """
        limit = self.sequence_limit
        if limit is None or not texts:
            return None
        lengths = self.token_lengths(texts)
        if lengths is None:
            return None
        return sum(1 for length in lengths if length > limit)

    def embed_one(self, text: str) -> list[float]:
        """Return a vector for one query text."""
        vectors = self.embed_query([text])
        if not vectors:
            raise ValueError("Embedding provider returned no vectors for query text")
        return vectors[0]

    @classmethod
    def create(cls, config: EmbeddingConfig) -> "EmbeddingTool":
        """Factory for provider-specific embedding tools."""
        if config.provider == EmbeddingProvider.HUGGINGFACE:
            return HuggingFaceEmbeddingTool(config=config)
        if config.provider == EmbeddingProvider.OPENAI:
            return OpenAIEmbeddingTool(config=config)
        if config.provider == EmbeddingProvider.OLLAMA:
            return OllamaEmbeddingTool(config=config)
        raise ValueError(f"Unsupported embedding provider: {config.provider}")

Attributes

config = Field(default_factory=EmbeddingConfig) class-attribute instance-attribute
sequence_limit property

Tokens this provider accepts before it silently truncates, if known.

Truncation is the failure mode with no symptom: the provider returns a vector of the right shape for a prefix of the text, and the caller cannot tell that the tail was dropped. Exposing the limit is what lets a caller report it instead of discovering it as unexplained recall loss.

Returns:

Type Description
int | None

int | None: The limit, or None where the provider does not state one.

Methods:

count_over_limit(texts)

How many of texts exceed :attr:sequence_limit.

Returns:

Type Description
int | None

int | None: The count, or None when the limit or the tokenizer is

int | None

unknown.

Source code in ontocast/tool/vector_store/embedding.py
def count_over_limit(self, texts: list[str]) -> int | None:
    """How many of ``texts`` exceed :attr:`sequence_limit`.

    Returns:
        int | None: The count, or None when the limit or the tokenizer is
        unknown.
    """
    limit = self.sequence_limit
    if limit is None or not texts:
        return None
    lengths = self.token_lengths(texts)
    if lengths is None:
        return None
    return sum(1 for length in lengths if length > limit)
create(config) classmethod

Factory for provider-specific embedding tools.

Source code in ontocast/tool/vector_store/embedding.py
@classmethod
def create(cls, config: EmbeddingConfig) -> "EmbeddingTool":
    """Factory for provider-specific embedding tools."""
    if config.provider == EmbeddingProvider.HUGGINGFACE:
        return HuggingFaceEmbeddingTool(config=config)
    if config.provider == EmbeddingProvider.OPENAI:
        return OpenAIEmbeddingTool(config=config)
    if config.provider == EmbeddingProvider.OLLAMA:
        return OllamaEmbeddingTool(config=config)
    raise ValueError(f"Unsupported embedding provider: {config.provider}")
embed(texts)

Return vectors for all given texts as documents.

Serialisation, where it is needed, belongs to whatever owns the model — the shared encoder for local checkpoints, nothing for remote providers.

Source code in ontocast/tool/vector_store/embedding.py
def embed(self, texts: list[str]) -> list[list[float]]:
    """Return vectors for all given texts as *documents*.

    Serialisation, where it is needed, belongs to whatever owns the model —
    the shared encoder for local checkpoints, nothing for remote providers.
    """
    if not texts:
        return []
    return self._embed_raw(self._apply(self.config.document_prefix, texts))
embed_one(text)

Return a vector for one query text.

Source code in ontocast/tool/vector_store/embedding.py
def embed_one(self, text: str) -> list[float]:
    """Return a vector for one query text."""
    vectors = self.embed_query([text])
    if not vectors:
        raise ValueError("Embedding provider returned no vectors for query text")
    return vectors[0]
embed_query(texts)

Return vectors for all given texts as queries.

Asymmetric retrieval models are trained with distinct query and document instructions and lose accuracy when both sides are encoded identically. With empty prefixes — the default, suiting a symmetric paraphrase model — this is exactly :meth:embed.

Source code in ontocast/tool/vector_store/embedding.py
def embed_query(self, texts: list[str]) -> list[list[float]]:
    """Return vectors for all given texts as *queries*.

    Asymmetric retrieval models are trained with distinct query and document
    instructions and lose accuracy when both sides are encoded identically. With
    empty prefixes — the default, suiting a symmetric paraphrase model — this is
    exactly :meth:`embed`.
    """
    if not texts:
        return []
    return self._embed_raw(self._apply(self.config.query_prefix, texts))
token_lengths(texts)

Word pieces each text costs this encoder, or None if unknowable.

Lengths rather than a count of overflows, because the two answer different questions: a count says how many queries were cut, while the distribution says whether a budget is nearly right or wildly wrong -- and only the latter can be used to size one.

Returns:

Type Description
list[int] | None

list[int] | None: One length per text, or None where the provider

list[int] | None

exposes no tokenizer. A caller must read None as "cannot tell", never

list[int] | None

as zero.

Source code in ontocast/tool/vector_store/embedding.py
def token_lengths(self, texts: list[str]) -> list[int] | None:
    """Word pieces each text costs this encoder, or None if unknowable.

    Lengths rather than a count of overflows, because the two answer different
    questions: a count says how many queries were cut, while the distribution
    says whether a budget is nearly right or wildly wrong -- and only the
    latter can be used to size one.

    Returns:
        list[int] | None: One length per text, or None where the provider
        exposes no tokenizer. A caller must read None as "cannot tell", never
        as zero.
    """
    return None

FastembedBm25SparseTool

Bases: Tool

BM25-style sparse text embeddings via fastembed (Qdrant-compatible).

Source code in ontocast/tool/vector_store/embedding.py
class FastembedBm25SparseTool(Tool):
    """BM25-style sparse text embeddings via fastembed (Qdrant-compatible)."""

    config: EmbeddingConfig = Field(default_factory=EmbeddingConfig)
    _embedder: Any = PrivateAttr(default=None)

    def _get_embedder(self) -> Any:
        if self._embedder is not None:
            return self._embedder
        fastembed_mod = require("fastembed", feature="BM25 sparse embeddings")
        sparse_cls = getattr(fastembed_mod, "SparseTextEmbedding", None)
        if sparse_cls is None:
            raise ImportError("fastembed.SparseTextEmbedding is not available")
        self._embedder = sparse_cls(model_name=self.config.bm25_model_name)
        return self._embedder

    def embed_sparse(self, texts: list[str]) -> list[SparseVector]:
        """Return Qdrant sparse vectors for indexing all given texts (thread-safe)."""
        if not texts:
            return []
        with _SPARSE_EMBED_LOCK:
            return self._embed_sparse_unlocked(texts)

    def embed_sparse_query(self, texts: list[str]) -> list[SparseVector]:
        """Return Qdrant sparse vectors for *querying* with all given texts.

        BM25 is asymmetric: documents carry term-frequency saturation weights, queries
        carry flat per-term weights, and the IDF factor is applied by the store. Encoding
        queries with the document encoder instead squares the term-frequency weighting and
        drops the query/document distinction entirely.
        """
        if not texts:
            return []
        with _SPARSE_EMBED_LOCK:
            return self._embed_sparse_unlocked(texts, query=True)

    def _embed_sparse_unlocked(
        self, texts: list[str], *, query: bool = False
    ) -> list[SparseVector]:
        model = self._get_embedder()
        encode = model.query_embed if query else model.embed
        out: list[SparseVector] = []
        for sparse_emb in encode(texts):
            payload = sparse_emb.as_object()
            indices_raw = payload["indices"]
            values_raw = payload["values"]
            indices_list = indices_raw.tolist()
            values_list = values_raw.tolist()
            out.append(
                SparseVector(
                    indices=[int(i) for i in indices_list],
                    values=[float(v) for v in values_list],
                )
            )
        if len(out) != len(texts):
            raise ValueError("BM25 embedder returned mismatched sparse vector count")
        return out

    def embed_one_sparse(self, text: str) -> SparseVector:
        vectors = self.embed_sparse_query([text])
        if not vectors:
            raise ValueError("BM25 embedder returned no sparse vector for query text")
        return vectors[0]

Attributes

config = Field(default_factory=EmbeddingConfig) class-attribute instance-attribute

Methods:

embed_one_sparse(text)
Source code in ontocast/tool/vector_store/embedding.py
def embed_one_sparse(self, text: str) -> SparseVector:
    vectors = self.embed_sparse_query([text])
    if not vectors:
        raise ValueError("BM25 embedder returned no sparse vector for query text")
    return vectors[0]
embed_sparse(texts)

Return Qdrant sparse vectors for indexing all given texts (thread-safe).

Source code in ontocast/tool/vector_store/embedding.py
def embed_sparse(self, texts: list[str]) -> list[SparseVector]:
    """Return Qdrant sparse vectors for indexing all given texts (thread-safe)."""
    if not texts:
        return []
    with _SPARSE_EMBED_LOCK:
        return self._embed_sparse_unlocked(texts)
embed_sparse_query(texts)

Return Qdrant sparse vectors for querying with all given texts.

BM25 is asymmetric: documents carry term-frequency saturation weights, queries carry flat per-term weights, and the IDF factor is applied by the store. Encoding queries with the document encoder instead squares the term-frequency weighting and drops the query/document distinction entirely.

Source code in ontocast/tool/vector_store/embedding.py
def embed_sparse_query(self, texts: list[str]) -> list[SparseVector]:
    """Return Qdrant sparse vectors for *querying* with all given texts.

    BM25 is asymmetric: documents carry term-frequency saturation weights, queries
    carry flat per-term weights, and the IDF factor is applied by the store. Encoding
    queries with the document encoder instead squares the term-frequency weighting and
    drops the query/document distinction entirely.
    """
    if not texts:
        return []
    with _SPARSE_EMBED_LOCK:
        return self._embed_sparse_unlocked(texts, query=True)

HuggingFaceEmbeddingTool

Bases: EmbeddingTool

Local HuggingFace/SentenceTransformer embeddings.

Source code in ontocast/tool/vector_store/embedding.py
class HuggingFaceEmbeddingTool(EmbeddingTool):
    """Local HuggingFace/SentenceTransformer embeddings."""

    _embedder: SharedEncoder | None = PrivateAttr(default=None)

    def _get_embedder(self) -> SharedEncoder:
        if self._embedder is not None:
            return self._embedder
        # Shared process-wide with entity clustering and semantic chunking, which
        # default to the same or a configurable checkpoint. The handle owns the
        # lock, so every one of those consumers is serialised on the same model
        # without any of them having to know about the others.
        self._embedder = get_shared_encoder(
            self.config.model_name,
            feature=(
                "Local HuggingFace embeddings. For a light install, set "
                "EMBEDDING_PROVIDER=openai or =ollama to embed via an API instead"
            ),
        )
        return self._embedder

    def _embed_raw(self, texts: list[str]) -> list[list[float]]:
        vectors = self._get_embedder().encode(
            texts, convert_to_numpy=True, show_progress_bar=len(texts) > 100
        )
        return [vector.tolist() for vector in vectors]

    @property
    def sequence_limit(self) -> int | None:
        """The checkpoint's ``max_seq_length``.

        Frequently far below what the tokenizer's own ``model_max_length`` reports,
        and it is this value that governs: sentence-transformers truncates to it
        before the model sees the text.
        """
        try:
            limit = self._get_embedder().model.max_seq_length
        except Exception:  # noqa: BLE001 - a missing attribute must not break retrieval
            return None
        return int(limit) if limit else None

    def token_lengths(self, texts: list[str]) -> list[int] | None:
        """Word pieces per text, from the checkpoint's own tokenizer.

        Tokenizes without encoding, which is cheap beside the forward pass this
        accompanies. Prefixes are applied first, because an instruction prefix
        counts against the same budget as the text it introduces.
        """
        if not texts:
            return []
        try:
            tokenizer = self._get_embedder().model.tokenizer
            prefixed = self._apply(self.config.query_prefix, texts)
            encoded = tokenizer(prefixed, add_special_tokens=True)["input_ids"]
        except Exception:  # noqa: BLE001 - telemetry must never break retrieval
            return None
        return [len(ids) for ids in encoded]

Attributes

sequence_limit property

The checkpoint's max_seq_length.

Frequently far below what the tokenizer's own model_max_length reports, and it is this value that governs: sentence-transformers truncates to it before the model sees the text.

Methods:

token_lengths(texts)

Word pieces per text, from the checkpoint's own tokenizer.

Tokenizes without encoding, which is cheap beside the forward pass this accompanies. Prefixes are applied first, because an instruction prefix counts against the same budget as the text it introduces.

Source code in ontocast/tool/vector_store/embedding.py
def token_lengths(self, texts: list[str]) -> list[int] | None:
    """Word pieces per text, from the checkpoint's own tokenizer.

    Tokenizes without encoding, which is cheap beside the forward pass this
    accompanies. Prefixes are applied first, because an instruction prefix
    counts against the same budget as the text it introduces.
    """
    if not texts:
        return []
    try:
        tokenizer = self._get_embedder().model.tokenizer
        prefixed = self._apply(self.config.query_prefix, texts)
        encoded = tokenizer(prefixed, add_special_tokens=True)["input_ids"]
    except Exception:  # noqa: BLE001 - telemetry must never break retrieval
        return None
    return [len(ids) for ids in encoded]

OllamaEmbeddingTool

Bases: _LangChainEmbeddingTool

Ollama embeddings using either LangChain or direct API fallback.

Source code in ontocast/tool/vector_store/embedding.py
class OllamaEmbeddingTool(_LangChainEmbeddingTool):
    """Ollama embeddings using either LangChain or direct API fallback."""

    def _build_embedder(self) -> Embeddings:
        OllamaEmbeddings = require(
            "langchain_ollama.embeddings", feature="Ollama embeddings"
        ).OllamaEmbeddings
        return OllamaEmbeddings(
            model=self.config.model_name,
            base_url=self.config.base_url,
        )

    def _embed_raw(self, texts: list[str]) -> list[list[float]]:
        try:
            return super()._embed_raw(texts)
        except Exception as exc:
            # Log the real cause: a bad base URL, an auth failure and an absent
            # langchain integration all reach the fallback identically, and if
            # the HTTP path then fails too the user is shown an httpx error
            # unrelated to what actually went wrong.
            logger.debug("Ollama langchain embedding failed, using HTTP: %s", exc)
            return self._embed_via_http(texts)

    def _embed_via_http(self, texts: list[str]) -> list[list[float]]:
        base_url = self.config.base_url or "http://localhost:11434"
        endpoint = f"{base_url.rstrip('/')}/api/embeddings"
        vectors: list[list[float]] = []
        with httpx.Client(timeout=30.0) as client:
            for text in texts:
                response = client.post(
                    endpoint,
                    json={"model": self.config.model_name, "prompt": text},
                )
                response.raise_for_status()
                payload = response.json()
                vector = payload.get("embedding")
                if not isinstance(vector, list):
                    raise ValueError(
                        "Ollama embedding response missing 'embedding' vector"
                    )
                vectors.append(vector)
        return vectors

OpenAIEmbeddingTool

Bases: _LangChainEmbeddingTool

OpenAI embeddings via langchain-openai.

Source code in ontocast/tool/vector_store/embedding.py
class OpenAIEmbeddingTool(_LangChainEmbeddingTool):
    """OpenAI embeddings via langchain-openai."""

    def _build_embedder(self) -> Embeddings:
        api_key = (
            SecretStr(self.config.api_key) if self.config.api_key is not None else None
        )
        OpenAIEmbeddings = require(
            "langchain_openai", feature="OpenAI embeddings"
        ).OpenAIEmbeddings
        return OpenAIEmbeddings(
            model=self.config.model_name,
            api_key=api_key,
            base_url=self.config.base_url,
        )

Functions: