Skip to content

ontocast.tool.vector_store.util

Backend-agnostic helpers for ontology vector storage.

Attributes

META_EMBEDDING_DIMENSION = 'embedding_dimension' module-attribute

META_EMBEDDING_MODEL = 'embedding_model' module-attribute

Classes

EmbeddingContractMismatchError

Bases: ValueError

Embedding vectors or store metadata disagree with the active embedding config.

Source code in ontocast/tool/vector_store/util.py
class EmbeddingContractMismatchError(ValueError):
    """Embedding vectors or store metadata disagree with the active embedding config."""

Functions:

atom_from_payload(payload, *, score=None, default_id='')

Source code in ontocast/tool/vector_store/util.py
def atom_from_payload(
    payload: Mapping[str, Any],
    *,
    score: float | None = None,
    default_id: str = "",
) -> GraphAtom:
    created_at_raw = payload.get("created_at")
    return GraphAtom(
        atom_id=str(payload.get("atom_id", default_id)),
        ontology_iri=str(payload.get("ontology_iri", "")),
        ontology_id=payload.get("ontology_id"),
        ontology_hash=payload.get("ontology_hash"),
        ontology_version=payload.get("ontology_version"),
        iri=str(payload.get("iri", "")),
        entity_role=canonicalize_entity_role(payload.get("entity_role")),
        core_representation=str(payload.get("core_representation", "")),
        minimal_representation=str(payload.get("minimal_representation", "")),
        neighborhood_representation=str(payload.get("neighborhood_representation", "")),
        lexical_triggers=_payload_str_list(payload, "lexical_triggers"),
        symbol_surfaces=_payload_str_list(payload, "symbol_surfaces"),
        created_at=parse_created_at(created_at_raw),
        score=score,
    )

atom_payload(atom)

Source code in ontocast/tool/vector_store/util.py
def atom_payload(atom: GraphAtom) -> dict[str, Any]:
    return {
        "atom_id": atom.atom_id,
        "ontology_iri": atom.ontology_iri,
        "ontology_id": atom.ontology_id,
        "ontology_hash": atom.ontology_hash,
        "ontology_version": atom.ontology_version,
        "iri": atom.iri,
        "entity_role": canonicalize_entity_role(atom.entity_role),
        "core_representation": atom.core_representation,
        "minimal_representation": atom.minimal_representation,
        "neighborhood_representation": atom.neighborhood_representation,
        "lexical_triggers": list(atom.lexical_triggers),
        "symbol_surfaces": list(atom.symbol_surfaces),
        "created_at": atom.created_at.isoformat(),
    }

atom_scope_fingerprint(store_config)

Fingerprint fragment for settings that change what gets stored per atom.

Covers both which entities become atoms and which literals become their surface forms and lexical triggers. All of these change the stored payload, so serving an index built under different values silently degrades retrieval instead of raising.

Returns None at the defaults, so collections built under them keep the fingerprint they already have and need no reindex on upgrade.

Parameters:

Name Type Description Default
store_config VectorStoreConfig

Active vector-store settings.

required

Returns:

Type Description
str | None

str | None: Compact divergence marker, or None when nothing diverges.

Source code in ontocast/tool/vector_store/util.py
def atom_scope_fingerprint(store_config: VectorStoreConfig) -> str | None:
    """Fingerprint fragment for settings that change what gets stored per atom.

    Covers both *which entities* become atoms and *which literals* become their
    surface forms and lexical triggers. All of these change the stored payload,
    so serving an index built under different values silently degrades
    retrieval instead of raising.

    Returns ``None`` at the defaults, so collections built under them keep the
    fingerprint they already have and need no reindex on upgrade.

    Args:
        store_config: Active vector-store settings.

    Returns:
        str | None: Compact divergence marker, or ``None`` when nothing diverges.
    """
    defaults = VectorStoreConfig.model_fields
    parts: list[str] = []
    if store_config.index_undescribed_iris:
        parts.append("undescribed")
    if store_config.embed_standard_vocab_iris:
        parts.append("stdvocab")
    for prefix in sorted(store_config.extra_excluded_namespace_prefixes):
        parts.append(f"x:{prefix}")

    def _diverges(name: str) -> bool:
        factory = defaults[name].default_factory
        default = factory() if factory is not None else defaults[name].default
        return getattr(store_config, name) != default

    # The surface-form and trigger settings are pushed into the atomizer and
    # decide what lands in the stored payload, so they belong in the identity of
    # the vectors. They were previously omitted, which meant changing one served
    # a stale index rather than raising EmbeddingContractMismatchError.
    for name in ("label_predicates", "symbol_predicates", "lexical_trigger_predicates"):
        if _diverges(name):
            joined = ",".join(sorted(getattr(store_config, name)))
            parts.append(f"{name}={render_text_hash(joined)[:12]}")
    for name in (
        "lexical_trigger_enabled",
        "lexical_trigger_heuristic_enabled",
        "lexical_trigger_min_len",
        "lexical_trigger_max_len",
        "lexical_trigger_heuristic_max_per_entity",
    ):
        if _diverges(name):
            parts.append(f"{name}={getattr(store_config, name)}")
    return ",".join(parts) if parts else None

coerce_metadata_int(value, *, field, collection)

Source code in ontocast/tool/vector_store/util.py
def coerce_metadata_int(value: Any, *, field: str, collection: str) -> int:
    if type(value) is bool:
        raise ValueError(
            f"Vector store '{collection}' metadata {field!r} has invalid type"
        )
    if isinstance(value, int):
        return value
    if isinstance(value, float) and value.is_integer():
        return int(value)
    if isinstance(value, str):
        try:
            return int(value.strip(), 10)
        except ValueError as exc:
            raise ValueError(
                f"Vector store '{collection}' metadata {field!r} is not an integer"
            ) from exc
    raise ValueError(f"Vector store '{collection}' metadata {field!r} has invalid type")

collection_embedding_metadata(embedding_config, *, metadata_dim, minimal_label_limit=None, atom_scope=None)

Source code in ontocast/tool/vector_store/util.py
def collection_embedding_metadata(
    embedding_config: EmbeddingConfig,
    *,
    metadata_dim: int,
    minimal_label_limit: int | None = None,
    atom_scope: str | None = None,
) -> dict[str, Any]:
    return {
        META_EMBEDDING_DIMENSION: metadata_dim,
        META_EMBEDDING_MODEL: embedding_model_fingerprint(
            embedding_config,
            minimal_label_limit=minimal_label_limit,
            atom_scope=atom_scope,
        ),
    }

dedupe_hits_by_identity(hits, *, store_config)

Source code in ontocast/tool/vector_store/util.py
def dedupe_hits_by_identity(
    hits: list[OntologySearchHit],
    *,
    store_config: VectorStoreConfig,
) -> list[OntologySearchHit]:
    if not hits:
        return []
    best_by_key: dict[str, OntologySearchHit] = {}
    order_index: dict[str, int] = {}
    for index, hit in enumerate(hits):
        key = identity_key_for_atom(hit.atom, store_config=store_config)
        previous = best_by_key.get(key)
        if previous is None:
            best_by_key[key] = hit
            order_index[key] = index
            continue
        if float(hit.score) > float(previous.score):
            best_by_key[key] = hit
    deduped = list(best_by_key.values())
    deduped.sort(
        key=lambda h: (
            -float(h.score),
            order_index[identity_key_for_atom(h.atom, store_config=store_config)],
        )
    )
    return deduped

effective_bm25_top_k(store_config, top_k)

Depth of the sparse lane, which need not match the dense lanes'.

Fusion is by reciprocal rank, so a channel's depth is a weight in disguise: a sparse list of length N hands out ranks 1..N at full lane weight however weak its tail is. The dense and sparse lanes fail differently -- dense retrieval degrades gracefully into topical near-misses, lexical retrieval into unrelated documents that share a token -- so the depth at which each stops being useful is not the same number, and tying them together means tuning one mis-tunes the other.

Returns:

Name Type Description
int int

bm25_top_k when set, else whatever the dense lanes use.

Source code in ontocast/tool/vector_store/util.py
def effective_bm25_top_k(store_config: VectorStoreConfig, top_k: int | None) -> int:
    """Depth of the sparse lane, which need not match the dense lanes'.

    Fusion is by reciprocal rank, so a channel's *depth* is a weight in disguise: a
    sparse list of length N hands out ranks 1..N at full lane weight however weak its
    tail is. The dense and sparse lanes fail differently -- dense retrieval degrades
    gracefully into topical near-misses, lexical retrieval into unrelated documents
    that share a token -- so the depth at which each stops being useful is not the
    same number, and tying them together means tuning one mis-tunes the other.

    Returns:
        int: ``bm25_top_k`` when set, else whatever the dense lanes use.
    """
    if store_config.bm25_top_k is not None:
        return store_config.bm25_top_k
    return effective_top_k(store_config, top_k)

effective_top_k(store_config, top_k)

Source code in ontocast/tool/vector_store/util.py
def effective_top_k(store_config: VectorStoreConfig, top_k: int | None) -> int:
    if top_k is not None:
        return top_k
    return store_config.top_k

embedding_contract_help(*, backend='vector store')

Source code in ontocast/tool/vector_store/util.py
def embedding_contract_help(*, backend: str = "vector store") -> str:
    return (
        f"Align EmbeddingConfig (EMBEDDING_*) with the {backend}: use the same model "
        "and dimension as when the store was created, or drop the ontology table/"
        "collection and let initialize() recreate it."
    )

embedding_fingerprint_matches(stored, embedding_config, *, minimal_label_limit=None, atom_scope=None)

Whether stored is the fingerprint the given config would produce.

Takes the same optional components as :func:embedding_model_fingerprint. Omitting them previously made this disagree with validate_embedding_contract_metadata for any non-default collection -- it would report a match the validator rejects.

Source code in ontocast/tool/vector_store/util.py
def embedding_fingerprint_matches(
    stored: str,
    embedding_config: EmbeddingConfig,
    *,
    minimal_label_limit: int | None = None,
    atom_scope: str | None = None,
) -> bool:
    """Whether ``stored`` is the fingerprint the given config would produce.

    Takes the same optional components as :func:`embedding_model_fingerprint`.
    Omitting them previously made this disagree with
    ``validate_embedding_contract_metadata`` for any non-default collection --
    it would report a match the validator rejects.
    """
    return stored == embedding_model_fingerprint(
        embedding_config,
        minimal_label_limit=minimal_label_limit,
        atom_scope=atom_scope,
    )

embedding_model_fingerprint(embedding_config, *, minimal_label_limit=None, atom_scope=None)

Identity of the vectors a config produces, stored alongside the collection.

Query/document prefixes belong here: they change the embedded text, so an index built without them is not comparable to queries issued with them, and the mismatch would otherwise show up only as quietly degraded retrieval. The sparse surface-form cap is included for the same reason -- it decides how many of a term's aliases enter the BM25 text. It contributes only when set to a non-default value, so collections built under the default keep their existing fingerprint. atom_scope follows the same rule for settings that decide which entities are atomized at all.

The surface-form contract (sf=) is separate and always contributes: it records which literals become surface forms and which entities become atoms, both of which change the stored index even at default settings.

Parameters:

Name Type Description Default
embedding_config EmbeddingConfig

Dense/sparse model configuration.

required
minimal_label_limit int | None

Sparse surface-form cap, when it differs from the default.

None
atom_scope str | None

Atom-scope divergence from :func:atom_scope_fingerprint, if any.

None

Returns:

Name Type Description
str str

Stable fingerprint stored alongside the collection.

Source code in ontocast/tool/vector_store/util.py
def embedding_model_fingerprint(
    embedding_config: EmbeddingConfig,
    *,
    minimal_label_limit: int | None = None,
    atom_scope: str | None = None,
) -> str:
    """Identity of the vectors a config produces, stored alongside the collection.

    Query/document prefixes belong here: they change the embedded text, so an index
    built without them is not comparable to queries issued with them, and the mismatch
    would otherwise show up only as quietly degraded retrieval. The sparse surface-form
    cap is included for the same reason -- it decides how many of a term's aliases enter
    the BM25 text. It contributes only when set to a non-default value, so collections
    built under the default keep their existing fingerprint. ``atom_scope`` follows the
    same rule for settings that decide which entities are atomized at all.

    The surface-form contract (``sf=``) is separate and always contributes: it records
    *which* literals become surface forms and which entities become atoms, both of which
    change the stored index even at default settings.

    Args:
        embedding_config: Dense/sparse model configuration.
        minimal_label_limit: Sparse surface-form cap, when it differs from the default.
        atom_scope: Atom-scope divergence from :func:`atom_scope_fingerprint`, if any.

    Returns:
        str: Stable fingerprint stored alongside the collection.
    """
    ec = embedding_config
    dense_part = f"dense:{ec.provider.value}:{ec.model_name}"
    affixes = f"|q={ec.query_prefix}|d={ec.document_prefix}"
    fingerprint = (
        f"{dense_part}|bm25={ec.bm25_model_name}{affixes}|sf={_SURFACE_FORM_CONTRACT}"
    )
    if (
        minimal_label_limit is not None
        and minimal_label_limit != _DEFAULT_MINIMAL_LABEL_LIMIT
    ):
        fingerprint += f"|minlabels={minimal_label_limit}"
    if atom_scope:
        fingerprint += f"|atoms={atom_scope}"
    return fingerprint

identity_key_for_atom(atom, *, store_config)

Source code in ontocast/tool/vector_store/util.py
def identity_key_for_atom(
    atom: GraphAtom,
    *,
    store_config: VectorStoreConfig,
) -> str:
    if store_config.dedup_mode == VectorStoreDedupMode.ATOM_ID:
        return atom.atom_id
    parts: list[str] = [
        atom.ontology_iri or "",
        atom.iri or "",
    ]
    if store_config.dedup_include_version:
        parts.append(atom.ontology_version or "")
    if store_config.dedup_include_hash:
        parts.append(atom.ontology_hash or "")
    return "|".join(parts)

iter_batches(items, batch_size)

Source code in ontocast/tool/vector_store/util.py
def iter_batches(items: list[Any], batch_size: int) -> list[list[Any]]:
    batches: list[list[Any]] = []
    for index in range(0, len(items), batch_size):
        batches.append(items[index : index + batch_size])
    return batches

normalized_core_neighborhood_weights(store_config)

Source code in ontocast/tool/vector_store/util.py
def normalized_core_neighborhood_weights(
    store_config: VectorStoreConfig,
) -> tuple[float, float]:
    cw, nw, _ = normalized_fusion_weights(store_config)
    total = cw + nw
    if total <= 0.0:
        return (0.5, 0.5)
    return (cw / total, nw / total)

normalized_fusion_weights(store_config)

Source code in ontocast/tool/vector_store/util.py
def normalized_fusion_weights(
    store_config: VectorStoreConfig,
) -> tuple[float, float, float]:
    cw = store_config.fusion_core_weight
    nw = store_config.fusion_neighborhood_weight
    bw = store_config.fusion_bm25_weight
    total = cw + nw + bw
    if total <= 0.0:
        return (1.0 / 3.0, 1.0 / 3.0, 1.0 / 3.0)
    return (cw / total, nw / total, bw / total)

parse_created_at(value)

Source code in ontocast/tool/vector_store/util.py
def parse_created_at(value: Any) -> datetime:
    if isinstance(value, datetime):
        return value
    if isinstance(value, str):
        try:
            return datetime.fromisoformat(value.replace("Z", "+00:00"))
        except ValueError:
            pass
    return datetime.now(timezone.utc)

point_id(atom_id)

Source code in ontocast/tool/vector_store/util.py
def point_id(atom_id: str) -> str:
    try:
        return str(uuid.UUID(atom_id))
    except ValueError:
        return str(uuid.uuid5(uuid.NAMESPACE_URL, atom_id))

point_id_for_atom(atom, *, store_config)

Source code in ontocast/tool/vector_store/util.py
def point_id_for_atom(
    atom: GraphAtom,
    *,
    store_config: VectorStoreConfig,
) -> str:
    if store_config.dedup_mode == VectorStoreDedupMode.ATOM_ID:
        return point_id(atom.atom_id)
    return point_id(identity_key_for_atom(atom, store_config=store_config))

rank_fuse_channel_hits(core_hits, neighborhood_hits, bm25_hits, *, core_weight, neighborhood_weight, bm25_weight, limit, rank_constant=0.0)

Fuse three ranked channels into one list by weighted reciprocal rank.

Each channel contributes weight / (rank_constant + rank) per atom, summed across channels. Raw channel scores never enter the fused score -- they are only a tiebreak -- which is what lets an uncalibrated BM25 scale sit beside cosine without either dominating by units alone.

rank_constant is the smoothing term. At 0 (the default) a rank-2 hit is worth exactly half a rank-1 hit and rank 3 a third, so the fused order is decided almost entirely by which channel put what first; a deep list of weak matches still hands out ranks 1..N at full lane weight. Raising it flattens that decay, so agreement across channels outweighs position within one -- which is the property reciprocal-rank fusion is usually chosen for.

Parameters:

Name Type Description Default
core_hits list[OntologySearchHit]

Core-lane hits, best first.

required
neighborhood_hits list[OntologySearchHit]

Neighborhood-lane hits, best first.

required
bm25_hits list[OntologySearchHit]

Sparse-lane hits, best first.

required
core_weight float

Normalized weight for the core lane.

required
neighborhood_weight float

Normalized weight for the neighborhood lane.

required
bm25_weight float

Normalized weight for the sparse lane.

required
limit int

Maximum hits to return.

required
rank_constant float

Added to each rank before the reciprocal.

0.0

Returns:

Type Description
list[OntologySearchHit]

list[OntologySearchHit]: Fused hits, best first, each carrying the fused

list[OntologySearchHit]

score in place of its channel score.

Source code in ontocast/tool/vector_store/util.py
def rank_fuse_channel_hits(
    core_hits: list[OntologySearchHit],
    neighborhood_hits: list[OntologySearchHit],
    bm25_hits: list[OntologySearchHit],
    *,
    core_weight: float,
    neighborhood_weight: float,
    bm25_weight: float,
    limit: int,
    rank_constant: float = 0.0,
) -> list[OntologySearchHit]:
    """Fuse three ranked channels into one list by weighted reciprocal rank.

    Each channel contributes ``weight / (rank_constant + rank)`` per atom, summed
    across channels. Raw channel scores never enter the fused score -- they are only
    a tiebreak -- which is what lets an uncalibrated BM25 scale sit beside cosine
    without either dominating by units alone.

    ``rank_constant`` is the smoothing term. At 0 (the default) a rank-2 hit is worth exactly half a rank-1 hit and rank 3 a third,
    so the fused order is decided almost entirely by which channel put what first;
    a deep list of weak matches still hands out ranks 1..N at full lane weight.
    Raising it flattens that decay, so agreement *across* channels outweighs
    position *within* one -- which is the property reciprocal-rank fusion is
    usually chosen for.

    Args:
        core_hits: Core-lane hits, best first.
        neighborhood_hits: Neighborhood-lane hits, best first.
        bm25_hits: Sparse-lane hits, best first.
        core_weight: Normalized weight for the core lane.
        neighborhood_weight: Normalized weight for the neighborhood lane.
        bm25_weight: Normalized weight for the sparse lane.
        limit: Maximum hits to return.
        rank_constant: Added to each rank before the reciprocal.

    Returns:
        list[OntologySearchHit]: Fused hits, best first, each carrying the fused
        score in place of its channel score.
    """
    rank_scores: dict[str, float] = {}
    best_hit_by_id: dict[str, OntologySearchHit] = {}

    def fold(hits: list[OntologySearchHit], weight: float) -> None:
        for rank, hit in enumerate(hits, start=1):
            atom_id = hit.atom.atom_id
            rank_scores[atom_id] = rank_scores.get(atom_id, 0.0) + (
                weight / (rank_constant + rank)
            )
            prev = best_hit_by_id.get(atom_id)
            if prev is None or hit.score > prev.score:
                best_hit_by_id[atom_id] = hit

    fold(core_hits, core_weight)
    fold(neighborhood_hits, neighborhood_weight)
    fold(bm25_hits, bm25_weight)

    ranked_atom_ids = sorted(
        rank_scores.keys(),
        key=lambda atom_id: (
            rank_scores[atom_id],
            float(best_hit_by_id[atom_id].score),
            atom_id,
        ),
        reverse=True,
    )[:limit]
    out: list[OntologySearchHit] = []
    for atom_id in ranked_atom_ids:
        source_hit = best_hit_by_id[atom_id]
        atom = source_hit.atom.model_copy(update={"score": rank_scores[atom_id]})
        out.append(OntologySearchHit(atom=atom, score=rank_scores[atom_id]))
    return out

require_embedding_vector_length(vector, *, role, expected)

Source code in ontocast/tool/vector_store/util.py
def require_embedding_vector_length(
    vector: list[float],
    *,
    role: str,
    expected: int,
) -> None:
    if len(vector) != expected:
        raise EmbeddingContractMismatchError(
            f"{role} vector length {len(vector)} does not match the configured "
            f"embedding dimension {expected}. " + embedding_contract_help()
        )

sync_atomizer_from_store_config(atomizer, store_config)

Mirror vector-store representation settings onto the atomizer.

Source code in ontocast/tool/vector_store/util.py
def sync_atomizer_from_store_config(
    atomizer: GraphAtomizer, store_config: VectorStoreConfig
) -> None:
    """Mirror vector-store representation settings onto the atomizer."""
    atomizer.minimal_representation_label_limit = store_config.minimal_label_limit
    atomizer.label_predicates = list(store_config.label_predicates)
    atomizer.symbol_predicates = list(store_config.symbol_predicates)
    atomizer.index_undescribed_iris = store_config.index_undescribed_iris
    atomizer.embed_standard_vocab_iris = store_config.embed_standard_vocab_iris
    atomizer.extra_excluded_namespace_prefixes = list(
        store_config.extra_excluded_namespace_prefixes
    )
    atomizer.lexical_trigger_enabled = store_config.lexical_trigger_enabled
    atomizer.lexical_trigger_predicates = list(store_config.lexical_trigger_predicates)
    atomizer.lexical_trigger_heuristic_enabled = (
        store_config.lexical_trigger_heuristic_enabled
    )
    atomizer.lexical_trigger_min_len = store_config.lexical_trigger_min_len
    atomizer.lexical_trigger_max_len = store_config.lexical_trigger_max_len
    atomizer.lexical_trigger_heuristic_max_per_entity = (
        store_config.lexical_trigger_heuristic_max_per_entity
    )

validate_embedding_contract_metadata(collection, raw_metadata, *, embedding_config, expected_meta_dim, minimal_label_limit=None, atom_scope=None)

Source code in ontocast/tool/vector_store/util.py
def validate_embedding_contract_metadata(
    collection: str,
    raw_metadata: Mapping[str, Any] | None,
    *,
    embedding_config: EmbeddingConfig,
    expected_meta_dim: int,
    minimal_label_limit: int | None = None,
    atom_scope: str | None = None,
) -> None:
    if raw_metadata is None:
        meta: dict[str, Any] = {}
    else:
        meta = dict(raw_metadata)
    dim_key = META_EMBEDDING_DIMENSION
    model_key = META_EMBEDDING_MODEL
    if dim_key not in meta or model_key not in meta:
        raise EmbeddingContractMismatchError(
            f"Vector store '{collection}' is missing OntoCast embedding metadata "
            f"({dim_key!r}, {model_key!r}). Drop and recreate the store. "
            + embedding_contract_help()
        )
    stored_dim = coerce_metadata_int(
        meta[dim_key], field=dim_key, collection=collection
    )
    stored_model = meta[model_key]
    if not isinstance(stored_model, str):
        raise ValueError(
            f"Vector store '{collection}' metadata {model_key!r} must be a string"
        )
    expected_model = embedding_model_fingerprint(
        embedding_config,
        minimal_label_limit=minimal_label_limit,
        atom_scope=atom_scope,
    )
    if stored_dim != expected_meta_dim or stored_model != expected_model:
        raise EmbeddingContractMismatchError(
            f"Vector store '{collection}' embedding contract mismatch: "
            f"store has dimension={stored_dim}, model={stored_model!r}; "
            f"current config expects dimension={expected_meta_dim}, "
            f"model={expected_model!r}. " + embedding_contract_help()
        )