Skip to content

ontocast.stategraph.context_resolver

Attributes

logger = logging.getLogger(__name__) module-attribute

Classes

UnitOntologyContext

Bases: BaseModel

Assembled prompt context: snapshot view + writable catalog IRIs for apply.

Source code in ontocast/stategraph/context_resolver.py
class UnitOntologyContext(BaseModel):
    """Assembled prompt context: snapshot view + writable catalog IRIs for apply."""

    snapshot: OntologySnapshot
    writable_iris: list[str] = Field(default_factory=list)
    confidence: float = 0.0

    @property
    def assembly_mode(self) -> OntologyAssemblyMode:
        return self.snapshot.assembly_mode

    @property
    def patch_sources(self) -> list[str]:
        return list(self.snapshot.source_iris)

    @property
    def primary_writable_iri(self) -> str:
        """Primary catalog IRI for metrics (first writable, else null)."""
        if self.writable_iris:
            return self.writable_iris[0]
        return NULL_ONTOLOGY.iri

Attributes

assembly_mode property
confidence = 0.0 class-attribute instance-attribute
patch_sources property
primary_writable_iri property

Primary catalog IRI for metrics (first writable, else null).

snapshot instance-attribute
writable_iris = Field(default_factory=list) class-attribute instance-attribute

Functions:

aggregate_writable_metrics(unit_contexts)

Aggregate per-unit writable IRI / source / mode metrics.

Accepts either :class:UnitOntologyContext or legacy (primary_iri, patch_sources, mode) tuples for map-stage collect.

Source code in ontocast/stategraph/context_resolver.py
def aggregate_writable_metrics(
    unit_contexts: dict[int, UnitOntologyContext]
    | dict[int, tuple[str, list[str], OntologyAssemblyMode]],
) -> tuple[
    dict[int, str],
    dict[int, list[str]],
    dict[int, OntologyAssemblyMode],
    dict[str, int],
]:
    """Aggregate per-unit writable IRI / source / mode metrics.

    Accepts either :class:`UnitOntologyContext` or legacy
    ``(primary_iri, patch_sources, mode)`` tuples for map-stage collect.
    """
    unit_primary_assignment: dict[int, str] = {}
    unit_patch_sources: dict[int, list[str]] = {}
    unit_context_mode_used: dict[int, OntologyAssemblyMode] = {}
    primary_counts: Counter[str] = Counter()
    for unit_index, context in unit_contexts.items():
        if isinstance(context, tuple):
            primary_iri, patch_sources, assembly_mode = context
        else:
            primary_iri = context.primary_writable_iri
            patch_sources = context.patch_sources
            assembly_mode = context.assembly_mode
        unit_primary_assignment[unit_index] = primary_iri
        unit_patch_sources[unit_index] = patch_sources
        unit_context_mode_used[unit_index] = assembly_mode
        primary_counts[primary_iri] += 1
    return (
        unit_primary_assignment,
        unit_patch_sources,
        unit_context_mode_used,
        dict(primary_counts),
    )

build_merged_document_ontology_context(context)

Build merged ontology context from reduced document artifacts.

The result depends only on document-level state, so it should be computed once per document. "ctx/merge_document_ontology.calls" on the budget tracker exists to make a regression to per-unit calls visible.

Source code in ontocast/stategraph/context_resolver.py
def build_merged_document_ontology_context(
    context: UnitLoopContext,
) -> UnitOntologyContext | None:
    """Build merged ontology context from reduced document artifacts.

    The result depends only on document-level state, so it should be computed
    once per document. ``"ctx/merge_document_ontology.calls"`` on the budget
    tracker exists to make a regression to per-unit calls visible.
    """
    started = time.perf_counter()
    context.budget_tracker.incr("ctx/merge_document_ontology.calls")
    artifacts = [
        ontology
        for ontology in context.reduced_artifacts()
        if not ontology.is_null() and len(ontology.graph) > 0
    ]
    if not artifacts:
        context.budget_tracker.add_duration(
            "ctx/merge_document_ontology", time.perf_counter() - started
        )
        return None

    sorted_artifacts = sorted(artifacts, key=lambda ontology: ontology.iri or "")
    merged_graph = RDFGraph()
    patch_sources: list[str] = []
    for ontology in sorted_artifacts:
        merged_graph += ontology.graph
        if ontology.iri:
            patch_sources.append(ontology.iri)
    merged_graph.sanitize_prefixes_namespaces()

    snapshot = OntologySnapshot.from_graph(
        merged_graph,
        source_iris=patch_sources,
        assembly_mode=OntologyAssemblyMode.DOCUMENT_MERGED_REDUCED,
        title="Merged document ontology context",
        description=(
            "Deterministic merge of reduced ontology artifacts used for facts context."
        ),
        strip_headers=True,
    )
    context.budget_tracker.add_duration(
        "ctx/merge_document_ontology", time.perf_counter() - started
    )
    context.retrieval_metrics[RetrievalMetric.ONTOLOGY_SNAPSHOT_TRIPLES] = len(
        snapshot.graph
    )
    return UnitOntologyContext(
        snapshot=snapshot,
        writable_iris=list(patch_sources),
        confidence=1.0,
    )

build_unioned_document_ontology_context(context, tools, units) async

Resolve every unit's context once and union them into one shared context.

Per-unit retrieval gives each unit a smaller chapter than the union would be, and that is a real saving on the first call for a unit. It is a loss on every call after it: the chapter is the bulk of a facts prompt, no two units get the same one, and a provider's prefix cache can therefore serve none of them -- so a document pays the chapter once per unit at full price instead of once at full price and N-1 times at the cached rate.

Unioning is the trade that makes the second arrangement available. It is recall-safe by construction: the union contains every atom each unit's own retrieval selected, so no unit is shown less than it would have been. What it costs is precision -- a unit also sees its siblings' terms -- and the per-call token count, which is why it is a setting and not the default.

Retrieval runs concurrently, bounded the same way the unit fan-out is. In the vector modes this is embedding and graph work with no LLM call; the LLM-selection mode does spend one call per unit here, exactly as it would have spent inside the fan-out.

Parameters:

Name Type Description Default
context UnitLoopContext

Document-level loop inputs.

required
tools ToolBox

Toolbox holding the catalog and retrieval.

required
units Sequence[SourceUnit]

The document's content units.

required

Returns:

Type Description
UnitOntologyContext | None

The unioned context, or None when no unit resolved anything -- which

UnitOntologyContext | None

leaves the caller on the per-unit path rather than handing every unit an

UnitOntologyContext | None

empty snapshot.

Source code in ontocast/stategraph/context_resolver.py
async def build_unioned_document_ontology_context(
    context: UnitLoopContext,
    tools: ToolBox,
    units: Sequence[SourceUnit],
) -> UnitOntologyContext | None:
    """Resolve every unit's context once and union them into one shared context.

    Per-unit retrieval gives each unit a smaller chapter than the union would
    be, and that is a real saving on the *first* call for a unit. It is a loss
    on every call after it: the chapter is the bulk of a facts prompt, no two
    units get the same one, and a provider's prefix cache can therefore serve
    none of them -- so a document pays the chapter once per unit at full price
    instead of once at full price and N-1 times at the cached rate.

    Unioning is the trade that makes the second arrangement available. It is
    recall-safe by construction: the union contains every atom each unit's own
    retrieval selected, so no unit is shown less than it would have been. What
    it costs is precision -- a unit also sees its siblings' terms -- and the
    per-call token count, which is why it is a setting and not the default.

    Retrieval runs concurrently, bounded the same way the unit fan-out is. In
    the vector modes this is embedding and graph work with no LLM call; the
    LLM-selection mode does spend one call per unit here, exactly as it would
    have spent inside the fan-out.

    Args:
        context: Document-level loop inputs.
        tools: Toolbox holding the catalog and retrieval.
        units: The document's content units.

    Returns:
        The unioned context, or None when no unit resolved anything -- which
        leaves the caller on the per-unit path rather than handing every unit an
        empty snapshot.
    """
    if not units:
        return None
    started = time.perf_counter()
    context.budget_tracker.incr("ctx/union_document_ontology.calls")

    limit = max(1, tools.config.server.parallel_workers)
    semaphore = asyncio.Semaphore(limit)

    async def resolve(unit: SourceUnit) -> UnitOntologyContext | None:
        async with semaphore:
            try:
                return await resolve_unit_ontology_context(context, tools, unit)
            except EmptyOntologyContextError:
                # A unit whose own context is empty must not void the document's:
                # the guard exists to catch a catalog that did not load, and the
                # union is exactly the evidence that it did.
                return None

    resolved = await asyncio.gather(*(resolve(unit) for unit in units))

    merged_graph = RDFGraph()
    writable: list[str] = []
    sources: list[str] = []
    for ctx in resolved:
        if ctx is None or not len(ctx.snapshot.graph):
            continue
        merged_graph += ctx.snapshot.graph
        writable.extend(ctx.writable_iris)
        sources.extend(ctx.snapshot.source_iris)
    if not len(merged_graph):
        context.budget_tracker.add_duration(
            "ctx/union_document_ontology", time.perf_counter() - started
        )
        return None

    merged_graph.sanitize_prefixes_namespaces()
    snapshot = OntologySnapshot.from_graph(
        merged_graph,
        # Sorted and de-duplicated so the same document renders the same
        # chapter twice: an unstable source order is an unstable prompt, and an
        # unstable prompt is the thing this whole path exists to avoid.
        source_iris=sorted(set(sources)),
        assembly_mode=OntologyAssemblyMode.DOCUMENT_MERGED_REDUCED,
        title="Unioned document ontology context",
        description=(
            "Union of the per-unit retrieved contexts, shared by every unit so "
            "the ontology chapter is identical across the fan-out."
        ),
        strip_headers=True,
    )
    context.budget_tracker.add_duration(
        "ctx/union_document_ontology", time.perf_counter() - started
    )
    context.retrieval_metrics[RetrievalMetric.ONTOLOGY_SNAPSHOT_TRIPLES] = len(
        snapshot.graph
    )
    logger.info(
        "Unioned ontology context over %d unit(s): %d triples, %d source(s).",
        len(units),
        len(snapshot.graph),
        len(snapshot.source_iris),
    )
    return UnitOntologyContext(
        snapshot=snapshot,
        writable_iris=sorted(set(writable)),
        confidence=1.0,
    )

resolve_unit_ontology_context(context, tools, unit, *, can_create_vocabulary=False) async

Assemble the ontology context one content unit is rendered against.

Parameters:

Name Type Description Default
context UnitLoopContext

Document-level loop inputs.

required
tools ToolBox

Toolbox holding the catalog and retrieval.

required
unit SourceUnit

The content unit being rendered.

required
can_create_vocabulary bool

Whether the caller can act on an empty context by inventing vocabulary. True for the ontology loop, which answers an empty seed with render_ontology_fresh; false for the facts loop, which can only fall back on generic terms.

False

Returns:

Type Description
UnitOntologyContext

The resolved context, possibly empty.

Raises:

Type Description
EmptyOntologyContextError

The context is empty, the caller cannot create vocabulary, and this deployment requires a context.

Source code in ontocast/stategraph/context_resolver.py
async def resolve_unit_ontology_context(
    context: UnitLoopContext,
    tools: ToolBox,
    unit: SourceUnit,
    *,
    can_create_vocabulary: bool = False,
) -> UnitOntologyContext:
    """Assemble the ontology context one content unit is rendered against.

    Args:
        context: Document-level loop inputs.
        tools: Toolbox holding the catalog and retrieval.
        unit: The content unit being rendered.
        can_create_vocabulary: Whether the caller can act on an empty context
            by inventing vocabulary. True for the ontology loop, which answers
            an empty seed with ``render_ontology_fresh``; false for the facts
            loop, which can only fall back on generic terms.

    Returns:
        The resolved context, possibly empty.

    Raises:
        EmptyOntologyContextError: The context is empty, the caller cannot
            create vocabulary, and this deployment requires a context.
    """
    mode = context.ontology_context_mode
    context.retrieval_metrics[RetrievalMetric.ONTOLOGY_CONTEXT_MODE] = mode.value
    if mode == OntologyContextMode.SELECTED_SINGLE_ONTOLOGY:
        resolved = await _resolve_selected_single_ontology_context(context, tools, unit)
    elif mode == OntologyContextMode.FIXED_SINGLE_ONTOLOGY:
        resolved = await _resolve_fixed_single_ontology_context(context, tools, unit)
    elif mode == OntologyContextMode.SELECTED_VECTOR_SEARCH_ONTOLOGY:
        require_vector_retrieval(tools)
        resolved = await _resolve_ensemble_context(context, tools, unit)
    else:
        raise ValueError(f"Unknown ontology_context_mode: {mode!r}")
    # Recorded here rather than in each resolver so no mode can be added without
    # a size: the two that bound nothing were also the two that reported nothing.
    context.retrieval_metrics[RetrievalMetric.ONTOLOGY_SNAPSHOT_TRIPLES] = len(
        resolved.snapshot.graph
    )
    # Checked here, not per mode, for the same reason the size is recorded here:
    # every mode can return an empty context, and the two that bound nothing
    # were also the two that reported nothing.
    #
    # Two exemptions, both because an empty context is not a fault for them:
    #
    # * A unit with no retrievable text has nothing to extract either way.
    # * A caller that can create vocabulary. The ontology renderer branches on
    #   exactly this condition -- an empty seed sends it to
    #   ``render_ontology_fresh``, which mints a new catalog ontology from the
    #   text -- so raising here made the one path designed for an empty catalog
    #   unreachable, and turned "this corpus has no ontology yet" into a
    #   deployment error. It also stopped a populated-catalog run whenever the
    #   selector honestly reported that no catalog ontology fits.
    if (
        not len(resolved.snapshot.graph)
        and unit.text.strip()
        and not can_create_vocabulary
        and tools.config.server.ontology_context_required
    ):
        reason = context.retrieval_metrics.get(
            RetrievalMetric.EMPTY_SNAPSHOT_REASON, "no ontology context was assembled"
        )
        raise EmptyOntologyContextError(
            f"Ontology context for this content unit is empty: {reason}. "
            "Extraction would fall back on generic vocabulary and the "
            "conformance gate would then have no node to constrain, reporting "
            "a vacuous pass. Fix the catalog, or set "
            "ONTOLOGY_CONTEXT_REQUIRED=false to extract without one "
            "deliberately."
        )
    if not len(resolved.snapshot.graph) and can_create_vocabulary:
        logger.info(
            "No ontology context for this unit; rendering a fresh ontology from "
            "its text"
        )
    return resolved