Skip to content

ontocast.agent.render_facts

Fact rendering agent for OntoCast.

This module provides functionality for rendering facts from RDF graphs into human-readable formats, making the extracted knowledge more accessible and understandable.

Attributes

logger = logging.getLogger(__name__) module-attribute

Classes

Functions:

render_facts(state, tools, supplemental_ontologies=None) async

Extract a unit's facts from its text.

There is one renderer because there is one render. The loop retries this only when it fails; a unit that rendered successfully is improved by the critic passes, which apply a compiled patch rather than asking for the graph to be written again. The update-mode renderer this used to dispatch to had become unreachable: it keyed on the unit graph being non-empty, and nothing populates that except a successful render, after which the render loop is already done.

The ontology renderer still has both modes, and for a reason that does not apply here: it keys on the retrieved snapshot, a different field from the one it writes, so its update mode is the normal path whenever a catalog exists.

Parameters:

Name Type Description Default
state UnitFactsState

The current unit facts state.

required
tools AtomicToolBox

The toolbox containing necessary tools.

required
supplemental_ontologies Sequence[Ontology] | None

Extra ontologies for prefix resolution.

None

Returns:

Name Type Description
UnitFactsState UnitFactsState

Updated state with rendered facts.

Source code in ontocast/agent/render_facts.py
async def render_facts(
    state: UnitFactsState,
    tools: AtomicToolBox,
    supplemental_ontologies: Sequence[Ontology] | None = None,
) -> UnitFactsState:
    """Extract a unit's facts from its text.

    There is one renderer because there is one render. The loop retries this
    only when it *fails*; a unit that rendered successfully is improved by the
    critic passes, which apply a compiled patch rather than asking for the graph
    to be written again. The update-mode renderer this used to dispatch to had
    become unreachable: it keyed on the unit graph being non-empty, and nothing
    populates that except a successful render, after which the render loop is
    already done.

    The ontology renderer still has both modes, and for a reason that does not
    apply here: it keys on the retrieved *snapshot*, a different field from the
    one it writes, so its update mode is the normal path whenever a catalog
    exists.

    Args:
        state: The current unit facts state.
        tools: The toolbox containing necessary tools.
        supplemental_ontologies: Extra ontologies for prefix resolution.

    Returns:
        UnitFactsState: Updated state with rendered facts.
    """
    logger.info(f"Render facts for {state.get_content_unit_progress_string()}")
    return await render_facts_fresh(
        state, tools, supplemental_ontologies=list(supplemental_ontologies or ())
    )

render_facts_fresh(state, tools, supplemental_ontologies=None) async

Render fresh facts from the current chunk into Turtle format.

Parameters:

Name Type Description Default
state UnitFactsState

The current unit facts state containing the chunk to render.

required
tools AtomicToolBox

The toolbox instance providing utility functions.

required

Returns:

Name Type Description
UnitFactsState UnitFactsState

Updated state with rendered facts.

Source code in ontocast/agent/render_facts.py
async def render_facts_fresh(
    state: UnitFactsState,
    tools: AtomicToolBox,
    supplemental_ontologies: Sequence[Ontology] | None = None,
) -> UnitFactsState:
    """Render fresh facts from the current chunk into Turtle format.

    Args:
        state: The current unit facts state containing the chunk to render.
        tools: The toolbox instance providing utility functions.

    Returns:
        UnitFactsState: Updated state with rendered facts.
    """
    logger.info("Rendering fresh facts")
    state.quarantined_literal_triples = []
    llm_tool = await tools.get_llm_tool(state.budget_tracker)
    profile = get_graph_format_profile(
        state.llm_graph_format,
        ontology_chapter_format=state.ontology_chapter_format,
        output_layout=state.llm_output_layout,
    )
    parser = PydanticOutputParser(pydantic_object=FactsRenderReport)

    access = ontology_access_for_unit_facts(state)

    known_prefixes = build_llm_prefix_map(
        access.ontology_for_prefixes(),
        supplemental_ontologies or (),
    )

    web_search_enabled = tools.web_grounding_enabled_for_node(
        WorkflowNode.TEXT_TO_FACTS
    )
    prompt_data = _prepare_prompt_data(
        state,
        access,
        profile,
        citation_vocabulary=tools.citation_vocabulary,
        quantity_fallback_vocabulary=tools.quantity_fallback_vocabulary,
        search_guidelines=search_guidelines_for(
            WorkflowNode.TEXT_TO_FACTS, web_search_enabled
        ),
    )
    prompt_data_fresh = {
        "preamble": preamble,
        "output_instruction": profile.render_fresh_output_instruction(target="facts"),
    }
    prompt_data.update(prompt_data_fresh)

    prompt = _create_prompt_template()

    previous_prefixes = RDFGraph.get_known_prefixes()
    try:
        # Set known prefixes in context before parsing
        RDFGraph.set_known_prefixes(known_prefixes if known_prefixes else None)

        render_report: FactsRenderReport = await call_llm_with_retry(
            llm_tool=llm_tool,
            prompt=prompt,
            parser=parser,
            prompt_kwargs={
                "format_instructions": profile.format_instructions(
                    FactsRenderReport,
                    web_search_enabled=web_search_enabled,
                ),
                **prompt_data,
            },
            llm_graph_format=state.llm_graph_format,
        )
        persist_search_request(
            state,
            WorkflowNode.TEXT_TO_FACTS,
            render_report.external_evidence_request,
            web_search_enabled,
        )
        render_report.semantic_graph.sanitize_prefixes_namespaces()
        clean_graph, rejected = finalize_llm_graph(render_report.semantic_graph)
        ontology_context_graph = access.effective_ontology_for_prompt().graph
        clean_graph, repair_records = _normalize_and_repair_graph(
            clean_graph,
            ontology_context_graph,
            tools,
            budget_tracker=state.budget_tracker,
        )
        state.applied_repairs.extend(repair_records)
        if tools.object_property_literal_check:
            clean_graph, op_rejected = partition_object_property_literal_triples(
                clean_graph, ontology_context_graph
            )
            rejected = rejected + op_rejected
        state.content_unit.graph = clean_graph
        state.quarantined_literal_triples = rejected
        if rejected:
            logger.warning(
                "Fresh facts quarantined %d triple(s) with invalid literals",
                len(rejected),
            )

        # Track triples in budget tracker (fresh facts)
        num_triples = len(clean_graph)
        logger.info(f"Fresh facts generated with {num_triples} triple(s).")
        state.budget_tracker.add_facts_update(num_operations=1, num_triples=num_triples)

        state.clear_failure()
        state.set_node_status(WorkflowNode.TEXT_TO_FACTS, Status.SUCCESS)
        return state

    except LLMConfigurationError:
        # The provider rejects the request itself, not this attempt at
        # it: every other unit is about to be rejected identically.
        # Failing one unit here turns a configuration fault into an
        # empty run that reports success.
        raise
    except Exception as e:
        return _handle_rendering_error(state, e, FailureStage.GENERATE_TTL_FOR_FACTS)
    finally:
        # Restore the loop-level catalog map (or whatever was there), rather
        # than clearing to None — the critic and completion passes still need
        # it after this render returns.
        RDFGraph.set_known_prefixes(previous_prefixes)