Skip to content

ontocast.agent.criticise_facts

Fact criticism agent.

This module provides functionality for analyzing and validating extracted facts.

Attributes

logger = logging.getLogger(__name__) module-attribute

Classes

Functions:

criticise_facts(state, tools) async

Critically analyze facts in the current content unit.

Parameters:

Name Type Description Default
state UnitFactsState

The current unit facts state containing the chunk to analyze.

required
tools AtomicToolBox

The toolbox instance providing utility functions.

required

Returns:

Name Type Description
UnitFactsState UnitFactsState

Updated state with analysis results.

Source code in ontocast/agent/criticise_facts.py
async def criticise_facts(
    state: UnitFactsState, tools: AtomicToolBox
) -> UnitFactsState:
    """Critically analyze facts in the current content unit.

    Args:
        state: The current unit facts state containing the chunk to analyze.
        tools: The toolbox instance providing utility functions.

    Returns:
        UnitFactsState: Updated state with analysis results.
    """
    if not state.content_unit:
        logger.warning("No current content unit to analyze")
        return state

    progress_info = state.get_content_unit_progress_string()
    logger.info(
        f"Facts critic for {progress_info}: visit {state.node_visits[WorkflowNode.CRITICISE_FACTS]}/{state.max_visits_per_node}"
    )

    llm_tool = await tools.get_llm_tool(state.budget_tracker)
    # Same profile the renderer used, chapter override included: the memoised
    # ontology chapter is only shared between the two calls when both ask for
    # the same syntax.
    profile = get_graph_format_profile(
        state.llm_graph_format,
        ontology_chapter_format=state.ontology_chapter_format,
        output_layout=state.llm_output_layout,
    )
    parser = PydanticOutputParser(pydantic_object=FactsCritiqueReport)

    ctx = ontology_access_for_unit_facts(state).effective_ontology_for_prompt()
    # Same chapter the renderer gets, index appendix included. Building it
    # without the suffix left the critic reading opaque IRIs while guideline 6a
    # told the renderer to resolve them through the TERM INDEX -- so the critic
    # judged term choices it could not read. Also memoised on the shared
    # snapshot, so this stops re-serialising the ontology on every visit.
    ontology_chapter = ctx.prompt_chapter(
        profile,
        max_triples=state.ontology_context_max_triples,
        text_caps=state.ontology_text_caps,
    )
    # Every statement gets a citable id, and the index is kept on the state so
    # the fixes that come back can be resolved by lookup. The critic used to be
    # asked to requote the statements it wanted changed, which it reproduces
    # correctly only a minority of the time -- for a bare removal, almost never.
    indexed_facts = profile.format_facts_chapter_indexed(state.content_unit.graph)
    state.prompt_triple_index = indexed_facts.index
    facts_chapter = indexed_facts.text + _build_quarantine_chapter(state)

    text_chapter = text_template.format(text=state.content_unit.extraction_text)

    user_instruction = (
        user_template.format(user_instruction=state.facts_user_instruction)
        if state.facts_user_instruction
        else ""
    )

    prompt = PromptTemplate(
        template=template_prompt,
        input_variables=[
            "preamble",
            "evaluation_instruction",
            "user_instruction",
            "ontology_chapter",
            "conformance_chapter",
            "facts_chapter",
            "text_chapter",
            "graph_format_instruction",
            "format_instructions",
        ],
    )

    graph_format_instruction = profile.critique_graph_instruction()
    web_search_enabled = tools.web_grounding_enabled_for_node(
        WorkflowNode.CRITICISE_FACTS
    )
    search_guidelines = search_guidelines_for(
        WorkflowNode.CRITICISE_FACTS, web_search_enabled
    )
    evaluation_instruction_str = evaluation_instruction
    if search_guidelines:
        evaluation_instruction_str = f"{evaluation_instruction}\n\n{search_guidelines}"

    prompt_data = {
        "preamble": preamble,
        "evaluation_instruction": evaluation_instruction_str,
        "user_instruction": user_instruction,
        "ontology_chapter": ontology_chapter,
        # Same rulebook the gate validates against; critique and render
        # share one contract.
        "conformance_chapter": state.conformance_chapter,
        "facts_chapter": facts_chapter,
        "text_chapter": text_chapter,
        "graph_format_instruction": graph_format_instruction,
        "format_instructions": profile.format_instructions(
            FactsCritiqueReport,
            web_search_enabled=web_search_enabled,
        ),
    }

    try:
        critique: FactsCritiqueReport = await call_llm_with_retry(
            llm_tool=llm_tool,
            prompt=prompt,
            parser=parser,
            prompt_kwargs=prompt_data,
            llm_graph_format=state.llm_graph_format,
        )
        persist_search_request(
            state,
            WorkflowNode.CRITICISE_FACTS,
            critique.external_evidence_request,
            web_search_enabled,
        )
        state.critic_outcome = "reviewed"

        logger.debug(
            f"Parsed critique report - success: {critique.success}, "
            f"score: {critique.score}"
        )

        # Acceptance is decided from defects that can be pointed at: the
        # deterministic findings already collected against this graph, plus the
        # critic's own fixes at the configured severity. `score` and `success`
        # are recorded and no longer consulted -- see acceptance.py for what the
        # score gate measured and why it could not be calibrated.
        defects = material_defects(
            state.deterministic_findings,
            critique.actionable_triple_fixes,
            tools.acceptance_policy,
        )
        reason = accept_reason(defects)

        state.attempt_log.append(
            LoopAttempt(
                render_attempt=state.node_visits[WorkflowNode.TEXT_TO_FACTS],
                critic_attempt=state.node_visits[WorkflowNode.CRITICISE_FACTS],
                kind="critic",
                score=critique.score,
                success=not defects,
                accept_reason=reason,
                n_actionable_fixes=len(critique.actionable_triple_fixes),
                severity_counts=Counter(
                    fix.severity for fix in critique.actionable_triple_fixes
                ),
                action_severity_counts=Counter(
                    f"{fix.action}:{fix.severity}"
                    for fix in critique.actionable_triple_fixes
                ),
                n_deterministic_findings=len(state.deterministic_findings),
                n_mandatory_findings=sum(
                    1 for finding in state.deterministic_findings if finding.mandatory
                ),
                triple_count=len(state.content_unit.graph),
            )
        )

        if not defects:
            state.status = Status.SUCCESS
            state.set_node_status(WorkflowNode.CRITICISE_FACTS, Status.SUCCESS)
            # Accepting means "no defect worth another render", NOT "the
            # critique was empty". The fixes are kept: the repair lane compiles
            # the mechanical ones for free and records the rest as residual.
            # Clearing them here used to discard the entire critique of every
            # accepted render -- the bulk of everything the critic produced,
            # since a REMOVE fix can never make a render blocking.
            state.suggestions = Suggestions.from_critique_report(critique)
            logger.info(
                "Facts critique passed (score %s, no material defect)",
                critique.score,
            )
        else:
            state.status = Status.FAILED
            state.set_node_status(WorkflowNode.CRITICISE_FACTS, Status.FAILED)
            state.failure_stage = FailureStage.FACTS_CRITIQUE
            state.suggestions = Suggestions.from_critique_report(critique)
            state.failure_reason = f"Facts unit has {len(defects)} material defect(s)"
            logger.info(
                "Facts critique rejected on %s: %s (score %s)",
                reason,
                "; ".join(defect.message for defect in defects[:3]),
                critique.score,
            )

        return state

    except LLMConfigurationError:
        # A rejected request is not a critic that failed to answer: the
        # next unit's critic will be rejected the same way.
        raise
    except Exception as e:
        # A critic that did not answer -- timeout, transport error, a response
        # that never parsed -- is not a critic that accepted. The unit leaves
        # the loop FAILED at the critique stage with its render intact, and
        # the attempt is on the record as a billed call that produced no
        # critique; the loop reads ``critic_outcome`` and applies no patch.
        logger.error(f"Failed to criticize facts: {str(e)}")
        state.critic_outcome = "unavailable"
        state.attempt_log.append(
            LoopAttempt(
                render_attempt=state.node_visits[WorkflowNode.TEXT_TO_FACTS],
                critic_attempt=state.node_visits[WorkflowNode.CRITICISE_FACTS],
                kind="critic",
                success=False,
                accept_reason="critic_unavailable",
                failure_stage=str(FailureStage.FACTS_CRITIQUE),
                failure_reason=str(e),
                n_deterministic_findings=len(state.deterministic_findings),
                n_mandatory_findings=sum(
                    1 for finding in state.deterministic_findings if finding.mandatory
                ),
                triple_count=len(state.content_unit.graph),
            )
        )
        state.set_failure(FailureStage.FACTS_CRITIQUE, str(e))
        state.set_node_status(WorkflowNode.CRITICISE_FACTS, Status.FAILED)
        return state