async def criticise_facts(
state: UnitFactsState, tools: AtomicToolBox
) -> UnitFactsState:
"""Critically analyze facts in the current content unit.
Args:
state: The current unit facts state containing the chunk to analyze.
tools: The toolbox instance providing utility functions.
Returns:
UnitFactsState: Updated state with analysis results.
"""
if not state.content_unit:
logger.warning("No current content unit to analyze")
return state
progress_info = state.get_content_unit_progress_string()
logger.info(
f"Facts critic for {progress_info}: visit {state.node_visits[WorkflowNode.CRITICISE_FACTS]}/{state.max_visits_per_node}"
)
llm_tool = await tools.get_llm_tool(state.budget_tracker)
# Same profile the renderer used, chapter override included: the memoised
# ontology chapter is only shared between the two calls when both ask for
# the same syntax.
profile = get_graph_format_profile(
state.llm_graph_format,
ontology_chapter_format=state.ontology_chapter_format,
output_layout=state.llm_output_layout,
)
parser = PydanticOutputParser(pydantic_object=FactsCritiqueReport)
ctx = ontology_access_for_unit_facts(state).effective_ontology_for_prompt()
# Same chapter the renderer gets, index appendix included. Building it
# without the suffix left the critic reading opaque IRIs while guideline 6a
# told the renderer to resolve them through the TERM INDEX -- so the critic
# judged term choices it could not read. Also memoised on the shared
# snapshot, so this stops re-serialising the ontology on every visit.
ontology_chapter = ctx.prompt_chapter(
profile,
max_triples=state.ontology_context_max_triples,
text_caps=state.ontology_text_caps,
)
# Every statement gets a citable id, and the index is kept on the state so
# the fixes that come back can be resolved by lookup. The critic used to be
# asked to requote the statements it wanted changed, which it reproduces
# correctly only a minority of the time -- for a bare removal, almost never.
indexed_facts = profile.format_facts_chapter_indexed(state.content_unit.graph)
state.prompt_triple_index = indexed_facts.index
facts_chapter = indexed_facts.text + _build_quarantine_chapter(state)
text_chapter = text_template.format(text=state.content_unit.extraction_text)
user_instruction = (
user_template.format(user_instruction=state.facts_user_instruction)
if state.facts_user_instruction
else ""
)
prompt = PromptTemplate(
template=template_prompt,
input_variables=[
"preamble",
"evaluation_instruction",
"user_instruction",
"ontology_chapter",
"conformance_chapter",
"facts_chapter",
"text_chapter",
"graph_format_instruction",
"format_instructions",
],
)
graph_format_instruction = profile.critique_graph_instruction()
web_search_enabled = tools.web_grounding_enabled_for_node(
WorkflowNode.CRITICISE_FACTS
)
search_guidelines = search_guidelines_for(
WorkflowNode.CRITICISE_FACTS, web_search_enabled
)
evaluation_instruction_str = evaluation_instruction
if search_guidelines:
evaluation_instruction_str = f"{evaluation_instruction}\n\n{search_guidelines}"
prompt_data = {
"preamble": preamble,
"evaluation_instruction": evaluation_instruction_str,
"user_instruction": user_instruction,
"ontology_chapter": ontology_chapter,
# Same rulebook the gate validates against; critique and render
# share one contract.
"conformance_chapter": state.conformance_chapter,
"facts_chapter": facts_chapter,
"text_chapter": text_chapter,
"graph_format_instruction": graph_format_instruction,
"format_instructions": profile.format_instructions(
FactsCritiqueReport,
web_search_enabled=web_search_enabled,
),
}
try:
critique: FactsCritiqueReport = await call_llm_with_retry(
llm_tool=llm_tool,
prompt=prompt,
parser=parser,
prompt_kwargs=prompt_data,
llm_graph_format=state.llm_graph_format,
)
persist_search_request(
state,
WorkflowNode.CRITICISE_FACTS,
critique.external_evidence_request,
web_search_enabled,
)
state.critic_outcome = "reviewed"
logger.debug(
f"Parsed critique report - success: {critique.success}, "
f"score: {critique.score}"
)
# Acceptance is decided from defects that can be pointed at: the
# deterministic findings already collected against this graph, plus the
# critic's own fixes at the configured severity. `score` and `success`
# are recorded and no longer consulted -- see acceptance.py for what the
# score gate measured and why it could not be calibrated.
defects = material_defects(
state.deterministic_findings,
critique.actionable_triple_fixes,
tools.acceptance_policy,
)
reason = accept_reason(defects)
state.attempt_log.append(
LoopAttempt(
render_attempt=state.node_visits[WorkflowNode.TEXT_TO_FACTS],
critic_attempt=state.node_visits[WorkflowNode.CRITICISE_FACTS],
kind="critic",
score=critique.score,
success=not defects,
accept_reason=reason,
n_actionable_fixes=len(critique.actionable_triple_fixes),
severity_counts=Counter(
fix.severity for fix in critique.actionable_triple_fixes
),
action_severity_counts=Counter(
f"{fix.action}:{fix.severity}"
for fix in critique.actionable_triple_fixes
),
n_deterministic_findings=len(state.deterministic_findings),
n_mandatory_findings=sum(
1 for finding in state.deterministic_findings if finding.mandatory
),
triple_count=len(state.content_unit.graph),
)
)
if not defects:
state.status = Status.SUCCESS
state.set_node_status(WorkflowNode.CRITICISE_FACTS, Status.SUCCESS)
# Accepting means "no defect worth another render", NOT "the
# critique was empty". The fixes are kept: the repair lane compiles
# the mechanical ones for free and records the rest as residual.
# Clearing them here used to discard the entire critique of every
# accepted render -- the bulk of everything the critic produced,
# since a REMOVE fix can never make a render blocking.
state.suggestions = Suggestions.from_critique_report(critique)
logger.info(
"Facts critique passed (score %s, no material defect)",
critique.score,
)
else:
state.status = Status.FAILED
state.set_node_status(WorkflowNode.CRITICISE_FACTS, Status.FAILED)
state.failure_stage = FailureStage.FACTS_CRITIQUE
state.suggestions = Suggestions.from_critique_report(critique)
state.failure_reason = f"Facts unit has {len(defects)} material defect(s)"
logger.info(
"Facts critique rejected on %s: %s (score %s)",
reason,
"; ".join(defect.message for defect in defects[:3]),
critique.score,
)
return state
except LLMConfigurationError:
# A rejected request is not a critic that failed to answer: the
# next unit's critic will be rejected the same way.
raise
except Exception as e:
# A critic that did not answer -- timeout, transport error, a response
# that never parsed -- is not a critic that accepted. The unit leaves
# the loop FAILED at the critique stage with its render intact, and
# the attempt is on the record as a billed call that produced no
# critique; the loop reads ``critic_outcome`` and applies no patch.
logger.error(f"Failed to criticize facts: {str(e)}")
state.critic_outcome = "unavailable"
state.attempt_log.append(
LoopAttempt(
render_attempt=state.node_visits[WorkflowNode.TEXT_TO_FACTS],
critic_attempt=state.node_visits[WorkflowNode.CRITICISE_FACTS],
kind="critic",
success=False,
accept_reason="critic_unavailable",
failure_stage=str(FailureStage.FACTS_CRITIQUE),
failure_reason=str(e),
n_deterministic_findings=len(state.deterministic_findings),
n_mandatory_findings=sum(
1 for finding in state.deterministic_findings if finding.mandatory
),
triple_count=len(state.content_unit.graph),
)
)
state.set_failure(FailureStage.FACTS_CRITIQUE, str(e))
state.set_node_status(WorkflowNode.CRITICISE_FACTS, Status.FAILED)
return state