Skip to content

ontocast.config.settings

Configuration management for OntoCast.

This module provides hierarchical configuration classes that map to the environment variables and usage patterns in the OntoCast system.

Attributes

CONVERTER_PROFILE_PRESETS = {'fast': {'do_ocr': False, 'table_mode': 'fast', 'do_formula_enrichment': False}, 'lean': {'do_ocr': False, 'table_mode': 'fast', 'do_formula_enrichment': True}, 'ocr': {'do_ocr': True, 'table_mode': 'accurate', 'do_formula_enrichment': False}} module-attribute

DEFAULT_CONVERTER_EXTENSIONS = ('.pdf', '.docx', '.pptx', '.xlsx', '.html', '.htm', '.md', '.csv', '.adoc', '.png', '.jpg', '.jpeg', '.tif', '.tiff') module-attribute

LLMModelName = OpenAIModel | OllamaModel | ClaudeModel | GeminiModel | str module-attribute

READ_WITHOUT_CONVERTER_EXTENSIONS = frozenset({'.txt', '.json', '.jsonl'}) module-attribute

__all__ = ['OntologyValidationConfig', 'AggregationConfig', 'ChunkConfig', 'ClaudeModel', 'Config', 'ConverterConfig', 'CrossQueryMergeMode', 'DomainConfig', 'EmbeddingConfig', 'EmbeddingProvider', 'FactsValidationConfig', 'FusekiConfig', 'GeminiModel', 'InducedSubgraphSeedOrder', 'LLMConfig', 'LLMModelName', 'LLMModelNameAbstract', 'LLMProvider', 'LanceDBConfig', 'LexicalTriggerFusion', 'OllamaModel', 'OpenAIModel', 'PatchRetrievalConfig', 'PathConfig', 'QdrantConfig', 'ServerConfig', 'SiblingGuardScope', 'SymbolCaseMismatchPolicy', 'ToolConfig', 'VectorStoreConfig', 'VectorStoreDedupMode', 'WebSearchConfig', 'WebSearchProvider'] module-attribute

logger = logging.getLogger(__name__) module-attribute

Classes

AggregationConfig

Bases: BaseSettings

Aggregation settings for entity clustering/disambiguation.

Source code in ontocast/config/settings.py
class AggregationConfig(BaseSettings):
    """Aggregation settings for entity clustering/disambiguation."""

    embedding_model: str = Field(
        default="sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
        description=(
            "Sentence-transformers model name used for entity embeddings. "
            "Spelled with the org prefix to match CHUNK_EMBEDDING_MODEL and "
            "EMBEDDING_MODEL_NAME: SharedEncoder keys its process-wide cache on "
            "the literal string, so the same checkpoint written two ways loads "
            "twice."
        ),
    )
    similarity_threshold: float = Field(
        default=0.80,
        ge=0.0,
        le=1.0,
        description=(
            "Cosine threshold of the cross-graph entity aligner when the caller "
            "names none: POST /match/entities, match-graphs and the "
            "ontocast_align_entities agent tool. The in-pipeline aggregator uses "
            "AGG_CANDIDATE_SIMILARITY_THRESHOLD; this setting does not affect it."
        ),
    )
    candidate_similarity_threshold: float = Field(
        default=0.70,
        ge=0.0,
        le=1.0,
        description=(
            "Cosine threshold of the in-pipeline aggregator: DBSCAN candidate "
            "clustering and the pairwise gate both use it. Deliberately "
            "permissive — candidates are validated symbolically afterwards."
        ),
    )
    literal_conflict_guard: bool = Field(
        default=True,
        description=(
            "Veto identity merges between entities asserting disjoint literal "
            "values on a shared predicate (numeric/temporal disjointness, or "
            "string sets with no compatible cross-pair). Turning it off "
            "isolates this guard's contribution to rejected merges."
        ),
    )
    initials_distinct_guard: bool = Field(
        default=True,
        description=(
            "Veto identity merges between entities whose labels are identical "
            "except for conflicting initials or single-letter identifiers "
            "('company S.' vs 'company T.') — the shape authors write to "
            "distinguish entities."
        ),
    )
    natural_key_merge: bool = Field(
        default=True,
        description=(
            "Positive identity evidence from natural keys: instances sharing "
            "an identical short string value on a single-valued "
            "identifier-like predicate (schema max-1, or observed "
            "single-valued on every subject) become merge candidates even "
            "when their labels and embeddings disagree. All distinctness "
            "guards still apply."
        ),
    )
    type_guard_untyped: Literal["permissive", "strict"] = Field(
        default="permissive",
        description=(
            "Type-compatibility guard behaviour for untyped entities. "
            "'permissive' (default) lets a typed entity merge with an untyped "
            "one; 'strict' fails typed-vs-untyped pairs closed (two untyped "
            "entities stay comparable in both modes)."
        ),
    )
    lexical_label_jaccard: float = Field(
        default=0.5,
        ge=0.0,
        le=1.0,
        description=(
            "Minimum label token-set Jaccard for the fuzzy lexical-alias merge tier."
        ),
    )
    lexical_sequence_ratio: float = Field(
        default=0.90,
        ge=0.0,
        le=1.0,
        description=(
            "Minimum SequenceMatcher ratio on URI normal forms for the fuzzy "
            "lexical-alias merge tier."
        ),
    )
    lexical_token_jaccard: float = Field(
        default=0.75,
        ge=0.0,
        le=1.0,
        description=(
            "Minimum normal-form token Jaccard for the fuzzy lexical-alias "
            "merge tier (both sides >= 2 tokens)."
        ),
    )
    functional_min_empirical_support: int = Field(
        default=2,
        ge=1,
        description=(
            "Minimum distinct subjects a predicate must be observed on "
            "before it counts as empirically single-valued for the "
            "functional-object merge guard."
        ),
    )
    sibling_guard_scope: SiblingGuardScope = Field(
        default=SiblingGuardScope.SUBJECT,
        description=(
            "Co-object sibling guard scope: 'subject' forbids merging any "
            "two objects of one subject; 'predicate' restricts the "
            "prohibition to objects sharing the same predicate."
        ),
    )
    unit_scoped_fact_iris: bool = Field(
        default=True,
        description=(
            "Suffix every minted fact IRI with the index of the unit that "
            "minted it (<local>__u<index>) before aggregation. Units mint "
            "instance IRIs independently, so without this the same local "
            "name from two units is one node before any merge guard runs, "
            "and the validation gate cannot split a singleton. With it, the "
            "pair is a merge candidate like any alias pair; final IRIs never "
            "carry the suffix. Off reproduces name-keyed fusion."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="AGG_",
        case_sensitive=False,
    )

Attributes

candidate_similarity_threshold = Field(default=0.7, ge=0.0, le=1.0, description='Cosine threshold of the in-pipeline aggregator: DBSCAN candidate clustering and the pairwise gate both use it. Deliberately permissive — candidates are validated symbolically afterwards.') class-attribute instance-attribute
embedding_model = Field(default='sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2', description='Sentence-transformers model name used for entity embeddings. Spelled with the org prefix to match CHUNK_EMBEDDING_MODEL and EMBEDDING_MODEL_NAME: SharedEncoder keys its process-wide cache on the literal string, so the same checkpoint written two ways loads twice.') class-attribute instance-attribute
functional_min_empirical_support = Field(default=2, ge=1, description='Minimum distinct subjects a predicate must be observed on before it counts as empirically single-valued for the functional-object merge guard.') class-attribute instance-attribute
initials_distinct_guard = Field(default=True, description="Veto identity merges between entities whose labels are identical except for conflicting initials or single-letter identifiers ('company S.' vs 'company T.') — the shape authors write to distinguish entities.") class-attribute instance-attribute
lexical_label_jaccard = Field(default=0.5, ge=0.0, le=1.0, description='Minimum label token-set Jaccard for the fuzzy lexical-alias merge tier.') class-attribute instance-attribute
lexical_sequence_ratio = Field(default=0.9, ge=0.0, le=1.0, description='Minimum SequenceMatcher ratio on URI normal forms for the fuzzy lexical-alias merge tier.') class-attribute instance-attribute
lexical_token_jaccard = Field(default=0.75, ge=0.0, le=1.0, description='Minimum normal-form token Jaccard for the fuzzy lexical-alias merge tier (both sides >= 2 tokens).') class-attribute instance-attribute
literal_conflict_guard = Field(default=True, description="Veto identity merges between entities asserting disjoint literal values on a shared predicate (numeric/temporal disjointness, or string sets with no compatible cross-pair). Turning it off isolates this guard's contribution to rejected merges.") class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='AGG_', case_sensitive=False) class-attribute instance-attribute
natural_key_merge = Field(default=True, description='Positive identity evidence from natural keys: instances sharing an identical short string value on a single-valued identifier-like predicate (schema max-1, or observed single-valued on every subject) become merge candidates even when their labels and embeddings disagree. All distinctness guards still apply.') class-attribute instance-attribute
sibling_guard_scope = Field(default=SiblingGuardScope.SUBJECT, description="Co-object sibling guard scope: 'subject' forbids merging any two objects of one subject; 'predicate' restricts the prohibition to objects sharing the same predicate.") class-attribute instance-attribute
similarity_threshold = Field(default=0.8, ge=0.0, le=1.0, description='Cosine threshold of the cross-graph entity aligner when the caller names none: POST /match/entities, match-graphs and the ontocast_align_entities agent tool. The in-pipeline aggregator uses AGG_CANDIDATE_SIMILARITY_THRESHOLD; this setting does not affect it.') class-attribute instance-attribute
type_guard_untyped = Field(default='permissive', description="Type-compatibility guard behaviour for untyped entities. 'permissive' (default) lets a typed entity merge with an untyped one; 'strict' fails typed-vs-untyped pairs closed (two untyped entities stay comparable in both modes).") class-attribute instance-attribute
unit_scoped_fact_iris = Field(default=True, description='Suffix every minted fact IRI with the index of the unit that minted it (<local>__u<index>) before aggregation. Units mint instance IRIs independently, so without this the same local name from two units is one node before any merge guard runs, and the validation gate cannot split a singleton. With it, the pair is a merge candidate like any alias pair; final IRIs never carry the suffix. Off reproduces name-keyed fusion.') class-attribute instance-attribute

ChunkConfig

Bases: BaseSettings

Chunking configuration settings.

Source code in ontocast/config/settings.py
class ChunkConfig(BaseSettings):
    """Chunking configuration settings."""

    min_size: int = Field(
        default=3000,
        description=(
            "Smallest chunk in characters. Shorter neighbouring pieces are "
            "merged until they reach it, without passing CHUNK_MAX_SIZE."
        ),
    )
    max_size: int = Field(
        default=12000,
        description=(
            "Largest chunk in characters. Raising it gives each LLM call more "
            "context and makes fewer calls; lower it if the model loses track "
            "of long chunks."
        ),
    )
    embedding_model: str = Field(
        default="sentence-transformers/paraphrase-multilingual-mpnet-base-v2",
        description=(
            "Sentence-transformers checkpoint for semantic chunking and "
            "embedding-based schema detection. Shared process-wide with "
            "EMBEDDING_MODEL_NAME and AGG_EMBEDDING_MODEL when the names match, "
            "so aligning all three halves resident local-model memory. Changing "
            "it invalidates the on-disk chunk cache and shifts chunk boundaries."
        ),
    )
    segmenter: Literal["semantic", "docling"] = Field(
        default="semantic",
        description=(
            "Primary segmenter: 'semantic' splits the markdown export inside "
            "detected section boundaries with the built-in semantic chunker "
            "(naive fallback without torch extras); 'docling' uses docling's "
            "HybridChunker structural segments."
        ),
    )
    section_classifier: Literal["llm", "heuristic", "heading", "off"] = Field(
        default="heuristic",
        description=(
            "Chunk section classification cascade, in increasing cost: "
            "'off' = no section tagging (disables section filters and schema "
            "default exclusions); 'heading' = document outline plus heading "
            "pattern/keyword matching; 'heuristic' (default) = heading plus "
            "content-density classification for regions with no usable "
            "heading; 'llm' = heuristic plus a batched LLM pass over whatever "
            "remains unlabeled. Only 'llm' makes LLM calls during chunking."
        ),
    )
    section_tag_min_chars: int = Field(
        default=80,
        description=(
            "Min stripped length for LLM section tagging; smaller segments merge "
            "into neighbors before tagging"
        ),
    )
    section_text_headings: bool = Field(
        default=True,
        description=(
            "Detect headings from plain-text layout (short, blank-line "
            "delimited, upper-case or numbered lines) in documents whose "
            "conversion produced no markdown heading structure at all."
        ),
    )
    section_density: Literal["off", "conservative", "aggressive"] = Field(
        default="conservative",
        description=(
            "Content-density section classification for regions with no usable "
            "heading. 'conservative' (default) recognises only reference lists "
            "and acknowledgements, whose surface form is near-unique. "
            "'aggressive' also guesses methods/results/introduction from "
            "figure-reference, quantity and citation densities -- these do not "
            "separate those sections cleanly, and a wrong label is silently "
            "acted on by the section filters, so it is opt-in. Requires "
            "CHUNK_SECTION_CLASSIFIER=heuristic or llm."
        ),
    )
    section_schema_detect: Literal["off", "lexical", "headings", "auto"] = Field(
        default="headings",
        description=(
            "How to infer the document-type schema when the request names "
            "none and its document_type_hint matches none: 'off' uses the "
            "manifest default; 'lexical' scores headings against each "
            "schema's vocabulary; 'headings' adds an embedding tier when the "
            "semantic extras are installed; 'auto' also allows a weaker "
            "content-based tier for documents with almost no headings. An "
            "explicit schema or a matching hint always wins, and detection "
            "falls back to the default rather than guess."
        ),
    )
    section_schema_detect_min_score: float = Field(
        default=2.0,
        description=(
            "Minimum distinctive evidence (headings recognised by exactly one "
            "candidate schema) before a detection is accepted."
        ),
    )
    section_schema_detect_min_margin: float = Field(
        default=1.8,
        description=(
            "Factor by which the winning schema's score must exceed the "
            "runner-up's; below it detection falls back to the default "
            "schema. Raise it to detect less often and more surely."
        ),
    )
    section_schema_detect_content_min_margin: float = Field(
        default=4.0,
        description=(
            "Stricter margin for the content-based tier, which is measurably "
            "less reliable than the heading tiers: body prose from one domain "
            "readily resembles another (scientific prose reads like a technical "
            "specification). Only used when CHUNK_SECTION_SCHEMA_DETECT=auto."
        ),
    )
    section_llm_batch_size: int = Field(
        default=40,
        description=(
            "Excerpts per LLM call when CHUNK_SECTION_CLASSIFIER=llm. One call "
            "covers a whole document's residual instead of one call per chunk; "
            "0 restores per-chunk calls."
        ),
    )
    section_filter_on_empty: Literal["warn", "error"] = Field(
        default="warn",
        description=(
            "What to do when a section selection removes every segment. 'warn' "
            "(default) logs and continues, which yields an empty facts graph "
            "indistinguishable from a document that genuinely had nothing to "
            "extract; 'error' fails the request instead (HTTP 422, non-zero "
            "exit for a batch run). Covers both the target_sections / "
            "summarize_sections allowlist and the exclude_sections denylist, "
            "including a schema's default_exclude."
        ),
    )
    bibliography_mode: Literal["domain_facts", "citations_only", "skip"] = Field(
        default="skip",
        description=(
            "Routing for chunks detected as bibliography/reference lists "
            "(section label or citation-density heuristics): 'skip' (default) "
            "drops the chunks before extraction, 'citations_only' extracts "
            "bibliographic metadata only, 'domain_facts' disables special "
            "handling."
        ),
    )
    min_unit_chars: int = Field(
        default=0,
        ge=0,
        description=(
            "Drop content units shorter than this many characters before "
            "extraction; 0 disables the floor. Unlike CHUNK_MIN_SIZE, which "
            "the chunker only aims at, this is enforced: a heading stub or "
            "caption fragment would otherwise cost a retrieval, a render and "
            "a critic call. Size it from the per-unit node durations in the "
            "budget summary."
        ),
    )
    non_content_mode: Literal["extract", "skip"] = Field(
        default="extract",
        description=(
            "What to do with front or back matter that states no domain "
            "facts: a unit headed by author information, notes, ORCID, data "
            "availability, competing interests, licence or similar that "
            "contains no number with a unit, or a unit made mostly of emails,"
            " URLs, ORCIDs and initials. 'extract' keeps it and marks it "
            "is_non_content; 'skip' drops it before extraction and counts it "
            "in the run manifest. A measurement anywhere in the unit keeps "
            "it."
        ),
    )
    max_measurements_per_unit: int = Field(
        default=0,
        ge=0,
        description=(
            "Split a sized unit at the sentence boundary nearest its midpoint, "
            "recursively, while it states more unit-adjacent numbers than "
            "this; 0 (default) disables. Extraction loss tracks how densely a "
            "unit packs measurements rather than how long it is, so this "
            "targets the dense units without shrinking every unit's share of "
            "the prompt. Pieces never go below min_size: a dense unit shorter "
            "than twice min_size is left whole."
        ),
    )
    citation_vocabulary: dict[str, str] = Field(
        default_factory=lambda: {
            "work_class": "schema:ScholarlyArticle",
            "fallback_class": "schema:CreativeWork",
            "title": "schema:name",
            "author": "schema:author",
            "author_name": "schema:name",
            "date_published": "schema:datePublished",
            "venue": "schema:isPartOf",
            "identifier": "schema:identifier",
            "cites": "schema:citation",
        },
        description=(
            "Terms the citation-metadata prompt uses in 'citations_only' mode, "
            "by role. Bibliographic entries are not domain facts, so unlike the "
            "rest of the pipeline these terms are not retrieved from the "
            "catalog -- they default to schema.org and are overridden here for "
            "catalogs that model citations with another vocabulary (e.g. "
            "bibo, FaBiO, DCMI). Keys are fixed roles; values are CURIEs or "
            "IRIs. Setting an empty mapping drops the vocabulary guidance."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="CHUNK_",
        case_sensitive=False,
    )

Attributes

bibliography_mode = Field(default='skip', description="Routing for chunks detected as bibliography/reference lists (section label or citation-density heuristics): 'skip' (default) drops the chunks before extraction, 'citations_only' extracts bibliographic metadata only, 'domain_facts' disables special handling.") class-attribute instance-attribute
citation_vocabulary = Field(default_factory=lambda: {'work_class': 'schema:ScholarlyArticle', 'fallback_class': 'schema:CreativeWork', 'title': 'schema:name', 'author': 'schema:author', 'author_name': 'schema:name', 'date_published': 'schema:datePublished', 'venue': 'schema:isPartOf', 'identifier': 'schema:identifier', 'cites': 'schema:citation'}, description="Terms the citation-metadata prompt uses in 'citations_only' mode, by role. Bibliographic entries are not domain facts, so unlike the rest of the pipeline these terms are not retrieved from the catalog -- they default to schema.org and are overridden here for catalogs that model citations with another vocabulary (e.g. bibo, FaBiO, DCMI). Keys are fixed roles; values are CURIEs or IRIs. Setting an empty mapping drops the vocabulary guidance.") class-attribute instance-attribute
embedding_model = Field(default='sentence-transformers/paraphrase-multilingual-mpnet-base-v2', description='Sentence-transformers checkpoint for semantic chunking and embedding-based schema detection. Shared process-wide with EMBEDDING_MODEL_NAME and AGG_EMBEDDING_MODEL when the names match, so aligning all three halves resident local-model memory. Changing it invalidates the on-disk chunk cache and shifts chunk boundaries.') class-attribute instance-attribute
max_measurements_per_unit = Field(default=0, ge=0, description="Split a sized unit at the sentence boundary nearest its midpoint, recursively, while it states more unit-adjacent numbers than this; 0 (default) disables. Extraction loss tracks how densely a unit packs measurements rather than how long it is, so this targets the dense units without shrinking every unit's share of the prompt. Pieces never go below min_size: a dense unit shorter than twice min_size is left whole.") class-attribute instance-attribute
max_size = Field(default=12000, description='Largest chunk in characters. Raising it gives each LLM call more context and makes fewer calls; lower it if the model loses track of long chunks.') class-attribute instance-attribute
min_size = Field(default=3000, description='Smallest chunk in characters. Shorter neighbouring pieces are merged until they reach it, without passing CHUNK_MAX_SIZE.') class-attribute instance-attribute
min_unit_chars = Field(default=0, ge=0, description='Drop content units shorter than this many characters before extraction; 0 disables the floor. Unlike CHUNK_MIN_SIZE, which the chunker only aims at, this is enforced: a heading stub or caption fragment would otherwise cost a retrieval, a render and a critic call. Size it from the per-unit node durations in the budget summary.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='CHUNK_', case_sensitive=False) class-attribute instance-attribute
non_content_mode = Field(default='extract', description="What to do with front or back matter that states no domain facts: a unit headed by author information, notes, ORCID, data availability, competing interests, licence or similar that contains no number with a unit, or a unit made mostly of emails, URLs, ORCIDs and initials. 'extract' keeps it and marks it is_non_content; 'skip' drops it before extraction and counts it in the run manifest. A measurement anywhere in the unit keeps it.") class-attribute instance-attribute
section_classifier = Field(default='heuristic', description="Chunk section classification cascade, in increasing cost: 'off' = no section tagging (disables section filters and schema default exclusions); 'heading' = document outline plus heading pattern/keyword matching; 'heuristic' (default) = heading plus content-density classification for regions with no usable heading; 'llm' = heuristic plus a batched LLM pass over whatever remains unlabeled. Only 'llm' makes LLM calls during chunking.") class-attribute instance-attribute
section_density = Field(default='conservative', description="Content-density section classification for regions with no usable heading. 'conservative' (default) recognises only reference lists and acknowledgements, whose surface form is near-unique. 'aggressive' also guesses methods/results/introduction from figure-reference, quantity and citation densities -- these do not separate those sections cleanly, and a wrong label is silently acted on by the section filters, so it is opt-in. Requires CHUNK_SECTION_CLASSIFIER=heuristic or llm.") class-attribute instance-attribute
section_filter_on_empty = Field(default='warn', description="What to do when a section selection removes every segment. 'warn' (default) logs and continues, which yields an empty facts graph indistinguishable from a document that genuinely had nothing to extract; 'error' fails the request instead (HTTP 422, non-zero exit for a batch run). Covers both the target_sections / summarize_sections allowlist and the exclude_sections denylist, including a schema's default_exclude.") class-attribute instance-attribute
section_llm_batch_size = Field(default=40, description="Excerpts per LLM call when CHUNK_SECTION_CLASSIFIER=llm. One call covers a whole document's residual instead of one call per chunk; 0 restores per-chunk calls.") class-attribute instance-attribute
section_schema_detect = Field(default='headings', description="How to infer the document-type schema when the request names none and its document_type_hint matches none: 'off' uses the manifest default; 'lexical' scores headings against each schema's vocabulary; 'headings' adds an embedding tier when the semantic extras are installed; 'auto' also allows a weaker content-based tier for documents with almost no headings. An explicit schema or a matching hint always wins, and detection falls back to the default rather than guess.") class-attribute instance-attribute
section_schema_detect_content_min_margin = Field(default=4.0, description='Stricter margin for the content-based tier, which is measurably less reliable than the heading tiers: body prose from one domain readily resembles another (scientific prose reads like a technical specification). Only used when CHUNK_SECTION_SCHEMA_DETECT=auto.') class-attribute instance-attribute
section_schema_detect_min_margin = Field(default=1.8, description="Factor by which the winning schema's score must exceed the runner-up's; below it detection falls back to the default schema. Raise it to detect less often and more surely.") class-attribute instance-attribute
section_schema_detect_min_score = Field(default=2.0, description='Minimum distinctive evidence (headings recognised by exactly one candidate schema) before a detection is accepted.') class-attribute instance-attribute
section_tag_min_chars = Field(default=80, description='Min stripped length for LLM section tagging; smaller segments merge into neighbors before tagging') class-attribute instance-attribute
section_text_headings = Field(default=True, description='Detect headings from plain-text layout (short, blank-line delimited, upper-case or numbered lines) in documents whose conversion produced no markdown heading structure at all.') class-attribute instance-attribute
segmenter = Field(default='semantic', description="Primary segmenter: 'semantic' splits the markdown export inside detected section boundaries with the built-in semantic chunker (naive fallback without torch extras); 'docling' uses docling's HybridChunker structural segments.") class-attribute instance-attribute

ClaudeModel

Bases: LLMModelNameAbstract

Anthropic Claude model names

Source code in ontocast/config/settings.py
class ClaudeModel(LLMModelNameAbstract):
    """Anthropic Claude model names"""

    CLAUDE_FABLE_5_1 = "claude-fable-5-1"
    CLAUDE_OPUS_5_5 = "claude-opus-5-5"
    CLAUDE_OPUS_5 = "claude-opus-5"
    CLAUDE_OPUS_4_8 = "claude-opus-4-8"
    CLAUDE_OPUS_4_7 = "claude-opus-4-7"
    CLAUDE_OPUS_4_6 = "claude-opus-4-6"
    CLAUDE_SONNET_5_5 = "claude-sonnet-5-5"
    CLAUDE_SONNET_5 = "claude-sonnet-5"
    CLAUDE_SONNET_4_6 = "claude-sonnet-4-6"
    CLAUDE_HAIKU_4_5 = "claude-haiku-4-5"

Attributes

CLAUDE_FABLE_5_1 = 'claude-fable-5-1' class-attribute instance-attribute
CLAUDE_HAIKU_4_5 = 'claude-haiku-4-5' class-attribute instance-attribute
CLAUDE_OPUS_4_6 = 'claude-opus-4-6' class-attribute instance-attribute
CLAUDE_OPUS_4_7 = 'claude-opus-4-7' class-attribute instance-attribute
CLAUDE_OPUS_4_8 = 'claude-opus-4-8' class-attribute instance-attribute
CLAUDE_OPUS_5 = 'claude-opus-5' class-attribute instance-attribute
CLAUDE_OPUS_5_5 = 'claude-opus-5-5' class-attribute instance-attribute
CLAUDE_SONNET_4_6 = 'claude-sonnet-4-6' class-attribute instance-attribute
CLAUDE_SONNET_5 = 'claude-sonnet-5' class-attribute instance-attribute
CLAUDE_SONNET_5_5 = 'claude-sonnet-5-5' class-attribute instance-attribute

Config

Bases: BaseSettings

Main OntoCast configuration.

This class aggregates all configuration sections and provides a unified interface for accessing configuration values.

Source code in ontocast/config/settings.py
class Config(BaseSettings):
    """Main OntoCast configuration.

    This class aggregates all configuration sections and provides
    a unified interface for accessing configuration values.
    """

    # Tool configuration (for ToolBox)
    tool_config: ToolConfig = Field(default_factory=ToolConfig)

    # Server configuration (for server.py)
    server: ServerConfig = Field(default_factory=ServerConfig)

    # Additional settings
    logging_level: str | None = Field(
        default=None,
        description=(
            "Log level for OntoCast: debug, info, warning or error. Other "
            "libraries log one level higher. Unset leaves logging as Python "
            "configures it."
        ),
    )
    clean: bool = Field(
        default=False,
        description=(
            "When true, ``ontocast process`` batch mode flushes the triple store "
            "(configured datasets) before loading ontologies."
        ),
    )

    model_config = SettingsConfigDict(
        case_sensitive=False,
        extra="ignore",
    )

    @model_validator(mode="after")
    def warn_when_max_triples_cannot_bind(self) -> "Config":
        """Warn when a raised ``ONTOLOGY_CONTEXT_MAX_TRIPLES`` cannot take effect.

        In vector mode with unit scope the induced subgraph is already capped
        at ``VECTOR_STORE_INDUCED_SUBGRAPH_MAX_TOTAL_TRIPLES``, so a larger
        context budget changes nothing. Only a value moved off the default is
        reported: the default sits above the induced cap by design.
        """
        server = self.server
        max_triples = server.ontology_context_max_triples
        default = ServerConfig.model_fields["ontology_context_max_triples"].default
        induced_cap = self.tool_config.vector_store.induced_subgraph_max_total_triples
        if (
            server.ontology_context_mode
            == OntologyContextMode.SELECTED_VECTOR_SEARCH_ONTOLOGY
            and server.ontology_context_scope == OntologyContextScope.UNIT
            and max_triples is not None
            and max_triples != default
            and max_triples >= induced_cap
        ):
            logger.warning(
                "ONTOLOGY_CONTEXT_MAX_TRIPLES=%s cannot take effect: in vector "
                "mode the context is already capped at "
                "VECTOR_STORE_INDUCED_SUBGRAPH_MAX_TOTAL_TRIPLES=%s. Raise that "
                "instead to widen the context.",
                max_triples,
                induced_cap,
            )
        return self

    @model_validator(mode="after")
    def warn_when_a_per_unit_chapter_defeats_the_shared_prefix(self) -> "Config":
        """Warn when a per-unit conformance chapter cancels a document-scoped one.

        ``ONTOLOGY_CONTEXT_SCOPE=document`` exists to give every unit in the
        fan-out one byte-identical prompt prefix, so a provider's prefix cache
        can serve every call after the first. The prefix runs from the preamble
        through the end of the ontology chapter -- and the *conformance* chapter
        sits inside it. A shapes contract of ``context`` selects requirements per
        unit, which makes that chapter differ per call and defeats the scope
        setting entirely.

        This is a warning rather than an error because ``auto`` reaches
        ``context`` on its own once a catalog outgrows the line budget, so a
        deployment can arrive here by adding shapes rather than by setting
        anything -- and refusing to start would be the wrong answer to that.
        The two settings live on different config objects, so this is the only
        place that can see both.
        """
        if self.server.ontology_context_scope != OntologyContextScope.DOCUMENT:
            return self
        contract = self.tool_config.facts_validation.shapes_prompt_contract
        if contract == "context":
            logger.warning(
                "FACTS_SHAPES_PROMPT_CONTRACT=context makes the conformance "
                "chapter per-unit, and that chapter sits inside the prompt "
                "prefix ONTOLOGY_CONTEXT_SCOPE=document is trying to share -- "
                "so no two calls in a document will share a prefix and "
                "prefix_cache_hit_rate will not improve. Use "
                "FACTS_SHAPES_PROMPT_CONTRACT=full to keep the prefix stable, "
                "or drop the document scope."
            )
        elif contract == "auto":
            logger.info(
                "ONTOLOGY_CONTEXT_SCOPE=document with "
                "FACTS_SHAPES_PROMPT_CONTRACT=auto: if the shapes catalog "
                "outgrows FACTS_SHAPES_PROMPT_MAX_LINES the contract resolves "
                "to 'context', which makes the conformance chapter per-unit and "
                "defeats the shared prompt prefix. Read "
                "validation_config.shapes_prompt_selection in the run manifest "
                "to see which way it resolved."
            )
        return self

    @classmethod
    def in_memory(cls, **overrides: Any) -> "Config":
        """Build a configuration that needs no external services.

        Selects the in-memory **triple** store (a full pyoxigraph SPARQL
        engine) and disables vector retrieval, so the whole pipeline runs
        inside the calling process with no server and no embedding index. This
        is the recommended starting point for embedding OntoCast in another
        application:

        ```python
        tools = await ToolBox.acreate(Config.in_memory())
        ```

        Ontology context then comes from a single working ontology per unit --
        the default :class:`~ontocast.onto.enum.OntologyContextMode`. Vector
        retrieval needs one of the two supported backends, Qdrant
        (``QDRANT_URI``) or LanceDB (``LANCEDB_ENABLED``), each of which is its
        own optional extra.

        Environment variables still populate any section not named in
        ``overrides``; only the store selection is forced. The vector backend
        is left on ``auto``, so enabling LanceDB on the result takes effect.

        Args:
            **overrides: Fields to set on the returned ``Config``.

        Returns:
            A configuration bound to the process-local backends.
        """
        config = cls(**overrides)
        config.tool_config.fuseki.uri = None
        config.tool_config.qdrant.uri = None
        config.tool_config.lancedb.enabled = False
        config.tool_config.vector_store.backend = VectorStoreBackend.AUTO
        return config

    def for_tenancy(self, tenant: str, project: str) -> "Config":
        """Return a deep copy of this config bound to ``tenant`` / ``project``.

        The copy is what makes per-scope isolation real. Vector store managers
        receive ``tool_config.vector_store`` and ``tool_config.qdrant`` **by
        reference** (`tool/vector_store/factory.py`) and mutate them when
        tenancy is applied, so two scopes sharing a ``Config`` would alias each
        other's collection names.

        Args:
            tenant: Tenant identifier.
            project: Project identifier within the tenant.

        Returns:
            An independent ``Config`` with dataset, collection and table names
            resolved for the requested partition.

        Raises:
            ValueError: If either identifier is blank.
        """
        scope = TenancyScope.build(tenant, project)
        copy = self.model_copy(deep=True)
        tool_config = copy.tool_config

        tool_config.fuseki.dataset = scope.facts_name
        tool_config.fuseki.ontologies_dataset = scope.ontologies_name
        tool_config.fuseki.shapes_dataset = scope.shapes_name
        tool_config.qdrant.facts_collection = scope.facts_name
        tool_config.qdrant.ontology_collection = scope.ontologies_name
        tool_config.lancedb.facts_table = scope.facts_name
        tool_config.lancedb.ontology_table = scope.ontologies_name
        tool_config.vector_store.facts_table = scope.facts_name
        tool_config.vector_store.ontology_table = scope.ontologies_name
        return copy

    def get_tool_config(self) -> ToolConfig:
        """Get tool configuration.

        Returns:
            ToolConfig: Configuration for tools
        """
        return self.tool_config

    def validate_llm_config(self) -> None:
        """Validate LLM configuration and raise errors for missing required settings."""
        provider = self.tool_config.llm_config.provider
        if (
            provider
            in (
                LLMProvider.OPENAI,
                LLMProvider.ANTHROPIC,
                LLMProvider.GOOGLE,
            )
            and not self.tool_config.llm_config.api_key
        ):
            raise ValueError(
                f"LLM_API_KEY environment variable is required for {provider.value} provider"
            )

Attributes

clean = Field(default=False, description='When true, ``ontocast process`` batch mode flushes the triple store (configured datasets) before loading ontologies.') class-attribute instance-attribute
logging_level = Field(default=None, description='Log level for OntoCast: debug, info, warning or error. Other libraries log one level higher. Unset leaves logging as Python configures it.') class-attribute instance-attribute
model_config = SettingsConfigDict(case_sensitive=False, extra='ignore') class-attribute instance-attribute
server = Field(default_factory=ServerConfig) class-attribute instance-attribute
tool_config = Field(default_factory=ToolConfig) class-attribute instance-attribute

Methods:

for_tenancy(tenant, project)

Return a deep copy of this config bound to tenant / project.

The copy is what makes per-scope isolation real. Vector store managers receive tool_config.vector_store and tool_config.qdrant by reference (tool/vector_store/factory.py) and mutate them when tenancy is applied, so two scopes sharing a Config would alias each other's collection names.

Parameters:

Name Type Description Default
tenant str

Tenant identifier.

required
project str

Project identifier within the tenant.

required

Returns:

Type Description
'Config'

An independent Config with dataset, collection and table names

'Config'

resolved for the requested partition.

Raises:

Type Description
ValueError

If either identifier is blank.

Source code in ontocast/config/settings.py
def for_tenancy(self, tenant: str, project: str) -> "Config":
    """Return a deep copy of this config bound to ``tenant`` / ``project``.

    The copy is what makes per-scope isolation real. Vector store managers
    receive ``tool_config.vector_store`` and ``tool_config.qdrant`` **by
    reference** (`tool/vector_store/factory.py`) and mutate them when
    tenancy is applied, so two scopes sharing a ``Config`` would alias each
    other's collection names.

    Args:
        tenant: Tenant identifier.
        project: Project identifier within the tenant.

    Returns:
        An independent ``Config`` with dataset, collection and table names
        resolved for the requested partition.

    Raises:
        ValueError: If either identifier is blank.
    """
    scope = TenancyScope.build(tenant, project)
    copy = self.model_copy(deep=True)
    tool_config = copy.tool_config

    tool_config.fuseki.dataset = scope.facts_name
    tool_config.fuseki.ontologies_dataset = scope.ontologies_name
    tool_config.fuseki.shapes_dataset = scope.shapes_name
    tool_config.qdrant.facts_collection = scope.facts_name
    tool_config.qdrant.ontology_collection = scope.ontologies_name
    tool_config.lancedb.facts_table = scope.facts_name
    tool_config.lancedb.ontology_table = scope.ontologies_name
    tool_config.vector_store.facts_table = scope.facts_name
    tool_config.vector_store.ontology_table = scope.ontologies_name
    return copy
get_tool_config()

Get tool configuration.

Returns:

Name Type Description
ToolConfig ToolConfig

Configuration for tools

Source code in ontocast/config/settings.py
def get_tool_config(self) -> ToolConfig:
    """Get tool configuration.

    Returns:
        ToolConfig: Configuration for tools
    """
    return self.tool_config
in_memory(**overrides) classmethod

Build a configuration that needs no external services.

Selects the in-memory triple store (a full pyoxigraph SPARQL engine) and disables vector retrieval, so the whole pipeline runs inside the calling process with no server and no embedding index. This is the recommended starting point for embedding OntoCast in another application:

tools = await ToolBox.acreate(Config.in_memory())

Ontology context then comes from a single working ontology per unit -- the default :class:~ontocast.onto.enum.OntologyContextMode. Vector retrieval needs one of the two supported backends, Qdrant (QDRANT_URI) or LanceDB (LANCEDB_ENABLED), each of which is its own optional extra.

Environment variables still populate any section not named in overrides; only the store selection is forced. The vector backend is left on auto, so enabling LanceDB on the result takes effect.

Parameters:

Name Type Description Default
**overrides Any

Fields to set on the returned Config.

{}

Returns:

Type Description
'Config'

A configuration bound to the process-local backends.

Source code in ontocast/config/settings.py
@classmethod
def in_memory(cls, **overrides: Any) -> "Config":
    """Build a configuration that needs no external services.

    Selects the in-memory **triple** store (a full pyoxigraph SPARQL
    engine) and disables vector retrieval, so the whole pipeline runs
    inside the calling process with no server and no embedding index. This
    is the recommended starting point for embedding OntoCast in another
    application:

    ```python
    tools = await ToolBox.acreate(Config.in_memory())
    ```

    Ontology context then comes from a single working ontology per unit --
    the default :class:`~ontocast.onto.enum.OntologyContextMode`. Vector
    retrieval needs one of the two supported backends, Qdrant
    (``QDRANT_URI``) or LanceDB (``LANCEDB_ENABLED``), each of which is its
    own optional extra.

    Environment variables still populate any section not named in
    ``overrides``; only the store selection is forced. The vector backend
    is left on ``auto``, so enabling LanceDB on the result takes effect.

    Args:
        **overrides: Fields to set on the returned ``Config``.

    Returns:
        A configuration bound to the process-local backends.
    """
    config = cls(**overrides)
    config.tool_config.fuseki.uri = None
    config.tool_config.qdrant.uri = None
    config.tool_config.lancedb.enabled = False
    config.tool_config.vector_store.backend = VectorStoreBackend.AUTO
    return config
validate_llm_config()

Validate LLM configuration and raise errors for missing required settings.

Source code in ontocast/config/settings.py
def validate_llm_config(self) -> None:
    """Validate LLM configuration and raise errors for missing required settings."""
    provider = self.tool_config.llm_config.provider
    if (
        provider
        in (
            LLMProvider.OPENAI,
            LLMProvider.ANTHROPIC,
            LLMProvider.GOOGLE,
        )
        and not self.tool_config.llm_config.api_key
    ):
        raise ValueError(
            f"LLM_API_KEY environment variable is required for {provider.value} provider"
        )
warn_when_a_per_unit_chapter_defeats_the_shared_prefix()

Warn when a per-unit conformance chapter cancels a document-scoped one.

ONTOLOGY_CONTEXT_SCOPE=document exists to give every unit in the fan-out one byte-identical prompt prefix, so a provider's prefix cache can serve every call after the first. The prefix runs from the preamble through the end of the ontology chapter -- and the conformance chapter sits inside it. A shapes contract of context selects requirements per unit, which makes that chapter differ per call and defeats the scope setting entirely.

This is a warning rather than an error because auto reaches context on its own once a catalog outgrows the line budget, so a deployment can arrive here by adding shapes rather than by setting anything -- and refusing to start would be the wrong answer to that. The two settings live on different config objects, so this is the only place that can see both.

Source code in ontocast/config/settings.py
@model_validator(mode="after")
def warn_when_a_per_unit_chapter_defeats_the_shared_prefix(self) -> "Config":
    """Warn when a per-unit conformance chapter cancels a document-scoped one.

    ``ONTOLOGY_CONTEXT_SCOPE=document`` exists to give every unit in the
    fan-out one byte-identical prompt prefix, so a provider's prefix cache
    can serve every call after the first. The prefix runs from the preamble
    through the end of the ontology chapter -- and the *conformance* chapter
    sits inside it. A shapes contract of ``context`` selects requirements per
    unit, which makes that chapter differ per call and defeats the scope
    setting entirely.

    This is a warning rather than an error because ``auto`` reaches
    ``context`` on its own once a catalog outgrows the line budget, so a
    deployment can arrive here by adding shapes rather than by setting
    anything -- and refusing to start would be the wrong answer to that.
    The two settings live on different config objects, so this is the only
    place that can see both.
    """
    if self.server.ontology_context_scope != OntologyContextScope.DOCUMENT:
        return self
    contract = self.tool_config.facts_validation.shapes_prompt_contract
    if contract == "context":
        logger.warning(
            "FACTS_SHAPES_PROMPT_CONTRACT=context makes the conformance "
            "chapter per-unit, and that chapter sits inside the prompt "
            "prefix ONTOLOGY_CONTEXT_SCOPE=document is trying to share -- "
            "so no two calls in a document will share a prefix and "
            "prefix_cache_hit_rate will not improve. Use "
            "FACTS_SHAPES_PROMPT_CONTRACT=full to keep the prefix stable, "
            "or drop the document scope."
        )
    elif contract == "auto":
        logger.info(
            "ONTOLOGY_CONTEXT_SCOPE=document with "
            "FACTS_SHAPES_PROMPT_CONTRACT=auto: if the shapes catalog "
            "outgrows FACTS_SHAPES_PROMPT_MAX_LINES the contract resolves "
            "to 'context', which makes the conformance chapter per-unit and "
            "defeats the shared prompt prefix. Read "
            "validation_config.shapes_prompt_selection in the run manifest "
            "to see which way it resolved."
        )
    return self
warn_when_max_triples_cannot_bind()

Warn when a raised ONTOLOGY_CONTEXT_MAX_TRIPLES cannot take effect.

In vector mode with unit scope the induced subgraph is already capped at VECTOR_STORE_INDUCED_SUBGRAPH_MAX_TOTAL_TRIPLES, so a larger context budget changes nothing. Only a value moved off the default is reported: the default sits above the induced cap by design.

Source code in ontocast/config/settings.py
@model_validator(mode="after")
def warn_when_max_triples_cannot_bind(self) -> "Config":
    """Warn when a raised ``ONTOLOGY_CONTEXT_MAX_TRIPLES`` cannot take effect.

    In vector mode with unit scope the induced subgraph is already capped
    at ``VECTOR_STORE_INDUCED_SUBGRAPH_MAX_TOTAL_TRIPLES``, so a larger
    context budget changes nothing. Only a value moved off the default is
    reported: the default sits above the induced cap by design.
    """
    server = self.server
    max_triples = server.ontology_context_max_triples
    default = ServerConfig.model_fields["ontology_context_max_triples"].default
    induced_cap = self.tool_config.vector_store.induced_subgraph_max_total_triples
    if (
        server.ontology_context_mode
        == OntologyContextMode.SELECTED_VECTOR_SEARCH_ONTOLOGY
        and server.ontology_context_scope == OntologyContextScope.UNIT
        and max_triples is not None
        and max_triples != default
        and max_triples >= induced_cap
    ):
        logger.warning(
            "ONTOLOGY_CONTEXT_MAX_TRIPLES=%s cannot take effect: in vector "
            "mode the context is already capped at "
            "VECTOR_STORE_INDUCED_SUBGRAPH_MAX_TOTAL_TRIPLES=%s. Raise that "
            "instead to widen the context.",
            max_triples,
            induced_cap,
        )
    return self

ConverterConfig

Bases: BaseSettings

Document-conversion settings for Docling-backed inputs.

Source code in ontocast/config/settings.py
class ConverterConfig(BaseSettings):
    """Document-conversion settings for Docling-backed inputs."""

    profile: Literal["auto", "fast", "lean", "ocr"] = Field(
        default="auto",
        description=(
            "Conversion preset. 'auto' picks per PDF: 'fast' when the PDF has a "
            "text layer (born-digital, or a scan with an OCR text layer), 'ocr' "
            "when its pages are images only. 'fast' turns OCR off and uses the "
            "fast table model. 'lean' is 'fast' plus equations decoded to LaTeX, "
            "at one model call per equation; choose it, or set "
            "CONVERTER_DO_FORMULA_ENRICHMENT, when equations matter. 'ocr' is "
            "Docling's own defaults: OCR on, accurate tables, no formula decoding. "
            "A preset sets only the fields not set explicitly. The resolved "
            "profile joins the converter cache key."
        ),
    )
    pdf_backend: Literal["docling_parse", "pypdfium2"] = Field(
        default="docling_parse",
        description="PDF backend used by Docling for standard pipeline conversion.",
    )
    do_ocr: bool = Field(
        default=True,
        description=(
            "Enable OCR in Docling's standard PDF pipeline. Set by the profile "
            "unless given explicitly."
        ),
    )
    do_table_structure: bool = Field(
        default=True,
        description="Enable table structure extraction in Docling's standard pipeline.",
    )
    force_backend_text: bool = Field(
        default=False,
        description=(
            "Prefer deterministic backend text extraction when available instead of "
            "model-based page reconstruction."
        ),
    )
    table_cell_matching: bool = Field(
        default=True,
        description="Enable Docling table cell matching during table extraction.",
    )
    table_mode: Literal["accurate", "fast"] = Field(
        default="accurate",
        description=(
            "TableFormer mode: 'accurate' or 'fast'. Set by the profile unless "
            "given explicitly."
        ),
    )
    do_formula_enrichment: bool = Field(
        default=False,
        description=(
            "Decode display equations to LaTeX with Docling's formula model, "
            "instead of a placeholder. The model is downloaded on first use and "
            "runs once per detected equation, so conversion time grows with the "
            "equation count. On only in the 'lean' profile; set explicitly, it "
            "applies to every PDF, scanned ones included."
        ),
    )
    layout_model: Literal[
        "heron",
        "heron_101",
        "egret_medium",
        "egret_large",
        "egret_xlarge",
        "v2",
    ] = Field(
        default="heron",
        description="Docling layout model preset for the standard PDF pipeline.",
    )
    ocr_engine: Literal[
        "auto",
        "easyocr",
        "rapidocr",
        "tesseract_cli",
        "tesseract",
    ] = Field(
        default="auto",
        description="OCR engine used when OCR is enabled in the standard PDF pipeline.",
    )
    ocr_lang: list[str] = Field(
        default_factory=list,
        description=(
            "OCR language codes passed to the selected Docling OCR engine; leave empty "
            "to use engine defaults."
        ),
    )
    force_full_page_ocr: bool = Field(
        default=False,
        description="Force full-page OCR instead of region-limited OCR.",
    )
    ocr_bitmap_area_threshold: float = Field(
        default=0.05,
        ge=0.0,
        le=1.0,
        description="Minimum bitmap area ratio before Docling runs OCR on a region.",
    )
    repair_ligature_gaps: bool = Field(
        default=False,
        description=(
            "Repair ASCII fi/fl/ff-style ligature gaps that some publisher PDFs "
            "emit after Docling extraction, mostly through the pypdfium2 "
            "backend. Participates in the converter cache key. Removal "
            "condition: Docling normalises these "
            "gap patterns itself, at which point this becomes a no-op that can "
            "be dropped in a breaking release."
        ),
    )
    repair_numeric_artifacts: bool = Field(
        default=False,
        description=(
            "Repair conversion artifacts in the extracted text before "
            "chunking: escaped HTML entities, carriage-return column wraps, "
            "flattened exponents ('2 x 10 6' -> '2 × 10^6') and ligature gaps"
            " such as 'signifi cant'. Duplicated superscripts and citation "
            "markers fused into values are left alone, since the text cannot "
            "recover them. Part of the converter cache key, so enabling it "
            "re-converts."
        ),
    )

    supported_extensions: list[str] = Field(
        default_factory=lambda: list(DEFAULT_CONVERTER_EXTENSIONS),
        description=(
            "File suffixes converted with Docling, as a JSON list. Narrow it to "
            "refuse formats; a suffix Docling does not support fails at "
            "conversion. .txt, .json and .jsonl are read without Docling and "
            "cannot be listed. Not part of the converter cache key."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="CONVERTER_",
        case_sensitive=False,
    )

    @field_validator("profile", mode="before")
    @classmethod
    def _refuse_removed_profiles(cls, value: Any) -> Any:
        if isinstance(value, str) and value in _REMOVED_PROFILES:
            raise ValueError(
                f"converter profile '{value}' was removed; {_REMOVED_PROFILES[value]}"
            )
        return value

    @field_validator("supported_extensions")
    @classmethod
    def _normalise_extensions(cls, value: list[str]) -> list[str]:
        normalised: list[str] = []
        for raw in value:
            suffix = raw.strip().lower()
            if not suffix:
                continue
            suffix = suffix if suffix.startswith(".") else f".{suffix}"
            if suffix not in normalised:
                normalised.append(suffix)
        reserved = sorted(set(normalised) & READ_WITHOUT_CONVERTER_EXTENSIONS)
        if reserved:
            raise ValueError(
                f"{', '.join(reserved)} are read without Docling and cannot be "
                "converter extensions"
            )
        return normalised

    def resolved(self, profile: Literal["fast", "lean", "ocr"]) -> ConverterConfig:
        """This config with ``profile``'s preset applied to every field not set
        explicitly (in the constructor or the environment).

        ``profile`` must be the configured one unless that is ``auto``.
        """
        if self.profile not in ("auto", profile):
            raise ValueError(f"profile is {self.profile!r}, cannot resolve {profile!r}")
        values = self.model_dump()
        for field, value in CONVERTER_PROFILE_PRESETS[profile].items():
            if field not in self.model_fields_set:
                values[field] = value
        values["profile"] = profile
        return ConverterConfig.model_construct(
            _fields_set=self.model_fields_set | {"profile"}, **values
        )

Attributes

do_formula_enrichment = Field(default=False, description="Decode display equations to LaTeX with Docling's formula model, instead of a placeholder. The model is downloaded on first use and runs once per detected equation, so conversion time grows with the equation count. On only in the 'lean' profile; set explicitly, it applies to every PDF, scanned ones included.") class-attribute instance-attribute
do_ocr = Field(default=True, description="Enable OCR in Docling's standard PDF pipeline. Set by the profile unless given explicitly.") class-attribute instance-attribute
do_table_structure = Field(default=True, description="Enable table structure extraction in Docling's standard pipeline.") class-attribute instance-attribute
force_backend_text = Field(default=False, description='Prefer deterministic backend text extraction when available instead of model-based page reconstruction.') class-attribute instance-attribute
force_full_page_ocr = Field(default=False, description='Force full-page OCR instead of region-limited OCR.') class-attribute instance-attribute
layout_model = Field(default='heron', description='Docling layout model preset for the standard PDF pipeline.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='CONVERTER_', case_sensitive=False) class-attribute instance-attribute
ocr_bitmap_area_threshold = Field(default=0.05, ge=0.0, le=1.0, description='Minimum bitmap area ratio before Docling runs OCR on a region.') class-attribute instance-attribute
ocr_engine = Field(default='auto', description='OCR engine used when OCR is enabled in the standard PDF pipeline.') class-attribute instance-attribute
ocr_lang = Field(default_factory=list, description='OCR language codes passed to the selected Docling OCR engine; leave empty to use engine defaults.') class-attribute instance-attribute
pdf_backend = Field(default='docling_parse', description='PDF backend used by Docling for standard pipeline conversion.') class-attribute instance-attribute
profile = Field(default='auto', description="Conversion preset. 'auto' picks per PDF: 'fast' when the PDF has a text layer (born-digital, or a scan with an OCR text layer), 'ocr' when its pages are images only. 'fast' turns OCR off and uses the fast table model. 'lean' is 'fast' plus equations decoded to LaTeX, at one model call per equation; choose it, or set CONVERTER_DO_FORMULA_ENRICHMENT, when equations matter. 'ocr' is Docling's own defaults: OCR on, accurate tables, no formula decoding. A preset sets only the fields not set explicitly. The resolved profile joins the converter cache key.") class-attribute instance-attribute
repair_ligature_gaps = Field(default=False, description='Repair ASCII fi/fl/ff-style ligature gaps that some publisher PDFs emit after Docling extraction, mostly through the pypdfium2 backend. Participates in the converter cache key. Removal condition: Docling normalises these gap patterns itself, at which point this becomes a no-op that can be dropped in a breaking release.') class-attribute instance-attribute
repair_numeric_artifacts = Field(default=False, description="Repair conversion artifacts in the extracted text before chunking: escaped HTML entities, carriage-return column wraps, flattened exponents ('2 x 10 6' -> '2 × 10^6') and ligature gaps such as 'signifi cant'. Duplicated superscripts and citation markers fused into values are left alone, since the text cannot recover them. Part of the converter cache key, so enabling it re-converts.") class-attribute instance-attribute
supported_extensions = Field(default_factory=lambda: list(DEFAULT_CONVERTER_EXTENSIONS), description='File suffixes converted with Docling, as a JSON list. Narrow it to refuse formats; a suffix Docling does not support fails at conversion. .txt, .json and .jsonl are read without Docling and cannot be listed. Not part of the converter cache key.') class-attribute instance-attribute
table_cell_matching = Field(default=True, description='Enable Docling table cell matching during table extraction.') class-attribute instance-attribute
table_mode = Field(default='accurate', description="TableFormer mode: 'accurate' or 'fast'. Set by the profile unless given explicitly.") class-attribute instance-attribute

Methods:

resolved(profile)

This config with profile's preset applied to every field not set explicitly (in the constructor or the environment).

profile must be the configured one unless that is auto.

Source code in ontocast/config/settings.py
def resolved(self, profile: Literal["fast", "lean", "ocr"]) -> ConverterConfig:
    """This config with ``profile``'s preset applied to every field not set
    explicitly (in the constructor or the environment).

    ``profile`` must be the configured one unless that is ``auto``.
    """
    if self.profile not in ("auto", profile):
        raise ValueError(f"profile is {self.profile!r}, cannot resolve {profile!r}")
    values = self.model_dump()
    for field, value in CONVERTER_PROFILE_PRESETS[profile].items():
        if field not in self.model_fields_set:
            values[field] = value
    values["profile"] = profile
    return ConverterConfig.model_construct(
        _fields_set=self.model_fields_set | {"profile"}, **values
    )

CrossQueryMergeMode

Bases: StrEnum

How per-query fused hits are merged across proposition windows.

Source code in ontocast/config/settings.py
class CrossQueryMergeMode(StrEnum):
    """How per-query fused hits are merged across proposition windows."""

    MAX_SCORE = "max_score"
    SUM_SCORE = "sum_score"

Attributes

MAX_SCORE = 'max_score' class-attribute instance-attribute
SUM_SCORE = 'sum_score' class-attribute instance-attribute

DomainConfig

Bases: BaseSettings

Domain and URI configuration.

Reads the same CURRENT_DOMAIN variable that :class:~ontocast.onto.state.AgentState defaults from. Previously this class declared its own unrelated placeholder default and was never read by anything, so the documented knob and the value the pipeline actually used could not agree.

Source code in ontocast/config/settings.py
class DomainConfig(BaseSettings):
    """Domain and URI configuration.

    Reads the same ``CURRENT_DOMAIN`` variable that
    :class:`~ontocast.onto.state.AgentState` defaults from. Previously this
    class declared its own unrelated placeholder default and was never read by
    anything, so the documented knob and the value the pipeline actually used
    could not agree.
    """

    current_domain: str = Field(
        default=DEFAULT_DOMAIN,
        validation_alias=AliasChoices("current_domain", "CURRENT_DOMAIN"),
        description=(
            "IRI stem from which document namespaces are formed. Used by "
            "AgentState when no explicit value is supplied."
        ),
    )

    model_config = SettingsConfigDict(
        case_sensitive=False,
    )

Attributes

current_domain = Field(default=DEFAULT_DOMAIN, validation_alias=AliasChoices('current_domain', 'CURRENT_DOMAIN'), description='IRI stem from which document namespaces are formed. Used by AgentState when no explicit value is supplied.') class-attribute instance-attribute
model_config = SettingsConfigDict(case_sensitive=False) class-attribute instance-attribute

EmbeddingConfig

Bases: BaseSettings

Embedding provider settings used by vector stores.

Source code in ontocast/config/settings.py
class EmbeddingConfig(BaseSettings):
    """Embedding provider settings used by vector stores."""

    provider: EmbeddingProvider = Field(
        default=EmbeddingProvider.HUGGINGFACE,
        description=(
            "Where vector-store embeddings are computed: huggingface runs a "
            "local sentence-transformers model, openai and ollama call a "
            "service."
        ),
    )
    model_name: str = Field(
        default="sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2",
        description=(
            "Embedding model identifier used by the selected provider. Spelled "
            "with the org prefix so it shares one SharedEncoder slot with "
            "AGG_EMBEDDING_MODEL and CHUNK_EMBEDDING_MODEL when aligned."
        ),
    )
    api_key: str | None = Field(
        default=None, description="Provider API key for hosted embedding services."
    )
    base_url: str | None = Field(
        default=None, description="Provider base URL (for Ollama-compatible endpoints)."
    )
    dimension: int = Field(
        default=384,
        ge=1,
        description="Expected dense embedding vector size for core and neighborhood vectors.",
    )
    bm25_model_name: str = Field(
        default="Qdrant/bm25",
        description="fastembed SparseTextEmbedding model id for the BM25 sparse lane.",
    )
    query_prefix: str = Field(
        default="",
        description=(
            "Prefix prepended to text embedded as a *query*. Asymmetric retrieval models "
            "underperform their spec without it — BGE wants "
            "'Represent this sentence for searching relevant passages: ', E5 wants "
            "'query: '. Empty (default) suits the symmetric paraphrase model. Part of "
            "the stored embedding contract: changing it requires a reindex."
        ),
    )
    document_prefix: str = Field(
        default="",
        description=(
            "Prefix prepended to text embedded as a *document* during indexing "
            "(E5 wants 'passage: '; BGE wants nothing). Part of the stored embedding "
            "contract: changing it requires a reindex."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="EMBEDDING_",
        case_sensitive=False,
    )

Attributes

api_key = Field(default=None, description='Provider API key for hosted embedding services.') class-attribute instance-attribute
base_url = Field(default=None, description='Provider base URL (for Ollama-compatible endpoints).') class-attribute instance-attribute
bm25_model_name = Field(default='Qdrant/bm25', description='fastembed SparseTextEmbedding model id for the BM25 sparse lane.') class-attribute instance-attribute
dimension = Field(default=384, ge=1, description='Expected dense embedding vector size for core and neighborhood vectors.') class-attribute instance-attribute
document_prefix = Field(default='', description="Prefix prepended to text embedded as a *document* during indexing (E5 wants 'passage: '; BGE wants nothing). Part of the stored embedding contract: changing it requires a reindex.") class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='EMBEDDING_', case_sensitive=False) class-attribute instance-attribute
model_name = Field(default='sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2', description='Embedding model identifier used by the selected provider. Spelled with the org prefix so it shares one SharedEncoder slot with AGG_EMBEDDING_MODEL and CHUNK_EMBEDDING_MODEL when aligned.') class-attribute instance-attribute
provider = Field(default=EmbeddingProvider.HUGGINGFACE, description='Where vector-store embeddings are computed: huggingface runs a local sentence-transformers model, openai and ollama call a service.') class-attribute instance-attribute
query_prefix = Field(default='', description="Prefix prepended to text embedded as a *query*. Asymmetric retrieval models underperform their spec without it — BGE wants 'Represent this sentence for searching relevant passages: ', E5 wants 'query: '. Empty (default) suits the symmetric paraphrase model. Part of the stored embedding contract: changing it requires a reindex.") class-attribute instance-attribute

EmbeddingProvider

Bases: StrEnum

Supported embedding providers.

Source code in ontocast/config/settings.py
class EmbeddingProvider(StrEnum):
    """Supported embedding providers."""

    OPENAI = "openai"
    HUGGINGFACE = "huggingface"
    OLLAMA = "ollama"

Attributes

HUGGINGFACE = 'huggingface' class-attribute instance-attribute
OLLAMA = 'ollama' class-attribute instance-attribute
OPENAI = 'openai' class-attribute instance-attribute

FactsValidationConfig

Bases: BaseSettings

Deterministic post-checks applied to LLM-rendered facts graphs.

Source code in ontocast/config/settings.py
2601
2602
2603
2604
2605
2606
2607
2608
2609
2610
2611
2612
2613
2614
2615
2616
2617
2618
2619
2620
2621
2622
2623
2624
2625
2626
2627
2628
2629
2630
2631
2632
2633
2634
2635
2636
2637
2638
2639
2640
2641
2642
2643
2644
2645
2646
2647
2648
2649
2650
2651
2652
2653
2654
2655
2656
2657
2658
2659
2660
2661
2662
2663
2664
2665
2666
2667
2668
2669
2670
2671
2672
2673
2674
2675
2676
2677
2678
2679
2680
2681
2682
2683
2684
2685
2686
2687
2688
2689
2690
2691
2692
2693
2694
2695
2696
2697
2698
2699
2700
2701
2702
2703
2704
2705
2706
2707
2708
2709
2710
2711
2712
2713
2714
2715
2716
2717
2718
2719
2720
2721
2722
2723
2724
2725
2726
2727
2728
2729
2730
2731
2732
2733
2734
2735
2736
2737
2738
2739
2740
2741
2742
2743
2744
2745
2746
2747
2748
2749
2750
2751
2752
2753
2754
2755
2756
2757
2758
2759
2760
2761
2762
2763
2764
2765
2766
2767
2768
2769
2770
2771
2772
2773
2774
2775
2776
2777
2778
2779
2780
2781
2782
2783
2784
2785
2786
2787
2788
2789
2790
2791
2792
2793
2794
2795
2796
2797
2798
2799
2800
2801
2802
2803
2804
2805
2806
2807
2808
2809
2810
2811
2812
2813
2814
2815
2816
2817
2818
2819
2820
2821
2822
2823
2824
2825
2826
2827
2828
2829
2830
2831
2832
2833
2834
2835
2836
2837
2838
2839
2840
2841
2842
2843
2844
2845
2846
2847
2848
2849
2850
2851
2852
2853
2854
2855
2856
2857
2858
2859
2860
2861
2862
2863
2864
2865
2866
2867
2868
2869
2870
2871
2872
2873
2874
2875
2876
2877
2878
2879
2880
2881
2882
2883
2884
2885
2886
2887
2888
2889
2890
2891
2892
2893
2894
2895
2896
2897
2898
2899
2900
2901
2902
2903
2904
2905
2906
2907
2908
2909
2910
2911
2912
2913
2914
2915
2916
2917
2918
2919
2920
2921
2922
2923
2924
2925
2926
2927
2928
2929
2930
2931
2932
2933
2934
2935
2936
2937
2938
2939
2940
2941
2942
2943
2944
2945
2946
2947
2948
2949
2950
2951
2952
2953
2954
2955
2956
2957
2958
2959
2960
2961
2962
2963
2964
2965
2966
2967
2968
2969
2970
2971
2972
2973
class FactsValidationConfig(BaseSettings):
    """Deterministic post-checks applied to LLM-rendered facts graphs."""

    object_property_literal_check: bool = Field(
        default=True,
        description=(
            "Quarantine string literals sitting on predicates whose schema range "
            "is a class (e.g. qudt:unit with range qudt:Unit). Quarantined triples "
            "are surfaced to the facts critic so the renderer resolves the token "
            "to an IRI from the ontology context."
        ),
    )
    critic_passes: int = Field(
        default=1,
        ge=0,
        description=(
            "Review-and-patch passes per facts unit, in **LLM calls**. Each "
            "pass re-runs the deterministic checks for free, sends the graph "
            "and its findings to the critic, and applies what comes back as a "
            "compiled patch. At the default of 1 a unit costs two provider "
            "calls: one extraction, one review. Set 0 for extraction only, "
            "leaving findings to the LLM-free repairs and the gate."
        ),
    )
    critic_max_delete_share: float = Field(
        default=0.25,
        ge=0.0,
        le=1.0,
        description=(
            "Largest share of a unit graph one critic pass may remove. Beyond"
            " it, fixes that remove statements are returned as residual "
            "findings and only pure additions are applied: a critique that "
            "removes this much is rewriting the graph rather than correcting "
            "it."
        ),
    )
    critic_min_deletes: int = Field(
        default=5,
        ge=0,
        description=(
            "Deletions always permitted regardless of share. Without a floor "
            "the share cap is strictest on short units, where a single "
            "legitimate correction is already a large fraction of the graph."
        ),
    )
    critic_allow_subject_rename: bool = Field(
        default=False,
        description=(
            "Whether a critic REPLACE fix may delete statements about one "
            "subject while writing about another. That is a rename, and "
            "applied literally it orphans the old node and leaves the new one"
            " bare."
        ),
    )
    property_alias_min_ratio: float = Field(
        default=0.95,
        ge=0.0,
        le=1.0,
        description=(
            "Similarity floor (SequenceMatcher ratio) for choosing among "
            "candidates in the near-miss property rewrite. A predicate found "
            "neither in the unit's context nor in the catalog is rewritten "
            "only to a catalog term whose name tokens contain, are contained "
            "in, or equal its own; when several qualify, the best wins if it "
            "clears this ratio. Similarity alone never triggers a rewrite."
        ),
    )
    merge_repair_passes: int = Field(
        default=1,
        ge=0,
        description=(
            "Deterministic un-merge budget at the post-aggregation validation "
            "gate: error findings on merged subjects turn into full-cluster "
            "pair vetoes and the facts units are re-aggregated, up to this "
            "many passes. 0 records findings without repairing."
        ),
    )
    accept_blocking_severity: Literal["critical", "important", "never"] = Field(
        default="critical",
        description=(
            "Which critic-assigned severities keep a unit in the "
            "render/critic loop. Deterministic mandatory findings always "
            "block; this threshold applies only to the severity label the LLM"
            " critic gives its own fixes. The critic labels most fixes "
            "'important', so 'critical' is the only level that discriminates."
            " 'never' lets deterministic findings decide alone."
        ),
    )
    numeric_identifier_guard: bool = Field(
        default=True,
        description=(
            "Leave digit groups that belong to an identifier, such as a file "
            "number, a date or a citation, out of the numeric-coverage "
            "inventory, so the critic is not asked to model them as "
            "quantities. Only digits joined to an identifier inside one token"
            " are dropped; a number with a unit attached ('5mg') still "
            "counts. false lists every digit group."
        ),
    )
    context_from_units: bool = Field(
        default=True,
        description=(
            "In facts-only runs, build the document-level ontology context "
            "from the snapshots the units resolved. With no ontology stage "
            "that context is otherwise empty: entity merging loses the type "
            "and functionality declarations its guards read, and validation "
            "skips every check that needs a vocabulary (reported as "
            "validated_without_ontology_context in the retrieval metrics)."
        ),
    )
    suspect_multi_value_severity: Literal["error", "warning"] = Field(
        default="error",
        description=(
            "Severity of SUSPECT_MULTI_VALUE gate findings (multiple distinct "
            "numeric values on one predicate, or multiple objects on a "
            "dominantly single-valued predicate). Only error findings drive "
            "the un-merge repair."
        ),
    )
    suspect_multi_value_require_cross_unit: bool = Field(
        default=False,
        description=(
            "Report a multi-valued IRI predicate as an error only when the "
            "values came from merging different units; otherwise report a "
            "warning. Errors trigger the un-merge repair, which would remove "
            "a statement that one unit genuinely made with two objects. "
            "Numeric and string values are not affected: two distinct "
            "quantities on one node are always a defect."
        ),
    )
    domain_adherence_min_share: float = Field(
        default=0.15,
        ge=0.0,
        le=1.0,
        description=(
            "Minimum fraction of a render's distinct schema terms (predicates"
            " and rdf:type objects, excluding minted instances and RDF, RDFS,"
            " OWL, XSD, SKOS, DC and PROV) that must come from the unit's "
            "ontology context; below it a mandatory DOMAIN_ADHERENCE finding "
            "asks for a rewrite. It catches renders that use a generic "
            "vocabulary throughout, which every per-triple check and shape "
            "accepts. 0 disables it; keep it disabled when extracting without"
            " a catalog, and calibrate it from domain_adherence in the facts "
            "findings."
        ),
    )
    domain_adherence_min_terms: int = Field(
        default=4,
        ge=0,
        description=(
            "Fewest distinct schema terms a render must use before its catalog "
            "share is judged at all. A share over one or two terms is noise: a "
            "front-matter unit that types an identifier and an author with "
            "generic vocabulary has not abandoned the catalog, and the "
            "mandatory finding it raised drove the critic into retyping the "
            "identifier as a quantity value. 0 judges every non-empty render."
        ),
    )
    additional_standard_namespaces: list[str] = Field(
        default_factory=lambda: ["https://schema.org/", "http://schema.org/"],
        description=(
            "Namespaces exempt from UNKNOWN_TERM findings in addition to the "
            "RDF/OWL substrate and annotation/provenance terms. Only "
            "meta-vocabularies are built in; a domain vocabulary a deployment "
            "genuinely shares across catalogs (SOSA/SSN, CSVW, FOAF, "
            "schema.org, Dublin Core application profiles) is exempted here. "
            "schema.org is the default because the shipped citation "
            "vocabulary uses it."
        ),
    )
    quantity_fallback_vocabulary: dict[str, str] = Field(
        default_factory=lambda: {
            "value_class": "qudt:QuantityValue",
            "numeric_value": "qudt:numericValue",
            "unit": "qudt:unit",
        },
        description=(
            "Vocabulary the facts prompt offers for quantities when the "
            "retrieved context has no suitable class, as a role-to-IRI "
            "mapping: value_class, numeric_value and unit, plus optional "
            "lower_bound, upper_bound and roles containing 'inclusive'. "
            "Defaults to QUDT; an empty mapping forbids the fallback. Terms "
            "named here are exempt from UNKNOWN_TERM and "
            "NON_CATALOG_VOCABULARY. When numeric_value and both bounds are "
            "set, a range with equal bounds becomes a single value; the unit "
            "role also drives the LABEL_ONLY_NUMBER finding."
        ),
    )
    functional_min_single_support: int = Field(
        default=3,
        ge=1,
        description=(
            "Minimum number of single-valued subjects a predicate needs before "
            "the gate treats it as empirically functional. Below this the "
            "evidence is too thin to call a second value a violation."
        ),
    )
    literal_variant_dedupe: bool = Field(
        default=True,
        description=(
            "LLM-free gate repair: collapse duplicate literals that differ "
            "only in language tag or datatype on one (subject, predicate) — "
            "'X'@en alongside 'X'^^xsd:string alongside 'X'. The language-"
            "tagged form wins, then the plain form; reified provenance moves "
            "to the surviving triple."
        ),
    )
    shapes_dir: str | None = Field(
        default=None,
        description=(
            "Directory of SHACL shape files (.ttl, searched recursively) "
            "loaded at startup into the tenant's shapes partition of the "
            "triple store, as ONTOCAST_ONTOLOGY_DIRECTORY is for ontologies. "
            "Validation reads the partition, so shapes uploaded through "
            "/shapes apply as well. Requires the 'shacl' extra; without it, "
            "or with no readable shapes, a warning is logged."
        ),
    )
    shacl_inference: Literal["none", "rdfs", "owlrl"] = Field(
        default="rdfs",
        description=(
            "Inference pyshacl applies before evaluating shapes. 'rdfs' "
            "(default) lets a shape that names a superproperty match the more"
            " specific predicate the renderer emits, which SHACL property "
            "paths do not do on their own; turning it off raises the "
            "violation count. Use 'none' for shapes written against exactly "
            "the terms the graph uses, or when validation time dominates."
        ),
    )
    shacl_advanced: bool = Field(
        default=True,
        description=(
            "Enable the SHACL Advanced Features extension (sh:sparql "
            "constraints, node expressions). Shapes that do not use it are "
            "unaffected."
        ),
    )
    shapes_prompt_contract: Literal["off", "auto", "full", "context"] = Field(
        default="auto",
        description=(
            "Show the loaded SHACL shapes to the facts renderer and critic as"
            " a conformance chapter, so they are prompted with the rules "
            "validation applies. Each shape contributes its sh:message, or a "
            "generated line when it has none. 'off': no chapter. 'full': "
            "every shape, up to shapes_prompt_max_lines. 'context': only "
            "shapes whose targets appear in the unit's ontology snapshot. "
            "'auto': 'full' while the catalog fits the line cap, 'context' "
            "once it does not. Without shapes the prompt is the same in every"
            " mode. Terms the shapes require are exempt from UNKNOWN_TERM."
        ),
    )
    shapes_prompt_max_lines: int = Field(
        default=60,
        ge=1,
        description=(
            "Cap on rule lines in the shapes conformance chapter. A size "
            "guard, not a ranking; when it truncates, the chapter says so, so"
            " the model does not read a missing rule as no rule."
        ),
    )
    numeric_coverage_limit: int = Field(
        default=30,
        ge=0,
        description=(
            "Cap on missing-numeric mentions listed in a NUMERIC_COVERAGE "
            "finding. Bounds prompt size; ordering is shortest-first "
            "presentation order, not relevance. 0 disables the finding "
            "entirely."
        ),
    )
    numeric_coverage_mandatory: Literal["off", "measurements", "all"] = Field(
        default="off",
        description=(
            "Which NUMERIC_COVERAGE findings block a unit's acceptance: 'off'"
            " keeps them advisory; 'measurements' blocks on numbers written "
            "with a unit that are missing from the graph; 'all' also blocks "
            "on bare numbers. true and false are accepted as 'all' and 'off'."
            " Advisory by default because the critic decides per mention "
            "whether a number is a quantity."
        ),
    )

    @field_validator("numeric_coverage_mandatory", mode="before")
    @classmethod
    def _coverage_mode_from_bool(cls, value: object) -> object:
        """Read a boolean as a mode: ``true`` is ``all``, ``false`` is ``off``.

        Env values arrive as strings, so the textual booleans are folded too.
        """
        if isinstance(value, bool):
            return "all" if value else "off"
        if isinstance(value, str):
            lowered = value.strip().lower()
            if lowered in ("true", "1", "yes", "on"):
                return "all"
            if lowered in ("false", "0", "no", ""):
                return "off"
            return lowered
        return value

    critic_min_triples: int = Field(
        default=1,
        ge=0,
        description=(
            "Skip the facts critic for a unit whose render holds fewer triples "
            "than this. A critic shown an empty graph scores it perfect and "
            "bills a call for nothing; the default skips exactly the empty "
            "renders, which are then recorded as skipped rather than reviewed. "
            "0 reviews every unit."
        ),
    )
    completion_passes: int = Field(
        default=0,
        ge=0,
        description=(
            "Insert-only completion passes per facts unit, in LLM calls, run "
            "after the critic loop when numbers written with a unit are still"
            " missing from the graph. A pass sees a term sheet, the unit's "
            "typed subjects, the text and the missing measurements, and each "
            "subject it adds is kept or rolled back by the same regression "
            "check as a critic fix. 0 disables it."
        ),
    )
    shacl_max_triples: int = Field(
        default=200_000,
        ge=0,
        description=(
            "Skip SHACL validation, with a warning, for graphs larger than "
            "this. pyshacl cost grows with graph x shapes, and a skipped run "
            "must be visible rather than read as 'conforms'. 0 disables the "
            "guard."
        ),
    )
    shacl_autofix: Literal["off", "rewrite", "prune"] = Field(
        default="prune",
        description=(
            "Repair of SHACL violations without an LLM call. 'rewrite' "
            "retypes a literal to the sh:datatype it parses as, and replaces "
            "a string with the catalog IRI whose label it matches exactly and"
            " uniquely. 'prune' also drops placeholder nodes that violate "
            "sh:minCount and state nothing beyond a type and label. Neither "
            "invents a value: a node with real data but a missing property "
            "stays a finding. 'off' reports only."
        ),
    )
    shacl_autofix_passes: int = Field(
        default=1,
        ge=0,
        description=(
            "Bounded validate -> autofix -> revalidate loop at the gate. A pass "
            "is kept only if it strictly reduces the violation count, so a "
            "repair that trades conformance for nothing is reverted."
        ),
    )
    code_predicates: list[str] = Field(
        default_factory=lambda: [
            "http://qudt.org/schema/qudt/ucumCode",
            "http://qudt.org/schema/qudt/symbol",
            "http://www.w3.org/2004/02/skos/core#notation",
        ],
        description=(
            "Predicates whose literal objects are machine codes (UCUM codes, "
            "symbols, notations). A code the model emitted, such as "
            "qudt:ucumCode 'd' on a value node with no qudt:unit, is resolved"
            " to the one catalog individual that declares it. Matching is "
            "exact and case-sensitive."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="FACTS_",
        case_sensitive=False,
    )

Attributes

accept_blocking_severity = Field(default='critical', description="Which critic-assigned severities keep a unit in the render/critic loop. Deterministic mandatory findings always block; this threshold applies only to the severity label the LLM critic gives its own fixes. The critic labels most fixes 'important', so 'critical' is the only level that discriminates. 'never' lets deterministic findings decide alone.") class-attribute instance-attribute
additional_standard_namespaces = Field(default_factory=lambda: ['https://schema.org/', 'http://schema.org/'], description='Namespaces exempt from UNKNOWN_TERM findings in addition to the RDF/OWL substrate and annotation/provenance terms. Only meta-vocabularies are built in; a domain vocabulary a deployment genuinely shares across catalogs (SOSA/SSN, CSVW, FOAF, schema.org, Dublin Core application profiles) is exempted here. schema.org is the default because the shipped citation vocabulary uses it.') class-attribute instance-attribute
code_predicates = Field(default_factory=lambda: ['http://qudt.org/schema/qudt/ucumCode', 'http://qudt.org/schema/qudt/symbol', 'http://www.w3.org/2004/02/skos/core#notation'], description="Predicates whose literal objects are machine codes (UCUM codes, symbols, notations). A code the model emitted, such as qudt:ucumCode 'd' on a value node with no qudt:unit, is resolved to the one catalog individual that declares it. Matching is exact and case-sensitive.") class-attribute instance-attribute
completion_passes = Field(default=0, ge=0, description="Insert-only completion passes per facts unit, in LLM calls, run after the critic loop when numbers written with a unit are still missing from the graph. A pass sees a term sheet, the unit's typed subjects, the text and the missing measurements, and each subject it adds is kept or rolled back by the same regression check as a critic fix. 0 disables it.") class-attribute instance-attribute
context_from_units = Field(default=True, description='In facts-only runs, build the document-level ontology context from the snapshots the units resolved. With no ontology stage that context is otherwise empty: entity merging loses the type and functionality declarations its guards read, and validation skips every check that needs a vocabulary (reported as validated_without_ontology_context in the retrieval metrics).') class-attribute instance-attribute
critic_allow_subject_rename = Field(default=False, description='Whether a critic REPLACE fix may delete statements about one subject while writing about another. That is a rename, and applied literally it orphans the old node and leaves the new one bare.') class-attribute instance-attribute
critic_max_delete_share = Field(default=0.25, ge=0.0, le=1.0, description='Largest share of a unit graph one critic pass may remove. Beyond it, fixes that remove statements are returned as residual findings and only pure additions are applied: a critique that removes this much is rewriting the graph rather than correcting it.') class-attribute instance-attribute
critic_min_deletes = Field(default=5, ge=0, description='Deletions always permitted regardless of share. Without a floor the share cap is strictest on short units, where a single legitimate correction is already a large fraction of the graph.') class-attribute instance-attribute
critic_min_triples = Field(default=1, ge=0, description='Skip the facts critic for a unit whose render holds fewer triples than this. A critic shown an empty graph scores it perfect and bills a call for nothing; the default skips exactly the empty renders, which are then recorded as skipped rather than reviewed. 0 reviews every unit.') class-attribute instance-attribute
critic_passes = Field(default=1, ge=0, description='Review-and-patch passes per facts unit, in **LLM calls**. Each pass re-runs the deterministic checks for free, sends the graph and its findings to the critic, and applies what comes back as a compiled patch. At the default of 1 a unit costs two provider calls: one extraction, one review. Set 0 for extraction only, leaving findings to the LLM-free repairs and the gate.') class-attribute instance-attribute
domain_adherence_min_share = Field(default=0.15, ge=0.0, le=1.0, description="Minimum fraction of a render's distinct schema terms (predicates and rdf:type objects, excluding minted instances and RDF, RDFS, OWL, XSD, SKOS, DC and PROV) that must come from the unit's ontology context; below it a mandatory DOMAIN_ADHERENCE finding asks for a rewrite. It catches renders that use a generic vocabulary throughout, which every per-triple check and shape accepts. 0 disables it; keep it disabled when extracting without a catalog, and calibrate it from domain_adherence in the facts findings.") class-attribute instance-attribute
domain_adherence_min_terms = Field(default=4, ge=0, description='Fewest distinct schema terms a render must use before its catalog share is judged at all. A share over one or two terms is noise: a front-matter unit that types an identifier and an author with generic vocabulary has not abandoned the catalog, and the mandatory finding it raised drove the critic into retyping the identifier as a quantity value. 0 judges every non-empty render.') class-attribute instance-attribute
functional_min_single_support = Field(default=3, ge=1, description='Minimum number of single-valued subjects a predicate needs before the gate treats it as empirically functional. Below this the evidence is too thin to call a second value a violation.') class-attribute instance-attribute
literal_variant_dedupe = Field(default=True, description="LLM-free gate repair: collapse duplicate literals that differ only in language tag or datatype on one (subject, predicate) — 'X'@en alongside 'X'^^xsd:string alongside 'X'. The language-tagged form wins, then the plain form; reified provenance moves to the surviving triple.") class-attribute instance-attribute
merge_repair_passes = Field(default=1, ge=0, description='Deterministic un-merge budget at the post-aggregation validation gate: error findings on merged subjects turn into full-cluster pair vetoes and the facts units are re-aggregated, up to this many passes. 0 records findings without repairing.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='FACTS_', case_sensitive=False) class-attribute instance-attribute
numeric_coverage_limit = Field(default=30, ge=0, description='Cap on missing-numeric mentions listed in a NUMERIC_COVERAGE finding. Bounds prompt size; ordering is shortest-first presentation order, not relevance. 0 disables the finding entirely.') class-attribute instance-attribute
numeric_coverage_mandatory = Field(default='off', description="Which NUMERIC_COVERAGE findings block a unit's acceptance: 'off' keeps them advisory; 'measurements' blocks on numbers written with a unit that are missing from the graph; 'all' also blocks on bare numbers. true and false are accepted as 'all' and 'off'. Advisory by default because the critic decides per mention whether a number is a quantity.") class-attribute instance-attribute
numeric_identifier_guard = Field(default=True, description="Leave digit groups that belong to an identifier, such as a file number, a date or a citation, out of the numeric-coverage inventory, so the critic is not asked to model them as quantities. Only digits joined to an identifier inside one token are dropped; a number with a unit attached ('5mg') still counts. false lists every digit group.") class-attribute instance-attribute
object_property_literal_check = Field(default=True, description='Quarantine string literals sitting on predicates whose schema range is a class (e.g. qudt:unit with range qudt:Unit). Quarantined triples are surfaced to the facts critic so the renderer resolves the token to an IRI from the ontology context.') class-attribute instance-attribute
property_alias_min_ratio = Field(default=0.95, ge=0.0, le=1.0, description="Similarity floor (SequenceMatcher ratio) for choosing among candidates in the near-miss property rewrite. A predicate found neither in the unit's context nor in the catalog is rewritten only to a catalog term whose name tokens contain, are contained in, or equal its own; when several qualify, the best wins if it clears this ratio. Similarity alone never triggers a rewrite.") class-attribute instance-attribute
quantity_fallback_vocabulary = Field(default_factory=lambda: {'value_class': 'qudt:QuantityValue', 'numeric_value': 'qudt:numericValue', 'unit': 'qudt:unit'}, description="Vocabulary the facts prompt offers for quantities when the retrieved context has no suitable class, as a role-to-IRI mapping: value_class, numeric_value and unit, plus optional lower_bound, upper_bound and roles containing 'inclusive'. Defaults to QUDT; an empty mapping forbids the fallback. Terms named here are exempt from UNKNOWN_TERM and NON_CATALOG_VOCABULARY. When numeric_value and both bounds are set, a range with equal bounds becomes a single value; the unit role also drives the LABEL_ONLY_NUMBER finding.") class-attribute instance-attribute
shacl_advanced = Field(default=True, description='Enable the SHACL Advanced Features extension (sh:sparql constraints, node expressions). Shapes that do not use it are unaffected.') class-attribute instance-attribute
shacl_autofix = Field(default='prune', description="Repair of SHACL violations without an LLM call. 'rewrite' retypes a literal to the sh:datatype it parses as, and replaces a string with the catalog IRI whose label it matches exactly and uniquely. 'prune' also drops placeholder nodes that violate sh:minCount and state nothing beyond a type and label. Neither invents a value: a node with real data but a missing property stays a finding. 'off' reports only.") class-attribute instance-attribute
shacl_autofix_passes = Field(default=1, ge=0, description='Bounded validate -> autofix -> revalidate loop at the gate. A pass is kept only if it strictly reduces the violation count, so a repair that trades conformance for nothing is reverted.') class-attribute instance-attribute
shacl_inference = Field(default='rdfs', description="Inference pyshacl applies before evaluating shapes. 'rdfs' (default) lets a shape that names a superproperty match the more specific predicate the renderer emits, which SHACL property paths do not do on their own; turning it off raises the violation count. Use 'none' for shapes written against exactly the terms the graph uses, or when validation time dominates.") class-attribute instance-attribute
shacl_max_triples = Field(default=200000, ge=0, description="Skip SHACL validation, with a warning, for graphs larger than this. pyshacl cost grows with graph x shapes, and a skipped run must be visible rather than read as 'conforms'. 0 disables the guard.") class-attribute instance-attribute
shapes_dir = Field(default=None, description="Directory of SHACL shape files (.ttl, searched recursively) loaded at startup into the tenant's shapes partition of the triple store, as ONTOCAST_ONTOLOGY_DIRECTORY is for ontologies. Validation reads the partition, so shapes uploaded through /shapes apply as well. Requires the 'shacl' extra; without it, or with no readable shapes, a warning is logged.") class-attribute instance-attribute
shapes_prompt_contract = Field(default='auto', description="Show the loaded SHACL shapes to the facts renderer and critic as a conformance chapter, so they are prompted with the rules validation applies. Each shape contributes its sh:message, or a generated line when it has none. 'off': no chapter. 'full': every shape, up to shapes_prompt_max_lines. 'context': only shapes whose targets appear in the unit's ontology snapshot. 'auto': 'full' while the catalog fits the line cap, 'context' once it does not. Without shapes the prompt is the same in every mode. Terms the shapes require are exempt from UNKNOWN_TERM.") class-attribute instance-attribute
shapes_prompt_max_lines = Field(default=60, ge=1, description='Cap on rule lines in the shapes conformance chapter. A size guard, not a ranking; when it truncates, the chapter says so, so the model does not read a missing rule as no rule.') class-attribute instance-attribute
suspect_multi_value_require_cross_unit = Field(default=False, description='Report a multi-valued IRI predicate as an error only when the values came from merging different units; otherwise report a warning. Errors trigger the un-merge repair, which would remove a statement that one unit genuinely made with two objects. Numeric and string values are not affected: two distinct quantities on one node are always a defect.') class-attribute instance-attribute
suspect_multi_value_severity = Field(default='error', description='Severity of SUSPECT_MULTI_VALUE gate findings (multiple distinct numeric values on one predicate, or multiple objects on a dominantly single-valued predicate). Only error findings drive the un-merge repair.') class-attribute instance-attribute

FusekiConfig

Bases: BaseSettings

Fuseki triple store configuration.

Source code in ontocast/config/settings.py
class FusekiConfig(BaseSettings):
    """Fuseki triple store configuration."""

    uri: str | None = Field(
        default=None,
        description=(
            "Fuseki HTTP server root (e.g. http://localhost:3030), not a dataset "
            "path or #/dataset/... UI URL; use FUSEKI_DATASET for the dataset name."
        ),
    )
    auth: str | None = Field(
        default=None,
        description=(
            "Fuseki credentials as user/password or user:password. Optional: "
            "FUSEKI_URI alone selects Fuseki, unauthenticated."
        ),
    )
    dataset: str | None = Field(
        default=None,
        description=(
            "Facts dataset name; defaults to the name derived from "
            f"{DEFAULT_TENANT!r}/{DEFAULT_PROJECT!r}. "
            "ontocast serve and ontocast process replace it with the name "
            "derived from the tenant and project; only an embedded ToolBox reads it."
        ),
    )
    ontologies_dataset: str | None = Field(
        default=None,
        description=(
            "Ontologies dataset name; derived like FUSEKI_DATASET, and replaced "
            "the same way by ontocast serve and ontocast process."
        ),
    )
    shapes_dataset: str | None = Field(
        default=None,
        description=(
            "SHACL shapes dataset name; derived like FUSEKI_DATASET, and "
            "replaced the same way by ontocast serve and ontocast process. Kept apart from "
            "the ontologies dataset because catalog discovery claims every "
            "named graph carrying an owl:Ontology subject, and a shapes "
            "document declares one."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="FUSEKI_",
        case_sensitive=False,
    )

    @model_validator(mode="after")
    def _resolve_fuseki_datasets(self) -> FusekiConfig:
        if self.dataset is None:
            self.dataset = tenant_project_facts_name(DEFAULT_TENANT, DEFAULT_PROJECT)
        if self.ontologies_dataset is None:
            self.ontologies_dataset = tenant_project_ontologies_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        if self.shapes_dataset is None:
            self.shapes_dataset = tenant_project_shapes_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        return self

Attributes

auth = Field(default=None, description='Fuseki credentials as user/password or user:password. Optional: FUSEKI_URI alone selects Fuseki, unauthenticated.') class-attribute instance-attribute
dataset = Field(default=None, description=f'Facts dataset name; defaults to the name derived from {DEFAULT_TENANT!r}/{DEFAULT_PROJECT!r}. ontocast serve and ontocast process replace it with the name derived from the tenant and project; only an embedded ToolBox reads it.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='FUSEKI_', case_sensitive=False) class-attribute instance-attribute
ontologies_dataset = Field(default=None, description='Ontologies dataset name; derived like FUSEKI_DATASET, and replaced the same way by ontocast serve and ontocast process.') class-attribute instance-attribute
shapes_dataset = Field(default=None, description='SHACL shapes dataset name; derived like FUSEKI_DATASET, and replaced the same way by ontocast serve and ontocast process. Kept apart from the ontologies dataset because catalog discovery claims every named graph carrying an owl:Ontology subject, and a shapes document declares one.') class-attribute instance-attribute
uri = Field(default=None, description='Fuseki HTTP server root (e.g. http://localhost:3030), not a dataset path or #/dataset/... UI URL; use FUSEKI_DATASET for the dataset name.') class-attribute instance-attribute

GeminiModel

Bases: LLMModelNameAbstract

Google Gemini model names

Source code in ontocast/config/settings.py
class GeminiModel(LLMModelNameAbstract):
    """Google Gemini model names"""

    # Frontier Intelligence & Reasoning
    GEMINI_3_1_PRO = "gemini-3.1-pro"
    GEMINI_3_1_PRO_PREVIEW = "gemini-3.1-pro-preview"

    # Speed & Multimodal Agents
    GEMINI_3_7_FLASH = "gemini-3.7-flash"
    GEMINI_3_5_FLASH = "gemini-3.5-flash"
    GEMINI_3_FLASH = "gemini-3-flash"
    GEMINI_3_FLASH_PREVIEW = "gemini-3-flash-preview"

    # Ultra Budget & Low-Latency
    GEMINI_3_1_FLASH_LITE = "gemini-3.1-flash-lite"

Attributes

GEMINI_3_1_FLASH_LITE = 'gemini-3.1-flash-lite' class-attribute instance-attribute
GEMINI_3_1_PRO = 'gemini-3.1-pro' class-attribute instance-attribute
GEMINI_3_1_PRO_PREVIEW = 'gemini-3.1-pro-preview' class-attribute instance-attribute
GEMINI_3_5_FLASH = 'gemini-3.5-flash' class-attribute instance-attribute
GEMINI_3_7_FLASH = 'gemini-3.7-flash' class-attribute instance-attribute
GEMINI_3_FLASH = 'gemini-3-flash' class-attribute instance-attribute
GEMINI_3_FLASH_PREVIEW = 'gemini-3-flash-preview' class-attribute instance-attribute

InducedSubgraphSeedOrder

Bases: StrEnum

Seed expansion order for induced-subgraph triple budgeting.

Source code in ontocast/config/settings.py
class InducedSubgraphSeedOrder(StrEnum):
    """Seed expansion order for induced-subgraph triple budgeting."""

    SCORE = "score"
    ONTOLOGY_ROUND_ROBIN = "ontology_round_robin"

Attributes

ONTOLOGY_ROUND_ROBIN = 'ontology_round_robin' class-attribute instance-attribute
SCORE = 'score' class-attribute instance-attribute

LLMConfig

Bases: BaseSettings

LLM configuration settings.

Source code in ontocast/config/settings.py
class LLMConfig(BaseSettings):
    """LLM configuration settings."""

    provider: LLMProvider = Field(
        default=LLMProvider.OPENAI,
        description=(
            "Provider the LLM calls go to. Each needs its install extra; "
            "ollama also needs LLM_BASE_URL, the others LLM_API_KEY."
        ),
    )
    model_name: LLMModelName = Field(
        default=OpenAIModel.GPT5_6_LUNA,
        description=(
            "Model name passed to the provider. Any name the provider accepts "
            "works; a name OntoCast does not know is passed through with a "
            "warning."
        ),
    )
    temperature: float = Field(
        default=0.0,
        description=(
            "Sampling temperature. Keep 0.0 so a run is repeatable and its "
            "cached responses stay valid. OpenAI gpt-5, gpt-5-mini and "
            "gpt-5-nano accept only 1.0, which is applied for them."
        ),
    )
    base_url: str | None = Field(
        default=None, description="LLM base URL (for ollama, etc.)"
    )
    api_key: str | None = Field(
        default=None,
        description=(
            "Key for OpenAI, Anthropic or Google. Only this variable is read, "
            "not the provider's own (OPENAI_API_KEY and so on). Not needed "
            "for ollama."
        ),
    )
    prompt_cache_key: str | None = Field(
        default=None,
        description=(
            "OpenAI prompt_cache_key: a routing hint that sends requests "
            "sharing a prompt prefix to the same cache shard, so a wide "
            "simultaneous fan-out hits the provider's prefix cache instead of"
            " building one entry per machine. Use one stable string per "
            "deployment, never one per request. Changes routing only, never "
            "the response, so it is not part of the LLM disk-cache key. "
            "OpenAI only."
        ),
    )
    cache_enabled: bool = Field(
        default=True,
        description="When true, read and write LLM response disk cache entries.",
    )
    cache_read_only: bool = Field(
        default=False,
        description="When true, use cached responses but do not write new entries.",
    )
    llm_max_inflight: int = Field(
        default=16,
        ge=1,
        # Documented as LLM_MAX_INFLIGHT, but the LLM_ env_prefix would otherwise
        # make the real variable LLM_LLM_MAX_INFLIGHT -- the documented name was a
        # silent no-op. validation_alias bypasses env_prefix, so the aliases bind
        # the literal variables LLM_MAX_INFLIGHT and (unprefixed) MAX_INFLIGHT.
        validation_alias=AliasChoices("llm_max_inflight", "max_inflight"),
        description=(
            "Maximum concurrent provider LLM requests shared across all documents."
        ),
    )
    request_timeout_seconds: float | None = Field(
        default=180.0,
        gt=0,
        description=(
            "Per-request timeout for a provider call, in seconds. A hung call "
            "otherwise holds both a unit-worker slot and an LLM_MAX_INFLIGHT "
            "slot indefinitely, so a couple of them permanently shrink the "
            "pipeline's effective width. Set to None to wait forever."
        ),
    )
    requests_per_second: float | None = Field(
        default=None,
        gt=0,
        description=(
            "Sustained provider request rate, paced by a per-process token "
            "bucket on request starts (langchain InMemoryRateLimiter). "
            "Complements LLM_MAX_INFLIGHT, which caps *concurrency* but not "
            "rate: a fan-out of short calls can exceed a provider tier's "
            "requests-per-minute while never holding many connections at "
            "once. Set from the deployment's provider tier; None (default) "
            "means unpaced. A throttle that slips through anyway is counted "
            "as llm/rate_limited in the budget."
        ),
    )
    max_retries: int | None = Field(
        default=None,
        ge=0,
        description=(
            "Retry budget handed to the provider SDK for transport-level "
            "failures (rate limits, connection resets), which back off and "
            "honour Retry-After. None (default) keeps each SDK's own default."
            " This is the knob to raise when a tier throttles -- the pipeline"
            " itself deliberately never retries transport failures (that "
            "multiplies request rate exactly when the provider asks for "
            "less). Ignored by the Ollama provider, which exposes no retry "
            "budget."
        ),
    )
    think: bool | None = Field(
        default=None,
        description=(
            "Controls thinking/reasoning mode for Ollama thinking models "
            "(e.g. qwen3, deepseek-r1). "
            "False disables thinking and ensures a non-empty content response. "
            "True enables thinking and captures it separately in reasoning_content. "
            "None uses the model's default behaviour (thinking tags may appear "
            "inline in content, or the response may be empty if all tokens are "
            "consumed during reasoning)."
        ),
    )
    num_predict: int | None = Field(
        default=None,
        description=(
            "Maximum number of tokens to generate (Ollama only). "
            "None uses Ollama's default (unlimited). "
            "Increase this when using thinking models to ensure enough tokens "
            "remain for the actual response after the reasoning phase."
        ),
    )
    num_ctx: int | None = Field(
        default=None,
        description=(
            "Context window size in tokens (Ollama only). "
            "Controls the total KV-cache window: prompt tokens + output tokens must "
            "fit within this budget. Ollama's default is model-dependent (often "
            "2048–4096). For large prompts set this to 16384 or higher. "
            "Directly affects VRAM usage on the inference server."
        ),
    )
    json_mode: bool = Field(
        default=False,
        description=(
            "Constrain OpenAI decoding to valid JSON (response_format "
            "json_object). Every response is parsed as a JSON envelope "
            "whatever LLM_GRAPH_FORMAT is, so this rules out envelope syntax "
            "errors rather than repairing them. Off by default because OpenAI"
            " rejects the request unless the prompt contains the word JSON. "
            "OpenAI only; other providers ignore it."
        ),
    )
    reasoning_effort: (
        Literal["none", "minimal", "low", "medium", "high", "xhigh", "max"] | None
    ) = Field(
        default=None,
        description=(
            "How much the model reasons before answering: none, minimal, low,"
            " medium, high, xhigh or max. Sent to OpenAI reasoning models as "
            "reasoning_effort and to Gemini 3+ as thinking_level; which "
            "levels a model accepts is the provider's decision, and a "
            "rejected level fails the request. Reasoning tokens are billed as"
            " output (reasoning_share_of_output in the budget). Unset keeps "
            "the provider default. Part of the LLM cache key. Ollama and "
            "Anthropic ignore it with a warning."
        ),
    )
    thinking_budget: int | None = Field(
        default=None,
        ge=-1,
        description=(
            "Thinking-token budget for Gemini 2.5 models: 0 disables thinking"
            " where the model allows it, -1 lets the model choose, a positive"
            " value caps it. Gemini 3 and later take LLM_REASONING_EFFORT "
            "instead; the two cannot be combined. Unset keeps the provider "
            "default. Part of the LLM cache key. Other providers ignore it "
            "with a warning."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="LLM_",
        case_sensitive=False,
    )

    @field_validator("model_name")
    @classmethod
    def validate_model_name(cls, v: LLMModelName, info) -> LLMModelName:
        """Warn when model_name is not a known preset for the provider.

        Deliberately a warning and not an error. A model outside the provider's
        enum is now a legitimate case in two ways: a release newer than this
        package, and an OpenAI-compatible endpoint reached through
        :attr:`LLMConfig.base_url`, where the useful model names are another
        vendor's entirely. Neither is distinguishable from a typo at config
        time, and the provider rejects a genuinely bad name on the first call
        with a better message than this validator could produce.
        """
        provider = info.data.get("provider")
        if provider is None:
            return v

        expected = _PRESET_MODELS_BY_PROVIDER.get(provider)
        if expected is None or isinstance(v, expected):
            return v

        # A bare string that names a preset exactly is not a stranger -- with
        # ``str`` in the union pydantic has no reason to prefer the enum, so
        # without this every env-supplied model name warned about itself.
        try:
            return expected(str(v))
        except ValueError:
            pass

        logger.warning(
            "Model %r is not a known %s preset (%s). Passing it through -- the "
            "provider decides whether it exists.",
            str(v),
            provider.value,
            expected.__name__,
        )
        return v

    @model_validator(mode="after")
    def validate_reasoning_knobs(self) -> "LLMConfig":
        """Reject the two Gemini reasoning spellings being set together.

        The Gemini API treats ``thinking_level`` and ``thinking_budget`` as
        mutually exclusive and the client resolves the clash by dropping the
        budget with a warning. Absorbing that silently is worse here than
        failing: the run would bill one reasoning setting while the manifest
        recorded the other, and every arm read off that manifest afterwards
        would be attributing cost to a budget that never applied.
        """
        if (
            self.provider == LLMProvider.GOOGLE
            and self.reasoning_effort is not None
            and self.thinking_budget is not None
        ):
            raise ValueError(
                "LLM_REASONING_EFFORT and LLM_THINKING_BUDGET are mutually "
                "exclusive on Google: Gemini 3+ reads the first as "
                "thinking_level, Gemini 2.5 reads the second. Set whichever "
                "matches the model generation, not both."
            )
        return self

Attributes

api_key = Field(default=None, description="Key for OpenAI, Anthropic or Google. Only this variable is read, not the provider's own (OPENAI_API_KEY and so on). Not needed for ollama.") class-attribute instance-attribute
base_url = Field(default=None, description='LLM base URL (for ollama, etc.)') class-attribute instance-attribute
cache_enabled = Field(default=True, description='When true, read and write LLM response disk cache entries.') class-attribute instance-attribute
cache_read_only = Field(default=False, description='When true, use cached responses but do not write new entries.') class-attribute instance-attribute
json_mode = Field(default=False, description='Constrain OpenAI decoding to valid JSON (response_format json_object). Every response is parsed as a JSON envelope whatever LLM_GRAPH_FORMAT is, so this rules out envelope syntax errors rather than repairing them. Off by default because OpenAI rejects the request unless the prompt contains the word JSON. OpenAI only; other providers ignore it.') class-attribute instance-attribute
llm_max_inflight = Field(default=16, ge=1, validation_alias=AliasChoices('llm_max_inflight', 'max_inflight'), description='Maximum concurrent provider LLM requests shared across all documents.') class-attribute instance-attribute
max_retries = Field(default=None, ge=0, description="Retry budget handed to the provider SDK for transport-level failures (rate limits, connection resets), which back off and honour Retry-After. None (default) keeps each SDK's own default. This is the knob to raise when a tier throttles -- the pipeline itself deliberately never retries transport failures (that multiplies request rate exactly when the provider asks for less). Ignored by the Ollama provider, which exposes no retry budget.") class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='LLM_', case_sensitive=False) class-attribute instance-attribute
model_name = Field(default=OpenAIModel.GPT5_6_LUNA, description='Model name passed to the provider. Any name the provider accepts works; a name OntoCast does not know is passed through with a warning.') class-attribute instance-attribute
num_ctx = Field(default=None, description="Context window size in tokens (Ollama only). Controls the total KV-cache window: prompt tokens + output tokens must fit within this budget. Ollama's default is model-dependent (often 2048–4096). For large prompts set this to 16384 or higher. Directly affects VRAM usage on the inference server.") class-attribute instance-attribute
num_predict = Field(default=None, description="Maximum number of tokens to generate (Ollama only). None uses Ollama's default (unlimited). Increase this when using thinking models to ensure enough tokens remain for the actual response after the reasoning phase.") class-attribute instance-attribute
prompt_cache_key = Field(default=None, description="OpenAI prompt_cache_key: a routing hint that sends requests sharing a prompt prefix to the same cache shard, so a wide simultaneous fan-out hits the provider's prefix cache instead of building one entry per machine. Use one stable string per deployment, never one per request. Changes routing only, never the response, so it is not part of the LLM disk-cache key. OpenAI only.") class-attribute instance-attribute
provider = Field(default=LLMProvider.OPENAI, description='Provider the LLM calls go to. Each needs its install extra; ollama also needs LLM_BASE_URL, the others LLM_API_KEY.') class-attribute instance-attribute
reasoning_effort = Field(default=None, description="How much the model reasons before answering: none, minimal, low, medium, high, xhigh or max. Sent to OpenAI reasoning models as reasoning_effort and to Gemini 3+ as thinking_level; which levels a model accepts is the provider's decision, and a rejected level fails the request. Reasoning tokens are billed as output (reasoning_share_of_output in the budget). Unset keeps the provider default. Part of the LLM cache key. Ollama and Anthropic ignore it with a warning.") class-attribute instance-attribute
request_timeout_seconds = Field(default=180.0, gt=0, description="Per-request timeout for a provider call, in seconds. A hung call otherwise holds both a unit-worker slot and an LLM_MAX_INFLIGHT slot indefinitely, so a couple of them permanently shrink the pipeline's effective width. Set to None to wait forever.") class-attribute instance-attribute
requests_per_second = Field(default=None, gt=0, description="Sustained provider request rate, paced by a per-process token bucket on request starts (langchain InMemoryRateLimiter). Complements LLM_MAX_INFLIGHT, which caps *concurrency* but not rate: a fan-out of short calls can exceed a provider tier's requests-per-minute while never holding many connections at once. Set from the deployment's provider tier; None (default) means unpaced. A throttle that slips through anyway is counted as llm/rate_limited in the budget.") class-attribute instance-attribute
temperature = Field(default=0.0, description='Sampling temperature. Keep 0.0 so a run is repeatable and its cached responses stay valid. OpenAI gpt-5, gpt-5-mini and gpt-5-nano accept only 1.0, which is applied for them.') class-attribute instance-attribute
think = Field(default=None, description="Controls thinking/reasoning mode for Ollama thinking models (e.g. qwen3, deepseek-r1). False disables thinking and ensures a non-empty content response. True enables thinking and captures it separately in reasoning_content. None uses the model's default behaviour (thinking tags may appear inline in content, or the response may be empty if all tokens are consumed during reasoning).") class-attribute instance-attribute
thinking_budget = Field(default=None, ge=-1, description='Thinking-token budget for Gemini 2.5 models: 0 disables thinking where the model allows it, -1 lets the model choose, a positive value caps it. Gemini 3 and later take LLM_REASONING_EFFORT instead; the two cannot be combined. Unset keeps the provider default. Part of the LLM cache key. Other providers ignore it with a warning.') class-attribute instance-attribute

Methods:

validate_model_name(v, info) classmethod

Warn when model_name is not a known preset for the provider.

Deliberately a warning and not an error. A model outside the provider's enum is now a legitimate case in two ways: a release newer than this package, and an OpenAI-compatible endpoint reached through :attr:LLMConfig.base_url, where the useful model names are another vendor's entirely. Neither is distinguishable from a typo at config time, and the provider rejects a genuinely bad name on the first call with a better message than this validator could produce.

Source code in ontocast/config/settings.py
@field_validator("model_name")
@classmethod
def validate_model_name(cls, v: LLMModelName, info) -> LLMModelName:
    """Warn when model_name is not a known preset for the provider.

    Deliberately a warning and not an error. A model outside the provider's
    enum is now a legitimate case in two ways: a release newer than this
    package, and an OpenAI-compatible endpoint reached through
    :attr:`LLMConfig.base_url`, where the useful model names are another
    vendor's entirely. Neither is distinguishable from a typo at config
    time, and the provider rejects a genuinely bad name on the first call
    with a better message than this validator could produce.
    """
    provider = info.data.get("provider")
    if provider is None:
        return v

    expected = _PRESET_MODELS_BY_PROVIDER.get(provider)
    if expected is None or isinstance(v, expected):
        return v

    # A bare string that names a preset exactly is not a stranger -- with
    # ``str`` in the union pydantic has no reason to prefer the enum, so
    # without this every env-supplied model name warned about itself.
    try:
        return expected(str(v))
    except ValueError:
        pass

    logger.warning(
        "Model %r is not a known %s preset (%s). Passing it through -- the "
        "provider decides whether it exists.",
        str(v),
        provider.value,
        expected.__name__,
    )
    return v
validate_reasoning_knobs()

Reject the two Gemini reasoning spellings being set together.

The Gemini API treats thinking_level and thinking_budget as mutually exclusive and the client resolves the clash by dropping the budget with a warning. Absorbing that silently is worse here than failing: the run would bill one reasoning setting while the manifest recorded the other, and every arm read off that manifest afterwards would be attributing cost to a budget that never applied.

Source code in ontocast/config/settings.py
@model_validator(mode="after")
def validate_reasoning_knobs(self) -> "LLMConfig":
    """Reject the two Gemini reasoning spellings being set together.

    The Gemini API treats ``thinking_level`` and ``thinking_budget`` as
    mutually exclusive and the client resolves the clash by dropping the
    budget with a warning. Absorbing that silently is worse here than
    failing: the run would bill one reasoning setting while the manifest
    recorded the other, and every arm read off that manifest afterwards
    would be attributing cost to a budget that never applied.
    """
    if (
        self.provider == LLMProvider.GOOGLE
        and self.reasoning_effort is not None
        and self.thinking_budget is not None
    ):
        raise ValueError(
            "LLM_REASONING_EFFORT and LLM_THINKING_BUDGET are mutually "
            "exclusive on Google: Gemini 3+ reads the first as "
            "thinking_level, Gemini 2.5 reads the second. Set whichever "
            "matches the model generation, not both."
        )
    return self

LLMModelNameAbstract

Bases: StrEnum

Abstract base class for all model names.

Source code in ontocast/config/settings.py
class LLMModelNameAbstract(StrEnum):
    """Abstract base class for all model names."""

LLMProvider

Bases: StrEnum

Supported LLM providers.

Source code in ontocast/config/settings.py
class LLMProvider(StrEnum):
    """Supported LLM providers."""

    OPENAI = "openai"
    OLLAMA = "ollama"
    ANTHROPIC = "anthropic"
    GOOGLE = "google"

Attributes

ANTHROPIC = 'anthropic' class-attribute instance-attribute
GOOGLE = 'google' class-attribute instance-attribute
OLLAMA = 'ollama' class-attribute instance-attribute
OPENAI = 'openai' class-attribute instance-attribute

LanceDBConfig

Bases: BaseSettings

Embedded LanceDB vector store settings.

Source code in ontocast/config/settings.py
class LanceDBConfig(BaseSettings):
    """Embedded LanceDB vector store settings."""

    enabled: bool = Field(
        default=False,
        description=(
            "Enable embedded LanceDB when QDRANT_URI is unset. "
            "Uses a local directory via lancedb.connect(data_dir)."
        ),
    )
    data_dir: Path | str = Field(
        default="~/.lancedb_data",
        description=(
            "Local filesystem directory passed to lancedb.connect(...) "
            "(supports ~ expansion)."
        ),
    )
    ontology_table: str | None = Field(
        default=None,
        description=(
            "Lance table for ontology atom vectors; derived like FUSEKI_DATASET, "
            "and replaced the same way by ontocast serve and ontocast process."
        ),
    )
    facts_table: str | None = Field(
        default=None,
        description="Lance table reserved for future fact vectors; created on init.",
    )

    model_config = SettingsConfigDict(
        env_prefix="LANCEDB_",
        case_sensitive=False,
    )

    @model_validator(mode="after")
    def _resolve_lancedb_tables(self) -> LanceDBConfig:
        if self.ontology_table is None:
            self.ontology_table = tenant_project_ontologies_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        if self.facts_table is None:
            self.facts_table = tenant_project_facts_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        return self

Attributes

data_dir = Field(default='~/.lancedb_data', description='Local filesystem directory passed to lancedb.connect(...) (supports ~ expansion).') class-attribute instance-attribute
enabled = Field(default=False, description='Enable embedded LanceDB when QDRANT_URI is unset. Uses a local directory via lancedb.connect(data_dir).') class-attribute instance-attribute
facts_table = Field(default=None, description='Lance table reserved for future fact vectors; created on init.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='LANCEDB_', case_sensitive=False) class-attribute instance-attribute
ontology_table = Field(default=None, description='Lance table for ontology atom vectors; derived like FUSEKI_DATASET, and replaced the same way by ontocast serve and ontocast process.') class-attribute instance-attribute

LexicalTriggerFusion

Bases: StrEnum

How lexical-trigger hits combine with semantic retrieval hits.

Source code in ontocast/config/settings.py
class LexicalTriggerFusion(StrEnum):
    """How lexical-trigger hits combine with semantic retrieval hits."""

    MAX_MERGE = "max_merge"
    APPEND = "append"

Attributes

APPEND = 'append' class-attribute instance-attribute
MAX_MERGE = 'max_merge' class-attribute instance-attribute

OllamaModel

Bases: LLMModelNameAbstract

Ollama model names

Source code in ontocast/config/settings.py
class OllamaModel(LLMModelNameAbstract):
    """Ollama model names"""

    # Meta
    MUSE_GLIMMER = "muse-glimmer"
    LLAMA4_SCOUT = "llama4-scout:17b"
    LLAMA3_3 = "llama3.3"
    LLAMA3_3_70B = "llama3.3:70b"
    LLAMA3_1 = "llama3.1"
    LLAMA3_1_70B = "llama3.1:70b"

    # Alibaba Qwen
    QWEN3_8 = "qwen3.8"
    QWEN3_8_27B = "qwen3.8:27b"
    QWEN3_8_FLASH_NEXT = "qwen3.8-flash-next"
    QWEN3_6 = "qwen3.6"
    QWEN3_6_LATEST = "qwen3.6:latest"
    QWEN3_6_27B = "qwen3.6:27b"
    QWEN3_6_35B = "qwen3.6:35b"
    QWEN3_5 = "qwen3.5"
    QWEN3 = "qwen3"
    QWEN3_CODER = "qwen3-coder"
    QWEN3_CODER_NEXT = "qwen3-coder-next"
    QWEN2_5 = "qwen2.5"
    QWEN2_5_CODER = "qwen2.5-coder"
    QWEN2_5_72B = "qwen2.5:72b"

    # Zhipu GLM
    GLM5_3 = "glm-5.3"
    GLM5_3_FLASH = "glm-5.3-flash"

    # IBM Granite
    GRANITE4_2_3B = "granite4.2:3b"
    GRANITE4_2_8B = "granite4.2:8b"
    GRANITE4_2_30B = "granite4.2:30b"
    GRANITE4_1_3B = "granite4.1:3b"
    GRANITE4_1_8B = "granite4.1:8b"
    GRANITE4_1_30B = "granite4.1:30b"

    # NVIDIA / OpenAI open weights
    NEMOTRON3_5_LIGHTNING = "nemotron-3.5-lightning"
    GPT_OSS_20B = "gpt-oss:20b"

    # Moonshot / DeepSeek
    DEEPSEEK_V4_1_FLASH = "deepseek-v4.1-flash"
    DEEPSEEK_R1 = "deepseek-r1"
    DEEPSEEK_V3 = "deepseek-v3"
    KIMI_K3 = "kimi-k3"
    KIMI_K2_7_CODE = "kimi-k2.7-code"
    KIMI_K2_6 = "kimi-k2.6"
    KIMI_K2_6_CLOUD = "kimi-k2.6:cloud"
    KIMI_K2_5 = "kimi-k2.5"

Attributes

DEEPSEEK_R1 = 'deepseek-r1' class-attribute instance-attribute
DEEPSEEK_V3 = 'deepseek-v3' class-attribute instance-attribute
DEEPSEEK_V4_1_FLASH = 'deepseek-v4.1-flash' class-attribute instance-attribute
GLM5_3 = 'glm-5.3' class-attribute instance-attribute
GLM5_3_FLASH = 'glm-5.3-flash' class-attribute instance-attribute
GPT_OSS_20B = 'gpt-oss:20b' class-attribute instance-attribute
GRANITE4_1_30B = 'granite4.1:30b' class-attribute instance-attribute
GRANITE4_1_3B = 'granite4.1:3b' class-attribute instance-attribute
GRANITE4_1_8B = 'granite4.1:8b' class-attribute instance-attribute
GRANITE4_2_30B = 'granite4.2:30b' class-attribute instance-attribute
GRANITE4_2_3B = 'granite4.2:3b' class-attribute instance-attribute
GRANITE4_2_8B = 'granite4.2:8b' class-attribute instance-attribute
KIMI_K2_5 = 'kimi-k2.5' class-attribute instance-attribute
KIMI_K2_6 = 'kimi-k2.6' class-attribute instance-attribute
KIMI_K2_6_CLOUD = 'kimi-k2.6:cloud' class-attribute instance-attribute
KIMI_K2_7_CODE = 'kimi-k2.7-code' class-attribute instance-attribute
KIMI_K3 = 'kimi-k3' class-attribute instance-attribute
LLAMA3_1 = 'llama3.1' class-attribute instance-attribute
LLAMA3_1_70B = 'llama3.1:70b' class-attribute instance-attribute
LLAMA3_3 = 'llama3.3' class-attribute instance-attribute
LLAMA3_3_70B = 'llama3.3:70b' class-attribute instance-attribute
LLAMA4_SCOUT = 'llama4-scout:17b' class-attribute instance-attribute
MUSE_GLIMMER = 'muse-glimmer' class-attribute instance-attribute
NEMOTRON3_5_LIGHTNING = 'nemotron-3.5-lightning' class-attribute instance-attribute
QWEN2_5 = 'qwen2.5' class-attribute instance-attribute
QWEN2_5_72B = 'qwen2.5:72b' class-attribute instance-attribute
QWEN2_5_CODER = 'qwen2.5-coder' class-attribute instance-attribute
QWEN3 = 'qwen3' class-attribute instance-attribute
QWEN3_5 = 'qwen3.5' class-attribute instance-attribute
QWEN3_6 = 'qwen3.6' class-attribute instance-attribute
QWEN3_6_27B = 'qwen3.6:27b' class-attribute instance-attribute
QWEN3_6_35B = 'qwen3.6:35b' class-attribute instance-attribute
QWEN3_6_LATEST = 'qwen3.6:latest' class-attribute instance-attribute
QWEN3_8 = 'qwen3.8' class-attribute instance-attribute
QWEN3_8_27B = 'qwen3.8:27b' class-attribute instance-attribute
QWEN3_8_FLASH_NEXT = 'qwen3.8-flash-next' class-attribute instance-attribute
QWEN3_CODER = 'qwen3-coder' class-attribute instance-attribute
QWEN3_CODER_NEXT = 'qwen3-coder-next' class-attribute instance-attribute

OntologyValidationConfig

Bases: BaseSettings

Deterministic post-checks applied to LLM-rendered ontology deltas.

Source code in ontocast/config/settings.py
class OntologyValidationConfig(BaseSettings):
    """Deterministic post-checks applied to LLM-rendered ontology deltas."""

    reconcile_minted_terms: Literal["off", "detect", "rewrite"] = Field(
        default="detect",
        description=(
            "What to do when a newly minted term's label or notation exactly "
            "matches an existing term of compatible role in the full catalog."
            " Under vector retrieval the renderer sees only part of the "
            "catalog, so it can mint a duplicate of a term it was not shown. "
            "'detect' logs each pair and changes nothing; 'rewrite' also "
            "replaces the minted IRI with the catalog IRI in the merged "
            "update (enable it once 'detect' shows the matches are true "
            "duplicates); 'off' skips the check."
        ),
    )
    critic_passes: int = Field(
        default=0,
        ge=0,
        description=(
            "Review-and-patch passes per ontology unit, in LLM calls. 0 "
            "(default) disables the ontology critic; each pass adds one call "
            "per unit."
        ),
    )
    critic_max_delete_share: float = Field(
        default=0.10,
        ge=0.0,
        le=1.0,
        description=(
            "Largest share of the delta one critic pass may remove. Stricter "
            "than the facts equivalent because an ontology delete propagates "
            "onto shared, versioned catalog terminals: its blast radius is "
            "every document using the term, not this unit."
        ),
    )
    critic_min_deletes: int = Field(
        default=3,
        ge=0,
        description="Deletions always permitted regardless of share.",
    )
    accept_blocking_finding_kinds: list[str] = Field(
        default_factory=lambda: [
            "foreign_delete",
            "foreign_namespace",
            "subclass_cycle",
            "role_confusion",
        ],
        description=(
            "Deterministic ontology findings that block acceptance. The default "
            "is the destructive-or-lossy subset only. Blocking on every "
            "mandatory finding would put `missing_label` in the set, which "
            "fires whenever a render mints a term without a label -- routine, "
            "and a permanent per-unit tax rather than a defect signal."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="ONTOLOGY_",
        case_sensitive=False,
    )

Attributes

accept_blocking_finding_kinds = Field(default_factory=lambda: ['foreign_delete', 'foreign_namespace', 'subclass_cycle', 'role_confusion'], description='Deterministic ontology findings that block acceptance. The default is the destructive-or-lossy subset only. Blocking on every mandatory finding would put `missing_label` in the set, which fires whenever a render mints a term without a label -- routine, and a permanent per-unit tax rather than a defect signal.') class-attribute instance-attribute
critic_max_delete_share = Field(default=0.1, ge=0.0, le=1.0, description='Largest share of the delta one critic pass may remove. Stricter than the facts equivalent because an ontology delete propagates onto shared, versioned catalog terminals: its blast radius is every document using the term, not this unit.') class-attribute instance-attribute
critic_min_deletes = Field(default=3, ge=0, description='Deletions always permitted regardless of share.') class-attribute instance-attribute
critic_passes = Field(default=0, ge=0, description='Review-and-patch passes per ontology unit, in LLM calls. 0 (default) disables the ontology critic; each pass adds one call per unit.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='ONTOLOGY_', case_sensitive=False) class-attribute instance-attribute
reconcile_minted_terms = Field(default='detect', description="What to do when a newly minted term's label or notation exactly matches an existing term of compatible role in the full catalog. Under vector retrieval the renderer sees only part of the catalog, so it can mint a duplicate of a term it was not shown. 'detect' logs each pair and changes nothing; 'rewrite' also replaces the minted IRI with the catalog IRI in the merged update (enable it once 'detect' shows the matches are true duplicates); 'off' skips the check.") class-attribute instance-attribute

OpenAIModel

Bases: LLMModelNameAbstract

OpenAI model names

Source code in ontocast/config/settings.py
class OpenAIModel(LLMModelNameAbstract):
    """OpenAI model names"""

    # GPT-6: astra (most capable), sol (balanced), luna (high volume)
    GPT6_ASTRA = "gpt-6-astra"
    GPT6_SOL = "gpt-6-sol"
    GPT6_LUNA = "gpt-6-luna"

    # GPT-5.6: sol (deepest reasoning; the bare alias), terra, luna (cheapest)
    GPT5_6 = "gpt-5.6"
    GPT5_6_SOL = "gpt-5.6-sol"
    GPT5_6_TERRA = "gpt-5.6-terra"
    GPT5_6_LUNA = "gpt-5.6-luna"

    GPT5_4_PRO = "gpt-5.4-pro"
    GPT5_4_THINKING = "gpt-5.4-thinking"
    GPT5_4 = "gpt-5.4"
    GPT5_4_MINI = "gpt-5.4-mini"
    GPT5_4_NANO = "gpt-5.4-nano"

    GPT5 = "gpt-5"
    GPT5_MINI = "gpt-5-mini"
    GPT5_NANO = "gpt-5-nano"
    GPT4_1 = "gpt-4.1"
    GPT4_1_MINI = "gpt-4.1-mini"
    GPT4_O = "gpt-4o"
    GPT4_O_MINI = "gpt-4o-mini"

Attributes

GPT4_1 = 'gpt-4.1' class-attribute instance-attribute
GPT4_1_MINI = 'gpt-4.1-mini' class-attribute instance-attribute
GPT4_O = 'gpt-4o' class-attribute instance-attribute
GPT4_O_MINI = 'gpt-4o-mini' class-attribute instance-attribute
GPT5 = 'gpt-5' class-attribute instance-attribute
GPT5_4 = 'gpt-5.4' class-attribute instance-attribute
GPT5_4_MINI = 'gpt-5.4-mini' class-attribute instance-attribute
GPT5_4_NANO = 'gpt-5.4-nano' class-attribute instance-attribute
GPT5_4_PRO = 'gpt-5.4-pro' class-attribute instance-attribute
GPT5_4_THINKING = 'gpt-5.4-thinking' class-attribute instance-attribute
GPT5_6 = 'gpt-5.6' class-attribute instance-attribute
GPT5_6_LUNA = 'gpt-5.6-luna' class-attribute instance-attribute
GPT5_6_SOL = 'gpt-5.6-sol' class-attribute instance-attribute
GPT5_6_TERRA = 'gpt-5.6-terra' class-attribute instance-attribute
GPT5_MINI = 'gpt-5-mini' class-attribute instance-attribute
GPT5_NANO = 'gpt-5-nano' class-attribute instance-attribute
GPT6_ASTRA = 'gpt-6-astra' class-attribute instance-attribute
GPT6_LUNA = 'gpt-6-luna' class-attribute instance-attribute
GPT6_SOL = 'gpt-6-sol' class-attribute instance-attribute

PatchRetrievalConfig

Bases: BaseSettings

Scoring, filtering, and capping of ontology atoms after vector search (backend-agnostic).

The path is intentionally simple: per-window channel fusion → max-score IRI dedupe → per-ontology round-robin → window-scaled hard cap. Merged-score ratio and MMR remain available as advanced opt-in (non-default) controls.

Source code in ontocast/config/settings.py
class PatchRetrievalConfig(BaseSettings):
    """Scoring, filtering, and capping of ontology atoms after vector search (backend-agnostic).

    The path is intentionally simple: per-window channel fusion → max-score IRI
    dedupe → per-ontology round-robin → window-scaled hard cap. Merged-score
    ratio and MMR remain available as advanced opt-in (non-default) controls.
    """

    min_merged_max_score: float = Field(
        default=0.18,
        ge=0.0,
        description=(
            "Relevance floor below which a unit is treated as having no "
            "relevant ontology and gets an empty patch. A fraction of the "
            "best fused score a window can reach (an atom ranked first in "
            "every lane), not an absolute score, so it stays meaningful when "
            "lane weights or VECTOR_STORE_FUSION_RANK_CONSTANT change. 0 "
            "disables it."
        ),
    )
    merged_score_ratio: float = Field(
        default=0.0,
        ge=0.0,
        le=1.0,
        description=(
            "Advanced: after merging hits across queries, keep atoms whose score is at "
            "least this fraction of the merged top score. 0 disables (default)."
        ),
    )
    cross_query_merge_mode: CrossQueryMergeMode = Field(
        default=CrossQueryMergeMode.MAX_SCORE,
        description=(
            "Cross-window merge: max_score (default; entity best score across "
            "windows) or sum_score (sum of per-window scores, so a term several "
            "windows agree on outranks one window's top hit). Both are followed "
            "by the same round-robin / cap stage, and single-window retrieval "
            "makes them identical."
        ),
    )
    per_ontology_seed_quota: int = Field(
        default=0,
        ge=0,
        description=(
            "Maximum seeds kept per ontology when filling the seed list "
            "round-robin. 0 means no per-ontology cap: seeds are taken in "
            "global score order, so the budget is not spread across "
            "ontologies that merely scored something."
        ),
    )
    per_ontology_atom_floor: int = Field(
        default=2,
        ge=0,
        description=(
            "Reserve pass before the global fill: each ontology contributing "
            "candidates is guaranteed min(floor, its candidate count) seed "
            "slots, allocated round-robin. Unlike per_ontology_seed_quota "
            "(a ceiling), the floor protects small modules from being starved "
            "by one dominant ontology at the atom cap. 0 disables."
        ),
    )
    small_module_closure_max_triples: int = Field(
        default=300,
        ge=0,
        description=(
            "Include a source ontology whole (header stripped) when it has at"
            " least one retrieved atom and at most this many triples. A small"
            " vocabulary shown in part pushes the renderer to invent "
            "near-miss property names, and modules such as qualified-quantity"
            " or observation patterns are only useful whole. Takes effect "
            "only for modules that win a seed, so it pairs with "
            "per_ontology_atom_floor. 0 disables it."
        ),
    )
    small_module_closure_max_total_triples: int | None = Field(
        default=None,
        ge=0,
        description=(
            "Ceiling on the triples that whole-module inclusions may add to "
            "one snapshot; unset means no ceiling. Modules are admitted in "
            "order of their best atom's retrieval score until the budget is "
            "spent, and a module too large for what remains is skipped so a "
            "smaller one can still fit. Ordering by score rather than seed "
            "count keeps small, sharply relevant vocabularies in. See "
            "module_closure_iris, module_closure_declined_iris and "
            "module_closure_triples in the run manifest."
        ),
    )
    per_role_atom_floor: int = Field(
        default=12,
        ge=0,
        description=(
            "Reserve pass guaranteeing predicate-role atoms a share of the seed "
            "budget before the global fill, in the same floor-not-ceiling shape "
            "as per_ontology_atom_floor. Dense similarity between prose and a "
            "noun phrase beats a verb phrase, so classes and individuals win a "
            "shared ranking and the properties carrying the graph structure are "
            "crowded out. 0 disables."
        ),
    )
    schema_closure_max_entities: int = Field(
        default=32,
        ge=0,
        description=(
            "Cap on terms admitted by rdfs:domain/rdfs:range closure over the "
            "retrieved seeds: properties whose domain or range names an admitted "
            "class (or its ancestors), and the domain/range classes of admitted "
            "properties. A class with no property that can link it is inert "
            "context. 0 disables."
        ),
    )
    schema_closure_ancestor_depth: int = Field(
        default=2,
        ge=0,
        description=(
            "How far to walk rdfs:subClassOf upward when matching a property's "
            "declared domain/range against an admitted class. Properties are "
            "usually declared on an ancestor of the class the text mentions."
        ),
    )
    mmr_lambda: float = Field(
        default=1.0,
        ge=0.0,
        le=1.0,
        description=(
            "MMR trade-off over dense core+neighborhood vectors: 1.0 keeps pure relevance "
            "(default; skips MMR), lower values increase diversity."
        ),
    )
    seeds_per_window: int = Field(
        default=4,
        ge=1,
        description=(
            "Target seeds per proposition window when scaling the effective atom cap: "
            "min(max_atoms, max(max_atoms_base, seeds_per_window * n_queries))."
        ),
    )
    max_atoms_base: int = Field(
        default=96,
        ge=0,
        description=(
            "Minimum effective atom cap before window scaling (0 defers "
            "entirely to seeds_per_window * n_queries). The cap does not grow"
            " with catalog size, so a floor below max_atoms discards "
            "candidates the per-lane top_k has already retrieved."
        ),
    )
    max_atoms: int = Field(
        default=96,
        ge=0,
        description=(
            "Hard cap on atoms kept after merging and optional MMR; 0 means "
            "unlimited. The effective cap is min(max_atoms, "
            "max(max_atoms_base, seeds_per_window * n_queries)). On "
            "multi-window input this is the main lever on how many relevant "
            "terms reach the snapshot; above it, the induced-subgraph triple "
            "budget becomes the limit."
        ),
    )

    dump_ontology_ranks: bool = Field(
        default=False,
        description=(
            "Collect per-ontology rank diagnostics (best rank/score per channel, fused "
            "rank, whether the ontology survived the atom cut) into retrieval metrics "
            "under 'ontology_rank_diagnostics'. Diagnostic only: it walks every channel "
            "hit list per query and does not change retrieval behaviour."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="ONTOLOGY_PATCH_",
        case_sensitive=False,
    )

    def effective_max_atoms(self, n_queries: int) -> int:
        """Window-scaled atom budget: min(hard_cap, max(base, seeds_per_window * n))."""
        if self.max_atoms <= 0:
            return 0
        windows = max(n_queries, 1)
        scaled = max(self.max_atoms_base, self.seeds_per_window * windows)
        return min(self.max_atoms, scaled)

Attributes

cross_query_merge_mode = Field(default=CrossQueryMergeMode.MAX_SCORE, description="Cross-window merge: max_score (default; entity best score across windows) or sum_score (sum of per-window scores, so a term several windows agree on outranks one window's top hit). Both are followed by the same round-robin / cap stage, and single-window retrieval makes them identical.") class-attribute instance-attribute
dump_ontology_ranks = Field(default=False, description="Collect per-ontology rank diagnostics (best rank/score per channel, fused rank, whether the ontology survived the atom cut) into retrieval metrics under 'ontology_rank_diagnostics'. Diagnostic only: it walks every channel hit list per query and does not change retrieval behaviour.") class-attribute instance-attribute
max_atoms = Field(default=96, ge=0, description='Hard cap on atoms kept after merging and optional MMR; 0 means unlimited. The effective cap is min(max_atoms, max(max_atoms_base, seeds_per_window * n_queries)). On multi-window input this is the main lever on how many relevant terms reach the snapshot; above it, the induced-subgraph triple budget becomes the limit.') class-attribute instance-attribute
max_atoms_base = Field(default=96, ge=0, description='Minimum effective atom cap before window scaling (0 defers entirely to seeds_per_window * n_queries). The cap does not grow with catalog size, so a floor below max_atoms discards candidates the per-lane top_k has already retrieved.') class-attribute instance-attribute
merged_score_ratio = Field(default=0.0, ge=0.0, le=1.0, description='Advanced: after merging hits across queries, keep atoms whose score is at least this fraction of the merged top score. 0 disables (default).') class-attribute instance-attribute
min_merged_max_score = Field(default=0.18, ge=0.0, description='Relevance floor below which a unit is treated as having no relevant ontology and gets an empty patch. A fraction of the best fused score a window can reach (an atom ranked first in every lane), not an absolute score, so it stays meaningful when lane weights or VECTOR_STORE_FUSION_RANK_CONSTANT change. 0 disables it.') class-attribute instance-attribute
mmr_lambda = Field(default=1.0, ge=0.0, le=1.0, description='MMR trade-off over dense core+neighborhood vectors: 1.0 keeps pure relevance (default; skips MMR), lower values increase diversity.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='ONTOLOGY_PATCH_', case_sensitive=False) class-attribute instance-attribute
per_ontology_atom_floor = Field(default=2, ge=0, description='Reserve pass before the global fill: each ontology contributing candidates is guaranteed min(floor, its candidate count) seed slots, allocated round-robin. Unlike per_ontology_seed_quota (a ceiling), the floor protects small modules from being starved by one dominant ontology at the atom cap. 0 disables.') class-attribute instance-attribute
per_ontology_seed_quota = Field(default=0, ge=0, description='Maximum seeds kept per ontology when filling the seed list round-robin. 0 means no per-ontology cap: seeds are taken in global score order, so the budget is not spread across ontologies that merely scored something.') class-attribute instance-attribute
per_role_atom_floor = Field(default=12, ge=0, description='Reserve pass guaranteeing predicate-role atoms a share of the seed budget before the global fill, in the same floor-not-ceiling shape as per_ontology_atom_floor. Dense similarity between prose and a noun phrase beats a verb phrase, so classes and individuals win a shared ranking and the properties carrying the graph structure are crowded out. 0 disables.') class-attribute instance-attribute
schema_closure_ancestor_depth = Field(default=2, ge=0, description="How far to walk rdfs:subClassOf upward when matching a property's declared domain/range against an admitted class. Properties are usually declared on an ancestor of the class the text mentions.") class-attribute instance-attribute
schema_closure_max_entities = Field(default=32, ge=0, description='Cap on terms admitted by rdfs:domain/rdfs:range closure over the retrieved seeds: properties whose domain or range names an admitted class (or its ancestors), and the domain/range classes of admitted properties. A class with no property that can link it is inert context. 0 disables.') class-attribute instance-attribute
seeds_per_window = Field(default=4, ge=1, description='Target seeds per proposition window when scaling the effective atom cap: min(max_atoms, max(max_atoms_base, seeds_per_window * n_queries)).') class-attribute instance-attribute
small_module_closure_max_total_triples = Field(default=None, ge=0, description="Ceiling on the triples that whole-module inclusions may add to one snapshot; unset means no ceiling. Modules are admitted in order of their best atom's retrieval score until the budget is spent, and a module too large for what remains is skipped so a smaller one can still fit. Ordering by score rather than seed count keeps small, sharply relevant vocabularies in. See module_closure_iris, module_closure_declined_iris and module_closure_triples in the run manifest.") class-attribute instance-attribute
small_module_closure_max_triples = Field(default=300, ge=0, description='Include a source ontology whole (header stripped) when it has at least one retrieved atom and at most this many triples. A small vocabulary shown in part pushes the renderer to invent near-miss property names, and modules such as qualified-quantity or observation patterns are only useful whole. Takes effect only for modules that win a seed, so it pairs with per_ontology_atom_floor. 0 disables it.') class-attribute instance-attribute

Methods:

effective_max_atoms(n_queries)

Window-scaled atom budget: min(hard_cap, max(base, seeds_per_window * n)).

Source code in ontocast/config/settings.py
def effective_max_atoms(self, n_queries: int) -> int:
    """Window-scaled atom budget: min(hard_cap, max(base, seeds_per_window * n))."""
    if self.max_atoms <= 0:
        return 0
    windows = max(n_queries, 1)
    scaled = max(self.max_atoms_base, self.seeds_per_window * windows)
    return min(self.max_atoms, scaled)

PathConfig

Bases: BaseSettings

Path and directory configuration.

Source code in ontocast/config/settings.py
class PathConfig(BaseSettings):
    """Path and directory configuration."""

    ontology_directory: Path | None = Field(
        default=None,
        description=(
            "Directory of seed ontology *.ttl files, read once at startup. "
            "Read-only: ingestion never writes here and deletion never removes "
            "files from it"
        ),
    )
    cache_dir: Path | None = Field(
        default=None, description="Cache directory for LLM responses and tool outputs"
    )
    cache_max_bytes: int | None = Field(
        default=DEFAULT_CACHE_MAX_BYTES,
        description=(
            "Size ceiling for the whole cache directory. Once exceeded, "
            "least-recently-used entries are deleted until the total fits. "
            "Accepts a byte count or a human size such as '1GB' or '500MB'. "
            "Set to 0 to disable automatic pruning."
        ),
    )
    cache_ttl_days: int | None = Field(
        default=None,
        description=(
            "Delete cache entries not used for this many days, applied before"
            " the size ceiling. None disables the age cut."
        ),
    )
    cache_prune_every: int = Field(
        default=DEFAULT_CACHE_PRUNE_EVERY,
        ge=1,
        description=(
            "Re-check the cache size ceiling after this many writes. The check "
            "walks the cache tree, so it is amortised rather than run per write."
        ),
    )

    @field_validator("cache_max_bytes", mode="before")
    @classmethod
    def _parse_cache_max_bytes(cls, value: object) -> object:
        """Accept human-readable sizes ('1GB', '500MB') alongside raw byte counts."""
        if not isinstance(value, str):
            return value
        text = value.strip().upper().replace("IB", "B")
        if not text:
            return None
        units = {"B": 1, "KB": 1024, "MB": 1024**2, "GB": 1024**3, "TB": 1024**4}
        for suffix, factor in sorted(units.items(), key=lambda kv: -len(kv[0])):
            if text.endswith(suffix):
                number = text[: -len(suffix)].strip()
                if not number:
                    break
                return int(float(number) * factor)
        return int(float(text))

    model_config = SettingsConfigDict(
        env_prefix="ONTOCAST_",
        case_sensitive=False,
    )

Attributes

cache_dir = Field(default=None, description='Cache directory for LLM responses and tool outputs') class-attribute instance-attribute
cache_max_bytes = Field(default=DEFAULT_CACHE_MAX_BYTES, description="Size ceiling for the whole cache directory. Once exceeded, least-recently-used entries are deleted until the total fits. Accepts a byte count or a human size such as '1GB' or '500MB'. Set to 0 to disable automatic pruning.") class-attribute instance-attribute
cache_prune_every = Field(default=DEFAULT_CACHE_PRUNE_EVERY, ge=1, description='Re-check the cache size ceiling after this many writes. The check walks the cache tree, so it is amortised rather than run per write.') class-attribute instance-attribute
cache_ttl_days = Field(default=None, description='Delete cache entries not used for this many days, applied before the size ceiling. None disables the age cut.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='ONTOCAST_', case_sensitive=False) class-attribute instance-attribute
ontology_directory = Field(default=None, description='Directory of seed ontology *.ttl files, read once at startup. Read-only: ingestion never writes here and deletion never removes files from it') class-attribute instance-attribute

QdrantConfig

Bases: BaseSettings

Qdrant-specific vector store connection settings.

Source code in ontocast/config/settings.py
class QdrantConfig(BaseSettings):
    """Qdrant-specific vector store connection settings."""

    uri: str | None = Field(
        default=None,
        description=(
            "Qdrant server URL, such as http://localhost:6333. Setting it "
            "selects Qdrant as the vector store."
        ),
    )
    api_key: str | None = Field(
        default=None, description="API key for a Qdrant server that requires one."
    )
    ontology_collection: str | None = Field(
        default=None,
        description=(
            "Qdrant collection for ontology atom vectors; derived like "
            "FUSEKI_DATASET, and replaced the same way by ontocast serve and "
            "ontocast process."
        ),
    )
    facts_collection: str | None = Field(
        default=None,
        description=(
            "Qdrant collection reserved for future fact vectors; created on init."
        ),
    )
    grpc_port: int = Field(
        default=6334, description="Qdrant gRPC port, used when QDRANT_USE_GRPC is on."
    )
    use_grpc: bool = Field(
        default=False,
        description="Talk to Qdrant over gRPC instead of HTTP.",
    )
    vector_size: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Vector size override. When set, must equal EmbeddingConfig.dimension; "
            "when unset, the embedding dimension is used."
        ),
    )
    distance: VectorDistance = Field(
        default=VectorDistance.COSINE,
        description=(
            "Qdrant vector distance when creating collections "
            "(Cosine, Dot, Euclid, Manhattan; same as qdrant_client Distance)."
        ),
    )
    upsert_batch_size: int = Field(
        default=256,
        ge=1,
        description="Batch size used for Qdrant upsert operations.",
    )
    timeout_seconds: int = Field(
        default=30,
        ge=1,
        description=(
            "Per-request timeout for Qdrant calls, in whole seconds (the client "
            "accepts nothing finer). Without one, an unreachable or hung Qdrant "
            "blocks a pipeline worker indefinitely."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="QDRANT_",
        case_sensitive=False,
    )

    @model_validator(mode="after")
    def _resolve_qdrant_collections(self) -> QdrantConfig:
        if self.ontology_collection is None:
            self.ontology_collection = tenant_project_ontologies_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        if self.facts_collection is None:
            self.facts_collection = tenant_project_facts_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        return self

Attributes

api_key = Field(default=None, description='API key for a Qdrant server that requires one.') class-attribute instance-attribute
distance = Field(default=VectorDistance.COSINE, description='Qdrant vector distance when creating collections (Cosine, Dot, Euclid, Manhattan; same as qdrant_client Distance).') class-attribute instance-attribute
facts_collection = Field(default=None, description='Qdrant collection reserved for future fact vectors; created on init.') class-attribute instance-attribute
grpc_port = Field(default=6334, description='Qdrant gRPC port, used when QDRANT_USE_GRPC is on.') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='QDRANT_', case_sensitive=False) class-attribute instance-attribute
ontology_collection = Field(default=None, description='Qdrant collection for ontology atom vectors; derived like FUSEKI_DATASET, and replaced the same way by ontocast serve and ontocast process.') class-attribute instance-attribute
timeout_seconds = Field(default=30, ge=1, description='Per-request timeout for Qdrant calls, in whole seconds (the client accepts nothing finer). Without one, an unreachable or hung Qdrant blocks a pipeline worker indefinitely.') class-attribute instance-attribute
upsert_batch_size = Field(default=256, ge=1, description='Batch size used for Qdrant upsert operations.') class-attribute instance-attribute
uri = Field(default=None, description='Qdrant server URL, such as http://localhost:6333. Setting it selects Qdrant as the vector store.') class-attribute instance-attribute
use_grpc = Field(default=False, description='Talk to Qdrant over gRPC instead of HTTP.') class-attribute instance-attribute
vector_size = Field(default=None, ge=1, description='Vector size override. When set, must equal EmbeddingConfig.dimension; when unset, the embedding dimension is used.') class-attribute instance-attribute

ServerConfig

Bases: BaseSettings

Server configuration settings.

Source code in ontocast/config/settings.py
class ServerConfig(BaseSettings):
    """Server configuration settings."""

    host: str = Field(
        default="127.0.0.1",
        description=(
            "Interface the server binds to. Defaults to loopback: the server "
            "has no authentication and exposes a destructive /flush, so "
            "binding every interface must be a deliberate choice. Set to "
            "0.0.0.0 for containers."
        ),
    )
    port: int = Field(
        default=8999, ge=1, le=65535, description="Port the server listens on."
    )
    max_visits_per_node: int = Field(
        default=1,
        ge=1,
        description=(
            "Retries of a render that failed outright (unparseable response, "
            "provider error). A render that succeeds is not repeated; "
            "improving it is the critic's job (FACTS_CRITIC_PASSES, "
            "ONTOLOGY_CRITIC_PASSES)."
        ),
        validation_alias=AliasChoices("max_visits_per_node", "max_visits"),
    )
    render_mode: RenderMode = Field(
        default=RenderMode.ONTOLOGY_AND_FACTS,
        description="Rendering mode: ontology, facts, or ontology_and_facts.",
    )
    llm_graph_format: LLMGraphFormat = Field(
        default=LLMGraphFormat.TURTLE,
        description=(
            "Format the LLM writes RDF graphs in: 'turtle' (Turtle strings) or "
            "'jsonld' (compact JSON-LD objects). Turtle spends fewer tokens per "
            "triple and cannot write an IRI object as a string by accident; "
            "'jsonld' suits providers whose structured output handles long "
            "strings worse than nested objects."
        ),
    )
    llm_output_layout: LLMOutputLayout = Field(
        default=LLMOutputLayout.COMPACT,
        description=(
            "Whitespace the LLM is asked to use in its structured responses: "
            "'compact' asks for minified JSON and one-line-per-subject Turtle "
            "without indentation; 'free' gives no instruction (models "
            "typically indent JSON). Indentation is billed as output tokens "
            "and carries nothing the parser reads. Applies to every call that "
            "emits a graph payload."
        ),
    )
    ontology_chapter_format: OntologyChapterFormat = Field(
        default=OntologyChapterFormat.AUTO,
        description=(
            "Syntax of the ontology chapter in the facts render and critic "
            "prompts. 'inherit' follows LLM_GRAPH_FORMAT; 'turtle' always "
            "uses Turtle, which is shorter than JSON-LD; 'term_sheet' lists "
            "one line per term (name, labels, type, hierarchy, domain and "
            "range, usage) and is the shortest. 'term_sheet' needs "
            "RENDER_MODE=facts, because the ontology loop patches the "
            "statements in its chapter. 'auto' picks 'term_sheet' for "
            "facts-only runs and 'inherit' otherwise. Only the prompt context"
            " changes; the model's output keeps LLM_GRAPH_FORMAT. Part of the"
            " LLM cache key for facts calls."
        ),
    )
    ontology_text_max_chars_naming: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Character cap on each rdfs:label, skos:prefLabel and "
            "skos:altLabel in the ontology chapter; unset disables it. Long "
            "names are clipped at a word boundary with a visible marker, so "
            "the model can tell a clipped name from a complete one. Part of "
            "the LLM cache key."
        ),
    )
    ontology_text_max_chars_contract: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Character cap on skos:scopeNote / skos:definition, the "
            "statements saying when a term applies. Worth clipping rather "
            "than dropping: a scope note's first sentence usually carries the"
            " contract and the rest elaborates. Joins the LLM cache key. None"
            " disables the cap."
        ),
    )
    ontology_text_max_chars_prose: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Character cap on rdfs:comment and the remaining SKOS notes -- "
            "description aimed at someone browsing the ontology rather than "
            "at an extractor. Joins the LLM cache key. None disables the cap."
        ),
    )
    ontology_text_total_budget: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Ceiling on the total length of all text literals in one ontology "
            "chapter, for catalogs with very many short terms. Over budget, "
            "literals are shortened, never removed: prose first, then usage "
            "contracts, then names, each to the largest cap that meets the "
            "budget, down to a floor; a chapter that still does not fit is used "
            "as is. Reported in budget.counters as chapter/text_chars_before, "
            "chapter/text_chars_after, chapter/literals_clipped and "
            "chapter/text_over_budget. Part of the LLM cache key."
        ),
    )
    ontology_context_mode: OntologyContextMode = Field(
        default=OntologyContextMode.SELECTED_SINGLE_ONTOLOGY,
        description=(
            "Per-unit ontology context: selected_single_ontology (LLM-picked catalog; "
            "costs one extra LLM call per content unit), selected_vector_search_ontology "
            "(vector-store stitched ensemble; Qdrant or LanceDB), or fixed_single_ontology "
            "(catalog ontology_id; requires ontology_context_fixed_ontology_id)."
        ),
    )
    ontology_context_fixed_ontology_id: str = Field(
        default="",
        description=(
            "Catalog ontology (IRI, ontology_id or author prefix) used in "
            "fixed_single_ontology mode, by ontocast process and by HTTP requests "
            "that name none. Setting it does not change the mode."
        ),
    )
    ontology_max_triples: int | None = Field(
        default=None,
        ge=1,
        description="Runaway-growth backstop on the per-unit ontology working "
        "graph: an update whose result would exceed this is skipped with a "
        "warning, all-or-nothing. Not a prompt bound -- use "
        "ontology_context_max_triples for context size. None (default) disables it.",
    )
    ontology_context_required: bool = Field(
        default=False,
        description=(
            "Fail the run when a facts unit's ontology context is empty, instead "
            "of extracting without a catalog. With no context the renderer falls "
            "back on generic vocabulary and SHACL has nothing to check, so the "
            "run would report a meaningless pass. Turn it on when extracting "
            "against a curated catalog. Off by default because the default render"
            " mode builds ontologies as it goes. Never applies to ontology units,"
            " for which an empty context means 'create a new ontology'."
        ),
    )

    @property
    def ontology_text_caps(self) -> TextCaps:
        """The four text-cap knobs as one value, for the chapter builders.

        Returns a :class:`TextCaps` whose ``active`` is False when nothing is
        set, which the chapter path treats as "leave every literal as authored"
        -- byte-for-byte, so a deployment that sets none of these cannot see its
        prompts or its cache keys move.
        """
        return TextCaps(
            naming=self.ontology_text_max_chars_naming,
            contract=self.ontology_text_max_chars_contract,
            prose=self.ontology_text_max_chars_prose,
            total_budget=self.ontology_text_total_budget,
        )

    ontology_context_scope: OntologyContextScope = Field(
        default=OntologyContextScope.UNIT,
        description=(
            "Resolve the ontology chapter per content unit ('unit') or once "
            "per document ('document'). Per-unit chapters are smaller but all"
            " different, so no call shares a prompt prefix with another. "
            "'document' shows every unit the union of the per-unit contexts: "
            "larger, but identical across the document, so a provider's "
            "prefix cache serves every call after the first. Each unit still "
            "sees every term its own retrieval chose, plus its siblings'. "
            "Takes effect only in facts-only runs: after an ontology stage, "
            "facts units already share one merged document context. Pair it "
            "with FANOUT_WARMUP_UNITS."
        ),
    )
    fanout_warmup_units: int = Field(
        default=0,
        ge=0,
        description=(
            "Content units to complete before the rest are run concurrently; "
            "0 runs all at once. A provider's prefix cache is filled by a "
            "completed request, so calls sent together all miss it. Useful "
            "only when calls share a prefix "
            "(ONTOLOGY_CONTEXT_SCOPE=document); costs the time of the warm-up"
            " units."
        ),
    )
    ontology_context_max_triples: int | None = Field(
        default=4000,
        ge=1,
        description="Triple budget for the ontology context serialized into a "
        "prompt, in every ontology_context_mode. Over budget, the least "
        "load-bearing triples are dropped first (header/list noise, then "
        "redundant structure, then comments and definitions); labels, types, "
        "hierarchy and domain/range are never dropped, so this is best-effort "
        "and a graph that cannot fit is passed through with a warning. "
        "In selected_vector_search_ontology with unit scope, "
        "VECTOR_STORE_INDUCED_SUBGRAPH_MAX_TOTAL_TRIPLES caps the context first. "
        "None disables condensing.",
    )
    parallel_workers: int = Field(
        default=16,
        ge=1,
        description=(
            "Maximum content units processed at once within one document. A "
            "unit makes one LLM call at a time, so this is also the "
            "concurrency one document puts on the provider. LLM_MAX_INFLIGHT "
            "caps calls across all documents and is the one to lower when a "
            "provider rate-limits. If a stage's loop_lag_total in "
            "budget.node_durations is a large share of its wall-clock time, "
            "more workers will slow it down."
        ),
    )
    enable_ontology_consolidation: bool = Field(
        default=False,
        description="Run optional ontology consolidation pass after normalization",
    )
    max_concurrent_processes: int | None = Field(
        default=None,
        ge=1,
        description=(
            "When set, limit concurrent /process and /process_unit handlers. "
            "Requests beyond the limit queue until a slot frees up; they are "
            "not rejected."
        ),
    )
    max_tenancy_scopes: int = Field(
        default=16,
        ge=1,
        description=(
            "How many tenant/project ToolBoxes to keep resident. Each holds a "
            "triple store connection and an ontology catalog; the expensive "
            "tools (LLM client, converter, embedding model) are shared across "
            "all of them. Least-recently-used scopes are evicted and closed. "
            "Bounded because scopes come from request parameters."
        ),
    )

    model_config = SettingsConfigDict(
        case_sensitive=False,
    )

    @model_validator(mode="after")
    def validate_ontology_chapter_format(self) -> "ServerConfig":
        """Resolve 'auto' against the render mode, and reject an illegal ask.

        The facts renderer reads its chapter and writes an unrelated graph, so
        the chapter is free to be any representation that names the terms. The
        ontology renderer and its critic write a *patch against the statements
        in the chapter*, which a line-per-term listing cannot express -- there
        is nothing to insert into or delete from.

        So the cheapest legal chapter differs by mode, and ``auto`` picks it
        rather than asking an operator to know which. It is resolved **here**,
        not at the point of use: every consumer reads this field, and a value
        that still meant "decide later" would reach the prompt profile, the
        LLM cache key and the run manifest as a name for no chapter in
        particular.

        An *explicit* ``term_sheet`` on a mode that cannot read one still
        fails. Falling back would be silent, and the run would spend an
        ontology pass producing patches nobody could apply while the manifest
        recorded a setting that never took effect -- which is exactly what
        ``auto`` exists to make unnecessary.
        """
        if self.ontology_chapter_format == OntologyChapterFormat.AUTO:
            self.ontology_chapter_format = (
                OntologyChapterFormat.TERM_SHEET
                if self.render_mode == RenderMode.FACTS
                else OntologyChapterFormat.INHERIT
            )
            return self
        if (
            self.ontology_chapter_format == OntologyChapterFormat.TERM_SHEET
            and self.render_mode != RenderMode.FACTS
        ):
            raise ValueError(
                "ONTOLOGY_CHAPTER_FORMAT=term_sheet requires RENDER_MODE=facts: "
                f"got render_mode={self.render_mode.value}. The ontology loop "
                "emits a patch against the statements in its chapter, so that "
                "chapter has to be a graph. Use 'turtle' for a cheaper chapter "
                "that stays one, or 'auto' to take whichever the mode allows."
            )
        return self

Attributes

enable_ontology_consolidation = Field(default=False, description='Run optional ontology consolidation pass after normalization') class-attribute instance-attribute
fanout_warmup_units = Field(default=0, ge=0, description="Content units to complete before the rest are run concurrently; 0 runs all at once. A provider's prefix cache is filled by a completed request, so calls sent together all miss it. Useful only when calls share a prefix (ONTOLOGY_CONTEXT_SCOPE=document); costs the time of the warm-up units.") class-attribute instance-attribute
host = Field(default='127.0.0.1', description='Interface the server binds to. Defaults to loopback: the server has no authentication and exposes a destructive /flush, so binding every interface must be a deliberate choice. Set to 0.0.0.0 for containers.') class-attribute instance-attribute
llm_graph_format = Field(default=LLMGraphFormat.TURTLE, description="Format the LLM writes RDF graphs in: 'turtle' (Turtle strings) or 'jsonld' (compact JSON-LD objects). Turtle spends fewer tokens per triple and cannot write an IRI object as a string by accident; 'jsonld' suits providers whose structured output handles long strings worse than nested objects.") class-attribute instance-attribute
llm_output_layout = Field(default=LLMOutputLayout.COMPACT, description="Whitespace the LLM is asked to use in its structured responses: 'compact' asks for minified JSON and one-line-per-subject Turtle without indentation; 'free' gives no instruction (models typically indent JSON). Indentation is billed as output tokens and carries nothing the parser reads. Applies to every call that emits a graph payload.") class-attribute instance-attribute
max_concurrent_processes = Field(default=None, ge=1, description='When set, limit concurrent /process and /process_unit handlers. Requests beyond the limit queue until a slot frees up; they are not rejected.') class-attribute instance-attribute
max_tenancy_scopes = Field(default=16, ge=1, description='How many tenant/project ToolBoxes to keep resident. Each holds a triple store connection and an ontology catalog; the expensive tools (LLM client, converter, embedding model) are shared across all of them. Least-recently-used scopes are evicted and closed. Bounded because scopes come from request parameters.') class-attribute instance-attribute
max_visits_per_node = Field(default=1, ge=1, description="Retries of a render that failed outright (unparseable response, provider error). A render that succeeds is not repeated; improving it is the critic's job (FACTS_CRITIC_PASSES, ONTOLOGY_CRITIC_PASSES).", validation_alias=AliasChoices('max_visits_per_node', 'max_visits')) class-attribute instance-attribute
model_config = SettingsConfigDict(case_sensitive=False) class-attribute instance-attribute
ontology_chapter_format = Field(default=OntologyChapterFormat.AUTO, description="Syntax of the ontology chapter in the facts render and critic prompts. 'inherit' follows LLM_GRAPH_FORMAT; 'turtle' always uses Turtle, which is shorter than JSON-LD; 'term_sheet' lists one line per term (name, labels, type, hierarchy, domain and range, usage) and is the shortest. 'term_sheet' needs RENDER_MODE=facts, because the ontology loop patches the statements in its chapter. 'auto' picks 'term_sheet' for facts-only runs and 'inherit' otherwise. Only the prompt context changes; the model's output keeps LLM_GRAPH_FORMAT. Part of the LLM cache key for facts calls.") class-attribute instance-attribute
ontology_context_fixed_ontology_id = Field(default='', description='Catalog ontology (IRI, ontology_id or author prefix) used in fixed_single_ontology mode, by ontocast process and by HTTP requests that name none. Setting it does not change the mode.') class-attribute instance-attribute
ontology_context_max_triples = Field(default=4000, ge=1, description='Triple budget for the ontology context serialized into a prompt, in every ontology_context_mode. Over budget, the least load-bearing triples are dropped first (header/list noise, then redundant structure, then comments and definitions); labels, types, hierarchy and domain/range are never dropped, so this is best-effort and a graph that cannot fit is passed through with a warning. In selected_vector_search_ontology with unit scope, VECTOR_STORE_INDUCED_SUBGRAPH_MAX_TOTAL_TRIPLES caps the context first. None disables condensing.') class-attribute instance-attribute
ontology_context_mode = Field(default=OntologyContextMode.SELECTED_SINGLE_ONTOLOGY, description='Per-unit ontology context: selected_single_ontology (LLM-picked catalog; costs one extra LLM call per content unit), selected_vector_search_ontology (vector-store stitched ensemble; Qdrant or LanceDB), or fixed_single_ontology (catalog ontology_id; requires ontology_context_fixed_ontology_id).') class-attribute instance-attribute
ontology_context_required = Field(default=False, description="Fail the run when a facts unit's ontology context is empty, instead of extracting without a catalog. With no context the renderer falls back on generic vocabulary and SHACL has nothing to check, so the run would report a meaningless pass. Turn it on when extracting against a curated catalog. Off by default because the default render mode builds ontologies as it goes. Never applies to ontology units, for which an empty context means 'create a new ontology'.") class-attribute instance-attribute
ontology_context_scope = Field(default=OntologyContextScope.UNIT, description="Resolve the ontology chapter per content unit ('unit') or once per document ('document'). Per-unit chapters are smaller but all different, so no call shares a prompt prefix with another. 'document' shows every unit the union of the per-unit contexts: larger, but identical across the document, so a provider's prefix cache serves every call after the first. Each unit still sees every term its own retrieval chose, plus its siblings'. Takes effect only in facts-only runs: after an ontology stage, facts units already share one merged document context. Pair it with FANOUT_WARMUP_UNITS.") class-attribute instance-attribute
ontology_max_triples = Field(default=None, ge=1, description='Runaway-growth backstop on the per-unit ontology working graph: an update whose result would exceed this is skipped with a warning, all-or-nothing. Not a prompt bound -- use ontology_context_max_triples for context size. None (default) disables it.') class-attribute instance-attribute
ontology_text_caps property

The four text-cap knobs as one value, for the chapter builders.

Returns a :class:TextCaps whose active is False when nothing is set, which the chapter path treats as "leave every literal as authored" -- byte-for-byte, so a deployment that sets none of these cannot see its prompts or its cache keys move.

ontology_text_max_chars_contract = Field(default=None, ge=1, description="Character cap on skos:scopeNote / skos:definition, the statements saying when a term applies. Worth clipping rather than dropping: a scope note's first sentence usually carries the contract and the rest elaborates. Joins the LLM cache key. None disables the cap.") class-attribute instance-attribute
ontology_text_max_chars_naming = Field(default=None, ge=1, description='Character cap on each rdfs:label, skos:prefLabel and skos:altLabel in the ontology chapter; unset disables it. Long names are clipped at a word boundary with a visible marker, so the model can tell a clipped name from a complete one. Part of the LLM cache key.') class-attribute instance-attribute
ontology_text_max_chars_prose = Field(default=None, ge=1, description='Character cap on rdfs:comment and the remaining SKOS notes -- description aimed at someone browsing the ontology rather than at an extractor. Joins the LLM cache key. None disables the cap.') class-attribute instance-attribute
ontology_text_total_budget = Field(default=None, ge=1, description='Ceiling on the total length of all text literals in one ontology chapter, for catalogs with very many short terms. Over budget, literals are shortened, never removed: prose first, then usage contracts, then names, each to the largest cap that meets the budget, down to a floor; a chapter that still does not fit is used as is. Reported in budget.counters as chapter/text_chars_before, chapter/text_chars_after, chapter/literals_clipped and chapter/text_over_budget. Part of the LLM cache key.') class-attribute instance-attribute
parallel_workers = Field(default=16, ge=1, description="Maximum content units processed at once within one document. A unit makes one LLM call at a time, so this is also the concurrency one document puts on the provider. LLM_MAX_INFLIGHT caps calls across all documents and is the one to lower when a provider rate-limits. If a stage's loop_lag_total in budget.node_durations is a large share of its wall-clock time, more workers will slow it down.") class-attribute instance-attribute
port = Field(default=8999, ge=1, le=65535, description='Port the server listens on.') class-attribute instance-attribute
render_mode = Field(default=RenderMode.ONTOLOGY_AND_FACTS, description='Rendering mode: ontology, facts, or ontology_and_facts.') class-attribute instance-attribute

Methods:

validate_ontology_chapter_format()

Resolve 'auto' against the render mode, and reject an illegal ask.

The facts renderer reads its chapter and writes an unrelated graph, so the chapter is free to be any representation that names the terms. The ontology renderer and its critic write a patch against the statements in the chapter, which a line-per-term listing cannot express -- there is nothing to insert into or delete from.

So the cheapest legal chapter differs by mode, and auto picks it rather than asking an operator to know which. It is resolved here, not at the point of use: every consumer reads this field, and a value that still meant "decide later" would reach the prompt profile, the LLM cache key and the run manifest as a name for no chapter in particular.

An explicit term_sheet on a mode that cannot read one still fails. Falling back would be silent, and the run would spend an ontology pass producing patches nobody could apply while the manifest recorded a setting that never took effect -- which is exactly what auto exists to make unnecessary.

Source code in ontocast/config/settings.py
@model_validator(mode="after")
def validate_ontology_chapter_format(self) -> "ServerConfig":
    """Resolve 'auto' against the render mode, and reject an illegal ask.

    The facts renderer reads its chapter and writes an unrelated graph, so
    the chapter is free to be any representation that names the terms. The
    ontology renderer and its critic write a *patch against the statements
    in the chapter*, which a line-per-term listing cannot express -- there
    is nothing to insert into or delete from.

    So the cheapest legal chapter differs by mode, and ``auto`` picks it
    rather than asking an operator to know which. It is resolved **here**,
    not at the point of use: every consumer reads this field, and a value
    that still meant "decide later" would reach the prompt profile, the
    LLM cache key and the run manifest as a name for no chapter in
    particular.

    An *explicit* ``term_sheet`` on a mode that cannot read one still
    fails. Falling back would be silent, and the run would spend an
    ontology pass producing patches nobody could apply while the manifest
    recorded a setting that never took effect -- which is exactly what
    ``auto`` exists to make unnecessary.
    """
    if self.ontology_chapter_format == OntologyChapterFormat.AUTO:
        self.ontology_chapter_format = (
            OntologyChapterFormat.TERM_SHEET
            if self.render_mode == RenderMode.FACTS
            else OntologyChapterFormat.INHERIT
        )
        return self
    if (
        self.ontology_chapter_format == OntologyChapterFormat.TERM_SHEET
        and self.render_mode != RenderMode.FACTS
    ):
        raise ValueError(
            "ONTOLOGY_CHAPTER_FORMAT=term_sheet requires RENDER_MODE=facts: "
            f"got render_mode={self.render_mode.value}. The ontology loop "
            "emits a patch against the statements in its chapter, so that "
            "chapter has to be a graph. Use 'turtle' for a cheaper chapter "
            "that stays one, or 'auto' to take whichever the mode allows."
        )
    return self

SiblingGuardScope

Bases: StrEnum

Scope of the co-object sibling merge guard.

Source code in ontocast/config/settings.py
class SiblingGuardScope(StrEnum):
    """Scope of the co-object sibling merge guard."""

    SUBJECT = "subject"
    PREDICATE = "predicate"

Attributes

PREDICATE = 'predicate' class-attribute instance-attribute
SUBJECT = 'subject' class-attribute instance-attribute

SymbolCaseMismatchPolicy

Bases: StrEnum

Treatment of hits whose only symbol evidence is case-mismatched.

Source code in ontocast/config/settings.py
class SymbolCaseMismatchPolicy(StrEnum):
    """Treatment of hits whose only symbol evidence is case-mismatched."""

    OFF = "off"
    DEMOTE = "demote"
    DROP = "drop"

Attributes

DEMOTE = 'demote' class-attribute instance-attribute
DROP = 'drop' class-attribute instance-attribute
OFF = 'off' class-attribute instance-attribute

ToolConfig

Bases: BaseSettings

Configuration for tools (LLM, triple stores, paths, chunking).

Source code in ontocast/config/settings.py
class ToolConfig(BaseSettings):
    """Configuration for tools (LLM, triple stores, paths, chunking)."""

    llm_config: LLMConfig = Field(default_factory=LLMConfig)
    chunk_config: ChunkConfig = Field(default_factory=ChunkConfig)
    converter_config: ConverterConfig = Field(default_factory=ConverterConfig)
    path_config: PathConfig = Field(default_factory=PathConfig)
    fuseki: FusekiConfig = Field(default_factory=FusekiConfig)
    domain: DomainConfig = Field(default_factory=DomainConfig)
    web_search: WebSearchConfig = Field(default_factory=WebSearchConfig)
    aggregation: AggregationConfig = Field(default_factory=AggregationConfig)
    embedding: EmbeddingConfig = Field(default_factory=EmbeddingConfig)
    patch_retrieval: PatchRetrievalConfig = Field(
        default_factory=PatchRetrievalConfig,
        description="Ontology patch retrieval: post-vector scoring, MMR, and limits.",
    )
    facts_validation: FactsValidationConfig = Field(
        default_factory=FactsValidationConfig,
        description="Deterministic post-checks on LLM-rendered facts graphs.",
    )
    ontology_validation: OntologyValidationConfig = Field(
        default_factory=OntologyValidationConfig,
        description="Deterministic post-checks on LLM-rendered ontology deltas.",
    )
    vector_store: VectorStoreConfig = Field(default_factory=VectorStoreConfig)
    qdrant: QdrantConfig = Field(default_factory=QdrantConfig)
    lancedb: LanceDBConfig = Field(default_factory=LanceDBConfig)

    @model_validator(mode="after")
    def _reject_dual_vector_backends(self) -> ToolConfig:
        if self.qdrant.uri and self.lancedb.enabled:
            raise ValueError(
                "Configure only one vector store backend: set QDRANT_URI or "
                "LANCEDB_ENABLED=true, not both."
            )
        return self

    @model_validator(mode="after")
    def _reject_mmr_with_atom_floors(self) -> ToolConfig:
        """MMR and the atom floors are two selection policies for one budget.

        MMR replaces the selection outright, so a non-zero floor is not
        enforced under it -- and the failure is silent, because a guarantee
        that stops holding produces no error and no metric. Reserving the floor
        slots first is not a fix either: the reserve is taken in score order and
        would leave MMR nothing to choose whenever one ontology supplies the
        candidates, which is the common case.

        So the combination is rejected rather than resolved. Both remedies are
        one setting away, and which one is wanted is the operator's decision,
        not a default this can guess.
        """
        pc = self.patch_retrieval
        if pc.mmr_lambda >= 1.0:
            return self
        floors = {
            "ONTOLOGY_PATCH_PER_ONTOLOGY_ATOM_FLOOR": pc.per_ontology_atom_floor,
            "ONTOLOGY_PATCH_PER_ROLE_ATOM_FLOOR": pc.per_role_atom_floor,
        }
        engaged = {name: value for name, value in floors.items() if value > 0}
        if not engaged:
            return self
        named = ", ".join(f"{name}={value}" for name, value in engaged.items())
        raise ValueError(
            f"ONTOLOGY_PATCH_MMR_LAMBDA={pc.mmr_lambda} enables MMR reranking, "
            f"which selects the whole atom budget itself and therefore cannot "
            f"honour the atom floors ({named}). Set the floors to 0 to choose "
            f"MMR, or MMR_LAMBDA=1.0 to keep the floors."
        )

    @model_validator(mode="after")
    def _warn_on_unreachable_retrieval_settings(self) -> ToolConfig:
        """Warn where a retrieval setting cannot bind, before a run pays for it.

        Not an error -- the configuration runs and produces a result. It is
        reported because the result is not the one the setting asks for, and
        nothing downstream says so: a sweep spends points on an axis that moves
        nothing, and an operator reads a number back as evidence about a knob
        that never applied.
        """
        pc, sc = self.patch_retrieval, self.vector_store

        # The window count is capped, so the scaled budget has a ceiling; above
        # it the hard cap is unreachable and every value is the same run.
        max_scaled = max(
            pc.max_atoms_base, pc.seeds_per_window * sc.proposition_max_windows
        )
        if pc.max_atoms > max_scaled:
            logger.warning(
                "ONTOLOGY_PATCH_MAX_ATOMS=%d cannot bind: the effective cap is "
                "min(max_atoms, max(MAX_ATOMS_BASE=%d, SEEDS_PER_WINDOW=%d * "
                "windows)), and windows is capped by "
                "VECTOR_STORE_PROPOSITION_MAX_WINDOWS=%d, so the cap never "
                "exceeds %d. Raise MAX_ATOMS_BASE to raise the budget.",
                pc.max_atoms,
                pc.max_atoms_base,
                pc.seeds_per_window,
                sc.proposition_max_windows,
                max_scaled,
            )

        return self

Attributes

aggregation = Field(default_factory=AggregationConfig) class-attribute instance-attribute
chunk_config = Field(default_factory=ChunkConfig) class-attribute instance-attribute
converter_config = Field(default_factory=ConverterConfig) class-attribute instance-attribute
domain = Field(default_factory=DomainConfig) class-attribute instance-attribute
embedding = Field(default_factory=EmbeddingConfig) class-attribute instance-attribute
facts_validation = Field(default_factory=FactsValidationConfig, description='Deterministic post-checks on LLM-rendered facts graphs.') class-attribute instance-attribute
fuseki = Field(default_factory=FusekiConfig) class-attribute instance-attribute
lancedb = Field(default_factory=LanceDBConfig) class-attribute instance-attribute
llm_config = Field(default_factory=LLMConfig) class-attribute instance-attribute
ontology_validation = Field(default_factory=OntologyValidationConfig, description='Deterministic post-checks on LLM-rendered ontology deltas.') class-attribute instance-attribute
patch_retrieval = Field(default_factory=PatchRetrievalConfig, description='Ontology patch retrieval: post-vector scoring, MMR, and limits.') class-attribute instance-attribute
path_config = Field(default_factory=PathConfig) class-attribute instance-attribute
qdrant = Field(default_factory=QdrantConfig) class-attribute instance-attribute
vector_store = Field(default_factory=VectorStoreConfig) class-attribute instance-attribute

VectorStoreConfig

Bases: BaseSettings

Backend-agnostic vector store retrieval and indexing settings.

Source code in ontocast/config/settings.py
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
2016
2017
2018
2019
2020
2021
2022
2023
2024
2025
2026
2027
2028
2029
2030
2031
2032
2033
2034
2035
2036
2037
2038
2039
2040
2041
2042
2043
2044
2045
2046
2047
2048
2049
2050
2051
2052
2053
2054
2055
2056
2057
2058
2059
2060
2061
2062
2063
2064
2065
2066
2067
2068
2069
2070
2071
2072
2073
2074
2075
2076
2077
2078
2079
2080
2081
2082
2083
2084
2085
2086
2087
2088
2089
2090
2091
2092
2093
2094
2095
2096
2097
2098
2099
2100
2101
2102
2103
2104
2105
2106
2107
2108
2109
2110
2111
2112
2113
2114
2115
2116
2117
2118
2119
2120
2121
2122
2123
2124
2125
2126
2127
2128
2129
2130
2131
2132
2133
2134
2135
2136
2137
2138
2139
2140
2141
2142
2143
2144
2145
2146
2147
2148
2149
2150
2151
2152
2153
2154
2155
2156
2157
2158
2159
2160
2161
2162
2163
2164
2165
2166
2167
2168
2169
2170
2171
2172
2173
2174
2175
2176
2177
2178
2179
2180
2181
2182
2183
2184
2185
2186
2187
2188
2189
2190
2191
2192
2193
2194
2195
2196
2197
2198
2199
2200
2201
2202
2203
2204
2205
2206
2207
2208
2209
2210
2211
2212
2213
2214
2215
2216
2217
2218
2219
2220
2221
2222
2223
2224
2225
2226
2227
2228
2229
2230
2231
2232
2233
2234
2235
2236
2237
2238
2239
2240
2241
2242
2243
2244
2245
2246
2247
2248
2249
2250
2251
2252
2253
2254
2255
2256
2257
2258
2259
2260
2261
2262
2263
2264
2265
2266
2267
2268
2269
2270
2271
2272
2273
2274
2275
2276
2277
2278
2279
2280
2281
2282
2283
2284
2285
2286
2287
2288
2289
2290
2291
2292
2293
2294
2295
2296
2297
2298
2299
2300
2301
2302
2303
2304
2305
2306
2307
2308
2309
2310
2311
2312
2313
2314
2315
2316
2317
2318
2319
2320
2321
2322
2323
2324
2325
2326
2327
2328
2329
2330
2331
2332
2333
2334
2335
2336
2337
2338
2339
2340
2341
2342
2343
2344
2345
2346
2347
2348
2349
2350
2351
2352
2353
2354
2355
2356
2357
2358
2359
2360
2361
2362
2363
2364
2365
2366
2367
2368
2369
2370
2371
2372
2373
2374
2375
2376
2377
2378
2379
2380
2381
2382
2383
2384
2385
2386
2387
2388
2389
2390
2391
2392
2393
2394
2395
2396
2397
2398
2399
2400
2401
2402
2403
2404
2405
2406
2407
2408
2409
2410
2411
2412
2413
2414
2415
2416
2417
2418
2419
2420
2421
2422
2423
2424
2425
2426
2427
2428
2429
2430
2431
2432
2433
2434
2435
2436
2437
2438
2439
2440
2441
2442
2443
2444
2445
2446
2447
2448
2449
2450
2451
2452
2453
2454
2455
2456
2457
2458
2459
2460
2461
2462
2463
2464
2465
2466
2467
2468
2469
class VectorStoreConfig(BaseSettings):
    """Backend-agnostic vector store retrieval and indexing settings."""

    backend: VectorStoreBackend = Field(
        default=VectorStoreBackend.AUTO,
        description=(
            "Which vector store implementation to use: 'auto' (infer from "
            "QDRANT_URI / LANCEDB_ENABLED, disabling vector retrieval when "
            "neither is configured), 'qdrant', 'lancedb', or 'none' "
            "to disable vector retrieval entirely."
        ),
    )
    top_k: int = Field(
        default=40,
        ge=1,
        description=(
            "Fused hits each query window offers to ontology-patch retrieval."
            " This is the candidate pool, not the result size: "
            "ONTOLOGY_PATCH_MAX_ATOMS caps what is kept, so a deeper pool "
            "fills the same budget from a wider field. Costs search time, not"
            " prompt size."
        ),
    )
    bm25_top_k: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Depth of the sparse (BM25) lane; unset uses TOP_K. Lanes are "
            "fused by reciprocal rank, so every hit in a lane votes at full "
            "lane weight and depth acts as weight. Lower it to keep the "
            "sparse lane for what it finds best (symbols, notations, "
            "formulae) without letting its weak tail vote."
        ),
    )
    induced_subgraph_depth: int = Field(
        default=2,
        ge=0,
        description="Neighborhood expansion depth for induced subgraph retrieval.",
    )
    induced_subgraph_hub_seed_count: int = Field(
        default=16,
        ge=0,
        description=(
            "Induced subgraph: number of top-relevance seeds that receive full BFS hub "
            "expansion. 0 disables hub-only BFS (all seeds expand)."
        ),
    )
    induced_subgraph_ancestor_closure_depth: int = Field(
        default=3,
        ge=0,
        description=(
            "Induced subgraph schema shell: max rdfs:subClassOf hops upward per class seed."
        ),
    )
    induced_subgraph_max_total_triples: int = Field(
        default=1200,
        ge=1,
        description=(
            "Hard cap on triples returned for induced subgraph retrieval. This, "
            "not the atom cap, is what binds in practice: set it too low and "
            "every seed-side knob (top_k, max_atoms, MMR, the atom floors) is "
            "flat, because the snapshot is already pinned at the cap. Raise "
            "this before tuning anything below it; it saturates."
        ),
    )
    induced_subgraph_estimated_triples_per_query: int = Field(
        default=24,
        ge=1,
        description=(
            "Estimated triples per query window, used to divide the induced "
            "subgraph's triple budget among seed entities."
        ),
    )
    induced_subgraph_type_promotion_score_factor: float = Field(
        default=1.0,
        ge=0.0,
        le=1.0,
        description=(
            "Fraction of a retrieved seed's score inherited by its promoted rdf:type "
            "IRIs during induced-subgraph budgeting. The seed always keeps its own "
            "score; this only scales the copy banked on the type. (Transferring the "
            "score to the type and zeroing the individual collapsed all typed "
            "individuals into a relevance-0 tie broken by raw IRI order, which "
            "starved high-ranked seeds under tight triple budgets.)"
        ),
    )
    induced_subgraph_seed_order: InducedSubgraphSeedOrder = Field(
        default=InducedSubgraphSeedOrder.SCORE,
        description=(
            "Seed expansion order under the induced-subgraph triple budget: 'score' "
            "expands in global relevance order; 'ontology_round_robin' interleaves "
            "seeds across source ontologies so no ontology is starved by another's "
            "high scorers. 'score' is the default."
        ),
    )
    induced_subgraph_symbol_predicates: list[str] = Field(
        default_factory=lambda: [
            "http://www.w3.org/2004/02/skos/core#notation",
            "http://qudt.org/schema/qudt/symbol",
            "http://qudt.org/schema/qudt/ucumCode",
        ],
        description=(
            "Predicate IRIs admitted as seed descriptions in the induced subgraph, "
            "between names and glosses (default mirrors lexical_trigger_predicates). "
            "Without them a unit individual reaches the prompt label-only and the "
            "LLM cannot map surface tokens like 'meV' to its IRI. Empty disables."
        ),
    )
    induced_subgraph_candidate_pushdown: bool = Field(
        default=False,
        description=(
            "Build the induced-subgraph working graph from a SPARQL CONSTRUCT of the "
            "seeds' bounded neighborhood instead of the merged ontology graphs. Bounds "
            "memory and wire volume on large catalogs; on small ones the neighborhood is "
            "essentially the whole ontology and there is nothing to gain. Requires a "
            "backend with supports_sparql_construct(); falls back silently otherwise."
        ),
    )
    proposition_window_sentences: int = Field(
        default=2,
        ge=1,
        description=(
            "Sentence window size used for proposition-level retrieval slicing. "
            "Bounded in practice by the embedding model's sequence limit, not by this "
            "setting: a window longer than the encoder accepts is truncated by the "
            "encoder, silently, so widening past that point discards query text rather "
            "than matching more of it. Watch chapter/query truncation in the retrieval "
            "metrics when raising it."
        ),
    )
    proposition_window_stride: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Sentences advanced between query windows; unset strides by the "
            "full window, so windows do not overlap. A smaller stride "
            "overlaps them, so a statement split across a window boundary "
            "still lands in one window, at the cost of more queries."
        ),
    )
    proposition_window_max_chars: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Characters per retrieval query window, used instead of "
            "PROPOSITION_WINDOW_SENTENCES when set. Length, not sentence "
            "count, decides whether a query works: sentences vary widely in "
            "length, and the encoder silently truncates long input. Windows "
            "take sentences until the budget is met, which also joins short "
            "fragments. Keep it below the encoder's sequence limit "
            "(characters per token are in the retrieval metrics). Query side "
            "only: no reindex needed."
        ),
    )
    proposition_window_max_tokens: int | None = Field(
        default=None,
        ge=1,
        description=(
            "Encoder tokens per query window; when set it takes precedence "
            "over PROPOSITION_WINDOW_SENTENCES and "
            "PROPOSITION_WINDOW_MAX_CHARS. Set below the encoder's sequence "
            "limit, it makes truncation impossible, and it splits a sentence "
            "longer than the budget at a word boundary. Needs an embedding "
            "provider that exposes its tokenizer; otherwise it falls back to "
            "a character estimate and logs that it did. Query side only: no "
            "reindex needed."
        ),
    )
    proposition_window_overlap: float = Field(
        default=0.0,
        ge=0.0,
        lt=1.0,
        description=(
            "Fraction of a window repeated at the start of the next one, under a "
            "character or token budget. 0.0 (default) leaves windows disjoint. "
            "PROPOSITION_WINDOW_STRIDE is the sentence-granular spelling of the same "
            "idea and does not apply under a budget, where a sentence says nothing "
            "about how much text is shared. Overlap multiplies queries, so it costs "
            "embedding time and, once PROPOSITION_MAX_WINDOWS binds, coverage "
            "elsewhere in the unit."
        ),
    )
    proposition_abbreviation_aware: bool = Field(
        default=False,
        description=(
            "Rejoin window fragments that the sentence splitter cut at an "
            "abbreviation, an initial or a citation, where a window can "
            "otherwise hold a few characters with nothing to retrieve. Uses "
            "general English and bibliographic patterns only, never a domain "
            "vocabulary. Off by default because it changes every window's "
            "boundaries."
        ),
    )
    proposition_measurement_aware: bool = Field(
        default=False,
        description=(
            "Forbid a window break between a number and the unit it is written with, "
            "or inside a range, using the number/unit shapes in the shared "
            "measurement lexicon (shapes, not a unit vocabulary). Only binds where a "
            "cut inside a sentence is possible, which means PROPOSITION_WINDOW_MAX_"
            "TOKENS with a sentence over budget: a window ending on 'a red shift of "
            "~10' retrieves nothing that the number and its unit together would."
        ),
    )
    proposition_max_windows: int = Field(
        default=16,
        ge=1,
        description=(
            "Upper bound on proposition windows generated per document excerpt. Over "
            "the bound windows are subsampled evenly rather than truncated, so the "
            "excerpt stays covered end to end -- but the text in the dropped windows "
            "reaches no dense or sparse lane at all."
        ),
    )
    proposition_retrieval_enabled: bool = Field(
        default=True,
        description="Enable proposition-level multi-query retrieval for induced graph mode.",
    )
    consistency_critic_min_fused_score: float = Field(
        default=0.5,
        ge=0.0,
        le=1.0,
        description=(
            "Minimum fused retrieval score for the consistency critic to report "
            "a possible conflict between ontologies. A weighted reciprocal-rank "
            "score summed over the core, neighborhood and BM25 lanes, not a "
            "cosine similarity: each lane adds its normalized weight divided by "
            "the rank. With BM25 enabled and default weights, a rank-1 hit in "
            "one lane alone scores below 0.5 (about 0.42 for the core lane), so "
            "0.5 requires at least two lanes to agree on the term."
        ),
    )
    embedding_batch_size: int = Field(
        default=64,
        ge=1,
        description="Batch size used for embedding requests during indexing.",
    )
    reindex_concurrency: int = Field(
        default=2,
        ge=1,
        description=(
            "Max ontologies to materialize/reindex concurrently during ToolBox "
            "initialize. Dense embeds are serialized via a process-wide lock; "
            "higher values mainly overlap triple-store I/O and BM25 with waits."
        ),
    )
    wipe_on_init: bool = Field(
        default=False,
        description=(
            "When true, ToolBox.initialize drops the current ontology/facts "
            "vector partition before recreating schema and reindexing. Use for "
            "clean-slate recovery (e.g. after embedding-model changes)."
        ),
    )
    prune_orphan_iris_on_init: bool = Field(
        default=True,
        description=(
            "When true, ToolBox.initialize deletes indexed ontology IRIs that "
            "are not in the synchronized catalog (covers IRI renames without a "
            "full wipe)."
        ),
    )
    fusion_core_weight: float = Field(
        default=0.7,
        ge=0.0,
        le=1.0,
        description=(
            "Core vector score weight for dual-vector ranking fusion. Weights are "
            "normalized across the three lanes before use, so only their ratio matters."
        ),
    )
    fusion_neighborhood_weight: float = Field(
        default=0.15,
        ge=0.0,
        le=1.0,
        description=(
            "Weight of the neighborhood lane in rank fusion. The neighborhood"
            " text describes a term's relations rather than the term, so it "
            "mainly corroborates the core lane; keep it below the core "
            "weight."
        ),
    )
    fusion_bm25_weight: float = Field(
        default=0.8,
        ge=0.0,
        le=1.0,
        description=(
            "Weight of the sparse (BM25) lane in rank fusion, normalized with"
            " the core and neighborhood weights when BM25 is enabled. Terms "
            "whose surface form is a symbol or notation (unit symbols, "
            "chemical formulae, gene symbols) are often found only by this "
            "lane, so a low weight lets any dense hit outvote them."
        ),
    )
    fusion_rank_constant: float = Field(
        default=0.0,
        ge=0.0,
        description=(
            "Constant added to each rank in lane fusion: a lane contributes "
            "weight / (constant + rank). At 0 the first rank dominates, so "
            "fusion follows whichever lane put a term first. Raising it makes"
            " agreement across lanes count for more than position within one;"
            " far above TOP_K it makes all ranks nearly equal."
        ),
    )
    minimal_label_limit: int = Field(
        default=5,
        ge=0,
        description=(
            "Maximum declared surface forms (rdfs:label, skos:prefLabel, dcterms:title, "
            "skos:altLabel) folded into each atom's sparse BM25 text. A vocabulary may "
            "declare more aliases than this; symbol aliases sort last and are dropped "
            "first, so raising this widens what the sparse lane can match. Changing it "
            "changes stored sparse vectors and requires a reindex."
        ),
    )
    index_undescribed_iris: bool = Field(
        default=False,
        description=(
            "Index every IRI in an ontology, including those that appear only"
            " as an object or predicate. By default only terms the ontology "
            "describes (as a subject, or with a label) are indexed: a merely "
            "referenced IRI has no text but its local name, and such strings "
            "embed as generic hubs that match every query and crowd out real "
            "terms. Referenced IRIs stay reachable through induced-subgraph "
            "expansion. Changing this requires a reindex."
        ),
    )
    embed_standard_vocab_iris: bool = Field(
        default=False,
        description=(
            "If True, atomize focal IRIs in standard RDF/OWL/SKOS/DC/SHACL/schema.org "
            "namespaces instead of skipping them. These are scaffolding an ontology "
            "reuses rather than terms it defines, so they carry no retrieval signal for "
            "the document being processed. Changing this requires a reindex."
        ),
    )
    extra_excluded_namespace_prefixes: list[str] = Field(
        default_factory=list,
        description=(
            "IRI prefixes never indexed from ontology sources, in addition to"
            " the standard vocabularies. Use it for an upper ontology or "
            "external vocabulary that a catalog includes but that should not "
            "compete in retrieval (BFO, SOSA, OM-2); merely referenced "
            "vocabularies are already skipped while index_undescribed_iris is"
            " false. Changing this requires a reindex."
        ),
    )
    dedup_mode: VectorStoreDedupMode = Field(
        default=VectorStoreDedupMode.IRI,
        description=(
            "Row/point identity policy for ontology vectors: 'iri' stores one logical "
            "record per entity key, while 'atom_id' keeps every atom variant separate."
        ),
    )
    dedup_include_version: bool = Field(
        default=True,
        description=(
            "When dedup_mode='iri', include ontology_version in the identity key so "
            "different ontology versions remain isolated."
        ),
    )
    dedup_include_hash: bool = Field(
        default=True,
        description=(
            "When dedup_mode='iri', include ontology_hash in the identity key so "
            "different ontology snapshots remain isolated."
        ),
    )
    dedup_query_hits_by_iri: bool = Field(
        default=True,
        description=(
            "Drop duplicate retrieval hits sharing the same logical IRI key and keep "
            "the best-scoring one."
        ),
    )
    ontology_table: str | None = Field(
        default=None,
        description=(
            "Ontology atom table/collection name; derived like FUSEKI_DATASET, "
            "and replaced the same way by ontocast serve and ontocast process."
        ),
    )
    facts_table: str | None = Field(
        default=None,
        description=(
            "Facts table/collection reserved for future fact vectors; created on init."
        ),
    )
    label_predicates: list[str] = Field(
        default_factory=lambda: [
            "http://www.w3.org/2000/01/rdf-schema#label",
            "http://www.w3.org/2004/02/skos/core#prefLabel",
            "http://purl.org/dc/terms/title",
            "http://www.w3.org/2004/02/skos/core#altLabel",
            "http://purl.org/dc/terms/alternative",
        ],
        description=(
            "Predicate IRIs whose literal objects are indexed as declared labels, "
            "in descending priority (default: rdfs:label, skos:prefLabel, "
            "dcterms:title, skos:altLabel, dcterms:alternative). Changing this "
            "changes stored vectors and requires a reindex."
        ),
    )
    symbol_predicates: list[str] = Field(
        default_factory=lambda: [
            "http://www.w3.org/2004/02/skos/core#notation",
            "http://qudt.org/schema/qudt/symbol",
            "http://qudt.org/schema/qudt/ucumCode",
        ],
        description=(
            "Predicates whose literal objects are indexed as symbols or "
            "notations (default: skos:notation, qudt:symbol, qudt:ucumCode). "
            "The indexing counterpart of INDUCED_SUBGRAPH_SYMBOL_PREDICATES; "
            "set both together. Changing this requires a reindex."
        ),
    )
    lexical_trigger_enabled: bool = Field(
        default=True,
        description=(
            "Enable the lexical-trigger lane: scan raw chunk text for notation/symbol "
            "tokens and inject matching atoms as additive retrieval seeds."
        ),
    )
    lexical_trigger_predicates: list[str] = Field(
        default_factory=lambda: [
            "http://www.w3.org/2004/02/skos/core#notation",
            "http://qudt.org/schema/qudt/symbol",
            "http://qudt.org/schema/qudt/ucumCode",
        ],
        description=(
            "Predicate IRIs whose literal objects become case-preserved lexical triggers "
            "(default: skos:notation, qudt:symbol, qudt:ucumCode)."
        ),
    )
    lexical_trigger_heuristic_enabled: bool = Field(
        default=True,
        description=(
            "Promote bare code-shaped rdfs:label/skos:altLabel values as triggers when "
            "no predicate-declared notation exists for the entity."
        ),
    )
    lexical_trigger_min_len: int = Field(
        default=2,
        ge=1,
        description="Minimum length for heuristic label/altLabel trigger promotion.",
    )
    lexical_trigger_max_len: int = Field(
        default=24,
        ge=1,
        description="Maximum length for heuristic label/altLabel trigger promotion.",
    )
    lexical_trigger_heuristic_max_per_entity: int = Field(
        default=2,
        ge=0,
        description="Cap on heuristic triggers per entity.",
    )
    lexical_trigger_max_atoms: int = Field(
        default=16,
        ge=0,
        description=(
            "Maximum lexical-trigger atoms injected per retrieval call, additive to the "
            "semantic atom budget."
        ),
    )
    lexical_trigger_score: float = Field(
        default=0.35,
        ge=0.0,
        le=1.0,
        description=(
            "Score given to lexical-trigger hits, on the fused reciprocal-rank "
            "scale: with BM25 enabled and default weights, a rank-1 core hit "
            "alone scores about 0.42. Keep it below that, so trigger hits join "
            "the semantic seeds rather than outrank all of them."
        ),
    )
    lexical_trigger_fusion: LexicalTriggerFusion = Field(
        default=LexicalTriggerFusion.MAX_MERGE,
        description=(
            "How lexical-trigger hits combine with semantic hits: 'max_merge'"
            " raises an already retrieved atom to the higher of its two "
            "scores and appends unseen atoms; 'append' only appends unseen "
            "atoms, so the trigger adds nothing to an atom retrieval already "
            "found."
        ),
    )
    query_unit_signals_enabled: bool = Field(
        default=True,
        description=(
            "Match the tokens that follow numbers in the unit text ('4-15 "
            "days', '200 kV', '0.5 %') against catalog labels, symbols and "
            "UCUM codes, ignoring case and plurals, and add the matched terms"
            " as seeds at lexical_trigger_score, outside the semantic atom "
            "budget. Recovers the units and qualifiers a measurement needs. "
            "Query side only: no reindex needed. Turn it off when the facts "
            "extracted are not quantities, or the catalog's labels are not in"
            " Latin script."
        ),
    )
    symbol_case_mismatch_policy: SymbolCaseMismatchPolicy = Field(
        default=SymbolCaseMismatchPolicy.DEMOTE,
        description=(
            "Treatment of retrieved atoms whose symbol (skos:notation, "
            "qudt:symbol, qudt:ucumCode) matches a query token only when case"
            " is ignored. The BM25 index is case-folded, so 'meV' in the text"
            " also retrieves the unit with symbol 'MeV', a factor of 10^9 "
            "apart. 'demote' multiplies the score by "
            "symbol_case_mismatch_demote_factor, 'drop' removes the atom, "
            "'off' keeps it. Exact-case and label matches are never affected."
        ),
    )
    symbol_case_mismatch_demote_factor: float = Field(
        default=0.5,
        ge=0.0,
        le=1.0,
        description=(
            "Factor a case-mismatched symbol's score is multiplied by when "
            "VECTOR_STORE_SYMBOL_CASE_MISMATCH_POLICY is demote. 0 ranks it "
            "last, 1 leaves it unchanged."
        ),
    )

    model_config = SettingsConfigDict(
        env_prefix="VECTOR_STORE_",
        case_sensitive=False,
    )

    @model_validator(mode="after")
    def _resolve_table_names(self) -> VectorStoreConfig:
        if self.ontology_table is None:
            self.ontology_table = tenant_project_ontologies_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        if self.facts_table is None:
            self.facts_table = tenant_project_facts_name(
                DEFAULT_TENANT, DEFAULT_PROJECT
            )
        return self

Attributes

backend = Field(default=VectorStoreBackend.AUTO, description="Which vector store implementation to use: 'auto' (infer from QDRANT_URI / LANCEDB_ENABLED, disabling vector retrieval when neither is configured), 'qdrant', 'lancedb', or 'none' to disable vector retrieval entirely.") class-attribute instance-attribute
bm25_top_k = Field(default=None, ge=1, description='Depth of the sparse (BM25) lane; unset uses TOP_K. Lanes are fused by reciprocal rank, so every hit in a lane votes at full lane weight and depth acts as weight. Lower it to keep the sparse lane for what it finds best (symbols, notations, formulae) without letting its weak tail vote.') class-attribute instance-attribute
consistency_critic_min_fused_score = Field(default=0.5, ge=0.0, le=1.0, description='Minimum fused retrieval score for the consistency critic to report a possible conflict between ontologies. A weighted reciprocal-rank score summed over the core, neighborhood and BM25 lanes, not a cosine similarity: each lane adds its normalized weight divided by the rank. With BM25 enabled and default weights, a rank-1 hit in one lane alone scores below 0.5 (about 0.42 for the core lane), so 0.5 requires at least two lanes to agree on the term.') class-attribute instance-attribute
dedup_include_hash = Field(default=True, description="When dedup_mode='iri', include ontology_hash in the identity key so different ontology snapshots remain isolated.") class-attribute instance-attribute
dedup_include_version = Field(default=True, description="When dedup_mode='iri', include ontology_version in the identity key so different ontology versions remain isolated.") class-attribute instance-attribute
dedup_mode = Field(default=VectorStoreDedupMode.IRI, description="Row/point identity policy for ontology vectors: 'iri' stores one logical record per entity key, while 'atom_id' keeps every atom variant separate.") class-attribute instance-attribute
dedup_query_hits_by_iri = Field(default=True, description='Drop duplicate retrieval hits sharing the same logical IRI key and keep the best-scoring one.') class-attribute instance-attribute
embed_standard_vocab_iris = Field(default=False, description='If True, atomize focal IRIs in standard RDF/OWL/SKOS/DC/SHACL/schema.org namespaces instead of skipping them. These are scaffolding an ontology reuses rather than terms it defines, so they carry no retrieval signal for the document being processed. Changing this requires a reindex.') class-attribute instance-attribute
embedding_batch_size = Field(default=64, ge=1, description='Batch size used for embedding requests during indexing.') class-attribute instance-attribute
extra_excluded_namespace_prefixes = Field(default_factory=list, description='IRI prefixes never indexed from ontology sources, in addition to the standard vocabularies. Use it for an upper ontology or external vocabulary that a catalog includes but that should not compete in retrieval (BFO, SOSA, OM-2); merely referenced vocabularies are already skipped while index_undescribed_iris is false. Changing this requires a reindex.') class-attribute instance-attribute
facts_table = Field(default=None, description='Facts table/collection reserved for future fact vectors; created on init.') class-attribute instance-attribute
fusion_bm25_weight = Field(default=0.8, ge=0.0, le=1.0, description='Weight of the sparse (BM25) lane in rank fusion, normalized with the core and neighborhood weights when BM25 is enabled. Terms whose surface form is a symbol or notation (unit symbols, chemical formulae, gene symbols) are often found only by this lane, so a low weight lets any dense hit outvote them.') class-attribute instance-attribute
fusion_core_weight = Field(default=0.7, ge=0.0, le=1.0, description='Core vector score weight for dual-vector ranking fusion. Weights are normalized across the three lanes before use, so only their ratio matters.') class-attribute instance-attribute
fusion_neighborhood_weight = Field(default=0.15, ge=0.0, le=1.0, description="Weight of the neighborhood lane in rank fusion. The neighborhood text describes a term's relations rather than the term, so it mainly corroborates the core lane; keep it below the core weight.") class-attribute instance-attribute
fusion_rank_constant = Field(default=0.0, ge=0.0, description='Constant added to each rank in lane fusion: a lane contributes weight / (constant + rank). At 0 the first rank dominates, so fusion follows whichever lane put a term first. Raising it makes agreement across lanes count for more than position within one; far above TOP_K it makes all ranks nearly equal.') class-attribute instance-attribute
index_undescribed_iris = Field(default=False, description='Index every IRI in an ontology, including those that appear only as an object or predicate. By default only terms the ontology describes (as a subject, or with a label) are indexed: a merely referenced IRI has no text but its local name, and such strings embed as generic hubs that match every query and crowd out real terms. Referenced IRIs stay reachable through induced-subgraph expansion. Changing this requires a reindex.') class-attribute instance-attribute
induced_subgraph_ancestor_closure_depth = Field(default=3, ge=0, description='Induced subgraph schema shell: max rdfs:subClassOf hops upward per class seed.') class-attribute instance-attribute
induced_subgraph_candidate_pushdown = Field(default=False, description="Build the induced-subgraph working graph from a SPARQL CONSTRUCT of the seeds' bounded neighborhood instead of the merged ontology graphs. Bounds memory and wire volume on large catalogs; on small ones the neighborhood is essentially the whole ontology and there is nothing to gain. Requires a backend with supports_sparql_construct(); falls back silently otherwise.") class-attribute instance-attribute
induced_subgraph_depth = Field(default=2, ge=0, description='Neighborhood expansion depth for induced subgraph retrieval.') class-attribute instance-attribute
induced_subgraph_estimated_triples_per_query = Field(default=24, ge=1, description="Estimated triples per query window, used to divide the induced subgraph's triple budget among seed entities.") class-attribute instance-attribute
induced_subgraph_hub_seed_count = Field(default=16, ge=0, description='Induced subgraph: number of top-relevance seeds that receive full BFS hub expansion. 0 disables hub-only BFS (all seeds expand).') class-attribute instance-attribute
induced_subgraph_max_total_triples = Field(default=1200, ge=1, description='Hard cap on triples returned for induced subgraph retrieval. This, not the atom cap, is what binds in practice: set it too low and every seed-side knob (top_k, max_atoms, MMR, the atom floors) is flat, because the snapshot is already pinned at the cap. Raise this before tuning anything below it; it saturates.') class-attribute instance-attribute
induced_subgraph_seed_order = Field(default=InducedSubgraphSeedOrder.SCORE, description="Seed expansion order under the induced-subgraph triple budget: 'score' expands in global relevance order; 'ontology_round_robin' interleaves seeds across source ontologies so no ontology is starved by another's high scorers. 'score' is the default.") class-attribute instance-attribute
induced_subgraph_symbol_predicates = Field(default_factory=lambda: ['http://www.w3.org/2004/02/skos/core#notation', 'http://qudt.org/schema/qudt/symbol', 'http://qudt.org/schema/qudt/ucumCode'], description="Predicate IRIs admitted as seed descriptions in the induced subgraph, between names and glosses (default mirrors lexical_trigger_predicates). Without them a unit individual reaches the prompt label-only and the LLM cannot map surface tokens like 'meV' to its IRI. Empty disables.") class-attribute instance-attribute
induced_subgraph_type_promotion_score_factor = Field(default=1.0, ge=0.0, le=1.0, description="Fraction of a retrieved seed's score inherited by its promoted rdf:type IRIs during induced-subgraph budgeting. The seed always keeps its own score; this only scales the copy banked on the type. (Transferring the score to the type and zeroing the individual collapsed all typed individuals into a relevance-0 tie broken by raw IRI order, which starved high-ranked seeds under tight triple budgets.)") class-attribute instance-attribute
label_predicates = Field(default_factory=lambda: ['http://www.w3.org/2000/01/rdf-schema#label', 'http://www.w3.org/2004/02/skos/core#prefLabel', 'http://purl.org/dc/terms/title', 'http://www.w3.org/2004/02/skos/core#altLabel', 'http://purl.org/dc/terms/alternative'], description='Predicate IRIs whose literal objects are indexed as declared labels, in descending priority (default: rdfs:label, skos:prefLabel, dcterms:title, skos:altLabel, dcterms:alternative). Changing this changes stored vectors and requires a reindex.') class-attribute instance-attribute
lexical_trigger_enabled = Field(default=True, description='Enable the lexical-trigger lane: scan raw chunk text for notation/symbol tokens and inject matching atoms as additive retrieval seeds.') class-attribute instance-attribute
lexical_trigger_fusion = Field(default=LexicalTriggerFusion.MAX_MERGE, description="How lexical-trigger hits combine with semantic hits: 'max_merge' raises an already retrieved atom to the higher of its two scores and appends unseen atoms; 'append' only appends unseen atoms, so the trigger adds nothing to an atom retrieval already found.") class-attribute instance-attribute
lexical_trigger_heuristic_enabled = Field(default=True, description='Promote bare code-shaped rdfs:label/skos:altLabel values as triggers when no predicate-declared notation exists for the entity.') class-attribute instance-attribute
lexical_trigger_heuristic_max_per_entity = Field(default=2, ge=0, description='Cap on heuristic triggers per entity.') class-attribute instance-attribute
lexical_trigger_max_atoms = Field(default=16, ge=0, description='Maximum lexical-trigger atoms injected per retrieval call, additive to the semantic atom budget.') class-attribute instance-attribute
lexical_trigger_max_len = Field(default=24, ge=1, description='Maximum length for heuristic label/altLabel trigger promotion.') class-attribute instance-attribute
lexical_trigger_min_len = Field(default=2, ge=1, description='Minimum length for heuristic label/altLabel trigger promotion.') class-attribute instance-attribute
lexical_trigger_predicates = Field(default_factory=lambda: ['http://www.w3.org/2004/02/skos/core#notation', 'http://qudt.org/schema/qudt/symbol', 'http://qudt.org/schema/qudt/ucumCode'], description='Predicate IRIs whose literal objects become case-preserved lexical triggers (default: skos:notation, qudt:symbol, qudt:ucumCode).') class-attribute instance-attribute
lexical_trigger_score = Field(default=0.35, ge=0.0, le=1.0, description='Score given to lexical-trigger hits, on the fused reciprocal-rank scale: with BM25 enabled and default weights, a rank-1 core hit alone scores about 0.42. Keep it below that, so trigger hits join the semantic seeds rather than outrank all of them.') class-attribute instance-attribute
minimal_label_limit = Field(default=5, ge=0, description="Maximum declared surface forms (rdfs:label, skos:prefLabel, dcterms:title, skos:altLabel) folded into each atom's sparse BM25 text. A vocabulary may declare more aliases than this; symbol aliases sort last and are dropped first, so raising this widens what the sparse lane can match. Changing it changes stored sparse vectors and requires a reindex.") class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='VECTOR_STORE_', case_sensitive=False) class-attribute instance-attribute
ontology_table = Field(default=None, description='Ontology atom table/collection name; derived like FUSEKI_DATASET, and replaced the same way by ontocast serve and ontocast process.') class-attribute instance-attribute
proposition_abbreviation_aware = Field(default=False, description="Rejoin window fragments that the sentence splitter cut at an abbreviation, an initial or a citation, where a window can otherwise hold a few characters with nothing to retrieve. Uses general English and bibliographic patterns only, never a domain vocabulary. Off by default because it changes every window's boundaries.") class-attribute instance-attribute
proposition_max_windows = Field(default=16, ge=1, description='Upper bound on proposition windows generated per document excerpt. Over the bound windows are subsampled evenly rather than truncated, so the excerpt stays covered end to end -- but the text in the dropped windows reaches no dense or sparse lane at all.') class-attribute instance-attribute
proposition_measurement_aware = Field(default=False, description="Forbid a window break between a number and the unit it is written with, or inside a range, using the number/unit shapes in the shared measurement lexicon (shapes, not a unit vocabulary). Only binds where a cut inside a sentence is possible, which means PROPOSITION_WINDOW_MAX_TOKENS with a sentence over budget: a window ending on 'a red shift of ~10' retrieves nothing that the number and its unit together would.") class-attribute instance-attribute
proposition_retrieval_enabled = Field(default=True, description='Enable proposition-level multi-query retrieval for induced graph mode.') class-attribute instance-attribute
proposition_window_max_chars = Field(default=None, ge=1, description="Characters per retrieval query window, used instead of PROPOSITION_WINDOW_SENTENCES when set. Length, not sentence count, decides whether a query works: sentences vary widely in length, and the encoder silently truncates long input. Windows take sentences until the budget is met, which also joins short fragments. Keep it below the encoder's sequence limit (characters per token are in the retrieval metrics). Query side only: no reindex needed.") class-attribute instance-attribute
proposition_window_max_tokens = Field(default=None, ge=1, description="Encoder tokens per query window; when set it takes precedence over PROPOSITION_WINDOW_SENTENCES and PROPOSITION_WINDOW_MAX_CHARS. Set below the encoder's sequence limit, it makes truncation impossible, and it splits a sentence longer than the budget at a word boundary. Needs an embedding provider that exposes its tokenizer; otherwise it falls back to a character estimate and logs that it did. Query side only: no reindex needed.") class-attribute instance-attribute
proposition_window_overlap = Field(default=0.0, ge=0.0, lt=1.0, description='Fraction of a window repeated at the start of the next one, under a character or token budget. 0.0 (default) leaves windows disjoint. PROPOSITION_WINDOW_STRIDE is the sentence-granular spelling of the same idea and does not apply under a budget, where a sentence says nothing about how much text is shared. Overlap multiplies queries, so it costs embedding time and, once PROPOSITION_MAX_WINDOWS binds, coverage elsewhere in the unit.') class-attribute instance-attribute
proposition_window_sentences = Field(default=2, ge=1, description="Sentence window size used for proposition-level retrieval slicing. Bounded in practice by the embedding model's sequence limit, not by this setting: a window longer than the encoder accepts is truncated by the encoder, silently, so widening past that point discards query text rather than matching more of it. Watch chapter/query truncation in the retrieval metrics when raising it.") class-attribute instance-attribute
proposition_window_stride = Field(default=None, ge=1, description='Sentences advanced between query windows; unset strides by the full window, so windows do not overlap. A smaller stride overlaps them, so a statement split across a window boundary still lands in one window, at the cost of more queries.') class-attribute instance-attribute
prune_orphan_iris_on_init = Field(default=True, description='When true, ToolBox.initialize deletes indexed ontology IRIs that are not in the synchronized catalog (covers IRI renames without a full wipe).') class-attribute instance-attribute
query_unit_signals_enabled = Field(default=True, description="Match the tokens that follow numbers in the unit text ('4-15 days', '200 kV', '0.5 %') against catalog labels, symbols and UCUM codes, ignoring case and plurals, and add the matched terms as seeds at lexical_trigger_score, outside the semantic atom budget. Recovers the units and qualifiers a measurement needs. Query side only: no reindex needed. Turn it off when the facts extracted are not quantities, or the catalog's labels are not in Latin script.") class-attribute instance-attribute
reindex_concurrency = Field(default=2, ge=1, description='Max ontologies to materialize/reindex concurrently during ToolBox initialize. Dense embeds are serialized via a process-wide lock; higher values mainly overlap triple-store I/O and BM25 with waits.') class-attribute instance-attribute
symbol_case_mismatch_demote_factor = Field(default=0.5, ge=0.0, le=1.0, description="Factor a case-mismatched symbol's score is multiplied by when VECTOR_STORE_SYMBOL_CASE_MISMATCH_POLICY is demote. 0 ranks it last, 1 leaves it unchanged.") class-attribute instance-attribute
symbol_case_mismatch_policy = Field(default=SymbolCaseMismatchPolicy.DEMOTE, description="Treatment of retrieved atoms whose symbol (skos:notation, qudt:symbol, qudt:ucumCode) matches a query token only when case is ignored. The BM25 index is case-folded, so 'meV' in the text also retrieves the unit with symbol 'MeV', a factor of 10^9 apart. 'demote' multiplies the score by symbol_case_mismatch_demote_factor, 'drop' removes the atom, 'off' keeps it. Exact-case and label matches are never affected.") class-attribute instance-attribute
symbol_predicates = Field(default_factory=lambda: ['http://www.w3.org/2004/02/skos/core#notation', 'http://qudt.org/schema/qudt/symbol', 'http://qudt.org/schema/qudt/ucumCode'], description='Predicates whose literal objects are indexed as symbols or notations (default: skos:notation, qudt:symbol, qudt:ucumCode). The indexing counterpart of INDUCED_SUBGRAPH_SYMBOL_PREDICATES; set both together. Changing this requires a reindex.') class-attribute instance-attribute
top_k = Field(default=40, ge=1, description='Fused hits each query window offers to ontology-patch retrieval. This is the candidate pool, not the result size: ONTOLOGY_PATCH_MAX_ATOMS caps what is kept, so a deeper pool fills the same budget from a wider field. Costs search time, not prompt size.') class-attribute instance-attribute
wipe_on_init = Field(default=False, description='When true, ToolBox.initialize drops the current ontology/facts vector partition before recreating schema and reindexing. Use for clean-slate recovery (e.g. after embedding-model changes).') class-attribute instance-attribute

VectorStoreDedupMode

Bases: StrEnum

How vector-store row/point identity is derived during upsert.

Source code in ontocast/config/settings.py
class VectorStoreDedupMode(StrEnum):
    """How vector-store row/point identity is derived during upsert."""

    ATOM_ID = "atom_id"
    IRI = "iri"

Attributes

ATOM_ID = 'atom_id' class-attribute instance-attribute
IRI = 'iri' class-attribute instance-attribute

WebSearchConfig

Bases: BaseSettings

Optional web-search settings for ontology grounding.

Source code in ontocast/config/settings.py
class WebSearchConfig(BaseSettings):
    """Optional web-search settings for ontology grounding."""

    enabled: bool = Field(
        default=False,
        description=(
            "Enable optional web grounding. Node execution still starts without "
            "search and only searches when node output requests it."
        ),
    )
    provider: WebSearchProvider = Field(
        default=WebSearchProvider.DUCKDUCKGO,
        description="Search engine queried for evidence when web search is enabled.",
    )
    top_k: int = Field(
        default=3,
        ge=1,
        le=10,
        description="Search results fetched per query.",
    )
    timeout_seconds: float = Field(
        default=8.0,
        ge=1.0,
        le=60.0,
        description=(
            "Seconds to wait for one search request, rounded down to whole seconds."
        ),
    )
    max_snippet_chars: int = Field(
        default=400,
        ge=80,
        le=2000,
        description="Characters kept from each result's snippet; the rest is cut.",
    )
    max_total_chars: int = Field(
        default=1800,
        ge=200,
        le=10000,
        description=(
            "Characters of search evidence added to one prompt, across all results."
        ),
    )
    ontology_render_enabled: bool = Field(
        default=True,
        description=(
            "Allow search-eligible retries for ontology render prompts "
            "(first pass remains no-search)."
        ),
    )
    ontology_critic_enabled: bool = Field(
        default=True,
        description=(
            "Allow search-eligible retries for ontology critic prompts "
            "(first pass remains no-search)."
        ),
    )
    facts_render_enabled: bool = Field(
        default=False,
        description=(
            "Allow search-eligible retries for facts render prompts "
            "(first pass remains no-search)."
        ),
    )
    facts_critic_enabled: bool = Field(
        default=False,
        description=(
            "Allow search-eligible retries for facts critic prompts "
            "(first pass remains no-search)."
        ),
    )
    planner_enabled: bool = Field(
        default=True, description="Enable LLM planner for web-search decisions"
    )
    planner_max_queries: int = Field(
        default=3, ge=1, le=8, description="Maximum focused search queries per node"
    )
    planner_min_query_chars: int = Field(
        default=12,
        ge=3,
        le=100,
        description="Minimum query length accepted by guardrails",
    )
    planner_min_confidence: float = Field(
        default=0.35,
        ge=0.0,
        le=1.0,
        description="Minimum planner confidence to run search",
    )
    reuse_evidence_across_attempt: bool = Field(
        default=True,
        description=("Reuse node-scoped evidence between retries for the same unit."),
    )
    min_snippet_chars: int = Field(
        default=40,
        ge=0,
        le=1000,
        description="Minimum snippet length to keep a search hit",
    )
    # NoDecode hands the raw environment string to `parse_domains`; without it
    # the settings source JSON-decodes first and rejects the comma-separated
    # and empty forms.
    allowed_domains: Annotated[list[str], NoDecode] = Field(
        default_factory=list,
        description=(
            "Optional allowlist of source domains for evidence: comma-separated "
            "or a JSON list"
        ),
    )
    blocked_domains: Annotated[list[str], NoDecode] = Field(
        default_factory=list,
        description=(
            "Optional blocklist of source domains for evidence: comma-separated "
            "or a JSON list"
        ),
    )
    region: str = Field(
        default="wt-wt",
        description=(
            "DuckDuckGo region code, such as us-en or de-de; wt-wt means no region."
        ),
    )
    safesearch: str = Field(
        default="moderate",
        description="DuckDuckGo safe-search mode: on, moderate or off.",
    )

    @field_validator("allowed_domains", "blocked_domains", mode="before")
    @classmethod
    def parse_domains(cls, value: str | list[str]) -> list[str]:
        if isinstance(value, str) and value.lstrip().startswith("["):
            value = json.loads(value)
        if isinstance(value, list):
            return [entry.strip().lower() for entry in value if entry.strip()]
        if isinstance(value, str):
            raw_values = [entry.strip().lower() for entry in value.split(",")]
            return [entry for entry in raw_values if entry]
        return []

    model_config = SettingsConfigDict(
        env_prefix="WEB_SEARCH_",
        case_sensitive=False,
    )

Attributes

allowed_domains = Field(default_factory=list, description='Optional allowlist of source domains for evidence: comma-separated or a JSON list') class-attribute instance-attribute
blocked_domains = Field(default_factory=list, description='Optional blocklist of source domains for evidence: comma-separated or a JSON list') class-attribute instance-attribute
enabled = Field(default=False, description='Enable optional web grounding. Node execution still starts without search and only searches when node output requests it.') class-attribute instance-attribute
facts_critic_enabled = Field(default=False, description='Allow search-eligible retries for facts critic prompts (first pass remains no-search).') class-attribute instance-attribute
facts_render_enabled = Field(default=False, description='Allow search-eligible retries for facts render prompts (first pass remains no-search).') class-attribute instance-attribute
max_snippet_chars = Field(default=400, ge=80, le=2000, description="Characters kept from each result's snippet; the rest is cut.") class-attribute instance-attribute
max_total_chars = Field(default=1800, ge=200, le=10000, description='Characters of search evidence added to one prompt, across all results.') class-attribute instance-attribute
min_snippet_chars = Field(default=40, ge=0, le=1000, description='Minimum snippet length to keep a search hit') class-attribute instance-attribute
model_config = SettingsConfigDict(env_prefix='WEB_SEARCH_', case_sensitive=False) class-attribute instance-attribute
ontology_critic_enabled = Field(default=True, description='Allow search-eligible retries for ontology critic prompts (first pass remains no-search).') class-attribute instance-attribute
ontology_render_enabled = Field(default=True, description='Allow search-eligible retries for ontology render prompts (first pass remains no-search).') class-attribute instance-attribute
planner_enabled = Field(default=True, description='Enable LLM planner for web-search decisions') class-attribute instance-attribute
planner_max_queries = Field(default=3, ge=1, le=8, description='Maximum focused search queries per node') class-attribute instance-attribute
planner_min_confidence = Field(default=0.35, ge=0.0, le=1.0, description='Minimum planner confidence to run search') class-attribute instance-attribute
planner_min_query_chars = Field(default=12, ge=3, le=100, description='Minimum query length accepted by guardrails') class-attribute instance-attribute
provider = Field(default=WebSearchProvider.DUCKDUCKGO, description='Search engine queried for evidence when web search is enabled.') class-attribute instance-attribute
region = Field(default='wt-wt', description='DuckDuckGo region code, such as us-en or de-de; wt-wt means no region.') class-attribute instance-attribute
reuse_evidence_across_attempt = Field(default=True, description='Reuse node-scoped evidence between retries for the same unit.') class-attribute instance-attribute
safesearch = Field(default='moderate', description='DuckDuckGo safe-search mode: on, moderate or off.') class-attribute instance-attribute
timeout_seconds = Field(default=8.0, ge=1.0, le=60.0, description='Seconds to wait for one search request, rounded down to whole seconds.') class-attribute instance-attribute
top_k = Field(default=3, ge=1, le=10, description='Search results fetched per query.') class-attribute instance-attribute

Methods:

parse_domains(value) classmethod
Source code in ontocast/config/settings.py
@field_validator("allowed_domains", "blocked_domains", mode="before")
@classmethod
def parse_domains(cls, value: str | list[str]) -> list[str]:
    if isinstance(value, str) and value.lstrip().startswith("["):
        value = json.loads(value)
    if isinstance(value, list):
        return [entry.strip().lower() for entry in value if entry.strip()]
    if isinstance(value, str):
        raw_values = [entry.strip().lower() for entry in value.split(",")]
        return [entry for entry in raw_values if entry]
    return []

WebSearchProvider

Bases: StrEnum

Supported web-search providers.

Source code in ontocast/config/settings.py
class WebSearchProvider(StrEnum):
    """Supported web-search providers."""

    DUCKDUCKGO = "duckduckgo"

Attributes

DUCKDUCKGO = 'duckduckgo' class-attribute instance-attribute

Functions: