Skip to content

ontocast.cli.server

OntoCast CLI entry point: serve (API) and process (local batch).

Example

Start the API server

ontocast serve

Process files locally without starting the server

ontocast process --input-path ./document.pdf --output-dir ./out

Attributes

F = TypeVar('F', bound=Callable[..., object]) module-attribute

logger = logging.getLogger(__name__) module-attribute

run = cli module-attribute

Classes

BootstrappedRuntime dataclass

Shared ToolBox + config after tenancy init.

Source code in ontocast/cli/server.py
@dataclass
class BootstrappedRuntime:
    """Shared ToolBox + config after tenancy init."""

    config: Config
    tools: ToolBox
    tenant: str
    project: str
    ontology_context_mode: OntologyContextMode

Attributes

config instance-attribute
ontology_context_mode instance-attribute
project instance-attribute
tenant instance-attribute
tools instance-attribute

Methods:

__init__(config, tools, tenant, project, ontology_context_mode)

LLMConfigAbort

Bases: ClickException

A provider rejected the request as configured; the run stopped.

Exits EX_CONFIG (78) rather than the generic 1, so a driver looping over configurations can tell a broken configuration from documents that merely failed to extract.

Source code in ontocast/cli/server.py
class LLMConfigAbort(click.ClickException):
    """A provider rejected the request as configured; the run stopped.

    Exits ``EX_CONFIG`` (78) rather than the generic 1, so a driver looping
    over configurations can tell a broken configuration from documents that
    merely failed to extract.
    """

    exit_code = 78

Attributes

exit_code = 78 class-attribute instance-attribute

Functions:

cli(env_files)

Source code in ontocast/cli/server.py
@click.group()
@click.option(
    "--env-file",
    "env_files",
    multiple=True,
    type=click.Path(exists=True, dir_okay=False, path_type=pathlib.Path),
    help=(
        "Dotenv file to load before running; repeatable. Later files override "
        "earlier ones, and variables exported in the shell override them all."
    ),
)
def cli(env_files: tuple[pathlib.Path, ...]) -> None:
    """OntoCast: start the API server or process local files in batch mode."""
    if env_files:
        apply_env_files(env_files)

get_next_level(level)

Source code in ontocast/cli/server.py
def get_next_level(level: int) -> int:
    levels = [
        logging.DEBUG,
        logging.INFO,
        logging.WARNING,
        logging.ERROR,
        logging.CRITICAL,
    ]

    try:
        idx = levels.index(level)
        return levels[min(idx + 1, len(levels) - 1)]
    except ValueError:
        return level

process(head_chunks, max_visits, tenant, project, ontology_dir, shapes_dir, wipe_vector_store, input_path, output_dir, facts_output_dir, ontology_output_dir, use_unit_pipeline, target_sections, exclude_sections, facts_user_instruction, summarize_sections, summary_max_sentences, document_type_hint, section_schema_id, keep_provenance, document_metadata)

Process local files through the extraction pipeline (no HTTP server).

Source code in ontocast/cli/server.py
@cli.command("process")
@_shared_runtime_options
@click.option(
    "--input-path",
    type=click.Path(path_type=pathlib.Path),
    required=True,
    help="File or directory to process locally (no HTTP server).",
)
@click.option(
    "--output-dir",
    type=click.Path(path_type=pathlib.Path),
    default=None,
    help=(
        "Shared directory for facts and ontology Turtle dumps. "
        "When omitted (and no per-kind override), dumps are written next to each input."
    ),
)
@click.option(
    "--facts-output-dir",
    type=click.Path(path_type=pathlib.Path),
    default=None,
    help="Override directory for ``*.facts.ttl`` dumps (defaults to --output-dir).",
)
@click.option(
    "--ontology-output-dir",
    type=click.Path(path_type=pathlib.Path),
    default=None,
    help=(
        "Override directory for ``*.ontology.ttl`` dumps (defaults to --output-dir)."
    ),
)
@click.option(
    "--use-unit-pipeline/--no-use-unit-pipeline",
    default=False,
    help=(
        "Run convert_document + run_unit_pipeline instead of the full workflow graph."
    ),
)
@click.option(
    "--target-sections",
    type=str,
    default=None,
    help=(
        "Comma-separated section labels to keep when chunking (e.g. results,methods). "
        "Enables section tagging in the workflow graph."
    ),
)
@click.option(
    "--exclude-sections",
    type=str,
    default=None,
    help=(
        "Comma-separated section labels to drop when chunking (e.g. "
        "acknowledgements,appendix). Unset = the resolved schema's defaults; "
        "pass an empty string to disable exclusion."
    ),
)
@click.option(
    "--facts-user-instruction",
    type=str,
    default="",
    help=(
        "Deployment-specific guidance appended to the facts render and "
        "critic prompts (the same per-request slot the HTTP API exposes). "
        "The library prompt stays domain-neutral; domain refinements belong "
        "here or in the shapes."
    ),
)
@click.option(
    "--summarize-sections",
    type=str,
    default=None,
    help=(
        "Comma-separated section labels to summarize before extraction, or '*' / empty "
        "for all chunks. Summaries are written per content unit during extraction."
    ),
)
@click.option(
    "--summary-max-sentences",
    type=int,
    default=5,
    show_default=True,
    help="Max sentences per chunk summary when --summarize-sections is set.",
)
@click.option(
    "--document-type-hint",
    type=str,
    default=None,
    help=(
        "Optional free-text hint about the source material (e.g. 'SEC 10-K', "
        "'journal article') to resolve section label schema and LLM tagging."
    ),
)
@click.option(
    "--section-schema-id",
    type=str,
    default=None,
    help=(
        "Section label schema id (academic, financial, legal, clinical, manual, "
        "fiction, patent, standard, news, general). Overrides "
        "--document-type-hint when set."
    ),
)
@click.option(
    "--keep-provenance/--strip-provenance",
    "keep_provenance",
    default=False,
    show_default=True,
    help=(
        "Keep chunk-level provenance in the dumped facts Turtle. Provenance is "
        "what lets a statement be traced back to its source span and "
        "re-verified against the document."
    ),
)
@click.option(
    "--document-metadata",
    type=str,
    default=None,
    help=(
        "JSON object of caller-asserted document identity metadata "
        '(e.g. \'{"doi":"10.1234/example","title":"…"}\'). '
        "When omitted, the filename is used as dcterms:title "
        "(file:line for JSONL records)."
    ),
)
def process(
    head_chunks: int | None,
    max_visits: int | None,
    tenant: str | None,
    project: str | None,
    ontology_dir: str | None,
    shapes_dir: str | None,
    wipe_vector_store: bool | None,
    input_path: pathlib.Path,
    output_dir: pathlib.Path | None,
    facts_output_dir: pathlib.Path | None,
    ontology_output_dir: pathlib.Path | None,
    use_unit_pipeline: bool,
    target_sections: str | None,
    exclude_sections: str | None,
    facts_user_instruction: str,
    summarize_sections: str | None,
    summary_max_sentences: int,
    document_type_hint: str | None,
    section_schema_id: str | None,
    keep_provenance: bool,
    document_metadata: str | None,
) -> None:
    """Process local files through the extraction pipeline (no HTTP server)."""
    runtime = _bootstrap_tools(
        tenant=tenant,
        project=project,
        wipe_vector_store=wipe_vector_store,
        ontology_dir=ontology_dir,
        shapes_dir=shapes_dir,
        flush_on_clean=True,
        batch=True,
    )
    # The parsers are shared with the HTTP layer and signal bad input by
    # raising; surface that as a click usage error rather than a traceback.
    try:
        parsed_target_sections = (
            parse_sections_list_param(target_sections, param="target-sections")
            if target_sections is not None
            else None
        )
        parsed_exclude_sections = (
            parse_sections_list_param(exclude_sections, param="exclude-sections")
            if exclude_sections is not None
            else None
        )
        parsed_summarize_sections = (
            parse_sections_list_param(summarize_sections, param="summarize-sections")
            if summarize_sections is not None
            else None
        )
        parsed_summary_max_sentences = parse_summary_max_sentences_param(
            summary_max_sentences,
            default=5,
        )
        parsed_document_type_hint = parse_document_type_hint_param(document_type_hint)
        parsed_section_schema_id = parse_section_schema_id_param(section_schema_id)
        parsed_max_visits = parse_max_visits_param(
            max_visits,
            default=runtime.config.server.max_visits_per_node,
        )
        parsed_document_metadata = parse_document_metadata_param(document_metadata)
    except ValueError as exc:
        raise click.BadParameter(str(exc)) from exc
    runtime.config.server.max_visits_per_node = parsed_max_visits

    workflow: CompiledStateGraph = create_agent_graph(runtime.tools)
    input_path = input_path.expanduser()
    out_dir = output_dir.expanduser() if output_dir is not None else None
    facts_dir = facts_output_dir.expanduser() if facts_output_dir is not None else None
    ontology_out_dir = (
        ontology_output_dir.expanduser() if ontology_output_dir is not None else None
    )
    supported_suffixes = get_batch_input_extensions(runtime.tools)
    try:
        files = sorted(crawl_directories(input_path, suffixes=supported_suffixes))
    except ValueError as exc:
        raise click.BadParameter(str(exc), param_hint="--input-path") from exc
    if not files:
        # An empty crawl used to exit 0 with no output, which reads as success.
        raise click.ClickException(
            f"No supported input files under {input_path} "
            f"(looking for {', '.join(supported_suffixes)})."
        )
    try:
        failed_files = asyncio.run(
            process_files_input(
                files,
                config=runtime.config,
                head_chunks=head_chunks,
                use_unit_pipeline=use_unit_pipeline,
                tools=runtime.tools,
                workflow=workflow,
                ontology_context_mode_value=runtime.ontology_context_mode,
                tenant=runtime.tenant,
                project=runtime.project,
                target_sections=parsed_target_sections,
                exclude_sections=parsed_exclude_sections,
                summarize_sections=parsed_summarize_sections,
                summary_max_sentences=parsed_summary_max_sentences,
                document_type_hint=parsed_document_type_hint,
                section_schema_id=parsed_section_schema_id,
                max_visits=parsed_max_visits,
                document_metadata=parsed_document_metadata,
                facts_user_instruction=facts_user_instruction,
                output_dir=out_dir,
                facts_output_dir=facts_dir,
                ontology_output_dir=ontology_out_dir,
                strip_provenance=not keep_provenance,
            )
        )
    except LLMConfigurationError as exc:
        # Not one failed file among many: the request as configured is one the
        # provider will never accept, so the batch stopped where it stood. A
        # dedicated exit code lets a benchmark driver tell "fix your LLM_*
        # settings" from "some documents did not extract".
        raise LLMConfigAbort(
            f"{exc} -- fix the LLM_* configuration and re-run."
        ) from exc
    if failed_files:
        # Exit non-zero so a scripted pipeline can tell a partial or total
        # failure from a clean run.
        raise click.ClickException(
            f"{len(failed_files)} of {len(files)} input file(s) failed: "
            + ", ".join(str(path) for path in failed_files[:5])
            + (" ..." if len(failed_files) > 5 else "")
        )

serve(head_chunks, max_visits, tenant, project, ontology_dir, shapes_dir, wipe_vector_store)

Start the OntoCast API server.

Source code in ontocast/cli/server.py
@cli.command("serve")
@_shared_runtime_options
def serve(
    head_chunks: int | None,
    max_visits: int | None,
    tenant: str | None,
    project: str | None,
    ontology_dir: str | None,
    shapes_dir: str | None,
    wipe_vector_store: bool | None,
) -> None:
    """Start the OntoCast API server."""
    runtime = _bootstrap_tools(
        tenant=tenant,
        project=project,
        wipe_vector_store=wipe_vector_store,
        ontology_dir=ontology_dir,
        shapes_dir=shapes_dir,
        flush_on_clean=False,
    )
    parsed_max_visits = parse_max_visits_param(
        max_visits,
        default=runtime.config.server.max_visits_per_node,
    )
    runtime.config.server.max_visits_per_node = parsed_max_visits
    app = create_app(
        tools=runtime.tools,
        server_config=runtime.config.server,
        head_chunks=head_chunks,
        active_tenant=runtime.tenant,
        active_project=runtime.project,
    )
    bind_host = runtime.config.server.host
    logger.info(
        "Starting Ontocast server on %s:%s", bind_host, runtime.config.server.port
    )
    if bind_host not in {"127.0.0.1", "localhost", "::1"}:
        logger.warning(
            "Binding %s: the server has no authentication and /flush is "
            "destructive. Put it behind a proxy that authenticates.",
            bind_host,
        )
    uvicorn.run(
        app,
        host=bind_host,
        port=runtime.config.server.port,
        log_level="info",
    )

store_names_replaced_by_tenancy(config, tenant, project)

Name the dataset and table settings that tenancy will replace.

serve and process derive every dataset, collection and table name from the tenant and project. A configured name that is neither the default scope's nor the target scope's was set by the operator and is about to be ignored.

Returns:

Type Description
list[str]

The environment names of those settings, in declaration order.

Source code in ontocast/cli/server.py
def store_names_replaced_by_tenancy(
    config: Config, tenant: str, project: str
) -> list[str]:
    """Name the dataset and table settings that tenancy will replace.

    ``serve`` and ``process`` derive every dataset, collection and table name
    from the tenant and project. A configured name that is neither the default
    scope's nor the target scope's was set by the operator and is about to be
    ignored.

    Returns:
        The environment names of those settings, in declaration order.
    """
    default = TenancyScope.build(DEFAULT_TENANT, DEFAULT_PROJECT)
    target = TenancyScope.build(tenant, project)
    tool_config = config.tool_config
    groups = (
        (
            tool_config.fuseki,
            (
                ("dataset", "facts_name"),
                ("ontologies_dataset", "ontologies_name"),
                ("shapes_dataset", "shapes_name"),
            ),
        ),
        (
            tool_config.vector_store,
            (("ontology_table", "ontologies_name"), ("facts_table", "facts_name")),
        ),
        (
            tool_config.qdrant,
            (
                ("ontology_collection", "ontologies_name"),
                ("facts_collection", "facts_name"),
            ),
        ),
        (
            tool_config.lancedb,
            (("ontology_table", "ontologies_name"), ("facts_table", "facts_name")),
        ),
    )
    replaced: list[str] = []
    for group, fields in groups:
        for field, scope_attr in fields:
            value = getattr(group, field)
            derived = {getattr(default, scope_attr), getattr(target, scope_attr)}
            if value is not None and value not in derived:
                replaced.append(env_names(type(group), field)[0])
    return replaced