@cli.command("process")
@_shared_runtime_options
@click.option(
"--input-path",
type=click.Path(path_type=pathlib.Path),
required=True,
help="File or directory to process locally (no HTTP server).",
)
@click.option(
"--output-dir",
type=click.Path(path_type=pathlib.Path),
default=None,
help=(
"Shared directory for facts and ontology Turtle dumps. "
"When omitted (and no per-kind override), dumps are written next to each input."
),
)
@click.option(
"--facts-output-dir",
type=click.Path(path_type=pathlib.Path),
default=None,
help="Override directory for ``*.facts.ttl`` dumps (defaults to --output-dir).",
)
@click.option(
"--ontology-output-dir",
type=click.Path(path_type=pathlib.Path),
default=None,
help=(
"Override directory for ``*.ontology.ttl`` dumps (defaults to --output-dir)."
),
)
@click.option(
"--use-unit-pipeline/--no-use-unit-pipeline",
default=False,
help=(
"Run convert_document + run_unit_pipeline instead of the full workflow graph."
),
)
@click.option(
"--target-sections",
type=str,
default=None,
help=(
"Comma-separated section labels to keep when chunking (e.g. results,methods). "
"Enables section tagging in the workflow graph."
),
)
@click.option(
"--exclude-sections",
type=str,
default=None,
help=(
"Comma-separated section labels to drop when chunking (e.g. "
"acknowledgements,appendix). Unset = the resolved schema's defaults; "
"pass an empty string to disable exclusion."
),
)
@click.option(
"--summarize-sections",
type=str,
default=None,
help=(
"Comma-separated section labels to summarize before extraction, or '*' / empty "
"for all chunks. When set, summarization runs inside chunk preparation."
),
)
@click.option(
"--summary-max-sentences",
type=int,
default=5,
show_default=True,
help="Max sentences per chunk summary when --summarize-sections is set.",
)
@click.option(
"--document-type-hint",
type=str,
default=None,
help=(
"Optional free-text hint about the source material (e.g. 'SEC 10-K', "
"'journal article') to resolve section label schema and LLM tagging."
),
)
@click.option(
"--section-schema-id",
type=str,
default=None,
help=(
"Section label schema id (academic, financial, legal, clinical, manual, "
"fiction, general). Overrides --document-type-hint when set."
),
)
@click.option(
"--document-metadata",
type=str,
default=None,
help=(
"JSON object of caller-asserted document identity metadata "
'(e.g. \'{"doi":"10.1234/example","title":"…"}\'). '
"When omitted, the filename is used as dcterms:title "
"(file:line for JSONL records)."
),
)
def process(
head_chunks: int | None,
max_visits: int | None,
tenant: str | None,
project: str | None,
wipe_vector_store: bool | None,
input_path: pathlib.Path,
output_dir: pathlib.Path | None,
facts_output_dir: pathlib.Path | None,
ontology_output_dir: pathlib.Path | None,
use_unit_pipeline: bool,
target_sections: str | None,
exclude_sections: str | None,
summarize_sections: str | None,
summary_max_sentences: int,
document_type_hint: str | None,
section_schema_id: str | None,
document_metadata: str | None,
) -> None:
"""Process local files through the extraction pipeline (no HTTP server)."""
runtime = _bootstrap_tools(
tenant=tenant,
project=project,
wipe_vector_store=wipe_vector_store,
flush_on_clean=True,
)
# The parsers are shared with the HTTP layer and signal bad input by
# raising; surface that as a click usage error rather than a traceback.
try:
parsed_target_sections = (
parse_sections_list_param(target_sections, param="target-sections")
if target_sections is not None
else None
)
parsed_exclude_sections = (
parse_sections_list_param(exclude_sections, param="exclude-sections")
if exclude_sections is not None
else None
)
parsed_summarize_sections = (
parse_sections_list_param(summarize_sections, param="summarize-sections")
if summarize_sections is not None
else None
)
parsed_summary_max_sentences = parse_summary_max_sentences_param(
summary_max_sentences,
default=5,
)
parsed_document_type_hint = parse_document_type_hint_param(document_type_hint)
parsed_section_schema_id = parse_section_schema_id_param(section_schema_id)
parsed_max_visits = parse_max_visits_param(
max_visits,
default=runtime.config.server.max_visits_per_node,
)
parsed_document_metadata = parse_document_metadata_param(document_metadata)
except ValueError as exc:
raise click.BadParameter(str(exc)) from exc
runtime.config.server.max_visits_per_node = parsed_max_visits
workflow: CompiledStateGraph = create_agent_graph(runtime.tools)
input_path = input_path.expanduser()
out_dir = output_dir.expanduser() if output_dir is not None else None
facts_dir = facts_output_dir.expanduser() if facts_output_dir is not None else None
ontology_dir = (
ontology_output_dir.expanduser() if ontology_output_dir is not None else None
)
supported_suffixes = get_supported_input_extensions(runtime.tools)
try:
files = sorted(crawl_directories(input_path, suffixes=supported_suffixes))
except ValueError as exc:
raise click.BadParameter(str(exc), param_hint="--input-path") from exc
if not files:
# An empty crawl used to exit 0 with no output, which reads as success.
raise click.ClickException(
f"No supported input files under {input_path} "
f"(looking for {', '.join(supported_suffixes)})."
)
failed_files = asyncio.run(
process_files_input(
files,
config=runtime.config,
head_chunks=head_chunks,
use_unit_pipeline=use_unit_pipeline,
tools=runtime.tools,
workflow=workflow,
ontology_context_mode_value=runtime.ontology_context_mode,
tenant=runtime.tenant,
project=runtime.project,
target_sections=parsed_target_sections,
exclude_sections=parsed_exclude_sections,
summarize_sections=parsed_summarize_sections,
summary_max_sentences=parsed_summary_max_sentences,
document_type_hint=parsed_document_type_hint,
section_schema_id=parsed_section_schema_id,
max_visits=parsed_max_visits,
document_metadata=parsed_document_metadata,
output_dir=out_dir,
facts_output_dir=facts_dir,
ontology_output_dir=ontology_dir,
)
)
if failed_files:
# Exit non-zero so a scripted pipeline can tell a partial or total
# failure from a clean run.
raise click.ClickException(
f"{len(failed_files)} of {len(files)} input file(s) failed: "
+ ", ".join(str(path) for path in failed_files[:5])
+ (" ..." if len(failed_files) > 5 else "")
)