Skip to content

graflo.data_source.rdf_identity

Identity of RDF nodes that have no stable IRI of their own.

  • A blank node is keyed on a digest of its outgoing triples, so one content gets one key on every read, whatever label the parser or the endpoint gave it.
  • IRIs joined by owl:sameAs form a component, and its smallest IRI stands for every member.

Nothing here depends on an RDF library: terms arrive as :class:Term.

Attributes

BLANK_PREFIX = '_:' module-attribute

OWL_SAME_AS = 'http://www.w3.org/2002/07/owl#sameAs' module-attribute

Outgoing = Callable[[str], Iterable[tuple[str, Term]]] module-attribute

XSD_STRING = 'http://www.w3.org/2001/XMLSchema#string' module-attribute

Classes

BlankNodeKeys

Content keys for blank nodes.

The key is a digest of the node's outgoing triples. A blank-node object counts by its own key, so a nested structure is keyed as a whole. A reference back to a node on the way down counts by how far back it points, which ends a cycle without naming a label.

Two blank nodes with the same content have the same key.

Source code in graflo/data_source/rdf_identity.py
class BlankNodeKeys:
    """Content keys for blank nodes.

    The key is a digest of the node's outgoing triples. A blank-node object
    counts by its own key, so a nested structure is keyed as a whole. A
    reference back to a node on the way down counts by how far back it
    points, which ends a cycle without naming a label.

    Two blank nodes with the same content have the same key.
    """

    def __init__(self, outgoing: Outgoing) -> None:
        self._outgoing = outgoing
        # A node on no cycle has one digest wherever it is met. A node on a
        # cycle is keyed from itself, and that key stands for it only when it
        # is the node asked for.
        self._acyclic: dict[str, str] = {}
        self._cyclic: dict[str, str] = {}

    def key(self, label: str) -> str:
        """The key of the blank node *label*."""
        known = self._acyclic.get(label) or self._cyclic.get(label)
        if known is not None:
            return known

        stack = [_Frame(label, self._outgoing(label), 0)]
        depth_of = {label: 0}
        while True:
            frame = stack[-1]
            depth = len(stack) - 1
            descended = False
            for predicate, obj in frame.pending:
                if obj.kind != "bnode":
                    frame.lines.append(f"<{predicate}> {term_text(obj)}")
                elif obj.value in self._acyclic:
                    frame.lines.append(f"<{predicate}> _:{self._acyclic[obj.value]}")
                elif obj.value in depth_of:
                    back = depth_of[obj.value]
                    frame.lowest = min(frame.lowest, back)
                    frame.lines.append(f"<{predicate}> _:^{depth - back}")
                else:
                    frame.waiting = predicate
                    depth_of[obj.value] = depth + 1
                    stack.append(
                        _Frame(obj.value, self._outgoing(obj.value), depth + 1)
                    )
                    descended = True
                    break
            if descended:
                continue

            digest = hashlib.sha256(
                "\n".join(sorted(frame.lines)).encode("utf-8")
            ).hexdigest()[:_KEY_LENGTH]
            stack.pop()
            del depth_of[frame.label]
            on_cycle = frame.lowest <= depth
            if not on_cycle:
                self._acyclic[frame.label] = digest
            if not stack:
                if on_cycle:
                    self._cyclic[frame.label] = digest
                return digest
            parent = stack[-1]
            parent.lines.append(f"<{parent.waiting}> _:{digest}")
            parent.lowest = min(parent.lowest, frame.lowest)

Methods:

__init__(outgoing)
Source code in graflo/data_source/rdf_identity.py
def __init__(self, outgoing: Outgoing) -> None:
    self._outgoing = outgoing
    # A node on no cycle has one digest wherever it is met. A node on a
    # cycle is keyed from itself, and that key stands for it only when it
    # is the node asked for.
    self._acyclic: dict[str, str] = {}
    self._cyclic: dict[str, str] = {}
key(label)

The key of the blank node label.

Source code in graflo/data_source/rdf_identity.py
def key(self, label: str) -> str:
    """The key of the blank node *label*."""
    known = self._acyclic.get(label) or self._cyclic.get(label)
    if known is not None:
        return known

    stack = [_Frame(label, self._outgoing(label), 0)]
    depth_of = {label: 0}
    while True:
        frame = stack[-1]
        depth = len(stack) - 1
        descended = False
        for predicate, obj in frame.pending:
            if obj.kind != "bnode":
                frame.lines.append(f"<{predicate}> {term_text(obj)}")
            elif obj.value in self._acyclic:
                frame.lines.append(f"<{predicate}> _:{self._acyclic[obj.value]}")
            elif obj.value in depth_of:
                back = depth_of[obj.value]
                frame.lowest = min(frame.lowest, back)
                frame.lines.append(f"<{predicate}> _:^{depth - back}")
            else:
                frame.waiting = predicate
                depth_of[obj.value] = depth + 1
                stack.append(
                    _Frame(obj.value, self._outgoing(obj.value), depth + 1)
                )
                descended = True
                break
        if descended:
            continue

        digest = hashlib.sha256(
            "\n".join(sorted(frame.lines)).encode("utf-8")
        ).hexdigest()[:_KEY_LENGTH]
        stack.pop()
        del depth_of[frame.label]
        on_cycle = frame.lowest <= depth
        if not on_cycle:
            self._acyclic[frame.label] = digest
        if not stack:
            if on_cycle:
                self._cyclic[frame.label] = digest
            return digest
        parent = stack[-1]
        parent.lines.append(f"<{parent.waiting}> _:{digest}")
        parent.lowest = min(parent.lowest, frame.lowest)

SameAs

Components of owl:sameAs statements between IRIs.

Source code in graflo/data_source/rdf_identity.py
class SameAs:
    """Components of ``owl:sameAs`` statements between IRIs."""

    def __init__(self, pairs: Iterable[tuple[str, str]]) -> None:
        parent: dict[str, str] = {}

        def find(iri: str) -> str:
            root = iri
            while parent.setdefault(root, root) != root:
                root = parent[root]
            while parent[iri] != root:
                parent[iri], iri = root, parent[iri]
            return root

        for left, right in pairs:
            a, b = find(left), find(right)
            if a != b:
                # The smaller IRI is the root, so a root is its component's smallest.
                parent[max(a, b)] = min(a, b)

        self._canonical = {iri: find(iri) for iri in parent}
        self._members: dict[str, list[str]] = {}
        for iri, root in self._canonical.items():
            self._members.setdefault(root, []).append(iri)
        for members in self._members.values():
            members.sort()

    def __bool__(self) -> bool:
        return bool(self._canonical)

    def canonical(self, iri: str) -> str:
        """The IRI that stands for *iri*: the smallest of its component."""
        return self._canonical.get(iri, iri)

    def members(self, iri: str) -> list[str]:
        """Every IRI of the component of *iri*, sorted; ``[iri]`` when it is alone."""
        return self._members.get(self.canonical(iri), [iri])

Methods:

__bool__()
Source code in graflo/data_source/rdf_identity.py
def __bool__(self) -> bool:
    return bool(self._canonical)
__init__(pairs)
Source code in graflo/data_source/rdf_identity.py
def __init__(self, pairs: Iterable[tuple[str, str]]) -> None:
    parent: dict[str, str] = {}

    def find(iri: str) -> str:
        root = iri
        while parent.setdefault(root, root) != root:
            root = parent[root]
        while parent[iri] != root:
            parent[iri], iri = root, parent[iri]
        return root

    for left, right in pairs:
        a, b = find(left), find(right)
        if a != b:
            # The smaller IRI is the root, so a root is its component's smallest.
            parent[max(a, b)] = min(a, b)

    self._canonical = {iri: find(iri) for iri in parent}
    self._members: dict[str, list[str]] = {}
    for iri, root in self._canonical.items():
        self._members.setdefault(root, []).append(iri)
    for members in self._members.values():
        members.sort()
canonical(iri)

The IRI that stands for iri: the smallest of its component.

Source code in graflo/data_source/rdf_identity.py
def canonical(self, iri: str) -> str:
    """The IRI that stands for *iri*: the smallest of its component."""
    return self._canonical.get(iri, iri)
members(iri)

Every IRI of the component of iri, sorted; [iri] when it is alone.

Source code in graflo/data_source/rdf_identity.py
def members(self, iri: str) -> list[str]:
    """Every IRI of the component of *iri*, sorted; ``[iri]`` when it is alone."""
    return self._members.get(self.canonical(iri), [iri])

Term

Bases: NamedTuple

An RDF term, independent of the library that read it.

value is the IRI, the blank-node label, or the literal's lexical form.

Source code in graflo/data_source/rdf_identity.py
class Term(NamedTuple):
    """An RDF term, independent of the library that read it.

    ``value`` is the IRI, the blank-node label, or the literal's lexical form.
    """

    kind: Literal["iri", "bnode", "literal"]
    value: str
    datatype: str | None = None
    lang: str | None = None

Attributes

datatype = None class-attribute instance-attribute
kind instance-attribute
lang = None class-attribute instance-attribute
value instance-attribute

Functions:

term_text(term)

One spelling per IRI or literal; a plain literal is an xsd:string.

Source code in graflo/data_source/rdf_identity.py
def term_text(term: Term) -> str:
    """One spelling per IRI or literal; a plain literal is an ``xsd:string``."""
    if term.kind == "iri":
        return f"<{term.value}>"
    text = json.dumps(term.value, ensure_ascii=False)
    if term.lang:
        return f"{text}@{term.lang.lower()}"
    if term.datatype and term.datatype != XSD_STRING:
        return f"{text}^^<{term.datatype}>"
    return text