Skip to content

Core Modules

Everything below is importable from the top-level suthing package.

File Handling

Read and write files whose format is inferred from the file name.

Supported formats are listed in :class:FileType; the extension → format map is :data:EXTENSIONS. Any of them can be compressed with .gz, .bz2, .xz or .zst (see :mod:suthing.fs).

FileHandle

Format-aware file I/O: one call to read or write a whole file.

The format comes from the extension (:data:EXTENSIONS) unless how is given, and a compression suffix is handled transparently. An unknown extension is an error rather than a guess.

Formats read to / write from:

  • YAML, JSON (.jsonld too): any YAML/JSON value. YAML loads with safe_load; values json cannot serialise go through :func:suthing.to_jsonable.
  • JSONL: a list of values, one per line.
  • CSV, TSV: a pandas DataFrame (extra keyword arguments go to read_csv / to_csv).
  • TXT: a str.
  • ENV: a dict[str, str]; loading does not touch os.environ (use :func:suthing.load_env for that).
  • PICKLE: any picklable object. Only unpickle files you trust.
Example

data = FileHandle.load("config.yaml") FileHandle.dump(data, "out/config.json.gz", mkdir=True)

Source code in suthing/file_handle.py
class FileHandle:
    """Format-aware file I/O: one call to read or write a whole file.

    The format comes from the extension (:data:`EXTENSIONS`) unless ``how`` is
    given, and a compression suffix is handled transparently. An unknown
    extension is an error rather than a guess.

    Formats read to / write from:

    - YAML, JSON (``.jsonld`` too): any YAML/JSON value. YAML loads with
      ``safe_load``; values json cannot serialise go through
      :func:`suthing.to_jsonable`.
    - JSONL: a list of values, one per line.
    - CSV, TSV: a pandas ``DataFrame`` (extra keyword arguments go to
      ``read_csv`` / ``to_csv``).
    - TXT: a ``str``.
    - ENV: a ``dict[str, str]``; loading does *not* touch ``os.environ``
      (use :func:`suthing.load_env` for that).
    - PICKLE: any picklable object. Only unpickle files you trust.

    Example:
        >>> data = FileHandle.load("config.yaml")
        >>> FileHandle.dump(data, "out/config.json.gz", mkdir=True)
    """

    @classmethod
    def load(
        cls,
        path: PathLike | None = None,
        *,
        how: FileType | str | None = None,
        fpath: PathLike | None = None,
        **kwargs: Any,
    ) -> Any:
        """Read a file from disk.

        Args:
            path: File to read; ``~`` is expanded.
            how: Format override; by default inferred from the extension.
            fpath: Deprecated spelling of *path*, kept so code written for
                suthing 0.5 keeps working; emits ``DeprecationWarning``.
            **kwargs: Passed to ``pandas.read_csv`` for CSV/TSV. Any other
                format rejects extra arguments.

        Returns:
            The parsed contents (see the class docstring for types).

        Raises:
            ValueError: If the format cannot be inferred.
            TypeError: For keyword arguments the format does not take, or
                when neither or both of *path* and *fpath* are given.
        """
        if fpath is not None:
            warnings.warn(
                "FileHandle.load(fpath=...) is deprecated; pass the path positionally",
                DeprecationWarning,
                stacklevel=2,
            )
            if path is not None:
                raise TypeError("pass the path once: positionally or as fpath=")
            path = fpath
        if path is None:
            raise TypeError("FileHandle.load() missing the file path")
        fmt, _ = _resolve(path, how)
        with open_compressed(path, "rb") as stream:
            return _read(stream, fmt, **kwargs)

    @classmethod
    def load_resource(
        cls,
        package: str,
        name: str,
        *,
        how: FileType | str | None = None,
        **kwargs: Any,
    ) -> Any:
        """Read a data file shipped inside an importable package.

        Args:
            package: Dotted package name, e.g. ``"mypkg.data"``.
            name: File name relative to the package; may contain ``/``.
            how: Format override; by default inferred from the extension.
            **kwargs: As for :meth:`load`.

        Returns:
            The parsed contents.
        """
        fmt, compression = _resolve(name, how)
        data = resources.files(package).joinpath(name).read_bytes()
        with io.BytesIO(data) as raw:
            if compression is None:
                return _read(raw, fmt, **kwargs)
            with wrap_compressed(raw, compression, "rb") as stream:
                return _read(stream, fmt, **kwargs)

    @classmethod
    def iter(
        cls,
        path: PathLike,
        *,
        how: FileType | str | None = None,
        chunksize: int = 10_000,
        **kwargs: Any,
    ) -> Iterator[Any]:
        """Stream a file instead of loading it whole.

        Args:
            path: File to read.
            how: Format override; by default inferred from the extension.
            chunksize: Rows per ``DataFrame`` chunk for CSV/TSV.
            **kwargs: Passed to ``pandas.read_csv`` for CSV/TSV, and to
                :func:`suthing.jsonl.iter_jsonl_stream` (``strict``,
                ``require_object``) for JSONL.

        Yields:
            JSONL: one value per line. CSV/TSV: ``DataFrame`` chunks.
            TXT: lines without their trailing newline.

        Raises:
            ValueError: For a format that cannot be streamed.
        """
        fmt, _ = _resolve(path, how)
        if fmt not in (FileType.JSONL, FileType.CSV, FileType.TSV, FileType.TXT):
            raise ValueError(f"{fmt.name} files cannot be streamed; use load()")
        with open_compressed(path, "rb") as stream:
            if fmt is FileType.JSONL:
                yield from iter_jsonl_stream(stream, **kwargs)
            elif fmt is FileType.TXT:
                _no_kwargs(fmt, kwargs)
                for raw in stream:
                    yield raw.decode("utf-8").rstrip("\r\n")
            else:
                import pandas as pd

                if fmt is FileType.TSV:
                    kwargs.setdefault("sep", "\t")
                with pd.read_csv(stream, chunksize=chunksize, **kwargs) as reader:
                    yield from reader

    @classmethod
    def dump(
        cls,
        item: Any,
        path: PathLike,
        *,
        how: FileType | str | None = None,
        atomic: bool = True,
        mkdir: bool = False,
        **kwargs: Any,
    ) -> pathlib.Path:
        """Write *item* to a file, compressing by suffix.

        Args:
            item: Value to write (see the class docstring for accepted types).
            path: Destination; ``~`` is expanded.
            how: Format override; by default inferred from the extension.
            atomic: Replace *path* in one step, so readers never see a
                partial file (see :func:`suthing.fs.atomic_open`).
            mkdir: Create missing parent directories first.
            **kwargs: Passed to ``DataFrame.to_csv`` for CSV/TSV. Any other
                format rejects extra arguments.

        Returns:
            The path written.

        Raises:
            ValueError: If the format cannot be inferred.
            TypeError: If *item* does not fit the format, or for keyword
                arguments the format does not take.
        """
        fmt, compression = _resolve(path, how)
        dest = pathlib.Path(path).expanduser()
        if mkdir:
            dest.parent.mkdir(parents=True, exist_ok=True)
        if fmt is FileType.JSONL:
            _no_kwargs(fmt, kwargs)
            write = lambda stream: write_jsonl_stream(item, stream)
        else:
            # Encode first: a type error must not leave a truncated file behind.
            payload = _encode(item, fmt, **kwargs)
            write = lambda stream: stream.write(payload)

        if not atomic:
            with open_compressed(dest, "wb") as stream:
                write(stream)
            return dest
        with atomic_open(dest) as raw:
            if compression is None:
                write(raw)
            else:
                with wrap_compressed(raw, compression, "wb") as stream:
                    write(stream)
        return dest

dump(item, path, *, how=None, atomic=True, mkdir=False, **kwargs) classmethod

Write item to a file, compressing by suffix.

Parameters:

Name Type Description Default
item Any

Value to write (see the class docstring for accepted types).

required
path PathLike

Destination; ~ is expanded.

required
how FileType | str | None

Format override; by default inferred from the extension.

None
atomic bool

Replace path in one step, so readers never see a partial file (see :func:suthing.fs.atomic_open).

True
mkdir bool

Create missing parent directories first.

False
**kwargs Any

Passed to DataFrame.to_csv for CSV/TSV. Any other format rejects extra arguments.

{}

Returns:

Type Description
Path

The path written.

Raises:

Type Description
ValueError

If the format cannot be inferred.

TypeError

If item does not fit the format, or for keyword arguments the format does not take.

Source code in suthing/file_handle.py
@classmethod
def dump(
    cls,
    item: Any,
    path: PathLike,
    *,
    how: FileType | str | None = None,
    atomic: bool = True,
    mkdir: bool = False,
    **kwargs: Any,
) -> pathlib.Path:
    """Write *item* to a file, compressing by suffix.

    Args:
        item: Value to write (see the class docstring for accepted types).
        path: Destination; ``~`` is expanded.
        how: Format override; by default inferred from the extension.
        atomic: Replace *path* in one step, so readers never see a
            partial file (see :func:`suthing.fs.atomic_open`).
        mkdir: Create missing parent directories first.
        **kwargs: Passed to ``DataFrame.to_csv`` for CSV/TSV. Any other
            format rejects extra arguments.

    Returns:
        The path written.

    Raises:
        ValueError: If the format cannot be inferred.
        TypeError: If *item* does not fit the format, or for keyword
            arguments the format does not take.
    """
    fmt, compression = _resolve(path, how)
    dest = pathlib.Path(path).expanduser()
    if mkdir:
        dest.parent.mkdir(parents=True, exist_ok=True)
    if fmt is FileType.JSONL:
        _no_kwargs(fmt, kwargs)
        write = lambda stream: write_jsonl_stream(item, stream)
    else:
        # Encode first: a type error must not leave a truncated file behind.
        payload = _encode(item, fmt, **kwargs)
        write = lambda stream: stream.write(payload)

    if not atomic:
        with open_compressed(dest, "wb") as stream:
            write(stream)
        return dest
    with atomic_open(dest) as raw:
        if compression is None:
            write(raw)
        else:
            with wrap_compressed(raw, compression, "wb") as stream:
                write(stream)
    return dest

iter(path, *, how=None, chunksize=10000, **kwargs) classmethod

Stream a file instead of loading it whole.

Parameters:

Name Type Description Default
path PathLike

File to read.

required
how FileType | str | None

Format override; by default inferred from the extension.

None
chunksize int

Rows per DataFrame chunk for CSV/TSV.

10000
**kwargs Any

Passed to pandas.read_csv for CSV/TSV, and to :func:suthing.jsonl.iter_jsonl_stream (strict, require_object) for JSONL.

{}

Yields:

Name Type Description
JSONL Any

one value per line. CSV/TSV: DataFrame chunks.

TXT Any

lines without their trailing newline.

Raises:

Type Description
ValueError

For a format that cannot be streamed.

Source code in suthing/file_handle.py
@classmethod
def iter(
    cls,
    path: PathLike,
    *,
    how: FileType | str | None = None,
    chunksize: int = 10_000,
    **kwargs: Any,
) -> Iterator[Any]:
    """Stream a file instead of loading it whole.

    Args:
        path: File to read.
        how: Format override; by default inferred from the extension.
        chunksize: Rows per ``DataFrame`` chunk for CSV/TSV.
        **kwargs: Passed to ``pandas.read_csv`` for CSV/TSV, and to
            :func:`suthing.jsonl.iter_jsonl_stream` (``strict``,
            ``require_object``) for JSONL.

    Yields:
        JSONL: one value per line. CSV/TSV: ``DataFrame`` chunks.
        TXT: lines without their trailing newline.

    Raises:
        ValueError: For a format that cannot be streamed.
    """
    fmt, _ = _resolve(path, how)
    if fmt not in (FileType.JSONL, FileType.CSV, FileType.TSV, FileType.TXT):
        raise ValueError(f"{fmt.name} files cannot be streamed; use load()")
    with open_compressed(path, "rb") as stream:
        if fmt is FileType.JSONL:
            yield from iter_jsonl_stream(stream, **kwargs)
        elif fmt is FileType.TXT:
            _no_kwargs(fmt, kwargs)
            for raw in stream:
                yield raw.decode("utf-8").rstrip("\r\n")
        else:
            import pandas as pd

            if fmt is FileType.TSV:
                kwargs.setdefault("sep", "\t")
            with pd.read_csv(stream, chunksize=chunksize, **kwargs) as reader:
                yield from reader

load(path=None, *, how=None, fpath=None, **kwargs) classmethod

Read a file from disk.

Parameters:

Name Type Description Default
path PathLike | None

File to read; ~ is expanded.

None
how FileType | str | None

Format override; by default inferred from the extension.

None
fpath PathLike | None

Deprecated spelling of path, kept so code written for suthing 0.5 keeps working; emits DeprecationWarning.

None
**kwargs Any

Passed to pandas.read_csv for CSV/TSV. Any other format rejects extra arguments.

{}

Returns:

Type Description
Any

The parsed contents (see the class docstring for types).

Raises:

Type Description
ValueError

If the format cannot be inferred.

TypeError

For keyword arguments the format does not take, or when neither or both of path and fpath are given.

Source code in suthing/file_handle.py
@classmethod
def load(
    cls,
    path: PathLike | None = None,
    *,
    how: FileType | str | None = None,
    fpath: PathLike | None = None,
    **kwargs: Any,
) -> Any:
    """Read a file from disk.

    Args:
        path: File to read; ``~`` is expanded.
        how: Format override; by default inferred from the extension.
        fpath: Deprecated spelling of *path*, kept so code written for
            suthing 0.5 keeps working; emits ``DeprecationWarning``.
        **kwargs: Passed to ``pandas.read_csv`` for CSV/TSV. Any other
            format rejects extra arguments.

    Returns:
        The parsed contents (see the class docstring for types).

    Raises:
        ValueError: If the format cannot be inferred.
        TypeError: For keyword arguments the format does not take, or
            when neither or both of *path* and *fpath* are given.
    """
    if fpath is not None:
        warnings.warn(
            "FileHandle.load(fpath=...) is deprecated; pass the path positionally",
            DeprecationWarning,
            stacklevel=2,
        )
        if path is not None:
            raise TypeError("pass the path once: positionally or as fpath=")
        path = fpath
    if path is None:
        raise TypeError("FileHandle.load() missing the file path")
    fmt, _ = _resolve(path, how)
    with open_compressed(path, "rb") as stream:
        return _read(stream, fmt, **kwargs)

load_resource(package, name, *, how=None, **kwargs) classmethod

Read a data file shipped inside an importable package.

Parameters:

Name Type Description Default
package str

Dotted package name, e.g. "mypkg.data".

required
name str

File name relative to the package; may contain /.

required
how FileType | str | None

Format override; by default inferred from the extension.

None
**kwargs Any

As for :meth:load.

{}

Returns:

Type Description
Any

The parsed contents.

Source code in suthing/file_handle.py
@classmethod
def load_resource(
    cls,
    package: str,
    name: str,
    *,
    how: FileType | str | None = None,
    **kwargs: Any,
) -> Any:
    """Read a data file shipped inside an importable package.

    Args:
        package: Dotted package name, e.g. ``"mypkg.data"``.
        name: File name relative to the package; may contain ``/``.
        how: Format override; by default inferred from the extension.
        **kwargs: As for :meth:`load`.

    Returns:
        The parsed contents.
    """
    fmt, compression = _resolve(name, how)
    data = resources.files(package).joinpath(name).read_bytes()
    with io.BytesIO(data) as raw:
        if compression is None:
            return _read(raw, fmt, **kwargs)
        with wrap_compressed(raw, compression, "rb") as stream:
            return _read(stream, fmt, **kwargs)

FileType

Bases: str, Enum

Formats :class:FileHandle reads and writes.

Source code in suthing/file_handle.py
class FileType(str, Enum):
    """Formats :class:`FileHandle` reads and writes."""

    YAML = "yaml"
    JSON = "json"
    JSONL = "jsonl"
    PICKLE = "pkl"
    CSV = "csv"
    TSV = "tsv"
    TXT = "txt"
    ENV = "env"

detect_format(path)

Infer the format and compression of path from its name.

Parameters:

Name Type Description Default
path PathLike

File path or name.

required

Returns:

Type Description
FileType | None

(format, compression); format is None for an unknown

str | None

extension and compression is None for an uncompressed file.

Source code in suthing/file_handle.py
def detect_format(path: PathLike) -> tuple[FileType | None, str | None]:
    """Infer the format and compression of *path* from its name.

    Args:
        path: File path or name.

    Returns:
        ``(format, compression)``; ``format`` is ``None`` for an unknown
        extension and ``compression`` is ``None`` for an uncompressed file.
    """
    fmt, compression = split_compression(path)
    return EXTENSIONS.get(fmt), compression

JSON Lines

JSON Lines (.jsonl / .ndjson): one JSON value per line.

All functions accept compressed files (.jsonl.gz etc., see :mod:suthing.fs). Blank lines are skipped on read.

JsonlError

Bases: NamedTuple

A line that could not be read.

Attributes:

Name Type Description
line int

1-based line number.

message str

What was wrong with it.

Source code in suthing/jsonl.py
class JsonlError(NamedTuple):
    """A line that could not be read.

    Attributes:
        line: 1-based line number.
        message: What was wrong with it.
    """

    line: int
    message: str

    def __str__(self) -> str:
        return f"line {self.line}: {self.message}"

encode_jsonl_row(row)

Encode one value as a UTF-8 JSON line, newline included.

Non-ASCII text is written as-is and values json cannot serialise go through :func:suthing.to_jsonable.

Source code in suthing/jsonl.py
def encode_jsonl_row(row: Any) -> bytes:
    """Encode one value as a UTF-8 JSON line, newline included.

    Non-ASCII text is written as-is and values json cannot serialise go through
    :func:`suthing.to_jsonable`.
    """
    return (json.dumps(row, ensure_ascii=False, default=to_jsonable) + "\n").encode(
        "utf-8"
    )

iter_jsonl(path, *, strict=True, require_object=False)

Stream the values of a JSON Lines file without loading it whole.

Parameters:

Name Type Description Default
path PathLike

File to read; may be compressed.

required
strict bool

Raise on a bad line. When False, bad lines are logged at WARNING level and skipped.

True
require_object bool

Treat any value that is not a JSON object as a bad line.

False

Yields:

Type Description
Any

One parsed value per non-blank line.

Raises:

Type Description
ValueError

On a bad line when strict is set; the message carries the line number.

Source code in suthing/jsonl.py
def iter_jsonl(
    path: PathLike, *, strict: bool = True, require_object: bool = False
) -> Iterator[Any]:
    """Stream the values of a JSON Lines file without loading it whole.

    Args:
        path: File to read; may be compressed.
        strict: Raise on a bad line. When ``False``, bad lines are logged at
            WARNING level and skipped.
        require_object: Treat any value that is not a JSON object as a bad line.

    Yields:
        One parsed value per non-blank line.

    Raises:
        ValueError: On a bad line when *strict* is set; the message carries the
            line number.
    """
    with open_compressed(path, "rb") as stream:
        yield from iter_jsonl_stream(
            stream, strict=strict, require_object=require_object
        )

iter_jsonl_stream(stream, *, strict=True, require_object=False)

Parse JSON Lines from an open binary stream; see :func:iter_jsonl.

Source code in suthing/jsonl.py
def iter_jsonl_stream(
    stream: IO[bytes], *, strict: bool = True, require_object: bool = False
) -> Iterator[Any]:
    """Parse JSON Lines from an open binary stream; see :func:`iter_jsonl`."""
    for line_no, value, error in _parse(stream, require_object=require_object):
        if error is None:
            yield value
        elif strict:
            raise ValueError(f"line {line_no}: {error}")
        else:
            logger.warning("skipping line %d: %s", line_no, error)

read_jsonl(path, *, require_object=False)

Read a whole JSON Lines file, collecting bad lines instead of raising.

Parameters:

Name Type Description Default
path PathLike

File to read; may be compressed.

required
require_object bool

Treat any value that is not a JSON object as a bad line.

False

Returns:

Type Description
list[Any]

(rows, errors): the values that parsed, in file order, and one

list[JsonlError]
Source code in suthing/jsonl.py
def read_jsonl(
    path: PathLike, *, require_object: bool = False
) -> tuple[list[Any], list[JsonlError]]:
    """Read a whole JSON Lines file, collecting bad lines instead of raising.

    Args:
        path: File to read; may be compressed.
        require_object: Treat any value that is not a JSON object as a bad line.

    Returns:
        ``(rows, errors)``: the values that parsed, in file order, and one
        :class:`JsonlError` per line that did not.
    """
    rows: list[Any] = []
    errors: list[JsonlError] = []
    with open_compressed(path, "rb") as stream:
        for line_no, value, error in _parse(stream, require_object=require_object):
            if error is None:
                rows.append(value)
            else:
                errors.append(JsonlError(line_no, error))
    return rows, errors

write_jsonl(rows, path, *, append=False, atomic=True, mkdir=False)

Write rows as JSON Lines, compressing by suffix.

Parameters:

Name Type Description Default
rows Iterable[Any]

Values to write, one per line; any iterable, consumed once.

required
path PathLike

Destination file.

required
append bool

Add to the end of an existing file instead of replacing it. A compressed file gains a new compressed member, which every reader of that format accepts.

False
atomic bool

Replace path in one step (see :func:suthing.fs.atomic_open). Ignored when appending.

True
mkdir bool

Create missing parent directories first.

False

Returns:

Type Description
int

The number of rows written.

Source code in suthing/jsonl.py
def write_jsonl(
    rows: Iterable[Any],
    path: PathLike,
    *,
    append: bool = False,
    atomic: bool = True,
    mkdir: bool = False,
) -> int:
    """Write *rows* as JSON Lines, compressing by suffix.

    Args:
        rows: Values to write, one per line; any iterable, consumed once.
        path: Destination file.
        append: Add to the end of an existing file instead of replacing it.
            A compressed file gains a new compressed member, which every
            reader of that format accepts.
        atomic: Replace *path* in one step (see :func:`suthing.fs.atomic_open`).
            Ignored when appending.
        mkdir: Create missing parent directories first.

    Returns:
        The number of rows written.
    """
    dest = pathlib.Path(path).expanduser()
    if mkdir:
        dest.parent.mkdir(parents=True, exist_ok=True)
    if append or not atomic:
        with open_compressed(dest, "ab" if append else "wb") as stream:
            return write_jsonl_stream(rows, stream)
    _, compression = split_compression(dest)
    with atomic_open(dest) as raw:
        if compression is None:
            return write_jsonl_stream(rows, raw)
        with wrap_compressed(raw, compression, "wb") as stream:
            return write_jsonl_stream(rows, stream)

write_jsonl_stream(rows, stream)

Write rows to an open binary stream; returns the number written.

Source code in suthing/jsonl.py
def write_jsonl_stream(rows: Iterable[Any], stream: IO[bytes]) -> int:
    """Write *rows* to an open binary stream; returns the number written."""
    if isinstance(rows, (str, bytes, dict)):
        raise TypeError(
            f"JSON Lines needs an iterable of rows, got {type(rows).__name__}"
        )
    count = 0
    for row in rows:
        stream.write(encode_jsonl_row(row))
        count += 1
    return count

File System

File-system primitives: path expansion, transparent compression, atomic writes.

Compression is chosen from the file name: .gz, .bz2 and .xz use the standard library; .zst needs the optional zstandard package (pip install suthing[zstd]).

atomic_open(path, *, mkdir=False, durable=False)

Open a binary stream whose contents replace path only on success.

Data goes to a temporary file in the same directory, which is moved over path with :func:os.replace when the block exits without an exception. Readers therefore see either the old file or the complete new one, and concurrent writers never interleave (the last one to finish wins). On an exception the temporary file is removed and path is left untouched.

The new file keeps the permissions of the file it replaces, or gets the usual 0o666 & ~umask if path did not exist.

Parameters:

Name Type Description Default
path PathLike

Destination file; ~ is expanded.

required
mkdir bool

Create missing parent directories first.

False
durable bool

fsync the data before the rename, so the new contents survive a crash, not just a concurrent reader.

False

Yields:

Type Description
IO[bytes]

A binary file object to write to.

Source code in suthing/fs.py
@contextlib.contextmanager
def atomic_open(
    path: PathLike, *, mkdir: bool = False, durable: bool = False
) -> Iterator[IO[bytes]]:
    """Open a binary stream whose contents replace *path* only on success.

    Data goes to a temporary file in the same directory, which is moved over
    *path* with :func:`os.replace` when the block exits without an exception.
    Readers therefore see either the old file or the complete new one, and
    concurrent writers never interleave (the last one to finish wins). On an
    exception the temporary file is removed and *path* is left untouched.

    The new file keeps the permissions of the file it replaces, or gets the
    usual ``0o666 & ~umask`` if *path* did not exist.

    Args:
        path: Destination file; ``~`` is expanded.
        mkdir: Create missing parent directories first.
        durable: ``fsync`` the data before the rename, so the new contents
            survive a crash, not just a concurrent reader.

    Yields:
        A binary file object to write to.
    """
    dest = pathlib.Path(path).expanduser()
    if mkdir:
        dest.parent.mkdir(parents=True, exist_ok=True)
    try:
        mode = dest.stat().st_mode & 0o7777
    except FileNotFoundError:
        mode = _default_file_mode()
    fd, tmp_name = tempfile.mkstemp(
        dir=dest.parent, prefix=f".{dest.name}.", suffix=".tmp"
    )
    tmp = pathlib.Path(tmp_name)
    try:
        with os.fdopen(fd, "wb") as f:
            yield f
            f.flush()
            if durable:
                os.fsync(f.fileno())
        os.chmod(tmp, mode)
        os.replace(tmp, dest)
    except BaseException:
        tmp.unlink(missing_ok=True)
        raise

atomic_write(path, data, *, encoding='utf-8', mkdir=False, durable=False)

Write data to path atomically (see :func:atomic_open).

The data is written as-is: a .gz name is not compressed. Use :meth:suthing.FileHandle.dump for format- and compression-aware writes.

Parameters:

Name Type Description Default
path PathLike

Destination file.

required
data bytes | str

Bytes, or text encoded with encoding.

required
encoding str

Encoding for text data.

'utf-8'
mkdir bool

Create missing parent directories first.

False
durable bool

fsync before the rename.

False

Returns:

Type Description
Path

The destination path.

Source code in suthing/fs.py
def atomic_write(
    path: PathLike,
    data: bytes | str,
    *,
    encoding: str = "utf-8",
    mkdir: bool = False,
    durable: bool = False,
) -> pathlib.Path:
    """Write *data* to *path* atomically (see :func:`atomic_open`).

    The data is written as-is: a ``.gz`` name is not compressed. Use
    :meth:`suthing.FileHandle.dump` for format- and compression-aware writes.

    Args:
        path: Destination file.
        data: Bytes, or text encoded with *encoding*.
        encoding: Encoding for text data.
        mkdir: Create missing parent directories first.
        durable: ``fsync`` before the rename.

    Returns:
        The destination path.
    """
    payload = data.encode(encoding) if isinstance(data, str) else data
    dest = pathlib.Path(path).expanduser()
    with atomic_open(dest, mkdir=mkdir, durable=durable) as f:
        f.write(payload)
    return dest

expand_path(path)

Expand ~ and make path absolute, resolving symlinks.

Parameters:

Name Type Description Default
path PathLike

Path as a string or path-like object.

required

Returns:

Type Description
Path

The resolved absolute path.

Source code in suthing/fs.py
def expand_path(path: PathLike) -> pathlib.Path:
    """Expand ``~`` and make *path* absolute, resolving symlinks.

    Args:
        path: Path as a string or path-like object.

    Returns:
        The resolved absolute path.
    """
    return pathlib.Path(path).expanduser().resolve()

open_compressed(path, mode='rb')

Open path as a binary stream, decompressing or compressing by suffix.

Parameters:

Name Type Description Default
path PathLike

File to open; ~ is expanded.

required
mode str

One of "rb", "wb", "ab".

'rb'

Yields:

Type Description
IO[bytes]

A binary file object.

Raises:

Type Description
ValueError

If mode is not a binary read/write/append mode.

Source code in suthing/fs.py
@contextlib.contextmanager
def open_compressed(path: PathLike, mode: str = "rb") -> Iterator[IO[bytes]]:
    """Open *path* as a binary stream, decompressing or compressing by suffix.

    Args:
        path: File to open; ``~`` is expanded.
        mode: One of ``"rb"``, ``"wb"``, ``"ab"``.

    Yields:
        A binary file object.

    Raises:
        ValueError: If *mode* is not a binary read/write/append mode.
    """
    if mode not in ("rb", "wb", "ab"):
        raise ValueError(f"mode must be 'rb', 'wb' or 'ab', got {mode!r}")
    p = pathlib.Path(path).expanduser()
    _, compression = split_compression(p)
    with open(p, mode) as raw:
        if compression is None:
            yield raw
        else:
            with wrap_compressed(raw, compression, mode) as stream:
                yield stream

split_compression(path)

Split a file name into its format suffix and compression suffix.

The format suffix is the last extension once any compression suffix is removed. A dotfile such as .env counts as its own suffix.

Parameters:

Name Type Description Default
path PathLike

File path or name.

required

Returns:

Type Description
str

(format_suffix, compression_suffix), lower-cased; format_suffix

str | None

is "" when there is none and compression_suffix is None for

tuple[str, str | None]

an uncompressed name.

Example

split_compression("data/rows.jsonl.gz") ('.jsonl', '.gz') split_compression("config/.env") ('.env', None)

Source code in suthing/fs.py
def split_compression(path: PathLike) -> tuple[str, str | None]:
    """Split a file name into its format suffix and compression suffix.

    The format suffix is the last extension once any compression suffix is
    removed. A dotfile such as ``.env`` counts as its own suffix.

    Args:
        path: File path or name.

    Returns:
        ``(format_suffix, compression_suffix)``, lower-cased; ``format_suffix``
        is ``""`` when there is none and ``compression_suffix`` is ``None`` for
        an uncompressed name.

    Example:
        >>> split_compression("data/rows.jsonl.gz")
        ('.jsonl', '.gz')
        >>> split_compression("config/.env")
        ('.env', None)
    """
    name = pathlib.PurePath(path).name.lower()
    compression = None
    for suffix in COMPRESSORS:
        if name.endswith(suffix) and len(name) > len(suffix):
            compression = suffix
            name = name[: -len(suffix)]
            break
    fmt = pathlib.PurePath(name).suffix
    if not fmt and name.startswith(".") and name.count(".") == 1:
        fmt = name
    return fmt, compression

wrap_compressed(fileobj, compression, mode)

Wrap an open binary file object in a (de)compressing stream.

Parameters:

Name Type Description Default
fileobj IO[bytes]

Underlying binary file object.

required
compression str

A key of :data:COMPRESSORS, e.g. ".gz".

required
mode str

"rb", "wb" or "ab".

required

Returns:

Type Description
IO[bytes]

A binary stream over fileobj; close it before closing fileobj so

IO[bytes]

compressed trailers get flushed.

Source code in suthing/fs.py
def wrap_compressed(fileobj: IO[bytes], compression: str, mode: str) -> IO[bytes]:
    """Wrap an open binary file object in a (de)compressing stream.

    Args:
        fileobj: Underlying binary file object.
        compression: A key of :data:`COMPRESSORS`, e.g. ``".gz"``.
        mode: ``"rb"``, ``"wb"`` or ``"ab"``.

    Returns:
        A binary stream over *fileobj*; close it before closing *fileobj* so
        compressed trailers get flushed.
    """
    return COMPRESSORS[compression](fileobj, mode)

Timing

Wall-clock timing and timestamps.

Timer

Bases: ContextDecorator

Measure the wall-clock time of a block or a function.

Uses :func:time.perf_counter. elapsed is live inside the block and frozen once it exits. On exit the timer reports itself to log if given, otherwise logs at DEBUG level when it has a label.

As a decorator (@Timer("load")) the same instance is re-entered on every call, so it reports the most recent call; it is not safe for recursive or concurrent calls of the decorated function.

Example

with Timer() as t: ... do_work() print(t.elapsed_str) with Timer("ingest", log=print): ... ingest() # prints "ingest: 1.23 sec" on exit

Source code in suthing/timer.py
class Timer(ContextDecorator):
    """Measure the wall-clock time of a block or a function.

    Uses :func:`time.perf_counter`. ``elapsed`` is live inside the block and
    frozen once it exits. On exit the timer reports itself to *log* if given,
    otherwise logs at DEBUG level when it has a *label*.

    As a decorator (``@Timer("load")``) the same instance is re-entered on
    every call, so it reports the most recent call; it is not safe for
    recursive or concurrent calls of the decorated function.

    Example:
        >>> with Timer() as t:
        ...     do_work()
        >>> print(t.elapsed_str)
        >>> with Timer("ingest", log=print):
        ...     ingest()  # prints "ingest: 1.23 sec" on exit
    """

    def __init__(
        self, label: str | None = None, log: Callable[[str], object] | None = None
    ) -> None:
        """Create a timer.

        Args:
            label: Name used when reporting on exit.
            log: Called with ``"<label>: <duration>"`` on exit, e.g.
                ``logger.info`` or ``print``.
        """
        self.label = label
        self.log = log
        self._start: float | None = None
        self._end: float | None = None

    def __enter__(self) -> Self:
        self._start = perf_counter()
        self._end = None
        return self

    def __exit__(self, *exc: object) -> None:
        self._end = perf_counter()
        message = f"{self.label or 'elapsed'}: {self}"
        if self.log is not None:
            self.log(message)
        elif self.label is not None:
            logger.debug(message)

    @property
    def elapsed(self) -> float:
        """Seconds since entering; 0.0 if the timer never started."""
        if self._start is None:
            return 0.0
        end = self._end if self._end is not None else perf_counter()
        return end - self._start

    @property
    def elapsed_ms(self) -> int:
        """:attr:`elapsed` in whole milliseconds."""
        return int(self.elapsed * 1000)

    @property
    def mins(self) -> int:
        """Whole minutes of :attr:`elapsed`. Deprecated: use :attr:`elapsed`."""
        warnings.warn(
            "Timer.mins is deprecated; use Timer.elapsed or Timer.format()",
            DeprecationWarning,
            stacklevel=2,
        )
        return int(self.elapsed // SECONDS_PER_MINUTE)

    @property
    def secs(self) -> int:
        """Whole seconds past :attr:`mins`. Deprecated: use :attr:`elapsed`."""
        warnings.warn(
            "Timer.secs is deprecated; use Timer.elapsed or Timer.format()",
            DeprecationWarning,
            stacklevel=2,
        )
        return int(self.elapsed % SECONDS_PER_MINUTE)

    @property
    def running(self) -> bool:
        """Whether the timer has started and not yet exited."""
        return self._start is not None and self._end is None

    def format(self, digits: int = 2) -> str:
        """:attr:`elapsed` formatted by :func:`format_duration`."""
        return format_duration(self.elapsed, digits)

    @property
    def elapsed_str(self) -> str:
        """:attr:`elapsed` formatted with two decimals, e.g. ``"2 min 7.12 sec"``."""
        return self.format()

    def __str__(self) -> str:
        return self.format()

    def __repr__(self) -> str:
        state = "running" if self.running else f"elapsed={self.elapsed:.6f}"
        return f"Timer(label={self.label!r}, {state})"

elapsed property

Seconds since entering; 0.0 if the timer never started.

elapsed_ms property

:attr:elapsed in whole milliseconds.

elapsed_str property

:attr:elapsed formatted with two decimals, e.g. "2 min 7.12 sec".

mins property

Whole minutes of :attr:elapsed. Deprecated: use :attr:elapsed.

running property

Whether the timer has started and not yet exited.

secs property

Whole seconds past :attr:mins. Deprecated: use :attr:elapsed.

__init__(label=None, log=None)

Create a timer.

Parameters:

Name Type Description Default
label str | None

Name used when reporting on exit.

None
log Callable[[str], object] | None

Called with "<label>: <duration>" on exit, e.g. logger.info or print.

None
Source code in suthing/timer.py
def __init__(
    self, label: str | None = None, log: Callable[[str], object] | None = None
) -> None:
    """Create a timer.

    Args:
        label: Name used when reporting on exit.
        log: Called with ``"<label>: <duration>"`` on exit, e.g.
            ``logger.info`` or ``print``.
    """
    self.label = label
    self.log = log
    self._start: float | None = None
    self._end: float | None = None

format(digits=2)

:attr:elapsed formatted by :func:format_duration.

Source code in suthing/timer.py
def format(self, digits: int = 2) -> str:
    """:attr:`elapsed` formatted by :func:`format_duration`."""
    return format_duration(self.elapsed, digits)

format_duration(seconds, digits=2)

Format a duration as "1 min 30.5 sec" (minutes only when non-zero).

Parameters:

Name Type Description Default
seconds float

Duration in seconds.

required
digits int

Decimal places for the seconds part.

2

Returns:

Type Description
str

The formatted duration.

Source code in suthing/timer.py
def format_duration(seconds: float, digits: int = 2) -> str:
    """Format a duration as ``"1 min 30.5 sec"`` (minutes only when non-zero).

    Args:
        seconds: Duration in seconds.
        digits: Decimal places for the seconds part.

    Returns:
        The formatted duration.
    """
    mins = int(seconds // SECONDS_PER_MINUTE)
    secs = round(seconds - mins * SECONDS_PER_MINUTE, digits)
    text = f"{secs} sec"
    return f"{mins} min {text}" if mins > 0 else text

utc_now_iso(timespec='seconds')

Current UTC time as ISO 8601 text, e.g. "2026-09-22T10:15:00+00:00".

Parameters:

Name Type Description Default
timespec str

Precision, as for :meth:datetime.datetime.isoformat ("seconds", "milliseconds", "microseconds", ...).

'seconds'

Returns:

Type Description
str

The timestamp, with an explicit +00:00 offset.

Source code in suthing/timer.py
def utc_now_iso(timespec: str = "seconds") -> str:
    """Current UTC time as ISO 8601 text, e.g. ``"2026-09-22T10:15:00+00:00"``.

    Args:
        timespec: Precision, as for :meth:`datetime.datetime.isoformat`
            (``"seconds"``, ``"milliseconds"``, ``"microseconds"``, ...).

    Returns:
        The timestamp, with an explicit ``+00:00`` offset.
    """
    return datetime.now(UTC).isoformat(timespec=timespec)

Profiling

Opt-in function profiling.

Decorate functions with :func:profiled; they are timed only while a :class:Profiler is active, and cost one context-variable lookup otherwise::

@profiled(key_args="batch_size")
def ingest(rows, batch_size=100): ...

with Profiler() as prof:
    ingest(rows, batch_size=50)
    ingest(rows, batch_size=500)
prof.summary()  # {"ingest(batch_size=50)": ProfileStats(...), ...}

The active profiler lives in a :class:contextvars.ContextVar, so it follows async tasks. Threads start with no active profiler unless they run in a copied context (contextvars.copy_context().run).

ProfileStats

Bases: NamedTuple

Timing statistics for one profiling key, in seconds.

Source code in suthing/profiling.py
class ProfileStats(NamedTuple):
    """Timing statistics for one profiling key, in seconds."""

    count: int
    total: float
    mean: float
    p50: float
    max: float

Profiler

Collects timings recorded by :func:profiled functions while active.

Source code in suthing/profiling.py
class Profiler:
    """Collects timings recorded by :func:`profiled` functions while active."""

    def __init__(self) -> None:
        self._samples: defaultdict[str, list[float]] = defaultdict(list)
        self._tokens: list[Token[Profiler | None]] = []

    def __enter__(self) -> Self:
        self._tokens.append(_active.set(self))
        return self

    def __exit__(self, *exc: object) -> None:
        _active.reset(self._tokens.pop())

    def record(self, key: str, seconds: float) -> None:
        """Add one timing under *key*."""
        self._samples[key].append(seconds)

    def samples(self) -> dict[str, list[float]]:
        """A copy of every recorded timing, by key."""
        return {k: list(v) for k, v in self._samples.items()}

    def summary(self) -> dict[str, ProfileStats]:
        """Count, total, mean, median and max timing per key."""
        return {
            k: ProfileStats(
                count=len(v),
                total=sum(v),
                mean=statistics.fmean(v),
                p50=statistics.median(v),
                max=max(v),
            )
            for k, v in self._samples.items()
        }

    def reset(self) -> None:
        """Drop all recorded timings."""
        self._samples.clear()

record(key, seconds)

Add one timing under key.

Source code in suthing/profiling.py
def record(self, key: str, seconds: float) -> None:
    """Add one timing under *key*."""
    self._samples[key].append(seconds)

reset()

Drop all recorded timings.

Source code in suthing/profiling.py
def reset(self) -> None:
    """Drop all recorded timings."""
    self._samples.clear()

samples()

A copy of every recorded timing, by key.

Source code in suthing/profiling.py
def samples(self) -> dict[str, list[float]]:
    """A copy of every recorded timing, by key."""
    return {k: list(v) for k, v in self._samples.items()}

summary()

Count, total, mean, median and max timing per key.

Source code in suthing/profiling.py
def summary(self) -> dict[str, ProfileStats]:
    """Count, total, mean, median and max timing per key."""
    return {
        k: ProfileStats(
            count=len(v),
            total=sum(v),
            mean=statistics.fmean(v),
            p50=statistics.median(v),
            max=max(v),
        )
        for k, v in self._samples.items()
    }

active_profiler()

The profiler recording in the current context, if any.

Source code in suthing/profiling.py
def active_profiler() -> Profiler | None:
    """The profiler recording in the current context, if any."""
    return _active.get()

profiled(func=None, /, *, key_args=(), name=None)

profiled(func: Callable[P, R]) -> Callable[P, R]
profiled(*, key_args: str | Sequence[str] = (), name: str | None = None) -> Callable[[Callable[P, R]], Callable[P, R]]

Time calls to a function whenever a :class:Profiler is active.

Usable bare (@profiled) or with options (@profiled(key_args="x")).

Parameters:

Name Type Description Default
func Callable[P, R] | None

Function to wrap (bare use).

None
key_args str | Sequence[str]

Parameter name(s) whose values become part of the key, so calls with different arguments are reported separately ("load(size=100)"). Defaults are filled in.

()
name str | None

Key prefix; defaults to the function's qualified name.

None

Returns:

Type Description
Callable[P, R] | Callable[[Callable[P, R]], Callable[P, R]]

The wrapped function, or a decorator.

Raises:

Type Description
ValueError

At decoration time, if a key_args name is not a parameter of the function.

Source code in suthing/profiling.py
def profiled(
    func: Callable[P, R] | None = None,
    /,
    *,
    key_args: str | Sequence[str] = (),
    name: str | None = None,
) -> Callable[P, R] | Callable[[Callable[P, R]], Callable[P, R]]:
    """Time calls to a function whenever a :class:`Profiler` is active.

    Usable bare (``@profiled``) or with options (``@profiled(key_args="x")``).

    Args:
        func: Function to wrap (bare use).
        key_args: Parameter name(s) whose values become part of the key, so
            calls with different arguments are reported separately
            (``"load(size=100)"``). Defaults are filled in.
        name: Key prefix; defaults to the function's qualified name.

    Returns:
        The wrapped function, or a decorator.

    Raises:
        ValueError: At decoration time, if a *key_args* name is not a
            parameter of the function.
    """
    args_tuple = (key_args,) if isinstance(key_args, str) else tuple(key_args)

    def decorate(f: Callable[P, R]) -> Callable[P, R]:
        label = name or getattr(f, "__qualname__", None) or repr(f)
        key_of = _key_builder(f, label, args_tuple)

        @functools.wraps(f)
        def wrapper(*args: P.args, **kwargs: P.kwargs) -> R:
            prof = _active.get()
            if prof is None:
                return f(*args, **kwargs)
            start = perf_counter()
            try:
                return f(*args, **kwargs)
            finally:
                prof.record(key_of(*args, **kwargs), perf_counter() - start)

        return wrapper

    if func is not None:
        return decorate(func)
    return decorate

Comparison

Deep comparison of nested data, with the location of every difference.

Difference

Bases: NamedTuple

One place where two structures differ.

Attributes:

Name Type Description
path str

Where, e.g. $.users[1].name.

expected Any

The value on the expected side, or :data:MISSING.

actual Any

The value on the actual side, or :data:MISSING.

reason str

Short description of the mismatch.

Source code in suthing/compare.py
class Difference(NamedTuple):
    """One place where two structures differ.

    Attributes:
        path: Where, e.g. ``$.users[1].name``.
        expected: The value on the expected side, or :data:`MISSING`.
        actual: The value on the actual side, or :data:`MISSING`.
        reason: Short description of the mismatch.
    """

    path: str
    expected: Any
    actual: Any
    reason: str

    def __str__(self) -> str:
        return (
            f"{self.path}: {self.reason}"
            f" (expected={self.expected!r}, actual={self.actual!r})"
        )

diff(expected, actual, *, rel_tol=0.0, abs_tol=0.0, ignore_order=False, max_diffs=None)

List every difference between two nested structures.

Mappings are compared key by key; sets as sets; other non-string iterables (lists, tuples, generators, arrays) item by item, so a length mismatch is reported as missing or unexpected items. Everything else is compared with ==. Numbers of different types compare by value, nan equals nan, and bool is not treated as a number.

Parameters:

Name Type Description Default
expected Any

Reference value.

required
actual Any

Value under test.

required
rel_tol float

Relative tolerance for numbers (:func:math.isclose).

0.0
abs_tol float

Absolute tolerance for numbers.

0.0
ignore_order bool

Match sequence items regardless of position (as a multiset). Quadratic in sequence length.

False
max_diffs int | None

Stop after this many differences.

None

Returns:

Type Description
list[Difference]

The differences, empty when the structures match.

Example
for d in diff({"a": [1, 2]}, {"a": [1]}):
    print(d)
# $.a[1]: missing item (expected=2, actual=<missing>)
Source code in suthing/compare.py
def diff(
    expected: Any,
    actual: Any,
    *,
    rel_tol: float = 0.0,
    abs_tol: float = 0.0,
    ignore_order: bool = False,
    max_diffs: int | None = None,
) -> list[Difference]:
    """List every difference between two nested structures.

    Mappings are compared key by key; sets as sets; other non-string iterables
    (lists, tuples, generators, arrays) item by item, so a length mismatch is
    reported as missing or unexpected items. Everything else is compared with
    ``==``. Numbers of different types compare by value, ``nan`` equals
    ``nan``, and ``bool`` is not treated as a number.

    Args:
        expected: Reference value.
        actual: Value under test.
        rel_tol: Relative tolerance for numbers (:func:`math.isclose`).
        abs_tol: Absolute tolerance for numbers.
        ignore_order: Match sequence items regardless of position (as a
            multiset). Quadratic in sequence length.
        max_diffs: Stop after this many differences.

    Returns:
        The differences, empty when the structures match.

    Example:
        ```python
        for d in diff({"a": [1, 2]}, {"a": [1]}):
            print(d)
        # $.a[1]: missing item (expected=2, actual=<missing>)
        ```
    """
    differ = _Differ(rel_tol, abs_tol, ignore_order, max_diffs)
    try:
        differ.walk(expected, actual, "$")
    except _Enough:
        pass
    return differ.found

equals(a, b, *, rel_tol=0.0, abs_tol=0.0, ignore_order=False)

Whether two nested structures match; see :func:diff for the rules.

Stops at the first difference.

Source code in suthing/compare.py
def equals(
    a: Any,
    b: Any,
    *,
    rel_tol: float = 0.0,
    abs_tol: float = 0.0,
    ignore_order: bool = False,
) -> bool:
    """Whether two nested structures match; see :func:`diff` for the rules.

    Stops at the first difference.
    """
    return not diff(
        a,
        b,
        rel_tol=rel_tol,
        abs_tol=abs_tol,
        ignore_order=ignore_order,
        max_diffs=1,
    )

Hashing

Stable content hashes: of JSON-like values, text, bytes, files and directory trees.

Every function takes an algorithm name understood by :func:hashlib.new ("sha256" by default) and returns a hex digest.

bytes_hash(data, *, length=None, algorithm='sha256')

Hex digest of data, optionally truncated to length characters.

Source code in suthing/hashing.py
def bytes_hash(
    data: bytes, *, length: int | None = None, algorithm: str = "sha256"
) -> str:
    """Hex digest of *data*, optionally truncated to *length* characters."""
    return _truncate(hashlib.new(algorithm, data).hexdigest(), length)

canonical_json(obj, *, default=None)

Render obj as canonical JSON: sorted keys, no whitespace, ASCII-escaped.

Equal values give identical text regardless of dict insertion order, which makes the result suitable for hashing and for use as a cache key.

Parameters:

Name Type Description Default
obj Any

JSON-serialisable value.

required
default Callable[[Any], Any] | None

Hook for values json cannot serialise (e.g. :func:suthing.to_jsonable). Without it such values raise TypeError, which keeps an accidental repr out of a hash.

None

Returns:

Type Description
str

The canonical JSON text.

Source code in suthing/hashing.py
def canonical_json(obj: Any, *, default: Callable[[Any], Any] | None = None) -> str:
    """Render *obj* as canonical JSON: sorted keys, no whitespace, ASCII-escaped.

    Equal values give identical text regardless of dict insertion order, which
    makes the result suitable for hashing and for use as a cache key.

    Args:
        obj: JSON-serialisable value.
        default: Hook for values json cannot serialise (e.g.
            :func:`suthing.to_jsonable`). Without it such values raise
            ``TypeError``, which keeps an accidental ``repr`` out of a hash.

    Returns:
        The canonical JSON text.
    """
    return json.dumps(obj, sort_keys=True, separators=(",", ":"), default=default)

file_hash(path, *, length=None, algorithm='sha256')

Hex digest of a file's raw bytes, read in chunks.

Source code in suthing/hashing.py
def file_hash(
    path: PathLike, *, length: int | None = None, algorithm: str = "sha256"
) -> str:
    """Hex digest of a file's raw bytes, read in chunks."""
    h = hashlib.new(algorithm)
    with open(pathlib.Path(path).expanduser(), "rb") as f:
        while chunk := f.read(_CHUNK):
            h.update(chunk)
    return _truncate(h.hexdigest(), length)

stable_hash(obj, *, length=None, algorithm='sha256', default=None)

Hex digest of :func:canonical_json of obj.

Equal to hashlib.sha256(json.dumps(obj, sort_keys=True, separators=(",", ":")).encode("utf-8")).hexdigest(), so it can replace that expression without changing any stored hash.

Parameters:

Name Type Description Default
obj Any

JSON-serialisable value.

required
length int | None

Keep only the first length hex characters.

None
algorithm str

:mod:hashlib algorithm name.

'sha256'
default Callable[[Any], Any] | None

Passed to :func:canonical_json.

None

Returns:

Type Description
str

The hex digest.

Source code in suthing/hashing.py
def stable_hash(
    obj: Any,
    *,
    length: int | None = None,
    algorithm: str = "sha256",
    default: Callable[[Any], Any] | None = None,
) -> str:
    """Hex digest of :func:`canonical_json` of *obj*.

    Equal to ``hashlib.sha256(json.dumps(obj, sort_keys=True,
    separators=(",", ":")).encode("utf-8")).hexdigest()``, so it can replace
    that expression without changing any stored hash.

    Args:
        obj: JSON-serialisable value.
        length: Keep only the first *length* hex characters.
        algorithm: :mod:`hashlib` algorithm name.
        default: Passed to :func:`canonical_json`.

    Returns:
        The hex digest.
    """
    return text_hash(
        canonical_json(obj, default=default), length=length, algorithm=algorithm
    )

text_hash(text, *, length=None, algorithm='sha256', encoding='utf-8')

Hex digest of text encoded with encoding, optionally truncated.

Source code in suthing/hashing.py
def text_hash(
    text: str,
    *,
    length: int | None = None,
    algorithm: str = "sha256",
    encoding: str = "utf-8",
) -> str:
    """Hex digest of *text* encoded with *encoding*, optionally truncated."""
    return bytes_hash(text.encode(encoding), length=length, algorithm=algorithm)

tree_hash(root, *, pattern='*', length=None, algorithm='sha256')

Hex digest of every file under root and its path relative to root.

Files are visited in sorted order of their POSIX relative paths, so the result does not depend on the file system's listing order. Renaming or moving a file changes the hash; so does changing its contents. Directories and symlinks to directories contribute only through the files they hold.

Parameters:

Name Type Description Default
root PathLike

Directory to hash.

required
pattern str

Glob matched recursively against file names (rglob).

'*'
length int | None

Keep only the first length hex characters.

None
algorithm str

:mod:hashlib algorithm name.

'sha256'

Returns:

Type Description
str

The hex digest.

Raises:

Type Description
NotADirectoryError

If root is not a directory.

Source code in suthing/hashing.py
def tree_hash(
    root: PathLike,
    *,
    pattern: str = "*",
    length: int | None = None,
    algorithm: str = "sha256",
) -> str:
    """Hex digest of every file under *root* and its path relative to *root*.

    Files are visited in sorted order of their POSIX relative paths, so the
    result does not depend on the file system's listing order. Renaming or
    moving a file changes the hash; so does changing its contents. Directories
    and symlinks to directories contribute only through the files they hold.

    Args:
        root: Directory to hash.
        pattern: Glob matched recursively against file names (``rglob``).
        length: Keep only the first *length* hex characters.
        algorithm: :mod:`hashlib` algorithm name.

    Returns:
        The hex digest.

    Raises:
        NotADirectoryError: If *root* is not a directory.
    """
    base = pathlib.Path(root).expanduser()
    if not base.is_dir():
        raise NotADirectoryError(str(base))
    files = sorted(
        (p.relative_to(base).as_posix(), p) for p in base.rglob(pattern) if p.is_file()
    )
    h = hashlib.new(algorithm)
    for rel, p in files:
        h.update(rel.encode("utf-8"))
        h.update(b"\0")
        h.update(bytes.fromhex(file_hash(p, algorithm=algorithm)))
    return _truncate(h.hexdigest(), length)

JSON Conversion

Coerce Python values into what :func:json.dumps accepts.

to_jsonable(obj, *, nan=None)

Recursively convert obj into JSON-serialisable built-in types.

Conversions:

  • dict/mapping → dict with str keys; list, tuple, set and frozenset → list (sets are sorted when they can be, so the output is deterministic)
  • Enum → its value; subclasses of str/int/float → the plain built-in
  • datetime/date/time → ISO 8601 text; timedelta → seconds
  • Decimal → float; UUID and paths → str
  • dataclass instances → dict of their fields
  • numpy scalars and arrays → Python scalars and nested lists (numpy is only consulted when it is already imported)
  • non-finite floats (nan, inf) → nan, since JSON has no such value

Can also be passed as default= to :func:json.dumps, where it is called only for values json cannot serialise itself.

Parameters:

Name Type Description Default
obj Any

Value to convert.

required
nan Any

Replacement for non-finite floats.

None

Returns:

Type Description
Any

A structure of dict, list, str, int, float,

Any

bool and None.

Raises:

Type Description
TypeError

For a value with no known conversion.

Source code in suthing/jsonable.py
def to_jsonable(obj: Any, *, nan: Any = None) -> Any:
    """Recursively convert *obj* into JSON-serialisable built-in types.

    Conversions:

    - ``dict``/mapping → ``dict`` with ``str`` keys; ``list``, ``tuple``,
      ``set`` and ``frozenset`` → ``list`` (sets are sorted when they can be,
      so the output is deterministic)
    - ``Enum`` → its ``value``; subclasses of ``str``/``int``/``float`` → the
      plain built-in
    - ``datetime``/``date``/``time`` → ISO 8601 text; ``timedelta`` → seconds
    - ``Decimal`` → ``float``; ``UUID`` and paths → ``str``
    - dataclass instances → ``dict`` of their fields
    - numpy scalars and arrays → Python scalars and nested lists (numpy is
      only consulted when it is already imported)
    - non-finite floats (``nan``, ``inf``) → *nan*, since JSON has no such value

    Can also be passed as ``default=`` to :func:`json.dumps`, where it is called
    only for values json cannot serialise itself.

    Args:
        obj: Value to convert.
        nan: Replacement for non-finite floats.

    Returns:
        A structure of ``dict``, ``list``, ``str``, ``int``, ``float``,
        ``bool`` and ``None``.

    Raises:
        TypeError: For a value with no known conversion.
    """
    if obj is None or isinstance(obj, bool):
        return obj
    if isinstance(obj, Enum):
        return to_jsonable(obj.value, nan=nan)
    if isinstance(obj, str):
        return str(obj)
    if isinstance(obj, int):
        return int(obj)
    if isinstance(obj, float):
        return float(obj) if math.isfinite(obj) else nan
    if isinstance(obj, Mapping):
        return {_key(k): to_jsonable(v, nan=nan) for k, v in obj.items()}
    if isinstance(obj, (list, tuple)):
        return [to_jsonable(v, nan=nan) for v in obj]
    if isinstance(obj, (set, frozenset)):
        items = [to_jsonable(v, nan=nan) for v in obj]
        try:
            return sorted(items)
        except TypeError:
            return items
    if isinstance(obj, (datetime, date, time)):
        return obj.isoformat()
    if isinstance(obj, timedelta):
        return obj.total_seconds()
    if isinstance(obj, Decimal):
        return to_jsonable(float(obj), nan=nan)
    if isinstance(obj, (UUID, pathlib.PurePath)):
        return str(obj)
    if dataclasses.is_dataclass(obj) and not isinstance(obj, type):
        return {
            f.name: to_jsonable(getattr(obj, f.name), nan=nan)
            for f in dataclasses.fields(obj)
        }
    np = sys.modules.get("numpy")
    if np is not None:
        if isinstance(obj, np.ndarray):
            return to_jsonable(obj.tolist(), nan=nan)
        if isinstance(obj, np.generic):
            return to_jsonable(obj.item(), nan=nan)
    raise TypeError(f"Object of type {type(obj).__name__} is not JSON serializable")

Iteration

Iteration helpers.

batched(iterable, n)

Split iterable into consecutive lists of n items; the last may be shorter.

Works on any iterable, including generators, and consumes it lazily. Unlike :func:itertools.batched (Python 3.12+) it yields lists, which callers can index, extend or pass straight to bulk-insert APIs.

Parameters:

Name Type Description Default
iterable Iterable[T]

Items to split.

required
n int

Batch size.

required

Yields:

Type Description
list[T]

Lists of at most n items.

Raises:

Type Description
ValueError

If n is less than 1.

Example

list(batched(range(5), 2)) [[0, 1], [2, 3], [4]]

Source code in suthing/iterx.py
def batched(iterable: Iterable[T], n: int) -> Iterator[list[T]]:
    """Split *iterable* into consecutive lists of *n* items; the last may be shorter.

    Works on any iterable, including generators, and consumes it lazily.
    Unlike :func:`itertools.batched` (Python 3.12+) it yields lists, which
    callers can index, extend or pass straight to bulk-insert APIs.

    Args:
        iterable: Items to split.
        n: Batch size.

    Yields:
        Lists of at most *n* items.

    Raises:
        ValueError: If *n* is less than 1.

    Example:
        >>> list(batched(range(5), 2))
        [[0, 1], [2, 3], [4]]
    """
    if n < 1:
        raise ValueError(f"batch size must be at least 1, got {n}")
    it = iter(iterable)
    while batch := list(islice(it, n)):
        yield batch

Text

Text helpers.

slugify(text, *, sep='-', fallback='item', lower=False, ascii_fold=False, max_length=None)

Turn text into a token safe for file names and URL path segments.

Every run of characters outside A-Za-z0-9._- becomes one sep, and sep is stripped from both ends.

Parameters:

Name Type Description Default
text str

Input text.

required
sep str

Replacement for each run of unsafe characters.

'-'
fallback str

Returned when nothing safe is left.

'item'
lower bool

Lower-case the result.

False
ascii_fold bool

Replace accented letters by their base letter ("é" → "e") instead of treating them as unsafe.

False
max_length int | None

Cut the result to at most this many characters, then strip sep again.

None

Returns:

Type Description
str

The slug, or fallback.

Example

slugify(" Person / Company ") 'Person-Company' slugify("Café Menü", sep="_", lower=True, ascii_fold=True) 'cafe_menu'

Source code in suthing/text.py
def slugify(
    text: str,
    *,
    sep: str = "-",
    fallback: str = "item",
    lower: bool = False,
    ascii_fold: bool = False,
    max_length: int | None = None,
) -> str:
    """Turn *text* into a token safe for file names and URL path segments.

    Every run of characters outside ``A-Za-z0-9._-`` becomes one *sep*, and
    *sep* is stripped from both ends.

    Args:
        text: Input text.
        sep: Replacement for each run of unsafe characters.
        fallback: Returned when nothing safe is left.
        lower: Lower-case the result.
        ascii_fold: Replace accented letters by their base letter (``"é"`` →
            ``"e"``) instead of treating them as unsafe.
        max_length: Cut the result to at most this many characters, then strip
            *sep* again.

    Returns:
        The slug, or *fallback*.

    Example:
        >>> slugify("  Person / Company ")
        'Person-Company'
        >>> slugify("Café Menü", sep="_", lower=True, ascii_fold=True)
        'cafe_menu'
    """
    if ascii_fold:
        text = (
            unicodedata.normalize("NFKD", text)
            .encode("ascii", "ignore")
            .decode("ascii")
        )
    slug = _UNSAFE.sub(sep, text.strip()).strip(sep)
    if lower:
        slug = slug.lower()
    if max_length is not None:
        slug = slug[:max_length].strip(sep)
    return slug or fallback

Environment

Environment variables: boolean flags and .env files.

env_flag(name, default=False)

Read environment variable name as a boolean.

1/true/t/yes/y/on are true and 0/false/f/no/n/off are false, ignoring case and surrounding whitespace. Unset or empty gives default.

Parameters:

Name Type Description Default
name str

Variable name.

required
default bool

Value when the variable is unset or empty.

False

Returns:

Type Description
bool

The flag value.

Raises:

Type Description
ValueError

For any other value, so a typo does not silently read as default.

Source code in suthing/environ.py
def env_flag(name: str, default: bool = False) -> bool:
    """Read environment variable *name* as a boolean.

    ``1/true/t/yes/y/on`` are true and ``0/false/f/no/n/off`` are false,
    ignoring case and surrounding whitespace. Unset or empty gives *default*.

    Args:
        name: Variable name.
        default: Value when the variable is unset or empty.

    Returns:
        The flag value.

    Raises:
        ValueError: For any other value, so a typo does not silently read as
            *default*.
    """
    raw = os.environ.get(name, "").strip().lower()
    if not raw:
        return default
    if raw in _TRUE:
        return True
    if raw in _FALSE:
        return False
    raise ValueError(f"environment variable {name}={raw!r} is not a boolean")

load_env(path, *, override=False)

Load a dotenv file into :data:os.environ.

To read a dotenv file without touching the environment, use FileHandle.load(path), which returns its values as a dict.

Parameters:

Name Type Description Default
path PathLike

Dotenv file.

required
override bool

Replace variables that are already set. By default the existing environment wins.

False

Returns:

Type Description
dict[str, str]

The variables the file defines (with values None in the file

dict[str, str]

dropped).

Source code in suthing/environ.py
def load_env(path: PathLike, *, override: bool = False) -> dict[str, str]:
    """Load a dotenv file into :data:`os.environ`.

    To read a dotenv file *without* touching the environment, use
    ``FileHandle.load(path)``, which returns its values as a dict.

    Args:
        path: Dotenv file.
        override: Replace variables that are already set. By default the
            existing environment wins.

    Returns:
        The variables the file defines (with values ``None`` in the file
        dropped).
    """
    from dotenv import dotenv_values, load_dotenv

    p = pathlib.Path(path).expanduser()
    if not p.is_file():
        raise FileNotFoundError(str(p))
    load_dotenv(p, override=override)
    return {k: v for k, v in dotenv_values(p).items() if v is not None}

Logging

One-call logging configuration for scripts and CLIs.

setup_logging(level='INFO', *, config=None, fmt=DEFAULT_FORMAT, stream=None, force=False)

Configure the root logger, from a config file or with sensible defaults.

With config, the file decides everything: .conf/.ini files go to :func:logging.config.fileConfig and .yaml/.yml/.json files to :func:logging.config.dictConfig. Loggers that already exist are kept enabled in both cases. Without config, :func:logging.basicConfig is called with level, fmt and stream.

Parameters:

Name Type Description Default
level int | str

Root level, as a number or a name such as "DEBUG".

'INFO'
config PathLike | None

Optional logging config file.

None
fmt str

Format string when no config is given.

DEFAULT_FORMAT
stream TextIO | None

Output stream when no config is given (default stderr).

None
force bool

Replace handlers already attached to the root logger.

False

Raises:

Type Description
FileNotFoundError

If config does not exist.

ValueError

If config has an unsupported extension.

Source code in suthing/log.py
def setup_logging(
    level: int | str = "INFO",
    *,
    config: PathLike | None = None,
    fmt: str = DEFAULT_FORMAT,
    stream: TextIO | None = None,
    force: bool = False,
) -> None:
    """Configure the root logger, from a config file or with sensible defaults.

    With *config*, the file decides everything: ``.conf``/``.ini`` files go to
    :func:`logging.config.fileConfig` and ``.yaml``/``.yml``/``.json`` files
    to :func:`logging.config.dictConfig`. Loggers that already exist are kept
    enabled in both cases. Without *config*, :func:`logging.basicConfig` is
    called with *level*, *fmt* and *stream*.

    Args:
        level: Root level, as a number or a name such as ``"DEBUG"``.
        config: Optional logging config file.
        fmt: Format string when no *config* is given.
        stream: Output stream when no *config* is given (default ``stderr``).
        force: Replace handlers already attached to the root logger.

    Raises:
        FileNotFoundError: If *config* does not exist.
        ValueError: If *config* has an unsupported extension.
    """
    if config is None:
        logging.basicConfig(
            level=level, format=fmt, stream=stream or sys.stderr, force=force
        )
        return

    path = pathlib.Path(config).expanduser()
    if not path.is_file():
        raise FileNotFoundError(str(path))
    suffix = path.suffix.lower()
    if force:
        root = logging.getLogger()
        for handler in root.handlers[:]:
            root.removeHandler(handler)
            handler.close()
    if suffix in (".conf", ".ini", ".cfg"):
        logging.config.fileConfig(path, disable_existing_loggers=False)
    elif suffix in (".yaml", ".yml", ".json"):
        from suthing.file_handle import FileHandle

        spec = FileHandle.load(path)
        if not isinstance(spec, dict):
            raise ValueError(f"{path} does not hold a logging config mapping")
        spec.setdefault("version", 1)
        spec.setdefault("disable_existing_loggers", False)
        logging.config.dictConfig(spec)
    else:
        raise ValueError(
            f"unsupported logging config {path.name!r}:"
            " expected .conf/.ini/.cfg or .yaml/.yml/.json"
        )