"""The Table class: create, write, and read a H5Col column-oriented table.
A table is an HDF5 group carrying ``CLASS="COLUMN_TABLE"``. This module builds
such groups from :class:`~h5col.specs.TableSpec` / :class:`~h5col.specs.ColumnSpec`,
appends rows following the H5Col write protocol (extend every column, write,
then commit ``NROWS`` last), reads them back, and validates conformance.
"""
from __future__ import annotations
from collections.abc import Iterator, Mapping, Sequence
from typing import Any
import h5py
import numpy as np
from . import arrow, categorical, indexes, info, lists, query, references
from ._hdf5 import (
create_column_dataset,
extend_to,
prepare_column_data,
read_str_array_attr,
read_str_attr,
read_uint64_attr,
substitute_fill_for_none,
write_ascii_token_attr,
write_extra_attributes,
write_uint64_attr,
write_utf8_array_attr,
write_utf8_attr,
)
from .booleans import bool_dtype, is_bool_dtype
from .column import Column
from .exceptions import ConformanceError, SchemaError, VersionError
from .info import TableInfo
from .listcolumn import ListColumn
from .missing import recommended_fill, validate_fill_outside_range
from .reserved import (
ATTR_CATEGORIES,
ATTR_CLASS,
ATTR_COLUMN_ORDER,
ATTR_DESCRIPTION,
ATTR_ENCODING_TYPE,
ATTR_ENCODING_VERSION,
ATTR_GENERATION,
ATTR_INDEX,
ATTR_INDEX_COLUMNS,
ATTR_NROWS,
ATTR_TITLE,
ATTR_UNITS,
ATTR_UNITS_VOCABULARY,
ATTR_VALID_MAX,
ATTR_VALID_MIN,
ATTR_VERSION,
CLASS_COLUMN_TABLE,
CLASS_LIST_COLUMN,
GROUP_CATEGORIES,
GROUP_SEARCH_INDEXES,
KIND_BITMAP,
KIND_CHUNK_MINMAX,
KIND_SORTED_ROWS,
MEMBER_MASK,
validate_attribute_names,
validate_column_name,
)
from .searchindex import SearchIndex, wrap_index
from .specs import ColumnSpec, ListColumnSpec, TableSpec
from .strings import FixedString
#: HEP001 revision this implementation writes and the highest major it reads.
VERSION = "1.0"
SUPPORTED_MAJOR = 1
_SKIP_CHILDREN = frozenset({GROUP_CATEGORIES, GROUP_SEARCH_INDEXES})
[docs]
class Table:
"""A H5Col column-oriented table backed by an HDF5 group."""
def __init__(self, group: Any) -> None:
self._group = group
def __repr__(self) -> str:
try:
return f"<h5col.Table {self._group.name!r} nrows={self.nrows}>"
except Exception:
return "<h5col.Table (closed or invalid)>"
# -- construction ------------------------------------------------------- #
[docs]
@staticmethod
def is_table_group(group: Any) -> bool:
"""True if *group* is a H5Col table group (lenient CLASS check).
Parameters
----------
group:
Any h5py group. One that is not a table group answers False rather
than raising, so this is safe to use while walking a file.
"""
return read_str_attr(group, ATTR_CLASS) == CLASS_COLUMN_TABLE
[docs]
@classmethod
def open(cls, group: Any) -> Table:
"""Open an existing table group, checking its CLASS and VERSION major.
Parameters
----------
group:
An h5py group already written as a H5Col table. Opening does not
read any column data.
Raises
------
ConformanceError
If *group* is not a H5Col table group, or its ``VERSION`` is missing
or unparsable.
VersionError
If the table's ``VERSION`` major exceeds the supported major.
"""
if not cls.is_table_group(group):
raise ConformanceError(
f"{group.name!r} is not a H5Col table group "
f"(missing or wrong {ATTR_CLASS} attribute)"
)
table = cls(group)
table._check_version()
return table
[docs]
@classmethod
def create(
cls,
group: Any,
columns: TableSpec | Sequence[ColumnSpec | ListColumnSpec],
*,
title: str | None = None,
description: str | None = None,
index_columns: Sequence[str] | None = None,
column_order: Sequence[str] | None = None,
units_vocabulary: str | None = None,
encoding_type: str | None = None,
encoding_version: str | None = None,
default_chunk_bytes: int | None = None,
) -> Table:
"""Create a new, empty table (``NROWS = 0``) with the given columns.
Parameters
----------
group:
An h5py group to write the table into. It must not already be a
H5Col table group.
columns:
A :class:`~h5col.TableSpec`, or a sequence of
:class:`~h5col.ColumnSpec` and :class:`~h5col.ListColumnSpec`.
title:
Human-readable table title, stored as the ``TITLE`` attribute.
description:
Longer free text, stored as ``DESCRIPTION``.
index_columns:
Names of the columns that together identify a row, stored as
``INDEX_COLUMNS``. Every name must be a declared scalar column; a
list column cannot be an index column.
column_order:
The columns' logical order, stored as ``COLUMN_ORDER``. Defaults to
the order given in *columns*, which HDF5 itself does not preserve.
units_vocabulary:
The vocabulary the columns' ``units`` come from, e.g. ``UDUNITS-2``.
encoding_type:
Producer-defined encoding identifier for the table as a whole.
encoding_version:
Version string paired with *encoding_type*.
default_chunk_bytes:
Overrides the automatic (chunk-cache-scaled) target chunk size for
columns that do not set an explicit ``chunks`` shape.
Raises
------
SchemaError
If *group* is already a H5Col table group, a column spec is invalid,
or an ``index_columns`` name is not among the declared columns.
ReservedNameError
If a column name is a H5Col reserved name.
FillValueError
If a column's fill value lies inside its declared valid range.
"""
if cls.is_table_group(group):
raise SchemaError(f"{group.name!r} is already a H5Col table group")
if isinstance(columns, TableSpec):
spec = columns
else:
spec = TableSpec(
columns=list(columns),
title=title,
description=description,
index_columns=list(index_columns) if index_columns else [],
column_order=list(column_order) if column_order else None,
units_vocabulary=units_vocabulary,
encoding_type=encoding_type,
encoding_version=encoding_version,
)
# Pre-flight: validate every column BEFORE writing the CLASS identifier,
# so an invalid spec never leaves a group marked as a (broken) table
# group (H5Col forbids CLASS="COLUMN_TABLE" on a non-conformant group).
for col in spec.columns:
cls._validate_column_spec(col)
declared = {c.name for c in spec.columns}
for ic in spec.index_columns:
if ic not in declared:
raise SchemaError(f"index column {ic!r} is not a declared column")
# Mutate the group, rolling everything back on failure so a partially
# built table is never left behind.
written: list[str] = []
created: list[str] = []
cat_existed = GROUP_CATEGORIES in group
try:
write_ascii_token_attr(group, ATTR_CLASS, CLASS_COLUMN_TABLE)
written.append(ATTR_CLASS)
write_ascii_token_attr(group, ATTR_VERSION, VERSION)
written.append(ATTR_VERSION)
write_uint64_attr(group, ATTR_NROWS, 0)
written.append(ATTR_NROWS)
for attr, value in (
(ATTR_TITLE, spec.title),
(ATTR_DESCRIPTION, spec.description),
(ATTR_UNITS_VOCABULARY, spec.units_vocabulary),
(ATTR_ENCODING_TYPE, spec.encoding_type),
(ATTR_ENCODING_VERSION, spec.encoding_version),
):
if value is not None:
write_utf8_attr(group, attr, value)
written.append(attr)
for col in spec.columns:
cls._create_one_column(
group, col, default_chunk_bytes=default_chunk_bytes
)
created.append(col.name)
if spec.columns:
write_utf8_array_attr(group, ATTR_COLUMN_ORDER, spec.ordered_names)
written.append(ATTR_COLUMN_ORDER)
if spec.index_columns:
references.write_ref_array_attr(
group, ATTR_INDEX_COLUMNS, [group[n] for n in spec.index_columns]
)
written.append(ATTR_INDEX_COLUMNS)
write_utf8_attr(group, ATTR_INDEX, spec.index_columns[0])
written.append(ATTR_INDEX)
except BaseException:
for n in created:
if n in group:
del group[n]
for a in written:
if a in group.attrs:
del group.attrs[a]
if not cat_existed and GROUP_CATEGORIES in group:
del group[GROUP_CATEGORIES]
raise
return cls(group)
@staticmethod
def _validate_column_spec(col: ColumnSpec | ListColumnSpec) -> None:
"""Validate one column spec without touching the file.
Parameters
----------
col:
The column spec to check, left unmodified. Its extra attribute
names are checked first either way; a
:class:`~h5col.specs.ListColumnSpec` is then handed to
:func:`~h5col.lists.validate_list_column_spec` and nothing below
applies to it. A :class:`~h5col.specs.ColumnSpec` is checked here
for a legal column name and for a fill value consistent with the
column's kind: categorical, boolean, or plain.
"""
validate_attribute_names(col.attributes, col.name)
if isinstance(col, ListColumnSpec):
lists.validate_list_column_spec(col)
return
validate_column_name(col.name)
if col.is_categorical:
assert col.categories is not None
ncats = len(col.categories)
if len(set(col.categories)) != len(col.categories):
raise SchemaError(
f"categorical column {col.name!r} has duplicate categories"
)
dtype = col.resolved_dtype()
if ncats > 0 and int(np.iinfo(dtype).max) < ncats - 1:
raise SchemaError(
f"categorical column {col.name!r} code dtype {dtype} cannot "
f"index {ncats} categories"
)
fill = (
col.fill_value
if col.fill_value is not None
else categorical.default_categorical_fill(dtype)
)
try:
cast_fill = int(np.asarray(fill, dtype=dtype))
except (OverflowError, ValueError) as exc:
raise SchemaError(
f"categorical column {col.name!r} fill {fill!r} is not "
f"representable in code dtype {dtype}"
) from exc
if 0 <= cast_fill < ncats:
raise SchemaError(
f"categorical column {col.name!r} fill {cast_fill} collides "
f"with a valid code [0, {ncats})"
)
validate_fill_outside_range(cast_fill, col.valid_min, col.valid_max)
return
Table._checked_fill(col)
@staticmethod
def _checked_fill(col: ColumnSpec) -> Any:
"""The fill value to create ``col`` with, or ``None`` where it declares none.
The single place that decides both what a legal fill value is and which
value a column ends up with. Creation and validation each of
those answers need run at different moments.
Categorical columns do not come here: they carry their own fill rule,
checked in :meth:`_validate_column_spec` and applied in
:meth:`_create_categorical_column`.
Parameters
----------
col:
A non-categorical column spec, left unmodified.
Raises
------
SchemaError
If a boolean column declares a fill value or a valid range. H5Col
leaves a boolean no value outside its own two, so either is a
contradiction rather than a preference.
FillValueError
If the fill value lies inside the column's declared valid range.
"""
if col.is_boolean:
if col.fill_value is not None:
raise SchemaError(
f"boolean column {col.name!r} must not declare a fill value"
)
if col.valid_min is not None or col.valid_max is not None:
raise SchemaError(
f"boolean column {col.name!r} must not declare valid_min/valid_max"
)
return None
fill = (
col.fill_value
if col.fill_value is not None
else recommended_fill(col.resolved_dtype())
)
validate_fill_outside_range(fill, col.valid_min, col.valid_max)
return fill
@staticmethod
def _create_one_column(
group: Any,
col: ColumnSpec | ListColumnSpec,
default_chunk_bytes: int | None = None,
) -> Any:
"""Create one column from its spec and return the new HDF5 object.
Parameters
----------
group:
The table group to create the column in. It is only passed on to
the creation helpers; nothing is read from it here.
col:
The column's spec. A :class:`~h5col.specs.ListColumnSpec` is handed
to :func:`~h5col.lists.create_list_column` and none of the scalar
handling below applies to it. A categorical
:class:`~h5col.specs.ColumnSpec` goes to
:meth:`_create_categorical_column`. Any other one supplies the
dtype, chunking, filters, fill value, and the optional attributes
written on the dataset.
default_chunk_bytes:
A byte target for one chunk, used only when ``col.chunks`` is None.
"""
if isinstance(col, ListColumnSpec):
return lists.create_list_column(
group, col, default_chunk_bytes=default_chunk_bytes
)
name = validate_column_name(col.name)
dtype = col.resolved_dtype()
if col.is_categorical:
return Table._create_categorical_column(
group, col, name, dtype, default_chunk_bytes=default_chunk_bytes
)
fill = Table._checked_fill(col)
ds = create_column_dataset(
group,
name,
dtype,
chunks=col.chunks,
fill_value=fill,
filters=col.filters,
default_chunk_bytes=default_chunk_bytes,
)
if col.valid_min is not None:
ds.attrs.create(ATTR_VALID_MIN, np.asarray(col.valid_min, dtype=dtype))
if col.valid_max is not None:
ds.attrs.create(ATTR_VALID_MAX, np.asarray(col.valid_max, dtype=dtype))
if col.units is not None:
write_utf8_attr(ds, ATTR_UNITS, col.units)
if col.units_vocabulary is not None:
write_utf8_attr(ds, ATTR_UNITS_VOCABULARY, col.units_vocabulary)
if col.description is not None:
write_utf8_attr(ds, ATTR_DESCRIPTION, col.description)
write_extra_attributes(ds, col.attributes)
return ds
@staticmethod
def _create_categorical_column(
group: Any,
col: ColumnSpec,
name: str,
dtype: np.dtype,
default_chunk_bytes: int | None = None,
) -> Any:
"""Create a categorical column and its companion categories dataset.
Parameters
----------
group:
The table group to create the column dataset in. Its
``CATEGORIES`` child group is required (created if absent) to hold
the labels.
col:
The categorical column's spec. ``col.categories`` must not be
None, which the caller has already ensured. Supplies the fill
value, chunking, filters, and the optional attributes written on
the dataset.
name:
The dataset's link name within *group*, already validated by the
caller. The categories dataset is named ``f"{name}__CATEGORIES"``.
dtype:
The integer code dtype. The fill value and any ``valid_min`` /
``valid_max`` are cast to it.
default_chunk_bytes:
An explicit byte target for one chunk, used only when
``col.chunks`` is None.
"""
assert col.categories is not None
# Fill range, dtype fit, and representability are enforced by
# _validate_column_spec, which always runs before creation.
fill = (
col.fill_value
if col.fill_value is not None
else categorical.default_categorical_fill(dtype)
)
ds = create_column_dataset(
group,
name,
dtype,
chunks=col.chunks,
fill_value=np.asarray(fill, dtype=dtype),
filters=col.filters,
default_chunk_bytes=default_chunk_bytes,
)
cat_group = group.require_group(GROUP_CATEGORIES)
cat_ds = categorical.create_categories_dataset(
cat_group, f"{name}__CATEGORIES", col.categories, col.ordered
)
references.write_ref_attr(ds, ATTR_CATEGORIES, cat_ds)
if col.valid_min is not None:
ds.attrs.create(ATTR_VALID_MIN, np.asarray(col.valid_min, dtype=dtype))
if col.valid_max is not None:
ds.attrs.create(ATTR_VALID_MAX, np.asarray(col.valid_max, dtype=dtype))
if col.units is not None:
write_utf8_attr(ds, ATTR_UNITS, col.units)
if col.units_vocabulary is not None:
write_utf8_attr(ds, ATTR_UNITS_VOCABULARY, col.units_vocabulary)
if col.description is not None:
write_utf8_attr(ds, ATTR_DESCRIPTION, col.description)
return ds
[docs]
@classmethod
def from_arrays(
cls,
group: Any,
arrays: Mapping[str, Any],
*,
specs: Sequence[ColumnSpec] | None = None,
**table_kwargs: Any,
) -> Table:
"""Create a table from column arrays and write them in one call.
Parameters
----------
group:
An h5py group to write the table into, as for :meth:`create`.
arrays:
One array per column, keyed by column name. Every array must hold
the same number of rows, and the column order follows this mapping.
specs:
Column specs to create the table with. When omitted, one is
inferred per array — boolean, a fixed-length string sized to the
longest value, or the array's own dtype.
table_kwargs:
Passed through to :meth:`create`, so ``title``, ``index_columns``
and the rest are available here too.
"""
if specs is None:
specs = [_infer_column_spec(name, arr) for name, arr in arrays.items()]
table = cls.create(group, specs, **table_kwargs)
table.append(arrays)
return table
[docs]
@classmethod
def from_arrow(
cls,
group: Any,
table: Any,
*,
specs: Sequence[ColumnSpec | ListColumnSpec] | None = None,
**table_kwargs: Any,
) -> Table:
"""Create a table from a :class:`pyarrow.Table` and write its rows.
Arrow's model is wider than H5Col's, so anything without an exact
equivalent is refused rather than approximated, see
:func:`~h5col.specs_from_arrow`, which decides the mapping and which you
can call first to inspect or adjust it::
specs = h5col.specs_from_arrow(tbl)
specs[2].chunks = 8192
Table.from_arrow(group, tbl, specs=specs)
Rows are written batch by batch rather than all at once, so importing a
large table costs about one batch of memory rather than the whole of it.
Two chunks of one dictionary column may carry different dictionaries,
in which case the same code stands for two different labels. The codes
are never read directly for that reason — the labels are, against the
unified category set :func:`~h5col.specs_from_arrow` derives.
The fill-value checks run whichever way the specs arrived: a fill that
already occurs in a column would leave those rows reading as missing, so
supplying specs cannot skip it.
Needs the optional ``pyarrow`` dependency (``pip install h5col[arrow]``).
.. versionadded:: 0.4.0
Parameters
----------
group:
An h5py group to write the table into, as for :meth:`create`.
table:
The :class:`pyarrow.Table` to import.
specs:
A complete list of column specs naming exactly the table's columns.
None infers them. Chunking and filters have no Arrow equivalent, so
this is the only way to set them.
table_kwargs:
Passed through to :meth:`create`, so ``title``, ``index_columns``
and the rest are available here too.
Raises
------
SchemaError
If a column's type has no H5Col equivalent, if a fill value occurs
in its column's data, if a boolean column holds nulls, or if
*specs* does not name exactly the table's columns.
ReservedNameError
If a column name, or a producer metadata key, is one H5Col reserves.
"""
prepared = arrow.prepared_specs(table, specs)
out = cls.create(group, prepared, **table_kwargs)
by_name = {s.name: s for s in prepared}
for batch in table.to_batches():
if not batch.num_rows:
continue
out.append(
{
name: arrow.append_values(by_name[name], batch.column(name))
for name in batch.schema.names
}
)
return out
# -- introspection ------------------------------------------------------ #
@property
def group(self) -> Any:
"""The underlying h5py Group backing the table."""
return self._group
@property
def nrows(self) -> int:
"""The table's logical row count (its ``NROWS`` attribute).
Raises
------
ConformanceError
If the group carries no ``NROWS`` attribute.
"""
n = read_uint64_attr(self._group, ATTR_NROWS)
if n is None:
raise ConformanceError(f"table {self._group.name!r} has no NROWS attribute")
return n
@property
def version(self) -> str | None:
"""The table's H5Col ``VERSION`` string, or None when absent."""
return read_str_attr(self._group, ATTR_VERSION)
@property
def title(self) -> str | None:
"""The table's ``title`` attribute, or None when unset."""
return read_str_attr(self._group, ATTR_TITLE)
@property
def description(self) -> str | None:
"""The table's ``description`` attribute, or None when unset."""
return read_str_attr(self._group, ATTR_DESCRIPTION)
@property
def generation(self) -> int | None:
"""The table's ``GENERATION`` validity token, or None when absent.
A table acquires ``GENERATION`` when its first search index is built
and increments it on every subsequent mutation of committed data.
"""
return indexes.table_generation(self._group)
def _discover_columns(self) -> dict[str, Any]:
"""All columns by name: rank-1 datasets and ``CLASS=LIST_COLUMN`` groups."""
found: dict[str, Any] = {}
for name, obj in self._group.items():
if name in _SKIP_CHILDREN:
continue
if isinstance(obj, h5py.Dataset) and obj.ndim == 1:
found[name] = obj
elif isinstance(obj, h5py.Group):
if read_str_attr(obj, ATTR_CLASS) == CLASS_LIST_COLUMN:
found[name] = obj
return found
def _scalar_column_datasets(self) -> dict[str, Any]:
"""Only the column *datasets* (excludes list column groups)."""
return {
n: o
for n, o in self._discover_columns().items()
if isinstance(o, h5py.Dataset)
}
def _wrap(self, obj: Any) -> Column | ListColumn:
"""Wrap a discovered column object in its column class.
Parameters
----------
obj:
One of the objects found by :meth:`_discover_columns`. An h5py
dataset becomes a :class:`~h5col.Column`; anything else is taken
to be a list column group and becomes a
:class:`~h5col.ListColumn`. That is not re-checked here.
"""
if isinstance(obj, h5py.Dataset):
return Column(obj, self)
return ListColumn(obj, self)
@property
def column_names(self) -> list[str]:
"""Column names in logical order (``column-order`` if present)."""
order = read_str_array_attr(self._group, ATTR_COLUMN_ORDER)
columns = self._discover_columns()
if order is not None:
return [n for n in order if n in columns]
return sorted(columns)
@property
def index_columns(self) -> list[str]:
"""Names of the row-index columns, outermost first."""
if ATTR_INDEX_COLUMNS not in self._group.attrs:
return []
refs = self._group.attrs[ATTR_INDEX_COLUMNS]
names = []
for r in refs:
if references.is_null_ref(r):
continue
names.append(references.resolve(self._group, r).name.rsplit("/", 1)[-1])
return names
@property
def columns(self) -> dict[str, Column | ListColumn]:
"""The table's columns by name, in column order, as wrapper objects."""
cols = self._discover_columns()
return {n: self._wrap(cols[n]) for n in self.column_names}
[docs]
def __getitem__(self, name: str) -> Column | ListColumn:
"""The column named *name*, as a :class:`Column` or :class:`ListColumn`.
Parameters
----------
name:
A column name. Raises :class:`KeyError` if the table has no such
column.
"""
cols = self._discover_columns()
if name not in cols:
raise KeyError(name)
return self._wrap(cols[name])
[docs]
def __contains__(self, name: str) -> bool:
"""True if the table has a column of this name.
Parameters
----------
name:
A column name.
"""
return name in self._discover_columns()
[docs]
def __iter__(self) -> Iterator[str]:
"""Iterate the column names, in the table's logical column order."""
return iter(self.column_names)
[docs]
def __len__(self) -> int:
"""The number of columns in the table (its rows are :attr:`nrows`)."""
return len(self._discover_columns())
# -- writing ------------------------------------------------------------ #
[docs]
def append(
self, data: Mapping[str, Any], *, maintain_indexes: bool = False
) -> None:
"""Append rows following the H5Col write protocol.
Every provided column must supply the same number of rows ``K``. A scalar
column absent from *data* is extended and left as its fill value
(missing); a boolean column (which has no fill) must always be provided.
A list column absent from *data* must be nullable — its new rows become
null lists; a non-nullable list column must always be provided. ``NROWS``
is committed last, then the file is flushed.
``None`` in a column's values marks that row as missing and is stored as
the column's fill value (for a categorical column, its fill code). A
column with no fill to store — a boolean, which H5Col forbids from
declaring one — rejects ``None`` instead of coercing it.
By default, search indexes are **not** maintained: the ``GENERATION``
increment that publishes the append disables them, detectably, and
:meth:`refresh_indexes` restores them later — this keeps the hot append
path fast. With ``maintain_indexes=True``, every supported index is
rewritten inside the append protocol (future-valued tokens before
content) and remains valid after the commit; indexes this
implementation cannot rebuild — unsupported kinds, element dtypes the
builder does not handle, non-growable index datasets — are left
entirely untouched, tokens included.
Parameters
----------
data:
New rows, one entry per column, keyed by column name. A
:class:`numpy.ma.MaskedArray` is accepted and its masked elements
mean the same as ``None``.
maintain_indexes:
Rebuild the supported search indexes inside the write protocol so
they stay valid after the commit, as described above.
Raises
------
OversizedStringError
If a fixed-length string value's encoding exceeds the column's byte
budget (H5Col never silently truncates).
SchemaError
For unknown columns, values that are not 1-D, unequal column
lengths, an omitted fill-less/boolean column, an omitted
non-nullable list column, an unknown category label, or a ``None``
in a column that declares no fill value.
"""
cols = self._discover_columns()
unknown = set(data) - set(cols)
if unknown:
raise SchemaError(f"unknown columns in append data: {sorted(unknown)}")
scalar_ds = {n: o for n, o in cols.items() if isinstance(o, h5py.Dataset)}
list_grps = {n: o for n, o in cols.items() if not isinstance(o, h5py.Dataset)}
prepared: dict[str, np.ndarray] = {}
list_rows: dict[str, list[Any]] = {}
lengths: set[int] = set()
for name, values in data.items():
if name in scalar_ds:
ds = scalar_ds[name]
if ATTR_CATEGORIES in ds.attrs:
arr = categorical.encode_labels(self._group, ds, values)
else:
# None means "this row is missing", so it becomes the
# column's fill value before encoding.
arr = prepare_column_data(
ds.dtype, substitute_fill_for_none(ds, values, name)
)
if arr.ndim != 1:
raise SchemaError(
f"append values for column {name!r} must be a 1-D sequence, "
f"got {arr.ndim}-D"
)
prepared[name] = arr
lengths.add(arr.shape[0])
else:
rows = list(values)
list_rows[name] = rows
lengths.add(len(rows))
if len(lengths) > 1:
raise SchemaError(f"append columns have unequal lengths: {sorted(lengths)}")
k = lengths.pop() if lengths else 0
if k == 0:
return
# Validate before mutating. A scalar column absent from the append is
# filled with its fill value, so a fill-less boolean column must be
# provided. A list column absent from the append becomes null lists, so
# it must be nullable (carry a top-level MASK).
for name, ds in scalar_ds.items():
has_fill = ds.id.get_create_plist().fill_value_defined() == 2
if name not in prepared and not has_fill:
raise SchemaError(
f"column {name!r} has no fill value and must be provided "
"in the append data"
)
for name, g in list_grps.items():
if name not in list_rows:
if MEMBER_MASK not in g:
raise SchemaError(
f"list column {name!r} is not nullable and must be provided "
"in the append data"
)
list_rows[name] = [None] * k
# Append-protocol step 1: read the pre-append state. The strict
# (repairing) read keeps a malformed foreign GENERATION from being
# incremented as-is, which could collide with an index's residue
# tokens and spuriously validate it after the step-5 rewrite.
n_old = self.nrows
n_new = n_old + k
g_old = indexes.mutation_generation(self._group)
# Steps 2-3: extend + write every scalar column (equal-extent
# invariant); resize leaves absent columns' new rows at their fill.
for name, ds in scalar_ds.items():
extend_to(ds, n_new)
if name in prepared:
ds[n_old:n_new] = prepared[name]
# Encode + write every list column (leaf-first, top OFFSETS last).
for name, g in list_grps.items():
lists.append_list_column(g, list_rows[name], n_old)
# Flush the new column data before committing NROWS, so a crash cannot
# publish a larger NROWS that points at unwritten rows (H5Col ordering).
self._group.file.flush()
# Step 4: maintain the supported search indexes — future-valued tokens
# first, then content, then flush (tokens-before-content rule).
if maintain_indexes:
if g_old is None and indexes.search_index_datasets(self._group):
# A table with indexes must carry GENERATION (rule 12); repair.
g_old = indexes.ensure_generation(self._group)
if g_old is not None and indexes.append_refresh_indexes(
self._group, g_old, n_new
):
self._group.file.flush()
# Step 5: bump GENERATION whenever the table carries it, so indexes
# not maintained above fail the validity check.
if g_old is not None:
write_uint64_attr(self._group, ATTR_GENERATION, g_old + 1)
# Step 6: commit NROWS last, then flush again.
write_uint64_attr(self._group, ATTR_NROWS, n_new)
self._group.file.flush()
[docs]
def truncate(self, nrows: int, *, maintain_indexes: bool = False) -> None:
"""Shrink the logical table to *nrows* rows (H5Col logical truncation).
The truncation is logical: no column dataset changes extent, and the
rows ``[nrows, old_NROWS)`` become reserved storage that consumers
ignore. List columns need no extra writes — the smaller ``NROWS``
bounds their offsets recursively. Reclaiming physical space would
require rewriting each column to its new extent, which this
implementation does not do.
Index handling mirrors :meth:`append` (the spec applies the same
steps 4-6 with the new row count): by default every search index is
left detectably stale by the ``GENERATION`` bump; with
``maintain_indexes=True`` the supported indexes are rebuilt inside the
protocol and remain valid after the commit.
Truncating to the current row count is a no-op (nothing changes, so
nothing is published); growing is an error — that is what
:meth:`append` is for.
Parameters
----------
nrows:
The new row count. Must be between 0 and the current row count.
maintain_indexes:
As for :meth:`append`.
Raises
------
SchemaError
If *nrows* is negative or greater than the current row count.
"""
n_new = int(nrows)
if n_new < 0:
raise SchemaError(f"cannot truncate to a negative row count {n_new}")
n_old = self.nrows
if n_new > n_old:
raise SchemaError(
f"cannot truncate {n_old} rows to {n_new}; truncation only shrinks"
)
if n_new == n_old:
return
g_old = indexes.mutation_generation(self._group)
# Step 4: maintain the supported search indexes — future-valued tokens
# first, then content, then flush (tokens-before-content rule).
if maintain_indexes:
if g_old is None and indexes.search_index_datasets(self._group):
# A table with indexes must carry GENERATION (rule 12); repair.
g_old = indexes.ensure_generation(self._group)
if g_old is not None and indexes.append_refresh_indexes(
self._group, g_old, n_new
):
self._group.file.flush()
# Step 5: bump GENERATION whenever the table carries it, so indexes
# not maintained above fail the validity check.
if g_old is not None:
write_uint64_attr(self._group, ATTR_GENERATION, g_old + 1)
# Step 6: commit NROWS last, then flush.
write_uint64_attr(self._group, ATTR_NROWS, n_new)
self._group.file.flush()
# -- reading ------------------------------------------------------------ #
[docs]
def read(
self,
columns: Sequence[str] | None = None,
*,
where: Any = None,
explain: bool = False,
masked: bool = True,
) -> Any:
"""Read columns (default all) as ``{name: array}`` over ``[0, NROWS)``.
Parameters
----------
columns:
Names to read, in the order given. None (the default) reads every
column of the table.
where:
Restrict the result to matching rows. Takes the same forms as
:meth:`select`; None (the default) reads every row.
explain:
When True the return value becomes a ``(result, QueryPlan)`` pair,
the plan describing how the query was evaluated.
masked:
Return each scalar column as a :class:`numpy.ma.MaskedArray` whose
mask marks its missing rows (the default). List columns ignore it,
already spelling a null row ``None``. Pass False for plain arrays,
in which case a missing row holds the column's fill value with
nothing to distinguish it from data.
Raises
------
KeyError
If a requested column name is not a column of the table.
SchemaError
If ``where=`` is malformed or references an unknown column.
"""
if where is not None or explain:
sel = self.select(where)
result = sel.read(columns, masked=masked)
return (result, sel.explain()) if explain else result
names = list(columns) if columns is not None else self.column_names
cols = self.columns
out: dict[str, Any] = {}
for name in names:
if name not in cols:
raise KeyError(name)
col = cols[name]
# A list column has no mask to carry: it already spells a null row
# `None`, and a ragged column cannot be a MaskedArray at all.
out[name] = (
col.read() if isinstance(col, ListColumn) else col.read(masked=masked)
)
return out
[docs]
def to_arrow(
self, columns: Sequence[str] | None = None, *, where: Any = None
) -> Any:
"""Convert the table (default all columns) to a :class:`pyarrow.Table`.
The one export that carries the whole H5Col data model: missing rows
become real Arrow nulls instead of the fill value, a categorical becomes
a dictionary of the codes and labels already stored, a list column keeps
its nulls at every level of nesting, and each column's ``units``,
``description`` and valid-range attributes ride along as Arrow field
metadata under an ``h5col.`` prefix.
Needs the optional ``pyarrow`` dependency (``pip install h5col[arrow]``).
.. versionadded:: 0.2.0
Parameters
----------
columns:
Names to convert, in the order given. None (the default) converts
every column of the table.
where:
Restrict the result to matching rows. Takes the same forms as
:meth:`select`; None (the default) converts every row.
Raises
------
KeyError
If a requested column name is not a column of the table.
"""
if where is not None:
return self.select(where).to_arrow(columns)
return arrow.table_arrow(self, columns)
[docs]
def select(self, where: Any = None) -> query.Selection:
"""Build a lazy :class:`~h5col.query.Selection` over the table.
Parameters
----------
where:
A query :class:`~h5col.query.Expression`, a ``List[Tuple]`` (AND),
or a ``List[List[Tuple]]`` (OR-of-ANDs, pyarrow DNF). None (the
default) selects every row.
"""
return query.Selection(self, query._to_expression(where))
[docs]
def count(self, where: Any = None) -> int:
"""Number of rows matching *where*, without reading any column values.
Parameters
----------
where:
As for :meth:`select`. None (the default) counts every row.
"""
return self.select(where).count
[docs]
def build_index(
self,
column: str,
kind: str | None = None,
*,
name: str | None = None,
description: str | None = None,
) -> SearchIndex:
"""Build a search index over *column* (alias of :meth:`add_search_index`).
Parameters
----------
column:
As for :meth:`add_search_index`.
kind:
As for :meth:`add_search_index`.
name:
As for :meth:`add_search_index`.
description:
As for :meth:`add_search_index`.
"""
return self.add_search_index(column, kind, name=name, description=description)
# -- search indexes ------------------------------------------------------ #
[docs]
def info(self, *, storage: bool = False, full: bool = False) -> TableInfo:
"""Table content information, storage and compression details optional.
The result prints as an aligned block at a prompt and renders as a
collapsible view in a Jupyter notebook, and its records stay reachable
for anyone who wants the numbers rather than the picture.
Nothing here reads a column's values, and by default nothing measures
stored size or compression either, so even inspecting a large table
requires a handful of metadata reads.
.. versionadded:: 0.5.0
Parameters
----------
storage:
Whether to measure what each column costs in the file. Off by
default because measuring walks every column's chunk index.
full:
Whether to report each column as a labelled block of everything
known about it: description, valid range, per-dataset sizes, etc.
This is the same detail a notebook shows behind a disclosure
triangle. What a column costs is part of everything, so this sets
``storage=True`` as well.
"""
return info.gather(self, storage=storage, full=full)
@property
def search_indexes(self) -> dict[str, SearchIndex]:
"""Every search-index dataset under ``SEARCH_INDEXES``, wrapped by kind."""
return {
name: wrap_index(ds, self)
for name, ds in indexes.search_index_datasets(self._group).items()
}
[docs]
def add_search_index(
self,
column: str,
kind: str | None = None,
*,
name: str | None = None,
description: str | None = None,
) -> SearchIndex:
"""Build a search index over *column* and link it to the column.
With ``kind=None`` the family is picked automatically: ``BITMAP`` for
boolean and categorical columns (low cardinality, exact equality
answers), ``CHUNK_MINMAX`` for any other orderable column. Building an
index over an unchanged table is not a mutation: ``GENERATION`` is
created (``0``) if absent but never incremented. The default dataset
name is ``<column>__<kind, lowercased>`` — a readable convention only;
the linkage is the object reference in the column's
``SEARCH_INDEX_LIST``.
Parameters
----------
column:
Name of the column to index. It must be a scalar column; list
columns cannot be indexed.
kind:
``CHUNK_MINMAX``, ``SORTED_ROWS`` or ``BITMAP``. None (the default)
picks the family that suits the column's datatype, as above.
name:
Name for the index dataset under ``SEARCH_INDEXES``. None uses the
``<column>__<kind>`` default described above.
description:
Free text stored on the index as its ``DESCRIPTION`` attribute.
Raises
------
KeyError
If *column* is not a column of the table.
SchemaError
If *column* is a list column, no index family applies to its dtype
(``kind=None``), *kind* is unimplemented, or ``SEARCH_INDEXES``
already holds a dataset of the chosen name.
ReservedNameError
If *name* is a H5Col reserved name.
ConformanceError
If the table carries no ``NROWS`` attribute.
"""
cols = self._discover_columns()
if column not in cols:
raise KeyError(column)
ds = cols[column]
if not isinstance(ds, h5py.Dataset):
raise SchemaError(
f"search indexes over list columns are not permitted ({column!r})"
)
if kind is None:
if not indexes.supported_index_dtype(ds.dtype):
raise SchemaError(
f"no search-index family applies to column {column!r} with "
f"dtype {ds.dtype!r}"
)
if is_bool_dtype(ds.dtype) or ATTR_CATEGORIES in ds.attrs:
kind = KIND_BITMAP
else:
kind = KIND_CHUNK_MINMAX
if kind == KIND_CHUNK_MINMAX:
index_ds = indexes.create_chunk_minmax(
self._group, ds, name=name, description=description
)
elif kind == KIND_SORTED_ROWS:
index_ds = indexes.create_sorted_rows(
self._group, ds, name=name, description=description
)
elif kind == KIND_BITMAP:
index_ds = indexes.create_bitmap(
self._group, ds, name=name, description=description
)
else:
raise SchemaError(
f"search-index kind {kind!r} is not implemented "
"(CHUNK_BLOOM is sub-phase 4c)"
)
return wrap_index(index_ds, self, ds)
[docs]
def refresh_indexes(self) -> int:
"""Rebuild every supported search index against the current table state.
Restores indexes left stale by ``append(maintain_indexes=False)`` or by
any other mutation. Returns the number of indexes refreshed; indexes of
unsupported kinds are left untouched (and stay detectably stale).
"""
return indexes.refresh_all_indexes(self._group)
[docs]
def index_is_valid(self, index: SearchIndex | Any) -> bool:
"""The H5Col consumer validity check for *index*.
Parameters
----------
index:
A :class:`~h5col.SearchIndex` wrapper or the index dataset itself.
"""
ds = index.dataset if isinstance(index, SearchIndex) else index
return indexes.index_is_valid(ds, self._group)
# -- validation --------------------------------------------------------- #
def _check_version(self) -> None:
v = self.version
if v is None:
raise ConformanceError("table has no VERSION attribute")
try:
major = int(v.split(".")[0])
except (ValueError, IndexError) as exc:
raise ConformanceError(f"unparsable VERSION {v!r}") from exc
if major > SUPPORTED_MAJOR:
raise VersionError(
f"table VERSION major {major} exceeds supported {SUPPORTED_MAJOR}"
)
def _check_nrows_attr(self) -> None:
if ATTR_NROWS not in self._group.attrs:
raise ConformanceError("table has no NROWS attribute")
val = np.asarray(self._group.attrs[ATTR_NROWS])
if val.shape != ():
raise ConformanceError(
f"NROWS must be a scalar attribute, got shape {val.shape}"
)
if not (val.dtype.kind == "u" and val.dtype.itemsize == 8):
raise ConformanceError(f"NROWS must be uint64, got dtype {val.dtype}")
[docs]
def validate(self, *, deep: bool = False) -> None:
"""Check the H5Col consistency requirements, raising on any violation.
A stale index is never an error: the validity check disables it, as the
spec intends.
Parameters
----------
deep:
When True, additionally re-derive every *valid* search index from
its column and compare (consistency rule 9's semantic half), which
costs an index build apiece. The default run is structural only.
Raises
------
ConformanceError
On the first consistency violation found.
VersionError
If the table's ``VERSION`` major exceeds the supported major.
"""
if not self.is_table_group(self._group):
raise ConformanceError("missing/incorrect CLASS attribute")
self._check_version()
self._check_nrows_attr()
nrows = self.nrows
columns = self._discover_columns()
datasets = {n: o for n, o in columns.items() if isinstance(o, h5py.Dataset)}
list_grps = {
n: o for n, o in columns.items() if not isinstance(o, h5py.Dataset)
}
# Rule 2 applies to column datasets only; list column members have their
# own per-level extents (checked below).
extents = {ds.shape[0] for ds in datasets.values()}
if len(extents) > 1:
raise ConformanceError(
f"column datasets have unequal extents: {sorted(extents)}"
)
if extents and next(iter(extents)) < nrows:
raise ConformanceError("a column extent is smaller than NROWS")
# Rule 6: column-order lists every column (datasets AND list columns).
order = read_str_array_attr(self._group, ATTR_COLUMN_ORDER)
if order is not None:
if sorted(order) != sorted(columns):
raise ConformanceError(
"column-order does not list every column exactly once"
)
# Rules 10 and 11: every list column subtree is structurally conformant.
for lg in list_grps.values():
lists.validate_list_column(lg, nrows)
for name, ds in datasets.items():
user_fill = ds.id.get_create_plist().fill_value_defined() == 2
if FixedString.is_fixed_string(ds.dtype) or ds.dtype.kind not in ("b",):
if not user_fill and not _looks_boolean(ds):
raise ConformanceError(
f"non-boolean column {name!r} lacks a user-defined fill value"
)
if _looks_boolean(ds) and user_fill:
raise ConformanceError(
f"boolean column {name!r} must not declare a fill value"
)
# INDEX_COLUMNS: non-null references to direct-child column datasets
# (compared by HDF5 path, not basename); _index must agree with the first.
if ATTR_INDEX_COLUMNS in self._group.attrs:
child_paths = {ds.name for ds in datasets.values()}
resolved_names: list[str] = []
for r in self._group.attrs[ATTR_INDEX_COLUMNS]:
if references.is_null_ref(r):
raise ConformanceError("INDEX_COLUMNS contains a null reference")
obj = references.resolve(self._group, r)
if obj.name not in child_paths:
raise ConformanceError(
f"INDEX_COLUMNS entry {obj.name!r} is not a direct-child "
"column dataset"
)
resolved_names.append(obj.name.rsplit("/", 1)[-1])
idx = read_str_attr(self._group, ATTR_INDEX)
if idx is not None and resolved_names and idx != resolved_names[0]:
raise ConformanceError(
f"_index {idx!r} does not match INDEX_COLUMNS[0] "
f"{resolved_names[0]!r}"
)
# Categorical columns and the CATEGORIES subgroup (rules 5 and 8).
cat_group = self._group.get(GROUP_CATEGORIES)
referenced: set[str] = set()
for name, ds in datasets.items():
if ATTR_CATEGORIES not in ds.attrs:
continue
if ds.dtype.kind not in ("i", "u"):
raise ConformanceError(
f"categorical column {name!r} must have an integer datatype"
)
ref = ds.attrs[ATTR_CATEGORIES]
if references.is_null_ref(ref):
raise ConformanceError(
f"categorical column {name!r} has a null CATEGORIES reference"
)
cat_ds = references.resolve(self._group, ref)
if cat_group is None or not cat_ds.name.startswith(f"{cat_group.name}/"):
raise ConformanceError(
f"categorical column {name!r} CATEGORIES does not resolve into "
"the CATEGORIES subgroup"
)
referenced.add(cat_ds.name)
if ds.id.get_create_plist().fill_value_defined() == 2:
fill = int(np.asarray(ds.fillvalue))
ncats = cat_ds.shape[0]
if 0 <= fill < ncats:
raise ConformanceError(
f"categorical column {name!r} fill {fill} collides with a "
f"valid code [0, {ncats})"
)
if cat_group is not None:
for cname, cobj in cat_group.items():
if not isinstance(cobj, h5py.Dataset):
raise ConformanceError(
f"CATEGORIES contains a non-dataset object {cname!r}"
)
if cobj.name not in referenced:
raise ConformanceError(
f"categories dataset {cname!r} is not referenced by any "
"categorical column"
)
# Search indexes: rules 3, 4, 12, and rule 9 for every supported kind.
indexes.validate_search_indexes(self._group, nrows, deep=deep)
[docs]
def add_column(
self,
spec: ColumnSpec | ListColumnSpec,
*,
default_chunk_bytes: int | None = None,
) -> Column | ListColumn:
"""Add a new column to an existing table (schema evolution).
A scalar column is created and grown to the table's current extent; its
existing rows read as its fill value (missing). A fill-less (boolean)
column, or a list column (whose "missing" analogue is a null list that
would need explicit backfilling), cannot represent pre-existing rows, so
adding one to a table that already has rows is refused.
Parameters
----------
spec:
The new column's :class:`~h5col.ColumnSpec` or
:class:`~h5col.ListColumnSpec`. Its name must not already be in use.
default_chunk_bytes:
As for :meth:`create`: the target chunk size when the spec sets no
explicit ``chunks`` shape.
Raises
------
SchemaError
If a column of that name already exists, the spec is invalid, or the
column cannot backfill pre-existing rows (a boolean or list column on
a non-empty table).
"""
cols = self._discover_columns()
if spec.name in cols:
raise SchemaError(f"column {spec.name!r} already exists")
self._validate_column_spec(spec)
if isinstance(spec, ListColumnSpec):
is_list = True
cannot_backfill = True
else:
is_list = False
cannot_backfill = spec.is_boolean
if cannot_backfill and self.nrows > 0:
kind = "list" if is_list else "boolean"
raise SchemaError(
f"cannot add {kind} column {spec.name!r} to a table with rows: "
"it cannot represent the pre-existing rows as missing"
)
datasets = {n: o for n, o in cols.items() if isinstance(o, h5py.Dataset)}
obj = self._create_one_column(
self._group, spec, default_chunk_bytes=default_chunk_bytes
)
if not is_list:
extent = next(iter({d.shape[0] for d in datasets.values()}), self.nrows)
extend_to(obj, extent)
order = read_str_array_attr(self._group, ATTR_COLUMN_ORDER) or list(cols)
if ATTR_COLUMN_ORDER in self._group.attrs:
del self._group.attrs[ATTR_COLUMN_ORDER]
write_utf8_array_attr(self._group, ATTR_COLUMN_ORDER, [*order, spec.name])
return self._wrap(obj)
def _looks_boolean(dataset: Any) -> bool:
"""Return True if *dataset* holds an H5Col boolean column.
Parameters
----------
dataset:
An open column dataset. Only its ``dtype`` is read, and the decision
rests entirely with :func:`~h5col.booleans.is_bool_dtype`.
"""
from .booleans import is_bool_dtype
return is_bool_dtype(dataset.dtype)
def _infer_column_spec(name: str, arr: Any) -> ColumnSpec:
"""Derive a :class:`~h5col.ColumnSpec` for *name* from the values in *arr*.
Parameters
----------
name:
The name the returned spec carries. It is used for nothing else here
and is not validated; :meth:`Table.create` checks it later.
arr:
The column's values, read through ``np.asarray``. A boolean array
gives the H5Col boolean dtype. Strings and bytes — including an object
array whose values are all ``str`` or ``bytes`` — give a fixed-length
string sized to the longest value in UTF-8 bytes, and at least one
byte wide. Anything else keeps the array's own dtype.
"""
a = np.asarray(arr)
if a.dtype.kind == "b":
return ColumnSpec(name=name, dtype=bool_dtype())
if a.dtype.kind in ("U", "S") or (
a.dtype.kind == "O" and all(isinstance(v, str | bytes) for v in a.ravel())
):
vals = a.ravel().tolist()
max_bytes = max(
(len(v.encode("utf-8") if isinstance(v, str) else v) for v in vals),
default=1,
)
return ColumnSpec(name=name, dtype=FixedString(max(1, max_bytes)))
return ColumnSpec(name=name, dtype=a.dtype)