Source code for h5col.table

"""The Table class: create, write, and read a H5Col column-oriented table.

A table is an HDF5 group carrying ``CLASS="COLUMN_TABLE"``. This module builds
such groups from :class:`~h5col.specs.TableSpec` / :class:`~h5col.specs.ColumnSpec`,
appends rows following the H5Col write protocol (extend every column, write,
then commit ``NROWS`` last), reads them back, and validates conformance.
"""

from __future__ import annotations

from collections.abc import Iterator, Mapping, Sequence
from typing import Any

import h5py
import numpy as np

from . import arrow, categorical, indexes, info, lists, query, references
from ._hdf5 import (
    create_column_dataset,
    extend_to,
    prepare_column_data,
    read_str_array_attr,
    read_str_attr,
    read_uint64_attr,
    substitute_fill_for_none,
    write_ascii_token_attr,
    write_extra_attributes,
    write_uint64_attr,
    write_utf8_array_attr,
    write_utf8_attr,
)
from .booleans import bool_dtype, is_bool_dtype
from .column import Column
from .exceptions import ConformanceError, SchemaError, VersionError
from .info import TableInfo
from .listcolumn import ListColumn
from .missing import recommended_fill, validate_fill_outside_range
from .reserved import (
    ATTR_CATEGORIES,
    ATTR_CLASS,
    ATTR_COLUMN_ORDER,
    ATTR_DESCRIPTION,
    ATTR_ENCODING_TYPE,
    ATTR_ENCODING_VERSION,
    ATTR_GENERATION,
    ATTR_INDEX,
    ATTR_INDEX_COLUMNS,
    ATTR_NROWS,
    ATTR_TITLE,
    ATTR_UNITS,
    ATTR_UNITS_VOCABULARY,
    ATTR_VALID_MAX,
    ATTR_VALID_MIN,
    ATTR_VERSION,
    CLASS_COLUMN_TABLE,
    CLASS_LIST_COLUMN,
    GROUP_CATEGORIES,
    GROUP_SEARCH_INDEXES,
    KIND_BITMAP,
    KIND_CHUNK_MINMAX,
    KIND_SORTED_ROWS,
    MEMBER_MASK,
    validate_attribute_names,
    validate_column_name,
)
from .searchindex import SearchIndex, wrap_index
from .specs import ColumnSpec, ListColumnSpec, TableSpec
from .strings import FixedString

#: HEP001 revision this implementation writes and the highest major it reads.
VERSION = "1.0"
SUPPORTED_MAJOR = 1

_SKIP_CHILDREN = frozenset({GROUP_CATEGORIES, GROUP_SEARCH_INDEXES})


[docs] class Table: """A H5Col column-oriented table backed by an HDF5 group.""" def __init__(self, group: Any) -> None: self._group = group def __repr__(self) -> str: try: return f"<h5col.Table {self._group.name!r} nrows={self.nrows}>" except Exception: return "<h5col.Table (closed or invalid)>" # -- construction ------------------------------------------------------- #
[docs] @staticmethod def is_table_group(group: Any) -> bool: """True if *group* is a H5Col table group (lenient CLASS check). Parameters ---------- group: Any h5py group. One that is not a table group answers False rather than raising, so this is safe to use while walking a file. """ return read_str_attr(group, ATTR_CLASS) == CLASS_COLUMN_TABLE
[docs] @classmethod def open(cls, group: Any) -> Table: """Open an existing table group, checking its CLASS and VERSION major. Parameters ---------- group: An h5py group already written as a H5Col table. Opening does not read any column data. Raises ------ ConformanceError If *group* is not a H5Col table group, or its ``VERSION`` is missing or unparsable. VersionError If the table's ``VERSION`` major exceeds the supported major. """ if not cls.is_table_group(group): raise ConformanceError( f"{group.name!r} is not a H5Col table group " f"(missing or wrong {ATTR_CLASS} attribute)" ) table = cls(group) table._check_version() return table
[docs] @classmethod def create( cls, group: Any, columns: TableSpec | Sequence[ColumnSpec | ListColumnSpec], *, title: str | None = None, description: str | None = None, index_columns: Sequence[str] | None = None, column_order: Sequence[str] | None = None, units_vocabulary: str | None = None, encoding_type: str | None = None, encoding_version: str | None = None, default_chunk_bytes: int | None = None, ) -> Table: """Create a new, empty table (``NROWS = 0``) with the given columns. Parameters ---------- group: An h5py group to write the table into. It must not already be a H5Col table group. columns: A :class:`~h5col.TableSpec`, or a sequence of :class:`~h5col.ColumnSpec` and :class:`~h5col.ListColumnSpec`. title: Human-readable table title, stored as the ``TITLE`` attribute. description: Longer free text, stored as ``DESCRIPTION``. index_columns: Names of the columns that together identify a row, stored as ``INDEX_COLUMNS``. Every name must be a declared scalar column; a list column cannot be an index column. column_order: The columns' logical order, stored as ``COLUMN_ORDER``. Defaults to the order given in *columns*, which HDF5 itself does not preserve. units_vocabulary: The vocabulary the columns' ``units`` come from, e.g. ``UDUNITS-2``. encoding_type: Producer-defined encoding identifier for the table as a whole. encoding_version: Version string paired with *encoding_type*. default_chunk_bytes: Overrides the automatic (chunk-cache-scaled) target chunk size for columns that do not set an explicit ``chunks`` shape. Raises ------ SchemaError If *group* is already a H5Col table group, a column spec is invalid, or an ``index_columns`` name is not among the declared columns. ReservedNameError If a column name is a H5Col reserved name. FillValueError If a column's fill value lies inside its declared valid range. """ if cls.is_table_group(group): raise SchemaError(f"{group.name!r} is already a H5Col table group") if isinstance(columns, TableSpec): spec = columns else: spec = TableSpec( columns=list(columns), title=title, description=description, index_columns=list(index_columns) if index_columns else [], column_order=list(column_order) if column_order else None, units_vocabulary=units_vocabulary, encoding_type=encoding_type, encoding_version=encoding_version, ) # Pre-flight: validate every column BEFORE writing the CLASS identifier, # so an invalid spec never leaves a group marked as a (broken) table # group (H5Col forbids CLASS="COLUMN_TABLE" on a non-conformant group). for col in spec.columns: cls._validate_column_spec(col) declared = {c.name for c in spec.columns} for ic in spec.index_columns: if ic not in declared: raise SchemaError(f"index column {ic!r} is not a declared column") # Mutate the group, rolling everything back on failure so a partially # built table is never left behind. written: list[str] = [] created: list[str] = [] cat_existed = GROUP_CATEGORIES in group try: write_ascii_token_attr(group, ATTR_CLASS, CLASS_COLUMN_TABLE) written.append(ATTR_CLASS) write_ascii_token_attr(group, ATTR_VERSION, VERSION) written.append(ATTR_VERSION) write_uint64_attr(group, ATTR_NROWS, 0) written.append(ATTR_NROWS) for attr, value in ( (ATTR_TITLE, spec.title), (ATTR_DESCRIPTION, spec.description), (ATTR_UNITS_VOCABULARY, spec.units_vocabulary), (ATTR_ENCODING_TYPE, spec.encoding_type), (ATTR_ENCODING_VERSION, spec.encoding_version), ): if value is not None: write_utf8_attr(group, attr, value) written.append(attr) for col in spec.columns: cls._create_one_column( group, col, default_chunk_bytes=default_chunk_bytes ) created.append(col.name) if spec.columns: write_utf8_array_attr(group, ATTR_COLUMN_ORDER, spec.ordered_names) written.append(ATTR_COLUMN_ORDER) if spec.index_columns: references.write_ref_array_attr( group, ATTR_INDEX_COLUMNS, [group[n] for n in spec.index_columns] ) written.append(ATTR_INDEX_COLUMNS) write_utf8_attr(group, ATTR_INDEX, spec.index_columns[0]) written.append(ATTR_INDEX) except BaseException: for n in created: if n in group: del group[n] for a in written: if a in group.attrs: del group.attrs[a] if not cat_existed and GROUP_CATEGORIES in group: del group[GROUP_CATEGORIES] raise return cls(group)
@staticmethod def _validate_column_spec(col: ColumnSpec | ListColumnSpec) -> None: """Validate one column spec without touching the file. Parameters ---------- col: The column spec to check, left unmodified. Its extra attribute names are checked first either way; a :class:`~h5col.specs.ListColumnSpec` is then handed to :func:`~h5col.lists.validate_list_column_spec` and nothing below applies to it. A :class:`~h5col.specs.ColumnSpec` is checked here for a legal column name and for a fill value consistent with the column's kind: categorical, boolean, or plain. """ validate_attribute_names(col.attributes, col.name) if isinstance(col, ListColumnSpec): lists.validate_list_column_spec(col) return validate_column_name(col.name) if col.is_categorical: assert col.categories is not None ncats = len(col.categories) if len(set(col.categories)) != len(col.categories): raise SchemaError( f"categorical column {col.name!r} has duplicate categories" ) dtype = col.resolved_dtype() if ncats > 0 and int(np.iinfo(dtype).max) < ncats - 1: raise SchemaError( f"categorical column {col.name!r} code dtype {dtype} cannot " f"index {ncats} categories" ) fill = ( col.fill_value if col.fill_value is not None else categorical.default_categorical_fill(dtype) ) try: cast_fill = int(np.asarray(fill, dtype=dtype)) except (OverflowError, ValueError) as exc: raise SchemaError( f"categorical column {col.name!r} fill {fill!r} is not " f"representable in code dtype {dtype}" ) from exc if 0 <= cast_fill < ncats: raise SchemaError( f"categorical column {col.name!r} fill {cast_fill} collides " f"with a valid code [0, {ncats})" ) validate_fill_outside_range(cast_fill, col.valid_min, col.valid_max) return Table._checked_fill(col) @staticmethod def _checked_fill(col: ColumnSpec) -> Any: """The fill value to create ``col`` with, or ``None`` where it declares none. The single place that decides both what a legal fill value is and which value a column ends up with. Creation and validation each of those answers need run at different moments. Categorical columns do not come here: they carry their own fill rule, checked in :meth:`_validate_column_spec` and applied in :meth:`_create_categorical_column`. Parameters ---------- col: A non-categorical column spec, left unmodified. Raises ------ SchemaError If a boolean column declares a fill value or a valid range. H5Col leaves a boolean no value outside its own two, so either is a contradiction rather than a preference. FillValueError If the fill value lies inside the column's declared valid range. """ if col.is_boolean: if col.fill_value is not None: raise SchemaError( f"boolean column {col.name!r} must not declare a fill value" ) if col.valid_min is not None or col.valid_max is not None: raise SchemaError( f"boolean column {col.name!r} must not declare valid_min/valid_max" ) return None fill = ( col.fill_value if col.fill_value is not None else recommended_fill(col.resolved_dtype()) ) validate_fill_outside_range(fill, col.valid_min, col.valid_max) return fill @staticmethod def _create_one_column( group: Any, col: ColumnSpec | ListColumnSpec, default_chunk_bytes: int | None = None, ) -> Any: """Create one column from its spec and return the new HDF5 object. Parameters ---------- group: The table group to create the column in. It is only passed on to the creation helpers; nothing is read from it here. col: The column's spec. A :class:`~h5col.specs.ListColumnSpec` is handed to :func:`~h5col.lists.create_list_column` and none of the scalar handling below applies to it. A categorical :class:`~h5col.specs.ColumnSpec` goes to :meth:`_create_categorical_column`. Any other one supplies the dtype, chunking, filters, fill value, and the optional attributes written on the dataset. default_chunk_bytes: A byte target for one chunk, used only when ``col.chunks`` is None. """ if isinstance(col, ListColumnSpec): return lists.create_list_column( group, col, default_chunk_bytes=default_chunk_bytes ) name = validate_column_name(col.name) dtype = col.resolved_dtype() if col.is_categorical: return Table._create_categorical_column( group, col, name, dtype, default_chunk_bytes=default_chunk_bytes ) fill = Table._checked_fill(col) ds = create_column_dataset( group, name, dtype, chunks=col.chunks, fill_value=fill, filters=col.filters, default_chunk_bytes=default_chunk_bytes, ) if col.valid_min is not None: ds.attrs.create(ATTR_VALID_MIN, np.asarray(col.valid_min, dtype=dtype)) if col.valid_max is not None: ds.attrs.create(ATTR_VALID_MAX, np.asarray(col.valid_max, dtype=dtype)) if col.units is not None: write_utf8_attr(ds, ATTR_UNITS, col.units) if col.units_vocabulary is not None: write_utf8_attr(ds, ATTR_UNITS_VOCABULARY, col.units_vocabulary) if col.description is not None: write_utf8_attr(ds, ATTR_DESCRIPTION, col.description) write_extra_attributes(ds, col.attributes) return ds @staticmethod def _create_categorical_column( group: Any, col: ColumnSpec, name: str, dtype: np.dtype, default_chunk_bytes: int | None = None, ) -> Any: """Create a categorical column and its companion categories dataset. Parameters ---------- group: The table group to create the column dataset in. Its ``CATEGORIES`` child group is required (created if absent) to hold the labels. col: The categorical column's spec. ``col.categories`` must not be None, which the caller has already ensured. Supplies the fill value, chunking, filters, and the optional attributes written on the dataset. name: The dataset's link name within *group*, already validated by the caller. The categories dataset is named ``f"{name}__CATEGORIES"``. dtype: The integer code dtype. The fill value and any ``valid_min`` / ``valid_max`` are cast to it. default_chunk_bytes: An explicit byte target for one chunk, used only when ``col.chunks`` is None. """ assert col.categories is not None # Fill range, dtype fit, and representability are enforced by # _validate_column_spec, which always runs before creation. fill = ( col.fill_value if col.fill_value is not None else categorical.default_categorical_fill(dtype) ) ds = create_column_dataset( group, name, dtype, chunks=col.chunks, fill_value=np.asarray(fill, dtype=dtype), filters=col.filters, default_chunk_bytes=default_chunk_bytes, ) cat_group = group.require_group(GROUP_CATEGORIES) cat_ds = categorical.create_categories_dataset( cat_group, f"{name}__CATEGORIES", col.categories, col.ordered ) references.write_ref_attr(ds, ATTR_CATEGORIES, cat_ds) if col.valid_min is not None: ds.attrs.create(ATTR_VALID_MIN, np.asarray(col.valid_min, dtype=dtype)) if col.valid_max is not None: ds.attrs.create(ATTR_VALID_MAX, np.asarray(col.valid_max, dtype=dtype)) if col.units is not None: write_utf8_attr(ds, ATTR_UNITS, col.units) if col.units_vocabulary is not None: write_utf8_attr(ds, ATTR_UNITS_VOCABULARY, col.units_vocabulary) if col.description is not None: write_utf8_attr(ds, ATTR_DESCRIPTION, col.description) return ds
[docs] @classmethod def from_arrays( cls, group: Any, arrays: Mapping[str, Any], *, specs: Sequence[ColumnSpec] | None = None, **table_kwargs: Any, ) -> Table: """Create a table from column arrays and write them in one call. Parameters ---------- group: An h5py group to write the table into, as for :meth:`create`. arrays: One array per column, keyed by column name. Every array must hold the same number of rows, and the column order follows this mapping. specs: Column specs to create the table with. When omitted, one is inferred per array — boolean, a fixed-length string sized to the longest value, or the array's own dtype. table_kwargs: Passed through to :meth:`create`, so ``title``, ``index_columns`` and the rest are available here too. """ if specs is None: specs = [_infer_column_spec(name, arr) for name, arr in arrays.items()] table = cls.create(group, specs, **table_kwargs) table.append(arrays) return table
[docs] @classmethod def from_arrow( cls, group: Any, table: Any, *, specs: Sequence[ColumnSpec | ListColumnSpec] | None = None, **table_kwargs: Any, ) -> Table: """Create a table from a :class:`pyarrow.Table` and write its rows. Arrow's model is wider than H5Col's, so anything without an exact equivalent is refused rather than approximated, see :func:`~h5col.specs_from_arrow`, which decides the mapping and which you can call first to inspect or adjust it:: specs = h5col.specs_from_arrow(tbl) specs[2].chunks = 8192 Table.from_arrow(group, tbl, specs=specs) Rows are written batch by batch rather than all at once, so importing a large table costs about one batch of memory rather than the whole of it. Two chunks of one dictionary column may carry different dictionaries, in which case the same code stands for two different labels. The codes are never read directly for that reason — the labels are, against the unified category set :func:`~h5col.specs_from_arrow` derives. The fill-value checks run whichever way the specs arrived: a fill that already occurs in a column would leave those rows reading as missing, so supplying specs cannot skip it. Needs the optional ``pyarrow`` dependency (``pip install h5col[arrow]``). .. versionadded:: 0.4.0 Parameters ---------- group: An h5py group to write the table into, as for :meth:`create`. table: The :class:`pyarrow.Table` to import. specs: A complete list of column specs naming exactly the table's columns. None infers them. Chunking and filters have no Arrow equivalent, so this is the only way to set them. table_kwargs: Passed through to :meth:`create`, so ``title``, ``index_columns`` and the rest are available here too. Raises ------ SchemaError If a column's type has no H5Col equivalent, if a fill value occurs in its column's data, if a boolean column holds nulls, or if *specs* does not name exactly the table's columns. ReservedNameError If a column name, or a producer metadata key, is one H5Col reserves. """ prepared = arrow.prepared_specs(table, specs) out = cls.create(group, prepared, **table_kwargs) by_name = {s.name: s for s in prepared} for batch in table.to_batches(): if not batch.num_rows: continue out.append( { name: arrow.append_values(by_name[name], batch.column(name)) for name in batch.schema.names } ) return out
# -- introspection ------------------------------------------------------ # @property def group(self) -> Any: """The underlying h5py Group backing the table.""" return self._group @property def nrows(self) -> int: """The table's logical row count (its ``NROWS`` attribute). Raises ------ ConformanceError If the group carries no ``NROWS`` attribute. """ n = read_uint64_attr(self._group, ATTR_NROWS) if n is None: raise ConformanceError(f"table {self._group.name!r} has no NROWS attribute") return n @property def version(self) -> str | None: """The table's H5Col ``VERSION`` string, or None when absent.""" return read_str_attr(self._group, ATTR_VERSION) @property def title(self) -> str | None: """The table's ``title`` attribute, or None when unset.""" return read_str_attr(self._group, ATTR_TITLE) @property def description(self) -> str | None: """The table's ``description`` attribute, or None when unset.""" return read_str_attr(self._group, ATTR_DESCRIPTION) @property def generation(self) -> int | None: """The table's ``GENERATION`` validity token, or None when absent. A table acquires ``GENERATION`` when its first search index is built and increments it on every subsequent mutation of committed data. """ return indexes.table_generation(self._group) def _discover_columns(self) -> dict[str, Any]: """All columns by name: rank-1 datasets and ``CLASS=LIST_COLUMN`` groups.""" found: dict[str, Any] = {} for name, obj in self._group.items(): if name in _SKIP_CHILDREN: continue if isinstance(obj, h5py.Dataset) and obj.ndim == 1: found[name] = obj elif isinstance(obj, h5py.Group): if read_str_attr(obj, ATTR_CLASS) == CLASS_LIST_COLUMN: found[name] = obj return found def _scalar_column_datasets(self) -> dict[str, Any]: """Only the column *datasets* (excludes list column groups).""" return { n: o for n, o in self._discover_columns().items() if isinstance(o, h5py.Dataset) } def _wrap(self, obj: Any) -> Column | ListColumn: """Wrap a discovered column object in its column class. Parameters ---------- obj: One of the objects found by :meth:`_discover_columns`. An h5py dataset becomes a :class:`~h5col.Column`; anything else is taken to be a list column group and becomes a :class:`~h5col.ListColumn`. That is not re-checked here. """ if isinstance(obj, h5py.Dataset): return Column(obj, self) return ListColumn(obj, self) @property def column_names(self) -> list[str]: """Column names in logical order (``column-order`` if present).""" order = read_str_array_attr(self._group, ATTR_COLUMN_ORDER) columns = self._discover_columns() if order is not None: return [n for n in order if n in columns] return sorted(columns) @property def index_columns(self) -> list[str]: """Names of the row-index columns, outermost first.""" if ATTR_INDEX_COLUMNS not in self._group.attrs: return [] refs = self._group.attrs[ATTR_INDEX_COLUMNS] names = [] for r in refs: if references.is_null_ref(r): continue names.append(references.resolve(self._group, r).name.rsplit("/", 1)[-1]) return names @property def columns(self) -> dict[str, Column | ListColumn]: """The table's columns by name, in column order, as wrapper objects.""" cols = self._discover_columns() return {n: self._wrap(cols[n]) for n in self.column_names}
[docs] def __getitem__(self, name: str) -> Column | ListColumn: """The column named *name*, as a :class:`Column` or :class:`ListColumn`. Parameters ---------- name: A column name. Raises :class:`KeyError` if the table has no such column. """ cols = self._discover_columns() if name not in cols: raise KeyError(name) return self._wrap(cols[name])
[docs] def __contains__(self, name: str) -> bool: """True if the table has a column of this name. Parameters ---------- name: A column name. """ return name in self._discover_columns()
[docs] def __iter__(self) -> Iterator[str]: """Iterate the column names, in the table's logical column order.""" return iter(self.column_names)
[docs] def __len__(self) -> int: """The number of columns in the table (its rows are :attr:`nrows`).""" return len(self._discover_columns())
# -- writing ------------------------------------------------------------ #
[docs] def append( self, data: Mapping[str, Any], *, maintain_indexes: bool = False ) -> None: """Append rows following the H5Col write protocol. Every provided column must supply the same number of rows ``K``. A scalar column absent from *data* is extended and left as its fill value (missing); a boolean column (which has no fill) must always be provided. A list column absent from *data* must be nullable — its new rows become null lists; a non-nullable list column must always be provided. ``NROWS`` is committed last, then the file is flushed. ``None`` in a column's values marks that row as missing and is stored as the column's fill value (for a categorical column, its fill code). A column with no fill to store — a boolean, which H5Col forbids from declaring one — rejects ``None`` instead of coercing it. By default, search indexes are **not** maintained: the ``GENERATION`` increment that publishes the append disables them, detectably, and :meth:`refresh_indexes` restores them later — this keeps the hot append path fast. With ``maintain_indexes=True``, every supported index is rewritten inside the append protocol (future-valued tokens before content) and remains valid after the commit; indexes this implementation cannot rebuild — unsupported kinds, element dtypes the builder does not handle, non-growable index datasets — are left entirely untouched, tokens included. Parameters ---------- data: New rows, one entry per column, keyed by column name. A :class:`numpy.ma.MaskedArray` is accepted and its masked elements mean the same as ``None``. maintain_indexes: Rebuild the supported search indexes inside the write protocol so they stay valid after the commit, as described above. Raises ------ OversizedStringError If a fixed-length string value's encoding exceeds the column's byte budget (H5Col never silently truncates). SchemaError For unknown columns, values that are not 1-D, unequal column lengths, an omitted fill-less/boolean column, an omitted non-nullable list column, an unknown category label, or a ``None`` in a column that declares no fill value. """ cols = self._discover_columns() unknown = set(data) - set(cols) if unknown: raise SchemaError(f"unknown columns in append data: {sorted(unknown)}") scalar_ds = {n: o for n, o in cols.items() if isinstance(o, h5py.Dataset)} list_grps = {n: o for n, o in cols.items() if not isinstance(o, h5py.Dataset)} prepared: dict[str, np.ndarray] = {} list_rows: dict[str, list[Any]] = {} lengths: set[int] = set() for name, values in data.items(): if name in scalar_ds: ds = scalar_ds[name] if ATTR_CATEGORIES in ds.attrs: arr = categorical.encode_labels(self._group, ds, values) else: # None means "this row is missing", so it becomes the # column's fill value before encoding. arr = prepare_column_data( ds.dtype, substitute_fill_for_none(ds, values, name) ) if arr.ndim != 1: raise SchemaError( f"append values for column {name!r} must be a 1-D sequence, " f"got {arr.ndim}-D" ) prepared[name] = arr lengths.add(arr.shape[0]) else: rows = list(values) list_rows[name] = rows lengths.add(len(rows)) if len(lengths) > 1: raise SchemaError(f"append columns have unequal lengths: {sorted(lengths)}") k = lengths.pop() if lengths else 0 if k == 0: return # Validate before mutating. A scalar column absent from the append is # filled with its fill value, so a fill-less boolean column must be # provided. A list column absent from the append becomes null lists, so # it must be nullable (carry a top-level MASK). for name, ds in scalar_ds.items(): has_fill = ds.id.get_create_plist().fill_value_defined() == 2 if name not in prepared and not has_fill: raise SchemaError( f"column {name!r} has no fill value and must be provided " "in the append data" ) for name, g in list_grps.items(): if name not in list_rows: if MEMBER_MASK not in g: raise SchemaError( f"list column {name!r} is not nullable and must be provided " "in the append data" ) list_rows[name] = [None] * k # Append-protocol step 1: read the pre-append state. The strict # (repairing) read keeps a malformed foreign GENERATION from being # incremented as-is, which could collide with an index's residue # tokens and spuriously validate it after the step-5 rewrite. n_old = self.nrows n_new = n_old + k g_old = indexes.mutation_generation(self._group) # Steps 2-3: extend + write every scalar column (equal-extent # invariant); resize leaves absent columns' new rows at their fill. for name, ds in scalar_ds.items(): extend_to(ds, n_new) if name in prepared: ds[n_old:n_new] = prepared[name] # Encode + write every list column (leaf-first, top OFFSETS last). for name, g in list_grps.items(): lists.append_list_column(g, list_rows[name], n_old) # Flush the new column data before committing NROWS, so a crash cannot # publish a larger NROWS that points at unwritten rows (H5Col ordering). self._group.file.flush() # Step 4: maintain the supported search indexes — future-valued tokens # first, then content, then flush (tokens-before-content rule). if maintain_indexes: if g_old is None and indexes.search_index_datasets(self._group): # A table with indexes must carry GENERATION (rule 12); repair. g_old = indexes.ensure_generation(self._group) if g_old is not None and indexes.append_refresh_indexes( self._group, g_old, n_new ): self._group.file.flush() # Step 5: bump GENERATION whenever the table carries it, so indexes # not maintained above fail the validity check. if g_old is not None: write_uint64_attr(self._group, ATTR_GENERATION, g_old + 1) # Step 6: commit NROWS last, then flush again. write_uint64_attr(self._group, ATTR_NROWS, n_new) self._group.file.flush()
[docs] def truncate(self, nrows: int, *, maintain_indexes: bool = False) -> None: """Shrink the logical table to *nrows* rows (H5Col logical truncation). The truncation is logical: no column dataset changes extent, and the rows ``[nrows, old_NROWS)`` become reserved storage that consumers ignore. List columns need no extra writes — the smaller ``NROWS`` bounds their offsets recursively. Reclaiming physical space would require rewriting each column to its new extent, which this implementation does not do. Index handling mirrors :meth:`append` (the spec applies the same steps 4-6 with the new row count): by default every search index is left detectably stale by the ``GENERATION`` bump; with ``maintain_indexes=True`` the supported indexes are rebuilt inside the protocol and remain valid after the commit. Truncating to the current row count is a no-op (nothing changes, so nothing is published); growing is an error — that is what :meth:`append` is for. Parameters ---------- nrows: The new row count. Must be between 0 and the current row count. maintain_indexes: As for :meth:`append`. Raises ------ SchemaError If *nrows* is negative or greater than the current row count. """ n_new = int(nrows) if n_new < 0: raise SchemaError(f"cannot truncate to a negative row count {n_new}") n_old = self.nrows if n_new > n_old: raise SchemaError( f"cannot truncate {n_old} rows to {n_new}; truncation only shrinks" ) if n_new == n_old: return g_old = indexes.mutation_generation(self._group) # Step 4: maintain the supported search indexes — future-valued tokens # first, then content, then flush (tokens-before-content rule). if maintain_indexes: if g_old is None and indexes.search_index_datasets(self._group): # A table with indexes must carry GENERATION (rule 12); repair. g_old = indexes.ensure_generation(self._group) if g_old is not None and indexes.append_refresh_indexes( self._group, g_old, n_new ): self._group.file.flush() # Step 5: bump GENERATION whenever the table carries it, so indexes # not maintained above fail the validity check. if g_old is not None: write_uint64_attr(self._group, ATTR_GENERATION, g_old + 1) # Step 6: commit NROWS last, then flush. write_uint64_attr(self._group, ATTR_NROWS, n_new) self._group.file.flush()
# -- reading ------------------------------------------------------------ #
[docs] def read( self, columns: Sequence[str] | None = None, *, where: Any = None, explain: bool = False, masked: bool = True, ) -> Any: """Read columns (default all) as ``{name: array}`` over ``[0, NROWS)``. Parameters ---------- columns: Names to read, in the order given. None (the default) reads every column of the table. where: Restrict the result to matching rows. Takes the same forms as :meth:`select`; None (the default) reads every row. explain: When True the return value becomes a ``(result, QueryPlan)`` pair, the plan describing how the query was evaluated. masked: Return each scalar column as a :class:`numpy.ma.MaskedArray` whose mask marks its missing rows (the default). List columns ignore it, already spelling a null row ``None``. Pass False for plain arrays, in which case a missing row holds the column's fill value with nothing to distinguish it from data. Raises ------ KeyError If a requested column name is not a column of the table. SchemaError If ``where=`` is malformed or references an unknown column. """ if where is not None or explain: sel = self.select(where) result = sel.read(columns, masked=masked) return (result, sel.explain()) if explain else result names = list(columns) if columns is not None else self.column_names cols = self.columns out: dict[str, Any] = {} for name in names: if name not in cols: raise KeyError(name) col = cols[name] # A list column has no mask to carry: it already spells a null row # `None`, and a ragged column cannot be a MaskedArray at all. out[name] = ( col.read() if isinstance(col, ListColumn) else col.read(masked=masked) ) return out
[docs] def to_arrow( self, columns: Sequence[str] | None = None, *, where: Any = None ) -> Any: """Convert the table (default all columns) to a :class:`pyarrow.Table`. The one export that carries the whole H5Col data model: missing rows become real Arrow nulls instead of the fill value, a categorical becomes a dictionary of the codes and labels already stored, a list column keeps its nulls at every level of nesting, and each column's ``units``, ``description`` and valid-range attributes ride along as Arrow field metadata under an ``h5col.`` prefix. Needs the optional ``pyarrow`` dependency (``pip install h5col[arrow]``). .. versionadded:: 0.2.0 Parameters ---------- columns: Names to convert, in the order given. None (the default) converts every column of the table. where: Restrict the result to matching rows. Takes the same forms as :meth:`select`; None (the default) converts every row. Raises ------ KeyError If a requested column name is not a column of the table. """ if where is not None: return self.select(where).to_arrow(columns) return arrow.table_arrow(self, columns)
[docs] def select(self, where: Any = None) -> query.Selection: """Build a lazy :class:`~h5col.query.Selection` over the table. Parameters ---------- where: A query :class:`~h5col.query.Expression`, a ``List[Tuple]`` (AND), or a ``List[List[Tuple]]`` (OR-of-ANDs, pyarrow DNF). None (the default) selects every row. """ return query.Selection(self, query._to_expression(where))
[docs] def count(self, where: Any = None) -> int: """Number of rows matching *where*, without reading any column values. Parameters ---------- where: As for :meth:`select`. None (the default) counts every row. """ return self.select(where).count
[docs] def build_index( self, column: str, kind: str | None = None, *, name: str | None = None, description: str | None = None, ) -> SearchIndex: """Build a search index over *column* (alias of :meth:`add_search_index`). Parameters ---------- column: As for :meth:`add_search_index`. kind: As for :meth:`add_search_index`. name: As for :meth:`add_search_index`. description: As for :meth:`add_search_index`. """ return self.add_search_index(column, kind, name=name, description=description)
# -- search indexes ------------------------------------------------------ #
[docs] def info(self, *, storage: bool = False, full: bool = False) -> TableInfo: """Table content information, storage and compression details optional. The result prints as an aligned block at a prompt and renders as a collapsible view in a Jupyter notebook, and its records stay reachable for anyone who wants the numbers rather than the picture. Nothing here reads a column's values, and by default nothing measures stored size or compression either, so even inspecting a large table requires a handful of metadata reads. .. versionadded:: 0.5.0 Parameters ---------- storage: Whether to measure what each column costs in the file. Off by default because measuring walks every column's chunk index. full: Whether to report each column as a labelled block of everything known about it: description, valid range, per-dataset sizes, etc. This is the same detail a notebook shows behind a disclosure triangle. What a column costs is part of everything, so this sets ``storage=True`` as well. """ return info.gather(self, storage=storage, full=full)
@property def search_indexes(self) -> dict[str, SearchIndex]: """Every search-index dataset under ``SEARCH_INDEXES``, wrapped by kind.""" return { name: wrap_index(ds, self) for name, ds in indexes.search_index_datasets(self._group).items() }
[docs] def add_search_index( self, column: str, kind: str | None = None, *, name: str | None = None, description: str | None = None, ) -> SearchIndex: """Build a search index over *column* and link it to the column. With ``kind=None`` the family is picked automatically: ``BITMAP`` for boolean and categorical columns (low cardinality, exact equality answers), ``CHUNK_MINMAX`` for any other orderable column. Building an index over an unchanged table is not a mutation: ``GENERATION`` is created (``0``) if absent but never incremented. The default dataset name is ``<column>__<kind, lowercased>`` — a readable convention only; the linkage is the object reference in the column's ``SEARCH_INDEX_LIST``. Parameters ---------- column: Name of the column to index. It must be a scalar column; list columns cannot be indexed. kind: ``CHUNK_MINMAX``, ``SORTED_ROWS`` or ``BITMAP``. None (the default) picks the family that suits the column's datatype, as above. name: Name for the index dataset under ``SEARCH_INDEXES``. None uses the ``<column>__<kind>`` default described above. description: Free text stored on the index as its ``DESCRIPTION`` attribute. Raises ------ KeyError If *column* is not a column of the table. SchemaError If *column* is a list column, no index family applies to its dtype (``kind=None``), *kind* is unimplemented, or ``SEARCH_INDEXES`` already holds a dataset of the chosen name. ReservedNameError If *name* is a H5Col reserved name. ConformanceError If the table carries no ``NROWS`` attribute. """ cols = self._discover_columns() if column not in cols: raise KeyError(column) ds = cols[column] if not isinstance(ds, h5py.Dataset): raise SchemaError( f"search indexes over list columns are not permitted ({column!r})" ) if kind is None: if not indexes.supported_index_dtype(ds.dtype): raise SchemaError( f"no search-index family applies to column {column!r} with " f"dtype {ds.dtype!r}" ) if is_bool_dtype(ds.dtype) or ATTR_CATEGORIES in ds.attrs: kind = KIND_BITMAP else: kind = KIND_CHUNK_MINMAX if kind == KIND_CHUNK_MINMAX: index_ds = indexes.create_chunk_minmax( self._group, ds, name=name, description=description ) elif kind == KIND_SORTED_ROWS: index_ds = indexes.create_sorted_rows( self._group, ds, name=name, description=description ) elif kind == KIND_BITMAP: index_ds = indexes.create_bitmap( self._group, ds, name=name, description=description ) else: raise SchemaError( f"search-index kind {kind!r} is not implemented " "(CHUNK_BLOOM is sub-phase 4c)" ) return wrap_index(index_ds, self, ds)
[docs] def refresh_indexes(self) -> int: """Rebuild every supported search index against the current table state. Restores indexes left stale by ``append(maintain_indexes=False)`` or by any other mutation. Returns the number of indexes refreshed; indexes of unsupported kinds are left untouched (and stay detectably stale). """ return indexes.refresh_all_indexes(self._group)
[docs] def index_is_valid(self, index: SearchIndex | Any) -> bool: """The H5Col consumer validity check for *index*. Parameters ---------- index: A :class:`~h5col.SearchIndex` wrapper or the index dataset itself. """ ds = index.dataset if isinstance(index, SearchIndex) else index return indexes.index_is_valid(ds, self._group)
# -- validation --------------------------------------------------------- # def _check_version(self) -> None: v = self.version if v is None: raise ConformanceError("table has no VERSION attribute") try: major = int(v.split(".")[0]) except (ValueError, IndexError) as exc: raise ConformanceError(f"unparsable VERSION {v!r}") from exc if major > SUPPORTED_MAJOR: raise VersionError( f"table VERSION major {major} exceeds supported {SUPPORTED_MAJOR}" ) def _check_nrows_attr(self) -> None: if ATTR_NROWS not in self._group.attrs: raise ConformanceError("table has no NROWS attribute") val = np.asarray(self._group.attrs[ATTR_NROWS]) if val.shape != (): raise ConformanceError( f"NROWS must be a scalar attribute, got shape {val.shape}" ) if not (val.dtype.kind == "u" and val.dtype.itemsize == 8): raise ConformanceError(f"NROWS must be uint64, got dtype {val.dtype}")
[docs] def validate(self, *, deep: bool = False) -> None: """Check the H5Col consistency requirements, raising on any violation. A stale index is never an error: the validity check disables it, as the spec intends. Parameters ---------- deep: When True, additionally re-derive every *valid* search index from its column and compare (consistency rule 9's semantic half), which costs an index build apiece. The default run is structural only. Raises ------ ConformanceError On the first consistency violation found. VersionError If the table's ``VERSION`` major exceeds the supported major. """ if not self.is_table_group(self._group): raise ConformanceError("missing/incorrect CLASS attribute") self._check_version() self._check_nrows_attr() nrows = self.nrows columns = self._discover_columns() datasets = {n: o for n, o in columns.items() if isinstance(o, h5py.Dataset)} list_grps = { n: o for n, o in columns.items() if not isinstance(o, h5py.Dataset) } # Rule 2 applies to column datasets only; list column members have their # own per-level extents (checked below). extents = {ds.shape[0] for ds in datasets.values()} if len(extents) > 1: raise ConformanceError( f"column datasets have unequal extents: {sorted(extents)}" ) if extents and next(iter(extents)) < nrows: raise ConformanceError("a column extent is smaller than NROWS") # Rule 6: column-order lists every column (datasets AND list columns). order = read_str_array_attr(self._group, ATTR_COLUMN_ORDER) if order is not None: if sorted(order) != sorted(columns): raise ConformanceError( "column-order does not list every column exactly once" ) # Rules 10 and 11: every list column subtree is structurally conformant. for lg in list_grps.values(): lists.validate_list_column(lg, nrows) for name, ds in datasets.items(): user_fill = ds.id.get_create_plist().fill_value_defined() == 2 if FixedString.is_fixed_string(ds.dtype) or ds.dtype.kind not in ("b",): if not user_fill and not _looks_boolean(ds): raise ConformanceError( f"non-boolean column {name!r} lacks a user-defined fill value" ) if _looks_boolean(ds) and user_fill: raise ConformanceError( f"boolean column {name!r} must not declare a fill value" ) # INDEX_COLUMNS: non-null references to direct-child column datasets # (compared by HDF5 path, not basename); _index must agree with the first. if ATTR_INDEX_COLUMNS in self._group.attrs: child_paths = {ds.name for ds in datasets.values()} resolved_names: list[str] = [] for r in self._group.attrs[ATTR_INDEX_COLUMNS]: if references.is_null_ref(r): raise ConformanceError("INDEX_COLUMNS contains a null reference") obj = references.resolve(self._group, r) if obj.name not in child_paths: raise ConformanceError( f"INDEX_COLUMNS entry {obj.name!r} is not a direct-child " "column dataset" ) resolved_names.append(obj.name.rsplit("/", 1)[-1]) idx = read_str_attr(self._group, ATTR_INDEX) if idx is not None and resolved_names and idx != resolved_names[0]: raise ConformanceError( f"_index {idx!r} does not match INDEX_COLUMNS[0] " f"{resolved_names[0]!r}" ) # Categorical columns and the CATEGORIES subgroup (rules 5 and 8). cat_group = self._group.get(GROUP_CATEGORIES) referenced: set[str] = set() for name, ds in datasets.items(): if ATTR_CATEGORIES not in ds.attrs: continue if ds.dtype.kind not in ("i", "u"): raise ConformanceError( f"categorical column {name!r} must have an integer datatype" ) ref = ds.attrs[ATTR_CATEGORIES] if references.is_null_ref(ref): raise ConformanceError( f"categorical column {name!r} has a null CATEGORIES reference" ) cat_ds = references.resolve(self._group, ref) if cat_group is None or not cat_ds.name.startswith(f"{cat_group.name}/"): raise ConformanceError( f"categorical column {name!r} CATEGORIES does not resolve into " "the CATEGORIES subgroup" ) referenced.add(cat_ds.name) if ds.id.get_create_plist().fill_value_defined() == 2: fill = int(np.asarray(ds.fillvalue)) ncats = cat_ds.shape[0] if 0 <= fill < ncats: raise ConformanceError( f"categorical column {name!r} fill {fill} collides with a " f"valid code [0, {ncats})" ) if cat_group is not None: for cname, cobj in cat_group.items(): if not isinstance(cobj, h5py.Dataset): raise ConformanceError( f"CATEGORIES contains a non-dataset object {cname!r}" ) if cobj.name not in referenced: raise ConformanceError( f"categories dataset {cname!r} is not referenced by any " "categorical column" ) # Search indexes: rules 3, 4, 12, and rule 9 for every supported kind. indexes.validate_search_indexes(self._group, nrows, deep=deep)
[docs] def add_column( self, spec: ColumnSpec | ListColumnSpec, *, default_chunk_bytes: int | None = None, ) -> Column | ListColumn: """Add a new column to an existing table (schema evolution). A scalar column is created and grown to the table's current extent; its existing rows read as its fill value (missing). A fill-less (boolean) column, or a list column (whose "missing" analogue is a null list that would need explicit backfilling), cannot represent pre-existing rows, so adding one to a table that already has rows is refused. Parameters ---------- spec: The new column's :class:`~h5col.ColumnSpec` or :class:`~h5col.ListColumnSpec`. Its name must not already be in use. default_chunk_bytes: As for :meth:`create`: the target chunk size when the spec sets no explicit ``chunks`` shape. Raises ------ SchemaError If a column of that name already exists, the spec is invalid, or the column cannot backfill pre-existing rows (a boolean or list column on a non-empty table). """ cols = self._discover_columns() if spec.name in cols: raise SchemaError(f"column {spec.name!r} already exists") self._validate_column_spec(spec) if isinstance(spec, ListColumnSpec): is_list = True cannot_backfill = True else: is_list = False cannot_backfill = spec.is_boolean if cannot_backfill and self.nrows > 0: kind = "list" if is_list else "boolean" raise SchemaError( f"cannot add {kind} column {spec.name!r} to a table with rows: " "it cannot represent the pre-existing rows as missing" ) datasets = {n: o for n, o in cols.items() if isinstance(o, h5py.Dataset)} obj = self._create_one_column( self._group, spec, default_chunk_bytes=default_chunk_bytes ) if not is_list: extent = next(iter({d.shape[0] for d in datasets.values()}), self.nrows) extend_to(obj, extent) order = read_str_array_attr(self._group, ATTR_COLUMN_ORDER) or list(cols) if ATTR_COLUMN_ORDER in self._group.attrs: del self._group.attrs[ATTR_COLUMN_ORDER] write_utf8_array_attr(self._group, ATTR_COLUMN_ORDER, [*order, spec.name]) return self._wrap(obj)
def _looks_boolean(dataset: Any) -> bool: """Return True if *dataset* holds an H5Col boolean column. Parameters ---------- dataset: An open column dataset. Only its ``dtype`` is read, and the decision rests entirely with :func:`~h5col.booleans.is_bool_dtype`. """ from .booleans import is_bool_dtype return is_bool_dtype(dataset.dtype) def _infer_column_spec(name: str, arr: Any) -> ColumnSpec: """Derive a :class:`~h5col.ColumnSpec` for *name* from the values in *arr*. Parameters ---------- name: The name the returned spec carries. It is used for nothing else here and is not validated; :meth:`Table.create` checks it later. arr: The column's values, read through ``np.asarray``. A boolean array gives the H5Col boolean dtype. Strings and bytes — including an object array whose values are all ``str`` or ``bytes`` — give a fixed-length string sized to the longest value in UTF-8 bytes, and at least one byte wide. Anything else keeps the array's own dtype. """ a = np.asarray(arr) if a.dtype.kind == "b": return ColumnSpec(name=name, dtype=bool_dtype()) if a.dtype.kind in ("U", "S") or ( a.dtype.kind == "O" and all(isinstance(v, str | bytes) for v in a.ravel()) ): vals = a.ravel().tolist() max_bytes = max( (len(v.encode("utf-8") if isinstance(v, str) else v) for v in vals), default=1, ) return ColumnSpec(name=name, dtype=FixedString(max(1, max_bytes))) return ColumnSpec(name=name, dtype=a.dtype)