Skip to content

cells

champalimaud.cells

The cells of each type: which root ids belong to a type.

cells_of_types(census, sides, types, *, side='right')

Collect the root ids of the cells of each type.

Parameters:

Name Type Description Default
census DataFrame

Columns root_id and primary_type.

required
sides DataFrame

Columns root_id and side, and others.

required
types sequence of str

The primary_type values to collect.

required
side str or None

Keep the cells on that side. None keeps every cell, including a cell that sides does not list.

"right"

Returns:

Type Description
dict of str to list of int

The root ids of each type, sorted. A type with no cells maps to an empty list.

Examples:

>>> import polars as pl
>>> census = pl.DataFrame(
...     {
...         "root_id": [3, 1, 2],
...         "primary_type": ["A", "A", "B"],
...     }
... )
>>> sides = pl.DataFrame(
...     {"root_id": [1, 2, 3], "side": ["right", "right", "left"]}
... )
>>> cells_of_types(census, sides, ["A", "B", "C"])
{'A': [1], 'B': [2], 'C': []}
Source code in champalimaud/cells.py
def cells_of_types(
    census: pl.DataFrame,
    sides: pl.DataFrame,
    types: Sequence[str],
    *,
    side: str | None = "right",
) -> dict[str, list[int]]:
    """Collect the root ids of the cells of each type.

    Parameters
    ----------
    census : polars.DataFrame
        Columns ``root_id`` and ``primary_type``.
    sides : polars.DataFrame
        Columns ``root_id`` and ``side``, and others.
    types : sequence of str
        The ``primary_type`` values to collect.
    side : str or None, default "right"
        Keep the cells on that side.
        ``None`` keeps every cell, including a cell that `sides` does
        not list.

    Returns
    -------
    dict of str to list of int
        The root ids of each type, sorted.
        A type with no cells maps to an empty list.

    Examples
    --------
    >>> import polars as pl
    >>> census = pl.DataFrame(
    ...     {
    ...         "root_id": [3, 1, 2],
    ...         "primary_type": ["A", "A", "B"],
    ...     }
    ... )
    >>> sides = pl.DataFrame(
    ...     {"root_id": [1, 2, 3], "side": ["right", "right", "left"]}
    ... )
    >>> cells_of_types(census, sides, ["A", "B", "C"])
    {'A': [1], 'B': [2], 'C': []}
    """
    cells = census.filter(pl.col("primary_type").is_in(list(types)))
    if side is not None:
        cells = cells.join(
            sides.filter(pl.col("side") == side).select("root_id"),
            on="root_id",
        )
    return {
        name: cells.filter(pl.col("primary_type") == name)["root_id"]
        .sort()
        .to_list()
        for name in types
    }

type_table(cells_by_type)

Tabulate the type of each cell.

Parameters:

Name Type Description Default
cells_by_type dict of str to list

Maps a type name to the ids of its cells.

required

Returns:

Type Description
DataFrame

Columns root_id and type, one row per cell.

Raises:

Type Description
ValueError

For a cell listed under two types.

Examples:

>>> type_table({"A": [1, 2], "B": [3]}).rows()
[(1, 'A'), (2, 'A'), (3, 'B')]
Source code in champalimaud/cells.py
def type_table(cells_by_type: dict[str, list]) -> pl.DataFrame:
    """Tabulate the type of each cell.

    Parameters
    ----------
    cells_by_type : dict of str to list
        Maps a type name to the ids of its cells.

    Returns
    -------
    polars.DataFrame
        Columns ``root_id`` and ``type``, one row per cell.

    Raises
    ------
    ValueError
        For a cell listed under two types.

    Examples
    --------
    >>> type_table({"A": [1, 2], "B": [3]}).rows()
    [(1, 'A'), (2, 'A'), (3, 'B')]
    """
    rows = [
        (cell, name) for name, ids in cells_by_type.items() for cell in ids
    ]
    table = pl.DataFrame(rows, schema=["root_id", "type"], orient="row")
    if table["root_id"].is_duplicated().any():
        msg = "a cell is listed under two types (root_id repeats)"
        raise ValueError(msg)
    return table