Skip to content

helpers

Small functions that more than one notebook can use. Each one below has a small demo. A helper written for a single notebook says which one, with a link.

Sentences from tables

The sentences of a notebook that quote its tables take their numbers from the data. The demos use a table with one row per type.

shares = pl.DataFrame(
    {"type": ["LC10a", "LC9", "LC11"], "share": [0.08, 0.13, 0.11]}
)
shares
shape: (3, 2)
typeshare
strf64
"LC10a"0.08
"LC9"0.13
"LC11"0.11

and_list joins names into a phrase:

def and_list(items: Sequence[str]) -> str:
    """items joined as a phrase: "a", "a and b", "a, b, and c"."""
    if len(items) <= 2:
        return " and ".join(items)
    return ", ".join(items[:-1]) + ", and " + items[-1]
and_list(shares["type"].to_list())
'LC10a, LC9, and LC11'

span_of gives the smallest and the largest value of a column, each with its label from the column by, formatted by fmt:

def span_of(
    table: pl.DataFrame, column: str, *, by: str = "type", fmt: str = ""
) -> str:
    """The smallest and the largest value of column, each with its
    label from the column by, as "8% (LC10a) to 13% (LC9)".

    fmt is the format spec of the values.
    """
    ordered = table.sort(column)
    low = ordered.row(0, named=True)
    high = ordered.row(-1, named=True)
    return (
        f"{low[column]:{fmt}} ({low[by]}) to {high[column]:{fmt}} ({high[by]})"
    )
span_of(shares, "share", fmt=".0%")
'8% (LC10a) to 13% (LC9)'

value_list gives every row's value of a column with its label, for a sentence that reports one value per group. template says how one item reads, with {label} and {value} in it:

def value_list(
    table: pl.DataFrame,
    column: str,
    *,
    by: str = "type",
    fmt: str = "",
    template: str = "{label} {value}",
) -> str:
    """The rows of table as a phrase, each as template filled with its
    label (from the column by) and its value of column (formatted by
    fmt), in the order of the rows: "LC10a 8%, LC9 13%, and LC11 11%".
    """
    return and_list(
        [
            template.format(label=row[by], value=f"{row[column]:{fmt}}")
            for row in table.iter_rows(named=True)
        ]
    )
value_list(shares, "share", fmt=".0%")
'LC10a 8%, LC9 13%, and LC11 11%'

in_order sorts the rows of a table by a given order of its labels, so that a sentence lists the types the way the notebook does everywhere else and not alphabetically. A row whose label is not in the order goes last:

def in_order(
    table: pl.DataFrame, order: Sequence[str], *, by: str = "type"
) -> pl.DataFrame:
    """The rows of table sorted by the position of their value of the
    column by in order; rows with the same value keep their order, and
    a value not in order goes last."""
    return table.sort(
        pl.col(by).replace_strict(
            {value: n for n, value in enumerate(order)}, default=len(order)
        ),
        maintain_order=True,
    )
in_order(shares, ["LC9", "LC11", "LC10a"])
shape: (3, 2)
typeshare
strf64
"LC9"0.13
"LC11"0.11
"LC10a"0.08

Cells of a type

A census says which type each cell is, and a table of sides says which hemisphere it is in. cells_of_types lists the cells of the given types in one hemisphere, as a dictionary of type name to root ids. The demos use five cells of two types.

demo_census = pl.DataFrame(
    {
        "root_id": [1, 2, 3, 4, 5],
        "primary_type": ["A", "A", "A", "B", "C"],
    }
)
demo_sides = pl.DataFrame(
    {
        "root_id": [1, 2, 3, 4],
        "side": ["right", "left", "right", "right"],
    }
)
demo_census
shape: (5, 2)
root_idprimary_type
i64str
1"A"
2"A"
3"A"
4"B"
5"C"
def cells_of_types(
    census: pl.DataFrame,
    sides: pl.DataFrame,
    types: Sequence[str],
    *,
    side: str | None = "right",
) -> dict[str, list[int]]:
    """The root ids of the cells of each type, sorted, as {type: ids}.

    census is DataFrame[root_id, primary_type] and sides is
    DataFrame[root_id, side, ...].
    side keeps the cells on that side; None keeps every cell, and a
    cell that sides does not list is then kept too.
    A type with no cells maps to an empty list.
    """
    cells = census.filter(pl.col("primary_type").is_in(list(types)))
    if side is not None:
        cells = cells.join(
            sides.filter(pl.col("side") == side).select("root_id"),
            on="root_id",
        )
    return {
        name: cells.filter(pl.col("primary_type") == name)["root_id"]
        .sort()
        .to_list()
        for name in types
    }
cells_of_types(demo_census, demo_sides, ["A", "B", "C"])
  • A: list · 2 items
    • 0: 1
    • 1: 3
  • B: list · 1 item
    • 0: 4
  • C: []

Cell columns on connections

A table of connections has one row per pair of cells and names each cell only by its id, in pre_pt_root_id and post_pt_root_id. Whatever is known about a cell, such as its type or hemisphere, can be added as columns of the connections it starts or ends. The demos use three connections between four cells.

demo_connections = pl.DataFrame(
    {
        "pre_pt_root_id": [1, 1, 2],
        "post_pt_root_id": [3, 4, 3],
        "weight": [5, 2, 7],
    }
)
demo_connections
shape: (3, 3)
pre_pt_root_idpost_pt_root_idweight
i64i64i64
135
142
237

with_cell_columns takes a table of cells, a root_id column and any other columns, and adds the other columns to the connections, prefixed pre_ for the presynaptic cell or post_ for the postsynaptic one (end). A cell missing from the table gets nulls. Here the postsynaptic cells get a type and a hemisphere:

def with_cell_columns(
    connections: pl.DataFrame, cells: pl.DataFrame, *, end: str = "pre"
) -> pl.DataFrame:
    """connections with the columns of cells added for one end of each
    connection, named pre_<column> or post_<column>.

    cells is DataFrame[root_id, ...] with one row per cell, since a
    repeated root_id would repeat the connections of that cell.
    end is "pre" or "post"; a cell missing from cells gets nulls.
    """
    if end not in ("pre", "post"):
        msg = f"end must be 'pre' or 'post', not {end!r}"
        raise ValueError(msg)
    if cells["root_id"].is_duplicated().any():
        msg = "root_id repeats in cells; each cell needs one row"
        raise ValueError(msg)
    prefixed = cells.rename(
        {
            column: f"{end}_{column}"
            for column in cells.columns
            if column != "root_id"
        }
    )
    return connections.join(
        prefixed,
        left_on=f"{end}_pt_root_id",
        right_on="root_id",
        how="left",
        maintain_order="left",
    )
with_cell_columns(
    demo_connections,
    pl.DataFrame(
        {
            "root_id": [3, 4],
            "type": ["Tm1", "Tm2"],
            "side": ["right", "left"],
        }
    ),
    end="post",
)
shape: (3, 5)
pre_pt_root_idpost_pt_root_idweightpost_typepost_side
i64i64i64strstr
135"Tm1""right"
142"Tm2""left"
237"Tm1""right"

The common case is the type of a cell, given as a dictionary of type name to the ids of its cells. type_table turns such a dictionary into a table of cells, and a cell listed under two types is an error:

def type_table(cells_by_type: dict[str, list]) -> pl.DataFrame:
    """DataFrame[root_id, type], one row per cell of cells_by_type.

    Raises ValueError for a cell listed under two types.
    """
    rows = [
        (cell, name) for name, ids in cells_by_type.items() for cell in ids
    ]
    table = pl.DataFrame(rows, schema=["root_id", "type"], orient="row")
    if table["root_id"].is_duplicated().any():
        msg = "a cell is listed under two types (root_id repeats)"
        raise ValueError(msg)
    return table
type_table({"A": [1, 2], "B": [3]})
shape: (3, 2)
root_idtype
i64str
1"A"
2"A"
3"B"

with_pre_type and with_post_type add the type of the presynaptic or postsynaptic cell, in the column pre_type or post_type. A cell in no type gets a null:

def with_pre_type(
    connections: pl.DataFrame, cells_by_type: dict[str, list]
) -> pl.DataFrame:
    """connections with the column pre_type: the type of the
    presynaptic cell, null for a cell in no type.

    cells_by_type maps a type name to the ids of its cells; a cell may
    belong to only one type.
    """
    return with_cell_columns(connections, type_table(cells_by_type), end="pre")
def with_post_type(
    connections: pl.DataFrame, cells_by_type: dict[str, list]
) -> pl.DataFrame:
    """connections with the column post_type: the type of the
    postsynaptic cell, null for a cell in no type.

    cells_by_type maps a type name to the ids of its cells; a cell may
    belong to only one type.
    """
    return with_cell_columns(
        connections, type_table(cells_by_type), end="post"
    )
with_post_type(
    with_pre_type(demo_connections, {"A": [1], "B": [2]}),
    {"C": [3, 4]},
)
shape: (3, 5)
pre_pt_root_idpost_pt_root_idweightpre_typepost_type
i64i64i64strstr
135"A""C"
142"A""C"
237"B""C"

reversed_connections swaps the two ends of every connection, so the cells that send to a type become the ones it "starts at". Typing the reversed table with with_pre_type then labels the type of the receiving cell, and a computation over the cells a connection starts at (such as a strength) becomes one over the cells it ends at:

def reversed_connections(connections: pl.DataFrame) -> pl.DataFrame:
    """connections with pre_pt_root_id and post_pt_root_id swapped;
    every other column stays with its row."""
    return connections.rename(
        {
            "pre_pt_root_id": "post_pt_root_id",
            "post_pt_root_id": "pre_pt_root_id",
        }
    )
reversed_connections(demo_connections)
shape: (3, 3)
post_pt_root_idpre_pt_root_idweight
i64i64i64
135
142
237

Overlap of two sets

The Jaccard index of two sets is the share of their union that they have in common,

\[ J(A, B) = \frac{|A \cap B|}{|A \cup B|}, \]

1 for identical sets and 0 for disjoint ones. Two empty sets share nothing, so jaccard gives them 0.

def jaccard(a: set, b: set) -> float:
    """|a & b| / |a | b|, and 0 for two empty sets."""
    union = a | b
    if not union:
        return 0.0
    return len(a & b) / len(union)
# These two share two of their four members.
jaccard({"A", "B", "C"}, {"B", "C", "D"})
0.5

Complementary cumulative distributions

The complementary cumulative distribution function (CCDF) of a sample \(x_1, \dots, x_N\) is the fraction of the sample at or above a mark \(m\),

\[ F(m) = \frac{1}{N} \sum_{k=1}^{N} [x_k \ge m], \]

where \([x_k \ge m]\) is 1 when the condition holds and 0 otherwise. A heavy tail is easiest to read on log-log axes. Between two marks \(m_a < m_b\) the local slope of that curve is

\[ \beta(m_a, m_b) = \frac{\ln F(m_b) - \ln F(m_a)}{\ln m_b - \ln m_a}. \]

A power law has constant \(\beta\). The demos use ten values.

sample = np.array([1, 1, 1, 2, 2, 3, 5, 8, 13, 40])
sample
array([ 1,  1,  1,  2,  2,  3,  5,  8, 13, 40])

ccdf_at computes \(F\) at the given marks:

def ccdf_at(sample: ArrayLike, marks: Sequence[float]) -> np.ndarray:
    """F(m) for each m in marks: the fraction of the sample at or above
    m."""
    ordered = np.sort(np.asarray(sample))
    if len(ordered) == 0:
        msg = "ccdf_at needs at least one value"
        raise ValueError(msg)
    below = np.searchsorted(ordered, marks, side="left")
    return (len(ordered) - below) / len(ordered)
ccdf_at(sample, [1, 2, 5, 10])
array([1. , 0.7, 0.4, 0.2])

local_slopes computes \(\beta\) for each pair of successive marks:

def local_slopes(sample: ArrayLike, marks: Sequence[int]) -> pl.DataFrame:
    """DataFrame[lower, upper, slope]: the slope of the log-log CCDF
    between successive marks; NaN where no value reaches the upper
    mark."""
    fraction = ccdf_at(sample, marks)
    lower = np.array(marks[:-1], dtype=float)
    upper = np.array(marks[1:], dtype=float)
    with np.errstate(divide="ignore", invalid="ignore"):
        slope = np.log(fraction[1:] / fraction[:-1]) / np.log(upper / lower)
    slope[~np.isfinite(slope)] = np.nan
    return pl.DataFrame({"lower": lower, "upper": upper, "slope": slope})
local_slopes(sample, [1, 2, 5, 10])
shape: (3, 3)
lowerupperslope
f64f64f64
1.02.0-0.514573
2.05.0-0.61074
5.010.0-1.0

ccdf lists the curve itself, one row per distinct value with the fraction of the sample at or above it:

def ccdf(sample: ArrayLike) -> pl.DataFrame:
    """DataFrame[value, fraction]: for each distinct value, the
    fraction of the sample at or above it."""
    distinct = np.unique(np.asarray(sample))
    return pl.DataFrame(
        {"value": distinct, "fraction": ccdf_at(sample, distinct)}
    )
ccdf(sample)
shape: (7, 2)
valuefraction
i64f64
11.0
20.7
30.5
50.4
80.3
130.2
400.1