helpers¶
Small functions that more than one notebook can use. Each one below has a small demo. A helper written for a single notebook says which one, with a link.
Sentences from tables¶
The sentences of a notebook that quote its tables take their numbers from the data. The demos use a table with one row per type.
| type | share |
|---|---|
| str | f64 |
| "LC10a" | 0.08 |
| "LC9" | 0.13 |
| "LC11" | 0.11 |
and_list joins names into a phrase:
def and_list(items: Sequence[str]) -> str:
"""items joined as a phrase: "a", "a and b", "a, b, and c"."""
if len(items) <= 2:
return " and ".join(items)
return ", ".join(items[:-1]) + ", and " + items[-1]
'LC10a, LC9, and LC11'
span_of gives the smallest and the largest value of a column,
each with its label from the column by, formatted by fmt:
def span_of(
table: pl.DataFrame, column: str, *, by: str = "type", fmt: str = ""
) -> str:
"""The smallest and the largest value of column, each with its
label from the column by, as "8% (LC10a) to 13% (LC9)".
fmt is the format spec of the values.
"""
ordered = table.sort(column)
low = ordered.row(0, named=True)
high = ordered.row(-1, named=True)
return (
f"{low[column]:{fmt}} ({low[by]}) to {high[column]:{fmt}} ({high[by]})"
)
'8% (LC10a) to 13% (LC9)'
value_list gives every row's value of a column with its label,
for a sentence that reports one value per group.
template says how one item reads,
with {label} and {value} in it:
def value_list(
table: pl.DataFrame,
column: str,
*,
by: str = "type",
fmt: str = "",
template: str = "{label} {value}",
) -> str:
"""The rows of table as a phrase, each as template filled with its
label (from the column by) and its value of column (formatted by
fmt), in the order of the rows: "LC10a 8%, LC9 13%, and LC11 11%".
"""
return and_list(
[
template.format(label=row[by], value=f"{row[column]:{fmt}}")
for row in table.iter_rows(named=True)
]
)
'LC10a 8%, LC9 13%, and LC11 11%'
in_order sorts the rows of a table by a given order of its
labels, so that a sentence lists the types the way the notebook
does everywhere else and not alphabetically.
A row whose label is not in the order goes last:
def in_order(
table: pl.DataFrame, order: Sequence[str], *, by: str = "type"
) -> pl.DataFrame:
"""The rows of table sorted by the position of their value of the
column by in order; rows with the same value keep their order, and
a value not in order goes last."""
return table.sort(
pl.col(by).replace_strict(
{value: n for n, value in enumerate(order)}, default=len(order)
),
maintain_order=True,
)
| type | share |
|---|---|
| str | f64 |
| "LC9" | 0.13 |
| "LC11" | 0.11 |
| "LC10a" | 0.08 |
Cells of a type¶
A census says which type each cell is,
and a table of sides says which hemisphere it is in.
cells_of_types lists the cells of the given types in one
hemisphere, as a dictionary of type name to root ids.
The demos use five cells of two types.
demo_census = pl.DataFrame(
{
"root_id": [1, 2, 3, 4, 5],
"primary_type": ["A", "A", "A", "B", "C"],
}
)
demo_sides = pl.DataFrame(
{
"root_id": [1, 2, 3, 4],
"side": ["right", "left", "right", "right"],
}
)
demo_census
| root_id | primary_type |
|---|---|
| i64 | str |
| 1 | "A" |
| 2 | "A" |
| 3 | "A" |
| 4 | "B" |
| 5 | "C" |
def cells_of_types(
census: pl.DataFrame,
sides: pl.DataFrame,
types: Sequence[str],
*,
side: str | None = "right",
) -> dict[str, list[int]]:
"""The root ids of the cells of each type, sorted, as {type: ids}.
census is DataFrame[root_id, primary_type] and sides is
DataFrame[root_id, side, ...].
side keeps the cells on that side; None keeps every cell, and a
cell that sides does not list is then kept too.
A type with no cells maps to an empty list.
"""
cells = census.filter(pl.col("primary_type").is_in(list(types)))
if side is not None:
cells = cells.join(
sides.filter(pl.col("side") == side).select("root_id"),
on="root_id",
)
return {
name: cells.filter(pl.col("primary_type") == name)["root_id"]
.sort()
.to_list()
for name in types
}
- A: list · 2 items
- 0: 1
- 1: 3
- B: list · 1 item
- 0: 4
- C: []
Cell columns on connections¶
A table of connections has one row per pair of cells and names
each cell only by its id,
in pre_pt_root_id and post_pt_root_id.
Whatever is known about a cell,
such as its type or hemisphere,
can be added as columns of the connections it starts or ends.
The demos use three connections between four cells.
demo_connections = pl.DataFrame(
{
"pre_pt_root_id": [1, 1, 2],
"post_pt_root_id": [3, 4, 3],
"weight": [5, 2, 7],
}
)
demo_connections
| pre_pt_root_id | post_pt_root_id | weight |
|---|---|---|
| i64 | i64 | i64 |
| 1 | 3 | 5 |
| 1 | 4 | 2 |
| 2 | 3 | 7 |
with_cell_columns takes a table of cells,
a root_id column and any other columns,
and adds the other columns to the connections,
prefixed pre_ for the presynaptic cell or post_ for the
postsynaptic one (end).
A cell missing from the table gets nulls.
Here the postsynaptic cells get a type and a hemisphere:
def with_cell_columns(
connections: pl.DataFrame, cells: pl.DataFrame, *, end: str = "pre"
) -> pl.DataFrame:
"""connections with the columns of cells added for one end of each
connection, named pre_<column> or post_<column>.
cells is DataFrame[root_id, ...] with one row per cell, since a
repeated root_id would repeat the connections of that cell.
end is "pre" or "post"; a cell missing from cells gets nulls.
"""
if end not in ("pre", "post"):
msg = f"end must be 'pre' or 'post', not {end!r}"
raise ValueError(msg)
if cells["root_id"].is_duplicated().any():
msg = "root_id repeats in cells; each cell needs one row"
raise ValueError(msg)
prefixed = cells.rename(
{
column: f"{end}_{column}"
for column in cells.columns
if column != "root_id"
}
)
return connections.join(
prefixed,
left_on=f"{end}_pt_root_id",
right_on="root_id",
how="left",
maintain_order="left",
)
with_cell_columns(
demo_connections,
pl.DataFrame(
{
"root_id": [3, 4],
"type": ["Tm1", "Tm2"],
"side": ["right", "left"],
}
),
end="post",
)
| pre_pt_root_id | post_pt_root_id | weight | post_type | post_side |
|---|---|---|---|---|
| i64 | i64 | i64 | str | str |
| 1 | 3 | 5 | "Tm1" | "right" |
| 1 | 4 | 2 | "Tm2" | "left" |
| 2 | 3 | 7 | "Tm1" | "right" |
The common case is the type of a cell,
given as a dictionary of type name to the ids of its cells.
type_table turns such a dictionary into a table of cells,
and a cell listed under two types is an error:
def type_table(cells_by_type: dict[str, list]) -> pl.DataFrame:
"""DataFrame[root_id, type], one row per cell of cells_by_type.
Raises ValueError for a cell listed under two types.
"""
rows = [
(cell, name) for name, ids in cells_by_type.items() for cell in ids
]
table = pl.DataFrame(rows, schema=["root_id", "type"], orient="row")
if table["root_id"].is_duplicated().any():
msg = "a cell is listed under two types (root_id repeats)"
raise ValueError(msg)
return table
| root_id | type |
|---|---|
| i64 | str |
| 1 | "A" |
| 2 | "A" |
| 3 | "B" |
with_pre_type and with_post_type add the type of the
presynaptic or postsynaptic cell, in the column pre_type or
post_type.
A cell in no type gets a null:
def with_pre_type(
connections: pl.DataFrame, cells_by_type: dict[str, list]
) -> pl.DataFrame:
"""connections with the column pre_type: the type of the
presynaptic cell, null for a cell in no type.
cells_by_type maps a type name to the ids of its cells; a cell may
belong to only one type.
"""
return with_cell_columns(connections, type_table(cells_by_type), end="pre")
def with_post_type(
connections: pl.DataFrame, cells_by_type: dict[str, list]
) -> pl.DataFrame:
"""connections with the column post_type: the type of the
postsynaptic cell, null for a cell in no type.
cells_by_type maps a type name to the ids of its cells; a cell may
belong to only one type.
"""
return with_cell_columns(
connections, type_table(cells_by_type), end="post"
)
| pre_pt_root_id | post_pt_root_id | weight | pre_type | post_type |
|---|---|---|---|---|
| i64 | i64 | i64 | str | str |
| 1 | 3 | 5 | "A" | "C" |
| 1 | 4 | 2 | "A" | "C" |
| 2 | 3 | 7 | "B" | "C" |
reversed_connections swaps the two ends of every connection,
so the cells that send to a type become the ones it "starts at".
Typing the reversed table with with_pre_type then labels the type
of the receiving cell,
and a computation over the cells a connection starts at
(such as a strength) becomes one over the cells it ends at:
def reversed_connections(connections: pl.DataFrame) -> pl.DataFrame:
"""connections with pre_pt_root_id and post_pt_root_id swapped;
every other column stays with its row."""
return connections.rename(
{
"pre_pt_root_id": "post_pt_root_id",
"post_pt_root_id": "pre_pt_root_id",
}
)
| post_pt_root_id | pre_pt_root_id | weight |
|---|---|---|
| i64 | i64 | i64 |
| 1 | 3 | 5 |
| 1 | 4 | 2 |
| 2 | 3 | 7 |
Overlap of two sets¶
The Jaccard index of two sets is the share of their union that they have in common,
1 for identical sets and 0 for disjoint ones.
Two empty sets share nothing, so jaccard gives them 0.
def jaccard(a: set, b: set) -> float:
"""|a & b| / |a | b|, and 0 for two empty sets."""
union = a | b
if not union:
return 0.0
return len(a & b) / len(union)
0.5
Complementary cumulative distributions¶
The complementary cumulative distribution function (CCDF) of a sample \(x_1, \dots, x_N\) is the fraction of the sample at or above a mark \(m\),
where \([x_k \ge m]\) is 1 when the condition holds and 0 otherwise. A heavy tail is easiest to read on log-log axes. Between two marks \(m_a < m_b\) the local slope of that curve is
A power law has constant \(\beta\). The demos use ten values.
array([ 1, 1, 1, 2, 2, 3, 5, 8, 13, 40])
ccdf_at computes \(F\) at the given marks:
def ccdf_at(sample: ArrayLike, marks: Sequence[float]) -> np.ndarray:
"""F(m) for each m in marks: the fraction of the sample at or above
m."""
ordered = np.sort(np.asarray(sample))
if len(ordered) == 0:
msg = "ccdf_at needs at least one value"
raise ValueError(msg)
below = np.searchsorted(ordered, marks, side="left")
return (len(ordered) - below) / len(ordered)
array([1. , 0.7, 0.4, 0.2])
local_slopes computes \(\beta\) for each pair of successive marks:
def local_slopes(sample: ArrayLike, marks: Sequence[int]) -> pl.DataFrame:
"""DataFrame[lower, upper, slope]: the slope of the log-log CCDF
between successive marks; NaN where no value reaches the upper
mark."""
fraction = ccdf_at(sample, marks)
lower = np.array(marks[:-1], dtype=float)
upper = np.array(marks[1:], dtype=float)
with np.errstate(divide="ignore", invalid="ignore"):
slope = np.log(fraction[1:] / fraction[:-1]) / np.log(upper / lower)
slope[~np.isfinite(slope)] = np.nan
return pl.DataFrame({"lower": lower, "upper": upper, "slope": slope})
| lower | upper | slope |
|---|---|---|
| f64 | f64 | f64 |
| 1.0 | 2.0 | -0.514573 |
| 2.0 | 5.0 | -0.61074 |
| 5.0 | 10.0 | -1.0 |
ccdf lists the curve itself, one row per distinct value with the
fraction of the sample at or above it:
def ccdf(sample: ArrayLike) -> pl.DataFrame:
"""DataFrame[value, fraction]: for each distinct value, the
fraction of the sample at or above it."""
distinct = np.unique(np.asarray(sample))
return pl.DataFrame(
{"value": distinct, "fraction": ccdf_at(sample, distinct)}
)
| value | fraction |
|---|---|
| i64 | f64 |
| 1 | 1.0 |
| 2 | 0.7 |
| 3 | 0.5 |
| 5 | 0.4 |
| 8 | 0.3 |
| 13 | 0.2 |
| 40 | 0.1 |