load¶
Reads what collect wrote and reshapes it.
A missing file raises with the collect function that writes it,
instead of returning an empty table that would look like a real
result.
Tables come back as polars frames.
def require(path):
"""Return path, or stop with the collect function that writes it."""
if not path.exists():
msg = f"{path} missing; run collect.{WRITERS[path]}"
raise SystemExit(msg)
return path
What require says for each data file right now:
ok, or the error a reader would see.
_rows = []
for _path in WRITERS:
try:
require(_path)
_result = "ok"
except SystemExit as _e:
_result = str(_e)
_rows.append(
{"file": str(_path.relative_to(Cfg.ROOT)), "require": _result}
)
pl.DataFrame(_rows)
| file | require |
|---|---|
| str | str |
| "data/cell_types.csv.gz" | "ok" |
| "data/codex/visual_neuron_types… | "ok" |
| "data/codex/column_assignment.c… | "ok" |
| "data/codex/classification.csv.… | "ok" |
| "data/connections.parquet" | "ok" |
| "data/skeletons" | "ok" |
Census¶
A cell's identity is a root id plus a type name,
and load_census reads the type of every typed cell.
The file also has additional_type(s);
analysis keys on primary_type only.
def load_census() -> pl.DataFrame:
"""Cfg.CENSUS -> DataFrame[root_id, primary_type]."""
return pl.read_csv(
require(Cfg.CENSUS),
columns=["root_id", "primary_type"],
schema_overrides={"root_id": pl.Int64},
)
| root_id | primary_type |
|---|---|
| i64 | str |
| 720575940599457990 | "T4b" |
| 720575940599763910 | "T5b" |
| 720575940600623020 | "Dm3q" |
| 720575940602564320 | "KCg-m" |
| 720575940602720940 | "JO-CA" |
| … | … |
| 720575940661275009 | "KCab-p" |
| 720575940661281409 | "L4" |
| 720575940661285249 | "LPC2" |
| 720575940661304449 | "Lawf1" |
| 720575940661333889 | "CB2303" |
The census has 138,327 cells in 8,772 types.
root_ids_of_type lists the root ids of one type,
as Python ints because CAVE rejects numpy integers:
def root_ids_of_type(census: pl.DataFrame, cell_type: str) -> list[int]:
"""The root ids of one primary_type in the census."""
return census.filter(pl.col("primary_type") == cell_type)[
"root_id"
].to_list()
- 0: 720575940606173630
- 1: 720575940609602770
- 2: 720575940611945170
- 3: 720575940615124370
- 4: 720575940618214360
Proofread neurons¶
FlyWire segments the volume automatically and people proofread the
result.
A segment that has a synapse is a partner of the cells it connects
to whether or not it was proofread,
and a segment that is not in the classification table is an
unproofread fragment, possibly a piece of a neuron.
The census types (almost) only proofread neurons,
so a root id missing from the census is usually a fragment,
not an untyped neuron.
load_proofread_ids lists the proofread neurons:
def load_proofread_ids() -> pl.DataFrame:
"""Cfg.CLASSIFICATION -> DataFrame[root_id] of every proofread
neuron."""
return pl.read_csv(
require(Cfg.CLASSIFICATION),
columns=["root_id"],
schema_overrides={"root_id": pl.Int64},
)
proofread_ids = load_proofread_ids()
_typed = census["root_id"].is_in(proofread_ids["root_id"].to_list())
pl.DataFrame(
{
"proofread neurons": [proofread_ids.height],
"of them typed (in the census)": [int(_typed.sum())],
"census cells not proofread": [int((~_typed).sum())],
}
)
| proofread neurons | of them typed (in the census) | census cells not proofread |
|---|---|---|
| i64 | i64 | i64 |
| 139255 | 138327 | 0 |
Hemisphere and columns¶
visual_neuron_types assigns each root id a side.
FlyWire's left and right are proofread unevenly,
so analyses that pick one side read it from here
rather than guessing from coordinates.
column_assignment gives a columnar cell its visual-column address
\((x, y)\) on the left or right retinotopic grid.
LC cells are not in that table;
their retinotopy has to be inferred from columnar partners
(see lc_output_clusters).
def load_visual_types() -> pl.DataFrame:
"""Cfg.VISUAL_TYPES -> DataFrame[root_id, type, side, ...] (the
other columns as in the file)."""
return pl.read_csv(
require(Cfg.VISUAL_TYPES), schema_overrides={"root_id": pl.Int64}
)
visual_types = load_visual_types()
pl.DataFrame(
{
"census cells": [census.height],
"side labeled": [
int(
census["root_id"]
.is_in(visual_types["root_id"].to_list())
.sum()
)
],
"right": [
census.join(
visual_types.filter(pl.col("side") == "right"),
on="root_id",
).height
],
}
)
| census cells | side labeled | right |
|---|---|---|
| i64 | i64 | i64 |
| 138327 | 95046 | 47273 |
def load_columns() -> pl.DataFrame:
"""Cfg.COLUMNS -> DataFrame[root_id, hemisphere, type, column_id, x,
y, p, q] for columnar cells."""
return pl.read_csv(
require(Cfg.COLUMNS), schema_overrides={"root_id": pl.Int64}
)
| root_id | hemisphere | type | column_id | x | y | p | q |
|---|---|---|---|---|---|---|---|
| i64 | str | str | i64 | i64 | i64 | i64 | i64 |
| 720575940596125868 | "right" | "T5c" | 97 | -5 | 2 | 6 | -4 |
| 720575940599333574 | "right" | "Tm1" | 355 | -7 | -6 | 4 | -10 |
| 720575940599457990 | "right" | "T4b" | 247 | -4 | -15 | -4 | -11 |
| 720575940599459782 | "right" | "T5b" | 513 | -4 | -13 | -3 | -10 |
| 720575940599704006 | "right" | "T5a" | 331 | 1 | -15 | -9 | -6 |
| … | … | … | … | … | … | … | … |
| 720575940661281409 | "left" | "L4" | 385 | -3 | -18 | -6 | -12 |
| 720575940661284993 | "right" | "R8" | 694 | 4 | 3 | -3 | 6 |
| 720575940661297281 | "left" | "Tm21" | 514 | -4 | -10 | -1 | -9 |
| 720575940661319553 | "left" | "L3" | 634 | -3 | -15 | -5 | -10 |
| 720575940661323905 | "left" | "L4" | 647 | -4 | -7 | 0 | -7 |
Connections¶
A connection is a pair of cells with at least one synapse,
and its weight \(w(i \to j)\) is that synapse count,
as in the name of Codex's connections_princeton_no_threshold
table.
The table has one row per ordered pair with at least one proofread
end, sorted by presynaptic cell,
so it holds every connection of every proofread neuron.
The synapses counted are the set behind the Codex release,
so the weight between two proofread neurons equals the Codex
weight.
The table has about a hundred million rows,
so scan_connections returns a lazy frame,
and the notebook that reads it filters what it needs before
collecting;
polars then reads only the row groups that can hold those cells.
def scan_connections() -> pl.LazyFrame:
"""Cfg.CONNECTIONS, lazily:
LazyFrame[pre_pt_root_id, post_pt_root_id, weight], all Int64.
Filter before .collect()."""
return pl.scan_parquet(require(Cfg.CONNECTIONS)).with_columns(
pl.col("weight").cast(pl.Int64)
)
load_connections_of is the filter most notebooks apply:
the connections that start at some cells,
or end at them with end="post",
and at least min_weight synapses.
def load_connections_of(
root_ids: Sequence[int], *, end: str = "pre", min_weight: int = 1
) -> pl.DataFrame:
"""The connections of Cfg.CONNECTIONS that start at (end "pre") or
end at (end "post") one of root_ids, with a weight of at least
min_weight: DataFrame[pre_pt_root_id, post_pt_root_id, weight].
Raises ValueError for any other end.
"""
if end not in ("pre", "post"):
msg = f"end must be 'pre' or 'post', not {end!r}"
raise ValueError(msg)
return (
scan_connections()
.filter(
pl.col(f"{end}_pt_root_id").is_in(list(root_ids))
& (pl.col("weight") >= min_weight)
)
.collect()
)
| pre_pt_root_id | post_pt_root_id | weight |
|---|---|---|
| i64 | i64 | i64 |
| 720575940606173630 | 720575940606609696 | 24 |
| 720575940606173630 | 720575940613369586 | 25 |
| 720575940606173630 | 720575940613843522 | 10 |
| 720575940606173630 | 720575940615645034 | 16 |
| 720575940606173630 | 720575940617569403 | 13 |
| … | … | … |
| 720575940611945170 | 720575940623768840 | 19 |
| 720575940611945170 | 720575940626393361 | 10 |
| 720575940611945170 | 720575940626770258 | 12 |
| 720575940611945170 | 720575940630742521 | 14 |
| 720575940611945170 | 720575940654004897 | 22 |
The first ten connections of the smallest presynaptic root id; the filter is what keeps the read small:
_lazy = scan_connections()
_first = _lazy.select(pl.col("pre_pt_root_id").min()).collect().item()
_lazy.filter(pl.col("pre_pt_root_id") == _first).head(10).collect()
| pre_pt_root_id | post_pt_root_id | weight |
|---|---|---|
| i64 | i64 | i64 |
| 720575940379280990 | 720575940615583647 | 1 |
Most connections carry a synapse or two and a few carry hundreds, so the distribution of weights has a heavy tail. The chart counts the connections at each weight on a log axis, with the heaviest ones pooled in its last bar.
_clip = 100
_counts = (
scan_connections()
.select(pl.col("weight").clip(upper_bound=_clip))
.group_by("weight")
.len()
.collect()
)
alt.Chart(_counts).mark_bar().encode(
x=alt.X("weight:Q", title=f"weight (synapses), clipped at {_clip}"),
y=alt.Y(
"len:Q",
scale=alt.Scale(type="log"),
title="connections (log scale)",
),
tooltip=["weight:Q", alt.Tooltip("len:Q", format=",")],
).properties(width=420, height=160)
Skeletons¶
One SWC file per root id.
load_skeletons returns whatever collect stored for the
requested types, named <type>_<root_id>.
def load_skeletons(census: pl.DataFrame, types) -> list:
"""navis TreeNeurons for every stored skeleton of the given
types."""
folder = require(Cfg.SKELETONS)
type_of = dict(census.select("root_id", "primary_type").iter_rows())
neurons = []
for path in sorted(folder.glob("*.swc")):
root_id = int(path.stem)
cell_type = type_of.get(root_id)
if cell_type not in types:
continue
neuron = navis.read_swc(path)
neuron.name = f"{cell_type}_{root_id}"
neurons.append(neuron)
return neurons
- 0: 'LC16_720575940606173630'
- 1: 'LC16_720575940609602770'
- 2: 'LC16_720575940611945170'
- 3: 'LC16_720575940615124370'
- 4: 'LC16_720575940618214360'
- 5: 'LC16_720575940618798040'
Figures¶
figure_path is where a figure notebook writes a file,
and it creates the folder:
def figure_path(name: str):
"""Cfg.FIGURES / name, creating the folder."""
Cfg.FIGURES.mkdir(parents=True, exist_ok=True)
return Cfg.FIGURES / name
PosixPath('/home/jaeho/projects/champalimaud/figures/lc16_network.html')