Coverage for src/jointview/columns.py: 100%
32 statements
« prev ^ index » next coverage.py v7.15.4, created at 2026-10-01 04:48 +0000
« prev ^ index » next coverage.py v7.15.4, created at 2026-10-01 04:48 +0000
1"""Which columns the app can draw, and cutting a chosen pair down to its common sample.
3This is the vocabulary both halves of the app are built on, and it belongs to neither
4of them. :mod:`jointview.plot` draws what :func:`aligned` produces and
5:mod:`jointview.stats` summarises it, so the column names and the shaping rules live
6here rather than in either — two peers reaching into a third, instead of one reaching
7into the other.
9``PERIOD`` is the clearest case. It is the name :func:`aligned` writes the x-axis under
10and the name the statistics read it back out of: a data contract between the two, not
11a fact about plotting.
13``WINDOWS`` is here for the same reason one layer out. Cutting the sample to the year to
14date is not a fact about the drawing either: the window is taken off the frame once, and
15the lines, the tables and the base a rebased pair is indexed to all follow from that one
16cut instead of each making it themselves.
17"""
19from __future__ import annotations
21from collections.abc import Callable
23import polars as pl
25PERIOD = "period"
27WINDOW_ALL = "all"
29# The stretches of the sample the app offers above the plot, in the order it offers them,
30# each mapped to where it begins as a function of the last period in the frame.
31#
32# Measured back from that last period rather than from today: a parquet is as recent as
33# it is, and a window read off the clock would come back empty on a series that ends last
34# year — the same file being informative on Tuesday and blank on Wednesday.
35#
36# `None` is the whole sample. It is not "an offset of zero": it cuts nothing, and it is
37# also how a caller knows there is no cut to mention.
38WINDOWS: dict[str, Callable[[pl.Expr], pl.Expr] | None] = {
39 WINDOW_ALL: None,
40 # The turn of the year is a calendar boundary rather than a duration, which is why
41 # this one truncates where the others count backwards.
42 "ytd": lambda last: last.dt.truncate("1y"),
43 "12m": lambda last: last.dt.offset_by("-12mo"),
44 "36m": lambda last: last.dt.offset_by("-36mo"),
45}
48def series_columns(frame: pl.DataFrame) -> list[str]:
49 """The columns that can be drawn: every numeric one.
51 >>> import datetime as dt, polars as pl
52 >>> frame = pl.DataFrame({"date": [dt.date(2024, 1, 1)], "nav": [1.0], "label": ["a"]})
53 >>> series_columns(frame)
54 ['nav']
55 """
56 return [name for name, dtype in frame.schema.items() if dtype.is_numeric()]
59# Dates and datetimes, and deliberately not everything `dtype.is_temporal()` admits:
60# that predicate is also true of `Time` and `Duration`. A holding period or a time of day
61# is a temporal *quantity*, not a point on a calendar, and it cannot be the axis these
62# series are observed along — `aligned` writes whatever this returns under `PERIOD`, and
63# `stats` reads the annualisation factor off its spacing, so a column of timedeltas
64# arriving here is answered with a CAGR in the trillions rather than an error (#74).
65#
66# Compared with `==` rather than `isinstance`, which is what polars documents for asking
67# after a base type: `pl.Datetime` matches every time unit and time zone, so a
68# `Datetime("ns", "Europe/Zurich")` is admitted without any of that being spelled here.
69_DATE_LIKE = (pl.Date, pl.Datetime)
72def date_column(frame: pl.DataFrame) -> str | None:
73 """The first date or datetime column, which becomes the x-axis. None means row number.
75 >>> import datetime as dt, polars as pl
76 >>> date_column(pl.DataFrame({"when": [dt.date(2024, 1, 1)], "nav": [1.0]}))
77 'when'
78 >>> date_column(pl.DataFrame({"nav": [1.0]})) is None
79 True
81 A `Time` or `Duration` column is temporal but is not a date, and does not become the
82 axis — the rows get numbered instead, which is the documented fallback:
84 >>> date_column(pl.DataFrame({"held": [dt.timedelta(days=1)], "nav": [1.0]})) is None
85 True
86 """
87 return next((name for name, dtype in frame.schema.items() if dtype in _DATE_LIKE), None)
90def windowed(frame: pl.DataFrame, key: str = WINDOW_ALL) -> pl.DataFrame:
91 """``frame`` cut to one of :data:`WINDOWS`, measured back from its last period.
93 Applied to the frame rather than to the picture, so everything built from it agrees:
94 the lines, the numbers beside them, and the base a rebased pair is indexed to.
96 Raises:
97 KeyError: if ``key`` is not one of :data:`WINDOWS`.
99 >>> import datetime as dt, polars as pl
100 >>> frame = pl.DataFrame(
101 ... {
102 ... "date": [dt.date(2022, 12, 30), dt.date(2023, 12, 29), dt.date(2024, 6, 28)],
103 ... "nav": [90.0, 100.0, 110.0],
104 ... }
105 ... )
106 >>> windowed(frame, "ytd")["date"].to_list()
107 [datetime.date(2024, 6, 28)]
108 >>> windowed(frame, "12m").height
109 2
110 >>> windowed(frame, "all").height
111 3
113 A window is a stretch of calendar, so a frame numbered by row has nothing to cut
114 against and comes back whole:
116 >>> windowed(pl.DataFrame({"nav": [1.0, 2.0]}), "ytd").height
117 2
118 """
119 start = WINDOWS[key]
120 date = date_column(frame)
121 if start is None or date is None:
122 return frame
123 # `max()` rather than the last row: the frame reaching here need not be sorted, and
124 # `aligned` does its own sorting afterwards.
125 return frame.filter(pl.col(date) >= start(pl.col(date).max()))
128def default_pair(frame: pl.DataFrame) -> tuple[int, int]:
129 """Indices into :func:`series_columns` to open on — the first two series.
131 A frame with a single series opens on it twice, rather than refusing to draw.
133 >>> import polars as pl
134 >>> default_pair(pl.DataFrame({"a": [1.0], "b": [2.0]}))
135 (0, 1)
136 >>> default_pair(pl.DataFrame({"only": [1.0]}))
137 (0, 0)
138 """
139 names = series_columns(frame)
140 if not names:
141 raise ValueError("frame has no numeric columns to plot") # noqa: TRY003
142 return 0, 1 if len(names) > 1 else 0
145def aligned(frame: pl.DataFrame, a: str, b: str) -> pl.DataFrame:
146 """The two series on their common sample: ``period``, ``a``, ``b``, in order.
148 Both the picture and the summary tables are built from this, so the numbers
149 beside the chart always describe the lines in it. Renaming also sidesteps
150 Vega-Lite's field-shorthand escaping and lets ``a`` and ``b`` be the same column.
152 The three names are fixed whatever the columns were called, the rows come out
153 sorted by period, and a date where either series is missing is not part of the
154 sample — here the frame arrives unsorted and with a gap on the 2nd:
156 >>> import datetime as dt, polars as pl
157 >>> frame = pl.DataFrame(
158 ... {
159 ... "date": [dt.date(2024, 1, 3), dt.date(2024, 1, 1), dt.date(2024, 1, 2)],
160 ... "x": [3.0, 1.0, None],
161 ... "y": [30.0, 10.0, 20.0],
162 ... }
163 ... )
164 >>> pair = aligned(frame, "x", "y")
165 >>> pair.columns
166 ['period', 'a', 'b']
167 >>> pair["a"].to_list()
168 [1.0, 3.0]
169 """
170 for column in (a, b):
171 if column not in frame.columns:
172 raise KeyError(f"no column {column!r} in frame") # noqa: TRY003
173 if not frame.schema[column].is_numeric():
174 raise TypeError(f"column {column!r} is {frame.schema[column]}, which cannot be drawn") # noqa: TRY003
176 date = date_column(frame)
177 period = pl.col(date).alias(PERIOD) if date else pl.int_range(pl.len()).alias(PERIOD)
178 data = frame.select(period, pl.col(a).alias("a"), pl.col(b).alias("b"))
179 return data.drop_nulls().sort(PERIOD)