Coverage for src/jointview/columns.py: 100%

32 statements  

« prev     ^ index     » next       coverage.py v7.15.4, created at 2026-10-01 04:48 +0000

1"""Which columns the app can draw, and cutting a chosen pair down to its common sample. 

2 

3This is the vocabulary both halves of the app are built on, and it belongs to neither 

4of them. :mod:`jointview.plot` draws what :func:`aligned` produces and 

5:mod:`jointview.stats` summarises it, so the column names and the shaping rules live 

6here rather than in either — two peers reaching into a third, instead of one reaching 

7into the other. 

8 

9``PERIOD`` is the clearest case. It is the name :func:`aligned` writes the x-axis under 

10and the name the statistics read it back out of: a data contract between the two, not 

11a fact about plotting. 

12 

13``WINDOWS`` is here for the same reason one layer out. Cutting the sample to the year to 

14date is not a fact about the drawing either: the window is taken off the frame once, and 

15the lines, the tables and the base a rebased pair is indexed to all follow from that one 

16cut instead of each making it themselves. 

17""" 

18 

19from __future__ import annotations 

20 

21from collections.abc import Callable 

22 

23import polars as pl 

24 

25PERIOD = "period" 

26 

27WINDOW_ALL = "all" 

28 

29# The stretches of the sample the app offers above the plot, in the order it offers them, 

30# each mapped to where it begins as a function of the last period in the frame. 

31# 

32# Measured back from that last period rather than from today: a parquet is as recent as 

33# it is, and a window read off the clock would come back empty on a series that ends last 

34# year — the same file being informative on Tuesday and blank on Wednesday. 

35# 

36# `None` is the whole sample. It is not "an offset of zero": it cuts nothing, and it is 

37# also how a caller knows there is no cut to mention. 

38WINDOWS: dict[str, Callable[[pl.Expr], pl.Expr] | None] = { 

39 WINDOW_ALL: None, 

40 # The turn of the year is a calendar boundary rather than a duration, which is why 

41 # this one truncates where the others count backwards. 

42 "ytd": lambda last: last.dt.truncate("1y"), 

43 "12m": lambda last: last.dt.offset_by("-12mo"), 

44 "36m": lambda last: last.dt.offset_by("-36mo"), 

45} 

46 

47 

48def series_columns(frame: pl.DataFrame) -> list[str]: 

49 """The columns that can be drawn: every numeric one. 

50 

51 >>> import datetime as dt, polars as pl 

52 >>> frame = pl.DataFrame({"date": [dt.date(2024, 1, 1)], "nav": [1.0], "label": ["a"]}) 

53 >>> series_columns(frame) 

54 ['nav'] 

55 """ 

56 return [name for name, dtype in frame.schema.items() if dtype.is_numeric()] 

57 

58 

59# Dates and datetimes, and deliberately not everything `dtype.is_temporal()` admits: 

60# that predicate is also true of `Time` and `Duration`. A holding period or a time of day 

61# is a temporal *quantity*, not a point on a calendar, and it cannot be the axis these 

62# series are observed along — `aligned` writes whatever this returns under `PERIOD`, and 

63# `stats` reads the annualisation factor off its spacing, so a column of timedeltas 

64# arriving here is answered with a CAGR in the trillions rather than an error (#74). 

65# 

66# Compared with `==` rather than `isinstance`, which is what polars documents for asking 

67# after a base type: `pl.Datetime` matches every time unit and time zone, so a 

68# `Datetime("ns", "Europe/Zurich")` is admitted without any of that being spelled here. 

69_DATE_LIKE = (pl.Date, pl.Datetime) 

70 

71 

72def date_column(frame: pl.DataFrame) -> str | None: 

73 """The first date or datetime column, which becomes the x-axis. None means row number. 

74 

75 >>> import datetime as dt, polars as pl 

76 >>> date_column(pl.DataFrame({"when": [dt.date(2024, 1, 1)], "nav": [1.0]})) 

77 'when' 

78 >>> date_column(pl.DataFrame({"nav": [1.0]})) is None 

79 True 

80 

81 A `Time` or `Duration` column is temporal but is not a date, and does not become the 

82 axis — the rows get numbered instead, which is the documented fallback: 

83 

84 >>> date_column(pl.DataFrame({"held": [dt.timedelta(days=1)], "nav": [1.0]})) is None 

85 True 

86 """ 

87 return next((name for name, dtype in frame.schema.items() if dtype in _DATE_LIKE), None) 

88 

89 

90def windowed(frame: pl.DataFrame, key: str = WINDOW_ALL) -> pl.DataFrame: 

91 """``frame`` cut to one of :data:`WINDOWS`, measured back from its last period. 

92 

93 Applied to the frame rather than to the picture, so everything built from it agrees: 

94 the lines, the numbers beside them, and the base a rebased pair is indexed to. 

95 

96 Raises: 

97 KeyError: if ``key`` is not one of :data:`WINDOWS`. 

98 

99 >>> import datetime as dt, polars as pl 

100 >>> frame = pl.DataFrame( 

101 ... { 

102 ... "date": [dt.date(2022, 12, 30), dt.date(2023, 12, 29), dt.date(2024, 6, 28)], 

103 ... "nav": [90.0, 100.0, 110.0], 

104 ... } 

105 ... ) 

106 >>> windowed(frame, "ytd")["date"].to_list() 

107 [datetime.date(2024, 6, 28)] 

108 >>> windowed(frame, "12m").height 

109 2 

110 >>> windowed(frame, "all").height 

111 3 

112 

113 A window is a stretch of calendar, so a frame numbered by row has nothing to cut 

114 against and comes back whole: 

115 

116 >>> windowed(pl.DataFrame({"nav": [1.0, 2.0]}), "ytd").height 

117 2 

118 """ 

119 start = WINDOWS[key] 

120 date = date_column(frame) 

121 if start is None or date is None: 

122 return frame 

123 # `max()` rather than the last row: the frame reaching here need not be sorted, and 

124 # `aligned` does its own sorting afterwards. 

125 return frame.filter(pl.col(date) >= start(pl.col(date).max())) 

126 

127 

128def default_pair(frame: pl.DataFrame) -> tuple[int, int]: 

129 """Indices into :func:`series_columns` to open on — the first two series. 

130 

131 A frame with a single series opens on it twice, rather than refusing to draw. 

132 

133 >>> import polars as pl 

134 >>> default_pair(pl.DataFrame({"a": [1.0], "b": [2.0]})) 

135 (0, 1) 

136 >>> default_pair(pl.DataFrame({"only": [1.0]})) 

137 (0, 0) 

138 """ 

139 names = series_columns(frame) 

140 if not names: 

141 raise ValueError("frame has no numeric columns to plot") # noqa: TRY003 

142 return 0, 1 if len(names) > 1 else 0 

143 

144 

145def aligned(frame: pl.DataFrame, a: str, b: str) -> pl.DataFrame: 

146 """The two series on their common sample: ``period``, ``a``, ``b``, in order. 

147 

148 Both the picture and the summary tables are built from this, so the numbers 

149 beside the chart always describe the lines in it. Renaming also sidesteps 

150 Vega-Lite's field-shorthand escaping and lets ``a`` and ``b`` be the same column. 

151 

152 The three names are fixed whatever the columns were called, the rows come out 

153 sorted by period, and a date where either series is missing is not part of the 

154 sample — here the frame arrives unsorted and with a gap on the 2nd: 

155 

156 >>> import datetime as dt, polars as pl 

157 >>> frame = pl.DataFrame( 

158 ... { 

159 ... "date": [dt.date(2024, 1, 3), dt.date(2024, 1, 1), dt.date(2024, 1, 2)], 

160 ... "x": [3.0, 1.0, None], 

161 ... "y": [30.0, 10.0, 20.0], 

162 ... } 

163 ... ) 

164 >>> pair = aligned(frame, "x", "y") 

165 >>> pair.columns 

166 ['period', 'a', 'b'] 

167 >>> pair["a"].to_list() 

168 [1.0, 3.0] 

169 """ 

170 for column in (a, b): 

171 if column not in frame.columns: 

172 raise KeyError(f"no column {column!r} in frame") # noqa: TRY003 

173 if not frame.schema[column].is_numeric(): 

174 raise TypeError(f"column {column!r} is {frame.schema[column]}, which cannot be drawn") # noqa: TRY003 

175 

176 date = date_column(frame) 

177 period = pl.col(date).alias(PERIOD) if date else pl.int_range(pl.len()).alias(PERIOD) 

178 data = frame.select(period, pl.col(a).alias("a"), pl.col(b).alias("b")) 

179 return data.drop_nulls().sort(PERIOD)