Coverage for src/signalk_cli/history/_results.py: 91%
113 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-10-06 00:17 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-10-06 00:17 +0000
1"""Reshape History API /values responses into long rows, wide rows, and per-path statistics.
3Values keep their JSON types here (numbers, strings, objects, arrays); the CLI
4writers and the Arrow conversion decide how to represent them.
5"""
7import json
8import re
9from typing import Any
11POSITION_RE = re.compile(r"navigation.*\.position")
13WIDE_SCALAR_COLUMNS = ["min_value", "avg_value", "max_value"]
14_WIDE_METHOD_FOR_COLUMN = {
15 "min_value": "min",
16 "avg_value": "average",
17 "max_value": "max",
18}
20CARDINALITY_COLUMNS = [
21 "path",
22 "distinct_values",
23 "distinct_values_2_decimal_places",
24 "nulls",
25 "zeroes",
26 "min",
27 "max",
28 "average",
29]
32def _is_number(v: object) -> bool:
33 return isinstance(v, (int, float)) and not isinstance(v, bool)
36def _cell(row: list, col_idx: int) -> Any:
37 """Value at 0-based value column col_idx (row[0] is the timestamp), or None."""
38 i = col_idx + 1
39 return row[i] if i < len(row) else None
42def _paths_by_column(payload: dict) -> list[str]:
43 return [
44 col.get("path", f"col_{i}") for i, col in enumerate(payload.get("values", []))
45 ]
48# ---------------------------------------------------------------------------
49# Long: one row per (timestamp, path)
50# ---------------------------------------------------------------------------
53def long_rows(payload: dict) -> tuple[list[str], list[str], list[Any]]:
54 """Flatten a /values response into (timestamps, paths, values), skipping nulls."""
55 col_paths = _paths_by_column(payload)
56 timestamps: list[str] = []
57 paths: list[str] = []
58 values: list[Any] = []
59 for row in payload.get("data", []):
60 if not row: 60 ↛ 61line 60 didn't jump to line 61 because the condition on line 60 was never true
61 continue
62 for i, path in enumerate(col_paths):
63 value = _cell(row, i)
64 if value is None:
65 continue
66 timestamps.append(row[0])
67 paths.append(path)
68 values.append(value)
69 return timestamps, paths, values
72# ---------------------------------------------------------------------------
73# Wide: one row per (timestamp, path) with min/avg/max or array element columns
74# ---------------------------------------------------------------------------
77def _first_value(data_rows: list, col_idx: int) -> Any:
78 for row in data_rows: 78 ↛ 83line 78 didn't jump to line 83 because the loop on line 78 didn't complete
79 if row: 79 ↛ 78line 79 didn't jump to line 78 because the condition on line 79 was always true
80 v = _cell(row, col_idx)
81 if v is not None:
82 return v
83 return None
86def _array_col_names(path: str, length: int) -> list[str]:
87 if length == 2 and POSITION_RE.fullmatch(path):
88 return ["longitude", "latitude"]
89 return [f"value_{i}" for i in range(length)]
92def wide_rows(payload: dict) -> tuple[list[str], list[str], dict[str, list[Any]]]:
93 """Reshape a /values response into (timestamps, paths, value_columns).
95 Scalar paths fill `min_value`/`avg_value`/`max_value`. Array paths
96 fill one column per element: `longitude`/`latitude` for
97 `navigation.*.position`, otherwise `value_0`, `value_1`, ... Cells
98 that don't apply to a row's path are None, and rows where every cell is
99 null are dropped.
100 """
101 value_columns = payload.get("values", [])
102 data_rows = payload.get("data", [])
104 path_method_idx: dict[str, dict[str, int]] = {}
105 for i, col in enumerate(value_columns):
106 path = col.get("path", f"col_{i}")
107 path_method_idx.setdefault(path, {})[col.get("method", "")] = i
109 path_cols: dict[str, list[str]] = {}
110 path_is_array: dict[str, bool] = {}
111 for path, methods in path_method_idx.items():
112 sample = None
113 for col_idx in methods.values(): 113 ↛ 117line 113 didn't jump to line 117 because the loop on line 113 didn't complete
114 sample = _first_value(data_rows, col_idx)
115 if sample is not None: 115 ↛ 113line 115 didn't jump to line 113 because the condition on line 115 was always true
116 break
117 path_is_array[path] = isinstance(sample, list)
118 path_cols[path] = (
119 _array_col_names(path, len(sample))
120 if isinstance(sample, list)
121 else WIDE_SCALAR_COLUMNS
122 )
124 all_cols: list[str] = []
125 for cols in path_cols.values():
126 all_cols.extend(c for c in cols if c not in all_cols)
128 timestamps: list[str] = []
129 paths_out: list[str] = []
130 out: dict[str, list[Any]] = {c: [] for c in all_cols}
132 for row in data_rows:
133 if not row: 133 ↛ 134line 133 didn't jump to line 134 because the condition on line 133 was never true
134 continue
135 for path, methods in path_method_idx.items():
136 cells: dict[str, Any] = {}
137 if path_is_array[path]:
138 arr = next(
139 (
140 v
141 for v in (_cell(row, i) for i in methods.values())
142 if v is not None
143 ),
144 None,
145 )
146 if arr is None: 146 ↛ 147line 146 didn't jump to line 147 because the condition on line 146 was never true
147 continue
148 cells = dict(zip(path_cols[path], arr))
149 else:
150 cells = {
151 col: _cell(row, methods[method]) if method in methods else None
152 for col, method in _WIDE_METHOD_FOR_COLUMN.items()
153 }
154 if all(v is None for v in cells.values()):
155 continue
156 timestamps.append(row[0])
157 paths_out.append(path)
158 for col in all_cols:
159 out[col].append(cells.get(col))
161 return timestamps, paths_out, out
164# ---------------------------------------------------------------------------
165# Cardinality
166# ---------------------------------------------------------------------------
169def _distinct_2dp(vals: list) -> int | None:
170 if not vals:
171 return 0
172 if all(_is_number(v) for v in vals):
173 return len({round(float(v), 2) for v in vals})
174 if isinstance(vals[0], list): 174 ↛ 178line 174 didn't jump to line 178 because the condition on line 174 was always true
175 return len(
176 {tuple(round(x, 2) if _is_number(x) else x for x in v) for v in vals}
177 )
178 return None
181def cardinality(payload: dict) -> list[dict[str, Any]]:
182 """Per-path statistics: distinct values, nulls, zeroes, and min/max/average.
184 Each row has the keys in `CARDINALITY_COLUMNS`. `min`, `max` and
185 `average` are None unless every value of the path is a number;
186 `distinct_values_2_decimal_places` is None for paths whose values are
187 neither numbers nor arrays.
188 """
189 col_paths = _paths_by_column(payload)
190 ordered = list(dict.fromkeys(col_paths))
191 vals: dict[str, list] = {p: [] for p in ordered}
192 nulls = dict.fromkeys(ordered, 0)
193 zeroes = dict.fromkeys(ordered, 0)
195 for row in payload.get("data", []):
196 if not row: 196 ↛ 197line 196 didn't jump to line 197 because the condition on line 196 was never true
197 continue
198 for i, path in enumerate(col_paths):
199 v = _cell(row, i)
200 if v is None:
201 nulls[path] += 1
202 continue
203 if _is_number(v) and v == 0:
204 zeroes[path] += 1
205 vals[path].append(v)
207 rows = []
208 for path in ordered:
209 pv = vals[path]
210 numeric = bool(pv) and all(_is_number(v) for v in pv)
211 rows.append(
212 {
213 "path": path,
214 "distinct_values": len(
215 {
216 json.dumps(v, sort_keys=True)
217 if isinstance(v, (dict, list))
218 else str(v)
219 for v in pv
220 }
221 ),
222 "distinct_values_2_decimal_places": _distinct_2dp(pv),
223 "nulls": nulls[path],
224 "zeroes": zeroes[path],
225 "min": min(pv) if numeric else None,
226 "max": max(pv) if numeric else None,
227 "average": sum(pv) / len(pv) if numeric else None,
228 }
229 )
230 return rows