Coverage for src/signalk_cli/history/_results.py: 91%

113 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-10-06 00:17 +0000

1"""Reshape History API /values responses into long rows, wide rows, and per-path statistics. 

2 

3Values keep their JSON types here (numbers, strings, objects, arrays); the CLI 

4writers and the Arrow conversion decide how to represent them. 

5""" 

6 

7import json 

8import re 

9from typing import Any 

10 

11POSITION_RE = re.compile(r"navigation.*\.position") 

12 

13WIDE_SCALAR_COLUMNS = ["min_value", "avg_value", "max_value"] 

14_WIDE_METHOD_FOR_COLUMN = { 

15 "min_value": "min", 

16 "avg_value": "average", 

17 "max_value": "max", 

18} 

19 

20CARDINALITY_COLUMNS = [ 

21 "path", 

22 "distinct_values", 

23 "distinct_values_2_decimal_places", 

24 "nulls", 

25 "zeroes", 

26 "min", 

27 "max", 

28 "average", 

29] 

30 

31 

32def _is_number(v: object) -> bool: 

33 return isinstance(v, (int, float)) and not isinstance(v, bool) 

34 

35 

36def _cell(row: list, col_idx: int) -> Any: 

37 """Value at 0-based value column col_idx (row[0] is the timestamp), or None.""" 

38 i = col_idx + 1 

39 return row[i] if i < len(row) else None 

40 

41 

42def _paths_by_column(payload: dict) -> list[str]: 

43 return [ 

44 col.get("path", f"col_{i}") for i, col in enumerate(payload.get("values", [])) 

45 ] 

46 

47 

48# --------------------------------------------------------------------------- 

49# Long: one row per (timestamp, path) 

50# --------------------------------------------------------------------------- 

51 

52 

53def long_rows(payload: dict) -> tuple[list[str], list[str], list[Any]]: 

54 """Flatten a /values response into (timestamps, paths, values), skipping nulls.""" 

55 col_paths = _paths_by_column(payload) 

56 timestamps: list[str] = [] 

57 paths: list[str] = [] 

58 values: list[Any] = [] 

59 for row in payload.get("data", []): 

60 if not row: 60 ↛ 61line 60 didn't jump to line 61 because the condition on line 60 was never true

61 continue 

62 for i, path in enumerate(col_paths): 

63 value = _cell(row, i) 

64 if value is None: 

65 continue 

66 timestamps.append(row[0]) 

67 paths.append(path) 

68 values.append(value) 

69 return timestamps, paths, values 

70 

71 

72# --------------------------------------------------------------------------- 

73# Wide: one row per (timestamp, path) with min/avg/max or array element columns 

74# --------------------------------------------------------------------------- 

75 

76 

77def _first_value(data_rows: list, col_idx: int) -> Any: 

78 for row in data_rows: 78 ↛ 83line 78 didn't jump to line 83 because the loop on line 78 didn't complete

79 if row: 79 ↛ 78line 79 didn't jump to line 78 because the condition on line 79 was always true

80 v = _cell(row, col_idx) 

81 if v is not None: 

82 return v 

83 return None 

84 

85 

86def _array_col_names(path: str, length: int) -> list[str]: 

87 if length == 2 and POSITION_RE.fullmatch(path): 

88 return ["longitude", "latitude"] 

89 return [f"value_{i}" for i in range(length)] 

90 

91 

92def wide_rows(payload: dict) -> tuple[list[str], list[str], dict[str, list[Any]]]: 

93 """Reshape a /values response into (timestamps, paths, value_columns). 

94 

95 Scalar paths fill `min_value`/`avg_value`/`max_value`. Array paths 

96 fill one column per element: `longitude`/`latitude` for 

97 `navigation.*.position`, otherwise `value_0`, `value_1`, ... Cells 

98 that don't apply to a row's path are None, and rows where every cell is 

99 null are dropped. 

100 """ 

101 value_columns = payload.get("values", []) 

102 data_rows = payload.get("data", []) 

103 

104 path_method_idx: dict[str, dict[str, int]] = {} 

105 for i, col in enumerate(value_columns): 

106 path = col.get("path", f"col_{i}") 

107 path_method_idx.setdefault(path, {})[col.get("method", "")] = i 

108 

109 path_cols: dict[str, list[str]] = {} 

110 path_is_array: dict[str, bool] = {} 

111 for path, methods in path_method_idx.items(): 

112 sample = None 

113 for col_idx in methods.values(): 113 ↛ 117line 113 didn't jump to line 117 because the loop on line 113 didn't complete

114 sample = _first_value(data_rows, col_idx) 

115 if sample is not None: 115 ↛ 113line 115 didn't jump to line 113 because the condition on line 115 was always true

116 break 

117 path_is_array[path] = isinstance(sample, list) 

118 path_cols[path] = ( 

119 _array_col_names(path, len(sample)) 

120 if isinstance(sample, list) 

121 else WIDE_SCALAR_COLUMNS 

122 ) 

123 

124 all_cols: list[str] = [] 

125 for cols in path_cols.values(): 

126 all_cols.extend(c for c in cols if c not in all_cols) 

127 

128 timestamps: list[str] = [] 

129 paths_out: list[str] = [] 

130 out: dict[str, list[Any]] = {c: [] for c in all_cols} 

131 

132 for row in data_rows: 

133 if not row: 133 ↛ 134line 133 didn't jump to line 134 because the condition on line 133 was never true

134 continue 

135 for path, methods in path_method_idx.items(): 

136 cells: dict[str, Any] = {} 

137 if path_is_array[path]: 

138 arr = next( 

139 ( 

140 v 

141 for v in (_cell(row, i) for i in methods.values()) 

142 if v is not None 

143 ), 

144 None, 

145 ) 

146 if arr is None: 146 ↛ 147line 146 didn't jump to line 147 because the condition on line 146 was never true

147 continue 

148 cells = dict(zip(path_cols[path], arr)) 

149 else: 

150 cells = { 

151 col: _cell(row, methods[method]) if method in methods else None 

152 for col, method in _WIDE_METHOD_FOR_COLUMN.items() 

153 } 

154 if all(v is None for v in cells.values()): 

155 continue 

156 timestamps.append(row[0]) 

157 paths_out.append(path) 

158 for col in all_cols: 

159 out[col].append(cells.get(col)) 

160 

161 return timestamps, paths_out, out 

162 

163 

164# --------------------------------------------------------------------------- 

165# Cardinality 

166# --------------------------------------------------------------------------- 

167 

168 

169def _distinct_2dp(vals: list) -> int | None: 

170 if not vals: 

171 return 0 

172 if all(_is_number(v) for v in vals): 

173 return len({round(float(v), 2) for v in vals}) 

174 if isinstance(vals[0], list): 174 ↛ 178line 174 didn't jump to line 178 because the condition on line 174 was always true

175 return len( 

176 {tuple(round(x, 2) if _is_number(x) else x for x in v) for v in vals} 

177 ) 

178 return None 

179 

180 

181def cardinality(payload: dict) -> list[dict[str, Any]]: 

182 """Per-path statistics: distinct values, nulls, zeroes, and min/max/average. 

183 

184 Each row has the keys in `CARDINALITY_COLUMNS`. `min`, `max` and 

185 `average` are None unless every value of the path is a number; 

186 `distinct_values_2_decimal_places` is None for paths whose values are 

187 neither numbers nor arrays. 

188 """ 

189 col_paths = _paths_by_column(payload) 

190 ordered = list(dict.fromkeys(col_paths)) 

191 vals: dict[str, list] = {p: [] for p in ordered} 

192 nulls = dict.fromkeys(ordered, 0) 

193 zeroes = dict.fromkeys(ordered, 0) 

194 

195 for row in payload.get("data", []): 

196 if not row: 196 ↛ 197line 196 didn't jump to line 197 because the condition on line 196 was never true

197 continue 

198 for i, path in enumerate(col_paths): 

199 v = _cell(row, i) 

200 if v is None: 

201 nulls[path] += 1 

202 continue 

203 if _is_number(v) and v == 0: 

204 zeroes[path] += 1 

205 vals[path].append(v) 

206 

207 rows = [] 

208 for path in ordered: 

209 pv = vals[path] 

210 numeric = bool(pv) and all(_is_number(v) for v in pv) 

211 rows.append( 

212 { 

213 "path": path, 

214 "distinct_values": len( 

215 { 

216 json.dumps(v, sort_keys=True) 

217 if isinstance(v, (dict, list)) 

218 else str(v) 

219 for v in pv 

220 } 

221 ), 

222 "distinct_values_2_decimal_places": _distinct_2dp(pv), 

223 "nulls": nulls[path], 

224 "zeroes": zeroes[path], 

225 "min": min(pv) if numeric else None, 

226 "max": max(pv) if numeric else None, 

227 "average": sum(pv) / len(pv) if numeric else None, 

228 } 

229 ) 

230 return rows