1"""
2Printing a value in a failure report.
3
4Three things a report's reader is owed. A value is printed whole, or the
5report says how much was left out and how to see it. Two large values that
6were expected to be equal are printed as what differs between them. And
7printing never goes wrong: a `repr` that raises is reported as having
8raised, and doesn't take the report with it.
9
10A value is printed by `pprint`, which puts a container that doesn't fit on
11a line one item to a line. A package can print the values it owns better
12than their `repr` does, through its lifecycle's `describe_value()`.
13
14A fourth thing is owed to whoever the report is shown to: it doesn't print
15what is known to be secret. The environment is printed as its names, with
16no values. A package leaves the secrets it owns out of what it describes.
17That is done for a value wherever it is: on its own, or inside a list, a
18tuple, a dict or a set, however far in. A secret that is a string like any
19other by the time the test has it can't be told from one, and is printed.
20"""
21
22import difflib
23import os
24import pprint
25from collections.abc import Callable, ItemsView, Sequence, ValuesView
26from dataclasses import dataclass
27from typing import Any
28
29from .layout import without_test_module_names
30
31__all__ = []
32
33# The most of one value a report prints, in characters.
34VALUE_CAP = 2_000
35# The most of one diff a report prints, in lines.
36DIFF_LINE_CAP = 60
37# The flag that lifts both.
38FULL_VALUES_FLAG = "--full-values"
39
40# A value that prints on one line no longer than this is small: it is
41# printed whole, and two of them are not diffed.
42_ONE_LINE = 80
43
44# How much of two long one-line strings is shown on each side of the first
45# place they differ.
46_BEFORE_THE_DIFFERENCE = 30
47_AFTER_THE_DIFFERENCE = 50
48
49type Describer = Callable[[object], str | None]
50
51
52@dataclass(frozen=True, kw_only=True)
53class PrintedValue:
54 """A value as a report prints it."""
55
56 # One line or several.
57 text: str
58 # How many characters the cap left out. 0 when `text` is the whole value.
59 cut_characters: int = 0
60 # Set when the report has printed this value already, under this name.
61 # `text` says so, and the value isn't printed again.
62 same_as: str | None = None
63
64
65@dataclass(frozen=True, kw_only=True)
66class Diff:
67 """What differs between two values that were expected to be equal."""
68
69 # A unified diff: `-` is the left side's, `+` is the right side's.
70 lines: tuple[str, ...]
71 # How many lines the cap left out. 0 when `lines` is the whole diff.
72 cut_lines: int = 0
73
74
75class _GuardedPrinter(pprint.PrettyPrinter):
76 """A PrettyPrinter that a `repr` can't stop by raising."""
77
78 def format(
79 self, object: object, context: dict, maxlevels: int, level: int
80 ) -> tuple[str, bool, bool]:
81 try:
82 return super().format(object, context, maxlevels, level)
83 except Exception as error:
84 return what_repr_raised(object, error), False, False
85
86
87def what_repr_raised(value: object, error: Exception) -> str:
88 return (
89 f"<{type(value).__qualname__}: its repr raised "
90 f"{type(error).__qualname__}: {error}>"
91 )
92
93
94class _AlreadyText:
95 """Text printed as it is, where a value's `repr` would have been."""
96
97 def __init__(self, text: str) -> None:
98 self.text = text
99
100 def __repr__(self) -> str:
101 return self.text
102
103
104def _is_the_environments(key: object, item: object) -> bool:
105 return (
106 isinstance(key, str) and isinstance(item, str) and os.environ.get(key) == item
107 )
108
109
110def _names_only(environment: os._Environ) -> str:
111 names = ", ".join(sorted(str(name) for name in environment))
112 return f"environ({_counted(len(environment), 'name')}, values withheld: {names})"
113
114
115class ValuePrinter:
116 """
117 Prints the values of one failure. It remembers what it has printed, so
118 that a large value the failure has in it twice is printed once.
119 """
120
121 def __init__(
122 self, *, describers: Sequence[Describer] = (), full_values: bool = False
123 ) -> None:
124 self.describers = describers
125 self.full_values = full_values
126 # The text of each large value printed so far, and the name it was
127 # printed under.
128 self._printed_as: dict[str, str] = {}
129
130 def printed(self, value: object, *, name: str | None = None) -> PrintedValue:
131 """
132 The value, whole or cut at the cap. `name` is what the report calls
133 it: a value printed under a name is not printed again under
134 another, and the second is said to be the same as the first.
135 """
136 text = self._whole(value)
137
138 if name is not None and not _is_small(text):
139 first_name = self._printed_as.setdefault(text, name)
140 if first_name != name:
141 return PrintedValue(
142 text=f"<the same as {first_name}>", same_as=first_name
143 )
144
145 if self.full_values or len(text) <= VALUE_CAP:
146 return PrintedValue(text=text)
147 return PrintedValue(text=text[:VALUE_CAP], cut_characters=len(text) - VALUE_CAP)
148
149 def summarized(self, value: object) -> PrintedValue:
150 """What kind of value it is and how big, for one the diff shows."""
151 return PrintedValue(text=_summary(value))
152
153 def diff(
154 self, left: object, right: object, *, left_source: str, right_source: str
155 ) -> Diff | None:
156 """
157 What differs between two values, or None when both are small enough
158 to read side by side.
159 """
160 left_text = self._whole(left, sort_dicts=True)
161 right_text = self._whole(right, sort_dicts=True)
162
163 if isinstance(left, str) and isinstance(right, str):
164 # Text of more than one line is compared line by line however
165 # short it is. Its `repr` is one line with `\n` in it, which
166 # is small and is no way to read it.
167 has_lines = "\n" in left or "\n" in right
168 if not has_lines and _is_small(left_text) and _is_small(right_text):
169 return None
170 lines = _text_diff(
171 left, right, left_source=left_source, right_source=right_source
172 )
173 elif _is_small(left_text) and _is_small(right_text):
174 return None
175 else:
176 lines = _line_diff(
177 left_text.splitlines(),
178 right_text.splitlines(),
179 left_source=left_source,
180 right_source=right_source,
181 )
182
183 if self.full_values or len(lines) <= DIFF_LINE_CAP:
184 return Diff(lines=tuple(lines))
185 return Diff(
186 lines=tuple(lines[:DIFF_LINE_CAP]),
187 cut_lines=len(lines) - DIFF_LINE_CAP,
188 )
189
190 def _whole(self, value: object, *, sort_dicts: bool = False) -> str:
191 return without_test_module_names(self._as_python_prints_it(value, sort_dicts))
192
193 def _as_python_prints_it(self, value: object, sort_dicts: bool) -> str:
194 try:
195 fit_to_print = self._fit_to_print(value, inside=frozenset())
196 except RecursionError:
197 return f"<{type(value).__qualname__}: nested too deeply to print>"
198 if isinstance(fit_to_print, _AlreadyText):
199 return fit_to_print.text
200
201 printer = _GuardedPrinter(width=_ONE_LINE, sort_dicts=sort_dicts)
202 try:
203 return printer.pformat(fit_to_print)
204 except Exception as error:
205 # Not a repr this time: a `__len__` or an `__iter__` that raised
206 # while the printer was laying the value out.
207 return what_repr_raised(value, error)
208
209 def _fit_to_print(self, value: object, *, inside: frozenset[int]) -> object:
210 """
211 The value as it can be handed to `pprint`: itself, or the text a
212 package describes it by, or a copy of a container with the same
213 done to everything in it.
214 """
215 described = self._described(value)
216 if described is not None:
217 return _AlreadyText(described)
218
219 if isinstance(value, os._Environ):
220 return _AlreadyText(_names_only(value))
221 is_a_view = isinstance(value, ValuesView | ItemsView)
222 if is_a_view and getattr(value, "_mapping", None) is os.environ:
223 kind = type(value).__qualname__
224 return _AlreadyText(f"<{kind} of the environment, withheld>")
225
226 if id(value) in inside:
227 # One that is inside itself is left for `pprint`, which says so.
228 return value
229 inside = inside | {id(value)}
230
231 # A subclass of one of these prints as its own `repr` says, which
232 # isn't the printer's to take apart.
233 if isinstance(value, dict):
234 if type(value) is not dict:
235 return value
236 return self._dict_fit_to_print(value, inside=inside)
237 if isinstance(value, list | tuple | set | frozenset):
238 if type(value) not in (list, tuple, set, frozenset):
239 return value
240 return type(value)(
241 self._fit_to_print(item, inside=inside) for item in value
242 )
243 return value
244
245 def _dict_fit_to_print(self, value: dict, *, inside: frozenset[int]) -> dict:
246 """
247 A dict made from the environment, all of it or some of it, as
248 `{**os.environ, "DEBUG": "1"}` is for a subprocess, is printed
249 without what it took: that is the environment's still. What was
250 taken is said once, by name, after the items that are the dict's
251 own.
252 """
253 fit_to_print: dict[object, object] = {}
254 from_the_environment = []
255 for key, item in value.items():
256 if _is_the_environments(key, item):
257 from_the_environment.append(key)
258 else:
259 fit_to_print[key] = self._fit_to_print(item, inside=inside)
260
261 if from_the_environment:
262 names = ", ".join(sorted(from_the_environment))
263 taken = _counted(len(from_the_environment), "name")
264 fit_to_print[_AlreadyText(f"<{taken} from the environment>")] = (
265 _AlreadyText(f"<values withheld: {names}>")
266 )
267 return fit_to_print
268
269 def _described(self, value: object) -> str | None:
270 for describe in self.describers:
271 try:
272 described = describe(value)
273 except Exception:
274 # A describer that fails is a describer with nothing to say.
275 continue
276 if described is not None:
277 return str(described)
278 return None
279
280
281def _is_small(text: str) -> bool:
282 return "\n" not in text and len(text) <= _ONE_LINE
283
284
285def _summary(value: Any) -> str:
286 name = type(value).__qualname__
287 if isinstance(value, str):
288 characters = _counted(len(value), "character")
289 lines = _counted(len(value.splitlines()), "line")
290 return f"<{name}, {characters} in {lines}>"
291 if isinstance(value, bytes):
292 return f"<{name}, {_counted(len(value), 'byte')}>"
293 if isinstance(value, dict):
294 return f"<{name} with {_counted(len(value), 'key')}>"
295 if isinstance(value, list | tuple | set | frozenset):
296 return f"<{name} with {_counted(len(value), 'item')}>"
297 return f"<{name}>"
298
299
300def _counted(count: int, what: str) -> str:
301 return f"{count:,} {what}" if count == 1 else f"{count:,} {what}s"
302
303
304def _line_diff(
305 left_lines: list[str],
306 right_lines: list[str],
307 *,
308 left_source: str,
309 right_source: str,
310) -> list[str]:
311 return list(
312 difflib.unified_diff(
313 left_lines,
314 right_lines,
315 fromfile=left_source,
316 tofile=right_source,
317 n=2,
318 lineterm="",
319 )
320 )
321
322
323def _text_diff(
324 left: str, right: str, *, left_source: str, right_source: str
325) -> list[str]:
326 """
327 Two strings, line by line as they read. Where reading them wouldn't show
328 everything (a space at the end of a line, a line ending that is there on
329 one side only), each line's `repr`, which does.
330 """
331 left_lines = left.splitlines()
332 right_lines = right.splitlines()
333 if len(left_lines) <= 1 and len(right_lines) <= 1:
334 return _one_line_text_diff(left, right)
335
336 end_the_same_way = left.endswith("\n") == right.endswith("\n")
337 if _reads_as_it_is(left) and _reads_as_it_is(right) and end_the_same_way:
338 return _line_diff(
339 left_lines,
340 right_lines,
341 left_source=left_source,
342 right_source=right_source,
343 )
344 return _line_diff(
345 [repr(line) for line in left.splitlines(keepends=True)],
346 [repr(line) for line in right.splitlines(keepends=True)],
347 left_source=left_source,
348 right_source=right_source,
349 )
350
351
352def _reads_as_it_is(text: str) -> bool:
353 """Whether every character of a text's lines can be seen when printed."""
354 return all(
355 line == line.rstrip() and line.isprintable() for line in text.split("\n")
356 )
357
358
359def _one_line_text_diff(left: str, right: str) -> list[str]:
360 """
361 Two strings with no lines to go by: where they first differ, and what
362 is around that on each side.
363 """
364 at = 0
365 while at < min(len(left), len(right)) and left[at] == right[at]:
366 at += 1
367 start = max(0, at - _BEFORE_THE_DIFFERENCE)
368 stop = at + _AFTER_THE_DIFFERENCE
369
370 def around(text: str) -> str:
371 before = "..." if start > 0 else ""
372 after = "..." if stop < len(text) else ""
373 return f"{before}{text[start:stop]!r}{after}"
374
375 where = (
376 f"first difference at character {at:,}"
377 f" (left is {len(left):,} characters, right is {len(right):,})"
378 )
379 return [
380 where,
381 f"- {around(left)}",
382 f"+ {around(right)}",
383 ]