From 6177b1a6a91150c2bdf855e539a2b1a10aac0de8 Mon Sep 17 00:00:00 2001 From: Mathieu Benoit Date: Wed, 19 Aug 2026 05:37:14 -0400 Subject: [PATCH] =?UTF-8?q?[FIX]=20script=20todo:=20r=C3=A9sumer=20les=20e?= =?UTF-8?q?rreurs=20au=20lieu=20de=20chercher=20=C2=AB=20error=20=C2=BB?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Le volet « d » et le compteur du tableau de bord cherchaient la sous-chaîne « error ». Le journal de l'installation qui vient d'échouer — APK tué par le noyau sur erplibre-ubuntu-2604-gnome — n'en contient AUCUNE : 0 ligne sur 8765, mesuré. Le volet annonçait « aucune erreur détectée » sur une machine morte, et le tableau de bord 0 erreur. Comptent désormais les marqueurs qui ne disent jamais « error » : « ⚠ ÉCHEC », « FAILURE », une trace Python, un « fatal: » de git, une mort par mémoire. Le volet ouvre sur un résumé — l'étape en échec et son diagnostic, les signaux durs, puis les répétitions comptées par forme. Sur ce journal : 1 étape nommée, 2 signaux, là où il n'affichait rien. --- EN --- The "d" pane and the dashboard counter looked for the "error" substring. The log of the install that just failed — APK killed by the kernel on erplibre-ubuntu-2604-gnome — contains NONE: 0 lines out of 8765, measured. The pane said "no error detected" about a dead machine, and the dashboard 0 errors. Markers that never say "error" now count: "⚠ ÉCHEC", "FAILURE", a Python traceback, a git "fatal:", a death by memory. The pane opens on a summary — the failed step with its diagnostic, the hard signals, then repeats counted by shape. On that log: 1 named step and 2 signals, where it showed nothing. Assisted-by: Claude Opus 5 --- script/todo/qemu_install_monitor.py | 194 ++++++++++++++++++++++++-- script/todo/todo_i18n.py | 5 + test/test_qemu_install_summary.py | 204 ++++++++++++++++++++++++++++ 3 files changed, 395 insertions(+), 8 deletions(-) create mode 100644 test/test_qemu_install_summary.py diff --git a/script/todo/qemu_install_monitor.py b/script/todo/qemu_install_monitor.py index eb63c5c..f366285 100644 --- a/script/todo/qemu_install_monitor.py +++ b/script/todo/qemu_install_monitor.py @@ -395,6 +395,133 @@ _LST_IGNORE_ERROR = ( ) +# Signaux d'échec qui ne contiennent NI « error » NI « warning ». Sans eux, le +# scan par sous-chaîne rate des installations franchement ratées : le journal de +# la VM erplibre-ubuntu-2604-gnome, dont la compilation de l'APK a été tuée par +# le noyau, ne portait AUCUNE ligne « error » — mesuré, 0 sur 8765 lignes — +# pendant que « ⚠ ÉCHEC : APK debug (gradle) », « FAILURE: Build failed » et +# « daemon disappeared unexpectedly » y étaient. Le détail des erreurs annonçait +# donc « aucune erreur détectée » sur une installation en échec. +# +# Chaque motif est là parce qu'il est apparu dans un vrai journal, pas par +# précaution : Gradle dit « FAILURE », Python « Traceback », git « fatal: », apt +# « Unable to locate package », le noyau « Killed » ou « Cannot allocate +# memory », et nos propres étapes « ⚠ ÉCHEC ». +_LST_HARD_MARKERS = ( + "⚠ échec", + "failed:", + "failure", + "traceback (most recent call last)", + "fatal:", + "command not found", + # PAS « no such file or directory » : sur le journal de référence, 5 de ses + # 7 occurrences étaient des sondes bénignes (« cat: .odoo-version »), et le + # bruit dilue un résumé dont l'intérêt est justement d'être court. Un + # fichier vraiment manquant fait échouer une ÉTAPE, elle-même captée. + "permission denied", + "unable to locate package", + "disappeared unexpectedly", + "outofmemory", + "cannot allocate memory", + "segmentation fault", + "core dumped", + "killed process", +) +# Étape en échec, telle que la pose « mstep » : « ⚠ ÉCHEC : ». C'est +# le signal AUTORITAIRE — il nomme l'étape, là où « FAILURE » ne nomme que +# l'outil. +_RE_FAILED_STEP = re.compile(r"⚠\s*(?:ÉCHEC|FAILED)\s*:?\s*(.+)") +# Début d'une autre étape ou d'une section : borne du diagnostic qui suit. +_RE_STEP_BOUND = re.compile(r"^\s*(?:->|==)\s") + + +def _is_hard_signal(line: str) -> bool: + low = line.lower() + return any(m in low for m in _LST_HARD_MARKERS) + + +def _error_signature(line: str) -> str: + """Ligne réduite à sa FORME, pour regrouper les répétitions. + + Un journal d'installation répète la même erreur des centaines de fois avec + un chemin ou un numéro qui change. Regrouper sur cette forme donne « ×342 » + au lieu de 342 lignes à faire défiler.""" + sig = re.sub(r"\d+", "#", line) + sig = re.sub(r"0x[0-9a-fA-F]+", "#", sig) + sig = re.sub(r"/\S+", "/…", sig) + return re.sub(r"\s+", " ", sig).strip()[:160] + + +def scan_log_summary(log_path: str, diag_cap: int = 14) -> dict: + """Résumé d'un journal d'installation : ce qui a échoué, puis le reste. + + Rend {steps, hard, groups, nerr, nwarn} où « steps » liste les étapes en + échec AVEC leur diagnostic, « hard » les autres signaux durs dédupliqués, et + « groups » les lignes « error »/« warning » regroupées par forme et comptées. + + L'ordre n'est pas cosmétique : une étape en échec nommée vaut mille lignes, + et c'est elle qu'on veut lire d'abord.""" + try: + lines = Path(log_path).read_text(errors="replace").splitlines() + except OSError: + return {"steps": [], "hard": [], "groups": [], "nerr": 0, "nwarn": 0} + + steps, hard, groups = [], {}, {} + nerr = nwarn = 0 + for i, line in enumerate(lines, 1): + if EXIT_MARKER in line: + continue + low = line.lower() + match = _RE_FAILED_STEP.search(line) + if match: + # Le diagnostic suit l'échec, jusqu'à l'étape suivante : c'est lui + # qui porte la cause, l'échec ne portant que le nom. + diag = [] + for nxt in lines[i : i + 60]: + if _RE_STEP_BOUND.match(nxt) or _RE_FAILED_STEP.search(nxt): + break + if EXIT_MARKER in nxt: + continue + if nxt.strip() and len(diag) < diag_cap: + diag.append(nxt.rstrip()) + steps.append( + {"line": i, "label": match.group(1).strip(), "diag": diag} + ) + continue + if _is_hard_signal(line): + sig = _error_signature(line) + entry = hard.setdefault( + sig, {"line": i, "text": line.strip(), "count": 0} + ) + entry["count"] += 1 + continue + if "error" in low and not any(ig in line for ig in _LST_IGNORE_ERROR): + nerr += 1 + key = ("error", _error_signature(line)) + groups.setdefault( + key, {"line": i, "text": line.strip(), "count": 0} + )["count"] += 1 + if "warning" in low and not any( + ig in line for ig in _LST_IGNORE_WARNING + ): + nwarn += 1 + key = ("warning", _error_signature(line)) + groups.setdefault( + key, {"line": i, "text": line.strip(), "count": 0} + )["count"] += 1 + ordered = sorted( + ({"kind": k[0], **v} for k, v in groups.items()), + key=lambda g: (-g["count"], g["line"]), + ) + return { + "steps": steps, + "hard": sorted(hard.values(), key=lambda h: h["line"]), + "groups": ordered, + "nerr": nerr, + "nwarn": nwarn, + } + + def scan_log_error_lines(log_path: str, cap: int = 500) -> tuple[list, list]: """(lignes_erreur, lignes_avertissement) d'un log, même détection que scan_log_errors mais on RETIENT les lignes (bornées à `cap`) pour les @@ -408,6 +535,9 @@ def scan_log_error_lines(log_path: str, cap: int = 500) -> tuple[list, list]: if EXIT_MARKER in line: continue low = line.lower() + if _is_hard_signal(line) and len(errs) < cap: + errs.append(f"{i}: {line}") + continue if ( "error" in low and not any(ig in line for ig in _LST_IGNORE_ERROR) @@ -438,6 +568,13 @@ def scan_log_errors(log_path: str) -> tuple[int, int]: low = line.lower() if EXIT_MARKER in line: continue + # Un échec d'étape EST une erreur, même sans le mot « error » : sinon le + # tableau de bord affiche « 0 erreur » sur une installation ratée — + # mesuré sur erplibre-ubuntu-2604-gnome, 0 ligne « error » pour un APK + # tué par le noyau. + if _is_hard_signal(line): + nerr += 1 + continue if "error" in low and not any(ig in line for ig in _LST_IGNORE_ERROR): nerr += 1 if "warning" in low and not any( @@ -984,26 +1121,64 @@ def run_monitor(manifest_path: str, run_app: bool = True): ("q", "dismiss", "Fermer"), ] - def __init__(self, vm_name, errs, warns): + def __init__(self, vm_name, errs, warns, summary=None): super().__init__() self._vm = vm_name self._errs = errs self._warns = warns + self._sum = summary or {} def compose(self) -> ComposeResult: - with Vertical(id="errbox"): - yield Static( - f" {self._vm} — ⚠ {len(self._errs)} " + nsteps = len(self._sum.get("steps", [])) + head = ( + f" {self._vm} — ⚠ {len(self._errs)} " + f"{t('errors')} · ⚡ {len(self._warns)} {t('warnings')}" + ) + # Le nombre d'étapes en échec passe DEVANT : c'est la seule ligne du + # bandeau qui dise si l'installation a abouti. + if nsteps: + head = ( + f" {self._vm} — 🛑 {nsteps} " + f"{t('failed steps')} · ⚠ {len(self._errs)} " f"{t('errors')} · ⚡ {len(self._warns)} {t('warnings')}" - f" ({t('Esc to close')})", - id="errtitle", ) + with Vertical(id="errbox"): + yield Static(f"{head} ({t('Esc to close')})", id="errtitle") yield RichLog( id="errlog", highlight=False, markup=False, wrap=True ) def on_mount(self) -> None: log = self.query_one("#errlog", RichLog) + steps = self._sum.get("steps", []) + hard = self._sum.get("hard", []) + groups = self._sum.get("groups", []) + + # -- Le résumé, d'abord. Une étape nommée vaut mille lignes. + if steps: + log.write(f"── {t('Failed steps')} ──") + for st in steps: + log.write(f"🛑 {st['label']} ({t('line')} {st['line']})") + for line in st["diag"]: + log.write(f" {line.strip()}") + log.write("") + if hard: + log.write(f"── {t('Hard signals')} ──") + for h in hard: + mult = f" ×{h['count']}" if h["count"] > 1 else "" + log.write(f"{h['line']}:{mult} {h['text']}") + log.write("") + if groups: + # Regroupé par FORME : un journal répète la même erreur des + # centaines de fois avec un chemin qui change. + log.write(f"── {t('Grouped by shape')} ──") + for g in groups[:60]: + mark = "⚠" if g["kind"] == "error" else "⚡" + mult = f" ×{g['count']}" if g["count"] > 1 else "" + log.write(f"{mark} {g['line']}:{mult} {g['text']}") + log.write("") + + # -- Puis le détail brut, pour qui veut tout lire. if self._errs: log.write(f"── {t('errors').capitalize()} ──") for line in self._errs: @@ -1012,7 +1187,7 @@ def run_monitor(manifest_path: str, run_app: bool = True): log.write(f"── {t('warnings').capitalize()} ──") for line in self._warns: log.write(line) - if not self._errs and not self._warns: + if not (steps or hard or self._errs or self._warns): log.write(t("No error detected.")) def action_dismiss(self) -> None: @@ -1782,7 +1957,10 @@ def run_monitor(manifest_path: str, run_app: bool = True): if not vm: return errs, warns = scan_log_error_lines(vm["log"]) - self.push_screen(ErrorLinesScreen(vm["name"], errs, warns)) + summary = scan_log_summary(vm["log"]) + self.push_screen( + ErrorLinesScreen(vm["name"], errs, warns, summary) + ) def on_click(self, event) -> None: # Clic sur le sommaire de stats -> déplie / replie le détail. diff --git a/script/todo/todo_i18n.py b/script/todo/todo_i18n.py index c64a60c..e92a5cb 100644 --- a/script/todo/todo_i18n.py +++ b/script/todo/todo_i18n.py @@ -2253,6 +2253,11 @@ TRANSLATIONS = { "en": "(the hypervisor only relays; -J puts the VM last)", }, "Choice": {"fr": "Choix", "en": "Choice"}, + "failed steps": {"fr": "étapes en échec", "en": "failed steps"}, + "Failed steps": {"fr": "Étapes en échec", "en": "Failed steps"}, + "Hard signals": {"fr": "Signaux durs", "en": "Hard signals"}, + "Grouped by shape": {"fr": "Regroupé par forme", "en": "Grouped by shape"}, + "line": {"fr": "ligne", "en": "line"}, "Tick the Android emulator tool when deploying.": { "fr": "Cochez l'outil Émulateur Android au déploiement.", "en": "Tick the Android emulator tool when deploying.", diff --git a/test/test_qemu_install_summary.py b/test/test_qemu_install_summary.py new file mode 100644 index 0000000..0ddd710 --- /dev/null +++ b/test/test_qemu_install_summary.py @@ -0,0 +1,204 @@ +#!/usr/bin/env python3 +# © 2026 TechnoLibre (http://www.technolibre.ca) +# License AGPL-3.0 or later (http://www.gnu.org/licenses/agpl) +"""Résumé d'un journal d'installation : ce qui a échoué doit se voir. + +Le détail des erreurs cherchait la sous-chaîne « error ». Or le journal de +l'installation qui a réellement échoué — erplibre-ubuntu-2604-gnome, APK tué +par le noyau — ne contient AUCUNE ligne « error » : 0 sur 8765, mesuré. Le +volet annonçait donc « aucune erreur détectée » sur une installation ratée, +et le tableau de bord comptait 0 erreur. + +Ces tests fixent la règle inverse : une étape en échec, un « FAILURE » de +Gradle, une trace Python ou une mort par mémoire se voient, et le résumé les +présente AVANT les centaines de lignes du détail. +""" + +import sys +import tempfile +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "script/todo")) +import qemu_install_monitor as m # noqa: E402 + +# Journal réduit à sa forme réelle : les marqueurs de l'installation, puis +# l'échec tel que Gradle l'écrit. Aucune ligne ne contient « error ». +LOG_GRADLE_OOM = """\ +== ERPLibre mobile, SDK Android (long) == + -> venv ERPLibre (tout ce qui suit en dépend) + -> dépendances npm + -> APK debug (gradle) + ⚠ ÉCHEC : APK debug (gradle) + aucun motif connu, dernières lignes : + + FAILURE: Build failed with an exception. + + * What went wrong: + Gradle build daemon disappeared unexpectedly (it may have been killed) + ⚠ aucun APK produit +__ERPLIBRE_EXIT__ 1 +""" + + +def _log(text): + fh = tempfile.NamedTemporaryFile( + "w", suffix=".log", delete=False, encoding="utf-8" + ) + fh.write(text) + fh.close() + return fh.name + + +class TestFailedStepsAreSeen(unittest.TestCase): + def setUp(self): + self.path = _log(LOG_GRADLE_OOM) + + def tearDown(self): + Path(self.path).unlink(missing_ok=True) + + def test_the_reference_log_has_no_line_saying_error(self): + """La prémisse de tout le reste : la détection par sous-chaîne ne + pouvait RIEN trouver ici.""" + self.assertNotIn("error", LOG_GRADLE_OOM.lower()) + + def test_the_failed_step_is_named(self): + got = m.scan_log_summary(self.path) + self.assertEqual( + [s["label"] for s in got["steps"]], ["APK debug (gradle)"] + ) + + def test_the_step_carries_its_diagnostic(self): + """L'échec nomme l'étape ; c'est le diagnostic qui porte la cause.""" + diag = "\n".join(m.scan_log_summary(self.path)["steps"][0]["diag"]) + self.assertIn("FAILURE: Build failed", diag) + self.assertIn("daemon disappeared", diag) + + def test_the_exit_marker_is_not_a_diagnostic(self): + diag = "\n".join(m.scan_log_summary(self.path)["steps"][0]["diag"]) + self.assertNotIn(m.EXIT_MARKER, diag) + + def test_the_diagnostic_stops_at_the_next_step(self): + """Sinon le diagnostic avale la suite de l'installation et ne désigne + plus rien.""" + text = LOG_GRADLE_OOM + " -> étape suivante\n bruit\n" + path = _log(text) + try: + diag = "\n".join(m.scan_log_summary(path)["steps"][0]["diag"]) + finally: + Path(path).unlink(missing_ok=True) + self.assertNotIn("bruit", diag) + + def test_hard_signals_are_listed(self): + hard = " ".join( + h["text"] for h in m.scan_log_summary(self.path)["hard"] + ) + self.assertIn("FAILURE", hard) + self.assertIn("disappeared unexpectedly", hard) + + def test_the_dashboard_no_longer_counts_zero_errors(self): + """Le compte alimente le tableau de bord : « 0 erreur » sur une + installation morte est un mensonge, pas une nuance.""" + nerr, _ = m.scan_log_errors(self.path) + self.assertGreater(nerr, 0) + + def test_the_detail_pane_is_no_longer_empty(self): + errs, _ = m.scan_log_error_lines(self.path) + self.assertTrue(errs) + self.assertTrue(any("ÉCHEC" in e or "FAILURE" in e for e in errs)) + + +class TestOtherRealFailures(unittest.TestCase): + """Chaque motif dur est là parce qu'il est apparu dans un vrai journal.""" + + def _first_hard(self, line): + path = _log(f" -> étape\n{line}\n") + try: + return m.scan_log_summary(path)["hard"] + finally: + Path(path).unlink(missing_ok=True) + + def test_python_traceback(self): + self.assertTrue(self._first_hard("Traceback (most recent call last):")) + + def test_git_fatal(self): + self.assertTrue(self._first_hard("fatal: repository not found")) + + def test_apt_missing_package(self): + self.assertTrue( + self._first_hard("E: Unable to locate package python3.12-venv") + ) + + def test_kernel_oom(self): + self.assertTrue( + self._first_hard("Out of memory: Killed process 37603 (java)") + ) + + def test_missing_command(self): + self.assertTrue(self._first_hard("bash: emulator: command not found")) + + def test_a_benign_probe_is_not_a_hard_signal(self): + """« No such file or directory » sortait 5 fois sur 7 d'une sonde + bénigne (« cat: .odoo-version ») : le bruit dilue un résumé dont tout + l'intérêt est d'être court.""" + self.assertFalse( + self._first_hard("cat: .odoo-version: No such file or directory") + ) + + +class TestGrouping(unittest.TestCase): + def test_repeats_are_counted_not_repeated(self): + """Un journal répète la même erreur des centaines de fois avec un + chemin qui change : on veut « ×200 », pas 200 lignes.""" + lines = "\n".join( + f"ERROR: cannot read /var/lib/x/file{i}.txt" for i in range(200) + ) + path = _log(lines + "\n") + try: + groups = m.scan_log_summary(path)["groups"] + finally: + Path(path).unlink(missing_ok=True) + self.assertEqual(len(groups), 1) + self.assertEqual(groups[0]["count"], 200) + + def test_the_most_frequent_comes_first(self): + path = _log( + "ERROR: rare thing\n" + + "\n".join(f"ERROR: common {i}" for i in range(5)) + + "\n" + ) + try: + groups = m.scan_log_summary(path)["groups"] + finally: + Path(path).unlink(missing_ok=True) + self.assertEqual(groups[0]["count"], 5) + + def test_warnings_are_grouped_apart_from_errors(self): + path = _log("WARNING: a\nERROR: b\n") + try: + kinds = {g["kind"] for g in m.scan_log_summary(path)["groups"]} + finally: + Path(path).unlink(missing_ok=True) + self.assertEqual(kinds, {"error", "warning"}) + + +class TestQuietLogs(unittest.TestCase): + def test_a_clean_log_stays_clean(self): + """Le résumé ne doit pas inventer d'échec là où il n'y en a pas.""" + path = _log("== installation ==\n -> étape\n ✅ terminé\n") + try: + got = m.scan_log_summary(path) + finally: + Path(path).unlink(missing_ok=True) + self.assertEqual(got["steps"], []) + self.assertEqual(got["hard"], []) + self.assertEqual(got["groups"], []) + + def test_a_missing_log_is_not_a_crash(self): + got = m.scan_log_summary("/nonexistent/erplibre.log") + self.assertEqual(got["steps"], []) + self.assertEqual(got["nerr"], 0) + + +if __name__ == "__main__": + unittest.main()