[ADD] migration: request every public URL before calling it done

A migration can load every module, log nothing, and still serve 500s on
pages nobody thought to open. Measured on the real one: 2 of 33 public
URLs failed — a blog post and /contactus — after a bump the log called
successful.

The list is the sitemap, what Odoo publishes for search engines. It
cannot be read from odoo-bin shell: enumerate_pages() asks for
http.root.get_db_router(request.db) and raises « object unbound »
without a real request. So the server is started on its own port,
asked, and always stopped — a forgotten one holds the port and fails
the next bump.

Offered before the Selenium prompt, default no: it boots a server and
can take minutes.

--- FR ---

[ADD] migration : interroger chaque URL publique avant de conclure

Une migration peut charger tous ses modules, ne rien écrire au journal,
et servir quand même des 500 sur des pages que personne n'ouvre.
Mesuré : 2 URL publiques sur 33 échouaient — un billet de blogue et
/contactus — après un palier que le journal disait réussi.

La liste est le sitemap, celle qu'Odoo publie pour les moteurs. Elle ne
se lit pas depuis odoo-bin shell : enumerate_pages() réclame
http.root.get_db_router(request.db) et lève « object unbound » sans
requête réelle. Le serveur est donc démarré sur son propre port,
interrogé, et toujours arrêté — un serveur oublié tient le port et fait
échouer le palier suivant.

Proposé avant l'invite Selenium, par défaut non : cela démarre un
serveur et peut durer.

Assisted-by: Claude Opus 5
(cherry picked from commit 277e85b3d06173beeb92be792467751b31ac0696)
This commit is contained in:
Mathieu Benoit 2026-08-16 05:50:00 -04:00
parent ec6ac3a188
commit c65db269ed
4 changed files with 551 additions and 0 deletions

View file

@ -0,0 +1,239 @@
#!/usr/bin/env python3
# © 2021-2026 TechnoLibre (http://www.technolibre.ca)
# License AGPL-3.0 or later (http://www.gnu.org/licenses/agpl)
"""Request every public URL of a migrated database, and report what breaks.
Why this exists
---------------
A migration can finish, load every module, and still serve a 500 on a page
nobody thought to open. Measured on a real one: ``/blog/<blog>/post/<post>``
answered 500 because a COW copy frozen on the previous version no longer held
the section a child view xpaths into. Nothing in the migration log said so —
the module loading had succeeded.
Where the list comes from
-------------------------
``/sitemap.xml``: the list Odoo itself publishes for search engines, built by
``website.enumerate_pages()``. It covers the controller routes declared with
``sitemap=True`` and the records behind them — pages, blog posts, products.
It cannot be read from ``odoo-bin shell``: ``enumerate_pages()`` asks for
``http.root.get_db_router(request.db)`` and raises « object unbound » without
a real request. So the server is started, asked, and stopped.
What counts as a failure
------------------------
Any status >= 400. The sitemap is the PUBLIC list: a page listed there and
answering 403 or 404 is as wrong as one answering 500, just less loud.
Exit codes: 0 every URL answered, 1 some failed, 2 the tool failed.
"""
import argparse
import os
import re
import subprocess
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
sys.path.append(
os.path.normpath(os.path.join(os.path.dirname(__file__), "..", "..", ".."))
)
try:
from script.todo.todo_i18n import t
except Exception: # pragma: no cover - repli si i18n indisponible
def t(key: str) -> str:
return key
RE_LOC = re.compile(r"<loc>\s*([^<\s]+)\s*</loc>", re.I)
# Un port à part : la migration tourne souvent à côté d'une instance vivante,
# et lui voler 8069 ferait échouer le test pour une raison sans rapport.
DEFAULT_PORT = 8169
def start_server(database, port, config_path="./config.conf"):
"""Démarrer Odoo sur la base, sans écrire dans le terminal appelant."""
return subprocess.Popen(
[
"./run.sh",
"-c",
config_path,
"-d",
database,
"--http-port",
str(port),
"--log-level=warn",
],
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
)
def fetch(url, timeout=30):
"""(statut, corps). Statut 0 quand la connexion elle-même échoue."""
try:
with urllib.request.urlopen(url, timeout=timeout) as answer:
return answer.getcode(), answer.read().decode(
"utf-8", errors="replace"
)
except urllib.error.HTTPError as exc:
return exc.code, exc.read().decode("utf-8", errors="replace")
except Exception:
return 0, ""
def wait_ready(base_url, timeout=180, sleep=2):
"""Attendre que le serveur réponde. False s'il n'est jamais venu."""
deadline = time.time() + timeout
while time.time() < deadline:
status, _body = fetch(base_url + "/web/login", timeout=5)
if status:
return True
time.sleep(sleep)
return False
def sitemap_urls(base_url):
"""Les URL du sitemap, index compris, ramenées sur l'hôte local.
Le sitemap porte le domaine du site (technolibre.ca) ; on teste une base
servie en local. Garder le domaine ferait interroger la production —
c'est le genre d'erreur qui ne se voit qu'après.
"""
status, body = fetch(base_url + "/sitemap.xml")
if not status or status >= 400:
return [], status
lst_loc = RE_LOC.findall(body)
# Un index de sitemaps ne contient que des sitemaps : on descend d'un cran.
if "<sitemapindex" in body.lower():
lst_page = []
for loc in lst_loc:
_status, sub = fetch(local_url(base_url, loc))
lst_page.extend(RE_LOC.findall(sub))
lst_loc = lst_page
seen, lst_url = set(), []
for loc in lst_loc:
url = local_url(base_url, loc)
if url not in seen:
seen.add(url)
lst_url.append(url)
return lst_url, status
def local_url(base_url, loc):
"""Remplacer le schéma et l'hôte du sitemap par ceux qu'on teste."""
parsed = urllib.parse.urlparse(loc)
path = parsed.path or "/"
if parsed.query:
path += "?" + parsed.query
return base_url.rstrip("/") + path
def check_urls(lst_url, timeout=30):
"""[(url, statut)] pour celles qui ont échoué."""
lst_failure = []
for url in lst_url:
status, _body = fetch(url, timeout=timeout)
if status == 0 or status >= 400:
lst_failure.append((url, status))
return lst_failure
def render(lst_url, lst_failure):
if not lst_url:
return f"⚠️ {t('The sitemap listed no URL: nothing was tested.')}\n"
if not lst_failure:
return (
f"✅ -> {len(lst_url)} {t('public URL(s) answered without error.')}"
"\n"
)
lines = [
f"❌ {len(lst_failure)} {t('of')} {len(lst_url)}"
f" {t('public URL(s) failed')} :"
]
for url, status in lst_failure:
label = status or t("no answer")
lines.append(f" [{label}] {url}")
lines.append(
f" {t('A page listed for search engines that does not answer is')}"
f" {t('a page your visitors do not reach either.')}"
)
return "\n".join(lines) + "\n"
def run(database, port, config_path, limit=None, timeout=30, boot=180):
"""Démarrer, interroger, arrêter. Rend (urls, échecs)."""
base_url = f"http://127.0.0.1:{port}"
server = start_server(database, port, config_path)
try:
if not wait_ready(base_url, timeout=boot):
raise RuntimeError(
f"{t('The server never answered on')} {base_url}"
)
lst_url, status = sitemap_urls(base_url)
if not status:
raise RuntimeError(f"{t('Could not read')} {base_url}/sitemap.xml")
if limit:
lst_url = lst_url[:limit]
return lst_url, check_urls(lst_url, timeout=timeout)
finally:
server.terminate()
try:
server.wait(timeout=30)
except subprocess.TimeoutExpired:
server.kill()
def main(argv=None):
parser = argparse.ArgumentParser(
description=(
"Start Odoo on a database, request every URL of its sitemap, and"
" report those that fail."
)
)
parser.add_argument("-d", "--database", required=True)
parser.add_argument("-c", "--config", default="./config.conf")
parser.add_argument("-p", "--port", type=int, default=DEFAULT_PORT)
parser.add_argument(
"--limit", type=int, default=None, help="test only the first N URLs"
)
parser.add_argument("--timeout", type=int, default=30)
parser.add_argument(
"--boot-timeout",
type=int,
default=180,
help="how long to wait for the server to answer",
)
config = parser.parse_args(argv)
print(
f"⧖ {t('Starting Odoo on')} '{config.database}'"
f" ({t('port')} {config.port})…"
)
try:
lst_url, lst_failure = run(
config.database,
config.port,
config.config,
limit=config.limit,
timeout=config.timeout,
boot=config.boot_timeout,
)
except RuntimeError as exc:
print(f"❌ {exc}")
return 2
print(render(lst_url, lst_failure))
return 1 if lst_failure else 0
if __name__ == "__main__":
sys.exit(main())

View file

@ -4978,6 +4978,50 @@ TRANSLATIONS = {
"fr": "pièce(s) jointe(s) effacée(s).",
"en": "attachment(s) deleted.",
},
"Request every public URL of this database now?": {
"fr": "Interroger maintenant toutes les URL publiques de cette base ?",
"en": "Request every public URL of this database now?",
},
"The sitemap listed no URL: nothing was tested.": {
"fr": "Le sitemap n'a listé aucune URL : rien n'a été testé.",
"en": "The sitemap listed no URL: nothing was tested.",
},
"public URL(s) answered without error.": {
"fr": "URL publique(s) ont répondu sans erreur.",
"en": "public URL(s) answered without error.",
},
"public URL(s) failed": {
"fr": "URL publique(s) ont échoué",
"en": "public URL(s) failed",
},
"no answer": {
"fr": "aucune réponse",
"en": "no answer",
},
"A page listed for search engines that does not answer is": {
"fr": "Une page listée pour les moteurs de recherche qui ne répond pas est",
"en": "A page listed for search engines that does not answer is",
},
"a page your visitors do not reach either.": {
"fr": "une page que vos visiteurs n'atteignent pas non plus.",
"en": "a page your visitors do not reach either.",
},
"The server never answered on": {
"fr": "Le serveur n'a jamais répondu sur",
"en": "The server never answered on",
},
"Could not read": {
"fr": "Impossible de lire",
"en": "Could not read",
},
"Starting Odoo on": {
"fr": "Démarrage d'Odoo sur",
"en": "Starting Odoo on",
},
"port": {
"fr": "port",
"en": "port",
},
"Nothing to decide yet": {
"fr": "Rien à décider pour l'instant",
"en": "Nothing to decide yet",

View file

@ -2552,6 +2552,13 @@ class TodoUpgrade:
# Update config without OCA_OpenUpgrade
cmd_update_config = f"./script/git/git_repo_update_group.py && ./script/generate_config.sh"
self.todo_upgrade_execute(cmd_update_config)
# Une migration peut charger tous ses modules et servir
# quand même un 500 sur une page que personne n'ouvre. Mesuré
# ici : /blog/<blog>/post/<billet> et /contactus, alors que
# le journal de migration n'avait rien signalé.
self.prompt_smoke_public_url(database_name_upgrade)
print(f"[y] {t('Open the server with Selenium')}")
status = (
input(
@ -2857,6 +2864,33 @@ class TodoUpgrade:
f" -d {database_name} {args} --apply"
)
def prompt_smoke_public_url(self, database_name):
"""Proposer d'interroger toutes les pages publiques de la base.
La liste vient du sitemap — celle qu'Odoo publie pour les moteurs de
recherche. Une page qui y figure et ne répond pas est une page que
les visiteurs n'atteignent pas non plus.
« non » par défaut : cela démarre un serveur et peut prendre quelques
minutes sur un gros site, et rien n'oblige à le faire à chaque
palier.
"""
answer = (
self.ask_gate(
f"💬 {t('Request every public URL of this database now?')}"
f" (y/N, {t('(b = go back to a previous step)')}) : "
)
.strip()
.lower()
)
if answer != "y":
return
self.run_on_terminal(
f"{PYTHON_BIN}"
" ./script/odoo/migration/smoke_public_url.py"
f" -d {database_name}"
)
def show_cow_drift(self, database_name, next_version, mode="diff"):
"""Montre les copies COW à risque. Ne touche à rien.

234
test/test_smoke_public_url.py Executable file
View file

@ -0,0 +1,234 @@
#!/usr/bin/env python3
# © 2021-2026 TechnoLibre (http://www.technolibre.ca)
# License AGPL-3.0 or later (http://www.gnu.org/licenses/agpl)
"""Une migration peut réussir et servir quand même des 500.
Mesuré sur une vraie migration 12 → 13 : tous les modules chargés, aucune
erreur au journal, et pourtant deux pages publiques sur trente-trois
répondaient 500 — un billet de blogue et /contactus. Rien ne les distinguait
avant de les demander.
La liste vient du sitemap, celle qu'Odoo publie pour les moteurs de
recherche. Elle ne se lit PAS depuis « odoo-bin shell » : enumerate_pages()
réclame http.root.get_db_router(request.db) et lève « object unbound » sans
requête réelle. D'où le serveur démarré, interrogé, arrêté.
"""
import os
import sys
import unittest
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, os.path.join(REPO, "script", "odoo", "migration"))
import smoke_public_url as smoke # noqa: E402
class TestTheUrlAreBroughtHome(unittest.TestCase):
"""Le sitemap porte le domaine du site ; on teste une base locale."""
def test_the_host_is_replaced(self):
# Garder le domaine ferait interroger la PRODUCTION depuis une
# migration. C'est le genre d'erreur qui ne se voit qu'après.
url = smoke.local_url(
"http://127.0.0.1:8169", "https://technolibre.ca/blog/x-3/post/y-5"
)
self.assertEqual(url, "http://127.0.0.1:8169/blog/x-3/post/y-5")
def test_a_query_string_survives(self):
url = smoke.local_url("http://127.0.0.1:8169", "https://a.ca/p?x=1")
self.assertEqual(url, "http://127.0.0.1:8169/p?x=1")
def test_a_bare_domain_becomes_the_root(self):
url = smoke.local_url("http://127.0.0.1:8169", "https://a.ca")
self.assertEqual(url, "http://127.0.0.1:8169/")
def test_a_trailing_slash_on_the_base_does_not_double(self):
url = smoke.local_url("http://127.0.0.1:8169/", "https://a.ca/p")
self.assertEqual(url, "http://127.0.0.1:8169/p")
class TestReadingTheSitemap(unittest.TestCase):
def setUp(self):
self.answers = {}
self.original = smoke.fetch
smoke.fetch = lambda url, timeout=30: self.answers.get(url, (404, ""))
self.addCleanup(setattr, smoke, "fetch", self.original)
def test_a_plain_sitemap(self):
self.answers["http://h/sitemap.xml"] = (
200,
"<urlset><loc>https://a.ca/x</loc><loc>https://a.ca/y</loc></urlset>",
)
lst_url, status = smoke.sitemap_urls("http://h")
self.assertEqual(status, 200)
self.assertEqual(lst_url, ["http://h/x", "http://h/y"])
def test_an_index_is_followed(self):
# Odoo découpe le sitemap au-delà d'un certain nombre de pages : ne
# pas descendre d'un cran ferait tester deux fichiers XML au lieu du
# site, et conclure « tout va bien ».
self.answers["http://h/sitemap.xml"] = (
200,
"<sitemapindex><loc>https://a.ca/sitemap-1-1.xml</loc>"
"</sitemapindex>",
)
self.answers["http://h/sitemap-1-1.xml"] = (
200,
"<urlset><loc>https://a.ca/deep</loc></urlset>",
)
lst_url, _status = smoke.sitemap_urls("http://h")
self.assertEqual(lst_url, ["http://h/deep"])
def test_duplicates_are_dropped(self):
self.answers["http://h/sitemap.xml"] = (
200,
"<urlset><loc>https://a.ca/x</loc><loc>https://a.ca/x</loc></urlset>",
)
lst_url, _status = smoke.sitemap_urls("http://h")
self.assertEqual(lst_url, ["http://h/x"])
def test_an_unreachable_sitemap_is_reported_not_taken_as_empty(self):
# Sans cela, un sitemap injoignable se lirait comme « aucune page à
# tester », c'est-à-dire comme un succès.
lst_url, status = smoke.sitemap_urls("http://h")
self.assertEqual(lst_url, [])
self.assertEqual(status, 404)
class TestWhatCountsAsAFailure(unittest.TestCase):
def setUp(self):
self.answers = {}
self.original = smoke.fetch
smoke.fetch = lambda url, timeout=30: self.answers.get(url, (200, ""))
self.addCleanup(setattr, smoke, "fetch", self.original)
def test_a_500_fails(self):
self.answers["http://h/bad"] = (500, "")
self.assertEqual(
smoke.check_urls(["http://h/bad"]), [("http://h/bad", 500)]
)
def test_a_404_fails_too(self):
# Le sitemap est la liste PUBLIQUE : une page qui y figure et répond
# 404 est aussi fausse qu'un 500, juste plus discrète.
self.answers["http://h/gone"] = (404, "")
self.assertEqual(len(smoke.check_urls(["http://h/gone"])), 1)
def test_no_answer_at_all_fails(self):
self.answers["http://h/dead"] = (0, "")
self.assertEqual(
smoke.check_urls(["http://h/dead"]), [("http://h/dead", 0)]
)
def test_a_200_passes(self):
self.assertEqual(smoke.check_urls(["http://h/ok"]), [])
class TestTheReport(unittest.TestCase):
def setUp(self):
from script.todo import todo_i18n
self.addCleanup(
setattr, todo_i18n, "_current_lang", todo_i18n._current_lang
)
todo_i18n._current_lang = "en"
def test_all_green_says_how_many(self):
text = smoke.render(["a", "b"], [])
self.assertIn("2", text)
self.assertIn("✅", text)
def test_a_failure_shows_the_status_and_the_url(self):
text = smoke.render(["a"], [("http://h/blog/x", 500)])
self.assertIn("500", text)
self.assertIn("http://h/blog/x", text)
def test_an_empty_sitemap_is_not_a_success(self):
# « rien testé » ne doit pas se lire « rien de cassé ».
text = smoke.render([], [])
self.assertIn("⚠️", text)
class TestTheServerIsAlwaysStopped(unittest.TestCase):
"""Un serveur oublié tient le port et fait échouer le palier suivant."""
def test_it_is_terminated_even_when_the_sitemap_fails(self):
stopped = []
class FakeServer:
def terminate(self):
stopped.append("terminate")
def wait(self, timeout=None):
return 0
def kill(self):
stopped.append("kill")
original_start = smoke.start_server
original_wait = smoke.wait_ready
smoke.start_server = lambda db, port, cfg="./config.conf": FakeServer()
smoke.wait_ready = lambda base, timeout=180, sleep=2: False
self.addCleanup(setattr, smoke, "start_server", original_start)
self.addCleanup(setattr, smoke, "wait_ready", original_wait)
with self.assertRaises(RuntimeError):
smoke.run("db", 8169, "./config.conf")
self.assertEqual(stopped, ["terminate"])
def test_a_distinct_port_by_default(self):
# 8069 est souvent pris par l'instance de travail : le lui voler
# ferait échouer le test pour une raison sans rapport.
self.assertNotEqual(smoke.DEFAULT_PORT, 8069)
class TestTheMigrationOffersIt(unittest.TestCase):
def setUp(self):
from script.todo import todo_i18n
self.addCleanup(
setattr, todo_i18n, "_current_lang", todo_i18n._current_lang
)
todo_i18n._current_lang = "en"
def run_prompt(self, answer):
from script.todo.todo_upgrade import TodoUpgrade
upgrade = TodoUpgrade.__new__(TodoUpgrade)
upgrade.dct_progression = {}
upgrade.lst_command_executed = []
upgrade.write_config = lambda: None
lst_cmd = []
upgrade.run_on_terminal = lambda cmd: lst_cmd.append(cmd) or 0
upgrade.ask_gate = lambda prompt: answer
upgrade.prompt_smoke_public_url("db_upgrade_13")
return lst_cmd
def test_the_default_runs_nothing(self):
# Cela démarre un serveur et peut durer : pas à chaque palier sans
# qu'on l'ait demandé.
self.assertEqual(self.run_prompt(""), [])
def test_yes_runs_it_on_the_upgraded_database(self):
lst_cmd = self.run_prompt("y")
self.assertEqual(len(lst_cmd), 1)
self.assertIn("smoke_public_url.py", lst_cmd[0])
self.assertIn("-d db_upgrade_13", lst_cmd[0])
def test_it_is_asked_before_the_selenium_prompt(self):
# Après, la question arriverait une fois le navigateur ouvert : on
# aurait déjà cherché à la main ce que le test nomme.
import inspect
from script.todo.todo_upgrade import TodoUpgrade
source = inspect.getsource(TodoUpgrade.execute_odoo_upgrade)
self.assertLess(
source.index("prompt_smoke_public_url"),
source.index("Open the server with Selenium"),
)
if __name__ == "__main__":
unittest.main()