diff --git a/docker-compose.yml b/docker-compose.yml index 9d293b23..af678321 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -179,8 +179,10 @@ services: # Job Worker - background service for processing async queue jobs # Runs alongside openspp/openspp-dev to execute delayed imports, etc. - # The runner auto-discovers databases every 5 minutes, so after installing - # new modules, jobs will be picked up within 5 minutes (or restart to force). + # While openspp installs or upgrades modules (ODOO_INIT_MODULES / + # ODOO_UPDATE_MODULES), the worker's upgrade gate pauses it: no registry + # load and no new jobs until the install/upgrade has finished. It then + # resumes on its own and picks up the new modules. jobworker: image: openspp-dev profiles: @@ -214,6 +216,19 @@ services: "-c", "/etc/odoo/odoo.conf", ] + healthcheck: + # Fresh heartbeat file = supervisor loop alive and no database degraded. + test: + ["CMD", "python", "/mnt/extra-addons/odoo-job-worker/job_worker_healthcheck.py"] + interval: 30s + timeout: 10s + start_period: 60s + retries: 3 + # Docker's default of 10s SIGKILLs jobs the runner would give up to 30s + # *per database* to finish (databases are stopped one after another), and + # each killed job loses a retry attempt. 60s covers one or two databases; + # raise it if this worker serves more. + stop_grace_period: 60s networks: - openspp diff --git a/docker/README.md b/docker/README.md index ccf3d300..19f0926f 100644 --- a/docker/README.md +++ b/docker/README.md @@ -245,6 +245,52 @@ docker compose -f docker/docker-compose.production.yml build --no-cache docker compose -f docker/docker-compose.production.yml up -d ``` +#### Addon code lives in a named volume + +Both production stacks mount the `odoo_addons` named volume over `/mnt/extra-addons`. +Docker copies the image's files into a named volume only while the volume is empty. So +after the first deployment, the addons there are whatever the first image shipped, not +what a newly pulled or rebuilt image contains. That includes the job worker and its +`job_worker_healthcheck.py`. Check what the containers actually run after an update: + +```bash +docker compose -f docker/docker-compose.production.yml exec queue-worker \ + grep '"version"' /mnt/extra-addons/odoo-job-worker/job_worker/__manifest__.py +``` + +If it is out of date and nothing else was installed into the volume, stop the stack and +remove the volume (`docker volume rm _odoo_addons`). It is re-seeded from the +image on the next start. + +#### Upgrading modules with the queue worker running + +Nothing should run against a database while its modules are being installed or upgraded. +Running jobs, and requests and crons in the `odoo` service, hold locks that can stall +the upgrade's schema changes. They also read a schema that is changing under them. The +clean upgrade stops both, runs the upgrade as a one-shot, then starts both again. Users +see downtime for as long as the upgrade runs: + +```bash +COMPOSE="docker compose -f docker/docker-compose.production.yml" +$COMPOSE stop queue-worker odoo # SIGTERM; running jobs get stop_grace_period to finish +$COMPOSE run --rm odoo odoo -u --stop-after-init --no-http +$COMPOSE up -d odoo queue-worker +``` + +The one-shot takes the database from the container's generated `/etc/odoo/odoo.conf` +(`db_name = ${DB_NAME}`), so there is no `-d` to get right in the host shell. + +If the queue worker is left running, or restarts during an upgrade, it pauses by itself. +_This requires `job_worker` 19.0.1.3.0 or later (OpenSPP/odoo-job-worker#33). Images +built with the default `ODOO_JOB_WORKER_REF=19.0` include it; a build pinned to an older +commit does not, and neither does an `odoo_addons` volume seeded before it (see above)._ +The `job_worker` upgrade gate stops the worker from loading its registry or taking new +jobs while any module is `to install`/`to upgrade`/`to remove`, or while its code and +the database disagree on module versions. It resumes when the upgrade has finished. See +the `job_worker` docs, "Upgrading modules while the worker runs", for the log lines and +the `JOB_WORKER_UPGRADE_*` settings. A pause longer than an hour turns the queue +worker's healthcheck unhealthy. + ### Antivirus Scanning (Optional) ClamAV antivirus scanning is available as an optional profile. Enable it when: @@ -380,6 +426,12 @@ to be reproducible. The container exposes a health endpoint at `/web/health` on port 8069. +The queue worker has no HTTP endpoint. Its healthcheck runs `job_worker_healthcheck.py`, +which checks that the runner's heartbeat file is fresh. The runner stops refreshing it +when a database is quarantined or stuck in database-error recovery. With `job_worker` +19.0.1.3.0 or later, it also stops when a database has been paused for a module upgrade +for longer than `JOB_WORKER_UPGRADE_PAUSE_UNHEALTHY_AFTER` (default one hour). + ## Ports - `8069` - HTTP (Odoo web interface) diff --git a/docker/docker-compose.nginx.yml b/docker/docker-compose.nginx.yml index ff91c81f..5f73505a 100644 --- a/docker/docker-compose.nginx.yml +++ b/docker/docker-compose.nginx.yml @@ -375,6 +375,21 @@ services: networks: - openspp-prod restart: unless-stopped + healthcheck: + # Fresh heartbeat file = supervisor loop alive and no database degraded + # (including a pause for a module upgrade that never finishes). The file + # lives in /tmp, the tmpfs below. + test: + ["CMD", "python", "/mnt/extra-addons/odoo-job-worker/job_worker_healthcheck.py"] + interval: 30s + timeout: 10s + start_period: 60s + retries: 3 + # Docker's default of 10s SIGKILLs jobs the runner would give up to 30s + # *per database* to finish (databases are stopped one after another), and + # each killed job loses a retry attempt. 60s covers one or two databases; + # raise it if this worker serves more. + stop_grace_period: 60s # NOTE: read_only not used — same reason as odoo service (entrypoint writes config) tmpfs: - /tmp:size=256M diff --git a/docker/docker-compose.production.yml b/docker/docker-compose.production.yml index 0eba74f2..d07a8076 100644 --- a/docker/docker-compose.production.yml +++ b/docker/docker-compose.production.yml @@ -292,11 +292,19 @@ services: - openspp-prod restart: always healthcheck: - test: ["CMD-SHELL", "pgrep -f 'odoo.addons.job_worker.cli' || exit 1"] + # Fresh heartbeat file = supervisor loop alive and no database degraded + # (including a pause for a module upgrade that never finishes). + test: + ["CMD", "python", "/mnt/extra-addons/odoo-job-worker/job_worker_healthcheck.py"] interval: 30s timeout: 10s start_period: 60s retries: 3 + # Docker's default of 10s SIGKILLs jobs the runner would give up to 30s + # *per database* to finish (databases are stopped one after another), and + # each killed job loses a retry attempt. 60s covers one or two databases; + # raise it if this worker serves more. + stop_grace_period: 60s deploy: resources: limits: diff --git a/spp_hide_menus_base/README.rst b/spp_hide_menus_base/README.rst index 6ff430a3..5831cb7d 100644 --- a/spp_hide_menus_base/README.rst +++ b/spp_hide_menus_base/README.rst @@ -103,6 +103,21 @@ Dependencies Changelog ========= +19.0.2.1.1 +~~~~~~~~~~ + +- ``_register_hook`` now re-hides menus only after a registry load that + installed or updated modules (``registry.updated_modules``), which is + every path that can reset ``group_ids``. Other loads — each HTTP and + cron worker, and the job worker, reloading after another process + signals a change — no longer repeat the same ``ir.ui.menu`` / + ``spp.hide.menu`` writes, which raced the upgrading process for the + same rows during deploys. +- Operator-facing change: hidden menus that become visible for any + reason other than a module update (for example a manual edit of a + menu's groups) are no longer re-hidden by a plain restart. Upgrade the + module (``-u spp_hide_menus_base``) to re-apply hiding. + 19.0.2.1.0 ~~~~~~~~~~ diff --git a/spp_hide_menus_base/__manifest__.py b/spp_hide_menus_base/__manifest__.py index 79db6476..274d4282 100644 --- a/spp_hide_menus_base/__manifest__.py +++ b/spp_hide_menus_base/__manifest__.py @@ -5,7 +5,7 @@ { "name": "OpenSPP Hide Non-OpenSPP Menus: Base", "category": "OpenSPP", - "version": "19.0.2.1.0", + "version": "19.0.2.1.1", "summary": "Administrators can manage the visibility of OpenSPP navigation menus, streamlining the user interface for specific user groups. The module modifies ir.ui.menu records to control menu visibility, providing a foundation for other modules to selectively hide non-essential navigation items.", "sequence": 1, "author": "OpenSPP.org", diff --git a/spp_hide_menus_base/models/ir_module_module.py b/spp_hide_menus_base/models/ir_module_module.py index d2c0c628..5da89dbc 100644 --- a/spp_hide_menus_base/models/ir_module_module.py +++ b/spp_hide_menus_base/models/ir_module_module.py @@ -105,9 +105,24 @@ def _register_hook(self): # (button_immediate_*). Upgrades through the base.module.upgrade # wizard or the CLI (-u) reload menu XML — resetting group_ids via # noupdate="0" — but never call next(), leaving hidden menus visible. - # _register_hook runs at the end of every registry load (startup, - # install, upgrade — all paths), so re-applying hiding here keeps - # menus hidden regardless of how the upgrade was triggered. + # Each of those paths ends in a registry load that installed or + # updated modules, and _register_hook runs at the end of every + # registry load, so re-applying hiding here keeps menus hidden + # regardless of how the upgrade was triggered. + # + # The pass runs only in a registry that installed or updated modules, + # which Odoo lists in registry.updated_modules: only that can have + # reset group_ids. Every other load (a plain restart, or each HTTP and + # cron worker and the job worker reloading after another process + # signals a change) would repeat the same writes on the same + # ir.ui.menu and spp.hide.menu rows, racing the upgrading process for + # them while it may still be running. + # + # updated_modules lasts as long as that registry, so a later model + # re-setup in the updating process (adding a custom field, say) runs + # _register_hook again and repeats the pass. hide_menus() is + # idempotent, so that is harmless. res = super()._register_hook() - self.hide_menus() + if self.env.registry.updated_modules: + self.hide_menus() return res diff --git a/spp_hide_menus_base/readme/HISTORY.md b/spp_hide_menus_base/readme/HISTORY.md index a56bacf8..d3163e62 100644 --- a/spp_hide_menus_base/readme/HISTORY.md +++ b/spp_hide_menus_base/readme/HISTORY.md @@ -1,3 +1,16 @@ +### 19.0.2.1.1 + +- ``_register_hook`` now re-hides menus only after a registry load that + installed or updated modules (``registry.updated_modules``), which is every + path that can reset ``group_ids``. Other loads — each HTTP and cron worker, + and the job worker, reloading after another process signals a change — no + longer repeat the same ``ir.ui.menu`` / ``spp.hide.menu`` writes, which raced + the upgrading process for the same rows during deploys. +- Operator-facing change: hidden menus that become visible for any reason + other than a module update (for example a manual edit of a menu's groups) are + no longer re-hidden by a plain restart. Upgrade the module + (``-u spp_hide_menus_base``) to re-apply hiding. + ### 19.0.2.1.0 - Enforce ``UNIQUE(menu_id)`` on ``spp.hide.menu``: a second configuration row diff --git a/spp_hide_menus_base/static/description/index.html b/spp_hide_menus_base/static/description/index.html index 47178b92..74b8f41b 100644 --- a/spp_hide_menus_base/static/description/index.html +++ b/spp_hide_menus_base/static/description/index.html @@ -474,6 +474,22 @@

Changelog

+

19.0.2.1.1

+
    +
  • _register_hook now re-hides menus only after a registry load that +installed or updated modules (registry.updated_modules), which is +every path that can reset group_ids. Other loads — each HTTP and +cron worker, and the job worker, reloading after another process +signals a change — no longer repeat the same ir.ui.menu / +spp.hide.menu writes, which raced the upgrading process for the +same rows during deploys.
  • +
  • Operator-facing change: hidden menus that become visible for any +reason other than a module update (for example a manual edit of a +menu’s groups) are no longer re-hidden by a plain restart. Upgrade the +module (-u spp_hide_menus_base) to re-apply hiding.
  • +
+
+

19.0.2.1.0

  • Enforce UNIQUE(menu_id) on spp.hide.menu: a second @@ -495,7 +511,7 @@

    19.0.2.1.0

    Target the existing record or drop the seed.
-
+

19.0.2.0.1

  • Keep hidden menus hidden after a module upgrade resets their @@ -505,7 +521,7 @@

    19.0.2.0.1

    next().
-
+

19.0.2.0.0

  • Initial migration to OpenSPP2
  • diff --git a/spp_hide_menus_base/tests/test_hide_menu.py b/spp_hide_menus_base/tests/test_hide_menu.py index 014dfe95..7a21f465 100644 --- a/spp_hide_menus_base/tests/test_hide_menu.py +++ b/spp_hide_menus_base/tests/test_hide_menu.py @@ -251,16 +251,17 @@ def test_hide_menus_reapplies_after_reset(self): ) def test_register_hook_rehides_after_reset(self): - """``_register_hook()`` re-applies hiding on every registry load. + """``_register_hook()`` re-applies hiding after a load that updated modules. ``ir.module.module.next()`` only runs on the immediate install/upgrade path (button_immediate_*). Upgrades performed through the ``base.module.upgrade`` wizard or the CLI (``-u``) reload module XML — resetting ``group_ids`` — but never call - ``next()``. ``_register_hook`` runs at the end of *every* registry - load, so it must re-hide regardless of the upgrade path. We can't - run a real upgrade in a test, so we reset the groups by hand and - call the hook the way the loader does. + ``next()``. Every one of those paths ends in a registry load that + installed or updated modules, and ``_register_hook`` runs at the + end of it, so it must re-hide regardless of the upgrade path. We + can't run a real upgrade in a test, so we reset the groups by hand + and call the hook the way the loader does after an update. """ IrModuleModule = self.env["ir.module.module"] HideMenu = self.env["spp.hide.menu"] @@ -285,15 +286,43 @@ def test_register_hook_rehides_after_reset(self): target.write({"group_ids": [Command.set([reset_groups.id])]}) self.assertNotIn(hide_group, target.group_ids) - IrModuleModule._register_hook() + with patch.object(self.env.registry, "updated_modules", ["spp_hide_menus_base"]): + IrModuleModule._register_hook() self.assertIn( hide_group, target.group_ids, - "_register_hook() should re-hide menus on every registry load, " - "covering upgrade paths that never call next()", + "_register_hook() should re-hide menus after a registry load that " + "updated modules, covering upgrade paths that never call next()", ) + def test_register_hook_leaves_menus_alone_when_no_module_was_updated(self): + """A registry load that installed or updated nothing must not touch menus. + + Every process reloads its registry once another one signals a change: + HTTP workers, cron workers, and a job worker running beside them. The + menu writes belong to the one process that updated modules. The others + repeating them on the same ``ir.ui.menu`` and ``spp.hide.menu`` rows, + while an upgrade may still be running, only adds lock waits and + duplicate-key races to every deploy. + """ + IrModuleModule = self.env["ir.module.module"] + with ( + patch.object(self.env.registry, "updated_modules", []), + patch.object(type(IrModuleModule), "hide_menus") as hide_menus, + ): + IrModuleModule._register_hook() + hide_menus.assert_not_called() + + def test_register_hook_hides_menus_when_a_module_was_updated(self): + IrModuleModule = self.env["ir.module.module"] + with ( + patch.object(self.env.registry, "updated_modules", ["spp_programs"]), + patch.object(type(IrModuleModule), "hide_menus") as hide_menus, + ): + IrModuleModule._register_hook() + hide_menus.assert_called_once_with() + def test_hide_menus_skips_unknown_modules(self): """An ir.module.module record whose name isn't in MENU_APP must be ignored by hide_menus() — no spp.hide.menu record is created for it.