Repair monitor cluster upgrade without blocking K1 package configuration

This commit is contained in:
DCCONSTRUCTIONS
2026-09-08 10:44:27 +03:00
parent 6492ebcdee
commit 14f8375713
11 changed files with 160 additions and 21 deletions
+2 -2
View File
@@ -53,13 +53,13 @@ def watch(collector):
"+",
"_SYSTEMD_UNIT=mission-core-node-monitor.service",
"+",
"_SYSTEMD_UNIT=postgresql@16-ndc-monitor.service",
"_SYSTEMD_UNIT=postgresql@16-ndcmonitor.service",
]
for unit in (
"mission-core-node.service",
"mission-core-k1.service",
"mission-core-node-monitor.service",
"postgresql@16-ndc-monitor.service",
"postgresql@16-ndcmonitor.service",
):
args.extend(["+", "UNIT=" + unit])
while not collector.stop.is_set():
+1 -1
View File
@@ -296,7 +296,7 @@ class LinuxMetrics:
)
groups = self.paths("sys/fs/cgroup/system.slice/mission-core*.service")
groups += self.paths(
"sys/fs/cgroup/system.slice/system-postgresql.slice/postgresql@16-ndc-monitor.service"
"sys/fs/cgroup/system.slice/system-postgresql.slice/postgresql@16-ndcmonitor.service"
)
groups += self.paths(
"sys/fs/cgroup/user.slice/user-*.slice/user@*.service/app.slice/app-gnome-org.nodedc.MissionCoreNode-*.scope"
+1 -1
View File
@@ -11,7 +11,7 @@ import sys
ROOT = Path(__file__).resolve().parents[1]
VERSION = "0.8.15"
VERSION = "0.8.16"
sys.path.insert(0, str(ROOT.parents[1] / "scripts/packaging"))
from debian import package
@@ -30,11 +30,27 @@ fi
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-remove \
"$mc_release_dir/timescaledb-2-loader-postgresql-16_2.29.2~ubuntu24.04-1615_amd64.deb" \
"$mc_release_dir/timescaledb-2-oss-postgresql-16_2.29.2~ubuntu24.04-1615_amd64.deb" \
"$mc_release_dir/mission-core-node_0.8.15_amd64.deb" \
"$mc_release_dir/mission-core-node_0.8.16_amd64.deb" \
"$mc_release_dir/mission-core-xgrids-k1_0.1.14_amd64.deb"
systemctl is-active --quiet mission-core-node.service
systemctl is-active --quiet mission-core-k1.service
systemctl is-active --quiet postgresql@16-ndc-monitor.service
runuser -u mission-core-node -- /usr/bin/python3 - <<'K1CHECK'
import http.client,json,socket,time
deadline=time.monotonic()+20
while True:
c=http.client.HTTPConnection('k1',timeout=3)
c.sock=socket.socket(socket.AF_UNIX);c.sock.settimeout(3)
try:
c.sock.connect('/run/mission-core-k1/driver.sock')
c.request('GET','/status');r=c.getresponse();v=json.load(r)
if r.status==200 and v.get('available') is True:break
except (OSError, ValueError, http.client.HTTPException):pass
finally:c.close()
if time.monotonic()>deadline:raise SystemExit('K1 enrollment service did not become ready')
time.sleep(1)
print('K1 enrollment service: ready')
K1CHECK
systemctl is-active --quiet postgresql@16-ndcmonitor.service
systemctl is-active --quiet mission-core-node-monitor.service
runuser -u mission-core-monitor -- /usr/bin/python3 - <<'PY'
import http.client,json,socket,time
@@ -53,4 +69,4 @@ while True:
time.sleep(1)
print('Local telemetry: ready')
PY
printf '%s\n' 'Mission Core Node R18: package and local archive checks passed.'
printf '%s\n' 'Mission Core Node R19: package and local archive checks passed.'
@@ -1,7 +1,7 @@
[Unit]
Description=Mission Core onboard system telemetry collector
After=postgresql@16-ndc-monitor.service
Wants=postgresql@16-ndc-monitor.service
After=postgresql@16-ndcmonitor.service
Wants=postgresql@16-ndcmonitor.service
[Service]
Type=simple
+9 -1
View File
@@ -23,7 +23,15 @@ case "$1" in
fi
rm -f /run/mission-core-node-k1-upgrade-active
fi
/usr/lib/mission-core-node/setup-monitor
# A diagnostic service must not leave Node half-configured and prevent
# dpkg from configuring the separately packaged acquisition plugin.
if /usr/lib/mission-core-node/setup-monitor; then
rm -f /run/mission-core-monitor-setup-failed
else
mc_monitor_result=$?
printf '%s\n' "$mc_monitor_result" > /run/mission-core-monitor-setup-failed
echo "Mission Core Node: monitoring setup failed; device services remain independent." >&2
fi
fi
;;
esac
+2 -2
View File
@@ -54,8 +54,8 @@ case "$1" in
systemctl disable mission-core-node.service
systemctl stop mission-core-node-monitor.service
systemctl disable mission-core-node-monitor.service
systemctl stop postgresql@16-ndc-monitor.service
systemctl disable postgresql@16-ndc-monitor.service
systemctl stop postgresql@16-ndcmonitor.service
systemctl disable postgresql@16-ndcmonitor.service
fi
;;
esac
+18 -7
View File
@@ -9,11 +9,22 @@ install -d -m 0750 -o mission-core-monitor -g mission-core-node /var/lib/mission
install -d -m 0755 -o postgres -g postgres /run/mission-core-monitor-db
install -d -m 0755 /etc/tmpfiles.d
printf '%s\n' 'd /run/mission-core-monitor-db 0755 postgres postgres -' > /etc/tmpfiles.d/mission-core-monitor.conf
if [ ! -d /etc/postgresql/16/ndc-monitor ]; then
pg_createcluster 16 ndc-monitor --port=5433 --socketdir=/run/mission-core-monitor-db --datadir=/var/lib/mission-core-monitor-db --start-conf=auto -- --auth-local=peer --auth-host=reject
# R18 used a dash, which the PostgreSQL systemd template expands as a slash.
# Rename only our exact cluster. pg_renamecluster preserves this custom data
# directory, updates its configuration paths, and retains all database files.
if [ -d /etc/postgresql/16/ndc-monitor ]; then
test ! -d /etc/postgresql/16/ndcmonitor
test "$(pg_conftool -s 16 ndc-monitor show data_directory)" = /var/lib/mission-core-monitor-db
test "$(pg_conftool -s 16 ndc-monitor show port)" = 5433
pg_renamecluster 16 ndc-monitor ndcmonitor
fi
install -d -m 0755 /etc/postgresql/16/ndc-monitor/conf.d
cat > /etc/postgresql/16/ndc-monitor/conf.d/60-mission-core-monitor.conf <<'CONF'
if [ ! -d /etc/postgresql/16/ndcmonitor ]; then
pg_createcluster 16 ndcmonitor --port=5433 --socketdir=/run/mission-core-monitor-db --datadir=/var/lib/mission-core-monitor-db --start-conf=auto -- --auth-local=peer --auth-host=reject
fi
test "$(pg_conftool -s 16 ndcmonitor show data_directory)" = /var/lib/mission-core-monitor-db
test "$(pg_conftool -s 16 ndcmonitor show port)" = 5433
install -d -m 0755 /etc/postgresql/16/ndcmonitor/conf.d
cat > /etc/postgresql/16/ndcmonitor/conf.d/60-mission-core-monitor.conf <<'CONF'
listen_addresses = ''
unix_socket_directories = '/run/mission-core-monitor-db'
shared_preload_libraries = 'timescaledb'
@@ -32,8 +43,8 @@ statement_timeout = '5s'
log_statement = 'none'
log_min_error_statement = 'panic'
CONF
install -d -m 0755 /etc/systemd/system/postgresql@16-ndc-monitor.service.d
cat > /etc/systemd/system/postgresql@16-ndc-monitor.service.d/60-mission-core-monitor.conf <<'CONF'
install -d -m 0755 /etc/systemd/system/postgresql@16-ndcmonitor.service.d
cat > /etc/systemd/system/postgresql@16-ndcmonitor.service.d/60-mission-core-monitor.conf <<'CONF'
[Service]
ExecStartPre=+/usr/bin/install -d -m 0755 -o postgres -g postgres /run/mission-core-monitor-db
MemoryHigh=256M
@@ -43,7 +54,7 @@ TasksMax=32
Nice=10
CONF
systemctl daemon-reload
systemctl restart postgresql@16-ndc-monitor.service
systemctl restart postgresql@16-ndcmonitor.service
runuser -u postgres -- psql -X -v ON_ERROR_STOP=1 -h /run/mission-core-monitor-db -p 5433 -d postgres <<'SQL'
SELECT 'CREATE ROLE "mission-core-monitor" LOGIN NOSUPERUSER NOCREATEDB NOCREATEROLE' WHERE NOT EXISTS(SELECT FROM pg_roles WHERE rolname='mission-core-monitor') \gexec
SELECT 'CREATE DATABASE mission_core_monitor OWNER "mission-core-monitor"' WHERE NOT EXISTS(SELECT FROM pg_database WHERE datname='mission_core_monitor') \gexec
@@ -135,3 +135,30 @@ Pending owner sudo: installed service/version checks, socket-only cluster,
archive growth, measured collector overhead, Core durable ACK/backfill, and
browser QA with real telemetry. A new physical K1 scan and critical scenarios
remain operator qualification, not an automatically exercised acceptance step.
## R18 installation regression and R19 correction
Owner resumed 08.09.2026. R18 install-result is 100, Node 0.8.15 half-configured,
K1 0.1.14 unpacked, Node active but K1 stopped. The exact systemd assertion is
`/etc/postgresql/16/ndc/monitor/postgresql.conf` missing: the installed Debian
PostgreSQL template expands `%I` by replacing instance dashes with path separators.
The cluster name `ndc-monitor` caused the path mismatch. This is a packaging
error, independent of the powered-off K1. Native R18 archive qualification used
a temporary directly started PostgreSQL process and did not cover this systemd
instance contract; the old lifecycle stub also accepted every systemctl command.
R19 / Node 0.8.16 uses cluster `ndcmonitor`. A guarded `pg_renamecluster` migrates
only the predecessor with exact custom data directory and port 5433, preserving
data and updating configuration paths. It refuses a conflicting destination or
foreign data path. All service references/cgroup metrics/journal filters agree.
Monitoring bootstrap failure is now diagnostic: Node postinst records it and
returns success so dpkg can configure the separate K1 plugin. The release
installer still fails its monitoring acceptance if the archive is unavailable;
it also requires K1's read-only status endpoint to answer available=true.
Regression tests execute fresh setup, repeated R18 migration, foreign-cluster
refusal, and both successful/failing monitor setup across Node/K1 lifecycle
states. They implement the actual Debian `%I` assertion rule, so the previous
hyphenated name fails. Real root-installed service and archived samples remain
an explicit post-install gate, never inferred from stub or fixture success.
+5 -2
View File
@@ -9,9 +9,10 @@ import pytest
PACKAGING = Path(__file__).resolve().parents[1] / "apps/node-agent/packaging"
@pytest.mark.parametrize("monitor_fails", [False, True])
@pytest.mark.parametrize("active,configured", [(True, True), (False, True), (True, False)])
def test_node_update_restores_only_previously_running_configured_plugin(
tmp_path, active, configured,
tmp_path, active, configured, monitor_fails,
):
binary = tmp_path / "bin"
binary.mkdir()
@@ -34,7 +35,7 @@ case "$1" in
esac
''')
(binary / "getent").write_text("#!/bin/sh\nexit 0\n")
(binary / "setup-monitor").write_text("#!/bin/sh\nprintf '%s\\n' monitor-setup >> \"$TEST_EVENTS\"\n")
(binary / "setup-monitor").write_text("#!/bin/sh\nprintf '%s\\n' monitor-setup >> \"$TEST_EVENTS\"\nexit " + ("1" if monitor_fails else "0") + "\n")
(binary / "dpkg-query").write_text(
"#!/bin/sh\necho 'install ok " + ("installed" if configured else "unpacked") + "'\n"
)
@@ -56,3 +57,5 @@ esac
assert ("start mission-core-k1.service" in calls) is (active and configured)
assert state.exists() is (active and configured)
assert not (run / "mission-core-node-k1-upgrade-active").exists()
assert (run / "mission-core-monitor-setup-failed").exists() is monitor_fails
+74
View File
@@ -0,0 +1,74 @@
"""Exercise the monitor bootstrap against PostgreSQL's systemd path contract."""
import os
import subprocess
from pathlib import Path
import pytest
SETUP = Path(__file__).resolve().parents[1] / "apps/node-agent/packaging/setup-monitor"
@pytest.mark.parametrize("legacy,foreign", [(False, False), (True, False), (True, True)])
def test_cluster_path_and_r18_migration_are_repeatable(tmp_path, legacy, foreign):
binary = tmp_path / "bin"
binary.mkdir()
conf = tmp_path / "etc/postgresql/16"
conf.mkdir(parents=True)
if legacy:
(conf / "ndc-monitor").mkdir()
(conf / "ndc-monitor/postgresql.conf").write_text("preserved R18 configuration")
driver = binary / "driver"
driver.write_text("""#!/usr/bin/env python3
import os,sys,pathlib
root=pathlib.Path(os.environ['TEST_ROOT']);name=pathlib.Path(sys.argv[0]).name;args=sys.argv[1:]
with (root/'events').open('a') as log:log.write(name+' '+ ' '.join(args)+'\\n')
conf=root/'etc/postgresql/16'
if name=='id':print(0)
elif name=='install':pathlib.Path(args[-1]).mkdir(parents=True,exist_ok=True)
elif name=='pg_conftool':
print(os.environ.get('TEST_DATA_ROOT',str(root/'var/lib/mission-core-monitor-db')) if args[-1]=='data_directory' else '5433')
elif name=='pg_createcluster':
(conf/args[1]).mkdir();(conf/args[1]/'postgresql.conf').write_text('new database')
elif name=='pg_renamecluster':
(conf/args[1]).rename(conf/args[2])
elif name=='systemctl' and args[0]=='restart' and args[1].startswith('postgresql@'):
# Exact installed Debian template: AssertPathExists=/etc/postgresql/%I/postgresql.conf.
instance=args[1].split('@')[1].removesuffix('.service')
assert (root/'etc/postgresql'/instance.replace('-','/')/'postgresql.conf').is_file()
""")
driver.chmod(0o700)
for name in [
"id",
"getent",
"install",
"pg_conftool",
"pg_createcluster",
"pg_renamecluster",
"systemctl",
"runuser",
]:
(binary / name).symlink_to(driver)
script = tmp_path / "setup"
text = SETUP.read_text()
for path in ["/etc/", "/var/lib/", "/run/"]:
text = text.replace(path, str(tmp_path / path.lstrip("/")) + "/")
script.write_text(text)
env = dict(
os.environ, PATH=str(binary) + os.pathsep + os.environ["PATH"], TEST_ROOT=str(tmp_path)
)
if foreign:
env["TEST_DATA_ROOT"] = "/unrelated/database"
result = subprocess.run(["/bin/sh", str(script)], env=env, capture_output=True)
assert result.returncode != 0
assert (conf / "ndc-monitor/postgresql.conf").is_file()
assert "pg_renamecluster" not in (tmp_path / "events").read_text()
return
for _ in range(2):
subprocess.run(["/bin/sh", str(script)], env=env, check=True, capture_output=True)
calls = (tmp_path / "events").read_text()
assert calls.count("pg_renamecluster 16 ndc-monitor ndcmonitor") == int(legacy)
assert calls.count("pg_createcluster 16 ndcmonitor") == int(not legacy)
assert not (conf / "ndc-monitor").exists()
if legacy:
assert (conf / "ndcmonitor/postgresql.conf").read_text() == "preserved R18 configuration"