diff --git a/changes-entries/systemd-watchdog.txt b/changes-entries/systemd-watchdog.txt
new file mode 100644
index 00000000000..2fca28a6e4e
--- /dev/null
+++ b/changes-entries/systemd-watchdog.txt
@@ -0,0 +1,2 @@
+ *) mod_systemd: Support the systemd watchdog, sending the keep-alive
+ notification if the service unit sets WatchdogSec=. [Joe Orton]
diff --git a/docs/manual/mod/mod_systemd.xml b/docs/manual/mod/mod_systemd.xml
index 089d6296609..619ce8b5fa0 100644
--- a/docs/manual/mod/mod_systemd.xml
+++ b/docs/manual/mod/mod_systemd.xml
@@ -32,7 +32,7 @@
This module provides support for systemd integration. It allows
httpd to be used in a service with the systemd
- Type=notify (see Type=notify or Type=notify-reload (see systemd.service(5)
for more information). The module is activated if loaded.
@@ -69,13 +69,90 @@ WantedBy=multi-user.target
href="https://www.freedesktop.org/software/systemd/man/systemd.kill.html">systemd.kill(5)
for more information.
- This module does not provide support for Systemd socket activation.
+ A service manager from systemd 253 onwards offers
+ Type=notify-reload, which is worth using in preference.
+ Under Type=notify a systemctl reload
+ returns as soon as the ExecReload command has sent its
+ signal, which is before the new configuration has been read, and it
+ reports success whatever becomes of the restart afterwards.
+ Type=notify-reload instead holds the reload open until
+ the server reports it finished, so the command waits for the new
+ configuration to be in use and fails if it never is. mod_systemd
+ sends the RELOADING=1 notification the protocol expects
+ while the configuration is being read, stamped with the
+ MONOTONIC_USEC the service manager requires, and
+ READY=1 once it has been loaded.
+
+
+ Example of a service unit which reloads synchronously
+
+[Service]
+Type=notify-reload
+ReloadSignal=SIGCONT
+ExecStart=/usr/local/apache2/bin/httpd -D FOREGROUND -k start
+ExecReload=/usr/local/apache2/bin/httpd -k graceful
+KillMode=mixed
+
+
+
+ The service manager runs ExecReload first, and sends
+ the signal named by ReloadSignal only once that command
+ has exited successfully. Keeping ExecReload is what
+ makes the reload safe: httpd -k graceful parses the new
+ configuration in a process of its own and exits without signalling
+ anything if it does not parse, so the reload fails and the running
+ server carries on with the configuration it already has. Leaving
+ ExecReload out, and letting the service manager signal
+ the server directly, gives up that check: the running parent reads
+ the new configuration itself, and a configuration which does not
+ parse makes it exit, taking the server down.
+
+ The signal sent after ExecReload has run is then
+ redundant, so ReloadSignal should name one httpd does
+ not act on, such as SIGCONT. It matters that it is set:
+ the default is SIGHUP, which httpd takes as an
+ ungraceful restart, dropping the connections a reload is
+ meant to preserve. A unit which does leave out
+ ExecReload must set ReloadSignal=SIGUSR1,
+ the signal httpd restarts gracefully on.
+
+ Systemd socket activation is supported if httpd was built with
+ it. Each Listen port
+ must then be one passed in by systemd; a port which was not is a
+ fatal configuration error rather than one httpd opens for itself.
+ Socket activation is used only if this module is loaded, so it can
+ be built in and left unused.
ExtendedStatus is
enabled by default if the module is loaded. If ExtendedStatus is not disabled in
the configuration, run-time load and request statistics are made
available in the systemctl status output.
+
+ The systemd watchdog is supported. If the service unit sets
+ WatchdogSec=, the parent process sends the keep-alive
+ notification which tells systemd the server is still alive; a server
+ which stops sending it is terminated and, with a suitable
+ Restart= setting, restarted. The notification is sent
+ while the configuration is being read and again once it is loaded, so
+ that a reload is covered, and periodically from the parent process
+ while the server runs.
+
+ That periodic notification is sent about every ten seconds, which
+ is how often the parent process runs the hook it is sent from. A
+ WatchdogSec= of less than twice that cannot be met, and
+ would have systemd terminating a server which is working normally;
+ such a setting is reported as a warning at startup. Use a
+ WatchdogSec= of at least 20 seconds.
+
+
+ Adding watchdog supervision to either unit above
+
+[Service]
+WatchdogSec=30
+Restart=on-failure
+
+
diff --git a/modules/arch/unix/mod_systemd.c b/modules/arch/unix/mod_systemd.c
index 22482fd6bbc..6c03acc3a35 100644
--- a/modules/arch/unix/mod_systemd.c
+++ b/modules/arch/unix/mod_systemd.c
@@ -16,6 +16,7 @@
*/
#include
+#include
#include
#include "ap_mpm.h"
#include "ap_listen.h"
@@ -39,12 +40,65 @@
#include
#endif
+/* Microseconds on the clock systemd compares RELOADING=1 against, or
+ * zero if it cannot be read. */
+static apr_uint64_t monotonic_usec(void)
+{
+ struct timespec ts;
+
+ if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0) {
+ return 0;
+ }
+ return (apr_uint64_t)ts.tv_sec * APR_USEC_PER_SEC + ts.tv_nsec / 1000;
+}
+
+/* ap_run_monitor() is called once every INTERVAL_OF_WRITABLE_PROBES turns
+ * of the parent's one second loop in ap_wait_or_timeout(), so that is how
+ * often a keep-alive notification can be sent, and the shortest watchdog
+ * timeout which can be met is twice that: sd_watchdog_enabled(3) asks for
+ * a notification every half of the configured timeout. */
+#define WATCHDOG_INTERVAL_SEC (10)
+
+/* The WatchdogSec= of the service in microseconds, or zero if the service
+ * manager is not watching. Set in pre_config, before the first
+ * notification which could carry a keep-alive. */
+static apr_uint64_t watchdog_usec;
+
+/* A keep-alive assignment to paste into a notification, or nothing while
+ * the service manager is not asking for one. Sending WATCHDOG=1 when it
+ * is not expected is harmless, but saying so only when asked keeps what
+ * httpd reports the same as what the service was configured for. */
+static const char *watchdog_ping(void)
+{
+ return watchdog_usec ? "WATCHDOG=1\n" : "";
+}
+
static int systemd_pre_config(apr_pool_t *pconf, apr_pool_t *plog,
apr_pool_t *ptemp)
{
- sd_notify(0,
- "RELOADING=1\n"
- "STATUS=Reading configuration...\n");
+ apr_uint64_t usec = monotonic_usec(), wd_usec;
+
+ /* Read afresh on each configuration load, since a restart unloads and
+ * loads the module again, and without unsetting it as server/listen.c
+ * does for $LISTEN_FDS, which would stop the keep-alive at the first
+ * reload. */
+ watchdog_usec = sd_watchdog_enabled(0, &wd_usec) > 0 ? wd_usec : 0;
+
+ /* A Type=notify-reload service ignores a reload notification which
+ * does not say when it was sent. */
+ if (usec) {
+ sd_notifyf(0,
+ "RELOADING=1\n"
+ "MONOTONIC_USEC=%" APR_UINT64_T_FMT "\n"
+ "%s"
+ "STATUS=Reading configuration...\n", usec, watchdog_ping());
+ }
+ else {
+ sd_notifyf(0,
+ "RELOADING=1\n"
+ "%s"
+ "STATUS=Reading configuration...\n", watchdog_ping());
+ }
ap_extended_status = 1;
return OK;
}
@@ -63,6 +117,17 @@ static void log_selinux_context(void)
}
#endif
+/* pconf is also cleared on a restart, where the service is not stopping
+ * at all, so distinguish the two by the state of the process. */
+static apr_status_t systemd_stopping(void *unused)
+{
+ if (ap_state_query(AP_SQ_MAIN_STATE) == AP_SQ_MS_EXITING) {
+ sd_notify(0, "STOPPING=1\n"
+ "STATUS=Shutting down.\n");
+ }
+ return APR_SUCCESS;
+}
+
/* Report the service is ready in post_config, which could be during
* startup or after a reload. The server could still hit a fatal
* startup error after this point during ap_run_mpm(), so this is
@@ -80,8 +145,33 @@ static int systemd_post_config(apr_pool_t *pconf, apr_pool_t *plog,
log_selinux_context();
#endif
- sd_notify(0, "READY=1\n"
- "STATUS=Configuration loaded.\n");
+ /* Not reached by "httpd -k stop" and friends, which signal the
+ * running server and exit before post_config. */
+ apr_pool_cleanup_register(pconf, NULL, systemd_stopping,
+ apr_pool_cleanup_null);
+
+ /* A timeout the parent cannot meet would have the service manager
+ * killing a healthy server every WatchdogSec, so say so rather than
+ * leaving nothing in the log to explain it. */
+ if (watchdog_usec
+ && watchdog_usec / 2 < (apr_uint64_t)WATCHDOG_INTERVAL_SEC
+ * APR_USEC_PER_SEC) {
+ ap_log_error(APLOG_MARK, APLOG_WARNING, 0, main_server, APLOGNO(10621)
+ "WatchdogSec is %" APR_UINT64_T_FMT "us, but keep-alive "
+ "notifications are sent from the parent process only "
+ "every %ds; configure a WatchdogSec of at least %ds or "
+ "the service will be killed while it is healthy",
+ watchdog_usec, WATCHDOG_INTERVAL_SEC,
+ 2 * WATCHDOG_INTERVAL_SEC);
+ }
+
+ /* The keep-alive rides along with the notification which ends a
+ * reload: the configuration is read outside the parent's monitor loop,
+ * so nothing reports while it is being parsed, and the service manager
+ * keeps the timeout armed throughout. */
+ sd_notifyf(0, "READY=1\n"
+ "%s"
+ "STATUS=Configuration loaded.\n", watchdog_ping());
return OK;
}
@@ -100,21 +190,32 @@ static int systemd_monitor(apr_pool_t *p, server_rec *s)
apr_interval_time_t up_time;
char bps[5];
+ /* Before anything which might decline: reporting the server is alive
+ * does not depend on there being a status line to report with it. */
+ if (watchdog_usec) {
+ sd_notify(0, "WATCHDOG=1\n");
+ }
+
if (!ap_extended_status) {
/* Nothing useful to report with ExtendedStatus disabled. */
return DECLINED;
}
ap_get_sload(&sload);
- /* up_time in seconds */
- up_time = (apr_uint32_t) apr_time_sec(apr_time_now() -
- ap_scoreboard_image->global->restart_time);
+ /* up_time in seconds, and never zero: a restart resets restart_time,
+ * so this hook can run in the same second it was set. */
+ up_time = apr_time_sec(apr_time_now() -
+ ap_scoreboard_image->global->restart_time);
+ if (up_time < 1) {
+ up_time = 1;
+ }
- apr_strfsize((unsigned long)((float) (sload.bytes_served)
- / (float) up_time), bps);
+ apr_strfsize(sload.bytes_served / up_time, bps);
+ /* ap_get_sload() gives idle and busy as percentages of the workers
+ * available, not as counts. */
sd_notifyf(0, "READY=1\n"
- "STATUS=Total requests: %lu; Idle/Busy workers %d/%d;"
+ "STATUS=Total requests: %lu; Idle/Busy workers %d%%/%d%%; "
"Requests/sec: %.3g; Bytes served/sec: %sB/sec\n",
sload.access_count, sload.idle, sload.busy,
((float) sload.access_count) / (float) up_time, bps);
@@ -122,9 +223,31 @@ static int systemd_monitor(apr_pool_t *p, server_rec *s)
return DECLINED;
}
+/* The number of sockets passed by the service manager has to be
+ * remembered: the configuration is read again on restart, by which time
+ * the environment sd_listen_fds() reads has been cleared, and the module
+ * itself has been unloaded and loaded again. Hence retained data rather
+ * than a static. */
+static const char *const retained_key = "mod_systemd_listen_fds";
+
+static int ap_systemd_listen_fds(int unset_environment)
+{
+ int *fds = ap_retained_data_get(retained_key);
+
+ if (fds == NULL) {
+ fds = ap_retained_data_create(retained_key, sizeof(*fds));
+ *fds = sd_listen_fds(0);
+ }
+ if (unset_environment) {
+ /* Take the variables out of the environment, keeping the count. */
+ sd_listen_fds(1);
+ }
+ return *fds;
+}
+
static int ap_find_systemd_socket(process_rec * process, apr_port_t port) {
- int fdcount, fd;
- int sdc = sd_listen_fds(0);
+ int fd;
+ int sdc = ap_systemd_listen_fds(0);
if (sdc < 0) {
ap_log_perror(APLOG_MARK, APLOG_CRIT, sdc, process->pool, APLOGNO(02486)
@@ -139,8 +262,7 @@ static int ap_find_systemd_socket(process_rec * process, apr_port_t port) {
return -1;
}
- fdcount = atoi(getenv("LISTEN_FDS"));
- for (fd = SD_LISTEN_FDS_START; fd < SD_LISTEN_FDS_START + fdcount; fd++) {
+ for (fd = SD_LISTEN_FDS_START; fd < SD_LISTEN_FDS_START + sdc; fd++) {
if (sd_is_socket_inet(fd, 0, 0, -1, port) > 0) {
return fd;
}
@@ -149,10 +271,6 @@ static int ap_find_systemd_socket(process_rec * process, apr_port_t port) {
return -1;
}
-static int ap_systemd_listen_fds(int unset_environment){
- return sd_listen_fds(unset_environment);
-}
-
static void systemd_register_hooks(apr_pool_t *p)
{
APR_REGISTER_OPTIONAL_FN(ap_systemd_listen_fds);
diff --git a/test/modules/arch/__init__.py b/test/modules/arch/__init__.py
new file mode 100644
index 00000000000..e69de29bb2d
diff --git a/test/modules/arch/linux/README b/test/modules/arch/linux/README
new file mode 100644
index 00000000000..4bdd7af91d1
--- /dev/null
+++ b/test/modules/arch/linux/README
@@ -0,0 +1,109 @@
+mod_systemd tests
+=================
+
+What is different about this module
+-----------------------------------
+mod_systemd has no directives. Everything it does is driven by the
+environment httpd was started with, and almost everything it produces goes
+to the service manager rather than to a client:
+
+ - sd_notify(3) datagrams sent to $NOTIFY_SOCKET at four points: reading
+ the configuration (pre_config), configuration loaded (post_config),
+ the MPM starting (pre_mpm, the only notification carrying MAINPID),
+ and a status line refreshed by the monitor hook.
+ - listening sockets inherited from the service manager, found through
+ $LISTEN_FDS. mod_systemd registers the two optional functions
+ server/listen.c calls, so loading the module is what enables socket
+ activation and not loading it is what disables it.
+ - ap_extended_status forced on in pre_config, so that the monitor hook
+ has request counts to report.
+ - the watchdog keep-alive, WATCHDOG=1, sent while the service manager
+ asks for one. Which it does through the environment as well:
+ WatchdogSec= in the unit becomes $WATCHDOG_USEC, and $WATCHDOG_PID
+ names the process expected to report.
+
+None of that is observable over HTTP, so the tests observe it directly.
+
+How the tests run without systemd, and without privileges
+---------------------------------------------------------
+There is no need for a service manager to exercise the protocol. Only
+test_005, the last test of test_006, and test_007 involve systemd at all;
+the rest run anywhere, and need nothing from the systemd package beyond the
+libsystemd httpd itself is linked against.
+
+ test_001_notify.py $NOTIFY_SOCKET is an ordinary AF_UNIX datagram
+ socket the test binds itself (env.NotifyListener).
+ sd_notify writes to whatever that variable names, so
+ the test reads httpd's notifications straight off the
+ socket. HttpdTestEnv.set_httpd_env() puts the
+ variable into the environment apachectl passes on.
+
+ test_002_monitor.py The periodic status line. ap_run_monitor() is
+ called once every ten turns of the parent's ~1s
+ loop, so these tests wait up to 25 seconds.
+
+ test_003_extended_status.py
+ The ExtendedStatus side effect, observed through
+ mod_status.
+
+ test_004_socket_activation.py
+ The test opens the listening socket itself and
+ hands it to httpd as descriptor 3 with LISTEN_FDS
+ and LISTEN_PID set, which is the whole protocol.
+ systemd-socket-activate would do the same, but its
+ --now option is newer than the systemd on some
+ distributions, and doing it directly needs no
+ systemd tooling at all.
+
+ test_006_watchdog.py The keep-alive. Mostly the same stand-in socket,
+ with $WATCHDOG_USEC set in the environment httpd is
+ started with; env.ForegroundServer runs httpd in the
+ foreground where $WATCHDOG_PID can name it, which
+ apachectl cannot do for a parent it has not forked
+ yet. The last test uses a real unit with
+ WatchdogSec=, where systemd's own
+ WatchdogTimestampMonotonic is the evidence that the
+ pings arrived.
+
+ The keep-alive is sent from the monitor hook, so it
+ is only as frequent as that hook is: about every ten
+ seconds. A WatchdogSec shorter than twice that
+ cannot be met, and mod_systemd says so (AH10621)
+ rather than letting a healthy server be killed.
+
+ test_005_service.py The real thing: a transient Type=notify unit run
+ with "systemd-run --user". This is what checks that
+ systemd holds the unit in "activating" until READY=1
+ arrives, tracks the right MainPID, and shows the
+ reported STATUS= as the unit's status text.
+ Skipped when the user has no systemd manager.
+
+ test_007_notify_reload.py
+ The reload protocol, as a transient
+ Type=notify-reload unit: that "systemctl reload"
+ waits for the new configuration to be in use, that a
+ configuration which does not parse fails the reload
+ without disturbing the running server, and that a
+ reload restarts it gracefully and only once. Also
+ skipped where systemd is older than 253, which is
+ where Type=notify-reload arrived.
+
+The whole package is skipped unless mod_systemd was built, which needs
+configure --enable-systemd. A static module is enough for everything
+except the test which has to leave mod_systemd out of the configuration;
+that one needs --enable-systemd=shared.
+
+Running the systemd suites where there is no user session
+---------------------------------------------------------
+"systemctl --user" needs a per-user manager, which a login session has but
+a CI container does not; enabling lingering needs privileges. Where that
+is a problem, run the tests inside a container with systemd as pid 1,
+which rootless podman supports on a cgroup v2 host:
+
+ podman run --rm -it --systemd=always \
+ -v $PWD:/src:z -w /src registry.fedoraproject.org/fedora:latest \
+ /usr/sbin/init
+
+then, in another terminal, "podman exec" into it and run pytest as a
+non-root user with XDG_RUNTIME_DIR set. The other four suites need none of
+this and run anywhere.
diff --git a/test/modules/arch/linux/__init__.py b/test/modules/arch/linux/__init__.py
new file mode 100644
index 00000000000..e69de29bb2d
diff --git a/test/modules/arch/linux/conftest.py b/test/modules/arch/linux/conftest.py
new file mode 100644
index 00000000000..7b1a0d66cd7
--- /dev/null
+++ b/test/modules/arch/linux/conftest.py
@@ -0,0 +1,40 @@
+import logging
+import os
+import sys
+
+import pytest
+
+sys.path.append(os.path.join(os.path.dirname(__file__), '../../..'))
+
+from .env import SystemdTestEnv
+
+
+def pytest_report_header(config, start_path):
+ env = SystemdTestEnv()
+ return f"mod_systemd [apache: {env.get_httpd_version()}, " \
+ f"mpm: {env.mpm_module}, {env.prefix}]"
+
+
+@pytest.fixture(scope="package")
+def env(pytestconfig) -> SystemdTestEnv:
+ level = logging.INFO
+ console = logging.StreamHandler()
+ console.setLevel(level)
+ console.setFormatter(logging.Formatter('%(levelname)s: %(message)s'))
+ logging.getLogger('').addHandler(console)
+ logging.getLogger('').setLevel(level=level)
+ env = SystemdTestEnv(pytestconfig=pytestconfig)
+ if not env.has_systemd_module:
+ pytest.skip("mod_systemd is not built, configure with --enable-systemd")
+ env.setup_httpd()
+ env.apache_access_log_clear()
+ env.httpd_error_log.clear_log()
+ env.start_notify_listener()
+ yield env
+ env.stop_notify_listener()
+
+
+@pytest.fixture(autouse=True, scope="package")
+def _stop_package_scope(env):
+ yield
+ assert env.apache_stop() == 0
diff --git a/test/modules/arch/linux/env.py b/test/modules/arch/linux/env.py
new file mode 100644
index 00000000000..ce1c2464f27
--- /dev/null
+++ b/test/modules/arch/linux/env.py
@@ -0,0 +1,638 @@
+import inspect
+import logging
+import os
+import re
+import shutil
+import signal
+import socket
+import subprocess
+import threading
+import time
+from typing import Callable, Dict, List, Optional
+
+from pyhttpd.env import HttpdTestEnv, HttpdTestSetup
+
+log = logging.getLogger(__name__)
+
+
+class SystemdTestSetup(HttpdTestSetup):
+
+ def __init__(self, env: 'HttpdTestEnv'):
+ super().__init__(env=env)
+ self.add_source_dir(os.path.dirname(inspect.getfile(SystemdTestSetup)))
+ # mod_systemd is only built with --enable-systemd, so it must not be
+ # a hard requirement; the tests skip when it is absent.
+ self.add_optional_modules(["systemd"])
+
+
+class SystemdTestEnv(HttpdTestEnv):
+
+ def __init__(self, pytestconfig=None):
+ super().__init__(pytestconfig=pytestconfig)
+ self.add_httpd_log_modules(["core"])
+ self._notify = None
+
+ def setup_httpd(self, setup: HttpdTestSetup = None):
+ super().setup_httpd(setup=SystemdTestSetup(env=self))
+
+ @property
+ def systemd_is_dso(self) -> bool:
+ return os.path.isfile(os.path.join(self.libexec_dir, 'mod_systemd.so'))
+
+ @property
+ def has_systemd_module(self) -> bool:
+ """Whether mod_systemd is available, however it was built:
+ --enable-systemd links it statically, --enable-systemd=shared
+ builds the DSO."""
+ if self.systemd_is_dso:
+ return True
+ p = subprocess.run([self.httpd_bin, '-l'], capture_output=True,
+ text=True)
+ return re.search(r'^\s+mod_systemd\.c$', p.stdout, re.M) is not None
+
+ @property
+ def notify(self) -> 'NotifyListener':
+ """The stand-in notification socket httpd reports to."""
+ return self._notify
+
+ def start_notify_listener(self) -> 'NotifyListener':
+ """Bind the notification socket and point httpd's $NOTIFY_SOCKET at it.
+
+ This must happen before the server is first started, since libsystemd
+ reads $NOTIFY_SOCKET from the environment of the httpd process.
+ """
+ assert self._notify is None
+ self._notify = NotifyListener(
+ os.path.join(self.server_dir, 'systemd-notify.sock'))
+ self.set_httpd_env('NOTIFY_SOCKET', self._notify.path)
+ return self._notify
+
+ def stop_notify_listener(self):
+ if self._notify is not None:
+ self._notify.close()
+ self._notify = None
+
+ def server_env(self) -> Dict[str, str]:
+ """The environment httpd is started with, as apachectl gets it."""
+ return self._clean_path_env()
+
+ @property
+ def httpd_bin(self) -> str:
+ return os.path.join(self.bin_dir, 'httpd')
+
+
+class NotifyListener:
+ """A stand-in for the systemd notification socket.
+
+ sd_notify(3) does nothing more than send a datagram to the AF_UNIX
+ socket named by $NOTIFY_SOCKET, so an unconnected datagram socket is
+ enough to observe everything mod_systemd reports, with no systemd
+ instance and no privileges involved.
+
+ Datagrams are drained by a background thread so that the periodic
+ notifications from the monitor hook cannot fill the socket buffer
+ while a test is doing something else.
+ """
+
+ def __init__(self, path: str):
+ # sockaddr_un is limited to 108 bytes; well within reach for a
+ # source tree in a home directory, but check rather than fail
+ # obscurely inside bind().
+ assert len(path) < 100, f"notification socket path too long: {path}"
+ if os.path.exists(path):
+ os.unlink(path)
+ self.path = path
+ self._sock = socket.socket(socket.AF_UNIX, socket.SOCK_DGRAM)
+ self._sock.bind(path)
+ self._sock.settimeout(0.1)
+ self._lock = threading.Lock()
+ self._messages: List[Dict[str, str]] = []
+ self._stop = threading.Event()
+ self._thread = threading.Thread(target=self._drain, daemon=True)
+ self._thread.start()
+
+ def _drain(self):
+ while not self._stop.is_set():
+ try:
+ data = self._sock.recv(8192)
+ except socket.timeout:
+ continue
+ except OSError:
+ break
+ try:
+ msg = self.parse(data)
+ except Exception as ex:
+ # Never let one unreadable datagram stop the listener: the
+ # tests would then see silence rather than a failure.
+ log.warning(f"undecodable notification {data!r}: {ex}")
+ continue
+ log.debug(f"notify: {msg}")
+ with self._lock:
+ self._messages.append(msg)
+
+ def close(self):
+ self._stop.set()
+ self._thread.join(timeout=2)
+ self._sock.close()
+ if os.path.exists(self.path):
+ os.unlink(self.path)
+
+ @staticmethod
+ def parse(data: bytes) -> Dict[str, str]:
+ """Split one notification datagram into its NAME=VALUE assignments.
+
+ The last line carries no trailing newline in the MAINPID
+ notification, and a value may itself contain '='.
+ """
+ msg = {}
+ for line in data.decode(errors='replace').split('\n'):
+ name, sep, value = line.partition('=')
+ if sep:
+ msg[name] = value
+ return msg
+
+ @property
+ def messages(self) -> List[Dict[str, str]]:
+ with self._lock:
+ return list(self._messages)
+
+ def clear(self):
+ with self._lock:
+ self._messages.clear()
+
+ def wait_for(self, match: Callable[[Dict[str, str]], bool],
+ timeout: float = 5.0) -> Optional[Dict[str, str]]:
+ """Return the first message satisfying `match`, waiting for it to
+ arrive if it has not already. Returns None on timeout."""
+ end = time.time() + timeout
+ seen = 0
+ while True:
+ with self._lock:
+ pending = self._messages[seen:]
+ seen = len(self._messages)
+ for msg in pending:
+ if match(msg):
+ return msg
+ if time.time() >= end:
+ return None
+ time.sleep(0.05)
+
+ def wait_for_key(self, key: str, timeout: float = 5.0) \
+ -> Optional[Dict[str, str]]:
+ return self.wait_for(lambda m: key in m, timeout=timeout)
+
+ def wait_for_status(self, pattern: str, timeout: float = 5.0) \
+ -> Optional[Dict[str, str]]:
+ rx = re.compile(pattern)
+ return self.wait_for(lambda m: 'STATUS' in m and rx.search(m['STATUS']),
+ timeout=timeout)
+
+ def statuses(self) -> List[str]:
+ return [m['STATUS'] for m in self.messages if 'STATUS' in m]
+
+
+# The STATUS= line the monitor hook reports, from systemd_monitor() in
+# modules/arch/unix/mod_systemd.c. Idle and busy are the percentages
+# ap_get_sload() computes, and are -1 when there are no workers at all.
+MONITOR_STATUS = re.compile(
+ r'^Total requests: (?P\d+);\s*'
+ r'Idle/Busy workers (?P-?\d+)%/(?P-?\d+)%;\s*'
+ r'Requests/sec: (?P\S+);\s*'
+ r'Bytes served/sec: (?P.*)B/sec$')
+
+# ap_run_monitor() is called every INTERVAL_OF_WRITABLE_PROBES (10) turns of
+# the ~1s parent loop in ap_wait_or_timeout(), so a status update is up to
+# roughly 10 seconds away. Waiting for one costs that; waiting to be sure
+# none is coming costs the whole timeout, so keep it to a small multiple.
+MONITOR_TIMEOUT = 25.0
+
+# Showing that no report is coming costs the whole wait, so it only has to
+# comfortably outlast one turn of that cycle.
+NO_MONITOR_TIMEOUT = 15.0
+
+
+# The keep-alive notification is sent from the same monitor hook, so waiting
+# for one costs the same as waiting for a status report.
+WATCHDOG_TIMEOUT = MONITOR_TIMEOUT
+
+# The shortest WatchdogSec mod_systemd will accept without complaining, which
+# is twice the monitor interval: the recommended keep-alive period is half the
+# watchdog timeout, and half of anything shorter than this is out of reach of a
+# hook which runs every ten seconds. Keep in step with mod_systemd.c.
+SYSTEMD_MONITOR_INTERVAL = 10
+MIN_WATCHDOG_SEC = 2 * SYSTEMD_MONITOR_INTERVAL
+
+
+def is_watchdog_ping(msg: Dict[str, str]) -> bool:
+ """Whether a notification carries the keep-alive ping.
+
+ mod_systemd is free to send WATCHDOG=1 in a datagram of its own or
+ alongside whatever else it is reporting, so match on the assignment
+ rather than on the message being only that.
+ """
+ return msg.get('WATCHDOG') == '1'
+
+
+# A configuration for a server run directly rather than through apachectl,
+# sharing the server root, module list and error log with the rest of the
+# suite but with its own pid file and port.
+STANDALONE_CONF = """
+ServerRoot "${server_dir}"
+DefaultRuntimeDir logs
+PidFile "${pidfile}"
+Include "conf/${modules_conf}"
+ServerName standalone.test
+ErrorLog "logs/error_log"
+LogLevel ${loglevel}
+DocumentRoot "${server_dir}/htdocs"
+
+ Require all granted
+
+
+ SSLSessionCache "shmcb:ssl_gcache_data(32000)"
+
+${extra}
+Listen ${port}
+"""
+
+
+def write_server_conf(env: SystemdTestEnv, name: str, port: int,
+ modules_conf: str = 'modules.conf',
+ extra: str = '') -> str:
+ """Write a standalone configuration and return its path."""
+ path = os.path.join(env.server_conf_dir, f'{name}.conf')
+ with open(path, 'w') as fd:
+ fd.write(STANDALONE_CONF
+ .replace('${server_dir}', env.server_dir)
+ .replace('${pidfile}',
+ os.path.join(env.server_logs_dir, f'{name}.pid'))
+ .replace('${loglevel}', 'debug' if env.verbosity else 'warn')
+ .replace('${modules_conf}', modules_conf)
+ .replace('${extra}', extra)
+ .replace('${port}', str(port)))
+ return path
+
+
+def http_responds(port: int, timeout: float = 2.0) -> bool:
+ """One HTTP request, without curl, so that a listening socket which
+ nothing is serving cannot be mistaken for a running server: under socket
+ activation the listener exists before httpd does."""
+ try:
+ with socket.create_connection(('127.0.0.1', port), 1.0) as c:
+ c.settimeout(timeout)
+ c.sendall(b'GET / HTTP/1.0\r\nHost: standalone.test\r\n\r\n')
+ return c.recv(64).startswith(b'HTTP/1.')
+ except OSError:
+ return False
+
+
+def error_log_size(env: SystemdTestEnv) -> int:
+ """Where the shared error log ends now, so that a later read can take
+ only what one operation wrote."""
+ try:
+ return os.path.getsize(env.httpd_error_log.path)
+ except OSError:
+ return 0
+
+
+def error_log_since(env: SystemdTestEnv, pos: int) -> str:
+ """The error log written since error_log_size() returned pos."""
+ try:
+ with open(env.httpd_error_log.path, errors='replace') as fd:
+ fd.seek(pos)
+ return fd.read()
+ except OSError:
+ return ''
+
+
+class ActivatedServer:
+ """An httpd handed a listening socket the way a service manager does.
+
+ The protocol is only $LISTEN_FDS descriptors starting at 3, and
+ $LISTEN_PID naming the process they were meant for, so the test opens
+ the socket and speaks it directly. systemd-socket-activate would do
+ the same, but its --now option is too recent to rely on, and this
+ needs no systemd tooling at all.
+
+ apachectl cannot pass descriptors, so httpd is run directly, in the
+ foreground: LISTEN_PID has to be the process which calls
+ sd_listen_fds(), and a daemonised parent would not be it.
+ """
+
+ def __init__(self, env: SystemdTestEnv, port: int, name: str = 'activate',
+ extra: str = '', listen_port: int = None,
+ modules_conf: str = 'modules.conf'):
+ self.env = env
+ self.port = port
+ # The port systemd-socket-activate binds, which is the same as the
+ # configured one unless a test wants them to disagree.
+ self.listen_port = port if listen_port is None else listen_port
+ self.name = name
+ self.pid_file = os.path.join(env.server_logs_dir, f'{name}.pid')
+ self.proc = None
+ self.stdout = None
+ self.stderr = None
+ self.conf_file = write_server_conf(env, name, port,
+ modules_conf=modules_conf,
+ extra=extra)
+
+ @staticmethod
+ def modules_conf_without(env: SystemdTestEnv, module: str) -> str:
+ """Write a copy of the generated module list with one module left
+ out, to check what happens when it is not loaded."""
+ name = f'modules-no-{module}.conf'
+ src = os.path.join(env.server_conf_dir, 'modules.conf')
+ rx = re.compile(rf'^\s*LoadModule\s+{module}_module\b')
+ with open(src) as fd:
+ lines = [l for l in fd if not rx.match(l)]
+ with open(os.path.join(env.server_conf_dir, name), 'w') as fd:
+ fd.writelines(lines)
+ return name
+
+ def args(self, fd: int) -> List[str]:
+ # The shell moves the inherited socket to descriptor 3 and names
+ # itself in LISTEN_PID before exec'ing httpd in its place, which is
+ # the one thing this cannot do from the parent: the pid has to be
+ # the one which will call sd_listen_fds(). bash rather than sh
+ # because dash parses only one digit in a redirection, and the
+ # socket lands well above descriptor 9.
+ return [
+ 'bash', '-c',
+ f'exec 3<&{fd}; export LISTEN_FDS=1 LISTEN_FDNAMES=activate '
+ 'LISTEN_PID=$$; exec "$0" "$@"',
+ self.env.httpd_bin, '-DFOREGROUND',
+ '-d', self.env.server_dir, '-f', self.conf_file,
+ ]
+
+ def start(self) -> 'ActivatedServer':
+ lsock = socket.create_server(('', self.listen_port),
+ family=socket.AF_INET6,
+ dualstack_ipv6=True, backlog=128)
+ try:
+ # A new session so that the whole group can be signalled on the
+ # way out.
+ self.proc = subprocess.Popen(
+ self.args(lsock.fileno()), env=self.env.server_env(),
+ start_new_session=True, pass_fds=(lsock.fileno(),),
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE)
+ finally:
+ # The child holds it now; keeping a copy here would leave the
+ # port bound after the server is gone.
+ lsock.close()
+ return self
+
+ def _reap(self):
+ """Collect the output of a server which has exited."""
+ if self.stderr is None and self.proc.poll() is not None:
+ self.stdout, self.stderr = self.proc.communicate()
+
+ def wait_exit(self, timeout: float = 10.0) -> int:
+ """Wait for a server which is expected to fail to start."""
+ self.stdout, self.stderr = self.proc.communicate(timeout=timeout)
+ return self.proc.returncode
+
+ def is_live(self, timeout: float = 10.0) -> bool:
+ end = time.time() + timeout
+ while True:
+ if self.proc.poll() is not None:
+ # Collect its diagnostics, so that a test reporting the
+ # server did not come up can say why.
+ self._reap()
+ return False
+ if http_responds(self.port):
+ return True
+ if time.time() >= end:
+ return False
+ time.sleep(0.2)
+
+ def is_running(self, settle: float = 2.0) -> bool:
+ """Whether the server is still up once it has had time to fail.
+
+ A restart which cannot find its sockets takes a few milliseconds to
+ bring the parent down, and its old children go on serving after it,
+ so neither an immediate check nor a request proves anything.
+ """
+ end = time.time() + settle
+ while time.time() < end:
+ if self.proc.poll() is not None:
+ return False
+ time.sleep(0.1)
+ return True
+
+ def reload(self) -> int:
+ """Ask the running server to restart gracefully."""
+ r = self.env.run([self.env.httpd_bin, '-d', self.env.server_dir,
+ '-f', self.conf_file, '-k', 'graceful'],
+ env=self.env.server_env())
+ return r.exit_code
+
+ def _signal_group(self, sig: int) -> bool:
+ """Signal every process still in the server's group, reporting
+ whether any remained. start_new_session() made the process we
+ launched the group leader, so its pid is the group id whether or
+ not it is still alive."""
+ try:
+ os.killpg(self.proc.pid, sig)
+ return True
+ except OSError:
+ return False
+
+ def stop(self):
+ if self.proc is None:
+ return
+ self._signal_group(signal.SIGTERM)
+ if self.proc.poll() is None:
+ try:
+ self.stdout, self.stderr = self.proc.communicate(timeout=10)
+ except subprocess.TimeoutExpired:
+ self._signal_group(signal.SIGKILL)
+ self.stdout, self.stderr = self.proc.communicate()
+ # A parent which died during a failed restart leaves its children
+ # behind, still holding the listening socket and still answering.
+ # They have to go too, or the next test finds the port taken.
+ end = time.time() + 5
+ while self._signal_group(0):
+ if time.time() >= end:
+ self._signal_group(signal.SIGKILL)
+ break
+ time.sleep(0.1)
+
+ def __enter__(self) -> 'ActivatedServer':
+ return self.start()
+
+ def __exit__(self, *args):
+ self.stop()
+
+
+# Type=notify-reload and ReloadSignal= both arrived in systemd 253. An
+# unrecognised Type= does not degrade to anything, it stops the unit loading
+# at all, so there is nothing to fall back to and the tests are skipped.
+NOTIFY_RELOAD_VERSION = 253
+
+
+def systemd_version() -> int:
+ """The version of the systemd on this host, or 0 if it cannot be asked.
+ "systemctl --version" opens with "systemd 259 (259.8-1.fc44)"."""
+ try:
+ p = subprocess.run(['systemctl', '--version'], capture_output=True,
+ text=True, timeout=15)
+ except (OSError, subprocess.TimeoutExpired):
+ return 0
+ m = re.match(r'systemd (\d+)', p.stdout)
+ return int(m.group(1)) if m else 0
+
+
+class TransientService:
+ """httpd run as a real transient systemd unit, with systemd-run.
+
+ This is the only harness here which exercises the notification protocol
+ against systemd itself rather than a stand-in socket: systemd provides
+ NOTIFY_SOCKET, holds the service in "activating" until READY=1 arrives,
+ tracks MAINPID, and shows the reported STATUS= as the unit's status
+ text. It needs a per-user service manager, which a login session has
+ but a bare CI container does not.
+
+ service_type selects what is exercised: "notify" for startup and
+ shutdown, "notify-reload" for the reload protocol on top of them.
+ """
+
+ def __init__(self, env: SystemdTestEnv, port: int,
+ name: str = None, extra: str = '',
+ service_type: str = 'notify', exec_reload: bool = True,
+ properties: List[str] = None):
+ self.env = env
+ self.port = port
+ self.unit = name or f'httpd-test-{os.getpid()}'
+ self.service_type = service_type
+ self.exec_reload = exec_reload
+ self.conf_file = write_server_conf(env, 'transient', port, extra=extra)
+ self.pid_file = os.path.join(env.server_logs_dir, 'transient.pid')
+ # Extra --property arguments for the unit, such as WatchdogSec= or
+ # ReloadSignal=.
+ self.properties = list(properties or [])
+
+ def read_pid(self) -> Optional[int]:
+ try:
+ with open(self.pid_file) as fd:
+ return int(fd.read().strip())
+ except (OSError, ValueError):
+ return None
+
+ @staticmethod
+ def is_available() -> bool:
+ """Whether this user has a systemd manager to run services under."""
+ if shutil.which('systemd-run') is None:
+ return False
+ if not os.environ.get('XDG_RUNTIME_DIR'):
+ return False
+ # Talking to the manager at all is the test: without a login
+ # session, or with lingering off, there is none to talk to.
+ try:
+ return subprocess.run(['systemctl', '--user', 'show', '-p',
+ 'Version'], capture_output=True,
+ timeout=15).returncode == 0
+ except (OSError, subprocess.TimeoutExpired):
+ return False
+
+ def systemctl(self, *args,
+ timeout: float = 60.0) -> subprocess.CompletedProcess:
+ return subprocess.run(['systemctl', '--user', *args],
+ capture_output=True, text=True, timeout=timeout)
+
+ def show(self, prop: str) -> str:
+ r = self.systemctl('show', '-p', prop, '--value', f'{self.unit}.service')
+ return r.stdout.strip()
+
+ def start(self, timeout: float = 20.0) -> subprocess.CompletedProcess:
+ httpd = self.env.httpd_bin
+ props = [
+ f'--service-type={self.service_type}',
+ '--property=KillMode=mixed',
+ # A reload which is never reported finished holds the job open
+ # until this elapses, so keep it to the same bound as the start.
+ f'--property=TimeoutStartSec={int(timeout)}',
+ ]
+ if self.exec_reload:
+ props.append(f'--property=ExecReload={httpd} '
+ f'-d {self.env.server_dir} '
+ f'-f {self.conf_file} -k graceful')
+ props += [f'--property={p}' for p in self.properties]
+ r = subprocess.run([
+ 'systemd-run', '--user', '--collect', '--quiet',
+ '--unit', self.unit, *props,
+ httpd, '-DFOREGROUND',
+ '-d', self.env.server_dir, '-f', self.conf_file,
+ ], capture_output=True, text=True, timeout=timeout)
+ return r
+
+ def reload(self, timeout: float = 60.0) -> subprocess.CompletedProcess:
+ """Ask the manager to reload the unit. Under Type=notify-reload
+ this returns once the server has reported the reload finished;
+ under Type=notify, as soon as ExecReload= has exited."""
+ return self.systemctl('reload', f'{self.unit}.service',
+ timeout=timeout)
+
+ def rewrite_conf(self, extra: str = ''):
+ """Replace the configuration the unit reads on its next reload.
+ The path does not change, so ExecStart= and ExecReload= still name
+ it."""
+ self.conf_file = write_server_conf(self.env, 'transient', self.port,
+ extra=extra)
+
+ def wait_active(self, timeout: float = 20.0) -> bool:
+ end = time.time() + timeout
+ while time.time() < end:
+ if self.show('ActiveState') == 'active':
+ return True
+ time.sleep(0.2)
+ return False
+
+ def stop(self):
+ self.systemctl('stop', f'{self.unit}.service')
+ self.systemctl('reset-failed', f'{self.unit}.service')
+
+ def __enter__(self) -> 'TransientService':
+ return self
+
+ def __exit__(self, *args):
+ self.stop()
+
+
+class ForegroundServer(ActivatedServer):
+ """httpd run directly in the foreground, with extra environment.
+
+ The watchdog protocol is keyed to a process id: sd_watchdog_enabled(3)
+ ignores $WATCHDOG_USEC unless $WATCHDOG_PID is unset or names the
+ process reading it. apachectl cannot be used to set that, since the
+ variable has to name the parent httpd and the pid is not known until it
+ exists, so the value is assigned in a shell which then exec's httpd in
+ its own place -- the same trick ActivatedServer uses for $LISTEN_PID.
+
+ Assignments are shell words, so "$$" in a value is the pid httpd will
+ have.
+ """
+
+ def __init__(self, env: SystemdTestEnv, port: int,
+ name: str = 'foreground', extra: str = '',
+ setenv: Dict[str, str] = None,
+ modules_conf: str = 'modules.conf'):
+ super().__init__(env, port, name=name, extra=extra,
+ modules_conf=modules_conf)
+ self.setenv = dict(setenv or {})
+
+ def args(self, fd: int = None) -> List[str]:
+ assigns = ' '.join(f'{k}={v}' for k, v in self.setenv.items())
+ return [
+ 'bash', '-c',
+ f'export {assigns}; exec "$0" "$@"' if assigns else 'exec "$0" "$@"',
+ self.env.httpd_bin, '-DFOREGROUND',
+ '-d', self.env.server_dir, '-f', self.conf_file,
+ ]
+
+ def start(self) -> 'ForegroundServer':
+ # A new session, so the whole group can be signalled on the way out
+ # even when a failed restart leaves children behind.
+ self.proc = subprocess.Popen(
+ self.args(), env=self.env.server_env(), start_new_session=True,
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE)
+ return self
diff --git a/test/modules/arch/linux/test_001_notify.py b/test/modules/arch/linux/test_001_notify.py
new file mode 100644
index 00000000000..d79980bc0ac
--- /dev/null
+++ b/test/modules/arch/linux/test_001_notify.py
@@ -0,0 +1,152 @@
+import time
+
+import pytest
+
+from pyhttpd.conf import HttpdConf
+
+
+class TestSystemdNotify:
+ """The service notifications mod_systemd sends over $NOTIFY_SOCKET
+ across the server lifecycle."""
+
+ @pytest.fixture(autouse=True, scope='class')
+ def _class_scope(self, env):
+ conf = HttpdConf(env)
+ conf.add_vhost_test1()
+ conf.install()
+ # Each test drives startup itself, so leave the server down.
+ assert env.apache_stop() == 0
+
+ @staticmethod
+ def index_of(messages, match):
+ for i, msg in enumerate(messages):
+ if match(msg):
+ return i
+ return -1
+
+ def test_systemd_001_01_startup(self, env):
+ """Startup reports configuration reading, then readiness, then that
+ the MPM is serving."""
+ env.notify.clear()
+ assert env.apache_restart() == 0
+ assert env.notify.wait_for_status(r'^Reading configuration\.\.\.$'), \
+ f"no reload notification, got {env.notify.statuses()}"
+ assert env.notify.wait_for_status(r'^Configuration loaded\.$'), \
+ f"no ready notification, got {env.notify.statuses()}"
+ assert env.notify.wait_for_status(r'^Processing requests\.\.\.$'), \
+ f"no pre_mpm notification, got {env.notify.statuses()}"
+
+ def test_systemd_001_02_reloading_before_ready(self, env):
+ """RELOADING=1 is sent while the configuration is read, and READY=1
+ only once it has been."""
+ env.notify.clear()
+ assert env.apache_restart() == 0
+ assert env.notify.wait_for_status(r'^Configuration loaded\.$')
+ msgs = env.notify.messages
+ reloading = self.index_of(
+ msgs, lambda m: m.get('RELOADING') == '1'
+ and m.get('STATUS') == 'Reading configuration...')
+ ready = self.index_of(
+ msgs, lambda m: m.get('READY') == '1'
+ and m.get('STATUS') == 'Configuration loaded.')
+ assert reloading >= 0 and ready >= 0
+ assert reloading < ready, \
+ "READY=1 was reported before the configuration was read"
+
+ def test_systemd_001_03_mainpid(self, env):
+ """MAINPID is the pid of the parent process, the one httpd records
+ in its pid file."""
+ env.notify.clear()
+ assert env.apache_restart() == 0
+ msg = env.notify.wait_for_key('MAINPID')
+ assert msg, f"no MAINPID reported, got {env.notify.messages}"
+ assert msg.get('READY') == '1'
+ assert msg.get('STATUS') == 'Processing requests...'
+ assert int(msg['MAINPID']) == env.read_pid_file()
+
+ def test_systemd_001_04_reload(self, env):
+ """A graceful restart reports reading the configuration and then
+ being ready again."""
+ assert env.apache_restart() == 0
+ pid = env.read_pid_file()
+ env.notify.clear()
+ assert env.apache_reload() == 0
+ assert env.notify.wait_for_status(r'^Reading configuration\.\.\.$'), \
+ f"no reload notification, got {env.notify.statuses()}"
+ assert env.notify.wait_for_status(r'^Configuration loaded\.$'), \
+ f"no ready notification, got {env.notify.statuses()}"
+ # The parent survives a graceful restart, and the MPM is not started
+ # over, so the pid systemd tracks neither changes nor is re-reported.
+ assert env.read_pid_file() == pid
+ for msg in env.notify.messages:
+ assert 'MAINPID' not in msg or int(msg['MAINPID']) == pid
+
+ def test_systemd_001_05_hard_restart(self, env):
+ """An ungraceful restart starts the MPM over, and re-reports the
+ main pid, which is still that of the surviving parent."""
+ assert env.apache_restart() == 0
+ pid = env.read_pid_file()
+ env.notify.clear()
+ assert env.apache_hard_restart() == 0
+ assert env.notify.wait_for_status(r'^Configuration loaded\.$'), \
+ f"no ready notification, got {env.notify.statuses()}"
+ msg = env.notify.wait_for_key('MAINPID')
+ assert msg, f"no MAINPID after restart, got {env.notify.messages}"
+ assert int(msg['MAINPID']) == pid
+ assert env.read_pid_file() == pid
+
+ def test_systemd_001_06_notify_socket_kept(self, env):
+ """mod_systemd leaves NOTIFY_SOCKET in the environment, so a second
+ start after a stop is reported just like the first."""
+ assert env.apache_restart() == 0
+ assert env.apache_stop() == 0
+ env.notify.clear()
+ assert env.apache_restart() == 0
+ assert env.notify.wait_for_status(r'^Configuration loaded\.$')
+
+ def test_systemd_001_07_stopping(self, env):
+ """Shutdown is announced before the process goes away."""
+ assert env.apache_restart() == 0
+ env.notify.clear()
+ assert env.apache_stop() == 0
+ # The server is already gone, so anything it sent has arrived.
+ msg = env.notify.wait_for_key('STOPPING', timeout=1)
+ assert msg, f"no STOPPING=1 on shutdown, got {env.notify.messages}"
+ assert msg['STOPPING'] == '1'
+ assert msg.get('STATUS') == 'Shutting down.'
+
+ def test_systemd_001_08_reloading_monotonic(self, env):
+ assert env.apache_restart() == 0
+ env.notify.clear()
+ assert env.apache_reload() == 0
+ msg = env.notify.wait_for(lambda m: m.get('RELOADING') == '1')
+ assert msg, "no RELOADING=1 on graceful restart"
+ assert 'MONOTONIC_USEC' in msg, \
+ f"RELOADING=1 sent without MONOTONIC_USEC: {msg}"
+ # systemd compares this against its own reading of the same clock.
+ assert 0 < int(msg['MONOTONIC_USEC']) <= time.clock_gettime_ns(
+ time.CLOCK_MONOTONIC) // 1000
+
+ def test_systemd_001_09_watchdog(self, env):
+ """A server started with $WATCHDOG_USEC reports that it is alive.
+
+ $WATCHDOG_PID is left unset, which sd_watchdog_enabled(3) takes to
+ mean any process may report: it has to be, since apachectl starts a
+ parent whose pid nothing knew in advance. The keep-alive arrives
+ with the startup notifications rather than only from the monitor
+ hook ten seconds later, which is what makes this cheap to check.
+ test_006_watchdog.py covers the rest of the protocol.
+ """
+ assert env.apache_stop() == 0
+ # Long enough that mod_systemd does not object to it (AH10622).
+ env.set_httpd_env('WATCHDOG_USEC', str(60 * 1000000))
+ try:
+ env.notify.clear()
+ assert env.apache_restart() == 0
+ msg = env.notify.wait_for_key('WATCHDOG', timeout=4)
+ assert msg, f"no watchdog keepalive was sent, " \
+ f"got {env.notify.messages}"
+ assert msg['WATCHDOG'] == '1'
+ finally:
+ env.set_httpd_env('WATCHDOG_USEC', None)
+ assert env.apache_restart() == 0
diff --git a/test/modules/arch/linux/test_002_monitor.py b/test/modules/arch/linux/test_002_monitor.py
new file mode 100644
index 00000000000..368d83ae74f
--- /dev/null
+++ b/test/modules/arch/linux/test_002_monitor.py
@@ -0,0 +1,107 @@
+import re
+
+import pytest
+
+from pyhttpd.conf import HttpdConf
+
+from .env import MONITOR_STATUS, MONITOR_TIMEOUT
+
+
+class TestSystemdMonitor:
+ """The periodic STATUS= line the monitor hook reports, which is what
+ "systemctl status httpd" shows below the unit description."""
+
+ @pytest.fixture(autouse=True, scope='class')
+ def _class_scope(self, env):
+ conf = HttpdConf(env, extras={
+ 'base': """
+
+ SetHandler server-status
+
+ """
+ })
+ conf.add_vhost_test1()
+ conf.install()
+ assert env.apache_restart() == 0
+
+ @staticmethod
+ def wait_report(env, above=-1):
+ """Wait for a report from the monitor hook accounting for more than
+ `above` requests, and return its fields."""
+ msg = env.notify.wait_for(
+ lambda m: 'STATUS' in m
+ and (g := MONITOR_STATUS.match(m['STATUS'])) is not None
+ and int(g.group('requests')) > above,
+ timeout=MONITOR_TIMEOUT)
+ assert msg, f"no monitor report within {MONITOR_TIMEOUT}s, " \
+ f"server {'up' if env.is_live() else 'down'}, " \
+ f"got {env.notify.messages}"
+ return MONITOR_STATUS.match(msg['STATUS'])
+
+ @pytest.fixture(scope='class')
+ def reports(self, env, _class_scope):
+ """Two consecutive reports with requests served between them, and
+ the mod_status report as of the second.
+
+ The hook runs once every ten turns of the parent's one second
+ loop, so waiting for a report costs up to ten seconds. The tests
+ below share one pair rather than each waiting for its own.
+ """
+ env.notify.clear()
+ first = self.wait_report(env)
+ for _ in range(10):
+ r = env.curl_get(env.mkurl("http", "test1", "/"))
+ assert r.response['status'] == 200
+ second = self.wait_report(env, above=int(first.group('requests')))
+ r = env.curl_get(env.mkurl("http", "test1", "/server-status?auto"))
+ assert r.response['status'] == 200
+ auto = {}
+ for line in r.response['body'].decode().splitlines():
+ key, sep, value = line.partition(':')
+ if sep and key not in auto:
+ auto[key] = value.strip()
+ return {'first': first, 'second': second, 'auto': auto}
+
+ def test_systemd_002_01_report_format(self, reports):
+ m = reports['first']
+ assert int(m.group('requests')) >= 0
+ assert m.group('bps')
+
+ def test_systemd_002_02_requests_counted(self, reports):
+ """The request count reported to systemd tracks requests served."""
+ before = int(reports['first'].group('requests'))
+ after = int(reports['second'].group('requests'))
+ assert after >= before + 10, \
+ f"{after - before} requests reported, at least 10 were served"
+
+ def test_systemd_002_03_rates_are_finite(self, reports):
+ """Neither rate is inf or nan.
+
+ systemd_monitor() divides by an uptime in whole seconds, which is
+ zero for a report arriving in the first second after the scoreboard
+ records a restart.
+ """
+ for which in ('first', 'second'):
+ m = reports[which]
+ rate = m.group('rate')
+ assert re.match(r'^-?\d', rate), f"{which} request rate is {rate!r}"
+ assert float(rate) >= 0
+ assert 'inf' not in m.group('bps') and 'nan' not in m.group('bps'), \
+ f"{which} byte rate is {m.group('bps')!r}"
+
+ def test_systemd_002_04_report_repeats(self, env, reports):
+ """The status line keeps being refreshed while the server runs."""
+ assert reports['first'].group(0) != reports['second'].group(0)
+
+ def test_systemd_002_05_worker_percentages(self, reports):
+ """Idle and busy are percentages of the workers available, and are
+ reported as such."""
+ m, auto = reports['second'], reports['auto']
+ idle, busy = int(m.group('idle')), int(m.group('busy'))
+ assert 0 <= idle <= 100 and 0 <= busy <= 100
+ # Integer division loses at most one point between the two.
+ assert 99 <= idle + busy <= 100
+ # Busy workers are a minority of a mostly idle test server, which
+ # is what mod_status reports over the same scoreboard.
+ assert int(auto['IdleWorkers']) > int(auto['BusyWorkers'])
+ assert idle > busy
diff --git a/test/modules/arch/linux/test_003_extended_status.py b/test/modules/arch/linux/test_003_extended_status.py
new file mode 100644
index 00000000000..badc1483944
--- /dev/null
+++ b/test/modules/arch/linux/test_003_extended_status.py
@@ -0,0 +1,66 @@
+import pytest
+
+from pyhttpd.conf import HttpdConf
+
+from .env import NO_MONITOR_TIMEOUT
+
+
+class TestSystemdExtendedStatus:
+ """mod_systemd turns ExtendedStatus on so that it has request counts to
+ report, which changes what the rest of the server records too."""
+
+ @pytest.fixture(autouse=True, scope='class')
+ def _class_scope(self, env):
+ yield
+ conf = HttpdConf(env)
+ conf.add_vhost_test1()
+ conf.install()
+ assert env.apache_restart() == 0
+
+ def auto_report(self, env):
+ r = env.curl_get(env.mkurl("http", "test1", "/server-status?auto"))
+ assert r.response['status'] == 200
+ return r.response['body'].decode()
+
+ def install(self, env, extra=''):
+ conf = HttpdConf(env, extras={
+ 'base': f"""
+ {extra}
+
+ SetHandler server-status
+
+ """
+ })
+ conf.add_vhost_test1()
+ conf.install()
+ assert env.apache_restart() == 0
+
+ def test_systemd_003_01_enabled_by_default(self, env):
+ """Loading mod_systemd is enough to get extended status; no
+ ExtendedStatus directive is present in this configuration."""
+ self.install(env)
+ body = self.auto_report(env)
+ assert 'Total Accesses:' in body, \
+ "ExtendedStatus was not enabled by mod_systemd"
+ assert 'Total kBytes:' in body
+
+ def test_systemd_003_02_directive_wins(self, env):
+ """An explicit ExtendedStatus off still takes effect: mod_systemd
+ sets the default in pre_config, before the configuration is read."""
+ self.install(env, extra='ExtendedStatus off')
+ body = self.auto_report(env)
+ assert 'Total Accesses:' not in body, \
+ "ExtendedStatus off was overridden by mod_systemd"
+
+ def test_systemd_003_03_no_report_without_extended_status(self, env):
+ """With extended status off the monitor hook declines, so no status
+ line is reported to systemd."""
+ env.notify.clear()
+ self.install(env, extra='ExtendedStatus off')
+ # Startup notifications are still sent...
+ assert env.notify.wait_for_status(r'^Processing requests\.\.\.$',
+ timeout=10) is not None
+ # ...but the periodic status line is not.
+ assert env.notify.wait_for_status(r'^Total requests: ',
+ timeout=NO_MONITOR_TIMEOUT) is None, \
+ "a monitor report was sent with ExtendedStatus off"
diff --git a/test/modules/arch/linux/test_004_socket_activation.py b/test/modules/arch/linux/test_004_socket_activation.py
new file mode 100644
index 00000000000..b92fadd36c7
--- /dev/null
+++ b/test/modules/arch/linux/test_004_socket_activation.py
@@ -0,0 +1,89 @@
+import pytest
+
+from .env import ActivatedServer
+
+
+class TestSystemdSocketActivation:
+ """Listening sockets passed in by the service manager.
+
+ mod_systemd exports the two optional functions server/listen.c uses to
+ find them, so socket activation is enabled by loading the module and
+ disabled by not loading it, whatever the environment says.
+
+ apachectl cannot pass file descriptors, so these tests open the
+ listening socket themselves and run httpd directly, in the foreground,
+ with the descriptor and the environment a service manager would give
+ it. No systemd process is involved.
+ """
+
+ @pytest.fixture(autouse=True, scope='class')
+ def _class_scope(self, env):
+ # The activated servers use the same server root; keep the one
+ # started by other tests out of the way.
+ assert env.apache_stop() == 0
+ yield
+ assert env.apache_stop() == 0
+
+ def test_systemd_004_01_activated_listener(self, env):
+ """httpd serves on a socket it never opened itself."""
+ env.notify.clear()
+ with ActivatedServer(env, port=env.http_port2) as server:
+ assert server.is_live(), \
+ f"server did not come up: {server.stderr}"
+ r = env.curl_get(f"http://{env.http_addr}:{env.http_port2}/")
+ assert r.response['status'] == 200
+ # The notification handshake works the same way as when httpd opens
+ # its own sockets.
+ assert env.notify.wait_for_status(r'^Configuration loaded\.$', timeout=0)
+ msg = env.notify.wait_for_key('MAINPID', timeout=0)
+ assert msg, f"httpd sent no MAINPID, got {env.notify.messages}"
+ assert msg['STATUS'] == 'Processing requests...'
+
+ def test_systemd_004_02_no_socket_for_port(self, env):
+ """A Listen port the service manager did not pass is an error, not
+ a port httpd quietly opens for itself."""
+ server = ActivatedServer(env, port=env.http_port2,
+ listen_port=env.proxy_port,
+ name='activate-wrongport')
+ with server:
+ assert server.wait_exit() != 0, "httpd started without a socket"
+ assert b'not configured in systemd' in server.stderr, \
+ f"unexpected startup diagnostic: {server.stderr}"
+
+ def test_systemd_004_03_disabled_without_module(self, env):
+ """Without mod_systemd the passed sockets are ignored and httpd
+ opens the configured port itself, the same arrangement that fails
+ in the test above."""
+ if not env.systemd_is_dso:
+ pytest.skip("mod_systemd is linked statically and cannot be "
+ "left out of the configuration")
+ modules_conf = ActivatedServer.modules_conf_without(env, 'systemd')
+ server = ActivatedServer(env, port=env.http_port2,
+ listen_port=env.proxy_port,
+ name='activate-nomodule',
+ modules_conf=modules_conf)
+ with server:
+ assert server.is_live(), \
+ f"server did not come up: {server.stderr}"
+ r = env.curl_get(f"http://{env.http_addr}:{env.http_port2}/")
+ assert r.response['status'] == 200
+
+ def test_systemd_004_04_graceful_restart(self, env):
+ """An activated server survives a graceful restart, which means
+ finding the passed sockets again after the environment naming them
+ has been cleared."""
+ with ActivatedServer(env, port=env.http_port2,
+ name='activate-reload') as server:
+ assert server.is_live(), \
+ f"server did not come up: {server.stderr}"
+ server.reload()
+ env.httpd_error_log.ignore_recent(lognos=['AH02487'])
+ # Check the parent first: when it dies here its children carry
+ # on holding the listening socket and answering, so a request
+ # succeeding proves nothing on its own.
+ assert server.is_running(), \
+ "the parent exited on graceful restart"
+ assert server.is_live(timeout=5), \
+ "the server did not survive a graceful restart"
+ r = env.curl_get(f"http://{env.http_addr}:{env.http_port2}/")
+ assert r.response['status'] == 200
diff --git a/test/modules/arch/linux/test_005_service.py b/test/modules/arch/linux/test_005_service.py
new file mode 100644
index 00000000000..ef220426302
--- /dev/null
+++ b/test/modules/arch/linux/test_005_service.py
@@ -0,0 +1,89 @@
+import time
+
+import pytest
+
+from .env import MONITOR_STATUS, MONITOR_TIMEOUT, TransientService, http_responds
+
+pytestmark = pytest.mark.skipif(
+ not TransientService.is_available(),
+ reason="no per-user systemd manager to run a transient service under")
+
+
+class TestSystemdService:
+ """httpd as a real systemd service, end to end.
+
+ Everything else here checks what mod_systemd sends. These check what
+ systemd does with it: hold the unit in "activating" until httpd is
+ ready, track the right process, and show the reported status text.
+ """
+
+ @pytest.fixture(autouse=True, scope='class')
+ def _class_scope(self, env):
+ assert env.apache_stop() == 0
+ yield
+ assert env.apache_stop() == 0
+
+ @pytest.fixture
+ def service(self, env) -> TransientService:
+ svc = TransientService(env, port=env.http_port2)
+ yield svc
+ svc.stop()
+
+ def test_systemd_005_01_type_notify(self, env, service):
+ """systemd-run returns once the unit is active, which for a
+ Type=notify service means READY=1 has been received, which
+ mod_systemd sends only after the configuration is loaded."""
+ r = service.start()
+ assert r.returncode == 0, f"systemd-run failed: {r.stderr}"
+ assert service.show('ActiveState') == 'active'
+ assert http_responds(env.http_port2), \
+ "the unit was active before the server would answer"
+
+ def test_systemd_005_02_main_pid(self, env, service):
+ """The process systemd tracks is the httpd parent."""
+ assert service.start().returncode == 0
+ assert service.wait_active()
+ main_pid = int(service.show('MainPID'))
+ assert main_pid > 0
+ assert main_pid == service.read_pid()
+
+ def test_systemd_005_03_status_text(self, env, service):
+ """The status line "systemctl status" shows comes from the monitor
+ hook, and is refreshed while the server runs."""
+ assert service.start().returncode == 0
+ assert service.wait_active()
+ # First the post_config report, then the periodic one.
+ assert service.show('StatusText') in ('Configuration loaded.',
+ 'Processing requests...')
+ end = time.time() + MONITOR_TIMEOUT
+ text = None
+ while time.time() < end:
+ text = service.show('StatusText')
+ if MONITOR_STATUS.match(text):
+ break
+ time.sleep(0.5)
+ assert MONITOR_STATUS.match(text), \
+ f"status text was never refreshed by the monitor hook: {text!r}"
+
+ def test_systemd_005_04_reload(self, env, service):
+ """systemctl reload runs httpd -k graceful and the unit stays
+ active throughout."""
+ assert service.start().returncode == 0
+ assert service.wait_active()
+ pid = int(service.show('MainPID'))
+ r = service.systemctl('reload', f'{service.unit}.service')
+ assert r.returncode == 0, f"reload failed: {r.stderr}"
+ assert service.show('ActiveState') == 'active'
+ assert int(service.show('MainPID')) == pid
+ assert http_responds(env.http_port2)
+
+ def test_systemd_005_05_stop(self, env, service):
+ """The unit stops cleanly, without systemd having to time out and
+ kill it."""
+ assert service.start().returncode == 0
+ assert service.wait_active()
+ r = service.systemctl('stop', f'{service.unit}.service')
+ assert r.returncode == 0, f"stop failed: {r.stderr}"
+ assert service.show('ActiveState') == 'inactive'
+ assert service.show('Result') == 'success'
+ assert not http_responds(env.http_port2)
diff --git a/test/modules/arch/linux/test_006_watchdog.py b/test/modules/arch/linux/test_006_watchdog.py
new file mode 100644
index 00000000000..d4fa8653bf6
--- /dev/null
+++ b/test/modules/arch/linux/test_006_watchdog.py
@@ -0,0 +1,211 @@
+import os
+import re
+import time
+
+import pytest
+
+from .env import (ForegroundServer, TransientService, MIN_WATCHDOG_SEC,
+ NO_MONITOR_TIMEOUT, WATCHDOG_TIMEOUT, is_watchdog_ping)
+
+
+def watchdog_env(usec: int, pid: str = '$$') -> dict:
+ """The environment a service manager sets for a watched service.
+
+ $WATCHDOG_PID names the process expected to report; "$$" is the shell
+ which exec's httpd, so it is the pid httpd will run as.
+ """
+ env = {'WATCHDOG_USEC': str(usec)}
+ if pid is not None:
+ env['WATCHDOG_PID'] = pid
+ return env
+
+
+class TestSystemdWatchdog:
+ """The keep-alive ping of the systemd watchdog protocol.
+
+ A service whose unit sets WatchdogSec= is started with $WATCHDOG_USEC
+ holding that timeout in microseconds, and $WATCHDOG_PID holding the pid
+ expected to report. The service must then send "WATCHDOG=1" to the
+ notification socket more often than the timeout, or systemd puts the
+ unit into a failed state with Result=watchdog. The timeout is armed
+ once start-up completes and, as test_006_06 relies on, stays armed
+ across a reload.
+
+ Nothing here needs systemd: the ping goes to $NOTIFY_SOCKET like every
+ other notification, so the same stand-in socket observes it. Only the
+ last test involves a service manager.
+ """
+
+ @pytest.fixture(autouse=True, scope='class')
+ def _class_scope(self, env):
+ # These run their own servers on the second port; the shared one
+ # would only add its notifications to the same socket.
+ assert env.apache_stop() == 0
+ yield
+ assert env.apache_stop() == 0
+
+ @pytest.fixture
+ def server_factory(self, env):
+ started = []
+
+ def make(usec=None, pid='$$', extra='', name='watchdog'):
+ setenv = watchdog_env(usec, pid) if usec is not None else {}
+ srv = ForegroundServer(env, port=env.http_port2, name=name,
+ extra=extra, setenv=setenv)
+ started.append(srv)
+ env.notify.clear()
+ srv.start()
+ assert srv.is_live(), \
+ f"server did not come up: {srv.stderr!r}"
+ return srv
+
+ yield make
+ for srv in started:
+ srv.stop()
+
+ def test_systemd_006_01_no_ping_when_unwatched(self, env, server_factory):
+ """A server the service manager is not watching sends no keep-alive.
+
+ The monitor hook still runs -- its status report is the evidence of
+ that -- so the absence of a ping is a decision and not silence.
+ """
+ server_factory()
+ assert env.notify.wait_for_status(r'^Total requests: ',
+ timeout=WATCHDOG_TIMEOUT), \
+ "the monitor hook never ran, so this proves nothing"
+ assert not [m for m in env.notify.messages if is_watchdog_ping(m)], \
+ "a keep-alive was sent with $WATCHDOG_USEC unset"
+
+ def test_systemd_006_02_ping_when_watched(self, env, server_factory):
+ """With the watchdog enabled the keep-alive is sent."""
+ server_factory(usec=60 * 1000000)
+ assert env.notify.wait_for(is_watchdog_ping,
+ timeout=WATCHDOG_TIMEOUT), \
+ f"no keep-alive within {WATCHDOG_TIMEOUT}s, " \
+ f"got {env.notify.messages}"
+
+ def test_systemd_006_03_ping_repeats(self, env, server_factory):
+ """The keep-alive is periodic, which is the whole point of it: one
+ ping would satisfy a test but not systemd."""
+ server_factory(usec=60 * 1000000)
+ for n in range(2):
+ assert env.notify.wait_for(is_watchdog_ping,
+ timeout=WATCHDOG_TIMEOUT), \
+ f"only {n} keep-alives arrived, got {env.notify.messages}"
+ env.notify.clear()
+
+ def test_systemd_006_04_ping_without_extended_status(self, env,
+ server_factory):
+ """The keep-alive does not depend on ExtendedStatus.
+
+ The monitor hook declines early when it has no request counts to
+ report (test_003_03), and a server whose status line is switched
+ off must still be reported as alive.
+ """
+ server_factory(usec=60 * 1000000, extra='ExtendedStatus off')
+ assert env.notify.wait_for_status(r'^Total requests: ',
+ timeout=NO_MONITOR_TIMEOUT) is None, \
+ "ExtendedStatus off did not stop the status report"
+ assert env.notify.wait_for(is_watchdog_ping,
+ timeout=WATCHDOG_TIMEOUT), \
+ "no keep-alive with ExtendedStatus off"
+
+ def test_systemd_006_05_no_ping_for_another_pid(self, env, server_factory):
+ """$WATCHDOG_PID naming a different process means the variables were
+ set for something further up the process tree, and must be ignored.
+ """
+ server_factory(usec=60 * 1000000, pid='1')
+ assert env.notify.wait_for_status(r'^Total requests: ',
+ timeout=WATCHDOG_TIMEOUT), \
+ "the monitor hook never ran, so this proves nothing"
+ assert not [m for m in env.notify.messages if is_watchdog_ping(m)], \
+ "a keep-alive was sent although $WATCHDOG_PID was another process"
+
+ def test_systemd_006_06_ping_across_reload(self, env, server_factory):
+ """The keep-alive survives a graceful restart.
+
+ The parent re-reads its configuration in the same process, but
+ mod_systemd is unloaded and loaded again with it, so anything it
+ remembered about the watchdog is gone by the time the monitor hook
+ runs again. systemd keeps the timeout armed throughout.
+ """
+ srv = server_factory(usec=60 * 1000000)
+ assert env.notify.wait_for(is_watchdog_ping, timeout=WATCHDOG_TIMEOUT)
+ env.notify.clear()
+ assert srv.reload() == 0
+ assert srv.is_live()
+ assert env.notify.wait_for(is_watchdog_ping,
+ timeout=WATCHDOG_TIMEOUT), \
+ f"no keep-alive after a reload, got {env.notify.messages}"
+
+ def test_systemd_006_07_ping_covers_the_reload_window(self, env,
+ server_factory):
+ """A keep-alive is sent as the configuration is read, and again once
+ it is loaded.
+
+ Reading the configuration happens outside the parent's monitor loop,
+ so nothing else reports during it. The watchdog stays armed while
+ the unit reloads, and a configuration which takes longer to parse
+ than the timeout would otherwise be killed halfway through.
+ """
+ srv = server_factory(usec=60 * 1000000)
+ assert env.notify.wait_for(is_watchdog_ping, timeout=WATCHDOG_TIMEOUT)
+ env.notify.clear()
+ assert srv.reload() == 0
+ assert env.notify.wait_for(
+ lambda m: 'RELOADING' in m and is_watchdog_ping(m),
+ timeout=WATCHDOG_TIMEOUT), \
+ f"no keep-alive as the configuration was read, " \
+ f"got {env.notify.messages}"
+ assert env.notify.wait_for(
+ lambda m: m.get('READY') == '1' and is_watchdog_ping(m),
+ timeout=WATCHDOG_TIMEOUT), \
+ f"no keep-alive once the configuration was loaded, " \
+ f"got {env.notify.messages}"
+
+ def test_systemd_006_08_short_timeout_warned(self, env, server_factory):
+ """A WatchdogSec the parent cannot meet is reported.
+
+ The keep-alive is sent from the monitor hook, which runs once every
+ ten turns of the parent's one second loop. A timeout of a few
+ seconds cannot be met however the module is written, and failing
+ silently would leave the server being killed and restarted with
+ nothing in the log to say why.
+ """
+ server_factory(usec=2 * 1000000)
+ assert env.httpd_error_log.scan_recent(
+ re.compile(r'.*AH10621: .*[Ww]atchdog.*'), timeout=10), \
+ "no warning about a watchdog timeout that cannot be met"
+ env.httpd_error_log.ignore_recent(lognos=['AH10621'])
+
+ def test_systemd_006_09_workable_timeout_not_warned(self, env,
+ server_factory):
+ """A timeout the parent can meet is not complained about."""
+ server_factory(usec=MIN_WATCHDOG_SEC * 1000000)
+ assert env.notify.wait_for(is_watchdog_ping, timeout=WATCHDOG_TIMEOUT)
+ with pytest.raises(TimeoutError):
+ env.httpd_error_log.scan_recent(
+ re.compile(r'.*AH10621: .*'), timeout=1)
+
+ @pytest.mark.skipif(not TransientService.is_available(),
+ reason="no per-user systemd manager")
+ def test_systemd_006_10_service_keeps_watchdog_alive(self, env):
+ """The real thing: systemd records each keep-alive it receives, and
+ the unit stays active rather than failing with Result=watchdog."""
+ with TransientService(env, port=env.http_port2,
+ properties=[f'WatchdogSec={MIN_WATCHDOG_SEC}s'],
+ name=f'httpd-wd-{os.getpid()}') as svc:
+ assert svc.start().returncode == 0
+ assert svc.wait_active()
+ first = svc.show('WatchdogTimestampMonotonic')
+ assert first and int(first) > 0, \
+ "systemd recorded no keep-alive at all"
+ end = time.time() + WATCHDOG_TIMEOUT
+ while time.time() < end:
+ if svc.show('WatchdogTimestampMonotonic') != first:
+ break
+ time.sleep(0.5)
+ assert svc.show('WatchdogTimestampMonotonic') != first, \
+ "systemd received no further keep-alive"
+ assert svc.show('ActiveState') == 'active'
+ assert svc.show('Result') == 'success'
diff --git a/test/modules/arch/linux/test_007_notify_reload.py b/test/modules/arch/linux/test_007_notify_reload.py
new file mode 100644
index 00000000000..93833299c63
--- /dev/null
+++ b/test/modules/arch/linux/test_007_notify_reload.py
@@ -0,0 +1,139 @@
+import os
+import time
+
+import pytest
+
+from .env import (NOTIFY_RELOAD_VERSION, TransientService, error_log_since,
+ error_log_size, http_responds, systemd_version)
+
+pytestmark = [
+ pytest.mark.skipif(
+ not TransientService.is_available(),
+ reason="no per-user systemd manager to run a transient service under"),
+ pytest.mark.skipif(
+ systemd_version() < NOTIFY_RELOAD_VERSION,
+ reason=f"Type=notify-reload needs systemd {NOTIFY_RELOAD_VERSION}"),
+]
+
+
+class TestNotifyReload:
+ """httpd as a Type=notify-reload unit.
+
+ Under Type=notify a reload is only whatever ExecReload= does, and
+ systemctl returns as soon as that command exits. For "httpd -k
+ graceful" that is as soon as the signal has been sent, long before the
+ new configuration is in use, and it reports success however the restart
+ turns out. Type=notify-reload holds the reload job open until the
+ service sends RELOADING=1 and then READY=1, which mod_systemd sends
+ from pre_config and post_config, so the result is reported once it is
+ known.
+
+ The unit here keeps ExecReload= and sets ReloadSignal=SIGCONT, which
+ httpd ignores. The manager runs ExecReload= first and sends
+ ReloadSignal= only once it has exited successfully, so:
+
+ - "httpd -k graceful" stays the thing which restarts the server, and
+ because it parses the new configuration in its own process before
+ signalling, a configuration which does not parse fails the reload
+ without the running server being signalled at all. A bare
+ ReloadSignal= has the running parent read the new configuration and
+ exit if it does not parse.
+
+ - the signal the manager then sends is redundant, so it is pointed at
+ one httpd does not act on. The default is SIGHUP, which httpd
+ takes as an *ungraceful* restart.
+ """
+
+ @pytest.fixture(autouse=True, scope='class')
+ def _class_scope(self, env):
+ assert env.apache_stop() == 0
+ yield
+ assert env.apache_stop() == 0
+
+ @pytest.fixture
+ def service(self, env) -> TransientService:
+ svc = TransientService(env, port=env.http_port2,
+ name=f'httpd-reload-{os.getpid()}',
+ service_type='notify-reload',
+ properties=['ReloadSignal=SIGCONT'])
+ yield svc
+ svc.stop()
+
+ def test_systemd_007_01_reload_waits_for_the_new_config(self, env, service):
+ """systemctl reload returns only once the reloaded server is
+ serving: the port added to the configuration answers without the
+ test waiting for it, because READY=1 is sent from post_config, by
+ which point the listeners are open."""
+ assert service.start().returncode == 0
+ assert service.wait_active()
+ assert http_responds(env.http_port2)
+
+ service.rewrite_conf(extra=f'Listen {env.proxy_port}')
+ r = service.reload()
+ assert r.returncode == 0, f"reload failed: {r.stderr}"
+ assert http_responds(env.proxy_port, timeout=10.0), \
+ "reload returned before the newly configured port was served"
+ assert http_responds(env.http_port2)
+
+ def test_systemd_007_02_broken_config_fails_the_reload(self, env, service):
+ """A configuration which does not parse fails the reload and leaves
+ the running server alone. ExecReload= reads it in a process of its
+ own and exits without signalling, and the manager sends no reload
+ signal of its own once that has failed, so the parent never reads
+ it."""
+ assert service.start().returncode == 0
+ assert service.wait_active()
+ pid = int(service.show('MainPID'))
+
+ service.rewrite_conf(extra='ThisDirectiveDoesNotExist on')
+ r = service.reload()
+ assert r.returncode != 0, \
+ "reload of an unparseable configuration reported success"
+ assert service.show('ActiveState') == 'active', \
+ "a failed reload took the unit out of active"
+ # The parent exits when a restart re-reads a configuration which
+ # does not parse, and its children go on serving for a while
+ # afterwards, so answering a request is not on its own proof that
+ # the server survived: check the process the manager tracks.
+ assert int(service.show('MainPID')) == pid, \
+ "the parent was restarted despite the configuration not parsing"
+ assert http_responds(env.http_port2)
+
+ def test_systemd_007_03_reload_restarts_gracefully_once(self, env, service):
+ """One reload is one graceful restart, and no ungraceful one.
+ Leaving ReloadSignal= at its SIGHUP default is what this catches:
+ httpd restarts ungracefully on SIGHUP, dropping the connections a
+ reload is supposed to keep."""
+ if env.mpm_module not in ('mpm_event', 'mpm_worker'):
+ pytest.skip(f"{env.mpm_module} does not log the graceful restart")
+ assert service.start().returncode == 0
+ assert service.wait_active()
+
+ pos = error_log_size(env)
+ assert service.reload().returncode == 0
+ # The restart is logged before the configuration is re-read, so it
+ # is written by the time READY=1 ends the reload; give the
+ # redundant signal which follows a moment to land as well.
+ time.sleep(2)
+ log = error_log_since(env, pos)
+ assert log.count("Attempting to restart") == 0, \
+ f"the reload restarted the server ungracefully:\n{log}"
+ assert log.count("Doing graceful restart") == 1, \
+ f"expected one graceful restart, log said:\n{log}"
+
+ def test_systemd_007_04_reload_repeats(self, env, service):
+ """Reloading twice works. The manager ignores a RELOADING=1 which
+ is not stamped later than the reload it asked for, so a server
+ which got MONOTONIC_USEC wrong could report the first reload and
+ then leave the second to time out."""
+ assert service.start().returncode == 0
+ assert service.wait_active()
+ pid = int(service.show('MainPID'))
+
+ for i in range(2):
+ r = service.reload()
+ assert r.returncode == 0, f"reload {i + 1} failed: {r.stderr}"
+ assert service.show('ActiveState') == 'active'
+ assert int(service.show('MainPID')) == pid, \
+ "a graceful restart replaced the parent process"
+ assert http_responds(env.http_port2)
diff --git a/test/pyhttpd/conf/httpd.conf.template b/test/pyhttpd/conf/httpd.conf.template
index 255b88ad05f..92cef2e06d2 100644
--- a/test/pyhttpd/conf/httpd.conf.template
+++ b/test/pyhttpd/conf/httpd.conf.template
@@ -1,6 +1,9 @@
ServerName localhost
ServerRoot "${server_dir}"
+DefaultRuntimeDir logs
+PidFile httpd.pid
+
Include "conf/modules.conf"
DocumentRoot "${server_dir}/htdocs"
diff --git a/test/pyhttpd/conf/stop.conf.template b/test/pyhttpd/conf/stop.conf.template
index 21bae845f8d..5e76b2d6e24 100644
--- a/test/pyhttpd/conf/stop.conf.template
+++ b/test/pyhttpd/conf/stop.conf.template
@@ -5,6 +5,11 @@
ServerName localhost
ServerRoot "${server_dir}"
+# Must agree with httpd.conf, or the running server cannot be found:
+# the built-in default varies between httpd versions.
+DefaultRuntimeDir logs
+PidFile httpd.pid
+
Include "conf/modules.conf"
DocumentRoot "${server_dir}/htdocs"
diff --git a/test/pyhttpd/env.py b/test/pyhttpd/env.py
index a30a027fb93..695176d54c7 100644
--- a/test/pyhttpd/env.py
+++ b/test/pyhttpd/env.py
@@ -303,6 +303,7 @@ def __init__(self, pytestconfig=None):
self._verbosity = pytestconfig.option.verbose if pytestconfig is not None else 0
self._test_conf = os.path.join(self._server_conf_dir, "test.conf")
self._httpd_base_conf = []
+ self._httpd_env = {}
self._httpd_log_modules = ['aptest']
self._log_interesting = None
self._setup = None
@@ -327,6 +328,19 @@ def add_httpd_conf(self, lines: List[str]):
def add_httpd_log_modules(self, modules: List[str]):
self._httpd_log_modules.extend(modules)
+ def set_httpd_env(self, name: str, value: Optional[str]):
+ """Add a variable to the environment httpd is started with, or
+ remove it again when passed None.
+
+ Used by tests for modules which take their input from the
+ environment rather than from the configuration, such as
+ mod_systemd reading $NOTIFY_SOCKET.
+ """
+ if value is None:
+ self._httpd_env.pop(name, None)
+ else:
+ self._httpd_env[name] = value
+
def issue_certs(self):
if self._ca is None:
self._ca = HttpdTestCA.create_root(name=self.http_tld,
@@ -690,6 +704,7 @@ def _clean_path_env(self) -> dict:
parts.insert(0, venv_bin)
env = os.environ.copy()
env['PATH'] = os.pathsep.join(parts)
+ env.update(self._httpd_env)
return env
def _run_apachectl(self, cmd) -> ExecResult:
@@ -741,6 +756,26 @@ def apache_fail(self):
rv = 0
return rv
+ def apache_hard_restart(self) -> int:
+ """Restart without the "graceful" flag, so the MPM starts over."""
+ r = self._run_apachectl("restart")
+ if r.exit_code == 0:
+ return 0 if self.is_live(self._http_base,
+ timeout=timedelta(seconds=10)) else -1
+ return r.exit_code
+
+ def read_pid_file(self, name: str = 'httpd.pid') -> Optional[int]:
+ # Where PidFile lands depends on how the httpd under test resolves
+ # a relative path against DefaultRuntimeDir, which has differed
+ # between versions; look in both places rather than assume.
+ for d in (self._server_logs_dir, self._server_dir):
+ try:
+ with open(os.path.join(d, name)) as fd:
+ return int(fd.read().strip())
+ except (OSError, ValueError):
+ continue
+ return None
+
def apache_access_log_clear(self):
if os.path.isfile(self._server_access_log):
os.remove(self._server_access_log)