diff --git a/changes-entries/systemd-watchdog.txt b/changes-entries/systemd-watchdog.txt new file mode 100644 index 00000000000..2fca28a6e4e --- /dev/null +++ b/changes-entries/systemd-watchdog.txt @@ -0,0 +1,2 @@ + *) mod_systemd: Support the systemd watchdog, sending the keep-alive + notification if the service unit sets WatchdogSec=. [Joe Orton] diff --git a/docs/manual/mod/mod_systemd.xml b/docs/manual/mod/mod_systemd.xml index 089d6296609..619ce8b5fa0 100644 --- a/docs/manual/mod/mod_systemd.xml +++ b/docs/manual/mod/mod_systemd.xml @@ -32,7 +32,7 @@

This module provides support for systemd integration. It allows httpd to be used in a service with the systemd - Type=notify (see Type=notify or Type=notify-reload (see systemd.service(5) for more information). The module is activated if loaded.

@@ -69,13 +69,90 @@ WantedBy=multi-user.target href="https://www.freedesktop.org/software/systemd/man/systemd.kill.html">systemd.kill(5) for more information.

-

This module does not provide support for Systemd socket activation.

+

A service manager from systemd 253 onwards offers + Type=notify-reload, which is worth using in preference. + Under Type=notify a systemctl reload + returns as soon as the ExecReload command has sent its + signal, which is before the new configuration has been read, and it + reports success whatever becomes of the restart afterwards. + Type=notify-reload instead holds the reload open until + the server reports it finished, so the command waits for the new + configuration to be in use and fails if it never is. mod_systemd + sends the RELOADING=1 notification the protocol expects + while the configuration is being read, stamped with the + MONOTONIC_USEC the service manager requires, and + READY=1 once it has been loaded.

+ + + Example of a service unit which reloads synchronously +
+[Service]
+Type=notify-reload
+ReloadSignal=SIGCONT
+ExecStart=/usr/local/apache2/bin/httpd -D FOREGROUND -k start
+ExecReload=/usr/local/apache2/bin/httpd -k graceful
+KillMode=mixed
+    
+
+ +

The service manager runs ExecReload first, and sends + the signal named by ReloadSignal only once that command + has exited successfully. Keeping ExecReload is what + makes the reload safe: httpd -k graceful parses the new + configuration in a process of its own and exits without signalling + anything if it does not parse, so the reload fails and the running + server carries on with the configuration it already has. Leaving + ExecReload out, and letting the service manager signal + the server directly, gives up that check: the running parent reads + the new configuration itself, and a configuration which does not + parse makes it exit, taking the server down.

+ +

The signal sent after ExecReload has run is then + redundant, so ReloadSignal should name one httpd does + not act on, such as SIGCONT. It matters that it is set: + the default is SIGHUP, which httpd takes as an + ungraceful restart, dropping the connections a reload is + meant to preserve. A unit which does leave out + ExecReload must set ReloadSignal=SIGUSR1, + the signal httpd restarts gracefully on.

+ +

Systemd socket activation is supported if httpd was built with + it. Each Listen port + must then be one passed in by systemd; a port which was not is a + fatal configuration error rather than one httpd opens for itself. + Socket activation is used only if this module is loaded, so it can + be built in and left unused.

ExtendedStatus is enabled by default if the module is loaded. If ExtendedStatus is not disabled in the configuration, run-time load and request statistics are made available in the systemctl status output.

+ +

The systemd watchdog is supported. If the service unit sets + WatchdogSec=, the parent process sends the keep-alive + notification which tells systemd the server is still alive; a server + which stops sending it is terminated and, with a suitable + Restart= setting, restarted. The notification is sent + while the configuration is being read and again once it is loaded, so + that a reload is covered, and periodically from the parent process + while the server runs.

+ +

That periodic notification is sent about every ten seconds, which + is how often the parent process runs the hook it is sent from. A + WatchdogSec= of less than twice that cannot be met, and + would have systemd terminating a server which is working normally; + such a setting is reported as a warning at startup. Use a + WatchdogSec= of at least 20 seconds.

+ + + Adding watchdog supervision to either unit above +
+[Service]
+WatchdogSec=30
+Restart=on-failure
+    
+
diff --git a/modules/arch/unix/mod_systemd.c b/modules/arch/unix/mod_systemd.c index 22482fd6bbc..6c03acc3a35 100644 --- a/modules/arch/unix/mod_systemd.c +++ b/modules/arch/unix/mod_systemd.c @@ -16,6 +16,7 @@ */ #include +#include #include #include "ap_mpm.h" #include "ap_listen.h" @@ -39,12 +40,65 @@ #include #endif +/* Microseconds on the clock systemd compares RELOADING=1 against, or + * zero if it cannot be read. */ +static apr_uint64_t monotonic_usec(void) +{ + struct timespec ts; + + if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0) { + return 0; + } + return (apr_uint64_t)ts.tv_sec * APR_USEC_PER_SEC + ts.tv_nsec / 1000; +} + +/* ap_run_monitor() is called once every INTERVAL_OF_WRITABLE_PROBES turns + * of the parent's one second loop in ap_wait_or_timeout(), so that is how + * often a keep-alive notification can be sent, and the shortest watchdog + * timeout which can be met is twice that: sd_watchdog_enabled(3) asks for + * a notification every half of the configured timeout. */ +#define WATCHDOG_INTERVAL_SEC (10) + +/* The WatchdogSec= of the service in microseconds, or zero if the service + * manager is not watching. Set in pre_config, before the first + * notification which could carry a keep-alive. */ +static apr_uint64_t watchdog_usec; + +/* A keep-alive assignment to paste into a notification, or nothing while + * the service manager is not asking for one. Sending WATCHDOG=1 when it + * is not expected is harmless, but saying so only when asked keeps what + * httpd reports the same as what the service was configured for. */ +static const char *watchdog_ping(void) +{ + return watchdog_usec ? "WATCHDOG=1\n" : ""; +} + static int systemd_pre_config(apr_pool_t *pconf, apr_pool_t *plog, apr_pool_t *ptemp) { - sd_notify(0, - "RELOADING=1\n" - "STATUS=Reading configuration...\n"); + apr_uint64_t usec = monotonic_usec(), wd_usec; + + /* Read afresh on each configuration load, since a restart unloads and + * loads the module again, and without unsetting it as server/listen.c + * does for $LISTEN_FDS, which would stop the keep-alive at the first + * reload. */ + watchdog_usec = sd_watchdog_enabled(0, &wd_usec) > 0 ? wd_usec : 0; + + /* A Type=notify-reload service ignores a reload notification which + * does not say when it was sent. */ + if (usec) { + sd_notifyf(0, + "RELOADING=1\n" + "MONOTONIC_USEC=%" APR_UINT64_T_FMT "\n" + "%s" + "STATUS=Reading configuration...\n", usec, watchdog_ping()); + } + else { + sd_notifyf(0, + "RELOADING=1\n" + "%s" + "STATUS=Reading configuration...\n", watchdog_ping()); + } ap_extended_status = 1; return OK; } @@ -63,6 +117,17 @@ static void log_selinux_context(void) } #endif +/* pconf is also cleared on a restart, where the service is not stopping + * at all, so distinguish the two by the state of the process. */ +static apr_status_t systemd_stopping(void *unused) +{ + if (ap_state_query(AP_SQ_MAIN_STATE) == AP_SQ_MS_EXITING) { + sd_notify(0, "STOPPING=1\n" + "STATUS=Shutting down.\n"); + } + return APR_SUCCESS; +} + /* Report the service is ready in post_config, which could be during * startup or after a reload. The server could still hit a fatal * startup error after this point during ap_run_mpm(), so this is @@ -80,8 +145,33 @@ static int systemd_post_config(apr_pool_t *pconf, apr_pool_t *plog, log_selinux_context(); #endif - sd_notify(0, "READY=1\n" - "STATUS=Configuration loaded.\n"); + /* Not reached by "httpd -k stop" and friends, which signal the + * running server and exit before post_config. */ + apr_pool_cleanup_register(pconf, NULL, systemd_stopping, + apr_pool_cleanup_null); + + /* A timeout the parent cannot meet would have the service manager + * killing a healthy server every WatchdogSec, so say so rather than + * leaving nothing in the log to explain it. */ + if (watchdog_usec + && watchdog_usec / 2 < (apr_uint64_t)WATCHDOG_INTERVAL_SEC + * APR_USEC_PER_SEC) { + ap_log_error(APLOG_MARK, APLOG_WARNING, 0, main_server, APLOGNO(10621) + "WatchdogSec is %" APR_UINT64_T_FMT "us, but keep-alive " + "notifications are sent from the parent process only " + "every %ds; configure a WatchdogSec of at least %ds or " + "the service will be killed while it is healthy", + watchdog_usec, WATCHDOG_INTERVAL_SEC, + 2 * WATCHDOG_INTERVAL_SEC); + } + + /* The keep-alive rides along with the notification which ends a + * reload: the configuration is read outside the parent's monitor loop, + * so nothing reports while it is being parsed, and the service manager + * keeps the timeout armed throughout. */ + sd_notifyf(0, "READY=1\n" + "%s" + "STATUS=Configuration loaded.\n", watchdog_ping()); return OK; } @@ -100,21 +190,32 @@ static int systemd_monitor(apr_pool_t *p, server_rec *s) apr_interval_time_t up_time; char bps[5]; + /* Before anything which might decline: reporting the server is alive + * does not depend on there being a status line to report with it. */ + if (watchdog_usec) { + sd_notify(0, "WATCHDOG=1\n"); + } + if (!ap_extended_status) { /* Nothing useful to report with ExtendedStatus disabled. */ return DECLINED; } ap_get_sload(&sload); - /* up_time in seconds */ - up_time = (apr_uint32_t) apr_time_sec(apr_time_now() - - ap_scoreboard_image->global->restart_time); + /* up_time in seconds, and never zero: a restart resets restart_time, + * so this hook can run in the same second it was set. */ + up_time = apr_time_sec(apr_time_now() - + ap_scoreboard_image->global->restart_time); + if (up_time < 1) { + up_time = 1; + } - apr_strfsize((unsigned long)((float) (sload.bytes_served) - / (float) up_time), bps); + apr_strfsize(sload.bytes_served / up_time, bps); + /* ap_get_sload() gives idle and busy as percentages of the workers + * available, not as counts. */ sd_notifyf(0, "READY=1\n" - "STATUS=Total requests: %lu; Idle/Busy workers %d/%d;" + "STATUS=Total requests: %lu; Idle/Busy workers %d%%/%d%%; " "Requests/sec: %.3g; Bytes served/sec: %sB/sec\n", sload.access_count, sload.idle, sload.busy, ((float) sload.access_count) / (float) up_time, bps); @@ -122,9 +223,31 @@ static int systemd_monitor(apr_pool_t *p, server_rec *s) return DECLINED; } +/* The number of sockets passed by the service manager has to be + * remembered: the configuration is read again on restart, by which time + * the environment sd_listen_fds() reads has been cleared, and the module + * itself has been unloaded and loaded again. Hence retained data rather + * than a static. */ +static const char *const retained_key = "mod_systemd_listen_fds"; + +static int ap_systemd_listen_fds(int unset_environment) +{ + int *fds = ap_retained_data_get(retained_key); + + if (fds == NULL) { + fds = ap_retained_data_create(retained_key, sizeof(*fds)); + *fds = sd_listen_fds(0); + } + if (unset_environment) { + /* Take the variables out of the environment, keeping the count. */ + sd_listen_fds(1); + } + return *fds; +} + static int ap_find_systemd_socket(process_rec * process, apr_port_t port) { - int fdcount, fd; - int sdc = sd_listen_fds(0); + int fd; + int sdc = ap_systemd_listen_fds(0); if (sdc < 0) { ap_log_perror(APLOG_MARK, APLOG_CRIT, sdc, process->pool, APLOGNO(02486) @@ -139,8 +262,7 @@ static int ap_find_systemd_socket(process_rec * process, apr_port_t port) { return -1; } - fdcount = atoi(getenv("LISTEN_FDS")); - for (fd = SD_LISTEN_FDS_START; fd < SD_LISTEN_FDS_START + fdcount; fd++) { + for (fd = SD_LISTEN_FDS_START; fd < SD_LISTEN_FDS_START + sdc; fd++) { if (sd_is_socket_inet(fd, 0, 0, -1, port) > 0) { return fd; } @@ -149,10 +271,6 @@ static int ap_find_systemd_socket(process_rec * process, apr_port_t port) { return -1; } -static int ap_systemd_listen_fds(int unset_environment){ - return sd_listen_fds(unset_environment); -} - static void systemd_register_hooks(apr_pool_t *p) { APR_REGISTER_OPTIONAL_FN(ap_systemd_listen_fds); diff --git a/test/modules/arch/__init__.py b/test/modules/arch/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/test/modules/arch/linux/README b/test/modules/arch/linux/README new file mode 100644 index 00000000000..4bdd7af91d1 --- /dev/null +++ b/test/modules/arch/linux/README @@ -0,0 +1,109 @@ +mod_systemd tests +================= + +What is different about this module +----------------------------------- +mod_systemd has no directives. Everything it does is driven by the +environment httpd was started with, and almost everything it produces goes +to the service manager rather than to a client: + + - sd_notify(3) datagrams sent to $NOTIFY_SOCKET at four points: reading + the configuration (pre_config), configuration loaded (post_config), + the MPM starting (pre_mpm, the only notification carrying MAINPID), + and a status line refreshed by the monitor hook. + - listening sockets inherited from the service manager, found through + $LISTEN_FDS. mod_systemd registers the two optional functions + server/listen.c calls, so loading the module is what enables socket + activation and not loading it is what disables it. + - ap_extended_status forced on in pre_config, so that the monitor hook + has request counts to report. + - the watchdog keep-alive, WATCHDOG=1, sent while the service manager + asks for one. Which it does through the environment as well: + WatchdogSec= in the unit becomes $WATCHDOG_USEC, and $WATCHDOG_PID + names the process expected to report. + +None of that is observable over HTTP, so the tests observe it directly. + +How the tests run without systemd, and without privileges +--------------------------------------------------------- +There is no need for a service manager to exercise the protocol. Only +test_005, the last test of test_006, and test_007 involve systemd at all; +the rest run anywhere, and need nothing from the systemd package beyond the +libsystemd httpd itself is linked against. + + test_001_notify.py $NOTIFY_SOCKET is an ordinary AF_UNIX datagram + socket the test binds itself (env.NotifyListener). + sd_notify writes to whatever that variable names, so + the test reads httpd's notifications straight off the + socket. HttpdTestEnv.set_httpd_env() puts the + variable into the environment apachectl passes on. + + test_002_monitor.py The periodic status line. ap_run_monitor() is + called once every ten turns of the parent's ~1s + loop, so these tests wait up to 25 seconds. + + test_003_extended_status.py + The ExtendedStatus side effect, observed through + mod_status. + + test_004_socket_activation.py + The test opens the listening socket itself and + hands it to httpd as descriptor 3 with LISTEN_FDS + and LISTEN_PID set, which is the whole protocol. + systemd-socket-activate would do the same, but its + --now option is newer than the systemd on some + distributions, and doing it directly needs no + systemd tooling at all. + + test_006_watchdog.py The keep-alive. Mostly the same stand-in socket, + with $WATCHDOG_USEC set in the environment httpd is + started with; env.ForegroundServer runs httpd in the + foreground where $WATCHDOG_PID can name it, which + apachectl cannot do for a parent it has not forked + yet. The last test uses a real unit with + WatchdogSec=, where systemd's own + WatchdogTimestampMonotonic is the evidence that the + pings arrived. + + The keep-alive is sent from the monitor hook, so it + is only as frequent as that hook is: about every ten + seconds. A WatchdogSec shorter than twice that + cannot be met, and mod_systemd says so (AH10621) + rather than letting a healthy server be killed. + + test_005_service.py The real thing: a transient Type=notify unit run + with "systemd-run --user". This is what checks that + systemd holds the unit in "activating" until READY=1 + arrives, tracks the right MainPID, and shows the + reported STATUS= as the unit's status text. + Skipped when the user has no systemd manager. + + test_007_notify_reload.py + The reload protocol, as a transient + Type=notify-reload unit: that "systemctl reload" + waits for the new configuration to be in use, that a + configuration which does not parse fails the reload + without disturbing the running server, and that a + reload restarts it gracefully and only once. Also + skipped where systemd is older than 253, which is + where Type=notify-reload arrived. + +The whole package is skipped unless mod_systemd was built, which needs +configure --enable-systemd. A static module is enough for everything +except the test which has to leave mod_systemd out of the configuration; +that one needs --enable-systemd=shared. + +Running the systemd suites where there is no user session +--------------------------------------------------------- +"systemctl --user" needs a per-user manager, which a login session has but +a CI container does not; enabling lingering needs privileges. Where that +is a problem, run the tests inside a container with systemd as pid 1, +which rootless podman supports on a cgroup v2 host: + + podman run --rm -it --systemd=always \ + -v $PWD:/src:z -w /src registry.fedoraproject.org/fedora:latest \ + /usr/sbin/init + +then, in another terminal, "podman exec" into it and run pytest as a +non-root user with XDG_RUNTIME_DIR set. The other four suites need none of +this and run anywhere. diff --git a/test/modules/arch/linux/__init__.py b/test/modules/arch/linux/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/test/modules/arch/linux/conftest.py b/test/modules/arch/linux/conftest.py new file mode 100644 index 00000000000..7b1a0d66cd7 --- /dev/null +++ b/test/modules/arch/linux/conftest.py @@ -0,0 +1,40 @@ +import logging +import os +import sys + +import pytest + +sys.path.append(os.path.join(os.path.dirname(__file__), '../../..')) + +from .env import SystemdTestEnv + + +def pytest_report_header(config, start_path): + env = SystemdTestEnv() + return f"mod_systemd [apache: {env.get_httpd_version()}, " \ + f"mpm: {env.mpm_module}, {env.prefix}]" + + +@pytest.fixture(scope="package") +def env(pytestconfig) -> SystemdTestEnv: + level = logging.INFO + console = logging.StreamHandler() + console.setLevel(level) + console.setFormatter(logging.Formatter('%(levelname)s: %(message)s')) + logging.getLogger('').addHandler(console) + logging.getLogger('').setLevel(level=level) + env = SystemdTestEnv(pytestconfig=pytestconfig) + if not env.has_systemd_module: + pytest.skip("mod_systemd is not built, configure with --enable-systemd") + env.setup_httpd() + env.apache_access_log_clear() + env.httpd_error_log.clear_log() + env.start_notify_listener() + yield env + env.stop_notify_listener() + + +@pytest.fixture(autouse=True, scope="package") +def _stop_package_scope(env): + yield + assert env.apache_stop() == 0 diff --git a/test/modules/arch/linux/env.py b/test/modules/arch/linux/env.py new file mode 100644 index 00000000000..ce1c2464f27 --- /dev/null +++ b/test/modules/arch/linux/env.py @@ -0,0 +1,638 @@ +import inspect +import logging +import os +import re +import shutil +import signal +import socket +import subprocess +import threading +import time +from typing import Callable, Dict, List, Optional + +from pyhttpd.env import HttpdTestEnv, HttpdTestSetup + +log = logging.getLogger(__name__) + + +class SystemdTestSetup(HttpdTestSetup): + + def __init__(self, env: 'HttpdTestEnv'): + super().__init__(env=env) + self.add_source_dir(os.path.dirname(inspect.getfile(SystemdTestSetup))) + # mod_systemd is only built with --enable-systemd, so it must not be + # a hard requirement; the tests skip when it is absent. + self.add_optional_modules(["systemd"]) + + +class SystemdTestEnv(HttpdTestEnv): + + def __init__(self, pytestconfig=None): + super().__init__(pytestconfig=pytestconfig) + self.add_httpd_log_modules(["core"]) + self._notify = None + + def setup_httpd(self, setup: HttpdTestSetup = None): + super().setup_httpd(setup=SystemdTestSetup(env=self)) + + @property + def systemd_is_dso(self) -> bool: + return os.path.isfile(os.path.join(self.libexec_dir, 'mod_systemd.so')) + + @property + def has_systemd_module(self) -> bool: + """Whether mod_systemd is available, however it was built: + --enable-systemd links it statically, --enable-systemd=shared + builds the DSO.""" + if self.systemd_is_dso: + return True + p = subprocess.run([self.httpd_bin, '-l'], capture_output=True, + text=True) + return re.search(r'^\s+mod_systemd\.c$', p.stdout, re.M) is not None + + @property + def notify(self) -> 'NotifyListener': + """The stand-in notification socket httpd reports to.""" + return self._notify + + def start_notify_listener(self) -> 'NotifyListener': + """Bind the notification socket and point httpd's $NOTIFY_SOCKET at it. + + This must happen before the server is first started, since libsystemd + reads $NOTIFY_SOCKET from the environment of the httpd process. + """ + assert self._notify is None + self._notify = NotifyListener( + os.path.join(self.server_dir, 'systemd-notify.sock')) + self.set_httpd_env('NOTIFY_SOCKET', self._notify.path) + return self._notify + + def stop_notify_listener(self): + if self._notify is not None: + self._notify.close() + self._notify = None + + def server_env(self) -> Dict[str, str]: + """The environment httpd is started with, as apachectl gets it.""" + return self._clean_path_env() + + @property + def httpd_bin(self) -> str: + return os.path.join(self.bin_dir, 'httpd') + + +class NotifyListener: + """A stand-in for the systemd notification socket. + + sd_notify(3) does nothing more than send a datagram to the AF_UNIX + socket named by $NOTIFY_SOCKET, so an unconnected datagram socket is + enough to observe everything mod_systemd reports, with no systemd + instance and no privileges involved. + + Datagrams are drained by a background thread so that the periodic + notifications from the monitor hook cannot fill the socket buffer + while a test is doing something else. + """ + + def __init__(self, path: str): + # sockaddr_un is limited to 108 bytes; well within reach for a + # source tree in a home directory, but check rather than fail + # obscurely inside bind(). + assert len(path) < 100, f"notification socket path too long: {path}" + if os.path.exists(path): + os.unlink(path) + self.path = path + self._sock = socket.socket(socket.AF_UNIX, socket.SOCK_DGRAM) + self._sock.bind(path) + self._sock.settimeout(0.1) + self._lock = threading.Lock() + self._messages: List[Dict[str, str]] = [] + self._stop = threading.Event() + self._thread = threading.Thread(target=self._drain, daemon=True) + self._thread.start() + + def _drain(self): + while not self._stop.is_set(): + try: + data = self._sock.recv(8192) + except socket.timeout: + continue + except OSError: + break + try: + msg = self.parse(data) + except Exception as ex: + # Never let one unreadable datagram stop the listener: the + # tests would then see silence rather than a failure. + log.warning(f"undecodable notification {data!r}: {ex}") + continue + log.debug(f"notify: {msg}") + with self._lock: + self._messages.append(msg) + + def close(self): + self._stop.set() + self._thread.join(timeout=2) + self._sock.close() + if os.path.exists(self.path): + os.unlink(self.path) + + @staticmethod + def parse(data: bytes) -> Dict[str, str]: + """Split one notification datagram into its NAME=VALUE assignments. + + The last line carries no trailing newline in the MAINPID + notification, and a value may itself contain '='. + """ + msg = {} + for line in data.decode(errors='replace').split('\n'): + name, sep, value = line.partition('=') + if sep: + msg[name] = value + return msg + + @property + def messages(self) -> List[Dict[str, str]]: + with self._lock: + return list(self._messages) + + def clear(self): + with self._lock: + self._messages.clear() + + def wait_for(self, match: Callable[[Dict[str, str]], bool], + timeout: float = 5.0) -> Optional[Dict[str, str]]: + """Return the first message satisfying `match`, waiting for it to + arrive if it has not already. Returns None on timeout.""" + end = time.time() + timeout + seen = 0 + while True: + with self._lock: + pending = self._messages[seen:] + seen = len(self._messages) + for msg in pending: + if match(msg): + return msg + if time.time() >= end: + return None + time.sleep(0.05) + + def wait_for_key(self, key: str, timeout: float = 5.0) \ + -> Optional[Dict[str, str]]: + return self.wait_for(lambda m: key in m, timeout=timeout) + + def wait_for_status(self, pattern: str, timeout: float = 5.0) \ + -> Optional[Dict[str, str]]: + rx = re.compile(pattern) + return self.wait_for(lambda m: 'STATUS' in m and rx.search(m['STATUS']), + timeout=timeout) + + def statuses(self) -> List[str]: + return [m['STATUS'] for m in self.messages if 'STATUS' in m] + + +# The STATUS= line the monitor hook reports, from systemd_monitor() in +# modules/arch/unix/mod_systemd.c. Idle and busy are the percentages +# ap_get_sload() computes, and are -1 when there are no workers at all. +MONITOR_STATUS = re.compile( + r'^Total requests: (?P\d+);\s*' + r'Idle/Busy workers (?P-?\d+)%/(?P-?\d+)%;\s*' + r'Requests/sec: (?P\S+);\s*' + r'Bytes served/sec: (?P.*)B/sec$') + +# ap_run_monitor() is called every INTERVAL_OF_WRITABLE_PROBES (10) turns of +# the ~1s parent loop in ap_wait_or_timeout(), so a status update is up to +# roughly 10 seconds away. Waiting for one costs that; waiting to be sure +# none is coming costs the whole timeout, so keep it to a small multiple. +MONITOR_TIMEOUT = 25.0 + +# Showing that no report is coming costs the whole wait, so it only has to +# comfortably outlast one turn of that cycle. +NO_MONITOR_TIMEOUT = 15.0 + + +# The keep-alive notification is sent from the same monitor hook, so waiting +# for one costs the same as waiting for a status report. +WATCHDOG_TIMEOUT = MONITOR_TIMEOUT + +# The shortest WatchdogSec mod_systemd will accept without complaining, which +# is twice the monitor interval: the recommended keep-alive period is half the +# watchdog timeout, and half of anything shorter than this is out of reach of a +# hook which runs every ten seconds. Keep in step with mod_systemd.c. +SYSTEMD_MONITOR_INTERVAL = 10 +MIN_WATCHDOG_SEC = 2 * SYSTEMD_MONITOR_INTERVAL + + +def is_watchdog_ping(msg: Dict[str, str]) -> bool: + """Whether a notification carries the keep-alive ping. + + mod_systemd is free to send WATCHDOG=1 in a datagram of its own or + alongside whatever else it is reporting, so match on the assignment + rather than on the message being only that. + """ + return msg.get('WATCHDOG') == '1' + + +# A configuration for a server run directly rather than through apachectl, +# sharing the server root, module list and error log with the rest of the +# suite but with its own pid file and port. +STANDALONE_CONF = """ +ServerRoot "${server_dir}" +DefaultRuntimeDir logs +PidFile "${pidfile}" +Include "conf/${modules_conf}" +ServerName standalone.test +ErrorLog "logs/error_log" +LogLevel ${loglevel} +DocumentRoot "${server_dir}/htdocs" + + Require all granted + + + SSLSessionCache "shmcb:ssl_gcache_data(32000)" + +${extra} +Listen ${port} +""" + + +def write_server_conf(env: SystemdTestEnv, name: str, port: int, + modules_conf: str = 'modules.conf', + extra: str = '') -> str: + """Write a standalone configuration and return its path.""" + path = os.path.join(env.server_conf_dir, f'{name}.conf') + with open(path, 'w') as fd: + fd.write(STANDALONE_CONF + .replace('${server_dir}', env.server_dir) + .replace('${pidfile}', + os.path.join(env.server_logs_dir, f'{name}.pid')) + .replace('${loglevel}', 'debug' if env.verbosity else 'warn') + .replace('${modules_conf}', modules_conf) + .replace('${extra}', extra) + .replace('${port}', str(port))) + return path + + +def http_responds(port: int, timeout: float = 2.0) -> bool: + """One HTTP request, without curl, so that a listening socket which + nothing is serving cannot be mistaken for a running server: under socket + activation the listener exists before httpd does.""" + try: + with socket.create_connection(('127.0.0.1', port), 1.0) as c: + c.settimeout(timeout) + c.sendall(b'GET / HTTP/1.0\r\nHost: standalone.test\r\n\r\n') + return c.recv(64).startswith(b'HTTP/1.') + except OSError: + return False + + +def error_log_size(env: SystemdTestEnv) -> int: + """Where the shared error log ends now, so that a later read can take + only what one operation wrote.""" + try: + return os.path.getsize(env.httpd_error_log.path) + except OSError: + return 0 + + +def error_log_since(env: SystemdTestEnv, pos: int) -> str: + """The error log written since error_log_size() returned pos.""" + try: + with open(env.httpd_error_log.path, errors='replace') as fd: + fd.seek(pos) + return fd.read() + except OSError: + return '' + + +class ActivatedServer: + """An httpd handed a listening socket the way a service manager does. + + The protocol is only $LISTEN_FDS descriptors starting at 3, and + $LISTEN_PID naming the process they were meant for, so the test opens + the socket and speaks it directly. systemd-socket-activate would do + the same, but its --now option is too recent to rely on, and this + needs no systemd tooling at all. + + apachectl cannot pass descriptors, so httpd is run directly, in the + foreground: LISTEN_PID has to be the process which calls + sd_listen_fds(), and a daemonised parent would not be it. + """ + + def __init__(self, env: SystemdTestEnv, port: int, name: str = 'activate', + extra: str = '', listen_port: int = None, + modules_conf: str = 'modules.conf'): + self.env = env + self.port = port + # The port systemd-socket-activate binds, which is the same as the + # configured one unless a test wants them to disagree. + self.listen_port = port if listen_port is None else listen_port + self.name = name + self.pid_file = os.path.join(env.server_logs_dir, f'{name}.pid') + self.proc = None + self.stdout = None + self.stderr = None + self.conf_file = write_server_conf(env, name, port, + modules_conf=modules_conf, + extra=extra) + + @staticmethod + def modules_conf_without(env: SystemdTestEnv, module: str) -> str: + """Write a copy of the generated module list with one module left + out, to check what happens when it is not loaded.""" + name = f'modules-no-{module}.conf' + src = os.path.join(env.server_conf_dir, 'modules.conf') + rx = re.compile(rf'^\s*LoadModule\s+{module}_module\b') + with open(src) as fd: + lines = [l for l in fd if not rx.match(l)] + with open(os.path.join(env.server_conf_dir, name), 'w') as fd: + fd.writelines(lines) + return name + + def args(self, fd: int) -> List[str]: + # The shell moves the inherited socket to descriptor 3 and names + # itself in LISTEN_PID before exec'ing httpd in its place, which is + # the one thing this cannot do from the parent: the pid has to be + # the one which will call sd_listen_fds(). bash rather than sh + # because dash parses only one digit in a redirection, and the + # socket lands well above descriptor 9. + return [ + 'bash', '-c', + f'exec 3<&{fd}; export LISTEN_FDS=1 LISTEN_FDNAMES=activate ' + 'LISTEN_PID=$$; exec "$0" "$@"', + self.env.httpd_bin, '-DFOREGROUND', + '-d', self.env.server_dir, '-f', self.conf_file, + ] + + def start(self) -> 'ActivatedServer': + lsock = socket.create_server(('', self.listen_port), + family=socket.AF_INET6, + dualstack_ipv6=True, backlog=128) + try: + # A new session so that the whole group can be signalled on the + # way out. + self.proc = subprocess.Popen( + self.args(lsock.fileno()), env=self.env.server_env(), + start_new_session=True, pass_fds=(lsock.fileno(),), + stdout=subprocess.PIPE, stderr=subprocess.PIPE) + finally: + # The child holds it now; keeping a copy here would leave the + # port bound after the server is gone. + lsock.close() + return self + + def _reap(self): + """Collect the output of a server which has exited.""" + if self.stderr is None and self.proc.poll() is not None: + self.stdout, self.stderr = self.proc.communicate() + + def wait_exit(self, timeout: float = 10.0) -> int: + """Wait for a server which is expected to fail to start.""" + self.stdout, self.stderr = self.proc.communicate(timeout=timeout) + return self.proc.returncode + + def is_live(self, timeout: float = 10.0) -> bool: + end = time.time() + timeout + while True: + if self.proc.poll() is not None: + # Collect its diagnostics, so that a test reporting the + # server did not come up can say why. + self._reap() + return False + if http_responds(self.port): + return True + if time.time() >= end: + return False + time.sleep(0.2) + + def is_running(self, settle: float = 2.0) -> bool: + """Whether the server is still up once it has had time to fail. + + A restart which cannot find its sockets takes a few milliseconds to + bring the parent down, and its old children go on serving after it, + so neither an immediate check nor a request proves anything. + """ + end = time.time() + settle + while time.time() < end: + if self.proc.poll() is not None: + return False + time.sleep(0.1) + return True + + def reload(self) -> int: + """Ask the running server to restart gracefully.""" + r = self.env.run([self.env.httpd_bin, '-d', self.env.server_dir, + '-f', self.conf_file, '-k', 'graceful'], + env=self.env.server_env()) + return r.exit_code + + def _signal_group(self, sig: int) -> bool: + """Signal every process still in the server's group, reporting + whether any remained. start_new_session() made the process we + launched the group leader, so its pid is the group id whether or + not it is still alive.""" + try: + os.killpg(self.proc.pid, sig) + return True + except OSError: + return False + + def stop(self): + if self.proc is None: + return + self._signal_group(signal.SIGTERM) + if self.proc.poll() is None: + try: + self.stdout, self.stderr = self.proc.communicate(timeout=10) + except subprocess.TimeoutExpired: + self._signal_group(signal.SIGKILL) + self.stdout, self.stderr = self.proc.communicate() + # A parent which died during a failed restart leaves its children + # behind, still holding the listening socket and still answering. + # They have to go too, or the next test finds the port taken. + end = time.time() + 5 + while self._signal_group(0): + if time.time() >= end: + self._signal_group(signal.SIGKILL) + break + time.sleep(0.1) + + def __enter__(self) -> 'ActivatedServer': + return self.start() + + def __exit__(self, *args): + self.stop() + + +# Type=notify-reload and ReloadSignal= both arrived in systemd 253. An +# unrecognised Type= does not degrade to anything, it stops the unit loading +# at all, so there is nothing to fall back to and the tests are skipped. +NOTIFY_RELOAD_VERSION = 253 + + +def systemd_version() -> int: + """The version of the systemd on this host, or 0 if it cannot be asked. + "systemctl --version" opens with "systemd 259 (259.8-1.fc44)".""" + try: + p = subprocess.run(['systemctl', '--version'], capture_output=True, + text=True, timeout=15) + except (OSError, subprocess.TimeoutExpired): + return 0 + m = re.match(r'systemd (\d+)', p.stdout) + return int(m.group(1)) if m else 0 + + +class TransientService: + """httpd run as a real transient systemd unit, with systemd-run. + + This is the only harness here which exercises the notification protocol + against systemd itself rather than a stand-in socket: systemd provides + NOTIFY_SOCKET, holds the service in "activating" until READY=1 arrives, + tracks MAINPID, and shows the reported STATUS= as the unit's status + text. It needs a per-user service manager, which a login session has + but a bare CI container does not. + + service_type selects what is exercised: "notify" for startup and + shutdown, "notify-reload" for the reload protocol on top of them. + """ + + def __init__(self, env: SystemdTestEnv, port: int, + name: str = None, extra: str = '', + service_type: str = 'notify', exec_reload: bool = True, + properties: List[str] = None): + self.env = env + self.port = port + self.unit = name or f'httpd-test-{os.getpid()}' + self.service_type = service_type + self.exec_reload = exec_reload + self.conf_file = write_server_conf(env, 'transient', port, extra=extra) + self.pid_file = os.path.join(env.server_logs_dir, 'transient.pid') + # Extra --property arguments for the unit, such as WatchdogSec= or + # ReloadSignal=. + self.properties = list(properties or []) + + def read_pid(self) -> Optional[int]: + try: + with open(self.pid_file) as fd: + return int(fd.read().strip()) + except (OSError, ValueError): + return None + + @staticmethod + def is_available() -> bool: + """Whether this user has a systemd manager to run services under.""" + if shutil.which('systemd-run') is None: + return False + if not os.environ.get('XDG_RUNTIME_DIR'): + return False + # Talking to the manager at all is the test: without a login + # session, or with lingering off, there is none to talk to. + try: + return subprocess.run(['systemctl', '--user', 'show', '-p', + 'Version'], capture_output=True, + timeout=15).returncode == 0 + except (OSError, subprocess.TimeoutExpired): + return False + + def systemctl(self, *args, + timeout: float = 60.0) -> subprocess.CompletedProcess: + return subprocess.run(['systemctl', '--user', *args], + capture_output=True, text=True, timeout=timeout) + + def show(self, prop: str) -> str: + r = self.systemctl('show', '-p', prop, '--value', f'{self.unit}.service') + return r.stdout.strip() + + def start(self, timeout: float = 20.0) -> subprocess.CompletedProcess: + httpd = self.env.httpd_bin + props = [ + f'--service-type={self.service_type}', + '--property=KillMode=mixed', + # A reload which is never reported finished holds the job open + # until this elapses, so keep it to the same bound as the start. + f'--property=TimeoutStartSec={int(timeout)}', + ] + if self.exec_reload: + props.append(f'--property=ExecReload={httpd} ' + f'-d {self.env.server_dir} ' + f'-f {self.conf_file} -k graceful') + props += [f'--property={p}' for p in self.properties] + r = subprocess.run([ + 'systemd-run', '--user', '--collect', '--quiet', + '--unit', self.unit, *props, + httpd, '-DFOREGROUND', + '-d', self.env.server_dir, '-f', self.conf_file, + ], capture_output=True, text=True, timeout=timeout) + return r + + def reload(self, timeout: float = 60.0) -> subprocess.CompletedProcess: + """Ask the manager to reload the unit. Under Type=notify-reload + this returns once the server has reported the reload finished; + under Type=notify, as soon as ExecReload= has exited.""" + return self.systemctl('reload', f'{self.unit}.service', + timeout=timeout) + + def rewrite_conf(self, extra: str = ''): + """Replace the configuration the unit reads on its next reload. + The path does not change, so ExecStart= and ExecReload= still name + it.""" + self.conf_file = write_server_conf(self.env, 'transient', self.port, + extra=extra) + + def wait_active(self, timeout: float = 20.0) -> bool: + end = time.time() + timeout + while time.time() < end: + if self.show('ActiveState') == 'active': + return True + time.sleep(0.2) + return False + + def stop(self): + self.systemctl('stop', f'{self.unit}.service') + self.systemctl('reset-failed', f'{self.unit}.service') + + def __enter__(self) -> 'TransientService': + return self + + def __exit__(self, *args): + self.stop() + + +class ForegroundServer(ActivatedServer): + """httpd run directly in the foreground, with extra environment. + + The watchdog protocol is keyed to a process id: sd_watchdog_enabled(3) + ignores $WATCHDOG_USEC unless $WATCHDOG_PID is unset or names the + process reading it. apachectl cannot be used to set that, since the + variable has to name the parent httpd and the pid is not known until it + exists, so the value is assigned in a shell which then exec's httpd in + its own place -- the same trick ActivatedServer uses for $LISTEN_PID. + + Assignments are shell words, so "$$" in a value is the pid httpd will + have. + """ + + def __init__(self, env: SystemdTestEnv, port: int, + name: str = 'foreground', extra: str = '', + setenv: Dict[str, str] = None, + modules_conf: str = 'modules.conf'): + super().__init__(env, port, name=name, extra=extra, + modules_conf=modules_conf) + self.setenv = dict(setenv or {}) + + def args(self, fd: int = None) -> List[str]: + assigns = ' '.join(f'{k}={v}' for k, v in self.setenv.items()) + return [ + 'bash', '-c', + f'export {assigns}; exec "$0" "$@"' if assigns else 'exec "$0" "$@"', + self.env.httpd_bin, '-DFOREGROUND', + '-d', self.env.server_dir, '-f', self.conf_file, + ] + + def start(self) -> 'ForegroundServer': + # A new session, so the whole group can be signalled on the way out + # even when a failed restart leaves children behind. + self.proc = subprocess.Popen( + self.args(), env=self.env.server_env(), start_new_session=True, + stdout=subprocess.PIPE, stderr=subprocess.PIPE) + return self diff --git a/test/modules/arch/linux/test_001_notify.py b/test/modules/arch/linux/test_001_notify.py new file mode 100644 index 00000000000..d79980bc0ac --- /dev/null +++ b/test/modules/arch/linux/test_001_notify.py @@ -0,0 +1,152 @@ +import time + +import pytest + +from pyhttpd.conf import HttpdConf + + +class TestSystemdNotify: + """The service notifications mod_systemd sends over $NOTIFY_SOCKET + across the server lifecycle.""" + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + conf = HttpdConf(env) + conf.add_vhost_test1() + conf.install() + # Each test drives startup itself, so leave the server down. + assert env.apache_stop() == 0 + + @staticmethod + def index_of(messages, match): + for i, msg in enumerate(messages): + if match(msg): + return i + return -1 + + def test_systemd_001_01_startup(self, env): + """Startup reports configuration reading, then readiness, then that + the MPM is serving.""" + env.notify.clear() + assert env.apache_restart() == 0 + assert env.notify.wait_for_status(r'^Reading configuration\.\.\.$'), \ + f"no reload notification, got {env.notify.statuses()}" + assert env.notify.wait_for_status(r'^Configuration loaded\.$'), \ + f"no ready notification, got {env.notify.statuses()}" + assert env.notify.wait_for_status(r'^Processing requests\.\.\.$'), \ + f"no pre_mpm notification, got {env.notify.statuses()}" + + def test_systemd_001_02_reloading_before_ready(self, env): + """RELOADING=1 is sent while the configuration is read, and READY=1 + only once it has been.""" + env.notify.clear() + assert env.apache_restart() == 0 + assert env.notify.wait_for_status(r'^Configuration loaded\.$') + msgs = env.notify.messages + reloading = self.index_of( + msgs, lambda m: m.get('RELOADING') == '1' + and m.get('STATUS') == 'Reading configuration...') + ready = self.index_of( + msgs, lambda m: m.get('READY') == '1' + and m.get('STATUS') == 'Configuration loaded.') + assert reloading >= 0 and ready >= 0 + assert reloading < ready, \ + "READY=1 was reported before the configuration was read" + + def test_systemd_001_03_mainpid(self, env): + """MAINPID is the pid of the parent process, the one httpd records + in its pid file.""" + env.notify.clear() + assert env.apache_restart() == 0 + msg = env.notify.wait_for_key('MAINPID') + assert msg, f"no MAINPID reported, got {env.notify.messages}" + assert msg.get('READY') == '1' + assert msg.get('STATUS') == 'Processing requests...' + assert int(msg['MAINPID']) == env.read_pid_file() + + def test_systemd_001_04_reload(self, env): + """A graceful restart reports reading the configuration and then + being ready again.""" + assert env.apache_restart() == 0 + pid = env.read_pid_file() + env.notify.clear() + assert env.apache_reload() == 0 + assert env.notify.wait_for_status(r'^Reading configuration\.\.\.$'), \ + f"no reload notification, got {env.notify.statuses()}" + assert env.notify.wait_for_status(r'^Configuration loaded\.$'), \ + f"no ready notification, got {env.notify.statuses()}" + # The parent survives a graceful restart, and the MPM is not started + # over, so the pid systemd tracks neither changes nor is re-reported. + assert env.read_pid_file() == pid + for msg in env.notify.messages: + assert 'MAINPID' not in msg or int(msg['MAINPID']) == pid + + def test_systemd_001_05_hard_restart(self, env): + """An ungraceful restart starts the MPM over, and re-reports the + main pid, which is still that of the surviving parent.""" + assert env.apache_restart() == 0 + pid = env.read_pid_file() + env.notify.clear() + assert env.apache_hard_restart() == 0 + assert env.notify.wait_for_status(r'^Configuration loaded\.$'), \ + f"no ready notification, got {env.notify.statuses()}" + msg = env.notify.wait_for_key('MAINPID') + assert msg, f"no MAINPID after restart, got {env.notify.messages}" + assert int(msg['MAINPID']) == pid + assert env.read_pid_file() == pid + + def test_systemd_001_06_notify_socket_kept(self, env): + """mod_systemd leaves NOTIFY_SOCKET in the environment, so a second + start after a stop is reported just like the first.""" + assert env.apache_restart() == 0 + assert env.apache_stop() == 0 + env.notify.clear() + assert env.apache_restart() == 0 + assert env.notify.wait_for_status(r'^Configuration loaded\.$') + + def test_systemd_001_07_stopping(self, env): + """Shutdown is announced before the process goes away.""" + assert env.apache_restart() == 0 + env.notify.clear() + assert env.apache_stop() == 0 + # The server is already gone, so anything it sent has arrived. + msg = env.notify.wait_for_key('STOPPING', timeout=1) + assert msg, f"no STOPPING=1 on shutdown, got {env.notify.messages}" + assert msg['STOPPING'] == '1' + assert msg.get('STATUS') == 'Shutting down.' + + def test_systemd_001_08_reloading_monotonic(self, env): + assert env.apache_restart() == 0 + env.notify.clear() + assert env.apache_reload() == 0 + msg = env.notify.wait_for(lambda m: m.get('RELOADING') == '1') + assert msg, "no RELOADING=1 on graceful restart" + assert 'MONOTONIC_USEC' in msg, \ + f"RELOADING=1 sent without MONOTONIC_USEC: {msg}" + # systemd compares this against its own reading of the same clock. + assert 0 < int(msg['MONOTONIC_USEC']) <= time.clock_gettime_ns( + time.CLOCK_MONOTONIC) // 1000 + + def test_systemd_001_09_watchdog(self, env): + """A server started with $WATCHDOG_USEC reports that it is alive. + + $WATCHDOG_PID is left unset, which sd_watchdog_enabled(3) takes to + mean any process may report: it has to be, since apachectl starts a + parent whose pid nothing knew in advance. The keep-alive arrives + with the startup notifications rather than only from the monitor + hook ten seconds later, which is what makes this cheap to check. + test_006_watchdog.py covers the rest of the protocol. + """ + assert env.apache_stop() == 0 + # Long enough that mod_systemd does not object to it (AH10622). + env.set_httpd_env('WATCHDOG_USEC', str(60 * 1000000)) + try: + env.notify.clear() + assert env.apache_restart() == 0 + msg = env.notify.wait_for_key('WATCHDOG', timeout=4) + assert msg, f"no watchdog keepalive was sent, " \ + f"got {env.notify.messages}" + assert msg['WATCHDOG'] == '1' + finally: + env.set_httpd_env('WATCHDOG_USEC', None) + assert env.apache_restart() == 0 diff --git a/test/modules/arch/linux/test_002_monitor.py b/test/modules/arch/linux/test_002_monitor.py new file mode 100644 index 00000000000..368d83ae74f --- /dev/null +++ b/test/modules/arch/linux/test_002_monitor.py @@ -0,0 +1,107 @@ +import re + +import pytest + +from pyhttpd.conf import HttpdConf + +from .env import MONITOR_STATUS, MONITOR_TIMEOUT + + +class TestSystemdMonitor: + """The periodic STATUS= line the monitor hook reports, which is what + "systemctl status httpd" shows below the unit description.""" + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + conf = HttpdConf(env, extras={ + 'base': """ + + SetHandler server-status + + """ + }) + conf.add_vhost_test1() + conf.install() + assert env.apache_restart() == 0 + + @staticmethod + def wait_report(env, above=-1): + """Wait for a report from the monitor hook accounting for more than + `above` requests, and return its fields.""" + msg = env.notify.wait_for( + lambda m: 'STATUS' in m + and (g := MONITOR_STATUS.match(m['STATUS'])) is not None + and int(g.group('requests')) > above, + timeout=MONITOR_TIMEOUT) + assert msg, f"no monitor report within {MONITOR_TIMEOUT}s, " \ + f"server {'up' if env.is_live() else 'down'}, " \ + f"got {env.notify.messages}" + return MONITOR_STATUS.match(msg['STATUS']) + + @pytest.fixture(scope='class') + def reports(self, env, _class_scope): + """Two consecutive reports with requests served between them, and + the mod_status report as of the second. + + The hook runs once every ten turns of the parent's one second + loop, so waiting for a report costs up to ten seconds. The tests + below share one pair rather than each waiting for its own. + """ + env.notify.clear() + first = self.wait_report(env) + for _ in range(10): + r = env.curl_get(env.mkurl("http", "test1", "/")) + assert r.response['status'] == 200 + second = self.wait_report(env, above=int(first.group('requests'))) + r = env.curl_get(env.mkurl("http", "test1", "/server-status?auto")) + assert r.response['status'] == 200 + auto = {} + for line in r.response['body'].decode().splitlines(): + key, sep, value = line.partition(':') + if sep and key not in auto: + auto[key] = value.strip() + return {'first': first, 'second': second, 'auto': auto} + + def test_systemd_002_01_report_format(self, reports): + m = reports['first'] + assert int(m.group('requests')) >= 0 + assert m.group('bps') + + def test_systemd_002_02_requests_counted(self, reports): + """The request count reported to systemd tracks requests served.""" + before = int(reports['first'].group('requests')) + after = int(reports['second'].group('requests')) + assert after >= before + 10, \ + f"{after - before} requests reported, at least 10 were served" + + def test_systemd_002_03_rates_are_finite(self, reports): + """Neither rate is inf or nan. + + systemd_monitor() divides by an uptime in whole seconds, which is + zero for a report arriving in the first second after the scoreboard + records a restart. + """ + for which in ('first', 'second'): + m = reports[which] + rate = m.group('rate') + assert re.match(r'^-?\d', rate), f"{which} request rate is {rate!r}" + assert float(rate) >= 0 + assert 'inf' not in m.group('bps') and 'nan' not in m.group('bps'), \ + f"{which} byte rate is {m.group('bps')!r}" + + def test_systemd_002_04_report_repeats(self, env, reports): + """The status line keeps being refreshed while the server runs.""" + assert reports['first'].group(0) != reports['second'].group(0) + + def test_systemd_002_05_worker_percentages(self, reports): + """Idle and busy are percentages of the workers available, and are + reported as such.""" + m, auto = reports['second'], reports['auto'] + idle, busy = int(m.group('idle')), int(m.group('busy')) + assert 0 <= idle <= 100 and 0 <= busy <= 100 + # Integer division loses at most one point between the two. + assert 99 <= idle + busy <= 100 + # Busy workers are a minority of a mostly idle test server, which + # is what mod_status reports over the same scoreboard. + assert int(auto['IdleWorkers']) > int(auto['BusyWorkers']) + assert idle > busy diff --git a/test/modules/arch/linux/test_003_extended_status.py b/test/modules/arch/linux/test_003_extended_status.py new file mode 100644 index 00000000000..badc1483944 --- /dev/null +++ b/test/modules/arch/linux/test_003_extended_status.py @@ -0,0 +1,66 @@ +import pytest + +from pyhttpd.conf import HttpdConf + +from .env import NO_MONITOR_TIMEOUT + + +class TestSystemdExtendedStatus: + """mod_systemd turns ExtendedStatus on so that it has request counts to + report, which changes what the rest of the server records too.""" + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + yield + conf = HttpdConf(env) + conf.add_vhost_test1() + conf.install() + assert env.apache_restart() == 0 + + def auto_report(self, env): + r = env.curl_get(env.mkurl("http", "test1", "/server-status?auto")) + assert r.response['status'] == 200 + return r.response['body'].decode() + + def install(self, env, extra=''): + conf = HttpdConf(env, extras={ + 'base': f""" + {extra} + + SetHandler server-status + + """ + }) + conf.add_vhost_test1() + conf.install() + assert env.apache_restart() == 0 + + def test_systemd_003_01_enabled_by_default(self, env): + """Loading mod_systemd is enough to get extended status; no + ExtendedStatus directive is present in this configuration.""" + self.install(env) + body = self.auto_report(env) + assert 'Total Accesses:' in body, \ + "ExtendedStatus was not enabled by mod_systemd" + assert 'Total kBytes:' in body + + def test_systemd_003_02_directive_wins(self, env): + """An explicit ExtendedStatus off still takes effect: mod_systemd + sets the default in pre_config, before the configuration is read.""" + self.install(env, extra='ExtendedStatus off') + body = self.auto_report(env) + assert 'Total Accesses:' not in body, \ + "ExtendedStatus off was overridden by mod_systemd" + + def test_systemd_003_03_no_report_without_extended_status(self, env): + """With extended status off the monitor hook declines, so no status + line is reported to systemd.""" + env.notify.clear() + self.install(env, extra='ExtendedStatus off') + # Startup notifications are still sent... + assert env.notify.wait_for_status(r'^Processing requests\.\.\.$', + timeout=10) is not None + # ...but the periodic status line is not. + assert env.notify.wait_for_status(r'^Total requests: ', + timeout=NO_MONITOR_TIMEOUT) is None, \ + "a monitor report was sent with ExtendedStatus off" diff --git a/test/modules/arch/linux/test_004_socket_activation.py b/test/modules/arch/linux/test_004_socket_activation.py new file mode 100644 index 00000000000..b92fadd36c7 --- /dev/null +++ b/test/modules/arch/linux/test_004_socket_activation.py @@ -0,0 +1,89 @@ +import pytest + +from .env import ActivatedServer + + +class TestSystemdSocketActivation: + """Listening sockets passed in by the service manager. + + mod_systemd exports the two optional functions server/listen.c uses to + find them, so socket activation is enabled by loading the module and + disabled by not loading it, whatever the environment says. + + apachectl cannot pass file descriptors, so these tests open the + listening socket themselves and run httpd directly, in the foreground, + with the descriptor and the environment a service manager would give + it. No systemd process is involved. + """ + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + # The activated servers use the same server root; keep the one + # started by other tests out of the way. + assert env.apache_stop() == 0 + yield + assert env.apache_stop() == 0 + + def test_systemd_004_01_activated_listener(self, env): + """httpd serves on a socket it never opened itself.""" + env.notify.clear() + with ActivatedServer(env, port=env.http_port2) as server: + assert server.is_live(), \ + f"server did not come up: {server.stderr}" + r = env.curl_get(f"http://{env.http_addr}:{env.http_port2}/") + assert r.response['status'] == 200 + # The notification handshake works the same way as when httpd opens + # its own sockets. + assert env.notify.wait_for_status(r'^Configuration loaded\.$', timeout=0) + msg = env.notify.wait_for_key('MAINPID', timeout=0) + assert msg, f"httpd sent no MAINPID, got {env.notify.messages}" + assert msg['STATUS'] == 'Processing requests...' + + def test_systemd_004_02_no_socket_for_port(self, env): + """A Listen port the service manager did not pass is an error, not + a port httpd quietly opens for itself.""" + server = ActivatedServer(env, port=env.http_port2, + listen_port=env.proxy_port, + name='activate-wrongport') + with server: + assert server.wait_exit() != 0, "httpd started without a socket" + assert b'not configured in systemd' in server.stderr, \ + f"unexpected startup diagnostic: {server.stderr}" + + def test_systemd_004_03_disabled_without_module(self, env): + """Without mod_systemd the passed sockets are ignored and httpd + opens the configured port itself, the same arrangement that fails + in the test above.""" + if not env.systemd_is_dso: + pytest.skip("mod_systemd is linked statically and cannot be " + "left out of the configuration") + modules_conf = ActivatedServer.modules_conf_without(env, 'systemd') + server = ActivatedServer(env, port=env.http_port2, + listen_port=env.proxy_port, + name='activate-nomodule', + modules_conf=modules_conf) + with server: + assert server.is_live(), \ + f"server did not come up: {server.stderr}" + r = env.curl_get(f"http://{env.http_addr}:{env.http_port2}/") + assert r.response['status'] == 200 + + def test_systemd_004_04_graceful_restart(self, env): + """An activated server survives a graceful restart, which means + finding the passed sockets again after the environment naming them + has been cleared.""" + with ActivatedServer(env, port=env.http_port2, + name='activate-reload') as server: + assert server.is_live(), \ + f"server did not come up: {server.stderr}" + server.reload() + env.httpd_error_log.ignore_recent(lognos=['AH02487']) + # Check the parent first: when it dies here its children carry + # on holding the listening socket and answering, so a request + # succeeding proves nothing on its own. + assert server.is_running(), \ + "the parent exited on graceful restart" + assert server.is_live(timeout=5), \ + "the server did not survive a graceful restart" + r = env.curl_get(f"http://{env.http_addr}:{env.http_port2}/") + assert r.response['status'] == 200 diff --git a/test/modules/arch/linux/test_005_service.py b/test/modules/arch/linux/test_005_service.py new file mode 100644 index 00000000000..ef220426302 --- /dev/null +++ b/test/modules/arch/linux/test_005_service.py @@ -0,0 +1,89 @@ +import time + +import pytest + +from .env import MONITOR_STATUS, MONITOR_TIMEOUT, TransientService, http_responds + +pytestmark = pytest.mark.skipif( + not TransientService.is_available(), + reason="no per-user systemd manager to run a transient service under") + + +class TestSystemdService: + """httpd as a real systemd service, end to end. + + Everything else here checks what mod_systemd sends. These check what + systemd does with it: hold the unit in "activating" until httpd is + ready, track the right process, and show the reported status text. + """ + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + assert env.apache_stop() == 0 + yield + assert env.apache_stop() == 0 + + @pytest.fixture + def service(self, env) -> TransientService: + svc = TransientService(env, port=env.http_port2) + yield svc + svc.stop() + + def test_systemd_005_01_type_notify(self, env, service): + """systemd-run returns once the unit is active, which for a + Type=notify service means READY=1 has been received, which + mod_systemd sends only after the configuration is loaded.""" + r = service.start() + assert r.returncode == 0, f"systemd-run failed: {r.stderr}" + assert service.show('ActiveState') == 'active' + assert http_responds(env.http_port2), \ + "the unit was active before the server would answer" + + def test_systemd_005_02_main_pid(self, env, service): + """The process systemd tracks is the httpd parent.""" + assert service.start().returncode == 0 + assert service.wait_active() + main_pid = int(service.show('MainPID')) + assert main_pid > 0 + assert main_pid == service.read_pid() + + def test_systemd_005_03_status_text(self, env, service): + """The status line "systemctl status" shows comes from the monitor + hook, and is refreshed while the server runs.""" + assert service.start().returncode == 0 + assert service.wait_active() + # First the post_config report, then the periodic one. + assert service.show('StatusText') in ('Configuration loaded.', + 'Processing requests...') + end = time.time() + MONITOR_TIMEOUT + text = None + while time.time() < end: + text = service.show('StatusText') + if MONITOR_STATUS.match(text): + break + time.sleep(0.5) + assert MONITOR_STATUS.match(text), \ + f"status text was never refreshed by the monitor hook: {text!r}" + + def test_systemd_005_04_reload(self, env, service): + """systemctl reload runs httpd -k graceful and the unit stays + active throughout.""" + assert service.start().returncode == 0 + assert service.wait_active() + pid = int(service.show('MainPID')) + r = service.systemctl('reload', f'{service.unit}.service') + assert r.returncode == 0, f"reload failed: {r.stderr}" + assert service.show('ActiveState') == 'active' + assert int(service.show('MainPID')) == pid + assert http_responds(env.http_port2) + + def test_systemd_005_05_stop(self, env, service): + """The unit stops cleanly, without systemd having to time out and + kill it.""" + assert service.start().returncode == 0 + assert service.wait_active() + r = service.systemctl('stop', f'{service.unit}.service') + assert r.returncode == 0, f"stop failed: {r.stderr}" + assert service.show('ActiveState') == 'inactive' + assert service.show('Result') == 'success' + assert not http_responds(env.http_port2) diff --git a/test/modules/arch/linux/test_006_watchdog.py b/test/modules/arch/linux/test_006_watchdog.py new file mode 100644 index 00000000000..d4fa8653bf6 --- /dev/null +++ b/test/modules/arch/linux/test_006_watchdog.py @@ -0,0 +1,211 @@ +import os +import re +import time + +import pytest + +from .env import (ForegroundServer, TransientService, MIN_WATCHDOG_SEC, + NO_MONITOR_TIMEOUT, WATCHDOG_TIMEOUT, is_watchdog_ping) + + +def watchdog_env(usec: int, pid: str = '$$') -> dict: + """The environment a service manager sets for a watched service. + + $WATCHDOG_PID names the process expected to report; "$$" is the shell + which exec's httpd, so it is the pid httpd will run as. + """ + env = {'WATCHDOG_USEC': str(usec)} + if pid is not None: + env['WATCHDOG_PID'] = pid + return env + + +class TestSystemdWatchdog: + """The keep-alive ping of the systemd watchdog protocol. + + A service whose unit sets WatchdogSec= is started with $WATCHDOG_USEC + holding that timeout in microseconds, and $WATCHDOG_PID holding the pid + expected to report. The service must then send "WATCHDOG=1" to the + notification socket more often than the timeout, or systemd puts the + unit into a failed state with Result=watchdog. The timeout is armed + once start-up completes and, as test_006_06 relies on, stays armed + across a reload. + + Nothing here needs systemd: the ping goes to $NOTIFY_SOCKET like every + other notification, so the same stand-in socket observes it. Only the + last test involves a service manager. + """ + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + # These run their own servers on the second port; the shared one + # would only add its notifications to the same socket. + assert env.apache_stop() == 0 + yield + assert env.apache_stop() == 0 + + @pytest.fixture + def server_factory(self, env): + started = [] + + def make(usec=None, pid='$$', extra='', name='watchdog'): + setenv = watchdog_env(usec, pid) if usec is not None else {} + srv = ForegroundServer(env, port=env.http_port2, name=name, + extra=extra, setenv=setenv) + started.append(srv) + env.notify.clear() + srv.start() + assert srv.is_live(), \ + f"server did not come up: {srv.stderr!r}" + return srv + + yield make + for srv in started: + srv.stop() + + def test_systemd_006_01_no_ping_when_unwatched(self, env, server_factory): + """A server the service manager is not watching sends no keep-alive. + + The monitor hook still runs -- its status report is the evidence of + that -- so the absence of a ping is a decision and not silence. + """ + server_factory() + assert env.notify.wait_for_status(r'^Total requests: ', + timeout=WATCHDOG_TIMEOUT), \ + "the monitor hook never ran, so this proves nothing" + assert not [m for m in env.notify.messages if is_watchdog_ping(m)], \ + "a keep-alive was sent with $WATCHDOG_USEC unset" + + def test_systemd_006_02_ping_when_watched(self, env, server_factory): + """With the watchdog enabled the keep-alive is sent.""" + server_factory(usec=60 * 1000000) + assert env.notify.wait_for(is_watchdog_ping, + timeout=WATCHDOG_TIMEOUT), \ + f"no keep-alive within {WATCHDOG_TIMEOUT}s, " \ + f"got {env.notify.messages}" + + def test_systemd_006_03_ping_repeats(self, env, server_factory): + """The keep-alive is periodic, which is the whole point of it: one + ping would satisfy a test but not systemd.""" + server_factory(usec=60 * 1000000) + for n in range(2): + assert env.notify.wait_for(is_watchdog_ping, + timeout=WATCHDOG_TIMEOUT), \ + f"only {n} keep-alives arrived, got {env.notify.messages}" + env.notify.clear() + + def test_systemd_006_04_ping_without_extended_status(self, env, + server_factory): + """The keep-alive does not depend on ExtendedStatus. + + The monitor hook declines early when it has no request counts to + report (test_003_03), and a server whose status line is switched + off must still be reported as alive. + """ + server_factory(usec=60 * 1000000, extra='ExtendedStatus off') + assert env.notify.wait_for_status(r'^Total requests: ', + timeout=NO_MONITOR_TIMEOUT) is None, \ + "ExtendedStatus off did not stop the status report" + assert env.notify.wait_for(is_watchdog_ping, + timeout=WATCHDOG_TIMEOUT), \ + "no keep-alive with ExtendedStatus off" + + def test_systemd_006_05_no_ping_for_another_pid(self, env, server_factory): + """$WATCHDOG_PID naming a different process means the variables were + set for something further up the process tree, and must be ignored. + """ + server_factory(usec=60 * 1000000, pid='1') + assert env.notify.wait_for_status(r'^Total requests: ', + timeout=WATCHDOG_TIMEOUT), \ + "the monitor hook never ran, so this proves nothing" + assert not [m for m in env.notify.messages if is_watchdog_ping(m)], \ + "a keep-alive was sent although $WATCHDOG_PID was another process" + + def test_systemd_006_06_ping_across_reload(self, env, server_factory): + """The keep-alive survives a graceful restart. + + The parent re-reads its configuration in the same process, but + mod_systemd is unloaded and loaded again with it, so anything it + remembered about the watchdog is gone by the time the monitor hook + runs again. systemd keeps the timeout armed throughout. + """ + srv = server_factory(usec=60 * 1000000) + assert env.notify.wait_for(is_watchdog_ping, timeout=WATCHDOG_TIMEOUT) + env.notify.clear() + assert srv.reload() == 0 + assert srv.is_live() + assert env.notify.wait_for(is_watchdog_ping, + timeout=WATCHDOG_TIMEOUT), \ + f"no keep-alive after a reload, got {env.notify.messages}" + + def test_systemd_006_07_ping_covers_the_reload_window(self, env, + server_factory): + """A keep-alive is sent as the configuration is read, and again once + it is loaded. + + Reading the configuration happens outside the parent's monitor loop, + so nothing else reports during it. The watchdog stays armed while + the unit reloads, and a configuration which takes longer to parse + than the timeout would otherwise be killed halfway through. + """ + srv = server_factory(usec=60 * 1000000) + assert env.notify.wait_for(is_watchdog_ping, timeout=WATCHDOG_TIMEOUT) + env.notify.clear() + assert srv.reload() == 0 + assert env.notify.wait_for( + lambda m: 'RELOADING' in m and is_watchdog_ping(m), + timeout=WATCHDOG_TIMEOUT), \ + f"no keep-alive as the configuration was read, " \ + f"got {env.notify.messages}" + assert env.notify.wait_for( + lambda m: m.get('READY') == '1' and is_watchdog_ping(m), + timeout=WATCHDOG_TIMEOUT), \ + f"no keep-alive once the configuration was loaded, " \ + f"got {env.notify.messages}" + + def test_systemd_006_08_short_timeout_warned(self, env, server_factory): + """A WatchdogSec the parent cannot meet is reported. + + The keep-alive is sent from the monitor hook, which runs once every + ten turns of the parent's one second loop. A timeout of a few + seconds cannot be met however the module is written, and failing + silently would leave the server being killed and restarted with + nothing in the log to say why. + """ + server_factory(usec=2 * 1000000) + assert env.httpd_error_log.scan_recent( + re.compile(r'.*AH10621: .*[Ww]atchdog.*'), timeout=10), \ + "no warning about a watchdog timeout that cannot be met" + env.httpd_error_log.ignore_recent(lognos=['AH10621']) + + def test_systemd_006_09_workable_timeout_not_warned(self, env, + server_factory): + """A timeout the parent can meet is not complained about.""" + server_factory(usec=MIN_WATCHDOG_SEC * 1000000) + assert env.notify.wait_for(is_watchdog_ping, timeout=WATCHDOG_TIMEOUT) + with pytest.raises(TimeoutError): + env.httpd_error_log.scan_recent( + re.compile(r'.*AH10621: .*'), timeout=1) + + @pytest.mark.skipif(not TransientService.is_available(), + reason="no per-user systemd manager") + def test_systemd_006_10_service_keeps_watchdog_alive(self, env): + """The real thing: systemd records each keep-alive it receives, and + the unit stays active rather than failing with Result=watchdog.""" + with TransientService(env, port=env.http_port2, + properties=[f'WatchdogSec={MIN_WATCHDOG_SEC}s'], + name=f'httpd-wd-{os.getpid()}') as svc: + assert svc.start().returncode == 0 + assert svc.wait_active() + first = svc.show('WatchdogTimestampMonotonic') + assert first and int(first) > 0, \ + "systemd recorded no keep-alive at all" + end = time.time() + WATCHDOG_TIMEOUT + while time.time() < end: + if svc.show('WatchdogTimestampMonotonic') != first: + break + time.sleep(0.5) + assert svc.show('WatchdogTimestampMonotonic') != first, \ + "systemd received no further keep-alive" + assert svc.show('ActiveState') == 'active' + assert svc.show('Result') == 'success' diff --git a/test/modules/arch/linux/test_007_notify_reload.py b/test/modules/arch/linux/test_007_notify_reload.py new file mode 100644 index 00000000000..93833299c63 --- /dev/null +++ b/test/modules/arch/linux/test_007_notify_reload.py @@ -0,0 +1,139 @@ +import os +import time + +import pytest + +from .env import (NOTIFY_RELOAD_VERSION, TransientService, error_log_since, + error_log_size, http_responds, systemd_version) + +pytestmark = [ + pytest.mark.skipif( + not TransientService.is_available(), + reason="no per-user systemd manager to run a transient service under"), + pytest.mark.skipif( + systemd_version() < NOTIFY_RELOAD_VERSION, + reason=f"Type=notify-reload needs systemd {NOTIFY_RELOAD_VERSION}"), +] + + +class TestNotifyReload: + """httpd as a Type=notify-reload unit. + + Under Type=notify a reload is only whatever ExecReload= does, and + systemctl returns as soon as that command exits. For "httpd -k + graceful" that is as soon as the signal has been sent, long before the + new configuration is in use, and it reports success however the restart + turns out. Type=notify-reload holds the reload job open until the + service sends RELOADING=1 and then READY=1, which mod_systemd sends + from pre_config and post_config, so the result is reported once it is + known. + + The unit here keeps ExecReload= and sets ReloadSignal=SIGCONT, which + httpd ignores. The manager runs ExecReload= first and sends + ReloadSignal= only once it has exited successfully, so: + + - "httpd -k graceful" stays the thing which restarts the server, and + because it parses the new configuration in its own process before + signalling, a configuration which does not parse fails the reload + without the running server being signalled at all. A bare + ReloadSignal= has the running parent read the new configuration and + exit if it does not parse. + + - the signal the manager then sends is redundant, so it is pointed at + one httpd does not act on. The default is SIGHUP, which httpd + takes as an *ungraceful* restart. + """ + + @pytest.fixture(autouse=True, scope='class') + def _class_scope(self, env): + assert env.apache_stop() == 0 + yield + assert env.apache_stop() == 0 + + @pytest.fixture + def service(self, env) -> TransientService: + svc = TransientService(env, port=env.http_port2, + name=f'httpd-reload-{os.getpid()}', + service_type='notify-reload', + properties=['ReloadSignal=SIGCONT']) + yield svc + svc.stop() + + def test_systemd_007_01_reload_waits_for_the_new_config(self, env, service): + """systemctl reload returns only once the reloaded server is + serving: the port added to the configuration answers without the + test waiting for it, because READY=1 is sent from post_config, by + which point the listeners are open.""" + assert service.start().returncode == 0 + assert service.wait_active() + assert http_responds(env.http_port2) + + service.rewrite_conf(extra=f'Listen {env.proxy_port}') + r = service.reload() + assert r.returncode == 0, f"reload failed: {r.stderr}" + assert http_responds(env.proxy_port, timeout=10.0), \ + "reload returned before the newly configured port was served" + assert http_responds(env.http_port2) + + def test_systemd_007_02_broken_config_fails_the_reload(self, env, service): + """A configuration which does not parse fails the reload and leaves + the running server alone. ExecReload= reads it in a process of its + own and exits without signalling, and the manager sends no reload + signal of its own once that has failed, so the parent never reads + it.""" + assert service.start().returncode == 0 + assert service.wait_active() + pid = int(service.show('MainPID')) + + service.rewrite_conf(extra='ThisDirectiveDoesNotExist on') + r = service.reload() + assert r.returncode != 0, \ + "reload of an unparseable configuration reported success" + assert service.show('ActiveState') == 'active', \ + "a failed reload took the unit out of active" + # The parent exits when a restart re-reads a configuration which + # does not parse, and its children go on serving for a while + # afterwards, so answering a request is not on its own proof that + # the server survived: check the process the manager tracks. + assert int(service.show('MainPID')) == pid, \ + "the parent was restarted despite the configuration not parsing" + assert http_responds(env.http_port2) + + def test_systemd_007_03_reload_restarts_gracefully_once(self, env, service): + """One reload is one graceful restart, and no ungraceful one. + Leaving ReloadSignal= at its SIGHUP default is what this catches: + httpd restarts ungracefully on SIGHUP, dropping the connections a + reload is supposed to keep.""" + if env.mpm_module not in ('mpm_event', 'mpm_worker'): + pytest.skip(f"{env.mpm_module} does not log the graceful restart") + assert service.start().returncode == 0 + assert service.wait_active() + + pos = error_log_size(env) + assert service.reload().returncode == 0 + # The restart is logged before the configuration is re-read, so it + # is written by the time READY=1 ends the reload; give the + # redundant signal which follows a moment to land as well. + time.sleep(2) + log = error_log_since(env, pos) + assert log.count("Attempting to restart") == 0, \ + f"the reload restarted the server ungracefully:\n{log}" + assert log.count("Doing graceful restart") == 1, \ + f"expected one graceful restart, log said:\n{log}" + + def test_systemd_007_04_reload_repeats(self, env, service): + """Reloading twice works. The manager ignores a RELOADING=1 which + is not stamped later than the reload it asked for, so a server + which got MONOTONIC_USEC wrong could report the first reload and + then leave the second to time out.""" + assert service.start().returncode == 0 + assert service.wait_active() + pid = int(service.show('MainPID')) + + for i in range(2): + r = service.reload() + assert r.returncode == 0, f"reload {i + 1} failed: {r.stderr}" + assert service.show('ActiveState') == 'active' + assert int(service.show('MainPID')) == pid, \ + "a graceful restart replaced the parent process" + assert http_responds(env.http_port2) diff --git a/test/pyhttpd/conf/httpd.conf.template b/test/pyhttpd/conf/httpd.conf.template index 255b88ad05f..92cef2e06d2 100644 --- a/test/pyhttpd/conf/httpd.conf.template +++ b/test/pyhttpd/conf/httpd.conf.template @@ -1,6 +1,9 @@ ServerName localhost ServerRoot "${server_dir}" +DefaultRuntimeDir logs +PidFile httpd.pid + Include "conf/modules.conf" DocumentRoot "${server_dir}/htdocs" diff --git a/test/pyhttpd/conf/stop.conf.template b/test/pyhttpd/conf/stop.conf.template index 21bae845f8d..5e76b2d6e24 100644 --- a/test/pyhttpd/conf/stop.conf.template +++ b/test/pyhttpd/conf/stop.conf.template @@ -5,6 +5,11 @@ ServerName localhost ServerRoot "${server_dir}" +# Must agree with httpd.conf, or the running server cannot be found: +# the built-in default varies between httpd versions. +DefaultRuntimeDir logs +PidFile httpd.pid + Include "conf/modules.conf" DocumentRoot "${server_dir}/htdocs" diff --git a/test/pyhttpd/env.py b/test/pyhttpd/env.py index a30a027fb93..695176d54c7 100644 --- a/test/pyhttpd/env.py +++ b/test/pyhttpd/env.py @@ -303,6 +303,7 @@ def __init__(self, pytestconfig=None): self._verbosity = pytestconfig.option.verbose if pytestconfig is not None else 0 self._test_conf = os.path.join(self._server_conf_dir, "test.conf") self._httpd_base_conf = [] + self._httpd_env = {} self._httpd_log_modules = ['aptest'] self._log_interesting = None self._setup = None @@ -327,6 +328,19 @@ def add_httpd_conf(self, lines: List[str]): def add_httpd_log_modules(self, modules: List[str]): self._httpd_log_modules.extend(modules) + def set_httpd_env(self, name: str, value: Optional[str]): + """Add a variable to the environment httpd is started with, or + remove it again when passed None. + + Used by tests for modules which take their input from the + environment rather than from the configuration, such as + mod_systemd reading $NOTIFY_SOCKET. + """ + if value is None: + self._httpd_env.pop(name, None) + else: + self._httpd_env[name] = value + def issue_certs(self): if self._ca is None: self._ca = HttpdTestCA.create_root(name=self.http_tld, @@ -690,6 +704,7 @@ def _clean_path_env(self) -> dict: parts.insert(0, venv_bin) env = os.environ.copy() env['PATH'] = os.pathsep.join(parts) + env.update(self._httpd_env) return env def _run_apachectl(self, cmd) -> ExecResult: @@ -741,6 +756,26 @@ def apache_fail(self): rv = 0 return rv + def apache_hard_restart(self) -> int: + """Restart without the "graceful" flag, so the MPM starts over.""" + r = self._run_apachectl("restart") + if r.exit_code == 0: + return 0 if self.is_live(self._http_base, + timeout=timedelta(seconds=10)) else -1 + return r.exit_code + + def read_pid_file(self, name: str = 'httpd.pid') -> Optional[int]: + # Where PidFile lands depends on how the httpd under test resolves + # a relative path against DefaultRuntimeDir, which has differed + # between versions; look in both places rather than assume. + for d in (self._server_logs_dir, self._server_dir): + try: + with open(os.path.join(d, name)) as fd: + return int(fd.read().strip()) + except (OSError, ValueError): + continue + return None + def apache_access_log_clear(self): if os.path.isfile(self._server_access_log): os.remove(self._server_access_log)