diff --git a/board/common/post-build.sh b/board/common/post-build.sh index eef3e9492..52fba85c9 100755 --- a/board/common/post-build.sh +++ b/board/common/post-build.sh @@ -146,3 +146,14 @@ mkuserguide() if [ "$BR2_PACKAGE_WEBUI" = "y" ]; then mkuserguide fi + +# The common rootfs skeleton carries confs for optional daemons, drop +# them when the daemon is not part of this image. +if [ "$BR2_PACKAGE_TTYD" != "y" ]; then + rm -f "$TARGET_DIR/etc/finit.d/available/ttyd.conf" \ + "$TARGET_DIR/etc/nginx/available/ttyd.conf" +fi + +# Drop dangling Finit enabled/*.conf symlinks, e.g., optional services +# not part of this image, they cause noise at every initctl reload. +find "$TARGET_DIR/etc/finit.d/enabled" -xtype l -delete 2>/dev/null diff --git a/doc/ChangeLog.md b/doc/ChangeLog.md index 7c825ef34..f0f101274 100644 --- a/doc/ChangeLog.md +++ b/doc/ChangeLog.md @@ -10,15 +10,51 @@ All notable changes to the project are documented in this file. - Upgrade Linux kernel to 6.18.42 (LTS) - Upgrade Buildroot to 2025.02.15 (LTS) +- Upgrade mdns-alias to [v1.3][ma13]: fixes crash on hostname change while + disconnected from Avahi, treats entry group failures and CNAME collisions + as transient (retried instead of exiting), and quieter logs by default - Add support for firewall address-set (ipset): named sets of IP addresses and networks, usable as zone sources for per-IP access control, issue #1189 - Build RPi64 SD card images in release builds - Include .pkg files in release builds +- The `statd` service now logs at `notice` level by default, like other + services, and supports `-v ` to adjust verbosity at runtime ### Fixes - Fix annoying "cannot deselect all services" or reset to YANG default in the web interface's firewall configuration page +- Fix sporadic slow response, or timeouts, when reading device status while + mDNS neighbors are being discovered, e.g., after an mDNS restart. Updates + to the neighbor table are now batched, and politely retried when other users + or services keep the system busy, logged as: + + statd[3558]: mdns: operational datastore busy, retrying ... + +- Fix interface setup failures after an interrupted or failed configuration + change. Leftover interfaces could break all subsequent changes to the + interface configuration, until reboot, logged as: + + dagger[2599]: Aborting: /run/net/131/action/init/br0/50-init.ip failed with exitcode 1 + confd[2599]: Failed to apply interface configuration + + with `RTNETLINK answers: File exists` in the failing script's log. + Creating and deleting interfaces is now tolerant to such leftovers +- Fix slow response, or timeouts, when configuring the system or reading + status while a periodic status snapshot is in progress. On slower systems + with a big configuration, the snapshot, taken every five minutes, could + hold up other users for minutes. Snapshots now run in a separate + low-priority process, `statd-journal`, reading status in small chunks to + let other users interleave +- Fix noisy logs on minimal builds, repeated on every configuration change: + + finit[1]: Skipping /etc/finit.d/enabled/webui.conf, dangling symlink: No such file or directory + finit[1]: service_register():/etc/finit.d/enabled/ttyd.conf: skipping ttyd: No such file or directory + + Optional services not included in the image are now skipped when enabled + in the configuration, and leftover confs are dropped at build time + +[ma13]: https://github.com/troglobit/mdns-alias/releases/tag/v1.3 [v26.06.0][] - 2026-07-01 ------------------------- diff --git a/package/mdns-alias/mdns-alias.hash b/package/mdns-alias/mdns-alias.hash index 7263ae672..7e4e9510e 100644 --- a/package/mdns-alias/mdns-alias.hash +++ b/package/mdns-alias/mdns-alias.hash @@ -1,5 +1,5 @@ # From GitHub release -sha256 9f194fa0b6e34fd915054394ef5b820a4f6b1755ace5ed1011bfba6df550accf mdns-alias-1.2.tar.gz +sha256 8186f0758f184cbdcab1033e4945117a587356c323e53bcdd19d47911ee2567b mdns-alias-1.3.tar.gz # Locally generated sha256 3d6f910b5e198f3daab48047b8ee6949040f7abee3927daf2e231f265faf7d91 LICENSE diff --git a/package/mdns-alias/mdns-alias.mk b/package/mdns-alias/mdns-alias.mk index f17147658..5fdeb91b0 100644 --- a/package/mdns-alias/mdns-alias.mk +++ b/package/mdns-alias/mdns-alias.mk @@ -4,7 +4,7 @@ # ################################################################################ -MDNS_ALIAS_VERSION = 1.2 +MDNS_ALIAS_VERSION = 1.3 MDNS_ALIAS_SITE = https://github.com/troglobit/mdns-alias/releases/download/v$(MDNS_ALIAS_VERSION) MDNS_ALIAS_LICENSE = ISC MDNS_ALIAS_LICENSE_FILES = LICENSE diff --git a/package/statd/statd.conf b/package/statd/statd.conf index 5d76de659..cbc1e9d9c 100644 --- a/package/statd/statd.conf +++ b/package/statd/statd.conf @@ -1,2 +1,2 @@ #set DEBUG=1 -service name:statd [12345] statd -f -p /run/statd.pid -n -- Status daemon +service name:statd [12345] statd -- Status daemon diff --git a/src/confd/src/core.c b/src/confd/src/core.c index a0569b688..ac58c20a1 100644 --- a/src/confd/src/core.c +++ b/src/confd/src/core.c @@ -57,6 +57,12 @@ int finit_enable(const char *svc) (int)(at - svc), svc); } + if (!fexist(src)) { + /* Optional service not part of this image, avoid dangling symlink */ + INFO("%s is not available in this image, cannot enable", svc); + return 0; + } + snprintf(dst, sizeof(dst), FINIT_RCSD "/enabled/%s.conf", svc); if (symlink(src, dst) && errno != EEXIST) { ERRNO("failed enabling %s", svc); diff --git a/src/confd/src/interfaces.c b/src/confd/src/interfaces.c index db01484c7..4f90ff4df 100644 --- a/src/confd/src/interfaces.c +++ b/src/confd/src/interfaces.c @@ -507,13 +507,26 @@ static int eth_gen_del(struct lyd_node *dif, FILE *ip) return 0; } -static int link_gen_del(struct lyd_node *dif, FILE *ip) +/* + * Tolerate the interface already being gone, e.g., leftover state from + * an earlier, partially applied generation -- a failed teardown would + * abort the whole generation. + */ +static int link_gen_del(struct dagger *net, struct lyd_node *dif) { - fprintf(ip, "link del dev %s\n", lydx_get_cattr(dif, "name")); + const char *ifname = lydx_get_cattr(dif, "name"); + FILE *sh; + + sh = dagger_fopen_net_exit(net, ifname, NETDAG_EXIT, "exit-del.sh"); + if (!sh) + return -EIO; + + fprintf(sh, "ip link del dev %s 2>/dev/null || true\n", ifname); + fclose(sh); return 0; } -static int veth_gen_del(struct lyd_node *dif, FILE *sh) +static int veth_gen_del(struct dagger *net, struct lyd_node *dif) { if (!veth_is_primary(dif)) return 0; @@ -526,7 +539,7 @@ static int veth_gen_del(struct lyd_node *dif, FILE *sh) if (lydx_get_child(dif, "container-network")) return 0; - return link_gen_del(dif, sh); + return link_gen_del(net, dif); } static int netdag_gen_iface_del(struct dagger *net, struct lyd_node *dif, @@ -538,10 +551,6 @@ static int netdag_gen_iface_del(struct dagger *net, struct lyd_node *dif, DEBUG_IFACE(dif, ""); - ip = dagger_fopen_net_exit(net, ifname, NETDAG_EXIT, "exit.ip"); - if (!ip) - return -EIO; - type = iftype_from_iface(dif); if (type == IFT_UNKNOWN) /* The interface is still in running, so we need to @@ -554,11 +563,14 @@ static int netdag_gen_iface_del(struct dagger *net, struct lyd_node *dif, switch (type) { case IFT_ETH: case IFT_LO: + ip = dagger_fopen_net_exit(net, ifname, NETDAG_EXIT, "exit.ip"); + if (!ip) + return -EIO; eth_gen_del(dif, ip); + fclose(ip); break; case IFT_VETH: - veth_gen_del(dif, ip); - break; + return veth_gen_del(net, dif); case IFT_WIFI: wifi_del_iface(dif, net); break; @@ -571,11 +583,9 @@ static int netdag_gen_iface_del(struct dagger *net, struct lyd_node *dif, case IFT_VXLAN: case IFT_WIREGUARD: case IFT_UNKNOWN: - link_gen_del(dif, ip); - break; + return link_gen_del(net, dif); } - fclose(ip); return 0; } @@ -617,6 +627,49 @@ static sr_error_t netdag_gen_iface_timeout(struct dagger *net, const char *ifnam return SR_ERR_OK; } +/* + * A netlink-created interface may linger from an earlier, partially + * applied generation, causing our `link add` to fail with EEXIST and + * abort the whole generation. Remove any leftover before creating. + */ +static int netdag_gen_ensure_absent(struct dagger *net, struct lyd_node *cif) +{ + const char *ifname = lydx_get_cattr(cif, "name"); + const char *peer = NULL; + FILE *sh; + + switch (iftype_from_iface(cif)) { + case IFT_BRIDGE: + case IFT_DUMMY: + case IFT_GRE: + case IFT_GRETAP: + case IFT_LAG: + case IFT_VLAN: + case IFT_VXLAN: + case IFT_WIREGUARD: + break; + case IFT_VETH: + /* primary's `link add` creates both ends */ + if (!veth_is_primary(cif)) + return 0; + peer = lydx_get_cattr(lydx_get_child(cif, "veth"), "peer"); + break; + default: + return 0; + } + + sh = dagger_fopen_net_init(net, ifname, NETDAG_INIT_PRE, "ensure-absent.sh"); + if (!sh) + return -EIO; + + fprintf(sh, "ip link del dev %s 2>/dev/null || true\n", ifname); + if (peer) + fprintf(sh, "ip link del dev %s 2>/dev/null || true\n", peer); + fclose(sh); + + return 0; +} + static sr_error_t netdag_gen_iface(sr_session_ctx_t *session, struct dagger *net, struct lyd_node *dif, struct lyd_node *cif) { @@ -683,7 +736,8 @@ static sr_error_t netdag_gen_iface(sr_session_ctx_t *session, struct dagger *net } if (op == LYDX_OP_CREATE) { - err = netdag_gen_afspec_add(session, net, dif, cif, ip); + err = netdag_gen_ensure_absent(net, cif); + err = err ? : netdag_gen_afspec_add(session, net, dif, cif, ip); if (err) goto err_close_ip; } diff --git a/src/confd/src/main.c b/src/confd/src/main.c index f679d9df6..cb908623b 100644 --- a/src/confd/src/main.c +++ b/src/confd/src/main.c @@ -655,10 +655,11 @@ int main(int argc, char **argv) else if (!strcmp(optarg, "warning")) log_level = LOG_WARNING; else if (!strcmp(optarg, "info")) - log_level = LOG_NOTICE; - else if (!strcmp(optarg, "debug")) + log_level = LOG_INFO; + else if (!strcmp(optarg, "debug")) { log_level = LOG_DEBUG; - else { + debug = 1; + } else { fprintf(stderr, "confd error: Invalid verbosity \"%s\"\n", optarg); return EXIT_FAILURE; } diff --git a/src/statd/Makefile.am b/src/statd/Makefile.am index 727583daa..6b4488522 100644 --- a/src/statd/Makefile.am +++ b/src/statd/Makefile.am @@ -4,6 +4,7 @@ ACLOCAL_AMFLAGS = -I m4 sbin_PROGRAMS = statd statd_SOURCES = statd.c shared.c shared.h journal.c journal_retention.c journal.h avahi.c avahi.h statd_CPPFLAGS = -D_DEFAULT_SOURCE -D_GNU_SOURCE +statd_CPPFLAGS += -DSTATD_VERSION=\"$(PACKAGE_VERSION)\" statd_CFLAGS = -W -Wall -Wextra statd_CFLAGS += $(jansson_CFLAGS) $(libyang_CFLAGS) $(sysrepo_CFLAGS) statd_CFLAGS += $(libsrx_CFLAGS) $(libite_CFLAGS) diff --git a/src/statd/avahi.c b/src/statd/avahi.c index b6619cc5e..1063dcee8 100644 --- a/src/statd/avahi.c +++ b/src/statd/avahi.c @@ -354,6 +354,57 @@ static int sr_setstr(sr_session_ctx_t *ses, const char *xpath, const char *val) return err; } +/* + * Resolver events arrive in bursts, e.g., browse storms after an avahi + * restart. Instead of one sr_apply_changes() per event, coalesce all + * edits staged on ctx->sr_ses and apply once the burst settles. On + * datastore contention, back off and retry rather than block -- this + * loop also serves all operational get callbacks. + */ +#define MDNS_APPLY_DEBOUNCE 0.5 +#define MDNS_APPLY_TIMEOUT 1000 /* ms */ +#define MDNS_APPLY_RETRY_MAX 6 /* caps backoff at 0.5 * 2^6 = 32 s */ + +static void ds_apply_cb(struct ev_loop *loop, ev_timer *w, int revents) +{ + struct mdns_ctx *ctx = (struct mdns_ctx *) + ((char *)w - offsetof(struct mdns_ctx, apply_timer)); + int err; + + (void)loop; + (void)revents; + + err = sr_apply_changes(ctx->sr_ses, MDNS_APPLY_TIMEOUT); + switch (err) { + case SR_ERR_OK: + ctx->apply_retries = 0; + break; + case SR_ERR_TIME_OUT: + case SR_ERR_LOCKED: + if (ctx->apply_retries < MDNS_APPLY_RETRY_MAX) + ctx->apply_retries++; + if (ctx->apply_retries == 3) + NOTE("mdns: operational datastore busy, retrying ..."); + ev_timer_set(&ctx->apply_timer, MDNS_APPLY_DEBOUNCE * (1 << ctx->apply_retries), 0.0); + ev_timer_start(ctx->loop, &ctx->apply_timer); + break; + default: + ERROR("mdns: sr_apply_changes: %s", sr_strerror(err)); + sr_discard_changes(ctx->sr_ses); + ctx->apply_retries = 0; + break; + } +} + +static void ds_schedule_apply(struct mdns_ctx *ctx) +{ + if (ev_is_active(&ctx->apply_timer)) + return; + + ev_timer_init(&ctx->apply_timer, ds_apply_cb, MDNS_APPLY_DEBOUNCE, 0.0); + ev_timer_start(ctx->loop, &ctx->apply_timer); +} + /* * Return an XPath string literal quoting val: single-quoted unless val * contains a single quote, in which case double quotes are used instead. @@ -437,13 +488,12 @@ static void ds_push_resolver(struct mdns_ctx *ctx, struct avahi_service *svc, } if (err) { + /* drops any coalesced edits too, later events repopulate */ sr_discard_changes(ctx->sr_ses); return; } - err = sr_apply_changes(ctx->sr_ses, 0); - if (err) - ERROR("mdns: sr_apply_changes: %s", sr_strerror(err)); + ds_schedule_apply(ctx); } static void ds_delete_service(struct mdns_ctx *ctx, const char *hostname, const char *name) @@ -470,7 +520,7 @@ static void ds_delete_neighbor(struct mdns_ctx *ctx, const char *hostname) static void ds_clear_all(struct mdns_ctx *ctx) { sr_delete_item(ctx->sr_ses, XPATH_BASE, 0); - sr_apply_changes(ctx->sr_ses, 0); + ds_schedule_apply(ctx); } /* -------------------------------------------------------------------------- @@ -641,7 +691,7 @@ static void service_browser_cb(AvahiServiceBrowser *b, } } - sr_apply_changes(ctx->sr_ses, 0); + ds_schedule_apply(ctx); break; } @@ -788,6 +838,7 @@ static void reconn_cb(struct ev_loop *loop, ev_timer *w, int revents) * that a normal daemon restart cancels this timer before it fires. */ #define MDNS_WARN_DELAY 10.0 +#define MDNS_FAIL_ESCALATE 3 /* NOTE level after 3 x MDNS_WARN_DELAY */ static void mdns_retry_cb(struct ev_loop *loop, ev_timer *w, int revents) { @@ -798,8 +849,12 @@ static void mdns_retry_cb(struct ev_loop *loop, ev_timer *w, int revents) (void)revents; ctx->fail_count++; - if (mdns_is_enabled(ctx)) - WARN("mdns: mDNS daemon not responding, will reconnect automatically"); + if (mdns_is_enabled(ctx)) { + if (ctx->fail_count >= MDNS_FAIL_ESCALATE) + NOTE("mdns: mDNS daemon still not responding, will keep trying"); + else + INFO("mdns: mDNS daemon not responding, will reconnect automatically"); + } } static void client_cb(AvahiClient *c, AvahiClientState state, void *userdata) @@ -813,7 +868,10 @@ static void client_cb(AvahiClient *c, AvahiClientState state, void *userdata) if (ctx->fail_count > 0) { ev_timer_stop(ctx->loop, &ctx->reconn_timer); ev_timer_stop(ctx->loop, &ctx->retry_timer); - NOTE("mdns: mDNS daemon reconnected"); + if (ctx->fail_count >= MDNS_FAIL_ESCALATE) + NOTE("mdns: mDNS daemon reconnected"); + else + INFO("mdns: mDNS daemon reconnected"); ctx->fail_count = 0; } INFO("mdns: client running"); @@ -850,16 +908,15 @@ static void client_cb(AvahiClient *c, AvahiClientState state, void *userdata) ev_timer_start(ctx->loop, &ctx->retry_timer); } - { + while (!LIST_EMPTY(&ctx->type_entries)) { struct avahi_type_entry *te; - while (!LIST_EMPTY(&ctx->type_entries)) { - te = LIST_FIRST(&ctx->type_entries); - avahi_service_browser_free(te->browser); - LIST_REMOVE(te, link); - free(te); - } + te = LIST_FIRST(&ctx->type_entries); + avahi_service_browser_free(te->browser); + LIST_REMOVE(te, link); + free(te); } + if (ctx->type_browser) { avahi_service_type_browser_free(ctx->type_browser); ctx->type_browser = NULL; @@ -929,11 +986,11 @@ void mdns_ctx_reconnect(struct mdns_ctx *ctx) int avahi_err; if (!mdns_is_enabled(ctx)) { - NOTE("mdns: mDNS is disabled, ignoring reconnect request"); + INFO("mdns: mDNS is disabled, ignoring reconnect request"); return; } - NOTE("mdns: reconnecting on request"); + INFO("mdns: reconnecting on request"); ev_timer_stop(ctx->loop, &ctx->reconn_timer); ev_timer_stop(ctx->loop, &ctx->retry_timer); @@ -973,6 +1030,8 @@ void mdns_ctx_exit(struct mdns_ctx *ctx) ev_timer_stop(ctx->loop, &ctx->reconn_timer); if (ev_is_active(&ctx->retry_timer)) ev_timer_stop(ctx->loop, &ctx->retry_timer); + if (ev_is_active(&ctx->apply_timer)) + ev_timer_stop(ctx->loop, &ctx->apply_timer); /* Free browsers explicitly before freeing the client */ while (!LIST_EMPTY(&ctx->type_entries)) { @@ -991,7 +1050,9 @@ void mdns_ctx_exit(struct mdns_ctx *ctx) } if (ctx->sr_ses) { - ds_clear_all(ctx); + /* event loop is going away, flush synchronously */ + sr_delete_item(ctx->sr_ses, XPATH_BASE, 0); + sr_apply_changes(ctx->sr_ses, MDNS_APPLY_TIMEOUT); sr_session_stop(ctx->sr_ses); ctx->sr_ses = NULL; } diff --git a/src/statd/avahi.h b/src/statd/avahi.h index 88f06a288..598bef793 100644 --- a/src/statd/avahi.h +++ b/src/statd/avahi.h @@ -61,6 +61,8 @@ struct mdns_ctx { unsigned int fail_count; /* Non-zero while avahi-daemon is absent */ ev_timer reconn_timer; /* Free+recreate client after brief delay */ ev_timer retry_timer; /* Deferred warn-log timer */ + ev_timer apply_timer; /* Debounced DS apply, with retry */ + unsigned int apply_retries; LIST_HEAD(, avahi_neighbor) neighbors; LIST_HEAD(, avahi_service) services; /* Flat list; keyed by 5-tuple */ LIST_HEAD(, avahi_type_entry) type_entries; diff --git a/src/statd/journal.c b/src/statd/journal.c index b515c1f1e..6a762a6b6 100644 --- a/src/statd/journal.c +++ b/src/statd/journal.c @@ -1,31 +1,46 @@ /* SPDX-License-Identifier: BSD-3-Clause */ +/* + * Periodic snapshots of the operational datastore for post-mortem and + * trend analysis: /var/lib/statd/operational.json is always the latest, + * with gzipped timestamped archives kept according to the retention + * policy in journal_retention.c. + * + * The work runs in a forked child, renamed statd-journal, at reduced + * priority. This keeps statd's event loop free to serve operational + * get callbacks -- including those triggered by the snapshot itself. + * The dump is chunked per YANG module, releasing all datastore locks + * between each read, so configuration changes and status queries from + * interactive users interleave with the dump instead of queueing up + * behind one long read. + */ + +#include +#include #include #include -#include -#include -#include #include -#include -#include +#include #include -#include -#include +#include +#include +#include #include +#include +#include +#include + #include #include "journal.h" -#define JOURNAL_DIR "/var/lib/statd" -#define DUMP_FILE "/var/lib/statd/operational.json" -#define DUMP_INTERVAL 300.0 /* 5 minutes in seconds */ - -static void journal_stop_cb(struct ev_loop *loop, struct ev_async *, int) -{ - DEBUG("Journal thread stop signal received"); - ev_break(loop, EVBREAK_ALL); -} +#define JOURNAL_DIR "/var/lib/statd" +#define DUMP_FILE JOURNAL_DIR "/operational.json" +#define DUMP_INTERVAL 300.0 /* seconds of rest between snapshots */ +#define CHUNK_DELAY 50000 /* us breather between module reads */ +#define CHUNK_TIMEOUT 10000 /* ms, keep short: the read holds module locks + * that configuration changes wait on */ static void get_timestamp_filename(char *buf, size_t len, time_t ts) { @@ -102,131 +117,187 @@ static int create_snapshot(const struct lyd_node *tree) return 0; } -static void journal_timer_cb(struct ev_loop *, struct ev_timer *w, int) +/* + * Read operational data one module at a time, merging into a single + * tree. Every sr_get_data() releases its locks on return, giving + * other datastore users a chance to run between chunks. + */ +static struct lyd_node *dump_modules(sr_session_ctx_t *ses, const struct ly_ctx *ctx, + int *skipped) +{ + const struct lys_module *mod; + struct lyd_node *tree = NULL; + uint32_t idx = 0; + + while ((mod = ly_ctx_get_module_iter(ctx, &idx))) { + char xpath[300]; + sr_data_t *data; + int err; + + if (!mod->implemented || !mod->compiled || !mod->compiled->data) + continue; + + snprintf(xpath, sizeof(xpath), "/%s:*", mod->name); + err = sr_get_data(ses, xpath, 0, CHUNK_TIMEOUT, 0, &data); + if (err) { + INFO("Skipping %s: %s", mod->name, sr_strerror(err)); + (*skipped)++; + continue; + } + + if (data) { + if (data->tree && lyd_merge_siblings(&tree, data->tree, 0)) + ERROR("Error, merging %s data", mod->name); + sr_release_data(data); + } + + usleep(CHUNK_DELAY); + } + + return tree; +} + +/* + * Forked child: fresh sysrepo connection, dump, archive, retention, + * then _exit() -- never touch inherited statd state. + */ +static void snapshot_process(void) { - struct journal_ctx *jctx = (struct journal_ctx *)w->data; - struct timespec start, end; struct snapshot *snapshots = NULL; - sr_conn_ctx_t *con; + sr_session_ctx_t *ses = NULL; + sr_conn_ctx_t *conn = NULL; + struct timespec start, end; const struct ly_ctx *ctx; - sr_data_t *sr_data = NULL; - sr_error_t err; - int snapshot_count = 0; - long duration_ms; - + struct lyd_node *tree; + int rc = EXIT_FAILURE; + int skipped = 0; + int count = 0; + long ms; + + prctl(PR_SET_NAME, "statd-journal", 0, 0, 0); + closelog(); /* drop log connection inherited from statd */ + openlog("statd-journal", LOG_PID | LOG_NDELAY | (debug ? LOG_PERROR : 0), LOG_DAEMON); + nice(10); + + NOTE("Starting operational datastore snapshot"); clock_gettime(CLOCK_MONOTONIC, &start); - DEBUG("Starting operational datastore dump"); - con = sr_session_get_connection(jctx->sr_query_ses); - if (!con) { - ERROR("Error, getting sr connection for dump"); - return; + if (mkdir(JOURNAL_DIR, 0755) && errno != EEXIST) + ERROR("Error, creating directory " JOURNAL_DIR ": %s", strerror(errno)); + + if (sr_connect(SR_CONN_DEFAULT, &conn)) { + ERROR("Error, connecting to sysrepo"); + _exit(rc); + } + if (sr_session_start(conn, SR_DS_OPERATIONAL, &ses)) { + ERROR("Error, starting session"); + goto done; } - ctx = sr_acquire_context(con); + ctx = sr_acquire_context(conn); if (!ctx) { - ERROR("Error, acquiring context for dump"); - return; + ERROR("Error, acquiring context"); + goto done; } - /* Query ALL operational data via second session - * This triggers our own operational callbacks running in main thread - */ - DEBUG("Calling sr_get_data on session %p", jctx->sr_query_ses); - err = sr_get_data(jctx->sr_query_ses, "/*", 0, 0, 0, &sr_data); - if (err != SR_ERR_OK) { - ERROR("Error, getting operational data: %s", sr_strerror(err)); - sr_release_context(con); - return; - } - DEBUG("sr_get_data succeeded, got data tree: %p", sr_data ? sr_data->tree : NULL); - - /* Create timestamped snapshot */ - if (sr_data && sr_data->tree) { - if (create_snapshot(sr_data->tree) != 0) { - sr_release_data(sr_data); - sr_release_context(con); - return; - } + tree = dump_modules(ses, ctx, &skipped); + if (tree) { + rc = create_snapshot(tree) ? EXIT_FAILURE : EXIT_SUCCESS; + lyd_free_all(tree); } else { DEBUG("No operational data to dump"); + rc = EXIT_SUCCESS; } + sr_release_context(conn); - sr_release_data(sr_data); - sr_release_context(con); - - /* Apply retention policy */ - if (journal_scan_snapshots(JOURNAL_DIR, &snapshots, &snapshot_count) == 0) { - DEBUG("Applying retention policy to %d snapshots", snapshot_count); - journal_apply_retention_policy(JOURNAL_DIR, snapshots, snapshot_count, time(NULL)); + if (journal_scan_snapshots(JOURNAL_DIR, &snapshots, &count) == 0) { + DEBUG("Applying retention policy to %d snapshots", count); + journal_apply_retention_policy(JOURNAL_DIR, snapshots, count, time(NULL)); free(snapshots); } clock_gettime(CLOCK_MONOTONIC, &end); - duration_ms = (end.tv_sec - start.tv_sec) * 1000 + - (end.tv_nsec - start.tv_nsec) / 1000000; + ms = (end.tv_sec - start.tv_sec) * 1000 + + (end.tv_nsec - start.tv_nsec) / 1000000; + if (skipped) + NOTE("Snapshot created and retention applied (took %ld ms, %d modules busy, skipped)", + ms, skipped); + else + NOTE("Snapshot created and retention applied (took %ld ms)", ms); +done: + if (ses) + sr_session_stop(ses); + sr_disconnect(conn); + _exit(rc); +} - INFO("Journal snapshot created and retention applied (took %ld ms)", duration_ms); +/* + * The timer is one-shot, re-armed only when the previous snapshot has + * finished. Snapshots can thus never overlap, and DUMP_INTERVAL is + * the rest between them rather than a fixed cadence -- on a slow, or + * busy, system snapshots are simply taken further apart. + */ +static void journal_rearm(struct journal_ctx *jctx) +{ + ev_timer_set(&jctx->timer, DUMP_INTERVAL, 0.0); + ev_timer_start(jctx->loop, &jctx->timer); } -static void *journal_thread_fn(void *arg) +static void journal_child_cb(struct ev_loop *loop, struct ev_child *w, int revents) { - struct journal_ctx *jctx = (struct journal_ctx *)arg; - struct ev_timer journal_timer; + struct journal_ctx *jctx = (struct journal_ctx *) + ((char *)w - offsetof(struct journal_ctx, child)); - INFO("Journal thread started"); + (void)revents; - if (mkdir("/var/lib/statd", 0755) != 0 && errno != EEXIST) { - ERROR("Error, creating directory /var/lib/statd: %s", strerror(errno)); - } + ev_child_stop(loop, w); + jctx->pid = 0; - jctx->journal_loop = ev_loop_new(EVFLAG_AUTO); - if (!jctx->journal_loop) { - ERROR("Error, creating journal thread event loop"); - return NULL; - } + if (!WIFEXITED(w->rstatus) || WEXITSTATUS(w->rstatus)) + ERROR("Journal snapshot failed, status %d", w->rstatus); - /* Setup async watcher for stop signal */ - ev_async_init(&jctx->journal_stop, journal_stop_cb); - ev_async_start(jctx->journal_loop, &jctx->journal_stop); + journal_rearm(jctx); +} - /* Setup timer for periodic dumps */ - ev_timer_init(&journal_timer, journal_timer_cb, DUMP_INTERVAL, DUMP_INTERVAL); - journal_timer.data = jctx; - ev_timer_start(jctx->journal_loop, &journal_timer); +static void journal_timer_cb(struct ev_loop *loop, ev_timer *w, int revents) +{ + struct journal_ctx *jctx = (struct journal_ctx *) + ((char *)w - offsetof(struct journal_ctx, timer)); + pid_t pid; - DEBUG("Journal thread entering event loop"); - ev_run(jctx->journal_loop, 0); + (void)revents; - ev_timer_stop(jctx->journal_loop, &journal_timer); - ev_async_stop(jctx->journal_loop, &jctx->journal_stop); - ev_loop_destroy(jctx->journal_loop); + pid = fork(); + if (pid < 0) { + ERRNO("Failed forking journal snapshot process"); + journal_rearm(jctx); + return; + } + if (!pid) + snapshot_process(); /* never returns */ - INFO("Journal thread exiting"); - return NULL; + jctx->pid = pid; + ev_child_init(&jctx->child, journal_child_cb, pid, 0); + ev_child_start(loop, &jctx->child); } -int journal_start(struct journal_ctx *jctx, sr_session_ctx_t *sr_query_ses) +int journal_start(struct journal_ctx *jctx, struct ev_loop *loop) { - int err; + jctx->loop = loop; + jctx->pid = 0; - jctx->sr_query_ses = sr_query_ses; - jctx->journal_thread_running = 1; + ev_timer_init(&jctx->timer, journal_timer_cb, DUMP_INTERVAL, 0.0); + ev_timer_start(loop, &jctx->timer); - err = pthread_create(&jctx->journal_thread, NULL, journal_thread_fn, jctx); - if (err) { - ERROR("Error, creating journal thread: %s", strerror(err)); - return err; - } - - INFO("Periodic operational dump enabled (every %.0f seconds)", DUMP_INTERVAL); + NOTE("Periodic operational snapshot enabled (every %.0f seconds)", DUMP_INTERVAL); return 0; } void journal_stop(struct journal_ctx *jctx) { - /* Signal thread to exit immediately via async watcher */ - jctx->journal_thread_running = 0; - ev_async_send(jctx->journal_loop, &jctx->journal_stop); - pthread_join(jctx->journal_thread, NULL); + ev_timer_stop(jctx->loop, &jctx->timer); + + /* Snapshot in progress completes on its own, reaped by init */ + if (jctx->pid) + ev_child_stop(jctx->loop, &jctx->child); } diff --git a/src/statd/journal.h b/src/statd/journal.h index dc5784fa7..7080854f5 100644 --- a/src/statd/journal.h +++ b/src/statd/journal.h @@ -3,27 +3,28 @@ #ifndef STATD_JOURNAL_H_ #define STATD_JOURNAL_H_ -#include -#include -#include #include -/* Snapshot structure for tracking journal files */ -struct snapshot { - char filename[256]; - time_t timestamp; -}; +#ifndef JOURNAL_RETENTION_STUB +#include +#include struct journal_ctx { - sr_session_ctx_t *sr_query_ses; /* Consumer session for queries */ - struct ev_loop *journal_loop; /* Event loop for journal thread */ - pthread_t journal_thread; /* Thread for periodic dumps */ - struct ev_async journal_stop; /* Signal to stop journal thread */ - volatile int journal_thread_running; /* Flag to stop journal thread */ + struct ev_loop *loop; + ev_timer timer; /* Periodic snapshot trigger */ + struct ev_child child; /* Reaper for the snapshot process */ + pid_t pid; /* Non-zero while a snapshot is running */ }; -int journal_start(struct journal_ctx *jctx, sr_session_ctx_t *sr_query_ses); +int journal_start(struct journal_ctx *jctx, struct ev_loop *loop); void journal_stop(struct journal_ctx *jctx); +#endif + +/* Snapshot structure for tracking journal files */ +struct snapshot { + char filename[256]; + time_t timestamp; +}; int journal_scan_snapshots(const char *dir, struct snapshot **snapshots, int *count); void journal_apply_retention_policy(const char *dir, struct snapshot *snapshots, int count, time_t now); diff --git a/src/statd/statd.c b/src/statd/statd.c index dac055836..e338725f0 100644 --- a/src/statd/statd.c +++ b/src/statd/statd.c @@ -1,15 +1,16 @@ /* SPDX-License-Identifier: BSD-3-Clause */ +#include #include #include #include #include +#include #include #include #include #include #include -#include #include #include @@ -67,10 +68,9 @@ struct sub { struct statd { struct sub_head subs; sr_session_ctx_t *sr_ses; /* Provider session with callbacks */ - sr_session_ctx_t *sr_query_ses; /* Consumer session for queries */ sr_conn_ctx_t *sr_conn; /* Connection (owns YANG context) */ struct ev_loop *ev_loop; - struct journal_ctx journal; /* Journal thread context */ + struct journal_ctx journal; /* Periodic operational snapshots */ struct mdns_ctx mdns; /* mDNS neighbor monitor */ }; @@ -97,7 +97,8 @@ static int ly_add_yanger_data(const struct ly_ctx *ctx, struct lyd_node **parent err = fsystemv(yanger_args, NULL, stream, NULL); if (err) { - ERROR("Error, running yanger"); + ERROR("Error calling yanger %s%s%s, exit code %d", yanger_args[1], + yanger_args[3] ? " " : "", yanger_args[3] ?: "", err); fclose(stream); return SR_ERR_SYS; } @@ -112,7 +113,7 @@ static int ly_add_yanger_data(const struct ly_ctx *ctx, struct lyd_node **parent err = lyd_parse_data_fd(ctx, fd, LYD_JSON, LYD_PARSE_ONLY, 0, parent); if (err) - ERROR("Error, parsing yanger data (%d): %s", err, ly_errmsg(ctx)); + ERROR("Failed parsing yanger data (%d): %s", err, ly_errmsg(ctx)); fclose(stream); /* Note: fclose() already closes the underlying fd from fdopen() */ @@ -135,12 +136,12 @@ static char *xpath_extract(const char *xpath, const char *key) end = strchr(ptr, '\''); if (!end) { - ERROR("Can't find end quote for %s (sanity check)", key); + ERROR("Cannot find end quote for %s (sanity check)", key); return NULL; } if ((end - ptr) >= XPATH_MAX) { - ERROR("Value for %s is to long (sanity check)", key); + ERROR("Value for %s is too long (sanity check)", key); return NULL; } @@ -174,13 +175,13 @@ static int sr_iface_cb(sr_session_ctx_t *session, uint32_t, const char *model, con = sr_session_get_connection(session); if (!con) { - ERROR("Error, getting sr connection"); + ERROR("Error getting sysrepo connection"); return SR_ERR_INTERNAL; } ctx = sr_acquire_context(con); if (!ctx) { - ERROR("Error, acquiring context"); + ERROR("Failed acquiring sysrepo context"); return SR_ERR_INTERNAL; } @@ -191,8 +192,9 @@ static int sr_iface_cb(sr_session_ctx_t *session, uint32_t, const char *model, } err = ly_add_yanger_data(ctx, parent, yanger_args); if (err) - ERROR("Error adding interface yanger data"); + ERROR("Failed adding yanger data for %s", ifname ?: model); + free(ifname); sr_release_context(con); return SR_ERR_OK; @@ -215,19 +217,19 @@ static int sr_generic_cb(sr_session_ctx_t *session, uint32_t, const char *model, con = sr_session_get_connection(session); if (!con) { - ERROR("Error, getting sr connection"); + ERROR("Error getting sysrepo connection"); return SR_ERR_INTERNAL; } ctx = sr_acquire_context(con); if (!ctx) { - ERROR("Error, acquiring context"); + ERROR("Failed acquiring sysrepo context"); return SR_ERR_INTERNAL; } err = ly_add_yanger_data(ctx, parent, yanger_args); if (err) - ERROR("Error adding yanger data"); + ERROR("Failed adding yanger data for %s", yanger_args[1]); sr_release_context(con); @@ -251,19 +253,19 @@ static int sr_ospf_cb(sr_session_ctx_t *session, uint32_t, const char *, con = sr_session_get_connection(session); if (!con) { - ERROR("Error, getting sr connection"); + ERROR("Error getting sysrepo connection"); return SR_ERR_INTERNAL; } ctx = sr_acquire_context(con); if (!ctx) { - ERROR("Error, acquiring context"); + ERROR("Failed acquiring sysrepo context"); return SR_ERR_INTERNAL; } err = ly_add_yanger_data(ctx, parent, yanger_args); if (err) - ERROR("Error adding yanger data"); + ERROR("Failed adding yanger data for %s", yanger_args[1]); sr_release_context(con); @@ -283,23 +285,23 @@ static int sr_rip_cb(sr_session_ctx_t *session, uint32_t, const char *, sr_conn_ctx_t *con; sr_error_t err; - DEBUG("Incoming rip query for xpath: %s", xpath); + DEBUG("Incoming RIP query for xpath: %s", xpath); con = sr_session_get_connection(session); if (!con) { - ERROR("Error, getting sr connection"); + ERROR("Error getting sysrepo connection"); return SR_ERR_INTERNAL; } ctx = sr_acquire_context(con); if (!ctx) { - ERROR("Error, acquiring context"); + ERROR("Failed acquiring sysrepo context"); return SR_ERR_INTERNAL; } err = ly_add_yanger_data(ctx, parent, yanger_args); if (err) - ERROR("Error adding yanger data"); + ERROR("Failed adding yanger data for %s", yanger_args[1]); sr_release_context(con); @@ -323,19 +325,19 @@ static int sr_bfd_cb(sr_session_ctx_t *session, uint32_t, const char *, con = sr_session_get_connection(session); if (!con) { - ERROR("Error, getting sr connection"); + ERROR("Error getting sysrepo connection"); return SR_ERR_INTERNAL; } ctx = sr_acquire_context(con); if (!ctx) { - ERROR("Error, acquiring context"); + ERROR("Failed acquiring sysrepo context"); return SR_ERR_INTERNAL; } err = ly_add_yanger_data(ctx, parent, yanger_args); if (err) - ERROR("Error adding yanger data"); + ERROR("Failed adding yanger data for %s", yanger_args[1]); sr_release_context(con); @@ -384,14 +386,14 @@ static int subscribe(struct statd *statd, char *model, char *xpath, SR_SUBSCR_DEFAULT | SR_SUBSCR_NO_THREAD | SR_SUBSCR_DONE_ONLY, &sub->sr_sub); if (err) { - ERROR("Error, subscribing to path \"%s\": %s", xpath, sr_strerror(err)); + ERROR("Failed subscribing to path \"%s\": %s", xpath, sr_strerror(err)); free(sub); return err; } err = sr_get_event_pipe(sub->sr_sub, &sr_ev_pipe); if (err) { - ERROR("Error, getting sysrepo event pipe: %s", sr_strerror(err)); + ERROR("Error getting sysrepo event pipe: %s", sr_strerror(err)); sr_unsubscribe(sub->sr_sub); free(sub); return err; @@ -463,21 +465,86 @@ static int subscribe_to_all(struct statd *statd) return SR_ERR_OK; } +static void version_print(void) +{ + printf("statd - status daemon v%s, compiled with libsysrepo v%s\n\n", + STATD_VERSION, SR_VERSION); +} + +static void help_print(void) +{ + printf("Usage:\n" + " statd [-h] [-V] [-v ]\n" + "\n" + "Options:\n" + " -h, --help Prints usage help.\n" + " -V, --version Prints version information.\n" + " -v, --verbosity \n" + " Change verbosity to a level (none, error, warning, info, debug).\n" + "\n"); +} + int main(int argc, char *argv[]) { struct ev_signal sigint_watcher, sigusr1_watcher, sighup_watcher; int log_opts = LOG_PID | LOG_NDELAY; + int log_level = LOG_NOTICE; struct statd statd = {}; - const char *env; + int opt; int err; - env = getenv("DEBUG"); - if (env || (argc > 1 && !strcmp(argv[1], "-d"))) { + struct option options[] = { + {"help", no_argument, NULL, 'h'}, + {"version", no_argument, NULL, 'V'}, + {"verbosity", required_argument, NULL, 'v'}, + {NULL, 0, NULL, 0}, + }; + + opterr = 0; + while ((opt = getopt_long(argc, argv, "hVv:", options, NULL)) != -1) { + switch (opt) { + case 'h': + version_print(); + help_print(); + return EXIT_SUCCESS; + case 'V': + version_print(); + return EXIT_SUCCESS; + case 'v': + if (!strcmp(optarg, "none")) + log_level = LOG_EMERG; + else if (!strcmp(optarg, "error")) + log_level = LOG_ERR; + else if (!strcmp(optarg, "warning")) + log_level = LOG_WARNING; + else if (!strcmp(optarg, "info")) + log_level = LOG_INFO; + else if (!strcmp(optarg, "debug")) { + log_level = LOG_DEBUG; + debug = 1; + } else { + fprintf(stderr, "statd error: Invalid verbosity \"%s\"\n", optarg); + return EXIT_FAILURE; + } + break; + default: + fprintf(stderr, "statd error: Invalid option or missing argument: -%c\n", optopt); + return EXIT_FAILURE; + } + } + + if (optind < argc) { + fprintf(stderr, "statd error: Redundant parameters\n"); + return EXIT_FAILURE; + } + + if (getenv("DEBUG")) { log_opts |= LOG_PERROR; debug = 1; } openlog("statd", log_opts, LOG_DAEMON); + setlogmask(LOG_UPTO(log_level)); TAILQ_INIT(&statd.subs); statd.ev_loop = EV_DEFAULT; @@ -491,7 +558,7 @@ int main(int argc, char *argv[]) } DEBUG("Connected to sysrepo"); - /* Session 1: Provider with operational callbacks */ + /* Provider session with operational callbacks */ err = sr_session_start(statd.sr_conn, SR_DS_OPERATIONAL, &statd.sr_ses); if (err) { ERROR("Error, start provider session: %s", sr_strerror(err)); @@ -500,19 +567,8 @@ int main(int argc, char *argv[]) } DEBUG("Provider session started (%p)", statd.sr_ses); - /* Session 2: Consumer for querying operational data */ - err = sr_session_start(statd.sr_conn, SR_DS_OPERATIONAL, &statd.sr_query_ses); - if (err) { - ERROR("Error, start query session: %s", sr_strerror(err)); - sr_session_stop(statd.sr_ses); - sr_disconnect(statd.sr_conn); - return EXIT_FAILURE; - } - DEBUG("Query session started (%p)", statd.sr_query_ses); - err = subscribe_to_all(&statd); if (err) { - sr_session_stop(statd.sr_query_ses); sr_session_stop(statd.sr_ses); sr_disconnect(statd.sr_conn); return EXIT_FAILURE; @@ -530,9 +586,8 @@ int main(int argc, char *argv[]) sighup_watcher.data = &statd; ev_signal_start(statd.ev_loop, &sighup_watcher); - err = journal_start(&statd.journal, statd.sr_query_ses); + err = journal_start(&statd.journal, statd.ev_loop); if (err) { - sr_session_stop(statd.sr_query_ses); sr_session_stop(statd.sr_ses); sr_disconnect(statd.sr_conn); return EXIT_FAILURE; @@ -554,7 +609,6 @@ int main(int argc, char *argv[]) journal_stop(&statd.journal); unsub_to_all(&statd); - sr_session_stop(statd.sr_query_ses); sr_session_stop(statd.sr_ses); sr_disconnect(statd.sr_conn); diff --git a/test/case/statd/system/system/run/initctl_-j b/test/case/statd/system/system/run/initctl_-j index 4e9bf0100..f995b21bb 100644 --- a/test/case/statd/system/system/run/initctl_-j +++ b/test/case/statd/system/system/run/initctl_-j @@ -214,7 +214,7 @@ "forking": false, "status": "running", "origin": "/etc/finit.d/enabled/statd.conf", - "command": "statd -f -p /run/statd.pid -n", + "command": "statd", "condition": [ "+pid/confd" ], "restarts": 0, "pidfile": "/run/statd.pid",