From 6e607092832c9e50711a721d2dd2568d6828945b Mon Sep 17 00:00:00 2001 From: Hedy88 Date: Thu, 27 Aug 2026 22:50:44 +0100 Subject: [PATCH] more things --- .clang-format | 6 + README.md | 92 ++++- configure.py | 66 +++- etc/init.rc | 14 +- src/bctl_main.cpp | 93 +++++ src/config.cpp | 116 ++++-- src/config.hpp | 47 ++- src/logger.cpp | 93 ++++- src/logger.hpp | 15 +- src/main.cpp | 28 +- src/selinux_setup.cpp | 88 +++++ src/selinux_setup.hpp | 29 ++ src/supervisor.cpp | 851 +++++++++++++++++++++++++++++++++--------- src/supervisor.hpp | 20 +- tools/run_vm.py | 151 +++++++- 15 files changed, 1409 insertions(+), 300 deletions(-) create mode 100644 .clang-format create mode 100644 src/bctl_main.cpp create mode 100644 src/selinux_setup.cpp create mode 100644 src/selinux_setup.hpp diff --git a/.clang-format b/.clang-format new file mode 100644 index 0000000..3c7b4e4 --- /dev/null +++ b/.clang-format @@ -0,0 +1,6 @@ +BasedOnStyle: LLVM +IndentWidth: 4 +ColumnLimit: 0 +SortIncludes: false +PointerAlignment: Left +ReflowComments: false \ No newline at end of file diff --git a/README.md b/README.md index 5afa5cb..61f5e71 100644 --- a/README.md +++ b/README.md @@ -15,14 +15,13 @@ format. roadmap: -- user/group privilege drop (`user`, `group`, supplementary groups) +- `SIGCHLD` crash-window limiting (rate-limited restarts; `crash-threshold`/ + `crash-window` are parsed but not yet enforced by the reaper) - dependency ordering between services -- `SIGCHLD` crash-window limiting (rate-limited restarts) - property triggers (`property:=`) and `setprop`/`getprop` - per-service logging to files - `reboot`/`poweroff` path with ordered unmount -- `SIGHUP` config reload -- SELinux +- readiness/socket activation ## building @@ -62,17 +61,18 @@ See [`etc/init.rc`](etc/init.rc) for a complete example. ```rc service NAME /path/to/exe [args...] - user root|other # privilege level (drop, planned) - group GROUP [GROUP...] - oneshot # run once and exit, never respawn - disabled # not started by the boot sequence - console # bind stdio to /dev/console - class NAME # grouping (default "default") - respawn never|on-failure|always # restart policy (default always) - crash-threshold N # restarts allowed per window - crash-window SECS - setenv K=V # extra environment (repeatable) - cwd /path + user = root|other # uid after the privilege drop (name or number) + group = GROUP [GROUP...] # primary gid + supplementary groups (names/numbers) + oneshot # run once and exit, never respawn + disabled # not started by the boot sequence + console # bind stdio to /dev/console + class = NAME # grouping (default "default") + respawn = never|on-failure|always # restart policy (default always) + crash-threshold = N # restarts allowed per window + crash-window = SECS + seclabel = CONTEXT # SELinux exec context (--selinux build) + setenv = K=V # extra environment (repeatable) + cwd = /path ``` ### actions @@ -91,6 +91,62 @@ on TRIGGER log message ``` -boot triggers fire in order: `early-init`, `init`, `boot`. `shutdown` is -reserved (planned wiring to the signal path). property/`service-*` triggers -are on the roadmap. +boot triggers fire in order: `early-init`, `init`, `boot`. `shutdown` triggers +fire when the system is winding down. property/`service-*` triggers are on the +roadmap. + +Services run as `root` by default; `user`/`group` trigger a full privilege +drop (supplementary groups, then gid, then uid) before exec. + +## control + +A running init listens on an abstract unix socket (`@bajia`). The bundled +`bctl` client drives it: + +```sh +bctl status # list services + state +bctl start NAME # start a service +bctl stop NAME # graceful stop (SIGTERM) +bctl restart NAME # restart a service +bctl trigger EVENT # fire an action trigger +bctl reload # re-parse init.rc and reconcile services +bctl shutdown [poweroff|reboot] +``` + +`reload` (also `kill -HUP 1`) re-parses the rc files: removed services are +stopped, added services registered, and running services whose definition +changed are restarted with the new definition. A parse error rejects the +reload and keeps the live config. + +## SELinux (experimental) + +SELinux support is opt-in (`configure.py --selinux`, adds `-DBAJIA_SELINUX` ++ `-lselinux`). With it enabled, PID 1 mounts selinuxfs, loads the policy +from `/etc/selinux/config`, calls `selinux_restorecon` on the core tree, and +applies a per-service exec label via the `seclabel = CONTEXT` service option. + +To bring up a policy without hand-writing one, reuse the host's installed +policy (Fedora/SELinux hosts have one at `/etc/selinux//`) in +permissive mode: + +```sh +python3 tools/run_vm.py --selinux +``` + +This bundles the host `policy.policy.` and `file_contexts` into the +initramfs, writes `SELINUX=permissive`, and boots `selinux=1 enforcing=0`. +Watch the serial console for `selinux: policy loaded, enforcing=0`; a +`selinux-probe` service prints the runtime exec contexts: + +``` +probe-ctx=system_u:system_r:init_t:s0 init-ctx=system_u:system_r:kernel_t:s0 +``` + +The guest kernel must support SELinux and the policy version must match the +kernel's (`cat /sys/fs/selinux/policyvers`). Once permissive is stable, +read `avc: denied` lines from dmesg and iterate toward a minimal custom +policy with `checkpolicy`/`audit2allow`, then flip to `enforcing=1`. + +Limitations: uses the dynamic libselinux (no static build on Fedora), so the +`--selinux` init is dynamically linked and the loader + libs (`libselinux`, +`libpcre2-8`, glibc) are bundled into the initramfs. diff --git a/configure.py b/configure.py index a7e4e3f..034e32e 100644 --- a/configure.py +++ b/configure.py @@ -5,9 +5,13 @@ usage: python3 configure.py python3 configure.py --asan # enable address/undefined sanitizers python3 configure.py --debug # -O0 instead of -O2 + python3 configure.py --selinux # enable SELinux integration + python3 configure.py --compile-commands # also write build/compile_commands.json """ import argparse +import json import os +import shlex import shutil import sys from pathlib import Path @@ -54,10 +58,19 @@ format: ) -def emit_ninja(cxx, cxxflags, dst): +def emit_ninja(cxx, cxxflags, dst, write_cc=False): sources = discover_sources() objs = ["obj/" + Path(s).stem + ".o" for s in sources] target = "bajia" + ctl_target = "bctl" + # bctl_main.cpp is a second shipped binary (the control client), + # excluded from the init objects by the *_main.cpp convention. + ctl_source = SRC / "bctl_main.cpp" + ctl_obj = "obj/bctl_main.o" + + # (source, output) pairs for compile_commands.json. + cc_entries = [(SRC / s, o) for s, o in zip(sources, objs)] + cc_entries.append((ctl_source, ctl_obj)) rule_cxx = ( "rule cxx\n" @@ -77,22 +90,49 @@ def emit_ninja(cxx, cxxflags, dst): lines.append(rule_cxx) lines.append(rule_link) lines.append('build {target}: link {objs}'.format(target=target, objs=" ".join(objs))) + lines.append('build {ctl}: link {ctl_obj}'.format(ctl=ctl_target, ctl_obj=ctl_obj)) lines.append("") for o, s in zip(objs, sources): lines.append('build {o}: cxx {src}/{s}'.format(o=o, src=SRC, s=s)) + lines.append('build {ctl_obj}: cxx {ctl_src}'.format(ctl_obj=ctl_obj, ctl_src=ctl_source)) lines.append("") - lines.append('build all: phony {target}'.format(target=target)) + lines.append( + 'build all: phony {target} {ctl}'.format(target=target, ctl=ctl_target)) lines.append("default all") lines.append("") + if write_cc: + write_compile_commands(cxx, cxxflags, cc_entries) + (BUILD / "obj").mkdir(exist_ok=True) dst.write_text("\n".join(lines) + "\n", encoding="utf-8") +def write_compile_commands(cxx, cxxflags, src_out_pairs): + """Write build/compile_commands.json for clangd / clang-tidy.""" + entries = [] + for src, out in src_out_pairs: + args = [cxx] + list(cxxflags) + ["-c", str(src), "-o", out] + entries.append({ + "directory": str(BUILD.resolve()), + "file": str(src.resolve()), + "output": out, + "arguments": args, + "command": shlex.join(args), + }) + (BUILD / "compile_commands.json").write_text( + json.dumps(entries, indent=1) + "\n", encoding="utf-8") + + def parse_args(): p = argparse.ArgumentParser(description="Configure bajia build (generates build.ninja)") p.add_argument("--asan", action="store_true", help="enable address + UB sanitizers") p.add_argument("--debug", action="store_true", help="disable -O2, enable -O0") p.add_argument("--clean", action="store_true", help="remove build dir") + p.add_argument("--selinux", action="store_true", + help="enable SELinux integration (needs libselinux-dev; " + "see tools/run_vm.py --selinux for a VM test path)") + p.add_argument("--compile-commands", action="store_true", + help="write build/compile_commands.json (for clangd)") return p.parse_args() def main(): @@ -106,6 +146,20 @@ def main(): print("error: ninja not found in PATH", file=sys.stderr) return 1 + # Wipe stale objects when compiler flags change: ninja doesn't track + # cflags, so a re-run with a different --selinux/--asan/--debug profile + # would otherwise link a half-stale tree. + stamp = {"selinux": args.selinux, "asan": args.asan, "debug": args.debug} + stamp_file = BUILD / ".configure.json" + if stamp_file.exists(): + try: + prev = json.loads(stamp_file.read_text()) + except (json.JSONDecodeError, OSError): + prev = None + if prev != stamp: + print("build configuration changed; cleaning build/") + shutil.rmtree(BUILD, ignore_errors=True) + BUILD.mkdir(exist_ok=True) cxxflags = list(CXXFLAGS) @@ -117,14 +171,20 @@ def main(): if args.asan: cxxflags += ["-fsanitize=address,undefined", "-fno-omit-frame-pointer"] linkflags += ["-fsanitize=address,undefined"] + if args.selinux: + cxxflags += ["-DBAJIA_SELINUX"] + linkflags += ["-lselinux"] cxxflags = [f for f in cxxflags if f not in WARNINGS] cxxflags = WARNINGS + cxxflags - emit_ninja(CXX, cxxflags, BUILD / "build.ninja") + emit_ninja(CXX, cxxflags, BUILD / "build.ninja", write_cc=args.compile_commands) write_makefile_convenience() + stamp_file.write_text(json.dumps(stamp, indent=1) + "\n", encoding="utf-8") print(f"configured {len(discover_sources())} sources -> build/build.ninja") + if args.compile_commands: + print("wrote build/compile_commands.json") print("run: ninja -C build") return 0 diff --git a/etc/init.rc b/etc/init.rc index c995b57..c13fe8b 100644 --- a/etc/init.rc +++ b/etc/init.rc @@ -2,7 +2,9 @@ # # constructs: # service NAME /path/to/exe [args...] -# user|group|oneshot|disabled|console|class|respawn|crash-*|setenv|cwd +# user = ... | group = ... | class = ... | respawn = ... | crash-threshold = ... +# crash-window = ... | setenv = ... | cwd = ... | seclabel = ... +# oneshot | disabled | console (flag options, no value) # # on TRIGGER # start NAME | stop NAME | restart NAME @@ -39,16 +41,16 @@ on boot # the console service: bind stdio to /dev/console, always respawn. service console /sbin/getty -L ttyS0 115200 vt100 - class core + class = core console - user root + user = root # a long-lived example daemon. respawning is the default (always). service watchdog /usr/sbin/watchdog - class core - respawn always + class = core + respawn = always # a one-shot job: runs once, exits, never respawns. service boot-logo /usr/bin/show-boot-logo - class late + class = late oneshot diff --git a/src/bctl_main.cpp b/src/bctl_main.cpp new file mode 100644 index 0000000..d991069 --- /dev/null +++ b/src/bctl_main.cpp @@ -0,0 +1,93 @@ +// bctl.cpp - tiny line-based client for the bajia control socket (@bajia). +// +// talks to a running bajia (PID 1) to start/stop/restart services, fire +// triggers, query service status, and request shutdown +// +// wired to the abstract unix socket "@bajia" served by +// Supervisor::open_control_socket(). +#include +#include +#include +#include +#include +#include +#include + +namespace { + +constexpr const char kCtlName[] = "bajia"; + +// NOLINTNEXTLINE(cppcoreguidelines-pro-type-vararg) +void usage(const char* argv0) { + std::fprintf( + stderr, + "usage: %s [args...]\n" + " start NAMES start a service\n" + " stop NAME gracefully stop a service (SIGTERM)\n" + " restart NAME restart a service\n" + " trigger EVENT fire a trigger (e.g. boot, shutdown)\n" + " reload re-parse the rc files and reconcile services\n" + " status list services and their state\n" + " shutdown [kind] shut down (kind: poweroff|reboot)\n" + " ping sanity check that init is alive\n", + argv0); +} + +} // namespace + +int main(int argc, char** argv) { + if (argc < 2) { + usage(argv[0]); + return 2; + } + + std::string line; + for (int i = 1; i < argc; ++i) { + if (i > 1) + line += ' '; + line += argv[i]; + } + line += '\n'; + + const int fd = ::socket(AF_UNIX, SOCK_STREAM | SOCK_CLOEXEC, 0); + if (fd < 0) { + std::perror("bctl: socket"); + return 1; + } + + sockaddr_un addr{}; + addr.sun_family = AF_UNIX; + std::memcpy(addr.sun_path + 1, kCtlName, + sizeof kCtlName); // leading NUL = abstract + const socklen_t len = static_cast(offsetof(sockaddr_un, sun_path) + + 1 + sizeof kCtlName); + if (::connect(fd, reinterpret_cast(&addr), len) != 0) { + std::fprintf(stderr, "bctl: cannot connect to @%s: %s\n", kCtlName, + std::strerror(errno)); + ::close(fd); + return 1; + } + + const ssize_t nw = ::write(fd, line.data(), line.size()); + if (nw != static_cast(line.size())) { + std::perror("bctl: write"); + ::close(fd); + return 1; + } + ::shutdown(fd, SHUT_WR); // tell init no more commands; read the reply + + std::string out; + char buf[512]; + ssize_t n; + while ((n = ::read(fd, buf, sizeof buf)) > 0) { + out.append(buf, static_cast(n)); + } + ::close(fd); + + std::fwrite(out.data(), 1, out.size(), stdout); + + // a line starting with "ERR" means failure (status lines end with OK). + const bool failed = + out.find("\nERR ") != std::string::npos || out.rfind("ERR ", 0) == 0; + return failed ? 1 : 0; +} \ No newline at end of file diff --git a/src/config.cpp b/src/config.cpp index e455c8c..8345812 100644 --- a/src/config.cpp +++ b/src/config.cpp @@ -12,7 +12,7 @@ namespace { std::vector tokenize(const std::string& line) { std::vector out; std::string cur; - + bool in_q = false; bool need_quote_close = false; @@ -48,7 +48,8 @@ std::vector tokenize(const std::string& line) { } if (in_q) { // unterminated quote: best effort, keep what we have. - if (!cur.empty()) out.push_back(cur); + if (!cur.empty()) + out.push_back(cur); } else if (!cur.empty()) { out.push_back(cur); } @@ -57,8 +58,10 @@ std::vector tokenize(const std::string& line) { } RespawnPolicy parse_respawn(const std::string& s) { - if (s == "always") return RespawnPolicy::Always; - if (s == "on-failure") return RespawnPolicy::OnFailure; + if (s == "always") + return RespawnPolicy::Always; + if (s == "on-failure") + return RespawnPolicy::OnFailure; return RespawnPolicy::Never; } @@ -66,14 +69,16 @@ RespawnPolicy parse_respawn(const std::string& s) { Service* Config::find_service(const std::string& name) { for (auto& s : services) { - if (s.name == name) return &s; + if (s.name == name) + return &s; } return nullptr; } const Service* Config::find_service(const std::string& name) const { for (auto& s : services) { - if (s.name == name) return &s; + if (s.name == name) + return &s; } return nullptr; } @@ -81,10 +86,11 @@ const Service* Config::find_service(const std::string& name) const { // public entry point Config parse_config(const std::vector& files) { Config cfg; + cfg.sources = files; int line = 0; - std::string section_kind; // "service" or "action" - Service* cur_svc = nullptr; // service being configured - Action* cur_act = nullptr; // action being configured + std::string section_kind; // "service" or "action" + Service* cur_svc = nullptr; // service being configured + Action* cur_act = nullptr; // action being configured for (const auto& file : files) { std::ifstream in(file); @@ -100,7 +106,8 @@ Config parse_config(const std::vector& files) { while (std::getline(in, raw)) { ++line; auto toks = tokenize(raw); - if (toks.empty()) continue; + if (toks.empty()) + continue; std::string first = toks[0]; size_t indent = raw.find_first_not_of(" \t"); @@ -133,42 +140,75 @@ Config parse_config(const std::vector& files) { } // service options (must already be inside a service section). + // value-taking keywords require `keyword = value` syntax. if (section_kind == "service" && cur_svc) { - if (first == "user" && toks.size() >= 2) cur_svc->uid = toks[1]; - else if (first == "group" && toks.size() >= 2) { - cur_svc->gid = toks[1]; - for (size_t i = 2; i < toks.size(); ++i) cur_svc->groups.push_back(toks[i]); + if (first == "oneshot") + cur_svc->oneshot = true; + else if (first == "disabled") + cur_svc->disabled = true; + else if (first == "console") + cur_svc->console = true; + else if (first == "user" || first == "group" || first == "class" || + first == "respawn" || first == "crash-threshold" || + first == "crash-window" || first == "setenv" || + first == "cwd" || first == "seclabel") { + if (toks.size() < 3 || toks[1] != "=") { + throw std::runtime_error(file + ":" + std::to_string(line) + + ": option '" + first + + "' requires '= value' syntax"); + } + if (first == "user") + cur_svc->uid = toks[2]; + else if (first == "group") { + cur_svc->gid = toks[2]; + for (size_t i = 3; i < toks.size(); ++i) + cur_svc->groups.push_back(toks[i]); + } else if (first == "class") + cur_svc->service_class = toks[2]; + else if (first == "respawn") + cur_svc->respawn = parse_respawn(toks[2]); + else if (first == "crash-threshold") + cur_svc->crash_threshold = std::stoi(toks[2]); + else if (first == "crash-window") + cur_svc->crash_window_secs = std::stoi(toks[2]); + else if (first == "setenv") + cur_svc->env.push_back(toks[2]); + else if (first == "cwd") + cur_svc->cwd = toks[2]; + else if (first == "seclabel") + cur_svc->seclabel = toks[2]; } - else if (first == "oneshot") cur_svc->oneshot = true; - else if (first == "disabled") cur_svc->disabled = true; - else if (first == "console") cur_svc->console = true; - else if (first == "class" && toks.size() >= 2) cur_svc->service_class = toks[1]; - else if (first == "respawn" && toks.size() >= 2) - cur_svc->respawn = parse_respawn(toks[1]); - else if (first == "crash-threshold" && toks.size() >= 2) - cur_svc->crash_threshold = std::stoi(toks[1]); - else if (first == "crash-window" && toks.size() >= 2) - cur_svc->crash_window_secs = std::stoi(toks[1]); - else if (first == "setenv" && toks.size() >= 2) cur_svc->env.push_back(toks[1]); - else if (first == "cwd" && toks.size() >= 2) cur_svc->cwd = toks[1]; + // unknown option keys are ignored. continue; } // action command (must be inside an action section). if (section_kind == "action" && cur_act) { Command cmd; - if (first == "start" && toks.size() >= 2) cmd.kind = Command::Kind::Start; - else if (first == "stop" && toks.size() >= 2) cmd.kind = Command::Kind::Stop; - else if (first == "restart" && toks.size() >= 2) cmd.kind = Command::Kind::Restart; - else if (first == "exec") cmd.kind = Command::Kind::Exec; - else if (first == "mkdir") cmd.kind = Command::Kind::Mkdir; - else if (first == "chmod") cmd.kind = Command::Kind::Chmod; - else if (first == "chown") cmd.kind = Command::Kind::Chown; - else if (first == "setenv") cmd.kind = Command::Kind::Setenv; - else if (first == "write") cmd.kind = Command::Kind::Write; - else if (first == "symlink") cmd.kind = Command::Kind::Symlink; - else if (first == "mount") cmd.kind = Command::Kind::Mount; - else if (first == "log") cmd.kind = Command::Kind::Log; + if (first == "start" && toks.size() >= 2) + cmd.kind = Command::Kind::Start; + else if (first == "stop" && toks.size() >= 2) + cmd.kind = Command::Kind::Stop; + else if (first == "restart" && toks.size() >= 2) + cmd.kind = Command::Kind::Restart; + else if (first == "exec") + cmd.kind = Command::Kind::Exec; + else if (first == "mkdir") + cmd.kind = Command::Kind::Mkdir; + else if (first == "chmod") + cmd.kind = Command::Kind::Chmod; + else if (first == "chown") + cmd.kind = Command::Kind::Chown; + else if (first == "setenv") + cmd.kind = Command::Kind::Setenv; + else if (first == "write") + cmd.kind = Command::Kind::Write; + else if (first == "symlink") + cmd.kind = Command::Kind::Symlink; + else if (first == "mount") + cmd.kind = Command::Kind::Mount; + else if (first == "log") + cmd.kind = Command::Kind::Log; else { throw std::runtime_error(file + ":" + std::to_string(line) + ": unknown action command '" + first + "'"); diff --git a/src/config.hpp b/src/config.hpp index 3ba3914..cf18fbe 100644 --- a/src/config.hpp +++ b/src/config.hpp @@ -4,12 +4,13 @@ // Two top-level constructs: // // service NAME /path/to/exec args... -// user root -// group root +// user = root # uid after the privilege drop +// group = root [g...] # primary gid + supplementary groups // oneshot // disabled -// class main -// respawn never|on-failure|always +// class = main +// respawn = never|on-failure|always +// seclabel = CONTEXT // console // // on TRIGGER @@ -30,6 +31,11 @@ namespace bajia { +// -------------------------------------------------------------------------- +// version +// -------------------------------------------------------------------------- +constexpr const char* kBajiaVersion = "0.1"; + // -------------------------------------------------------------------------- // Service // -------------------------------------------------------------------------- @@ -41,19 +47,24 @@ enum class RespawnPolicy { struct Service { std::string name; - std::vector args; // executable path + arguments + std::vector args; // executable path + arguments std::string cwd = "/"; - std::string uid = "root"; // resolved in supervisor + std::string uid = "root"; // resolved in supervisor std::string gid = "root"; std::vector groups; - bool oneshot = false; // run once, don't keep alive - bool disabled = false; // not started automatically - bool console = false; // bind stdio to the console + bool oneshot = false; // run once, don't keep alive + bool disabled = false; // not started automatically + bool console = false; // bind stdio to the console std::string service_class = "default"; RespawnPolicy respawn = RespawnPolicy::Always; - int crash_threshold = 4; // max restarts within window - int crash_window_secs = 30; // before giving up - std::vector env; // "K=V" pairs + int crash_threshold = 4; // max restarts within window + int crash_window_secs = 30; // before giving up + std::vector env; // "K=V" pairs + std::string seclabel; // SELinux exec context (optional) + + // Transient runtime flag: this service was stopped by a config reload + // because its definition changed; respawn it once with the new definition. + bool restart_on_reap = false; // Runtime state int pid = 0; @@ -69,7 +80,7 @@ struct Command { Start, Stop, Restart, - Exec, // run a synchronous command to completion + Exec, // run a synchronous command to completion Mkdir, Chmod, Chown, @@ -80,11 +91,11 @@ struct Command { Log, }; Kind kind; - std::vector args; // command-specific arguments + std::vector args; // command-specific arguments }; struct Action { - std::string trigger; // e.g. "boot", "early-init" + std::string trigger; // e.g. "boot", "early-init" std::vector commands; }; @@ -95,9 +106,11 @@ struct Config { std::vector services; std::vector actions; std::string hostname; + // the .rc files this config was parsed from (used for SIGHUP reload). + std::vector sources; - Service* find_service(const std::string& name); - const Service* find_service(const std::string& name) const; + Service* find_service(const std::string& name); + const Service* find_service(const std::string& name) const; }; // Parse a set of .rc files into a Config. Throws std::runtime_error on diff --git a/src/logger.cpp b/src/logger.cpp index cbf3e66..176e13d 100644 --- a/src/logger.cpp +++ b/src/logger.cpp @@ -6,6 +6,7 @@ #include #include #include +#include #include namespace bajia { @@ -14,31 +15,75 @@ namespace { std::mutex g_lock; int g_fd = -1; // console fd (>=0 when open) +int g_scr_fd = + -1; // mirrored VT fd (/dev/tty1); systemd-style boot text on the screen std::array g_ring; size_t g_ring_pos = 0; size_t g_ring_count = 0; LogLevel g_min_level = LogLevel::Info; +// /dev/console follows the kernel's *preferred* console; with +// `console=tty1 console=ttyS0` that is ttyS0, so the VGA screen would stay +// empty. Mirror to /dev/tty1 (the screen) whenever it is a different device. +void open_screen() { + if (g_scr_fd >= 0) + return; + int fd = ::open("/dev/tty1", O_WRONLY | O_NOCTTY | O_CLOEXEC); + if (fd < 0) + return; // retried lazily on the next write (devtmpfs may not + // be mounted yet during the very first boot actions) + if (g_fd >= 0) { + struct stat a{}, b{}; + if (::fstat(g_fd, &a) == 0 && ::fstat(fd, &b) == 0 && + a.st_rdev == b.st_rdev) { + ::close(fd); + return; // /dev/console is already the VT + } + } + g_scr_fd = fd; +} + const char* level_name(LogLevel l) { switch (l) { - case LogLevel::Debug: return "DBG"; - case LogLevel::Info: return "INF"; - case LogLevel::Warn: return "WRN"; - case LogLevel::Err: return "ERR"; + case LogLevel::Debug: + return "DBG"; + case LogLevel::Info: + return "INF"; + case LogLevel::Warn: + return "WRN"; + case LogLevel::Err: + return "ERR"; } return "???"; } +void raw_write(int fd, const std::string& s) { + size_t off = 0; + while (off < s.size()) { + ssize_t n = ::write(fd, s.data() + off, s.size() - off); + if (n < 0) + break; + off += static_cast(n); + } +} + void write_all(const std::string& s) { if (g_fd >= 0) { - size_t off = 0; - while (off < s.size()) { - ssize_t n = ::write(g_fd, s.data() + off, s.size() - off); - if (n < 0) break; - off += static_cast(n); - } + raw_write(g_fd, s); } else { - ::write(STDERR_FILENO, s.data(), s.size()); + raw_write(STDERR_FILENO, s); + } +} + +// status banners (`[ OK ]` / `[FAILED]` / welcome) go to /dev/console and +// are mirrored to the VT screen so a VM with a display shows boot progress; +// the timestamped [INF]/[WRN] log stream (`write_all`) stays console-only. +void write_status(const std::string& s) { + write_all(s); + if (g_fd >= 0) { + open_screen(); + if (g_scr_fd >= 0) + raw_write(g_scr_fd, s); } } @@ -71,9 +116,10 @@ void log_init(const std::string& console_path, LogLevel min_level) { void log_set_level(LogLevel level) { g_min_level = level; } void log_msg(LogLevel level, const std::string& tag, const std::string& msg) { - if (level < g_min_level) return; - std::string line = "[" + timestamp() + "] [" + level_name(level) + "] " + tag + ": " + - msg + "\n"; + if (level < g_min_level) + return; + std::string line = "[" + timestamp() + "] [" + level_name(level) + "] " + + tag + ": " + msg + "\n"; std::lock_guard lk(g_lock); if (g_fd < 0 && g_ring_count < g_ring.size()) { g_ring[g_ring_pos] = line; @@ -89,4 +135,23 @@ void log_msg(LogLevel level, const std::string& tag, const std::string& msg) { write_all(line); } +void log_status(LogStatus st, const std::string& msg) { + std::string line; + if (st == LogStatus::Banner) { + line = msg + "\n"; + } else { + bool tty = g_fd >= 0 && ::isatty(g_fd); + if (st == LogStatus::Ok) { + line = "[" + std::string(tty ? "\x1b[32m" : "") + " OK " + + (tty ? "\x1b[0m" : "") + "] "; + } else { + line = "[" + std::string(tty ? "\x1b[31m" : "") + "FAILED" + + (tty ? "\x1b[0m" : "") + "] "; + } + line += msg + "\n"; + } + std::lock_guard lk(g_lock); + write_status(line); +} + } // namespace bajia diff --git a/src/logger.hpp b/src/logger.hpp index fafda3b..2076ae9 100644 --- a/src/logger.hpp +++ b/src/logger.hpp @@ -10,12 +10,23 @@ namespace bajia { -enum class LogLevel { Debug = 0, Info = 1, Warn = 2, Err = 3 }; +enum class LogLevel { Debug = 0, + Info = 1, + Warn = 2, + Err = 3 }; -void log_init(const std::string& console_path = "/dev/console", LogLevel min_level = LogLevel::Info); +// high-visibility console status banners in the systemd `[ OK ]` style. +// console-only (not kmsg/ring-buffer), colored when the console is a tty. +enum class LogStatus { Banner, + Ok, + Failed }; + +void log_init(const std::string& console_path = "/dev/console", + LogLevel min_level = LogLevel::Info); void log_set_level(LogLevel level); void log_msg(LogLevel level, const std::string& tag, const std::string& msg); +void log_status(LogStatus status, const std::string& msg); template void log_info(const std::string& tag, Args&&... args) { diff --git a/src/main.cpp b/src/main.cpp index 05da2e2..f9f46a1 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -5,6 +5,7 @@ // `init=/path/to/bajia`. #include "config.hpp" #include "logger.hpp" +#include "selinux_setup.hpp" #include "supervisor.hpp" #include @@ -37,26 +38,28 @@ int main(int argc, char** argv) { // PID 1 note: the kernel may pass extra args after the init program name. for (int i = 1; i < argc; ++i) { if (argv[i][0] == '-') { - if (std::strcmp(argv[i], "-h") == 0 || std::strcmp(argv[i], "--help") == 0) { + if (std::strcmp(argv[i], "-h") == 0 || + std::strcmp(argv[i], "--help") == 0) { usage(argv[0]); return 0; } + if (std::strcmp(argv[i], "--run-as-user") == 0) { run_as_user = true; continue; } + usage(argv[0]); return 2; } + files.emplace_back(argv[i]); } if (::getpid() != 1 && !run_as_user) { ::fprintf(stderr, - "bajia: refusing to run as pid %d (not PID 1). bajia is an " - "init system and would re-fire boot triggers, mount " - "filesystems and spawn services on top of a running system. " - "Launch it via the kernel (init=...), or pass --run-as-user " + "bajia: refusing to run as pid %d (not PID 1). " + "launch it via the kernel (init=...), or pass --run-as-user " "for a development run.\n", ::getpid()); return 1; @@ -64,9 +67,11 @@ int main(int argc, char** argv) { if (files.empty()) { // default: prefer the single init.rc, and also load /etc/bajia.d/*.rc. - if (::access("/etc/bajia/init.rc", R_OK) == 0) files.emplace_back("/etc/bajia/init.rc"); + if (::access("/etc/bajia/init.rc", R_OK) == 0) + files.emplace_back("/etc/bajia/init.rc"); if (files.empty()) { - ::fprintf(stderr, "bajia: no config given and /etc/bajia/init.rc not found.\n"); + ::fprintf(stderr, + "bajia: no config given and /etc/bajia/init.rc not found.\n"); return 1; } } @@ -79,12 +84,19 @@ int main(int argc, char** argv) { return 1; } - // logger targets the console; falls back to stderr until the console is ready. + // logger targets the console; falls back to stderr until the console is + // ready. log_init("/dev/console", LogLevel::Info); + log_status(LogStatus::Banner, + std::string("Welcome to bajia ") + kBajiaVersion); log_info("init", "loaded ", std::to_string(files.size()), " config file(s), ", std::to_string(config.services.size()), " services, ", std::to_string(config.actions.size()), " actions"); + if (::getpid() == 1) { + selinux_setup(); + } + Supervisor supervisor(std::move(config)); supervisor.run(); // [[noreturn]] } diff --git a/src/selinux_setup.cpp b/src/selinux_setup.cpp new file mode 100644 index 0000000..d91e92a --- /dev/null +++ b/src/selinux_setup.cpp @@ -0,0 +1,88 @@ +// selinux_setup.cpp - setup SELinux policies (only when built with +// `configure.py --selinux`; otherwise this compiles to an empty TU). +#include "selinux_setup.hpp" + +#ifdef BAJIA_SELINUX + +#include "logger.hpp" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace bajia { + +int selinux_log_callback(int type, const char* fmt, ...) { + (void)type; + va_list ap; + va_start(ap, fmt); + + va_list copy; + va_copy(copy, ap); + int len = vsnprintf(nullptr, 0, fmt, copy); + va_end(copy); + + if (len >= 0) { + std::string msg(static_cast(len), '\0'); + vsnprintf(msg.data(), msg.size() + 1, fmt, ap); + log_info("selinux", msg.c_str()); + } + + va_end(ap); + return 0; +} + +int selinux_audit_callback(void* auditdata, security_class_t cls, char* msg, + size_t msglen) { + (void)auditdata; + (void)msglen; + log_info("selinux-audit", "class: ", std::to_string(cls), " msg: ", msg); + return 0; +} + +void selinux_setup() { + selinux_callback cb{}; + cb.func_log = selinux_log_callback; + selinux_set_callback(SELINUX_CB_LOG, cb); + cb.func_audit = selinux_audit_callback; + selinux_set_callback(SELINUX_CB_AUDIT, cb); + + // make sure /sys is mounted, then let selinux_init_load_policy() do the + // real work: it mounts selinuxfs, parses /etc/selinux/config + the + // enforcing= cmdline flag and loads the policy. (libselinux's own + // is_selinux_enabled() needs a prior config parse, so we don't gate on it + // here; its return value is authoritative.) + ::mkdir("/sys", 0755); + if (::mount("sysfs", "/sys", "sysfs", 0, nullptr) != 0 && errno != EBUSY) { + log_info("selinux", "cannot mount sysfs: ", std::strerror(errno)); + return; + } + + int enforce = -1; + if (selinux_init_load_policy(&enforce) != 0) { + log_info("selinux", + "policy load failed (no selinuxfs, kernel disabled " + "via selinux=0, and/or no policy in /etc/selinux//policy)"); + return; + } + log_info("selinux", + "policy loaded, enforcing=", std::to_string(enforce == 1)); + + // relabel the core tree so every path matches the policy's file_contexts. + for (const char* path : {"/dev", "/run", "/etc", "/bin", "/sbin", "/usr"}) { + if (selinux_restorecon(path, SELINUX_RESTORECON_RECURSE) != 0) { + log_info("selinux", "restorecon ", path, ": ", std::strerror(errno)); + } + } + log_info("selinux", "setup complete"); +} + +} // namespace bajia + +#endif // BAJIA_SELINUX diff --git a/src/selinux_setup.hpp b/src/selinux_setup.hpp new file mode 100644 index 0000000..eef1ff6 --- /dev/null +++ b/src/selinux_setup.hpp @@ -0,0 +1,29 @@ +#pragma once + +// SELinux integration is optional: it pulls in libselinux, which is only +// useful on a real Android-like system with policies. Compile with +// `configure.py --selinux` to enable; otherwise these are no-ops so the +// embedded/VM builds stay self-contained. +#ifdef BAJIA_SELINUX + +#include + +namespace bajia { + +int selinux_log_callback(int type, const char* fmt, ...); +int selinux_audit_callback(void* auditdata, security_class_t cls, char* msg, + size_t msglen); + +void selinux_setup(); + +} // namespace bajia + +#else // !BAJIA_SELINUX + +namespace bajia { + +inline void selinux_setup() {} + +} // namespace bajia + +#endif // BAJIA_SELINUX \ No newline at end of file diff --git a/src/supervisor.cpp b/src/supervisor.cpp index 6116674..c8a9315 100644 --- a/src/supervisor.cpp +++ b/src/supervisor.cpp @@ -3,7 +3,10 @@ #include "logger.hpp" +#include #include +#include +#include #include #include #include @@ -16,11 +19,17 @@ #include #include #include +#include #include #include #include #include #include +#include +#include +#ifdef BAJIA_SELINUX +#include +#endif namespace bajia { @@ -30,33 +39,89 @@ constexpr const char* kTag = "supv"; // shutdown timing constexpr int kStopGraceSecs = 5; // wait this long for a clean SIGTERM exit -constexpr int kKillGraceSecs = 2; // then this long after SIGKILL before giving up +constexpr int kKillGraceSecs = + 2; // then this long after SIGKILL before giving up // console fd shared with spawned services flagged `console`. int g_open_console_fd = -1; +// resolves a uid/gid: numeric values pass through, names are looked up via +// getpwnam/getgrnam (reads /etc/passwd, /etc/group). (uid_t)-1 / (gid_t)-1 +// means "not found". +uid_t resolve_user(const std::string& s) { + if (!s.empty()) { + char* end = nullptr; + long v = std::strtol(s.c_str(), &end, 10); + if (end != s.c_str() && *end == '\0') + return static_cast(v); + } + const struct passwd* pw = ::getpwnam(s.c_str()); + return pw ? pw->pw_uid : static_cast(-1); +} + +gid_t resolve_group(const std::string& s) { + if (!s.empty()) { + char* end = nullptr; + long v = std::strtol(s.c_str(), &end, 10); + if (end != s.c_str() && *end == '\0') + return static_cast(v); + } + const struct group* g = ::getgrnam(s.c_str()); + return g ? g->gr_gid : static_cast(-1); +} + +// true when a service's definition differs between the live and a reloaded +// config (name is matched separately by the caller). +bool service_changed(const Service& a, const Service& b) { + return a.args != b.args || a.cwd != b.cwd || a.uid != b.uid || + a.gid != b.gid || a.groups != b.groups || a.oneshot != b.oneshot || + a.disabled != b.disabled || a.console != b.console || + a.service_class != b.service_class || a.respawn != b.respawn || + a.crash_threshold != b.crash_threshold || + a.crash_window_secs != b.crash_window_secs || a.env != b.env || + a.seclabel != b.seclabel; +} + std::string join(const std::vector& v, const std::string& sep) { std::string out; for (size_t i = 0; i < v.size(); ++i) { - if (i) out += sep; + if (i) + out += sep; out += v[i]; } return out; } +std::string join_gids(const std::vector& v) { + std::string out; + for (size_t i = 0; i < v.size(); ++i) { + if (i) + out += ','; + out += std::to_string(v[i]); + } + return out; +} + } // namespace Supervisor::Supervisor(Config config) : config_(std::move(config)) {} Supervisor::~Supervisor() { - if (sigfd_ >= 0) ::close(sigfd_); - if (epfd_ >= 0) ::close(epfd_); + for (auto& [fd, buf] : ctl_clients_) + ::close(fd); + ctl_clients_.clear(); + if (ctl_fd_ >= 0) + ::close(ctl_fd_); + if (sigfd_ >= 0) + ::close(sigfd_); + if (epfd_ >= 0) + ::close(epfd_); } void Supervisor::setup_signals() { sigset_t mask; sigemptyset(&mask); - + // block these for ALL threads/processes; signalfd delivers them to us. sigaddset(&mask, SIGCHLD); sigaddset(&mask, SIGTERM); @@ -80,7 +145,7 @@ void Supervisor::setup_signals() { _exit(1); } - struct epoll_event ev {}; + struct epoll_event ev{}; ev.events = EPOLLIN; ev.data.fd = sigfd_; if (::epoll_ctl(epfd_, EPOLL_CTL_ADD, sigfd_, &ev) != 0) { @@ -99,12 +164,39 @@ void Supervisor::spawn_service(Service& svc, bool missing_ok) { return; } - if (::access(svc.args[0].c_str(), X_OK) != 0) { - if (missing_ok) { - log_info(kTag, svc.name, ": executable missing (", svc.args[0], "), ignored"); + // resolve the identity before forking so a bad lookup never leaves a + // half-forked child around. + const uid_t uid = resolve_user(svc.uid); + if (uid == static_cast(-1)) { + log_info(kTag, svc.name, ": unknown user '", svc.uid, "'"); + return; + } + const gid_t gid = resolve_group(svc.gid); + if (gid == static_cast(-1)) { + log_info(kTag, svc.name, ": unknown group '", svc.gid, "'"); + return; + } + std::vector supp; + supp.reserve(svc.groups.size()); + for (const auto& g : svc.groups) { + const gid_t gg = resolve_group(g); + if (gg == static_cast(-1)) { + log_info(kTag, svc.name, ": unknown group '", g, "'"); return; } - log_info(kTag, svc.name, ": cannot exec ", svc.args[0], ": ", std::strerror(errno)); + supp.push_back(gg); + } + log_info(kTag, svc.name, ": uid=", std::to_string(uid), " gid=", + std::to_string(gid), " groups=[", join_gids(supp), "]"); + + if (::access(svc.args[0].c_str(), X_OK) != 0) { + if (missing_ok) { + log_info(kTag, svc.name, ": executable missing (", svc.args[0], + "), ignored"); + return; + } + log_info(kTag, svc.name, ": cannot exec ", svc.args[0], ": ", + std::strerror(errno)); return; } @@ -116,32 +208,66 @@ void Supervisor::spawn_service(Service& svc, bool missing_ok) { if (pid == 0) { // child + // PID 1 blocks SIGTERM/SIGINT/etc for its signalfd; unblock before + // exec or every service would run with those signals permanently + // blocked (bctl stop / shutdown / reload restarts would never land). + sigset_t empty; + sigemptyset(&empty); + ::sigprocmask(SIG_SETMASK, &empty, nullptr); + if (svc.console && g_open_console_fd >= 0) { ::dup2(g_open_console_fd, 0); ::dup2(g_open_console_fd, 1); ::dup2(g_open_console_fd, 2); } - if (::chdir(svc.cwd.c_str()) != 0) { - _exit(126); - } - // environment: inherit, then apply K=V entries. std::vector kvs = svc.env; std::vector envp; - for (char** e = environ; e && *e; ++e) envp.push_back(*e); - for (auto& kv : kvs) envp.push_back(const_cast(kv.c_str())); + for (char** e = environ; e && *e; ++e) + envp.push_back(*e); + for (auto& kv : kvs) + envp.push_back(const_cast(kv.c_str())); envp.push_back(nullptr); std::vector argv; - for (auto& a : svc.args) argv.push_back(const_cast(a.c_str())); + for (auto& a : svc.args) + argv.push_back(const_cast(a.c_str())); argv.push_back(nullptr); - // uid/gid handling intentionally left to a drop-privileges pass; - // for a first version we run as-is. Placeholder to avoid unused warn. - (void)svc.uid; - (void)svc.gid; - (void)svc.groups; +#ifdef BAJIA_SELINUX + if (!svc.seclabel.empty()) { + if (::setexeccon(svc.seclabel.c_str()) != 0) { + log_info(kTag, "setexeccon(", svc.seclabel, + ") failed: ", std::strerror(errno)); + } + } +#else + (void)svc.seclabel; +#endif + + // privilege drop: supplementary groups first, then gid, then uid. + const int gr = ::setgroups(supp.size(), supp.data()); + if (gr != 0) { + log_info(kTag, svc.name, ": setgroups failed: ", std::strerror(errno)); + _exit(126); + } + if (::setgid(gid) != 0) { + log_info(kTag, svc.name, ": setgid(", std::to_string(gid), + ") failed: ", std::strerror(errno)); + _exit(126); + } + if (::setuid(uid) != 0) { + log_info(kTag, svc.name, ": setuid(", std::to_string(uid), + ") failed: ", std::strerror(errno)); + _exit(126); + } + + if (::chdir(svc.cwd.c_str()) != 0) { + log_info(kTag, svc.name, ": chdir(", svc.cwd, + ") failed: ", std::strerror(errno)); + _exit(126); + } ::execvpe(argv[0], argv.data(), envp.data()); // exec failed in child @@ -152,6 +278,11 @@ void Supervisor::spawn_service(Service& svc, bool missing_ok) { svc.pid = pid; svc.running = true; log_info(kTag, svc.name, " started (pid ", std::to_string(pid), ")"); + // status banner only for explicit starts (missing_ok=false == started via + // `start NAME`); crash respawns are suppressed so fault loops stay quiet. + if (!missing_ok) { + log_status(LogStatus::Ok, "Started " + svc.name + "."); + } } void Supervisor::start_service(const std::string& name) { @@ -160,13 +291,15 @@ void Supervisor::start_service(const std::string& name) { log_info(kTag, "start ", name, ": no such service"); return; } - if (svc->running) return; + if (svc->running) + return; spawn_service(*svc, false); } void Supervisor::stop_service(const std::string& name, bool kill) { Service* svc = config_.find_service(name); - if (!svc || !svc->running) return; + if (!svc || !svc->running) + return; if (svc->pid > 0) { ::kill(svc->pid, kill ? SIGKILL : SIGTERM); } @@ -174,14 +307,94 @@ void Supervisor::stop_service(const std::string& name, bool kill) { void Supervisor::restart_service(const std::string& name) { Service* svc = config_.find_service(name); - if (!svc) return; + if (!svc) + return; if (svc->running && svc->pid > 0) { ::kill(svc->pid, SIGTERM); - // It will be respawned by reap logic for non-oneshot services; for + // it will be respawned by reap logic for non-oneshot services; for // simplicity, mark for immediate respawn below. } // If not running, start now. - if (!svc->running) spawn_service(*svc, false); + if (!svc->running) + spawn_service(*svc, false); +} + +void Supervisor::reload_config() { + if (shutdown_requested_) { + log_info(kTag, "reload ignored (shutdown in progress)"); + return; + } + Config fresh; + try { + fresh = parse_config(config_.sources); + } catch (const std::exception& e) { + log_status(LogStatus::Failed, + "Config reload denied: " + std::string(e.what())); + return; + } + log_info(kTag, "reload: ", std::to_string(config_.services.size()), " -> ", + std::to_string(fresh.services.size()), " services"); + log_status(LogStatus::Ok, "Reloading configuration."); + + // 1. removed services: stop and drop from the live table so the reaper + // treats their (reparented) exit as an orphan. + const auto removed = [&]() { + std::vector names; + for (const auto& svc : config_.services) { + if (!fresh.find_service(svc.name)) + names.push_back(svc.name); + } + return names; + }(); + for (const auto& name : removed) { + Service* svc = config_.find_service(name); + if (svc && svc->running && svc->pid > 0) { + log_info(kTag, "reload: stopping removed service ", name, " (pid ", + std::to_string(svc->pid), ")"); + ::kill(svc->pid, SIGTERM); + } + config_.services.erase( + std::remove_if(config_.services.begin(), config_.services.end(), + [&](const Service& s) { return s.name == name; }), + config_.services.end()); + } + + // 2. added / changed services. Runtime state (pid, running) is preserved + // across a definition swap; changed-but-running services are stopped and + // respawned with the new definition by the reaper. + for (auto& nsvc : fresh.services) { + Service* osvc = config_.find_service(nsvc.name); + if (!osvc) { + log_info(kTag, "reload: added service ", nsvc.name); + config_.services.push_back(std::move(nsvc)); + continue; + } + if (!service_changed(*osvc, nsvc)) { + log_info(kTag, "reload: unchanged ", nsvc.name); + continue; + } + const bool was_running = osvc->running && osvc->pid > 0; + const int pid = osvc->pid; + const bool running = osvc->running; + const int exit_code = osvc->exit_code; + *osvc = std::move(nsvc); + osvc->pid = pid; + osvc->running = running; + osvc->exit_code = exit_code; + if (was_running) { + osvc->restart_on_reap = true; + log_info(kTag, "reload: restarting ", osvc->name, + " (definition changed, pid ", std::to_string(pid), ")"); + ::kill(pid, SIGTERM); + } else { + log_info(kTag, "reload: updated definition of ", osvc->name); + } + } + + // 3. future actions/hostname use the fresh parse; already-fired triggers + // are not replayed. + config_.actions = std::move(fresh.actions); + config_.hostname = std::move(fresh.hostname); } void Supervisor::reap_children() { @@ -193,26 +406,42 @@ void Supervisor::reap_children() { for (auto& svc : config_.services) { if (svc.pid == pid) { matched = true; - int code = WIFEXITED(status) ? WEXITSTATUS(status) - : (WIFSIGNALED(status) ? 128 + WTERMSIG(status) : -1); + int code = WIFEXITED(status) + ? WEXITSTATUS(status) + : (WIFSIGNALED(status) ? 128 + WTERMSIG(status) : -1); svc.exit_code = code; svc.pid = 0; svc.running = false; log_info(kTag, svc.name, " exited with code ", std::to_string(code)); bool success = (code == 0); + if (svc.restart_on_reap) { + // stopped by a config reload; respawn once with the new + // definition regardless of oneshot/respawn policy. + svc.restart_on_reap = false; + if (!shutdown_requested_) + spawn_service(svc, true); + break; + } if (svc.oneshot) { // oneshot services terminate on their own; never respawn. continue; } bool should_respawn = false; switch (svc.respawn) { - case RespawnPolicy::Always: should_respawn = true; break; - case RespawnPolicy::OnFailure: should_respawn = !success; break; - case RespawnPolicy::Never: should_respawn = false; break; + case RespawnPolicy::Always: + should_respawn = true; + break; + case RespawnPolicy::OnFailure: + should_respawn = !success; + break; + case RespawnPolicy::Never: + should_respawn = false; + break; } if (should_respawn) { - if (shutdown_requested_) break; + if (shutdown_requested_) + break; spawn_service(svc, true); } break; @@ -227,12 +456,18 @@ void Supervisor::reap_children() { void Supervisor::run_exec_command(const Command& cmd) { // synchronously run an action 'exec' and wait for it to complete. - if (cmd.args.empty()) return; + if (cmd.args.empty()) + return; pid_t pid = ::fork(); - if (pid < 0) return; + if (pid < 0) + return; if (pid == 0) { + sigset_t empty; + sigemptyset(&empty); + ::sigprocmask(SIG_SETMASK, &empty, nullptr); std::vector argv; - for (auto& a : cmd.args) argv.push_back(const_cast(a.c_str())); + for (auto& a : cmd.args) + argv.push_back(const_cast(a.c_str())); argv.push_back(nullptr); ::execvpe(argv[0], argv.data(), environ); _exit(127); @@ -243,80 +478,325 @@ void Supervisor::run_exec_command(const Command& cmd) { WIFEXITED(status) ? std::to_string(WEXITSTATUS(status)) : "signal"); } -void Supervisor::run_command(Command& cmd) { +bool Supervisor::run_command(Command& cmd) { using K = Command::Kind; switch (cmd.kind) { - case K::Start: - if (!cmd.args.empty()) start_service(cmd.args[0]); - break; - case K::Stop: - if (!cmd.args.empty()) stop_service(cmd.args[0], true); - break; - case K::Restart: - if (!cmd.args.empty()) restart_service(cmd.args[0]); - break; - case K::Exec: - run_exec_command(cmd); - break; - case K::Mkdir: - if (cmd.args.size() >= 1) { - mode_t m = cmd.args.size() >= 2 ? static_cast(std::stoul(cmd.args[1], nullptr, 8)) : 0755; - ::mkdir(cmd.args[0].c_str(), m); + case K::Start: + if (!cmd.args.empty()) + start_service(cmd.args[0]); + return true; + case K::Stop: + if (!cmd.args.empty()) + stop_service(cmd.args[0], true); + return true; + case K::Restart: + if (!cmd.args.empty()) + restart_service(cmd.args[0]); + return true; + case K::Exec: + run_exec_command(cmd); + return true; + case K::Mkdir: { + if (cmd.args.size() >= 1) { + mode_t m = cmd.args.size() >= 2 + ? static_cast(std::stoul(cmd.args[1], nullptr, 8)) + : 0755; + int rc = ::mkdir(cmd.args[0].c_str(), m); + int e = errno; + if (rc != 0 && e != EEXIST) { + log_status(LogStatus::Failed, + "Failed to mkdir " + cmd.args[0] + ": " + std::strerror(e)); + return false; } - break; - case K::Chmod: - if (cmd.args.size() >= 2) ::chmod(cmd.args[0].c_str(), - static_cast(std::stoul(cmd.args[1], nullptr, 8))); - break; - case K::Chown: - if (cmd.args.size() >= 3) { - uid_t uid = static_cast(std::stoul(cmd.args[1])); - gid_t gid = static_cast(std::stoul(cmd.args[2])); - ::chown(cmd.args[0].c_str(), uid, gid); - } - break; - case K::Setenv: - if (cmd.args.size() >= 1) ::setenv(cmd.args[0].c_str(), cmd.args[1].c_str(), 1); - break; - case K::Write: { - if (cmd.args.size() >= 2) { - int fd = ::open(cmd.args[0].c_str(), O_WRONLY | O_CREAT | O_TRUNC, 0644); - if (fd >= 0) { - ::write(fd, cmd.args[1].c_str(), cmd.args[1].size()); - ::close(fd); - } - } - break; + return true; // created, or already present (EEXIST) } - case K::Symlink: - if (cmd.args.size() >= 2) ::symlink(cmd.args[0].c_str(), cmd.args[1].c_str()); - break; - case K::Mount: - if (cmd.args.size() >= 3) ::mount(cmd.args[0].c_str(), cmd.args[1].c_str(), - cmd.args[2].c_str(), 0, nullptr); - break; - case K::Log: - log_info(kTag, "action: ", join(cmd.args, " ")); - break; + return true; } + case K::Chmod: { + if (cmd.args.size() >= 2) { + int rc = + ::chmod(cmd.args[0].c_str(), + static_cast(std::stoul(cmd.args[1], nullptr, 8))); + int e = errno; + if (rc != 0) + log_status(LogStatus::Failed, + "Failed to chmod " + cmd.args[0] + ": " + std::strerror(e)); + return rc == 0; + } + return true; + } + case K::Chown: { + if (cmd.args.size() >= 3) { + uid_t uid = static_cast(std::stoul(cmd.args[1])); + gid_t gid = static_cast(std::stoul(cmd.args[2])); + int rc = ::chown(cmd.args[0].c_str(), uid, gid); + int e = errno; + if (rc != 0) + log_status(LogStatus::Failed, + "Failed to chown " + cmd.args[0] + ": " + std::strerror(e)); + return rc == 0; + } + return true; + } + case K::Setenv: + if (cmd.args.size() >= 1) + ::setenv(cmd.args[0].c_str(), cmd.args[1].c_str(), 1); + return true; + case K::Write: { + if (cmd.args.size() >= 2) { + int fd = ::open(cmd.args[0].c_str(), O_WRONLY | O_CREAT | O_TRUNC, 0644); + if (fd < 0) { + int e = errno; + log_status(LogStatus::Failed, + "Failed to write " + cmd.args[0] + ": " + std::strerror(e)); + return false; + } + bool ok = ::write(fd, cmd.args[1].c_str(), cmd.args[1].size()) >= 0; + int e = errno; + ::close(fd); + if (!ok) { + log_status(LogStatus::Failed, + "Failed to write " + cmd.args[0] + ": " + std::strerror(e)); + } + return ok; + } + return true; + } + case K::Symlink: { + if (cmd.args.size() >= 2) { + int rc = ::symlink(cmd.args[0].c_str(), cmd.args[1].c_str()); + int e = errno; + if (rc != 0) + log_status(LogStatus::Failed, "Failed to symlink " + cmd.args[1] + + ": " + std::strerror(e)); + return rc == 0; + } + return true; + } + case K::Mount: { + if (cmd.args.size() >= 3) { + int rc = ::mount(cmd.args[0].c_str(), cmd.args[1].c_str(), + cmd.args[2].c_str(), 0, nullptr); + int e = errno; + if (rc != 0 && e != EBUSY) { + log_status(LogStatus::Failed, "Failed to mount " + cmd.args[1] + " (" + + cmd.args[2] + + "): " + std::strerror(e)); + return false; + } + return true; // success, or already mounted (EBUSY) + } + return true; + } + case K::Log: + log_info(kTag, "action: ", join(cmd.args, " ")); + return true; + } + return true; } void Supervisor::execute_action(Action& action) { log_info(kTag, "trigger: ", action.trigger); + bool ok = true; for (auto& cmd : action.commands) { - run_command(cmd); + if (!run_command(cmd)) + ok = false; + } + if (action.trigger == "shutdown") + return; // no OK banner at the end of life + if (ok) { + log_status(LogStatus::Ok, "Reached target '" + action.trigger + "'."); + } else { + log_status(LogStatus::Failed, "Failed to reach target '" + action.trigger + + "' (see errors above)."); } } // The console fd, opened once PID1 realizes it's on a real console. Provided // so spawn_service can rebind stdio for services flagged `console`. void Supervisor::open_console() { - if (g_open_console_fd >= 0) return; + if (g_open_console_fd >= 0) + return; g_open_console_fd = ::open("/dev/console", O_RDWR | O_NOCTTY | O_CLOEXEC); } +namespace { + +// abstract unix socket name for the control channel (no filesystem entry). +constexpr const char kCtlName[] = "bajia"; + +std::string status_line(const Service& svc) { + std::string out = svc.name; + if (svc.running && svc.pid > 0) { + out += " running pid " + std::to_string(svc.pid); + } else { + out += " stopped (last exit " + std::to_string(svc.exit_code) + ")"; + } + out += svc.oneshot ? " oneshot" : ""; + out += " class " + svc.service_class; + return out + "\n"; +} + +} // namespace + +void Supervisor::open_control_socket() { + ctl_fd_ = ::socket(AF_UNIX, SOCK_STREAM | SOCK_NONBLOCK | SOCK_CLOEXEC, 0); + if (ctl_fd_ < 0) { + log_info(kTag, "ctl: socket: ", std::strerror(errno)); + return; + } + + sockaddr_un addr{}; + addr.sun_family = AF_UNIX; + std::memcpy(addr.sun_path + 1, kCtlName, + sizeof kCtlName); // leading NUL = abstract + const socklen_t len = static_cast(offsetof(sockaddr_un, sun_path) + + 1 + sizeof kCtlName); + if (::bind(ctl_fd_, reinterpret_cast(&addr), len) != 0) { + log_info(kTag, "ctl: bind @", kCtlName, ": ", std::strerror(errno)); + ::close(ctl_fd_); + ctl_fd_ = -1; + return; + } + if (::listen(ctl_fd_, 4) != 0) { + log_info(kTag, "ctl: listen: ", std::strerror(errno)); + ::close(ctl_fd_); + ctl_fd_ = -1; + return; + } + epoll_event ev{}; + ev.events = EPOLLIN; + ev.data.fd = ctl_fd_; + if (::epoll_ctl(epfd_, EPOLL_CTL_ADD, ctl_fd_, &ev) != 0) { + log_info(kTag, "ctl: epoll_ctl: ", std::strerror(errno)); + ::close(ctl_fd_); + ctl_fd_ = -1; + return; + } + log_info(kTag, "control socket ready (@", kCtlName, ")"); +} + +void Supervisor::ctl_accept() { + for (;;) { + int fd = ::accept4(ctl_fd_, nullptr, nullptr, SOCK_NONBLOCK | SOCK_CLOEXEC); + if (fd < 0) { + if (errno == EAGAIN || errno == EINTR) + return; + log_info(kTag, "ctl: accept: ", std::strerror(errno)); + return; + } + epoll_event ev{}; + ev.events = EPOLLIN; + ev.data.fd = fd; + if (::epoll_ctl(epfd_, EPOLL_CTL_ADD, fd, &ev) != 0) { + ::close(fd); + continue; + } + ctl_clients_[fd] = std::string(); + } +} + +void Supervisor::ctl_handle_client(int fd) { + auto drop = [&]() { + epoll_ctl(epfd_, EPOLL_CTL_DEL, fd, nullptr); + ctl_clients_.erase(fd); + ::close(fd); + }; + + char buf[256]; + for (;;) { + const ssize_t n = ::read(fd, buf, sizeof buf); + if (n < 0) { + if (errno == EAGAIN) + return; // no more pending input + drop(); // read error + return; + } + if (n == 0) { // peer closed + drop(); + return; + } + ctl_clients_[fd].append(buf, static_cast(n)); + + size_t pos; + while ((pos = ctl_clients_[fd].find('\n')) != std::string::npos) { + std::string line = ctl_clients_[fd].substr(0, pos); + ctl_clients_[fd].erase(0, pos + 1); + const std::string resp = ctl_execute(line); + size_t off = 0; + while (off < resp.size()) { // small responses; best-effort + const ssize_t w = ::write(fd, resp.data() + off, resp.size() - off); + if (w <= 0) { + drop(); + return; + } + off += static_cast(w); + } + } + } +} + +std::string Supervisor::ctl_execute(const std::string& line) { + std::vector toks; + if (!line.empty()) { + char* save = nullptr; + std::string copy = line; + for (char* p = ::strtok_r(copy.data(), " ", &save); p; + p = ::strtok_r(nullptr, " ", &save)) { + toks.emplace_back(p); + } + } + if (toks.empty()) + return "OK\n"; + + const std::string& cmd = toks[0]; + if (cmd == "ping") + return "OK pong\n"; + + if (cmd == "start" && toks.size() >= 2) { + start_service(toks[1]); + return "OK\n"; + } + if (cmd == "stop" && toks.size() >= 2) { + stop_service(toks[1], false); // graceful SIGTERM + return "OK\n"; + } + if (cmd == "restart" && toks.size() >= 2) { + restart_service(toks[1]); + return "OK\n"; + } + if (cmd == "trigger" && toks.size() >= 2) { + for (auto& action : config_.actions) { + if (action.trigger == toks[1]) + execute_action(action); + } + return "OK\n"; + } + if (cmd == "status") { + std::string out; + for (const auto& svc : config_.services) + out += status_line(svc); + return out + "OK\n"; + } + if (cmd == "shutdown") { + const ShutdownKind kind = toks.size() >= 2 && toks[1] == "bootloader-reboot" + ? ShutdownKind::Reboot + : toks.size() >= 2 && toks[1] == "reboot" + ? ShutdownKind::Reboot + : ShutdownKind::PowerOff; + begin_shutdown(kind); + return "OK\n"; + } + if (cmd == "reload") { + reload_config(); + return "OK\n"; + } + return "ERR unknown command: " + cmd + "\n"; +} + void Supervisor::begin_shutdown(ShutdownKind kind) { - if (shutdown_requested_) return; // already winding down + if (shutdown_requested_) + return; // already winding down shutdown_requested_ = true; shutdown_kind_ = kind; shutdown_state_ = ShutdownState::FiringActions; @@ -326,25 +806,30 @@ void Supervisor::begin_shutdown(ShutdownKind kind) { bool Supervisor::any_running() const { for (const auto& svc : config_.services) { - if (svc.running) return true; + if (svc.running) + return true; } return false; } int Supervisor::shutdown_timeout_ms() const { // millis until the next state transition, or -1 for "wait forever". - if (shutdown_state_ == ShutdownState::Running) return -1; + if (shutdown_state_ == ShutdownState::Running) + return -1; const auto now = std::chrono::steady_clock::now(); - if (now >= shutdown_deadline_) return 0; + if (now >= shutdown_deadline_) + return 0; const auto ms = std::chrono::duration_cast( - shutdown_deadline_ - now).count(); + shutdown_deadline_ - now) + .count(); return ms >= INT_MAX ? INT_MAX : static_cast(ms); } void Supervisor::unmount_filesystems() { FILE* f = ::fopen("/proc/self/mounts", "r"); if (!f) { - log_info(kTag, "unmount: cannot open /proc/self/mounts: ", std::strerror(errno)); + log_info(kTag, + "unmount: cannot open /proc/self/mounts: ", std::strerror(errno)); return; } std::vector mounts; // mountpoints in mount order @@ -356,7 +841,8 @@ void Supervisor::unmount_filesystems() { p = ::strtok_r(nullptr, " \t\n", &save)) { toks.emplace_back(p); } - if (toks.size() >= 2 && toks[1] != "/") mounts.push_back(toks[1]); + if (toks.size() >= 2 && toks[1] != "/") + mounts.push_back(toks[1]); } ::fclose(f); for (auto it = mounts.rbegin(); it != mounts.rend(); ++it) { // deepest last @@ -370,71 +856,73 @@ void Supervisor::unmount_filesystems() { bool Supervisor::advance_shutdown() { switch (shutdown_state_) { - case ShutdownState::Running: - return false; + case ShutdownState::Running: + return false; - case ShutdownState::FiringActions: - log_info(kTag, "firing shutdown actions"); - for (auto& action : config_.actions) { - if (action.trigger == "shutdown") execute_action(action); + case ShutdownState::FiringActions: + log_info(kTag, "firing shutdown actions"); + for (auto& action : config_.actions) { + if (action.trigger == "shutdown") + execute_action(action); + } + for (auto& svc : config_.services) { + if (svc.running && svc.pid > 0) { + log_info(kTag, "stopping ", svc.name, " (pid ", std::to_string(svc.pid), + ")"); + ::kill(svc.pid, SIGTERM); } + } + shutdown_deadline_ = + std::chrono::steady_clock::now() + std::chrono::seconds(kStopGraceSecs); + shutdown_state_ = ShutdownState::StoppingServices; + return false; + + case ShutdownState::StoppingServices: + if (!any_running()) { + shutdown_state_ = ShutdownState::Finalizing; + return false; + } + if (std::chrono::steady_clock::now() >= shutdown_deadline_) { + log_info(kTag, "grace elapsed; forcing kill"); for (auto& svc : config_.services) { if (svc.running && svc.pid > 0) { - log_info(kTag, "stopping ", svc.name, " (pid ", + log_info(kTag, "killing ", svc.name, " (pid ", std::to_string(svc.pid), ")"); - ::kill(svc.pid, SIGTERM); + ::kill(svc.pid, SIGKILL); } } shutdown_deadline_ = std::chrono::steady_clock::now() + - std::chrono::seconds(kStopGraceSecs); - shutdown_state_ = ShutdownState::StoppingServices; - return false; - - case ShutdownState::StoppingServices: - if (!any_running()) { - shutdown_state_ = ShutdownState::Finalizing; - return false; - } - if (std::chrono::steady_clock::now() >= shutdown_deadline_) { - log_info(kTag, "grace elapsed; forcing kill"); - for (auto& svc : config_.services) { - if (svc.running && svc.pid > 0) { - log_info(kTag, "killing ", svc.name, " (pid ", - std::to_string(svc.pid), ")"); - ::kill(svc.pid, SIGKILL); - } - } - shutdown_deadline_ = std::chrono::steady_clock::now() + - std::chrono::seconds(kKillGraceSecs); - shutdown_state_ = ShutdownState::ForcingKill; - } - return false; - - case ShutdownState::ForcingKill: - if (!any_running() || std::chrono::steady_clock::now() >= shutdown_deadline_) { - shutdown_state_ = ShutdownState::Finalizing; // give up on the rest - } - return false; - - case ShutdownState::Finalizing: { - bool reboot = shutdown_kind_ == ShutdownKind::Reboot; - log_info(kTag, "finalizing: ", reboot ? "reboot" : "poweroff"); - ::sync(); - unmount_filesystems(); - ::sync(); - if (reboot) { - log_info(kTag, "reboot()"); - if (::reboot(RB_AUTOBOOT) != 0) { - log_info(kTag, "reboot failed: ", std::strerror(errno)); - } - } else { - log_info(kTag, "power off"); - if (::reboot(RB_POWER_OFF) != 0) { - log_info(kTag, "poweroff failed: ", std::strerror(errno)); - } - } - return true; // nothing left do to; run() will _exit(0) + std::chrono::seconds(kKillGraceSecs); + shutdown_state_ = ShutdownState::ForcingKill; } + return false; + + case ShutdownState::ForcingKill: + if (!any_running() || + std::chrono::steady_clock::now() >= shutdown_deadline_) { + shutdown_state_ = ShutdownState::Finalizing; // give up on the rest + } + return false; + + case ShutdownState::Finalizing: { + bool reboot = shutdown_kind_ == ShutdownKind::Reboot; + log_info(kTag, "finalizing: ", reboot ? "reboot" : "poweroff"); + ::sync(); + unmount_filesystems(); + ::sync(); + if (reboot) { + log_info(kTag, "reboot()"); + if (::reboot(RB_AUTOBOOT) != 0) { + log_info(kTag, "reboot failed: ", std::strerror(errno)); + } + } else { + log_info(kTag, "power off"); + if (::reboot(RB_POWER_OFF) != 0) { + log_info(kTag, "poweroff failed: ", std::strerror(errno)); + } + } + return true; // nothing left do to; run() will _exit(0) + } } return false; } @@ -442,27 +930,31 @@ bool Supervisor::advance_shutdown() { void Supervisor::handle_sigchld() { struct signalfd_siginfo si; ssize_t n; - while ((n = ::read(sigfd_, &si, sizeof si)) == static_cast(sizeof si)) { + while ((n = ::read(sigfd_, &si, sizeof si)) == + static_cast(sizeof si)) { switch (si.ssi_signo) { - case SIGCHLD: - reap_children(); - break; - case SIGTERM: - case SIGINT: - // SIGTERM -> reboot, SIGINT -> poweroff (distinct paths to test). - begin_shutdown(si.ssi_signo == SIGTERM ? ShutdownKind::Reboot - : ShutdownKind::PowerOff); - break; - case SIGQUIT: - log_info(kTag, "SIGQUIT: emergency exit"); - _exit(1); - break; - case SIGHUP: - case SIGUSR1: - log_info(kTag, "reload requested (not yet implemented)"); - break; - default: - break; + case SIGCHLD: + reap_children(); + break; + case SIGTERM: + case SIGINT: + // SIGTERM -> reboot, SIGINT -> poweroff (distinct paths to test). + begin_shutdown(si.ssi_signo == SIGTERM ? ShutdownKind::Reboot + : ShutdownKind::PowerOff); + break; + case SIGQUIT: + log_info(kTag, "SIGQUIT: emergency exit"); + _exit(1); + break; + case SIGHUP: + log_info(kTag, "SIGHUP: reloading config"); + reload_config(); + break; + case SIGUSR1: + log_info(kTag, "SIGUSR1 received"); + break; + default: + break; } } } @@ -470,13 +962,15 @@ void Supervisor::handle_sigchld() { [[noreturn]] void Supervisor::run() { setup_signals(); open_console(); + open_control_socket(); log_info(kTag, "bajia init starting (pid 1)"); // boot sequence: fire ordered triggers. for (const char* ev : {"early-init", "init", "boot"}) { for (auto& action : config_.actions) { - if (action.trigger == ev) execute_action(action); + if (action.trigger == ev) + execute_action(action); } } @@ -484,12 +978,13 @@ void Supervisor::handle_sigchld() { for (;;) { // while shutting down, wake up exactly when the next state transition // is due; otherwise block indefinitely. - int timeout = shutdown_state_ == ShutdownState::Running ? -1 - : shutdown_timeout_ms(); + int timeout = + shutdown_state_ == ShutdownState::Running ? -1 : shutdown_timeout_ms(); int n = ::epoll_wait(epfd_, events, 8, timeout); if (n < 0) { - if (errno == EINTR) continue; + if (errno == EINTR) + continue; log_info(kTag, "epoll_wait: ", std::strerror(errno)); _exit(1); } @@ -497,6 +992,10 @@ void Supervisor::handle_sigchld() { for (int i = 0; i < n; ++i) { if (events[i].data.fd == sigfd_) { handle_sigchld(); + } else if (events[i].data.fd == ctl_fd_) { + ctl_accept(); + } else if (ctl_clients_.find(events[i].data.fd) != ctl_clients_.end()) { + ctl_handle_client(events[i].data.fd); } } diff --git a/src/supervisor.hpp b/src/supervisor.hpp index 91e2b75..2522e7d 100644 --- a/src/supervisor.hpp +++ b/src/supervisor.hpp @@ -8,6 +8,7 @@ #include #include +#include namespace bajia { @@ -18,7 +19,7 @@ enum class ShutdownKind { }; class Supervisor { -public: + public: explicit Supervisor(Config config); ~Supervisor(); @@ -35,7 +36,7 @@ public: // not return normally. [[noreturn]] void run(); -private: + private: // shutdown is a state machine driven from the event loop so that SIGKILL // grace periods and child reaping keep working while we wind down. enum class ShutdownState { @@ -49,6 +50,8 @@ private: Config config_; int epfd_ = -1; int sigfd_ = -1; + int ctl_fd_ = -1; + std::unordered_map ctl_clients_; bool shutdown_requested_ = false; ShutdownKind shutdown_kind_ = ShutdownKind::PowerOff; ShutdownState shutdown_state_ = ShutdownState::Running; @@ -60,7 +63,18 @@ private: void reap_children(); void execute_action(Action& action); void run_exec_command(const Command& cmd); - void run_command(Command& cmd); + bool run_command(Command& cmd); + // re-parse the .rc files and reconcile live services: removed services are + // stopped, added ones registered, changed ones restarted. Called from + // SIGHUP (and the `reload` control command). + void reload_config(); + + // control socket ("@bajia") so userland can drive init: start/stop/restart + // services, fire triggers, query status, request shutdown. + void open_control_socket(); + void ctl_accept(); + void ctl_handle_client(int fd); + std::string ctl_execute(const std::string& line); void begin_shutdown(ShutdownKind kind); // step the shutdown state machine; returns true when the loop should exit. diff --git a/tools/run_vm.py b/tools/run_vm.py index 03840ac..bc49989 100644 --- a/tools/run_vm.py +++ b/tools/run_vm.py @@ -11,12 +11,19 @@ examples: python3 tools/run_vm.py --fetch-busybox --nographic python3 tools/run_vm.py --kernel /boot/vmlinuz-$(uname -r) --gdb python3 tools/run_vm.py --config my-init.rc + python3 tools/run_vm.py --selinux # boot with the host SELinux policy (permissive) inside the guest: log in as `root` (passwordless by default, or use ---root-password), then `kill -TERM 1` -> reboot path, `kill -INT 1` -> +--root-password). `bctl status` / `bctl shutdown poweroff` drive the +init over its control socket; `kill -TERM 1` -> reboot path, `kill -INT 1` -> poweroff path. bajia is built statically by default: a dynamic binary cannot exec inside the initramfs (no libc there). Use --no-static only if you ship the libs too. +--selinux flips bajia to a dynamic build (there is no static libselinux on +Fedora), bundles libselinux/libpcre2/glibc + the loader into the initramfs, +and copies the host's /etc/selinux/ policy (loaded permissively). This +is the first stage of bootstrapping a real policy: boot, read the `avc: +denied` lines, refine, then flip to enforcing. """ from __future__ import annotations @@ -60,11 +67,11 @@ on shutdown service console-serial /bin/getty -L ttyS0 115200 vt100 console - respawn always + respawn = always service console-tty1 /bin/getty -L 38400 tty1 vt100 console - respawn always + respawn = always """ # source tarballs of busybox (github.com/mirror/busybox); a pinned tag is @@ -77,18 +84,19 @@ BUSYBOX_URLS = [ BUSYBOX_APPLETS = ["sh", "getty", "mount", "sync", "ls", "cat", "kill", "ps", "poweroff", "reboot", "mkdir", "mknod", "login"] - def run(cmd, **kw) -> subprocess.CompletedProcess: print("$", " ".join(str(c) for c in cmd)) return subprocess.run(cmd, **kw) - -def build_bajia(static: bool) -> Path: +def build_bajia(static: bool, selinux: bool = False) -> Path: env = dict(os.environ) + cfg = ["python3", "configure.py"] + if selinux: + cfg.append("--selinux") if static: print("building bajia (statically linked)...") env["CXX"] = env.get("CXX", "g++") + " -static" - r = run(["python3", "configure.py"], env=env) + r = run(cfg, env=env) if r.returncode != 0: sys.exit("configure.py failed") r = run(["ninja", "-C", str(BUILD)]) @@ -96,6 +104,64 @@ def build_bajia(static: bool) -> Path: sys.exit("ninja build failed") return BAJIA +SELINUX_CONFIG = Path("/etc/selinux/config") + +def host_selinux_type() -> str: + if not SELINUX_CONFIG.is_file(): + return "targeted" + for line in SELINUX_CONFIG.read_text().splitlines(): + line = line.strip() + if line.startswith("SELINUXTYPE="): + return line.split("=", 1)[1].strip().strip('"') + return "targeted" + +def selinux_policy_files() -> list[tuple[Path, str]]: + """Return (host_path, initramfs_relative_path) pairs for the policy payload.""" + typ = host_selinux_type() + base = Path("/etc/selinux") / typ + pols = sorted((base / "policy").glob("policy.*")) + if not pols: + sys.exit(f"--selinux: no policy under {base / 'policy'}/ " + "(install selinux-policy-targeted, a Fedora SELinux host is assumed)") + # libselinux 3.x looks for the restorecon table at contexts/files/file_contexts + # (older versions used contexts/file_contexts); bundle whichever exists. + fc = next((p for p in (base / "contexts" / "files" / "file_contexts", + base / "contexts" / "file_contexts") if p.is_file()), None) + if not fc: + sys.exit(f"--selinux: missing file_contexts under {base / 'contexts'}/") + fc_rel = "etc/selinux/" + typ + "/" + str(fc.relative_to(base)) + prefix = f"etc/selinux/{typ}/" + return [(pols[-1], prefix + f"policy/{pols[-1].name}"), + (fc, fc_rel)] + +def bundle_dynamic_libs(init: Path, root: Path) -> None: + """Copy the dynamic loader + resolved .so deps into the initramfs, + mirroring their absolute paths so the interpreter finds them.""" + out = run(["ldd", str(init)], capture_output=True, text=True) + if out.returncode != 0: + sys.exit("ldd failed on " + str(init)) + libs: list[str] = [] + for line in out.stdout.splitlines(): + line = line.strip() + if "=>" in line: + path = line.split("=>", 1)[1].strip().split(" ", 1)[0].strip() + else: # "linux-vdso" or the loader line "/lib64/ld-linux-x86-64.so.2 (0x...)" + path = line.split(" ", 1)[0].strip() + if path.startswith("/") and Path(path).is_file(): + libs.append(path) + seen: set[str] = set() + for lib in libs: # preserve first-seen order + if lib in seen: + continue + seen.add(lib) + dst = root / lib.lstrip("/") + dst.parent.mkdir(parents=True, exist_ok=True) + shutil.copy(lib, dst) + dst.chmod(0o755) + print("bundled dynamic libs:", ", ".join(seen)) + if not seen: + sys.exit("ldd reported no libraries - unexpected for a dynamic binary") + def find_kernel() -> Path | None: p = Path("/boot/vmlinuz-" + os.uname().release) if p.is_file(): @@ -194,8 +260,20 @@ def root_passwd_line(password: str | None) -> str: field = crypt_password(password) if password else "" return f"root:{field}:0:0:root:/:/bin/sh\n" +# appended to the test init.rc; started on boot, prints the exec context of a +# seclabel'd service and of init itself, then exits. +SELINUX_RC_PROBE = """ +service selinux-probe /bin/sh -c "echo probe-ctx=$(cat /proc/self/attr/current) init-ctx=$(cat /proc/1/attr/current)" + console + seclabel = system_u:system_r:init_t:s0 + respawn = never + +on boot + start selinux-probe +""" + def build_initramfs(init: Path, busybox: Path, rc_text: str, root_password: str | None, - keep: bool) -> Path: + selinux: bool, keep: bool) -> Path: if not shutil.which("cpio"): sys.exit("cpio not found (install cpio)") root = Path(tempfile.mkdtemp(prefix="bajia-root-")) @@ -204,19 +282,52 @@ def build_initramfs(init: Path, busybox: Path, rc_text: str, root_password: str "dev", "proc", "sys", "run", "tmp"): (root / sub).mkdir(parents=True) + if selinux: + for src, rel in selinux_policy_files(): + dst = root / rel + dst.parent.mkdir(parents=True, exist_ok=True) + shutil.copy(src, dst) + selcfg = root / "etc" / "selinux" / "config" + selcfg.write_text(f"SELINUX=permissive\nSELINUXTYPE={host_selinux_type()}\n") + bundle_dynamic_libs(init, root) + elif not is_static(init): + print("warning: bajia is dynamically linked; /init will fail to exec " + "inside the initramfs (error -2). Rebuild with --no-static " + "unset (static is the default) or drop --no-build.") + + if selinux: + rc_text = rc_text + SELINUX_RC_PROBE + shutil.copy(init, root / "init") (root / "init").chmod(0o755) + ctl = BUILD / "bctl" + if ctl.is_file(): + shutil.copy(ctl, root / "bin" / "bctl") + (root / "bin" / "bctl").chmod(0o755) + shutil.copy(busybox, root / "bin" / "busybox") (root / "bin" / "busybox").chmod(0o755) for applet in BUSYBOX_APPLETS: (root / "bin" / applet).symlink_to("busybox") (root / "etc" / "bajia" / "init.rc").write_text(rc_text) - (root / "etc" / "passwd").write_text(root_passwd_line(root_password)) + (root / "etc" / "passwd").write_text( + root_passwd_line(root_password) + + "nobody:x:65534:65534:nobody:/:/bin/sh\n") + (root / "etc" / "group").write_text( + "root:x:0:\n" + "nobody:x:65534:\n" + "daemon:x:1:\n") + # The staging tree is owned by the host user and mkdtemp makes the + # top dir 0700; GNU cpio preserves both, so without this the guest's + # "/" would be mode 0700 owned by uid 1000 -- fine for root services, + # but dropped-privilege services couldn't traverse it. Stamp owner + # root:root (no host chown needed) and make the root traversable. p = run(["bash", "-c", - "cd \"$1\" && find . -print0 | cpio --null -o -H newc", + "cd \"$1\" && chmod 0755 . && " + "find . -print0 | cpio --null -o -H newc --owner=0:0", "bajia-initramfs", str(root)], stdout=subprocess.PIPE) if p.returncode != 0: sys.exit("cpio packing failed") @@ -235,6 +346,9 @@ def qemu_command(kernel: Path, initrd: Path, args: argparse.Namespace) -> list[s display = args.display if display is None: display = "gtk" if os.environ.get("DISPLAY") else "none" + append = (f"console=tty1 console=ttyS0 rdinit=/init loglevel={args.loglevel}") + if args.selinux: + append += " selinux=1 enforcing=0" cmd = [ qemu, "-M", args.machine, @@ -242,13 +356,15 @@ def qemu_command(kernel: Path, initrd: Path, args: argparse.Namespace) -> list[s "-smp", str(args.smp), "-kernel", str(kernel), "-initrd", str(initrd), - "-append", - f"console=tty1 console=ttyS0 rdinit=/init loglevel={args.loglevel}", + "-append", append, "-display", display, "-serial", "stdio", ] if args.nographic: cmd[cmd.index("-display") + 1] = "none" + if args.serial_log: + cmd[cmd.index("-display") + 1] = "none" + cmd[cmd.index("-serial") + 1] = f"file:{args.serial_log}" if args.gdb or args.wait_gdb: cmd += ["-gdb", "tcp::1234", "-S"] if args.wait_gdb else ["-s"] cmd += ["-no-reboot", "-no-shutdown"] @@ -287,12 +403,17 @@ def main() -> int: help="pause the machine until a gdb client attaches") ap.add_argument("--keep-initramfs", action="store_true", help="don't delete the initramfs staging tree") + ap.add_argument("--selinux", action="store_true", + help="bundle the host SELinux policy + libs, boot permissive") + ap.add_argument("--serial-log", type=Path, + help="write the serial console to this file (forces -display none)") args = ap.parse_args() - init = BAJIA if args.no_build else build_bajia(not args.no_static) + static = not args.no_static and not args.selinux + init = BAJIA if args.no_build else build_bajia(static, args.selinux) if not init.is_file(): sys.exit(f"bajia not built at {init} (drop --no-build)") - if not is_static(init): + if not static and not args.selinux and not is_static(init): print("warning: bajia is dynamically linked; /init will fail to exec " "inside the initramfs (error -2). Rebuild with --no-static " "unset (static is the default) or drop --no-build.") @@ -315,7 +436,7 @@ def main() -> int: rc_text = args.config.read_text() if args.config else DEFAULT_RC initrd = build_initramfs(init, busybox, rc_text, args.root_password, - keep=args.keep_initramfs) + selinux=args.selinux, keep=args.keep_initramfs) print("initramfs:", initrd, f"({initrd.stat().st_size / 1024:.0f} KB)") cmd = qemu_command(kernel, initrd, args)