From 9070801f12cb37356aef85cff4e9f4e0c6435530 Mon Sep 17 00:00:00 2001 From: sunrisepeak Date: Sun, 30 Aug 2026 20:53:26 +0800 Subject: [PATCH 1/5] 0.10.0 --- one spawn, a working directory, and a job that terminate reaches MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ⭐ THE SAVING IS NOT ONLY IN THE HEADER. This file held THREE bodies of sixty lines that differed by four: every fix to the shared part --- the exec-report pipe among them --- had to be made three times, or be made once and be wrong twice. They are one function now, and the modifiers are positions in `kal_spawn'. Two of those positions are new work rather than rearrangement: `work' --- the directory the started program runs in, an `fchdir' in the duplicate before the replacement. `execveat' takes `base' as a dirfd, but that RESOLVES the name, and resolving a name is not entering a directory --- which is what the comment here used to claim and the code never did. ⚠️ A failed `fchdir' does NOT reach `execveat': running the right program in the wrong directory is precisely the silent wrongness this exists to remove, so it goes back through the pipe an exec failure uses. `KAL_SPAWN_OWN_JOB' --- `setpgid(0, 0)' in the duplicate, asked for there and not from the parent, whose own call races the replacement and loses. ⭐ AND `kal_process_terminate' NOW REACHES THE WHOLE JOB, WITHOUT THE HANDLE CARRYING ANY STATE TO SAY SO. A program that formed its own job has a group identifier equal to its own pid; one that did not inherited this implementation's, which is some other process. `getpgid(pid) == pid' distinguishes them exactly. ⚠️ Without this half the flag would do nothing a caller could see --- forming the job matters only because terminating then reaches what the started program itself started. Conformance against openkal 0.11: 140 held, 0 did not hold. --- mcpp.toml | 4 +- src/process.cpp | 312 ++++++++++++++++-------------------------------- src/sys.h | 4 + 3 files changed, 110 insertions(+), 210 deletions(-) diff --git a/mcpp.toml b/mcpp.toml index d7e690a..8cba798 100644 --- a/mcpp.toml +++ b/mcpp.toml @@ -1,7 +1,7 @@ [package] namespace = "mcpplibs" name = "openkal-linux" -version = "0.9.0" +version = "0.10.0" description = "The reference implementation of openkal for Linux, written on the kernel's own system-call interface so that it can be placed beneath a C library as well as above one." license = "Apache-2.0" @@ -18,7 +18,7 @@ authors = ["mcpplibs"] repo = "https://github.com/mcpplibs/openkal-linux" [dependencies] -openkal = "0.10.0" +openkal = "0.11.0" # The package contributes definitions and no modules. The interface it # implements is declared by the specification package, which this package diff --git a/src/process.cpp b/src/process.cpp index cad665b..2bbd7ed 100644 --- a/src/process.cpp +++ b/src/process.cpp @@ -159,128 +159,37 @@ inline void reap(okl_long child) { extern "C" { -int kal_process_spawn(kal_dir base, +// Starting a program. ⭐ ONE FUNCTION SINCE 0.11, AND THE SAVING IS NOT ONLY IN +// THE HEADER: this file used to hold THREE bodies of sixty lines that differed +// by four. Every fix to the shared part --- and there have been several, the +// exec-report pipe among them --- had to be made three times or be made once and +// be wrong twice. +// +// The modifiers are now positions in `how': a working directory, a set of +// grants, and two flags. They compose, which the three declarations could not +// do: there was no way to grant directories AND bind a lifetime, and no way at +// all to say the two things a shell runner needs together. +int kal_process_spawn(const kal_spawn* how, const char* path, kal_uintptr path_len, const char** argv, const kal_uintptr* argv_lens, kal_uintptr argc, const char** envp, const kal_uintptr* envp_lens, kal_uintptr envc, const kal_spawn_streams* streams, kal_process* out) { - const int b = okl::unpack(base.h); - if (b < 0 || out == nullptr) return kal_err_invalid; - if (!okl::acceptable(path, path_len)) return kal_err_invalid; - okl::terminated p(path, path_len); - if (!p.ok) return kal_err_invalid; - - // The vector is passed unaltered. Clause 7.6: argv[0] is the name the - // started program observes as its own, and it is the caller's to choose --- - // the started program reads it through kal_env_arg(0), so a caller that did - // not supply it could not predict what the program would read. - vector args, envs; - if (!args.build(argv, argv_lens, argc)) return kal_err_no_memory; - if (!envs.build(envp, envp_lens, envc)) return kal_err_no_memory; - - const okl_long in = streams ? static_cast(streams->in.h) : 0; - const okl_long ou = streams ? static_cast(streams->out.h) : 0; - const okl_long er = streams ? static_cast(streams->err.h) : 0; - - // Opened before the duplication, so that the duplicate inherits it. A pipe - // that cannot be made is not a reason to refuse the spawn: this answers as - // it did before the channel existed. - exec_report report; - report.open(0); - - // The image is duplicated and then replaced. openkal has no operation that - // duplicates the calling image, and this is why: the duplicate is not a - // resource the caller receives, it exists for the length of two system - // calls, and no environment without it could be asked to reproduce it. - const okl_long child = okl::sys(okl::nr_clone, 17 /* SIGCHLD */, 0, 0, 0, 0); - if (okl::failed(child)) { report.close_both(); return okl::translate(child); } - - if (child == 0) { - if (in != 0) okl::sys(okl::nr_dup3, in, 0, 0); - if (ou != 0) okl::sys(okl::nr_dup3, ou, 1, 0); - if (er != 0) okl::sys(okl::nr_dup3, er, 2, 0); - // ⚠️⚠️ THIS COMMENT USED TO CLAIM A PROPERTY THIS CODE DOES NOT HAVE. - // It said the started program's working directory is the directory - // supplied here. It is not. `b' is the directory the NAME resolves - // against and nothing more --- `execveat' takes a dirfd to resolve - // `p.buf', and resolving a name is not entering a directory. The - // started program's working directory is this implementation's own, - // whatever that happens to be, inherited across the clone above. - // - // ⭐ Found by a consumer's test rather than by reading, which is the - // point: `chdir' then start a program, ask it for its working - // directory, and it answers the directory the caller left --- against - // a host as control, which answers the one the caller entered. - // - // ⇒ AND IT IS NOT FIXABLE HERE. An `fchdir(b)' before the replacement - // would make the sentence true and the behaviour no better: `b' is - // whichever preopen the name resolved under --- for a program named - // `/usr/bin/sh' that is the root --- so the started program would get - // an arbitrary directory instead of a different arbitrary directory. - // Naming the program and naming where it runs are two directories, and - // openkal has an argument for one of them. openkal-musl's `chdir' - // therefore rebinds its own table and cannot do better; the interface - // has no operation that carries the second directory across a spawn. - // - // There is deliberately no operation that changes a working directory - // afterwards, because a working directory that can be changed is - // shared mutable state between execution contexts. That refusal is - // sound and is NOT what is missing --- what is missing is a way to say, - // at the moment of starting, which directory the program starts in. - const okl_long why = - okl::sys(okl::nr_execveat, b, reinterpret_cast(p.buf), - reinterpret_cast(args.slots), - reinterpret_cast(envs.slots), 0); - // Reached only when the replacement did not happen, because when it does - // there is nothing here to reach. - report.say(why); - okl::sys(okl::nr_exit_group, 127); - for (;;) { } - } + if (how == nullptr || out == nullptr) return kal_err_invalid; - if (const okl_long why = report.heard()) { - reap(child); - return okl::translate(why); - } + const int b = okl::unpack(how->base.h); + const int w = okl::unpack(how->work.h); + if (b < 0 || w < 0) return kal_err_invalid; + if (!okl::acceptable(path, path_len)) return kal_err_invalid; + if (how->grant_count > 0 && how->grants == nullptr) return kal_err_invalid; - *out = kal_process{ static_cast(child) }; - return kal_ok; -} + // ⚠️ REFUSED BEFORE ANYTHING IS STARTED, not after. A caller that asked for a + // bound lifetime and received a program without one has been given a program + // that outlives it --- which is the failure the flag exists to remove --- so an + // unclaimed position is an error and not a thing to proceed without. + constexpr kal_uintptr known = KAL_SPAWN_BOUND_LIFETIME | KAL_SPAWN_OWN_JOB; + if (how->flags & ~known) return kal_err_not_supported; -// The same, with the started program's lifetime bound to this one's. 0.10. -// -// ⭐⭐ WHY THIS EXISTS, AND IT IS NOT A CONVENIENCE. -// -// A C library asked for `execve' composes it out of what this interface has: -// start the program, wait for it, end with its status. That composition leaves -// THREE images where a system with the operation has two --- the caller, a copy -// that waits, and the program --- and `kal_process_terminate' upon the identifier -// the caller holds reaches the WAITER. Measured with a host as control: identical -// status words, opposite outcomes; the caller is told the program died on the -// signal it sent, while the program runs to completion, unsupervised. -// openkal-linux#13. -// -// ⚠️ THE BINDING IS SET IN THE STARTED IMAGE AND NOT FROM HERE, which is why it -// is a second spawn rather than an operation applied to a handle. This kernel's -// facility answers "end this context when the one that started it ends", and only -// that context can ask for it. -// -// ⚠️ AND IT IS ASKED FOR BEFORE THE REPLACEMENT AND CHECKED AFTER: the setting -// survives the replacement, but the parent could have ended in between --- in -// which case the signal has already been delivered and there is nothing to -// notice. Reading the parent's identity after arming closes that window: if it -// is no longer the one that armed, this image ends now rather than becoming the -// orphan the caller asked not to have. -int kal_process_spawn_bound(kal_dir base, - const char* path, kal_uintptr path_len, - const char** argv, const kal_uintptr* argv_lens, kal_uintptr argc, - const char** envp, const kal_uintptr* envp_lens, kal_uintptr envc, - const kal_spawn_streams* streams, - kal_process* out) { - const int b = okl::unpack(base.h); - if (b < 0 || out == nullptr) return kal_err_invalid; - if (!okl::acceptable(path, path_len)) return kal_err_invalid; okl::terminated p(path, path_len); if (!p.ok) return kal_err_invalid; @@ -288,14 +197,28 @@ int kal_process_spawn_bound(kal_dir base, if (!args.build(argv, argv_lens, argc)) return kal_err_no_memory; if (!envs.build(envp, envp_lens, envc)) return kal_err_no_memory; + // Resolved before the duplication, because a failure after it would leave a + // child to be reaped and a caller with an error it cannot act upon. + constexpr kal_uintptr max_grants = 16; + if (how->grant_count > max_grants) return kal_err_invalid; + int granted[max_grants]; + for (kal_uintptr i = 0; i < how->grant_count; ++i) { + granted[i] = okl::unpack(how->grants[i].dir.h); + if (granted[i] < 0) return kal_err_invalid; + } + const okl_long in = streams ? static_cast(streams->in.h) : 0; const okl_long ou = streams ? static_cast(streams->out.h) : 0; const okl_long er = streams ? static_cast(streams->err.h) : 0; - const okl_long mine = okl::sys(okl::nr_getpid); + const bool bind = (how->flags & KAL_SPAWN_BOUND_LIFETIME) != 0; + const bool job = (how->flags & KAL_SPAWN_OWN_JOB) != 0; + const okl_long mine = bind ? okl::sys(okl::nr_getpid) : 0; + + // The bound is 3 + grant_count, because the placements below reach that far. exec_report report; - report.open(0); + report.open(how->grant_count); const okl_long child = okl::sys(okl::nr_clone, 17 /* SIGCHLD */, 0, 0, 0, 0); if (okl::failed(child)) { report.close_both(); return okl::translate(child); } @@ -305,19 +228,57 @@ int kal_process_spawn_bound(kal_dir base, if (ou != 0) okl::sys(okl::nr_dup3, ou, 1, 0); if (er != 0) okl::sys(okl::nr_dup3, er, 2, 0); - // 9 is SIGKILL: the binding must not be something the started program - // can decline, because the caller asked for a program that does not - // outlive it and not for one that is invited not to. - okl::sys(okl::nr_prctl, okl::pr_set_pdeathsig, 9, 0, 0, 0); - // The window: if the caller ended between the clone and the line above, - // the signal is already spent and this image would survive it. - if (okl::sys(okl::nr_getppid) != mine) + // ⚠️ dup3 REFUSES A DUPLICATION ONTO ITSELF, which the ordinary case + // reaches whenever a granted directory already occupies the number it + // is destined for. Refusing there is correct of dup3 --- the flags could + // not be applied --- and here it means the descriptor is already in + // place, so it is left alone rather than treated as a failure. + for (kal_uintptr i = 0; i < how->grant_count; ++i) { + const okl_long want = static_cast(3 + i); + if (granted[i] != want) + okl::sys(okl::nr_dup3, granted[i], want, 0); + } + + // ⭐ THE DIRECTORY THE PROGRAM RUNS IN, AND THIS LINE IS THE WHOLE OF IT. + // + // `execveat' below takes `b' as a dirfd, but that only RESOLVES the + // name --- resolving a name is not entering a directory, which is what + // the comment here used to get wrong. Until 0.11 there was no second + // directory to enter, and a started program ran wherever this + // implementation happened to be. + // + // ⚠️ A FAILURE HERE MUST NOT REACH `execveat'. Running the right program + // in the wrong directory is precisely the silent wrongness this exists to + // remove, so it is reported through the same pipe an exec failure uses. + if (const okl_long e = okl::sys(okl::nr_fchdir, w); okl::failed(e)) { + report.say(e); okl::sys(okl::nr_exit_group, 127); + for (;;) { } + } + + // ⭐ A JOB OF ITS OWN, so that terminating it reaches what it starts. + // Asked for here and not from the parent: the parent's own `setpgid' on + // this child races the replacement below, and loses it once the program + // has been replaced. + if (job) okl::sys(okl::nr_setpgid, 0, 0); + + if (bind) { + // 9 is SIGKILL: the binding must not be something the started program + // can decline, because the caller asked for a program that does not + // outlive it and not for one that is invited not to. + okl::sys(okl::nr_prctl, okl::pr_set_pdeathsig, 9, 0, 0, 0); + // The window: if the caller ended between the clone and the line + // above, the signal is already spent and this image would survive it. + if (okl::sys(okl::nr_getppid) != mine) + okl::sys(okl::nr_exit_group, 127); + } const okl_long why = okl::sys(okl::nr_execveat, b, reinterpret_cast(p.buf), reinterpret_cast(args.slots), reinterpret_cast(envs.slots), 0); + // Reached only when the replacement did not happen, because when it does + // there is nothing here to reach. report.say(why); okl::sys(okl::nr_exit_group, 127); for (;;) { } @@ -375,89 +336,6 @@ void kal_process_channel_close(kal_stream s) { okl::sys(okl::nr_close, fd); } -// Starting a program that receives exactly the directories named. -// -// THE GRANTS ARE PLACED AS DESCRIPTORS THREE AND UPWARD, which is the -// arrangement kal_fs_preopen reads them back from. The inverse relationship -// clause 7.11 describes is therefore between this operation and that one, and -// it is why the two must agree about the numbering rather than each choosing. -// -// A COUNT OF ZERO IS NOT THE SAME AS kal_process_spawn. It starts a program with -// no preopens at all, which is the whole reason a caller reaches for this -// operation, so the loop below is not skipped when there is nothing to place --- -// what matters is that nothing else is inherited either. -int kal_process_spawn_with(kal_dir base, - const char* path, kal_uintptr path_len, - const char** argv, const kal_uintptr* argv_lens, kal_uintptr argc, - const char** envp, const kal_uintptr* envp_lens, kal_uintptr envc, - const kal_spawn_streams* streams, - const kal_preopen* grants, kal_uintptr grant_count, - kal_process* out) { - const int b = okl::unpack(base.h); - if (b < 0 || out == nullptr) return kal_err_invalid; - if (!okl::acceptable(path, path_len)) return kal_err_invalid; - if (grant_count > 0 && grants == nullptr) return kal_err_invalid; - okl::terminated p(path, path_len); - if (!p.ok) return kal_err_invalid; - - vector args, envs; - if (!args.build(argv, argv_lens, argc)) return kal_err_no_memory; - if (!envs.build(envp, envp_lens, envc)) return kal_err_no_memory; - - // Resolved before the fork, because a failure after it would leave a child - // to be reaped and a caller with an error it cannot act upon. - constexpr kal_uintptr max_grants = 16; - if (grant_count > max_grants) return kal_err_invalid; - int granted[max_grants]; - for (kal_uintptr i = 0; i < grant_count; ++i) { - granted[i] = okl::unpack(grants[i].dir.h); - if (granted[i] < 0) return kal_err_invalid; - } - - const okl_long in = streams ? static_cast(streams->in.h) : 0; - const okl_long ou = streams ? static_cast(streams->out.h) : 0; - const okl_long er = streams ? static_cast(streams->err.h) : 0; - - // The bound is 3 + grant_count, because the placements below reach that far. - exec_report report; - report.open(grant_count); - - const okl_long child = okl::sys(okl::nr_clone, 17 /* SIGCHLD */, 0, 0, 0, 0); - if (okl::failed(child)) { report.close_both(); return okl::translate(child); } - - if (child == 0) { - if (in != 0) okl::sys(okl::nr_dup3, in, 0, 0); - if (ou != 0) okl::sys(okl::nr_dup3, ou, 1, 0); - if (er != 0) okl::sys(okl::nr_dup3, er, 2, 0); - - // ⚠️ dup3 REFUSES A DUPLICATION ONTO ITSELF, which the ordinary case - // reaches whenever a granted directory already occupies the number it - // is destined for. Refusing there is correct of dup3 --- the flags could - // not be applied --- and here it means the descriptor is already in - // place, so it is left alone rather than treated as a failure. - for (kal_uintptr i = 0; i < grant_count; ++i) { - const okl_long want = static_cast(3 + i); - if (granted[i] != want) - okl::sys(okl::nr_dup3, granted[i], want, 0); - } - - const okl_long why = - okl::sys(okl::nr_execveat, b, reinterpret_cast(p.buf), - reinterpret_cast(args.slots), - reinterpret_cast(envs.slots), 0); - report.say(why); - okl::sys(okl::nr_exit_group, 127); - for (;;) { } - } - - if (const okl_long why = report.heard()) { - reap(child); - return okl::translate(why); - } - - *out = kal_process{ static_cast(child) }; - return kal_ok; -} int kal_process_wait(kal_process h, int* status, int* terminated_by_environment) { if (h.h == 0) return kal_err_invalid; @@ -483,9 +361,26 @@ int kal_process_wait(kal_process h, int* status, int* terminated_by_environment) return kal_ok; } +// ⭐ REACHES THE WHOLE JOB WHEN THERE IS ONE, AND THE HANDLE DOES NOT HAVE TO SAY +// SO --- the kernel already knows. +// +// A program started with KAL_SPAWN_OWN_JOB called `setpgid(0, 0)', so its group +// identifier IS its own. A program started without it inherited this +// implementation's group, whose identifier is some other process. So +// `getpgid(pid) == pid' distinguishes the two exactly, and no state has to be +// carried on the handle to remember which spawn produced it. +// +// ⚠️ WITHOUT THIS THE FLAG WOULD DO NOTHING VISIBLE. Forming the job in the +// started program is half of it; the half that matters to a caller is that +// terminating reaches what that program itself started. A shell that backgrounds +// work is killed and its background work survives --- which is the failure the +// flag exists to remove, and it would have survived the flag too. int kal_process_terminate(kal_process h) { if (h.h == 0) return kal_err_invalid; - const okl_long r = okl::sys(okl::nr_kill, static_cast(h.h), 15 /* SIGTERM */); + const okl_long pid = static_cast(h.h); + const okl_long pgid = okl::sys(okl::nr_getpgid, pid); + const okl_long target = (!okl::failed(pgid) && pgid == pid) ? -pid : pid; + const okl_long r = okl::sys(okl::nr_kill, target, 15 /* SIGTERM */); return okl::failed(r) ? okl::translate(r) : kal_ok; } @@ -497,7 +392,8 @@ kal_uintptr kal_process_props(void) { return KAL_PROCESS_PROP_TERMINATE | KAL_PROCESS_PROP_STREAM_PASSING | KAL_PROCESS_PROP_EXIT_STATUS | KAL_PROCESS_PROP_CHANNEL | KAL_PROCESS_PROP_GRANT_DIR - | KAL_PROCESS_PROP_BOUND_LIFETIME; + | KAL_PROCESS_PROP_BOUND_LIFETIME + | KAL_PROCESS_PROP_OWN_JOB; } } diff --git a/src/sys.h b/src/sys.h index 98c8a8f..635b2d0 100644 --- a/src/sys.h +++ b/src/sys.h @@ -99,6 +99,8 @@ enum : okl_long { nr_renameat = 264, nr_readlinkat = 267, nr_dup3 = 292, nr_execveat = 322, nr_dup2 = 33, nr_utimensat = 280, nr_symlinkat = 266, nr_fstatfs = 138, nr_getrandom = 318, + // openkal 0.11: the directory a started program runs in, and the job it forms + nr_fchdir = 81, nr_setpgid = 109, nr_getpgid = 121, // openkal.net and openkal.datagram nr_socket = 41, nr_connect = 42, nr_accept = 43, nr_sendto = 44, nr_recvfrom = 45, nr_shutdown = 48, nr_bind = 49, nr_listen = 50, @@ -183,6 +185,8 @@ enum : okl_long { nr_prctl = 167, nr_sched_getaffinity = 123, nr_getppid = 173, nr_arch_prctl = -1, nr_utimensat = 88, nr_symlinkat = 36, nr_fstatfs = 44, nr_getrandom = 278, + // openkal 0.11: the directory a started program runs in, and the job it forms + nr_fchdir = 50, nr_setpgid = 154, nr_getpgid = 155, // openkal.net and openkal.datagram nr_socket = 198, nr_connect = 203, nr_accept = 202, nr_sendto = 206, nr_recvfrom = 207, nr_shutdown = 210, nr_bind = 200, nr_listen = 201, From 4128bdab766e0fd19b08b4a447c0656e51d61e8b Mon Sep 17 00:00:00 2001 From: sunrisepeak Date: Sun, 30 Aug 2026 21:36:36 +0800 Subject: [PATCH 2/5] Take up openkal 0.11's unit: a handle the caller holds, not a flag The identity is established at the first start --- the first member forms the group and its identifier is reported --- so nothing has to be remembered and no registry appears. `kal_process_terminate' is one program again; the unit has its own operations. --- src/process.cpp | 74 ++++++++++++++++++++++++++++++++----------------- src/sys.h | 8 +++--- 2 files changed, 52 insertions(+), 30 deletions(-) diff --git a/src/process.cpp b/src/process.cpp index 2bbd7ed..d31110c 100644 --- a/src/process.cpp +++ b/src/process.cpp @@ -187,8 +187,7 @@ int kal_process_spawn(const kal_spawn* how, // bound lifetime and received a program without one has been given a program // that outlives it --- which is the failure the flag exists to remove --- so an // unclaimed position is an error and not a thing to proceed without. - constexpr kal_uintptr known = KAL_SPAWN_BOUND_LIFETIME | KAL_SPAWN_OWN_JOB; - if (how->flags & ~known) return kal_err_not_supported; + if (how->flags & ~KAL_SPAWN_BOUND_LIFETIME) return kal_err_not_supported; okl::terminated p(path, path_len); if (!p.ok) return kal_err_invalid; @@ -212,7 +211,13 @@ int kal_process_spawn(const kal_spawn* how, const okl_long er = streams ? static_cast(streams->err.h) : 0; const bool bind = (how->flags & KAL_SPAWN_BOUND_LIFETIME) != 0; - const bool job = (how->flags & KAL_SPAWN_OWN_JOB) != 0; + + // ⭐ THE UNIT, WHOSE IDENTITY HERE IS A PROCESS GROUP'S --- which is to say, + // the identifier of whichever program formed it first. `join' is zero for the + // first member, and the child then makes the group its own; a later member is + // given the number to join. + const okl_long join = how->job ? static_cast(how->job->h) : 0; + const bool unit = how->job != nullptr; const okl_long mine = bind ? okl::sys(okl::nr_getpid) : 0; @@ -256,11 +261,12 @@ int kal_process_spawn(const kal_spawn* how, for (;;) { } } - // ⭐ A JOB OF ITS OWN, so that terminating it reaches what it starts. - // Asked for here and not from the parent: the parent's own `setpgid' on - // this child races the replacement below, and loses it once the program - // has been replaced. - if (job) okl::sys(okl::nr_setpgid, 0, 0); + // ⭐ THE UNIT, ENTERED HERE AND NOT FROM THE PARENT: the parent's own + // `setpgid' on this child races the replacement below and loses once the + // program has been replaced. Zero means "your own", which is how a group + // comes into existence at all --- there is nothing to create beforehand, + // which is why the interface reports the identity rather than taking it. + if (unit) okl::sys(okl::nr_setpgid, 0, join); if (bind) { // 9 is SIGKILL: the binding must not be something the started program @@ -289,6 +295,12 @@ int kal_process_spawn(const kal_spawn* how, return okl::translate(why); } + // ⚠️ WRITTEN ONLY AFTER THE START HAS SUCCEEDED, and only when the unit was + // new. The first member's identifier IS the group's, so this is where the + // caller learns it; a later member joins one the caller already holds and + // there is nothing to report. + if (unit && join == 0) how->job->h = static_cast(child); + *out = kal_process{ static_cast(child) }; return kal_ok; } @@ -361,29 +373,39 @@ int kal_process_wait(kal_process h, int* status, int* terminated_by_environment) return kal_ok; } -// ⭐ REACHES THE WHOLE JOB WHEN THERE IS ONE, AND THE HANDLE DOES NOT HAVE TO SAY -// SO --- the kernel already knows. +// ⭐ ONE PROGRAM, WHATEVER UNIT IT IS IN. // -// A program started with KAL_SPAWN_OWN_JOB called `setpgid(0, 0)', so its group -// identifier IS its own. A program started without it inherited this -// implementation's group, whose identifier is some other process. So -// `getpgid(pid) == pid' distinguishes the two exactly, and no state has to be -// carried on the handle to remember which spawn produced it. -// -// ⚠️ WITHOUT THIS THE FLAG WOULD DO NOTHING VISIBLE. Forming the job in the -// started program is half of it; the half that matters to a caller is that -// terminating reaches what that program itself started. A shell that backgrounds -// work is killed and its background work survives --- which is the failure the -// flag exists to remove, and it would have survived the flag too. +// An earlier draft made this reach the whole group when the started program had +// formed one, recovering that fact with `getpgid(pid) == pid'. It worked, and it +// was the wrong shape: the meaning of this operation then turned on a property of +// the handle that no caller could see. The unit has its own operation below, and +// the caller says which of the two it means. int kal_process_terminate(kal_process h) { if (h.h == 0) return kal_err_invalid; - const okl_long pid = static_cast(h.h); - const okl_long pgid = okl::sys(okl::nr_getpgid, pid); - const okl_long target = (!okl::failed(pgid) && pgid == pid) ? -pid : pid; - const okl_long r = okl::sys(okl::nr_kill, target, 15 /* SIGTERM */); + const okl_long r = okl::sys(okl::nr_kill, static_cast(h.h), 15 /* SIGTERM */); + return okl::failed(r) ? okl::translate(r) : kal_ok; +} + +// Every program in the unit, including ones this implementation never held a +// handle to --- which is the whole reason a unit exists. +// +// ⚠️ A GROUP IS NAMED BY A PROCESS IDENTIFIER, AND THOSE ARE REUSED. Once the +// program that formed the group has ended and the numbers have wrapped, this can +// reach a different group. That is what this system does --- every program that +// calls `killpg' lives with it --- and the interface records it rather than +// reading as though it were not so. +int kal_process_job_terminate(kal_job j) { + if (j.h == 0) return kal_err_invalid; + const okl_long r = okl::sys(okl::nr_kill, -static_cast(j.h), 15 /* SIGTERM */); return okl::failed(r) ? okl::translate(r) : kal_ok; } +// ⚠️ RELEASES NOTHING AND ENDS NOTHING. A group here is a number, not a resource, +// so there is no handle to close --- and the operation exists so that a caller +// need not know that. Where the unit IS a resource, releasing it must still not +// end its members; the interface says so at the declaration. +void kal_process_job_close(kal_job) { } + // Releasing the handle does not affect the program. A program that has not been // waited for continues, and this environment collects it when the caller exits. void kal_process_close(kal_process) { } @@ -393,7 +415,7 @@ kal_uintptr kal_process_props(void) { | KAL_PROCESS_PROP_EXIT_STATUS | KAL_PROCESS_PROP_CHANNEL | KAL_PROCESS_PROP_GRANT_DIR | KAL_PROCESS_PROP_BOUND_LIFETIME - | KAL_PROCESS_PROP_OWN_JOB; + | KAL_PROCESS_PROP_JOB; } } diff --git a/src/sys.h b/src/sys.h index 635b2d0..a736f2a 100644 --- a/src/sys.h +++ b/src/sys.h @@ -99,8 +99,8 @@ enum : okl_long { nr_renameat = 264, nr_readlinkat = 267, nr_dup3 = 292, nr_execveat = 322, nr_dup2 = 33, nr_utimensat = 280, nr_symlinkat = 266, nr_fstatfs = 138, nr_getrandom = 318, - // openkal 0.11: the directory a started program runs in, and the job it forms - nr_fchdir = 81, nr_setpgid = 109, nr_getpgid = 121, + // openkal 0.11: the directory a started program runs in, and the unit it joins + nr_fchdir = 81, nr_setpgid = 109, // openkal.net and openkal.datagram nr_socket = 41, nr_connect = 42, nr_accept = 43, nr_sendto = 44, nr_recvfrom = 45, nr_shutdown = 48, nr_bind = 49, nr_listen = 50, @@ -185,8 +185,8 @@ enum : okl_long { nr_prctl = 167, nr_sched_getaffinity = 123, nr_getppid = 173, nr_arch_prctl = -1, nr_utimensat = 88, nr_symlinkat = 36, nr_fstatfs = 44, nr_getrandom = 278, - // openkal 0.11: the directory a started program runs in, and the job it forms - nr_fchdir = 50, nr_setpgid = 154, nr_getpgid = 155, + // openkal 0.11: the directory a started program runs in, and the unit it joins + nr_fchdir = 50, nr_setpgid = 154, // openkal.net and openkal.datagram nr_socket = 198, nr_connect = 203, nr_accept = 202, nr_sendto = 206, nr_recvfrom = 207, nr_shutdown = 210, nr_bind = 200, nr_listen = 201, From 67ee038f522a663a1eba77ac9d9e8013103cc0aa Mon Sep 17 00:00:00 2001 From: sunrisepeak Date: Sun, 30 Aug 2026 22:25:13 +0800 Subject: [PATCH 3/5] Enter a unit, terminate one for good, and stop a signal openkal cannot report MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `kal_process_job_enter' is `setpgid(0, join)': a program joins a unit or forms one and learns its identity. It exists because `kal_spawn.job' cannot say it --- that field places a program the caller STARTS, and a copy wishing to lead a unit must say so about ITSELF before it replaces itself, which is what every shell and every runner with a timeout is written as. ⚠️⚠️ `kal_process_job_terminate' NOW USES THE SIGNAL THAT CANNOT BE DECLINED, and that is a decision. A unit exists because its members include programs the caller never held a handle to; a request any one of them may ignore does not terminate the unit, it terminates the part that agreed. Measured with a consumer's own test: a shell that traps the polite signal and loops outlived every deadline, and the caller's escalation could not help --- openkal has no vocabulary for "and this time I mean it". What it costs is stated at the call: a member gets no chance to clean up, and a caller that wants that still has `kal_process_terminate' upon a member it holds. ⭐⭐ AND SIGPIPE IS IGNORED AT STARTUP, WHICH IS THE OTHER HALF OF THE SAME IDEA. openkal defines no signals. `kal_stream_write' is required to REPORT that a stream's far end has gone --- and this kernel instead delivers SIGPIPE, whose default action ends the program. So a program that wrote to a closed stream was not told; it stopped, with a status no operation here produced. ⚠️ A C library above answers `signal(SIGPIPE, SIG_IGN)' and TRUTHFULLY reports success, because openkal has no signals and there is nothing for it to set. The program is killed anyway, four layers below the call it made to prevent exactly that. Measured: a consumer's test exited 141 having printed none of its own assertions. Ignored here because here is the only place that can; the write then fails with EPIPE, which is what the interface said would happen all along. The earlier `getpgid(pid) == pid' trick is gone with the flag it served: `kal_process_terminate' means one program again, always. --- src/env.cpp | 32 ++++++++++++++++++++++++++++++++ src/process.cpp | 38 +++++++++++++++++++++++++++++++++++++- src/sys.h | 4 ++-- 3 files changed, 71 insertions(+), 3 deletions(-) diff --git a/src/env.cpp b/src/env.cpp index d513aa2..3480733 100644 --- a/src/env.cpp +++ b/src/env.cpp @@ -100,6 +100,38 @@ bool recover(char*** argv_out, int* argc_out, char*** envp_out) { return false; } +// ⚠️⚠️ A PROGRAM ABOVE openkal SHALL NOT BE ENDED BY SOMETHING openkal NEVER +// TOLD IT ABOUT, AND WITHOUT THIS LINE ONE WAS. +// +// openkal defines no signals. `kal_stream_write' is required to REPORT that the +// far end of a stream is gone --- there is a condition for it --- and this kernel +// instead delivers SIGPIPE, whose default action ends the program. So a program +// that wrote to a closed stream did not receive the error the interface promises; +// it stopped, with a status no operation here produced and no wording anywhere in +// the specification. +// +// ⭐ MEASURED THROUGH A CONSUMER, AND THE SHAPE IS WHY IT TOOK SO LONG TO SEE. A +// C library above this one answers `signal(SIGPIPE, SIG_IGN)' --- openkal has no +// signals, so the library has nothing to set and truthfully reports success. The +// program is then killed anyway, four layers below the call it made to prevent +// exactly that. Exit 141 in a test whose own assertions never printed. +// +// ⇒ Ignored HERE and not there, because here is the only place that can: the +// interface has no operation a C library could use to say it. The write then +// fails with EPIPE, which `kal_stream_write' translates and reports, which is +// what the interface said would happen all along. +// +// ⚠️ NOT A POLICY CHOICE ABOUT SIGNALS IN GENERAL. This is the one signal an +// ordinary openkal operation provokes; the rest are left exactly as this program +// was started with. +[[gnu::constructor(101)]] void quiet_the_signal_openkal_cannot_report() { + // struct k_sigaction as this kernel takes it: handler, flags, restorer, mask. + struct { void* handler; unsigned long flags; void* restorer; unsigned long mask; } + ignore{ reinterpret_cast(1) /* SIG_IGN */, 0, nullptr, 0 }; + okl::sys(okl::nr_rt_sigaction, 13 /* SIGPIPE */, + reinterpret_cast(&ignore), 0, sizeof ignore.mask); +} + [[gnu::constructor(101)]] void capture(int argc, char** argv, char** envp) { if (okl::g_argv != nullptr) return; if (plausible(argc, argv, envp)) { okl::record(argc, argv, envp); return; } diff --git a/src/process.cpp b/src/process.cpp index d31110c..8e2e5e8 100644 --- a/src/process.cpp +++ b/src/process.cpp @@ -386,9 +386,45 @@ int kal_process_terminate(kal_process h) { return okl::failed(r) ? okl::translate(r) : kal_ok; } +// This program itself joins or forms a unit --- the operation `kal_spawn.job' +// cannot express, because that one places a program the caller STARTS and a copy +// wishing to lead a unit must say so about ITSELF before it replaces itself. +int kal_process_job_enter(kal_job* j) { + if (j == nullptr) return kal_err_invalid; + const okl_long join = static_cast(j->h); + const okl_long r = okl::sys(okl::nr_setpgid, 0, join); + if (okl::failed(r)) return okl::translate(r); + // The identity of a group is its leader's, so a program that has just formed + // one reports its own. Read back rather than assumed: `setpgid(0, 0)' makes + // this program the leader, and `getpid' is that leader's identifier. + if (join == 0) j->h = static_cast(okl::sys(okl::nr_getpid)); + return kal_ok; +} + // Every program in the unit, including ones this implementation never held a // handle to --- which is the whole reason a unit exists. // +// ⚠️⚠️ AND IT IS THE SIGNAL THAT CANNOT BE DECLINED, WHICH IS A DECISION AND NOT +// A DETAIL. +// +// `kal_process_terminate' upon ONE program uses the polite one: a caller holds +// that program's handle, can wait for it, and can terminate it again. None of +// that is true of a unit. A unit exists because its members include programs the +// caller never held a handle to and cannot enumerate --- and a request that any +// one of them may ignore does not terminate the unit, it terminates the part of +// it that agreed. +// +// ⭐ Measured with a consumer's own test: a shell that traps the polite signal +// and loops. Asked politely, the unit outlived every deadline; the caller's +// escalation could not help, because openkal has no vocabulary for "and this +// time I mean it" --- it has no signals at all. +// +// ⇒ So the operation does what its name says. ⚠️ WHAT THIS COSTS IS REAL: a +// member gets no chance to clean up, where on a system programmed directly a +// caller would send the polite signal first and wait. A caller that wants that +// still has it --- `kal_process_terminate' upon the member it holds --- and what it +// cannot do is ask a unit politely. +// // ⚠️ A GROUP IS NAMED BY A PROCESS IDENTIFIER, AND THOSE ARE REUSED. Once the // program that formed the group has ended and the numbers have wrapped, this can // reach a different group. That is what this system does --- every program that @@ -396,7 +432,7 @@ int kal_process_terminate(kal_process h) { // reading as though it were not so. int kal_process_job_terminate(kal_job j) { if (j.h == 0) return kal_err_invalid; - const okl_long r = okl::sys(okl::nr_kill, -static_cast(j.h), 15 /* SIGTERM */); + const okl_long r = okl::sys(okl::nr_kill, -static_cast(j.h), 9 /* SIGKILL */); return okl::failed(r) ? okl::translate(r) : kal_ok; } diff --git a/src/sys.h b/src/sys.h index a736f2a..14a3ec5 100644 --- a/src/sys.h +++ b/src/sys.h @@ -100,7 +100,7 @@ enum : okl_long { nr_dup2 = 33, nr_utimensat = 280, nr_symlinkat = 266, nr_fstatfs = 138, nr_getrandom = 318, // openkal 0.11: the directory a started program runs in, and the unit it joins - nr_fchdir = 81, nr_setpgid = 109, + nr_fchdir = 81, nr_setpgid = 109, nr_rt_sigaction = 13, // openkal.net and openkal.datagram nr_socket = 41, nr_connect = 42, nr_accept = 43, nr_sendto = 44, nr_recvfrom = 45, nr_shutdown = 48, nr_bind = 49, nr_listen = 50, @@ -186,7 +186,7 @@ enum : okl_long { nr_arch_prctl = -1, nr_utimensat = 88, nr_symlinkat = 36, nr_fstatfs = 44, nr_getrandom = 278, // openkal 0.11: the directory a started program runs in, and the unit it joins - nr_fchdir = 50, nr_setpgid = 154, + nr_fchdir = 50, nr_setpgid = 154, nr_rt_sigaction = 134, // openkal.net and openkal.datagram nr_socket = 198, nr_connect = 203, nr_accept = 202, nr_sendto = 206, nr_recvfrom = 207, nr_shutdown = 210, nr_bind = 200, nr_listen = 201, From 2b0c5712d3272d0a8e90c7c00bad667a08373e5f Mon Sep 17 00:00:00 2001 From: sunrisepeak Date: Sun, 30 Aug 2026 22:43:40 +0800 Subject: [PATCH 4/5] Answer the stop-request word, with the restorer this architecture needs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A disposition for the two polite signals, storing a word and waking whoever waits on it --- both of which are safe to do from one. `kal_task_wait' is what a caller waits with, so the wake is the raw call rather than the library one. ⚠️ A DISPOSITION AND NOT A WAITING CONTEXT. Consuming these from a context of its own needs them blocked in EVERY context, and blocking is per-context and inherited: a program that already had contexts running would have some that still take the default action, and blocking at startup would make a program that never asks unkillable. A disposition is per PROGRAM and can be installed at any moment --- so it is armed on the first enquiry, and a program that never asks is unaffected. ⚠️ THE RESTORER IS SUPPLIED HERE ON x86_64 AND BY THE KERNEL ON aarch64. Without SA_RESTORER the return from a handler faults; the C library normally supplies those three instructions and this implementation has none beneath it. --- src/process.cpp | 68 ++++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 67 insertions(+), 1 deletion(-) diff --git a/src/process.cpp b/src/process.cpp index 8e2e5e8..a265064 100644 --- a/src/process.cpp +++ b/src/process.cpp @@ -446,12 +446,78 @@ void kal_process_job_close(kal_job) { } // waited for continues, and this environment collects it when the caller exits. void kal_process_close(kal_process) { } +// ⭐⭐ A WORD THE ENVIRONMENT SETS WHEN SOMEBODY HAS ASKED THIS PROGRAM TO END. +// +// ⚠️ A HANDLER AND NOT A WAITING CONTEXT, AND THE REASON IS WHICH ONE CAN BE +// ARMED WITHOUT DISTURBING A PROGRAM THAT NEVER ASKS. Consuming these signals +// from a context of its own would require them BLOCKED IN EVERY context, and +// blocking is per-context and inherited: a program that already had contexts +// running when it first asked would have some that still take the default +// action, and a program that never asks would have been made unkillable at +// startup. A disposition is per PROGRAM and can be installed at any moment. +// +// The handler does two things, and both are safe to do from one: store a word, +// and wake whoever waits on it. `kal_task_wait' is what a caller waits with, so +// the wake is the same operation `kal_task_wake' performs --- issued here as the +// raw call, because a handler may not enter code that takes a lock. +// +// ⚠️ THE RESTORER IS SUPPLIED HERE ON ONE ARCHITECTURE AND BY THE KERNEL ON THE +// OTHER. On x86_64 a disposition installed without SA_RESTORER faults on return +// from the handler --- the C library normally supplies the three instructions, +// and this implementation has no C library beneath it. On aarch64 the kernel +// supplies it and the flag must NOT be set. +namespace { + +kal_u32 g_stop_word = 0; +int g_stop_armed = 0; + +#if defined(__x86_64__) +extern "C" void okl_sigreturn_trampoline(void); +asm(".globl okl_sigreturn_trampoline\n" + "okl_sigreturn_trampoline:\n" + " movq $15, %rax\n" // rt_sigreturn + " syscall\n"); +constexpr unsigned long sa_restorer_flag = 0x04000000u; // SA_RESTORER +#endif + +void stop_handler(int) { + __atomic_store_n(&g_stop_word, 1u, __ATOMIC_RELEASE); + okl::sys(okl::nr_futex, reinterpret_cast(&g_stop_word), + 1 /* FUTEX_WAKE */, 0x7fffffff, 0, 0, 0); +} + +void arm_one(int signo) { + struct { void* handler; unsigned long flags; void* restorer; unsigned long mask; } act {}; + act.handler = reinterpret_cast(&stop_handler); +#if defined(__x86_64__) + act.flags = sa_restorer_flag; + act.restorer = reinterpret_cast(&okl_sigreturn_trampoline); +#endif + okl::sys(okl::nr_rt_sigaction, signo, + reinterpret_cast(&act), 0, sizeof act.mask); +} + +} // namespace + +// ⚠️ ARMED ON THE FIRST ENQUIRY AND NOT AT STARTUP. A program that never asks +// keeps the default action, which is what every program that has never heard of +// this operation expects --- and it is the only arrangement under which adding +// this operation changes nothing for anyone who does not use it. +const kal_u32* kal_process_stop_requested(void) { + if (!__atomic_exchange_n(&g_stop_armed, 1, __ATOMIC_ACQ_REL)) { + arm_one(15); // SIGTERM + arm_one(2); // SIGINT + } + return &g_stop_word; +} + kal_uintptr kal_process_props(void) { return KAL_PROCESS_PROP_TERMINATE | KAL_PROCESS_PROP_STREAM_PASSING | KAL_PROCESS_PROP_EXIT_STATUS | KAL_PROCESS_PROP_CHANNEL | KAL_PROCESS_PROP_GRANT_DIR | KAL_PROCESS_PROP_BOUND_LIFETIME - | KAL_PROCESS_PROP_JOB; + | KAL_PROCESS_PROP_JOB + | KAL_PROCESS_PROP_STOP_REQUESTED; } } From 94f76956565f2042717da12af8b8c3e2573483d2 Mon Sep 17 00:00:00 2001 From: sunrisepeak Date: Sun, 30 Aug 2026 23:28:46 +0800 Subject: [PATCH 5/5] Update this repository's own tests for the 0.11 spawn record MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ⚠️ Missed on the first pass because I updated src/ and grepped src/. The tests call kal_process_spawn too, and CI is what said so. --- tests/conformance_process_task.cpp | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/tests/conformance_process_task.cpp b/tests/conformance_process_task.cpp index 1925594..ce255a2 100644 --- a/tests/conformance_process_task.cpp +++ b/tests/conformance_process_task.cpp @@ -53,6 +53,11 @@ int main() { check(have_root, "a directory covering the file system is supplied"); if (have_root) { + // How every start below is described. `work' is the same directory as + // `base', which is what a caller that does not care about the working + // directory passes --- openkal has no ambient one for a default to mean. + const kal_spawn how{ slash, slash, nullptr, nullptr, 0, 0 }; + // The program that succeeds and the program that fails are at // different places on different systems. The test locates them rather // than assuming, because assuming would make it a test of one system. @@ -66,7 +71,7 @@ int main() { const kal_uintptr lens[] = { 7 }; int rc = kal_err_invalid; for (int i = 0; i < 2 && rc != kal_ok; ++i) - rc = kal_process_spawn(slash, true_paths[i], true_lens[i], argv, lens, 1, + rc = kal_process_spawn(&how, true_paths[i], true_lens[i], argv, lens, 1, nullptr, nullptr, 0, nullptr, &p); check(rc == kal_ok, "a program is started"); if (rc == kal_ok) { @@ -83,7 +88,7 @@ int main() { const char* qargv[] = { "openkal" }; int qrc = kal_err_invalid; for (int i = 0; i < 2 && qrc != kal_ok; ++i) - qrc = kal_process_spawn(slash, false_paths[i], false_lens[i], qargv, lens, 1, + qrc = kal_process_spawn(&how, false_paths[i], false_lens[i], qargv, lens, 1, nullptr, nullptr, 0, nullptr, &q); if (qrc == kal_ok) { int status = -1, terminated = -1; @@ -114,7 +119,7 @@ int main() { const kal_uintptr rlens[] = { 22, 2, script_len }; int rrc = kal_err_invalid; for (int i = 0; i < 2 && rrc != kal_ok; ++i) - rrc = kal_process_spawn(slash, sh_paths[i], sh_lens[i], rargv, rlens, 3, + rrc = kal_process_spawn(&how, sh_paths[i], sh_lens[i], rargv, rlens, 3, nullptr, nullptr, 0, nullptr, &r); check(rrc == kal_ok, "a shell is started"); if (rrc == kal_ok) { @@ -128,7 +133,9 @@ int main() { // A name that ascends is refused here as it is in the file system. kal_process bad{}; - check(kal_process_spawn(kal::fs::working(), "../bin/true", 11, + const kal_spawn escape{ kal::fs::working(), kal::fs::working(), + nullptr, nullptr, 0, 0 }; + check(kal_process_spawn(&escape, "../bin/true", 11, nullptr, nullptr, 0, nullptr, nullptr, 0, nullptr, &bad) != kal_ok, "an ascending program name is refused");