liken

InputUnitCoveredTotalPercent
Gostatements144561671586.5%

Go

14456 of 16715 statements, 86.5%.

FileCovered statementsTotal statementsPercent
api/api.go33100.0%
api/conditions.go1717100.0%
api/versions.go77100.0%
cli/approve.go434987.8%
cli/cluster.go313686.1%
cli/completion.go7171100.0%
cli/dispatch.go1111100.0%
cli/main.go829982.8%
cli/plugins.go708384.3%
cli/request.go273284.4%
cluster-operator/channel.go454893.8%
cluster-operator/events.go77100.0%
cluster-operator/fleet.go14615395.4%
cluster-operator/flux.go9511384.1%
cluster-operator/heartbeats.go353989.7%
cluster-operator/janitor.go555894.8%
cluster-operator/leader.go21612.5%
cluster-operator/main.go8112067.5%
cluster-operator/metrics.go232495.8%
cluster-operator/phase.go66100.0%
cluster-operator/rollout.go14414599.3%
cluster-operator/steward.go404393.0%
cluster-operator/stuckpods.go202195.2%
cluster-operator/watches.go9910594.3%
cluster/changes.go3737100.0%
cluster/cluster.go566191.8%
cluster/features.go8787100.0%
cluster/registries.go1111100.0%
cluster/runtime.go899098.9%
disks/fat32.go39540198.5%
disks/fat32_read.go10311986.6%
disks/fat32_state.go273284.4%
disks/fat32_write.go20322590.2%
disks/gpt.go18820293.1%
disks/section.go0150.0%
hardware/aliases.go2929100.0%
hardware/delivery.go798296.3%
hardware/inventory.go646795.5%
hardware/names.go3535100.0%
hardware/pciids.go2626100.0%
hardware/settle.go1212100.0%
hardware/sysfs.go495196.1%
hardware/uevent.go616889.7%
hardware/unclaimed.go526777.6%
hardware/unclaimedserio.go99100.0%
identity/adopt.go273479.4%
identity/kubeconfig.go516479.7%
identity/mint.go486178.7%
image/cpio.go182281.8%
image/cpio_read.go455384.9%
image/layer.go9911090.0%
image/media.go606888.2%
image/stick.go9912181.8%
init/actuator.go6966.7%
init/attended.go131492.9%
init/bootentries.go9310786.9%
init/ciphers.go252889.3%
init/claim.go12212696.8%
init/cluster.go505296.2%
init/cmdline.go1313100.0%
init/components.go545598.2%
init/console.go525889.7%
init/crash.go15418185.1%
init/diskids.go717792.2%
init/disklinks.go14416388.3%
init/diskpaths.go515396.2%
init/disks.go5959100.0%
init/diskuuid.go182281.8%
init/durable.go294072.5%
init/efi.go9410688.7%
init/efiactuator.go212487.5%
init/ext4.go384977.6%
init/facts.go39340297.8%
init/failstop.go2424100.0%
init/fatstate.go555993.2%
init/features.go849291.3%
init/grow.go718583.5%
init/grubactuator.go718385.5%
init/grubcfg.go505198.0%
init/grubenv.go505296.2%
init/grubinstall.go404588.9%
init/hardware.go687491.9%
init/imports.go364383.7%
init/install.go12315380.4%
init/k3s.go25328887.8%
init/liveload.go9910693.4%
init/loadoption.go889295.7%
init/logrotate.go465288.5%
init/main.go04000.0%
init/manifests.go15417289.5%
init/moduleparams.go3030100.0%
init/modules.go13114590.3%
init/network.go17319787.8%
init/podlogs.go171989.5%
init/proving.go9712577.6%
init/quiesce.go283775.7%
init/radios.go14415692.3%
init/reboot.go3711033.6%
init/registries.go687294.4%
init/reinstall.go64712.8%
init/report.go16017094.1%
init/reportdisks.go848697.7%
init/reportlayout.go848697.7%
init/reportproposal.go17818297.8%
init/resolve.go313296.9%
init/responder.go758390.4%
init/restart.go9210290.2%
init/rlimits.go2121100.0%
init/rootimage.go558664.0%
init/schemarevision.go10111885.6%
init/serio.go788492.9%
init/serioattach.go899296.7%
init/seriolines.go323982.1%
init/serioretry.go363797.3%
init/seriostatus.go12813098.5%
init/seriotty.go0140.0%
init/slotloader.go414885.4%
init/slots.go101376.9%
init/softdeps.go536088.3%
init/storage.go15317686.9%
init/supervisor.go13023455.6%
init/switchroot.go4911443.0%
init/system.go84119.5%
init/time.go12715482.5%
init/versions.go283190.3%
init/wireless.go22324491.4%
init/worldreport.go152171.4%
init/wpaconfig.go848697.7%
init/wpactrl.go10711791.5%
kubernetes/apiclient.go182090.0%
kubernetes/clusters.go91181.8%
kubernetes/fakeapi/fakeapi.go11912595.2%
kubernetes/heartbeat.go7272100.0%
kubernetes/helmcharts.go44100.0%
kubernetes/kubeconfig.go152171.4%
kubernetes/machines.go141593.3%
kubernetes/ownwrite.go1717100.0%
kubernetes/pods.go121392.3%
kubernetes/resourceclaims.go4580.0%
kubernetes/resourceslices.go515494.4%
kubernetes/secrets.go77100.0%
kubernetes/services.go88100.0%
kubernetes/watch/reads.go586096.7%
kubernetes/watch/wake.go4141100.0%
kubernetes/workloads.go262796.3%
logs/cursor.go101190.9%
logs/envelope.go131492.9%
logs/kmsg.go677787.0%
logs/lift.go5151100.0%
logs/main.go0180.0%
logs/tail.go10211390.3%
machine-operator/backstop.go1919100.0%
machine-operator/cdi.go848796.6%
machine-operator/cluster.go879591.6%
machine-operator/conditions.go159159100.0%
machine-operator/converge.go10411094.5%
machine-operator/demotion.go2727100.0%
machine-operator/disruptions.go66100.0%
machine-operator/download.go10312085.8%
machine-operator/dra.go767996.2%
machine-operator/drain.go9393100.0%
machine-operator/draplugin.go9611087.3%
machine-operator/endpoint.go55100.0%
machine-operator/events.go2323100.0%
machine-operator/fetch.go676898.5%
machine-operator/held.go1010100.0%
machine-operator/hosts.go354087.5%
machine-operator/imports.go616396.8%
machine-operator/labels.go3434100.0%
machine-operator/liveness.go313296.9%
machine-operator/loop.go14414996.6%
machine-operator/machineevents.go1515100.0%
machine-operator/main.go0850.0%
machine-operator/metrics.go646894.1%
machine-operator/node.go6875.0%
machine-operator/outcome.go404197.6%
machine-operator/ownedconditions.go66100.0%
machine-operator/ownstatus.go121392.3%
machine-operator/phase.go2626100.0%
machine-operator/protection.go1313100.0%
machine-operator/publishing.go6969100.0%
machine-operator/publishingserio.go2525100.0%
machine-operator/rebinding.go394292.9%
machine-operator/rebootrequest.go252792.6%
machine-operator/reconcile.go34635298.3%
machine-operator/registries.go646697.0%
machine-operator/release.go848697.7%
machine-operator/retraction.go485194.1%
machine-operator/retry.go293096.7%
machine-operator/seeding.go313588.6%
machine-operator/staging.go4949100.0%
machine-operator/staleness.go1111100.0%
machine-operator/taints.go5050100.0%
machine-operator/userspace.go1313100.0%
machine-operator/waits.go12012397.6%
machine-operator/watches.go9510095.0%
machine/channel.go91090.0%
machine/drift.go144144100.0%
machine/factsread.go24932377.1%
machine/factsreadboot.go11314080.7%
machine/factsserio.go384290.5%
machine/factstree.go12714289.4%
machine/factswrite.go12514983.9%
machine/failstop.go161794.1%
machine/hosts.go55100.0%
machine/imports.go202195.2%
machine/inotify.go12113788.3%
machine/known.go677095.7%
machine/layer.go172085.0%
machine/machine.go4141100.0%
machine/moduleparameters.go2323100.0%
machine/reboot.go373897.4%
machine/registries.go1616100.0%
machine/release.go2929100.0%
machine/rlimit.go585998.3%
machine/serio.go343597.1%
machine/staging.go869887.8%
machine/status.go2121100.0%
machine/storage.go646598.5%
machine/sysctl.go222395.7%
machine/systemrelease.go1212100.0%
machine/wireless.go1414100.0%
metrics/listener.go222395.7%
metrics/metrics.go242596.0%
mount/args.go536877.9%
mount/main.go143836.8%
mount/options.go1616100.0%
plugins/image.go374386.0%
plugins/list.go444793.6%
plugins/plugins.go4580.0%
plugins/remove.go11100.0%
plugins/sync.go274165.9%
releases/bundle.go748884.1%
releases/download.go303293.8%
releases/fetch.go687887.2%
releases/index.go8610086.0%
releases/serve.go151788.2%
releases/versions.go66100.0%
scaffold/scaffold.go13116380.4%
api/api.go 100.0%
1// Package api is the grammar of the liken.sh API group. It holds the2// shapes and the vocabulary that every liken document shares, and3// nothing more.4//5// liken's documents include the Machine, the Cluster, and the6// smaller records that use the same machinery. Kubernetes resources7// shape all of them, and this package defines that shape. It8// declares the group and version, and the metadata each document9// contains. It declares the conditions and phases each status uses,10// the role vocabulary both documents use, and the version grammar11// that names releases. The machine package and the cluster package12// each own one document, and both build on this package. Neither13// package imports the other. Kubernetes uses the same layering:14// every typed API embeds metav1, and no typed API imports another.15//16// This package covers a narrow scope, chosen on purpose: only facts17// that every document shares belong here. Nothing behavioral belongs18// here. Nothing about storage or staging belongs here. A type that19// names one document is in the wrong package.20//21// k8s.io/apimachinery already defines these shapes, as22// metav1.ObjectMeta and metav1.Condition. liken redeclares them on23// purpose. The kubernetes package speaks to the API server with only24// net/http, for the same reason: liken's only Kubernetes dependency25// is the YAML converter. Importing apimachinery for two structs26// would add its whole dependency tree and release cadence. The27// narrow copies also do real work. metav1.ObjectMeta has about28// twenty-five fields, and liken uses four of them. Every liken29// document parses strictly, so liken refuses metadata it would30// otherwise ignore silently: labels and owner references in a31// hand-written manifest. The wire format stays compatible with the32// Kubernetes convention. The Go types represent only the fields33// liken uses.34package api3536// APIVersion is the full group/version string every liken document37// declares. It is also the URL segment operators use to speak to the38// API server: /apis/liken.sh/v1alpha1/machines. CRD groups are DNS39// names, and liken owns liken.sh.40const APIVersion = "liken.sh/v1alpha1"4142// ObjectMeta is the slice of Kubernetes object metadata that liken43// uses. Name identifies the object. ResourceVersion is the cluster's44// optimistic-concurrency counter. The operator sends it back when it45// watches, and the server resumes the stream from that point.46// Generation counts changes to the spec. The API server increases47// generation on spec writes and leaves it unchanged on status48// writes. This lets a condition record which version of the spec it49// judged.50//51// Annotations carry the signals that are not desired state. A spec52// declares what should hold, and a controller works until it holds.53// A request such as "poll the channel now" does not fit that54// contract: it asks for one action, not a standing state. Kubernetes55// puts these signals in annotations (kubectl writes restartedAt to56// request a Deployment rollout the same way), and liken follows.57// The operators that render canonical documents build their metadata58// fresh, from the name alone, so an annotation can never change a59// staged document's hash, and an annotation edit can never reboot a60// machine.61//62// UID names one instance of an object, so an owner reference to it63// does not also name an object that is deleted and created again under64// the same name. The machine operator reads its Machine's UID to name65// the Machine as the owner of its heartbeat lease.66type ObjectMeta struct {67	Name            string            `json:"name"`68	UID             string            `json:"uid,omitempty"`69	ResourceVersion string            `json:"resourceVersion,omitempty"`70	Generation      int64             `json:"generation,omitempty"`71	Annotations     map[string]string `json:"annotations,omitempty"`72}7374// Meta is the part of an object's metadata that a watch's store and an75// operator's memo of its own writes read: the shared module's76// memo.Meta. Both are aliases of the same interface literal, so the77// Machine and the Cluster answer memo.Meta with no import of that78// module. init links this package, and the module's client would bring79// net/http into init.80type Meta = interface {81	GetNamespace() string82	GetName() string83	GetResourceVersion() string84}8586// GetName answers the object's name.87func (m ObjectMeta) GetName() string { return m.Name }8889// GetNamespace answers the empty namespace, because every liken kind90// is cluster-scoped.91func (m ObjectMeta) GetNamespace() string { return "" }9293// GetResourceVersion answers the version of the copy.94func (m ObjectMeta) GetResourceVersion() string { return m.ResourceVersion }9596// Phase summarizes a document's state in one word. Go models a97// closed vocabulary with a named string type. The constants below98// are the only legal values. The compiler catches an error when code99// passes a Phase where a Role belongs. The wire format stays100// unchanged, because a named string marshals exactly like a bare101// string. Kubernetes' own API types use the same method: v1.PodPhase102// and metav1.ConditionStatus.103type Phase string104105// These are the phases a machine can report, listed from most severe106// to least severe. Each phase summarizes the conditions; it is not a107// fact on its own. The operator derives the phase from the108// conditions on every pass. The table that does this work,109// machine-operator/phase.go, maps each condition to the phase it110// produces. Lost is the exception to this rule. A machine that is111// Lost cannot write its own status, so the cluster operator writes112// Lost on the machine's behalf when the machine's heartbeat stops.113// The Cluster's status reuses this same vocabulary (Ready, Updating,114// Degraded), so a fleet and its machines use the same words to115// describe their state.116const (117	PhaseUnknown       Phase = "Unknown"       // the facts are unreadable, so the operator cannot judge the machine's state118	PhaseBooting       Phase = "Booting"       // init has not yet finished publishing this boot's record119	PhaseLost          Phase = "Lost"          // the heartbeat stopped, and the cluster operator wrote this, not the machine120	PhaseBlocked       Phase = "Blocked"       // drift exists but the system cannot stage it; it needs a different edit, not more time121	PhaseUpdating      Phase = "Updating"      // a reboot is under way to apply a staged change122	PhaseUpdatePending Phase = "UpdatePending" // a change is staged and waits for a Manual reboot123	PhaseDegraded      Phase = "Degraded"      // something is wrong that does not match one of the specific states above124	PhaseDownloading   Phase = "Downloading"   // a release downloads to the inactive slot while the machine serves its workloads125	PhaseReady         Phase = "Ready"         // every condition is True126)127128// Role is what a machine is in its cluster. There are exactly two129// roles. Leaders run a control plane: an API server, a scheduler,130// and the datastore. Followers run workloads and take direction from131// the leaders. k3s calls these roles "server" and "agent". liken132// translates between the two names in exactly one place, the moment133// it execs k3s (init/supervisor.go), and uses leader and follower134// everywhere else. Both documents share this vocabulary directly:135// the Cluster's spec declares the leaders, and each Machine's status136// reports the role it derived.137type Role string138139const (140	RoleLeader   Role = "leader"141	RoleFollower Role = "follower"142)
api/conditions.go 100.0%
1package api23// Conditions are how a liken document's status carries observations:4// a set of typed, timestamped verdicts ("Ready", "SysctlsApplied")5// that controllers maintain and humans and tooling read. The shape6// and the rules here mirror metav1.Condition, the method Kubernetes7// uses everywhere (Pods, Nodes, and Deployments all carry these), so8// anyone who reads `kubectl describe` output already knows how to9// read a liken document.1011import (12	"slices"13	"time"1415	"github.com/liken-sh/liken/kubernetes/conditions"16)1718// The condition type is the one every liken component declares: the19// shared module's conditions.Condition, which has the JSON shape of20// metav1.Condition. The aliases keep the names that the Machine and21// Cluster packages, the operators, and the CLI already use. The22// shared package imports only time, so init, which reads these types23// and must link no HTTP client, links nothing more.24//25// ConditionStatus is a condition's verdict. It is a string rather26// than a bool because there is a third state: a controller must be27// able to say when it currently cannot tell.28type ConditionStatus = conditions.Status2930const (31	ConditionTrue    = conditions.True32	ConditionFalse   = conditions.False33	ConditionUnknown = conditions.Unknown34)3536// Condition mirrors metav1.Condition. ObservedGeneration records37// which metadata.generation the condition judged. Generation counts38// spec edits, so a reader can tell "Ready, for the spec as it stands"39// apart from "Ready, but for a spec two edits ago". That difference40// matters in liken, where edits wait for a reboot to take effect. (The41// convergence conditions make the stronger, content-hashed version of42// this claim. The generation is for tooling that reads the43// convention.)44//45// The CRDs require a reason of at least one character, and accept an46// empty message, so a condition encodes the same way with or without47// omitempty on those two fields.48type Condition = conditions.Condition4950// SetCondition adds or updates a condition by type, through51// conditions.Set, at the time now. It preserves the Kubernetes rule52// that makes lastTransitionTime meaningful: the time moves only when53// Status flips, not on every write. This is what lets `kubectl get`54// answer "how long has this machine been Ready?" instead of only55// "when did the operator last say so?".56func SetCondition(list []Condition, c Condition, now time.Time) []Condition {57	c.LastTransitionTime = now58	conditions.Set(&list, c)59	return list60}6162// Transitions answers each condition in after that transitioned from63// before, by the rule of conditions.Set: the condition is new, or its64// status or its reason changed. An operator composes a whole status,65// writes it, and then posts one Event for each transition. Comparing66// the written list with the stored one, after the write lands, means67// a refused write posts nothing, and the next pass finds the same68// transition again.69func Transitions(before, after []Condition) []Condition {70	var out []Condition71	for _, c := range after {72		held := FindCondition(before, c.Type)73		if held == nil || held.Status != c.Status || held.Reason != c.Reason {74			out = append(out, c)75		}76	}77	return out78}7980// FindCondition returns the condition of the named type, or nil when81// the list holds none.82func FindCondition(list []Condition, conditionType string) *Condition {83	for i := range list {84		if list[i].Type == conditionType {85			return &list[i]86		}87	}88	return nil89}9091// RemoveCondition drops the condition of the named type. Most92// conditions are observations: they stay in the list and flip93// between True and False. Removal exists for the conditions that are94// grants. A grant is present while it is extended and gone when it95// is revoked. Its absence carries the meaning, so there is no False96// state for other machinery to misread as trouble.97func RemoveCondition(list []Condition, conditionType string) []Condition {98	for i := range list {99		if list[i].Type == conditionType {100			return slices.Delete(list, i, i+1)101		}102	}103	return list104}
api/versions.go 100.0%
1package api23// This file defines liken's version grammar and the ordering over it.4//5// A liken version is a calendar date and a serial: yyyy.mm.dd-nnn,6// with every field zero-padded, and the serial starts at 001 and7// counts up within the day. The number says when a release was cut8// and nothing else. liken does not use semantic versioning on9// purpose: an OS release carries a kernel version and a Kubernetes10// version of its own, and a semver major on the outside would only11// invite readings like "liken 2.0 must mean kubernetes 2". There is12// also no compatibility boundary for a major version to mark. Every13// release is expected to take over from the release before it, and14// to carry the machine's layer and on-disk state forward, so the15// code keeps compatibility, and the version number does not report16// it. What shipped inside a release is recorded where it belongs, in17// the release document's components.18//19// Versions are immutable: a new serial supersedes a bad release; the20// system never rebuilds a release under the same name. The catalog21// on the Cluster names releases by these versions, and two questions22// need an ordering over them: which catalog entry is the newest (the23// NEWEST printer column), and whether a machine is behind its24// target.2526import (27	"fmt"28	"regexp"29	"strings"30	"time"31)3233// versionShape is the grammar, shared verbatim with the Cluster34// CRD's pattern on spec.version and the catalog: a four-digit year,35// a two-digit month and day, a dash, and a three-digit serial.36var versionShape = regexp.MustCompile(`^\d{4}\.\d{2}\.\d{2}-\d{3}$`)3738// ValidVersion checks a version against the grammar at authoring39// time. This lets the system refuse a malformed version when it40// bundles a release, rather than let it be discovered when a machine41// fails to fetch it. Beyond the shape, the date must be a real42// calendar date: the CRD's pattern cannot check that 2026.02.30 never43// happened, but this function can.44func ValidVersion(v string) error {45	if !versionShape.MatchString(v) {46		return fmt.Errorf("version %q is not yyyy.mm.dd-nnn (for example 2026.07.11-001)", v)47	}48	date, _, _ := strings.Cut(v, "-")49	if _, err := time.Parse("2006.01.02", date); err != nil {50		return fmt.Errorf("version %q does not name a real date", v)51	}52	return nil53}5455// CompareVersions orders two version strings. It returns a negative56// number when a is older, zero when they name the same version, and57// a positive number when a is newer. Every field of the grammar is58// zero-padded to a fixed width, so plain string comparison gives the59// correct order. The padding exists exactly so that no field ever60// needs numeric parsing to sort: as bare numbers, "010" and "9"61// would sort backwards.62func CompareVersions(a, b string) int {63	return strings.Compare(a, b)64}
cli/approve.go 87.8%
1package main23// approve-reboot is the one cluster command that carries liken's own4// meaning instead of handing the terminal to another tool. A machine5// whose rebootPolicy is Manual stages a change and waits for a6// person. This command is how the person answers: it reads7// status.pending, reports what is waiting, and writes the8// approve-disruption annotation with the staged document's hash. The9// hash makes the grant one-shot (machine.ApproveDisruptionAnnotation10// explains why), and writing the same annotation twice is the same11// grant, so the command is repeatable.1213import (14	"flag"15	"fmt"16	"io"17	"strings"1819	"github.com/liken-sh/liken/liken/api"20	"github.com/liken-sh/liken/liken/kubernetes"21	"github.com/liken-sh/liken/liken/machine"22)2324// chooseApproval picks which pending entry to approve. A reboot25// applies every staged document at once, so when a reboot-class26// change is pending, approving it covers the rest. Otherwise the27// first entry stands for the list: with several restart-class28// changes pending, each applies on its own grant, and running the29// command again approves the next.30func chooseApproval(pending []machine.PendingDisruption) *machine.PendingDisruption {31	for i := range pending {32		if pending[i].Kind == machine.DisruptionReboot {33			return &pending[i]34		}35	}36	if len(pending) > 0 {37		return &pending[0]38	}39	return nil40}4142// renderPending reports what the machine waits for, in the shape a43// person reads before deciding to grant: each change's condition and44// reason, its summary with the short hash a grant names, and what45// kind of disruption applies it.46func renderPending(m *machine.Machine) string {47	name := m.Metadata.Name48	pending := m.Status.Pending49	if len(pending) == 0 {50		return fmt.Sprintf("%s is converged; nothing is waiting\n", name)51	}52	count := fmt.Sprintf("%d changes", len(pending))53	if len(pending) == 1 {54		count = "one change"55	}56	var b strings.Builder57	fmt.Fprintf(&b, "%s is waiting on %s:\n", name, count)58	for _, p := range pending {59		condition := p.Condition60		if c := api.FindCondition(m.Status.Conditions, p.Condition); c != nil {61			condition = fmt.Sprintf("%s  %s", p.Condition, c.Reason)62		}63		applies := "a reboot applies this, and every other staged change with it"64		if p.Kind == machine.DisruptionRestart {65			applies = "a k3s restart applies this; the machine does not reboot"66		}67		fmt.Fprintf(&b, "  %s\n  %s (%.12s)\n  %s\n", condition, p.Summary, p.Hash, applies)68	}69	return b.String()70}7172// approveReboot is the command: resolve the credential, read the73// machine, report, and grant.74func approveReboot(args []string, out io.Writer) error {75	fs := flag.NewFlagSet("approve-reboot", flag.ContinueOnError)76	server := fs.String("server", "", "the API server address, when cluster.yaml's endpoint is not reachable from here")77	if err := fs.Parse(args); err != nil {78		return err79	}80	if fs.NArg() != 2 {81		return fmt.Errorf("usage: liken approve-reboot [-server URL] <deployment-dir> <machine>")82	}83	kubeconfigPath, err := writeKubeconfig(fs.Arg(0), *server, io.Discard)84	if err != nil {85		return err86	}87	c, err := kubernetes.KubeconfigClient(kubeconfigPath)88	if err != nil {89		return err90	}91	m, err := kubernetes.GetMachine(c, fs.Arg(1))92	if err != nil {93		return err94	}95	fmt.Fprint(out, renderPending(m))96	choice := chooseApproval(m.Status.Pending)97	if choice == nil {98		return nil99	}100	short := fmt.Sprintf("%.12s", choice.Hash)101	patch := fmt.Sprintf(`{"metadata":{"annotations":{%q:%q}}}`, machine.ApproveDisruptionAnnotation, short)102	if err := kubernetes.PatchJSON(c, kubernetes.MachinesPath+"/"+m.Metadata.Name, []byte(patch)); err != nil {103		return err104	}105	fmt.Fprintf(out, "\napproved: %s=%s\n", machine.ApproveDisruptionAnnotation, short)106	return nil107}
cli/cluster.go 86.1%
1package main23// This file is the cluster half of the toolkit: how a command on an4// operator's workstation finds the cluster and holds a credential5// for it.6//7// The one positional argument these commands share is the8// deployment directory, the layout GETTING-STARTED.md documents:9// identity sits at <dir>/identity, and the endpoint comes from10// <dir>/cluster.yaml. The endpoint needs a caveat, and -server11// exists because of it. spec.endpoint is the address a follower12// joins through from inside the cluster's own network segment. The13// operator's workstation often cannot reach that address: a dev lab14// reaches its guests through a forwarded localhost port instead.15// So cluster.yaml is the default and -server overrides it.16//17// The kubeconfig goes to a predictable path that these commands18// rewrite and reuse: <dir>/identity/kubeconfig, with 060019// permissions, beside the identity it is derived from, because that20// is already the directory an operator guards. A fresh file per run21// would leak, because the passthrough commands replace this process22// with the tool (exec), so no deferred cleanup ever runs.2324import (25	"flag"26	"fmt"27	"io"28	"os"29	"os/exec"30	"path/filepath"31	"strings"3233	"github.com/liken-sh/liken/liken/cluster"34	"github.com/liken-sh/liken/liken/identity"35	"golang.org/x/sys/unix"36)3738// resolveCluster decides where the cluster is and what it is named.39// The endpoint is the -server flag when given, otherwise the40// endpoint in the deployment's cluster.yaml. A deployment with no41// cluster.yaml at all is a single machine on its own (see42// cluster.LoadCluster), which never declares an endpoint, so that43// case also asks for -server. The name comes from the same44// document's metadata, so that the kubeconfig for each cluster an45// operator holds labels its entries with that cluster's own name. A46// deployment with no cluster.yaml has no name, and falls back to47// liken.48func resolveCluster(dir, server string) (name, endpoint string, err error) {49	c, err := cluster.LoadCluster(filepath.Join(dir, "cluster.yaml"))50	if err != nil {51		return "", "", fmt.Errorf("reading the deployment's cluster.yaml: %w", err)52	}53	name = "liken"54	if c != nil && c.Metadata.Name != "" {55		name = c.Metadata.Name56	}57	if server != "" {58		return name, server, nil59	}60	if c == nil || c.Spec.Endpoint == "" {61		return "", "", fmt.Errorf("%s/cluster.yaml declares no endpoint; pass -server", dir)62	}63	return name, c.Spec.Endpoint, nil64}6566// writeKubeconfig resolves the cluster's name and endpoint, mints67// the admin credential into the deployment's identity directory, and68// returns the kubeconfig's path.69func writeKubeconfig(dir, server string, out io.Writer) (string, error) {70	name, endpoint, err := resolveCluster(dir, server)71	if err != nil {72		return "", err73	}74	identityDir := filepath.Join(dir, "identity")75	if err := identity.Kubeconfig(identityDir, name, endpoint, out); err != nil {76		return "", err77	}78	return filepath.Join(identityDir, "kubeconfig"), nil79}8081// execTool is exec(2): it replaces this process with the tool, so82// the tool owns the terminal and its exit code is liken's. It is a83// variable so tests can observe the call instead of vanishing into84// the exec.85var execTool = unix.Exec8687// envWith returns the environment with one variable bound to one88// value, whether or not it was set before. Appending a duplicate89// would not do: when a variable appears twice in an environment,90// which copy a program reads is the program's own accident.91func envWith(environ []string, key, value string) []string {92	kept := make([]string, 0, len(environ)+1)93	for _, entry := range environ {94		if !strings.HasPrefix(entry, key+"=") {95			kept = append(kept, entry)96		}97	}98	return append(kept, key+"="+value)99}100101// passthrough runs one of the tools an operator already uses,102// against the right cluster: it resolves the credential, sets103// KUBECONFIG, and replaces this process with the tool from PATH.104// liken does not reimplement kubectl, and it does not vendor these105// binaries either: every command here is a credential and an106// endpoint handed to a program the operator already chose to107// install.108func passthrough(tool string, args []string) error {109	fs := flag.NewFlagSet(tool, flag.ContinueOnError)110	server := fs.String("server", "", "the API server address, when cluster.yaml's endpoint is not reachable from here")111	if err := fs.Parse(args); err != nil {112		return err113	}114	if fs.NArg() < 1 {115		return fmt.Errorf("usage: liken %s [-server URL] <deployment-dir> [args...]", tool)116	}117	kubeconfigPath, err := writeKubeconfig(fs.Arg(0), *server, io.Discard)118	if err != nil {119		return err120	}121	path, err := exec.LookPath(tool)122	if err != nil {123		return fmt.Errorf("%s is not on PATH; install it and run this again", tool)124	}125	return execTool(path,126		append([]string{tool}, fs.Args()[1:]...),127		envWith(os.Environ(), "KUBECONFIG", kubeconfigPath))128}
cli/completion.go 100.0%
1package main23// How the base CLI completes a command line for the shell, and4// the contract the per-operator CLIs copy. kubectl completes a plugin5// by running an executable named kubectl_complete-<name> on PATH, and6// that shim runs this binary's hidden __complete verb. The verb answers7// in cobra's completion protocol: one candidate per line, then a final8// ":N" line that carries the ShellCompDirective bitmask. The base CLI9// answers from static lists, because its verbs, its plugins10// subcommands, and its plugin domains are all known without a cluster.1112import (13	"fmt"14	"io"15	"sort"16	"strings"17)1819// The ShellCompDirective bits this CLI emits, cobra's own20// numbering. The default is 0, which leaves the shell's own file21// completion in place. NoFileComp stops the shell from offering file22// names when the CLI has a closed set of its own, and Error marks a23// request the CLI could not answer.24const (25	compDirectiveError      = 126	compDirectiveNoFileComp = 427	compDirectiveDefault    = 028)2930// The plugins group, the one verb with subcommands of its own.31const pluginsVerb = "plugins"3233// Every top-level verb the dispatcher answers, minus completion34// and __complete, which serve the shell and never a person. A new verb35// in run() joins this list so the shell offers it.36var topLevelVerbs = []string{37	"new", "mint", "adopt", "kubeconfig", "kubectl", "stern", "flux",38	"approve-reboot", "request-reboot", "plugins", "layer", "fetch",39	"media", "stick", "bundle", "index", "serve", "version",40}4142// The plugin domains the walk dispatches to a43// kubectl-liken-<domain> binary. The list is static because the base44// CLI ships with a release and completes before any cluster is reached,45// so the shell offers a domain the same whether its CLI is installed or46// not.47var pluginDomains = []string{"audio", "bluetooth", "display", "library", "media"}4849// The plugins group's own subcommands.50var pluginsSubcommands = []string{"list", "remove", "sync"}5152// topLevelCandidates is the verbs and the plugin domains as one53// sorted set with no repeat, so "liken <TAB>" offers a command and a54// domain side by side. media names both a verb and a domain, and it55// appears once.56func topLevelCandidates() []string {57	seen := map[string]bool{}58	var all []string59	for _, item := range topLevelVerbs {60		if !seen[item] {61			seen[item] = true62			all = append(all, item)63		}64	}65	for _, item := range pluginDomains {66		if !seen[item] {67			seen[item] = true68			all = append(all, item)69		}70	}71	sort.Strings(all)72	return all73}7475// One flag the CLI accepts, and whether it reads a value from the76// next word. Completion walks the words with this table so it can tell a77// flag's value apart from a positional argument.78type completionFlag struct {79	name       string80	takesValue bool81}8283// Every flag the subcommands bind, in the completion table's own84// form. The base CLI parses each subcommand with the standard flag85// package, which reads one and two dashes alike, so the table names the86// one-dash spelling the usage prints. Each of these reads a value.87var completionFlags = []completionFlag{88	{name: "-server", takesValue: true},89	{name: "-digest", takesValue: true},90	{name: "-source", takesValue: true},91	{name: "-slot-size", takesValue: true},92	{name: "-console", takesValue: true},93}9495// What the words before the cursor amount to: the positional96// arguments already given, and the one flag whose value the cursor is97// about to type.98type completionParse struct {99	positionals      []string100	pendingValueFlag string101}102103// runComplete answers a __complete request and writes the protocol to104// stdout. It stays silent on every failure, because a completer that105// prints an error corrupts the command line the shell is drawing.106func runComplete(args []string, stdout io.Writer) error {107	candidates, directive := completeArgs(args)108	for _, candidate := range candidates {109		fmt.Fprintln(stdout, candidate)110	}111	fmt.Fprintf(stdout, ":%d\n", directive)112	return nil113}114115// completeArgs reads the request words and returns the candidates. The last116// word is the one under the cursor, and the words before it are already117// typed. It is the whole of the completion logic, so a test drives it118// with words alone.119func completeArgs(args []string) ([]string, int) {120	if len(args) == 0 {121		return topLevelCandidates(), compDirectiveNoFileComp122	}123	toComplete := args[len(args)-1]124	parse := parseCompletionArgs(args[:len(args)-1])125126	// The word under the cursor is the value of a flag that reads127	// one, so complete that flag's values, not a positional.128	if parse.pendingValueFlag != "" {129		return completeFlagValue(parse.pendingValueFlag, toComplete)130	}131132	// The word under the cursor starts a flag, so complete flag133	// names.134	if strings.HasPrefix(toComplete, "-") {135		return filterByPrefix(flagNames(), toComplete), compDirectiveNoFileComp136	}137138	// The word is a positional. With none given yet it names a139	// verb or a plugin domain; after the plugins verb it names a plugins140	// subcommand; a verb's own argument is a path, so the shell's file141	// completion answers it.142	switch {143	case len(parse.positionals) == 0:144		return filterByPrefix(topLevelCandidates(), toComplete), compDirectiveNoFileComp145	case len(parse.positionals) == 1 && parse.positionals[0] == pluginsVerb:146		return filterByPrefix(pluginsSubcommands, toComplete), compDirectiveNoFileComp147	default:148		return nil, compDirectiveDefault149	}150}151152// parseCompletionArgs walks the words before the cursor into their153// positionals, and marks a flag left waiting for its value.154func parseCompletionArgs(prior []string) completionParse {155	var parse completionParse156	for i := 0; i < len(prior); i++ {157		token := prior[i]158		if len(token) > 1 && token[0] == '-' {159			name, _, hasEquals := splitFlag(token)160			flag, ok := lookupCompletionFlag(name)161			if !ok || !flag.takesValue {162				continue163			}164			if hasEquals {165				continue166			}167			if i+1 < len(prior) {168				i++169				continue170			}171			parse.pendingValueFlag = name172			continue173		}174		parse.positionals = append(parse.positionals, token)175	}176	return parse177}178179// splitFlag splits a "-name=value" word into its name and value, and180// reports a bare "-name" as a name with no value. The standard flag181// package reads one and two dashes alike, so a "--name" spelling splits182// the same way.183func splitFlag(token string) (name, value string, hasEquals bool) {184	if equals := strings.IndexByte(token, '='); equals >= 0 {185		return token[:equals], token[equals+1:], true186	}187	return token, "", false188}189190// lookupCompletionFlag finds a flag by its exact name, and reads a191// "--name" spelling as the "-name" the table holds.192func lookupCompletionFlag(name string) (completionFlag, bool) {193	name = "-" + strings.TrimLeft(name, "-")194	for _, flag := range completionFlags {195		if flag.name == name {196			return flag, true197		}198	}199	return completionFlag{}, false200}201202// flagNames lists every flag name completion can offer.203func flagNames() []string {204	names := make([]string, 0, len(completionFlags))205	for _, flag := range completionFlags {206		names = append(names, flag.name)207	}208	return names209}210211// completeFlagValue offers the values a flag accepts. Every base212// flag names an address, a digest, a size, or a console, none of which213// this CLI can enumerate, so it offers nothing and leaves the file214// fallback off.215func completeFlagValue(flag, toComplete string) ([]string, int) {216	return nil, compDirectiveNoFileComp217}218219// filterByPrefix keeps the candidates the cursor's word begins.220func filterByPrefix(items []string, prefix string) []string {221	var kept []string222	for _, item := range items {223		if strings.HasPrefix(item, prefix) {224			kept = append(kept, item)225		}226	}227	return kept228}229230// completionScript prints the bash script that completes a direct231// invocation of the binary, under the liken name and the kubectl-liken232// name both. The shell sources it, and from then on it runs the hidden233// __complete verb and reads the protocol back. kubectl needs none of234// this: it runs the kubectl_complete-liken shim instead, which runs the235// same verb.236func completionScript(shell string, stdout io.Writer) error {237	if shell != "bash" {238		return fmt.Errorf("completion supports bash, not %q", shell)239	}240	fmt.Fprint(stdout, bashCompletionScript)241	return nil242}243244// The bash completion for a direct liken or kubectl-liken call. It245// mirrors the small part of cobra's generated script this CLI needs: run246// __complete with the words before the cursor and the word under it, read247// the candidates, and read the final ":N" directive. A per-operator CLI248// copies this text and changes only the binary name and the two function249// names.250const bashCompletionScript = `# bash completion for liken and kubectl-liken251__kubectl_liken_complete()252{253    local cur args out comp directive254    COMPREPLY=()255    cur="${COMP_WORDS[COMP_CWORD]}"256257    # The words after the binary name and before the cursor, then the258    # word under it, go to the hidden __complete verb.259    args=("${COMP_WORDS[@]:1:COMP_CWORD-1}")260    out="$("${COMP_WORDS[0]}" __complete "${args[@]}" "$cur" 2>/dev/null)"261262    # The last line is ":N", the ShellCompDirective bitmask.263    directive="${out##*$'\n'}"264    directive="${directive#:}"265    [[ "$directive" =~ ^[0-9]+$ ]] || directive=0266267    # Every other line is a candidate; a tab and a description may follow268    # the value, so keep only the value.269    while IFS= read -r comp; do270        [[ -z "$comp" || "$comp" == :* ]] && continue271        COMPREPLY+=("${comp%%$'\t'*}")272    done <<< "${out%$'\n'*}"273274    # Bit 2 keeps the cursor against the completion; bit 4 stops the file275    # fallback the "complete -o default" registration would otherwise add.276    (( (directive & 2) != 0 )) && compopt -o nospace 2>/dev/null277    (( (directive & 4) != 0 )) && compopt +o default 2>/dev/null278}279complete -o default -F __kubectl_liken_complete liken kubectl-liken280`
cli/dispatch.go 100.0%
1package main23// This file is the base binary's own prefix walk. kubectl dispatches4// kubectl liken audio capture by finding kubectl-liken-audio on PATH,5// so the two-layer command falls out of kubectl's own rule with no6// code here. The walk below gives the bare liken and kubectl-liken7// names the same behavior: liken audio capture finds the same binary8// and runs it with capture. The walk reads the domain from the9// arguments, never from this program's own name, so it behaves the10// same under either name.1112import (13	"os"14	"os/exec"15	"path/filepath"1617	"github.com/liken-sh/liken/liken/plugins"18)1920// findPlugin locates a domain's CLI. It reads PATH first, because a21// person who added the plugin directory to PATH expects that copy,22// and then the plugin directory itself, which is not always on PATH.23// It returns the path, or the empty string when neither holds the24// CLI.25func findPlugin(binDir, domain string) string {26	name := plugins.Name(domain)27	if path, err := exec.LookPath(name); err == nil {28		return path29	}30	candidate := filepath.Join(binDir, name)31	if info, err := os.Stat(candidate); err == nil && !info.IsDir() {32		return candidate33	}34	return ""35}3637// dispatchPlugin runs a domain's CLI with the remaining arguments, or38// reports that no plugin answers the domain. It replaces this process39// with the plugin (execTool), so the plugin owns the terminal and its40// exit code is the base binary's.41func dispatchPlugin(binDir, domain string, rest []string) (bool, error) {42	path := findPlugin(binDir, domain)43	if path == "" {44		return false, nil45	}46	return true, execTool(path, append([]string{plugins.Name(domain)}, rest...), os.Environ())47}
cli/main.go 82.8%
1// The liken CLI is the toolkit for producing and operating a2// deployment of liken.3//4// This program does everything an operator needs to do to a5// deployment that is not running a machine: minting or adopting a6// cluster identity, computing credentials from that identity,7// packing a deployment layer, and assembling install media from a8// public release. The other Go programs in this repo run inside the9// machine: init as PID 1, and the operators as pods. This program10// runs on the operator's workstation, and it ships with public11// releases, so producing a cluster never requires this repo or a12// build.13//14// The command is a thin dispatcher. The logic lives with the domain15// that owns it (the identity package, and later the image and16// releases packages). Because of this, the CLI stays a table of17// names, while each capability keeps its own file, its own tests,18// and its own documentation.19package main2021import (22	"bufio"23	"flag"24	"fmt"25	"io"26	"os"27	"strings"2829	"github.com/liken-sh/liken/liken/identity"30	"github.com/liken-sh/liken/liken/image"31	"github.com/liken-sh/liken/liken/machine"32	"github.com/liken-sh/liken/liken/plugins"33	"github.com/liken-sh/liken/liken/releases"34	"github.com/liken-sh/liken/liken/scaffold"35)3637// A consoleList collects repeated -console flags in order.38type consoleList []string3940func (c *consoleList) String() string { return fmt.Sprint([]string(*c)) }4142func (c *consoleList) Set(v string) error {43	*c = append(*c, v)44	return nil45}4647const usage = `liken: the toolkit for setting up and running a liken cluster4849usage:5051  liken new <directory>52      Start a deployment: answer a few questions and get a directory53      of manifests: cluster.yaml and one file per machine, with54      comments that teach every field. The other commands build on55      this directory.5657  liken mint <identity-dir>58      Create a new cluster identity: the certificates and join token59      that every machine in one cluster shares.6061  liken adopt <harvest-dir> <identity-dir>62      Take identity files copied off an existing cluster's disk and63      arrange them as an identity directory. The cluster does not64      have to be a liken cluster: any k3s cluster's identity can be65      adopted, so liken machines can join a cluster you already run.6667  liken kubeconfig [-server URL] <deployment-dir>68      Write an admin kubeconfig: the credential kubectl uses to69      administer the cluster. The endpoint comes from the70      deployment's cluster.yaml; -server overrides it, for when71      that address is not reachable from this machine.7273  liken approve-reboot [-server URL] <deployment-dir> <machine>74      Report what a machine waits for, and grant it the one75      disruption its rebootPolicy withholds. The grant is an76      annotation naming the staged change's hash, so it spends77      itself when the change applies, and running this twice is78      the same grant. When nothing waits, this writes nothing.7980  liken request-reboot [-server URL] <deployment-dir> <machine>81      Ask a machine to reboot when no change asks it to: a driver82      bound to the wrong device, or a machine you are experimenting83      on. The machine still takes its turn from the cluster,84      cordons, and drains. On rebootPolicy: Auto that is the whole85      procedure. On Manual the request waits for approve-reboot.86      The request names the running boot, so it spends itself when87      the machine comes back.8889  liken kubectl [-server URL] <deployment-dir> [args...]90  liken stern   [-server URL] <deployment-dir> [args...]91  liken flux    [-server URL] <deployment-dir> [args...]92      Run the named tool from your PATH against the deployment's93      cluster: resolve the credential, set KUBECONFIG, and hand the94      terminal to the tool. Give liken's -server before the95      directory; everything after the directory goes to the tool.9697  liken layer <manifests-dir> <identity-dir> <output.cpio>98      Pack your cluster's half of the operating system into one small99      archive: your cluster and machine manifests, and your identity.100      Kernel modules are not part of it: the system image carries the101      kernel's whole module tree, and spec.modules loads from that102      tree at boot.103104  liken fetch [-digest sha256:<hex>] <source-url> <version|latest> <channel-dir>105      Download a published release from a channel into a local106      channel directory, verifying every artifact against the107      release's document. Pass "latest" to take whatever the channel108      currently names newest. -digest pins the document itself to a109      catalog entry's digest, closing the trust chain end to end.110111  liken media <release-dir> <deployment.cpio> <output.cpio>112      Build a bootable install image from a downloaded release and113      your deployment layer. Machines install themselves from it.114115  liken stick [-console ttyS0] <release-dir> <deployment.cpio> <output.img>116      Build the USB install stick's disk image from a downloaded117      release and your deployment layer: one stick for the whole118      deployment, with a boot menu listing every machine by name.119      Boot it, pick the machine you're standing at, and follow the120      console: the install holds until you remove the stick and press121      Enter, then powers the machine off. -console (repeatable) adds a122      console= argument to every entry; the machines keep it123      permanently.124125  liken bundle [-slot-size 1Gi] <vmlinuz> <liken.sqfs> <boot.cpio> <microcode.cpio> <liken-cli> <systemd-boot.efi> <grub-boot.img> <grub-core.img> <licenses.md> <channel-dir> <version> [component=version ...]126      Lay out a release: copy the nine files into the channel and127      write the release.yaml that names each one by its digest. The128      version is a calendar date and serial (2026.07.11-001); the129      component=version pairs record which upstreams shipped inside,130      since the date deliberately doesn't say.131132  liken serve <channel-dir> [address]133      Share a release channel over plain HTTP so machines can134      download from it. The address defaults to :8017.135136  liken index -source <url> <output-dir> < keys137      Render a channel's index: a front page listing every release, a138      page per release giving its catalog entry and its artifacts, a139      page for the source mirror, and the versions.yaml document that140      lists every release for a script. Read the channel's object keys141      from standard input, one per line, and read each release's142      document from the channel itself. The index is derived, so143      rendering it again over the same channel repairs whatever is144      stale.145146  liken plugins sync|list|remove [-server URL] <deployment-dir>147      The plugins group. sync reads the operators the cluster148      runs, pulls each one's CLI from the registry into149      ~/.liken/plugins/bin, and prints the line to add that directory150      to PATH. list reports each installed CLI's version and the151      operator version it faces, and marks drift. remove <domain>152      deletes one installed CLI. A domain that names no command, like153      "liken audio capture", walks to kubectl-liken-<domain>.154155  liken completion bash156      Print the bash completion script for this toolkit.157158  liken version159      Print this toolkit's version.160161An identity directory holds the certificates and join token that162make a cluster one cluster. Some of the files are private keys, so163keep the directory out of version control.164165A deployment layer is a small archive holding everything about the166operating system that is yours and not liken's. A machine boots the167generic liken image and your layer together, and the kernel joins168them into one system.169170A release channel is a directory any web server can share: one171subdirectory per version, each holding the release's files and a172release.yaml that names every file by its sha256 digest, so a173machine can check that what it downloaded is what was published.174`175176func main() {177	if err := run(os.Args[1:]); err != nil {178		fmt.Fprintf(os.Stderr, "liken: %v\n", err)179		os.Exit(1)180	}181}182183func run(args []string) error {184	// The __complete verb answers the shell's completion request. It185	// runs before the command check, because a request for the first186	// word carries no command and a person waits on the answer.187	if len(args) > 0 && args[0] == "__complete" {188		return runComplete(args[1:], os.Stdout)189	}190	if len(args) == 0 {191		fmt.Fprint(os.Stderr, usage)192		return fmt.Errorf("a command is required")193	}194	switch args[0] {195	case "new":196		if len(args) != 2 {197			return fmt.Errorf("usage: liken new <directory>")198		}199		return scaffold.New(args[1], os.Stdin, os.Stdout)200	case "mint":201		if len(args) != 2 {202			return fmt.Errorf("usage: liken mint <identity-dir>")203		}204		return identity.Mint(args[1], os.Stdout)205	case "adopt":206		if len(args) != 3 {207			return fmt.Errorf("usage: liken adopt <harvest-dir> <identity-dir>")208		}209		return identity.Adopt(args[1], args[2], os.Stdout)210	case "kubeconfig":211		fs := flag.NewFlagSet("kubeconfig", flag.ContinueOnError)212		server := fs.String("server", "", "the API server address, when cluster.yaml's endpoint is not reachable from here")213		if err := fs.Parse(args[1:]); err != nil {214			return err215		}216		if fs.NArg() != 1 {217			return fmt.Errorf("usage: liken kubeconfig [-server URL] <deployment-dir>")218		}219		_, err := writeKubeconfig(fs.Arg(0), *server, os.Stdout)220		return err221	case "kubectl", "stern", "flux":222		return passthrough(args[0], args[1:])223	case "approve-reboot":224		return approveReboot(args[1:], os.Stdout)225	case "request-reboot":226		return requestReboot(args[1:], os.Stdout)227	case "plugins":228		return pluginsCommand(args[1:], os.Stdout)229	case "layer":230		if len(args) != 4 {231			return fmt.Errorf("usage: liken layer <manifests-dir> <identity-dir> <output.cpio>")232		}233		return image.Layer(args[1], args[2], args[3])234	case "fetch":235		fs := flag.NewFlagSet("fetch", flag.ContinueOnError)236		digest := fs.String("digest", "", "pin the release document to a catalog entry's sha256:<hex> digest")237		if err := fs.Parse(args[1:]); err != nil {238			return err239		}240		if fs.NArg() != 3 {241			return fmt.Errorf("usage: liken fetch [-digest sha256:<hex>] <source-url> <version|latest> <channel-dir>")242		}243		return releases.Fetch(fs.Arg(0), fs.Arg(1), *digest, fs.Arg(2), os.Stdout)244	case "media":245		if len(args) != 4 {246			return fmt.Errorf("usage: liken media <release-dir> <deployment.cpio> <output.cpio>")247		}248		return image.Media(args[1], args[2], args[3], os.Stdout)249	case "stick":250		// The CLI's first flags. This uses the standard library's251		// flag package over a subcommand's own FlagSet, so the252		// positional arguments stay positional.253		fs := flag.NewFlagSet("stick", flag.ContinueOnError)254		var consoles consoleList255		fs.Var(&consoles, "console", "add console=<value> to every menu entry (repeatable)")256		if err := fs.Parse(args[1:]); err != nil {257			return err258		}259		if fs.NArg() != 3 {260			return fmt.Errorf("usage: liken stick [-console ttyS0] <release-dir> <deployment.cpio> <output.img>")261		}262		return image.Stick(fs.Arg(0), fs.Arg(1), fs.Arg(2), consoles, os.Stdout)263	case "bundle":264		fs := flag.NewFlagSet("bundle", flag.ContinueOnError)265		slotSize := fs.String("slot-size", "1Gi", "the boot slot size that every artifact, plus headroom, must fit")266		if err := fs.Parse(args[1:]); err != nil {267			return err268		}269		if fs.NArg() < 11 {270			return fmt.Errorf("usage: liken bundle [-slot-size 1Gi] <vmlinuz> <liken.sqfs> <boot.cpio> <microcode.cpio> <liken-cli> <systemd-boot.efi> <grub-boot.img> <grub-core.img> <licenses.md> <channel-dir> <version> [component=version ...]")271		}272		var components []machine.ReleaseComponent273		for _, arg := range fs.Args()[11:] {274			name, version, ok := strings.Cut(arg, "=")275			if !ok || name == "" || version == "" {276				return fmt.Errorf("component %q must be name=version", arg)277			}278			components = append(components, machine.ReleaseComponent{Name: name, Version: version})279		}280		return releases.Bundle(fs.Arg(0), fs.Arg(1), fs.Arg(2), fs.Arg(3), fs.Arg(4), fs.Arg(5), fs.Arg(6), fs.Arg(7), fs.Arg(8), fs.Arg(9), fs.Arg(10), *slotSize, components, os.Stdout)281	case "index":282		fs := flag.NewFlagSet("index", flag.ContinueOnError)283		source := fs.String("source", "", "the channel's base URL, where the pages will be served")284		if err := fs.Parse(args[1:]); err != nil {285			return err286		}287		if fs.NArg() != 1 || *source == "" {288			return fmt.Errorf("usage: liken index -source <url> <output-dir> < keys")289		}290		keys, err := readLines(os.Stdin)291		if err != nil {292			return err293		}294		return releases.Index(*source, keys, fs.Arg(0), os.Stdout)295	case "serve":296		addr := ":8017"297		switch len(args) {298		case 2:299		case 3:300			addr = args[2]301		default:302			return fmt.Errorf("usage: liken serve <channel-dir> [address]")303		}304		return releases.Serve(args[1], addr)305	case "version":306		fmt.Println(machine.Version)307		return nil308	case "completion":309		shell := ""310		if len(args) > 1 {311			shell = args[1]312		}313		return completionScript(shell, os.Stdout)314	default:315		// A first argument that names no command is read as a plugin316		// domain: liken audio capture walks the prefix to317		// kubectl-liken-audio and runs it, the same walk kubectl runs318		// for kubectl liken audio capture.319		if binDir, err := plugins.BinDir(); err == nil {320			if dispatched, err := dispatchPlugin(binDir, args[0], args[1:]); dispatched {321				return err322			}323		}324		fmt.Fprint(os.Stderr, usage)325		return fmt.Errorf("unknown command %q", args[0])326	}327}328329// readLines collects a list given on standard input, one entry per330// line, and drops blank lines. The index command takes a channel's331// object keys this way, because the command that lists a bucket holds332// a credential and this program holds none.333func readLines(r io.Reader) ([]string, error) {334	var lines []string335	scanner := bufio.NewScanner(r)336	for scanner.Scan() {337		if line := strings.TrimSpace(scanner.Text()); line != "" {338			lines = append(lines, line)339		}340	}341	return lines, scanner.Err()342}
cli/plugins.go 84.3%
1package main23// The plugins command group installs and manages the per-operator4// CLIs that give liken its two-layer commands. Each operator ships a5// binary named kubectl-liken-<domain> beside its operator image at6// the same version. sync reads the operators the cluster runs, pulls7// each CLI out of the registry, and writes it into the plugin8// directory. list reports what is installed against what the cluster9// runs. remove deletes one. The registry pull, the filesystem layout,10// and the comparison live in the plugins package; this file is the11// cluster read and the dispatch.1213import (14	"flag"15	"fmt"16	"io"17	"os"18	"os/exec"19	"path/filepath"20	"runtime"21	"strings"2223	"github.com/liken-sh/liken/kubernetes/apiclient"24	"github.com/liken-sh/liken/liken/kubernetes"25	"github.com/liken-sh/liken/liken/plugins"26)2728// clusterClient resolves the deployment's credential and builds a29// client for its cluster, the same resolution the other cluster30// commands use (cluster.go).31func clusterClient(dir, server string) (*apiclient.Client, error) {32	kubeconfigPath, err := writeKubeconfig(dir, server, io.Discard)33	if err != nil {34		return nil, err35	}36	return kubernetes.KubeconfigClient(kubeconfigPath)37}3839// operatorsFromWorkloads maps the plugin workloads to the domains and40// images the sync pulls, dropping any that name no domain or no image.41func operatorsFromWorkloads(workloads []kubernetes.Workload) []plugins.Operator {42	var operators []plugins.Operator43	for _, w := range workloads {44		domain, image := w.PluginDomain(), w.OperatorImage()45		if domain == "" || image == "" {46			continue47		}48		operators = append(operators, plugins.Operator{Domain: domain, Image: image})49	}50	return operators51}5253// operatorVersions maps each domain to the version its operator image54// carries, the version every CLI compares against.55func operatorVersions(workloads []kubernetes.Workload) map[string]string {56	versions := map[string]string{}57	for _, w := range workloads {58		domain, image := w.PluginDomain(), w.OperatorImage()59		if domain == "" || image == "" {60			continue61		}62		if version, err := plugins.ImageVersion(image); err == nil {63			versions[domain] = version64		}65	}66	return versions67}6869// pluginVersion asks an installed CLI for its stamped version. It is70// a variable so tests observe the call instead of running a real71// binary.72var pluginVersion = execPluginVersion7374func execPluginVersion(binDir, domain string) (string, error) {75	out, err := exec.Command(filepath.Join(binDir, plugins.Name(domain)), "--version").Output()76	if err != nil {77		return "", err78	}79	return strings.TrimSpace(string(out)), nil80}8182// installedVersions reads the stamped version of each installed CLI.83// A CLI that does not answer --version reports an empty version, so84// the list still names it rather than dropping it.85func installedVersions(binDir string, domains []string) map[string]string {86	versions := map[string]string{}87	for _, domain := range domains {88		version, err := pluginVersion(binDir, domain)89		if err != nil {90			version = ""91		}92		versions[domain] = version93	}94	return versions95}9697func pluginsCommand(args []string, out io.Writer) error {98	if len(args) == 0 {99		return fmt.Errorf("usage: liken plugins sync|list|remove")100	}101	switch args[0] {102	case "sync":103		return pluginsSync(args[1:], out)104	case "list":105		return pluginsList(args[1:], out)106	case "remove":107		return pluginsRemove(args[1:], out)108	default:109		return fmt.Errorf("usage: liken plugins sync|list|remove")110	}111}112113func pluginsSync(args []string, out io.Writer) error {114	fs := flag.NewFlagSet("plugins sync", flag.ContinueOnError)115	server := fs.String("server", "", "the API server address, when cluster.yaml's endpoint is not reachable from here")116	if err := fs.Parse(args); err != nil {117		return err118	}119	if fs.NArg() != 1 {120		return fmt.Errorf("usage: liken plugins sync [-server URL] <deployment-dir>")121	}122	c, err := clusterClient(fs.Arg(0), *server)123	if err != nil {124		return err125	}126	workloads, err := kubernetes.ListPluginWorkloads(c)127	if err != nil {128		return err129	}130	binDir, err := plugins.BinDir()131	if err != nil {132		return err133	}134	return plugins.Sync(operatorsFromWorkloads(workloads), runtime.GOARCH, binDir, os.Getenv("PATH"), out)135}136137func pluginsList(args []string, out io.Writer) error {138	fs := flag.NewFlagSet("plugins list", flag.ContinueOnError)139	server := fs.String("server", "", "the API server address, when cluster.yaml's endpoint is not reachable from here")140	if err := fs.Parse(args); err != nil {141		return err142	}143	if fs.NArg() != 1 {144		return fmt.Errorf("usage: liken plugins list [-server URL] <deployment-dir>")145	}146	c, err := clusterClient(fs.Arg(0), *server)147	if err != nil {148		return err149	}150	workloads, err := kubernetes.ListPluginWorkloads(c)151	if err != nil {152		return err153	}154	binDir, err := plugins.BinDir()155	if err != nil {156		return err157	}158	domains, err := plugins.Installed(binDir)159	if err != nil {160		return err161	}162	entries := plugins.Merge(installedVersions(binDir, domains), operatorVersions(workloads))163	plugins.Render(entries, out)164	return nil165}166167func pluginsRemove(args []string, out io.Writer) error {168	if len(args) != 1 {169		return fmt.Errorf("usage: liken plugins remove <domain>")170	}171	binDir, err := plugins.BinDir()172	if err != nil {173		return err174	}175	if err := plugins.Remove(binDir, args[0]); err != nil {176		return err177	}178	fmt.Fprintf(out, "removed %s\n", plugins.Name(args[0]))179	return nil180}
cli/request.go 84.4%
1package main23// request-reboot is approve-reboot's counterpart. Every other reboot4// liken performs applies a staged change, so a machine that agrees5// with every document it was given has no way to reboot, and a6// machine with no shell has no other way to be told. This command is7// how a person asks: it reads status.boot.time, derives the running8// boot's identity, and writes the request-reboot annotation naming9// it. Naming the running boot makes the request one-shot10// (machine.RequestRebootAnnotation explains why), so a later request11// works with no cleanup in between.12//13// The request skips no coordination. The machine still takes its14// turn from the cluster's disruption budget, cordons, and drains. On15// rebootPolicy: Manual it parks where a staged change parks, and16// liken approve-reboot releases it.1718import (19	"flag"20	"fmt"21	"io"22	"strings"2324	"github.com/liken-sh/liken/liken/kubernetes"25	"github.com/liken-sh/liken/liken/machine"26)2728// renderRequest reports what the request set in motion. What happens29// next depends on the machine's rebootPolicy, so the report says30// which of the two paths this machine is on, and names the command31// that finishes the second one.32func renderRequest(m *machine.Machine, dir, short string) string {33	var b strings.Builder34	fmt.Fprintf(&b, "requested: %s=%s\n", machine.RequestRebootAnnotation, short)35	if m.Spec.RebootPolicyOrDefault() == machine.RebootAuto {36		fmt.Fprintf(&b, "%s takes a reboot turn from the cluster, drains its workloads, and reboots.\n", m.Metadata.Name)37	} else {38		fmt.Fprintf(&b, "%s has rebootPolicy: Manual, so the request waits for you. Release it with:\n", m.Metadata.Name)39		fmt.Fprintf(&b, "  liken approve-reboot %s %s\n", dir, m.Metadata.Name)40	}41	fmt.Fprint(&b, "Nothing is staged, so the machine comes back on the documents it runs now.\n")42	return b.String()43}4445// requestReboot is the command: resolve the credential, read the46// machine, name its boot, and ask.47func requestReboot(args []string, out io.Writer) error {48	fs := flag.NewFlagSet("request-reboot", flag.ContinueOnError)49	server := fs.String("server", "", "the API server address, when cluster.yaml's endpoint is not reachable from here")50	if err := fs.Parse(args); err != nil {51		return err52	}53	if fs.NArg() != 2 {54		return fmt.Errorf("usage: liken request-reboot [-server URL] <deployment-dir> <machine>")55	}56	kubeconfigPath, err := writeKubeconfig(fs.Arg(0), *server, io.Discard)57	if err != nil {58		return err59	}60	c, err := kubernetes.KubeconfigClient(kubeconfigPath)61	if err != nil {62		return err63	}64	m, err := kubernetes.GetMachine(c, fs.Arg(1))65	if err != nil {66		return err67	}68	// A machine that reports no boot time has published nothing for69	// the request to name. That is a machine still starting, or one70	// that has never reported at all, and either way there is no71	// boot to ask about yet.72	bootID := machine.BootID(m.Status.Boot)73	if bootID == "" {74		return fmt.Errorf("%s reports no boot time, so there is no boot to request a reboot of; the machine is still starting, or it has never reported", m.Metadata.Name)75	}76	short := fmt.Sprintf("%.12s", bootID)77	patch := fmt.Sprintf(`{"metadata":{"annotations":{%q:%q}}}`, machine.RequestRebootAnnotation, short)78	if err := kubernetes.PatchJSON(c, kubernetes.MachinesPath+"/"+m.Metadata.Name, []byte(patch)); err != nil {79		return err80	}81	fmt.Fprint(out, renderRequest(m, fs.Arg(0), short))82	return nil83}
cluster-operator/channel.go 93.8%
1package main23// This poller checks the release channel. It shows how a cluster4// detects that a new version exists.5//6// The channel's root document, channel.yaml, names the latest7// published version. This poller fetches that document so the sweep8// can show the answer in the Cluster's status.9//10// Two facts shape this design. First, the answer is advisory. The11// machine package's channel.go explains why the design keeps the12// answer outside the trust chain. Because the answer is advisory, a13// stale or missing answer costs nothing. The poller keeps the last14// answer it received and tries again later. Second, the sweep runs15// every ten seconds, but the channel changes about once a week. So16// the poller fetches the document on a long interval instead of every17// sweep. Two things make the poller fetch immediately: a new source,18// or a new value in the Cluster's liken.sh/check-releases annotation.19// The annotation is a declared request to poll immediately (the20// cluster package's CheckReleasesAnnotation explains why the request21// is an annotation and not a spec field).22//23// The fetch runs on its own goroutine. The machine operator's release24// fetcher uses the same method. The sweep must keep judging the fleet25// at its own speed, and a slow or dead release server must never26// delay a Lost verdict. The Observe function determines what to do27// and returns immediately; the goroutine reports its result back28// through the mutex.2930import (31	"fmt"32	"io"33	"net/http"34	"strings"35	"sync"36	"time"3738	"github.com/liken-sh/liken/liken/cluster"39	"github.com/liken-sh/liken/liken/machine"40)4142// channelPollInterval sets how long an answer stays fresh, so the43// poller does not ask again during this time. People publish releases44// on a human timescale, and the poll answers the question "should45// someone think about upgrading?", not a real-time telemetry46// question. A person who just published a release and wants the47// answer now can set the check-releases annotation to request an48// immediate poll.49const channelPollInterval = 6 * time.Hour5051type channelPoller struct {52	mu        sync.Mutex53	source    string    // the channel this poller last asked54	check     string    // the last check-releases annotation this poller handled55	polled    time.Time // when the last attempt started56	inFlight  bool57	available string // the channel's last known answer5859	// fetch performs the actual HTTP request. Tests can replace it60	// with a fake function, so they do not need a real channel61	// server.62	fetch func(url string) ([]byte, error)63}6465func newChannelPoller() *channelPoller {66	return &channelPoller{fetch: fetchChannelDocument}67}6869// Observe runs once per sweep. It receives the spec's current70// releases section and the Cluster's check-releases annotation. It71// determines whether a poll is due: the source or the annotation72// changed, or the last answer is too old. If a poll is due, Observe73// starts it in the background. Observe never blocks.74func (p *channelPoller) Observe(releases cluster.ClusterReleasesSpec, check string, now time.Time) {75	p.mu.Lock()76	defer p.mu.Unlock()7778	// If there is no source, there is nothing to poll. A cluster79	// without a channel has no available version, and it does not80	// keep an available version from a channel it had before.81	if releases.Source == "" {82		p.source, p.check, p.available = "", "", ""83		return84	}8586	if releases.Source != p.source {87		// A different channel's answer does not apply here. The88		// poller drops it and asks the new channel immediately.89		p.source, p.check, p.available = releases.Source, check, ""90		p.polled = time.Time{}91	} else if check != p.check {92		// This is the request to poll immediately. The poller honors93		// it once, by marking the last answer as stale right away.94		// The last answer stays visible until the fresh answer95		// replaces it.96		p.check = check97		p.polled = time.Time{}98	}99100	if p.inFlight || now.Sub(p.polled) < channelPollInterval {101		return102	}103	// The poller records the time of this attempt before it runs.104	// This way, a dead server is asked again on the interval, not on105	// every ten-second sweep.106	p.polled = now107	p.inFlight = true108	go p.poll(p.source)109}110111// Available returns the latest version the channel last announced.112// It is empty until a poll succeeds.113func (p *channelPoller) Available() string {114	p.mu.Lock()115	defer p.mu.Unlock()116	return p.available117}118119// poll fetches and parses the channel document. It keeps the answer120// only if the sweep is still asking about the same source.121func (p *channelPoller) poll(source string) {122	url := strings.TrimSuffix(source, "/") + "/channel.yaml"123	raw, err := p.fetch(url)124	var latest string125	if err == nil {126		var channel *machine.Channel127		if channel, err = machine.ParseChannel(raw); err == nil {128			latest = channel.Latest129		}130	}131132	p.mu.Lock()133	defer p.mu.Unlock()134	p.inFlight = false135	if source != p.source {136		return137	}138	if err != nil {139		// The last answer stands. Advisory data may be stale. The140		// interval, or an edit to the check value, will trigger141		// another poll.142		fmt.Printf("polling the release channel %s: %v\n", url, err)143		return144	}145	p.available = latest146}147148// fetchChannelDocument sends a GET request for the channel document149// through the default transport.150func fetchChannelDocument(url string) ([]byte, error) {151	return fetchChannelDocumentVia(nil, url)152}153154// fetchChannelDocumentVia sends a GET request for the channel document155// through transport, or through the default transport when transport156// is nil, using a client with a bounded size and time limit. The157// document is only a few lines of YAML, so the size limit is generous.158// The time limit stops a stuck server from keeping the poll in progress159// forever. A test passes a transport that reaches its fake server160// without a socket.161func fetchChannelDocumentVia(transport http.RoundTripper, url string) ([]byte, error) {162	client := &http.Client{Transport: transport, Timeout: 30 * time.Second}163	resp, err := client.Get(url)164	if err != nil {165		return nil, err166	}167	defer resp.Body.Close()168	if resp.StatusCode != http.StatusOK {169		return nil, fmt.Errorf("%s: %s", url, resp.Status)170	}171	return io.ReadAll(io.LimitReader(resp.Body, 1<<20))172}
cluster-operator/events.go 100.0%
1package main23// The Events the cluster operator posts about the Machines and the4// Cluster. `kubectl describe` lists them under the status, so a person5// reads the history of a rollout, or of a machine that went silent,6// with no log to open. kubernetes/events writes them, and root plan 787// gives the rule for a condition, an Event, and a log line.8//9// Both kinds are cluster-scoped, so their Events are in the namespace10// default. `kubectl describe` finds them there, and `kubectl events11// --for cluster/<name>` finds them only with -n default or -A.12//13//   - Each condition transition of the Cluster posts one Event, with14//     the condition's reason, after the status write lands: a fleet15//     that goes MachinesDegraded or AllMachinesReady, and a rollout16//     that goes RollingOut, RolloutStalled, or RolloutComplete.17//   - The verdicts this program writes onto a Machine post one Event18//     each on that Machine: MachineLost, RebootTurnGranted, and19//     RebootTurnReclaimed. Each one is an action this program takes on20//     an object that another program owns, so each Event names the21//     action and not the condition's reason.22//   - The flux feature's one-off actions post one Event each on the23//     Cluster: the mint of the deploy key and the planting of the24//     engine.25//26// Nothing that repeats on each sweep posts an Event: a poll of the27// release channel, an eviction the budget refuses, or the sweep28// itself.2930import (31	"github.com/liken-sh/liken/kubernetes/conditions"32	"github.com/liken-sh/liken/kubernetes/events"33	"github.com/liken-sh/liken/liken/api"34	"github.com/liken-sh/liken/liken/cluster"35	"github.com/liken-sh/liken/liken/machine"36)3738// The reasons of the Events that are not condition transitions. A39// transition's Event takes the condition's own reason.40const (41	reasonMachineLost         = "MachineLost"42	reasonRebootTurnGranted   = "RebootTurnGranted"43	reasonRebootTurnReclaimed = "RebootTurnReclaimed"44	reasonFluxDeployKeyMinted = "FluxDeployKeyMinted"45	reasonFluxEngineSeeded    = "FluxEngineSeeded"46)4748// machineReference answers the object reference of a Machine, for an49// Event about it.50func machineReference(m *machine.Machine) events.ObjectReference {51	return events.ObjectReference{52		APIVersion: api.APIVersion, Kind: machineKind,53		Name: m.Metadata.Name, UID: m.Metadata.UID,54	}55}5657// clusterReference answers the object reference of the Cluster, for an58// Event about it.59func clusterReference(c *cluster.Cluster) events.ObjectReference {60	return events.ObjectReference{61		APIVersion: api.APIVersion, Kind: clusterKind,62		Name: c.Metadata.Name, UID: c.Metadata.UID,63	}64}6566// postClusterTransitions posts one Event for each Cluster condition67// that transitioned from stored to written, with the condition's reason68// and message.69func postClusterTransitions(recorder *events.Recorder, c *cluster.Cluster, stored, written []api.Condition) {70	for _, t := range api.Transitions(stored, written) {71		recorder.Transition(clusterReference(c), t, clusterBadStatus(t))72	}73}7475// clusterBadStatus answers the status of a Cluster condition that76// needs a person, for events.Recorder.Transition. A degraded machine,77// a stalled rollout, and a declined flux teardown each need one. A78// fleet in the middle of a rollout is MachinesReady False too, and79// that needs no one.80func clusterBadStatus(c api.Condition) conditions.Status {81	switch c.Reason {82	case "MachinesDegraded", "RolloutStalled", "NotPlantedByLiken":83		return c.Status84	}85	return ""86}
cluster-operator/fleet.go 95.4%
1package main23// The fleet sweep is the cluster operator's main task.4//5// Each machine's operator reports on itself. This leaves one gap6// that nothing else covers: a dead machine cannot report that it is7// dead. Its last written status stays in the API and reads Ready8// forever, which is worse than no status at all. Kubernetes has this9// same problem with kubelets, and it solves the problem with10// heartbeats. The kubelet renews a lease every few seconds, and the11// node controller turns a silent lease into a NotReady Node. The12// Machine object gets the same treatment here. Each machine's13// operator renews its own heartbeat lease (see the kubernetes14// package), and this sweep marks a machine with the Lost phase when15// its lease goes silent. The sweep also produces the cluster's16// headcount. The same pass over the Machine list produces the17// ready-out-of-total tally that the Cluster's status carries.18//19// Writing another machine's status breaks the one-writer-per-object20// rule that the machine operators otherwise follow. So the sweep is21// careful about when it writes: it only writes a machine whose22// heartbeat is already stale. A stale heartbeat means the machine's23// own operator has stopped writing, so the two writers can never24// actually write to the same object at the same time. The moment the25// machine returns, its own operator's next pass overwrites the Lost26// verdict with fresh observations. No cleanup step is needed.2728import (29	"errors"30	"fmt"31	"slices"32	"strings"33	"time"3435	"github.com/liken-sh/liken/kubernetes/apiclient"36	"github.com/liken-sh/liken/liken/api"37	"github.com/liken-sh/liken/liken/cluster"38	"github.com/liken-sh/liken/liken/machine"39)4041// A fleetSweep holds one pass's verdict over the whole fleet: which42// machines to declare Lost, the headcount for the Cluster's status,43// the MachinesReady condition that carries the full detail, and the44// phase that summarizes the verdict in one word. Every Machine uses45// this same conditions-then-phase arrangement, applied here to the46// fleet. The function decideFleetSweep only computes the verdict; the47// function sweepFleet carries it out.48type fleetSweep struct {49	lost      []string50	tally     cluster.MachineTally51	condition api.Condition52	phase     api.Phase5354	// The two counts that the fleet's metrics publish (metrics.go).55	// They are part of the verdict because this loop is the one place56	// that judges every machine at once, and a second pass over the57	// same list could only ever repeat this one.58	phases    map[api.Phase]int59	approvals int60}6162// decideFleetSweep judges each machine by its effective phase. A63// machine counts toward ready only when its phase is Ready and its64// heartbeat is fresh. The verdict sorts each machine that is not65// Ready into one of two groups: mid-transition (rebooting into a66// change, waiting for a reboot, or booting) or unwell (Lost, Blocked,67// or otherwise degraded). Unwell outranks mid-transition. The68// MachinesReady condition names the affected machines, so nobody has69// to search for them.70func decideFleetSweep(machines []machine.Machine, heard map[string]time.Time, now time.Time) fleetSweep {71	s := fleetSweep{tally: cluster.MachineTally{Total: len(machines)}, phases: map[api.Phase]int{}}72	var transitioning, unwell []string73	for i := range machines {74		m := &machines[i]75		effective := effectivePhase(m, heard, now)76		if effective == api.PhaseLost && m.Status.Phase != api.PhaseLost {77			s.lost = append(s.lost, m.Metadata.Name)78		}79		s.phases[effective]++80		if awaitsApproval(m) {81			s.approvals++82		}8384		switch effective {85		case api.PhaseReady:86			s.tally.Ready++87		case api.PhaseUpdating, api.PhaseUpdatePending, api.PhaseDownloading, api.PhaseBooting:88			transitioning = append(transitioning, m.Metadata.Name)89		default:90			unwell = append(unwell, m.Metadata.Name)91		}92	}93	s.tally.Summary = fmt.Sprintf("%d/%d", s.tally.Ready, s.tally.Total)9495	switch {96	case len(unwell) > 0:97		s.phase = api.PhaseDegraded98		s.condition = api.Condition{99			Type: "MachinesReady", Status: api.ConditionFalse, Reason: "MachinesDegraded",100			Message: fmt.Sprintf("%s machines ready; unwell: %s", s.tally.Summary, strings.Join(unwell, ", ")),101		}102	case len(transitioning) > 0:103		s.phase = api.PhaseUpdating104		s.condition = api.Condition{105			Type: "MachinesReady", Status: api.ConditionFalse, Reason: "MachinesUpdating",106			Message: fmt.Sprintf("%s machines ready; mid-transition: %s", s.tally.Summary, strings.Join(transitioning, ", ")),107		}108	default:109		s.phase = api.PhaseReady110		s.condition = api.Condition{111			Type: "MachinesReady", Status: api.ConditionTrue, Reason: "AllMachinesReady",112			Message: fmt.Sprintf("all %d machines are ready", s.tally.Total),113		}114	}115	return s116}117118// awaitsApproval reports whether this machine holds a staged change119// that waits for a person. A machine on the Manual reboot policy120// stages its change and stops there, and the reason on the121// convergence condition says so. The cluster operator counts these,122// because the count is a fleet fact: it says how many machines a123// person has to visit before the fleet converges.124func awaitsApproval(m *machine.Machine) bool {125	for _, c := range m.Status.Conditions {126		if c.Reason == "RebootPending" || c.Reason == "RestartPending" {127			return true128		}129	}130	return false131}132133// sweepFleet carries out the sweep. It lists the fleet and its134// heartbeats, determines the verdict, marks the silent machines Lost,135// and publishes the verdict on the Cluster. The available parameter136// is the channel poller's last answer, probe holds when the flux137// engine probe last asked, and steward holds the terminating pods the138// steward has reported. The caller passes all three in, so the sweep139// itself stays a function of its arguments.140//141// A pass that cannot read the fleet returns the reason. The reading142// is the whole pass here: with no list of machines there is no143// verdict to reach, so this is the failure that the layer 2 error144// counter counts (metrics.go).145//146// The order of the writes matters. The Cluster's status goes first,147// because it carries the grant ledger, and a grant goes out only after148// the API server stores the grant's entry (rollout.go). The Machine149// writes follow it: the grants, the confirms, the reclaims, and the150// Lost verdicts.151func sweepFleet(reads *fleetReader, clusterDoc *cluster.Cluster, available string, probe *engineProbe,152	steward *podSteward, cm *clusterMetrics, now time.Time) error {153	c := reads.client154	machines, err := reads.machines()155	if err != nil {156		fmt.Printf("listing machines for the fleet sweep: %v\n", err)157		return err158	}159	// Each Machine whose ledger entry shows no grant in its copy comes160	// from the API server, because the decision to free the entry's161	// slot needs the Machine's current version (rollout.go).162	machines, err = reads.readTurns(clusterDoc.Status.RebootTurns, machines)163	if err != nil {164		fmt.Printf("reading the machines that hold reboot turns: %v\n", err)165		return err166	}167	heard, err := reads.heartbeats(now)168	if err != nil {169		fmt.Printf("listing heartbeats for the fleet sweep: %v\n", err)170		return err171	}172	s := decideFleetSweep(machines, heard, now)173174	// The rollout decision uses the same listing: which machines may175	// take their reboot turn now, and which spent grants return to176	// the budget (see rollout.go). This sequencing happens here for177	// the same reason the tally does: the sweep is the one place that178	// sees the whole fleet at once. appliedVersion is the release the179	// machine-operator DaemonSet's template actually carries180	// (daemonSetVersion, steward.go); decideRollout compares it181	// against the fleet's target so a leader can go first while the182	// applied template lags, because only a leader's boot can advance183	// it.184	appliedVersion, err := daemonSetVersion(reads, machineOperatorDaemonSet)185	r := decideRollout(machines, heard, reads.sightings.since(), clusterDoc, appliedVersion, now)186	if err != nil {187		fmt.Printf("reading the %s DaemonSet for the rollout: %v\n", machineOperatorDaemonSet, err)188		r = r.withoutGrants(err)189	}190191	// The OS's own pods, the operator's pods and the log relay pods,192	// are also fleet state. An upgraded machine keeps pods from193	// before its upgrade until the steward refreshes them (see194	// steward.go).195	stewardOSPods(reads, machines, steward, now)196197	// A retracted feature leaves its workloads behind. k3s only198	// deletes an addon when it sees the addon's manifest disappear199	// while k3s is running, but retraction removes the manifest while200	// k3s is down (see janitor.go). The flux feature's teardown is201	// always the janitor's, in a deliberate order (janitorFlux). That202	// teardown can also decline, when liken did not plant the Flux it203	// found, and it reports the refusal on the Cluster.204	janitorFeatureWorkloads(reads, clusterDoc)205	fluxTeardown := janitorFlux(c, clusterDoc)206207	// The flux feature's deploy key is fleet state too: minted once,208	// then read on every pass so the status always carries the209	// public half (see flux.go). The engine gets the same standing210	// care on a slower cadence: the probe asks once a minute, and gone211	// means re-planted on the pass that asked.212	publicKey := ensureFluxDeployKey(c, reads.recorder, clusterDoc)213	if seed, err := engineSeed(); err != nil {214		if clusterDoc.FeatureEnabled(cluster.FeatureFlux) {215			fmt.Printf("this build carries no engine seed (a plain go build outside make): %v\n", err)216		}217	} else {218		ensureFluxEngine(c, reads.recorder, clusterDoc, seed, probe, now)219	}220221	// The Cluster's status goes first, ahead of every Machine write222	// (see above).223	stored := publishClusterStatus(reads, clusterDoc, s, r, fluxTeardown, available, publicKey, now)224	carryOutRollout(reads, machines, r, stored, now)225	markLost(reads, machines, s.lost, heard, now)226227	// The fleet's metrics come from the verdict that the write above228	// carries, so the graph and the Cluster's status report one229	// observation (metrics.go).230	cm.observeSweep(s, r)231	return nil232}233234// markLost writes the Lost verdict onto each machine that the sweep235// found silent. Each verdict that lands posts MachineLost on the236// machine, with the time this program last saw its heartbeat lease237// renewed, because the Ready condition holds only that the heartbeat238// is stale, not since when.239func markLost(reads *fleetReader, machines []machine.Machine, lost []string, heard map[string]time.Time, now time.Time) {240	for _, m := range machines {241		if !slices.Contains(lost, m.Metadata.Name) {242			continue243		}244		// Everything else in the status stays as the machine last245		// wrote it. Those fields are still the machine's last known246		// facts, and only the machine can revise them. The phase and247		// the Ready condition change because they are claims about248		// the present, and the machine has stopped making those249		// claims.250		status := m.Status251		status.Phase = api.PhaseLost252		// The clone matters here. status.Conditions still shares its253		// backing array with the machines slice, and SetCondition254		// rewrites an existing entry in place.255		status.Conditions = api.SetCondition(slices.Clone(m.Status.Conditions), api.Condition{256			Type: "Ready", Status: api.ConditionUnknown, Reason: "HeartbeatStale",257			ObservedGeneration: m.Metadata.Generation,258			Message:            "the machine's operator has stopped renewing its heartbeat lease; the machine is presumed down",259		}, now)260		// A conflict here means the machine just came back and wrote261		// its own status first. That is the exact outcome this write262		// exists to allow, so the sweep skips this machine. The sweep263		// never retries a write onto another machine's status.264		if err := reads.publishStatus(&m, &status); errors.Is(err, apiclient.ErrConflict) {265			continue266		} else if err != nil {267			fmt.Printf("marking %s lost: %v\n", m.Metadata.Name, err)268		} else {269			fmt.Printf("machine %s has gone silent; marked Lost\n", m.Metadata.Name)270			reads.recorder.Warning(machineReference(&m), reasonMachineLost, lostMessage(heard, m.Metadata.Name))271		}272	}273}274275// lostMessage answers the message of a MachineLost Event: when this276// program last saw the machine renew its heartbeat lease, on this277// program's clock, or that it never did. The time is the one the278// verdict measured from, so it is HeartbeatStaleAfter or more before279// the Event.280func lostMessage(heard map[string]time.Time, name string) string {281	renewed, ok := heard[name]282	if !ok {283		return "the machine has never renewed a heartbeat lease; marked Lost"284	}285	return "the heartbeat lease was last renewed at " + renewed.UTC().Format(time.RFC3339) + "; marked Lost"286}287288// publishClusterStatus publishes the sweep's verdict on the Cluster.289// The MachinesReady condition carries the observation, stamped with290// the generation of the spec it judged. The Progressing condition291// reports the rollout, and the FluxTeardown condition appears only292// while the flux janitor has a refusal to report. The phase293// summarizes the first two conditions, and the tally is the headcount294// that the printer shows. This function also295// derives the release fields: the catalog's newest version, and the296// channel's last polled version. The sweep is the only writer of the297// Cluster's status, so deriving every field is its job. The same write298// carries the grant ledger, status.rebootTurns. The function299// writes the status only when something actually changed, so a300// settled fleet causes no write. Each condition that transitioned301// posts its Event after the write lands, so a refused write posts302// nothing and the next sweep finds the same transition again.303//304// It answers the ledger that the API server holds after the sweep: the305// one the Cluster already held when nothing changed, the one the306// write's answer holds, or none when the write failed. A 409 means307// another copy wrote the status since this sweep read it, so this308// sweep's grants must not go out.309func publishClusterStatus(reads *fleetReader, clusterDoc *cluster.Cluster, s fleetSweep, r rollout, fluxTeardown *api.Condition, available, publicKey string, now time.Time) []cluster.RebootTurn {310	newest := cluster.NewestVersion(clusterDoc.Spec.Releases.Catalog)311	s.condition.ObservedGeneration = clusterDoc.Metadata.Generation312	r.progressing.ObservedGeneration = clusterDoc.Metadata.Generation313	conditions := api.SetCondition(slices.Clone(clusterDoc.Status.Conditions), s.condition, now)314	conditions = api.SetCondition(conditions, r.progressing, now)315	// The flux janitor reports only a refusal, so this condition is316	// present exactly while there is a refusal to report. Removing it317	// when the janitor returns nothing keeps the list steady, which is318	// what the comparison below needs to leave a settled cluster319	// unwritten.320	if fluxTeardown != nil {321		fluxTeardown.ObservedGeneration = clusterDoc.Metadata.Generation322		conditions = api.SetCondition(conditions, *fluxTeardown, now)323	} else {324		conditions = api.RemoveCondition(conditions, fluxTeardownCondition)325	}326	// The deploy key's public half. A declared feature with an empty327	// answer means this pass could not read the Secret (a permission328	// still being seeded, a lost mint race); the last published329	// value is still the fleet's key, so the status keeps it. A330	// retracted feature clears the section: the Secret keeps the key331	// for a re-enable, but a status should not advertise a feature332	// the spec no longer declares.333	flux := clusterDoc.Status.Flux334	if !clusterDoc.FeatureEnabled(cluster.FeatureFlux) {335		flux = cluster.ClusterFluxStatus{}336	} else if publicKey != "" {337		flux = cluster.ClusterFluxStatus{PublicKey: publicKey}338	}339	if clusterDoc.Status.Machines != s.tally || clusterDoc.Status.Phase != s.phase ||340		clusterDoc.Status.Releases.Newest != newest ||341		clusterDoc.Status.Releases.Available != available ||342		clusterDoc.Status.Flux != flux ||343		clusterDoc.Status.ObservedGeneration != clusterDoc.Metadata.Generation ||344		!sameTurns(r.turns, clusterDoc.Status.RebootTurns) ||345		!slices.Equal(conditions, clusterDoc.Status.Conditions) {346		updated := *clusterDoc347		updated.Status.Machines = s.tally348		updated.Status.Phase = s.phase349		updated.Status.Releases.Newest = newest350		updated.Status.Releases.Available = available351		updated.Status.Flux = flux352		updated.Status.ObservedGeneration = clusterDoc.Metadata.Generation353		updated.Status.Conditions = conditions354		updated.Status.RebootTurns = r.turns355		stored, err := reads.publishClusterStatus(&updated)356		if err != nil {357			fmt.Printf("publishing cluster status: %v\n", err)358			return nil359		}360		postClusterTransitions(reads.recorder, clusterDoc, clusterDoc.Status.Conditions, conditions)361		if stored == nil {362			return nil363		}364		if !sameTurns(r.turns, stored.Status.RebootTurns) {365			return schemaPrunedTurns(reads, stored, now)366		}367		return stored.Status.RebootTurns368	}369	return clusterDoc.Status.RebootTurns370}371372// schemaPrunedTurns reports a Cluster write whose answer lacks the373// ledger that it sent. The served CRD then comes from a release that374// does not declare status.rebootTurns, and the API server pruned the375// field. That happens while a newer cluster-operator runs against an376// older CRD: in the seconds after the first leader boots a new release,377// or when a leader on an older release starts k3s again and applies378// its own copy of the CRD. The sweep grants nothing until the newer379// schema returns, and the Progressing message says why. The reason380// stays RollingOut, so the change posts no Event. It answers the381// ledger the API server stored.382func schemaPrunedTurns(reads *fleetReader, stored *cluster.Cluster, now time.Time) []cluster.RebootTurn {383	fmt.Println("the served Cluster schema does not store status.rebootTurns; granting no reboot turn")384	updated := *stored385	updated.Status.Conditions = api.SetCondition(slices.Clone(stored.Status.Conditions), api.Condition{386		Type: "Progressing", Status: api.ConditionTrue, Reason: "RollingOut",387		ObservedGeneration: stored.Metadata.Generation,388		Message: "reboot turns are waiting for a Cluster schema that stores status.rebootTurns; " +389			"the served CustomResourceDefinition comes from an older release",390	}, now)391	if _, err := reads.publishClusterStatus(&updated); err != nil {392		fmt.Printf("publishing cluster status: %v\n", err)393	}394	return stored.Status.RebootTurns395}396397// sameTurns reports whether two ledgers hold the same entries in the398// same order. It compares times with Equal, because a time decoded399// from the API server carries another location than one the sweep400// made.401func sameTurns(a, b []cluster.RebootTurn) bool {402	return slices.EqualFunc(a, b, func(x, y cluster.RebootTurn) bool {403		return x.Machine == y.Machine && x.MachineResourceVersion == y.MachineResourceVersion && x.Since.Equal(y.Since)404	})405}
cluster-operator/flux.go 84.1%
1package main23// The deploy key minter: the cluster operator's part of the flux4// feature.5//6// The flux feature syncs the fleet's declared state from a git7// repository, and that repository is expected to be private, so the8// sync engine needs a credential. liken mints that credential inside9// the cluster, instead of asking a person to generate a key and10// carry it in. The person's whole job is to copy the public half11// from the Cluster's status and register it at the forge as a deploy12// key. Private material never travels, never sits on a USB stick,13// and never depends on someone getting key formats right by hand.14//15// The key is one per cluster, not one per machine, because a16// narrower key would not narrow anything: the private half lives in17// a Secret, every Secret lives in the cluster datastore, and every18// leader carries that datastore. The datastore is the unit of19// exposure, so one key per datastore states the truth. Rotation is a20// person's decision, made by deleting the Secret; the next sweep21// mints a fresh pair and publishes the new half to register.22//23// This job belongs to the cluster operator because the credential is24// cluster-scoped: the sweep is the one writer of Cluster status, and25// the flux-system Secret is a cluster-level object that any leader's26// sweep may create. Init cannot do it, because init runs before k3s27// exists. The permission to touch the Secret arrives with the28// feature itself: the flux feature's manifests carry a Role in29// flux-system (flux/manifests/flux-system.yaml), seeded only while30// the feature is declared, so this operator holds no standing Secret31// access on fleets that never asked for GitOps.3233import (34	"crypto/ed25519"35	"embed"36	"encoding/json"37	"encoding/pem"38	"errors"39	"fmt"40	"net/http"41	"strings"42	"time"4344	"github.com/liken-sh/liken/kubernetes/apiclient"45	"github.com/liken-sh/liken/kubernetes/events"46	"github.com/liken-sh/liken/liken/cluster"47	"github.com/liken-sh/liken/liken/kubernetes"48	"golang.org/x/crypto/ssh"49	"sigs.k8s.io/yaml"50)5152// The Secret's home, by Flux's conventions: the flux-system53// namespace, a Secret named flux-system, and the identity /54// identity.pub keys that Flux's git client reads. Following the55// conventions means the sync engine finds the credential with no56// configuration pointing at it.57const (58	fluxSecretPath  = "/api/v1/namespaces/flux-system/secrets/flux-system"59	fluxSecretsPath = "/api/v1/namespaces/flux-system/secrets"60)6162// ensureFluxDeployKey makes sure the fleet's deploy key exists when63// the flux feature is declared, and returns the public half for the64// status to carry. It returns "" when the feature is off, and also65// on any API failure, because the sweep runs again and a missing66// status one pass long is cheaper than a wrong one. The first passes67// after the feature turns on can fail here legitimately: the Role68// that grants this operator its Secret access is itself seeded by69// the feature, and k3s may not have applied it yet. The sweep's70// level-triggered loop absorbs that window; nothing here needs to71// wait or retry.72//73// A mint posts FluxDeployKeyMinted on the Cluster, because the person74// must register the new public half at the forge before the first75// sync.76func ensureFluxDeployKey(c *apiclient.Client, recorder *events.Recorder, clusterDoc *cluster.Cluster) string {77	if !clusterDoc.FeatureEnabled(cluster.FeatureFlux) {78		return ""79	}80	// The declared configuration supplies known_hosts. A81	// malformed declaration does not block the mint: the key is82	// per-cluster and parameter-independent, and init's feature pass83	// already reports the configuration problem.84	cfg, _ := clusterDoc.FluxConfig()8586	secret := &kubernetes.Secret{}87	err := c.RequestJSON(http.MethodGet, fluxSecretPath, nil, secret)88	if err == nil {89		// The key exists and never changes here, but the Secret's90		// known_hosts entry follows the declaration: a forge that91		// rotates its host key must not strand the sync on a stale92		// pin.93		if cfg != nil && string(secret.Data["known_hosts"]) != cfg.KnownHosts {94			patch, _ := json.Marshal(map[string]any{95				"stringData": map[string]string{"known_hosts": cfg.KnownHosts},96			})97			if err := kubernetes.PatchJSON(c, fluxSecretPath, patch); err != nil {98				fmt.Printf("refreshing the flux known_hosts: %v\n", err)99			}100		}101		return string(secret.Data["identity.pub"])102	}103	if !errors.Is(err, apiclient.ErrNotFound) {104		fmt.Printf("reading the flux deploy key: %v\n", err)105		return ""106	}107108	pub, priv, err := mintDeployKey(clusterDoc.Metadata.Name)109	if err != nil {110		fmt.Printf("minting the flux deploy key: %v\n", err)111		return ""112	}113	knownHosts := ""114	if cfg != nil {115		knownHosts = cfg.KnownHosts116	}117	body, err := json.Marshal(map[string]any{118		"apiVersion": "v1",119		"kind":       "Secret",120		"metadata": map[string]string{121			"name":      "flux-system",122			"namespace": "flux-system",123		},124		"type": "Opaque",125		// stringData is the write-side convenience: the API server126		// encodes it into data, so this code never handles base64.127		"stringData": map[string]string{128			"identity":     priv,129			"identity.pub": pub,130			"known_hosts":  knownHosts,131		},132	})133	if err != nil {134		fmt.Printf("encoding the flux deploy key secret: %v\n", err)135		return ""136	}137	if err := c.RequestJSON(http.MethodPost, fluxSecretsPath, body, nil); err != nil {138		// A conflict means another leader's sweep minted first. That139		// copy's key is the fleet's key; the next pass reads it.140		if !errors.Is(err, apiclient.ErrConflict) {141			fmt.Printf("creating the flux deploy key secret: %v\n", err)142		}143		return ""144	}145	fmt.Printf("minted the flux deploy key; register the public half from the Cluster's status at the forge\n")146	recorder.Normal(clusterReference(clusterDoc), reasonFluxDeployKeyMinted,147		"minted the flux deploy key; register the public half from status.flux.publicKey at the forge")148	return pub149}150151// The engine seed: the pinned gotk-components.yaml that the flux152// domain renders at build time and the Makefile copies here for153// the embed below, which cannot reach outside the package. The154// directory always holds a README, so a plain `go build` works155// before the flux domain has fetched anything; a binary built that156// way carries no seed, and the planter reports it instead of157// planting nothing silently.158//159//go:embed seed160var seedFS embed.FS161162func engineSeed() ([]byte, error) {163	return seedFS.ReadFile("seed/gotk-components.yaml")164}165166// engineProbePath names the one object whose absence means the167// engine is gone: the kustomize-controller Deployment. That168// controller is the applier that heals everything else from git, so169// while it exists, git owns the engine, whatever its state; when it170// does not, nothing can heal, and liken re-plants the seed.171const engineProbePath = "/apis/apps/v1/namespaces/flux-system/deployments/kustomize-controller"172173// seedObject is one document of the seed: its addressing fields,174// and its whole body as JSON, ready to send.175type seedObject struct {176	APIVersion string177	Kind       string178	Name       string179	Namespace  string180	body       []byte181}182183// parseSeed splits the seed into its documents. The seed is184// flux-rendered YAML with clean document separators, so the split is185// by separator lines; each document converts to JSON once, here,186// and never again on the apply path.187func parseSeed(raw []byte) ([]seedObject, error) {188	var objects []seedObject189	for _, doc := range strings.Split("\n"+string(raw), "\n---\n") {190		if strings.TrimSpace(stripYAMLComments(doc)) == "" {191			continue192		}193		body, err := yaml.YAMLToJSON([]byte(doc))194		if err != nil {195			return nil, fmt.Errorf("parsing a seed document: %w", err)196		}197		var meta struct {198			APIVersion string `json:"apiVersion"`199			Kind       string `json:"kind"`200			Metadata   struct {201				Name      string `json:"name"`202				Namespace string `json:"namespace"`203			} `json:"metadata"`204		}205		if err := json.Unmarshal(body, &meta); err != nil {206			return nil, err207		}208		objects = append(objects, seedObject{209			APIVersion: meta.APIVersion,210			Kind:       meta.Kind,211			Name:       meta.Metadata.Name,212			Namespace:  meta.Metadata.Namespace,213			body:       body,214		})215	}216	return objects, nil217}218219// stripYAMLComments drops comment and blank lines, so a document220// that is only commentary (the seed's generated-file header) reads221// as empty.222func stripYAMLComments(doc string) string {223	var kept []string224	for _, line := range strings.Split(doc, "\n") {225		trimmed := strings.TrimSpace(line)226		if trimmed == "" || strings.HasPrefix(trimmed, "#") {227			continue228		}229		kept = append(kept, line)230	}231	return strings.Join(kept, "\n")232}233234// seedKinds maps each kind the seed may contain to its REST235// resource. liken's API client has no discovery machinery, and it236// does not need any: the seed's vocabulary is small and known, and237// an unknown kind is a loud error naming what to add, which happens238// exactly when a new Flux version grows a new kind.239var seedKinds = map[string]struct {240	resource   string241	namespaced bool242}{243	"Namespace":                {"namespaces", false},244	"ResourceQuota":            {"resourcequotas", true},245	"NetworkPolicy":            {"networkpolicies", true},246	"CustomResourceDefinition": {"customresourcedefinitions", false},247	"ServiceAccount":           {"serviceaccounts", true},248	"ClusterRole":              {"clusterroles", false},249	"ClusterRoleBinding":       {"clusterrolebindings", false},250	"Role":                     {"roles", true},251	"RoleBinding":              {"rolebindings", true},252	"Service":                  {"services", true},253	"Deployment":               {"deployments", true},254}255256// collectionPath builds the create URL for one seed object: the core257// group lives under /api/v1, every other group under /apis/<group>/258// <version>, with the namespace segment for namespaced kinds.259func collectionPath(o seedObject) (string, error) {260	kind, known := seedKinds[o.Kind]261	if !known {262		return "", fmt.Errorf("the seed contains a %s, a kind the planter's table does not map; add it to seedKinds", o.Kind)263	}264	base := "/apis/" + o.APIVersion265	if !strings.Contains(o.APIVersion, "/") {266		base = "/api/" + o.APIVersion267	}268	if kind.namespaced {269		return base + "/namespaces/" + o.Namespace + "/" + kind.resource, nil270	}271	return base + "/" + kind.resource, nil272}273274// engineProbeInterval sets how long the sweep waits between engine275// probes. The probe is a backstop for one failure: someone deletes276// the kustomize-controller `Deployment` while the flux feature is277// on. This program watches only Machines, so no event announces that278// deletion. The sweep runs at least every ten seconds and again after279// each Machine change, and a probe on every sweep would make this GET280// the most frequent request `liken` sends, for an answer that almost281// never changes. One GET a minute costs less than a second watch and282// its restart handling. An engine that heals within a minute is283// still far faster than the only other repair, which is this284// program's next start.285const engineProbeInterval = 60 * time.Second286287// engineProbe holds the one fact the engine's care must remember288// across passes: when the sweep last asked whether the engine is289// there. The channel poller (channel.go) keeps its state the same290// way, and for the same reason. The sweep stays level-triggered and291// keeps nothing itself, so what must outlive a pass lives in a value292// the caller threads through. This one needs no mutex, because the293// probe runs inline on the sweep's goroutine and one pass runs at a294// time, while the channel poller's fetch runs on its own goroutine.295type engineProbe struct {296	asked time.Time297}298299// TryAsk reports whether this sweep may probe, and records the ask300// when it may. The zero value probes on the first sweep. A feature301// that is off does not use up an interval, because the caller returns302// before it calls this. Planting the seed counts as an ask, because303// the probe records the ask before it reads the answer. That keeps304// the state at one timestamp, and the interval then holds with no305// exception. A plant that only partly landed waits for the next306// probe, inside the same minute this interval promises.307func (p *engineProbe) TryAsk(now time.Time) bool {308	if now.Sub(p.asked) < engineProbeInterval {309		return false310	}311	p.asked = now312	return true313}314315// ensureFluxEngine plants the engine seed when the engine is gone.316// The probe asks once an interval, so a deleted engine heals within317// about a minute, not at the next boot. Each object is a plain318// create, and a conflict means the object already exists, which the319// planter leaves exactly as it found it: the seed only ever fills320// absence. Present but broken stays the repository's problem on321// purpose; liken answers only for gone. A planting posts322// FluxEngineSeeded on the Cluster, with how many objects it created.323func ensureFluxEngine(c *apiclient.Client, recorder *events.Recorder, clusterDoc *cluster.Cluster, seed []byte, probe *engineProbe, now time.Time) {324	if !clusterDoc.FeatureEnabled(cluster.FeatureFlux) {325		return326	}327	if !probe.TryAsk(now) {328		return329	}330	err := c.RequestJSON(http.MethodGet, engineProbePath, nil, nil)331	if err == nil {332		return333	}334	if !errors.Is(err, apiclient.ErrNotFound) {335		fmt.Printf("probing for the flux engine: %v\n", err)336		return337	}338	objects, err := parseSeed(seed)339	if err != nil {340		fmt.Printf("reading the engine seed: %v\n", err)341		return342	}343	fmt.Printf("the flux engine is absent; planting the seed (%d objects)\n", len(objects))344	created, present, failed := 0, 0, 0345	for _, o := range objects {346		path, err := collectionPath(o)347		if err != nil {348			fmt.Printf("planting the engine seed: %v\n", err)349			failed++350			continue351		}352		err = c.RequestJSON(http.MethodPost, path, o.body, nil)353		switch {354		case errors.Is(err, apiclient.ErrConflict):355			present++ // it already exists; whatever is there stays356		case err != nil:357			fmt.Printf("planting %s %s: %v\n", o.Kind, o.Name, err)358			failed++359		default:360			created++361		}362	}363	recorder.Normal(clusterReference(clusterDoc), reasonFluxEngineSeeded, fmt.Sprintf(364		"the flux engine was absent; planted its seed: %d objects created, %d present already, %d failed",365		created, present, failed))366}367368// mintDeployKey generates the pair: an ed25519 key, the modern SSH369// default, small enough to read aloud and accepted by every current370// forge. The public half is one authorized_keys line, commented with371// the cluster's name so a forge's key list stays legible. The372// private half is OpenSSH PEM, the format Flux's git client parses.373func mintDeployKey(clusterName string) (publicKey, privateKey string, err error) {374	pub, priv, err := ed25519.GenerateKey(nil)375	if err != nil {376		return "", "", err377	}378	comment := "liken:" + clusterName379	pemBlock, err := ssh.MarshalPrivateKey(priv, comment)380	if err != nil {381		return "", "", err382	}383	sshPub, err := ssh.NewPublicKey(pub)384	if err != nil {385		return "", "", err386	}387	// MarshalAuthorizedKey ends the line with a newline and no388	// comment; the status reads better with the comment and without389	// the newline.390	line := string(ssh.MarshalAuthorizedKey(sshPub))391	line = line[:len(line)-1] + " " + comment392	return line, string(pem.EncodeToMemory(pemBlock)), nil393}
cluster-operator/heartbeats.go 89.7%
1package main23// This file records when this program saw each machine's heartbeat4// change, so a sweep judges the heartbeat's age on one clock.5//6// A machine's operator writes the renewTime of its heartbeat Lease from7// the machine's own clock. A sweep that subtracted that time from this8// program's clock would add the difference between the two clocks to9// every heartbeat's age. A machine whose clock runs 45 seconds behind10// would read Lost while it renews every 8 seconds, and a machine whose11// clock runs ahead would read Lost late after it stopped. Clocks12// differ most at boot, before the first time synchronization, and on a13// machine with a wrong hardware clock and no time server.14//15// So this program reads renewTime only to learn that it changed. The16// Leases' watch calls the handler below for each write, and the handler17// records the moment on this program's clock. The node lifecycle18// controller in Kubernetes judges a kubelet's lease the same way19// (pkg/controller/nodelifecycle/node_lifecycle_controller.go).20//21// The record is a cache of what the watch delivered, so a program that22// starts records each Lease as seen at that moment. This program opens23// its watches only after it takes the lead, and exits when it loses24// the lead, so a new leader is always a new process with an empty25// record. A machine that is already gone then reads Lost26// HeartbeatStaleAfter after that, and until then it reads its last27// written status, which is usually Ready. So the record also keeps the28// time of the first sweep's read of the Leases, and the rollout grants29// no reboot turn until the record is HeartbeatStaleAfter older than30// that read (decideRollout). Without the hold, a new leader that31// started a minute after the old leader's machine lost power could32// count that dead machine as up and grant a second leader its turn.3334import (35	"sync"36	"time"3738	"github.com/liken-sh/liken/kubernetes/informer"39	"github.com/liken-sh/liken/liken/kubernetes"40	"k8s.io/client-go/tools/cache"41)4243// heartbeatSightings maps each Lease's name to the last renewTime this44// program saw, and when it first saw it. The watch's handler writes it45// from the informer's goroutine and a sweep reads it from the loop's46// goroutine, so a lock guards it. Its zero value is an empty record.47type heartbeatSightings struct {48	mu   sync.Mutex49	seen map[string]sighting5051	// began is when the first sweep read the Leases, on this program's52	// clock. That read holds every Lease, so each Lease that existed53	// then has a sighting no later than began, and each Lease that54	// stayed unchanged since reads stale after began plus55	// HeartbeatStaleAfter. A Lease that appears later was written by a56	// machine that was up to write it.57	began time.Time58}5960type sighting struct {61	// renewed is the Lease's renewTime, from the machine's clock. It is62	// compared only for equality.63	renewed time.Time64	// at is when this program first saw that renewTime, from its own65	// clock.66	at time.Time67}6869// observe records that the Lease held renewed at now, and answers when70// this program first saw that renewTime. A renewTime that differs from71// the recorded one, earlier or later, is a renewal: a machine's clock72// can step back.73func (s *heartbeatSightings) observe(name string, renewed, now time.Time) time.Time {74	s.mu.Lock()75	defer s.mu.Unlock()76	if last, ok := s.seen[name]; ok && last.renewed.Equal(renewed) {77		return last.at78	}79	if s.seen == nil {80		s.seen = map[string]sighting{}81	}82	s.seen[name] = sighting{renewed: renewed, at: now}83	return now84}8586func (s *heartbeatSightings) forget(name string) {87	s.mu.Lock()88	defer s.mu.Unlock()89	delete(s.seen, name)90}9192// heard observes each renewal of one read of the Leases, and maps each93// Lease's name to when this program saw its renewTime change. A sweep94// reads the Leases from the watch's copy or from the API server, and95// both reads pass through here. The handler has already recorded each96// change the watch delivered, at the moment it arrived, so this records97// only a change that a direct read finds first.98func (s *heartbeatSightings) heard(renewals map[string]time.Time, now time.Time) map[string]time.Time {99	s.mu.Lock()100	if s.began.IsZero() {101		s.began = now102	}103	s.mu.Unlock()104	heard := make(map[string]time.Time, len(renewals))105	for name, renewed := range renewals {106		heard[name] = s.observe(name, renewed, now)107	}108	return heard109}110111// since answers when the first sweep read the Leases, or the zero time112// before any sweep has read them.113func (s *heartbeatSightings) since() time.Time {114	s.mu.Lock()115	defer s.mu.Unlock()116	return s.began117}118119// handler records each Lease the watch delivers. A Lease whose120// renewTime does not parse carries no liveness claim121// (kubernetes.Renewals), so it records nothing. A deleted Lease122// leaves the record, so a Lease created again later counts as seen123// when it arrives.124func (s *heartbeatSightings) handler(source informer.Source) cache.ResourceEventHandler {125	seen := func(object any) {126		lease, err := informer.Convert[kubernetes.Lease](object)127		if err != nil {128			informer.Report(source.String(), err)129			return130		}131		now := time.Now()132		for name, renewed := range kubernetes.Renewals([]kubernetes.Lease{lease}) {133			s.observe(name, renewed, now)134		}135	}136	return cache.ResourceEventHandlerFuncs{137		AddFunc:    func(object any) { seen(object) },138		UpdateFunc: func(_, object any) { seen(object) },139		DeleteFunc: func(object any) {140			// A tombstone can hold no copy of the Lease, but its key141			// always names it.142			key, err := cache.DeletionHandlingMetaNamespaceKeyFunc(object)143			if err != nil {144				informer.Report(source.String(), err)145				return146			}147			if _, name, err := cache.SplitMetaNamespaceKey(key); err == nil {148				s.forget(name)149			}150		},151	}152}
cluster-operator/janitor.go 94.8%
1package main23// The feature janitor deletes the workloads that belong to retracted4// features.5//6// A feature's workload manifests ship in the image. init seeds them7// into k3s's auto-deploy directory only while the cluster document8// declares the feature (see init/features.go). Retraction removes the9// file, and that removal deletes nothing on its own.10//11// State that plainly, because the opposite is easy to assume: k3s12// never deletes an addon's objects because its manifest file went13// away. The deploy controller walks the files that exist, and applies14// them.15// Nothing reconciles an Addon against a source file that is no longer16// there, so a deleted manifest leaves both the Addon and everything17// it created in place. k3s does delete an addon's objects when the18// component is on its disable list, because then it still has the19// file to read and can act on what the file declares. A liken feature20// is not a k3s component and is never on that list.21//22// So the janitor is not a fallback for one boot path. It is the only23// thing that removes a retracted feature's workloads, on every path.24// Every feature-seeded workload carries a liken.sh/feature annotation25// that names the feature it belongs to. Each sweep deletes any26// liken-system workload whose annotation names a feature that the27// cluster document no longer declares. Whether the feature is28// declared is the only question the janitor asks. It does not wait29// for the retraction to roll through the fleet: once the document no30// longer claims a workload's feature, the fleet does not need that31// workload running anywhere. The janitor does not act on objects that32// carry no annotation. The operator and log-relay DaemonSets live in33// the same namespace, and no feature owns them.3435import (36	"errors"37	"fmt"38	"net/http"39	"net/url"4041	"github.com/liken-sh/liken/kubernetes/apiclient"42	"github.com/liken-sh/liken/liken/api"43	"github.com/liken-sh/liken/liken/cluster"44	"github.com/liken-sh/liken/liken/kubernetes"45)4647// featureAnnotation names the feature a workload belongs to. It is48// the janitor's whole contract with the manifests: a feature's49// manifests carry this annotation (for example,50// open-iscsi/manifests/iscsid.yaml), and everything else omits it.51const featureAnnotation = "liken.sh/feature"5253// featureWorkloadKinds lists the workload kinds the janitor sweeps.54// Each kind is a list endpoint in liken-system. Features seed55// DaemonSets today. When a feature ships its first Deployment or56// Service, add that kind here, in the same change.57var featureWorkloadKinds = []struct {58	kind     string59	listPath string60}{61	{"daemonset", daemonSetsPath},62}6364// featureWorkload holds the small amount of information the janitor65// needs about an object: its name, and which feature, if any, claims66// it.67type featureWorkload struct {68	Metadata struct {69		Name        string            `json:"name"`70		Annotations map[string]string `json:"annotations"`71	} `json:"metadata"`72}7374// decideRetractions computes which workloads belong to features that75// the document no longer declares. Presence in spec.features is the76// same opt-in test that init uses, so the two gates always agree.77func decideRetractions(features map[string]*cluster.FeatureConfig, workloads []featureWorkload) []featureWorkload {78	var retracted []featureWorkload79	for _, w := range workloads {80		slug := w.Metadata.Annotations[featureAnnotation]81		if slug == "" {82			continue83		}84		if _, declared := features[slug]; !declared {85			retracted = append(retracted, w)86		}87	}88	return retracted89}9091// janitorFeatureWorkloads carries out the janitor's work, once per92// sweep: it lists each swept kind, selects which workloads to93// retract, and deletes them. Deletion is by name with background94// propagation, so the workload's pods are deleted along with it.95func janitorFeatureWorkloads(r *fleetReader, clusterDoc *cluster.Cluster) {96	for _, k := range featureWorkloadKinds {97		workloads, err := r.workloads(k.listPath)98		if err != nil {99			fmt.Printf("listing %ss for the feature janitor: %v\n", k.kind, err)100			continue101		}102		for _, w := range decideRetractions(clusterDoc.Spec.Features, workloads) {103			name := w.Metadata.Name104			slug := w.Metadata.Annotations[featureAnnotation]105			path := k.listPath + "/" + name + "?propagationPolicy=Background"106			if err := r.client.RequestJSON(http.MethodDelete, path, nil, nil); err != nil {107				fmt.Printf("deleting %s %s for the retracted %s feature: %v\n", k.kind, name, slug, err)108			} else {109				fmt.Printf("the cluster no longer declares the %s feature; deleted %s %s\n", slug, k.kind, name)110			}111		}112	}113}114115// The flux janitor: the teardown half of the seed-once engine.116//117// The generic janitor above deletes only a workload that carries the118// feature annotation, and the flux janitor makes that same ownership119// test first, because here the answer is not always yes. An adopted120// cluster may already run a Flux that somebody else installed, long121// before liken arrived. The document an operator writes during122// adoption describes machines, but spec.features is an opt-in list,123// so the same document is also a claim over everything else in the124// cluster, and a slug left out of it is a retraction. So the janitor125// reads the ownership mark first, and tears down only the126// installation liken planted.127//128// The rest of the teardown cannot use the generic janitor's shape,129// because these objects are not inert. Deleting the Kustomization130// while its controller still runs triggers the engine's own deletion131// finalizer, and with prune on, that finalizer garbage-collects132// everything the repository ever applied: the workloads, the Machine133// documents, and the Cluster document itself. So flux's retraction is134// ordered, and the order is the whole design: kill the controllers135// first, so the finalizer can never fire, then remove the finalizers136// by hand, and only then delete the objects.137//138// Each sweep advances one stage and returns, so the stages are139// separated by real observations, never by in-process waits:140//141//  0. The flux-system Namespace carries no liken.sh/feature mark:142//     delete nothing, and report the refusal on the Cluster.143//  1. Engine Deployments still exist: delete them.144//  2. Controller pods still exist: wait. A Deployment's deletion is145//     asynchronous, and a controller that is still terminating could146//     still process a finalizer.147//  3. Controllers provably gone: strip the sync objects' finalizers,148//     delete them, and delete the engine's cluster-scoped remains:149//     the CRDs, the engine's RBAC, the planter's grant, and the150//     namespace.151//152// What survives is deliberate: nothing. Retraction burns the deploy153// key with the namespace, and a re-enabled feature mints a fresh154// key to register. Keeping the key was considered and rejected: it155// would mean k3s's addon machinery must never touch the namespace,156// and the ground would outlive the feature as unowned state. Off157// means off. The repository's own workloads also survive, as158// orphans: stopping the sync must not undeploy what the sync159// deployed.160//161// The janitor's rights are standing, in the operator's own manifest,162// unlike the planter's, which arrive with the feature. This is163// deliberate too: the janitor's job begins exactly when the164// feature's delivered grants disappear, so delivered rights could165// never clean up after the feature that delivered them. The166// standing rights are deletes on exact names, powerless to create167// or read anything.168169// The engine's namespace, by name and by path. The Namespace is also170// the ownership record: liken's own feature manifest declares it with171// the feature annotation (flux/manifests/flux-system.yaml), and this172// janitor reads that annotation before it deletes anything. k3s173// applies that manifest as an addon the moment the document declares174// the feature, so the mark lands with the namespace itself, ahead of175// anything the cluster operator does.176const (177	fluxNamespace     = "flux-system"178	fluxNamespacePath = "/api/v1/namespaces/" + fluxNamespace179)180181// fluxTeardownCondition is the type of the Cluster condition that182// reports a declined teardown. It is present only while there is a183// refusal to report, so the sweep removes it whenever the janitor184// returns nothing (see publishClusterStatus in fleet.go).185const fluxTeardownCondition = "FluxTeardown"186187// fluxTeardownPaths are the delete targets of the final stage, in188// order. The sync objects come first, finalizers already stripped;189// the namespace's own deletion then sweeps everything namespaced190// that remains, the deploy key Secret included; the cluster-scoped191// remains close it out. The CRD and RBAC names must match the192// engine seed; the parity test in flux_test.go holds them to it.193var fluxTeardownPaths = []string{194	"/apis/kustomize.toolkit.fluxcd.io/v1/namespaces/flux-system/kustomizations/flux-system",195	"/apis/source.toolkit.fluxcd.io/v1/namespaces/flux-system/gitrepositories/flux-system",196	fluxNamespacePath,197	"/apis/apiextensions.k8s.io/v1/customresourcedefinitions/buckets.source.toolkit.fluxcd.io",198	"/apis/apiextensions.k8s.io/v1/customresourcedefinitions/externalartifacts.source.toolkit.fluxcd.io",199	"/apis/apiextensions.k8s.io/v1/customresourcedefinitions/gitrepositories.source.toolkit.fluxcd.io",200	"/apis/apiextensions.k8s.io/v1/customresourcedefinitions/helmcharts.source.toolkit.fluxcd.io",201	"/apis/apiextensions.k8s.io/v1/customresourcedefinitions/helmrepositories.source.toolkit.fluxcd.io",202	"/apis/apiextensions.k8s.io/v1/customresourcedefinitions/ocirepositories.source.toolkit.fluxcd.io",203	"/apis/apiextensions.k8s.io/v1/customresourcedefinitions/kustomizations.kustomize.toolkit.fluxcd.io",204	"/apis/rbac.authorization.k8s.io/v1/clusterroles/crd-controller-flux-system",205	"/apis/rbac.authorization.k8s.io/v1/clusterroles/flux-edit-flux-system",206	"/apis/rbac.authorization.k8s.io/v1/clusterroles/flux-view-flux-system",207	"/apis/rbac.authorization.k8s.io/v1/clusterrolebindings/cluster-reconciler-flux-system",208	"/apis/rbac.authorization.k8s.io/v1/clusterrolebindings/crd-controller-flux-system",209	"/apis/rbac.authorization.k8s.io/v1/clusterroles/liken-engine-planter",210	"/apis/rbac.authorization.k8s.io/v1/clusterrolebindings/liken-engine-planter",211}212213// fluxControllerPodsPath lists the engine's controller pods, alive214// or terminating. The engine's own labels select them.215var fluxControllerPodsPath = "/api/v1/namespaces/flux-system/pods?labelSelector=" +216	url.QueryEscape("app in (source-controller,kustomize-controller)")217218// declinedFluxTeardown is the report the janitor returns when it219// refuses to tear down a Flux that liken did not plant. The message220// carries the remedy, because an operator who really does want liken221// to own this installation needs one command and should not have to222// find it.223func declinedFluxTeardown() *api.Condition {224	mark := featureAnnotation + "=" + cluster.FeatureFlux225	return &api.Condition{226		Type:   fluxTeardownCondition,227		Status: api.ConditionFalse,228		Reason: "NotPlantedByLiken",229		Message: "the cluster no longer declares flux, and the " + fluxNamespace +230			" Namespace carries no " + mark + " annotation, so liken did not plant this" +231			" installation and deleted nothing. To give liken this installation, run:" +232			" kubectl annotate namespace " + fluxNamespace + " " + mark,233	}234}235236// janitorFlux tears the flux feature down when the cluster document237// no longer declares it, and only when liken planted it. Every call238// is one stage at most; the sweep calls it again within ten seconds,239// and silence is the converged state. The returned condition reports240// a refusal, and nil means there is nothing to report.241func janitorFlux(c *apiclient.Client, clusterDoc *cluster.Cluster) *api.Condition {242	if clusterDoc.FeatureEnabled(cluster.FeatureFlux) {243		return nil244	}245246	// Stage 0: ownership, before a single delete. An installation247	// liken did not plant is not one liken removes, so the Namespace's248	// mark gates everything below. An absent Namespace means there is249	// no installation to tear down, and a read that fails outright250	// says nothing about ownership, so both answers stop the pass251	// without a verdict.252	//253	// This gate has one real cost, and it falls on a cluster that254	// liken founded before the mark existed: that cluster's255	// flux-system Namespace carries no annotation, so retracting the256	// feature now deletes nothing and raises the condition instead.257	// The operator annotates the Namespace and the teardown proceeds258	// on the next sweep. Failing closed is the correct direction here,259	// because the failure in the other direction deletes somebody260	// else's GitOps installation, and the condition names the one261	// command that clears this one.262	namespace := &featureWorkload{}263	err := c.RequestJSON(http.MethodGet, fluxNamespacePath, nil, namespace)264	if errors.Is(err, apiclient.ErrNotFound) {265		return nil266	}267	if err != nil {268		fmt.Printf("reading the %s namespace for the flux teardown: %v\n", fluxNamespace, err)269		return nil270	}271	if namespace.Metadata.Annotations[featureAnnotation] != cluster.FeatureFlux {272		return declinedFluxTeardown()273	}274275	// Stage 1: the controllers must die before anything else is276	// touched. A successful delete means this pass's work is done;277	// the pod check below needs a fresh observation anyway. A delete278	// that failed also ends the pass, because the Deployment still279	// exists: its pods can be gone for a moment, and a controller280	// that comes back would run the prune finalizers that stage 3281	// strips.282	stop := false283	for _, name := range []string{"source-controller", "kustomize-controller"} {284		path := "/apis/apps/v1/namespaces/flux-system/deployments/" + name285		if err := c.RequestJSON(http.MethodGet, path, nil, nil); errors.Is(err, apiclient.ErrNotFound) {286			continue287		}288		err := c.RequestJSON(http.MethodDelete, path+"?propagationPolicy=Background", nil, nil)289		switch {290		case err == nil:291			fmt.Printf("liken planted this flux and the cluster no longer declares it; deleted the %s Deployment\n", name)292			stop = true293		case !errors.Is(err, apiclient.ErrNotFound):294			fmt.Printf("flux teardown, deleting the %s Deployment: %v\n", name, err)295			stop = true296		}297	}298	if stop {299		return nil300	}301302	// Stage 2: wait out terminating controller pods. This gate is303	// what makes the finalizer unreachable: no controller process304	// exists past it.305	pods, err := kubernetes.List[featureWorkload](c, fluxControllerPodsPath)306	if err != nil || len(pods) > 0 {307		return nil308	}309310	// Stage 3: nothing can react anymore. Strip the sync objects'311	// finalizers so their deletion, and the namespace's, completes312	// instead of waiting forever for a controller that no longer313	// exists. Then delete what remains, and let 404s stay silent:314	// this function runs on every sweep, and the converged state is315	// nothing but 404s.316	for _, path := range fluxTeardownPaths[:2] {317		_ = kubernetes.PatchJSON(c, path, []byte(`{"metadata": {"finalizers": null}}`))318	}319	// Each line states why the delete was safe to make, because a320	// reader who meets these lines in a log is asking exactly that:321	// liken planted this installation, so it is liken's to remove, and322	// no controller survives to run a prune finalizer over it.323	for _, path := range fluxTeardownPaths {324		err := c.RequestJSON(http.MethodDelete, path, nil, nil)325		if err == nil {326			fmt.Printf("liken planted this flux and the cluster no longer declares it; the controllers are gone, so no prune can fire; deleted %s\n", path)327		} else if !errors.Is(err, apiclient.ErrNotFound) && !errors.Is(err, apiclient.ErrConflict) {328			fmt.Printf("flux teardown, deleting %s: %v\n", path, err)329		}330	}331	return nil332}
cluster-operator/leader.go 12.5%
1package main23// Leader election: only one copy of this program acts on the fleet.4//5// The sweep writes Lost verdicts, the Cluster's status, and reboot6// grants, and it evicts and deletes. Two copies that sweep at the same7// time can each grant a reboot turn from a different view of the8// fleet, and together exceed the disruption budget. A rolling update, a9// node partition, and a replica count above one each run a second10// copy, so this program elects one acting copy through11// kubernetes/election, on a coordination.k8s.io Lease named12// liken-cluster-operator in liken-system. That package explains the13// election, its timings, and its limits.14//15// The election is not a fence. A former leader that paused can still16// land a grant it sent before the pause, after another copy took the17// Lease. The grant ledger in the Cluster's status, not the election,18// keeps the disruption budget against that grant (rollout.go).19//20// Only the copy that holds the Lease opens its watches and sweeps. A21// copy that the API server refuses on the Lease waits, as for any other22// failure of the election. The OS images carry this binary, so a23// rollback runs it under the RBAC that the target release's manifests24// grant. The Cluster CRD refuses a target older than 2026.09.28-001, the25// first release whose RBAC grants this program the create and update of26// its Lease, so every release that a rollback can reach lets a copy27// take the Lease.28//29// The election links client-go's typed clientset. That costs this30// binary about 13 MB, and it is accepted here because the cluster31// operator runs as one pod in the fleet. The machine operator runs on32// every machine and never imports the package.3334import (35	"context"36	"errors"37	"fmt"38	"os"3940	"k8s.io/client-go/rest"4142	"github.com/liken-sh/liken/kubernetes/election"43)4445// leaseName names the Lease the copies compete for, and leaseNamespace46// is the namespace the whole OS uses. The sweep also reads this Lease47// from its watch, to judge how fresh its copies are (watches.go).48const (49	leaseName      = "liken-cluster-operator"50	leaseNamespace = "liken-system"51)5253// electionOptions names this program's Lease and pod for the election.54// report receives each line with this program's prefix.55func electionOptions(pod string, report func(string)) election.Options {56	return election.Options{57		Name:      leaseName,58		Namespace: leaseNamespace,59		Pod:       pod,60		Report:    func(line string) { report(component + ": " + line) },61	}62}6364// lead blocks until this process may act. It exits the process when65// stop ends first, and whenever the Lease is lost after that. The pod's66// hostname is its name, which makes the identity readable in `kubectl67// get lease`.68func lead(stop context.Context) *election.Lead {69	pod, err := os.Hostname()70	if err != nil {71		fatal("reading the pod's name for the leader election: %v", err)72	}73	config, err := rest.InClusterConfig()74	if err != nil {75		fatal("in-cluster config for the leader election: %v", err)76	}77	options := electionOptions(pod, func(line string) { fmt.Println(line) })78	l, err := election.Start(stop, config, options)79	if errors.Is(err, election.ErrStopped) {80		os.Exit(0)81	}82	if err != nil {83		fatal("leader election: %v", err)84	}85	return l86}
cluster-operator/main.go 67.5%
1// liken-cluster-operator is the program that watches the fleet.2//3// Each machine's operator reports on itself. This leaves verdicts4// that no single machine can write: a dead machine cannot report5// that it is dead, no machine can total a headcount that includes6// itself without conflicting with the other machines' operators, and7// someone who can see every machine's request must hand out reboot8// turns. This program writes exactly those verdicts: the Lost phase9// on silent machines, the Cluster's status, and the rollout's10// grants. It also refreshes the OS's own DaemonSet pods after11// upgrades (see steward.go).12//13// The machine operator is privileged and node-local: a DaemonSet14// with hostPath mounts, running on the machine it manages. This15// program is different. It is an ordinary workload: a single-replica16// Deployment with no mounts, no host network, and no privilege. The17// Kubernetes API is its only input and its only output. This is what18// lets its RBAC role match exactly what a fleet observer needs: read19// machines and heartbeats, write statuses, evict stale OS pods.20//21// Only one copy of this program acts at a time. The Deployment rolls a22// new pod in beside the old one, and each copy competes for a leader23// election Lease; only the copy that holds it watches the fleet and24// writes. leader.go names the Lease and the reason for the election.25package main2627import (28	"context"29	"errors"30	"flag"31	"fmt"32	"os"33	"os/signal"34	"syscall"35	"time"3637	"k8s.io/client-go/dynamic"3839	"github.com/liken-sh/liken/kubernetes/apiclient"40	"github.com/liken-sh/liken/kubernetes/election"41	"github.com/liken-sh/liken/kubernetes/events"42	"github.com/liken-sh/liken/kubernetes/informer"43	"github.com/liken-sh/liken/liken/cluster"44	"github.com/liken-sh/liken/liken/kubernetes"45	"github.com/liken-sh/liken/liken/kubernetes/watch"46	"github.com/liken-sh/liken/liken/machine"47	"github.com/liken-sh/liken/liken/metrics"48)4950// component is this program's name in every metric that carries one.51const component = "liken-cluster-operator"5253// metricsAddress is where this operator answers a Prometheus scrape.54// The default is the port that55// `plans/completed/65-prometheus-metrics.md` at the top of the56// repository gives every process on the cluster network, so the binary57// carries the contract and the pod template only has to name the port58// it exposes. An empty value turns the listener off, for an owner who59// runs no Prometheus.60var metricsAddress = flag.String("metrics-address", ":9200",61	"the address to serve /metrics on; empty serves no metrics")6263func main() {64	flag.Parse()65	fmt.Println(component, machine.Version)6667	// A SIGTERM from the kubelet ends the loop after the sweep in68	// flight, and the process then releases the Lease, so the copy that69	// waits beside it takes over at once.70	stop, _ := signal.NotifyContext(context.Background(), syscall.SIGTERM, os.Interrupt)7172	// The metrics registry outlives every pass, because a counter's73	// whole value is that it accumulates (metrics.go). A copy that74	// waits for the Lease serves its runtime metrics too.75	operatorMetrics, clusterLayer := serveMetrics(*metricsAddress)7677	// A failure during setup ends the process deliberately. This is78	// the same crash-only method the machine operator uses: kubelet79	// restarts the pod with backoff, and the failure shows in80	// `kubectl get pods`.81	client, err := kubernetes.InClusterClient("")82	if err != nil {83		fatal("in-cluster config: %v", err)84	}85	watcher, err := informer.InCluster()86	if err != nil {87		fatal("in-cluster config for the watches: %v", err)88	}8990	// Nothing below runs until this copy holds the Lease.91	leader := lead(stop)9293	// The ticker is a clock. A heartbeat ages past the staleness limit94	// with no event, because an aging Lease is not a write. A granted95	// reboot turn passes rolloutStallAfter the same way, and the96	// channel poller and the engine probe keep their own intervals.97	// Ten seconds keeps those verdicts within one sweep of the moment98	// they come due. The objects a sweep judges come from99	// the watches' copies, so a tick costs the API server one read: the100	// flux deploy key Secret while the feature is declared, or the101	// flux-system Namespace while it is not (watches.go says why those102	// two stay direct).103	ticker := time.NewTicker(10 * time.Second)104	run(stop, context.Background(), leader, client, watcher, ticker.C, operatorMetrics, clusterLayer)105}106107// run is the work of a copy that may act: it opens the watches, sweeps108// until stop ends, and then steps down. The watches live until watches109// ends, which in main is the end of the process, so the sweep in110// flight at a shutdown still reads from them.111func run(stop, watches context.Context, leader *election.Lead, client *apiclient.Client,112	watcher dynamic.Interface, ticks <-chan time.Time, om *metrics.Operator, cm *clusterMetrics) {113	// Every write asks the election first (election.Lead.MayWrite).114	client = client.WithWriteGuard(leader.MayWrite)115116	// The watches start with the lead, so a copy that has never acted117	// holds no copy of the fleet and opens no stream (watches.go names each watch and118	// what wakes the loop). The wake channel has one slot, so a burst119	// of Machine changes makes one wake, and one sweep over the newest120	// state answers the whole burst.121	wakes := make(chan struct{}, 1)122	fleet := watchFleet(watches, watcher, client, watch.Signal(wakes), om.WatchRestarted)123	// The recorder posts the Events about the Machines and the Cluster124	// (events.go). It writes through the guarded client, so a replica125	// that lost the election posts nothing, and from its own goroutine,126	// so a sweep never waits on an Event. It outlives stop, so the127	// Events of the last sweep are written.128	fleet.recorder = events.New(context.Background(), client, component, events.Options{})129130	operate(stop, fleet, wakes, ticks, om, cm)131	// The sweep has returned, so no write of the sweep is in flight.132	// The recorder writes the Events the sweep posted, and Shutdown133	// returns after its last request has returned. So the Lease is134	// released after the last write, and a waiting copy takes it on its135	// next try. A recorder whose last write did not return in time136	// leaves the Lease to expire, after that write can land.137	flush, cancel := context.WithTimeout(context.Background(), events.ShutdownTimeout)138	defer cancel()139	if err := fleet.recorder.Shutdown(flush); err != nil {140		fmt.Printf("%v; leaving the Lease to expire\n", err)141		return142	}143	leader.StepDown()144}145146// operate finds the Cluster and sweeps the fleet until stop ends. A147// wake or a tick starts the next sweep, and stop ends the loop only148// between two sweeps, so the sweep in flight finishes its writes.149func operate(stop context.Context, fleet *fleetReader,150	wakes <-chan struct{}, ticks <-chan time.Time, om *metrics.Operator, cm *clusterMetrics) {151	// This program takes no configuration at all. It finds the152	// Cluster it operates in the Clusters' copy, because a fleet has153	// exactly one Cluster, and the machine operators seed it from the154	// image. Until the CRD is served and some machine's seed lands, in155	// the first minutes of a brand-new cluster, there is nothing to156	// operate, so this program waits.157	clusterDoc := awaitCluster(fleet, wakes, stop)158	if clusterDoc == nil {159		return160	}161	name := clusterDoc.Metadata.Name162	fmt.Printf("operating cluster %s\n", name)163164	// Three values outlive a pass, the same way the machine operator's165	// release fetcher does. The sweep stays level-triggered and166	// stateless, and each of these remembers only what it needs to167	// avoid asking or saying the same thing too often. The channel168	// poller keeps when it last asked and the channel's last answer.169	// The flux engine probe keeps when it last asked whether the engine170	// is still there. The pod steward keeps the terminating pods it has171	// already reported.172	poller := newChannelPoller()173	probe := &engineProbe{}174	steward := &podSteward{}175176	for stop.Err() == nil {177		started := time.Now()178		err := sweep(fleet, name, poller, probe, steward, cm)179		om.ObserveReconcile(clusterKind, time.Since(started), err)180		select {181		case <-wakes:182		case <-ticks:183		case <-stop.Done():184		}185	}186}187188// sweep runs one pass of the cluster operator's whole job, always189// starting from the cluster's current state. It reads the Cluster190// fresh, because its spec drives the rollout. It gives the channel191// poller its look at the spec. Then it lets the fleet sweep list the192// fleet, judge it, and write the result, carrying the engine probe193// along so the flux engine's care keeps its own cadence.194func sweep(reads *fleetReader, name string, poller *channelPoller, probe *engineProbe, steward *podSteward, cm *clusterMetrics) error {195	clusterDoc, err := reads.cluster(name)196	if err != nil {197		fmt.Printf("reading cluster %s: %v\n", name, err)198		return err199	}200	poller.Observe(clusterDoc.Spec.Releases,201		clusterDoc.Metadata.Annotations[cluster.CheckReleasesAnnotation], time.Now())202	return sweepFleet(reads, clusterDoc, poller.Available(), probe, steward, cm, time.Now())203}204205// awaitCluster waits until a Cluster exists, and answers it. A 404206// response only means the CRD is not served yet. An empty list means207// no machine has seeded the object yet. Both conditions resolve208// themselves as the fleet boots. The Clusters' copy wakes the loop209// when the first Cluster arrives. While the copy cannot answer, each210// look reads the API server, and the ticker's pace bounds those reads.211// It answers nil when stop ends first.212func awaitCluster(reads *fleetReader, wakes <-chan struct{}, stop context.Context) *cluster.Cluster {213	retry := time.NewTicker(10 * time.Second)214	defer retry.Stop()215	for {216		clusters, err := reads.clusters()217		if err != nil && !errors.Is(err, apiclient.ErrNotFound) {218			fmt.Printf("listing clusters: %v\n", err)219		}220		if len(clusters) > 0 {221			return &clusters[0]222		}223		select {224		case <-wakes:225		case <-retry.C:226		case <-stop.Done():227			return nil228		}229	}230}231232func fatal(format string, args ...any) {233	fmt.Fprintf(os.Stderr, format+"\n", args...)234	os.Exit(1)235}
cluster-operator/metrics.go 95.8%
1package main23// Layer 3 for the cluster operator: the metrics about the fleet.4//5// Each machine's operator publishes its own facts, and this program6// publishes the three that only a program which sees every machine7// at once can state. Every one of them comes from the sweep's own8// verdict, the value that this same pass writes onto the Cluster's9// status, so a graph and a `kubectl get cluster` can never disagree10// (fleet.go and rollout.go).1112import (13	"fmt"14	"os"1516	"github.com/liken-sh/liken/liken/api"17	"github.com/liken-sh/liken/liken/machine"18	"github.com/liken-sh/liken/liken/metrics"19	"github.com/prometheus/client_golang/prometheus"20)2122// clusterKind is the resource this operator's loop reconciles, and23// the label its layer 2 duration and error series carry.24const clusterKind = "Cluster"2526// machineKind is what this operator watches. The watch spans the27// whole fleet, because any machine's transition can change the28// Cluster's phase, so the watch restart counter carries this kind29// and the reconcile counters carry Cluster.30const machineKind = "Machine"3132// serveMetrics builds this operator's whole registry, the contract's33// layer 1 and layer 2 beside the fleet's layer 3, and starts the34// listener that answers a scrape. A listener that cannot bind35// reports itself and nothing more: the fleet must keep converging36// whether or not anybody watches it.37func serveMetrics(address string) (*metrics.Operator, *clusterMetrics) {38	o := metrics.NewOperator(component, machine.Version,39		[]string{clusterKind}, watchKinds)40	layer := newClusterMetrics(o)41	if addr, err := o.Serve(address); err != nil {42		fmt.Fprintf(os.Stderr, "the metrics listener is not serving: %v\n", err)43	} else if addr != nil {44		fmt.Printf("serving metrics on %s/metrics\n", addr)45	}46	return o, layer47}4849// clusterMetrics holds this operator's layer 3.50type clusterMetrics struct {51	machines  *prometheus.GaugeVec52	approvals prometheus.Gauge53	behind    prometheus.Gauge54}5556// newClusterMetrics registers layer 3 on the operator's registry.57// The machine gauge starts with a zero for every phase liken defines.58// A fleet reports nothing at all in a phase that no machine holds, so59// without these zeros the first machine to go Lost would draw its60// first point with no earlier point to fall from.61func newClusterMetrics(o *metrics.Operator) *clusterMetrics {62	m := &clusterMetrics{63		machines: prometheus.NewGaugeVec(prometheus.GaugeOpts{64			Name: metrics.Prefix + "machines",65			Help: "Machines in the fleet, by the phase the sweep judged them to hold.",66		}, []string{"phase"}),67		approvals: prometheus.NewGauge(prometheus.GaugeOpts{68			Name: metrics.Prefix + "disruption_approvals_pending",69			Help: "Machines that hold a staged change and wait for a person to approve it.",70		}),71		behind: prometheus.NewGauge(prometheus.GaugeOpts{72			Name: metrics.Prefix + "machines_behind_target",73			Help: "Machines that do not run the release the Cluster's spec targets.",74		}),75	}76	o.Registry().MustRegister(m.machines, m.approvals, m.behind)77	for _, phase := range fleetPhases {78		m.machines.WithLabelValues(string(phase)).Set(0)79	}80	return m81}8283// fleetPhases is the phase vocabulary a machine can hold. The api84// package defines it and the CRD's enum enforces it, so this label85// is bounded by a list that only a schema change can grow.86var fleetPhases = []api.Phase{87	api.PhaseReady,88	api.PhaseBooting,89	api.PhaseDownloading,90	api.PhaseUpdating,91	api.PhaseUpdatePending,92	api.PhaseBlocked,93	api.PhaseDegraded,94	api.PhaseLost,95	api.PhaseUnknown,96}9798// observeSweep publishes the verdict that this pass is about to99// write onto the Cluster. The phases are the effective phases, the100// ones the sweep judged from each machine's status and its101// heartbeat, so a machine that has gone silent counts as Lost here102// at the same moment it counts as Lost in the headcount.103func (m *clusterMetrics) observeSweep(s fleetSweep, r rollout) {104	counts := map[api.Phase]int{}105	for phase, count := range s.phases {106		// A Machine that exists and has never published a status107		// holds no phase at all. Unknown is what that machine is:108		// the fleet has an object for it and no observation of it.109		if phase == "" {110			phase = api.PhaseUnknown111		}112		counts[phase] += count113	}114	for _, phase := range fleetPhases {115		m.machines.WithLabelValues(string(phase)).Set(float64(counts[phase]))116	}117	m.approvals.Set(float64(s.approvals))118	m.behind.Set(float64(r.behind))119}
cluster-operator/phase.go 100.0%
1package main23// This file computes a machine's phase as the fleet should read it.4//5// Each machine derives its own phase from its own conditions (see6// machine-operator/phase.go). But a written status is only as7// current as the machine that wrote it, and a silent machine may no8// longer exist. This file makes the fleet-side correction: trust the9// machine's claim while its heartbeat is fresh, and read silence as10// Lost. The one phase a machine can never derive for itself is11// exactly the phase this program exists to write.1213import (14	"time"1516	"github.com/liken-sh/liken/liken/api"17	"github.com/liken-sh/liken/liken/kubernetes"18	"github.com/liken-sh/liken/liken/machine"19)2021// effectivePhase returns a machine's phase, corrected for liveness.22// It returns the machine's own claim when its heartbeat is fresh,23// and Lost when the machine has gone silent. heard maps each heartbeat24// Lease to when this program last saw its renewTime change, on this25// program's clock (heartbeats.go), so a machine whose clock differs26// from this program's reads neither Lost early nor Lost late. A machine27// with no lease at all has never been heard from. This function judges28// every machine the same way, including whichever machine hosts this29// program's pod: this program is not itself a machine, so no machine30// is exempt. Both the fleet sweep and the rollout use this function31// to judge machines.32//33// Silence does not always mean trouble. A machine holding a reboot34// grant (see rollout.go) was told to go down. So until the grant is35// old enough to count as a stall, the sweep treats the machine's36// silence as a reboot in progress.37func effectivePhase(m *machine.Machine, heard map[string]time.Time, now time.Time) api.Phase {38	changed, ok := heard[m.Metadata.Name]39	if ok && now.Sub(changed) <= kubernetes.HeartbeatStaleAfter {40		return m.Status.Phase41	}42	if grant := api.FindCondition(m.Status.Conditions, machine.RebootApprovedCondition); grant != nil &&43		now.Sub(grant.LastTransitionTime) <= rolloutStallAfter {44		return api.PhaseUpdating45	}46	return api.PhaseLost47}
cluster-operator/rollout.go 99.3%
1package main23// The rollout conductor: how the cluster sequences its fleet's4// reboots.5//6// A staged change that needs a reboot creates drift on every7// affected machine at once. If each machine rebooted the moment it8// was ready, they would all reboot at the same time, taking the9// whole fleet down at once, which risks quorum. Kubernetes workloads10// get protection from this problem through PodDisruptionBudgets and11// kubectl drain. Machines need the same protection, so the Cluster12// carries a machine-level maxUnavailable value (spec.disruption),13// and the conductor hands out reboot turns one budget slot at a14// time.15//16// Turns normally go to workers before leaders, because a worker17// mistake costs little while every leader carries a share of quorum.18// One case reorders that default: a version rollout, while the19// machine-operator DaemonSet's applied template lags the fleet's20// target release. Only a leader's boot rewrites the AddOn manifests21// that produce that template (cluster-operator/steward.go), so a22// follower that reboots into the new release first only extends the23// lag, running its new binary inside the old template until the pod24// steward can refresh it (the guard in machine-operator/conditions.go25// covers that follower in the meantime). Sending a leader first ends26// the lag instead, so while it lasts, the gate holds workers back and27// grants a waiting leader its turn.28//29// The coordination happens through conditions. A machine that needs30// a reboot sets reason AwaitingTurn on its convergence condition and31// waits. This program responds by writing a RebootApproved condition32// onto that Machine. That condition is a grant: it is present while33// the turn is extended, it is removed when the turn is spent, and it34// is never set to False. Writing one condition type that you own35// onto an object that another controller manages is a standard36// arrangement: the scheduler writes PodScheduled onto Pods that the37// kubelet owns. The machine's own operator carries the grant along38// untouched, because its status writes preserve condition types it39// does not set, and it acts on the grant: it cordons, drains, and40// reboots the machine (see drain.go).41//42// The budget counts all unavailability, planned or not. A machine43// that is Lost or Degraded occupies a slot just like a machine that44// is rebooting on request. So a fleet that already has machines down45// pauses its own rollout instead of making things worse. The leaders46// have a stricter floor that no budget can override: only one leader47// may be down or granted a turn at a time. The datastore keeps48// quorum only while a majority of leaders is up, and letting a49// second leader go down could break that majority.50//51// The cluster records each turn before it grants it. The Cluster's52// status holds the grant ledger, status.rebootTurns: each machine that53// holds a turn or is about to receive one, with the resourceVersion of54// the Machine that the grant write names. A sweep adds an entry in its55// Cluster status write, which names the Cluster's resourceVersion, and56// writes the grant only after the API server stored the entry. Two57// copies of this program that read the same Cluster cannot both add58// an entry, because the second write gets 409 Conflict, so the budget59// is one compare-and-swap on one object. The PodDisruptionBudget does60// the same for evictions: the API server records each evicted pod in61// status.disruptedPods before it deletes the pod.62//63// The election alone cannot keep the budget. A leader that pauses64// after its write guard allowed a grant, on a stalled node or under65// SIGSTOP, sends the grant when it resumes, and the API server can66// commit a write after the client gave up on it. That grant names the67// Machine's version from before the pause, and nothing else wrote the68// Machine since, so the API server accepts it. Without the ledger, a69// new leader could grant another machine in the meantime, and the two70// grants together would exceed the budget.71//72// An entry counts against the budget until no grant that it covers can73// land. That is true when the Machine is gone, or holds no grant at a74// version other than the recorded one, because a resourceVersion never75// returns to an earlier value. The sweep reads such a Machine from the76// API server, not from its watch's copy, because an older copy also77// differs from the entry (fleetReader.readTurns). An entry whose78// Machine holds no grant at the recorded version is confirmed: this79// program writes the grant itself, from that copy. A late grant from80// the copy that recorded the entry names the same version, so the API81// server accepts one of the two writes. A grant that no entry names,82// from a release before the ledger or from a ledger that an older83// schema pruned, enters the ledger at the Machine's current version.84// These rules read only the API's state, so a new leader applies them85// to the entries of the leader before it the same way.86//87// The budget counts a machine as down only when its liveness verdict88// says so. A new process of this program records each heartbeat Lease89// as seen when it starts (heartbeats.go), so for its first90// HeartbeatStaleAfter it reads a machine that is already down by that91// machine's last written status, which is usually Ready. So the conductor grants no turn and92// reclaims no grant until its record of the heartbeats is that old,93// and a machine that was already down reads Lost before the first94// grant.9596import (97	"fmt"98	"slices"99	"strings"100	"time"101102	"github.com/liken-sh/liken/liken/api"103	"github.com/liken-sh/liken/liken/cluster"104	"github.com/liken-sh/liken/liken/kubernetes"105	"github.com/liken-sh/liken/liken/machine"106)107108// rolloutStallAfter sets how long a granted machine may stay109// unavailable before the rollout is marked stalled. This time is110// long compared to a normal reboot, which takes a couple of minutes.111// It is short compared to how long a person takes to notice a112// machine that never came back. While the rollout is stalled, it113// grants no new turns. A halted rollout that a person can see is114// better than an automated rollout that keeps granting turns across115// a fleet whose machines are not coming back.116const rolloutStallAfter = 10 * time.Minute117118// A rollout holds one sweep's sequencing verdict: which machines to119// grant a reboot turn, which spent grants to take back, and the120// Progressing condition that reports the rollout on the Cluster. The121// condition deliberately reuses Deployment vocabulary: Progressing,122// with False meaning the rollout has stopped making progress.123type rollout struct {124	grant       []string125	revoke      []string126	progressing api.Condition127128	// turns is the grant ledger that the sweep writes into the129	// Cluster's status before any grant, in name order. confirm names130	// each machine whose entry an earlier sweep recorded, whose grant131	// is absent, and whose Machine is still at the recorded version.132	turns   []cluster.RebootTurn133	confirm []string134135	// behind counts the machines that do not run the fleet's target136	// release yet. It belongs to the rollout, because the rollout is137	// what closes the gap, and this loop already reads every138	// machine's version. A cluster with no target counts nobody139	// behind: there is nothing to be behind (metrics.go).140	behind int141}142143// wantsTurn reports whether any of the machine's conditions carry144// the AwaitingTurn reason. This reason means the machine has a145// staged change and may take its disruption as soon as the cluster146// grants a turn. That readiness comes from either of two paths: the147// machine's rebootPolicy is Auto, or a person approved the change on148// a Manual machine through the liken.sh/approve-disruption149// annotation.150func wantsTurn(m *machine.Machine) bool {151	for _, c := range m.Status.Conditions {152		if c.Reason == "AwaitingTurn" {153			return true154		}155	}156	return false157}158159// available reports whether a machine is serving the cluster right160// now. Blocked and UpdatePending machines are up. Their trouble is161// administrative, not operational. A Downloading machine is up too:162// the release downloads in the background, and the machine goes163// down only on the turn that this conductor grants. Every other164// machine that is not Ready is either absent or unwell.165func available(phase api.Phase) bool {166	switch phase {167	case api.PhaseReady, api.PhaseUpdatePending, api.PhaseBlocked, api.PhaseDownloading:168		return true169	}170	return false171}172173// decideRollout computes the whole rollout decision, over the same174// inputs that the fleet sweep reads. It grants turns to workers175// first, then to leaders, each group in name order. Workers go first176// because a mistake with a worker costs little, while every leader177// carries a share of quorum. Name order makes the decision178// deterministic, so two sweeps of the same fleet agree.179//180// One exception reorders that default. appliedVersion is the release181// the machine-operator DaemonSet's template actually carries182// (daemonSetVersion, steward.go); an empty value, meaning no183// DaemonSet has applied anything yet, never triggers the exception.184// While appliedVersion lags clusterDoc.Spec.Version, and a leader185// either waits for a turn or holds an unspent grant, every waiting186// worker sits out the sweep: granting one would send it into the187// very template lag the leader's boot is about to end. The hold188// lasts through the leader's whole turn, because each sweep189// recomputes it from the grant that is still outstanding. When no190// leader waits and none holds a grant, the workers proceed in the191// default order, because holding them for a leader that cannot take192// or use a turn would stall the rollout; the guard in193// machine-operator/conditions.go is what makes a worker's lag194// survivable in that case.195//196// heardSince is when this program first read the heartbeat Leases197// (heartbeatSightings.since). Until HeartbeatStaleAfter has passed198// since then, a machine that is already down still reads its last199// status, so the decision grants no turn and reclaims no grant, and200// the Progressing message says why. A grant and a reclaim both trust201// that a machine reads available only when it is up. The zero time202// means the record is old enough.203func decideRollout(machines []machine.Machine, heard map[string]time.Time, heardSince time.Time,204	clusterDoc *cluster.Cluster, appliedVersion string, now time.Time) rollout {205	var r rollout206	judgedFrom := heardSince.Add(kubernetes.HeartbeatStaleAfter)207	listening := !heardSince.IsZero() && !now.After(judgedFrom)208	inFlight := 0 // budget slots occupied: unavailable machines and unspent grants209	leaderBusy := false210	leaderAdvancing := false // a leader holds an unspent grant; its boot advances the template211	var stalled, inProgress, waiting, workers, leaders []string212	ledger := map[string]cluster.RebootTurn{}213	for _, turn := range clusterDoc.Status.RebootTurns {214		ledger[turn.Machine] = turn215	}216	versions := map[string]string{}217	since := now.UTC().Truncate(time.Second)218219	for i := range machines {220		m := &machines[i]221		name := m.Metadata.Name222		versions[name] = m.Metadata.ResourceVersion223		leader := slices.Contains(clusterDoc.Spec.Leaders, name)224		phase := effectivePhase(m, heard, now)225		grant := api.FindCondition(m.Status.Conditions, machine.RebootApprovedCondition)226		if clusterDoc.Spec.Version != "" && m.Status.Version.Liken != clusterDoc.Spec.Version {227			r.behind++228		}229230		// The ledger. The sweep reads a machine whose entry shows no231		// grant from the API server (fleetReader.readTurns), so its232		// version here is current. At another version, no grant that233		// the entry covers can land, so the entry leaves and its slot234		// is free. A grant that no entry names, from an older binary235		// or from a ledger that an older schema pruned, enters the236		// ledger at the Machine's current version.237		turn, recorded := ledger[name]238		if recorded && grant == nil && m.Metadata.ResourceVersion != turn.MachineResourceVersion {239			recorded = false240		}241		if !recorded && grant != nil {242			turn, recorded = cluster.RebootTurn{Machine: name, MachineResourceVersion: m.Metadata.ResourceVersion, Since: since}, true243		}244		if recorded {245			r.turns = append(r.turns, turn)246		}247		// A pending entry records a turn whose grant this sweep does248		// not see. A grant write from the copy that recorded it can249		// still be in flight.250		pending := recorded && grant == nil251252		switch {253		case grant != nil && available(phase) && !wantsTurn(m):254			// The turn is spent, because the machine converged, or it255			// was never used, because the edit was reverted or the256			// policy changed to Manual. Either way, the machine is257			// back and no longer asking, so the grant returns to the258			// budget. Its entry keeps the slot until a later sweep259			// reads the Machine without the grant.260			r.revoke = append(r.revoke, name)261			inFlight++262			leaderBusy = leaderBusy || leader263		case grant != nil && !available(phase) && now.Sub(grant.LastTransitionTime) > rolloutStallAfter:264			// This machine was granted a turn, went down, and has265			// stayed down too long. That is no longer a reboot in266			// progress. It is an outage.267			stalled = append(stalled, name)268			inFlight++269		case grant != nil || !available(phase) || pending:270			// A machine that is mid-turn, recorded for a turn, or271			// unwell occupies a budget slot either way. The budget272			// counts unavailability itself, not whether the273			// unavailability was planned.274			inFlight++275			if grant != nil || pending {276				inProgress = append(inProgress, name)277			}278			// The confirm writes the grant from the copy at the279			// recorded version. A late grant from the copy that280			// recorded the entry names the same version, so the API281			// server accepts one of the two. A machine that is down282			// gets no confirm, because its grant would hide the Lost283			// verdict, and that verdict's own write moves the version.284			if pending && available(phase) && !listening {285				r.confirm = append(r.confirm, name)286			}287			if leader {288				leaderBusy = true289				// A leader that holds an unspent grant, or is recorded290				// for one, is mid-turn: its boot is the event that291				// advances the applied template, and the gate below292				// keys on that. A leader that is merely unavailable,293				// with no grant, advances nothing.294				leaderAdvancing = leaderAdvancing || grant != nil || pending295			}296		case wantsTurn(m):297			if leader {298				leaders = append(leaders, name)299			} else {300				workers = append(workers, name)301			}302		}303	}304305	slices.Sort(workers)306	slices.Sort(leaders)307	capacity := clusterDoc.Spec.Disruption.MaxUnavailableOrDefault() - inFlight308309	// The gate: while the applied template lags the fleet's target,310	// advancing it is worth more than a worker's turn. The hold must311	// outlive the sweep that grants the leader, because every sweep312	// recomputes this decision from scratch, and the grant's own313	// status write triggers the next sweep within milliseconds. So314	// the hold keys on durable state: a leader that can be granted315	// now, or a leader whose granted turn is still running. When316	// neither exists, because every leader is Manual or a leader is317	// stuck without a grant, the workers proceed: holding them for a318	// leader that cannot take or use a turn would stall the rollout,319	// and the machine operator's guard makes a worker's lag320	// survivable.321	gate := appliedVersion != "" && clusterDoc.Spec.Version != "" && appliedVersion != clusterDoc.Spec.Version322	holdWorkers := gate && ((len(leaders) > 0 && !leaderBusy) || leaderAdvancing)323324	if !holdWorkers {325		for _, name := range workers {326			if len(stalled) > 0 || listening || capacity <= 0 {327				waiting = append(waiting, name)328				continue329			}330			r.grant = append(r.grant, name)331			r.turns = append(r.turns, cluster.RebootTurn{Machine: name, MachineResourceVersion: versions[name], Since: since})332			inProgress = append(inProgress, name)333			capacity--334		}335	}336	for _, name := range leaders {337		if len(stalled) > 0 || listening || capacity <= 0 || leaderBusy {338			waiting = append(waiting, name)339			continue340		}341		r.grant = append(r.grant, name)342		r.turns = append(r.turns, cluster.RebootTurn{Machine: name, MachineResourceVersion: versions[name], Since: since})343		inProgress = append(inProgress, name)344		capacity--345		leaderBusy = true346	}347	slices.SortFunc(r.turns, func(a, b cluster.RebootTurn) int { return strings.Compare(a.Machine, b.Machine) })348	if holdWorkers {349		// Every waiting worker sits out the sweep: the gate holds the350		// whole group back rather than admit some of them ahead of the351		// leader turn that ends the lag.352		waiting = append(waiting, workers...)353	}354	if listening {355		// A reclaim waits too: the machine reads available from a356		// status that may be older than the machine's last power-off.357		r.revoke = nil358	}359360	switch {361	case len(stalled) > 0:362		r.progressing = api.Condition{363			Type: "Progressing", Status: api.ConditionFalse, Reason: "RolloutStalled",364			Message: fmt.Sprintf("granted a reboot turn more than %s ago and not back: %s; no further turns until it returns",365				rolloutStallAfter, strings.Join(stalled, ", ")),366		}367	case len(inProgress)+len(waiting) > 0:368		message := "taking a reboot turn: " + strings.Join(inProgress, ", ")369		if len(inProgress) == 0 {370			message = "reboot turns are waiting on the disruption budget"371			if listening {372				message = "reboot turns are waiting on the heartbeats"373			}374		}375		if len(waiting) > 0 {376			message += "; waiting: " + strings.Join(waiting, ", ")377		}378		if holdWorkers && len(workers) > 0 {379			message += fmt.Sprintf("; the applied system-pod template (%s) lags the fleet's target (%s); "+380				"a leader goes first to advance the template, and workers wait", appliedVersion, clusterDoc.Spec.Version)381		}382		if listening {383			message += fmt.Sprintf("; cluster-operator started reading the heartbeat leases at %s, "+384				"and grants no turn until %s, when a machine that is down reads Lost",385				heardSince.Format(time.RFC3339), judgedFrom.Format(time.RFC3339))386		}387		r.progressing = api.Condition{388			Type: "Progressing", Status: api.ConditionTrue, Reason: "RollingOut", Message: message,389		}390	default:391		r.progressing = api.Condition{392			Type: "Progressing", Status: api.ConditionTrue, Reason: "RolloutComplete",393			Message: "no machines are waiting for a reboot turn",394		}395	}396	return r397}398399// withoutGrants holds every new turn of a sweep that could not read400// the machine-operator DaemonSet, because the sweep cannot tell whether401// the applied template lags the fleet's target. The held turns leave402// the ledger too, so they take no slot. Reclaiming a spent grant does403// not depend on the template, so the reclaims stand. A confirm stands404// too: its entry records a decision that an earlier sweep made with405// that read.406func (r rollout) withoutGrants(err error) rollout {407	if len(r.grant) == 0 {408		return r409	}410	r.progressing = api.Condition{411		Type: "Progressing", Status: api.ConditionTrue, Reason: "RollingOut",412		Message: fmt.Sprintf("reboot turns are waiting on a read of the applied system-pod template: %v; waiting: %s",413			err, strings.Join(r.grant, ", ")),414	}415	r.turns = slices.DeleteFunc(slices.Clone(r.turns), func(turn cluster.RebootTurn) bool {416		return slices.Contains(r.grant, turn.Machine)417	})418	r.grant = nil419	return r420}421422// carryOutRollout writes the verdict onto the fleet: it adds grants423// to the named Machines, confirms the pending entries, and removes424// spent grants. These are writes to other machines' statuses, and they425// are safe for the same reason the Lost write is: each write touches426// only the one condition type that this writer owns. When a 409427// happens because of a crossing write, this function simply waits for428// the next sweep. Each grant, confirm, and reclaim that lands posts429// one Event on the machine.430//431// stored is the ledger that the API server holds after this sweep's432// Cluster write. A grant or a confirm is written only when stored433// holds its machine at the version of the copy the write sends. A434// Cluster write that got a 409, or whose answer an older schema435// pruned, stores no entry, so the sweep grants nothing. A reclaim436// adds no disruption, so it needs no entry.437func carryOutRollout(reads *fleetReader, machines []machine.Machine, r rollout, stored []cluster.RebootTurn, now time.Time) {438	for i := range machines {439		m := &machines[i]440		name := m.Metadata.Name441		status := m.Status442		recorded := slices.ContainsFunc(stored, func(turn cluster.RebootTurn) bool {443			return turn.Machine == name && turn.MachineResourceVersion == m.Metadata.ResourceVersion444		})445		switch {446		case recorded && (slices.Contains(r.grant, name) || slices.Contains(r.confirm, name)):447			status.Conditions = api.SetCondition(slices.Clone(m.Status.Conditions), api.Condition{448				Type: machine.RebootApprovedCondition, Status: api.ConditionTrue, Reason: "DisruptionBudgetAllows",449				ObservedGeneration: m.Metadata.Generation,450				Message:            "the cluster's disruption budget allows this machine to take its reboot turn now",451			}, now)452			message := "the cluster's disruption budget allows this machine to take its reboot turn now"453			if slices.Contains(r.confirm, name) {454				message = "the cluster recorded this machine's reboot turn in an earlier sweep, and grants it now"455			}456			if err := reads.publishStatus(m, &status); err != nil {457				fmt.Printf("granting %s its reboot turn: %v\n", name, err)458			} else {459				fmt.Printf("granted %s its reboot turn\n", name)460				reads.recorder.Normal(machineReference(m), reasonRebootTurnGranted, message)461			}462		case slices.Contains(r.revoke, name):463			status.Conditions = api.RemoveCondition(slices.Clone(m.Status.Conditions), machine.RebootApprovedCondition)464			if err := reads.publishStatus(m, &status); err != nil {465				fmt.Printf("reclaiming %s's reboot turn: %v\n", name, err)466			} else {467				reads.recorder.Normal(machineReference(m), reasonRebootTurnReclaimed,468					"the machine is available and asks for no reboot, so its turn returns to the disruption budget")469			}470		}471	}472}
cluster-operator/steward.go 93.0%
1package main23// The pod steward keeps the OS's own pods up to date with the4// operating systems that run under them.5//6// Two container images are part of the OS: the operator's image and7// the log relays' image. image/build.sh bakes both into the8// initramfs, k3s imports them into containerd at boot, and9// imagePullPolicy: Never means a node can only ever run the builds10// that its own OS carries, or builds that an earlier OS on that node11// left in containerd's persistent store. That breaks the usual12// Kubernetes assumption that any node can pull any image, and every13// OS DaemonSet's design accounts for it:14//15//   - Each template pins a *stable* tag (liken.sh/operator:installed,16//     liken.sh/logs:installed). Every release tags its own build17//     with this stable tag, so the same pod spec resolves to each18//     node's own baked image, and applying a new release's manifests19//     does not change the image field at all.20//   - updateStrategy is OnDelete, so applying manifests never21//     deletes a running pod. A rolling update would recreate pods on22//     nodes whose OS does not carry the new image yet. For the23//     operator, that would kill the very pod each machine needs to24//     drive its own upgrade, and leave the machine unable to ever25//     take one.26//27// What is left is freshness. After a machine reboots into a new28// release, its existing pod objects predate the new manifests, and29// this steward deletes them so each DaemonSet can recreate its pod30// from the current template. Every stewarded DaemonSet carries a31// liken.sh/os-version annotation that names the release that shipped32// it. The template stamps this annotation onto its pods too. The33// steward refreshes a pod exactly when its machine reports that34// version in its facts, but the pod predates it. Both halves of that35// condition matter. A machine still on the old OS keeps its old pod,36// because evicting it would leave the machine without that pod's37// function at all. A machine ahead of the applied manifests also38// keeps its old pod, because a refresh would recreate another stale39// pod, and this would repeat every sweep until a leader running the40// new release applies manifests that can actually satisfy the41// machine.4243import (44	"errors"45	"fmt"46	"time"4748	"github.com/liken-sh/liken/kubernetes/apiclient"49	"github.com/liken-sh/liken/liken/kubernetes"50	"github.com/liken-sh/liken/liken/machine"51)5253// osVersionAnnotation names the liken release that a manifest, and54// the pods created from it, shipped with. image/build.sh substitutes55// the real version into each DaemonSet when it bakes the image.56const osVersionAnnotation = "liken.sh/os-version"5758// machineOperatorDaemonSet names the stewarded DaemonSet whose59// applied os-version the rollout gate also reads (fleet.go and60// rollout.go). Both OS DaemonSets ship in the same manifests write,61// so one annotation answers for the applied release, and the machine62// operator's template is the one whose lag blocks a machine's63// actuation after a reboot.64const machineOperatorDaemonSet = "liken-machine-operator"6566// stewardedDaemonSets lists the OS's own DaemonSets: the operator and67// the log relays. Their images are baked into the initramfs, so68// their pods need the steward's refresh after an upgrade. Each69// DaemonSet is expected to label its pods app: <name>, matching the70// selector its manifest declares.71var stewardedDaemonSets = []string{72	machineOperatorDaemonSet,73	"machine-logs",74}7576// daemonSetsPath is the collection of DaemonSets in liken-system: the77// OS's own and the ones the features seed.78const daemonSetsPath = "/apis/apps/v1/namespaces/liken-system/daemonsets"7980func daemonSetPodsPath(name string) string {81	return "/api/v1/namespaces/liken-system/pods?labelSelector=app%3D" + name82}8384// decideRefresh computes the steward's whole judgment, over the85// sweep's inputs: which of one DaemonSet's pods to evict, so the86// DaemonSet recreates them from the current template. dsVersion is87// the os-version annotation on the DaemonSet itself. It names the88// release whose manifests are actually applied. An empty string,89// meaning no annotation or no DaemonSet, means there is no release90// to refresh toward.91//92// A pod that is already terminating is not evicted again. It stays in93// the pod copy until the kubelet stops it, and each sweep in that time94// would otherwise evict it once more.95func decideRefresh(dsVersion string, machines []machine.Machine, pods []kubernetes.Pod) []kubernetes.Pod {96	if dsVersion == "" {97		return nil98	}99	running := make(map[string]string, len(machines))100	for i := range machines {101		running[machines[i].Metadata.Name] = machines[i].Status.Version.Liken102	}103	var refresh []kubernetes.Pod104	for _, p := range pods {105		if p.Terminating() {106			continue // an earlier sweep evicted it, and the DaemonSet recreates it once it is gone107		}108		osVersion, known := running[p.Spec.NodeName]109		if !known || osVersion != dsVersion {110			continue // the machine is not yet running what the manifests shipped111		}112		if p.Metadata.Annotations[osVersionAnnotation] == dsVersion {113			continue // the pod already comes from this release's template114		}115		refresh = append(refresh, p)116	}117	return refresh118}119120// stewardOSPods carries out the steward's work, run once per sweep,121// over every stewarded DaemonSet in turn. The steward's memory holds122// the terminating pods it has reported (stuckpods.go).123func stewardOSPods(r *fleetReader, machines []machine.Machine, steward *podSteward, now time.Time) {124	complete := true125	for _, name := range stewardedDaemonSets {126		complete = stewardDaemonSet(r, machines, name, steward, now) && complete127	}128	steward.settle(complete)129}130131// daemonSetVersion reads one DaemonSet's os-version annotation: the132// release whose manifests are actually applied. It returns "" when133// the DaemonSet does not exist yet, or predates this annotation,134// because both cases mean there is nothing yet to compare a machine's135// running version against. Any other failure is an error, not "",136// because "" turns off the rollout's template gate, and a failed read137// does not show that the template is current. The rollout gate reads138// this same function for the machine-operator DaemonSet, on the same139// terms (fleet.go, rollout.go).140func daemonSetVersion(r *fleetReader, name string) (string, error) {141	ds, err := r.daemonSet(name)142	if errors.Is(err, apiclient.ErrNotFound) {143		return "", nil144	}145	if err != nil {146		return "", err147	}148	return ds.Metadata.Annotations[osVersionAnnotation], nil149}150151// stewardDaemonSet reads one DaemonSet's shipped version, lists its152// pods, and evicts the stale ones. It deliberately evicts pods153// instead of deleting them, because eviction is the same action154// liken already uses for drains, and the DaemonSet recreates the pod155// either way. For the relay pod, eviction also discards the pod's156// emptyDir resume cursors. So each OS upgrade re-sends the tail of157// that machine's log streams once, and the envelopes' seq field158// removes any duplicates.159//160// It answers whether it read the DaemonSet's pods, or found no161// DaemonSet to read them for.162func stewardDaemonSet(r *fleetReader, machines []machine.Machine, name string, steward *podSteward, now time.Time) bool {163	dsVersion, err := daemonSetVersion(r, name)164	if err != nil {165		fmt.Printf("reading the %s DaemonSet for the steward: %v\n", name, err)166		return false167	}168	if dsVersion == "" {169		return true // no DaemonSet, or nothing applied yet, to steward toward170	}171	pods, err := r.daemonSetPods(name)172	if err != nil {173		fmt.Printf("listing %s pods for the steward: %v\n", name, err)174		return false175	}176	steward.reportOverdue(pods, now)177	for _, p := range decideRefresh(dsVersion, machines, pods) {178		if err := kubernetes.EvictPod(r.client, p); err != nil {179			fmt.Printf("refreshing pod %s: %v\n", p.Metadata.Name, err)180		} else {181			fmt.Printf("pod %s on %s predates release %s; evicted for the DaemonSet to recreate\n",182				p.Metadata.Name, p.Spec.NodeName, dsVersion)183		}184	}185	return true186}
cluster-operator/stuckpods.go 95.2%
1package main23// The report of an OS pod that stays terminating.4//5// The steward evicts a stale OS pod and then waits: the DaemonSet6// creates the new pod only after the kubelet has stopped the old one.7// The steward does not evict a terminating pod again (steward.go), so8// a pod that the kubelet never finishes stopping would leave nothing9// in the log. That pod keeps its machine on the old template, and a10// person has to look at the kubelet on that machine. So the steward11// reports such a pod, once, when it is past its deletion deadline.12//13// The API server sets deletionTimestamp to the moment of the deletion14// plus the pod's grace period, so a pod that is still in the listing15// after that time has run past its grace period.1617import (18	"fmt"19	"time"2021	"github.com/liken-sh/liken/liken/kubernetes"22)2324// podSteward is what the steward remembers from one sweep to the25// next: the UIDs of the overdue pods it has reported. One sweep reads26// every OS DaemonSet's listing, and settle keeps only the UIDs that27// sweep found overdue, so a pod that leaves the listing leaves the28// memory with it.29type podSteward struct {30	reported map[string]bool31	held     map[string]bool32}3334// overdue returns the pods of one listing that are terminating past35// their deletion deadline and that no earlier sweep reported. A pod36// whose deadline does not parse is left out, because there is no37// deadline to judge it by.38func (s *podSteward) overdue(pods []kubernetes.Pod, now time.Time) []kubernetes.Pod {39	if s.held == nil {40		s.held = map[string]bool{}41	}42	var report []kubernetes.Pod43	for _, p := range pods {44		if !p.Terminating() {45			continue46		}47		deadline, err := time.Parse(time.RFC3339, p.Metadata.DeletionTimestamp)48		if err != nil || !now.After(deadline) {49			continue50		}51		s.held[p.Metadata.UID] = true52		if !s.reported[p.Metadata.UID] {53			report = append(report, p)54		}55	}56	return report57}5859// settle ends one sweep: the overdue pods it found are the ones the60// next sweep does not report again. A sweep that could not read every61// listing is not complete, and it keeps the earlier UIDs too: the pods62// of the listing it missed are still overdue, and a UID it dropped63// would be reported again.64func (s *podSteward) settle(complete bool) {65	if !complete {66		for uid := range s.reported {67			if s.held == nil {68				s.held = map[string]bool{}69			}70			s.held[uid] = true71		}72	}73	s.reported, s.held = s.held, nil74}7576// reportOverdue prints one line for each pod that overdue returns.77func (s *podSteward) reportOverdue(pods []kubernetes.Pod, now time.Time) {78	for _, p := range s.overdue(pods, now) {79		fmt.Printf("pod %s on %s is still terminating past its deletion deadline %s; "+80			"the DaemonSet recreates it only after the kubelet on %s stops it\n",81			p.Metadata.Name, p.Spec.NodeName, p.Metadata.DeletionTimestamp, p.Spec.NodeName)82	}83}
cluster-operator/watches.go 94.3%
1package main23// The cluster operator's watches, and the reads a sweep makes through4// them.5//6// A sweep judges the whole fleet: every Machine, every heartbeat Lease,7// the Cluster, the OS DaemonSets in liken-system, and the pods of the8// two stewarded DaemonSets. The sweep runs at least every ten seconds,9// because the ticker is the clock that ages heartbeats into Lost10// verdicts and granted turns into a stalled rollout. A sweep that read11// each of these from the API server would send about ten requests12// every ten seconds, and one more full sweep for every status write in13// the fleet. So the operator watches each collection, keeps a copy in14// memory, and the sweep reads the copies.15//16// A watch also decides whether a change wakes the loop at once:17//18//   - A Machine wakes the loop on every change. Any machine's19//     transition can change the Cluster's phase, a rollout's budget,20//     or a Lost verdict, and those are status writes.21//   - The Cluster wakes the loop only on an edit to its spec. This22//     program is the only writer of the Cluster's status, and its own23//     write must not start another sweep.24//   - A DaemonSet wakes the loop only on an edit to its spec. A25//     leader's boot writes the new release's template, and the steward26//     refreshes pods from it. The DaemonSet controller writes status27//     all the time, and none of it concerns the sweep.28//   - The heartbeat Leases and the pods wake nothing. The ticker is the29//     clock that judges a heartbeat's age, and the steward acts on a30//     machine's version, which arrives as a Machine change. The31//     Leases' handler records when each Lease's renewTime changed, on32//     this program's clock, and the sweep measures a heartbeat's age33//     from that record (heartbeats.go).34//35// A copy that cannot answer, because its watch has not synced or its36// last watch failed, is never read as the truth. The sweep reads the37// API server instead. That covers the first seconds after this process38// takes the lead, a release skew, where this binary runs under the39// previous release's RBAC, and an API server restart. After a failed40// watch, the reflector waits out a backoff of up to a minute before it41// watches again, and a copy then misses each write made in that time,42// such as a machine's Degraded status, which the budget must count.43//44// Two reads stay direct on every sweep. The flux deploy key Secret is45// read only while the flux feature is declared, and the permission to46// read it arrives with the feature's own manifests, so a watch would47// be refused on every fleet without GitOps. The flux-system Namespace48// is read only while the feature is not declared, to find an49// installation to tear down, and this program may read that one50// Namespace by name only.5152import (53	"context"54	"errors"55	"slices"56	"strings"57	"time"5859	"github.com/liken-sh/liken/kubernetes/apiclient"60	"github.com/liken-sh/liken/kubernetes/election"61	"github.com/liken-sh/liken/kubernetes/events"62	"github.com/liken-sh/liken/kubernetes/informer"63	"github.com/liken-sh/liken/kubernetes/memo"64	"github.com/liken-sh/liken/liken/api"65	"github.com/liken-sh/liken/liken/cluster"66	"github.com/liken-sh/liken/liken/kubernetes"67	"github.com/liken-sh/liken/liken/kubernetes/watch"68	"github.com/liken-sh/liken/liken/machine"69	"k8s.io/apimachinery/pkg/runtime/schema"70	"k8s.io/client-go/dynamic"71	"k8s.io/client-go/tools/cache"72)7374// The kinds this operator watches, as the dynamic client names them.75var (76	likenVersion      = schema.GroupVersion{Group: "liken.sh", Version: strings.TrimPrefix(api.APIVersion, "liken.sh/")}77	machineResource   = likenVersion.WithResource("machines")78	clusterResource   = likenVersion.WithResource("clusters")79	leaseResource     = schema.GroupVersionResource{Group: "coordination.k8s.io", Version: "v1", Resource: "leases"}80	daemonSetResource = schema.GroupVersionResource{Group: "apps", Version: "v1", Resource: "daemonsets"}81	podResource       = schema.GroupVersionResource{Version: "v1", Resource: "pods"}82)8384// The kind labels of liken_watch_restarts_total, one for each watch.85const (86	leaseKind     = "Lease"87	daemonSetKind = "DaemonSet"88	podKind       = "Pod"89)9091// watchKinds lists every kind label, so the counter reports a zero for92// each watch before its first restart.93var watchKinds = []string{machineKind, clusterKind, leaseKind, daemonSetKind, podKind}9495// appLabel is the label each stewarded DaemonSet puts on its pods. The96// pod copy is indexed by it, so the steward reads one DaemonSet's pods97// without a scan.98const appLabel = "app"99100// fleetReader reads each API object a sweep judges: from the copy a101// watch keeps, or from the API server when the copy cannot answer. A102// fleetReader with no copies at all reads the API server every time,103// which is what the tests of a sweep use.104type fleetReader struct {105	client *apiclient.Client106107	// recorder posts the Events about the Machines and the Cluster108	// (events.go). It is here because every part of a sweep that acts109	// reads through the fleet reader. A nil recorder posts nothing.110	recorder *events.Recorder111112	machineCopy   *informer.Collection113	clusterCopy   *informer.Collection114	leaseCopy     *informer.Collection115	daemonSetCopy *informer.Collection116	podCopy       *informer.Collection117118	// machineVersions and clusterVersions are the memos of this119	// program's own status writes to the Machines and the Cluster, and120	// of its reads of them from the API server (kubernetes/memo). The121	// watch delivers a write a moment after the API server answers it.122	// A sweep that decided from a copy without its own last grant would123	// count one fewer machine in flight, and could grant a turn beyond124	// the disruption budget. So a copy answers only at the version of125	// this program's last write or read of the object, and the sweep126	// otherwise reads that one object from the API server.127	machineVersions *memo.Versions128	clusterVersions *memo.Versions129130	// sightings records when this program saw each heartbeat Lease131	// change (heartbeats.go). The Leases' watch writes it, and each132	// read of the heartbeats passes through it.133	sightings heartbeatSightings134}135136// watchFleet opens the watches and returns the reader over their137// copies. Each change that needs a sweep calls wake.138func watchFleet(ctx context.Context, watcher dynamic.Interface, client *apiclient.Client,139	wake func(), restarted func(kind string)) *fleetReader {140	start := func(kind string, source informer.Source, handler cache.ResourceEventHandler, indexers cache.Indexers) *informer.Collection {141		return informer.Start(ctx, watcher, source, informer.Options{142			Handler:  handler,143			Synced:   wake,144			Reopened: func() { restarted(kind) },145			// A copy stops answering after any failed watch, as the146			// head of this file says.147			UnreadyOnWatchError: true,148			Indexers:            indexers,149		})150	}151	machines := informer.Source{Resource: machineResource}152	clusters := informer.Source{Resource: clusterResource}153	leases := informer.Source{Resource: leaseResource, Namespace: "liken-system"}154	daemonSets := informer.Source{Resource: daemonSetResource, Namespace: "liken-system"}155	pods := informer.Source{Resource: podResource, Namespace: "liken-system",156		LabelSelector: appLabel + " in (" + strings.Join(stewardedDaemonSets, ",") + ")"}157158	r := &fleetReader{159		client:          client,160		machineCopy:     start(machineKind, machines, watch.WakeOnChange[machine.Machine](machines, wake), nil),161		clusterCopy:     start(clusterKind, clusters, watch.WakeOnEdit[cluster.Cluster](clusters, wake), nil),162		daemonSetCopy:   start(daemonSetKind, daemonSets, watch.WakeOnEdit[featureWorkload](daemonSets, wake), nil),163		podCopy:         start(podKind, pods, nil, cache.Indexers{appLabel: watch.LabelIndex(appLabel)}),164		machineVersions: memo.New(),165		clusterVersions: memo.New(),166	}167	r.leaseCopy = start(leaseKind, leases, r.sightings.handler(leases), nil)168	return r169}170171// current answers the view of a copy the sweep may read, or a view172// that answers nothing when the copies may be behind the API server.173//174// Every watch of this process shares one HTTP/2 connection. When the175// API server behind it loses power or leaves the network, the stream176// goes quiet, and client-go closes it only about 45 seconds after the177// last frame. The copies look current the whole time. A sweep that178// judged heartbeats from them would find every machine's heartbeat179// aging, and mark live machines Lost after 40 seconds, through a180// write that reaches a live API server on another connection.181//182// The Leases' copy carries its own proof of freshness. The leader183// election renews this program's own Lease in liken-system every five184// seconds, and the Leases' copy receives each renewal. When the copy's185// view of that renewal is older than one renewal deadline and one retry186// period, fifteen seconds, the stream or the renewals have stopped, and187// the sweep reads the API server instead. A heartbeat then ages at most188// fifteen seconds in the copy, well inside its 40. client-go gives the189// election and the watches one shared connection, so a stalled190// connection stops the renewals too: the write guard then refuses every191// write after ten seconds, and the process exits when the election192// gives up.193//194// That proof covers a quiet connection, not a failed watch. Each195// reflector recovers from a failed watch on its own backoff, so a fresh196// Leases' copy says nothing about the Machines' copy. A failed watch197// makes its own copy stop answering instead198// (informer.Options.UnreadyOnWatchError), and this check comes on top199// of that.200//201// A copy that holds every object of its kind, with no selector, is202// Whole, so a list from it also reads each object this program wrote203// that the copy does not hold yet.204func (r *fleetReader) current(held *informer.Collection) informer.View {205	if held == nil {206		return informer.View{}207	}208	lease, found, ok := watch.Get[kubernetes.Lease](r.leaseCopy.View(), leaseNamespace+"/"+leaseName)209	if !ok || !found {210		return informer.View{}211	}212	renewed, err := time.Parse(time.RFC3339Nano, lease.Spec.RenewTime)213	if err != nil || time.Since(renewed) >= copyFreshness {214		return informer.View{}215	}216	view := held.View()217	view.Whole = held == r.machineCopy || held == r.clusterCopy218	return view219}220221// copyFreshness is how old the copy's view of this program's own222// leader Lease may be before the sweep stops reading the copies.223var copyFreshness = election.RenewDeadline + election.RetryPeriod224225// machines reads every Machine in the fleet. A copy at another version226// than this program's own last write or read of the Machine, such as a227// copy without the grant the last sweep wrote, is read again from the228// API server, which holds the write. While the copy cannot answer, the229// list comes from the API server and notes each Machine's version, so a230// copy that answers later at an older version is read again.231func (r *fleetReader) machines() ([]machine.Machine, error) {232	held := informer.Held{View: r.current(r.machineCopy), Versions: r.machineVersions}233	machines, err := informer.List[machine.Machine](r.client, held, kubernetes.MachinesPath, machinePath)234	if err != nil {235		return nil, err236	}237	settleMemo(held, machines)238	return machines, nil239}240241// readTurns reads from the API server each Machine that the ledger242// names and whose copy holds no grant, and answers the list with each243// fresh copy in place of the old one. A Machine the API server no244// longer holds leaves the list. The decision to free a ledger entry's245// slot compares the Machine's version with the entry's, and a copy246// older than the entry can differ from it while the grant can still247// land, so this read must see every write at once. A Machine whose248// copy holds a grant needs no read, because its slot stays in use249// either way. A settled fleet has an empty ledger, so this costs about250// one read for each finished turn.251func (r *fleetReader) readTurns(turns []cluster.RebootTurn, machines []machine.Machine) ([]machine.Machine, error) {252	machines = slices.Clone(machines)253	for _, turn := range turns {254		i := slices.IndexFunc(machines, func(m machine.Machine) bool { return m.Metadata.Name == turn.Machine })255		if i >= 0 && api.FindCondition(machines[i].Status.Conditions, machine.RebootApprovedCondition) != nil {256			continue257		}258		fresh, err := memo.ReadFresh[machine.Machine](r.client, r.machineVersions, turn.Machine, machinePath(turn.Machine))259		switch {260		case errors.Is(err, apiclient.ErrNotFound):261			if i >= 0 {262				machines = slices.Delete(machines, i, i+1)263			}264		case err != nil:265			return nil, err266		case i >= 0:267			machines[i] = *fresh268		default:269			machines = append(machines, *fresh)270		}271	}272	return machines, nil273}274275func machinePath(name string) string { return kubernetes.MachinesPath + "/" + name }276277func clusterPath(name string) string { return kubernetes.ClustersPath + "/" + name }278279// settleMemo drops the memo's record of each object that the list left280// out and the store no longer holds: a Machine somebody deleted, whose281// record would otherwise stay for the life of the process. It also282// drops the record of each object whose copy in the store is at the283// noted version (watch.Settle), so a later write from another writer284// costs no read. A list from the API server settles nothing, because285// no ready store compares with it. The records that list noted stay286// until a ready store holds their versions.287func settleMemo[T any, P informer.Object[T]](held informer.Held, items []T) {288	if !held.View.Ready() {289		return290	}291	listed := make(map[string]bool, len(items))292	keys := make([]string, 0, len(items))293	for i := range items {294		key := informer.Key(P(&items[i]).GetObjectMeta())295		listed[key] = true296		keys = append(keys, key)297	}298	held.Versions.ForgetGone(held.View.Store, listed)299	watch.Settle(held, keys...)300}301302// publishStatus writes a Machine's status, and notes the version the303// API server answered for the Machines' copy. A write that fails notes304// that this program holds no current copy: a request that timed out can305// still have landed, so the next read of that Machine goes to the API306// server.307func (r *fleetReader) publishStatus(m *machine.Machine, status *machine.MachineStatus) error {308	return r.machineVersions.Send(m.Metadata.Name, func() (string, error) {309		return kubernetes.PublishStatus(r.client, m, status)310	})311}312313// publishClusterStatus writes the Cluster's status, and notes the314// version for the Clusters' copy, so the next sweep does not compare315// its verdict against a status older than its own last write. It316// answers the Cluster the API server stored, or nil when the answer317// held none.318func (r *fleetReader) publishClusterStatus(clusterDoc *cluster.Cluster) (*cluster.Cluster, error) {319	var stored *cluster.Cluster320	err := r.clusterVersions.Send(clusterDoc.Metadata.Name, func() (string, error) {321		written, err := kubernetes.PublishClusterStatus(r.client, clusterDoc)322		if err != nil || written == nil {323			return "", err324		}325		stored = written326		return written.Metadata.ResourceVersion, nil327	})328	return stored, err329}330331// clusters reads every Cluster. A fleet has one.332func (r *fleetReader) clusters() ([]cluster.Cluster, error) {333	held := informer.Held{View: r.current(r.clusterCopy), Versions: r.clusterVersions}334	clusters, err := informer.List[cluster.Cluster](r.client, held, kubernetes.ClustersPath, clusterPath)335	if err != nil {336		return nil, err337	}338	settleMemo(held, clusters)339	return clusters, nil340}341342// cluster reads the Cluster this program operates.343func (r *fleetReader) cluster(name string) (*cluster.Cluster, error) {344	return watch.ReadOne[cluster.Cluster](r.client,345		informer.Held{View: r.current(r.clusterCopy), Versions: r.clusterVersions}, name, clusterPath(name))346}347348// heartbeats answers, for each machine's heartbeat Lease, when this349// program last saw its renewTime change, on this program's clock350// (heartbeats.go). now is the sweep's clock, and it stamps a change351// that this read finds before the watch delivers it.352func (r *fleetReader) heartbeats(now time.Time) (map[string]time.Time, error) {353	if leases, ok := watch.List[kubernetes.Lease](r.current(r.leaseCopy)); ok {354		return r.sightings.heard(kubernetes.Renewals(leases), now), nil355	}356	renewals, err := kubernetes.ListHeartbeats(r.client)357	if err != nil {358		return nil, err359	}360	return r.sightings.heard(renewals, now), nil361}362363// workloads reads one kind of workload in liken-system for the feature364// janitor. The DaemonSets come from their copy, which the steward365// reads too.366func (r *fleetReader) workloads(listPath string) ([]featureWorkload, error) {367	if listPath == daemonSetsPath {368		if daemonSets, ok := watch.List[featureWorkload](r.current(r.daemonSetCopy)); ok {369			return daemonSets, nil370		}371	}372	return kubernetes.List[featureWorkload](r.client, listPath)373}374375// daemonSet reads one DaemonSet in liken-system.376func (r *fleetReader) daemonSet(name string) (*featureWorkload, error) {377	if ds, found, ok := watch.Get[featureWorkload](r.current(r.daemonSetCopy), "liken-system/"+name); ok {378		if !found {379			return nil, apiclient.ErrNotFound380		}381		return ds, nil382	}383	return apiclient.Get[featureWorkload](r.client, daemonSetsPath+"/"+name)384}385386// daemonSetPods reads the pods of one stewarded DaemonSet.387func (r *fleetReader) daemonSetPods(name string) ([]kubernetes.Pod, error) {388	if pods, ok := watch.ByIndex[kubernetes.Pod](r.current(r.podCopy), appLabel, name); ok {389		return pods, nil390	}391	return kubernetes.List[kubernetes.Pod](r.client, daemonSetPodsPath(name))392}
cluster/changes.go 100.0%
1package cluster23// This file classifies a cluster-document edit by how the system4// must apply it.5//6// liken has four tiers of convergence, determined by where a setting7// is read and by what still acts on it once the boot that read it is8// over.9//10// Settings the kernel reads live (/proc/sys) reconcile in place.11//12// Settings k3s reads at process start (the boot drop-in,13// registries.yaml, the feature manifests, and the Go runtime14// environment init hands the process) apply by restarting the k3s15// child process.16//17// Settings a boot reads to reach a decision it never revisits apply18// by staging the document and waiting. The two are which datastore to19// join, and the URL to join through. The machine's next boot reads the20// staged copy, whenever that boot comes and for whatever reason. This21// tier asks for no disruption at all. NextBootApplies below names22// these fields and every reader of them. It also states the one value23// a running machine still holds after such an edit.24//25// Everything else a boot acted on applies by rebooting the machine.26// The running system carries the old value's effects, and only a boot27// undoes them. The address plan is in the node's addresses and routes.28// Storage is in the mounts. The time upstreams are the servers a29// leader's discipline loop queries for the life of the boot.30//31// A reboot works in place of every other tier, because a reboot is a32// k3s restart plus more, and it is also a next boot. So an edit that33// spans tiers is safe at the reboot tier, and classification must34// always err toward it. The one pair that needs its own argument is a35// restart-class field beside a next-boot-class one, which36// RestartApplies below sends to the restart tier.37//38// This file is the classifier for the two middle tiers. The operator39// consults it to determine whether a staged cluster document calls40// for a restart intent, a reboot intent, or no intent at all. Init41// consults RestartApplies before it acts on a restart intent, so the42// two programs can never disagree about what a restart may apply.4344import "encoding/json"4546// RestartApplies reports whether a k3s restart is enough to move a47// machine from the current spec to the desired one. The two specs48// must differ (no drift needs no disruption at all), and the49// difference must be confined to the restart-class fields, the ones50// k3s reads only at process start, and to the next-boot-class fields51// below them.52//53// The comparison works by subtraction rather than by a list of54// changed domains: it zeroes the restart-class fields on copies of55// both specs and asks whether anything else differs. Any remaining56// difference means the reboot tier. This makes the safety property57// structural: a future ClusterSpec field is reboot-class from the58// day it is added, with no classification table to remember to59// extend, and forgetting one could only ever cost an unnecessary60// reboot, never an under-applied restart.61//62// Both comparisons run over JSON renderings, the same bytes the63// document hash is built from, so this classification and the hash64// can never disagree about whether two specs differ. Version and65// Releases are excluded from both comparisons: canonical documents66// never carry them, because the operator strips them before hashing,67// and their actuation is a download, not a boot.68func RestartApplies(current, desired ClusterSpec) bool {69	current.Version, desired.Version = "", ""70	current.Releases, desired.Releases = ClusterReleasesSpec{}, ClusterReleasesSpec{}71	if jsonEqual(current, desired) {72		return false73	}74	// A feature that leaves host state behind is the one exception to75	// the rule that features are restart-class. Netfilter chains,76	// mounts, live storage sessions, and loaded modules outlive the77	// k3s process, so a restart stops the controller and leaves its78	// programming in force with nothing maintaining it. A boot79	// discards all of it. The machine also runs the controller right80	// up to that boot, so there is no interval where the programming81	// stands without the controller that maintains it.82	if RetractionLeavesHostState(&Cluster{Spec: current}, &Cluster{Spec: desired}) {83		return false84	}85	current.Features, desired.Features = nil, nil86	current.Registries, desired.Registries = RegistriesSpec{}, RegistriesSpec{}87	current.Runtime, desired.Runtime = ClusterRuntimeSpec{}, ClusterRuntimeSpec{}88	// The rest of the address plan is reboot-class, because those89	// fields set a machine's node IP and the ranges k3s hands out,90	// both of which a boot has already acted on by the time k3s91	// starts. The NodePort list is the exception: nothing reads it92	// before k3s does, so it is zeroed here with the other93	// restart-class fields rather than with its own section.94	current.Network.NodePortCIDRs, desired.Network.NodePortCIDRs = nil, nil95	// Origin and Endpoint belong to the next-boot tier96	// (NextBootApplies below), so they zero here too. An edit that97	// changes one of them beside a restart-class field converges by a98	// restart, not by a reboot. The restart re-renders the k3s drop-in99	// from the staged document (init/restart.go). That render writes100	// exactly the join keys a boot would write from the same document.101	// A restart re-runs every reader but one, the follower's time102	// source list, and leaving that list to the next boot is what the103	// next-boot tier promises anyway.104	current.Origin, desired.Origin = "", ""105	current.Endpoint, desired.Endpoint = "", ""106	return jsonEqual(current, desired)107}108109// NextBootApplies reports whether a machine can adopt the desired110// spec by staging it and waiting for its next boot, with no reboot111// and no k3s restart. The two specs must differ, and the difference112// must be confined to Origin and Endpoint.113//114// These two fields have three readers, and all three run during a115// boot. leaderJoinConfig (init/k3s.go) reads Origin to decide whether116// the founding leader renders cluster-init or joins a datastore that117// already exists, and it reads Endpoint for the URL a joining leader118// points at. k3sBootConfig (init/k3s.go) writes Endpoint into a119// follower's server: key. timeSources (init/time.go) puts the120// endpoint's host at the end of a follower's time sources, as the121// fallback for a leader that declares no address.122//123// Nothing re-reads either field after that. A joined member keeps a124// client-side load balancer that holds every leader's address, so it125// never asks the endpoint for anything again. A running k3s never126// reads the drop-in again.127//128// One value does outlive the boot that read it: the endpoint's host129// stays in a follower's time source list until the next boot. That130// entry is the last resort, behind every leader that declares an131// address inside nodeCIDR. It costs nothing on a fleet whose leaders132// declare their addresses. On a fleet whose leaders declare none, the133// endpoint's host is a follower's only time source, and the follower134// asks the old host until it boots. So keep the old address answering135// until every machine has adopted the edit. This is the one cost of136// the tier. The failure it risks is clock drift on that follower,137// not a lost cluster.138//139// The tier is the same for every machine, whatever its role. A140// follower, a founding leader, and an ordinary leader all read these141// fields only during a boot. No role needs a heavier tier than142// another.143//144// The staged document therefore costs little to hold. The machine145// keeps serving under the document it booted. The next boot, for an146// upgrade or for any other change, reads the staged copy and proves147// it.148//149// The comparison uses the same subtraction over JSON that150// RestartApplies uses, for the same reason. A field this function does151// not name stays in the comparison. So a future ClusterSpec field can152// never fall into this tier by accident.153func NextBootApplies(current, desired ClusterSpec) bool {154	current.Version, desired.Version = "", ""155	current.Releases, desired.Releases = ClusterReleasesSpec{}, ClusterReleasesSpec{}156	if jsonEqual(current, desired) {157		return false158	}159	current.Origin, desired.Origin = "", ""160	current.Endpoint, desired.Endpoint = "", ""161	return jsonEqual(current, desired)162}163164// jsonEqual compares two values by their JSON bytes. A marshal error165// reads as "differs", the safe direction under this file's rule.166// Every type compared here is a plain data struct that cannot167// actually fail to marshal. But if one somehow did, reading it as168// "differs" costs at most an unneeded reboot, while a false "equal"169// result could leave a change unapplied.170func jsonEqual(a, b any) bool {171	ra, errA := json.Marshal(a)172	rb, errB := json.Marshal(b)173	return errA == nil && errB == nil && string(ra) == string(rb)174}
cluster/cluster.go 91.8%
1// Package cluster is the Cluster API: the document that describes2// what the machines form together, as Go types.3//4// A Machine describes one computer. A Cluster describes the group.5// The choice of where a fact goes depends on who must agree on it.6// Every fact in a ClusterSpec is one of two kinds. Either every node7// must hold the fact identically (which machines run control8// planes, the address ranges that pods and services use), or the9// fact belongs to the group and no single machine owns it (the10// endpoint followers join through). Every fact specific to one11// machine (its interfaces, its addresses, its disks) stays on the12// Machine. The two document packages never import each other. The13// grammar they share (ObjectMeta, Phase, Condition, Role) is in14// the api package underneath both.15//16// Like the Machine manifest, the Cluster manifest arrives as a file17// in the image. The liken operator seeds it into the cluster as a18// custom resource once the API is up. Unlike the Machine manifest,19// every machine's image includes the same cluster.yaml. The document20// is cluster-scoped, so there is exactly one. liken already treats21// "same image" as "same cluster": the image contains the cluster's CA22// and join token.23//24// The type names repeat a word (cluster.ClusterSpec), for the same25// reason the machine package's type names do. They mirror the CRD26// kind and Kubernetes' XxxSpec/XxxStatus convention. Matching what27// `kubectl explain cluster.spec` shows is worth more than avoiding28// the repeated word.29package cluster3031import (32	"errors"33	"fmt"34	"io/fs"35	"net"36	"os"37	"slices"38	"time"3940	"github.com/liken-sh/liken/liken/api"41	"sigs.k8s.io/yaml"42)4344// ClusterManifestPath is where the image contains the cluster's45// manifest. Init reads it for this machine's role. The operator46// reads the same file through a hostPath mount, to seed the47// in-cluster Cluster resource.48const ClusterManifestPath = "/etc/liken/cluster.yaml"4950// BootClusterManifestPath is where init publishes the cluster51// document this boot actually derived its role from: the staged or52// proven copy from machineState, or the image's seed on a first53// boot. The operator needs the bytes, not only their hash, because54// drift detection compares documents by meaning. A hand-written seed55// and the operator's canonical rendering of the same spec are56// different bytes that say the same thing. A formatting difference57// must never reboot the fleet.58const BootClusterManifestPath = "/run/liken/cluster.yaml"5960type Cluster struct {61	APIVersion string         `json:"apiVersion"`62	Kind       string         `json:"kind"`63	Metadata   api.ObjectMeta `json:"metadata"`64	Spec       ClusterSpec    `json:"spec,omitzero"`65	Status     ClusterStatus  `json:"status,omitzero"`66}6768// GetObjectMeta answers the metadata that a watch's store and the69// operators' memo read (api.Meta).70func (m *Cluster) GetObjectMeta() api.Meta { return m.Metadata }7172// ClusterOrigin records how the cluster's datastore came to exist.73// Founded means liken created the datastore, through the founding74// leader's cluster-init. Adopted means the datastore already existed75// in a cluster liken did not create, and liken machines join it as76// members instead of starting one. This distinction matters in one77// place: the founding leader's datastore decision (init/k3s.go). An78// adopted cluster's founder joins like every other machine, because79// initializing a second datastore next to a live one would split the80// cluster in two.81// The values are CamelCase because Kubernetes API constants are82// CamelCase, the same convention the phase and policy vocabularies83// follow.84type ClusterOrigin string8586const (87	OriginFounded ClusterOrigin = "Founded"88	OriginAdopted ClusterOrigin = "Adopted"89)9091// ClusterStatus is what a reader can observe about the cluster as a92// whole. The cluster operator writes it (see cluster-operator/fleet.go);93// observing the fleet is that program's whole job. The staged and94// proven copies of the Cluster document carry spec only. Status95// never appears in the lifecycle bytes, because observations are not96// part of the document's identity.97//98// The conditions are the real observations, and the phase is their99// one-word summary, the same arrangement the Machine uses. Phase is100// Ready when every machine is Ready. Phase is Updating when the only101// machines not Ready are mid-transition: rebooting into a change,102// waiting on one, or booting. Phase is Degraded when any machine is103// Lost, Blocked, or otherwise unhealthy. This status can never show104// one state: quorum loss. Losing a majority of leaders takes the API105// server down with it, so nobody is left to write the status. When106// quorum is lost, the symptom is a status that stops updating.107type ClusterStatus struct {108	Phase    api.Phase             `json:"phase,omitempty"`109	Machines MachineTally          `json:"machines,omitzero"`110	Releases ClusterReleasesStatus `json:"releases,omitzero"`111	Flux     ClusterFluxStatus     `json:"flux,omitzero"`112113	// RebootTurns is the grant ledger: each machine that holds a114	// reboot turn, or is about to receive one. The cluster operator115	// adds a machine here in a write conditional on the Cluster's116	// resourceVersion before it grants the turn, so two copies of the117	// operator cannot both use one budget slot118	// (cluster-operator/rollout.go).119	RebootTurns []RebootTurn `json:"rebootTurns,omitempty"`120121	// ObservedGeneration is the metadata.generation of the spec that122	// this status judged, stamped by the sweep on every write. The123	// conditions each carry the same stamp, but a client that only124	// asks "has any controller seen my edit yet" should not have to125	// parse conditions for the answer. Kubernetes controllers publish126	// this field at the top of status for exactly that question, and127	// `kubectl rollout status` and the common wait libraries read it128	// there.129	ObservedGeneration int64 `json:"observedGeneration,omitempty"`130131	Conditions []api.Condition `json:"conditions,omitempty"`132}133134// ClusterReleasesStatus is what the sweep observes about releases.135// Newest is the catalog's highest version: spec-local, with no136// network call. Available is the version the channel's own document137// announces as its latest, learned by polling <source>/channel.yaml.138// It may name a version the catalog does not hold yet, which is139// exactly the signal it exists to surface. Available is advisory by140// design: adopting a release still means committing a digest-pinned141// catalog entry. Like every status field, both values are derived.142// They exist so printer columns can show the whole story next to143// spec.version, without anyone comparing versions at the terminal.144type ClusterReleasesStatus struct {145	Newest    string `json:"newest,omitempty"`146	Available string `json:"available,omitempty"`147}148149// ClusterFluxStatus is the flux feature's observable half.150// PublicKey is the fleet's deploy key: the public half, in the151// authorized_keys form a forge accepts. The private half lives only152// in the flux-system Secret, minted there by the cluster operator153// (cluster-operator/flux.go), so nobody ever handles private154// material to set the sync up: they register this value at the155// forge, and the fleet does the rest. An empty status means the156// feature is not declared, or the key is not minted yet.157type ClusterFluxStatus struct {158	PublicKey string `json:"publicKey,omitempty"`159}160161// A RebootTurn is one entry of the grant ledger. Machine names the162// Machine. MachineResourceVersion is the resourceVersion of the163// Machine that the grant write names: the API server accepts that164// write only while the Machine is still at this version, so once the165// Machine is at another version and holds no grant, no grant that this166// entry covers can land. Since is when the entry entered the ledger.167type RebootTurn struct {168	Machine                string    `json:"machine"`169	MachineResourceVersion string    `json:"machineResourceVersion"`170	Since                  time.Time `json:"since"`171}172173// MachineTally counts the cluster's Machines: how many are fully174// healthy (phase Ready, with a fresh heartbeat) out of how many175// exist. Summary holds the same two numbers formatted as "4/5". The176// struct stores this because a CRD printer column can read one177// field, but cannot combine two.178type MachineTally struct {179	Ready   int    `json:"ready"`180	Total   int    `json:"total"`181	Summary string `json:"summary,omitempty"`182}183184type ClusterSpec struct {185	// Origin is how the cluster's datastore came to exist: Founded186	// (the default when unset) or Adopted. An adopted cluster is one187	// liken is migrating into, rather than one it created. Its188	// machines join an existing datastore through the endpoint, and189	// no leader ever initializes a new one. The one legal edit is190	// the promotion, Adopted to Founded, made once the last foreign191	// member is gone. After that edit, a rebuild from scratch192	// behaves like any founded cluster, with the founder running193	// cluster-init.194	Origin ClusterOrigin `json:"origin,omitempty"`195196	// Leaders names the machines that run control planes, by their197	// Machine names. A machine's role is derived, never declared: a198	// machine is a leader exactly when its name appears here.199	// Promoting a follower is therefore one Cluster edit, not a200	// coordinated pair of Machine edits. One name means k3s keeps201	// its state in sqlite. More than one name means embedded etcd,202	// whose majority quorum calls for an odd count of leaders:203	// three, not two or four. No admission rule enforces the odd204	// count, on purpose. Growing from one leader to three in a205	// single edit never passes through two, and a transient even206	// state is not worth rejecting.207	Leaders []string `json:"leaders,omitempty"`208209	// Endpoint is the URL followers join the cluster through, for210	// example https://10.10.0.1:6443. The system uses it for first211	// contact only. After a follower joins, k3s agents keep a212	// client-side load balancer that holds the address of every213	// leader, so a dead endpoint strands only brand-new followers,214	// never running ones. (Followers' time queries ask each leader by215	// its own address and bypass the endpoint entirely.) Endpoint is216	// a single, explicit input, on purpose. An endpoint that should217	// outlive any single leader, such as a DNS name or a virtual IP,218	// is a choice for the deployment to make, never the OS.219	Endpoint string `json:"endpoint,omitempty"`220221	// Network holds the network facts that k3s requires every node222	// to agree on. These are cluster-scoped facts, and declaring223	// them per node would misstate them.224	Network ClusterNetworkSpec `json:"network,omitzero"`225226	// Time is the cluster's time hierarchy: where the leaders get227	// their time. It lives on the Cluster for the same reason the228	// network plan does. Clocks are a fact the whole fleet must229	// agree on, and TLS, and with it the cluster itself, stops230	// working when they do not agree.231	Time ClusterTimeSpec `json:"time,omitzero"`232233	// Disruption bounds how much of the fleet may be down at once234	// when the cluster sequences reboots (cluster-operator/rollout.go).235	Disruption ClusterDisruptionSpec `json:"disruption,omitzero"`236237	// Features is the cluster's set of opt-ins from liken's curated238	// feature vocabulary (features.go): optional capabilities that239	// the fleet as a whole offers. It lives on the Cluster because a240	// feature is a fact every node must agree on. A PersistentVolume241	// can attach to any node the scheduler picks, and even the k3s242	// disable list applies cluster-wide. Features is an object keyed243	// by feature slug, rather than a list of names, so a feature can244	// grow parameters without breaking the schema. The key's245	// presence is the opt-in, and a feature's zero configuration is246	// {}. The value is a pointer, so an explicit null arrives as247	// present-but-nil, and validateFeatures refuses it.248	// Everywhere else in Kubernetes, null means unset, so a bare249	// `traefik:` in a hand-written manifest must be an error. It250	// must never become a quiet enable, or an even quieter no-op.251	// Features converge by reboot like every other cluster fact.252	// They stay in the canonical staged document, so an edit changes253	// the document's hash and rolls through the fleet as staged254	// changes and granted reboots.255	Features map[string]*FeatureConfig `json:"features,omitempty"`256257	// Runtime is the discipline the cluster imposes on the k3s process258	// init launches, on the kubelet component inside it, and on259	// containerd beside it (runtime.go). The k3s subsection is that260	// process's Go environment and its log level. The kubelet261	// subsection is a configuration file that init names on the k3s262	// command line. The containerd subsection is a drop-in beside the263	// configuration that k3s renders for containerd. Like features and264	// registries, all three are read only when the k3s process starts,265	// so an edit converges by restarting k3s in place, not by266	// rebooting.267	Runtime ClusterRuntimeSpec `json:"runtime,omitzero"`268269	// Registries is how container images arrive on the fleet's270	// machines: mirror endpoints that containerd pulls through, and271	// k3s's embedded peer-to-peer registry. It lives on the Cluster272	// because any node may be asked to pull any image, so how images273	// arrive is a fact the whole fleet must agree on. Credentials274	// are deliberately not here. A spec is public, so credentials275	// enter through the registry-credentials Secret instead276	// (machine/registries.go tells that story). Like features,277	// registries stay in the canonical staged document. Both are278	// read only when the k3s process starts, so their edits converge279	// by restarting k3s in place, instead of rebooting the machine280	// (changes.go).281	Registries RegistriesSpec `json:"registries,omitzero"`282283	// Version is the fleet's target liken release: the one field an284	// upgrade edits. Machines carry no version in their specs.285	// Instead, each machine's operator compares the version its boot286	// reported (status.version.liken) against this target, live, and287	// moves toward it through the same staged-change-and-granted-reboot288	// machinery that applies every other change. Declaring the289	// version here, and only here, makes an upgrade one edit instead290	// of one edit per machine.291	Version string `json:"version,omitempty"`292293	// Releases is where release artifacts come from, and which294	// releases exist: the catalog that Version must name its target295	// in. An admission rule on the CRD enforces that requirement, so296	// the system rejects a mistyped target at admission, instead of297	// blocking machines.298	Releases ClusterReleasesSpec `json:"releases,omitzero"`299}300301// ClusterReleasesSpec is the cluster's release feed. The system302// reads Version and Releases live, on purpose: the operator reads303// them from the in-cluster resource on every pass, and the canonical304// cluster document that machines stage and reboot into excludes them305// (machine-operator/cluster.go's renderCluster). Because of this306// split, publishing a release or retargeting the fleet moves307// machines through downloads and sequenced reboots. It never stages308// a fleet-wide configuration change of its own.309type ClusterReleasesSpec struct {310	// Source is the base URL releases are served under. A release's311	// artifacts live at <source>/<version>/, starting with312	// release.yaml, the document that names every artifact by digest.313	Source string `json:"source,omitempty"`314315	// Catalog is the set of releases machines may be asked to run.316	// The digest is the sha256 of release.yaml's exact bytes, and it317	// is the root of the trust chain: the API names the document,318	// the document names the artifacts, and the system checks every319	// downloaded byte against one or the other.320	Catalog []ReleaseCatalogEntry `json:"catalog,omitempty"`321}322323// CheckReleasesAnnotation is the annotation that requests an324// immediate channel poll. The fleet observer polls the channel's325// root document on a lazy interval. Setting this annotation on the326// Cluster to any new value, such as a timestamp (the content itself327// means nothing), makes the very next sweep poll immediately. This328// is an annotation, not a spec field, because it asks for one329// action rather than declaring a standing state. kubectl requests a330// Deployment rollout with the restartedAt annotation in the same331// shape.332const CheckReleasesAnnotation = "liken.sh/check-releases"333334// ReleaseCatalogEntry names one release: its version and the digest335// of its release document, as "sha256:<64 hex digits>".336type ReleaseCatalogEntry struct {337	Version string `json:"version"`338	Digest  string `json:"digest"`339}340341// Entry finds one version's catalog entry. It returns nil when the342// catalog does not list the version. The CRD requires spec.version343// to be a catalog member at admission, so this lookup should always344// succeed for the target version. Callers still handle nil, rather345// than rely on that admission rule.346func (s ClusterReleasesSpec) Entry(version string) *ReleaseCatalogEntry {347	for i := range s.Catalog {348		if s.Catalog[i].Version == version {349			return &s.Catalog[i]350		}351	}352	return nil353}354355// NewestVersion is the catalog's highest version. It returns "" when356// the catalog is empty. The fleet sweep publishes this value as357// status.releases.newest, so a printer column can answer "is there358// something newer than the target?" at a glance. The ordering itself359// belongs to the version grammar (api.CompareVersions).360func NewestVersion(catalog []ReleaseCatalogEntry) string {361	newest := ""362	for _, entry := range catalog {363		if newest == "" || api.CompareVersions(entry.Version, newest) > 0 {364			newest = entry.Version365		}366	}367	return newest368}369370// ClusterDisruptionSpec is the machine-level equivalent of a371// workload's PodDisruptionBudget, reduced to the one number that372// matters for a fleet: how many machines may be voluntarily down at373// the same time. It governs only disruptions the cluster chooses,374// such as rolling reboots that apply staged changes. It cannot375// promise anything about machines that fail on their own. The376// rollout does count failed machines against the budget, though, so377// a fleet that is already losing machines pauses its own rollout.378type ClusterDisruptionSpec struct {379	// MaxUnavailable is how many machines may be unavailable at380	// once, planned and unplanned together. Zero means unset, and381	// the system defaults it to 1. One machine at a time is the382	// safest rollout, and the right answer for small fleets. The383	// leaders keep a stricter, automatic floor, regardless of this384	// number: only one leader may ever be down at a time. That floor385	// is not a policy choice. Losing more leaders at once could cost386	// the datastore its majority quorum, and no budget setting can387	// change that arithmetic.388	MaxUnavailable int `json:"maxUnavailable,omitempty"`389}390391// MaxUnavailableOrDefault applies the default: an unset budget is 1.392func (d ClusterDisruptionSpec) MaxUnavailableOrDefault() int {393	if d.MaxUnavailable < 1 {394		return 1395	}396	return d.MaxUnavailable397}398399// ClusterTimeSpec declares where time comes from. Only the leaders400// consult it. Followers sync from the leaders themselves, resolved401// from the fleet's Machine manifests, with the endpoint as the402// fallback. The hierarchy is therefore upstreams, then leaders, then403// everyone else.404type ClusterTimeSpec struct {405	// Upstreams are the NTP servers the cluster's leaders sync from,406	// as hostnames or addresses. There is no default. A distro that407	// shipped pool.ntp.org here would enroll every deployment's408	// machines in a volunteer-run service without asking. An empty409	// list is legal and means the fleet free-runs. The machines stay410	// consistent with each other, but they are correct only if the411	// hardware clocks happen to be correct too.412	Upstreams []string `json:"upstreams,omitempty"`413}414415// ClusterNetworkSpec is the cluster's address plan. Every field is416// optional. Whatever is left unset falls back to k3s's default. The417// value of writing a field here is that every node provably agrees418// on it.419type ClusterNetworkSpec struct {420	// NodeCIDR is the subnet the nodes address each other on: the421	// network all cluster traffic crosses. A machine may have422	// several interfaces, such as an internet uplink and a423	// management port. Kubernetes traffic uses the interface whose424	// address falls inside this subnet, and that address becomes425	// the machine's node IP.426	NodeCIDR string `json:"nodeCIDR,omitempty"`427428	// ClusterCIDR is the range pod addresses are drawn from429	// (k3s default 10.42.0.0/16).430	ClusterCIDR string `json:"clusterCIDR,omitempty"`431432	// ServiceCIDR is the range service addresses are drawn from433	// (k3s default 10.43.0.0/16).434	ServiceCIDR string `json:"serviceCIDR,omitempty"`435436	// ClusterDNS is the service address of the cluster's DNS resolver,437	// which must sit inside ServiceCIDR (k3s default 10.43.0.10).438	ClusterDNS string `json:"clusterDNS,omitempty"`439440	// ClusterDomain is the DNS suffix cluster-internal names live441	// under (k3s default cluster.local).442	ClusterDomain string `json:"clusterDomain,omitempty"`443444	// NodePortCIDRs are the networks a NodePort service answers on.445	// A NodePort is a port opened on the machines themselves, and446	// which of a machine's addresses it answers on is a choice.447	//448	// Left unset, a NodePort answers on the node IP alone: the449	// address on NodeCIDR, where the rest of the cluster expects to450	// find this machine. That is the narrow, correct default for a451	// cluster whose traffic arrives on the node network.452	//453	// A deployment sets this when its traffic arrives somewhere454	// else, which is the ordinary case for a cluster reached over a455	// tunnel or a second segment. Each entry is a subnet, and a456	// NodePort answers on every local address inside any of them.457	// The list replaces the default rather than adding to it, so the458	// document states the whole answer: a list that omits NodeCIDR459	// closes NodePorts on the node network, including for the other460	// machines in the cluster.461	//462	// Every other field here is immutable, because changing one463	// renumbers the cluster. This field is not. It only widens or464	// narrows which local addresses answer, it reverses, and k3s465	// reads it when its process starts, so an edit converges by466	// restarting k3s in place.467	NodePortCIDRs []string `json:"nodePortCIDRs,omitempty"`468}469470// Validate checks the address plan for the errors a person can fix471// in the manifest. The API server enforces the same rules on any472// document applied through it. This check exists for the documents473// that reach a machine another way: init also reads a cluster474// document that a person wrote by hand and carried in on a stick,475// and no API server ever saw that one.476//477// Only the NodePort list is checked here. The other fields are478// addresses that k3s itself refuses when they are malformed, and it479// refuses them before it serves anything.480func (n ClusterNetworkSpec) Validate() error {481	for i, entry := range n.NodePortCIDRs {482		if _, _, err := net.ParseCIDR(entry); err != nil {483			return fmt.Errorf("nodePortCIDRs[%d] %q is not a subnet; write each entry as an address and a prefix length, such as 10.0.0.0/24", i, entry)484		}485	}486	return nil487}488489// NodePortAddresses resolves what kube-proxy should be told about490// which addresses a NodePort answers on. It reports the cluster's491// list when the document names one, and kube-proxy's "primary"492// keyword otherwise, which is that program's own name for the node493// IP alone.494//495// Naming the default rather than leaving the setting out is496// deliberate. kube-proxy's real default is every local address,497// which would include the uplink and whatever interfaces the CNI498// creates later, and it warns about that default at startup.499func (n ClusterNetworkSpec) NodePortAddresses() []string {500	if len(n.NodePortCIDRs) == 0 {501		return []string{"primary"}502	}503	return slices.Clone(n.NodePortCIDRs)504}505506// Role is what a machine should be in this cluster. A nil Cluster507// answers leader. A machine with no cluster manifest is on its own,508// and a machine on its own runs as its own single-node cluster,509// which is liken's default arrangement.510func (c *Cluster) Role(machineName string) api.Role {511	if c == nil || slices.Contains(c.Spec.Leaders, machineName) {512		return api.RoleLeader513	}514	return api.RoleFollower515}516517// ParseCluster reads a Cluster manifest from its bytes, strictly,518// for the same reason Machine parsing is strict. A misspelled field519// should produce an error someone sees, rather than become a setting520// that silently never applies.521func ParseCluster(raw []byte) (*Cluster, error) {522	c := &Cluster{}523	if err := yaml.UnmarshalStrict(raw, c); err != nil {524		return nil, err525	}526	if c.Kind != "Cluster" {527		return nil, fmt.Errorf("expected kind Cluster, got %q", c.Kind)528	}529	if err := validateFeatures(c.Spec.Features); err != nil {530		return nil, err531	}532	if err := validateRegistries(c.Spec.Registries); err != nil {533		return nil, err534	}535	if err := c.Spec.Runtime.Validate(); err != nil {536		return nil, fmt.Errorf("spec.runtime: %w", err)537	}538	if err := c.Spec.Network.Validate(); err != nil {539		return nil, fmt.Errorf("spec.network: %w", err)540	}541	return c, nil542}543544// RuntimeSpec is the k3s runtime section, safe on a nil Cluster. A545// machine on its own, with no cluster document, gets the zero section,546// which imposes nothing, the same as an unset section.547func (c *Cluster) RuntimeSpec() K3sRuntimeSpec {548	if c == nil {549		return K3sRuntimeSpec{}550	}551	return c.Spec.Runtime.K3s552}553554// KubeletSpec is the kubelet section, safe on a nil Cluster, like555// RuntimeSpec above. A machine on its own gets the zero section, which556// names nothing, so the kubelet keeps its own policy.557func (c *Cluster) KubeletSpec() KubeletRuntimeSpec {558	if c == nil {559		return KubeletRuntimeSpec{}560	}561	return c.Spec.Runtime.Kubelet562}563564// ContainerdSpec is the containerd section, safe on a nil Cluster,565// like the two above. A machine on its own gets the zero section,566// which names no level, so containerd keeps its own.567func (c *Cluster) ContainerdSpec() ContainerdRuntimeSpec {568	if c == nil {569		return ContainerdRuntimeSpec{}570	}571	return c.Spec.Runtime.Containerd572}573574// EndpointOrEmpty is the join URL, safe on a nil Cluster like the575// accessors above. An empty answer means a machine alone: it is its576// own cluster and has no endpoint to reach, so nothing about it is577// unreachable.578func (c *Cluster) EndpointOrEmpty() string {579	if c == nil {580		return ""581	}582	return c.Spec.Endpoint583}584585// NodePortAddresses is the address plan's NodePort resolution, safe586// on a nil Cluster. A machine on its own answers NodePorts on its587// node IP, the same narrow default a document leaves unset.588func (c *Cluster) NodePortAddresses() []string {589	if c == nil {590		return ClusterNetworkSpec{}.NodePortAddresses()591	}592	return c.Spec.Network.NodePortAddresses()593}594595// LoadCluster reads the Cluster manifest from a file. A machine with596// no cluster manifest gets nil, which is a valid, single-node597// arrangement (see Role). A manifest that exists but does not parse598// is a configuration error, and the function reports it as one.599func LoadCluster(path string) (*Cluster, error) {600	raw, err := os.ReadFile(path)601	if errors.Is(err, fs.ErrNotExist) {602		return nil, nil603	}604	if err != nil {605		return nil, err606	}607	c, err := ParseCluster(raw)608	if err != nil {609		return nil, fmt.Errorf("%s: %w", path, err)610	}611	return c, nil612}
cluster/features.go 100.0%
1package cluster23// The feature vocabulary: the curated set of optional capabilities a4// cluster may opt into through spec.features (see ClusterSpec).5//6// liken is a minimum viable highly-available cluster, and7// capabilities that people may not need should not run on every8// deployment. But an OS must be able to offer more than its minimum,9// so the Cluster document carries a vocabulary of optional features,10// and this table is that vocabulary. Deployments name features. What11// a feature is made of is liken's concern, recorded here. Everything12// that must agree on the vocabulary agrees by consulting this table.13// Init validates the cluster document against it and renders the k3s14// disable list from it. The operator judges each machine's standing15// against it. A parity test holds the hand-written CRD16// (cluster/manifests/clusters-crd.yaml) to exactly these slugs.17//18// A feature is a slug plus any subset of these contributions:19//20//   - k3s configuration rendered at boot: today, membership in the21//     disable list that init computes into the leader's boot22//     drop-in (init/k3s.go).23//   - Vendored userspace binaries: a top-level domain in this24//     repository, with a pinned VERSION and a fetch.sh that produces25//     sha256-verified static binaries. These ship in every image and26//     stay inert until declared.27//   - Kernel modules: the domain's modules.conf, staged into the28//     image at /etc/liken/features/<slug>/modules.conf and loaded by29//     init only when the feature is declared. That file's presence30//     is also how init detects that the booted image carries the31//     payload, so no feature-to-modules mapping lives in Go.32//   - Workload manifests and OCI images: these ship in the image, and33//     init seeds them into k3s's auto-deploy directory only when the34//     feature is declared.35//   - An init boot hook: per-slug code gated on declaration, for the36//     few features with a boot-time contribution, such as an iSCSI37//     initiator name.38//   - Parameters: the feature's configuration object, empty for39//     every feature today (see FeatureConfig).40//41// Payloads ship in every image because they are small. Enabling a42// feature is then purely a runtime act: one Cluster edit that rolls43// through the fleet as staged changes and granted reboots, never an44// image rebuild. A feature too large to ship unconditionally, such45// as a GPU toolkit, would be the moment to introduce build-time46// conditioning, and not before.4748import (49	"fmt"50	"maps"51	"slices"52	"strconv"53	"strings"54)5556// FeatureKind is how a feature is delivered. Deployments never see57// this distinction. The machinery needs it, because the kind58// determines what enabling a feature actually does.59type FeatureKind string6061const (62	// FeatureBundled is a component the k3s binary already carries:63	// Traefik, the service load balancer, and metrics-server. The64	// image ships it whether or not anyone wants it, so opting in65	// costs nothing at build time. The whole actuation is init66	// leaving the component out of the disable list it renders into67	// the k3s boot drop-in.68	FeatureBundled FeatureKind = "Bundled"6970	// FeatureEmbedded is a controller compiled into the k3s server71	// process itself, rather than a component k3s deploys into the72	// cluster. Two features use this kind: the Helm controller,73	// which watches HelmChart resources and renders them into74	// workloads, and the network policy controller, which turns75	// NetworkPolicy resources into packet filtering, something the76	// flannel CNI cannot do alone. Each costs memory in the k3s77	// process whether or not any of its resources exist. An embedded78	// feature can never appear on the disable list, because that79	// list names only deployable components. Its actuation is a80	// dedicated disable key that init renders into the leader's boot81	// drop-in (init/k3s.go), and init omits that key only when the82	// feature is enabled.83	FeatureEmbedded FeatureKind = "Embedded"8485	// FeatureVendored is a payload this repository vendors as a86	// top-level domain: static binaries, kernel modules, and87	// sometimes a workload manifest. Every image includes the payload.88	// When /etc/liken/features/<slug>/modules.conf exists in the89	// booted image, init detects that the image has the payload. That90	// check matters when an older image boots against a cluster91	// document that declares a feature the image predates: the92	// machine reports the gap, instead of silently lacking the93	// capability.94	FeatureVendored FeatureKind = "Vendored"9596	// FeatureWorkload is a capability that runs entirely as cluster97	// workloads: manifests that ship in the image and seed into k3s's98	// auto-deploy directory when the feature is declared, with no99	// vendored host binaries and no kernel modules. The workload's100	// container images are ordinary registry pulls, not baked101	// payloads, so the feature needs no image-side presence check:102	// an image whose vocabulary contains the slug also carries its103	// manifests, because one build produces both. flux takes this104	// shape.105	FeatureWorkload FeatureKind = "Workload"106)107108// FeatureDefinition names one feature: its slug, which is the key109// deployments write in spec.features, and its kind. Requires names110// the features this one cannot work without. Enabling a feature111// enables everything it requires, so a deployment declares only the112// feature it needs and never has to name the dependency. Params113// names the parameters the feature's configuration accepts; most114// features have none, and their configuration is exactly {}. The CRD115// holds a parameterized feature to these names at admission, the116// parity test holds the CRD to this table, and ValidateParams is the117// same judgment at the file doors. Retraction states what must be118// true before the feature may stop.119type FeatureDefinition struct {120	Slug       string121	Kind       FeatureKind122	Requires   []string123	Params     []string124	Teardown   FeatureTeardown125	Retraction FeatureRetraction126}127128// FeatureRetraction is what must be true before a feature may stop.129//130// A feature's controller programs things outside the cluster's own131// records: netfilter chains, host ports, mounts, and the objects that132// other controllers hold finalizers on. Stopping the controller does133// not undo any of it. These fields are what hold a retraction until134// the cluster is ready for it.135//136// They state conditions, never a list of objects to delete. A list of137// another project's object names goes stale the first time that138// project renames one, and it goes stale silently. Each component139// already carries its own teardown, and the ordering here is what140// lets that teardown run: the component that owns the objects stays141// up until they are gone.142type FeatureRetraction struct {143	// Precondition is what the cluster must no longer hold before144	// this feature may stop. Until it holds, the machine operator145	// keeps the feature enabled and reports what stands in the way,146	// so the retraction waits whole instead of applying halfway147	// (machine-operator/retraction.go).148	Precondition Precondition149150	// HostState says that stopping this feature leaves state that151	// only a boot clears: netfilter chains and ipsets, mounts, live152	// storage sessions, loaded kernel modules. Such a retraction is153	// reboot-class (changes.go). That is not only how the state gets154	// cleared. It is how the machine avoids ever running without the155	// controller while the controller's programming is still in156	// force.157	HostState bool158}159160// Precondition names a fact about the cluster that must be false161// before a feature may stop. The machine operator evaluates these162// against live objects, because none of them can be seen in the163// document alone. This is why the refusal is a status condition and164// not a schema rule: a CRD can only judge what the object itself165// says.166type Precondition string167168const (169	// PreconditionNone is the ordinary case. Most features leave170	// nothing behind that another object depends on.171	PreconditionNone Precondition = ""172173	// NoHelmCharts holds when no HelmChart resource remains. The174	// Helm controller puts a removal finalizer on every HelmChart it175	// manages, so a HelmChart deleted after its controller stops176	// waits on a finalizer that nothing will ever clear, and the177	// release it installed keeps running. Retracting traefik is what178	// ordinarily clears this: once the disable list names traefik,179	// k3s deletes the HelmChart, and the Helm controller, still180	// running, uninstalls the release.181	NoHelmCharts Precondition = "NoHelmCharts"182183	// NoLoadBalancerServices holds when no Service of type184	// LoadBalancer remains. The cloud controller that servicelb runs185	// inside owns the DaemonSets behind those Services and the186	// cleanup finalizer on the Services themselves. Stopping it187	// stops the only thing that can finish deleting them. liken188	// cannot clear this precondition itself, because those Services189	// belong to the deployment, so this retraction waits for a190	// person.191	NoLoadBalancerServices Precondition = "NoLoadBalancerServices"192)193194// FeatureTeardown is who removes a retracted feature's objects from195// the cluster.196type FeatureTeardown string197198const (199	// TeardownAddon is the default: retraction removes the feature's200	// seeded manifests while k3s runs, and k3s deletes the addon's201	// objects itself. This is the right shape for a feature whose202	// objects are inert data and simple workloads.203	TeardownAddon FeatureTeardown = ""204205	// TeardownJanitor means k3s must never delete this feature's206	// objects: retraction removes the seeded manifests only while207	// k3s is down, and the cluster operator's janitor tears the208	// objects down in a deliberate order. flux needs this because209	// its objects are not inert: deleting the Kustomization while210	// its controller still runs triggers the engine's own garbage211	// collection, which would prune everything the repository ever212	// applied, the fleet's own documents included. The janitor kills213	// the controllers first, so that finalizer can never fire.214	TeardownJanitor FeatureTeardown = "Janitor"215)216217// Features is the vocabulary, listed in the order that explains it218// best. The bundled components' slugs are exactly k3s's names for219// them, the same words the disable list uses. A vendored feature's220// slug names the capability (iscsi), never the project that221// implements it (open-iscsi), because an implementation can change222// and an API should not have to change with it.223var Features = []FeatureDefinition{224	// traefik states no precondition of its own. Stopping it is what225	// deletes its HelmCharts, through k3s's disable path, and that226	// deletion is what clears helm's precondition below. So an edit227	// that drops both stops traefik first and helm one convergence228	// later.229	{Slug: "traefik", Kind: FeatureBundled, Requires: []string{"helm"}},230	{Slug: "servicelb", Kind: FeatureBundled,231		Retraction: FeatureRetraction{Precondition: NoLoadBalancerServices}},232	{Slug: "metrics-server", Kind: FeatureBundled},233	{Slug: "helm", Kind: FeatureEmbedded,234		Retraction: FeatureRetraction{Precondition: NoHelmCharts}},235	// The network policy controller programs the host's netfilter236	// tables. Its chains are keyed on pod addresses, and an address237	// outlives the pod that held it, so a chain left behind does not238	// fail safe. It enforces a policy nobody declared against239	// whatever workload receives that address next.240	{Slug: "network-policy", Kind: FeatureEmbedded,241		Retraction: FeatureRetraction{HostState: true}},242	// A live iSCSI session lives in the kernel, and iscsid recovers243	// it. Stopping the feature takes the daemon away and leaves the244	// session logged in with nothing to reconnect it.245	{Slug: "iscsi", Kind: FeatureVendored,246		Retraction: FeatureRetraction{HostState: true}},247	// The nfs feature is the loaded module. Nothing else gates an248	// NFS mount, because mount.nfs ships in every image, so a249	// retraction that does not reach a boot changes nothing at all.250	{Slug: "nfs", Kind: FeatureVendored,251		Retraction: FeatureRetraction{HostState: true}},252	// flux names the project, not the capability, and this is a253	// deliberate exception to the naming rule above. The rule exists254	// so an implementation can change behind a stable name, and for255	// iscsi that holds: the kernel interface is the capability, and256	// open-iscsi could be swapped behind it. A GitOps engine is257	// different. Its in-cluster resources and its repository258	// conventions are the interface the deployment builds its whole259	// repository against, so a generic gitops slug would promise a260	// swappability the design could never honor. A deployment that261	// needs a different engine needs a different feature.262	{Slug: "flux", Kind: FeatureWorkload, Teardown: TeardownJanitor,263		Params: []string{"repository", "path", "branch", "knownHosts", "prune"}},264}265266// FeatureConfig is one feature's configuration: the object under its267// slug in spec.features. {} is every feature's zero configuration,268// and a parameterized feature's parameters are its keys. The type is269// a plain map on purpose, never a struct of named fields. Each270// machine's binary carries only the parameter vocabulary its image271// was built with, and a fleet mid-upgrade holds several of those272// vocabularies at once. A struct parsed strictly would refuse a273// document from a newer vocabulary, and then a downgraded machine274// could not read its own proven document, could not derive its role,275// and would sit Blocked. So the file doors accept any parameters,276// and the judgment happens where a verdict can be reported: the CRD277// refuses a parameter its vocabulary does not list at admission, and278// init's feature pass reports a parameter this image cannot honor279// (ValidateParams below), leaving the machine degraded rather than280// down.281type FeatureConfig map[string]any282283// ValidateParams holds one feature's configuration to this284// definition's parameter vocabulary: every key must be a declared285// parameter, and every value must be a string, the one value type286// the vocabulary uses. A nil configuration validates like {}; the287// null refusal belongs to validateFeatures, at parse time. The error288// names both possible causes, because an unknown parameter reads the289// same from here whether a newer vocabulary defined it or a290// hand-written seed misspelled it.291func (def *FeatureDefinition) ValidateParams(cfg *FeatureConfig) error {292	if cfg == nil {293		return nil294	}295	for _, key := range slices.Sorted(maps.Keys(*cfg)) {296		if !slices.Contains(def.Params, key) {297			offers := "no parameters"298			if len(def.Params) > 0 {299				offers = strings.Join(def.Params, ", ")300			}301			return fmt.Errorf(302				"%s: this image's vocabulary has no %q parameter; upgrade to a release that carries it, or fix the name if it is a misspelling (this image offers: %s)",303				def.Slug, key, offers)304		}305		if _, ok := (*cfg)[key].(string); !ok {306			return fmt.Errorf("%s: %s must be a string", def.Slug, key)307		}308	}309	return nil310}311312// FeatureFlux is the GitOps feature's slug: the declaration that the313// fleet's workloads and configuration sync from a git repository314// through Flux.315const FeatureFlux = "flux"316317// The sync defaults: the repository's root, on the branch most318// forges create by default, with pruning on. A deployment that keeps319// several clusters in one repository sets path to its cluster's320// directory instead.321//322// Pruning defaults to on because a cluster liken founds starts empty.323// Every object in it came from the repository, so an object the324// repository no longer produces has no other author, and deleting it325// keeps the cluster equal to the repository.326const (327	FluxDefaultPath   = "."328	FluxDefaultBranch = "main"329	FluxDefaultPrune  = true330)331332// FluxConfig is the flux feature's configuration, typed: where the333// fleet's declared state lives, and which part of it this cluster334// syncs. KnownHosts is the forge's SSH host keys in known_hosts335// form, one line per key. It belongs in the spec because a host key is336// public material: it is the forge's identity, not a secret, and337// declaring it here is what lets the first clone verify the forge338// without anyone answering a trust prompt. The cluster operator339// keeps the flux-system Secret's known_hosts entry synchronized340// with it.341//342// Prune says whether the sync deletes an object that the repository343// no longer produces. It belongs in the document because init cannot344// work the answer out: seedFluxSync renders a file and holds no API345// client, so it can never read what the repository contains. Only the346// person who wrote the repository knows whether it carries its own347// copy of the flux-system Kustomization. If it does, that348// Kustomization appears in its own inventory, a build that stops349// producing it marks it for deletion, and the deletion's finalizer350// then removes everything the repository ever applied. Such a351// repository declares prune: "false".352type FluxConfig struct {353	Repository string354	Path       string355	Branch     string356	KnownHosts string357	Prune      bool358}359360// FluxConfig reads the flux feature's declaration, applies the sync361// defaults, and requires the one parameter that has no default: the362// repository. It returns nil with no error when the cluster does not363// declare the feature; that is the ordinary state, not a mistake. An364// empty string counts as unset, so `path: ""` gets the default365// rather than an impossible sync path.366func (c *Cluster) FluxConfig() (*FluxConfig, error) {367	if c == nil {368		return nil, nil369	}370	cfg, declared := c.Spec.Features[FeatureFlux]371	if !declared {372		return nil, nil373	}374	def := FeatureBySlug(FeatureFlux)375	if err := def.ValidateParams(cfg); err != nil {376		return nil, err377	}378	out := &FluxConfig{379		Path:   FluxDefaultPath,380		Branch: FluxDefaultBranch,381		Prune:  FluxDefaultPrune,382	}383	if cfg != nil {384		if s, _ := (*cfg)["repository"].(string); s != "" {385			out.Repository = s386		}387		if s, _ := (*cfg)["path"].(string); s != "" {388			out.Path = s389		}390		if s, _ := (*cfg)["branch"].(string); s != "" {391			out.Branch = s392		}393		if s, _ := (*cfg)["knownHosts"].(string); s != "" {394			out.KnownHosts = s395		}396		// Every parameter's value is a string, so a boolean arrives397		// spelled out and this is where it becomes one.398		if s, _ := (*cfg)["prune"].(string); s != "" {399			prune, err := strconv.ParseBool(s)400			if err != nil {401				return nil, fmt.Errorf(402					"flux: prune must be \"true\" or \"false\", not %q: write prune: \"false\" when the repository carries its own flux-system Kustomization",403					s)404			}405			out.Prune = prune406		}407	}408	if out.Repository == "" {409		return nil, fmt.Errorf(410			"flux: repository is required: the git URL the fleet's declared state syncs from, for example ssh://git@forge.example/fleet.git")411	}412	return out, nil413}414415// FeatureBySlug finds one feature's definition. It returns nil when416// the vocabulary does not include the slug.417func FeatureBySlug(slug string) *FeatureDefinition {418	for i := range Features {419		if Features[i].Slug == slug {420			return &Features[i]421		}422	}423	return nil424}425426// FeatureSlugs returns every slug in table order, for error messages427// and the CRD parity test.428func FeatureSlugs() []string {429	slugs := make([]string, len(Features))430	for i, f := range Features {431		slugs[i] = f.Slug432	}433	return slugs434}435436// EnabledFeatures returns the slugs this cluster's declarations437// enable, sorted. This is the declared opt-ins plus everything they438// require. For example, traefik pulls in helm, because k3s deploys439// Traefik through a HelmChart resource that only the Helm controller440// can render. The closure runs to a fixed point, so a requirement's441// own requirements are honored too. A nil Cluster, a machine with no442// cluster document, enables nothing. This is how the minimum viable443// cluster stays the default.444func (c *Cluster) EnabledFeatures() []string {445	if c == nil {446		return nil447	}448	enabled := map[string]bool{}449	for slug := range c.Spec.Features {450		enabled[slug] = true451	}452	for changed := true; changed; {453		changed = false454		for _, f := range Features {455			if !enabled[f.Slug] {456				continue457			}458			for _, req := range f.Requires {459				if !enabled[req] {460					enabled[req] = true461					changed = true462				}463			}464		}465	}466	return slices.Sorted(maps.Keys(enabled))467}468469// FeatureEnabled reports whether one feature is on for this cluster,470// declared or required by a declared one.471func (c *Cluster) FeatureEnabled(slug string) bool {472	return slices.Contains(c.EnabledFeatures(), slug)473}474475// RetractedFeatures returns the features that stop between two476// declarations, sorted. It compares enabled sets rather than declared477// ones, because a feature can leave without anyone naming it. Nothing478// but traefik requires helm, so an edit that removes traefik stops479// helm as well, and that unnamed stop is the one with the sharpest480// consequences: it is what strands a Helm release.481//482// The classifier calls this to pick a change's tier (changes.go), and483// the machine operator calls it to work out which preconditions an484// edit must satisfy before a machine may act on it.485func RetractedFeatures(current, desired *Cluster) []string {486	running := map[string]bool{}487	for _, slug := range current.EnabledFeatures() {488		running[slug] = true489	}490	for _, slug := range desired.EnabledFeatures() {491		delete(running, slug)492	}493	return slices.Sorted(maps.Keys(running))494}495496// RetractionLeavesHostState reports whether any feature stopping here497// leaves state that only a boot clears. Such a change takes the498// reboot tier, so the machine never runs without the controller while499// the controller's programming is still in force.500func RetractionLeavesHostState(current, desired *Cluster) bool {501	for _, slug := range RetractedFeatures(current, desired) {502		if def := FeatureBySlug(slug); def != nil && def.Retraction.HostState {503			return true504		}505	}506	return false507}508509// DisabledComponents computes the k3s disable list: every bundled510// component minus this cluster's opt-ins, sorted. It is always the511// complete list, never a fragment for k3s to merge with a default512// defined somewhere else. The rendered value therefore has exactly513// one author: init, which writes it into the boot drop-in on leaders514// (init/k3s.go). A nil Cluster disables everything bundled, which515// keeps the same default for a machine on its own.516func (c *Cluster) DisabledComponents() []string {517	var disabled []string518	for _, f := range Features {519		if f.Kind != FeatureBundled {520			continue521		}522		if c != nil {523			if _, enabled := c.Spec.Features[f.Slug]; enabled {524				continue525			}526		}527		disabled = append(disabled, f.Slug)528	}529	slices.Sort(disabled)530	return disabled531}532533// validateFeatures holds spec.features to its shape. ParseCluster534// calls it, so every file door, such as init vetting a staged or535// proven document, or the operator hashing a rendered one, refuses a536// null the same way the CRD refuses it at admission. Each refusal537// carries a message that says what to write instead.538//539// An unknown slug is deliberately not an error here, though the CRD540// refuses one at admission. The difference is the vocabulary each541// door holds. A fleet has exactly one vocabulary at its API, the542// newest image's CRD. But each machine's parser carries only the543// vocabulary its own image was built with, and a fleet mid-upgrade544// holds several of those vocabularies at once. A document that545// declares a feature this binary predates must still parse.546// Otherwise the machine could not read its own proven document after547// a downgrade, could not derive its role, and would sit Blocked on a548// document the rest of the fleet is running without trouble. The549// feature pass reports the unknown slug instead: FeaturesReady goes550// False, naming the slug and this image's vocabulary. This message551// covers both real causes, an image that predates the feature and a552// misspelling in a hand-written seed, and it leaves the machine553// degraded rather than down.554func validateFeatures(features map[string]*FeatureConfig) error {555	for _, slug := range slices.Sorted(maps.Keys(features)) {556		if features[slug] == nil {557			return fmt.Errorf("spec.features: %s is null, which in Kubernetes means unset; presence is the opt-in, so write %q to enable the feature or remove the key entirely",558				slug, slug+": {}")559		}560	}561	return nil562}
cluster/registries.go 100.0%
1package cluster23// Private registries, the cluster's half: spec.registries on the4// Cluster (RegistriesSpec) declares mirror endpoints that containerd5// pulls through, and k3s's embedded peer-to-peer registry. These are6// cluster facts, because any node may be asked to pull any image,7// and they converge inside the canonical cluster document like every8// other fleet-wide fact.9//10// Credentials are deliberately not declared here. A spec is public:11// anyone who can get the Cluster can read every field. Credentials12// enter through a Kubernetes Secret instead, which the machine13// operator renders into a per-machine credentials document that14// follows the staged/proven lifecycle (machine/registries.go15// explains it).1617import (18	"fmt"19	"maps"20	"net/url"21	"slices"22)2324// RegistriesSpec is the Cluster's declaration of how images arrive.25// The zero value means nothing is declared: no mirrors, no embedded26// registry, and no registries.yaml written at all.27type RegistriesSpec struct {28	// Mirrors maps a registry host, as an image reference names it29	// (docker.io, registry.example:5000), to the endpoint URLs that30	// containerd should try, in preference order, before it falls31	// back to the registry itself.32	Mirrors map[string][]string `json:"mirrors,omitempty"`3334	// Embedded turns on k3s's embedded registry mirror (Spegel).35	// Every node serves the images it already holds to its peers,36	// which keeps a fleet on a slow uplink from pulling the same37	// bytes once per machine.38	Embedded bool `json:"embedded,omitempty"`39}4041// validateRegistries holds spec.registries to its shape. The spec42// itself may be null or absent. Both decode to the zero struct and43// genuinely mean "no registries configuration", unlike44// spec.features, where a null could be mistaken for an opt-in. But45// the function refuses a mirror host with a null or empty endpoint46// list. A host with no endpoints mirrors nothing, and it is not the47// same as declaring no mirror at all. A bare `docker.io:` in48// hand-written YAML is a mistake, so the function names it rather49// than guess what the author meant.50func validateRegistries(r RegistriesSpec) error {51	for _, host := range slices.Sorted(maps.Keys(r.Mirrors)) {52		if host == "" {53			return fmt.Errorf("spec.registries.mirrors: a mirror's key must be the registry host it stands in for (docker.io, registry.example:5000)")54		}55		endpoints := r.Mirrors[host]56		if len(endpoints) == 0 {57			return fmt.Errorf("spec.registries.mirrors: %s lists no endpoints; list at least one endpoint URL, or remove the host entirely", host)58		}59		for _, endpoint := range endpoints {60			u, err := url.Parse(endpoint)61			if err != nil || u.Host == "" || (u.Scheme != "http" && u.Scheme != "https") {62				return fmt.Errorf("spec.registries.mirrors: %s endpoint %q must be an http:// or https:// URL", host, endpoint)63			}64		}65	}66	return nil67}
cluster/runtime.go 98.9%
1package cluster23// The runtime discipline the cluster imposes on the k3s process, on4// the components inside it, and on the container runtime beside it.5//6// liken supervises one long-lived process: k3s. This section is the7// operator's control over how that process runs, and it has one8// subsection per thing that reads a setting. The k3s subsection is the9// discipline of the process itself. The kubelet subsection is the10// configuration of the kubelet, which is a component compiled into11// that same process rather than a program of its own. The containerd12// subsection is the configuration of containerd, which is a separate13// program that k3s starts. The split follows the reader, so a path14// names the thing that acts on the value:15// spec.runtime.k3s.goMemoryLimit reads as "the runtime memory limit of16// the k3s process", spec.runtime.kubelet.imageGC.maximumAge reads as17// "the age ceiling of the kubelet's image collector", and18// spec.runtime.containerd.logLevel reads as "the log level of19// containerd".20//21// Every setting here is an opt-in. An unset field imposes nothing, so22// the reader keeps its own default: Go's runtime defaults for the k3s23// subsection, the kubelet's own image collection policy for the24// kubelet subsection, and containerd's own level for the containerd25// subsection. This matters for an upgrade. A cluster that names26// nothing renders the same bytes it rendered before, so no machine27// restarts k3s to gain a section it never asked for.28//29// Every setting here is also read only when the k3s process starts, so30// an edit converges by restarting k3s in place, the same tier as a31// features edit. changes.go classifies the whole section that way.32// containerd is included in that rule, because k3s starts containerd33// and stops it again, so a k3s restart is also a containerd restart.3435import (36	"fmt"37	"slices"38	"strconv"39	"strings"40	"time"41)4243// ClusterRuntimeSpec is the runtime discipline section of a44// ClusterSpec. It holds one subsection per thing that reads a45// setting: the k3s process, the kubelet inside it, and containerd46// beside it.47type ClusterRuntimeSpec struct {48	// K3s is the runtime discipline of the k3s process init launches:49	// the Go environment it runs under, and how much it prints.50	K3s K3sRuntimeSpec `json:"k3s,omitzero"`5152	// Kubelet is the configuration of the kubelet component that runs53	// inside the k3s process, on every machine of the cluster.54	Kubelet KubeletRuntimeSpec `json:"kubelet,omitzero"`5556	// Containerd is the configuration of containerd, the container57	// runtime that k3s starts on every machine of the cluster.58	Containerd ContainerdRuntimeSpec `json:"containerd,omitzero"`59}6061// K3sRuntimeSpec is the runtime discipline of the k3s process itself:62// the Go environment it runs under, and how much it prints. Every63// field is read only when k3s starts, so an edit converges by64// restarting k3s in place, the same tier as a features edit. An unset65// field imposes nothing, so k3s keeps its own default for it.66//67// The two Go values shape only the environment init hands the k3s68// process, where k3s keeps Go's own defaults: no memory ceiling, and a69// heap that grows to twice its live data before the collector runs.70// That trade is right on a machine with memory to spare. It is worth71// tuning on the small machines liken targets, where k3s is the72// dominant resident process, and every uncollected megabyte takes73// memory from the workloads. containerd and the shims k3s starts74// inherit that environment, because k3s is their parent. No other75// process reads it: not init, not the operators, not the workloads,76// which get their environment from their own pod specs.77//78// Debug is not part of that environment. It is a k3s setting of its79// own, and it reaches only the components compiled into the k3s80// process. containerd has a level of its own, in the containerd81// subsection.82type K3sRuntimeSpec struct {83	// GoMemoryLimit is the soft ceiling on everything the k3s84	// runtime manages: heap, stacks, and its own metadata (Go's85	// GOMEMLIMIT). It accepts three forms. "off" removes the ceiling.86	// A percent such as "25%" is that share of this machine's memory,87	// so one setting scales across a fleet of different sizes. An88	// absolute quantity such as "448Mi" is a fixed ceiling on every89	// machine. Left unset, k3s runs with no ceiling, the same as "off".90	GoMemoryLimit string `json:"goMemoryLimit,omitempty"`9192	// GoGC is the collector's everyday pace, as a percent of heap93	// growth between collections (Go's GOGC). Left unset, init sets no94	// GOGC, so k3s keeps Go's own pace of one hundred percent. It is a95	// pointer so an explicit value is told apart from unset, and the96	// file doors refuse a value below 1.97	GoGC *int `json:"goGC,omitempty"`9899	// Debug raises the k3s process to debug logging, the same thing100	// k3s's own --debug flag does. It reaches every Kubernetes101	// component compiled into the process: the API server, the102	// scheduler, the controllers, and the kubelet. Left unset, k3s103	// logs at info. Turn it on to read a decision that the info lines104	// do not explain, and turn it off again, because debug multiplies105	// the volume of a stream that a small machine has to store and106	// ship off itself.107	Debug bool `json:"debug,omitempty"`108}109110// GoGCPercent resolves the collector pace. It reports the set value and111// true, or zero and false when the cluster names none, so a caller can112// tell an explicit pace from Go's own default.113func (k K3sRuntimeSpec) GoGCPercent() (int, bool) {114	if k.GoGC == nil {115		return 0, false116	}117	return *k.GoGC, true118}119120// GoMemoryLimitBytes resolves the memory ceiling against this machine's121// memory. It returns the ceiling in bytes, whether the ceiling is off,122// and any error in the setting. An unset limit is no ceiling, the same123// as "off", so k3s runs on Go's own default. "off" returns off true and124// no ceiling.125func (k K3sRuntimeSpec) GoMemoryLimitBytes(memoryBytes uint64) (uint64, bool, error) {126	s := strings.TrimSpace(k.GoMemoryLimit)127	switch {128	case s == "" || s == "off":129		return 0, true, nil130	case strings.HasSuffix(s, "%"):131		pct, err := parseMemoryPercent(s)132		if err != nil {133			return 0, false, err134		}135		return memoryBytes * pct / 100, false, nil136	default:137		bytes, err := parseBinaryQuantity(s)138		return bytes, false, err139	}140}141142// parseMemoryPercent reads a whole-number percent in the range143// (0, 100]. Zero would ask for no memory at all, and a value above144// 100 would ask for more than the machine has.145func parseMemoryPercent(s string) (uint64, error) {146	n, err := strconv.ParseUint(strings.TrimSuffix(s, "%"), 10, 64)147	if err != nil {148		return 0, fmt.Errorf("goMemoryLimit %q: a percent must be a whole number, like %q", s, "25%")149	}150	if n < 1 || n > 100 {151		return 0, fmt.Errorf("goMemoryLimit %q: a percent must be between 1%% and 100%%", s)152	}153	return n, nil154}155156// parseBinaryQuantity reads an absolute ceiling: a plain byte count,157// or a Ki/Mi/Gi/Ti quantity. It accepts only the power-of-two158// suffixes, the same units the storage math uses, because mixing "2G"159// (decimal) with "2Gi" (binary) would invite a silent seven-percent160// mistake. A zero ceiling is always an error, not a quiet "off".161func parseBinaryQuantity(s string) (uint64, error) {162	units := []struct {163		suffix string164		factor uint64165	}{166		{"Ki", 1 << 10},167		{"Mi", 1 << 20},168		{"Gi", 1 << 30},169		{"Ti", 1 << 40},170	}171	digits := s172	var unit uint64 = 1173	for _, u := range units {174		if rest, ok := strings.CutSuffix(s, u.suffix); ok {175			digits, unit = rest, u.factor176			break177		}178	}179	n, err := strconv.ParseUint(digits, 10, 64)180	if err != nil {181		return 0, fmt.Errorf("goMemoryLimit %q: expected \"off\", a percent like %q, or a quantity like %q", s, "25%", "448Mi")182	}183	if n == 0 {184		return 0, fmt.Errorf("goMemoryLimit %q: a memory ceiling can't be zero; write \"off\" to remove the ceiling", s)185	}186	// The multiply must not wrap. An absurd count times a large unit187	// would otherwise wrap around to a small ceiling and starve k3s188	// silently, which is the worst possible reading of a typo.189	if n > ^uint64(0)/unit {190		return 0, fmt.Errorf("goMemoryLimit %q: the quantity is too large to be a memory size", s)191	}192	return n * unit, nil193}194195// KubeletRuntimeSpec is the configuration of the kubelet that runs196// inside the k3s process. The kubelet reads its configuration once,197// when the process starts, so an edit converges by restarting k3s in198// place.199type KubeletRuntimeSpec struct {200	// ImageGC is the policy for the kubelet's image collector, which201	// prunes containerd's image store.202	ImageGC ImageGCSpec `json:"imageGC,omitzero"`203}204205// ImageGCSpec is the cluster's image collection policy. containerd's206// image store grows with every image a node pulls, and nothing in the207// store is removed while a container uses it. The kubelet is the only208// thing that prunes it, and it prunes on two triggers: the disk209// thresholds, and the age of an unused image.210//211// The thresholds are percentages of the filesystem that holds the212// store, which on liken is clusterState. The kubelet measures that213// filesystem every five minutes. When usage passes the high threshold,214// it removes unused images, oldest first, until usage falls under the215// low threshold. A percentage rather than a byte count is what lets216// one setting serve a fleet of different disk sizes.217//218// The ages are the other trigger, and they work on one image at a219// time. An image unused for longer than MaximumAge is removed even220// when the disk is nearly empty, which keeps a store from carrying221// years of tags no workload names any more. An image unused for less222// than MinimumAge is kept even when the disk is full, which stops a223// node from re-pulling an image it just stopped using.224//225// The thresholds are pointers so that an explicit value is told apart226// from unset. An unset field leaves the kubelet's own default in227// place, and a section that names no field at all renders no228// configuration file, so the kubelet keeps its whole default policy.229type ImageGCSpec struct {230	// HighThresholdPercent is the disk usage that starts collection,231	// as a percent of the filesystem that holds containerd's image232	// store. The kubelet's own default is 85.233	HighThresholdPercent *int `json:"highThresholdPercent,omitempty"`234235	// LowThresholdPercent is the disk usage that stops collection,236	// as a percent of the same filesystem. It must be below237	// HighThresholdPercent. The kubelet's own default is 80.238	LowThresholdPercent *int `json:"lowThresholdPercent,omitempty"`239240	// MaximumAge is how long an unused image may stay in the store241	// before the kubelet removes it, regardless of disk usage, as a Go242	// duration such as "168h". The kubelet does no age check by243	// default.244	MaximumAge string `json:"maximumAge,omitempty"`245246	// MinimumAge is how long an unused image is kept before the247	// kubelet may remove it, as a Go duration such as "5m". The248	// kubelet's own default is two minutes.249	MinimumAge string `json:"minimumAge,omitempty"`250}251252// The kubelet's own image collection defaults, which hold for every253// field the cluster leaves unset. They are named here because the254// validation needs them: a document that sets one threshold and not255// the other is still a pair, and the pair has to make sense against256// the value the kubelet will supply for the missing half.257const (258	defaultImageGCHighThresholdPercent = 85259	defaultImageGCLowThresholdPercent  = 80260	defaultImageMinimumGCAge           = 2 * time.Minute261)262263// Validate holds the image collection policy to what the kubelet264// accepts when it starts. The kubelet refuses a threshold outside 0 to265// 100, a low threshold at or above the high one, and a maximum age at266// or below the minimum. It exits rather than starting without them, so267// a machine that boots a rejected configuration runs no control plane268// and serves no shell to repair it. This door is therefore at least as269// strict as the kubelet's own.270func (g ImageGCSpec) Validate() error {271	// A field the document leaves unset still takes part in the272	// comparisons below, because the kubelet supplies its own default273	// for the missing half and then compares the pair. A lone274	// highThresholdPercent of 70 sits under the default low of 80, and275	// the kubelet refuses that pair. Naming where each value came from276	// is what makes the error tell an operator which field to edit.277	const fromDefault = " (the kubelet's default)"278	high, highNote := defaultImageGCHighThresholdPercent, fromDefault279	if g.HighThresholdPercent != nil {280		high, highNote = *g.HighThresholdPercent, ""281		if err := validateThresholdPercent("highThresholdPercent", high); err != nil {282			return err283		}284	}285	low, lowNote := defaultImageGCLowThresholdPercent, fromDefault286	if g.LowThresholdPercent != nil {287		low, lowNote = *g.LowThresholdPercent, ""288		if err := validateThresholdPercent("lowThresholdPercent", low); err != nil {289			return err290		}291	}292	if low >= high {293		return fmt.Errorf("lowThresholdPercent %d%s must be below highThresholdPercent %d%s, so collection has a range to work in", low, lowNote, high, highNote)294	}295296	minimum, minimumNote := defaultImageMinimumGCAge, fromDefault297	if g.MinimumAge != "" {298		parsed, err := parseImageAge("minimumAge", g.MinimumAge)299		if err != nil {300			return err301		}302		minimum, minimumNote = parsed, ""303	}304	if g.MaximumAge != "" {305		maximum, err := parseImageAge("maximumAge", g.MaximumAge)306		if err != nil {307			return err308		}309		if maximum <= minimum {310			return fmt.Errorf("maximumAge %q must be greater than minimumAge %s%s, so an unused image has a span in which it is kept", g.MaximumAge, minimum, minimumNote)311		}312	}313	return nil314}315316// validateThresholdPercent bounds one threshold to 1 through 100. Zero317// would ask the kubelet to collect against a disk that is never empty318// enough, and a value above 100 names a fullness a filesystem cannot319// reach, so collection would never start.320func validateThresholdPercent(field string, value int) error {321	if value < 1 || value > 100 {322		return fmt.Errorf("%s %d: a threshold is a percent of the filesystem that holds the image store, so it must be between 1 and 100", field, value)323	}324	return nil325}326327// parseImageAge reads one age as a Go duration, the same grammar the328// kubelet's own configuration file uses. A zero or negative age is329// refused here rather than passed on, because the kubelet reads zero330// as "no age check" and would then run a policy the document did not331// ask for.332func parseImageAge(field, value string) (time.Duration, error) {333	age, err := time.ParseDuration(value)334	if err != nil {335		return 0, fmt.Errorf("%s %q: expected a Go duration, like %q or %q", field, value, "30m", "168h")336	}337	if age <= 0 {338		return 0, fmt.Errorf("%s %q: an age must be greater than zero", field, value)339	}340	return age, nil341}342343// Validate checks the kubelet section.344func (k KubeletRuntimeSpec) Validate() error {345	if err := k.ImageGC.Validate(); err != nil {346		return fmt.Errorf("imageGC: %w", err)347	}348	return nil349}350351// ContainerdRuntimeSpec is the configuration of containerd, the352// container runtime that runs beside the k3s process. containerd is a353// program of its own that k3s starts, not a component compiled into354// k3s, so it keeps its own configuration file and its own log level.355// Nothing in the k3s subsection reaches it.356//357// containerd is the loudest writer on a liken machine. At info it358// prints a line for each step of every pod's life, on every node, for359// as long as the node runs. That detail is worth having while a360// cluster is new and worth turning down once it is not, and a machine361// serves no shell in which to turn it down by hand.362type ContainerdRuntimeSpec struct {363	// LogLevel is how much containerd prints. Left unset, containerd364	// keeps its own default of info. Set "warn" to keep the failures365	// and drop the pod lifecycle lines.366	LogLevel string `json:"logLevel,omitempty"`367}368369// containerdLogLevels are the levels a cluster may name, from loudest370// to quietest. containerd itself takes three more (trace above debug,371// and fatal and panic below error), which liken does not offer. trace372// is a volume no machine should write to the disk it also runs on, and373// fatal and panic drop the error lines written before a crash, the374// lines that explain it.375var containerdLogLevels = []string{"debug", "info", "warn", "error"}376377// Validate holds the containerd section to the levels above. containerd378// exits when it reads a level it does not know, and k3s does not start379// without containerd, so a machine that boots a rejected level runs no380// workloads and serves no shell to repair it.381func (c ContainerdRuntimeSpec) Validate() error {382	if c.LogLevel == "" || slices.Contains(containerdLogLevels, c.LogLevel) {383		return nil384	}385	return fmt.Errorf("logLevel %q: expected one of %s", c.LogLevel, strings.Join(containerdLogLevels, ", "))386}387388// Validate holds the runtime section to its shape, so every file door389// refuses garbage the same way the CRD refuses it at admission. Each390// subsection names itself in the error, so a message reads as the path391// of the field that is wrong.392func (r ClusterRuntimeSpec) Validate() error {393	if err := r.K3s.Validate(); err != nil {394		return fmt.Errorf("k3s: %w", err)395	}396	if err := r.Kubelet.Validate(); err != nil {397		return fmt.Errorf("kubelet: %w", err)398	}399	if err := r.Containerd.Validate(); err != nil {400		return fmt.Errorf("containerd: %w", err)401	}402	return nil403}404405// Validate checks one k3s runtime section.406func (k K3sRuntimeSpec) Validate() error {407	if _, _, err := k.GoMemoryLimitBytes(1 << 30); err != nil {408		return err409	}410	if k.GoGC != nil && *k.GoGC < 1 {411		return fmt.Errorf("goGC %d: GOGC must be at least 1; a smaller value makes the collector run continuously and starve the process of CPU", *k.GoGC)412	}413	return nil414}
disks/fat32.go 98.5%
1package disks23// This file implements a FAT32 formatter, built from the4// specification.5//6// The system slots need FAT because the firmware reads them first.7// UEFI promises to understand exactly one filesystem, and this is8// it. Like the GPT (gpt.go), FAT32 is a small, fixed binary format,9// simple enough to write directly instead of shelling out to a tool.10// The whole filesystem is three structures:11//12//   - A boot sector that describes the geometry: how big a sector13//     is, how many sectors make a cluster (the allocation unit), and14//     where the tables live. Everything else is computed from these15//     numbers, so every reader (firmware, the kernel, this code)16//     derives the same layout from the same 90 bytes.17//18//   - The file allocation table itself, stored twice. The FAT is the19//     filesystem's entire record of allocation. It has one 32-bit20//     entry for each cluster. Each entry holds the number of the21//     next cluster in its file, forming a linked list stored in a22//     table, or holds an end-of-chain mark. The second copy is FAT's23//     only durability measure. There is no journal, only a spare24//     copy of the table.25//26//   - A root directory, which under FAT32 is an ordinary cluster27//     chain like any file's. By universal convention, it starts at28//     cluster 2. Clusters 0 and 1 do not exist. Their table entries29//     are repurposed as signature and health flags.30//31// What FAT lacks is also worth noting: no permissions, no owners, no32// symlinks, no journal. It is a format for data exchange between33// systems, which is why firmware standardized on it. This is also34// why the slots hold nothing but the OS artifacts themselves. Their35// integrity is proven by a digest, not trusted to the filesystem.3637import (38	"encoding/binary"39	"fmt"40	"io"41	"os"42)4344const (45	// fat32ReservedSectors is the number of sectors that precede the46	// first FAT. These sectors hold the boot sector, FSInfo, their47	// backups at sectors 6 and 7, and padding. 32 is the value that48	// every other formatter writes, and matching this convention49	// costs nothing.50	fat32ReservedSectors = 325152	fat32NumFATs = 25354	// fat32MinClusters is the cluster count below which a volume is55	// legally FAT16. Cluster count alone determines the FAT type.56	// Nothing in the header declares the type directly. A volume57	// below this count, even if formatted with FAT32 structures,58	// would be misread as FAT16 by any reader that follows the59	// specification.60	fat32MinClusters = 65_52561)6263// fat32SectorsPerCluster picks the allocation unit from the volume's64// size, following the table in Microsoft's specification. The first65// line matters most. At one sector per cluster, a volume barely past66// the 65,525-cluster minimum fits in about 33 MiB. This is what lets67// small partitions, such as the GRUB boot home, be FAT32 at68// all. Above 260 MB, the specification increases the cluster size,69// to keep the allocation table from growing without limit. liken's70// half-gigabyte slots land on 8 sectors (4 KiB), the same choice71// that every other formatter makes.72func fat32SectorsPerCluster(totalSectors uint64) uint64 {73	switch {74	case totalSectors <= 532_480: // ≤ 260 MB75		return 176	case totalSectors <= 16_777_216: // ≤ 8 GB77		return 878	case totalSectors <= 33_554_432: // ≤ 16 GB79		return 1680	case totalSectors <= 67_108_864: // ≤ 32 GB81		return 3282	default:83		return 6484	}85}8687// A Device is the surface that a format writes onto: positioned88// writes, plus a durability barrier. An *os.File is one example,89// either a partition device or a plain image file. A Section is90// another example: a partition's window inside an image file.91type Device interface {92	io.WriterAt93	Sync() error94}9596// FormatFAT32 writes a fresh, empty FAT32 filesystem across the97// given device. The volume ID is FAT's only identity field; FAT has98// no UUIDs. The label is cosmetic, up to 11 bytes, and appears in99// directory listings on other machines. Set it to something that100// says whose partition this is.101//102// Write order does not matter for crash safety. FormatFAT32 only103// ever runs against a partition being claimed or an image being104// built. If a write is torn, the volume stays unformatted, and the105// job simply runs again.106func FormatFAT32(f Device, totalBytes uint64, label string, volumeID uint32) error {107	totalSectors := totalBytes / SectorSize108	sectorsPerCluster := fat32SectorsPerCluster(totalSectors)109110	// This is Microsoft's own FAT-size formula, taken directly from111	// the specification. The table's size depends on the cluster112	// count, which depends on the space left after the table. The113	// formula resolves this circular dependency by slightly114	// overestimating. This wastes at most a sector or two of table115	// space that nothing will ever index.116	tmp1 := totalSectors - fat32ReservedSectors117	tmp2 := (256*sectorsPerCluster + fat32NumFATs) / 2118	fatSectors := (tmp1 + tmp2 - 1) / tmp2119120	dataStart := uint64(fat32ReservedSectors) + fat32NumFATs*fatSectors121	if totalSectors < dataStart {122		return fmt.Errorf("partition is too small for FAT32: %d bytes leaves no room past the tables", totalBytes)123	}124	clusters := (totalSectors - dataStart) / sectorsPerCluster125	if clusters < fat32MinClusters {126		return fmt.Errorf(127			"partition is too small for FAT32: %d bytes yields %d clusters and FAT32 requires %d (about 33Mi at one sector per cluster); the FAT type is determined by cluster count, so a smaller volume would misparse as FAT16",128			totalBytes, clusters, fat32MinClusters)129	}130	// This is the upper limit. Cluster numbers at and above131	// 0x0FFFFFF5 are reserved marks, such as bad cluster or end of132	// chain, and are not addresses.133	if clusters >= 0x0FFFFFF5 {134		return fmt.Errorf("partition is too large for FAT32: %d clusters", clusters)135	}136137	boot := buildFAT32BootSector(totalSectors, fatSectors, sectorsPerCluster, label, volumeID)138	// The FSInfo sector caches two hints: how many clusters are free,139	// and where to start looking for one. Readers are allowed to140	// distrust it, but a fresh count is exact: every cluster except141	// the root directory's one cluster.142	info := buildFAT32FSInfo(uint32(clusters-1), 3)143144	// This code zeros both tables and the root directory's cluster145	// before it writes anything meaningful. A claimed partition146	// inherits whatever bytes the disk held before. Stale data where147	// a reader expects free entries is filesystem corruption.148	zeroStart := uint64(fat32ReservedSectors) * SectorSize149	zeroLen := (fat32NumFATs*fatSectors + sectorsPerCluster) * SectorSize150	if err := zeroRange(f, int64(zeroStart), int64(zeroLen)); err != nil {151		return fmt.Errorf("zeroing the allocation tables: %w", err)152	}153154	// The first three FAT entries are flags, not chains. Entry 0155	// repeats the media byte. Entry 1 carries the clean-shutdown and156	// no-errors bits. Entry 2 ends the root directory's one-cluster157	// chain.158	head := make([]byte, 12)159	binary.LittleEndian.PutUint32(head[0:4], 0x0FFFFFF8)160	binary.LittleEndian.PutUint32(head[4:8], 0x0FFFFFFF)161	binary.LittleEndian.PutUint32(head[8:12], 0x0FFFFFFF)162163	// The label lives in two places: the boot sector field above, and164	// a special entry in the root directory. This is a 32-byte165	// directory entry whose attribute byte (0x08) marks it as the166	// volume's name, not a file. Tools expect the two labels to167	// match. fsck flags a volume that has one without the other.168	labelEntry := make([]byte, 32)169	copy(labelEntry[0:11], "           ")170	copy(labelEntry[0:11], label)171	labelEntry[11] = 0x08172173	// This writes primary and backup copies of everything. Writing174	// the backups is worth the cost. fsck tools do recover boot175	// sectors from sector 6.176	writes := []struct {177		lba  uint64178		data []byte179	}{180		{0, boot},181		{1, info},182		{6, boot},183		{7, info},184		{fat32ReservedSectors, head},185		{fat32ReservedSectors + fatSectors, head},186		{dataStart, labelEntry},187	}188	for _, w := range writes {189		if _, err := f.WriteAt(w.data, int64(w.lba*SectorSize)); err != nil {190			return fmt.Errorf("writing sector %d: %w", w.lba, err)191		}192	}193	return f.Sync()194}195196// buildFAT32BootSector lays out the 90 bytes of geometry that every197// FAT reader parses, padded to a full sector. The sector ends with198// the 0x55AA mark that distinguishes a structured sector from a199// blank one. An MBR ends with the same two bytes, which is why200// init's blank-disk check needs no special case for FAT.201func buildFAT32BootSector(totalSectors, fatSectors, sectorsPerCluster uint64, label string, volumeID uint32) []byte {202	bs := make([]byte, SectorSize)203204	// This is a jump instruction over the parameter block, from the205	// era when the BIOS executed the boot sector's code. Nothing206	// executes it here, but readers check its shape as a sanity207	// check.208	bs[0], bs[1], bs[2] = 0xEB, 0x58, 0x90209	copy(bs[3:11], "liken   ")210211	binary.LittleEndian.PutUint16(bs[11:13], SectorSize)212	bs[13] = byte(sectorsPerCluster)213	binary.LittleEndian.PutUint16(bs[14:16], fat32ReservedSectors)214	bs[16] = fat32NumFATs215	// Offsets 17-23 are the FAT12/16 fields: root entry count,216	// 16-bit totals, and 16-bit FAT size. All are zero on FAT32.217	// Zeroing them is one of the ways readers tell the FAT variants218	// apart. The media byte at offset 21 survives from the diskette219	// era. 0xF8 means "fixed disk" and must match FAT entry 0.220	bs[21] = 0xF8221	// These are sectors-per-track and head counts for CHS222	// addressing, which nothing has used in this century. These223	// values are conventional filler.224	binary.LittleEndian.PutUint16(bs[24:26], 32)225	binary.LittleEndian.PutUint16(bs[26:28], 64)226	// "Hidden sectors" records the partition's offset on its disk,227	// and only old CHS arithmetic reads it. Modern formatters write228	// zero for partitions addressed by LBA.229	binary.LittleEndian.PutUint32(bs[28:32], 0)230	binary.LittleEndian.PutUint32(bs[32:36], uint32(totalSectors))231232	binary.LittleEndian.PutUint32(bs[36:40], uint32(fatSectors))233	// This sets the extension flags (mirroring on: writes go to both234	// FATs) and version 0.0, the only version that exists.235	binary.LittleEndian.PutUint16(bs[40:42], 0)236	binary.LittleEndian.PutUint16(bs[42:44], 0)237	// The root directory starts at cluster 2, because clusters 0 and238	// 1 do not exist. Their FAT entries are the flag words above.239	binary.LittleEndian.PutUint32(bs[44:48], 2)240	binary.LittleEndian.PutUint16(bs[48:50], 1) // FSInfo lives at sector 1241	binary.LittleEndian.PutUint16(bs[50:52], 6) // its backup, and the boot sector's, at 6242	// This writes 12 reserved bytes, then the extended boot signature243	// block: a BIOS drive number (0x80, first fixed disk), the 0x29244	// mark that says the three fields after it are present, the245	// volume's ID and label, and the type string. The type string is246	// informational only, but every formatter writes it, and some247	// tools read it.248	bs[64] = 0x80249	bs[66] = 0x29250	binary.LittleEndian.PutUint32(bs[67:71], volumeID)251	copy(bs[71:82], "           ")252	copy(bs[71:82], label)253	copy(bs[82:90], "FAT32   ")254255	bs[510], bs[511] = 0x55, 0xAA256	return bs257}258259// buildFAT32FSInfo lays out the free-space hint sector. It has three260// signature words, with the free-cluster count and the next-free261// hint between them.262func buildFAT32FSInfo(freeClusters, nextFree uint32) []byte {263	info := make([]byte, SectorSize)264	binary.LittleEndian.PutUint32(info[0:4], 0x41615252)265	binary.LittleEndian.PutUint32(info[484:488], 0x61417272)266	binary.LittleEndian.PutUint32(info[488:492], freeClusters)267	binary.LittleEndian.PutUint32(info[492:496], nextFree)268	info[510], info[511] = 0x55, 0xAA269	return info270}271272// zeroRange writes zeros across a byte range in bounded chunks. This273// lets it zero a megabyte of allocation table without needing a274// megabyte of memory.275func zeroRange(f io.WriterAt, offset, length int64) error {276	zeros := make([]byte, 256<<10)277	for length > 0 {278		n := min(length, int64(len(zeros)))279		if _, err := f.WriteAt(zeros[:n], offset); err != nil {280			return err281		}282		offset += n283		length -= n284	}285	return nil286}287288// HasFAT32 checks a device for a FAT32 filesystem that liken wrote:289// the boot signature plus the type string. Microsoft's specification290// warns that the type string does not determine the FAT type;291// cluster count does that. But the question here is narrower: has292// liken already formatted this volume? liken writes both marks293// itself, so checking for them is enough.294func HasFAT32(devPath string) bool {295	f, err := os.Open(devPath)296	if err != nil {297		return false298	}299	defer f.Close()300	head := make([]byte, SectorSize)301	if _, err := io.ReadFull(io.NewSectionReader(f, 0, SectorSize), head); err != nil {302		return false303	}304	return head[510] == 0x55 && head[511] == 0xAA && string(head[82:90]) == "FAT32   "305}
disks/fat32_read.go 86.6%
1package disks23// This file implements reading a FAT32 volume back. It is the4// reverse of the writer, and it is kept deliberately small. Its5// consumers are verification: tests that check the writer's output,6// and tooling that looks inside an install image without mounting7// it. Because of this, it reads whole files and resolves long8// names, and does nothing else that a real driver would do.910import (11	"encoding/binary"12	"fmt"13	"io"14	"slices"15	"strings"16	"unicode/utf16"17)1819// A FATVolume is an opened FAT32 filesystem. It holds the volume's20// geometry, its allocation table, and the free-space accounting21// that the writer left.22type FATVolume struct {23	dev               io.ReaderAt24	sectorsPerCluster uint3225	dataStart         uint3226	fat               []uint322728	// FreeClusters and NextFree repeat the FSInfo sector, the29	// writer's accounting, so that callers can check it against the30	// table itself.31	FreeClusters uint3232	NextFree     uint3233}3435// OpenFATVolume parses a volume's boot sector and allocation table.36// It verifies that the two FAT copies agree. That pair is FAT's37// only redundancy, so a disagreement means a writer failed partway38// through.39func OpenFATVolume(dev io.ReaderAt) (*FATVolume, error) {40	boot := make([]byte, SectorSize)41	if _, err := dev.ReadAt(boot, 0); err != nil {42		return nil, fmt.Errorf("reading the boot sector: %w", err)43	}44	if boot[510] != 0x55 || boot[511] != 0xAA || string(boot[82:90]) != "FAT32   " {45		return nil, fmt.Errorf("no FAT32 boot sector")46	}47	reserved := uint32(binary.LittleEndian.Uint16(boot[14:16]))48	fatSectors := binary.LittleEndian.Uint32(boot[36:40])49	numFATs := uint32(boot[16])5051	v := &FATVolume{52		dev:               dev,53		sectorsPerCluster: uint32(boot[13]),54		dataStart:         reserved + numFATs*fatSectors,55	}5657	table := make([]byte, int64(fatSectors)*SectorSize)58	if _, err := dev.ReadAt(table, int64(reserved)*SectorSize); err != nil {59		return nil, fmt.Errorf("reading the FAT: %w", err)60	}61	second := make([]byte, len(table))62	if _, err := dev.ReadAt(second, int64(reserved+fatSectors)*SectorSize); err != nil {63		return nil, fmt.Errorf("reading the second FAT: %w", err)64	}65	if !slices.Equal(table, second) {66		return nil, fmt.Errorf("the two FAT copies disagree")67	}6869	totalSectors := binary.LittleEndian.Uint32(boot[32:36])70	clusters := (totalSectors - v.dataStart) / v.sectorsPerCluster71	v.fat = make([]uint32, clusters+2)72	for i := range v.fat {73		v.fat[i] = binary.LittleEndian.Uint32(table[i*4:]) & 0x0FFFFFFF74	}7576	info := make([]byte, SectorSize)77	if _, err := dev.ReadAt(info, 1*SectorSize); err != nil {78		return nil, fmt.Errorf("reading FSInfo: %w", err)79	}80	v.FreeClusters = binary.LittleEndian.Uint32(info[488:492])81	v.NextFree = binary.LittleEndian.Uint32(info[492:496])82	return v, nil83}8485// UsedClusters counts the occupied entries in the table itself, so86// that callers can check the FSInfo accounting against the actual87// table.88func (v *FATVolume) UsedClusters() uint32 {89	used := uint32(0)90	for _, entry := range v.fat[2:] {91		if entry != 0 {92			used++93		}94	}95	return used96}9798// chain reads a whole cluster chain's bytes, rounded up to full99// clusters. Callers trim the result using the directory record's100// size.101func (v *FATVolume) chain(first uint32) ([]byte, error) {102	var out []byte103	clusterBytes := int64(v.sectorsPerCluster) * SectorSize104	for cluster := first; cluster < 0x0FFFFFF8; cluster = v.fat[cluster] {105		if cluster < 2 || int(cluster) >= len(v.fat) {106			return nil, fmt.Errorf("cluster chain runs off the table at %d", cluster)107		}108		buf := make([]byte, clusterBytes)109		off := (int64(v.dataStart) + int64(cluster-2)*int64(v.sectorsPerCluster)) * SectorSize110		if _, err := v.dev.ReadAt(buf, off); err != nil {111			return nil, err112		}113		out = append(out, buf...)114	}115	return out, nil116}117118// A FATEntry is one directory child, its long name resolved.119type FATEntry struct {120	Name         string121	IsDir        bool122	FirstCluster uint32123	Size         uint32124}125126// Entries decodes one directory's records. It resolves and127// checksum-verifies long-name chains, and skips dot entries and the128// volume label.129func (v *FATVolume) Entries(firstCluster uint32) ([]FATEntry, error) {130	raw, err := v.chain(firstCluster)131	if err != nil {132		return nil, err133	}134	var out []FATEntry135	var longName []uint16136	for off := 0; off+32 <= len(raw); off += 32 {137		rec := raw[off : off+32]138		switch {139		case rec[0] == 0x00:140			return out, nil // end of directory141		case rec[11] == 0x0F:142			var units []uint16143			for _, span := range [][2]int{{1, 11}, {14, 26}, {28, 32}} {144				for i := span[0]; i < span[1]; i += 2 {145					units = append(units, binary.LittleEndian.Uint16(rec[i:]))146				}147			}148			longName = append(units, longName...)149		case rec[11] == 0x08:150			longName = nil // the volume label151		default:152			name := shortNameOf(rec)153			if longName != nil {154				sum := byte(0)155				for i := range 11 {156					sum = (sum>>1 | sum<<7) + rec[i]157				}158				if raw[off-32+13] != sum {159					return nil, fmt.Errorf("long-name checksum mismatch before %q", name)160				}161				var units []uint16162				for _, u := range longName {163					if u == 0 || u == 0xFFFF {164						break165					}166					units = append(units, u)167				}168				name = string(utf16.Decode(units))169				longName = nil170			}171			if name == "." || name == ".." {172				continue173			}174			out = append(out, FATEntry{175				Name:         name,176				IsDir:        rec[11]&0x10 != 0,177				FirstCluster: uint32(binary.LittleEndian.Uint16(rec[20:22]))<<16 | uint32(binary.LittleEndian.Uint16(rec[26:28])),178				Size:         binary.LittleEndian.Uint32(rec[28:32]),179			})180		}181	}182	return out, nil183}184185func shortNameOf(rec []byte) string {186	base := strings.TrimRight(string(rec[0:8]), " ")187	ext := strings.TrimRight(string(rec[8:11]), " ")188	if base == "." || base == ".." {189		return base190	}191	if ext == "" {192		return base193	}194	return base + "." + ext195}196197// Find walks a slash-separated path from the root.198func (v *FATVolume) Find(path string) (FATEntry, error) {199	cluster := uint32(2)200	parts := strings.Split(strings.Trim(path, "/"), "/")201	for i, part := range parts {202		entries, err := v.Entries(cluster)203		if err != nil {204			return FATEntry{}, err205		}206		idx := slices.IndexFunc(entries, func(e FATEntry) bool { return e.Name == part })207		if idx < 0 {208			return FATEntry{}, fmt.Errorf("%q not found (component %q)", path, part)209		}210		if i == len(parts)-1 {211			return entries[idx], nil212		}213		cluster = entries[idx].FirstCluster214	}215	return FATEntry{}, fmt.Errorf("empty path %q", path)216}217218// ReadFile reads one file's exact bytes.219func (v *FATVolume) ReadFile(path string) ([]byte, error) {220	e, err := v.Find(path)221	if err != nil {222		return nil, err223	}224	if e.IsDir {225		return nil, fmt.Errorf("%q is a directory", path)226	}227	raw, err := v.chain(e.FirstCluster)228	if err != nil {229		return nil, err230	}231	return raw[:e.Size], nil232}
disks/fat32_state.go 84.4%
1package disks23// The mark a FAT volume carries between mounts.4//5// FAT has no journal. The one thing it records about its own health is6// a single bit in the boot sector, and the driver manages it: the bit7// goes on when the volume is mounted for writing, and off when the8// volume is released. A volume found with the bit on was mounted and9// never released, so its directory entries and its allocation table10// may disagree with each other.11//12// A machine reads that bit to detect a stop that was not clean.13// Clearing it is how a machine says the volume has been dealt with.14// Both belong here, next to the formatter that lays the boot sector15// out, because both depend on the same field offsets.16//17// The driver will not clear the bit itself once it has found it set.18// That is deliberate: the bit is meant to survive until a repair, so a19// warning cannot be lost by a reboot. It also means a volume marked20// once stays marked forever unless something outside the driver clears21// it, which is why ClearFAT32Dirty exists at all.2223import (24	"fmt"25	"os"26)2728// fat32StateOffset is where the boot sector keeps the state byte on a29// FAT32 volume. FAT12 and FAT16 keep it at offset 0x25 instead,30// because their extended boot signature block starts earlier. liken31// formats its volumes as FAT32 and nothing else, so only this offset32// is needed here, and the callers below check the volume's type before33// they trust the byte.34const fat32StateOffset = 0x413536// fatStateDirty is the bit that says the volume was not released.37const fatStateDirty = 0x013839// readBootSector reads the first sector and confirms that it is a40// FAT32 boot sector liken would recognize. Confirming the type first41// matters: on a FAT16 volume the same offset falls inside the count of42// sectors per FAT, so a byte read there would be a number, not a flag.43func readBootSector(devPath string) ([]byte, error) {44	f, err := os.Open(devPath)45	if err != nil {46		return nil, err47	}48	defer f.Close()49	sector := make([]byte, SectorSize)50	if _, err := f.ReadAt(sector, 0); err != nil {51		return nil, fmt.Errorf("reading the boot sector of %s: %w", devPath, err)52	}53	if sector[510] != 0x55 || sector[511] != 0xAA || string(sector[82:90]) != "FAT32   " {54		return nil, fmt.Errorf("%s does not carry a FAT32 boot sector", devPath)55	}56	return sector, nil57}5859// FAT32Dirty reports whether a volume still carries the mark that says60// it was mounted and never released.61//62// Read this before mounting the volume. Mounting it for writing sets63// the mark, so a read afterwards says only that the volume is in use64// now, which is true of every mounted volume and answers nothing.65func FAT32Dirty(devPath string) (bool, error) {66	sector, err := readBootSector(devPath)67	if err != nil {68		return false, err69	}70	return sector[fat32StateOffset]&fatStateDirty != 0, nil71}7273// ClearFAT32Dirty clears the mark and puts the change on the platter.74//75// The volume must not be mounted. A mounted volume's boot sector lives76// in the kernel's own buffer, and the driver writes that buffer back77// on its own schedule, so a write underneath it is either lost or78// wins by accident. Neither outcome is a fix.79//80// The caller is responsible for whether clearing the mark is honest.81// This function only performs it, and it refuses a volume that is82// not FAT32, because on any other layout this byte means something83// else.84func ClearFAT32Dirty(devPath string) error {85	sector, err := readBootSector(devPath)86	if err != nil {87		return err88	}89	if sector[fat32StateOffset]&fatStateDirty == 0 {90		return nil91	}92	f, err := os.OpenFile(devPath, os.O_RDWR, 0)93	if err != nil {94		return err95	}96	defer f.Close()97	sector[fat32StateOffset] &^= fatStateDirty98	// One byte carries the change, and writing only that byte leaves99	// every other field exactly as it was found. A volume this machine100	// did not format may hold fields liken does not recognize.101	if _, err := f.WriteAt(sector[fat32StateOffset:fat32StateOffset+1], fat32StateOffset); err != nil {102		return fmt.Errorf("clearing the mark on %s: %w", devPath, err)103	}104	// FAT32 keeps a backup boot sector, named by the field at offset105	// 50. The driver reads the primary and never consults the backup,106	// so the backup is left alone: a repair tool compares the two, and107	// a difference here is a true record of what happened.108	if err := f.Sync(); err != nil {109		return fmt.Errorf("flushing the mark on %s: %w", devPath, err)110	}111	return nil112}
disks/fat32_write.go 90.2%
1package disks23// This file writes files into a FAT32 volume, without mounting it.4//5// On a running machine, the kernel's vfat driver does this. A build6// tool that lays out an install image has no kernel to ask, so it7// performs the driver's role directly. The job is smaller than it8// sounds, because of what a FAT filesystem is. The FAT itself is one9// 32-bit entry for each cluster, and each entry names the next10// cluster of its file. A directory is only a file whose bytes are11// 32-byte records. Writing a file means: pick clusters, chain them12// in the table, put the bytes in the clusters, and add a record to13// the parent directory's bytes.14//15// This writer is deliberately narrower than a driver. It only ever16// fills a freshly formatted volume, once. Because of this,17// allocation is a bump pointer: the next free cluster is always the18// next cluster in sequence. There are no deletes and no19// fragmentation, and this code can hold the whole directory tree in20// memory until Close writes it out.21//22// The one genuinely complex part is names. FAT's native names use23// the DOS 8.3 form: eleven uppercase bytes. Any name that is longer,24// or that uses lowercase letters, uses the VFAT extension instead: a25// chain of "long name" records placed ahead of the real one. Each26// long-name record carries thirteen UTF-16 characters, and carries27// an attribute combination (read-only + hidden + system + volume28// label) that DOS-era readers were guaranteed to skip. Firmware and29// every current OS read these long-name records. loader.conf and30// deployment.cpio do not fit the 8.3 form, so this writer produces31// long-name records for them.3233import (34	"encoding/binary"35	"fmt"36	"io"37	"path"38	"strings"39	"unicode/utf16"40)4142// endOfChain is the FAT entry that marks a file's last cluster.43// Every value at and above 0x0FFFFFF8 means end-of-chain. This is44// the conventional value that every formatter writes.45const endOfChain = 0x0FFFFFF7 + 14647// A fatFile is one file waiting for its directory record. It48// records where its chain starts and how long the file really is.49// FAT records exact sizes, but the chain rounds up to full clusters.50type fatFile struct {51	name         string52	firstCluster uint3253	size         int6454}5556// A fatDir accumulates a directory's children until Close writes57// the records. Directories get their clusters at Close, after every58// file has its chain, because a record needs its child's first59// cluster.60type fatDir struct {61	name         string62	parent       *fatDir63	files        []fatFile64	subdirs      []*fatDir65	firstCluster uint3266}6768// A FATWriter fills a formatted FAT32 volume with files. The69// geometry comes from the volume's own boot sector, so the writer70// and the format always agree about where everything is.71type FATWriter struct {72	dev interface {73		io.ReaderAt74		io.WriterAt75		Sync() error76	}77	sectorsPerCluster uint3278	reservedSectors   uint3279	fatSectors        uint3280	dataStart         uint32 // sector of cluster 281	fat               []uint3282	next              uint32 // the bump allocator: next free cluster83	root              *fatDir84	dirs              map[string]*fatDir85	labelEntry        []byte // the volume-label record the format wrote86}8788// NewFATWriter opens a freshly formatted volume for filling. It89// reads the geometry back from the boot sector that the format just90// wrote.91func NewFATWriter(dev interface {92	io.ReaderAt93	io.WriterAt94	Sync() error95}) (*FATWriter, error) {96	boot := make([]byte, SectorSize)97	if _, err := dev.ReadAt(boot, 0); err != nil {98		return nil, fmt.Errorf("reading the boot sector: %w", err)99	}100	if boot[510] != 0x55 || boot[511] != 0xAA || string(boot[82:90]) != "FAT32   " {101		return nil, fmt.Errorf("the volume does not carry a FAT32 boot sector")102	}103104	w := &FATWriter{105		dev:               dev,106		sectorsPerCluster: uint32(boot[13]),107		reservedSectors:   uint32(binary.LittleEndian.Uint16(boot[14:16])),108		fatSectors:        binary.LittleEndian.Uint32(boot[36:40]),109	}110	numFATs := uint32(boot[16])111	if numFATs != 2 {112		return nil, fmt.Errorf("expected 2 FATs, found %d", numFATs)113	}114	w.dataStart = w.reservedSectors + numFATs*w.fatSectors115116	totalSectors := binary.LittleEndian.Uint32(boot[32:36])117	clusters := (totalSectors - w.dataStart) / w.sectorsPerCluster118119	// This holds the whole table in memory: the clusters, plus the120	// two flag entries. Entries 0 and 1 repeat what the format121	// wrote. Cluster 2 is the root directory's single starting122	// cluster.123	w.fat = make([]uint32, clusters+2)124	w.fat[0] = 0x0FFFFFF8125	w.fat[1] = 0x0FFFFFFF126	w.fat[2] = endOfChain127	w.next = 3128129	// The format left exactly one record in the root: the volume130	// label. This code keeps it, so that Close can write the root131	// back with the label first, the way tools expect.132	w.labelEntry = make([]byte, 32)133	if _, err := dev.ReadAt(w.labelEntry, int64(w.dataStart)*SectorSize); err != nil {134		return nil, fmt.Errorf("reading the volume label: %w", err)135	}136	if w.labelEntry[11] != 0x08 {137		return nil, fmt.Errorf("the root directory does not begin with the volume label the format writes")138	}139140	w.root = &fatDir{firstCluster: 2}141	w.dirs = map[string]*fatDir{"": w.root}142	return w, nil143}144145func (w *FATWriter) clusterBytes() int64 {146	return int64(w.sectorsPerCluster) * SectorSize147}148149// clusterOffset is a cluster's first byte on the volume. Clusters 0150// and 1 do not exist. The data region begins at cluster 2.151func (w *FATWriter) clusterOffset(n uint32) int64 {152	return (int64(w.dataStart) + int64(n-2)*int64(w.sectorsPerCluster)) * SectorSize153}154155// allocate takes n consecutive clusters and chains them, and156// returns the first one. Because the writer writes each volume only157// once, allocate can work as a simple bump pointer.158func (w *FATWriter) allocate(n int64) (uint32, error) {159	if n == 0 {160		n = 1 // even an empty directory owns one cluster161	}162	first := w.next163	if int64(first)+n > int64(len(w.fat)) {164		return 0, fmt.Errorf("the volume is full: %d clusters wanted, %d left", n, int64(len(w.fat))-int64(w.next))165	}166	for i := int64(0); i < n-1; i++ {167		w.fat[first+uint32(i)] = first + uint32(i) + 1168	}169	w.fat[first+uint32(n-1)] = endOfChain170	w.next += uint32(n)171	return first, nil172}173174// Mkdir registers a directory. Its parent must already exist. This175// matches how the archive writers in this package expect to be176// called.177func (w *FATWriter) Mkdir(dirPath string) error {178	dirPath = strings.Trim(dirPath, "/")179	if _, exists := w.dirs[dirPath]; exists {180		return fmt.Errorf("directory %q already exists", dirPath)181	}182	parent, ok := w.dirs[parentOf(dirPath)]183	if !ok {184		return fmt.Errorf("directory %q has no parent yet", dirPath)185	}186	d := &fatDir{name: path.Base(dirPath), parent: parent}187	parent.subdirs = append(parent.subdirs, d)188	w.dirs[dirPath] = d189	return nil190}191192// WriteFile streams one file's bytes into freshly allocated193// clusters, and records the file for its directory record. Streaming194// matters here, because the files in an OS image can be hundreds of195// megabytes.196func (w *FATWriter) WriteFile(filePath string, r io.Reader, size int64) error {197	filePath = strings.Trim(filePath, "/")198	dir, ok := w.dirs[parentOf(filePath)]199	if !ok {200		return fmt.Errorf("file %q has no directory yet", filePath)201	}202203	clusterBytes := w.clusterBytes()204	first, err := w.allocate((size + clusterBytes - 1) / clusterBytes)205	if err != nil {206		return err207	}208209	buf := make([]byte, clusterBytes)210	remaining := size211	for cluster := first; remaining > 0; cluster++ {212		n := min(remaining, clusterBytes)213		if _, err := io.ReadFull(r, buf[:n]); err != nil {214			return fmt.Errorf("reading %q: %w", filePath, err)215		}216		if _, err := w.dev.WriteAt(buf[:n], w.clusterOffset(cluster)); err != nil {217			return fmt.Errorf("writing %q: %w", filePath, err)218		}219		remaining -= n220	}221222	dir.files = append(dir.files, fatFile{name: path.Base(filePath), firstCluster: first, size: size})223	return nil224}225226// Close writes the data that depends on the complete directory227// tree: each directory's records, which need every child's first228// cluster, both copies of the FAT, and the free-space accounting.229// Close assigns clusters to every directory first, in one pass, so230// that parent and child records can point at each other regardless231// of order.232func (w *FATWriter) Close() error {233	var all []*fatDir234	var walk func(d *fatDir)235	walk = func(d *fatDir) {236		all = append(all, d)237		for _, sub := range d.subdirs {238			walk(sub)239		}240	}241	walk(w.root)242243	// This is the assignment pass. Every directory gets a chain244	// sized for its records. The root's chain starts at its fixed245	// cluster 2, and Close extends it only if the label and children246	// outgrow one cluster.247	clusterBytes := w.clusterBytes()248	for _, d := range all {249		size := int64(len(w.dirRecords(d, 0))) // cluster numbers don't change the length250		if d == w.root {251			if size > clusterBytes {252				extra, err := w.allocate((size - clusterBytes + clusterBytes - 1) / clusterBytes)253				if err != nil {254					return err255				}256				w.fat[2] = extra257			}258			continue259		}260		first, err := w.allocate((size + clusterBytes - 1) / clusterBytes)261		if err != nil {262			return err263		}264		d.firstCluster = first265	}266267	// This is the content pass. With every cluster known, this code268	// writes each directory's records along its chain.269	for _, d := range all {270		records := w.dirRecords(d, d.firstCluster)271		cluster := d.firstCluster272		for off := int64(0); off < int64(len(records)); off += clusterBytes {273			end := min(off+clusterBytes, int64(len(records)))274			if _, err := w.dev.WriteAt(records[off:end], w.clusterOffset(cluster)); err != nil {275				return fmt.Errorf("writing directory records: %w", err)276			}277			cluster = w.fat[cluster]278		}279	}280281	// This writes both FATs, byte for byte the same. The second282	// copy is FAT's only redundancy, so writing it identically is283	// the whole purpose of the copy.284	table := make([]byte, int64(w.fatSectors)*SectorSize)285	for i, entry := range w.fat {286		binary.LittleEndian.PutUint32(table[i*4:], entry)287	}288	for copyN := range 2 {289		off := int64(w.reservedSectors+uint32(copyN)*w.fatSectors) * SectorSize290		if _, err := w.dev.WriteAt(table, off); err != nil {291			return fmt.Errorf("writing FAT copy %d: %w", copyN+1, err)292		}293	}294295	// This writes the primary and backup FSInfo sectors. The free296	// count is exact, because the bump allocator tracks precisely297	// what it used.298	free := uint32(len(w.fat)) - w.next299	info := buildFAT32FSInfo(free, w.next)300	for _, sector := range []int64{1, 7} {301		if _, err := w.dev.WriteAt(info, sector*SectorSize); err != nil {302			return fmt.Errorf("writing FSInfo: %w", err)303		}304	}305	return w.dev.Sync()306}307308// dirRecords lays out one directory's 32-byte records. It writes309// the label (root only), the dot entries (subdirectories only), and310// then every child. Each child record is preceded by its long-name311// records when the 8.3 form cannot carry the name alone. Short312// names must be unique within their directory, because fsck treats313// a duplicate as corruption. For this reason, when two truncated314// stand-in names collide, this code adds the classic ~N tail.315func (w *FATWriter) dirRecords(d *fatDir, self uint32) []byte {316	var records []byte317	if d == w.root {318		records = append(records, w.labelEntry...)319	} else {320		// "." and ".." are ordinary records with fixed names. By321		// specification, ".." pointing at the root writes cluster 0,322		// not 2. This is a DOS-era exception that every reader323		// expects.324		parent := d.parent.firstCluster325		if d.parent == w.root {326			parent = 0327		}328		records = append(records, plainRecord(".          ", 0x10, self, 0)...)329		records = append(records, plainRecord("..         ", 0x10, parent, 0)...)330	}331	taken := map[string]bool{}332	for _, sub := range d.subdirs {333		records = append(records, namedRecords(sub.name, 0x10, sub.firstCluster, 0, taken)...)334	}335	for _, f := range d.files {336		records = append(records, namedRecords(f.name, 0x20, f.firstCluster, f.size, taken)...)337	}338	return records339}340341// plainRecord is one 32-byte record whose 8.3 name is already342// known. The timestamps are all zero, for the same reason as the343// cpio writer's timestamps: the image's bytes should depend on its344// contents, not on when it was built. FAT cannot express a year345// before 1980, so a zero date reads back as 1980-00-00. This value346// is meaningless but harmless, because nothing on an install stick347// reads it.348func plainRecord(shortName string, attr byte, firstCluster uint32, size int64) []byte {349	r := make([]byte, 32)350	copy(r[0:11], shortName)351	r[11] = attr352	binary.LittleEndian.PutUint16(r[20:22], uint16(firstCluster>>16))353	binary.LittleEndian.PutUint16(r[26:28], uint16(firstCluster))354	binary.LittleEndian.PutUint32(r[28:32], uint32(size))355	return r356}357358// namedRecords builds a child's full record set: the plain record359// under its 8.3 name, preceded by long-name records when the real360// name needs them. taken tracks the directory's short names, so361// that no two children share one.362func namedRecords(name string, attr byte, firstCluster uint32, size int64, taken map[string]bool) []byte {363	short, needsLong := shortNameFor(name, taken)364	taken[short] = true365	record := plainRecord(short, attr, firstCluster, size)366	if !needsLong {367		return record368	}369	return append(longNameRecords(name, short), record...)370}371372// shortNameFor derives the 11-byte 8.3 form. A name that is already373// a valid uppercase 8.3 name needs no long-name records. Every374// other name gets an uppercased, truncated stand-in, and carries375// long-name records alongside it. The extension comes from the last376// dot, following the DOS convention. When a stand-in collides with377// an earlier sibling's, this code adds the classic ~N tail. Readers378// match names using the long-name records, but the short names379// still must be unique, or fsck reports the directory as corrupt.380func shortNameFor(name string, taken map[string]bool) (string, bool) {381	base, ext := name, ""382	if i := strings.LastIndex(name, "."); i > 0 {383		base, ext = name[:i], name[i+1:]384	}385	fits := len(base) >= 1 && len(base) <= 8 && len(ext) <= 3 && !strings.Contains(base, ".")386	upper := strings.ToUpper(name)387	simple := strings.IndexFunc(upper, func(r rune) bool {388		return (r < 'A' || r > 'Z') && (r < '0' || r > '9') && r != '.' && r != '-' && r != '_'389	}) < 0390	needsLong := !fits || !simple || name != upper391392	base = strings.ToUpper(strings.ReplaceAll(base, ".", ""))393	ext = strings.ToUpper(ext)394	if len(base) > 8 {395		base = base[:8]396	}397	if len(ext) > 3 {398		ext = ext[:3]399	}400	short := fmt.Sprintf("%-8s%-3s", base, ext)401	for tail := 1; taken[short]; tail++ {402		suffix := fmt.Sprintf("~%d", tail)403		trimmed := base[:min(len(base), 8-len(suffix))] + suffix404		short = fmt.Sprintf("%-8s%-3s", trimmed, ext)405		needsLong = true406	}407	return short, needsLong408}409410// longNameRecords encodes a name as VFAT long-name records. Each411// record holds thirteen UTF-16 characters, and the records are412// stored last piece first. The final piece is flagged 0x40. Every413// record carries a checksum of the 8.3 name that it belongs to, so414// a reader can tell a stray long-name record from one that belongs415// to a live entry.416func longNameRecords(name, short string) []byte {417	units := utf16.Encode([]rune(name))418	units = append(units, 0) // NUL-terminated, then padded with 0xFFFF419	for len(units)%13 != 0 {420		units = append(units, 0xFFFF)421	}422423	sum := byte(0)424	for i := range 11 {425		sum = (sum>>1 | sum<<7) + short[i]426	}427428	pieces := len(units) / 13429	var records []byte430	for piece := pieces; piece >= 1; piece-- {431		r := make([]byte, 32)432		r[0] = byte(piece)433		if piece == pieces {434			r[0] |= 0x40 // the last piece of the name, stored first435		}436		r[11] = 0x0F // the attribute combination DOS readers skip437		r[13] = sum438		chunk := units[(piece-1)*13 : piece*13]439		for i, u := range chunk[0:5] {440			binary.LittleEndian.PutUint16(r[1+2*i:], u)441		}442		for i, u := range chunk[5:11] {443			binary.LittleEndian.PutUint16(r[14+2*i:], u)444		}445		for i, u := range chunk[11:13] {446			binary.LittleEndian.PutUint16(r[28+2*i:], u)447		}448		records = append(records, r...)449	}450	return records451}452453func parentOf(p string) string {454	dir := path.Dir(p)455	if dir == "." || dir == "/" {456		return ""457	}458	return dir459}
disks/gpt.go 93.1%
1// Package disks writes and reads the on-disk formats that liken2// works with directly: GUID Partition Tables and FAT32 filesystems.3//4// These are the two formats that firmware itself understands. This5// is why liken handles them directly, instead of shelling out to6// tools. A machine's first boot has no tools available, and a7// workstation that builds an install image should not need any8// either. Both formats are a few hundred bytes at well-known9// offsets, simple enough that writing them directly shows exactly10// what they are.11//12// Two consumers share this package: init, which claims and grows a13// machine's real disks from PID 1, and the image package, which14// places the same structures into a plain file when it builds15// bootable install media.16package disks1718// This file implements reading and writing GUID Partition Tables,19// from scratch.20//21// A partition table is not an artifact of some tool. It is a few22// hundred bytes at well-known offsets that the kernel, firmware, and23// every OS can read. GPT, the modern format, is simple24// enough to handle directly, and doing so shows exactly what it is:25//26//	LBA 0        a "protective MBR": a legacy MBR whose single27//	             partition claims the whole disk, so that tools28//	             that do not understand GPT see an occupied disk29//	             instead of free space30//	LBA 1        the GPT header: where everything else is, plus31//	             CRC32 checksums of itself and of the entry array32//	LBA 2..33    the partition entry array: 128 slots of 12833//	             bytes, each naming a partition's type, unique34//	             GUID, extent, and 36-character name35//	...          the partitions themselves36//	end of disk  a mirror of the entries and header (in that37//	             order, with the header last at the very final38//	             sector), so that the table survives damage to39//	             LBA 0 and 140//41// The 36-character partition name is the field that liken cares42// about most. It carries a role's identity (liken:clusterState),43// which is what lets every boot after the first recognize a44// partition, no matter what device name the kernel assigns the disk45// that day.46//47// Writing comes in two forms. Claiming a blank disk creates a table48// from nothing, and mints fresh GUIDs. Growing a partition edits the49// table that already exists, which is why this file also contains a50// reader. An edit must carry every identity (the disk's GUID, and51// each partition's unique GUID) through unchanged. Other tools rely52// on those GUIDs to tell disks and partitions apart, so an edit must53// never regenerate them.5455import (56	"crypto/rand"57	"encoding/binary"58	"fmt"59	"hash/crc32"60	"io"61	"os"62	"slices"63	"unicode/utf16"6465	"golang.org/x/sys/unix"66)6768const (69	SectorSize = 5127071	// The entry array's size is part of the format: 128 entries of72	// 128 bytes equals 32 sectors. The MBR and header come before it,73	// so the first sector a partition may use is 34. The layout74	// mirrors at the tail, so the last usable sector is 34 sectors75	// from the end.76	entryCount   = 12877	entrySize    = 12878	entrySectors = entryCount * entrySize / SectorSize79	ReservedLBAs = 2 + entrySectors8081	// Partitions start on 1MiB boundaries (2048 sectors). This is82	// the alignment that every modern partitioner uses. This83	// alignment makes partitions line up with any plausible physical84	// block size or RAID stripe underneath.85	PartitionAlignment = 20488687	// GPT names use UTF-16LE, because UEFI predates the industry's88	// move to UTF-8, in a fixed 72-byte field of 36 code units.89	NameChars = 3690)9192// LastUsableLBA is the last sector that a partition may occupy.93// The backup entry array and header claim the tail of the disk.94func LastUsableLBA(totalSectors uint64) uint64 {95	return totalSectors - ReservedLBAs - 196}9798// AlignLBA rounds a sector number up to the next 1MiB boundary.99func AlignLBA(lba uint64) uint64 {100	return (lba + PartitionAlignment - 1) / PartitionAlignment * PartitionAlignment101}102103// A Partition is one entry as liken declares it when claiming a104// disk. A fresh unique GUID is filled in at write time. The type105// GUID is part of the plan. Most roles are ordinary Linux data, but106// the system slots must be typed as EFI system partitions, or the107// firmware will never look at them.108type Partition struct {109	Name     string110	FirstLBA uint64111	LastLBA  uint64112	TypeGUID [16]byte113}114115// An Entry is one occupied slot of the entry array. It holds116// everything the disk records about a partition. An edit preserves117// each field byte-for-byte, except the extent that the edit changes.118type Entry struct {119	TypeGUID   [16]byte120	UniqueGUID [16]byte121	FirstLBA   uint64122	LastLBA    uint64123	Attributes uint64124	Name       string125}126127// A Table is a whole partition table in memory. It is what128// ReadGPT returns and what SerializeGPT lays out. This struct keeps129// the two header fields, because they show how to detect a grown130// disk. A table whose alternate (backup) header is no longer at the131// disk's final sector was written when the disk was smaller.132type Table struct {133	DiskGUID      [16]byte134	Entries       []Entry135	LastUsableLBA uint64136	AlternateLBA  uint64137}138139// A Chunk is one run of bytes at one location. It is the unit that140// serialization produces and that writing consumes.141type Chunk struct {142	LBA  uint64143	Data []byte144}145146// mbrBootCodeSize is the number of bytes at the start of sector 0147// that precede the MBR's own fields: 440 bytes of x86 boot code,148// then the 4-byte disk signature and 2 reserved bytes. Everything149// before the partition entries at byte 446 belongs to the150// bootloader, not the partition table.151const mbrBootCodeSize = 446152153// BIOSBootPartition is the type GUID for GRUB's BIOS boot154// partition: a small raw partition, with no filesystem, that holds155// the core image that the MBR's 440 bytes of boot code jump into.156// The MBR has no room for a real program, and GPT, unlike the old157// MBR layout, leaves no dependable gap after sector 0. For this158// reason, GRUB's convention is to use a typed partition that159// nothing else will claim. The GUID spells "Hah!IdontNeedEFI" in160// ASCII. The GRUB developers chose this text as a joke, but every161// partitioning tool recognizes it as a genuine, well-known constant.162var BIOSBootPartition = MustGUID("21686148-6449-6E6F-744E-656564454649")163164// LinuxFilesystemData is the partition type GUID that means "an165// ordinary Linux filesystem". Types are well-known constants, not166// invented values. This exact GUID is what lsblk, blkid, and167// installers everywhere recognize as a Linux data partition.168var LinuxFilesystemData = MustGUID("0FC63DAF-8483-4772-8E79-3D69D8477DE4")169170// EFISystemPartition is the type GUID that makes a partition an171// ESP: the one partition type that UEFI firmware itself reads. The172// type GUID is the entire signal. Firmware does not inspect a173// partition's contents to find boot candidates. It looks for this174// GUID, and expects to find FAT inside. This is why liken plans the175// type for each role instead of assuming it.176var EFISystemPartition = MustGUID("C12A7328-F81F-11D2-BA4B-00A0C93EC93B")177178// MustGUID turns a GUID's canonical text into its 16 on-disk bytes.179// The encoding is a historical exception. The first three fields are180// little-endian, because GUIDs come from Microsoft by way of UEFI,181// while the rest is byte-for-byte. Because of this, the text and the182// bytes read in different orders. Getting this wrong makes every183// tool misread the type.184func MustGUID(s string) [16]byte {185	var canonical [16]byte186	n, err := fmt.Sscanf(s,187		"%02X%02X%02X%02X-%02X%02X-%02X%02X-%02X%02X-%02X%02X%02X%02X%02X%02X",188		&canonical[0], &canonical[1], &canonical[2], &canonical[3],189		&canonical[4], &canonical[5], &canonical[6], &canonical[7],190		&canonical[8], &canonical[9], &canonical[10], &canonical[11],191		&canonical[12], &canonical[13], &canonical[14], &canonical[15])192	if n != 16 || err != nil {193		panic("bad GUID literal: " + s)194	}195	return guidToDisk(canonical)196}197198func guidToDisk(canonical [16]byte) [16]byte {199	return [16]byte{200		canonical[3], canonical[2], canonical[1], canonical[0],201		canonical[5], canonical[4],202		canonical[7], canonical[6],203		canonical[8], canonical[9],204		canonical[10], canonical[11], canonical[12], canonical[13], canonical[14], canonical[15],205	}206}207208// RandomGUID generates a version-4 (random) GUID in on-disk209// encoding, for the disk itself and for each partition. This makes210// each one globally distinct from every other disk and partition in211// the world.212func RandomGUID() [16]byte {213	var canonical [16]byte214	if _, err := rand.Read(canonical[:]); err != nil {215		// A failure from crypto/rand means the kernel could not216		// supply random bytes at all. Other tools trust these GUIDs217		// as permanent identities that tell disks apart. Minting218		// them without randomness risks collisions, so this code219		// cannot continue. In init, this panic stops the kernel, and220		// panic=10 reboots the machine for another try. In the221		// toolkit, it ends the build.222		panic(err)223	}224	canonical[6] = canonical[6]&0x0F | 0x40 // version 4: random225	canonical[8] = canonical[8]&0x3F | 0x80 // variant: RFC 4122226	return guidToDisk(canonical)227}228229// SerializeGPT is the pure half of writing. It takes a table in, and230// produces the five on-disk chunks out: protective MBR, primary231// header, entries, backup entries, and backup header, with both232// CRCs computed. totalSectors sets where the backup lands and233// what lastUsable becomes. Because of this, serializing a table that234// was read from a smaller disk relocates its backup to the new end.235func SerializeGPT(t *Table, totalSectors uint64) ([]Chunk, error) {236	if len(t.Entries) > entryCount {237		return nil, fmt.Errorf("%d partitions won't fit a %d-entry GPT", len(t.Entries), entryCount)238	}239240	// This writes the entry array first, because both headers embed241	// its checksum.242	entries := make([]byte, entryCount*entrySize)243	for i, p := range t.Entries {244		e := entries[i*entrySize:]245		copy(e[0:16], p.TypeGUID[:])246		copy(e[16:32], p.UniqueGUID[:])247		binary.LittleEndian.PutUint64(e[32:40], p.FirstLBA)248		binary.LittleEndian.PutUint64(e[40:48], p.LastLBA)249		binary.LittleEndian.PutUint64(e[48:56], p.Attributes)250251		name := utf16.Encode([]rune(p.Name))252		if len(name) > NameChars {253			return nil, fmt.Errorf("partition name %q exceeds GPT's %d characters", p.Name, NameChars)254		}255		for j, r := range name {256			binary.LittleEndian.PutUint16(e[56+2*j:], r)257		}258	}259	entriesCRC := crc32.ChecksumIEEE(entries)260261	backupHeaderLBA := totalSectors - 1262	backupEntriesLBA := totalSectors - 1 - entrySectors263264	// Each header records its own location and the other copy's265	// location. The backup is not a byte-for-byte copy of the266	// primary. It holds the same facts, written from the other end267	// of the disk.268	header := func(currentLBA, otherLBA, entriesLBA uint64) []byte {269		h := make([]byte, SectorSize)270		copy(h[0:8], "EFI PART")271		binary.LittleEndian.PutUint32(h[8:12], 0x0001_0000) // revision 1.0272		binary.LittleEndian.PutUint32(h[12:16], 92)         // header size273		// h[16:20] is the header's own CRC. This code computes it274		// over the 92 bytes with this field zeroed, then writes the275		// result into the field.276		binary.LittleEndian.PutUint64(h[24:32], currentLBA)277		binary.LittleEndian.PutUint64(h[32:40], otherLBA)278		binary.LittleEndian.PutUint64(h[40:48], ReservedLBAs)279		binary.LittleEndian.PutUint64(h[48:56], LastUsableLBA(totalSectors))280		copy(h[56:72], t.DiskGUID[:])281		binary.LittleEndian.PutUint64(h[72:80], entriesLBA)282		binary.LittleEndian.PutUint32(h[80:84], entryCount)283		binary.LittleEndian.PutUint32(h[84:88], entrySize)284		binary.LittleEndian.PutUint32(h[88:92], entriesCRC)285		binary.LittleEndian.PutUint32(h[16:20], crc32.ChecksumIEEE(h[0:92]))286		return h287	}288289	// This builds the protective MBR: one legacy partition of type290	// 0xEE that spans the disk, capped at the 32-bit sector count291	// that an MBR can express. This makes GPT-unaware tools refuse292	// to touch the disk, instead of treating it as empty space.293	//294	// This code owns only the tail of sector 0: the partition295	// entries at byte 446 and the boot signature at byte 510. The296	// 446 bytes ahead of them are BIOS boot code, and they belong to297	// whichever component owns booting. On a BIOS machine, that is298	// GRUB's first stage, which install places there and init heals299	// when needed. This serialization writes zeros there, because a300	// freshly claimed disk boots nothing yet. writeTableBytes301	// preserves whatever the disk already carries there, so302	// rewriting the partition table never removes a machine's303	// ability to boot.304	mbr := make([]byte, SectorSize)305	span := min(totalSectors-1, 0xFFFF_FFFF)306	entry := mbr[446:]307	entry[1], entry[2], entry[3] = 0x00, 0x02, 0x00 // CHS start: legacy filler308	entry[4] = 0xEE                                 // type: "GPT protective"309	entry[5], entry[6], entry[7] = 0xFF, 0xFF, 0xFF // CHS end: "beyond CHS"310	binary.LittleEndian.PutUint32(entry[8:12], 1)311	binary.LittleEndian.PutUint32(entry[12:16], uint32(span))312	mbr[510], mbr[511] = 0x55, 0xAA313314	return []Chunk{315		{0, mbr},316		{1, header(1, backupHeaderLBA, 2)},317		{2, entries},318		{backupEntriesLBA, entries},319		{backupHeaderLBA, header(backupHeaderLBA, 1, backupEntriesLBA)},320	}, nil321}322323// readGPTCopy parses one copy of the table from its header sector.324// It verifies both checksums, and explains what disqualified the325// copy if it fails.326func readGPTCopy(r io.ReaderAt, headerLBA uint64) (*Table, error) {327	h := make([]byte, SectorSize)328	if _, err := r.ReadAt(h, int64(headerLBA)*SectorSize); err != nil {329		return nil, fmt.Errorf("reading the header at sector %d: %w", headerLBA, err)330	}331	if string(h[0:8]) != "EFI PART" {332		return nil, fmt.Errorf("no GPT signature at sector %d", headerLBA)333	}334335	// The header's CRC covers headerSize bytes, with the CRC field336	// itself zeroed. This code reads the size from the header itself337	// (92 today, but the format allows more), so the checksum covers338	// exactly what the writer covered.339	headerSize := binary.LittleEndian.Uint32(h[12:16])340	if headerSize < 92 || headerSize > SectorSize {341		return nil, fmt.Errorf("implausible header size %d at sector %d", headerSize, headerLBA)342	}343	scratch := slices.Clone(h[:headerSize])344	clear(scratch[16:20])345	if crc32.ChecksumIEEE(scratch) != binary.LittleEndian.Uint32(h[16:20]) {346		return nil, fmt.Errorf("header checksum mismatch at sector %d", headerLBA)347	}348349	// liken only ever writes the standard 128×128 array. A table350	// with any other geometry is not one that liken wrote. liken351	// refuses foreign disks at claim time, so this code refuses this352	// table too.353	gotCount := binary.LittleEndian.Uint32(h[80:84])354	gotSize := binary.LittleEndian.Uint32(h[84:88])355	if gotCount != entryCount || gotSize != entrySize {356		return nil, fmt.Errorf("unexpected entry geometry %d×%d at sector %d", gotCount, gotSize, headerLBA)357	}358359	entriesLBA := binary.LittleEndian.Uint64(h[72:80])360	entries := make([]byte, gotCount*gotSize)361	if _, err := r.ReadAt(entries, int64(entriesLBA)*SectorSize); err != nil {362		return nil, fmt.Errorf("reading the entry array at sector %d: %w", entriesLBA, err)363	}364	if crc32.ChecksumIEEE(entries) != binary.LittleEndian.Uint32(h[88:92]) {365		return nil, fmt.Errorf("entry array checksum mismatch (header at sector %d)", headerLBA)366	}367368	t := &Table{369		LastUsableLBA: binary.LittleEndian.Uint64(h[48:56]),370		AlternateLBA:  binary.LittleEndian.Uint64(h[32:40]),371	}372	copy(t.DiskGUID[:], h[56:72])373374	for i := range int(gotCount) {375		e := entries[i*int(gotSize):]376		var typeGUID [16]byte377		copy(typeGUID[:], e[0:16])378		if typeGUID == ([16]byte{}) {379			continue // an all-zero type GUID marks an unused slot380		}381		p := Entry{382			TypeGUID:   typeGUID,383			FirstLBA:   binary.LittleEndian.Uint64(e[32:40]),384			LastLBA:    binary.LittleEndian.Uint64(e[40:48]),385			Attributes: binary.LittleEndian.Uint64(e[48:56]),386		}387		copy(p.UniqueGUID[:], e[16:32])388		var units []uint16389		for j := range NameChars {390			u := binary.LittleEndian.Uint16(e[56+2*j:])391			if u == 0 {392				break393			}394			units = append(units, u)395		}396		p.Name = string(utf16.Decode(units))397		t.Entries = append(t.Entries, p)398	}399	return t, nil400}401402// ReadGPT parses a device's partition table. It tries the primary403// copy first, and falls back to the backup. ReadGPT resolves404// disagreements in the primary's favor. The kernel read the primary405// too, so recognition already trusted it, and the next rewrite makes406// the pair agree again.407func ReadGPT(r io.ReaderAt, totalSectors uint64) (*Table, error) {408	primary, perr := readGPTCopy(r, 1)409	backup, berr := readGPTCopy(r, totalSectors-1)410411	switch {412	case perr == nil && berr == nil:413		if primary.DiskGUID != backup.DiskGUID || !slices.Equal(primary.Entries, backup.Entries) {414			fmt.Println("liken: storage: the primary and backup partition tables disagree; trusting the primary (so did the kernel)")415		}416		return primary, nil417	case perr == nil:418		fmt.Printf("liken: storage: the backup partition table is unreadable (%v); the next rewrite restores it\n", berr)419		return primary, nil420	case berr == nil:421		fmt.Printf("liken: storage: the primary partition table is unreadable (%v); recovered from the backup\n", perr)422		return backup, nil423	default:424		// Redundancy cannot cover one case. If the disk was grown425		// while the primary was already unreadable, the backup is426		// still at the old end of the disk, where nothing looks for427		// it.428		return nil, fmt.Errorf("neither partition table copy is readable: primary: %v; backup at sector %d: %v (a grown disk's backup is no longer at the end)",429			perr, totalSectors-1, berr)430	}431}432433// WriteTable is the I/O half of writing. It serializes the table434// for this disk size, writes the chunks to their places, and then435// asks the kernel to re-read the result. WriteTable is deliberately436// thin. Everything interesting happens in SerializeGPT.437func WriteTable(devPath string, totalSectors uint64, t *Table) error {438	f, err := writeTableBytes(devPath, totalSectors, t)439	if err != nil {440		return err441	}442	defer f.Close()443444	// The bytes are on disk, but the kernel's view of the device445	// predates them. This ioctl asks the kernel to re-read the446	// table, which is what makes the vda1, vda2, ... devices appear447	// or grow.448	if _, err := unix.IoctlRetInt(int(f.Fd()), unix.BLKRRPART); err != nil {449		return fmt.Errorf("re-reading partition table: %w", err)450	}451	return nil452}453454// WriteTableInPlace writes a table's bytes without asking the455// kernel to re-read them. This is only correct when no partition's456// extent changed, such as when relocating the backup copy to the457// end of a grown disk. In that case, the kernel's existing view of458// the device already matches the new table. WriteTableInPlace exists459// because the kernel refuses to re-read a disk with any partition460// mounted, and the disk that carries the running system always has461// one mounted: the boot slot that the OS mounted its own root image462// from.463func WriteTableInPlace(devPath string, totalSectors uint64, t *Table) error {464	f, err := writeTableBytes(devPath, totalSectors, t)465	if err != nil {466		return err467	}468	return f.Close()469}470471// writeTableBytes serializes and writes a table's chunks, and472// returns the still-open device, so the caller can do more work on473// it before it closes the device.474func writeTableBytes(devPath string, totalSectors uint64, t *Table) (*os.File, error) {475	chunks, err := SerializeGPT(t, totalSectors)476	if err != nil {477		return nil, err478	}479480	f, err := os.OpenFile(devPath, os.O_RDWR, 0)481	if err != nil {482		return nil, err483	}484485	// Sector 0 is shared between two owners. SerializeGPT owns the486	// protective entry and the boot signature, but the boot code487	// ahead of them (446 bytes: code plus the MBR disk signature)488	// does not belong to the table. This code carries the disk's489	// existing bytes through, so that a GPT rewrite, most commonly490	// growth relocating the backup, never erases the machine's own491	// bootloader.492	bootCode := make([]byte, mbrBootCodeSize)493	if _, err := f.ReadAt(bootCode, 0); err != nil {494		f.Close()495		return nil, fmt.Errorf("reading the existing boot code: %w", err)496	}497	for _, chunk := range chunks {498		if chunk.LBA == 0 {499			copy(chunk.Data[:mbrBootCodeSize], bootCode)500		}501	}502503	for _, chunk := range chunks {504		if _, err := f.WriteAt(chunk.Data, int64(chunk.LBA)*SectorSize); err != nil {505			f.Close()506			return nil, fmt.Errorf("writing at LBA %d: %w", chunk.LBA, err)507		}508	}509	if err := f.Sync(); err != nil {510		f.Close()511		return nil, err512	}513	return f, nil514}515516// Write lays a brand-new partition table onto a blank disk being517// claimed. It creates fresh GUIDs for the disk and every partition,518// because claiming is when these identities are created.519func Write(devPath string, totalSectors uint64, parts []Partition) error {520	t := &Table{DiskGUID: RandomGUID()}521	for _, p := range parts {522		t.Entries = append(t.Entries, Entry{523			TypeGUID:   p.TypeGUID,524			UniqueGUID: RandomGUID(),525			FirstLBA:   p.FirstLBA,526			LastLBA:    p.LastLBA,527			Name:       p.Name,528		})529	}530	return WriteTable(devPath, totalSectors, t)531}
disks/section.go 0.0%
1package disks23// A Section addresses one partition inside a larger file. Building a4// disk image requires this: the image file is the whole disk, and a5// build tool must write a filesystem into the partition's window of6// it, at an offset, without reaching anything outside that window.7// On a running machine, the kernel provides this windowing itself;8// the /dev/vda1 device is a section of /dev/vda. A build tool that9// writes a plain file must implement this windowing itself.1011import (12	"fmt"13	"io"14	"os"15)1617// A Section is a bounded window into a file, at an offset. The18// formats in this package accept a Section wherever they accept a19// Device or a reader.20type Section struct {21	f      *os.File22	offset int6423	size   int6424}2526// NewSection creates a window of size bytes into f, starting at27// offset.28func NewSection(f *os.File, offset, size int64) *Section {29	return &Section{f: f, offset: offset, size: size}30}3132// Size returns the window's length in bytes.33func (s *Section) Size() int64 { return s.size }3435// WriteAt writes within the window. WriteAt refuses to write past36// the window's end. A filesystem that tries to write outside its37// partition has a bug, and this code stops it at the first byte.38func (s *Section) WriteAt(p []byte, off int64) (int, error) {39	if off < 0 || off+int64(len(p)) > s.size {40		return 0, fmt.Errorf("write of %d bytes at %d reaches outside the %d-byte section", len(p), off, s.size)41	}42	return s.f.WriteAt(p, s.offset+off)43}4445// ReadAt reads within the window, with the same bounds check as46// WriteAt.47func (s *Section) ReadAt(p []byte, off int64) (int, error) {48	if off < 0 || off >= s.size {49		return 0, io.EOF50	}51	if off+int64(len(p)) > s.size {52		p = p[:s.size-off]53		n, err := s.f.ReadAt(p, s.offset+off)54		if err == nil {55			err = io.EOF56		}57		return n, err58	}59	return s.f.ReadAt(p, s.offset+off)60}6162// Sync flushes the underlying file. Durability is a property of the63// whole file, not of the window.64func (s *Section) Sync() error { return s.f.Sync() }
hardware/aliases.go 100.0%
1// Package hardware is the part of liken that observes the machine's2// devices. It holds the sysfs walk that finds the devices, the3// modalias matching that names their drivers, and the databases4// that turn hex IDs into words.5//6// The kernel does nearly all device management by itself. It7// enumerates hardware, binds resident drivers, and creates the8// /dev nodes. But the kernel never loads a module. When a device9// appears and no resident driver claims it, the kernel sends a10// uevent that carries a MODALIAS fingerprint, and the device has no11// driver until a userspace program responds to that uevent. On most12// systems, udev responds by loading whatever module matches. liken's13// design is different: drivers are declared in spec.modules, and an14// undriven device produces a report that names the module that would15// drive it, instead of liken loading a driver without a declaration.16// This package does the observing half of that design. It changes17// nothing on the machine; it only reads.18package hardware1920// This file implements the read side of the modalias system. The21// modalias system is the kernel's own driver-matching database, but22// organized for lookup by device fingerprint instead of by driver.23// Every device the kernel enumerates gets a MODALIAS string, a24// one-line fingerprint of its identity in a bus-specific format:25//26//	usb:v46F4p0001d0100dc00dsc00dp00ic08isc06ip50in0027//	pci:v00001AF4d00001050sv00001AF4sd00001100bc03sc80i0028//29// Every module declares, in its .modinfo section, glob patterns for30// the fingerprints it can drive. At build time, depmod extracts those31// patterns into modules.alias, with one "alias <pattern> <module>"32// line for each pattern. Matching a device's fingerprint against that33// table is how udev, or this package, finds the module that drives34// the device.3536import (37	"fmt"38	"os"39	"slices"40	"strings"41)4243// AliasTable is a loaded modules.alias file. It holds every44// driver-matching pattern that the kernel build produced, in file45// order. A lookup scans the table from the start. The table has only46// a few tens of thousands of lines, and a machine checks it only47// when a device has no driver. For this size and use, an index would48// add complexity without a benefit.49type AliasTable struct {50	entries []aliasEntry51}5253type aliasEntry struct {54	pattern string55	module  string56}5758// LoadAliasTable reads a modules.alias file. A missing table59// produces an error, not an empty table. Without the table, every60// unclaimed device would incorrectly report "no driver known". An61// error is more accurate than that incorrect report.62func LoadAliasTable(path string) (*AliasTable, error) {63	raw, err := os.ReadFile(path)64	if err != nil {65		return nil, fmt.Errorf("reading the module alias table: %w", err)66	}67	table := &AliasTable{}68	for line := range strings.SplitSeq(string(raw), "\n") {69		pattern, module, found := strings.Cut(strings.TrimPrefix(line, "alias "), " ")70		if !found || strings.HasPrefix(line, "#") {71			continue72		}73		table.entries = append(table.entries, aliasEntry{pattern: pattern, module: module})74	}75	return table, nil76}7778// Candidates returns every module whose alias patterns match this79// modalias. It removes duplicates and keeps table order. More than80// one match is normal, not an error. For example, a USB mass-storage81// device matches both uas and usb_storage. udev's response to this82// is to load both modules, after which the modules themselves83// determine which one binds to the device. liken reports all84// matching modules instead of loading any of them, because the85// choice belongs to the person who writes spec.modules.86func (t *AliasTable) Candidates(modalias string) []string {87	var modules []string88	for _, e := range t.entries {89		if matchModalias(e.pattern, modalias) && !slices.Contains(modules, e.module) {90			modules = append(modules, e.module)91		}92	}93	return modules94}9596// matchModalias implements the glob dialect that modules.alias uses.97// In this dialect, '*' matches any run of characters, '?' matches98// exactly one character, and every other character is literal. This99// dialect is fnmatch without character classes, because modinfo100// emits only these two metacharacters. The whole pattern must match101// the whole value. For this reason, "usb:v0403" does not match102// "usb:v0403p6001".103func matchModalias(pattern, value string) bool {104	// This function matches '*' greedily and backtracks on a105	// mismatch: a two-pointer glob walk. It stores the position of106	// the last star and the value position where the star last107	// matched. On a mismatch, it retries from the star, one more108	// character into the value.109	p, v := 0, 0110	star, mark := -1, 0111	for v < len(value) {112		switch {113		case p < len(pattern) && (pattern[p] == value[v] || pattern[p] == '?'):114			p++115			v++116		case p < len(pattern) && pattern[p] == '*':117			star, mark = p, v118			p++119		case star >= 0:120			mark++121			p, v = star+1, mark122		default:123			return false124		}125	}126	for p < len(pattern) && pattern[p] == '*' {127		p++128	}129	return p == len(pattern)130}
hardware/delivery.go 96.3%
1package hardware23// This file finds what claiming a device would hand over.4//5// Delivering hardware to a workload means giving it device nodes:6// the /dev entries that a container can receive without any7// privilege. Sysfs records exactly which descendants of a device8// have nodes. Any directory that holds a `dev` file is one of these9// descendants, and its uevent file publishes the node's path under10// /dev as DEVNAME. So the question "what would claiming this device11// deliver" is answered by a subtree walk. A device whose subtree12// delivers nothing, such as a NIC or a bare controller, is not13// claimable inventory, even though the hardware is real.14//15// The walk stops at nested bus devices. Sysfs nests the physical16// topology: a USB controller's directory contains the hubs, which17// contain the devices, which contain the interfaces. Without this18// limit, every device node on the bus would count toward the19// controller that hosts it. Each PCI and USB device gets its own20// inventory decision. For this reason, a walk claims only the nodes21// between this device and the next bus device below it.22//23// The walk stops at a bluetooth device as well. Sysfs puts the HID24// device of a connected peripheral under the adapter's own USB25// interface, so the nodes below a bluetooth device are the26// peripherals' nodes. They come and go as people turn controllers and27// headsets on, and the air is not a bus this walk can descend to give28// each peripheral its own inventory decision.29//30// One node comes from above the walk instead of below it. A USB31// interface's driver publishes its own nodes, but the device that the32// interface belongs to also has a usbfs node, and that is the node33// every libusb program opens. The walk reports it beside the subtree's34// nodes, because both belong to the same claim.35//36// One more node can belong to a claim from outside any subtree. The37// kernel registers a misc device such as /dev/uhid under no bus38// device, so the walk cannot reach it, and MiscNode below reads the39// misc class directly. The publish policy decides which claim40// receives such a node.41//42// One set of nodes comes from no walk at all. The kernel fixes the43// numbers of the first 32 input event nodes, and a node in that range44// exists only while a device is registered at that minor. EvdevNodes45// below lists the whole range from those numbers, so a claim can hold a46// node for a minor that a peripheral registers after the claim was47// prepared.4849import (50	"fmt"51	"io/fs"52	"os"53	"path/filepath"54	"slices"55	"strconv"56	"strings"57)5859// DeliveredNode is one /dev entry a claim on a device would inject.60// The subsystem is the kernel's category for the node, and the61// category decides how the node behaves when two processes open it.62// Block carries the sysfs base name when the node is a block device,63// because the platform test matches storage roles by that name.64type DeliveredNode struct {65	Path      string66	Subsystem string67	Block     string68}6970// Delivery is the report that the walk produces: every device node71// that a claim on this device would inject, each paired with its72// kernel subsystem. The publish policy groups these nodes by73// subsystem, so the pairing is the record, and the flat views below74// derive from it.75//76// BusNode is the node that carries transfers to the whole device on77// its bus, rather than to one driver's interface: on USB, the usbfs78// node. It is not part of Nodes because it is not one of the kinds79// the policy sorts by. If it were among them, the policy could80// deliver it with a companion device, and a claim on one wire would81// reach the whole device.82type Delivery struct {83	Nodes   []DeliveredNode84	BusNode string85}8687// DevNodes lists the /dev paths in walk order.88func (d Delivery) DevNodes() []string {89	var paths []string90	for _, n := range d.Nodes {91		paths = append(paths, n.Path)92	}93	return paths94}9596// Blocks lists the block-device base names among the nodes. The97// platform test checks these against the storage roles' partitions.98func (d Delivery) Blocks() []string {99	var blocks []string100	for _, n := range d.Nodes {101		if n.Block != "" {102			blocks = append(blocks, n.Block)103		}104	}105	return blocks106}107108// Subsystems names each kind of node the delivery holds, once,109// sorted, so the same hardware always reports the same list.110func (d Delivery) Subsystems() []string {111	var kinds []string112	for _, n := range d.Nodes {113		if n.Subsystem != "" && !slices.Contains(kinds, n.Subsystem) {114			kinds = append(kinds, n.Subsystem)115		}116	}117	slices.Sort(kinds)118	return kinds119}120121// InspectDelivery walks one device's sysfs subtree and finds its122// device nodes. If the device is missing, InspectDelivery reports an123// empty delivery. Hardware can unplug between a discovery walk and124// this one, and an empty result is correct for hardware that is no125// longer there.126func InspectDelivery(sysRoot string, d Device) Delivery {127	root := filepath.Join(sysRoot, "bus", d.Bus, "devices", d.Address)128	// The bus entry is a symlink into the devices tree. The walk129	// needs the real directory, so that its children are also real130	// directories.131	resolved, err := filepath.EvalSymlinks(root)132	if err != nil {133		return Delivery{}134	}135	return Delivery{Nodes: SubtreeNodes(resolved), BusNode: usbfsNode(sysRoot, d)}136}137138// SubtreeNodes walks one sysfs directory and returns the device nodes139// under it, in walk order, stopping at the boundaries isSubtreeBoundary140// names. The directory must be a real directory, not a symlink to one,141// so that its children are real directories too. A directory that142// does not exist has no nodes.143func SubtreeNodes(dir string) []DeliveredNode {144	var nodes []DeliveredNode145	_ = filepath.WalkDir(dir, func(path string, entry fs.DirEntry, err error) error {146		if err != nil || !entry.IsDir() {147			return nil148		}149		if path != dir && isSubtreeBoundary(path) {150			return fs.SkipDir151		}152		if _, err := os.Stat(filepath.Join(path, "dev")); err != nil {153			return nil154		}155		devname := ueventDevName(path)156		if devname == "" {157			return nil158		}159		node := DeliveredNode{Path: "/dev/" + devname, Subsystem: subsystemName(path)}160		if node.Subsystem == "tty" && readAttr(path, "type") == "0" {161			// The serial core registers a tty for every port it162			// reserves, whether or not a UART answers there. The 8250163			// driver reserves 32 at boot, and a port with no UART164			// reports type 0, PORT_UNKNOWN. Its node opens and then165			// fails every read and write, so it delivers nothing.166			return nil167		}168		if node.Subsystem == "block" {169			node.Block = filepath.Base(path)170		}171		nodes = append(nodes, node)172		return nil173	})174	return nodes175}176177// usbfsNode reports the usbfs node of the USB device that an178// interface belongs to. usbfs gives every USB device one character179// node, /dev/bus/usb/<busnum>/<devnum>, with three digits in each180// number, and that node carries raw transfers to any endpoint of the181// device. A userspace driver built on libusb reads sysfs to enumerate182// the hardware, which needs no privilege, and then opens this node to183// talk to it. The nodes a kernel driver registers carry that driver's184// own protocol, so hidraw or a tty cannot take its place, and a185// libusb program in a container without it enumerates the device and186// then fails to open it.187//188// The walk adds this node only for an interface. Leaf drivers bind189// interfaces, usbfs names the usb_device above them, and the walk190// already reports that node for the usb_device itself.191//192// The kernel assigns devnum at enumeration, so the same hardware in193// the same port gets a different node after a replug. liken does not194// store the number: each walk reads it again.195func usbfsNode(sysRoot string, d Device) string {196	parent, _, isInterface := strings.Cut(d.Address, ":")197	if d.Bus != "usb" || !isInterface {198		return ""199	}200	dir := filepath.Join(sysRoot, "bus", "usb", "devices", parent)201	bus, err := strconv.Atoi(readAttr(dir, "busnum"))202	if err != nil {203		return ""204	}205	device, err := strconv.Atoi(readAttr(dir, "devnum"))206	if err != nil {207		return ""208	}209	return fmt.Sprintf("/dev/bus/usb/%03d/%03d", bus, device)210}211212// MiscNode reports the /dev node of a named misc device, or nothing213// while the module that registers the device is not loaded. The214// kernel lists every misc device under /sys/class/misc, where its215// uevent file publishes the node's path as DEVNAME, the same way a216// subtree's entries do, and devtmpfs creates the node the moment the217// class entry appears.218func MiscNode(sysRoot, name string) string {219	devname := ueventDevName(filepath.Join(sysRoot, "class", "misc", name))220	if devname == "" {221		return ""222	}223	return "/dev/" + devname224}225226// The kernel's numbering for the input subsystem's legacy event227// nodes: character major 13, one minor for each of the first 32228// event devices, counting up from 64.229const (230	evdevMajor     = 13231	evdevMinorBase = 64232	evdevMinors    = 32233)234235// EvdevNodes lists the 32 legacy input event nodes by the numbers the236// kernel fixes for them, with no sysfs entry behind them. A peripheral237// that connects over the air registers its node after a claim is238// prepared, and the container runtime injects a node only when the239// container starts. A claim that carries the whole range holds a node240// for every minor the kernel may use later, so a program in the241// container opens the device the moment it appears.242func EvdevNodes() []DeliveredNode {243	nodes := make([]DeliveredNode, 0, evdevMinors)244	for i := range evdevMinors {245		nodes = append(nodes, DeliveredNode{246			Path:      fmt.Sprintf("/dev/input/event%d", i),247			Subsystem: "input",248		})249	}250	return nodes251}252253// EvdevNumbers reports the character device numbers of a legacy input254// event node, and answers only for the first 32. Above that range the255// kernel allocates the minor when the device registers, so no fixed256// number exists to state before the node exists.257func EvdevNumbers(path string) (major, minor int, ok bool) {258	digits, found := strings.CutPrefix(path, "/dev/input/event")259	if !found {260		return 0, 0, false261	}262	index, err := strconv.Atoi(digits)263	if err != nil || index < 0 || index >= evdevMinors || strconv.Itoa(index) != digits {264		return 0, 0, false265	}266	return evdevMajor, evdevMinorBase + index, true267}268269// isSubtreeBoundary reports whether a sysfs directory ends the walk.270//271// A PCI or USB device is the boundary because another inventory272// decision begins there: the device below gets its own entry in the273// slice, and its nodes go with that entry.274//275// A bluetooth device is the boundary for a different reason. The276// nodes below it belong to the peripherals that are connected to the277// radio right now, not to the radio. A game controller's HID device278// lands under the adapter's own USB interface, so without this279// boundary a claim on the adapter receives the input and hidraw nodes280// of every controller a person turns on. The peripherals reach the281// radio over the air, which is not a bus this walk can descend, and282// they connect and disconnect while a pod holds the adapter.283func isSubtreeBoundary(path string) bool {284	switch subsystemName(path) {285	case "pci", "usb", "bluetooth":286		return true287	}288	return false289}290291// subsystemName reads the subsystem symlink that every sysfs device292// carries, and returns its base name: the kernel subsystem that the293// directory belongs to.294func subsystemName(path string) string {295	target, err := os.Readlink(filepath.Join(path, "subsystem"))296	if err != nil {297		return ""298	}299	return filepath.Base(target)300}301302// ueventDevName extracts DEVNAME from a device's uevent file.303// DEVNAME is the node's path relative to /dev, and devtmpfs mirrors304// this path exactly.305func ueventDevName(path string) string {306	raw, err := os.ReadFile(filepath.Join(path, "uevent"))307	if err != nil {308		return ""309	}310	for line := range strings.Lines(string(raw)) {311		if name, ok := strings.CutPrefix(strings.TrimSpace(line), "DEVNAME="); ok {312			return name313		}314	}315	return ""316}
hardware/inventory.go 95.5%
1package hardware23// This file finds the hardware a workload can claim: the devices that4// DiscoverDevices finds on the pci and usb buses, and the devices on5// the board that no bus enumerates.6//7// The unclaimed report reads the pci and usb buses alone, because a8// missing module is a problem to fix only there. The inventory asks a9// different question: which hardware delivers a device node that a10// pod could receive. Some of that hardware is on no pci or usb bus.11// The firmware describes it, through ACPI or the legacy PnP tables,12// and the kernel registers it as a platform or pnp device. A firmware13// TPM, a CMOS clock, a laptop's own keyboard behind the i804214// controller, the ACPI power and sleep buttons, and a serial port on15// the board are all of this kind. So the inventory starts from the16// nodes themselves, not from a list of buses. It reads every node the17// kernel registered and keeps the ones that no pci or usb device owns.18//19// Each such node belongs to the outermost device above it that has a20// driver. That device is the hardware: the TPM's node hangs under21// the platform device the TPM driver bound, and a keyboard's event22// node hangs under the input device, under the serio port, under the23// i8042 controller. The controller is the outermost driven device, so24// the controller is what a claim names, and the walk from it delivers25// every node beneath, the same way a pci device delivers the nodes26// beneath it.2728import (29	"os"30	"path/filepath"31	"slices"32	"strings"33)3435// DiscoverInventory lists the devices that the inventory considers:36// every device DiscoverDevices finds, the USB devices themselves, and37// then each device outside the pci and usb buses that owns a device38// node. naming may be nil, the same as for DiscoverDevices.39func DiscoverInventory(sysRoot string, naming *PCIIDs) []Device {40	devices := DiscoverDevices(sysRoot, naming)41	devices = append(devices, usbDevices(sysRoot)...)42	return append(devices, boardDevices(sysRoot)...)43}4445// usbDevices lists each USB device apart from its interfaces. The46// kernel gives a modalias to an interface and none to the device it47// belongs to, so DiscoverDevices, which reads a modalias as the mark48// of a device a module could drive, lists the interfaces alone. The49// inventory needs the device as well, because a device that no driver50// binds at all publishes whole, by the device's own port path.51func usbDevices(sysRoot string) []Device {52	dir := filepath.Join(sysRoot, "bus", "usb", "devices")53	entries, err := os.ReadDir(dir)54	if err != nil {55		return nil56	}57	var devices []Device58	for _, entry := range entries {59		name := entry.Name()60		path := filepath.Join(dir, name)61		if strings.Contains(name, ":") || readAttr(path, "modalias") != "" {62			continue63		}64		devices = append(devices, Device{65			Bus:     "usb",66			Address: name,67			Driver:  boundDriver(path),68			Name:    usbName(dir, name),69			Vendor:  readAttr(path, "idVendor"),70			Product: readAttr(path, "idProduct"),71			Serial:  readAttr(path, "serial"),72		})73	}74	return devices75}7677// InventoryEvent reports whether a uevent can change what78// DiscoverInventory and the delivery walks read. Every device outside79// /devices/virtual/ can, because the inventory starts from every node80// outside it. Under /devices/virtual/ the walks read only the misc81// class, where /dev/uhid and /dev/uinput appear when their modules82// load (MiscNode).83//84// The rest of /devices/virtual/ is the noise a node that runs85// Kubernetes makes. Each pod that starts or stops adds or removes a86// veth pair, and the pair and each of its queues announce themselves.87// The lab counted 132 such events in two minutes while seven pods88// started and four stopped, and nothing else under /devices/virtual/.89//90// A path outside /devices/ is a kernel object that is no device, and91// the walks read none. An NFS mount adds and removes RPC clients under92// /kernel/sunrpc/: a testbed machine that mounted NFS volumes counted93// 34 such events in 30 minutes. A module announces itself under94// /module/ when it loads, and the devices it creates send events of95// their own.96func InventoryEvent(event Uevent) bool {97	if !strings.HasPrefix(event.DevPath, "/devices/") {98		return false99	}100	if !strings.HasPrefix(event.DevPath, "/devices/virtual/") {101		return true102	}103	return strings.HasPrefix(event.DevPath, "/devices/virtual/misc/")104}105106// enumeratedBuses are the buses DiscoverDevices walks. A node under a107// device of either bus already belongs to that device's delivery.108var enumeratedBuses = map[string]bool{"pci": true, "usb": true}109110// boardDevices finds the outermost driven device above each node111// that no pci or usb device owns. /sys/dev/char and /sys/dev/block112// hold one link for every node the kernel registered, and each link113// resolves to the node's directory in the device tree.114func boardDevices(sysRoot string) []Device {115	// The links resolve to real paths, so the root they are compared116	// against must be real as well.117	devicesRoot, err := filepath.EvalSymlinks(filepath.Join(sysRoot, "devices"))118	if err != nil {119		return nil120	}121	owners := map[string]Device{}122	for _, kind := range []string{"char", "block"} {123		dir := filepath.Join(sysRoot, "dev", kind)124		entries, err := os.ReadDir(dir)125		if err != nil {126			continue127		}128		for _, entry := range entries {129			node, err := filepath.EvalSymlinks(filepath.Join(dir, entry.Name()))130			if err != nil {131				continue132			}133			owner, ok := boardOwner(devicesRoot, node)134			if !ok {135				continue136			}137			if _, seen := owners[owner]; seen {138				continue139			}140			bus := subsystemName(owner)141			address := filepath.Base(owner)142			// InspectDelivery finds a device through its bus's143			// devices directory, so an owner that its bus does not144			// list is one the delivery walk could not reach.145			if _, err := os.Stat(filepath.Join(sysRoot, "bus", bus, "devices", address)); err != nil {146				continue147			}148			owners[owner] = Device{149				Bus:      bus,150				Address:  address,151				Modalias: readAttr(owner, "modalias"),152				Driver:   boundDriver(owner),153			}154		}155	}156	devices := make([]Device, 0, len(owners))157	for _, d := range owners {158		devices = append(devices, d)159	}160	slices.SortFunc(devices, func(a, b Device) int {161		return strings.Compare(a.Bus+"/"+a.Address, b.Bus+"/"+b.Address)162	})163	return devices164}165166// boardOwner answers the device that owns one node's directory: the167// outermost directory between the device tree's root and the node168// that has a driver. A node under /devices/virtual/ has no hardware169// behind it, and a node under a pci or usb device belongs to that170// device, so neither has a board owner. A node with no driven device171// above it has none either, because no driver serves it.172func boardOwner(devicesRoot, node string) (string, bool) {173	rel, err := filepath.Rel(devicesRoot, node)174	if err != nil || rel == "." || strings.HasPrefix(rel, "..") {175		return "", false176	}177	parts := strings.Split(rel, string(filepath.Separator))178	if parts[0] == "virtual" {179		return "", false180	}181	owner := ""182	dir := devicesRoot183	for _, part := range parts {184		dir = filepath.Join(dir, part)185		if enumeratedBuses[subsystemName(dir)] {186			return "", false187		}188		if owner == "" && boundDriver(dir) != "" {189			owner = dir190		}191	}192	return owner, owner != ""193}
hardware/names.go 100.0%
1package hardware23// Class words: this file turns bus class codes into the general4// kind that a fleet listing can print. Both tables are small,5// spec-defined enums: the PCI SIG's class codes and the USB-IF's6// interface classes. For this reason, they exist here as Go tables7// instead of a database file. The vendor and product names are the8// opposite case. They have tens of thousands of entries that change9// every month, and pci.ids exists for that case.1011import "strings"1213// classCode normalizes a bus's class attribute to bare lowercase hex.14// The buses write it differently: PCI writes "0x038000" and USB writes15// "03". Both are published as they are, minus the prefix, so a16// selector matches what the bus said.17func classCode(class string) string {18	return strings.ToLower(strings.TrimPrefix(class, "0x"))19}2021// pciClassWord decodes sysfs's class attribute (for example,22// "0x038000" means class 03, subclass 80, interface 00) to the word23// for its class byte. An operator who decides whether to care about24// an unclaimed device needs a word like "display" or "network", not25// the subclass detail, so the vocabulary reads the base class, with26// exactly one subclass exception below. The test beside this file27// locks every word: a subclass earns a row of its own only when the28// base class's word would misdescribe a device liken must act on.29func pciClassWord(class string) string {30	hex := strings.TrimPrefix(class, "0x")31	if len(hex) < 2 {32		return ""33	}34	// Subclass 0805 is the SD and eMMC host controller. PCI files it35	// under base class 08, the system peripherals, but the device36	// behind it is a disk, and a machine whose only disk is an eMMC37	// module cannot install unless the report reads the controller as38	// storage. Every other subclass of base class 08 stays system,39	// because those devices really are the machine's own plumbing.40	if strings.HasPrefix(hex, "0805") {41		return "storage"42	}43	switch hex[:2] {44	case "01":45		return "storage"46	case "02":47		return "network"48	case "03":49		return "display"50	case "04":51		return "multimedia"52	case "05":53		return "memory"54	case "06":55		return "bridge"56	case "07":57		return "communication"58	case "08":59		return "system"60	case "09":61		return "input"62	case "0c":63		return "serial-bus"64	case "0d":65		return "wireless"66	case "10":67		return "encryption"68	case "12":69		return "accelerator"70	}71	return ""72}7374// usbClassWord decodes a USB interface's bInterfaceClass, two hex75// digits from the USB-IF's class code table, the same way.76func usbClassWord(class string) string {77	switch strings.ToLower(class) {78	case "01":79		return "audio"80	case "02":81		return "communications"82	case "03":83		return "hid"84	case "06":85		return "imaging"86	case "07":87		return "printer"88	case "08":89		return "mass-storage"90	case "0a":91		return "cdc-data"92	case "0b":93		return "smart-card"94	case "0e":95		return "video"96	case "10":97		return "audio-video"98	case "e0":99		return "wireless"100	case "ff":101		return "vendor-specific"102	}103	return ""104}
hardware/pciids.go 100.0%
1package hardware23// pci.ids is the naming database for PCI hardware.4//5// USB devices carry their own names: manufacturer and product6// strings that the kernel reads at enumeration. PCI devices carry7// only numbers. The names that lspci prints come from a community8// database, pci.ids, maintained by the hwdata project. liken vendors9// this database as a pinned flat file, and the hwdata domain owns10// the pin. Because of this, an unclaimed-device report can say "Red11// Hat, Inc. Virtio 1.0 GPU" instead of "1af4:1050". This dependency12// is soft by design. Every caller falls back to the numeric IDs when13// the file is missing, because naming adds convenience, and14// reporting is the required task.1516import (17	"os"18	"strings"19)2021// PCIIDs is the loaded database. It indexes vendors by their 4-digit22// hex ID, and devices by vendor plus device. This code does not23// parse the file's third level, the subsystem names, or its24// trailing class-code section. The report names devices, and class25// words come from the spec-defined table in names.go.26type PCIIDs struct {27	vendors map[string]string28	devices map[string]string29}3031// LoadPCIIDs parses the pci.ids format. Vendor lines start at the32// left margin (for example, "1af4  Red Hat, Inc."). Device lines33// start one tab in. Subsystem lines start two tabs in. Single-letter34// sections (for example, "C 03  Display controller") come after the35// vendor lines.36func LoadPCIIDs(path string) (*PCIIDs, error) {37	raw, err := os.ReadFile(path)38	if err != nil {39		return nil, err40	}41	ids := &PCIIDs{vendors: map[string]string{}, devices: map[string]string{}}42	vendor := ""43	for line := range strings.SplitSeq(string(raw), "\n") {44		switch {45		case line == "" || strings.HasPrefix(line, "#"):46		case strings.HasPrefix(line, "\t\t"):47			// Subsystem names give more detail than a status line needs.48		case strings.HasPrefix(line, "\t"):49			if id, name, found := strings.Cut(line[1:], "  "); found && vendor != "" {50				ids.devices[vendor+":"+id] = name51			}52		default:53			id, name, found := strings.Cut(line, "  ")54			// A non-hex first field, such as the "C 03" class55			// sections, ends the vendor list for this parser.56			if !found || len(id) != 4 || strings.ContainsFunc(id, notHex) {57				vendor = ""58				continue59			}60			vendor = strings.ToLower(id)61			ids.vendors[vendor] = name62		}63	}64	return ids, nil65}6667func notHex(r rune) bool {68	return !strings.ContainsRune("0123456789abcdefABCDEF", r)69}7071// Name renders a vendor:device pair in words. It returns both names72// when the database has an entry for the device. It returns the73// vendor name plus the numeric device ID when the database has an74// entry for the vendor only. It returns nothing for an unknown75// vendor, because the caller's numeric fallback reads better than a76// bare hex device ID with no vendor name.77func (p *PCIIDs) Name(vendor, device string) string {78	vendor, device = strings.ToLower(vendor), strings.ToLower(device)79	vendorName, ok := p.vendors[vendor]80	if !ok {81		return ""82	}83	if deviceName, ok := p.devices[vendor+":"+device]; ok {84		return vendorName + " " + deviceName85	}86	return vendorName + " device " + device87}
hardware/settle.go 100.0%
1package hardware23import (4	"context"5	"time"6)78// Settle drains further uevent signals until quiet lasts a full9// interval, so a burst of arrivals becomes one walk, but only up to10// a ceiling. Waiting for true silence does not work on a node that11// runs Kubernetes. Each pod that starts or stops adds or removes a12// veth pair, and the pair and each of its queues announce themselves,13// so the stream can stay busy for minutes. The lab observed this14// blocking a hot-plugged disk's report for minutes because of an15// unrelated crash-looping pod. Walks are cheap16// and idempotent, so when the stream will not go quiet, walking17// anyway is the correct move. Anything that changes during the walk18// sends another uevent signal. A closed channel means the listener19// stopped, so the wait ends at once, and the caller's next receive20// finds the close and opens the listener again.21func Settle(ctx context.Context, uevents <-chan struct{}, quiet, ceiling time.Duration) {22	deadline := time.NewTimer(ceiling)23	defer deadline.Stop()24	timer := time.NewTimer(quiet)25	defer timer.Stop()26	for {27		select {28		case <-ctx.Done():29			return30		case _, ok := <-uevents:31			if !ok {32				return33			}34			timer.Reset(quiet)35		case <-timer.C:36			return37		case <-deadline.C:38			return39		}40	}41}
hardware/sysfs.go 96.1%
1package hardware23// This file implements the sysfs walk. It reads the kernel's device4// tree the way udev's coldplug replay does, but only to observe, not5// to act.6//7// Sysfs is the kernel's live model of the hardware, with one8// directory per device, exported at /sys. Two facts about each9// device answer this package's whole question. The modalias file10// holds the device's identity fingerprint. The driver symlink exists11// exactly when a driver has bound the device. When the fingerprint12// is present and the symlink is absent, the device is undriven, and13// this is the raw material for the unclaimed report.14//15// The walk covers the pci and usb buses only. These are the buses16// where hardware arrives, either soldered to the board or plugged17// in at runtime, and where a missing module is a problem for the18// operator to fix. The other buses that sysfs shows are either19// devices built into the kernel itself (platform, acpi, and their20// many firmware-described stubs, most of which have no driver by21// design) or children of devices that the pci and usb buses already22// cover. Walking these other buses would report information that no23// change to spec.modules could act on. The device inventory asks a24// wider question, and inventory.go answers it from the nodes instead25// of the buses.2627import (28	"fmt"29	"os"30	"path/filepath"31	"strings"32)3334// Device is one bus device as sysfs presents it. It holds the35// device's fingerprint, the name of its driver (empty when it has36// none), and its identity in words. Every consumer gathers this data37// here — the unclaimed report, a console line, a ResourceSlice — so38// that each one describes hardware the same way.39//40// Address is the device's sysfs directory name: a PCI41// domain:bus:device.function or a USB port path. This name stays42// stable for as long as the hardware stays in place. PCI addresses43// are fixed by the board, and a USB path names the physical port44// chain, not the order in which devices were plugged in. This45// stability is what makes Address usable as a device name in a46// ResourceSlice.47//48// Serial is the identity that the hardware itself carries, when it49// carries one. USB devices often carry a serial number; PCI50// functions rarely do. Serial lets a claim pin one physical unit,51// rather than any unit of the same model.52//53// Vendor and Product are the numeric identity underneath Name: bare54// lowercase hex, without the 0x prefix. The struct keeps both forms55// because selectors match on numbers, while people read words.56//57// Class and ClassCode are the same fact at two depths. Class is the58// base class in one word, which is what a person reads. ClassCode is59// what the bus published, with no detail removed: six hex digits on60// PCI (base class, subclass, and programming interface) and two on61// USB (the interface class). The full code is what tells a VGA62// controller from a 3D one, and neither the kernel nor a database is63// needed to read it, so liken publishes it rather than growing a64// table of its own for every distinction a selector might need.65type Device struct {66	Bus       string67	Address   string68	Modalias  string69	Driver    string70	Name      string71	Class     string72	ClassCode string73	Serial    string74	Vendor    string75	Product   string76}7778// DiscoverDevices walks the pci and usb buses under one sysfs root.79// The root is a parameter so that tests can supply their own; a real80// machine has exactly one sysfs. naming may be nil. When it is nil,81// PCI devices fall back to their numeric IDs.82func DiscoverDevices(sysRoot string, naming *PCIIDs) []Device {83	var devices []Device84	for _, bus := range []string{"pci", "usb"} {85		dir := filepath.Join(sysRoot, "bus", bus, "devices")86		entries, err := os.ReadDir(dir)87		if err != nil {88			// A bus with no directory is not compiled into this89			// kernel. There is nothing to report.90			continue91		}92		for _, entry := range entries {93			path := filepath.Join(dir, entry.Name())94			modalias := readAttr(path, "modalias")95			if modalias == "" {96				continue97			}98			d := Device{Bus: bus, Address: entry.Name(), Modalias: modalias, Driver: boundDriver(path)}99			switch bus {100			case "pci":101				d.Vendor = strings.TrimPrefix(readAttr(path, "vendor"), "0x")102				d.Product = strings.TrimPrefix(readAttr(path, "device"), "0x")103				d.Name = pciName(d.Vendor, d.Product, naming)104				d.Class = pciClassWord(readAttr(path, "class"))105				d.ClassCode = classCode(readAttr(path, "class"))106			case "usb":107				d.Vendor = parentAttr(dir, entry.Name(), "idVendor")108				d.Product = parentAttr(dir, entry.Name(), "idProduct")109				d.Name = usbName(dir, entry.Name())110				d.Class = usbClassWord(readAttr(path, "bInterfaceClass"))111				d.ClassCode = classCode(readAttr(path, "bInterfaceClass"))112				d.Serial = parentAttr(dir, entry.Name(), "serial")113			}114			devices = append(devices, d)115		}116	}117	return devices118}119120// boundDriver reports which driver claimed a device: the driver121// symlink's target name. It returns empty when the symlink does not122// exist, which is how sysfs shows that no driver has bound the123// device.124func boundDriver(devicePath string) string {125	target, err := os.Readlink(filepath.Join(devicePath, "driver"))126	if err != nil {127		return ""128	}129	return filepath.Base(target)130}131132// readAttr reads one sysfs attribute file as a trimmed string. Sysfs133// attributes are single values with a trailing newline. A missing134// attribute reads as empty, because absence is normal. For example,135// only USB interfaces have bInterfaceClass, and only devices with136// strings have manufacturer.137func readAttr(devicePath, attr string) string {138	raw, err := os.ReadFile(filepath.Join(devicePath, attr))139	if err != nil {140		return ""141	}142	return strings.TrimSpace(string(raw))143}144145// pciName names a PCI device from the pci.ids database, when one is146// loaded. Otherwise, it falls back to the bare vendor:device hex.147// This fallback is numeric, but it is still enough to search by.148func pciName(vendor, device string, naming *PCIIDs) string {149	if vendor == "" {150		return ""151	}152	if naming != nil {153		if name := naming.Name(vendor, device); name != "" {154			return name155		}156	}157	return fmt.Sprintf("%s:%s", vendor, device)158}159160// usbName names a USB device from the strings that the hardware161// itself carries. These strings live on the device, but the162// undriven part is usually one of the device's interfaces. Leaf163// drivers bind interfaces, and the device node itself always164// belongs to usbcore, so an interface borrows its parent's name. The165// parent's directory name is the interface's directory name minus166// the :config.interface suffix, which is the USB sysfs naming167// convention.168func usbName(busDir, name string) string {169	parent, _, isInterface := strings.Cut(name, ":")170	dir := filepath.Join(busDir, name)171	if isInterface {172		dir = filepath.Join(busDir, parent)173	}174	words := strings.TrimSpace(175		readAttr(dir, "manufacturer") + " " + readAttr(dir, "product"))176	return words177}178179// parentAttr reads a USB attribute from the parent device when the180// entry is an interface. This works for the same reason that usbName181// borrows the parent's strings: identity attributes such as serial,182// idVendor, and idProduct live on the device, while the driver binds183// the interface.184func parentAttr(busDir, name, attr string) string {185	parent, _, isInterface := strings.Cut(name, ":")186	dir := filepath.Join(busDir, name)187	if isInterface {188		dir = filepath.Join(busDir, parent)189	}190	return readAttr(dir, attr)191}
hardware/uevent.go 89.7%
1package hardware23// Uevents are how the kernel reports that hardware changed.4//5// The kernel broadcasts every device add, remove, and driver bind on6// a netlink socket (NETLINK_KOBJECT_UEVENT, multicast group 1). This7// is the same channel that udev listens on, where udev exists. Each8// datagram is "action@devpath" followed by KEY=VALUE pairs,9// including the MODALIAS fingerprint, but this listener deliberately10// does not read what the event says about the device. A uevent only11// signals that something changed. The sysfs walk re-reads the whole12// state moments later. This is simpler and more accurate than13// incrementally mirroring kernel state from event payloads, because a14// mirror can drift out of sync, while a re-walk cannot. The listener15// reads where the event happened, the device path and the subsystem,16// so that a caller can drop the events that cannot change what its17// walk reads.1819import (20	"bytes"21	"context"22	"errors"23	"fmt"2425	"golang.org/x/sys/unix"26)2728// ListenForUevents opens the kernel's uevent socket and returns a29// channel. This channel signals whenever a device appears,30// disappears, or changes drivers. The channel holds one pending31// signal and drops the rest. For example, a burst of eleven uevents32// from one USB stick's enumeration needs one re-walk, not eleven.33//34// The channel closes when the listener stops for any reason other35// than the end of the context. A listener that stops loses every36// event after it, so a caller that receives from a closed channel37// opens the listener again and walks sysfs again, because changes may38// have happened while nothing listened.39//40// The socket is non-blocking. The reader waits for it in poll, not in41// a read, so it can also watch a cancel pipe in the same poll and stop42// the moment the context ends. See watchUevents and readUevents for the43// wake and the stop.44func ListenForUevents(ctx context.Context) (<-chan struct{}, error) {45	return ListenForUeventsMatching(ctx, nil)46}4748// ListenForUeventsMatching is ListenForUevents with one more test: a49// uevent wakes the channel only when match also accepts it. A nil50// match accepts every uevent. A lost datagram still wakes the channel,51// because the listener cannot tell what the lost datagram was.52func ListenForUeventsMatching(ctx context.Context, match func(Uevent) bool) (<-chan struct{}, error) {53	fd, err := unix.Socket(unix.AF_NETLINK, unix.SOCK_DGRAM|unix.SOCK_CLOEXEC|unix.SOCK_NONBLOCK, unix.NETLINK_KOBJECT_UEVENT)54	if err != nil {55		return nil, fmt.Errorf("opening the uevent socket: %w", err)56	}57	// Group 1 carries the kernel's own broadcasts. Group 2 carries58	// udev's re-broadcasts to libudev clients, but on a liken59	// machine, nothing sends to group 2.60	if err := unix.Bind(fd, &unix.SockaddrNetlink{Family: unix.AF_NETLINK, Groups: 1}); err != nil {61		unix.Close(fd)62		return nil, fmt.Errorf("binding the uevent socket: %w", err)63	}64	notify, err := watchUevents(ctx, fd, match)65	if err != nil {66		unix.Close(fd)67		return nil, err68	}69	return notify, nil70}7172// watchUevents starts the reader over the non-blocking socket fd and73// returns its wake channel. It owns fd from this point: the reader74// closes it on exit.75//76// Cancellation cannot rely on closing fd, because a close does not wake77// a thread already blocked in a read on that descriptor. So the reader78// never blocks in a read. It waits in poll over fd and the read end of79// a cancel pipe. When the context is done, a second goroutine closes80// the pipe's write end. That close puts a hangup on the read end, the81// poll wakes, and the reader returns. This split of ownership closes82// every descriptor once: the cancel goroutine closes the write end, and83// the reader closes fd and the read end as it leaves.84func watchUevents(ctx context.Context, fd int, match func(Uevent) bool) (<-chan struct{}, error) {85	var pipe [2]int86	if err := unix.Pipe2(pipe[:], unix.O_CLOEXEC|unix.O_NONBLOCK); err != nil {87		return nil, fmt.Errorf("opening the cancel pipe: %w", err)88	}89	notify := make(chan struct{}, 1)90	// The cancel goroutine also ends when the reader stops on its own,91	// so a listener that failed leaves no goroutine and no descriptor92	// behind for the life of the context.93	done := make(chan struct{})94	go func() {95		select {96		case <-ctx.Done():97		case <-done:98		}99		unix.Close(pipe[1])100	}()101	go func() {102		readUevents(fd, pipe[0], match, notify)103		close(done)104	}()105	return notify, nil106}107108// readUevents is the reader loop. It blocks in poll over the uevent109// socket and the cancel pipe. A ready socket means a datagram to read;110// a ready cancel pipe means the context is done and the loop returns. It111// closes the descriptors it owns as it leaves.112//113// Every other way out closes notify as well: a poll error, and a socket114// that poll reports is not open (POLLNVAL) or hung up (POLLHUP). The115// reader cannot recover from any of them, and it would spin on the last116// two, because poll keeps reporting them at once. A netlink socket never117// hangs up while it is open, and an overflow reports POLLERR, which the118// read below answers with ENOBUFS, so neither ends the reader. Only this119// goroutine sends on notify, so the close cannot race a send.120func readUevents(fd, cancelR int, match func(Uevent) bool, notify chan<- struct{}) {121	defer unix.Close(fd)122	defer unix.Close(cancelR)123	buf := make([]byte, 64<<10)124	fds := []unix.PollFd{125		{Fd: int32(fd), Events: unix.POLLIN},126		{Fd: int32(cancelR), Events: unix.POLLIN},127	}128	for {129		_, err := unix.Poll(fds, -1)130		if errors.Is(err, unix.EINTR) {131			// A signal interrupted the wait. Wait again.132			continue133		}134		if err == nil && fds[1].Revents != 0 {135			// The cancel pipe reports a hangup. The context is done.136			return137		}138		if err != nil || fds[0].Revents&(unix.POLLNVAL|unix.POLLHUP) != 0 {139			close(notify)140			return141		}142		size, _, err := unix.Recvfrom(fd, buf, 0)143		if err != nil {144			if recvErrorLostAUevent(err) {145				// ENOBUFS means the kernel's receive buffer overflowed and146				// it dropped datagrams before this call ever ran; any other147				// error here is one this reader has no known cause for on148				// this socket, and either way poll reported the socket149				// ready and this call still returned no datagram. Hardware150				// may have changed with the loss, so wake the sysfs walk151				// now rather than wait for an unrelated later uevent to152				// trigger it by accident.153				wake(notify)154			}155			// EAGAIN means the poll woke without a datagram to read, and156			// EINTR means a signal interrupted the read before it took157			// anything. Neither one loses a datagram, so wait for the158			// next event with no wake.159			continue160		}161		event, ok := parseUevent(buf[:size])162		if !ok || !hardwareChanged(event) || (match != nil && !match(event)) {163			continue164		}165		wake(notify)166	}167}168169// recvErrorLostAUevent reports whether a Recvfrom error on the uevent170// socket left a datagram unread. EAGAIN means poll woke with nothing171// queued, and EINTR means a signal interrupted the call before it read172// anything; neither one loses a datagram. Every other error does, most173// often ENOBUFS: poll reported the socket ready, and the call still did174// not return a datagram.175func recvErrorLostAUevent(err error) bool {176	return !errors.Is(err, unix.EAGAIN) && !errors.Is(err, unix.EINTR)177}178179// wake does one non-blocking send on the notify channel. The channel180// holds one pending wake, so a send while one is already queued is181// dropped: the walk that drains the channel reads the whole state, so182// coalescing loses nothing.183func wake(notify chan<- struct{}) {184	select {185	case notify <- struct{}{}:186	default:187	}188}189190// A Uevent is the part of one kernel uevent that the listener reads:191// what happened, to which device, in which subsystem. DevPath is the192// device's path under /sys, such as193// /devices/pci0000:00/0000:00:03.0/usb1/1-2.194type Uevent struct {195	Action    string196	DevPath   string197	Subsystem string198}199200// parseUevent reads one datagram from the kernel. The datagram is201// "action@devpath" and then KEY=VALUE pairs, each ended by a NUL. A202// datagram with no "@" in its first field is not the kernel's: udev's203// libudev format starts with "libudev".204func parseUevent(datagram []byte) (Uevent, bool) {205	fields := bytes.Split(datagram, []byte{0})206	action, devpath, found := bytes.Cut(fields[0], []byte("@"))207	if !found {208		return Uevent{}, false209	}210	event := Uevent{Action: string(action), DevPath: string(devpath)}211	for _, field := range fields[1:] {212		if value, ok := bytes.CutPrefix(field, []byte("SUBSYSTEM=")); ok {213			event.Subsystem = string(value)214		}215	}216	return event, true217}218219// hardwareChanged reports whether one uevent requires a re-walk. Add220// and remove events change what exists. Bind and unbind events change221// which driver is bound. Every other event, such as change, move, or222// the online and offline events for memory blocks, changes nothing223// that this package reports.224func hardwareChanged(event Uevent) bool {225	switch event.Action {226	case "add", "remove", "bind", "unbind":227		return true228	}229	return false230}
hardware/unclaimed.go 77.6%
1package hardware23// This file builds the unclaimed report. It judges which undriven4// devices are worth an operator's attention, and states what would5// fix each one.6//7// Not every driverless device is a problem. Sysfs contains many8// devices that legitimately have no driver, such as bridges, stubs,9// and firmware-described devices. A report that listed all of them10// would hide the one entry that matters, such as a plugged-in disk11// that nobody can use, among entries that nobody can act on. The12// filter is actionability. A device appears in the report exactly13// when some loadable module in the kernel build could drive it,14// because that is exactly when an edit to spec.modules, or a release15// that ships the module, would change something.1617import (18	"os"19	"path/filepath"20	"slices"21	"strings"2223	"github.com/liken-sh/liken/liken/machine"24)2526// Catalog holds everything needed to judge a device. It holds the27// kernel build's complete alias table, which lists the modules that28// could drive a device. It holds the shipped set, which lists which29// of those modules this image carries. It holds the builtin set,30// which lists the drivers that are resident in vmlinuz and need no31// loading. It optionally holds the PCI naming database. The catalog32// loads once at boot, and the judgments run again each time the33// hardware changes.34type Catalog struct {35	Aliases *AliasTable36	Shipped map[string]bool37	Builtin map[string]bool38	PCI     *PCIIDs39}4041// LoadCatalog assembles the catalog from a module directory42// (/lib/modules/<release>) and a pci.ids path. Only the alias table43// is essential. Without it, no candidate can be named, and the44// report would produce nothing. For this reason, only a missing45// alias table is an error. When the naming database is missing, it46// falls back to numeric IDs.47func LoadCatalog(moduleDir, pciIDsPath string) (*Catalog, error) {48	aliases, err := LoadAliasTable(filepath.Join(moduleDir, "modules.alias"))49	if err != nil {50		return nil, err51	}52	c := &Catalog{53		Aliases: aliases,54		Shipped: LoadShippedModules(moduleDir),55		Builtin: LoadModuleSet(filepath.Join(moduleDir, "modules.builtin")),56	}57	c.PCI, _ = LoadPCIIDs(pciIDsPath)58	return c, nil59}6061// Discover walks sysfs and returns the machine's current unclaimed62// devices. init makes this one call at boot, and again each time a63// uevent reports that the hardware changed. serio is the machine's64// spec.serio list, which decides whether a serial-line adapter is65// finished (unclaimedserio.go).66func (c *Catalog) Discover(sysRoot string, serio []machine.SerioAttachment) []machine.UnclaimedDevice {67	return c.Unclaimed(DiscoverDevices(sysRoot, c.PCI), serio)68}6970// Unclaimed judges a walked device list. Unclaimed reports a device71// when nothing drives it, and at least one loadable module's alias72// patterns match its fingerprint. Unclaimed excludes candidates that73// are already built into the kernel, because they are resident and74// loading is not the missing step. Unclaimed skips a device with no75// loadable candidate at all, because no action can fix it. Unclaimed76// sorts the result so that the report stays stable across walks. A77// status object that reorders on every read causes watches to fire78// for no reason.79//80// A driven device joins the report in one case: a serial-line adapter81// that works only after an attachment, which no spec.serio entry82// declares (unclaimedserio.go).83func (c *Catalog) Unclaimed(devices []Device, serio []machine.SerioAttachment) []machine.UnclaimedDevice {84	var unclaimed []machine.UnclaimedDevice85	for _, d := range devices {86		if d.Driver != "" {87			if u, ok := unattachedAdapter(d, serio); ok {88				unclaimed = append(unclaimed, u)89			}90			continue91		}92		var candidates, aboard []string93		for _, module := range c.Aliases.Candidates(d.Modalias) {94			key := strings.ReplaceAll(module, "-", "_")95			if c.Builtin[key] {96				continue97			}98			candidates = append(candidates, module)99			if c.Shipped[key] {100				aboard = append(aboard, module)101			}102		}103		if len(candidates) == 0 {104			continue105		}106		unclaimed = append(unclaimed, machine.UnclaimedDevice{107			Modalias:   d.Modalias,108			Bus:        d.Bus,109			Name:       d.Name,110			Class:      d.Class,111			Candidates: candidates,112			Message:    unclaimedMessage(candidates, aboard),113		})114	}115	slices.SortFunc(unclaimed, func(a, b machine.UnclaimedDevice) int {116		if a.Bus != b.Bus {117			return strings.Compare(a.Bus, b.Bus)118		}119		return strings.Compare(a.Modalias, b.Modalias)120	})121	return unclaimed122}123124// unclaimedMessage states the fix in words. When the image carries125// a candidate module, the fix is an edit, and the message names only126// the modules that a person can actually declare. On a stock image127// this is every candidate, because the image carries the kernel's128// whole module tree. When the image carries none (a composed image129// can remove modules), the fix is a different image, and the message130// still names the modules, so a person can find an image that131// carries them.132func unclaimedMessage(candidates, aboard []string) string {133	if len(aboard) > 0 {134		return "declare " + strings.Join(aboard, " or ") + " in spec.modules"135	}136	verb := "carries neither"137	if len(candidates) == 1 {138		verb = "doesn't carry it"139	} else if len(candidates) > 2 {140		verb = "carries none of them"141	}142	return strings.Join(candidates, " or ") +143		" would drive it, but this image " + verb + "; use an image that does"144}145146// LoadShippedModules finds which modules are actually present on147// the machine. It does this by checking, with stat, every file that148// modules.dep names, instead of trusting the index. This distinction149// matters on composed systems. A deployment layer that adds modules150// ships the kernel's complete index, so that dependency resolution151// covers everything present. This makes index membership mean152// "exists in the kernel build", not "exists on this machine".153// Telling a person to declare a module that the machine does not154// carry would send them through the same failed fix twice.155func LoadShippedModules(moduleDir string) map[string]bool {156	set := map[string]bool{}157	raw, err := os.ReadFile(filepath.Join(moduleDir, "modules.dep"))158	if err != nil {159		return set160	}161	for line := range strings.SplitSeq(string(raw), "\n") {162		path, _, found := strings.Cut(line, ":")163		if !found {164			continue165		}166		if _, err := os.Stat(filepath.Join(moduleDir, path)); err != nil {167			continue168		}169		name := strings.TrimSuffix(filepath.Base(path), ".zst")170		name = strings.TrimSuffix(name, ".ko")171		set[strings.ReplaceAll(name, "-", "_")] = true172	}173	return set174}175176// LoadModuleSet reads a depmod module list into a set of normalized177// names. The list can be modules.builtin, or any file of module178// paths, one per line, with optional ':'-terminated fields. A179// normalized name is the file's base name with its extensions180// dropped, and its hyphens changed to underscores. This is the same181// equivalence that the kernel applies. A missing file reads as an182// empty set. Judgment then degrades, so that every module looks183// unshipped, but reporting still continues.184func LoadModuleSet(path string) map[string]bool {185	set := map[string]bool{}186	raw, err := os.ReadFile(path)187	if err != nil {188		return set189	}190	for line := range strings.SplitSeq(string(raw), "\n") {191		name, _, _ := strings.Cut(line, ":")192		name = strings.TrimSpace(name)193		if name == "" {194			continue195		}196		name = strings.TrimSuffix(filepath.Base(name), ".zst")197		name = strings.TrimSuffix(name, ".ko")198		set[strings.ReplaceAll(name, "-", "_")] = true199	}200	return set201}
hardware/unclaimedserio.go 100.0%
1package hardware23// The unclaimed report's one entry for a device that has a driver.4//5// A USB-CEC adapter shows why the report needs it. Before cdc_acm6// binds the adapter, the ordinary report lists it and names cdc_acm.7// After cdc_acm binds it, the adapter has a driver and a serial line,8// and it still does nothing: its real driver is a serio driver, which9// binds only after init attaches the line (machine/serio.go). The10// ordinary rule drops a driven device, so without this entry the11// adapter would leave the report at the moment it is half done.12//13// A pod can also drive such an adapter itself over the tty, as libcec14// does, and then the adapter needs no entry. The report cannot tell15// the two uses apart, so its message names both, and the entry stays16// listed for an adapter a pod drives that way.17//18// The entry is narrow on purpose. It lists only an identity that the19// protocol table names as a known adapter, only the interface that20// owns the serial line, and only while no spec.serio entry matches21// the device. Any other driven device is working equipment.2223import (24	"fmt"2526	"github.com/liken-sh/liken/liken/machine"27)2829// cdcCommunicationsClass is the USB interface class of a CDC30// communications interface. A CDC ACM device has two interfaces that31// cdc_acm binds: the communications interface, which owns the tty,32// and the data interface beside it. The report lists the first, so33// one adapter is one entry.34const cdcCommunicationsClass = "02"3536// unattachedAdapter reports the entry for one driven device, when the37// device is a known serial-line adapter that no spec.serio entry38// attaches.39func unattachedAdapter(d Device, serio []machine.SerioAttachment) (machine.UnclaimedDevice, bool) {40	if d.Bus != "usb" || d.ClassCode != cdcCommunicationsClass {41		return machine.UnclaimedDevice{}, false42	}43	protocol, ok := machine.LookupSerioProtocol(machine.KnownSerioAdapter(d.Vendor, d.Product))44	if !ok || d.Driver != protocol.LineDriver {45		return machine.UnclaimedDevice{}, false46	}47	for _, entry := range serio {48		if entry.Matches(d.Vendor, d.Product, d.Serial) {49			return machine.UnclaimedDevice{}, false50		}51	}52	return machine.UnclaimedDevice{53		Modalias: d.Modalias,54		Bus:      d.Bus,55		Name:     d.Name,56		Class:    d.Class,57		// The entry names no candidates. The alias table did not put58		// it here, and the fix is two modules and a spec.serio entry59		// together, which the message states whole. The fix is only60		// one of two ways to use the adapter: a pod can also drive it61		// over the tty, as libcec does, and an entry withholds the tty,62		// so the message says which use it is for.63		Message: fmt.Sprintf("for the kernel CEC driver, declare %s and %s in spec.modules and a %s entry in spec.serio; "+64			"the entry withholds the tty from workloads, so leave it out when a pod drives the adapter over its serial line",65			protocol.Discipline, protocol.Driver, protocol.Name),66	}, true67}
identity/adopt.go 79.4%
1package identity23// This file implements adoption: taking on an existing cluster's4// identity. Adoption is the reverse of minting.5//6// The image carries the cluster's certificate authorities and join7// token. This is why machines built from the same image belong to8// the same cluster. Minting creates that identity before any machine9// exists, which works for a cluster that liken founds. For a cluster10// that liken did not create, such as any existing k3s cluster11// however it was set up, the identity already exists on that12// cluster's servers. It must be copied off one of them instead.13// Adoption takes that copy and places it into the deployment's14// identity directory exactly as minting would have. Because of this,15// everything downstream (the kubeconfig, the image build, and16// init's seeding of /var/lib/rancher/k3s/server/tls) is identical17// whether the identity was minted or adopted. An image built from an18// adopted identity joins the existing cluster. Its machines present19// the real token, and every certificate they see chains to the real20// CAs.21//22// To harvest the identity, run this as root on any server of the23// existing cluster:24//25//	cd /var/lib/rancher/k3s/server26//	tar czf /tmp/identity.tgz token \27//	    tls/server-ca.{crt,key} \28//	    tls/client-ca.{crt,key} \29//	    tls/request-header-ca.{crt,key} \30//	    tls/service.key \31//	    tls/etcd/server-ca.{crt,key} \32//	    tls/etcd/peer-ca.{crt,key}33//34// Then unpack that archive somewhere private, and point adoption at35// the directory. Only the certificate authorities and the token come36// over. The tls directory on a live server also holds the leaf37// certificates that k3s signed from them, such as the API server's38// serving certificate and the kubelet certificates. Those stay39// behind, because every server signs its own leaf certificates from40// the shared roots. The service.key file is included for the same41// reason it exists in minting. It signs every ServiceAccount token,42// and a control plane that verified tokens against a different key43// would reject every pod's identity.4445import (46	"crypto/sha256"47	"fmt"48	"io"49	"os"50	"path/filepath"51	"strings"52)5354// Adopt copies a harvested identity from harvest into dir. Adopt55// refuses anything that is not a complete, self-consistent bundle56// placed over an empty deployment.57func Adopt(harvest, dir string, out io.Writer) error {58	// This code refuses a partial harvest before it changes anything.59	// A bundle that is missing one CA would produce an image that60	// boots, and then fails later in a way that is hard to trace,61	// such as a control plane that cannot sign one kind of62	// certificate, or pods whose tokens do not verify.63	for _, f := range Bundle {64		if _, err := os.Stat(filepath.Join(harvest, f)); err != nil {65			return fmt.Errorf("harvest is missing %s; re-run the tar on the existing server", f)66		}67	}6869	// This code cross-checks the token: the token's embedded CA hash70	// must match the harvested server CA. The token file on a running71	// server is in k3s's "secure" format, K10<CA-HASH>::<user>:72	// <password>, where CA-HASH is the SHA256 hash of the cluster CA73	// certificate. This code checks that hash here. This check74	// catches a token harvested from one cluster mixed with CAs from75	// another, a mixup that would otherwise surface only later, as76	// every machine refuses to join. This is the same verification77	// that a joining machine performs before it trusts an endpoint,78	// done early, where the fix (re-harvest, from one server this79	// time) is cheap.80	token, err := os.ReadFile(filepath.Join(harvest, "token"))81	if err != nil {82		return err83	}84	if strings.HasPrefix(string(token), "K10") {85		crt, err := os.ReadFile(filepath.Join(harvest, "tls", "server-ca.crt"))86		if err != nil {87			return err88		}89		want := fmt.Sprintf("K10%x::", sha256.Sum256(crt))90		if !strings.HasPrefix(string(token), want) {91			return fmt.Errorf("the harvested token does not hash the harvested server CA; these came from different clusters")92		}93	}9495	// Replacing an identity is a deliberate act, the same as minting.96	// An image built from a mix of two identities could not join97	// either cluster. If the identity directory holds any file from98	// the bundle, this code stops and makes the operator choose.99	for _, f := range Bundle {100		if _, err := os.Stat(filepath.Join(dir, f)); err == nil {101			return fmt.Errorf("%s already exists; this deployment already holds an identity — delete the identity directory first if replacing it is really the intent", filepath.Join(dir, f))102		}103	}104105	for _, f := range Bundle {106		if err := copyPrivately(filepath.Join(harvest, f), filepath.Join(dir, f)); err != nil {107			return err108		}109		fmt.Fprintf(out, "adopted %s\n", f)110	}111112	// Private keys and the token are secrets, but the certificates113	// are not. Restricting the whole tree, including the identity114	// directory, is simpler than listing which paths need the115	// restriction.116	err = filepath.Walk(dir, func(path string, info os.FileInfo, err error) error {117		if err != nil {118			return err119		}120		return os.Chmod(path, info.Mode().Perm()&^0o077)121	})122	if err != nil {123		return err124	}125126	fmt.Fprintln(out, "the identity is adopted: images built from it join the existing cluster")127	return nil128}129130// copyPrivately copies one file, creating parents as needed.131func copyPrivately(src, dst string) error {132	data, err := os.ReadFile(src)133	if err != nil {134		return err135	}136	if err := os.MkdirAll(filepath.Dir(dst), 0o700); err != nil {137		return err138	}139	return os.WriteFile(dst, data, 0o600)140}
identity/kubeconfig.go 79.7%
1package identity23// This file computes the operator's kubeconfig offline from a4// deployment's identity. This code never asks the machine for a5// credential. Pre-seeding the CAs (mint.go) exists precisely so that6// this code can compute the credential without contacting the7// machine.8//9// A kubeconfig states three facts:10//11//  1. where the cluster is (a URL),12//  2. why to trust that this is really the cluster (the server CA13//     that signed its serving certificate),14//  3. who the client is (a client certificate that the cluster's15//     client CA signed).16//17// The identity in a client certificate lives in its subject. The API18// server reads CN as the username, and every O as a group. No user19// database exists behind this. Presenting a certificate with20// O=system:masters makes the bearer a cluster admin, because RBAC21// binds that group to cluster-admin. The certificates themselves are22// the only user records.2324import (25	"crypto/ecdsa"26	"crypto/elliptic"27	"crypto/rand"28	"crypto/x509"29	"crypto/x509/pkix"30	"encoding/base64"31	"encoding/pem"32	"fmt"33	"io"34	"math/big"35	"os"36	"path/filepath"37	"time"38)3940// Kubeconfig computes an admin credential from the identity in dir,41// and writes dir/kubeconfig, pointed at server. The caller resolves42// server: the CLI reads it from the deployment's cluster.yaml, or43// takes it from the operator's -server flag when that document's44// endpoint is not reachable from the workstation. The caller also45// resolves name, the cluster's own name from the same document,46// which labels the cluster and context entries in the file. Kubeconfig writes47// the result into the identity directory and nowhere else. liken48// never changes ~/.kube/config or any other kubeconfig that the49// operator already has. Point kubectl at the file explicitly:50//51//	kubectl --kubeconfig dev-cluster/identity/kubeconfig get nodes52//53// Each run mints a fresh keypair and certificate. The certificate54// lasts one year, which is generous for a development credential.55// Running Kubeconfig again replaces it in seconds.56func Kubeconfig(dir, name, server string, out io.Writer) error {57	tls := filepath.Join(dir, "tls")5859	caCert, caKey, err := readCA(tls, "client-ca")60	if err != nil {61		return err62	}6364	// This code creates the client identity: a fresh keypair, and a65	// certificate for it signed by the client CA. A client66	// certificate needs the clientAuth extended key usage. The API67	// server rejects certificates that do not declare their purpose.68	key, err := ecdsa.GenerateKey(elliptic.P256(), rand.Reader)69	if err != nil {70		return err71	}72	serial, err := rand.Int(rand.Reader, new(big.Int).Lsh(big.NewInt(1), 128))73	if err != nil {74		return err75	}76	now := time.Now()77	template := &x509.Certificate{78		SerialNumber: serial,79		Subject: pkix.Name{80			CommonName:   "admin",81			Organization: []string{"system:masters"},82		},83		NotBefore:   now,84		NotAfter:    now.AddDate(0, 0, 365),85		KeyUsage:    x509.KeyUsageDigitalSignature,86		ExtKeyUsage: []x509.ExtKeyUsage{x509.ExtKeyUsageClientAuth},87	}88	der, err := x509.CreateCertificate(rand.Reader, template, caCert, &key.PublicKey, caKey)89	if err != nil {90		return err91	}9293	sec1, err := x509.MarshalECPrivateKey(key)94	if err != nil {95		return err96	}97	adminCert := pem.EncodeToMemory(&pem.Block{Type: "CERTIFICATE", Bytes: der})98	adminKey := pem.EncodeToMemory(&pem.Block{Type: "EC PRIVATE KEY", Bytes: sec1})99100	serverCA, err := os.ReadFile(filepath.Join(tls, "server-ca.crt"))101	if err != nil {102		return err103	}104105	// This is the kubeconfig itself, with the certificates embedded106	// in base64, as every kubeconfig does. This makes the file107	// self-contained and portable.108	//109	// The cluster and context entries carry the cluster's own name,110	// because these labels are how kubectl tells clusters apart when111	// an operator merges kubeconfigs from more than one of them. The112	// user entry stays admin: the certificate's subject, not this113	// label, is the identity the API server reads.114	b64 := base64.StdEncoding.EncodeToString115	kubeconfig := fmt.Sprintf(`apiVersion: v1116kind: Config117clusters:118  - name: %[1]s119    cluster:120      server: %[2]s121      certificate-authority-data: %[3]s122contexts:123  - name: %[1]s124    context:125      cluster: %[1]s126      user: admin127current-context: %[1]s128users:129  - name: admin130    user:131      client-certificate-data: %[4]s132      client-key-data: %[5]s133`, name, server, b64(serverCA), b64(adminCert), b64(adminKey))134135	path := filepath.Join(dir, "kubeconfig")136	if err := os.WriteFile(path, []byte(kubeconfig), 0o600); err != nil {137		return err138	}139	fmt.Fprintf(out, "wrote %s for admin (O=system:masters) at %s\n", path, server)140	return nil141}142143// readCA loads one authority's certificate and private key from the144// tls tree.145func readCA(tls, name string) (*x509.Certificate, *ecdsa.PrivateKey, error) {146	certPEM, err := os.ReadFile(filepath.Join(tls, name+".crt"))147	if err != nil {148		return nil, nil, err149	}150	block, _ := pem.Decode(certPEM)151	if block == nil {152		return nil, nil, fmt.Errorf("%s.crt is not PEM", name)153	}154	cert, err := x509.ParseCertificate(block.Bytes)155	if err != nil {156		return nil, nil, err157	}158159	keyPEM, err := os.ReadFile(filepath.Join(tls, name+".key"))160	if err != nil {161		return nil, nil, err162	}163	block, _ = pem.Decode(keyPEM)164	if block == nil {165		return nil, nil, fmt.Errorf("%s.key is not PEM", name)166	}167	key, err := x509.ParseECPrivateKey(block.Bytes)168	if err != nil {169		return nil, nil, err170	}171	return cert, key, nil172}
identity/mint.go 78.7%
1// Package identity mints, adopts, and derives credentials from a2// deployment's identity. An identity is the certificate authorities3// and join token that make machines built from the same image4// members of the same cluster.5//6// Kubernetes trust is built from several small PKIs, each one7// covering a single relationship. k3s checks for these files before8// it generates its own. Placing them in9// /var/lib/rancher/k3s/server/tls before first start reverses the10// usual flow. Normally, the identity is an output that must be11// extracted from a running machine, something a machine with no12// shell could never provide. Here, the identity is an input that the13// image carries. Everything that k3s signs from this point on,14// including the API server's serving certificate and the kubelet15// certificates, chains up to keys that existed before the machine16// ever booted. This is what lets an operator's kubeconfig be17// computed offline (see kubeconfig.go).18//19// An identity belongs to a deployment, not to the OS. This package20// produces an identity, and the caller names the deployment it21// belongs to. The files are private keys and must never enter22// version-control history. Deployment directories list them in23// .gitignore.24package identity2526import (27	"crypto/ecdsa"28	"crypto/elliptic"29	"crypto/rand"30	"crypto/sha256"31	"crypto/x509"32	"crypto/x509/pkix"33	"encoding/hex"34	"encoding/pem"35	"fmt"36	"io"37	"math/big"38	"os"39	"path/filepath"40	"time"41)4243// Bundle lists the identity as paths relative to its directory. It44// lists everything that mint produces and adopt copies, and nothing45// more. The token lives beside the tls tree, not inside it, matching46// where k3s keeps its own token (/var/lib/rancher/k3s/server/token).47// The image package reads this list to place an identity into a48// deployment layer, so a new artifact added here reaches the49// machines with no other change. The kubeconfig is deliberately not50// part of the bundle. It is the operator's credential, and it must51// never be included in an image.52var Bundle = []string{53	"token",54	"tls/server-ca.crt", "tls/server-ca.key",55	"tls/client-ca.crt", "tls/client-ca.key",56	"tls/request-header-ca.crt", "tls/request-header-ca.key",57	"tls/service.key",58	"tls/etcd/server-ca.crt", "tls/etcd/server-ca.key",59	"tls/etcd/peer-ca.crt", "tls/etcd/peer-ca.key",60}6162// Mint creates the identity's artifacts in dir, and keeps any that63// already exist. The authorities:64//65//	server-ca          signs the API server's serving certificates.66//	                   kubectl checks this signature before it trusts67//	                   a connection.68//	client-ca          signs client certificates. The API server reads69//	                   the identity from the certificate subject. CN70//	                   is the username, and each O is a group.71//	request-header-ca  is the aggregation layer's trust root.72//	                   Extension API servers accept73//	                   proxied-authentication headers only from a74//	                   front proxy that presents a certificate from75//	                   this CA.76//	etcd/server-ca     are etcd's two PKIs. liken's k3s keeps its77//	etcd/peer-ca       state in sqlite through kine, not etcd, but k3s78//	                   still manages the full CA family as a set.79//	service.key        is not a CA. It is the key that signs every80//	                   ServiceAccount token. Whoever holds this key81//	                   can mint valid identities for any pod.82//83// Every key is ECDSA P-256, matching what k3s generates for itself.84// Every certificate has a ten-year lifetime. Rotating these roots is85// work that a later milestone will take up. Until then, a long86// lifetime keeps the identity simple to manage.87//88// Mint creates each artifact only if it does not already exist.89// Because of this, adding a new artifact, or running Mint again for90// any reason, never replaces an identity that machines already91// carry. Replacing the CAs would break every kubeconfig computed92// from them, and replacing the token would prevent any machine that93// has not joined yet from joining. Replacing the identity is a94// deliberate act. Delete the identity directory, and the next call95// to Mint creates a new one.96func Mint(dir string, out io.Writer) error {97	tls := filepath.Join(dir, "tls")98	if err := os.MkdirAll(filepath.Join(tls, "etcd"), 0o755); err != nil {99		return err100	}101102	authorities := []struct {103		path string104		cn   string105	}{106		{"server-ca", "liken server CA"},107		{"client-ca", "liken client CA"},108		{"request-header-ca", "liken request-header CA"},109		{"etcd/server-ca", "liken etcd server CA"},110		{"etcd/peer-ca", "liken etcd peer CA"},111	}112	for _, ca := range authorities {113		if _, err := os.Stat(filepath.Join(tls, ca.path+".crt")); err == nil {114			fmt.Fprintf(out, "keeping %s: %s\n", ca.path, ca.cn)115			continue116		}117		if err := newCA(tls, ca.path, ca.cn); err != nil {118			return fmt.Errorf("minting %s: %w", ca.path, err)119		}120		fmt.Fprintf(out, "minted %s: %s\n", ca.path, ca.cn)121	}122123	// The ServiceAccount signing key differs from the CAs above.124	// Tokens are JWTs, so there is no certificate, only a keypair125	// that the API server signs with and verifies against. The126	// encoding matters. kube-apiserver reads this file with a parser127	// that understands the older SEC1 encoding ("EC PRIVATE KEY") but128	// not PKCS#8 ("PRIVATE KEY"). Given a PKCS#8 file, kube-apiserver129	// fails at startup with an error that says the file contains no130	// valid keys.131	serviceKey := filepath.Join(tls, "service.key")132	if _, err := os.Stat(serviceKey); err == nil {133		fmt.Fprintln(out, "keeping service.key: the ServiceAccount token signing key")134	} else {135		key, err := ecdsa.GenerateKey(elliptic.P256(), rand.Reader)136		if err != nil {137			return err138		}139		sec1, err := x509.MarshalECPrivateKey(key)140		if err != nil {141			return err142		}143		if err := writePEM(serviceKey, "EC PRIVATE KEY", sec1, 0o600); err != nil {144			return err145		}146		fmt.Fprintln(out, "minted service.key: the ServiceAccount token signing key")147	}148149	// This is the cluster's join token, in k3s's "secure" format:150	//151	//	K10<CA-HASH>::<user>:<password>152	//153	// Normally, an operator must copy this token off a running154	// server (/var/lib/rancher/k3s/server/node-token), because the CA155	// that it hashes does not exist until k3s generates it at first156	// boot. liken reverses that order. The server CA is minted above,157	// before any machine exists, so this code can compute the whole158	// token right here. CA-HASH is the SHA256 hash of the cluster CA159	// certificate file. A joining machine fetches the server's CA160	// bundle, hashes it, and compares the hash before it trusts the161	// endpoint or presents the secret. Because of this, the token162	// authenticates in both directions: the machine proves its163	// identity to the cluster, and the cluster proves its identity to164	// the machine. The secret half is 32 hex characters of real165	// randomness, the same format that k3s generates. "server" is the166	// credential's username. Whoever holds this token may join167	// machines to the cluster.168	tokenPath := filepath.Join(dir, "token")169	if _, err := os.Stat(tokenPath); err == nil {170		fmt.Fprintln(out, "keeping token: the cluster join token")171	} else {172		crt, err := os.ReadFile(filepath.Join(tls, "server-ca.crt"))173		if err != nil {174			return err175		}176		secret := make([]byte, 16)177		if _, err := rand.Read(secret); err != nil {178			return err179		}180		token := fmt.Sprintf("K10%x::server:%s\n", sha256.Sum256(crt), hex.EncodeToString(secret))181		if err := os.WriteFile(tokenPath, []byte(token), 0o600); err != nil {182			return err183		}184		fmt.Fprintln(out, "minted token: the cluster join token")185	}186187	return nil188}189190// newCA creates one self-signed root. It creates a fresh P-256 key191// and a certificate whose extensions mark it as a CA that can sign192// other certificates.193func newCA(tls, path, cn string) error {194	key, err := ecdsa.GenerateKey(elliptic.P256(), rand.Reader)195	if err != nil {196		return err197	}198199	// Serial numbers must be unique for each issuer. In practice,200	// every CA satisfies this requirement with 128 random bits,201	// without keeping state.202	serial, err := rand.Int(rand.Reader, new(big.Int).Lsh(big.NewInt(1), 128))203	if err != nil {204		return err205	}206207	now := time.Now()208	template := &x509.Certificate{209		SerialNumber:          serial,210		Subject:               pkix.Name{CommonName: cn},211		NotBefore:             now,212		NotAfter:              now.AddDate(0, 0, 3650),213		IsCA:                  true,214		BasicConstraintsValid: true,215		KeyUsage: x509.KeyUsageDigitalSignature |216			x509.KeyUsageCertSign |217			x509.KeyUsageCRLSign,218	}219	der, err := x509.CreateCertificate(rand.Reader, template, template, &key.PublicKey, key)220	if err != nil {221		return err222	}223224	sec1, err := x509.MarshalECPrivateKey(key)225	if err != nil {226		return err227	}228	if err := writePEM(filepath.Join(tls, path+".key"), "EC PRIVATE KEY", sec1, 0o600); err != nil {229		return err230	}231	return writePEM(filepath.Join(tls, path+".crt"), "CERTIFICATE", der, 0o644)232}233234// writePEM writes one PEM block to a file.235func writePEM(path, blockType string, der []byte, mode os.FileMode) error {236	return os.WriteFile(path, pem.EncodeToMemory(&pem.Block{Type: blockType, Bytes: der}), mode)237}
image/cpio.go 81.8%
1// Package image assembles the pieces of liken into the archives a2// machine boots. build.sh produces the generic operating system.3// This package produces the deployment layer that stacks on top of4// it (layer.go explains the split and why concatenation joins them).5package image67// This file implements a cpio writer, for the one format that8// matters here: "newc", the only archive format the kernel's9// initramfs unpacker accepts.10//11// The format dates from 1990 and is simple, which is exactly why the12// kernel adopted it. Every entry is a fixed 110-byte ASCII header (a13// magic number and thirteen 8-digit hexadecimal fields), then the14// file name, then the file's bytes, each padded to a 4-byte15// boundary. A special final entry named TRAILER!!! marks the end.16// There is no index, no compression, and no seeking. The kernel17// reads the archive from start to finish, creating each node as it18// appears. This is why parent directories must be written before19// their contents.20//21// Two properties matter for an initramfs, so this writer fixes them22// rather than making them configurable. Every entry belongs to root23// (uid/gid 0), no matter who ran the build. And hardlink handling24// goes unused (nlink is 1 for files, 2 for directories, and every25// data length is real), so the unpacker's deduplication paths never26// run.2728import (29	"fmt"30	"io"31)3233type archive struct {34	w   io.Writer35	off int // the number of bytes written so far, used for 4-byte alignment36	ino int // a fresh inode number for each entry. The unpacker treats37	// (dev, ino) pairs as hardlink identity, so each entry's number must differ.38}3940func newArchive(w io.Writer) *archive {41	return &archive{w: w}42}4344// header writes one newc header, plus the entry's name, padded so45// the data that follows starts on a 4-byte boundary. The header46// holds thirteen fields, in this order: inode, mode, uid, gid,47// nlink, mtime, filesize, devmajor, devminor, rdevmajor, rdevminor,48// namesize, and check. check is always zero here; newc's sibling49// format, "crc", uses it, but newc does not.50func (a *archive) header(name string, mode, nlink, filesize int) error {51	a.ino++52	// mtime is zero on purpose. This way, the archive's bytes depend53	// only on its contents, not on when it was built. This keeps a54	// deployment layer reproducible and its digest stable.55	if err := a.write(fmt.Appendf(nil,56		"070701%08X%08X%08X%08X%08X%08X%08X%08X%08X%08X%08X%08X%08X",57		a.ino, mode, 0, 0, nlink, 0, filesize, 0, 0, 0, 0, len(name)+1, 0)); err != nil {58		return err59	}60	if err := a.write(append([]byte(name), 0)); err != nil {61		return err62	}63	return a.pad()64}6566// dir writes one directory entry. Directories carry no data. The67// S_IFDIR bits in the mode tell the unpacker to create the directory.68func (a *archive) dir(name string, perm int) error {69	return a.header(name, 0o040000|perm, 2, 0)70}7172// file writes one regular file entry: the header, then the bytes,73// then padding to keep the next header aligned.74func (a *archive) file(name string, data []byte, perm int) error {75	if err := a.header(name, 0o100000|perm, 1, len(data)); err != nil {76		return err77	}78	if err := a.write(data); err != nil {79		return err80	}81	return a.pad()82}8384// close writes the trailer entry that tells the unpacker the archive85// is complete.86func (a *archive) close() error {87	return a.header("TRAILER!!!", 0, 1, 0)88}8990func (a *archive) write(b []byte) error {91	n, err := a.w.Write(b)92	a.off += n93	return err94}9596func (a *archive) pad() error {97	if rem := a.off % 4; rem != 0 {98		return a.write(make([]byte, 4-rem))99	}100	return nil101}
image/cpio_read.go 84.9%
1package image23// This file reads newc archives: the inverse of cpio.go's writer,4// for the few places the build tools need to look inside an archive5// they did not just write. The stick builder reads a deployment6// layer to list the machines in it. The layer is what actually7// boots, so the install menu comes from the layer, not from a8// manifests directory whose content can differ from the layer.9//10// The reader is as strict as the writer is simple. A truncated or11// malformed archive is always an error, never a partial result,12// because every caller is about to make decisions from what it read.1314import (15	"fmt"16	"path"17	"slices"18	"strconv"19	"strings"20)2122// A cpioEntry is one file or directory read back from an archive.23type cpioEntry struct {24	name string25	mode uint3226	uid  uint3227	gid  uint3228	data []byte29}3031// readCPIO parses one newc archive into its entries, stopping at the32// trailer, and returns whatever bytes follow it. A composed image is33// several archives concatenated, so "the rest" is often the next34// archive. The data slices alias raw. A caller that keeps them also35// keeps the archive alive, which every current caller does anyway.36func readCPIO(raw []byte) ([]cpioEntry, []byte, error) {37	var entries []cpioEntry38	off := 039	pad4 := func(n int) int { return (n + 3) &^ 3 }40	for {41		if off+110 > len(raw) {42			return nil, nil, fmt.Errorf("archive ends mid-header at byte %d", off)43		}44		if string(raw[off:off+6]) != "070701" {45			return nil, nil, fmt.Errorf("no newc magic at byte %d", off)46		}47		field := func(i int) (uint32, error) {48			start := off + 6 + i*849			v, err := strconv.ParseUint(string(raw[start:start+8]), 16, 32)50			if err != nil {51				return 0, fmt.Errorf("header field %d at byte %d: %w", i, off, err)52			}53			return uint32(v), nil54		}55		mode, err := field(1)56		if err != nil {57			return nil, nil, err58		}59		uid, err := field(2)60		if err != nil {61			return nil, nil, err62		}63		gid, err := field(3)64		if err != nil {65			return nil, nil, err66		}67		filesize, err := field(6)68		if err != nil {69			return nil, nil, err70		}71		namesize, err := field(11)72		if err != nil {73			return nil, nil, err74		}75		if namesize == 0 || off+110+int(namesize) > len(raw) {76			return nil, nil, fmt.Errorf("archive ends mid-name at byte %d", off)77		}78		name := string(raw[off+110 : off+110+int(namesize)-1])79		dataStart := pad4(off + 110 + int(namesize))80		if dataStart+int(filesize) > len(raw) {81			return nil, nil, fmt.Errorf("archive ends mid-file in %q", name)82		}83		data := raw[dataStart : dataStart+int(filesize)]84		off = pad4(dataStart + int(filesize))85		if name == "TRAILER!!!" {86			return entries, raw[off:], nil87		}88		entries = append(entries, cpioEntry{name: name, mode: mode, uid: uid, gid: gid, data: data})89	}90}9192// machineNames lists the machines a deployment layer carries, from93// its etc/liken/machines/*.yaml entries. Init selects among these94// same files by the liken.machine= parameter, so the names here are95// exactly the names an install can be asked for. machineNames refuses96// a layer carrying no machines: install media with nothing to offer97// is a mistake worth catching while it is still a build.98func machineNames(layer []byte) ([]string, error) {99	entries, _, err := readCPIO(layer)100	if err != nil {101		return nil, fmt.Errorf("reading the deployment layer: %w", err)102	}103	var names []string104	for _, e := range entries {105		dir, file := path.Split(e.name)106		if dir == "etc/liken/machines/" && strings.HasSuffix(file, ".yaml") {107			names = append(names, strings.TrimSuffix(file, ".yaml"))108		}109	}110	if len(names) == 0 {111		return nil, fmt.Errorf("the deployment layer carries no machine manifests; there would be nobody to install")112	}113	slices.Sort(names)114	return names, nil115}
image/layer.go 90.0%
1package image23// This file builds the deployment layer: everything that makes a4// generic liken image one particular deployment's image.5//6// The generic archive that build.sh produces carries the operating7// system and nothing else: no cluster identity, no manifests. This8// file packs the rest, the deployment's half, as a second cpio9// archive. Concatenation joins the two archives: the kernel's10// initramfs unpacker processes concatenated archives in order into11// one filesystem, with later entries overriding earlier ones (the12// same mechanism the install image uses to carry its payload). This13// split is what makes liken releasable: the generic archive's digest14// never changes with the deployment, and producing a bootable image15// from a release is composition, not compilation.16//17// The layer holds these contents, at the paths init and k3s read:18//19//	etc/liken/cluster.yaml       the deployment's cluster document20//	etc/liken/machines/*.yaml    one manifest per machine21//	etc/liken/token              the join token (0600; init hands k3s22//	                             the path, never the value)23//	etc/liken/psk/<ssid>         one passphrase file per wireless24//	                             network (0600), the same class of25//	                             secret as the token26//	var/lib/rancher/k3s/         the certificate authorities, exactly27//	  server/tls/**              where k3s looks before generating28//	                             its own29//30// Kernel modules are deliberately not here. The system image carries31// the kernel build's whole module tree, so a machine's declared32// modules (spec.modules) are a boot-time load from that tree, never33// a build-time selection. The layer stays what its name says: the34// deployment's declarations and identity, nothing of the OS.35//36// The identity's kubeconfig, and the admin keypair inside it, stay37// behind on purpose. The kubeconfig is the operator's credential, not38// the machine's. A machine image carrying it would hand cluster-admin39// access to anyone who reads the disk.4041import (42	"errors"43	"fmt"44	"io/fs"45	"os"46	"path/filepath"47	"slices"48	"strings"4950	"github.com/liken-sh/liken/liken/cluster"51	"github.com/liken-sh/liken/liken/identity"52	"github.com/liken-sh/liken/liken/machine"53)5455// Layer packs a deployment's archive from its manifests and identity.56func Layer(manifests, identityDir, out string) error {57	// The documents are read and validated before the output file58	// exists, so a refusal leaves no truncated cpio behind for a later59	// step to pack into an image.60	d, err := readDeployment(manifests)61	if err != nil {62		return err63	}64	// The passphrases are read and cross-checked against the65	// manifests here, before the output exists, for the same reason66	// the documents are: a refusal must leave no truncated cpio67	// behind.68	passphrases, err := readPassphrases(identityDir, d)69	if err != nil {70		return err71	}7273	f, err := os.Create(out)74	if err != nil {75		return err76	}77	defer f.Close()78	a := newArchive(f)79	dirs := map[string]bool{}8081	// ensure writes the missing parents of a path, root first. This is82	// the order the kernel's unpacker needs to create them.83	ensure := func(path string) error {84		var parents []string85		for d := filepath.Dir(path); d != "."; d = filepath.Dir(d) {86			parents = append(parents, d)87		}88		slices.Reverse(parents)89		for _, d := range parents {90			if !dirs[d] {91				if err := a.dir(d, 0o755); err != nil {92					return err93				}94				dirs[d] = true95			}96		}97		return nil98	}99	place := func(dst string, data []byte, perm int) error {100		if err := ensure(dst); err != nil {101			return err102		}103		return a.file(dst, data, perm)104	}105	ship := func(src, dst string, perm int) error {106		data, err := os.ReadFile(src)107		if err != nil {108			return err109		}110		return place(dst, data, perm)111	}112113	// This stages the manifests at the paths init reads. It copies114	// them file by file rather than as a tree, so what the layer can115	// carry stays explicit: one cluster document, one manifest per116	// machine.117	if d.cluster != nil {118		if err := place("etc/liken/cluster.yaml", d.cluster, 0o644); err != nil {119			return err120		}121	}122	for _, m := range d.machines {123		if err := place(filepath.Join("etc/liken/machines", m.name), m.raw, 0o644); err != nil {124			return err125		}126	}127128	// This stages the identity: the CA tree where k3s looks for it,129	// and the token under /etc/liken, where no disk mount can hide it130	// underneath. The bundle list belongs to the identity package131	// itself, so the layer and the mint can never disagree about what132	// an identity is.133	for _, p := range identity.Bundle {134		src := filepath.Join(identityDir, p)135		dst := filepath.Join("var/lib/rancher/k3s/server", p)136		perm := 0o600137		if strings.HasSuffix(p, ".crt") {138			perm = 0o644139		}140		if p == "token" {141			dst = "etc/liken/token"142		}143		if err := ship(src, dst, perm); err != nil {144			return err145		}146	}147148	// The passphrases land beside the token, at the token's mode,149	// because init reads its cluster credentials from one place.150	for _, p := range passphrases {151		if err := place(filepath.Join("etc/liken/psk", p.name), p.raw, 0o600); err != nil {152			return err153		}154	}155156	return a.close()157}158159// deployment is the manifests directory read into memory: the cluster160// document, when the directory has one, and each machine manifest161// under the file name the layer keeps for it.162type deployment struct {163	cluster  []byte164	machines []deploymentMachine165}166167type deploymentMachine struct {168	name string169	raw  []byte170	// The layer packs the raw bytes and reads the parsed document171	// for the checks that need a field, so nothing parses twice.172	parsed *machine.Machine173}174175// passphraseFile is one wireless network's passphrase, read from the176// identity directory under the SSID that names it. These files are177// not part of identity.Bundle: the bundle is the fixed set the mint178// creates, while a passphrase is written by the operator, and a179// deployment with no wifi has none at all.180type passphraseFile struct {181	name string182	raw  []byte183}184185// readPassphrases reads every passphrase file in the identity186// directory, and refuses a deployment that declares a network no file187// answers for. The refusal belongs here, at build time, because the188// alternative is a boot that parks: the machine would install, come189// up with no passphrase, and hold on its console. The person building190// the stick can fix the file now; the person at the machine cannot.191func readPassphrases(identityDir string, d *deployment) ([]passphraseFile, error) {192	dir := filepath.Join(identityDir, "psk")193	// A deployment with no wireless machine has no psk directory at194	// all, and that is not a missing file.195	entries, err := os.ReadDir(dir)196	if err != nil && !errors.Is(err, fs.ErrNotExist) {197		return nil, err198	}199200	var files []passphraseFile201	held := map[string]bool{}202	for _, entry := range entries {203		// One file names one network, so a directory here names no204		// network and is skipped.205		if entry.IsDir() {206			continue207		}208		raw, err := os.ReadFile(filepath.Join(dir, entry.Name()))209		if err != nil {210			return nil, err211		}212		files = append(files, passphraseFile{name: entry.Name(), raw: raw})213		held[entry.Name()] = true214	}215216	for _, m := range d.machines {217		for _, ifc := range m.parsed.Spec.Network.Interfaces {218			w := ifc.Wireless219			// An open network takes no key, so it needs no file.220			if w == nil || w.SecurityOrDefault() != machine.WirelessWPAPSK || held[w.SSID] {221				continue222			}223			return nil, fmt.Errorf("%s: interface %s joins the network %q, and no file holds its passphrase; write it to %s",224				m.name, ifc.Name, w.SSID, filepath.Join(dir, w.SSID))225		}226	}227	return files, nil228}229230// readDeployment reads every document the layer will pack and runs the231// same parsers a machine runs at boot. This is where the refusal232// scales: it happens where the media is built, and a person is already233// reading the output there. A document that reaches a machine unread234// installs without complaint, and then stops every boot from the disk235// a few seconds in, with the reason on a console nobody is watching.236//237// The checks are the set that scaffold/scaffold.go runs over the238// documents it generates. An absent cluster.yaml is legal, because a239// machine alone is its own cluster.240func readDeployment(manifests string) (*deployment, error) {241	d := &deployment{}242243	if path := filepath.Join(manifests, "cluster.yaml"); fileExists(path) {244		raw, err := os.ReadFile(path)245		if err != nil {246			return nil, err247		}248		if _, err := cluster.ParseCluster(raw); err != nil {249			return nil, fmt.Errorf("%s: %w", path, err)250		}251		d.cluster = raw252	}253254	paths, err := filepath.Glob(filepath.Join(manifests, "machines", "*.yaml"))255	if err != nil {256		return nil, err257	}258	for _, path := range paths {259		raw, err := os.ReadFile(path)260		if err != nil {261			return nil, err262		}263		parsed, err := machine.Parse(raw)264		if err != nil {265			return nil, fmt.Errorf("%s: %w", path, err)266		}267		if err := parsed.Spec.Storage.Validate(); err != nil {268			return nil, fmt.Errorf("%s: %w", path, err)269		}270		if err := parsed.Spec.Network.Validate(); err != nil {271			return nil, fmt.Errorf("%s: %w", path, err)272		}273		d.machines = append(d.machines, deploymentMachine{name: filepath.Base(path), raw: raw, parsed: parsed})274	}275	return d, nil276}277278func fileExists(path string) bool {279	_, err := os.Stat(path)280	return err == nil281}
image/media.go 88.2%
1package image23// This file builds install media: a public release and a deployment4// layer become the image a machine boots once, to install itself.5//6// An install boot is a small one. The installer is init itself, and7// it never needs the running system, only the manifests that say how8// to partition, and the payload to copy. So the install image is9// four cpio archives concatenated: the release's microcode early10// cpio first (the kernel scans the very start of its initrd for11// microcode, and an install boot is a full kernel boot that deserves12// current microcode like any other), then the boot archive13// (boot.cpio: init and the early boot's modules), the deployment14// layer (the manifests whose storage specs drive the partitioning),15// and a wrapper carrying the release payload at16// /usr/share/liken/release. The kernel's initramfs unpacker17// processes concatenated archives in order into one filesystem, the18// same mechanism the machine's boot entries use to join their19// halves from the slot.20//21// The payload is the slot layout, exactly: every artifact the release22// document lists, the document itself byte for byte, and the layer23// beside its sidecar. Carrying the document verbatim, rather than24// generating one, is what lets the installed machine verify later25// downloads against the same catalog digest the release was26// published under. The stick and the internet then vouch for the27// same bytes.28//29// Everything is verified before a single byte is packed. An artifact30// that fails its document here would fail it again in the installer,31// on a machine, where the only remedy is new media.3233import (34	"crypto/sha256"35	"encoding/hex"36	"fmt"37	"io"38	"os"39	"path/filepath"4041	"github.com/liken-sh/liken/liken/machine"42)4344// payloadDir is the path where the wrapper archive carries the45// release. It is the path the installer reads (init's46// releasePayloadDir).47const payloadDir = "usr/share/liken/release"4849// Media assembles install media from a release directory (a public50// bundle: artifacts beside their release.yaml) and a deployment51// layer, and writes the bootable image to out.52func Media(releaseDir, layerPath, out string, log io.Writer) error {53	document, release, err := verifiedRelease(releaseDir)54	if err != nil {55		return err56	}5758	layer, err := os.ReadFile(layerPath)59	if err != nil {60		return fmt.Errorf("reading the deployment layer: %w", err)61	}6263	f, err := os.Create(out)64	if err != nil {65		return err66	}67	defer f.Close()6869	// This writes the microcode early cpio first (it must lead the70	// whole initrd), then the boot archive, then the layer, so the71	// layer's entries override at unpack.72	for _, name := range []string{microcodeArtifact, "boot.cpio"} {73		part, err := os.Open(filepath.Join(releaseDir, name))74		if err != nil {75			return fmt.Errorf("release %s carries no %s; use a newer release",76				release.Metadata.Name, name)77		}78		_, err = io.Copy(f, part)79		part.Close()80		if err != nil {81			return err82		}83	}84	if _, err := f.Write(layer); err != nil {85		return err86	}8788	if err := writePayload(f, releaseDir, release, document, layer); err != nil {89		return err90	}9192	info, err := f.Stat()93	if err != nil {94		return err95	}96	fmt.Fprintf(log, "install media for liken %s: %d MB (%d artifacts + the deployment layer)\n",97		release.Metadata.Name, info.Size()/(1<<20), len(release.Artifacts))98	return nil99}100101// verifiedRelease reads a release directory's document and proves102// every listed artifact against it. This is the same check the fetch103// path performs, done before a single byte is packed, because an104// artifact that fails its document here would fail it again on a105// machine, where the only remedy is new media.106func verifiedRelease(releaseDir string) ([]byte, *machine.Release, error) {107	document, err := os.ReadFile(filepath.Join(releaseDir, "release.yaml"))108	if err != nil {109		return nil, nil, fmt.Errorf("reading the release document: %w", err)110	}111	release, err := machine.ParseRelease(document)112	if err != nil {113		return nil, nil, err114	}115	for _, artifact := range release.Artifacts {116		f, err := os.Open(filepath.Join(releaseDir, artifact.Name))117		if err != nil {118			return nil, nil, fmt.Errorf("the release is missing an artifact its document lists: %w", err)119		}120		err = artifact.Verify(f)121		f.Close()122		if err != nil {123			return nil, nil, fmt.Errorf("the release does not match its own document: %w", err)124		}125	}126	return document, release, nil127}128129// writePayload packs the installer's payload as a wrapper archive:130// the slot layout, exactly. It includes every artifact the document131// lists, the document itself byte for byte, and the layer beside the132// sidecar computed from it. The artifacts were verified before this133// function was called, and this function reads them again here. The134// document and the layer travel as the bytes already in hand.135func writePayload(w io.Writer, releaseDir string, release *machine.Release, document, layer []byte) error {136	sum := sha256.Sum256(layer)137	sidecar := machine.FormatLayerSidecar(hex.EncodeToString(sum[:]))138139	payload := []struct {140		name string141		data []byte142	}{143		{"release.yaml", document},144		{machine.LayerName, layer},145		{machine.LayerSidecarName, sidecar},146	}147	for _, artifact := range release.Artifacts {148		data, err := os.ReadFile(filepath.Join(releaseDir, artifact.Name))149		if err != nil {150			return err151		}152		payload = append(payload, struct {153			name string154			data []byte155		}{artifact.Name, data})156	}157158	a := newArchive(w)159	for _, d := range []string{"usr", "usr/share", "usr/share/liken", payloadDir} {160		if err := a.dir(d, 0o755); err != nil {161			return err162		}163	}164	for _, file := range payload {165		if err := a.file(filepath.Join(payloadDir, file.name), file.data, 0o644); err != nil {166			return err167		}168	}169	return a.close()170}
image/stick.go 81.8%
1package image23// This file builds the install stick: one bootable disk image for4// each deployment.5//6// A stick turns a downloaded release and a deployment layer into7// running machines. Its disk image is a GPT with a single EFI system8// partition, and the partition holds two kinds of thing: the boot9// half (systemd-boot as the well-known \EFI\BOOT\BOOTX64.EFI that10// firmware runs from removable media, its menu configuration, and11// the files the menu entries boot) and the install payload that the12// booted installer copies onto the machine's own disk.13//14// The menu is the deployment's machine list. Every entry boots the15// same kernel and the same two initramfs archives. What differs is16// one argument, liken.machine=<name>, so the operator standing at a17// machine picks its name and everything else follows. One stick18// serves the whole fleet, so nobody ever needs to reflash it between19// machines.20//21// The payload duplicates the OS files that also sit beside it on the22// stick (about 160MB): the installer reads /usr/share/liken/release23// from its own initramfs and never reads the stick's filesystem24// again after that. Teaching the installer to read the stick would25// save this space, but this script deliberately does not do that, to26// keep the installer identical across the stick and the lab's27// direct-kernel path.2829import (30	"fmt"31	"io"32	"maps"33	"os"34	"path/filepath"35	"slices"36	"strings"3738	"github.com/liken-sh/liken/liken/disks"39	"github.com/liken-sh/liken/liken/machine"40)4142// bootMenuArtifact is systemd-boot's canonical name in a release.43const bootMenuArtifact = "systemd-bootx64.efi"4445// microcodeArtifact is the CPU microcode early cpio's canonical name46// in a release. Every boot's initrd list leads with it.47const microcodeArtifact = "microcode.cpio"4849// attendedParam is the word every menu entry carries to tell init that50// a person picked this boot. init/attended.go owns the meaning.51const attendedParam = "liken.attended"5253// ramRootParam makes the kernel build rootfs on ramfs, which has no54// capacity limit, instead of the default memory filesystem, which caps55// itself at half the memory the kernel has accounted when rootfs56// mounts. That cap is mm/shmem.c's shmem_default_max_blocks, and it57// returns totalram_pages() / 2. A stick boot's initrd carries the58// whole OS, and its unpack writes hundreds of megabytes into rootfs.59//60// On kernel 7.1.5 the cap bound far below a 4GB guest's real memory on61// some boots, and 3 of 6 boots failed the unpack partway. The file62// being written kept its full size with a corrupt tail, which no later63// check can tell from a good file until a mount rejects it. All 664// boots unpacked byte-perfect under this parameter.65//66// What made the cap bind low is not established, and the failure no67// longer reproduces. Twenty lab boots with this parameter removed, ten68// on 7.1.5 and ten on 7.2, all unpacked clean, and every one of them69// reported the same 2.95GB of a 4GB guest. So the guest kernel was70// never the variable, and neither was deferred page initialization:71// both kernels leave CONFIG_DEFERRED_STRUCT_PAGE_INIT off, and memory72// accounting finishes in mm_core_init before rootfs mounts from73// vfs_caches_init.74//75// The parameter stays anyway. The failure was measured, nobody found76// what caused it, and a machine with less memory than the lab guest77// reaches the cap sooner. ramfs bounds the unpack by the initrd's own78// contents, which the slot sizes already bound, so no cap can bind at79// all, whatever the cause was.80//81// Boots from an installed slot do not carry this parameter: their82// initrd holds only the boot archive and microcode, two orders of83// magnitude below any observed cap, and the entries that name them84// belong to the boot chain that init alone rewrites.85const ramRootParam = "rootfstype=ramfs"8687// sortKeySeparator divides a machine name from the action in a menu88// entry's sort-key. entryText explains why it must sort below every89// character a machine name can hold.90const sortKeySeparator = "+"9192// Stick builds the install image. releaseDir is a downloaded release93// (artifacts beside their release.yaml), layerPath is the94// deployment's layer, and out is the disk image to write. consoles95// adds console= arguments to every menu entry, and through the96// installer, to the machines' permanent boot entries. The default97// (none) leaves hardware on its own screen, and the lab passes98// ttyS0.99func Stick(releaseDir, layerPath, out string, consoles []string, log io.Writer) error {100	document, release, err := verifiedRelease(releaseDir)101	if err != nil {102		return err103	}104	for _, required := range []string{bootMenuArtifact, microcodeArtifact} {105		if !slices.ContainsFunc(release.Artifacts, func(a machine.ReleaseArtifact) bool {106			return a.Name == required107		}) {108			return fmt.Errorf("release %s carries no %s; use a newer release",109				release.Metadata.Name, required)110		}111	}112113	layer, err := os.ReadFile(layerPath)114	if err != nil {115		return fmt.Errorf("reading the deployment layer: %w", err)116	}117	machines, err := machineNames(layer)118	if err != nil {119		return err120	}121122	// The payload spills to a temp file next to the image. It is123	// hundreds of megabytes, and its exact size is needed before the124	// build can lay out the image.125	payload, err := os.CreateTemp(filepath.Dir(out), ".payload-*")126	if err != nil {127		return err128	}129	defer os.Remove(payload.Name())130	defer payload.Close()131	if err := writePayload(payload, releaseDir, release, document, layer); err != nil {132		return err133	}134	payloadInfo, err := payload.Stat()135	if err != nil {136		return err137	}138139	// This is everything the partition will hold. It generates the140	// loader files up front, so their sizes count too.141	loaderConf := loaderConfText()142	entries := map[string][]byte{}143	for _, name := range machines {144		entries["loader/entries/"+name+"-install.conf"] = entryText(name, false, consoles)145		entries["loader/entries/"+name+"-reinstall.conf"] = entryText(name, true, consoles)146	}147	entries["loader/entries/hardware-report.conf"] = reportEntryText(consoles)148149	sizes := int64(len(loaderConf)) + int64(payloadInfo.Size()) + int64(len(layer))150	for _, e := range entries {151		sizes += int64(len(e))152	}153	for _, a := range release.Artifacts {154		if a.Name != "liken" {155			sizes += a.Size156		}157	}158	// The CLI is in the payload but not laid on the stick's159	// filesystem. A person with the stick already ran the toolkit to160	// make it, and the installer copies the slot's own copy from the161	// payload.162	espBytes := espSize(sizes)163164	// This is the image: 1MiB of alignment before the partition, the165	// ESP, and the GPT's mirrored tail.166	totalBytes := int64(disks.PartitionAlignment)*disks.SectorSize + espBytes +167		int64(disks.ReservedLBAs+disks.PartitionAlignment)*disks.SectorSize168	f, err := os.Create(out)169	if err != nil {170		return err171	}172	defer f.Close()173	if err := f.Truncate(totalBytes); err != nil {174		return err175	}176177	totalSectors := uint64(totalBytes) / disks.SectorSize178	firstLBA := uint64(disks.PartitionAlignment)179	lastLBA := firstLBA + uint64(espBytes)/disks.SectorSize - 1180	table := &disks.Table{DiskGUID: disks.RandomGUID()}181	table.Entries = append(table.Entries, disks.Entry{182		TypeGUID:   disks.EFISystemPartition,183		UniqueGUID: disks.RandomGUID(),184		FirstLBA:   firstLBA,185		LastLBA:    lastLBA,186		Name:       "liken:install",187	})188	chunks, err := disks.SerializeGPT(table, totalSectors)189	if err != nil {190		return err191	}192	for _, chunk := range chunks {193		if _, err := f.WriteAt(chunk.Data, int64(chunk.LBA)*disks.SectorSize); err != nil {194			return fmt.Errorf("writing the partition table: %w", err)195		}196	}197198	// This is the filesystem, inside the partition's window of the199	// image.200	esp := disks.NewSection(f, int64(firstLBA)*disks.SectorSize, espBytes)201	if err := disks.FormatFAT32(esp, uint64(espBytes), "LIKEN-INST", 0x4C494B45); err != nil {202		return err203	}204	w, err := disks.NewFATWriter(esp)205	if err != nil {206		return err207	}208	for _, dir := range []string{"EFI", "EFI/BOOT", "loader", "loader/entries"} {209		if err := w.Mkdir(dir); err != nil {210			return err211		}212	}213214	copyIn := func(path, src string) error {215		in, err := os.Open(src)216		if err != nil {217			return err218		}219		defer in.Close()220		info, err := in.Stat()221		if err != nil {222			return err223		}224		return w.WriteFile(path, in, info.Size())225	}226	if err := copyIn("EFI/BOOT/BOOTX64.EFI", filepath.Join(releaseDir, bootMenuArtifact)); err != nil {227		return err228	}229	if err := w.WriteFile("loader/loader.conf", strings.NewReader(loaderConf), int64(len(loaderConf))); err != nil {230		return err231	}232	for _, path := range slices.Sorted(maps.Keys(entries)) {233		text := entries[path]234		if err := w.WriteFile(path, strings.NewReader(string(text)), int64(len(text))); err != nil {235			return err236		}237	}238	if err := copyIn("vmlinuz", filepath.Join(releaseDir, "vmlinuz")); err != nil {239		return err240	}241	if err := copyIn(microcodeArtifact, filepath.Join(releaseDir, microcodeArtifact)); err != nil {242		return err243	}244	if err := copyIn("boot.cpio", filepath.Join(releaseDir, "boot.cpio")); err != nil {245		return err246	}247	if err := w.WriteFile(machine.LayerName, strings.NewReader(string(layer)), int64(len(layer))); err != nil {248		return err249	}250	if _, err := payload.Seek(0, io.SeekStart); err != nil {251		return err252	}253	if err := w.WriteFile("payload.cpio", payload, payloadInfo.Size()); err != nil {254		return err255	}256	if err := w.Close(); err != nil {257		return err258	}259	if err := f.Sync(); err != nil {260		return err261	}262263	fmt.Fprintf(log, "install stick for liken %s: %d MB, %d machines on the menu\n",264		release.Metadata.Name, totalBytes/(1<<20), len(machines))265	fmt.Fprintf(log, "write it with: dd if=%s of=/dev/YOUR-STICK bs=4M oflag=direct status=progress\n", out)266	return nil267}268269// espSize is the partition size for a given content size: room for270// the allocation tables and directories, a little slack, rounded up271// to a whole MiB, and never below FAT32's floor of about 260MiB (the272// FAT type depends on the cluster count, so a smaller volume would273// not be FAT32 at all).274func espSize(contentBytes int64) int64 {275	size := contentBytes + contentBytes/10 + 8<<20276	size = (size + (1 << 20) - 1) &^ ((1 << 20) - 1)277	return max(size, 300<<20)278}279280// loaderConfText is the menu's one setting: wait for a person,281// forever. An installer must never pick a machine by timeout. The282// whole point of the menu is that a person says what this boot does.283func loaderConfText() string {284	return `# The liken installer's menu. Each machine has two entries: install285# it onto blank disks, or wipe and reinstall it over an existing liken286# install. The last entry is the hardware report, which writes a287# proposed manifest for the machine you are standing at and changes288# nothing on its disks. Pick the entry for what you mean to do. The289# "e" key edits an entry's options for one boot, if you ever need to.290timeout menu-force291`292}293294// entryText is one machine's menu entry, in systemd-boot's Boot Loader295// Specification form. Each machine has two entries: install onto blank296// disks (liken.install), and wipe and reinstall over an existing liken297// install (liken.reinstall). The installer only ever claims a blank298// disk, so reinstall is the escape hatch for a disk liken itself wrote;299// picking the entry at the keyboard is the confirmation the reinstall300// needs.301//302// The four initrd lines concatenate in order, microcode first because303// the kernel scans the very start of its initrd for microcode. This is304// the same composition an installed machine gets from the three initrd=305// parameters in its boot entries, plus the installer's payload.306//307// Every entry also carries liken.attended, the word that tells init a308// person is standing at this machine. The menu is the only thing that309// can say this truthfully: it waits forever, so a boot that starts from310// it started because somebody pressed a key. init holds the console at311// the end of an attended boot, and powers off without a prompt when the312// word is absent (init/attended.go).313//314// The sort-key keeps a machine's two entries together and in order.315// Both keys begin with the machine name, so all of one machine's316// entries sort before the next machine's, and "install" sorts before317// "reinstall" within the pair. Without a sort-key, systemd-boot orders318// entries the way it orders kernels, newest first, which would scatter319// the pairs.320//321// The character between the name and the action determines whether322// that grouping survives names that are prefixes of other names.323// systemd-boot compares sort keys as plain UTF-16 code units324// (strcmp16), and a machine name is an RFC-1123 label, so the lowest325// character a name can contain is "-" at 0x2D. sortKeySeparator sits326// one below it. That inequality is the whole mechanism: comparing327// "node+install" against "node-old+install" ends at "+" against "-",328// so both of node's keys land before both of node-old's. A separator at329// or above "-" would interleave the two machines instead, and a menu330// that shows "wipe and reinstall as node-old" directly above "wipe and331// reinstall as node" invites the one misread that wipes the wrong332// machine.333func entryText(name string, reinstall bool, consoles []string) []byte {334	word, action, sortSuffix := "liken.install", "install", "install"335	if reinstall {336		word, action, sortSuffix = "liken.reinstall", "wipe and reinstall", "reinstall"337	}338	options := []string{ramRootParam, "rdinit=/liken", "liken.machine=" + name, word, attendedParam}339	for _, c := range consoles {340		options = append(options, "console="+c)341	}342	return fmt.Appendf(nil, `# Boot this deployment's OS with the identity %q and %s343# this machine's own disks.344title %s as %s345sort-key %s%s%s346linux /vmlinuz347initrd /microcode.cpio348initrd /boot.cpio349initrd /%s350initrd /payload.cpio351options %s352`, name, action, action, name, name, sortKeySeparator, sortSuffix, machine.LayerName, strings.Join(options, " "))353}354355// reportEntryText is the stick-wide hardware report entry. It carries356// liken.report and no liken.machine=, because it describes the hardware357// in front of it, not a machine in the deployment. It boots the same358// kernel and initrd stack as the install entries, so the report reads359// its drivers from the same payload.360//361// The entry has no sort-key on purpose. systemd-boot shows every entry362// that carries a sort-key before every entry that does not, so the one363// report entry always lands after all the machine entries, wherever the364// machine names would otherwise sort it.365//366// Like the install entries, it carries liken.attended: the report ends367// by printing a proposal that a person reads, so it must hold the368// console until they have read it.369func reportEntryText(consoles []string) []byte {370	options := []string{ramRootParam, "rdinit=/liken", "liken.report", attendedParam}371	for _, c := range consoles {372		options = append(options, "console="+c)373	}374	return fmt.Appendf(nil, `# Describe this machine's hardware and write a proposed manifest to375# the stick as %s. This entry changes nothing on the machine's disks.376title liken hardware report377linux /vmlinuz378initrd /microcode.cpio379initrd /boot.cpio380initrd /%s381initrd /payload.cpio382options %s383`, "hardware-report.yaml", machine.LayerName, strings.Join(options, " "))384}
init/actuator.go 66.7%
1package main23// The boot actuator: the firmware interface the upgrade path uses.4//5// Blue-green upgrades need exactly three actions from the code that6// boots the machine. First, point the next boot, and only the next7// boot, at a slot on trial. Second, keep every later boot pointed at8// the proven slot. Third, report whether that standing preference is9// actually in effect. Everything else about upgrades stays the same10// on every machine and lives in proving.go: the staged/proven/11// attempted store, the verdicts, and the order of operations that12// stops a failing release from causing repeated reboots. Only these13// three actions differ by firmware. They form the whole interface14// between the lifecycle and the machine's boot hardware. Each15// firmware's implementation of the interface is called a dialect.16//17// A UEFI machine's firmware holds boot preferences itself, as18// variables in NVRAM. The UEFI specification defines the interface:19// BootNext is the one-shot trial, and BootOrder is the standing20// preference (efiactuator.go). A BIOS machine's firmware holds no21// boot preferences, so liken stores them where GRUB can read them:22// the environment block on the boot home (grubactuator.go). A23// machine chooses this path by declaring the biosBoot and bootHome24// storage roles. The installed grubenv file is the sign that the25// GRUB interface is in place.2627import (28	"errors"29	"os"30	"path/filepath"31	"sync"32	"sync/atomic"3334	"github.com/liken-sh/liken/liken/machine"35)3637// firmwareWrites serializes the two firmware writers that can overlap38// in time: the reboot path and the proving watch. Every other write39// happens at boot, before the machine plane starts, with nothing to40// race. The order of the reboot path's writes is load-bearing: it41// asserts the proven fallback and then arms the trial, and a write42// landing between or after those two leaves the attempted marker43// standing for a trial that the firmware will never run, which the44// next boot reads as a release that ran and fell back.45//46// Every holder releases this lock, including one that panics. The47// reboot path is reachable from inside a machine-plane component, and48// the plane restarts a component after a panic. A lock left held by49// no one would block every later reboot before the reboot syscall,50// and such a machine looks healthy while it ignores every reboot it51// is asked to take. Only a power cycle would recover it.52var firmwareWrites sync.Mutex5354// shuttingDown says the reboot path has claimed the firmware. The55// lock alone cannot say that, because the reboot path releases it56// after its turn, and the proving watch may still be running for a57// while after. The reboot path sets this before it takes the lock,58// and the watch tests it while holding the lock, so a watch that59// waited out the reboot path's turn sees the flag and leaves the60// firmware exactly as that turn left it.61var shuttingDown atomic.Bool6263// bootActuator lists the three actions the proving lifecycle asks of64// the firmware. Each method performs one of the three actions65// described above.66type bootActuator interface {67	// canArmTrial reports whether armTrial could point a boot at the68	// given slot. It changes nothing. It exists because of an69	// ordering rule in armProvingBoot: the code must write the70	// attempted marker before it arms the trial. A marker with no71	// trial behind it reads as "tried and fell back" on the next72	// boot. That is a false rejection of a release that never ran.73	// The code must find anything knowably wrong before it writes74	// the marker, and this method is that check.75	canArmTrial(slot string) error7677	// armTrial makes the next boot, and only the next boot, try the78	// given slot. The one-shot property makes a trial safe: the boot79	// that the arming triggers also consumes it. So any reset after80	// that boot (a panic, a watchdog, a power cut) returns to the81	// standing preference. On success, armTrial returns a82	// console-ready description of what it armed, in the firmware's83	// own terms.84	armTrial(slot string) (string, error)8586	// fallbackLeads reports whether the standing boot preference87	// leads with the given slot. It confirms this by reading the88	// preference back; it does not trust that an earlier write89	// worked. A trial is safe only when every reset lands on a90	// proven slot. The reboot path uses this method to check that91	// before it arms a trial.92	fallbackLeads(slot string) bool9394	// assertProven makes the standing boot preference lead with the95	// given slot, and corrects any drift. It runs on every boot and96	// after every promotion. This keeps the store on disk as the97	// authority; the firmware only ever holds a copy of it. Each98	// dialect also repairs whatever part of its boot path can be lost99	// while a machine runs, because a boot path that is damaged now100	// would stop the next reboot from coming back.101	assertProven(slot string)102}103104// chooseBootActuator picks the firmware interface this machine uses.105// It applies the same test the installer uses: /sys/firmware/efi106// exists only when UEFI booted this kernel. A BIOS machine uses107// GRUB when the boot home carries an installed environment block.108// Without one, for example on external media, a direct-kernel lab109// boot, or a machine installed without the GRUB roles, no firmware110// interface is available.111func chooseBootActuator() bootActuator {112	if firmwareIsUEFI() {113		return efiActuator{dir: efiVarsDir, machineName: bootParamValue("liken.machine")}114	}115	grubDir := filepath.Join(roleMounts[machine.BootHomeRole].path, "grub")116	if _, err := os.Stat(filepath.Join(grubDir, "grubenv")); err == nil {117		return grubActuator{grubDir: grubDir, machineName: bootParamValue("liken.machine")}118	}119	return noActuator{}120}121122// noActuator serves a machine with no way to choose its next boot.123// BIOS firmware holds no boot variables, and a boot from external124// media or a direct-kernel QEMU boot has no standing preference to125// manage. Every assertion does nothing, and no trial can ever arm.126// A machine that cannot guarantee its fallback must stay on the127// version that works.128type noActuator struct{}129130func (noActuator) canArmTrial(slot string) error {131	return errNoActuator132}133134func (noActuator) armTrial(slot string) (string, error) {135	return "", errNoActuator136}137138func (noActuator) fallbackLeads(slot string) bool { return false }139140func (noActuator) assertProven(slot string) {}141142var errNoActuator = errors.New("this machine's firmware holds no boot variables, so init has no way to arm a one-shot trial")
init/attended.go 92.9%
1package main23// Attended boots: the ones a person is standing at.4//5// liken's boots divide into two kinds, and the difference is not what6// the boot does but who is watching it. A machine booting from its own7// disk at three in the morning has no audience: it must never stop and8// ask a question, and panic=10 with the fall-back slot is its whole9// recovery story. A machine booting from an install stick has an10// audience by construction, because the stick's menu waits forever and11// somebody pressed a key on it.12//13// Every terminal state of an attended boot ends at a held console, so14// the machine's last word stays on screen instead of vanishing behind15// a power-off on a timer. On real hardware a dark screen and a dead16// machine could otherwise mean "done" or "never started".17//18// The two kinds must be told apart by something reliable, and the boot19// words are not it. liken.install, liken.reinstall, and liken.report20// say what a boot does; anything can write them, and the things that21// write them without a person are common (an image build in QEMU, a22// PXE server, a lab Makefile). So the presence of a person is its own23// word on the command line, written by the one thing that can vouch24// for it: the menu.2526import (27	"fmt"28	"os"29)3031// attendedParam marks a boot that a person started at a keyboard. The32// install stick writes it into every entry of its menu, because picking33// an entry from that menu is the act that proves a person is standing34// at the machine. Nothing else writes it: a command line assembled by a35// script, a Makefile, or a PXE server describes what the boot must do,36// never who is watching it.37const attendedParam = "liken.attended"3839// attended reports whether a person is at this machine's console and40// can answer a prompt.41func attended() bool {42	return bootParam(attendedParam)43}4445// consoleDevice is the device that /dev/console names: the console the46// kernel opens for init, and the one a person types on. It is a47// variable so tests can point the hold at a file of their own.48var consoleDevice = "/dev/console"4950// holdInstallerConsole ends one of the install menu's boots. It reports51// the boot's last word, and, when a person is there to read it, keeps52// that word on screen until they acknowledge it. A finished install and53// a refused one say different things, so the caller supplies the54// message.55//56// The hold happens only on an attended boot. A menu pick is what makes57// a boot attended, and the menu's entries say so with liken.attended.58// An install word alone does not: an image build, a lab guest, and a59// PXE server all write liken.install onto a command line with nobody60// watching, and a machine nobody is watching must never stop at a61// prompt. It would wait for a keypress that never comes, and the caller62// that started it would wait with it, forever. The same reasoning63// covers a headless server, where the console device opens and only the64// read blocks: what tells this code it may wait is the word on the65// command line, not the device answering.66//67// An attended boot's message goes out twice, to two different readers.68// The log copy goes through the kmsg pipeline, which the kernel echoes69// to every console= the command line named, so a person watching any70// one of them reads it in order with the rest of the boot. The direct71// copy goes to the console device, which is the last console= alone72// (the kernel's rule for /dev/console), and that is the device this73// code reads the keypress from. Neither copy reaches both readers, so74// both copies are written.75//76// The direct copy is written immediately. A pause here could only hide77// the race with the still-draining log, and no fixed pause can hide it:78// a serial console at 9600 baud takes seconds to transmit the hardware79// report's proposal. The ordered copy is what puts the message in its80// right place in the boot's story; the direct copy exists to reach the81// screen its reader must answer.82//83// The failure paths also print the disk inventory. The spec's device84// paths resolved against these disks, and an install boot has one more85// contestant in the naming race than the installed machine will: the86// install medium itself. A finished install needs no such evidence, so87// its message stays short.88func holdInstallerConsole(message string, inventory bool) {89	if inventory {90		reportBlockDevices()91	}92	fmt.Fprintln(os.Stderr, message)93	if !attended() {94		return95	}9697	console, err := os.OpenFile(consoleDevice, os.O_RDWR, 0)98	if err != nil {99		// With no console to prompt on, there is nobody to wait for,100		// whatever the command line claimed.101		fmt.Fprintf(os.Stderr, "liken: opening the console to wait: %v\n", err)102		return103	}104	defer console.Close()105	fmt.Fprintln(console, message)106	buf := make([]byte, 1)107	_, _ = console.Read(buf)108}
init/bootentries.go 86.9%
1package main23// The firmware boot entries that name this machine's system slots.4//5// A UEFI machine keeps its boot menu in NVRAM, as one variable per6// entry (efi.go reads and writes them, loadoption.go describes the7// record). liken writes one entry per system slot. Each entry pins a8// partition by its GPT unique GUID, loads \vmlinuz from it through the9// kernel's EFI stub, and carries the whole kernel command line in the10// entry's own optional data.11//12// That command line is the reason this file matters beyond the13// install. It carries the machine's identity, the slot letter, and the14// console. On a UEFI machine the firmware's entry is the only place15// that holds it, so a firmware whose variables reset to defaults16// leaves a complete installed disk that nothing can start. This file17// renders an entry from facts that live on the disk, so any boot can18// write the entry again. slotloader.go covers the other half: the19// machine that cannot boot at all because the entry is already gone.2021import (22	"bytes"23	"encoding/binary"24	"fmt"25	"os"26	"slices"27	"strings"2829	"github.com/liken-sh/liken/liken/machine"30)3132// slotEntryDescription is the name liken writes on a slot's entry.33// Everything here finds an entry by this description, never by its34// number: the number is a handle the firmware owns, and the35// description is the identity liken owns.36func slotEntryDescription(slot string) string { return "liken slot " + slot }3738// registerSlotEntries registers both slots with UEFI firmware. Slot39// B's entry points at files that do not exist yet, because its slot40// stays empty until the first upgrade. This is fine: a firmware that41// cannot load an entry's file moves on to the next entry in BootOrder.42func registerSlotEntries(slotA, slotB *slotPartition, machineName string) error {43	entryA, err := writeSlotBootEntry(efiVarsDir, "A", slotA, machineName)44	if err != nil {45		return err46	}47	entryB, err := writeSlotBootEntry(efiVarsDir, "B", slotB, machineName)48	if err != nil {49		return err50	}51	if err := writeBootOrder(efiVarsDir, bootOrderWith(efiVarsDir, []uint16{entryA, entryB})); err != nil {52		return fmt.Errorf("writing BootOrder: %w", err)53	}54	// The one-shot is pinned here for the same reason every later55	// boot pins it (assertBootNext): a firmware that drops the56	// BootOrder write across a reset would otherwise send the57	// machine's very first boot back to whatever it booted before58	// the install.59	assertBootNext(efiVarsDir, entryA, "A")60	fmt.Printf("liken: install: boot entries %s and %s written; BootOrder prefers slot A\n",61		bootEntryID(entryA), bootEntryID(entryB))62	return nil63}6465// bootOrderWith puts the given entries at the head of BootOrder, in66// the order given, and keeps whatever the firmware already had after67// them, in its previous order. The firmware's own entries (setup68// menus, shells) stay reachable. They are just never preferred.69func bootOrderWith(dir string, leaders []uint16) []uint16 {70	order := slices.Clone(leaders)71	for _, n := range readBootOrder(dir) {72		if !slices.Contains(order, n) {73			order = append(order, n)74		}75	}76	return order77}7879// writeSlotBootEntry writes one slot's firmware entry under the number80// that already carries its description, or under the lowest free81// number.82func writeSlotBootEntry(dir, slot string, part *slotPartition, machineName string) (uint16, error) {83	number, err := setBootEntry(dir, slotBootOption(slot, part, machineName))84	if err != nil {85		return 0, fmt.Errorf("writing the %s entry: %w", slotEntryDescription(slot), err)86	}87	return number, nil88}8990// slotBootOption renders one slot's whole boot entry: where the kernel91// is, which partition holds it, and the command line to start it with.92//93// The partition is pinned by its GPT unique GUID, so the entry stays94// correct when the disk moves to another controller or another port.95func slotBootOption(slot string, part *slotPartition, machineName string) loadOption {96	return loadOption{97		attributes:  loadOptionActive,98		description: slotEntryDescription(slot),99		hardDrive: &hardDriveNode{100			partitionNumber: part.number,101			firstLBA:        part.firstLBA,102			sectors:         part.lastLBA - part.firstLBA + 1,103			partitionGUID:   part.guid,104		},105		filePath: `\vmlinuz`,106		// The EFI stub reads its command line as UTF-16, the107		// firmware's native string type.108		optionalData: encodeUTF16Z(strings.Join(slotKernelArgs(machineName, slot), " ")),109	}110}111112// slotInitrds names the archives that load beside the kernel, in the113// order they must load. Microcode leads, because the kernel scans the114// very start of its initrd for a microcode update before it115// decompresses anything. boot.cpio carries init and the early boot's116// modules. The deployment layer comes last and belongs to this117// deployment alone. The system itself (liken.sqfs) is deliberately118// absent: init mounts it straight from the slot, so the loader stages119// megabytes instead of the whole OS. Composition at load time is what120// lets an upgrade replace the generic half without touching the layer.121func slotInitrds() []string {122	return []string{"microcode.cpio", "boot.cpio", machine.LayerName}123}124125// slotKernelArgs renders the command line a firmware boot entry126// carries. The EFI stub loads every initrd= file, in order, from the127// same filesystem it loaded the kernel from, so a firmware entry names128// them on the command line itself.129func slotKernelArgs(machineName, slot string) []string {130	var initrds []string131	for _, name := range slotInitrds() {132		initrds = append(initrds, `initrd=\`+name)133	}134	return slotArgs(machineName, slot, initrds)135}136137// slotArgs assembles a command line from scratch, so every argument is138// deliberate and none is inherited by accident:139//140//	console=...      copied from this boot, so the installed system141//	                 keeps using whatever console its operator wired142//	rdinit=/liken    run our program as PID 1143//	liken.machine=   the machine's identity144//	liken.slot=      which slot this entry boots, so a running system145//	                 reports which half of the blue-green pair it is on146//	panic=10         reboot ten seconds after a kernel panic, instead147//	                 of hanging forever. Upgrades depend on this: a148//	                 panicking trial slot must reset, so the firmware's149//	                 consumed BootNext can fall back to the proven slot150//151// The middle argument is what differs between the two things that boot152// a slot. slotKernelArgs supplies the initrd= parameters that an EFI153// stub needs, and the loader entry in slotloader.go supplies none,154// because that loader stages the archives itself.155func slotArgs(machineName, slot string, afterRdinit []string) []string {156	args := append(consoleArgs(), "rdinit=/liken")157	args = append(args, afterRdinit...)158	return append(args, "liken.machine="+machineName, "liken.slot="+slot, "panic=10")159}160161// consoleArgs copies every console= argument from the running command162// line. This machine was told where its console is, and everything163// liken writes should keep using the same console.164func consoleArgs() []string {165	var consoles []string166	for _, field := range cmdlineFields() {167		if strings.HasPrefix(field, "console=") {168			consoles = append(consoles, field)169		}170	}171	return consoles172}173174// healBootEntries brings the firmware's boot menu into agreement with175// this machine's slots. It renders each slot's entry from the GPT facts176// on the disk, writes back only what drifted, and puts both slots at177// the head of BootOrder with the preferred slot first.178//179// This runs on every boot and again on the way down before every180// reboot, so a machine that boots at all repairs its own boot menu.181// That covers every way a firmware loses an entry: an update that182// resets its variables, a dead NVRAM battery, a setup menu's "load183// defaults", and a variable store that filled up. It cannot cover a184// machine with no entry left to boot, because such a machine never runs185// this. slotloader.go covers that one.186//187// Every comparison happens before any write. NVRAM accepts a limited188// number of writes, and the entries and the order cost a healthy189// machine none of them. BootNext is the one exception, at one write190// per boot cycle: the firmware consumes the variable at power-on, so191// the boot-time assertion always finds it gone and writes it back,192// and the shutdown assertion then finds that write standing and193// writes nothing. assertBootNext says why the variable is written at194// all. A healthy machine's console still stays silent, because195// assertBootNext prints only when a write fails or does not read196// back, so a line from here still means something drifted.197//198// The two halves are separate because a boot that cannot write an entry199// can still order the entries that exist. Without a name there is200// nothing correct to render, because an entry carries identity, and a201// boot from external media or a direct-kernel lab boot has no name to202// render. The ordering still runs, so such a boot leaves the machine's203// fallback exactly as strong as it found it.204func healBootEntries(dir, machineName, preferred string) {205	if machineName != "" {206		writeSlotEntries(dir, machineName, preferred)207	}208	var leaders []uint16209	preferredEntry, havePreferred := uint16(0), false210	for _, slot := range slotOrder(preferred) {211		number, ok := findSlotEntry(dir, slot)212		if !ok {213			continue214		}215		leaders = append(leaders, number)216		if slot == preferred {217			preferredEntry, havePreferred = number, true218		}219	}220	if len(leaders) == 0 {221		fmt.Fprintf(os.Stderr, "liken: system: no boot entry answers to either slot and this boot cannot write one; leaving BootOrder alone\n")222		return223	}224	assertBootOrder(dir, leaders, preferred, havePreferred)225	// BootNext joins the assertion because a firmware exists that226	// accepts a BootOrder write, reads it back correctly, and still227	// resets to its old order, while it honors BootNext through every228	// reset. The one-shot is the only preference such a firmware229	// keeps, so the proven slot is pinned there too. The pin is written only230	// when the preferred slot's own entry exists: leaders[0] is the231	// other slot's entry when it does not, and a one-shot aimed there232	// would boot the wrong half of the pair.233	if havePreferred {234		assertBootNext(dir, preferredEntry, preferred)235	}236}237238// writeSlotEntries renders each slot's entry from the GPT facts on its239// disk and writes back only the ones that drifted. A slot whose240// partition cannot be read is skipped rather than guessed at, and the241// other slot still gets its entry: half a boot menu is worth more than242// none.243func writeSlotEntries(dir, machineName, preferred string) {244	parts := discoverPartitions()245	for _, slot := range slotOrder(preferred) {246		part, err := findSlotPartition(parts, slotStorageRole(slot))247		if err != nil {248			fmt.Fprintf(os.Stderr, "liken: system: %v; the %s entry cannot be checked\n",249				err, slotEntryDescription(slot))250			continue251		}252		want := slotBootOption(slot, part, machineName)253		if number, ok := findSlotEntry(dir, slot); ok {254			current, err := readEFIVar(dir, bootEntryID(number))255			if err == nil && bytes.Equal(current, encodeLoadOption(want)) {256				continue // the firmware already holds this exact entry257			}258		}259		number, err := setBootEntry(dir, want)260		if err != nil {261			fmt.Fprintf(os.Stderr, "liken: system: writing the %s entry: %v\n", slotEntryDescription(slot), err)262			continue263		}264		fmt.Printf("liken: system: %s disagreed with this machine's disk; wrote it as %s\n",265			slotEntryDescription(slot), bootEntryID(number))266	}267}268269// assertBootOrder makes the standing preference lead with the given270// entries. It trusts the readback, not the write. Some firmware accepts271// a write and then fails to hold it, and every later report would be272// wrong if this code took the write result at face value.273// preferredLeads says whether leaders[0] is the preferred slot's own274// entry. It is false when no entry answers to that slot, and the275// console line must not then claim the preferred slot leads, because276// what actually leads is the other slot's entry.277func assertBootOrder(dir string, leaders []uint16, preferred string, preferredLeads bool) {278	want := bootOrderWith(dir, leaders)279	if slices.Equal(want, readBootOrder(dir)) {280		return // the firmware already agrees281	}282	if err := writeBootOrder(dir, want); err != nil {283		fmt.Fprintf(os.Stderr, "liken: system: repairing BootOrder: %v\n", err)284		return285	}286	if readback := readBootOrder(dir); !slices.Equal(readback, want) {287		fmt.Fprintln(os.Stderr, "liken: system: BootOrder was written but does not read back; the firmware is not holding it")288		return289	}290	if !preferredLeads {291		fmt.Printf("liken: system: BootOrder now leads with %s; no entry answers to slot %s, which the store calls proven\n",292			bootEntryID(leaders[0]), preferred)293		return294	}295	fmt.Printf("liken: system: BootOrder now leads with %s (slot %s is proven)\n",296		bootEntryID(leaders[0]), preferred)297}298299// assertBootNext points the one-shot at the proven slot's entry, so a300// firmware that drops the BootOrder write across a reset still boots301// the proven slot: the one-shot is consumed and honored before the302// standing order is consulted at all. This costs one NVRAM write per303// boot, because the firmware deletes the variable at power-on, and304// the boot-time assertion always finds it gone. armTrial runs after305// this on the reboot path and overwrites the value, which is the306// contract: a staged trial takes the one shot for exactly one boot.307// The same overwrite clears a one-shot left over from a withdrawn308// trial, the way the GRUB dialect clears try_slot.309//310// The comparison decodes both sides to a number instead of comparing311// bytes, because a firmware may store the variable wider than the two312// bytes it needs. A byte comparison would then rewrite NVRAM on every313// call and warn about a firmware that is holding the value correctly.314// Like assertBootOrder, this trusts the readback and not the write.315func assertBootNext(dir string, entry uint16, preferred string) {316	if current, err := readEFIVar(dir, "BootNext"); err == nil && bootNextEntry(current) == int32(entry) {317		return // the firmware already aims at the proven slot318	}319	if err := writeEFIVar(dir, "BootNext", bootNextPayload(entry)); err != nil {320		fmt.Fprintf(os.Stderr, "liken: system: aiming BootNext at slot %s: %v\n", preferred, err)321		return322	}323	if readback, err := readEFIVar(dir, "BootNext"); err != nil || bootNextEntry(readback) != int32(entry) {324		fmt.Fprintln(os.Stderr, "liken: system: BootNext was written but does not read back; the firmware is not holding it")325	}326}327328// bootNextEntry decodes the variable's entry number, and returns -1329// for a payload too short to carry one. The result is signed so that330// a missing or truncated variable compares unequal to every real331// entry number, all of which are non-negative.332func bootNextEntry(payload []byte) int32 {333	if len(payload) < 2 {334		return -1335	}336	return int32(binary.LittleEndian.Uint16(payload))337}338339// bootNextPayload encodes one entry number the way BootNext carries it:340// a single 16-bit little-endian number, the same encoding BootOrder341// uses for each of its elements.342func bootNextPayload(entry uint16) []byte {343	return []byte{byte(entry), byte(entry >> 8)}344}345346// slotOrder puts the preferred slot first and the other slot second.347// Both slots lead the boot order, so a firmware that cannot load the348// preferred slot's kernel tries the other half of the pair before it349// falls back to its own setup menu. A machine with no proven record yet350// takes the installer's order.351func slotOrder(preferred string) []string {352	if preferred == "B" {353		return []string{"B", "A"}354	}355	return []string{"A", "B"}356}357358// slotStorageRole names the storage role that holds a slot.359func slotStorageRole(slot string) machine.StorageRoleName {360	if slot == "B" {361		return machine.SystemBRole362	}363	return machine.SystemARole364}365366// findSlotEntry locates a slot's firmware entry the same way367// everything in liken finds things: by the name written on it.368//369// Two entries can answer to one slot, because some firmware clones a370// boot option it did not write. The testbed's mini PC keeps liken's371// slot B entry and a copy of its own beside it, with the same372// description and the same command line, differing only in the device373// path: liken names the partition by its GPT GUID alone, and the374// firmware's copy carries the full PCI and SATA path ahead of it, the375// same shape it writes for its own Windows and Ubuntu entries.376//377// So this returns the lowest matching number, and never an arbitrary378// one. setBootEntry writes the lowest match too, so the two agree, and379// every reader that compares a number against BootOrder agrees with380// them. An arbitrary choice would make fallbackLeads compare the wrong381// entry against the head of BootOrder, and armProvingBoot would then382// refuse to arm a trial on a machine whose fallback is in fact real.383func findSlotEntry(efiDir, slot string) (uint16, bool) {384	found, ok := uint16(0), false385	for number, option := range listBootEntries(efiDir) {386		if option.description != slotEntryDescription(slot) {387			continue388		}389		if !ok || number < found {390			found, ok = number, true391		}392	}393	return found, ok394}
init/ciphers.go 89.3%
1package main23// The 802.11 stack encrypts with ciphers it instantiates at the4// moment the supplicant installs a key, and on an ordinary5// distribution those arrive by autoload: the kernel's crypto API6// asks userspace to run modprobe. liken ships no modprobe, so the7// request fails silently, the cipher never exists, and the key8// install fails with ENOENT after a handshake that succeeded. The9// supplicant reports that as WRONG_KEY, which reads as a wrong10// passphrase and would even park a radio-only machine. So the11// wireless bring-up loads the ciphers itself, before any supplicant12// starts (plans/completed/62-wifi.md names the incident that proved this on13// metal).1415import (16	"fmt"17	"os"18	"path/filepath"19	"strings"2021	"github.com/liken-sh/liken/liken/machine"22)2324// The three ciphers a join can need: ccm(aes) for CCMP, the WPA2 and25// WPA3 pairwise cipher; gcm(aes) for GCMP; and cmac(aes) for26// protected management frames. Which of them the kernel build ships27// as modules varies by build, and the pass below takes each as it28// finds it.29var wirelessCiphers = []string{"ccm", "gcm", "cmac"}3031// loadCipherModule is the real load behind the pass, a variable so a32// test can run the pass against a fabricated tree with no kernel.33var loadCipherModule = func(base, name string, deps map[string][]string, done map[string]bool) error {34	_, err := loadModule(base, name, "", deps, done)35	return err36}3738// loadWirelessCiphers loads the ciphers from the running kernel's39// own module tree. Only a spec that declares a wireless interface40// reaches this call (radios.go), which keeps the rule that nothing41// wireless runs unless the spec asks for it.42func loadWirelessCiphers() {43	loadWirelessCiphersFrom(filepath.Join("/lib/modules", kernelRelease()))44}4546// loadWirelessCiphersFrom is the same pass with the module tree as a47// parameter, so a test can hand it a fabricated one. Every outcome48// is a report: a builtin is a skip, a name the build has neither way49// is a skip with its own line, and a refused load goes to stderr.50// Nothing here stops a boot; a missing cipher fails the join later,51// and the join's own reporting carries that.52func loadWirelessCiphersFrom(base string) []machine.ModuleStatus {53	deps, err := readModulesDep(filepath.Join(base, "modules.dep"))54	if err != nil {55		fmt.Fprintf(os.Stderr, "liken: wireless: %v\n", err)56		deps = map[string][]string{}57	}58	builtin, err := readModulesBuiltin(filepath.Join(base, "modules.builtin"))59	if err != nil {60		fmt.Fprintf(os.Stderr, "liken: wireless: %v\n", err)61	}6263	done := map[string]bool{}64	// The ciphers ship with the release, not with the Machine, so65	// this pass declares no parameters; spec.moduleParameters covers66	// only modules the spec itself names.67	outcomes := declaredModuleOutcomes(wirelessCiphers, deps, builtin, nil, declaredPass{68		resident: func(name string) bool { return moduleIsResident(sysModuleDir, name) },69		load: func(name, parameters string) error {70			return loadCipherModule(base, name, deps, done)71		},72		readback: func(string) map[string]string { return nil },73	})74	for _, outcome := range outcomes {75		reportCipher(outcome)76	}77	return outcomes78}7980// reportCipher writes one cipher's console line. The state prints in81// every case, so a skip reads as a deliberate skip and not as a82// spelling mistake in this file.83func reportCipher(outcome machine.ModuleStatus) {84	state := strings.ToLower(string(outcome.State))85	if outcome.State == machine.ModuleFailed {86		fmt.Fprintf(os.Stderr, "liken: wireless: cipher %s: %s: %s\n", outcome.Name, state, outcome.Message)87		return88	}89	// A Missing outcome carries a message too: the state word alone90	// says a cipher is absent and never says what the pass looked91	// for, and the console is where a person diagnoses a join that92	// failed on a cipher the build lacks.93	if outcome.Message != "" {94		fmt.Printf("liken: wireless: cipher %s: %s: %s\n", outcome.Name, state, outcome.Message)95		return96	}97	fmt.Printf("liken: wireless: cipher %s: %s\n", outcome.Name, state)98}
init/claim.go 96.8%
1package main23// Claiming a blank disk.4//5// Claiming happens once in a disk's life; it is one half of storage6// reconciliation. storage.go owns the other half, which runs on every7// boot: recognition and mounting. grow.go owns growth. A machine's8// first boot finds declared roles with no partitions to recognize.9// Claiming is how those partitions come to exist: a blank disk gets a10// GPT, and each role's name is written into its partition entry. That11// name is the identity every later boot recognizes.12//13// Two properties make claiming safe to attempt on every boot:14//15//   - Only a blank disk may be claimed. Blank means no partition16//     table and no filesystem signature at all (isBlank below): a17//     disk that nothing, including liken, ever wrote to. The code18//     refuses anything else and states the reason on the console,19//     because a disk someone else formatted holds someone else's20//     data.21//22//   - Nothing is written until the code has validated every disk's23//     plan. planClaim lays out a claim entirely in memory first, so a24//     spec that cannot fit fails before the first byte lands on any25//     disk. The boot can then fall back to a different spec (the26//     proven manifest, after it rejects a staged one) against disks27//     that are exactly as it expects them.28//29// Claiming can also resume after an interruption. The role's name30// goes into the partition table first, before any filesystem exists.31// So a boot that dies between partitioning and mkfs leaves partitions32// that the next boot recognizes by name and finishes formatting33// (mountRole's half of the task, in storage.go).34//35// A claim reports which roles it created, because a partition created36// this boot always gets a new filesystem. The bytes under a new37// partition belong to whatever held the disk before liken claimed it.38//39// A role's spec names its disk by the declared device string, not by40// a device path guaranteed to work here directly: it may hold a41// stable name under /dev/disk/by-id/ or /dev/disk/by-path/ as well as42// a kernel name. reconcileStorage resolves the declared string to a43// disk before it ever reaches planClaim (resolve.go), by identity,44// through the same sysfs computation the by-id and by-path link trees45// use, rather than by reading /dev/disk/ itself.4647import (48	"fmt"49	"os"50	"strings"51	"time"5253	"github.com/liken-sh/liken/liken/disks"54	"github.com/liken-sh/liken/liken/machine"55)5657// isBlank reports whether a disk carries nothing recognizable: no MBR58// or GPT, no ext4 filesystem written straight to the device. Claiming59// is allowed only when a disk is blank. A disk something else60// formatted fails one of these checks, and the code leaves it alone.61func isBlank(devPath string) (bool, error) {62	f, err := os.Open(devPath)63	if err != nil {64		return false, err65	}66	defer f.Close()6768	// 2048 bytes covers all three signatures: the MBR's 0x55AA at69	// byte 510, GPT's "EFI PART" at byte 512, and ext4's magic at70	// byte 1080.71	head := make([]byte, 2048)72	if _, err := f.ReadAt(head, 0); err != nil {73		return false, err74	}75	switch {76	case head[510] == 0x55 && head[511] == 0xAA:77		return false, nil78	case string(head[512:520]) == "EFI PART":79		return false, nil80	case head[1080] == 0x53 && head[1081] == 0xEF:81		return false, nil82	}83	return true, nil84}8586// A claimPlan is one blank disk's pending partition table. The code87// validates and lays it out up front, and writes it only after every88// disk's plan has succeeded.89type claimPlan struct {90	device       string91	totalSectors uint6492	parts        []disks.Partition93	roles        []machine.DeclaredRole94}9596// planAllClaims lays out a claim for every disk that a still-missing97// role points at. It groups every declared role, found or not, by the98// disk its device resolves to (its kernel device path, from99// resolve.go), not by the declared string, because one disk can be100// declared two ways at once: its kernel name for one role, and its101// stable name for another. A found role joins this grouping too,102// because a disk that already carries one declared role's partition103// is not a blank disk, whichever other role's device also names it;104// only a group with no found role in it is a candidate to claim.105//106// A found role's own device does not have to resolve at all.107// Recognition finds a role's partition by the name written on it,108// never by the device the spec declares, so a stale or ambiguous109// device on an already-satisfied role changes nothing that is about110// to be written; this function only resolves it on the chance that it111// names a disk that another, still-missing role also names, one that112// recognition has not yet found. A still-missing role is different:113// its device is the only way to find the disk it names, so a device114// that resolves to no disk, or to two, fails reconciliation at once,115// before any group forms.116func planAllClaims(roles []machine.DeclaredRole, found map[machine.StorageRoleName]partition) ([]claimPlan, error) {117	type group struct {118		disk  *machine.BlockDevice119		roles []machine.DeclaredRole // the still-missing roles that declare this disk120		found *machine.DeclaredRole  // set when a found role also declares this disk121	}122	groups := map[string]*group{}123	var order []string124125	for _, role := range roles {126		_, isFound := found[role.Name]127		// A found role's disk resolves without waiting: its partition128		// is already recognized, nothing here is about to write to129		// it, and a stale device hint on it must not cost the boot130		// the whole deadline. A still-missing role names the disk the131		// claim is about to format, and an install boot reaches this132		// moments after the boot archive loads its controller133		// drivers, so that one waits for the disk to attach.134		disk, err := resolveDeclaredDisk(role.Device)135		if !isFound && err == nil && disk == nil {136			disk, err = awaitDeclaredDisk(role.Device,137				fmt.Sprintf("liken: storage: waiting for %s to attach", role.Device))138		}139		switch {140		case isFound:141			if err != nil || disk == nil {142				continue143			}144		case err != nil:145			return nil, err146		case disk == nil:147			return nil, fmt.Errorf("declared device %s is not attached", role.Device)148		}149150		key := devicePath(*disk)151		g, ok := groups[key]152		if !ok {153			g = &group{disk: disk}154			groups[key] = g155			order = append(order, key)156		}157		if isFound {158			if g.found == nil {159				g.found = &role160			}161			continue162		}163		g.roles = append(g.roles, role)164	}165166	var claims []claimPlan167	for _, key := range order {168		g := groups[key]169		if len(g.roles) == 0 {170			continue // every role naming this disk is already recognized171		}172		// A disk where the code recognizes one role but others still173		// need claiming is a disk whose table liken wrote, and174		// something later changed. It is not blank, so the code175		// cannot claim it, and it is not safe to repair automatically.176		if g.found != nil {177			return nil, fmt.Errorf("disk %s already carries %s but is missing other declared roles; refusing to modify it",178				key, g.found.PartitionName())179		}180		if err := oneSizelessRole(key, g.roles); err != nil {181			return nil, err182		}183		plan, err := planClaim(g.roles[0].Device, g.disk, g.roles)184		if err != nil {185			return nil, err186		}187		claims = append(claims, plan)188	}189	return claims, nil190}191192// oneSizelessRole enforces, within one disk's group of still-missing193// roles, the rule Validate already enforces on the literal spelling:194// only one role may take the rest of a disk. Validate cannot see this195// same violation when two roles spell one disk two different ways,196// because it groups by the declared string, and each spelling looks197// like a different disk to that check. This check runs against the198// resolved group instead, so a disk declared two ways still gets only199// one remainder. Without it, planPartitions would silently keep only200// the last sizeless role it saw and drop the other from the table201// entirely, and a disk would be claimed for a spec it cannot satisfy.202func oneSizelessRole(disk string, roles []machine.DeclaredRole) error {203	var remainder *machine.DeclaredRole204	for _, role := range roles {205		if role.Size != "" {206			continue207		}208		if remainder != nil {209			return fmt.Errorf("storage roles %s and %s both omit their size on %s; only one role per disk may omit its size",210				remainder.Name, role.Name, disk)211		}212		remainder = &role213	}214	return nil215}216217// planClaim validates that a resolved, blank disk may be claimed for218// the still-missing roles that name it, and lays out its table.219// planAllClaims is the only caller: it has already resolved the220// declared device string to the disk it names, checked that no other221// declared role already carries a recognized partition on this disk,222// and confirmed that at most one of these roles is sizeless. Sized223// roles come first, at their exact sizes, in canonical order. The224// remainder role, if any, takes whatever space is left. planClaim225// writes nothing.226func planClaim(declared string, disk *machine.BlockDevice, mine []machine.DeclaredRole) (claimPlan, error) {227	if disk == nil {228		return claimPlan{}, fmt.Errorf("declared device %s is not attached", declared)229	}230	device := devicePath(*disk)231	blank, err := isBlank(device)232	if err != nil {233		return claimPlan{}, fmt.Errorf("examining %s: %w", device, err)234	}235	if !blank {236		return claimPlan{}, fmt.Errorf("%s carries a partition table or filesystem liken doesn't recognize; refusing to touch it (wipe it yourself if it's expendable)", device)237	}238239	totalSectors := disk.SizeBytes / disks.SectorSize240	parts, err := planPartitions(device, mine, totalSectors)241	if err != nil {242		return claimPlan{}, err243	}244	return claimPlan{device: device, totalSectors: totalSectors, parts: parts, roles: mine}, nil245}246247// applyClaim writes one planned table and waits for the kernel to248// show its partitions.249func applyClaim(plan claimPlan) error {250	fmt.Printf("liken: storage: claiming %s (%s) for %d role(s)\n",251		plan.device, gib(plan.totalSectors*disks.SectorSize), len(plan.roles))252	if err := disks.Write(plan.device, plan.totalSectors, plan.parts); err != nil {253		return fmt.Errorf("partitioning %s: %w", plan.device, err)254	}255	return waitForPartitions(plan.parts)256}257258// planPartitions lays out a claimed disk's table. Sized roles pack259// from the front, in canonical order, and each partition start260// aligns to 1MiB. The single, validated, sizeless role takes the261// rest of the disk. The device name appears only in error messages;262// the layout math works purely in sectors.263func planPartitions(device string, mine []machine.DeclaredRole, totalSectors uint64) ([]disks.Partition, error) {264	lastUsable := disks.LastUsableLBA(totalSectors)265	var parts []disks.Partition266	next := uint64(disks.PartitionAlignment)267	var remainder *machine.DeclaredRole268	for _, role := range mine {269		if role.Size == "" {270			remainder = &role271			continue272		}273		bytes, _ := machine.ParseSize(role.Size) // validated before any disk is touched274		sectors := (bytes + disks.SectorSize - 1) / disks.SectorSize275		p := disks.Partition{Name: role.PartitionName(), FirstLBA: next, LastLBA: next + sectors - 1,276			TypeGUID: partitionTypeFor(role.Name)}277		if p.LastLBA > lastUsable {278			return nil, fmt.Errorf("disk %s is too small: %s requires %s at sector %d but the disk's usable space ends at %d",279				device, role.Name, role.Size, p.FirstLBA, lastUsable)280		}281		parts = append(parts, p)282		next = disks.AlignLBA(p.LastLBA + 1)283	}284	if remainder != nil {285		if next > lastUsable {286			return nil, fmt.Errorf("disk %s is too small: nothing left for %s", device, remainder.Name)287		}288		parts = append(parts, disks.Partition{Name: remainder.PartitionName(), FirstLBA: next, LastLBA: lastUsable,289			TypeGUID: partitionTypeFor(remainder.Name)})290	}291	return parts, nil292}293294// partitionPatience bounds the wait for a new table's partitions to295// appear at their planned sizes, after a claim or a growth writes the296// table.297const partitionPatience = 5 * time.Second298299// waitForPartitions gives the kernel a moment to show the devices for300// a table just written. BLKRRPART works synchronously, but the301// devtmpfs nodes and sysfs entries appear slightly later. Each302// partition must appear at its planned size. This lets the same wait303// serve growth as well as claiming: a partition still showing its old304// geometry is as wrong as one that has not appeared. If the deadline305// passes, the error names each partition that did not appear as306// expected.307func waitForPartitions(parts []disks.Partition) error {308	deadline := time.Now().Add(partitionPatience)309	for {310		visible := map[string]uint64{}311		for _, p := range discoverPartitions() {312			visible[p.partName] = p.sizeBytes313		}314		var missing []string315		for _, want := range parts {316			wantBytes := (want.LastLBA - want.FirstLBA + 1) * disks.SectorSize317			got, ok := visible[want.Name]318			switch {319			case !ok:320				missing = append(missing, want.Name)321			case got != wantBytes:322				missing = append(missing, fmt.Sprintf("%s (still %d bytes, want %d)", want.Name, got, wantBytes))323			}324		}325		if len(missing) == 0 {326			return nil327		}328		if time.Now().After(deadline) {329			return fmt.Errorf("partitions never appeared after their table was written: %s",330				strings.Join(missing, ", "))331		}332		time.Sleep(100 * time.Millisecond)333	}334}
init/cluster.go 96.2%
1package main23// The cluster document's boot-time selection.4//5// The Cluster manifest goes through the same lifecycle as the Machine6// manifest (staged, proven, seed), but its selection is simpler,7// because the boot needs the two documents at different moments. The8// Machine manifest drives storage reconciliation, so init must read9// it before the boot properly mounts any filesystem. The boot needs10// the Cluster document only after storage settles, because role11// derivation, k3s configuration, and time sources consume it. By the12// time the boot reads it, machineState is an ordinary mounted13// filesystem, and the code can read the store in place.14//15// The code vets a staged document before it tries the document, the16// same as with the Machine manifest. A staged document that will not17// parse, or is not a Cluster, will fail every future boot the same18// way. So the code rejects it without trying it, and the boot falls19// back to proven, then to seed. Init cannot prove a staged cluster20// document, because its failure modes appear downstream: a bad21// endpoint means the follower never joins, and that failure reaches22// init only as a k3s that never starts. Promotion therefore belongs23// to the operator. The operator's own existence as a pod is the proof24// that the join worked. The operator's half of the lifecycle, and the25// attempted marker, are described in cluster.go on the operator26// side.2728import (29	"errors"30	"fmt"31	"io/fs"32	"os"3334	"github.com/liken-sh/liken/liken/cluster"35	"github.com/liken-sh/liken/liken/machine"36)3738// checkSeedCluster parses the image's cluster document and reports why39// it cannot be used. An install boot calls this before it touches a40// disk, because the install copies these exact bytes into both slots,41// and chooseCluster below treats a seed that will not parse as fatal42// on every boot that follows. Catching it here turns a machine that43// powers off forever into a sentence on the installer's console.44//45// Finding no document is a valid answer, the same as it is below: a46// machine alone is its own cluster. A document that exists and cannot47// be read is not, because the install would copy an unreadable file48// and the next boot would fail on it.49func checkSeedCluster(seedPath string) error {50	raw, err := os.ReadFile(seedPath)51	if errors.Is(err, fs.ErrNotExist) {52		return nil53	}54	if err != nil {55		return fmt.Errorf("%w: %s: %v", errIdentity, seedPath, err)56	}57	if _, err := cluster.ParseCluster(raw); err != nil {58		return fmt.Errorf("%w: %s: %v", errIdentity, seedPath, err)59	}60	return nil61}6263// chooseCluster returns the cluster document this boot runs under,64// both parsed and as its exact bytes, and records the choice in the65// boot record. The code publishes the exact bytes to /run for the66// operator's drift detection. The preference order is staged67// (vetted), then proven, then the image's seed. A memory-backed68// machine has no durable store, so it reads only the seed, every69// boot. Finding no document anywhere is a valid answer, because a70// machine alone is its own cluster. A seed that exists but will not71// parse is an error the caller treats as fatal, because a machine72// that cannot tell its role must not guess it.73func chooseCluster(stateRoot, seedPath string, durable bool, boot *machine.BootStatus) (*cluster.Cluster, []byte, error) {74	if durable {75		store := machine.ClusterManifests(stateRoot)7677		// The code republishes the standing rejection into the boot78		// record every boot (rejectStagedDocument explains why).79		boot.ClusterRejection, _ = store.LoadRejection()8081		if raw, err := store.LoadStaged(); err != nil {82			fmt.Fprintf(os.Stderr, "liken: cluster: the staged document is unreadable: %v\n", err)83		} else if raw != nil {84			hash := machine.ManifestHash(raw)85			attempted, _ := store.LoadAttempted()86			c, perr := cluster.ParseCluster(raw)87			switch {88			case perr != nil:89				boot.ClusterRejection = rejectStagedDocument("cluster", "document", store.Reject,90					raw, fmt.Sprintf("the staged cluster document does not parse: %v", perr))91			case attempted == hash:92				// A previous boot ran this exact document, and nobody93				// promoted it. The machine never joined its cluster94				// under the document, and the operator that would95				// have carried the proof never ran. A staged document96				// gets exactly one trial boot.97				boot.ClusterRejection = rejectStagedDocument("cluster", "document", store.Reject,98					raw, "the last boot ran this staged cluster document and never joined the cluster under it")99			default:100				if err := store.WriteAttempted(hash); err != nil {101					fmt.Fprintf(os.Stderr, "liken: cluster: marking the staged document attempted: %v\n", err)102				}103				boot.ClusterManifestSource = machine.ManifestSourceStaged104				boot.ClusterManifestHash = hash105				fmt.Printf("liken: cluster: booting under the Staged document (%.12s); the operator's first pass is the proof\n", hash)106				return c, raw, nil107			}108		}109110		if raw, err := store.LoadProven(); err != nil {111			fmt.Fprintf(os.Stderr, "liken: cluster: the proven document is unreadable: %v\n", err)112		} else if raw != nil {113			c, perr := cluster.ParseCluster(raw)114			if perr != nil {115				// A proven document that will not parse is a116				// corrupted last-known-good copy. The code reports it117				// and falls through to the seed, rather than failing118				// the boot over a recovery file.119				fmt.Fprintf(os.Stderr, "liken: cluster: the proven document is unreadable: %v\n", perr)120			} else {121				boot.ClusterManifestSource = machine.ManifestSourceProven122				boot.ClusterManifestHash = machine.ManifestHash(raw)123				return c, raw, nil124			}125		}126	}127128	raw, err := os.ReadFile(seedPath)129	if errors.Is(err, fs.ErrNotExist) {130		return nil, nil, nil131	}132	if err != nil {133		return nil, nil, err134	}135	c, err := cluster.ParseCluster(raw)136	if err != nil {137		return nil, nil, fmt.Errorf("%s: %w", seedPath, err)138	}139	boot.ClusterManifestSource = machine.ManifestSourceSeed140	boot.ClusterManifestHash = machine.ManifestHash(raw)141	return c, raw, nil142}
init/cmdline.go 100.0%
1package main23// The kernel command line: the one input channel that exists before4// any filesystem does. rdinit= on the command line pointed the5// kernel at this program, and the command line is liken's channel6// for facts a machine needs before it has read a single file. The7// bootloader owns the command line, which is why it can carry8// identity: the bootloader configures the command line per machine,9// even when the image is shared by a fleet.1011import (12	"os"13	"slices"14	"strings"15)1617// cmdlinePath is a package variable rather than a constant so tests18// can point the parsers at a file of their own making.19var cmdlinePath = "/proc/cmdline"2021// cmdlineFields reads the kernel command line as its words, split on22// whitespace. Every parameter lookup starts from this shape. A23// command line that cannot be read yields no words. The file exists24// on any booted kernel, so its absence only ever means a test did not25// fake one, and every lookup then reports "not present".26func cmdlineFields() []string {27	raw, err := os.ReadFile(cmdlinePath)28	if err != nil {29		return nil30	}31	return strings.Fields(string(raw))32}3334// cmdlineRaw returns the whole command line as one line, with the35// trailing newline the kernel appends removed. The parsers above take36// the line apart; this is for reporting it whole, so that an operator37// reading Machine status sees exactly what the bootloader passed,38// including any parameter liken itself does not look for.39func cmdlineRaw() string {40	raw, err := os.ReadFile(cmdlinePath)41	if err != nil {42		return ""43	}44	return strings.TrimSpace(string(raw))45}4647// bootParamValue returns the value of a name=value parameter on the48// kernel command line ("" when absent). Examples are which machine49// this is (liken.machine=) and which system slot booted it50// (liken.slot=).51func bootParamValue(name string) string {52	for _, field := range cmdlineFields() {53		if value, ok := strings.CutPrefix(field, name+"="); ok {54			return value55		}56	}57	return ""58}5960// bootParam reports whether a word appears on the kernel command61// line. This is liken's channel for per-boot behavior that is not62// machine configuration; machine configuration belongs in the63// Machine manifest. Parameter names use the liken.* prefix, to stay64// clear of the kernel's own parameters.65func bootParam(name string) bool {66	return slices.Contains(cmdlineFields(), name)67}
init/components.go 98.2%
1package main23// The machine plane.4//5// liken has exactly two planes. Machine-plane concerns are the loops6// a machine needs to function as a machine: reaping, watching for7// reboot intents, keeping the clock synchronized. These loops run as8// goroutines inside init, registered here. Workload-plane software9// runs as processes under k3s, and k3s itself is the only child10// process init ever supervises. A traditional distribution keeps a11// middle tier of system daemons, for example a time daemon, a device12// daemon, and a log daemon, under a service manager. liken has none13// of these daemons on purpose, which is why this file replaces most14// of what a service manager does.15//16// A concern belongs on the machine plane only when k3s depends on it17// to exist. Anything the cluster could host for itself belongs in18// the cluster, where Kubernetes is the supervisor. Time qualifies for19// the machine plane because a machine with a skewed clock fails TLS20// and cannot join the cluster, so the cluster can never correct the21// clock for it. A concern that k3s does not depend on runs in the22// cluster instead of in init.23//24// Running everything in one process is a real trade-off. It buys25// simplicity: dependency ordering follows program order, restart26// policy is a for loop, state uses shared structs instead of IPC, and27// the whole machine plane ships and upgrades as one binary. It costs28// isolation: there is no privilege separation, because every29// goroutine runs as PID 1, with root privileges, and fault isolation30// is imperfect. recover catches a panic, but a fatal runtime error,31// for example concurrent map misuse or an out-of-memory condition,32// ends all of PID 1, and the kernel answers that with a panic of its33// own. liken accepts this cost because it is crash-only by design:34// the root filesystem is RAM, the code proves manifests before it35// trusts them, and a reboot lands the machine in a known-good state.36//37// The rule has one exception. A component becomes a child process38// only when it parses untrusted network input, needs fewer39// privileges than PID 1, or must not take the machine down when it40// fails fatally. The child process is this same binary, re-executed41// with an argv verb (the multi-call pattern), so the OS stays one42// artifact. The NTP responder, which reads unauthenticated UDP, runs43// as a goroutine today. It is the first candidate for promotion to44// its own supervised process.45//46// The contract with a component is small: a component is a function47// of a context, and it runs until its work is done or the context is48// cancelled. An error return means the component failed and the code49// should restart it. Returning nil means the work is complete, and a50// component that reports on a boot milestone simply returns when it51// reaches the milestone. The code logs a panic with its stack and52// restarts the component with backoff, instead of letting the panic53// unwind PID 1 into a kernel panic. Shutdown runs the dependency54// stack in reverse: k3s and its containers stop first, then the code55// cancels this plane, then filesystems unmount. The wait is bounded,56// so one stuck loop cannot stall a reboot.5758import (59	"context"60	"fmt"61	"math/rand/v2"62	"os"63	"runtime/debug"64	"slices"65	"strings"66	"sync"67	"time"68)6970// plane is the boot's one machine plane. It is package-level, the71// same as the reaper's registry, because init is a single program72// and these are its loops. Tests construct their own planes; this73// variable holds the real boot's plane.74var plane = newMachinePlane()7576type machinePlane struct {77	ctx    context.Context78	cancel context.CancelFunc79	wg     sync.WaitGroup8081	// The names still running, so that a shutdown that times out can82	// name which component ignored it. On a machine with no shell,83	// the console message is the only place this fact can appear.84	mu      sync.Mutex85	running map[string]bool86}8788// Restart pacing, matching the k3s supervisor's pacing: a component89// that keeps failing waits twice as long each time. The wait is90// capped, so a truly broken loop still retries every half minute, and91// a component that ran for a while before failing starts over at the92// initial delay.93const (94	componentBackoff    = time.Second95	componentMaxBackoff = 30 * time.Second96)9798func newMachinePlane() *machinePlane {99	ctx, cancel := context.WithCancel(context.Background())100	return &machinePlane{101		ctx:     ctx,102		cancel:  cancel,103		running: map[string]bool{},104	}105}106107// start registers a component and runs it until it finishes or the108// plane shuts down, and restarts it on failure. start returns109// immediately, and there is no way to await a component; this is by110// design. Nothing in a boot sequences on another component's111// progress, and main calls anything that must happen before k3s112// starts synchronously instead.113func (p *machinePlane) start(name string, run func(context.Context) error) {114	p.mu.Lock()115	p.running[name] = true116	p.mu.Unlock()117118	p.wg.Add(1)119	go func() {120		defer p.wg.Done()121		defer func() {122			p.mu.Lock()123			delete(p.running, name)124			p.mu.Unlock()125		}()126127		backoff := componentBackoff128		for {129			started := time.Now()130			err := runComponent(p.ctx, run)131			if p.ctx.Err() != nil || err == nil {132				return133			}134			fmt.Fprintf(os.Stderr, "liken: %s: %v\n", name, err)135136			if time.Since(started) > time.Minute {137				backoff = componentBackoff138			} else if backoff < componentMaxBackoff {139				backoff *= 2140			}141			delay := withJitter(backoff)142			fmt.Printf("liken: restarting %s in %s\n", name, delay.Round(time.Millisecond))143			select {144			case <-time.After(delay):145			case <-p.ctx.Done():146				return147			}148		}149	}()150}151152// runComponent is the recovery boundary: a panicking component153// surfaces as an ordinary error, with its stack attached. Without154// this boundary, any panic anywhere in the machine plane would155// unwind PID 1 and panic the kernel. runComponent cannot catch156// everything; a fatal runtime error still ends the process. This is157// the imperfect fault isolation the header comment describes.158func runComponent(ctx context.Context, run func(context.Context) error) (err error) {159	defer func() {160		if r := recover(); r != nil {161			err = fmt.Errorf("panicked: %v\n%s", r, debug.Stack())162		}163	}()164	return run(ctx)165}166167// withJitter randomizes a delay upward by as much as half again. The168// point is fleet behavior, not this one machine's behavior. A power169// event reboots every machine together, and their components fail170// together too, for example because the network is not up yet or a171// leader is not back yet. Identical backoff would have every machine172// retry at the same instants. A random share spreads the retries173// out, so recovery load arrives gradually rather than all at once.174func withJitter(d time.Duration) time.Duration {175	return d + rand.N(d/2)176}177178// sleepUnlessCancelled is the pause a polling component takes between179// looks. It reports false when the plane is shutting down, which180// tells the component to return instead of polling again. Plain181// time.Sleep is off limits inside a component for exactly this182// reason: a loop blocked in time.Sleep cannot observe its183// cancellation.184func sleepUnlessCancelled(ctx context.Context, d time.Duration) bool {185	select {186	case <-ctx.Done():187		return false188	case <-time.After(d):189		return true190	}191}192193// shutdown cancels every component and waits for them, but only for194// a bounded time. The plane shuts down because the machine is going195// down, and a component that ignores its context must not be able196// to block a reboot. shutdown names any component still running at197// the timeout on the console and leaves it behind. This is safe,198// because the machine kills every process and reboots moments199// later.200func (p *machinePlane) shutdown(timeout time.Duration) {201	p.cancel()202	done := make(chan struct{})203	go func() {204		p.wg.Wait()205		close(done)206	}()207	select {208	case <-done:209	case <-time.After(timeout):210		p.mu.Lock()211		names := make([]string, 0, len(p.running))212		for name := range p.running {213			names = append(names, name)214		}215		p.mu.Unlock()216		slices.Sort(names)217		fmt.Fprintf(os.Stderr, "liken: shutdown proceeding without: %s\n", strings.Join(names, ", "))218	}219}
init/console.go 89.7%
1package main23// init's own log lines go to /dev/kmsg, the kernel's log buffer,4// instead of straight to the serial console. The kernel echoes every5// buffered record to the console anyway, with its printk timestamps6// added, which the raw writes never had. So the console reads as it7// always did. What changes is that init's lines now also exist8// somewhere else too: as structured records in the ring buffer,9// interleaved with the kernel's own records in true order, where the10// liken-logs relay can read them into the cluster. Records written11// through /dev/kmsg carry syslog facility 1, where the kernel's12// records carry facility 0. So the two streams separate by a field,13// instead of by guessing from prefixes.14//15// The mechanism reassigns the os.Stdout and os.Stderr package16// variables, which every fmt.Printf in this program reads at each17// call. This gives one change point, with no call sites touched. The18// variables point at the write ends of two pipes, and a goroutine19// per pipe carries complete lines into /dev/kmsg. The underlying20// file descriptors 1 and 2 are deliberately left alone, still aimed21// at the console the kernel opened for init: the Go runtime writes22// panic reports to fd 2 directly, and a panic in PID 1 must reach23// the console even when everything built here has stopped working.24//25// The pipes create this program's sharpest remaining risk. If a26// drainer goroutine ever died, its pipe would fill, and the next27// fmt.Printf would block forever, blocking PID 1. That is why the28// drainers are not machine-plane components: a component restart29// waits out a backoff, and the plane logs failures to os.Stderr,30// which would deadlock into the dead drainer's own pipe. For the31// same reason, each drainer's loop does nothing but split lines and32// write them, and any failure inside falls back to writing the33// console directly, rather than returning.34//35// A few lines per exec print before /dev exists and cannot go to36// kmsg: the hello line, the PID-1 refusal, and switch_root's37// messages. These lines stay console-only and are not shipped38// anywhere else, because there is nowhere else for them to go yet.3940import (41	"bytes"42	"fmt"43	"io"44	"os"45	"time"4647	"github.com/liken-sh/liken/liken/machine"48)4950// console is the raw serial console: file descriptor 1, exactly as51// the kernel opened it, captured before any reassignment. Output52// from child processes, for example k3s and mke2fs, writes here53// directly, and bypasses kmsg. At the volume k3s produces, the54// 256KiB ring buffer would fill and discard the kernel's own records55// within seconds. Also, the code already tails k3s's output into the56// cluster from its log file, so sending it through kmsg too would57// ship every line twice.58var console io.Writer = os.Stdout5960const (61	// Syslog priorities for init's two streams: the userspace62	// facility shifted past the three severity bits, plus info for63	// stdout and warning for stderr. The facility and severity64	// numbers form the wire contract with the log relays, so they65	// live in the machine package that both binaries import.66	kmsgInfo    = machine.FacilityUser<<3 | machine.SeverityInfo67	kmsgWarning = machine.FacilityUser<<3 | machine.SeverityWarning6869	// The kernel rejects /dev/kmsg writes longer than about 1KB70	// (LOG_LINE_MAX) with EINVAL; it does not truncate them. 80071	// bytes of payload leaves room for the priority prefix and the72	// continuation marks around split chunks.73	kmsgPayloadLimit = 80074)7576// The three kernel files the redirect writes. They are variables only77// so that a test can point the redirect at files it reads back.78var (79	printkDevkmsgPath = "/proc/sys/kernel/printk_devkmsg"80	printkPath        = "/proc/sys/kernel/printk"81	kmsgPath          = "/dev/kmsg"82)8384// redirectToKmsg points init's own output at /dev/kmsg. It runs85// after mountEssentials, because it needs /proc/sys and /dev/kmsg,86// and before the first machine-plane component starts, so the87// package variables are reassigned while main is still the only88// goroutine that reads them. Any failure stops the whole redirect89// and leaves the direct console writes in place: a machine that90// cannot buffer its logs still reports its boot on the console.91func redirectToKmsg() {92	// The kernel rate-limits userspace kmsg writes by default, to a93	// small burst every few seconds, which would silently discard94	// most of the boot report. This sysctl turns rate-limiting off95	// for the whole machine. It must run first, and it is required,96	// not a tuning option: without it, the redirect would lose lines97	// that the console used to show.98	if err := os.WriteFile(printkDevkmsgPath, []byte("on\n"), 0); err != nil {99		fmt.Printf("liken: logs stay on the console: printk_devkmsg: %v\n", err)100		return101	}102	// Assert that the console echoes severity-6 records: writing a103	// single number to this file sets console_loglevel, and 7 means104	// everything below debug prints. The kernel's default config105	// already sets 7; this guards against a quiet= setting in a106	// future config. This is best effort, because the echo is a107	// convenience while the buffer is the record of truth.108	if err := os.WriteFile(printkPath, []byte("7"), 0); err != nil {109		fmt.Printf("liken: setting console loglevel: %v\n", err)110	}111112	kmsg, err := os.OpenFile(kmsgPath, os.O_WRONLY, 0)113	if err != nil {114		fmt.Printf("liken: logs stay on the console: /dev/kmsg: %v\n", err)115		return116	}117118	outR, outW, err := os.Pipe()119	if err != nil {120		fmt.Printf("liken: logs stay on the console: pipe: %v\n", err)121		return122	}123	errR, errW, err := os.Pipe()124	if err != nil {125		outR.Close()126		outW.Close()127		fmt.Printf("liken: logs stay on the console: pipe: %v\n", err)128		return129	}130131	go drainToKmsg(outR, kmsg, kmsgInfo)132	go drainToKmsg(errR, kmsg, kmsgWarning)133	os.Stdout = outW134	os.Stderr = errW135	fmt.Println("liken: init logs via /dev/kmsg from here on (earlier lines were console-only)")136}137138// drainToKmsg carries one stream from its pipe into the kernel's139// buffer, one record per line. The loop must never end while the140// machine runs; the write end is os.Stdout or os.Stderr, and the141// code never closes it. The loop must never let a failure stop it142// either. emitKmsgLine absorbs everything, so the only exit is a143// read error, and a read error can only mean the process is dying144// anyway.145func drainToKmsg(r io.Reader, kmsg io.Writer, priority int) {146	var buf bytes.Buffer147	chunk := make([]byte, 4096)148	for {149		n, err := r.Read(chunk)150		if n > 0 {151			buf.Write(chunk[:n])152			for {153				line, rerr := buf.ReadBytes('\n')154				if rerr != nil {155					// No newline yet. Put the fragment back and156					// wait for the rest.157					buf.Write(line)158					break159				}160				emitKmsgLine(kmsg, priority, line[:len(line)-1])161			}162		}163		if err != nil {164			return165		}166	}167}168169// emitKmsgLine writes one line as one or more kmsg records, and170// cannot fail. A write error, or anything worse, sends the line to171// the raw console instead, because losing a log line is acceptable172// only after both destinations have refused it.173func emitKmsgLine(kmsg io.Writer, priority int, line []byte) {174	defer func() {175		if recover() != nil {176			fmt.Fprintf(console, "%s\n", line)177		}178	}()179	for _, part := range splitKmsgLine(line, kmsgPayloadLimit) {180		if _, err := fmt.Fprintf(kmsg, "<%d>%s", priority, part); err != nil {181			fmt.Fprintf(console, "%s\n", part)182		}183	}184}185186// splitKmsgLine cuts a line into pieces the kernel will accept as187// records, and marks the cuts. A piece that continues gets a188// trailing " ...", and a piece that continues another piece gets a189// leading "... ". This lets a reader of any one record tell that it190// is looking at a fragment. Most lines fit in one untouched piece;191// the lines that do not fit are things like the kernel command line192// echo in the world report.193func splitKmsgLine(line []byte, limit int) [][]byte {194	if len(line) <= limit {195		return [][]byte{line}196	}197	var parts [][]byte198	for start := 0; start < len(line); start += limit {199		end := min(start+limit, len(line))200		var part []byte201		if start > 0 {202			part = append(part, "... "...)203		}204		part = append(part, line[start:end]...)205		if end < len(line) {206			part = append(part, " ..."...)207		}208		parts = append(parts, part)209	}210	return parts211}212213// syncLogs pauses long enough for the drainers to move the last214// lines out of the pipes and into the kernel's buffer. The shutdown215// paths call it just before rebooting or powering off, so the final216// lines, usually the explanation of why the machine is going down,217// reach the kernel's buffer instead of being lost in a pipe. The218// pipes drain in microseconds, so 50ms is a generous bound. When the219// redirect never happened, the pause costs nothing that matters next220// to a reboot.221func syncLogs() {222	time.Sleep(50 * time.Millisecond)223}
init/crash.go 85.1%
1package main23// What killed the machine, answered by the machine itself.4//5// A kernel panic is the one failure the rest of liken's observability6// cannot see. The trace goes to the serial console, the baked7// panic=10 argument reboots the machine ten seconds later, and the8// firmware falls back to the proven slot. The machine comes back9// healthy, and nothing anywhere says why it went down.10//11// pstore is the kernel's answer. At the moment of a panic or an12// oops, the kernel writes the tail of its own log to a platform13// store that survives the reboot. On UEFI machines that store is the14// firmware's variable memory, through the efi_pstore backend that15// the image's fixed module list loads early in every boot. On the16// next boot, the kernel serves whatever the store holds as plain17// files under /sys/fs/pstore, one file per record.18//19// This file is the boot step that reads those files. It preserves20// them under machineState's crash store, because firmware variable21// memory is a few hundred kilobytes shared with the boot entries: a22// journal that small must be emptied after every read, or the next23// crash finds no room to record itself. It then derives a one-line24// summary, the newest crash's time, reason, and the kernel's own25// message, and hands it to the facts tree, which is how the fact26// becomes status.lastCrash in the cluster.27//28// Two rules shape everything here. First, the copy must land before29// the clear: deleting a pstore file erases the backing firmware30// variable, the only copy of the evidence. Second, every boot31// re-derives the summary from the preserved records, never from32// memory of an earlier boot, so an erased Machine status rebuilds33// exactly (status.go's reconstructibility rule).34//35// One gap is inherent: a panic that lands before the fixed module36// list loads leaves no record, because the backend was not yet37// registered. The window is a few hundred milliseconds at the top38// of boot.3940import (41	"errors"42	"fmt"43	"io"44	"os"45	"path/filepath"46	"regexp"47	"slices"48	"strconv"49	"strings"50	"time"51	"unicode"5253	"golang.org/x/sys/unix"5455	"github.com/liken-sh/liken/liken/machine"56)5758// pstoreDir is where the kernel serves the platform store's records.59// A variable rather than a constant, so tests can stand up a60// directory of fake records.61var pstoreDir = "/sys/fs/pstore"6263const (64	// crashKeep bounds the crash store. A crash is a dozen kilobytes,65	// so this bound is about tidiness, not space. There is no age66	// bound: an old crash stays on record, and its timestamp says how67	// old the news is.68	crashKeep = 106970	// crashRecordCap bounds one record read. A real record is at most71	// the kernel's 10 KiB dump budget; a backend broken enough to72	// serve more must not feed PID 1 unbounded bytes.73	crashRecordCap = 1 << 207475	// crashMessageCap bounds the summary message, matching the CRD76	// schema's maxLength. The full text stays in the records.77	crashMessageCap = 10247879	// crashDirFormat names one crash's directory by its moment, in80	// compact UTC. No colons, so the name would survive even a copy81	// onto FAT media.82	crashDirFormat = "20060102T150405Z"8384	// crashGroupWindow is how close two records' timestamps must be85	// to belong to one dump. Parts of one dump are written in the86	// same instant; the kernel's per-boot ordinals repeat across87	// boots, so time is what separates this boot's Panic#1 from last88	// month's.89	crashGroupWindow = 5 * time.Second90)9192// crashStore is where machineState keeps preserved crashes: one93// directory per crash, named by crashDirFormat.94func crashStore(stateDir string) string {95	return filepath.Join(stateDir, "crash")96}9798// mountPstore mounts the platform store's filesystem. The kernel99// creates the mount point itself when pstore is built in. EBUSY100// means something already mounted it, which serves the same purpose.101// The records are static, so mounting after a backend registers102// loses nothing: the filesystem populates from whatever the store103// holds.104func mountPstore() {105	err := unix.Mount("pstore", pstoreDir, "pstore",106		unix.MS_NOSUID|unix.MS_NODEV|unix.MS_NOEXEC, "")107	switch {108	case err == nil:109		fmt.Printf("liken: mounted pstore on %s\n", pstoreDir)110	case errors.Is(err, unix.EBUSY):111	default:112		fmt.Fprintf(os.Stderr, "liken: mounting pstore: %v\n", err)113	}114}115116// pstoreRecord is one file from the store: its exact bytes, its117// mtime (which pstore stamps with the wall clock at the moment of118// the dump), and, when the file is a readable kmsg dump, the parsed119// header. Records that do not parse still preserve; they are120// evidence, just not summarizable.121type pstoreRecord struct {122	name      string123	time      time.Time124	reason    string125	ordinal   int126	part      int127	body      string128	raw       []byte129	parseable bool130}131132// readPstoreRecords reads every record in a directory. It works on133// the live pstore mount and on a preserved crash directory alike,134// because preservation keeps names, bytes, and mtimes.135func readPstoreRecords(dir string) []pstoreRecord {136	entries, err := os.ReadDir(dir)137	if err != nil {138		return nil139	}140	var recs []pstoreRecord141	for _, e := range entries {142		if e.IsDir() {143			continue144		}145		info, err := e.Info()146		if err != nil {147			continue148		}149		f, err := os.Open(filepath.Join(dir, e.Name()))150		if err != nil {151			fmt.Fprintf(os.Stderr, "liken: crash: reading %s: %v\n", e.Name(), err)152			continue153		}154		raw, err := io.ReadAll(io.LimitReader(f, crashRecordCap))155		_ = f.Close()156		if err != nil {157			fmt.Fprintf(os.Stderr, "liken: crash: reading %s: %v\n", e.Name(), err)158			continue159		}160		rec := pstoreRecord{name: e.Name(), time: info.ModTime(), raw: raw}161		// Only plain dmesg records carry a parseable dump. The162		// .enc.z suffix marks a record the kernel could not163		// decompress; its payload is raw deflate, preserved verbatim164		// and never parsed.165		if strings.HasPrefix(rec.name, "dmesg-") && !strings.HasSuffix(rec.name, ".enc.z") {166			rec.reason, rec.ordinal, rec.part, rec.body, rec.parseable = parseDumpHeader(raw)167		}168		recs = append(recs, rec)169	}170	return recs171}172173// dumpHeaderRE matches the first line the kernel writes into every174// kmsg dump: the reason word, a per-boot ordinal, and the part175// number, as in "Panic#1 Part3". The reason vocabulary belongs to176// the kernel; Panic and Oops are the two words this configuration177// dumps.178var dumpHeaderRE = regexp.MustCompile(`^([A-Za-z]+)#(\d+) Part(\d+)$`)179180// parseDumpHeader splits a record into its header and its body. A181// ramoops zone opens with its own "====" timestamp line ahead of the182// header; tolerating it costs one comparison and keeps the parser183// honest against every backend this kernel can register.184func parseDumpHeader(data []byte) (reason string, ordinal, part int, body string, ok bool) {185	text := string(data)186	if strings.HasPrefix(text, "====") {187		if _, rest, found := strings.Cut(text, "\n"); found {188			text = rest189		}190	}191	header, rest, _ := strings.Cut(text, "\n")192	m := dumpHeaderRE.FindStringSubmatch(strings.TrimRight(header, "\r"))193	if m == nil {194		return "", 0, 0, "", false195	}196	ordinal, _ = strconv.Atoi(m[2])197	part, _ = strconv.Atoi(m[3])198	return m[1], ordinal, part, rest, true199}200201// crashGroup is one crash: every part of one kmsg dump.202type crashGroup struct {203	reason  string204	ordinal int205	time    time.Time206	parts   []pstoreRecord207}208209// text reassembles the dump in chronological order. The kernel hands210// the log tail out newest-first, so part 1 holds the newest lines,211// including the panic message itself, and higher parts hold212// progressively older lines. Reading time order means reading parts213// in descending number.214func (g crashGroup) text() string {215	parts := slices.Clone(g.parts)216	slices.SortFunc(parts, func(a, b pstoreRecord) int { return b.part - a.part })217	var b strings.Builder218	for _, p := range parts {219		b.WriteString(p.body)220		if !strings.HasSuffix(p.body, "\n") {221			b.WriteString("\n")222		}223	}224	return b.String()225}226227// groupCrashes folds parseable records into crashes, newest first.228// Parts of one dump share a reason and an ordinal and were written229// in the same instant. The ordinal alone cannot identify a dump,230// because it counts per boot and restarts at one, so the timestamp231// window is what keeps two boots' first panics apart.232func groupCrashes(recs []pstoreRecord) []crashGroup {233	sorted := slices.Clone(recs)234	slices.SortFunc(sorted, func(a, b pstoreRecord) int { return b.time.Compare(a.time) })235	var groups []crashGroup236	for _, rec := range sorted {237		if !rec.parseable {238			continue239		}240		joined := false241		for i := range groups {242			g := &groups[i]243			if g.reason == rec.reason && g.ordinal == rec.ordinal &&244				g.time.Sub(rec.time) <= crashGroupWindow {245				g.parts = append(g.parts, rec)246				joined = true247				break248			}249		}250		if !joined {251			groups = append(groups, crashGroup{252				reason:  rec.reason,253				ordinal: rec.ordinal,254				time:    rec.time,255				parts:   []pstoreRecord{rec},256			})257		}258	}259	return groups260}261262// syslogPrefixRE strips the "<level>[ timestamp] " that kmsg dumps263// carry on every body line.264var syslogPrefixRE = regexp.MustCompile(`^<\d+>\[\s*\d+\.\d+\] ?`)265266// crashMessage picks the one line that says what happened, in the267// kernel's own words: the panic line when there is one, the oops's268// BUG line otherwise, then the oops code line, then the first line269// of the dump. Panic text can carry arbitrary bytes, and this string270// travels into a YAML file and a Kubernetes API write, so it leaves271// here printable and bounded or not at all.272func crashMessage(text string) string {273	var lines []string274	for line := range strings.SplitSeq(text, "\n") {275		line = syslogPrefixRE.ReplaceAllString(strings.TrimRight(line, "\r"), "")276		if strings.TrimSpace(line) != "" {277			lines = append(lines, line)278		}279	}280	for _, marker := range []string{"Kernel panic - not syncing:", "BUG:", "Oops:"} {281		for _, line := range lines {282			if idx := strings.Index(line, marker); idx >= 0 {283				return sanitizeCrashMessage(line[idx:])284			}285		}286	}287	if len(lines) > 0 {288		return sanitizeCrashMessage(lines[0])289	}290	return ""291}292293// sanitizeCrashMessage drops everything unprintable and cuts the294// string at the cap, on a rune boundary.295func sanitizeCrashMessage(s string) string {296	s = strings.Map(func(r rune) rune {297		if r == '\t' {298			return ' '299		}300		if !unicode.IsPrint(r) {301			return -1302		}303		return r304	}, strings.ToValidUTF8(s, ""))305	for len(s) > crashMessageCap {306		_, size := lastRune(s)307		s = s[:len(s)-size]308	}309	return strings.TrimSpace(s)310}311312// lastRune reports the final rune of a non-empty string and its313// width in bytes.314func lastRune(s string) (rune, int) {315	r := []rune(s)316	last := r[len(r)-1]317	return last, len(string(last))318}319320// crashSummary condenses one crash into the status stub, with the321// records field naming where the full text lives.322func crashSummary(g crashGroup, records string) *machine.CrashStatus {323	t := g.time324	return &machine.CrashStatus{325		Time:    &t,326		Reason:  machine.CrashReason(g.reason),327		Message: crashMessage(g.text()),328		Records: records,329	}330}331332// preserveCrashRecords copies every record, verbatim and with its333// crash-time mtime, into one directory per crash batch. The copies334// and their directory sync to disk before this function returns,335// because the caller's next step erases the originals. A directory336// that already exists does not prove the copy is whole: a prior boot337// can die, or fail a write, partway through the batch. So every338// record is written again, which costs a few kilobytes and happens339// only while the store still holds the batch.340func preserveCrashRecords(recs []pstoreRecord, dest string) error {341	if err := os.MkdirAll(dest, 0o755); err != nil {342		return err343	}344	for _, rec := range recs {345		path := filepath.Join(dest, rec.name)346		f, err := os.OpenFile(path, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0o644)347		if err != nil {348			return err349		}350		if _, err := f.Write(rec.raw); err != nil {351			_ = f.Close()352			return err353		}354		if err := f.Sync(); err != nil {355			_ = f.Close()356			return err357		}358		if err := f.Close(); err != nil {359			return err360		}361		if err := os.Chtimes(path, rec.time, rec.time); err != nil {362			return err363		}364	}365	// The batch's own entry in the crash store must reach the disk366	// too, or a power cut can lose the whole directory.367	for _, d := range []string{dest, filepath.Dir(dest)} {368		if err := syncDirectory(d); err != nil {369			return err370		}371	}372	return nil373}374375// clearPstore erases records from the platform store. Unlinking a376// pstore file is what deletes the backing firmware variable. A377// failed erase is reported and left; the record comes back next378// boot, and the preserve step's already-exists rule keeps the retry379// cheap.380func clearPstore(dir string, recs []pstoreRecord) {381	for _, rec := range recs {382		if err := os.Remove(filepath.Join(dir, rec.name)); err != nil {383			fmt.Fprintf(os.Stderr, "liken: crash: clearing %s: %v\n", rec.name, err)384		}385	}386}387388// pruneCrashStore keeps the newest crashes and removes the rest. The389// directory names are timestamps in a sortable format, so name order390// is time order.391func pruneCrashStore(dir string, keep int) {392	entries, err := os.ReadDir(dir)393	if err != nil {394		return395	}396	var names []string397	for _, e := range entries {398		if e.IsDir() {399			names = append(names, e.Name())400		}401	}402	slices.Sort(names)403	slices.Reverse(names)404	for _, name := range names[min(keep, len(names)):] {405		if err := os.RemoveAll(filepath.Join(dir, name)); err != nil {406			fmt.Fprintf(os.Stderr, "liken: crash: pruning %s: %v\n", name, err)407		}408	}409}410411// latestPreservedCrash re-derives the summary from the crash store:412// the newest preserved crash that parses. This runs on every boot,413// crash or no crash, and is what makes status.lastCrash414// reconstructible; the store is the fact, and the status is only a415// reading of it.416func latestPreservedCrash(store string) *machine.CrashStatus {417	entries, err := os.ReadDir(store)418	if err != nil {419		return nil420	}421	var names []string422	for _, e := range entries {423		if e.IsDir() {424			names = append(names, e.Name())425		}426	}427	slices.Sort(names)428	slices.Reverse(names)429	for _, name := range names {430		dir := filepath.Join(store, name)431		if groups := groupCrashes(readPstoreRecords(dir)); len(groups) > 0 {432			return crashSummary(groups[0], dir)433		}434	}435	return nil436}437438// settleCrashRecords is the boot step: read the platform store,439// preserve and clear it when machineState is durable, and report the440// newest crash on record. On a machine whose machineState fell back441// to memory, the platform store is the only durable storage there442// is, so the records stay in it; only crashes older than the newest443// leave, to keep the firmware's small variable memory from filling444// and silently blocking the next dump.445func settleCrashRecords(stateDir string, durable bool) *machine.CrashStatus {446	store := crashStore(stateDir)447	recs := readPstoreRecords(pstoreDir)448	groups := groupCrashes(recs)449450	// fresh is the summary of records that remain in pstore, either451	// because this machine has nowhere better to keep them, or452	// because preserving them failed.453	var fresh *machine.CrashStatus454	if len(recs) > 0 {455		if durable {456			dest := filepath.Join(store, newestRecordTime(recs).UTC().Format(crashDirFormat))457			if err := preserveCrashRecords(recs, dest); err != nil {458				fmt.Fprintf(os.Stderr, "liken: crash: preserving records: %v (they stay in pstore)\n", err)459				if len(groups) > 0 {460					fresh = crashSummary(groups[0], pstoreDir)461				}462			} else {463				fmt.Printf("liken: crash: preserved %d pstore records to %s\n", len(recs), dest)464				clearPstore(pstoreDir, recs)465			}466		} else {467			for _, g := range groups[min(1, len(groups)):] {468				clearPstore(pstoreDir, g.parts)469			}470			if len(groups) > 0 {471				fresh = crashSummary(groups[0], pstoreDir)472			}473		}474	}475476	var summary *machine.CrashStatus477	if durable {478		pruneCrashStore(store, crashKeep)479		summary = latestPreservedCrash(store)480	}481	if summary == nil || (fresh != nil && fresh.Time.After(*summary.Time)) {482		summary = fresh483	}484485	if summary != nil {486		// Console parity: the fact prints where an operator at the487		// serial port can see it, in the same words status carries.488		fmt.Printf("liken: crash: last kernel %s at %s: %s (records: %s)\n",489			summary.Reason, summary.Time.UTC().Format(time.RFC3339),490			summary.Message, summary.Records)491	}492	return summary493}494495// newestRecordTime is the batch's moment: the newest record's mtime,496// which names the batch's directory in the crash store.497func newestRecordTime(recs []pstoreRecord) time.Time {498	newest := recs[0].time499	for _, rec := range recs[1:] {500		if rec.time.After(newest) {501			newest = rec.time502		}503	}504	return newest505}
init/diskids.go 92.2%
1package main23// Stable identity names for local disks, computed from firmware4// values rather than kernel-assigned letters.5//6// A WWN (a world-unique ID number), a model string, and a serial7// number are values a disk's own controller reports over the wire.8// They live in the drive's firmware, not on its platters or its9// flash, so an image written onto a disk, or a whole disk cloned onto10// another, carries none of them: the copy still answers with its own11// controller's identity. That is what makes these values fit for12// naming a disk in a Machine spec. A kernel letter names a slot in13// probe order, which depends on which disk answered first, and a14// filesystem label is data on the disk, which an image write15// replaces.16//17// disklinks.go states why liken reads these values itself instead of18// running udev: everything udev publishes comes from files sysfs19// already exposes, so liken reads the same files. This holds here20// too. udev's by-id rules build several names from a disk's wwid,21// model, and serial attributes, in the sysfs directories this file22// reads.23//24// The ata-<model>_<serial> name a SATA disk answers to does not come25// from the model attribute sysfs publishes for it. That attribute26// holds the 16-byte SCSI INQUIRY string libata's SCSI translation27// layer reports, truncated to fit a field SCSI defines, not the28// 40-byte model string udev's ata_id helper decodes straight from the29// drive's own ATA IDENTIFY DEVICE data. A name built from the shorter30// string would not match the name udev gives the same disk. libata31// answers SCSI vital product data page 0x89, "ATA Information", by32// embedding the entire 512-byte IDENTIFY block in the response, so33// this file reads that page instead. The bytes are the same IDENTIFY34// data udev's helper asks the drive for with an ioctl.3536import (37	"fmt"38	"os"39	"path/filepath"40	"regexp"41	"strings"42)4344// diskIDNames lists the by-id names one disk answers to, in a fixed45// order: the WWN-derived names first, then the name specific to the46// disk's transport. The order is not significant to a reader of the47// names, but a fixed order keeps the tests, and any future diff of48// them, readable.49func diskIDNames(name string) []string {50	dir := filepath.Join(sysBlock, name)51	var names []string5253	// NVMe publishes its own wwid at the block level; SCSI and SATA54	// disks publish it on the bus device underneath. Either location55	// can hold a NAA-format World Wide Name, the SCSI standard's56	// world-unique disk identifier, encoded as hex.57	wwid := sysfsString(dir, "wwid", "device/wwid")58	if hex, ok := strings.CutPrefix(wwid, "naa."); ok {59		hex = sanitizeIDPart(hex)60		// udev publishes both a wwn- name (its own convention) and a61		// scsi-3 name (the SCSI standard's own prefix for a NAA62		// identifier) for the same value, and other software looks63		// for either, so liken computes both.64		names = append(names, "wwn-0x"+hex, "scsi-3"+hex)65	}6667	switch diskTransport(name) {68	case "sata":69		if model, serial := scsiVPD89ATAIdentity(dir); model != "" && serial != "" {70			names = append(names, "ata-"+sanitizeIDPart(model)+"_"+sanitizeIDPart(serial))71		}72	case "nvme":73		model := sysfsString(dir, "device/model")74		serial := sysfsString(dir, "serial", "device/serial")75		if serial == "" {76			// A drive that answers SCSI inquiries under NVMe's SCSI77			// translation layer, or that a virtualized controller78			// exposes without a plain serial attribute, still answers79			// vital product data page 0x80 with its serial.80			serial = scsiVPD80Serial(dir)81		}82		if model != "" && serial != "" {83			names = append(names, "nvme-"+sanitizeIDPart(model)+"_"+sanitizeIDPart(serial))84		}85		if hex, ok := strings.CutPrefix(wwid, "eui."); ok {86			names = append(names, "nvme-eui."+sanitizeIDPart(hex))87		}88	case "mmc":89		// An mmc card carries its product name and serial in its CID90		// register, which the card's own controller answers with, so91		// the name survives an image write the way the ata- names do.92		// The kernel publishes the serial as 0x-prefixed hex, and the93		// prefix stays in the name: udev's by-id rule substitutes the94		// attribute exactly as sysfs prints it, and real machines95		// carry links like mmc-SA16G_0xbee5b555. A name without the96		// prefix would not match the name udev gives the same disk.97		card := sysfsString(dir, "device/name")98		serial := sysfsString(dir, "device/serial")99		if card != "" && serial != "" {100			names = append(names, "mmc-"+sanitizeIDPart(card)+"_"+sanitizeIDPart(serial))101		}102	case "virtio":103		// virtio-blk publishes only a serial, directly on the disk,104		// because a virtual disk has no model or vendor to report.105		if serial := sysfsString(dir, "serial"); serial != "" {106			names = append(names, "virtio-"+sanitizeIDPart(serial))107		}108	case "usb":109		if usb := usbIDName(dir); usb != "" {110			names = append(names, usb)111		}112	}113114	return names115}116117// usbIDName builds the by-id name for a disk that reaches liken over118// USB, from the manufacturer, product, and serial the USB device119// itself publishes, and the SCSI LUN usb-storage assigned it.120func usbIDName(dir string) string {121	real, err := filepath.EvalSymlinks(dir)122	if err != nil {123		return ""124	}125	device := usbDeviceDir(real)126	if device == "" {127		return ""128	}129	manufacturer := sysfsString(device, "manufacturer")130	product := sysfsString(device, "product")131	serial := sysfsString(device, "serial")132	if manufacturer == "" || product == "" || serial == "" {133		return ""134	}135	return fmt.Sprintf("usb-%s_%s_%s-0:%s",136		sanitizeIDPart(manufacturer), sanitizeIDPart(product), sanitizeIDPart(serial),137		scsiLUN(real))138}139140// usbDeviceDir walks a resolved sysfs path upward to the directory141// that names the USB device itself, as against the interface, the142// SCSI host, or the target beneath it that also sit on this path.143// idVendor exists only on the device's own directory, because only144// the device, not the things usb-storage attaches beneath it, has a145// USB vendor ID to report.146func usbDeviceDir(path string) string {147	for dir := path; dir != "/" && dir != "."; dir = filepath.Dir(dir) {148		if _, err := os.Stat(filepath.Join(dir, "idVendor")); err == nil {149			return dir150		}151	}152	return ""153}154155// hctlPattern matches a SCSI address segment in a sysfs path: host,156// channel, target, and LUN, each a plain number, the way the kernel157// names a SCSI device's directory.158var hctlPattern = regexp.MustCompile(`/(\d+):(\d+):(\d+):(\d+)(/|$)`)159160// scsiLUN reads the LUN out of the h:c:t:l segment in a resolved161// sysfs path. usb-storage presents a USB mass-storage device to the162// kernel as a virtual SCSI host, and addresses each LUN under it as163// host:channel:target:lun, the way any SCSI host does; the final164// field is the LUN a multi-LUN device, such as a card reader with165// several slots, uses to tell its LUNs apart. A path with no such166// segment names LUN 0, the only LUN a device with no LUN addressing167// at all ever uses.168func scsiLUN(path string) string {169	if m := hctlPattern.FindStringSubmatch(path); m != nil {170		return m[4]171	}172	return "0"173}174175// scsiVPD80Serial reads a disk's serial out of vital product data176// page 0x80, a binary SCSI page that every SCSI target, and every177// SATA disk through libata's SCSI translation, answers to. The page178// is four header bytes (peripheral qualifier and type, the page179// code, then the ASCII payload's length as a two-byte big-endian180// count) followed by the serial itself. diskIDNames reads this page181// as a fallback, for a disk whose driver does not also mirror the182// serial into a plain sysfs attribute.183func scsiVPD80Serial(dir string) string {184	raw, err := os.ReadFile(filepath.Join(dir, "device", "vpd_pg80"))185	if err != nil || len(raw) < 4 {186		return ""187	}188	length := int(raw[2])<<8 | int(raw[3])189	if len(raw) < 4+length {190		return ""191	}192	return strings.Trim(string(raw[4:4+length]), "\x00 \t\n\r\v\f")193}194195// scsiVPD89ATAIdentity reads a SATA disk's model and serial out of196// vital product data page 0x89, "ATA Information", which libata197// answers by embedding the drive's full 512-byte ATA IDENTIFY DEVICE198// block after a 60-byte header. A page shorter than 572 bytes carries199// no complete IDENTIFY block, so this function reports neither field.200//201// IDENTIFY packs each string byte-swapped within its 16-bit words: a202// drive that reports "QEMU HARDDISK" stores it as "EQUMH RADDSI K".203// The serial sits at words 10-19 of the block, the model at words204// 27-46; both are ASCII, space-padded to their field width, so this205// function swaps each field's bytes back and trims the padding.206func scsiVPD89ATAIdentity(dir string) (model, serial string) {207	raw, err := os.ReadFile(filepath.Join(dir, "device", "vpd_pg89"))208	if err != nil || len(raw) < 572 {209		return "", ""210	}211	identify := raw[60:572]212	serial = ataIdentifyString(identify[20:40])213	model = ataIdentifyString(identify[54:94])214	return model, serial215}216217// ataIdentifyString decodes one fixed-width string out of an ATA218// IDENTIFY DEVICE block: it swaps every adjacent byte pair back into219// reading order, then trims the spaces and NUL bytes the drive pads220// the field with.221func ataIdentifyString(field []byte) string {222	swapped := make([]byte, len(field))223	for i := 0; i+1 < len(field); i += 2 {224		swapped[i], swapped[i+1] = field[i+1], field[i]225	}226	return strings.Trim(string(swapped), "\x00 \t\n\r\v\f")227}228229// idCharset is every byte udev leaves alone when it builds a by-id230// name out of a firmware string. udev's own replace_whitespace and231// blacklist rules keep a name shell-safe and safe as a single path232// segment: a value must not smuggle a slash into the file name it233// becomes, or whitespace that would need quoting to reference again.234const idCharset = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789#+.:=@_-"235236// sanitizeIDPart makes one firmware string, such as a model or237// serial, safe to fold into a by-id name. Every byte outside238// idCharset, including the spaces a fixed-width SCSI field pads out239// with and the NUL bytes some firmware pads with instead, becomes an240// underscore.241func sanitizeIDPart(s string) string {242	var b strings.Builder243	b.Grow(len(s))244	for i := 0; i < len(s); i++ {245		c := s[i]246		if strings.IndexByte(idCharset, c) >= 0 {247			b.WriteByte(c)248		} else {249			b.WriteByte('_')250		}251	}252	return b.String()253}
init/diskpaths.go 96.2%
1package main23// Local by-path names: udev's path_id for a disk on this machine's4// own PCI, SATA, USB, or virtio bus.5//6// A by-id name follows the disk. A by-path name follows the port: it7// stays the same when the disk moves, because it names the wire that8// connects the disk rather than the disk's own firmware. This is why a9// port name matters at all, on top of the identity names diskids.go10// already builds. Wiping and swapping in a spare drive is easier when11// an operator can say "whatever disk uses this bay" instead12// of tracking a serial number, and a spec that says so keeps working13// after the swap.14//15// udev builds the same name in its path_id builtin, by walking a16// disk's chain of parent devices in sysfs from the disk up to the17// bus. Every step of that chain is a directory name the kernel18// already publishes, in the resolved path behind /sys/block/<name>,19// so this file walks the same chain udev does and reads nothing udev20// would not also read.2122import (23	"fmt"24	"path/filepath"25	"regexp"26	"strings"27)2829// pciSlotPattern matches a PCI address segment: a four-digit domain, a30// two-digit bus, and a two-digit device and function, the way the31// kernel names a PCI device's directory. A disk's path can pass32// through more than one such segment, for example a root port and the33// storage controller behind it, so diskPathName keeps the last match:34// the slot closest to the disk is the one that identifies its port.35var pciSlotPattern = regexp.MustCompile(`^[0-9a-fA-F]{4}:[0-9a-fA-F]{2}:[0-9a-fA-F]{2}\.[0-9a-fA-F]$`)3637// ataSegmentPattern matches the ataN directory libata creates for38// each port it drives, one per SATA disk.39var ataSegmentPattern = regexp.MustCompile(`^ata(\d+)$`)4041// virtioSegmentPattern matches the virtioN directory the virtio bus42// creates for each device, disk or otherwise, attached to it.43var virtioSegmentPattern = regexp.MustCompile(`^virtio\d+$`)4445// usbPortSegmentPattern matches a USB device's own directory, named46// for its position in the bus: the bus number, a dash, and the port47// path down through any hubs to reach it. A USB device's interface48// is one level below it, in a directory named the same way with49// ":<interface>" appended, which this pattern excludes on purpose so50// it never matches the interface by mistake.51var usbPortSegmentPattern = regexp.MustCompile(`^\d+-[0-9.]+$`)5253// iscsiSessionPattern matches the sessionN directory scsi_transport_iscsi54// creates for each login. A session number is reassigned on every55// login, so a name built from it would change out from under a mount56// on the next reboot; disklinks.go already gives an iSCSI disk a57// stable by-path name built from its target and address instead.58var iscsiSessionPattern = regexp.MustCompile(`^session\d+$`)5960// nvmeNsidPattern reads the namespace ID off the end of an NVMe block61// device's own name, for example the 1 in nvme0n1. diskPathName only62// falls back to this parse when the kernel's own nsid attribute is63// absent, because the two numbers agree only when a controller's64// namespaces are created in order starting at 1: the block name's65// trailing digit is the Nth namespace this controller probed, while66// the nsid attribute is the namespace's own number as the controller67// reports it. A controller that lost namespace 1 and kept only68// namespace 2 still names its sole block device nvme0n1, because the69// kernel numbers by probe order, so a name built from that digit70// would claim namespace 1 while the disk is actually namespace 2.71var nvmeNsidPattern = regexp.MustCompile(`n(\d+)$`)7273// mmcHostSegment is the directory the mmc core creates under a host74// controller for the cards it manages. It is directly below the75// controller's own device, so the segment before it names the76// controller.77const mmcHostSegment = "mmc_host"7879// mmcPathName is udev's by-path answer for an eMMC module or an SD80// card, which is a name only when the host controller is a platform81// device.82//83// udev's path_id builtin has no mmc handler at all. It walks the84// parents of a block device and prepends a segment for each bus it85// recognizes, and mmc is not one of them. A card's parents are the mmc86// card, the mmc host, and then the controller, so the name a card gets87// is the name of its controller and nothing else. A controller the88// firmware enumerates through ACPI or a device tree is a platform89// device, and path_id names a platform device platform-<name>, with90// the PCI slot ahead of it when the platform device is below a PCI91// function. That is where an eMMC's by-path name comes from.92//93// A controller that enumerates over PCI on its own, which is what94// sdhci-pci drives, gets no name. path_id counts a PCI slot as a95// parent and not as a transport, and it refuses to name a block device96// whose walk found no transport, so udev writes no ID_PATH and builds97// no link under /dev/disk/by-path. liken publishes no name there98// either, because a by-path name that udev does not build is a name no99// other tool on the machine agrees with. Such a card is named by its100// by-id name alone (diskids.go).101func mmcPathName(segments []string, pciSlot string) string {102	host := -1103	for i, seg := range segments {104		if seg == mmcHostSegment {105			host = i - 1106			break107		}108	}109	if host < 0 {110		return ""111	}112	dir := strings.Join(segments[:host+1], string(filepath.Separator))113	subsystem, err := filepath.EvalSymlinks(filepath.Join(dir, "subsystem"))114	if err != nil || filepath.Base(subsystem) != "platform" {115		return ""116	}117	if pciSlot != "" {118		return "pci-" + pciSlot + "-platform-" + segments[host]119	}120	return "platform-" + segments[host]121}122123// diskPathName computes the local by-path name for one disk: the port124// it answers on, read from the resolved sysfs path behind125// /sys/block/<name>. It returns "" for a disk with no such name,126// which includes any disk this machine reaches over iSCSI, and any127// disk whose bus this function does not recognize.128func diskPathName(name string) string {129	real, err := filepath.EvalSymlinks(filepath.Join(sysBlock, name))130	if err != nil {131		return ""132	}133	segments := strings.Split(real, string(filepath.Separator))134135	for _, seg := range segments {136		if iscsiSessionPattern.MatchString(seg) {137			return ""138		}139	}140141	pciSlot := ""142	for _, seg := range segments {143		if pciSlotPattern.MatchString(seg) {144			pciSlot = seg145		}146	}147148	if name := mmcPathName(segments, pciSlot); name != "" {149		return name150	}151152	if pciSlot == "" {153		return ""154	}155156	for _, seg := range segments {157		if m := ataSegmentPattern.FindStringSubmatch(seg); m != nil {158			return "pci-" + pciSlot + "-ata-" + m[1]159		}160	}161	for _, seg := range segments {162		if seg == "nvme" {163			// udev's path_id builtin reads the namespace's own nsid164			// attribute rather than parsing the block name, so this165			// does the same and only parses the name on an old kernel166			// that predates the attribute.167			if nsid := sysfsString(filepath.Join(sysBlock, name), "nsid"); nsid != "" {168				return "pci-" + pciSlot + "-nvme-" + nsid169			}170			if nsid := nvmeNsidPattern.FindStringSubmatch(name); nsid != nil {171				return "pci-" + pciSlot + "-nvme-" + nsid[1]172			}173			return ""174		}175	}176	for _, seg := range segments {177		if !usbPortSegmentPattern.MatchString(seg) {178			continue179		}180		hctl := hctlPattern.FindStringSubmatch(real)181		if hctl == nil {182			return ""183		}184		portpath := strings.SplitN(seg, "-", 2)[1]185		return fmt.Sprintf("pci-%s-usb-0:%s:1.0-scsi-%s:%s:%s:%s",186			pciSlot, portpath, hctl[1], hctl[2], hctl[3], hctl[4])187	}188	for _, seg := range segments {189		if virtioSegmentPattern.MatchString(seg) {190			// The virtio segment names the device but contributes191			// nothing to the name beyond the PCI slot: a virtio-blk192			// device has no port structure of its own to describe, only193			// the PCI function QEMU or the hypervisor assigned it. This194			// matches udev's own path_id rule for virtio.195			return "pci-" + pciSlot196		}197	}198199	return ""200}
init/disks.go 100.0%
1package main23// The machine's disks, discovered by asking the kernel directly.4//5// A full distribution answers the question "what disks does this6// machine have?" with udev: a daemon that receives device events,7// runs a rules engine, and publishes its conclusions as a tree of8// symlinks under /dev/disk. liken does not need the daemon. udev9// gets all of its information by reading sysfs, the kernel's live10// object model, already mounted at /sys. liken is the only program11// here that needs the answer, so it reads the same files itself.12// Each sysfs attribute is a small text file that holds one value, so13// discovery is only directory walks and file reads.1415import (16	"fmt"17	"os"18	"path/filepath"19	"strconv"20	"strings"2122	"github.com/liken-sh/liken/liken/machine"23)2425// These are the roots discovery reads from. They are variables rather26// than constants so that tests can set up a fake machine in a27// tempdir. On a real boot, they never hold anything else.28var (29	sysBlock = "/sys/block"30	devRoot  = "/dev"31)3233// devicePath is the node devtmpfs maintains for a disk. The name34// under /dev is the same name sysfs lists; the kernel assigns it in35// both places.36func devicePath(d machine.BlockDevice) string {37	return devRoot + "/" + d.Name38}3940// discoverBlockDevices lists the machine's disks: one entry per41// directory in /sys/block that represents real storage. The result42// uses the status type directly, because this same inventory appears43// both in the world report and in the facts init publishes.44func discoverBlockDevices() []machine.BlockDevice {45	entries, err := os.ReadDir(sysBlock)46	if err != nil {47		fmt.Fprintf(os.Stderr, "liken: reading %s: %v\n", sysBlock, err)48		return nil49	}5051	var disks []machine.BlockDevice52	for _, entry := range entries {53		name := entry.Name()54		dir := filepath.Join(sysBlock, name)5556		// /sys/block lists every block device, including purely57		// virtual ones that the kernel can create without hardware,58		// for example loop, ram, and zram devices. The `device`59		// symlink marks real storage: it points back at the bus60		// device (PCI, virtio, USB) that provides the disk, and61		// virtual devices have no such parent.62		if _, err := os.Stat(filepath.Join(dir, "device")); err != nil {63			continue64		}6566		if mmcHardwareArea(dir) {67			continue68		}6970		d := machine.BlockDevice{Name: name}7172		// The size file counts sectors of 512 bytes. It always uses73		// 512, whatever the disk's real sector size is, because the74		// kernel's ABI fixed that unit decades ago.75		if raw, err := os.ReadFile(filepath.Join(dir, "size")); err == nil {76			if sectors, err := strconv.ParseUint(strings.TrimSpace(string(raw)), 10, 64); err == nil {77				d.SizeBytes = sectors * 51278			}79		}8081		// Which identifying attributes exist depends on the bus. NVMe82		// and SCSI disks publish a model, and often a serial, on the83		// bus device. virtio-blk publishes only a serial, directly on84		// the disk. The code reads whatever this disk offers; absence85		// is normal.86		d.Model = sysfsString(dir, "device/model")87		d.Serial = sysfsString(dir, "serial", "device/serial")88		if d.Serial == "" {89			// Some drivers, and some virtualized controllers, answer90			// SCSI inquiries but mirror no serial into a plain sysfs91			// attribute. Vital product data page 0x80 is the92			// fallback: every SCSI target, and every SATA disk93			// through libata's SCSI translation, answers it.94			d.Serial = scsiVPD80Serial(dir)95		}9697		d.StableNames = stableNames(name)9899		disks = append(disks, d)100	}101	return disks102}103104// mmcHardwareArea reports whether a block device is one of the areas105// an eMMC card presents beside its data area: the two boot areas, the106// general-purpose areas, and RPMB. No role may land on any of them.107// The boot areas hold what a board's firmware reads before Linux108// runs, RPMB is an authenticated mailbox rather than storage, and the109// general-purpose areas are partitions the card's own controller110// carved. Only the data area is a disk.111//112// The marker is structural, not a name pattern. The mmc block driver113// registers each hardware area as a child of the data area's own114// block device, so an area's `device` link leads to a device in the115// block subsystem, where every real disk's `device` link leads to the116// bus device that carries it (PCI, virtio, USB, mmc). Only the mmc117// driver builds that shape today, so on a machine with no mmc118// hardware this function returns false for every device and the walk119// is unchanged.120//121// RPMB needs no check of its own. Kernels from 3.8 to 4.14 built122// its block device with the data area as its parent, the same shape123// the boot areas have, so the structural check covers it there, and124// kernels from 4.15 on publish RPMB as a character device that125// never reaches this walk.126func mmcHardwareArea(dir string) bool {127	subsystem, err := filepath.EvalSymlinks(filepath.Join(dir, "device", "subsystem"))128	if err != nil {129		return false130	}131	return filepath.Base(subsystem) == "block"132}133134// stableDiskByID and stableDiskByPath are the directories udev's own135// rules populate under /dev/disk, and the ones this file's own link136// trees (disklinks.go and its siblings) populate in the same shape.137// discoverBlockDevices renders StableNames as paths under these138// directories so that a name in status is the exact path a spec can139// paste back in, whether or not the link that path names has been140// built yet: discovery computes a disk's identity from sysfs141// directly, the same way the link trees do, rather than reading the142// trees back.143const (144	stableDiskByID   = "/dev/disk/by-id/"145	stableDiskByPath = "/dev/disk/by-path/"146)147148// stableNames renders the full paths of every name that identifies149// one disk across boots: every by-id name first, because a by-id150// name follows the disk's own firmware, then the by-path name, when151// the disk's port resolves to one, because a by-path name follows152// the port instead. A disk with neither, for example a SATA disk153// with no WWN behind a bus diskPathName does not recognize, gets no154// stable names at all.155func stableNames(name string) []string {156	var names []string157	for _, id := range diskIDNames(name) {158		names = append(names, stableDiskByID+id)159	}160	if path := diskPathName(name); path != "" {161		names = append(names, stableDiskByPath+path)162	}163	return names164}165166// sysfsString reads the first of the named attributes that exists, as167// a trimmed string. The trimming matters, because padding reaches a168// value from more than one place and none of it belongs to the value.169// sysfs ends every attribute with a newline. SCSI model strings carry170// spaces out to their on-wire field width. A target that reports a171// serial from a fixed-width buffer fills the rest of the buffer with172// NUL bytes.173//174// The NUL needs its own place in the cutset, because a NUL is a175// control character rather than whitespace, and strings.TrimSpace176// leaves it where it is. A NUL that survives here causes two faults.177// It reaches the Machine status, where a serial ends in a byte no178// reader of the API expects. It also reaches the by-path link names,179// and the kernel refuses a path that holds a NUL, so the link for that180// disk never appears.181func sysfsString(dir string, names ...string) string {182	for _, name := range names {183		if raw, err := os.ReadFile(filepath.Join(dir, name)); err == nil {184			return strings.Trim(string(raw), "\x00 \t\n\r\v\f")185		}186	}187	return ""188}189190// reportBlockDevices prints the world report's storage section: every191// disk attached to this machine.192func reportBlockDevices() {193	disks := discoverBlockDevices()194	if len(disks) == 0 {195		fmt.Println("liken: no disks attached")196		return197	}198	for _, d := range disks {199		details := []string{gib(d.SizeBytes)}200		if d.Model != "" {201			details = append(details, d.Model)202		}203		if d.Serial != "" {204			details = append(details, "serial "+d.Serial)205		}206		if len(d.StableNames) > 0 {207			details = append(details, d.StableNames[0])208		}209		fmt.Printf("liken: disk %s: %s\n", devicePath(d), strings.Join(details, ", "))210	}211}212213// gib renders a byte count in binary gigabytes (GiB = 2^30). This is214// why a "20G" drive from QEMU reads as exactly 20.0. Retail drives215// are labeled in decimal gigabytes instead, so they read smaller216// than the number on the sticker.217func gib(b uint64) string {218	return fmt.Sprintf("%.1f GiB", float64(b)/(1<<30))219}
init/diskuuid.go 81.8%
1package main23// Reading a partition's filesystem UUID, for the by-uuid link tree.4//5// A filesystem UUID names the contents of a partition, not the disk that6// holds it. mke2fs assigns ext4's UUID, and liken assigns FAT32's volume7// ID (FormatFAT32, in the disks package), the moment each filesystem is8// created. Reformat the same disk and it gets a new identity; move a9// filesystem's bytes to a different disk and the identity follows the10// bytes. This is why a role's own device can never appear under11// by-uuid: claiming a role (claim.go) takes a disk that carries no12// partition table at all, and a disk with no partitions has no13// filesystem yet to carry a UUID.14//15// The tree exists for consumers that have no notion of liken's role16// names. A CSI driver, or a line in /etc/fstab, names a volume to mount17// by the UUID its filesystem carries, and expects to find it under18// /dev/disk/by-uuid, the way every other Linux distribution's udev19// publishes it.20//21// filesystemUUID reads a device's raw bytes, the same way hasExt4 and22// disks.HasFAT32 already do, rather than mounting the filesystem or23// shelling out to blkid. It returns "" for anything it does not24// recognize: a missing device, a read shorter than the structure it25// expects, or bytes that carry neither filesystem's magic. This is the26// ordinary answer for a disk a role has claimed but not yet formatted,27// and for a raw or blank partition generally, not a failure to report.2829import (30	"encoding/binary"31	"fmt"32	"os"3334	"github.com/liken-sh/liken/liken/disks"35)3637func filesystemUUID(devPath string) string {38	if hasExt4(devPath) {39		return ext4UUID(devPath)40	}41	if disks.HasFAT32(devPath) {42		return fat32VolumeID(devPath)43	}44	return ""45}4647// ext4SuperblockUUIDOffset is s_uuid's position within the superblock:48// 16 bytes at 0x68, right after the 4-byte fields the resize code reads49// (ext4.go). The layout is ext4's on-disk format, fixed permanently.50const ext4SuperblockUUIDOffset = ext4SuperblockOffset + 0x685152// ext4UUID reads s_uuid and renders it in the canonical 8-4-4-4-12 form53// that blkid and every other UUID-aware tool uses: lowercase hex, split54// by the fields the UUID standard defines, joined with dashes.55func ext4UUID(devPath string) string {56	f, err := os.Open(devPath)57	if err != nil {58		return ""59	}60	defer f.Close()61	uuid := make([]byte, 16)62	if _, err := f.ReadAt(uuid, ext4SuperblockUUIDOffset); err != nil {63		return ""64	}65	return fmt.Sprintf("%x-%x-%x-%x-%x",66		uuid[0:4], uuid[4:6], uuid[6:8], uuid[8:10], uuid[10:16])67}6869// fat32VolumeIDOffset is where the boot sector carries the volume ID:70// byte 67, right after the extended boot signature (fat32.go).71const fat32VolumeIDOffset = 677273// fat32VolumeID reads the 4-byte volume ID FAT keeps in place of a UUID74// and renders it the way Windows and every FAT tool display it: the75// high 16 bits, a dash, the low 16 bits, in uppercase hex.76func fat32VolumeID(devPath string) string {77	f, err := os.Open(devPath)78	if err != nil {79		return ""80	}81	defer f.Close()82	id := make([]byte, 4)83	if _, err := f.ReadAt(id, fat32VolumeIDOffset); err != nil {84		return ""85	}86	v := binary.LittleEndian.Uint32(id)87	return fmt.Sprintf("%04X-%04X", v>>16, v&0xFFFF)88}
init/durable.go 72.5%
1package main23// Durability: how the installer makes a write survive a power cut.4//5// The system slots are FAT32, because firmware can read FAT32 and can6// read almost nothing else. FAT has no journal, so nothing in the7// filesystem promises that a half-finished write is recognizable as8// half-finished. The discipline that makes a write safe therefore9// lives here, in the code that performs it, and it has three parts:10// write to a temporary name, force the bytes out, then rename. A11// power cut before the rename leaves a .partial file that no later12// boot looks at; a power cut after it leaves a complete file.13//14// "Force the bytes out" means fsync, not sync. sync writes dirty pages15// back to the driver and returns. A drive with a volatile write cache16// reports those writes complete while the bytes are still in the17// drive's own RAM. fsync is the call that asks the drive to empty that18// cache, and it is the difference between a file that exists after a19// power cut and one that does not.20//21// The same discipline serves the machine's other FAT filesystems: the22// boot home that carries GRUB's configuration, and the installation23// stick that the hardware report writes its proposal to.2425import (26	"os"2728	"github.com/liken-sh/liken/liken/machine"29)3031// verifyFile checks one file on disk against its release artifact.32func verifyFile(artifact machine.ReleaseArtifact, path string) error {33	f, err := os.Open(path)34	if err != nil {35		return err36	}37	defer f.Close()38	return artifact.Verify(f)39}4041// copyDurably copies through a temporary name, runs fsync, and42// renames the file, so the slot never holds a file that looks final43// but is not. FAT has no journal, so durability here depends entirely44// on this discipline. Without the explicit sync before the rename,45// the page cache may still hold the file's bytes when the rename46// happens, and a power cut then leaves a final-looking file with47// incomplete contents.48func copyDurably(source, dest string) error {49	src, err := os.Open(source)50	if err != nil {51		return err52	}53	defer src.Close()54	tmp := dest + ".partial"55	dst, err := os.Create(tmp)56	if err != nil {57		return err58	}59	if _, err := dst.ReadFrom(src); err != nil {60		dst.Close()61		return err62	}63	if err := dst.Sync(); err != nil {64		dst.Close()65		return err66	}67	if err := dst.Close(); err != nil {68		return err69	}70	return os.Rename(tmp, dest)71}7273// syncDirectory pushes one directory's own contents, the names of the74// files in it, all the way to the medium.75//76// The two functions above make each file's bytes durable, and stop77// there. The rename that gives a file its final name is a directory78// update, and nothing has forced that update out yet. On FAT, an fsync79// of a directory takes the same path as an fsync of a file80// (fat_file_fsync), which ends in a flush of the whole block device, so81// one call covers every rename made on the filesystem it names.82//83// The directory opens read-only, which is what fsync on a directory84// needs; writing to a directory is the kernel's job, not ours.85func syncDirectory(path string) error {86	d, err := os.Open(path)87	if err != nil {88		return err89	}90	defer d.Close()91	return d.Sync()92}9394// writeFileDurably writes bytes through a temporary name, then it95// runs fsync and renames the file. This is the same method that96// copyDurably applies to slot artifacts, for the same reason: FAT has97// no journal, so the code itself must enforce durability.98func writeFileDurably(path string, data []byte) error {99	tmp := path + ".partial"100	f, err := os.Create(tmp)101	if err != nil {102		return err103	}104	if _, err := f.Write(data); err != nil {105		f.Close()106		return err107	}108	if err := f.Sync(); err != nil {109		f.Close()110		return err111	}112	if err := f.Close(); err != nil {113		return err114	}115	return os.Rename(tmp, path)116}
init/efi.go 88.7%
1package main23// Reading and writing the firmware's variables.4//5// A UEFI machine's firmware keeps a small store of variables in6// non-volatile memory. This store is the modern equivalent of "BIOS7// settings", and the boot menu lives there (loadoption.go describes8// the records). The kernel exposes the store as efivarfs, a tiny9// filesystem where every variable is a file, so reading a variable10// is only reading its file. Each file's first four bytes are the11// variable's attribute flags (non-volatile, visible at boot time,12// visible at runtime); the payload follows those flags.13//14// None of this is guaranteed to exist: /sys/firmware/efi appears only15// when UEFI actually booted the kernel. Its absence is a fact worth16// reporting, not an error. A direct-kernel QEMU boot and an old BIOS17// server are both real machines that liken runs on. Everything here18// handles a machine with no variable store by reporting nothing.1920import (21	"encoding/binary"22	"errors"23	"fmt"24	"os"25	"path/filepath"26	"strings"2728	"golang.org/x/sys/unix"2930	"github.com/liken-sh/liken/liken/machine"31)3233// Every variable's filename carries its owner's GUID, because34// variable names are unique only per vendor. The boot manager's35// variables all belong to the specification's own GUID, fixed36// permanently as EFI_GLOBAL_VARIABLE.37const efiGlobalVariable = "8be4df61-93ca-11d2-aa0d-00e098032b8c"3839// efiSysDir is where the kernel exposes UEFI runtime services. Its40// existence is the whole test for which firmware kind is present. It41// is a variable rather than a constant, so tests can set which42// firmware kind they run under, regardless of the machine running43// them.44var efiSysDir = "/sys/firmware/efi"4546// efiVarsDir is a variable rather than a constant so tests can stand47// up a directory of fake variables and exercise everything above the48// mount itself.49var efiVarsDir = efiSysDir + "/efivars"5051// firmwareIsUEFI reports whether UEFI firmware booted this kernel.52// The kernel creates /sys/firmware/efi only when the EFI runtime53// services came up with it.54func firmwareIsUEFI() bool {55	_, err := os.Stat(efiSysDir)56	return err == nil57}5859// mountEFIVars mounts the firmware's variable store, when there is60// one. It does nothing on non-UEFI machines. EBUSY means something61// already mounted the store, which serves the same purpose.62func mountEFIVars() {63	if !firmwareIsUEFI() {64		return65	}66	err := unix.Mount("efivarfs", efiVarsDir, "efivarfs",67		unix.MS_NOSUID|unix.MS_NODEV|unix.MS_NOEXEC, "")68	switch {69	case err == nil:70		fmt.Printf("liken: mounted efivarfs on %s\n", efiVarsDir)71	case errors.Is(err, unix.EBUSY):72	default:73		fmt.Fprintf(os.Stderr, "liken: mounting efivarfs: %v\n", err)74	}75}7677// readEFIVar reads one global variable's payload, with the efivarfs78// attribute word stripped. Callers get the variable's value, in the79// form the specification describes.80func readEFIVar(dir, name string) ([]byte, error) {81	raw, err := os.ReadFile(filepath.Join(dir, name+"-"+efiGlobalVariable))82	if err != nil {83		return nil, err84	}85	if len(raw) < 4 {86		return nil, fmt.Errorf("variable %s is %d bytes; even an empty one carries 4 of attributes", name, len(raw))87	}88	return raw[4:], nil89}9091// fsImmutableFlag is the kernel's per-file immutable bit92// (FS_IMMUTABLE_FL), part of the fixed ioctl ABI that chattr uses.93// x/sys does not export it, so this file spells it out here, the94// same way it spells out the ext4 superblock offsets.95const fsImmutableFlag = 0x000000109697// efiVarAttrs is the attribute word every liken-written variable98// carries: stored in NVRAM (survives power loss), visible to the99// firmware's boot services, and visible to the running OS. Boot100// entries need all three attributes. An entry that the boot manager101// cannot see can never be chosen.102const efiVarAttrs = 0x00000007 // NON_VOLATILE | BOOTSERVICE_ACCESS | RUNTIME_ACCESS103104// writeEFIVar writes one global variable: the attribute word and the105// payload in a single write. This is the only way efivarfs lets a106// variable change, because a partial variable is worse than none, so107// the filesystem refuses piecemeal writes.108//109// The complication is the immutable flag. The kernel marks every110// variable file immutable, so that a stray `rm -rf /` cannot111// permanently disable the motherboard, a real failure mode on early112// UEFI machines. That flag is why this helper exists instead of a113// plain WriteFile. Clearing the flag takes two ioctls on the114// existing file. A variable that does not exist yet has no flag to115// clear.116func writeEFIVar(dir, name string, payload []byte) error {117	path := filepath.Join(dir, name+"-"+efiGlobalVariable)118	if f, err := os.Open(path); err == nil {119		flags, err := unix.IoctlGetInt(int(f.Fd()), unix.FS_IOC_GETFLAGS)120		if err == nil && flags&fsImmutableFlag != 0 {121			err = unix.IoctlSetPointerInt(int(f.Fd()), unix.FS_IOC_SETFLAGS, flags&^fsImmutableFlag)122			if err != nil {123				f.Close()124				return fmt.Errorf("clearing the immutable flag on %s: %w", name, err)125			}126		}127		f.Close()128	}129	b := make([]byte, 4, 4+len(payload))130	binary.LittleEndian.PutUint32(b, efiVarAttrs)131	b = append(b, payload...)132	if err := os.WriteFile(path, b, 0o644); err != nil {133		return fmt.Errorf("writing %s: %w", name, err)134	}135	return nil136}137138// listBootEntries reads every Boot#### variable, decodes it, and139// keys the result by entry number. It skips entries that will not140// decode; those belong to the firmware, and liken finds its own141// entries by description, never by assuming a number.142func listBootEntries(dir string) map[uint16]loadOption {143	entries := map[uint16]loadOption{}144	files, err := os.ReadDir(dir)145	if err != nil {146		return entries147	}148	for _, f := range files {149		var n uint16150		if _, err := fmt.Sscanf(f.Name(), "Boot%04X-"+efiGlobalVariable, &n); err != nil {151			continue152		}153		payload, err := readEFIVar(dir, bootEntryID(n))154		if err != nil {155			continue156		}157		option, err := parseLoadOption(payload)158		if err != nil {159			continue160		}161		entries[n] = option162	}163	return entries164}165166// setBootEntry writes a boot entry under the number that already167// carries its description, or under the lowest free number. This is168// recognition by name, the same as with partitions: the number is a169// handle the firmware owns, and the description is the identity170// liken owns.171func setBootEntry(dir string, option loadOption) (uint16, error) {172	entries := listBootEntries(dir)173	number := uint16(0)174	for {175		existing, taken := entries[number]176		if taken && existing.description == option.description {177			break // ours already; overwrite in place178		}179		if !taken {180			if _, err := readEFIVar(dir, bootEntryID(number)); err != nil {181				break // genuinely free, not just undecodable182			}183		}184		number++185	}186	return number, writeEFIVar(dir, bootEntryID(number), encodeLoadOption(option))187}188189// readBootOrder decodes the firmware's standing preference list.190// BootOrder is a packed array of 16-bit little-endian entry numbers,191// with the first preference first. readBootOrder returns nil when192// the variable is missing or unreadable; nil reads the same as an193// empty order everywhere that matters.194func readBootOrder(dir string) []uint16 {195	b, err := readEFIVar(dir, "BootOrder")196	if err != nil {197		return nil198	}199	var order []uint16200	for i := 0; i+2 <= len(b); i += 2 {201		order = append(order, binary.LittleEndian.Uint16(b[i:i+2]))202	}203	return order204}205206// writeBootOrder packs a preference list back into the firmware's207// format and writes it.208func writeBootOrder(dir string, order []uint16) error {209	payload := make([]byte, len(order)*2)210	for i, n := range order {211		binary.LittleEndian.PutUint16(payload[i*2:], n)212	}213	return writeEFIVar(dir, "BootOrder", payload)214}215216// firmwareFacts reads the machine's boot facts from wherever this217// machine keeps them: which mode it booted in, which entry the boot218// used, and the standing preference. The code decodes each fact to a219// name, so a fleet listing reads "liken slot A", not a hex dump. The220// console parity principle holds here as everywhere: reportFirmware221// prints these same facts at boot.222//223// The variable store's presence determines the mode: efivarfs exists224// only when UEFI booted this kernel. Everything else counts as225// "BIOS", and a BIOS machine's boot facts live on disk instead, in226// GRUB's environment block (biosFirmwareFacts).227func firmwareFacts(dir string) machine.FirmwareStatus {228	if _, err := os.Stat(dir); err != nil {229		return biosFirmwareFacts()230	}231	fw := machine.FirmwareStatus{Mode: machine.FirmwareUEFI}232233	// BootCurrent and BootNext are single entry numbers; BootOrder is234	// a list of them. All are 16-bit little-endian, and all are235	// optional: a firmware that direct-booted a kernel may have set236	// none of them.237	if b, err := readEFIVar(dir, "BootCurrent"); err == nil && len(b) >= 2 {238		fw.BootCurrent = describeBootEntry(dir, binary.LittleEndian.Uint16(b), "")239	}240	order := readBootOrder(dir)241	// A healthy firmware consumes BootNext at power-on, and this runs242	// after power-on, so any value found here deserves a word in the243	// report. Two kinds of value exist. An entry that repeats the244	// head of the order is the pin that assertBootNext leaves245	// (bootentries.go), which means the firmware did not consume the246	// variable. An entry that names anything else is a trial that has247	// not run. The note tells a reader which of the two they see.248	if b, err := readEFIVar(dir, "BootNext"); err == nil && len(b) >= 2 {249		next, note := binary.LittleEndian.Uint16(b), ""250		if len(order) > 0 && order[0] == next {251			note = "pinned at the proven slot"252		}253		fw.BootNext = describeBootEntry(dir, next, note)254	}255	for _, n := range order {256		fw.BootOrder = append(fw.BootOrder, describeBootEntry(dir, n, ""))257	}258	return fw259}260261// describeBootEntry renders one entry in a readable form: the262// firmware's own name for the variable, plus the entry's description263// when the code can decode it. An entry that is missing or corrupted264// still appears by its ID, so nothing in the order stays hidden.265//266// The note is what the caller knows about this entry that the entry267// itself does not say. It joins the description inside the same268// parentheses, and it appears even for an entry that will not269// decode, because what the caller knows is true either way.270func describeBootEntry(dir string, n uint16, note string) string {271	id, inside := bootEntryID(n), []string{}272	if payload, err := readEFIVar(dir, id); err == nil {273		if option, err := parseLoadOption(payload); err == nil && option.description != "" {274			inside = append(inside, option.description)275		}276	}277	if note != "" {278		inside = append(inside, note)279	}280	if len(inside) == 0 {281		return id282	}283	return fmt.Sprintf("%s (%s)", id, strings.Join(inside, ", "))284}285286// reportFirmware prints the firmware's facts on the console: the287// same facts firmwareFacts publishes, formatted for the world288// report. Both modes print the same shape of report, so reading a289// BIOS machine's console teaches the same things that reading a UEFI290// machine's console does. Only the mode line and the mechanisms291// behind the facts differ.292func reportFirmware() {293	fw := firmwareFacts(efiVarsDir)294	if fw.Mode == machine.FirmwareUEFI {295		fmt.Println("liken: firmware: UEFI")296		if fw.BootCurrent == "" {297			fmt.Println("liken: firmware: BootCurrent not set (a direct-kernel boot never picks an entry)")298		}299	} else {300		fmt.Println("liken: firmware: BIOS (no /sys/firmware/efi; boot preferences live with GRUB, when installed)")301	}302	if fw.BootCurrent != "" {303		fmt.Printf("liken: firmware: booted via %s\n", fw.BootCurrent)304	}305	if fw.BootNext != "" {306		fmt.Printf("liken: firmware: BootNext holds %s\n", fw.BootNext)307	}308	for _, entry := range fw.BootOrder {309		fmt.Printf("liken: firmware: boot order: %s\n", entry)310	}311}
init/efiactuator.go 87.5%
1package main23// The UEFI implementation of the boot actuator (actuator.go describes4// the interface and why it exists).5//6// UEFI firmware keeps its boot preferences as variables (efi.go), and7// the specification already provides exactly the two mechanisms8// blue-green upgrades need. BootNext is a one-shot: the firmware9// consumes it as it boots, so a slot on trial gets exactly one10// chance, and any reset after that falls back to BootOrder.11// BootOrder is the standing preference list, and keeping the proven12// slot at its head is what makes the fallback real.1314import (15	"fmt"16	"os"17)1819type efiActuator struct {20	// dir is the efivarfs mount, or a test's stand-in for one.21	dir string22	// machineName is rendered into every boot entry this dialect23	// writes. An entry carries identity, so an anonymous entry would be24	// wrong on every later boot. This value comes from the kernel25	// command line, where an earlier boot's entry put it.26	machineName string27}2829func (a efiActuator) canArmTrial(slot string) error {30	if _, ok := findSlotEntry(a.dir, slot); !ok {31		return fmt.Errorf("no boot entry answers to %q", "liken slot "+slot)32	}33	return nil34}3536func (a efiActuator) armTrial(slot string) (string, error) {37	entry, ok := findSlotEntry(a.dir, slot)38	if !ok {39		return "", fmt.Errorf("no boot entry answers to %q", "liken slot "+slot)40	}41	if err := writeEFIVar(a.dir, "BootNext", bootNextPayload(entry)); err != nil {42		return "", fmt.Errorf("arming BootNext: %w", err)43	}44	return "BootNext armed at " + bootEntryID(entry), nil45}4647func (a efiActuator) fallbackLeads(slot string) bool {48	leader, ok := findSlotEntry(a.dir, slot)49	if !ok {50		return false51	}52	order := readBootOrder(a.dir)53	return len(order) > 0 && order[0] == leader54}5556// assertProven brings this machine's whole boot path into agreement57// with the store: NVRAM holds an entry for each slot with the proven58// slot at the head of BootOrder (bootentries.go), and the proven slot59// alone answers the firmware's default boot path (slotloader.go).60//61// Both halves are needed, and they answer different failures. The62// entries are what an ordinary boot reads. The default path is what a63// firmware falls back on when it has no entries left to read.64//65// The two assertions run separately on purpose. A slot that cannot66// carry a loader must not stop the entries from healing, and a variable67// store that refuses a write must not stop the loader from landing.68//69// The first half asserts more than the standing order: it also aims70// the one-shot BootNext at the proven slot, for the firmware that71// holds BootNext through a reset but not BootOrder (assertBootNext in72// bootentries.go).73func (a efiActuator) assertProven(slot string) {74	healBootEntries(a.dir, a.machineName, slot)75	a.healSlotLoader(slot)76}7778// healSlotLoader keeps the fallback loader on the proven slot and79// nowhere else. A firmware at its defaults takes the first answer it80// finds, and the other slot holds an older release or nothing at all.81//82// The order of these two steps is the whole guarantee. The proven slot83// takes the loader first, and only a slot that took it lets the other84// slot lose its own. A machine is never left with neither.85func (a efiActuator) healSlotLoader(slot string) {86	mount := slotMountPath(slot)87	if mount == "" || a.machineName == "" {88		// A slot that is not mounted cannot take a loader, and a boot89		// with no name has nothing correct to write in the loader's90		// entry. Both cases leave the loader that is already there.91		return92	}93	if err := writeSlotLoader(mount, slot, a.machineName); err != nil {94		fmt.Fprintf(os.Stderr, "liken: system: %v; this machine has no fallback for a firmware that loses its boot entries\n", err)95		return96	}97	if other := slotMountPath(otherSlot(slot)); other != "" {98		removeSlotLoader(other)99	}100}
init/ext4.go 77.6%
1package main23// Growing ext4 filesystems, without resize2fs.4//5// A filesystem's size lives in its superblock, not in its partition.6// Growing the partition changes nothing until the code tells the7// filesystem that there is more room. The usual tool for that is8// resize2fs, but for online growth, meaning the filesystem stays9// mounted, which is the only state liken needs, resize2fs is only a10// thin wrapper. The kernel has done the actual work since ext3, and11// asking it takes one ioctl carrying the new block count. liken12// issues the ioctl directly. This spares the image a second13// e2fsprogs binary and shows what "resizing a filesystem" actually14// is.15//16// The superblock sits 1024 bytes into the device, whatever the block17// size is. The first KiB is left alone for boot sectors, a18// convention older than ext itself. Everything growth needs is in19// the superblock's first few hundred bytes.2021import (22	"fmt"23	"os"24	"unsafe"2526	"golang.org/x/sys/unix"2728	"github.com/liken-sh/liken/liken/machine"29)3031// hasExt4 checks a device for ext4's superblock magic: two bytes,32// 0xEF53 little-endian, at offset 1080. The superblock starts at33// 1024, and the magic sits 56 bytes into it. This is the same check34// that blkid makes; identifying a filesystem takes nothing more than35// this.36func hasExt4(devPath string) bool {37	f, err := os.Open(devPath)38	if err != nil {39		return false40	}41	defer f.Close()42	magic := make([]byte, 2)43	if _, err := f.ReadAt(magic, 1080); err != nil {44		return false45	}46	return magic[0] == 0x53 && magic[1] == 0xEF47}4849// ext4Geometry is the two superblock facts resizing needs: how big a50// block is, and how many blocks the filesystem's superblock records.51type ext4Geometry struct {52	blockSize  uint6453	blockCount uint6454}5556const ext4SuperblockOffset = 10245758// parseExt4Superblock reads geometry from a superblock's bytes. The59// offsets are the on-disk format, fixed permanently:60//61//	s_blocks_count_lo  u32 at 4    block count, low 32 bits62//	s_log_block_size   u32 at 24   block size = 1024 << this63//	s_magic            u16 at 56   0xEF5364//	s_feature_incompat u32 at 96   bit 0x80 = the 64bit feature65//	s_blocks_count_hi  u32 at 336  block count's high bits (64bit only)66func parseExt4Superblock(sb []byte) (ext4Geometry, error) {67	if len(sb) < 1024 {68		return ext4Geometry{}, fmt.Errorf("superblock truncated at %d bytes", len(sb))69	}70	if sb[56] != 0x53 || sb[57] != 0xEF {71		return ext4Geometry{}, fmt.Errorf("no ext4 magic in the superblock")72	}7374	logBlockSize := le32(sb[24:])75	// 1024 << 6 = 64KiB, ext4's ceiling. Anything above this value is76	// corruption.77	if logBlockSize > 6 {78		return ext4Geometry{}, fmt.Errorf("implausible block size exponent %d", logBlockSize)79	}80	g := ext4Geometry{81		blockSize:  1024 << logBlockSize,82		blockCount: uint64(le32(sb[4:])),83	}84	// With the 64bit feature, the block count has high bits in a85	// second field. Without the feature, those bytes belong to other86	// fields, and the code must not read them as a count.87	if le32(sb[96:])&0x80 != 0 {88		g.blockCount |= uint64(le32(sb[336:])) << 3289	}90	return g, nil91}9293func le32(b []byte) uint32 {94	return uint32(b[0]) | uint32(b[1])<<8 | uint32(b[2])<<16 | uint32(b[3])<<2495}9697// readExt4Geometry reads the superblock from a device node.98func readExt4Geometry(devPath string) (ext4Geometry, error) {99	f, err := os.Open(devPath)100	if err != nil {101		return ext4Geometry{}, err102	}103	defer f.Close()104	sb := make([]byte, 1024)105	if _, err := f.ReadAt(sb, ext4SuperblockOffset); err != nil {106		return ext4Geometry{}, fmt.Errorf("reading the superblock: %w", err)107	}108	return parseExt4Superblock(sb)109}110111// ext4ResizeFS is EXT4_IOC_RESIZE_FS: _IOW('f', 16, __u64), assembled112// the same way the kernel's ioctl macros assemble it: direction113// "write" (1<<30), argument size (8<<16), type ('f'<<8), and command114// number (16).115const ext4ResizeFS = (1 << 30) | (8 << 16) | ('f' << 8) | 16 // 0x40086610116117// growExt4 asks the kernel to grow the mounted filesystem to118// newBlocks. Any fd inside the mount identifies the filesystem; the119// mountpoint itself is the simplest one to use.120func growExt4(mountpoint string, newBlocks uint64) error {121	f, err := os.OpenFile(mountpoint, os.O_RDONLY|unix.O_DIRECTORY, 0)122	if err != nil {123		return err124	}125	defer f.Close()126	if _, _, errno := unix.Syscall(unix.SYS_IOCTL, f.Fd(), ext4ResizeFS, uintptr(unsafe.Pointer(&newBlocks))); errno != 0 {127		return fmt.Errorf("EXT4_IOC_RESIZE_FS to %d blocks: %w", newBlocks, errno)128	}129	return nil130}131132// maybeGrowFilesystem brings a mounted role's filesystem up to its133// partition's size, if the partition outgrew the filesystem.134// mke2fs's defaults reserve resize headroom (the resize_inode) for135// growth of roughly a thousandfold, far beyond anything a disk will136// actually do, so a grow that fits the partition is expected to137// succeed. Failure means the code cannot satisfy the declared138// capacity, and that fails reconciliation, the same as any other139// unsatisfiable role.140func maybeGrowFilesystem(role machine.DeclaredRole, p partition, mountpoint string) error {141	g, err := readExt4Geometry(devRoot + "/" + p.name)142	if err != nil {143		return fmt.Errorf("reading %s's filesystem geometry: %w", role.Name, err)144	}145	newBlocks := p.sizeBytes / g.blockSize146	if newBlocks <= g.blockCount {147		return nil148	}149	fmt.Printf("liken: storage: growing %s's filesystem from %s to %s\n",150		role.Name, gib(g.blockCount*g.blockSize), gib(newBlocks*g.blockSize))151	if err := growExt4(mountpoint, newBlocks); err != nil {152		return fmt.Errorf("growing %s's filesystem: %w", role.Name, err)153	}154	return nil155}
init/facts.go 97.8%
1package main23// Facts: init's half of the Machine status.4//5// Init is the only program that observes the boot: the DHCP exchange,6// and the hardware as the kernel first presents it. So it is the only7// program that can report those facts. It writes them under8// /run/liken/facts as a tree of small files (machine/factstree.go),9// shaped exactly like the Machine's status block. The liken operator,10// which runs in the cluster and cannot see any of this directly,11// reads the tree and publishes it to the API. Init never talks to12// Kubernetes; this tree is the entire interface between the two.13//14// No single owner holds the facts. Each fact has its own file, so each15// init component writes its own subtree with no shared lock. A boot16// step writes its subtree once, at the point where it discovered the17// fact. The long-lived components own the subtrees that change after18// boot: the clock owns time/, the hardware watch owns19// hardware/blockDevices/ and hardware/unclaimed/, the module loader20// owns modules/ and boot/manifest, and the restart path owns21// features/, registries/, and the boot/ manifest records it rewrites.22// factstree.go documents the whole ownership map.23//24// The boot's write-once facts land together, in publishBootFacts,25// rather than each at its own discovery line. Each boot step holds26// its discovered facts locally and hands them here, so the tree27// appears once, complete, instead of growing in pieces while the28// operator's watch reads it.2930import (31	"fmt"32	"os"33	"runtime"34	"strings"35	"time"3637	"golang.org/x/sys/unix"3839	"github.com/liken-sh/liken/liken/api"40	"github.com/liken-sh/liken/liken/cluster"41	"github.com/liken-sh/liken/liken/machine"42)4344// Where the facts tree and the boot manifest land. These are package45// variables rather than constants, so tests can publish into a46// tempdir. A real boot never points them anywhere but /run.47var (48	factsTree        = machine.FactsTree{Dir: machine.FactsDir, Report: reportFactsError}49	bootManifestPath = machine.BootManifestPath5051	// xtablesProbe is the command that reports the netfilter52	// userspace's version. It is a variable so tests can aim it at53	// nothing.54	xtablesProbe = "iptables"55)5657// reportFactsError reports a failed facts write to stderr. The facts58// tree calls it for each failed write, so a boot step calls the writers59// bare. A write failure must never stop the boot: the machine keeps60// running and the operator falls back to a partial tree, which reads a61// missing fact as its zero value. Losing a fact is a reporting gap, not62// a reason to halt PID 1.63func reportFactsError(err error) {64	fmt.Fprintf(os.Stderr, "liken: writing facts: %v\n", err)65}6667// bootFacts gathers the write-once facts that the boot discovered. It68// is a struct rather than a parameter list because nearly a dozen69// positional arguments invite mixed-up order, and named fields read70// correctly at the call site. It carries values the boot already71// holds; nothing routes through a shared owner.72type bootFacts struct {73	clusterDoc   *cluster.Cluster74	role         api.Role75	conns        []*connection76	storage      machine.StorageStatus77	boot         machine.BootStatus78	modules      []machine.ModuleStatus79	features     []machine.FeatureStatus80	registries   machine.RegistriesStatus81	time         machine.TimeStatus82	blockDevices []machine.BlockDevice83	unclaimed    []machine.UnclaimedDevice84	lastCrash    *machine.CrashStatus85	lastFailStop *machine.FailStop86}8788// publishBootFacts writes every fact the boot discovered once and never89// revisits. It calls one per-subtree writer for each, so each fact90// lands in its own file through the same atomic rename as every later91// write. The tree carries the same facts the boot printed to the92// console, the console-parity principle: anything reported only to the93// serial port is invisible to anyone operating the machine remotely, so94// the status must repeat what the console reports. The hardware,95// version, and firmware blocks are re-derived here rather than96// remembered from earlier boot steps.97func publishBootFacts(tree machine.FactsTree, in bootFacts) {98	now := time.Now()99	memoryBytes, bootedAt := machineUptimeFacts(now)100101	tree.WriteRole(in.role)102	// The crash stub arrives settled rather than being re-read here,103	// because settling it has side effects (preserving and clearing the104	// platform store) that belong to one moment early in boot.105	tree.WriteLastCrash(in.lastCrash)106	// The fail-stop record arrives read rather than being re-read here,107	// so that the console line and this fact come from one reading of108	// one file.109	tree.WriteLastFailStop(in.lastFailStop)110	tree.WriteVersion(versionFacts())111	// Network facts exist only for interfaces that came up; a machine112	// that failed DHCP still publishes the facts it has.113	tree.WriteNetwork(networkFacts(in.clusterDoc, in.conns))114	// The clock's state so far: the boot-time measurement, if one115	// succeeded, or an accurate unsynchronized or free-running report.116	// The clock loop owns time/ after this seed.117	tree.WriteTime(in.time)118	tree.WriteHardwareBasics(runtime.NumCPU(), memoryBytes)119	tree.WriteBlockDevices(in.blockDevices)120	tree.WriteUnclaimed(in.unclaimed)121	tree.WriteFirmware(firmwareFacts(efiVarsDir))122	tree.WriteStorage(in.storage)123	tree.WriteModules(in.modules)124	tree.WriteFeatures(in.features)125	tree.WriteRegistries(in.registries)126	tree.WriteRuntime(runtimeFacts(in.clusterDoc, memoryBytes))127	// The limits are read back from init itself rather than passed in,128	// like the hardware and firmware blocks above. Init is PID 1, so129	// its limits are the ones every process on the machine inherits,130	// and one Getrlimit here needs no earlier boot step to remember131	// anything. The names to read are the union of the table and the132	// spec, and the boot record carries the spec.133	tree.WriteRlimits(readRlimits(machine.OSRlimits, in.boot.Rlimits))134135	// The boot record: what this boot ran under. The four manifest136	// records seed here at boot; the module loader and the restart path137	// own the ones they rewrite afterward.138	tree.WriteBootTime(bootedAt)139	tree.WriteBootSlot(in.boot.Slot)140	// The command line is read here rather than passed in, like the141	// hardware and firmware blocks above, because the kernel holds it142	// and no earlier boot step has to remember it.143	tree.WriteBootCommandLine(cmdlineRaw())144	tree.WriteBootStorage(in.boot.Storage)145	tree.WriteBootNetwork(in.boot.Network)146	tree.WriteBootModules(in.boot.Modules)147	tree.WriteBootModuleParameters(in.boot.ModuleParameters)148	tree.WriteBootSerio(in.boot.Serio)149	tree.WriteBootRlimits(in.boot.Rlimits)150	tree.WriteBootManifest(in.boot.ManifestSource, in.boot.ManifestHash)151	tree.WriteBootClusterManifest(in.boot.ClusterManifestSource, in.boot.ClusterManifestHash)152	tree.WriteBootCredentials(in.boot.CredentialsSource, in.boot.CredentialsHash)153	tree.WriteBootImports(in.boot.ImportsSource, in.boot.ImportsHash, in.boot.ImportsDiscarded)154	tree.WriteRejection(machine.RejectMachine, in.boot.Rejection)155	tree.WriteRejection(machine.RejectCluster, in.boot.ClusterRejection)156	tree.WriteRejection(machine.RejectSystem, in.boot.SystemRejection)157	tree.WriteRejection(machine.RejectCredentials, in.boot.CredentialsRejection)158}159160// runtimeFacts resolves the runtime discipline init imposed on k3s, on161// the kubelet inside it, and on containerd beside it, so the facts162// carry the same values init printed to the console.163//164// The Go environment reports the resolved ceiling, not the spec string:165// an absolute quantity in MiB, or no ceiling at all when the cluster166// left the limit unset or turned it off. It reads the same section and167// memory that k3sRuntimeEnv reads, so the facts and the process168// environment can never disagree.169//170// The other values need no resolving, because each is already in the171// grammar of the file it lands in. The image collection policy is the172// lines that kubeletImageGCSettings wrote into the kubelet's file, the173// debug flag is the drop-in key, and the containerd level is the one174// the drop-in gives containerd. An unset field reports nothing, and an175// unset section yields a zero status, so WriteRuntime writes no files176// and status.runtime is absent.177func runtimeFacts(clusterDoc *cluster.Cluster, memoryBytes uint64) machine.RuntimeStatus {178	spec := clusterDoc.RuntimeSpec()179	st := machine.RuntimeStatus{}180	if gc, ok := spec.GoGCPercent(); ok {181		st.K3s.GoGC = gc182	}183	limit, off, err := spec.GoMemoryLimitBytes(memoryBytes)184	if err == nil && !off && limit > 0 {185		st.K3s.GoMemoryLimit = fmt.Sprintf("%dMi", limit/(1<<20))186	}187	st.K3s.Debug = spec.Debug188	st.Containerd.LogLevel = clusterDoc.ContainerdSpec().LogLevel189	imageGC := clusterDoc.KubeletSpec().ImageGC190	if imageGC.HighThresholdPercent != nil {191		st.Kubelet.ImageGC.HighThresholdPercent = *imageGC.HighThresholdPercent192	}193	if imageGC.LowThresholdPercent != nil {194		st.Kubelet.ImageGC.LowThresholdPercent = *imageGC.LowThresholdPercent195	}196	st.Kubelet.ImageGC.MaximumAge = imageGC.MaximumAge197	st.Kubelet.ImageGC.MinimumAge = imageGC.MinimumAge198	return st199}200201// machineMemoryBytes reports the machine's total memory in bytes from202// one syscall. The restart path uses it to resolve the runtime facts203// again, because a restart re-derives the k3s environment the same way204// a boot does.205func machineMemoryBytes() uint64 {206	var si unix.Sysinfo_t207	if err := unix.Sysinfo(&si); err != nil {208		return 0209	}210	return uint64(si.Totalram) * uint64(si.Unit)211}212213// machineUptimeFacts answers two questions from one syscall: how much214// memory the machine has, and the moment it booted. Sysinfo reports215// total memory and uptime; subtracting uptime from the clock gives the216// boot instant. The wall clock at this point comes from the217// hypervisor's RTC, because no NTP synchronization has happened yet.218func machineUptimeFacts(now time.Time) (memoryBytes uint64, bootedAt *time.Time) {219	var si unix.Sysinfo_t220	if err := unix.Sysinfo(&si); err != nil {221		return 0, nil222	}223	booted := now.Add(-time.Duration(si.Uptime) * time.Second)224	return uint64(si.Totalram) * uint64(si.Unit), &booted225}226227// versionFacts assembles the machine's version inventory. Two kinds of228// value live here. The kernel release and the netfilter userspace229// version are observed from the running machine, so the code asks the230// machine itself rather than copying a build pin. The rest, the boot231// artifacts and bundled payloads, have no version command of their own,232// so applyComponentFacts reports them from the record the image build233// staged beside the bytes (versions.go).234func versionFacts() machine.VersionStatus {235	v := machine.VersionStatus{Liken: machine.Version}236237	var u unix.Utsname238	if err := unix.Uname(&u); err == nil {239		v.Kernel = unix.ByteSliceToString(u.Release[:])240	}241242	// The netfilter userspace reports itself as "iptables vX.Y.Z243	// (legacy)"; the version and variant are the interesting part.244	if out, ok := run(xtablesProbe, "-V"); ok {245		v.Xtables = strings.TrimPrefix(out, "iptables ")246	}247248	// The CPU's running microcode revision is observed the same way.249	// The microcode pin says which early cpio the release carries; this250	// says which revision the CPU actually runs. The two agreeing is the251	// proof that the early cpio applied, and only real hardware can give252	// it: on a virtual machine the hypervisor owns the microcode, and253	// this reports the hypervisor's value.254	v.MicrocodeRevision = microcodeRevision(cpuinfoPath)255256	applyComponentFacts(&v)257	return v258}259260// publishBootManifest writes the Machine manifest this boot ran under,261// byte for byte. This is how the operator identifies which Machine it262// manages, and, on a first boot, the spec to seed the in-cluster263// Machine from. It stays one whole file, beside the facts tree, because264// it shares the tree's lifetime and the operator needs its exact bytes.265func publishBootManifest(choice *manifestChoice) {266	if len(choice.raw) == 0 {267		return268	}269	if err := os.WriteFile(bootManifestPath, choice.raw, 0o644); err != nil {270		fmt.Fprintf(os.Stderr, "liken: writing the boot manifest: %v\n", err)271	}272}273274// publishBootClusterManifest is the cluster document's version of the275// boot manifest publication: the exact bytes this boot derived its276// role from. The operator's drift detection needs these bytes, since277// it compares documents by meaning and needs bytes to parse, not278// just a hash.279func publishBootClusterManifest(raw []byte) {280	if len(raw) == 0 {281		return282	}283	if err := os.WriteFile(cluster.BootClusterManifestPath, raw, 0o644); err != nil {284		fmt.Fprintf(os.Stderr, "liken: writing the boot cluster manifest: %v\n", err)285	}286}287288// networkFacts folds every connection into a NetworkStatus. Every289// value comes from the connections themselves, and nothing reads a290// clock or the kernel, so the background radio pass can call this291// again after boot and every unchanged interface renders exactly the292// facts it rendered the first time.293func networkFacts(clusterDoc *cluster.Cluster, conns []*connection) machine.NetworkStatus {294	status := machine.NetworkStatus{}295	if len(conns) == 0 {296		return status297	}298299	for _, conn := range conns {300		status.Interfaces = append(status.Interfaces, interfaceFacts(conn))301	}302303	// The summary skips an addressless interface. The top-level304	// fields describe the interface that carries the machine's305	// traffic, and a radio that did not join carries none, so it306	// appears in the collection below and never as the summary.307	primary := conns[0]308	for _, conn := range conns {309		if conn.addr != nil {310			primary = conn311			break312		}313	}314	if _, ifname := nodeAddress(clusterDoc, conns); ifname != "" {315		for _, conn := range conns {316			if conn.ifname == ifname {317				primary = conn318			}319		}320	}321	summary := interfaceFacts(primary)322	status.Interface = summary.Name323	status.MAC = summary.MAC324	status.Addresses = []string{summary.Address}325	status.Gateway = summary.Gateway326	status.Nameservers = summary.Nameservers327	status.LeaseExpires = summary.LeaseExpires328	return status329}330331// interfaceFacts turns one connection into status: the same facts332// the console report prints, in a form other code can query.333func interfaceFacts(conn *connection) machine.InterfaceStatus {334	status := machine.InterfaceStatus{335		Name:        conn.ifname,336		MAC:         conn.mac.String(),337		Method:      conn.method,338		Nameservers: make([]string, 0, len(conn.nameservers)),339	}340	if conn.addr != nil {341		status.Address = conn.addr.String()342	}343	if conn.radio != nil {344		status.Wireless = conn.radio.wirelessStatus()345	}346	// applyLease fixed this instant when the ACK landed, so a later347	// rewrite reports the same expiry instead of inventing a new348	// one.349	if conn.method == machine.MethodDHCP && !conn.leaseExpires.IsZero() {350		expires := conn.leaseExpires351		status.LeaseExpires = &expires352	}353	if conn.gateway != nil {354		status.Gateway = conn.gateway.String()355	}356	for _, ns := range conn.nameservers {357		status.Nameservers = append(status.Nameservers, ns.String())358	}359	return status360}
init/failstop.go 100.0%
1package main23// The boot's half of the fail-stop record.4//5// failBoot prints the reason for a refusal and powers the machine off.6// On an install boot that is enough: a person picked the entry and is7// reading the console. On a boot from disk nobody is reading anything.8// The message goes to the kernel ring buffer, k3s never starts, so no9// log relay carries it off the machine, and the power-off erases the10// ring buffer. The machine then repeats the same few seconds at every11// power-on, and the only way to learn which of failBoot's call sites12// fired is to read init's source.13//14// This file closes that gap. recordFailStop writes the reason to15// machineState before the power-off, and reportFailStop reads it back16// on the next boot: to the console for a person at the serial port,17// and into status.lastFailStop for everyone else. machine/failstop.go18// owns the record itself.19//20// What a refusal can record follows from where the record lives, and21// the two fail-stops differ. An identity refusal always records,22// because storage has settled by the time the boot reads a cluster23// document. A storage refusal records only when it happens after24// machineState mounts.25//26// That excludes the most likely storage failure of all. settleStorage27// waits for every declared disk to attach before it mounts anything,28// so a disk that never appears stops the boot while the record still29// has nowhere to go. A machine whose state disk is fine and whose pod30// disk is dead therefore powers off with nothing written down, even31// though the partition that would hold the record is right there.32// Closing that gap means mounting machineState on the way out of a33// failed wait, which is a change to the settling order and not to this34// file.3536import (37	"fmt"38	"os"39	"time"40	"unicode/utf8"4142	"github.com/liken-sh/liken/liken/machine"43)4445// failStopReasonCap bounds the recorded reason, matching the maxLength46// that the CRD sets on status.lastFailStop.reason. The cut happens47// once, here, before the write, so the record on disk and the field in48// the cluster carry exactly the same words.49const failStopReasonCap = 10245051// recordFailStop writes this boot's refusal to machineState, when52// there is a mounted machineState to write to. A failure to record is53// reported and no more: the machine is already stopping, and losing54// the record must not stand in the way of the power-off.55//56// An install boot records nothing, for two reasons. A person picked57// that entry and is reading the console, which is the audience the58// record exists to replace. And the installer must leave nothing on a59// machine it did not finish installing: a record written from the60// stick would be read back by the first successful boot and reported61// as that machine's own history.62func recordFailStop(stateDir, reason string, now time.Time) {63	if installing() {64		return65	}66	if !machineStateWritable {67		fmt.Fprintln(os.Stderr, "liken: fail-stop: machineState is not mounted, so this refusal leaves no record")68		return69	}70	record := machine.FailStop{Reason: capReason(reason), Time: now}71	if err := machine.WriteFailStop(stateDir, record); err != nil {72		fmt.Fprintf(os.Stderr, "liken: fail-stop: recording the refusal: %v\n", err)73		return74	}75	fmt.Fprintln(os.Stderr, "liken: fail-stop: recorded the reason; the next boot reports it as status.lastFailStop")76}7778// reportFailStop reads the standing record and puts it on the console.79// It runs on every boot, refused or not, and it never removes the80// record: the field answers when this machine last refused to boot,81// which stays true until the next refusal overwrites it. Deriving the82// answer from the file on every boot is what makes the field83// reconstructible after somebody erases the Machine's status.84func reportFailStop(stateDir string) *machine.FailStop {85	record, err := machine.ReadFailStop(stateDir)86	if err != nil {87		fmt.Fprintf(os.Stderr, "liken: fail-stop: reading the record: %v\n", err)88		return nil89	}90	if record == nil {91		return nil92	}93	// Console parity: the fact prints where an operator at the serial94	// port can see it, in the same words status carries.95	fmt.Printf("liken: fail-stop: this machine last refused to boot at %s: %s\n",96		record.Time.UTC().Format(time.RFC3339), record.Reason)97	return record98}99100// capReason cuts a reason to the cap on a rune boundary, so a101// truncated message stays valid UTF-8 and the API server accepts it.102func capReason(reason string) string {103	if len(reason) <= failStopReasonCap {104		return reason105	}106	cut := failStopReasonCap107	for cut > 0 && !utf8.RuneStart(reason[cut]) {108		cut--109	}110	return reason[:cut]111}
init/fatstate.go 93.2%
1package main23// What a machine does about a FAT volume that was never released.4//5// The system slots and the boot home are FAT32, because firmware reads6// FAT32 and reads almost nothing else. FAT keeps no journal, so its7// whole account of its own health is one bit in the boot sector: on8// while the volume is mounted for writing, off once it is released.9// A volume found with the bit on was not released, and its directory10// entries and allocation table may disagree.11//12// Two things follow, and this file does both.13//14// The machine reports it. A boot reads the bit before it mounts the15// volume, because mounting for writing sets the bit and a later read16// would only say that the volume is in use now. The answer reaches17// Machine status as lastStopUnclean, so an operator sees that a18// machine did not stop cleanly without reading a serial console.19//20// "Before it mounts the volume" is earlier than it sounds for one of21// the volumes. The initramfs mounts the slot it boots from to reach22// the system image, so that slot's answer has to be read there and23// carried forward, rather than read where the other roles are read.24//25// The machine clears it, where it can say honestly that the volume is26// good. This is necessary rather than tidy: the driver will not clear27// a bit it found set, so one unclean stop marks a volume for the rest28// of its life and the warning stops meaning anything. liken clears the29// bit only after it has read every artifact the slot claims to hold30// and checked each one against the digest in the slot's own release31// document. That is a stronger check than a filesystem repair, which32// can prove that a chain of clusters is well formed but cannot tell33// whether the bytes in it are the kernel liken meant to boot.34//35// Three volumes cannot be treated the same way:36//37//   - The slot this machine booted is already mounted for writing by38//     the time storage reconciles, and the running root is a loop39//     device over a file on it. Its bit is reported and left alone.40//     It clears on the next boot from the other slot, when this one is41//     no longer in use.42//   - The other slot is not in use. It is checked and cleared here.43//   - The boot home holds GRUB's configuration and environment, which44//     liken rewrites from the manifest on every boot. It has no45//     release document to check against, so liken clears its bit on46//     the strength of rewriting its contents, and says so.4748import (49	"fmt"50	"os"51	"path/filepath"5253	"golang.org/x/sys/unix"5455	"github.com/liken-sh/liken/liken/disks"56	"github.com/liken-sh/liken/liken/machine"57)5859// fatCheckMount is where a slot is mounted read-only while its60// contents are checked. A read-only mount does not set the volume's61// mark, so the check leaves nothing behind to clean up. It is a62// package variable so a test can point the check at a directory of its63// own.64var fatCheckMount = "/run/liken/fat-check"6566// bootedSlotStopEnv carries one fact from the initramfs to the rest of67// the boot: what the slot this machine booted said about its last stop.68// The initramfs mounts that slot for writing to reach the system image,69// long before storage reconciles, and the mount sets the mark. So by70// the time reconciliation asks, the device only says that the volume is71// in use now, which is true of every boot and worth reporting on none.72//73// The fact travels in the environment because init re-executes itself74// from the new root at the end of switch_root (switchroot.go), which75// ends the process that read it. The environment survives that exec.76const bootedSlotStopEnv = "LIKEN_BOOTED_SLOT_STOP"7778const (79	stopMarkClean   = "clean"80	stopMarkUnclean = "unclean"81)8283// recordBootedSlotStop reads a slot's mark and keeps the answer for the84// rest of the boot. Call it before mounting the slot for writing, which85// is the act that destroys the answer.86//87// A read that fails records nothing. The fallback is a read of the88// device later, which reports a mark this boot set, and that is wrong89// in the safe direction: a machine whose boot sector cannot be read is90// about to fail its storage reconciliation for the same reason.91func recordBootedSlotStop(dev string) {92	unclean, err := disks.FAT32Dirty(dev)93	if err != nil {94		fmt.Fprintf(os.Stderr, "liken: storage: reading the booted slot's stop mark: %v\n", err)95		return96	}97	mark := stopMarkClean98	if unclean {99		mark = stopMarkUnclean100	}101	if err := os.Setenv(bootedSlotStopEnv, mark); err != nil {102		fmt.Fprintf(os.Stderr, "liken: storage: keeping the booted slot's stop mark: %v\n", err)103	}104}105106// fatStopMark answers whether a FAT role's volume already carried its107// mark when this boot found it.108//109// Every role but one still holds the answer on the device, because110// nothing has mounted it yet. The exception is the slot this machine111// booted, whose answer the initramfs read and recorded.112func fatStopMark(name machine.StorageRoleName, dev string) (bool, error) {113	if isSystemSlot(name) && string(name) == bootedSlotRole() {114		switch os.Getenv(bootedSlotStopEnv) {115		case stopMarkUnclean:116			return true, nil117		case stopMarkClean:118			return false, nil119		}120		// Nothing was recorded, so the initramfs mounted no slot. A boot121		// that takes the system image from RAM does this, and then the122		// device is still the place to ask.123	}124	return disks.FAT32Dirty(dev)125}126127// readFATStop reports whether a FAT role's volume still carries the128// mark from an earlier stop, and heals the volume when it can. It129// returns what belongs in status: whether the previous stop left this130// volume unreleased. Healing does not change that answer, because the131// answer describes the stop that already happened.132//133// A role that this boot claimed and formatted is new, so it is not134// asked. Anything that goes wrong here is reported and treated as no135// mark: a machine must boot even when it cannot read one byte of a136// boot sector, and storage reconciliation will fail for a real reason137// a moment later if the volume is truly unreadable.138func readFATStop(name machine.StorageRoleName, dev string, created bool) bool {139	if created {140		return false141	}142	unclean, err := fatStopMark(name, dev)143	if err != nil {144		fmt.Fprintf(os.Stderr, "liken: storage: reading %s's stop mark: %v\n", name, err)145		return false146	}147	if !unclean {148		return false149	}150	fmt.Printf("liken: storage: %s was not released at the last stop\n", name)151	healFATRole(name, dev)152	return true153}154155// healFATRole clears a volume's mark when liken can vouch for what is156// on it. It never reports failure to its caller: a volume that keeps157// its mark is the state the machine is already in, and the fact158// reaches status either way.159func healFATRole(name machine.StorageRoleName, dev string) {160	switch {161	case isSystemSlot(name) && string(name) == bootedSlotRole():162		fmt.Printf("liken: storage: %s is the slot this machine booted, so its mark stays until a boot from the other slot\n", name)163		return164	case isSystemSlot(name):165		if err := checkSlotArtifacts(dev); err != nil {166			fmt.Fprintf(os.Stderr, "liken: storage: %s keeps its mark: %v\n", name, err)167			return168		}169		fmt.Printf("liken: storage: %s holds every artifact its release document names; clearing its mark\n", name)170	case name == machine.BootHomeRole:171		// This boot rewrites GRUB's configuration and environment from172		// the manifest, so whatever the last stop left here is173		// replaced rather than trusted.174		fmt.Printf("liken: storage: %s is rewritten from the manifest on every boot; clearing its mark\n", name)175	default:176		return177	}178	if err := disks.ClearFAT32Dirty(dev); err != nil {179		fmt.Fprintf(os.Stderr, "liken: storage: clearing %s's mark: %v\n", name, err)180	}181}182183// bootedSlotRole names the role of the slot this machine booted, so184// the caller can tell it apart from the one that is idle. A boot with185// no slot parameter booted no slot at all, and then neither slot is186// in use.187func bootedSlotRole() string {188	switch bootParamValue("liken.slot") {189	case "A":190		return string(machine.SystemARole)191	case "B":192		return string(machine.SystemBRole)193	}194	return ""195}196197// checkSlotArtifacts reads a slot and checks every artifact its198// release document names against that document's digests. It mounts199// the slot read-only, which leaves the volume's mark untouched, and200// unmounts before returning so that the caller can write to the201// device.202//203// A slot with no release document has never been written by liken and204// is not something to vouch for.205func checkSlotArtifacts(dev string) error {206	if err := os.MkdirAll(fatCheckMount, 0o755); err != nil {207		return err208	}209	if err := mountFilesystem(dev, fatCheckMount, "vfat", unix.MS_RDONLY, ""); err != nil {210		return fmt.Errorf("mounting %s to check it: %w", dev, err)211	}212	err := verifySlotContents(fatCheckMount)213	if unmountErr := unmountFilesystem(fatCheckMount, 0); unmountErr != nil {214		// The check is worthless if the volume is still mounted, since215		// the device cannot be written underneath it.216		_ = unmountFilesystem(fatCheckMount, unix.MNT_DETACH)217		return fmt.Errorf("releasing %s after checking it: %w", dev, unmountErr)218	}219	return err220}221222// verifySlotContents hashes every artifact on a mounted slot against223// the release document that the slot carries.224func verifySlotContents(mount string) error {225	raw, err := os.ReadFile(filepath.Join(mount, "release.yaml"))226	if err != nil {227		return fmt.Errorf("this slot carries no release document: %w", err)228	}229	release, err := machine.ParseRelease(raw)230	if err != nil {231		return fmt.Errorf("this slot's release document does not parse: %w", err)232	}233	for _, artifact := range release.Artifacts {234		if err := verifyFile(artifact, filepath.Join(mount, artifact.Name)); err != nil {235			return fmt.Errorf("%s does not match the release document: %w", artifact.Name, err)236		}237	}238	return nil239}
init/features.go 91.3%
1package main23// Actuating the cluster's opt-in features on this machine.4//5// The Cluster document's spec.features lists the fleet's opt-ins from6// liken's curated vocabulary (the cluster package's features.go).7// This pass is init's half of carrying them out: one verdict per8// enabled feature, bound for status.features through the facts tree,9// with a console line for each, so the console and the API tell the10// same story. The code validates the cluster document against the11// vocabulary at parse time, so every slug that reaches this pass is a12// known one.13//14// Two kinds of feature arrive here, distinguished by the vocabulary15// table and never by the user. A bundled feature is a component the16// k3s binary already carries. Its whole actuation is the disable17// list this boot renders into the k3s drop-in (k3s.go); nothing18// about it can be missing from the image, and it always reports19// Active. A vendored feature is a payload the image ships inert:20// kernel modules listed at /etc/liken/features/<slug>/modules.conf,21// sometimes a workload manifest alongside them, sometimes a22// boot-time file that only this machine can write. This pass is the23// gate that makes a declared payload real. The absence of24// modules.conf is how a machine reports that its image predates a25// feature the cluster now declares.2627import (28	"errors"29	"fmt"30	"io/fs"31	"os"32	"path/filepath"33	"strings"3435	"github.com/liken-sh/liken/liken/cluster"36	"github.com/liken-sh/liken/liken/machine"37)3839// These are package variables rather than constants, so tests can40// point the actuation at trees of their own making.41var (42	featuresDir = "/etc/liken/features"43	// Every manifest liken writes for k3s to apply goes here, in the44	// subdirectory of k3s's auto-deploy directory that belongs to45	// liken alone. k3s stages the manifests of its own bundled46	// components at the top of that directory, and the teardown of a47	// component that a boot disables reads that component's file at48	// startup, so those files must stay where k3s put them: liken49	// writes and removes nothing outside this subdirectory50	// (init/k3s.go seeds it). k3s walks the whole tree, and an addon51	// takes its name from the file name and not from the directory52	// that holds it, so a manifest here reaches the cluster under the53	// same name it would have at the top.54	k3sManifestsDir = "/var/lib/rancher/" + likenManifestsRel55	iscsiDir        = "/etc/iscsi"56)5758func actuateFeatures(clusterDoc *cluster.Cluster, machineName string) []machine.FeatureStatus {59	slugs := clusterDoc.EnabledFeatures()60	if len(slugs) == 0 {61		return nil62	}63	moduleBase := filepath.Join("/lib/modules", kernelRelease())64	statuses := make([]machine.FeatureStatus, 0, len(slugs))65	for _, slug := range slugs {66		var status machine.FeatureStatus67		def := cluster.FeatureBySlug(slug)68		var paramsErr error69		if def != nil {70			paramsErr = def.ValidateParams(clusterDoc.Spec.Features[slug])71		}72		switch {73		case def == nil:74			// A slug that this binary's vocabulary does not include.75			// The parser deliberately lets it through (features.go in76			// the cluster package explains why a strict vocabulary77			// would disable a downgraded machine), so the report78			// happens here, naming both plausible causes: this image79			// predates the feature, or a hand-written seed misspelled80			// it.81			status = machine.FeatureStatus{82				Name:  slug,83				State: machine.FeatureMissing,84				Message: fmt.Sprintf(85					"this image's vocabulary has no %q feature; upgrade to a release that carries it, or fix the name if it is a misspelling (this image offers: %s)",86					slug, strings.Join(cluster.FeatureSlugs(), ", ")),87			}88		case paramsErr != nil:89			// A parameter this binary's vocabulary does not include,90			// on a slug it does. The parser lets it through for the91			// same downgrade reason as an unknown slug, and the same92			// two causes apply, so the report happens here too. The93			// feature fails whole, rather than actuating the part of94			// the declaration this image understands.95			status = machine.FeatureStatus{96				Name:    slug,97				State:   machine.FeatureFailed,98				Message: paramsErr.Error(),99			}100		case def.Kind == cluster.FeatureVendored:101			status = actuateVendoredFeature(moduleBase, slug, machineName)102		case def.Kind == cluster.FeatureWorkload:103			status = actuateWorkloadFeature(clusterDoc, slug)104		default:105			status = machine.FeatureStatus{Name: slug, State: machine.FeatureActive}106		}107		if status.Message != "" {108			fmt.Printf("liken: features: %s: %s: %s\n",109				status.Name, strings.ToLower(string(status.State)), status.Message)110		} else {111			fmt.Printf("liken: features: %s: %s\n",112				status.Name, strings.ToLower(string(status.State)))113		}114		statuses = append(statuses, status)115	}116	return statuses117}118119// actuateVendoredFeature makes one declared payload real: it loads120// the payload's modules, runs its boot hook, and seeds its workload121// manifests. Any step can fail without stopping the boot, because a122// machine missing a feature is degraded, not down. The status123// carries the story, and the message names the fix.124func actuateVendoredFeature(moduleBase, slug, machineName string) machine.FeatureStatus {125	status := machine.FeatureStatus{Name: slug, State: machine.FeatureActive}126	dir := filepath.Join(featuresDir, slug)127128	// The payload-shipped check. Every image that carries a vendored129	// feature stages its module list here, so the absence of the130	// file means the image itself is older than the feature. Nothing131	// on this machine can repair that; the fix is a release that132	// ships the payload.133	modulesConf := filepath.Join(dir, "modules.conf")134	if _, err := os.Stat(modulesConf); errors.Is(err, fs.ErrNotExist) {135		status.State = machine.FeatureMissing136		status.Message = fmt.Sprintf(137			"this image predates the %s feature; upgrade to a release whose image carries it", slug)138		return status139	}140141	// The feature's modules load through the same pipeline as the142	// spec's declared modules, and any verdict short of healthy143	// fails the whole feature: a storage client whose transport is144	// missing is not partly active.145	names, err := readModuleList(modulesConf)146	if err != nil {147		status.State = machine.FeatureFailed148		status.Message = err.Error()149		return status150	}151	for _, m := range loadDeclaredModulesFrom(moduleBase, names, nil) {152		if m.State == machine.ModuleLoaded || m.State == machine.ModuleBuiltin {153			continue154		}155		status.State = machine.FeatureFailed156		status.Message = fmt.Sprintf("module %s: %s", m.Name, m.Message)157		return status158	}159160	// The feature's boot hook, for the few files that only the161	// booting machine can write.162	if hook := featureBootHooks[slug]; hook != nil {163		if err := hook(machineName); err != nil {164			status.State = machine.FeatureFailed165			status.Message = err.Error()166			return status167		}168	}169170	// The feature's workload, if it has one, joins the manifests k3s171	// applies at startup. seedClusterState already reset liken's172	// subdirectory to exactly the image's own manifests when173	// clusterState mounted, so this pass adds only what this boot's174	// cluster document declares, and a retracted feature's manifest175	// simply never reappears. The code copies the manifests on every176	// machine, not just leaders, on purpose: only a leader's k3s177	// reads the directory, an extra file on a follower has no178	// effect, and contents that varied by role would be one more179	// thing to reason about during a promotion.180	if err := seedFeatureManifests(slug); err != nil {181		status.State = machine.FeatureFailed182		status.Message = err.Error()183		return status184	}185	return status186}187188// actuateWorkloadFeature makes one workload feature real. There is189// no payload gate like the vendored features' modules.conf check: an190// image whose vocabulary contains the slug also carries its manifests,191// because one build produces both. The flux feature also proves its192// configuration first, so a declaration that cannot sync reports the193// missing parameter instead of seeding workloads that would only194// fail in pods.195func actuateWorkloadFeature(clusterDoc *cluster.Cluster, slug string) machine.FeatureStatus {196	status := machine.FeatureStatus{Name: slug, State: machine.FeatureActive}197	if slug == cluster.FeatureFlux {198		cfg, err := clusterDoc.FluxConfig()199		if err == nil {200			err = seedFluxSync(cfg)201		}202		if err != nil {203			status.State = machine.FeatureFailed204			status.Message = err.Error()205			return status206		}207	}208	if err := seedFeatureManifests(slug); err != nil {209		status.State = machine.FeatureFailed210		status.Message = err.Error()211		return status212	}213	return status214}215216// seedFluxSync renders the flux feature's two sync objects from the217// declared parameters and seeds them beside the feature's manifests.218// These are the objects `flux bootstrap` would otherwise commit to219// the repository. Here they stay liken's forever: they re-render on220// every boot from the Cluster document, so editing the declaration221// is a real act, and the repository should not carry its own copies,222// or git and liken would fight over them. A repository that already223// carries them, such as one liken adopts, declares prune: "false",224// because the Kustomization would otherwise appear in its own225// inventory and delete everything the repository applied on the first226// build that stops producing it. The engine itself takes the opposite227// arrangement, planted once by the cluster operator and owned by the228// repository from then on (cluster-operator/flux.go).229func seedFluxSync(cfg *cluster.FluxConfig) error {230	// Flux spells sync paths relative to the repository root. The231	// default "." becomes "./", and a declared path gains the "./"232	// prefix the convention expects.233	path := "./"234	if cfg.Path != "." {235		path = "./" + strings.TrimPrefix(cfg.Path, "./")236	}237	// The rendered values are quoted with %q, which is JSON string238	// quoting, and JSON strings are valid YAML scalars. This keeps a239	// hostile-looking branch name or path from becoming YAML240	// structure. prune is the exception: it is a YAML boolean, and %t241	// renders it as true or false.242	rendered := fmt.Sprintf(`# Rendered by liken from the Cluster document's flux declaration.243# liken re-renders these two objects on every boot, so a copy in git244# would fight this one. A repository that carries its own copies must245# declare prune: "false" in the Cluster document.246apiVersion: source.toolkit.fluxcd.io/v1247kind: GitRepository248metadata:249  name: flux-system250  namespace: flux-system251spec:252  interval: 1m0s253  url: %q254  ref:255    branch: %q256  secretRef:257    name: flux-system258---259apiVersion: kustomize.toolkit.fluxcd.io/v1260kind: Kustomization261metadata:262  name: flux-system263  namespace: flux-system264spec:265  interval: 10m0s266  path: %q267  prune: %t268  sourceRef:269    kind: GitRepository270    name: flux-system271`, cfg.Repository, cfg.Branch, path, cfg.Prune)272	if err := os.MkdirAll(k3sManifestsDir, 0o755); err != nil {273		return err274	}275	return os.WriteFile(filepath.Join(k3sManifestsDir, "flux-sync.yaml"), []byte(rendered), 0o644)276}277278// featureBootHooks are the per-feature boot-time contributions, keyed279// by slug. The iscsi hook writes the initiator's identity. A feature280// with no boot-time file simply has no entry here.281var featureBootHooks = map[string]func(machineName string) error{282	"iscsi": writeInitiatorName,283}284285// writeInitiatorName gives the machine its iSCSI identity: the name286// this initiator presents when it logs in to a target, which storage287// arrays use in their access lists. The name derives from the288// machine name, so it stays the same on every boot with nothing to289// persist, and a reinstalled machine comes back as itself. The290// iqn.2026-07.sh.liken prefix follows the iSCSI naming convention: a291// date when the naming authority (the liken.sh domain) was292// registered, then the authority written in reverse, so that names293// under it can never collide with another owner's names. Both halves294// of the initiator read this file: the host's iscsiadm, which CSI295// drivers run through a chroot, directly, and the iscsid DaemonSet296// through its /etc/iscsi hostPath.297func writeInitiatorName(machineName string) error {298	if err := os.MkdirAll(iscsiDir, 0o755); err != nil {299		return err300	}301	name := fmt.Sprintf("InitiatorName=iqn.2026-07.sh.liken:%s\n", machineName)302	return os.WriteFile(filepath.Join(iscsiDir, "initiatorname.iscsi"), []byte(name), 0o600)303}304305// renderedFeatureManifests names the files a feature's actuation306// renders into the auto-deploy directory, beyond the manifests307// copied verbatim from the image. Retraction needs these names,308// because a glob over the image's manifests cannot see a rendered309// file.310var renderedFeatureManifests = map[string][]string{311	cluster.FeatureFlux: {"flux-sync.yaml"},312}313314// featureManifestPaths lists one feature's workload manifests as the315// image ships them. This is the single place that spells out the316// layout (featuresDir/<slug>/manifests/*.yaml), so seeding and317// retraction can never disagree about where a feature's workloads318// live. A feature with no manifests directory has no workload, which319// is fine; nfs takes exactly that shape.320func featureManifestPaths(slug string) ([]string, error) {321	return filepath.Glob(filepath.Join(featuresDir, slug, "manifests", "*.yaml"))322}323324// seedFeatureManifests copies a feature's workload manifests into325// k3s's auto-deploy directory.326func seedFeatureManifests(slug string) error {327	manifests, err := featureManifestPaths(slug)328	if err != nil {329		return err330	}331	for _, manifest := range manifests {332		raw, err := os.ReadFile(manifest)333		if err != nil {334			return err335		}336		if err := os.MkdirAll(k3sManifestsDir, 0o755); err != nil {337			return err338		}339		if err := os.WriteFile(filepath.Join(k3sManifestsDir, filepath.Base(manifest)), raw, 0o644); err != nil {340			return err341		}342	}343	return nil344}
init/grow.go 83.5%
1package main23// Growing partitions in place.4//5// Sizes in the storage spec are grow-only: a role may be declared6// larger than its partition, never smaller, because growing never7// moves data. A partition grows by rewriting its table entry's end8// sector and telling the filesystem inside (ext4.go's half of the9// job). Shrinking or moving would mean relocating live data, which10// is a data migration, and liken does not do migrations.11//12// Two rules bound what a grow can do:13//14//   - A partition can only grow into empty space directly after it.15//     If another partition starts there, the grow is unsatisfiable,16//     and an unsatisfiable spec fails reconciliation the same as any17//     other one. The machine does not quietly run with less than it18//     declared.19//20//   - Every table edit happens while nothing from that disk is21//     mounted. The kernel refuses to re-read the partition table of22//     a disk in use (BLKRRPART returns EBUSY), which is why growth23//     runs after recognition and before the code mounts any role.24//25// A disk that was itself grown, for example by the lab's qemu-img26// resize or a cloud volume expansion, needs its table rewritten even27// when no partition changes: the backup table belongs at the end of28// the disk, and the end just moved. Remainder roles, which have no29// declared size, grow to the new last usable sector when that30// happens. Their size was always defined as the rest of the disk, so31// when the disk grows, they grow with it.3233import (34	"fmt"35	"os"3637	"github.com/liken-sh/liken/liken/disks"38	"github.com/liken-sh/liken/liken/machine"39)4041// A growth is one entry's extension: which table slot, and the new42// final sector.43type growth struct {44	entryIndex int45	newLastLBA uint6446}4748// planGrowth compares each declared role recognized on this disk49// against its table entry, and determines what must grow. It is pure;50// the device name appears only in error messages. rewrite reports51// whether the table needs rewriting even when no extent changes,52// which is how a grown disk gets its backup table moved to the new53// end.54func planGrowth(device string, roles []machine.DeclaredRole, t *disks.Table, totalSectors uint64) (edits []growth, rewrite bool, err error) {55	lastUsable := disks.LastUsableLBA(totalSectors)56	for _, role := range roles {57		idx := -158		for i, e := range t.Entries {59			if e.Name == role.PartitionName() {60				idx = i61				break62			}63		}64		if idx < 0 {65			continue // this role's partition lives on another disk66		}67		e := t.Entries[idx]6869		var target uint6470		if role.Size == "" {71			// A remainder role's size is "the rest of the disk", so72			// growth for it means the disk itself grew.73			target = lastUsable74			if target <= e.LastLBA {75				continue76			}77		} else {78			bytes, _ := machine.ParseSize(role.Size) // validated before any disk is touched79			declared := (bytes + disks.SectorSize - 1) / disks.SectorSize80			if declared <= e.LastLBA-e.FirstLBA+1 {81				// Already at or above the declared size: satisfied.82				// (A shrink that the operator failed to refuse also83				// lands here, tolerated and never acted on.)84				continue85			}86			target = e.FirstLBA + declared - 187			if target > lastUsable {88				return nil, false, fmt.Errorf("disk %s is too small to grow %s to %s: needs sector %d but the disk's usable space ends at %d",89					device, role.Name, role.Size, target, lastUsable)90			}91		}9293		// Growing never moves data, so anything that starts in the94		// space this partition needs leaves the spec unsatisfiable.95		for j, other := range t.Entries {96			if j == idx {97				continue98			}99			if other.FirstLBA > e.LastLBA && other.FirstLBA <= target {100				return nil, false, fmt.Errorf("cannot grow %s on %s to sector %d: partition %q begins at sector %d, in the way (growing never moves data)",101					role.Name, device, target, other.Name, other.FirstLBA)102			}103		}104		edits = append(edits, growth{entryIndex: idx, newLastLBA: target})105	}106	return edits, len(edits) > 0 || t.AlternateLBA != totalSectors-1, nil107}108109// A growPlan is one disk's pending rewrite. The code computes it up110// front (the plan-everything-then-apply rule in storage.go's111// header), and applies it only after every disk's plan succeeds.112type growPlan struct {113	device       string114	totalSectors uint64115	table        *disks.Table116	edits        []growth117}118119// planAllGrowth reads each recognized disk's table and plans its120// growth, and writes nothing.121func planAllGrowth(roles []machine.DeclaredRole, found map[machine.StorageRoleName]partition) ([]growPlan, error) {122	byDisk := map[string][]machine.DeclaredRole{}123	var order []string124	for _, role := range roles {125		p, ok := found[role.Name]126		if !ok {127			continue128		}129		// The boot roles never grow: ext4 grows in place by ioctl, but130		// FAT's geometry is fixed at format time (the slots, bootHome),131		// and biosBoot's whole purpose is a layout that the MBR's132		// literal sector numbers can rely on. Their sizes are settled133		// on the day they are claimed. The code refuses a spec asking134		// for more here, at planning time, before it writes anything,135		// and the refusal follows the usual staged-spec path: the code136		// rejects the spec, and the boot falls back to the proven137		// manifest.138		if isFixedSizeRole(role.Name) {139			if role.Size != "" {140				if bytes, _ := machine.ParseSize(role.Size); bytes > p.sizeBytes {141					return nil, fmt.Errorf(142						"%s is %s and can't grow to %s: boot roles are fixed when claimed",143						role.Name, gib(p.sizeBytes), role.Size)144				}145			}146			continue147		}148		if _, seen := byDisk[p.disk]; !seen {149			order = append(order, p.disk)150		}151		byDisk[p.disk] = append(byDisk[p.disk], role)152	}153154	var plans []growPlan155	for _, diskName := range order {156		device := devRoot + "/" + diskName157		disk := diskByPath(device)158		if disk == nil {159			return nil, fmt.Errorf("recognized partitions on %s but the disk is not in the inventory", device)160		}161		totalSectors := disk.SizeBytes / disks.SectorSize162163		f, err := os.Open(device)164		if err != nil {165			return nil, fmt.Errorf("examining %s: %w", device, err)166		}167		t, err := disks.ReadGPT(f, totalSectors)168		f.Close()169		if err != nil {170			return nil, fmt.Errorf("reading the partition table on %s: %w", device, err)171		}172173		edits, rewrite, err := planGrowth(device, byDisk[diskName], t, totalSectors)174		if err != nil {175			return nil, err176		}177		if !rewrite {178			continue179		}180		plans = append(plans, growPlan{device: device, totalSectors: totalSectors, table: t, edits: edits})181	}182	return plans, nil183}184185// applyGrowth rewrites one disk's table with its planned extensions186// and waits for the kernel to show the new geometry.187func applyGrowth(plan growPlan) error {188	// A plan with no edits is a pure relocation: the disk grew, no189	// partition's extent changes, and only the backup copy of the190	// table needs to move to the new end. The kernel's view of the191	// partitions is already correct, so the code does not ask it to192	// re-read the table, and it could not ask anyway. On the disk193	// carrying the running system, the boot slot has been mounted194	// since early boot found the system image on it, and the kernel195	// refuses to re-read a disk in use. (This is the normal first196	// boot after the liken.sh deployment stamps its disk image onto197	// a slightly larger disk.)198	if len(plan.edits) == 0 {199		fmt.Printf("liken: storage: %s grew; relocating its backup partition table to the new end\n", plan.device)200		if err := disks.WriteTableInPlace(plan.device, plan.totalSectors, plan.table); err != nil {201			return fmt.Errorf("rewriting the partition table on %s: %w", plan.device, err)202		}203		return nil204	}205206	for _, g := range plan.edits {207		e := &plan.table.Entries[g.entryIndex]208		fmt.Printf("liken: storage: growing %s on %s from %s to %s\n",209			e.Name, plan.device,210			gib((e.LastLBA-e.FirstLBA+1)*disks.SectorSize),211			gib((g.newLastLBA-e.FirstLBA+1)*disks.SectorSize))212		e.LastLBA = g.newLastLBA213	}214	if err := disks.WriteTable(plan.device, plan.totalSectors, plan.table); err != nil {215		return fmt.Errorf("rewriting the partition table on %s: %w", plan.device, err)216	}217218	var expect []disks.Partition219	for _, e := range plan.table.Entries {220		expect = append(expect, disks.Partition{Name: e.Name, FirstLBA: e.FirstLBA, LastLBA: e.LastLBA})221	}222	return waitForPartitions(expect)223}
init/grubactuator.go 85.5%
1package main23// The GRUB dialect of the boot actuator (actuator.go describes the4// interface and explains why it exists).5//6// BIOS firmware holds no boot variables, so a BIOS machine keeps its7// boot preferences where GRUB can read them: the environment block on8// the boot home (grubenv.go describes the format). This dialect maps9// one to one onto the UEFI dialect. try_slot is the one-shot trial10// (grub.cfg consumes it before it loads a single kernel byte, in the11// same way that firmware consumes BootNext). default_slot is the12// standing preference, and it corresponds to the first entry in13// BootOrder.14//15// Both dialects heal, for the same reason and by different means. A16// BIOS machine's boot path lives on the disk itself: the MBR's boot17// code, GRUB's core image, and the config on the boot home. Cloud hosts18// are known to rewrite MBRs under running machines (Linode's boot-mode19// changes do exactly that). A UEFI machine's boot path lives in NVRAM,20// which a firmware update or a dead battery resets to defaults21// (efiactuator.go). So, when this dialect asserts the proven slot, it22// also re-derives every byte of the boot chain from the proven slot's23// own artifacts, and it puts back whatever disagrees. This healing runs24// on every boot and on the way down before every reboot, because a boot25// path that is zeroed while the machine runs must be healed before the26// reboot. Otherwise, the machine would never come back.2728import (29	"bytes"30	"fmt"31	"os"32	"path/filepath"3334	"github.com/liken-sh/liken/liken/machine"35)3637type grubActuator struct {38	// grubDir is the boot home's grub directory. It holds grub.cfg39	// and grubenv.40	grubDir string41	// machineName is rendered into grub.cfg when this code heals42	// grub.cfg. This value comes from the kernel command line, where43	// GRUB's own config put it.44	machineName string45}4647func (a grubActuator) envPath() string { return filepath.Join(a.grubDir, "grubenv") }4849func (a grubActuator) canArmTrial(slot string) error {50	if _, err := readGRUBEnv(a.envPath()); err != nil {51		return fmt.Errorf("the GRUB environment block is not usable: %w", err)52	}53	return nil54}5556func (a grubActuator) armTrial(slot string) (string, error) {57	if err := updateGRUBEnv(a.envPath(), map[string]string{"try_slot": slot}); err != nil {58		return "", fmt.Errorf("arming try_slot: %w", err)59	}60	return "try_slot=" + slot + " written to the GRUB environment block", nil61}6263func (a grubActuator) fallbackLeads(slot string) bool {64	env, err := readGRUBEnv(a.envPath())65	return err == nil && env["default_slot"] == slot66}6768// assertProven brings the whole GRUB boot path into agreement with69// the store. The environment block prefers the proven slot, and the70// boot chain on disk matches the proven slot's own artifacts.71func (a grubActuator) assertProven(slot string) {72	env, err := readGRUBEnv(a.envPath())73	switch {74	case err != nil:75		fmt.Fprintf(os.Stderr, "liken: system: reading the GRUB environment block: %v\n", err)76	case env["default_slot"] != slot || env["try_slot"] != "":77		// This call clears a leftover try_slot along with the78		// preference write. A one-shot trial that was armed for a79		// release since withdrawn must not run on a later reboot.80		if err := updateGRUBEnv(a.envPath(), map[string]string{"default_slot": slot, "try_slot": ""}); err != nil {81			fmt.Fprintf(os.Stderr, "liken: system: asserting default_slot in the GRUB environment block: %v\n", err)82		} else if readback, err := readGRUBEnv(a.envPath()); err != nil || readback["default_slot"] != slot {83			// This code trusts the readback, not the write, in the84			// same way that the UEFI dialect trusts the readback of85			// BootOrder.86			fmt.Fprintln(os.Stderr, "liken: system: the GRUB environment block was written but reads back unchanged")87		} else {88			fmt.Printf("liken: system: the GRUB environment block now prefers slot %s (proven)\n", slot)89		}90	}9192	// The environment block and the boot chain are separate93	// assertions on purpose. A torn environment block must not stop94	// the boot sectors from healing, and a problem in the boot95	// sectors must not stop the environment block assertion.96	a.healBootChain(slot)97}9899// healBootChain re-derives the on-disk boot chain from the proven100// slot's artifacts and rewrites whatever disagrees. The comparison101// runs every time. The write happens only when something drifted, so102// a healthy machine's console shows no message.103func (a grubActuator) healBootChain(slot string) {104	slotMount := slotMountPath(slot)105	if slotMount == "" {106		return107	}108	bootImg, err := os.ReadFile(filepath.Join(slotMount, "grub-boot.img"))109	if err != nil {110		// A proven release from before liken carried GRUB artifacts111		// cannot say what the boot sectors should hold. The chain112		// that booted this machine stays as it is.113		fmt.Printf("liken: system: slot %s carries no grub-boot.img; leaving the boot sectors alone\n", slot)114		return115	}116	coreImg, err := os.ReadFile(filepath.Join(slotMount, "grub-core.img"))117	if err != nil {118		fmt.Printf("liken: system: slot %s carries no grub-core.img; leaving the boot sectors alone\n", slot)119		return120	}121122	parts := discoverPartitions()123	biosBoot, err := findSlotPartition(parts, machine.BIOSBootRole)124	if err != nil {125		fmt.Fprintf(os.Stderr, "liken: system: %v; the boot sectors cannot be checked\n", err)126		return127	}128	diskDev, err := diskDevice(parts, machine.BIOSBootRole)129	if err != nil {130		fmt.Fprintf(os.Stderr, "liken: system: %v; the boot sectors cannot be checked\n", err)131		return132	}133	plan, err := planGRUBBootSectors(bootImg, coreImg, biosBoot)134	if err != nil {135		fmt.Fprintf(os.Stderr, "liken: system: planning the boot sectors: %v\n", err)136		return137	}138	disk, err := os.OpenFile(diskDev, os.O_RDWR, 0)139	if err != nil {140		fmt.Fprintf(os.Stderr, "liken: system: opening %s to check the boot sectors: %v\n", diskDev, err)141		return142	}143	defer disk.Close()144	ok, err := plan.inPlace(disk)145	if err != nil {146		fmt.Fprintf(os.Stderr, "liken: system: reading the boot sectors on %s: %v\n", diskDev, err)147		return148	}149	if !ok {150		if err := plan.write(disk); err != nil {151			fmt.Fprintf(os.Stderr, "liken: system: healing the boot sectors on %s: %v\n", diskDev, err)152			return153		}154		fmt.Printf("liken: system: the boot sectors on %s disagreed with slot %s's artifacts; healed\n", diskDev, slot)155	}156157	// grub.cfg is rendered, not copied, so it heals the same way:158	// this code re-renders it and compares the result. Without a159	// machine name, there is nothing correct to render, so this code160	// leaves the file alone rather than write an anonymous version.161	if a.machineName == "" {162		return163	}164	cfgPath := filepath.Join(a.grubDir, "grub.cfg")165	want := []byte(renderGRUBConfig(a.machineName, consoleArgs()))166	current, err := os.ReadFile(cfgPath)167	if err == nil && bytes.Equal(current, want) {168		return169	}170	if err := writeFileDurably(cfgPath, want); err != nil {171		fmt.Fprintf(os.Stderr, "liken: system: healing grub.cfg: %v\n", err)172		return173	}174	fmt.Println("liken: system: grub.cfg disagreed with this machine's rendering; healed")175}176177// slotMountPath translates a slot letter to its mountpoint. It uses178// the same table that storage reconciliation uses for mounts, so179// tests can substitute a temporary directory, the same way tests do180// for every role.181func slotMountPath(slot string) string {182	switch slot {183	case "A":184		return roleMounts[machine.SystemARole].path185	case "B":186		return roleMounts[machine.SystemBRole].path187	}188	return ""189}190191// biosFirmwareFacts reports a BIOS machine's boot configuration from192// where that machine actually keeps it: GRUB's environment block and193// the kernel command line that GRUB composed. The fields deliberately194// mirror the UEFI report. A fleet listing reads the same way for195// either firmware type, and each fact names the mechanism it came196// from.197func biosFirmwareFacts() machine.FirmwareStatus {198	fw := machine.FirmwareStatus{Mode: machine.FirmwareBIOS}199	if slot := bootParamValue("liken.slot"); slot != "" {200		fw.BootCurrent = fmt.Sprintf("liken slot %s (liken.slot= on the kernel command line)", slot)201	}202	grubDir := filepath.Join(roleMounts[machine.BootHomeRole].path, "grub")203	env, err := readGRUBEnv(filepath.Join(grubDir, "grubenv"))204	if err != nil {205		return fw206	}207	if try := env["try_slot"]; try != "" {208		fw.BootNext = fmt.Sprintf("liken slot %s (grubenv try_slot)", try)209	}210	if def := env["default_slot"]; def != "" {211		fw.BootOrder = []string{fmt.Sprintf("liken slot %s (grubenv default_slot)", def)}212	}213	return fw214}
init/grubcfg.go 98.0%
1package main23// Rendering grub.cfg: the BIOS machine's boot entries.4//5// On a UEFI machine, the installer writes two firmware boot entries6// (install.go's writeSlotBootEntry documents each command line flag)7// and the upgrade machinery steers between them with BootNext and8// BootOrder. This file renders the GRUB equivalent: one config on the9// boot home that reads the environment block (grubenv.go) and makes10// the same decisions that the firmware would have made.11//12// This code renders the config once, at install time, and the config13// stays static afterward. Everything that changes from boot to boot14// lives in the environment block, so steering the machine never means15// editing this file. What is per-machine here is exactly what is16// per-machine in a UEFI entry: the machine's name and its console.17//18// The sequence mirrors the firmware dialect:19//20//   - When try_slot is set, the previous boot armed a trial. The21//     config reads it and consumes it before it loads a single kernel22//     byte (set empty, then save_env). This matches how firmware23//     consumes BootNext. Any reset after this point, such as a panic,24//     a watchdog reset, or a power cut, boots default_slot. A crash in25//     the window between save_env and the kernel jump reads as "tried26//     and fell back". This wrongly rejects a release that never ran.27//     armProvingBoot documents why this tradeoff, a fixable false28//     rejection over an unfixable reboot loop, is the right one. The29//     same reasoning applies here.30//31//   - fallback=1 is the BootOrder fall-through. UEFI firmware moves32//     down BootOrder when an entry fails to load. GRUB instead drops33//     to an interactive prompt, which is a hang on a headless machine.34//     With a fallback entry, a chosen slot whose kernel cannot be35//     found or loaded falls through to the default slot instead.36//37//   - The empty-slot default (A) means that even a torn or38//     freshly-made environment block boots something: slot A is where39//     the installer put the first release.4041import (42	"strings"4344	"github.com/liken-sh/liken/liken/machine"45)4647func renderGRUBConfig(machineName string, consoles []string) string {48	cfg := "# Rendered by liken at install time; do not edit. Boot-to-boot\n" +49		"# state lives in grubenv, not here (init/grubcfg.go explains).\n"5051	// The console: GRUB's own output goes to the machine's console, so52	// boot problems are visible on the same wire that the kernel's53	// messages use. serialConsoleDirectives returns an empty string on54	// a machine with no serial console, and GRUB then uses its default,55	// the VGA text console.56	cfg += serialConsoleDirectives(consoles)5758	cfg += `59load_env6061if [ -n "$try_slot" ]; then62    set slot=$try_slot63    set try_slot=64    save_env try_slot65else66    set slot=$default_slot67fi68if [ -z "$slot" ]; then69    set slot=A70fi7172set default=073set timeout=074set fallback=17576`77	// The two entries differ only in which slot variable they read.78	// Each mirrors the UEFI entry's command line: the kernel and its79	// three initrds from the slot found by its label, liken.slot= to80	// tell the booted system which half of the blue-green pair it is81	// on, and panic=10 so that a panicking trial resets into the82	// fallback instead of hanging.83	kernelArgs := grubKernelArgs(machineName, consoles)84	cfg += grubMenuEntry("liken (chosen slot)", "$slot", kernelArgs)85	cfg += grubMenuEntry("liken (default slot)", "$default_slot", kernelArgs)86	return cfg87}8889// grubMenuEntry writes one slot's entry. GRUB's initrd directive90// takes a space-separated list and concatenates the files in order,91// so the microcode early cpio leads, the same order the UEFI entry's92// initrd= lines declare and for the same reason: the kernel scans93// the very start of its initrd for microcode.94func grubMenuEntry(title, slotExpr, kernelArgs string) string {95	return "menuentry '" + title + "' {\n" +96		"    search --no-floppy --label LIKEN-SYS-" + slotExpr + " --set=root\n" +97		"    linux /vmlinuz " + kernelArgs + " liken.slot=" + slotExpr + " panic=10\n" +98		"    initrd /microcode.cpio /boot.cpio /" + machine.LayerName + "\n" +99		"}\n"100}101102// grubKernelArgs assembles the command line that both entries share,103// from the same parts that writeSlotBootEntry uses: the machine's104// consoles, rdinit, and its name. (Each entry appends its own slot105// and panic arguments, since the slot differs between entries.)106func grubKernelArgs(machineName string, consoles []string) string {107	args := ""108	for _, console := range consoles {109		args += console + " "110	}111	return args + "rdinit=/liken liken.machine=" + machineName112}113114// serialConsoleDirectives turns console=ttyS<unit>[,<speed>...]115// arguments into GRUB's serial terminal setup, so that GRUB's menu116// and any error it prints reach the serial console that the machine117// is actually operated from. This function leaves other console forms118// (tty0, hvc0) to GRUB's default output.119func serialConsoleDirectives(consoles []string) string {120	for _, console := range consoles {121		rest, ok := strings.CutPrefix(console, "console=ttyS")122		if !ok {123			continue124		}125		unit, options, _ := strings.Cut(rest, ",")126		if unit == "" {127			continue128		}129		speed := "115200"130		if options != "" {131			// The kernel's serial options are <speed><parity><bits>.132			// Only the leading digits are the speed.133			digits := options134			for i, r := range options {135				if r < '0' || r > '9' {136					digits = options[:i]137					break138				}139			}140			if digits != "" {141				speed = digits142			}143		}144		return "serial --unit=" + unit + " --speed=" + speed + "\n" +145			"terminal_output serial console\n" +146			"terminal_input serial console\n"147	}148	return ""149}
init/grubenv.go 96.2%
1package main23// GRUB's environment block: the BIOS machine's boot variables.4//5// UEFI firmware gives liken a small durable store with exactly the6// right shape for blue-green boots: BootNext (try this once) and7// BootOrder (prefer this from now on). BIOS firmware stores nothing,8// so GRUB provides the equivalent: a preallocated 1024-byte file at a9// fixed location that GRUB itself can read and write at boot time,10// designed for exactly this bookkeeping. liken keeps two variables in11// it: default_slot, the proven slot that every unremarkable boot12// should run (it corresponds to BootOrder), and try_slot, the13// one-shot trial (it corresponds to BootNext), which grub.cfg14// consumes before it loads a single kernel byte.15//16// The format is fixed so that GRUB can rewrite the file in place17// through its own filesystem driver: a signature line, then18// name=value lines (lines starting with # are comments), padded with19// '#' characters to exactly 1024 bytes. The size never changes. This20// is what makes the boot-time write safe on FAT: the file's blocks21// are simply overwritten, and no allocation moves.22//23// liken writes the block from Go rather than shipping grub-editenv,24// because it is a 1 KiB documented format, and it fits the same25// write-it-by-hand approach that the GPT and FAT writers already26// follow. Writes from Linux go through the durable temp-and-rename27// path. Unlike GRUB, init has a real filesystem driver underneath,28// and GRUB re-resolves the file's blocks fresh each boot, so the29// in-place constraint applies only to GRUB's own save_env.3031import (32	"fmt"33	"maps"34	"os"35	"sort"36	"strings"37)3839const (40	grubEnvSize      = 102441	grubEnvSignature = "# GRUB Environment Block\n"42)4344// parseGRUBEnv reads an environment block's variables. Its strictness45// matches what GRUB itself accepts: exactly 1024 bytes, the signature46// first, comments ignored, and the padding after the last variable47// never parsed as content.48func parseGRUBEnv(block []byte) (map[string]string, error) {49	if len(block) != grubEnvSize {50		return nil, fmt.Errorf("a GRUB environment block is exactly %d bytes, not %d", grubEnvSize, len(block))51	}52	text := string(block)53	if !strings.HasPrefix(text, grubEnvSignature) {54		return nil, fmt.Errorf("the GRUB environment block signature is missing")55	}56	vars := map[string]string{}57	for line := range strings.SplitSeq(text, "\n") {58		if line == "" || strings.HasPrefix(line, "#") {59			continue60		}61		name, value, ok := strings.Cut(line, "=")62		if !ok || name == "" {63			return nil, fmt.Errorf("the GRUB environment block holds a line that is neither comment nor name=value: %q", line)64		}65		vars[name] = value66	}67	return vars, nil68}6970// renderGRUBEnv lays out a block holding exactly these variables, in71// sorted order so the same variables always produce the same bytes.72func renderGRUBEnv(vars map[string]string) ([]byte, error) {73	var b strings.Builder74	b.WriteString(grubEnvSignature)75	names := make([]string, 0, len(vars))76	for name := range vars {77		names = append(names, name)78	}79	sort.Strings(names)80	for _, name := range names {81		value := vars[name]82		if strings.ContainsAny(name, "=\n#") || name == "" {83			return nil, fmt.Errorf("%q cannot name a GRUB environment variable", name)84		}85		if strings.Contains(value, "\n") {86			return nil, fmt.Errorf("a GRUB environment value cannot span lines: %q", value)87		}88		b.WriteString(name)89		b.WriteString("=")90		b.WriteString(value)91		b.WriteString("\n")92	}93	if b.Len() > grubEnvSize {94		return nil, fmt.Errorf("the variables overflow the block: %d bytes into %d", b.Len(), grubEnvSize)95	}96	block := make([]byte, grubEnvSize)97	copy(block, b.String())98	for i := b.Len(); i < grubEnvSize; i++ {99		block[i] = '#'100	}101	return block, nil102}103104// updateGRUBEnv is the read-modify-write function that the actuator105// uses. It loads the block at path, applies the given values (an106// empty value still writes the variable, present but empty, which is107// how a one-shot reads after GRUB consumes it), and writes the result108// durably. A variable mapped to the empty string stays in the block on109// purpose: absent and empty read the same to grub.cfg's -n tests, and110// keeping the name visible makes the block easier to inspect.111func updateGRUBEnv(path string, set map[string]string) error {112	raw, err := os.ReadFile(path)113	if err != nil {114		return err115	}116	vars, err := parseGRUBEnv(raw)117	if err != nil {118		return err119	}120	maps.Copy(vars, set)121	block, err := renderGRUBEnv(vars)122	if err != nil {123		return err124	}125	return writeFileDurably(path, block)126}127128// readGRUBEnv loads and parses the block at path.129func readGRUBEnv(path string) (map[string]string, error) {130	raw, err := os.ReadFile(path)131	if err != nil {132		return nil, err133	}134	return parseGRUBEnv(raw)135}
init/grubinstall.go 88.9%
1package main23// Patching GRUB's boot sectors: what grub-bios-setup does, done by4// hand, in the same way that this repo writes partition tables and5// filesystems by hand. This lets the machine verify and heal its own6// boot sectors instead of depending on a rescue boot and a person7// running dd.8//9// BIOS boot is a chain of disk addresses set at install time. The10// firmware loads sector 0 and jumps into it. Those 440 bytes11// (boot.img) hold just enough code to load one more sector: the12// first sector of the core image, whose address is patched into13// boot.img at a fixed offset. That first sector, which GRUB calls14// diskboot, ends with a blocklist: the disk address and length of15// the rest of the core image, patched in the same way. Only from16// that point does GRUB have real filesystem drivers, and only then17// does it stop needing literal sector numbers.18//19// liken keeps every one of those addresses derivable. The core image20// lives at the start of the biosBoot partition, contiguous, so the21// whole chain is a pure function of boot.img, core.img, and the22// partition's first sector. This is what makes healing reliable: any23// boot can recompute the expected bytes from the proven slot's24// artifacts and compare them against the disk. When a Linode image25// deploy zeroes the MBR, which has happened twice, this code repairs26// it at the next opportunity instead of leaving the machine unable to27// boot.2829import (30	"bytes"31	"encoding/binary"32	"fmt"33	"io"3435	"github.com/liken-sh/liken/liken/disks"36)3738// The offsets that grub-bios-setup patches, fixed by boot.img's layout:39//40//	0x5c  GRUB_BOOT_MACHINE_KERNEL_SECTOR: the 64-bit LBA of the41//	      core image's first sector42//	0x66  GRUB_BOOT_MACHINE_DRIVE_CHECK: a workaround for BIOSes43//	      that pass a garbage boot drive in DL. On hard disks,44//	      grub-bios-setup disables the check by overwriting the two45//	      instruction bytes with NOPs.46//47// And the offsets in diskboot, fixed by the last 12 bytes of the core48// image's first sector: the blocklist entry that names where the rest49// of the image lives (start sector, sector count), and the real-mode50// segment it loads at (0x820, compiled in, asserted here and never51// written).52const (53	grubKernelSectorOffset = 0x5c54	grubDriveCheckOffset   = 0x6655	grubBlocklistStart     = 50056	grubBlocklistLength    = 50857	grubBlocklistSegment   = 51058	grubLoadSegment        = 0x82059	mbrBootCodeBytes       = 44060)6162// grubBootSectors is one machine's expected boot chain: the bytes63// that belong at LBA 0 and at the biosBoot partition, computed from a64// release's artifacts and the partition's location. Installing and65// healing perform the same operation on this value: write what66// should be there.67type grubBootSectors struct {68	mbr     []byte // the 440 boot-code bytes for sector 069	core    []byte // the patched core image for the partition70	coreLBA uint64 // the partition's first sector71}7273// planGRUBBootSectors patches copies of the release's grub-boot.img74// and grub-core.img for a machine whose biosBoot partition starts at75// coreLBA. This function validates everything, so a write can only76// ever put a coherent chain on disk.77func planGRUBBootSectors(bootImg, coreImg []byte, part *slotPartition) (*grubBootSectors, error) {78	if len(bootImg) != disks.SectorSize {79		return nil, fmt.Errorf("grub-boot.img is %d bytes; boot.img is one %d-byte sector", len(bootImg), disks.SectorSize)80	}81	if len(coreImg) < disks.SectorSize {82		return nil, fmt.Errorf("grub-core.img is %d bytes, not even one sector; it cannot carry a blocklist", len(coreImg))83	}84	coreSectors := (uint64(len(coreImg)) + disks.SectorSize - 1) / disks.SectorSize85	if available := part.lastLBA - part.firstLBA + 1; coreSectors > available {86		return nil, fmt.Errorf("grub-core.img needs %d sectors but the biosBoot partition holds %d", coreSectors, available)87	}8889	mbr := bytes.Clone(bootImg[:mbrBootCodeBytes])90	binary.LittleEndian.PutUint64(mbr[grubKernelSectorOffset:], part.firstLBA)91	// Two NOPs disable the buggy-BIOS drive check, as grub-bios-setup92	// does for any hard disk.93	mbr[grubDriveCheckOffset], mbr[grubDriveCheckOffset+1] = 0x90, 0x909495	core := bytes.Clone(coreImg)96	if seg := binary.LittleEndian.Uint16(core[grubBlocklistSegment:]); seg != grubLoadSegment {97		return nil, fmt.Errorf("grub-core.img's load segment is %#x, want %#x; this is not an i386-pc core image", seg, grubLoadSegment)98	}99	// The blocklist: the rest of the image follows its first sector100	// without gaps. The partition guarantees this.101	binary.LittleEndian.PutUint64(core[grubBlocklistStart:], part.firstLBA+1)102	binary.LittleEndian.PutUint16(core[grubBlocklistLength:], uint16(coreSectors-1))103104	return &grubBootSectors{mbr: mbr, core: core, coreLBA: part.firstLBA}, nil105}106107// inPlace reports whether the disk already carries this chain. This108// is the comparison half of healing, and it is cheap enough to run on109// every boot (440 bytes plus the core image).110func (s *grubBootSectors) inPlace(disk io.ReaderAt) (bool, error) {111	mbr := make([]byte, mbrBootCodeBytes)112	if _, err := disk.ReadAt(mbr, 0); err != nil {113		return false, fmt.Errorf("reading the boot code: %w", err)114	}115	if !bytes.Equal(mbr, s.mbr) {116		return false, nil117	}118	core := make([]byte, len(s.core))119	if _, err := disk.ReadAt(core, int64(s.coreLBA)*disks.SectorSize); err != nil {120		return false, fmt.Errorf("reading the core image: %w", err)121	}122	return bytes.Equal(core, s.core), nil123}124125// write puts the chain on disk: the core image first and synced, then126// the MBR's boot code last. This order makes a torn write safe. The127// MBR points at the partition's first sector, which never moves, so128// an old MBR over a new core image still boots. A new MBR over a129// half-written core image would jump into garbage instead.130func (s *grubBootSectors) write(disk disks.Device) error {131	if _, err := disk.WriteAt(s.core, int64(s.coreLBA)*disks.SectorSize); err != nil {132		return fmt.Errorf("writing the core image: %w", err)133	}134	if err := disk.Sync(); err != nil {135		return err136	}137	// This writes only the 440 boot-code bytes. Everything after them138	// in sector 0 (the disk signature, the protective MBR entry, the139	// boot signature) belongs to the partition table's writer.140	if _, err := disk.WriteAt(s.mbr, 0); err != nil {141		return fmt.Errorf("writing the boot code: %w", err)142	}143	return disk.Sync()144}
init/hardware.go 91.9%
1package main23// Hardware observation: the boot-time walk and the live watch that4// keep the unclaimed-device report correct.5//6// This approach comes from milestone 11: drivers are declared7// (spec.modules) and never auto-loaded, so a surprise device is an8// inert, reported fact. The kernel does everything else. A resident9// driver binds hot-plugged hardware without any userspace help. This10// leaves exactly one job here: notice undriven devices and report11// them, to the console and to the facts tree, where the operator12// lifts them into the Machine's status. One watcher produces both13// outputs, and the same watcher will one day feed ResourceSlices too.1415import (16	"context"17	"errors"18	"fmt"19	"path/filepath"20	"slices"21	"strings"22	"time"2324	"github.com/liken-sh/liken/liken/hardware"25	"github.com/liken-sh/liken/liken/machine"26)2728// The observation's inputs are variables, so tests can point them29// into fabricated trees. pciIDsPath is where the image stages30// hwdata's database. When this file is absent, devices show numeric31// names instead.32var (33	sysfsRoot  = "/sys"34	pciIDsPath = "/usr/share/hwdata/pci.ids"35)3637// loadHardwareCatalog loads the lookup tables once per boot. A nil38// catalog (an image without the full alias table) disables the39// report rather than the boot. The machine still runs; it just40// cannot name the devices that it is not driving.41func loadHardwareCatalog() *hardware.Catalog {42	moduleDir := filepath.Join("/lib/modules", kernelRelease())43	catalog, err := hardware.LoadCatalog(moduleDir, pciIDsPath)44	if err != nil {45		fmt.Printf("liken: hardware: no unclaimed-device reporting: %v\n", err)46		return nil47	}48	return catalog49}5051// discoverUnclaimed is the boot-time walk, nil-safe for images with52// no catalog. The declared serio entries decide whether a serial-line53// adapter still needs one (hardware/unclaimedserio.go).54func discoverUnclaimed(catalog *hardware.Catalog) []machine.UnclaimedDevice {55	if catalog == nil {56		return nil57	}58	return catalog.Discover(sysfsRoot, serioAttachments.declaredEntries())59}6061// watchHardware is the machine-plane component that keeps the report62// current. It waits on the kernel's uevent socket, and when the63// hardware changes, it re-walks sysfs, reports the difference to the64// console, and republishes the facts. The uevent only signals that65// something changed; the walk re-reads the whole truth, so a missed66// or coalesced event costs nothing. The watch owns67// hardware/blockDevices/ and hardware/unclaimed/, so it is the only68// writer of those subtrees.69//70// lastDisks is the boot's own disk snapshot, passed in rather than71// read back. The disk inventory has the same failure mode the watch72// exists to prevent: a boot-time snapshot goes stale the moment73// hardware moves. This inventory can even race the boot. A disk behind74// a just-loaded driver (a USB stick binding at boot) can finish its75// SCSI probe after the facts were first published, and the probe's own76// uevents bring the inventory current moments later. The baseline is77// the boot's snapshot, so a disk that appeared between the boot's walk78// and this watch's start still reads as a change worth publishing.79func watchHardware(catalog *hardware.Catalog, tree machine.FactsTree, last []machine.UnclaimedDevice, lastDisks []machine.BlockDevice) func(ctx context.Context) error {80	// walk reads sysfs and publishes what changed since the last walk.81	// The listener opens first and the walk runs right after, so a82	// change made before the listener opened, including one made while83	// a stopped listener was being replaced, still reaches the facts.84	walk := func() {85		devices := hardware.DiscoverDevices(sysfsRoot, catalog.PCI)86		unclaimed := catalog.Unclaimed(devices, serioAttachments.declaredEntries())87		disks := discoverBlockDevices()88		for _, line := range hardwareTransitions(last, unclaimed, devices) {89			fmt.Println(line)90		}91		if !slices.EqualFunc(last, unclaimed, unclaimedEqual) || !slices.EqualFunc(lastDisks, disks, blockDeviceEqual) {92			tree.WriteUnclaimed(unclaimed)93			tree.WriteBlockDevices(disks)94		}95		last, lastDisks = unclaimed, disks96	}97	return func(ctx context.Context) error {98		uevents, err := listenForUevents(ctx)99		if err != nil {100			return err101		}102		for {103			walk()104			select {105			case <-ctx.Done():106				return nil107			case _, ok := <-uevents:108				if !ok {109					return errUeventsStopped110				}111			}112			// One plugged-in device produces a burst of uevents (the lab113			// measured eleven for one USB stick). This code waits for114			// the burst to finish rather than walking once per event.115			hardware.Settle(ctx, uevents, time.Second, 5*time.Second)116		}117	}118}119120// listenForUevents opens the kernel's uevent socket. It is a package121// variable so a test can hand a component a listener that has stopped,122// or one whose wakes the test sends, because a test cannot make the123// kernel stop the real socket or send it a uevent.124var listenForUevents = hardware.ListenForUevents125126// errUeventsStopped ends a component whose uevent listener stopped.127// The machine plane starts the component again, which opens a new128// listener and walks sysfs again before it waits, so a change made129// while nothing listened is still reported.130var errUeventsStopped = errors.New("the uevent listener stopped")131132// hardwareTransitions describes what changed between two walks, in133// the same style as the rest of the boot's console report. An entry134// that appeared is a new gap. An entry that left either got its135// driver (this reports which driver) or was unplugged.136func hardwareTransitions(before, after []machine.UnclaimedDevice, devices []hardware.Device) []string {137	base := moduleBase()138	var lines []string139	for _, u := range after {140		if !slices.ContainsFunc(before, func(b machine.UnclaimedDevice) bool { return b.Modalias == u.Modalias }) {141			lines = append(lines, fmt.Sprintf("liken: hardware: unclaimed %s: %s", describeUnclaimed(u), unclaimedAdvice(base, u)))142		}143	}144	for _, u := range before {145		if slices.ContainsFunc(after, func(a machine.UnclaimedDevice) bool { return a.Modalias == u.Modalias }) {146			continue147		}148		driver := ""149		for _, d := range devices {150			if d.Modalias == u.Modalias && d.Driver != "" {151				driver = d.Driver152			}153		}154		// An entry with no candidates is a serial-line adapter that155		// already had its driver. It leaves the list when a156		// spec.serio entry declares it, not when a driver binds.157		if driver != "" && len(u.Candidates) == 0 {158			lines = append(lines, fmt.Sprintf("liken: hardware: %s is now declared in spec.serio", nameOrModalias(u)))159		} else if driver != "" {160			lines = append(lines, fmt.Sprintf("liken: hardware: %s is now driven by %s", nameOrModalias(u), driver))161		} else {162			lines = append(lines, fmt.Sprintf("liken: hardware: %s was removed", nameOrModalias(u)))163		}164	}165	return lines166}167168// describeUnclaimed renders one entry for a console line: bus and169// class when known, then the best name available.170func describeUnclaimed(u machine.UnclaimedDevice) string {171	description := u.Bus172	if u.Class != "" {173		description += " " + u.Class174	}175	return description + " device " + nameOrModalias(u)176}177178// softdepBase points the soft-dependency reader at a module tree.179// Empty means the running kernel's own tree; tests set it to a180// fixture so their advice does not depend on the host's modules.181var softdepBase = ""182183func moduleBase() string {184	if softdepBase != "" {185		return softdepBase186	}187	return filepath.Join("/lib/modules", kernelRelease())188}189190// unclaimedAdvice states the fix for one unclaimed device, and191// improves the stock advice with the soft dependencies the loader does192// not read. The catalog already named the candidate drivers and said193// to declare them in spec.modules. A candidate can name another194// module to load before it (r8169 names realtek), which modules.dep195// never records, so the plain advice would send a person to declare a196// driver that then binds to the wrong thing. This walks each197// candidate's soft dependency chain and names the full ordered list,198// so the advice reads "declare realtek, then r8169 in spec.modules".199//200// When no candidate gains a soft dependency, the catalog's own message201// stands unchanged. That keeps the wording for the case the catalog202// alone already handles, including the composed image that carries no203// candidate at all, where the fix is a different image and there is204// nothing to expand.205func unclaimedAdvice(base string, u machine.UnclaimedDevice) string {206	expanded := false207	choices := make([]string, 0, len(u.Candidates))208	for _, candidate := range u.Candidates {209		chain := softdepChain(base, candidate)210		if len(chain) == 1 {211			choices = append(choices, candidate)212			continue213		}214		expanded = true215		choices = append(choices, strings.Join(chain[:len(chain)-1], ", ")+", then "+chain[len(chain)-1])216	}217	if !expanded {218		return u.Message219	}220	return "declare " + strings.Join(choices, " or ") + " in spec.modules"221}222223func nameOrModalias(u machine.UnclaimedDevice) string {224	if u.Name != "" {225		return u.Name226	}227	return u.Modalias228}229230func unclaimedEqual(a, b machine.UnclaimedDevice) bool {231	return a.Modalias == b.Modalias && a.Bus == b.Bus && a.Name == b.Name &&232		a.Class == b.Class && a.Message == b.Message &&233		slices.Equal(a.Candidates, b.Candidates)234}235236// blockDeviceEqual compares two disk inventory entries field for237// field. StableNames is a slice, so BlockDevice cannot use slices.Equal238// directly; this function is its stand-in, the same role239// unclaimedEqual plays for UnclaimedDevice above.240func blockDeviceEqual(a, b machine.BlockDevice) bool {241	return a.Name == b.Name && a.SizeBytes == b.SizeBytes &&242		a.Model == b.Model && a.Serial == b.Serial &&243		slices.Equal(a.StableNames, b.StableNames)244}
init/imports.go 83.7%
1package main23// Crash-safe image imports: init's half of the protocol.4//5// The machine package's imports.go explains why this file exists:6// containerd's unpack is not crash-ordered. A machine that dies at7// the wrong moment keeps a store that says "unpacked" over torn8// files, and containerd never re-unpacks a digest that it has a9// record for, so the damage is permanent. This file makes the10// decision that prevents that damage, once per boot, before k3s can11// touch the store: trust the store, discard it, or put new tarballs12// on trial.13//14// The rule rests on one bit of state. A staged imports record stands15// from the moment a trial boots until the operator proves it, so a16// record still standing at the next boot means the previous boot died17// unproven and the store may be lying. The only safe move that18// depends on no other component's internals is to discard the store19// completely.20// Every OS image unpacks fresh from the tarballs that this boot21// carries. Workload images re-pull from their registries (cheaply,22// when the embedded registry shares them between peers). The agent's23// credentials re-mint from the join token. The cost is bounded and24// rare: only a machine that died inside a window of a few minutes,25// and only on boots that had something new to unpack, pays this cost.26//27// Most boots pay only one hash pass: the tarballs match the proven28// record, and the store is trusted.2930import (31	"errors"32	"fmt"33	"io/fs"34	"os"35	"path/filepath"3637	"github.com/liken-sh/liken/liken/machine"38)3940// These are package variables rather than constants, so tests can41// point the settle pass at trees of their own making. The images42// directory is read from clusterState (where seedClusterState43// refreshed it this boot) rather than from the image's baked copy,44// because clusterState mounts over the seed tree, and these are the45// exact bytes that k3s will hand to containerd. Hashing these bytes,46// rather than trusting a build-time claim about them, is also what47// catches a tarball whose own copy tore.48var (49	k3sAgentDir  = machine.K3sAgentDir50	k3sImagesDir = filepath.Join(machine.K3sAgentDir, "images")51)5253// settleImageImports determines whether this boot's container store54// can be trusted, before k3s ever starts. It runs only when both sides55// of the question are durable. Without machineState, there is nowhere56// to remember a trial. Without durable clusterState, the store resets57// with every boot and cannot get stuck in the first place.58func settleImageImports(stateRoot string, durable, clusterDurable bool, boot *machine.BootStatus) {59	if !durable || !clusterDurable {60		return61	}6263	digests, err := machine.HashImageTarballs(k3sImagesDir)64	if err != nil {65		fmt.Fprintf(os.Stderr, "liken: imports: hashing the image tarballs: %v\n", err)66		return67	}68	raw, hash, err := machine.RenderImportedImages(digests)69	if err != nil {70		fmt.Fprintf(os.Stderr, "liken: imports: rendering the record: %v\n", err)71		return72	}73	store := machine.ImportedImagesStore(stateRoot)7475	// A staged record's existence is the whole question. Its content76	// does not matter (even an unreadable record marks a dead trial),77	// so there is nothing to parse and nothing to reject. The78	// fallback is not an older document; it is a clean store.79	// Discarding is safe to interrupt for the same reason: the record80	// stays standing until the operator promotes it, so a partial81	// discard simply runs again on the next boot.82	staged, err := store.LoadStaged()83	if err != nil {84		fmt.Fprintf(os.Stderr, "liken: imports: reading the staged record: %v\n", err)85	}86	if staged != nil || err != nil {87		discardContainerStore()88		boot.ImportsDiscarded = true89		fmt.Println("liken: imports: the previous boot's imports were never proven; discarding the container store (OS images re-unpack from this boot's tarballs, workloads re-pull)")90	} else if proven, perr := store.LoadProven(); perr == nil && proven != nil && machine.ManifestHash(proven) == hash {91		// The quiet path, which covers almost every boot: the same92		// tarballs that this store already proved it can serve.93		boot.ImportsSource = machine.ManifestSourceProven94		boot.ImportsHash = hash95		fmt.Printf("liken: imports: %d image tarballs proven (%.12s)\n", len(digests), hash)96		return97	}9899	// A trial: new digests, a first boot, or the retry after a100	// discard. This call stages the record durably before k3s can101	// touch the store, so a death anywhere after this line reads102	// correctly on the next boot.103	if err := store.WriteStaged(raw); err != nil {104		fmt.Fprintf(os.Stderr, "liken: imports: staging the record: %v\n", err)105		return106	}107	boot.ImportsSource = machine.ManifestSourceStaged108	boot.ImportsHash = hash109	fmt.Printf("liken: imports: trialing %d image tarballs (%.12s); the operator proves them once they serve\n", len(digests), hash)110}111112// discardContainerStore empties the k3s agent directory, and spares113// only the images/ tarballs that this boot just seeded (k3s is about114// to import them, and they came from the image, not from the115// distrusted store). Everything else under agent/ is derived state116// that k3s re-creates from the join token and the cluster: the117// containerd store, the kubelet's credentials, and its caches. The118// same crash window can tear the kubelet's credentials too: a zeroed119// serving key stops the agent just as surely as a torn snapshot.120func discardContainerStore() {121	entries, err := os.ReadDir(k3sAgentDir)122	if errors.Is(err, fs.ErrNotExist) {123		return124	}125	if err != nil {126		fmt.Fprintf(os.Stderr, "liken: imports: reading %s: %v\n", k3sAgentDir, err)127		return128	}129	for _, entry := range entries {130		if entry.Name() == "images" {131			continue132		}133		if err := os.RemoveAll(filepath.Join(k3sAgentDir, entry.Name())); err != nil {134			fmt.Fprintf(os.Stderr, "liken: imports: discarding %s: %v\n", entry.Name(), err)135		}136	}137}
init/install.go 80.4%
1package main23// The installer: the USB-stick boot that puts liken on its own disk.4//5// A machine booted with liken.install on its command line is running6// from external media: QEMU's -kernel flag in the lab, or an7// installer stick or PXE on real hardware. Its job this boot is not8// to serve a cluster but to make external media unnecessary. It9// verifies the release payload it carries, copies it into system10// slot A, registers both slots with whatever governs booting11// (firmware boot entries on a UEFI machine, liken's own GRUB on a12// machine that declares the biosBoot and bootHome roles), and powers13// off. Every boot after this one comes from the disk.14//15// The installer is liken itself, not a separate program. It uses the16// same init, the same storage reconciliation (which claimed and17// formatted the slots moments earlier, on a fresh machine), and the18// same manifest selection. Two things differ: the destination, and19// the ending. An install boot ends in a power-off because the20// install medium must not boot again. With QEMU's -kernel flag21// present, a reboot would just run the installer again.22//23// Idempotence is what makes a crash safe. A power cut in the middle24// of an install leaves claimed slots (claiming is resumable by25// name), half-copied files that fail verification and are copied26// again, and boot entries that are found by description and27// rewritten in place. Running the installer twice converges to the28// same result; there is no state to clean up first.2930import (31	"fmt"32	"os"33	"path/filepath"34	"strconv"35	"strings"3637	"golang.org/x/sys/unix"3839	"github.com/liken-sh/liken/liken/disks"40	"github.com/liken-sh/liken/liken/machine"41)4243// releasePayloadDir is where the install image carries the release44// it installs: the artifacts that the release document lists, byte45// for byte, plus the deployment layer and its sidecar. image/media.go46// assembles this as a wrapper cpio archive, concatenated onto the47// composed system. The kernel unpacks concatenated archives in48// order, the same mechanism that early microcode updates use. This49// is a variable, so tests can supply a payload of their own.50var releasePayloadDir = "/usr/share/liken/release"5152// installSlot is the slot a fresh install lands on. An install always53// writes slot A and registers slot B empty, because the blue-green pair54// fills its second slot only at the first upgrade. installToDisk returns55// this letter so an attended install can name the slot in the message a56// person reads before the machine powers off.57const installSlot = "A"5859// installToDisk performs the whole install against the slots that60// storage reconciliation just mounted. It returns rather than powers61// off, so main can apply boot policy. On success it returns the slot62// letter it installed to, and the machine has nothing left to do but63// stop.64func installToDisk(machineName string) (string, error) {65	if machineName == "" {66		return "", fmt.Errorf("install: this machine has no name (liken.machine= or a manifest must supply one); boot entries carry identity, so an anonymous install would be wrong on every later boot")67	}6869	// The slots must both exist before anything is copied. The design70	// depends on a fallback slot being registered from the start, so71	// an install that can only register one slot must not proceed.72	parts := discoverPartitions()73	slotA, err := findSlotPartition(parts, machine.SystemARole)74	if err != nil {75		return "", err76	}77	slotB, err := findSlotPartition(parts, machine.SystemBRole)78	if err != nil {79		return "", err80	}8182	// This code verifies the payload before it copies a single byte.83	// The release document names each artifact's digest, and the84	// copies that this image carries must match it exactly.85	raw, err := os.ReadFile(filepath.Join(releasePayloadDir, "release.yaml"))86	if err != nil {87		return "", fmt.Errorf("install: reading the release document: %w", err)88	}89	release, err := machine.ParseRelease(raw)90	if err != nil {91		return "", fmt.Errorf("install: %w", err)92	}93	fmt.Printf("liken: install: release %s, %d artifacts\n", release.Metadata.Name, len(release.Artifacts))9495	slotMount := roleMounts[machine.SystemARole].path96	for _, artifact := range release.Artifacts {97		source := filepath.Join(releasePayloadDir, artifact.Name)98		if err := verifyFile(artifact, source); err != nil {99			return "", fmt.Errorf("install: the payload this image carries doesn't match its own release document: %w", err)100		}101		dest := filepath.Join(slotMount, artifact.Name)102		if err := copyDurably(source, dest); err != nil {103			return "", fmt.Errorf("install: copying %s: %w", artifact.Name, err)104		}105		// This verifies what was written, not what was meant to be106		// written. The copy is re-read from the slot and hashed107		// again, so a torn or corrupted write is caught now rather108		// than on a later boot.109		if err := verifyFile(artifact, dest); err != nil {110			return "", fmt.Errorf("install: the copy on the slot doesn't verify: %w", err)111		}112		fmt.Printf("liken: install: %s verified and installed (%d bytes)\n", artifact.Name, artifact.Size)113	}114115	// The deployment layer travels beside the listed artifacts, not116	// among them. The release document is the public one, and the117	// layer belongs to this deployment alone, vouched for by its118	// sidecar. The same discipline applies: verify the payload's119	// copy, copy it durably, and verify what landed. The sidecar is120	// written last, so a slot with a sidecar is a slot whose layer121	// was complete when it was written.122	if err := installLayer(slotMount); err != nil {123		return "", err124	}125126	// The release document lands after the artifacts it describes, so127	// that the slot records which release it holds without asking the128	// network. The fetcher writes the same file for the same reason129	// when it fills the other slot. A slot that cannot name its own130	// release is a slot nothing can later check, and a later boot has131	// to take its contents on trust (fatstate.go).132	if err := copyDurably(filepath.Join(releasePayloadDir, "release.yaml"),133		filepath.Join(slotMount, "release.yaml")); err != nil {134		return "", fmt.Errorf("install: writing the release document to the slot: %w", err)135	}136137	// The actuator half: register the slots with whatever will hold138	// this machine's boot choices. A UEFI machine gets firmware boot139	// entries. A machine whose spec declares the GRUB roles gets its140	// own bootloader written. Both at once is valid: a disk prepared141	// under UEFI can carry its GRUB for a BIOS life, and the lab's142	// dual-firmware disks do exactly that. But an install that no143	// firmware could ever boot is refused, before the power-off makes144	// it permanent.145	actuators := 0146	if firmwareIsUEFI() {147		// The fallback comes first, because it is the half that still148		// works when the firmware holds no variables at all149		// (slotloader.go).150		if err := writeSlotLoader(slotMount, installSlot, machineName); err != nil {151			return "", fmt.Errorf("install: %w", err)152		}153		if err := registerSlotEntries(slotA, slotB, machineName); err != nil {154			return "", fmt.Errorf("install: %w", err)155		}156		actuators++157	}158	if hasPartition(parts, machine.BIOSBootRole) {159		if err := installGRUB(parts, machineName, slotMount); err != nil {160			return "", err161		}162		actuators++163	}164	if actuators == 0 {165		return "", fmt.Errorf("install: this machine's firmware holds no boot variables (BIOS) and its spec declares no biosBoot/bootHome roles for GRUB; there is nothing that could boot the installed disk")166	}167168	// Everything must reach the disk before the caller announces169	// success, because an attended install holds at the console after170	// that announcement and a person may cut power instead of pressing171	// Enter. The per-file fsync in copyDurably makes each file's data172	// durable, but not the rename that gave the file its final name:173	// on FAT, that directory update sits in the page cache until174	// something flushes it. The power-off after the hold would flush175	// it, but the success message must be true at the moment it176	// prints, not at the moment the machine obeys it.177	//178	// Two calls, because they do different halves of the job. unix.Sync179	// walks every mounted filesystem and writes its dirty pages back to180	// the drivers. It stops there: it asks no drive to empty its own181	// write cache. syncDirectory does that for the slot.182	unix.Sync()183	if err := syncDirectory(slotMount); err != nil {184		return "", fmt.Errorf("install: flushing the slot's directory to the disk: %w", err)185	}186	return installSlot, nil187}188189// installGRUB writes the bootloader for a machine that boots190// BIOS-style: the patched boot chain into the MBR and the biosBoot191// partition (grubinstall.go handles the arithmetic), and the config192// and environment block onto the boot home. The GRUB artifacts come193// from the slot that the release was just verified onto. They194// arrived through the same verify-copy-verify pipeline as everything195// else. Like the rest of the installer, this converges on re-run:196// every write puts down the same bytes.197func installGRUB(parts []partition, machineName, slotMount string) error {198	biosBoot, err := findSlotPartition(parts, machine.BIOSBootRole)199	if err != nil {200		return err201	}202	diskDev, err := diskDevice(parts, machine.BIOSBootRole)203	if err != nil {204		return err205	}206207	bootImg, err := os.ReadFile(filepath.Join(slotMount, "grub-boot.img"))208	if err != nil {209		return fmt.Errorf("install: this release carries no grub-boot.img, so it cannot boot a BIOS machine: %w", err)210	}211	coreImg, err := os.ReadFile(filepath.Join(slotMount, "grub-core.img"))212	if err != nil {213		return fmt.Errorf("install: this release carries no grub-core.img, so it cannot boot a BIOS machine: %w", err)214	}215	plan, err := planGRUBBootSectors(bootImg, coreImg, biosBoot)216	if err != nil {217		return fmt.Errorf("install: %w", err)218	}219	disk, err := os.OpenFile(diskDev, os.O_RDWR, 0)220	if err != nil {221		return err222	}223	writeErr := plan.write(disk)224	if err := disk.Close(); writeErr == nil {225		writeErr = err226	}227	if writeErr != nil {228		return fmt.Errorf("install: writing the boot sectors on %s: %w", diskDev, writeErr)229	}230231	// The boot home: the rendered config and a fresh environment232	// block that name slot A as the default. Slot A is where this233	// install just put the release.234	home := roleMounts[machine.BootHomeRole].path235	grubDir := filepath.Join(home, "grub")236	if err := os.MkdirAll(grubDir, 0o755); err != nil {237		return fmt.Errorf("install: %w", err)238	}239	cfg := renderGRUBConfig(machineName, consoleArgs())240	if err := writeFileDurably(filepath.Join(grubDir, "grub.cfg"), []byte(cfg)); err != nil {241		return fmt.Errorf("install: writing grub.cfg: %w", err)242	}243	env, err := renderGRUBEnv(map[string]string{"default_slot": installSlot, "try_slot": ""})244	if err != nil {245		return err246	}247	if err := writeFileDurably(filepath.Join(grubDir, "grubenv"), env); err != nil {248		return fmt.Errorf("install: writing grubenv: %w", err)249	}250	// Both files landed through a rename, and a rename is a directory251	// update. This flush carries those updates, and the drive's write252	// cache with them, out to the medium. A BIOS machine whose boot home253	// lost its grub.cfg to a power cut has no bootloader configuration254	// at all, and GRUB stops at a rescue prompt.255	if err := syncDirectory(grubDir); err != nil {256		return fmt.Errorf("install: flushing the boot home to the disk: %w", err)257	}258259	fmt.Printf("liken: install: GRUB installed on %s; grub.cfg and grubenv on the boot home prefer slot A\n", diskDev)260	return nil261}262263// diskDevice names the whole-disk device that a role's partition264// lives on. Boot-sector writes address the disk, not the partition,265// because the MBR belongs to no partition at all.266func diskDevice(parts []partition, role machine.StorageRoleName) (string, error) {267	name := machine.DeclaredRole{Name: role}.PartitionName()268	for _, p := range parts {269		if p.partName == name {270			return devRoot + "/" + p.disk, nil271		}272	}273	return "", fmt.Errorf("no partition carries the name %q, so its disk cannot be found", name)274}275276// hasPartition reports whether a role's partition exists on this277// machine. For the installer, this is the sign that the machine's278// spec declared the role, since storage reconciliation claimed the279// partitions moments before the install began.280func hasPartition(parts []partition, role machine.StorageRoleName) bool {281	name := machine.DeclaredRole{Name: role}.PartitionName()282	for _, p := range parts {283		if p.partName == name {284			return true285		}286	}287	return false288}289290// installLayer copies the deployment layer and its sidecar from the291// payload to the slot. The sidecar is the layer's trust root. The292// release document cannot name the layer, because the document is293// public and the layer belongs to one deployment alone. Media that294// carries a layer without its sidecar, or a layer that its sidecar295// rejects, is incomplete, and the install refuses rather than install296// something that no later boot could verify.297func installLayer(slotMount string) error {298	sidecar, err := os.ReadFile(filepath.Join(releasePayloadDir, machine.LayerSidecarName))299	if err != nil {300		return fmt.Errorf("install: reading the layer's sidecar: %w", err)301	}302	digest, err := machine.ParseLayerSidecar(sidecar)303	if err != nil {304		return fmt.Errorf("install: %w", err)305	}306307	verify := func(path string) error {308		f, err := os.Open(path)309		if err != nil {310			return err311		}312		defer f.Close()313		return machine.VerifyLayer(digest, f)314	}315316	source := filepath.Join(releasePayloadDir, machine.LayerName)317	if err := verify(source); err != nil {318		return fmt.Errorf("install: the layer this image carries doesn't match its sidecar: %w", err)319	}320	dest := filepath.Join(slotMount, machine.LayerName)321	if err := copyDurably(source, dest); err != nil {322		return fmt.Errorf("install: copying %s: %w", machine.LayerName, err)323	}324	if err := verify(dest); err != nil {325		return fmt.Errorf("install: the layer on the slot doesn't verify: %w", err)326	}327	if err := copyDurably(filepath.Join(releasePayloadDir, machine.LayerSidecarName),328		filepath.Join(slotMount, machine.LayerSidecarName)); err != nil {329		return fmt.Errorf("install: copying %s: %w", machine.LayerSidecarName, err)330	}331	fmt.Printf("liken: install: %s verified and installed against its sidecar\n", machine.LayerName)332	return nil333}334335// findSlotPartition locates one slot by the name written on it at336// claim time, and reads its GPT identity: the unique GUID that a337// boot entry uses to pin the partition regardless of device338// position.339func findSlotPartition(parts []partition, role machine.StorageRoleName) (*slotPartition, error) {340	declared := machine.DeclaredRole{Name: role}341	for _, p := range parts {342		if p.partName != declared.PartitionName() {343			continue344		}345		device := devRoot + "/" + p.disk346		disk := diskByPath(device)347		if disk == nil {348			return nil, fmt.Errorf("install: %s found on %s but the disk is not in the inventory", role, device)349		}350		f, err := os.Open(device)351		if err != nil {352			return nil, err353		}354		table, err := disks.ReadGPT(f, disk.SizeBytes/disks.SectorSize)355		f.Close()356		if err != nil {357			return nil, fmt.Errorf("install: reading %s's partition table: %w", device, err)358		}359		for _, entry := range table.Entries {360			if entry.Name == declared.PartitionName() {361				number, err := partitionNumber(p)362				if err != nil {363					return nil, err364				}365				return &slotPartition{366					number:   number,367					firstLBA: entry.FirstLBA,368					lastLBA:  entry.LastLBA,369					guid:     entry.UniqueGUID,370				}, nil371			}372		}373		return nil, fmt.Errorf("install: %s appears in sysfs but not in %s's table; refusing to guess", role, device)374	}375	return nil, fmt.Errorf("install: no partition carries %s; the manifest must declare both system slots before a machine can install itself", declared.PartitionName())376}377378type slotPartition struct {379	number   uint32380	firstLBA uint64381	lastLBA  uint64382	guid     [16]byte383}384385// partitionNumber recovers the partition's index from the kernel's386// node name: the suffix after the disk's name (vdc1 means 1), with387// the "p" separator that NVMe-style names insert (nvme0n1p2 means 2).388// The kernel numbers partitions by their position in the GPT entry389// array, starting at 1. This function refuses a suffix that is not a390// number. The index goes into a firmware boot entry, and 0 is not a391// valid GPT slot, so a malformed name must stop the install rather392// than encode garbage that the firmware would trust.393func partitionNumber(p partition) (uint32, error) {394	suffix := strings.TrimPrefix(p.name, p.disk)395	suffix = strings.TrimPrefix(suffix, "p")396	n, err := strconv.Atoi(suffix)397	if err != nil || n < 1 {398		return 0, fmt.Errorf("install: cannot read a partition number from %q on disk %s", p.name, p.disk)399	}400	return uint32(n), nil401}
init/k3s.go 87.8%
1package main23// From one machine to a cluster: deciding what this machine is, and4// telling k3s.5//6// k3s draws the same line that liken does: leaders run a control7// plane, and followers run workloads and take direction (k3s's names8// for the same roles are "server" and "agent", and those words9// appear here only where k3s's own files and flags require them).10// Which role this machine should have is not the machine's own11// business to declare. The Cluster manifest names the leaders, and a12// machine derives its role by looking for its own name in that list.13// Everything role-specific about starting k3s follows from that one14// derivation.15//16// k3s is configured by file, not by flags (the supervisor's empty17// argument lists are deliberate), and its config loader has a18// feature built for exactly liken's situation: alongside a config19// file, k3s reads every *.yaml in a sibling <name>.yaml.d/ directory20// as drop-ins. So the split is this: what a person decided lives in21// the image's static files (/etc/rancher/k3s/config.yaml for22// leaders, agent.yaml for followers, both reviewable in the repo),23// and what only the boot can observe (this machine's node IP, the24// cluster's address plan, where the join token sits) lands in a25// drop-in that this file writes. Init never rewrites a file that a26// person wrote.2728import (29	"crypto/rand"30	"encoding/hex"31	"fmt"32	"maps"33	"net"34	"os"35	"path/filepath"36	"slices"37	"strings"3839	"golang.org/x/sys/unix"4041	"github.com/liken-sh/liken/liken/api"42	"github.com/liken-sh/liken/liken/cluster"43	"github.com/liken-sh/liken/liken/machine"44)4546// These are package variables rather than constants, so tests can47// point the derivations at files of their own making.48var (49	// The static halves, shipped in the image.50	k3sServerConfig = "/etc/rancher/k3s/config.yaml"51	k3sAgentConfig  = "/etc/rancher/k3s/agent.yaml"5253	// k3sKubeletConfig is where init writes the kubelet's own54	// configuration file, when the cluster names a setting that only55	// that file carries. It sits beside k3s's two config files rather56	// than in a drop-in directory, because k3s does not treat it as57	// k3s configuration: the drop-in only names its path.58	k3sKubeletConfig = "/etc/rancher/k3s/kubelet.yaml"5960	// k3sKubeletConfigCopy is the copy of that file that k3s makes61	// when it starts with the config argument. k3s hands the kubelet62	// one drop-in directory holding its own defaults and this copy,63	// and it refreshes the copy only when the argument is present. The64	// directory sits on clusterState, so a copy that nothing removes65	// survives every restart and reboot, and the kubelet would run a66	// retracted policy forever. The boot that stops naming the file67	// removes the copy with it.68	k3sKubeletConfigCopy = "/var/lib/rancher/k3s/agent/etc/kubelet.conf.d/10-cli-config.conf"6970	// k3sContainerdDropIn is where init writes containerd's log level,71	// when the cluster names one.72	//73	// containerd reads its level from its own configuration file, and74	// init cannot write that file: k3s renders it from a template each75	// time it starts containerd, so anything init wrote there would76	// last until the next start. But the file k3s renders imports77	// config-v3.toml.d/*.toml from the directory beside it, and78	// containerd merges each imported file over what it has already79	// read. So a drop-in in that directory sets the level, and k3s80	// keeps rendering its own configuration untouched.81	//82	// liken writes a drop-in instead of the other supported path, a83	// config-v3.toml.tmpl that overrides k3s's template. An override84	// is a Go template that must still parse and still render every85	// key containerd needs, so an override that breaks stops86	// containerd on a machine that serves no shell to repair it. A87	// drop-in carries only the level.88	// If a future k3s stops importing the directory, the level returns89	// to containerd's own default and containerd still starts. The90	// drop-in also depends on no template name, and it leaves the91	// override path free for the operator, who has only that one path92	// and would otherwise have to merge liken's needs into their own93	// template.94	//95	// The name carries containerd's configuration version, because96	// that is the directory k3s imports: containerd 2.x reads version97	// 3, and k3s renders config-v3.toml for it. The number orders the98	// file within the directory, which containerd reads in glob order.99	//100	// The directory sits on clusterState, so a drop-in that nothing101	// removes survives every restart and reboot. The boot that stops102	// naming a level removes it.103	k3sContainerdDropIn = "/var/lib/rancher/k3s/agent/etc/containerd/config-v3.toml.d/10-liken-log-level.toml"104105	// tokenPath is where the image carries the cluster's join token,106	// minted offline like the CAs it hashes (see the identity107	// package). This code hands the token to k3s as a token-file, so108	// the secret itself never appears in a config file or on a109	// command line.110	tokenPath = "/etc/liken/token"111112	// seedSourceDir is where the image bakes k3s's seed files, the113	// tree that seedClusterState copies onto the clusterState114	// filesystem.115	seedSourceDir = "/var/lib/rancher"116)117118// k3s's auto-deploy directory and liken's own subdirectory of it,119// both spelled relative to the root that holds them: the image's120// /var/lib/rancher on one side, the clusterState filesystem on the121// other. features.go builds the absolute path from the second of122// these, and explains why liken keeps to a subdirectory.123const (124	k3sManifestsRel   = "k3s/server/manifests"125	likenManifestsRel = k3sManifestsRel + "/liken"126)127128// leaderJoinConfig selects a leader's datastore keys, based on leader129// count. One leader is exactly the cluster that liken has always130// run: sqlite (via kine), no etcd, nothing to join. Keeping131// single-node cheap is deliberate. More than one leader means132// embedded etcd, and the first entry in spec.leaders is the founding133// leader. The founding leader renders cluster-init: true, which on134// the migration boot tells k3s to move the existing sqlite datastore135// into etcd in place (the documented path that made starting on136// sqlite safe rather than a dead end). Every other leader points137// server: at the founder, using the founder's declared address on138// the node network, or the endpoint when the founder declares no139// address, and joins. Rejoins keep the same flags every boot, which140// is k3s's recommended steady state.141//142// The founding leader matters only for deriving configuration: it is143// the first name in a list, nothing more. etcd's raft leader is144// elected and moves between members. The founder holds no ongoing145// special position, and once the cluster is up, it is an ordinary146// member among an odd number of them.147//148// An adopted cluster (spec.origin: Adopted) changes one assumption:149// the datastore already exists, in a cluster that liken did not150// create, and initializing a second one next to it would split the151// cluster in two. So under adoption, every leader joins: the founder152// through the endpoint (the one address that the existing control153// plane is known to be reachable at), and the others prefer the154// founder as usual. No leader renders cluster-init or falls back to155// sqlite, not even a lone leader. Each joining leader becomes an etcd156// member, and raft replicates the existing keyspace to it, so the157// cluster's state carries over without ever being exported or158// copied.159func leaderJoinConfig(clusterDoc *cluster.Cluster, name, manifestDir string) (clusterInit bool, joinURL string) {160	if clusterDoc == nil {161		return false, ""162	}163	if clusterDoc.Spec.Origin == cluster.OriginAdopted {164		if leaders := clusterDoc.Spec.Leaders; len(leaders) > 0 && name != leaders[0] {165			if addr := declaredNodeAddress(clusterDoc, manifestDir, leaders[0]); addr != "" {166				return false, fmt.Sprintf("https://%s:6443", addr)167			}168		}169		return false, clusterDoc.Spec.Endpoint170	}171	if len(clusterDoc.Spec.Leaders) < 2 {172		return false, ""173	}174	founder := clusterDoc.Spec.Leaders[0]175	if name == founder {176		return true, ""177	}178	if addr := declaredNodeAddress(clusterDoc, manifestDir, founder); addr != "" {179		return false, fmt.Sprintf("https://%s:6443", addr)180	}181	return false, clusterDoc.Spec.Endpoint182}183184// nodeAddress picks which of the machine's addresses is its node IP:185// the address that Kubernetes traffic uses, and the one that other186// nodes are told to reach it at. The Cluster's nodeCIDR determines187// this. The interface whose address falls inside nodeCIDR is the188// cluster-facing one. A machine with several interfaces needs this189// choice to be explicit. Left unconfigured, k3s selects the190// interface that holds the default route, which on a machine with an191// internet uplink is exactly the wrong interface.192func nodeAddress(clusterDoc *cluster.Cluster, conns []*connection) (ip, ifname string) {193	if !declaresNodeCIDR(clusterDoc) {194		return "", ""195	}196	_, subnet, err := net.ParseCIDR(clusterDoc.Spec.Network.NodeCIDR)197	if err != nil {198		fmt.Fprintf(os.Stderr, "liken: cluster nodeCIDR %q: %v\n", clusterDoc.Spec.Network.NodeCIDR, err)199		return "", ""200	}201	for _, conn := range conns {202		if conn.addr != nil && subnet.Contains(conn.addr.IP) {203			return conn.addr.IP.String(), conn.ifname204		}205	}206	return "", ""207}208209// declaresNodeCIDR reports whether the cluster names the subnet a node210// address must fall inside. The pass split asks it separately from211// nodeAddress because the two absences mean different things: no212// nodeCIDR means there is no address rule to satisfy, while a213// declared nodeCIDR that no interface matches means the rule exists214// and is not met yet.215func declaresNodeCIDR(clusterDoc *cluster.Cluster) bool {216	return clusterDoc != nil && clusterDoc.Spec.Network.NodeCIDR != ""217}218219// k3sBootInputs gathers everything that the drop-in needs and that220// only this boot could determine. writeK3sBootConfig fills it in, and221// k3sBootConfig renders it. This is a struct rather than a parameter222// list, because ten positional arguments with adjacent strings and223// bools invite mistakes. Named fields read correctly at the call224// site.225type k3sBootInputs struct {226	role          api.Role227	clusterDoc    *cluster.Cluster228	nodeIP        string229	nodeInterface string230	haveToken     bool231	clusterInit   bool232	joinURL       string233	nodeLabels    map[string]string234	nodeTaints    []machine.NodeTaint235	kubeletConfig string236	debug         bool237}238239// k3sBootConfig renders the drop-in: everything that k3s must be240// told and that only this boot could determine. It renders plain241// key: value lines, because every value here is a string that k3s242// maps onto one of its flags.243func k3sBootConfig(in k3sBootInputs) string {244	role, clusterDoc := in.role, in.clusterDoc245	var b strings.Builder246	b.WriteString("# Written by liken at boot: the configuration only the boot can\n")247	b.WriteString("# derive, joined with the static file this directory sits beside.\n")248249	// The join token applies to both roles: the leader requires250	// exactly this token from anyone joining, and a follower presents251	// it. Because the token embeds a hash of the cluster CA, a252	// follower also uses it to verify that it joins the cluster its253	// configuration names.254	if in.haveToken {255		fmt.Fprintf(&b, "token-file: %s\n", tokenPath)256	}257258	if role == api.RoleFollower {259		// "server" is k3s's config key for "the control plane I take260		// direction from": a follower points at the endpoint.261		fmt.Fprintf(&b, "server: %s\n", clusterDoc.Spec.Endpoint)262	} else {263		// The disable list: which of k3s's bundled components264		// (Traefik, the service load balancer, metrics-server) stay265		// off. liken disables them on principle: anything beyond the266		// control plane should be a declared, visible workload. The267		// Cluster's spec.features is that declaration, so this code268		// computes the list as the bundled set minus the cluster's269		// opt-ins (DisabledComponents, in the cluster package). This270		// renders on leaders only, because disable is a server-side271		// key that an agent would refuse. It always renders as the272		// complete list, never a fragment merged with a default273		// somewhere else, so the value has exactly one author. A274		// machine with no cluster document disables everything275		// bundled: the minimum viable cluster is the default, and276		// features are always an opt-in.277		if disabled := clusterDoc.DisabledComponents(); len(disabled) > 0 {278			b.WriteString("disable:\n")279			for _, name := range disabled {280				fmt.Fprintf(&b, "  - %s\n", name)281			}282		}283		// The Helm controller is not on the disable list, because it284		// is not a deployable component. It is a controller compiled285		// into the k3s server process. It watches for HelmChart286		// resources to render, and it holds informer caches whether287		// or not any exist. On a small machine, this memory use is288		// worth naming, so this controller follows the same rule as289		// the bundled components: off unless the cluster declares the290		// helm feature, or a feature that requires it, the way291		// traefik does, since k3s deploys Traefik through a292		// HelmChart.293		if !clusterDoc.FeatureEnabled("helm") {294			b.WriteString("disable-helm-controller: true\n")295		}296		// The embedded cloud controller manager is in the same297		// position: a controller inside the k3s server process,298		// spending memory and holding a leader-election lease on299		// every leader. On real clouds, an external provider replaces300		// it. On bare metal, its only real job here is running the301		// service load balancer (klipper-lb lives inside it, since302		// k3s moved ServiceLB there), so it runs exactly when303		// servicelb is declared. Without it, the kubelet initializes304		// the node itself, addresses included, which is the ordinary305		// arrangement for a machine that runs in no cloud.306		if !clusterDoc.FeatureEnabled("servicelb") {307			b.WriteString("disable-cloud-controller: true\n")308		}309		// The network policy controller is likewise embedded. It310		// turns NetworkPolicy resources into per-node packet311		// filtering, which the flannel CNI cannot do alone. A cluster312		// that requires that enforcement declares it. On a cluster that313		// does not, the controller would spend its memory watching314		// for resources that never come. Without this controller,315		// Kubernetes accepts NetworkPolicy documents and enforces316		// nothing, which matches flannel's own behavior. This is why317		// the feature's absence is a safe default rather than a318		// broken one.319		if !clusterDoc.FeatureEnabled("network-policy") {320			b.WriteString("disable-network-policy: true\n")321		}322		if clusterDoc != nil {323			// The datastore keys, selected by leaderJoinConfig: the324			// founding leader of a multi-leader cluster runs embedded325			// etcd (and creates it, on the migration boot), and the326			// other leaders join it. A single leader renders neither327			// key and stays on sqlite.328			if in.clusterInit {329				b.WriteString("cluster-init: true\n")330			} else if in.joinURL != "" {331				fmt.Fprintf(&b, "server: %s\n", in.joinURL)332			}333			// The embedded registry mirror (Spegel) is a server-side334			// key. The control plane runs the coordination, and every335			// node participates through the mirror entries that init336			// renders into registries.yaml (registries.go).337			if clusterDoc.Spec.Registries.Embedded {338				b.WriteString("embedded-registry: true\n")339			}340			// The cluster's address plan is leader configuration.341			// Followers receive it from the control plane that they342			// join.343			net := clusterDoc.Spec.Network344			for _, entry := range []struct{ key, value string }{345				{"cluster-cidr", net.ClusterCIDR},346				{"service-cidr", net.ServiceCIDR},347				{"cluster-dns", net.ClusterDNS},348				{"cluster-domain", net.ClusterDomain},349			} {350				if entry.value != "" {351					fmt.Fprintf(&b, "%s: %s\n", entry.key, entry.value)352				}353			}354		}355	}356357	// The node IP and the interface it lives on, when the Cluster's358	// nodeCIDR identified one. node-ip is what the kubelet359	// advertises. flannel-iface is which interface carries pod-to-pod360	// traffic. They must agree, and they must both point at the361	// cluster segment.362	if in.nodeIP != "" {363		fmt.Fprintf(&b, "node-ip: %s\n", in.nodeIP)364		fmt.Fprintf(&b, "flannel-iface: %s\n", in.nodeInterface)365	}366367	// Which of this machine's addresses a NodePort answers on. Both368	// roles render it, because a NodePort is opened by kube-proxy on369	// every node that runs one, leader and follower alike.370	//371	// The value is computed rather than written in the static file,372	// even though it is usually the same "primary" on every machine373	// in every cluster. A setting with two authors, a default in one374	// file and an override in another, cannot be read off either375	// one. Rendering the whole list here means the drop-in states376	// what kube-proxy will be told, and the console line below377	// prints it.378	//379	// kube-proxy takes this as a command-line flag, which k3s380	// exposes as kube-proxy-arg. A plain kube-proxy-arg key in a381	// drop-in replaces the list from the file it sits beside, which382	// is what makes one author possible here: nothing needs the +383	// suffix that appends.384	fmt.Fprintf(&b, "kube-proxy-arg:\n  - nodeport-addresses=%s\n",385		strings.Join(clusterDoc.NodePortAddresses(), ","))386387	// The kubelet's own configuration file, when the cluster named a388	// setting that goes in one (kubeletBootConfig below). Both roles389	// render this key, because the kubelet runs on every node, leader390	// and follower alike.391	//392	// kubelet-arg is k3s's pass-through to the kubelet's command line,393	// and it follows the same replace-or-append rule as kube-proxy-arg394	// above: a plain key in a drop-in replaces the list from the file395	// it sits beside. The static files name no kubelet-arg, so this396	// drop-in is the whole list, and nothing needs the + suffix.397	if in.kubeletConfig != "" {398		fmt.Fprintf(&b, "kubelet-arg:\n  - config=%s\n", in.kubeletConfig)399	}400401	// k3s's own log level. debug is k3s's configuration key for the402	// --debug flag, and it raises every Kubernetes component compiled403	// into the process. Both roles render it, because a follower runs404	// a kubelet and a kube-proxy that are as loud as a leader's. False405	// renders nothing, so a cluster that leaves the field unset gets406	// the same bytes it got before the field existed, and does not407	// restart k3s on the upgrade that added it.408	if in.debug {409		b.WriteString("debug: true\n")410	}411412	// The spec's node labels, so the node registers with its413	// scheduling identity already set. A freshly reinstalled machine414	// must not spend its first minutes as a blank node that workloads415	// wrongly select against. The + suffix is k3s's append syntax for416	// list values. A plain node-label key in a drop-in would replace417	// the static file's list, erasing liken.sh/machine=true, and418	// appending is the whole point of a drop-in. This code sorts the419	// labels, so the same spec always renders the same bytes.420	if len(in.nodeLabels) > 0 {421		b.WriteString("node-label+:\n")422		for _, name := range slices.Sorted(maps.Keys(in.nodeLabels)) {423			fmt.Fprintf(&b, "  - %s=%s\n", name, in.nodeLabels[name])424		}425	}426427	// The spec's node taints, so a node that registers for the first428	// time is born repelling. An untainted first minute accepts429	// exactly the pods that the taint exists to keep out, and those430	// pods are running workloads that a later taint would have to431	// evict. node-taint is k3s's key for the kubelet's432	// registerWithTaints setting, which k3s writes into the kubelet433	// configuration it generates rather than onto the kubelet's434	// command line, and the kubelet reads that setting only when it435	// creates the Node object. This block therefore does436	// its work on a first boot or after a reinstall. On every later437	// boot the Node object already exists, this block changes nothing,438	// and the operator's live reconciliation is the only mechanism439	// left. The + suffix appends, as it does for node-label above, so440	// this drop-in can never claim the whole list away from a static441	// file. An entry renders as key=value:Effect, or as key:Effect442	// when the value is empty, which is the kubelet's own taint443	// grammar. This code sorts by key and then by effect, so the same444	// spec always renders the same bytes.445	if len(in.nodeTaints) > 0 {446		taints := slices.Clone(in.nodeTaints)447		slices.SortFunc(taints, func(left, right machine.NodeTaint) int {448			if byKey := strings.Compare(left.Key, right.Key); byKey != 0 {449				return byKey450			}451			return strings.Compare(string(left.Effect), string(right.Effect))452		})453		b.WriteString("node-taint+:\n")454		for _, taint := range taints {455			entry := taint.Key456			if taint.Value != "" {457				entry += "=" + taint.Value458			}459			fmt.Fprintf(&b, "  - %s:%s\n", entry, taint.Effect)460		}461	}462	return b.String()463}464465// kubeletImageGCSettings renders the cluster's image collection policy466// as the lines of a KubeletConfiguration document, under the kubelet's467// own field names. It returns nothing when the cluster named nothing.468//469// The values pass through unresolved, because every one of them is470// already in the kubelet's own grammar: a whole percent, and a Go471// duration, which is what the kubelet's metav1.Duration parses. The472// file doors in cluster/runtime.go have already refused anything the473// kubelet would refuse, so this renders bytes rather than deciding474// anything.475func kubeletImageGCSettings(g cluster.ImageGCSpec) []string {476	var settings []string477	if g.HighThresholdPercent != nil {478		settings = append(settings, fmt.Sprintf("imageGCHighThresholdPercent: %d", *g.HighThresholdPercent))479	}480	if g.LowThresholdPercent != nil {481		settings = append(settings, fmt.Sprintf("imageGCLowThresholdPercent: %d", *g.LowThresholdPercent))482	}483	if g.MinimumAge != "" {484		settings = append(settings, fmt.Sprintf("imageMinimumGCAge: %s", g.MinimumAge))485	}486	if g.MaximumAge != "" {487		settings = append(settings, fmt.Sprintf("imageMaximumGCAge: %s", g.MaximumAge))488	}489	return settings490}491492// kubeletBootConfig wraps those settings in the KubeletConfiguration493// document that the kubelet reads, and returns the empty string when494// there are none, so a cluster that named no policy gets no file.495//496// A configuration file rather than flags, for two reasons. The age497// ceiling has no kubelet flag at all: imageMaximumGCAge exists only in498// this file, so a policy that names it has no other path. And a499// setting with one author is readable, so the three settings that do500// have flags travel with it rather than splitting the policy across501// two grammars. Mixing the two is safe here: a kubelet flag beats the502// file, and k3s passes no image collection flag of its own, so nothing503// this file says is overridden.504func kubeletBootConfig(settings []string) string {505	if len(settings) == 0 {506		return ""507	}508	var b strings.Builder509	b.WriteString("# Written by liken at boot: the kubelet settings that the\n")510	b.WriteString("# cluster's spec.runtime.kubelet section names.\n")511	b.WriteString("apiVersion: kubelet.config.k8s.io/v1beta1\n")512	b.WriteString("kind: KubeletConfiguration\n")513	for _, setting := range settings {514		b.WriteString(setting + "\n")515	}516	return b.String()517}518519// containerdLogLevelDropIn renders the drop-in that gives containerd520// the cluster's log level, and returns the empty string when the521// cluster names none, so a cluster that tunes nothing gets no file.522//523// The drop-in carries one table and nothing else. containerd merges524// each imported file over what it has already read, so this level wins525// over the rendered configuration's. There is nothing to lose in that526// merge: k3s's own template writes no [debug] table at all, so the527// rendered configuration names no level for this file to overwrite.528//529// The version line names version 3, the same version k3s renders,530// and that is the ceiling: containerd refuses an imported file whose531// version exceeds the importing file's. containerd's current version532// is higher than 3, so at each start it migrates the drop-in forward533// and prints one "Configuration migrated" line for it, beside the534// line it already prints for the file k3s renders. The cost is one535// log line per containerd start.536//537// The level goes here rather than on containerd's command line,538// because k3s builds that command line itself and passes no log level539// on it. The one thing that would beat this file is a540// CONTAINERD_LOG_LEVEL variable in k3s's environment, which k3s turns541// into a --log-level flag, and init sets no such variable.542func containerdLogLevelDropIn(level string) string {543	if level == "" {544		return ""545	}546	var b strings.Builder547	b.WriteString("# Written by liken at boot: the containerd log level that the\n")548	b.WriteString("# cluster's spec.runtime.containerd section names.\n")549	b.WriteString("version = 3\n")550	b.WriteString("\n[debug]\n")551	fmt.Fprintf(&b, "  level = %q\n", level)552	return b.String()553}554555// k3sServerDB is where k3s keeps the control plane's datastore556// (sqlite via kine, or embedded etcd) on the clusterState filesystem.557const k3sServerDB = "/var/lib/rancher/k3s/server/db"558559// purgeLeaderLeftovers removes a demoted machine's old control-plane560// datastore. A machine that served as a leader and was demoted keeps561// its etcd data on clusterState. etcd refuses to let a562// permanently-removed member rejoin with its old data directory, so563// a later re-promotion would fail against the leftover data. A564// follower has no reason to keep a datastore, and deleting it is565// what makes demotion truly reversible.566//567// The proven-source guard is what makes this safe to automate. A568// staged document that demotes this machine is still on trial. If it569// fails to prove (for example, an edit that wrongly demotes the only570// leader), the fallback boots the leader role again and needs its571// datastore exactly where it was. The cleanup happens only after a572// demotion has already proved out: the machine joined its cluster as573// a follower, and the operator promoted the document. The cleanup574// runs on the boot after that.575func purgeLeaderLeftovers(role api.Role, clusterManifestSource machine.ManifestSource, dbDir string) {576	if role != api.RoleFollower || clusterManifestSource != machine.ManifestSourceProven {577		return578	}579	if _, err := os.Stat(dbDir); err != nil {580		return581	}582	if err := os.RemoveAll(dbDir); err != nil {583		fmt.Fprintf(os.Stderr, "liken: purging the old control-plane datastore: %v\n", err)584		return585	}586	fmt.Println("liken: this follower once served as a leader; its old control-plane datastore is purged so a future promotion starts clean")587}588589// persistNodePassword gives k3s's node password durable storage. On590// its first join, a machine mints a random secret (its "node591// password"). The leader records it, and every reconnect after that592// must present the same secret. This is what stops a stranger from593// registering as an existing node and receiving its kubelet594// certificates. k3s keeps this secret at595// /etc/rancher/node/password, which on liken is the RAM root: the596// password would vanish on every reboot, and the machine would597// present the wrong secret when it tried to rejoin its own cluster.598// The password is machine identity, and the machine's durable599// identity data lives on machineState, so /etc/rancher/node becomes600// a symlink onto that filesystem. A machine whose machineState is601// memory-backed keeps the tmpfs default, since nothing about it602// survives reboots anyway.603func persistNodePassword(storage machine.StorageStatus) {604	if storage.MachineState.Backing != machine.BackingPartition {605		return606	}607	dir := filepath.Join(machine.MachineStateDir, "node")608	if err := os.MkdirAll(dir, 0o700); err != nil {609		fmt.Fprintf(os.Stderr, "liken: node identity: %v\n", err)610		return611	}612	if err := os.MkdirAll("/etc/rancher", 0o755); err != nil {613		fmt.Fprintf(os.Stderr, "liken: node identity: %v\n", err)614		return615	}616	if err := os.Symlink(dir, "/etc/rancher/node"); err != nil {617		fmt.Fprintf(os.Stderr, "liken: node identity: %v\n", err)618		return619	}620	if err := mintNodePassword(dir); err != nil {621		fmt.Fprintf(os.Stderr, "liken: node identity: %v\n", err)622		return623	}624	fmt.Printf("liken: node identity: /etc/rancher/node persists on machineState\n")625}626627// mintNodePassword writes the node password that k3s would otherwise628// mint for itself on first join. Letting k3s write it risks locking629// a machine out of its own cluster after a single power cut. k3s's630// write is a plain create, and a cut between the create and the data631// reaching disk leaves a 0-byte file. Every boot after that presents632// an empty password, and the cluster refuses it ("node password not633// set"), forever. k3s honors a password file that already exists, so634// init mints the password first, with the same atomic, fsynced write635// that the staging files get, in the same format that k3s generates636// (32 hex characters of real randomness).637//638// A password that already exists here is kept, because the cluster639// recorded it at registration and would refuse a replacement. A640// 0-byte file is the torn write described above, not a credential,641// so it counts as absent. This recovers a machine whose cut happened642// before registration; a cut after registration was unrecoverable643// either way.644func mintNodePassword(dir string) error {645	path := filepath.Join(dir, "password")646	if existing, err := os.ReadFile(path); err == nil && len(existing) > 0 {647		return nil648	}649	secret := make([]byte, 16)650	if _, err := rand.Read(secret); err != nil {651		return err652	}653	return machine.WriteDurable(path, []byte(hex.EncodeToString(secret)+"\n"))654}655656// unreachableNodePortsWarning is the console line for a cluster657// whose NodePort networks leave out the one every other machine uses658// to reach this one, and the empty string when there is nothing to659// say. The list replaces the default rather than adding to it, which660// is what lets a document state the whole answer, and it is also661// what makes this mistake possible: a deployment that names only its662// tunnel closes NodePorts on the node network. Nothing else in the663// system would report that. The machine still boots, because the664// deployment may have meant exactly this.665func unreachableNodePortsWarning(clusterDoc *cluster.Cluster, nodeIP string) string {666	if clusterDoc == nil || len(clusterDoc.Spec.Network.NodePortCIDRs) == 0 || nodeIP == "" {667		return ""668	}669	address := net.ParseIP(nodeIP)670	for _, entry := range clusterDoc.Spec.Network.NodePortCIDRs {671		_, subnet, err := net.ParseCIDR(entry)672		if err == nil && subnet.Contains(address) {673			return ""674		}675	}676	return fmt.Sprintf("liken: nodePortCIDRs (%s) does not include this machine's node IP %s; NodePort services will not answer on the node network",677		strings.Join(clusterDoc.Spec.Network.NodePortCIDRs, ","), nodeIP)678}679680// writeK3sBootConfig derives this machine's role and k3s681// configuration and writes the drop-in beside the role's static682// config file. It returns the role, which tells the supervisor which683// k3s to start.684func writeK3sBootConfig(clusterDoc *cluster.Cluster, m *machine.Machine, conns []*connection) (api.Role, error) {685	name := m.Metadata.Name686	// Role is safe to call with a nil cluster document by design: a687	// machine with no cluster document is a leader of one. This rule688	// is what keeps the follower branches below, which dereference689	// cluster freely, off the nil path. A follower can only be690	// derived from a document.691	role := clusterDoc.Role(name)692	if clusterDoc != nil {693		fmt.Printf("liken: this machine is a cluster %s (cluster %s)\n", role, clusterDoc.Metadata.Name)694	}695	if role == api.RoleFollower && clusterDoc.Spec.Endpoint == "" {696		return role, fmt.Errorf("this machine is a follower, but the cluster manifest declares no endpoint to join")697	}698699	haveToken := true700	if _, err := os.Stat(tokenPath); err != nil {701		haveToken = false702		if role == api.RoleFollower {703			return role, fmt.Errorf("this machine is a follower, but the image carries no join token at %s", tokenPath)704		}705	}706707	nodeIP, nodeInterface := nodeAddress(clusterDoc, conns)708	if nodeIP != "" {709		fmt.Printf("liken: node IP is %s on %s\n", nodeIP, nodeInterface)710	} else if role == api.RoleFollower {711		fmt.Fprintf(os.Stderr, "liken: no address falls inside the cluster's nodeCIDR; k3s will guess a node IP\n")712	}713714	if warning := unreachableNodePortsWarning(clusterDoc, nodeIP); warning != "" {715		fmt.Fprintln(os.Stderr, warning)716	}717718	clusterInit, joinURL := leaderJoinConfig(clusterDoc, name, machine.MachineManifestDir)719	if clusterInit {720		fmt.Println("liken: this machine is the founding leader; embedded etcd runs here")721	} else if joinURL != "" {722		fmt.Printf("liken: joining the control plane at %s\n", joinURL)723	}724725	base := k3sServerConfig726	if role == api.RoleFollower {727		base = k3sAgentConfig728	}729	dropInDir := base + ".d"730	if err := os.MkdirAll(dropInDir, 0o755); err != nil {731		return role, err732	}733734	// The kubelet's configuration file and the drop-in key that names735	// it are written together, in one function, because a drop-in that736	// names a file which is not there stops the kubelet from starting.737	// A cluster that names no policy writes no file and gets no key, so738	// its drop-in is the same bytes it was before the section existed.739	kubeletConfig := ""740	settings := kubeletImageGCSettings(clusterDoc.KubeletSpec().ImageGC)741	if document := kubeletBootConfig(settings); document != "" {742		if err := os.WriteFile(k3sKubeletConfig, []byte(document), 0o644); err != nil {743			return role, err744		}745		kubeletConfig = k3sKubeletConfig746	} else {747		// A cluster that retracts the whole section leaves the file this748		// machine wrote earlier in the same boot, because the restart749		// path runs this function again without a reboot to clear the750		// root. The drop-in no longer names the file, so the kubelet751		// would not read it, but a file that no longer describes the752		// machine misleads the next person who reads it.753		_ = os.Remove(k3sKubeletConfig)754		// The copy k3s made of it must go too, and this one is755		// load-bearing: the kubelet reads the whole drop-in directory756		// the copy sits in, and k3s refreshes the copy only when the757		// argument is present. Left in place, it would keep the758		// retracted policy running on every start after this one.759		_ = os.Remove(k3sKubeletConfigCopy)760	}761762	// containerd's log level, written as a drop-in beside the763	// configuration that k3s renders, because k3s renders that764	// configuration itself on every start (k3sContainerdDropIn765	// explains). The directory is k3s's to create, and on a first boot766	// k3s has not created it yet, so init makes it here.767	containerdLevel := clusterDoc.ContainerdSpec().LogLevel768	if document := containerdLogLevelDropIn(containerdLevel); document != "" {769		if err := os.MkdirAll(filepath.Dir(k3sContainerdDropIn), 0o755); err != nil {770			return role, err771		}772		if err := os.WriteFile(k3sContainerdDropIn, []byte(document), 0o644); err != nil {773			return role, err774		}775	} else {776		// The drop-in sits on clusterState, and containerd imports777		// whatever the directory holds on every start. A cluster that778		// stops naming a level must therefore have the drop-in removed,779		// or containerd would keep the retracted level on every start780		// after this one. This removes one named file, which liken781		// wrote and nothing else writes, so it needs no check of who782		// owns it. Every other file in the directory is the operator's783		// and stays.784		_ = os.Remove(k3sContainerdDropIn)785	}786787	content := k3sBootConfig(k3sBootInputs{788		role:          role,789		clusterDoc:    clusterDoc,790		nodeIP:        nodeIP,791		nodeInterface: nodeInterface,792		haveToken:     haveToken,793		clusterInit:   clusterInit,794		joinURL:       joinURL,795		nodeLabels:    m.Spec.NodeLabels,796		nodeTaints:    m.Spec.NodeTaints,797		kubeletConfig: kubeletConfig,798		debug:         clusterDoc.RuntimeSpec().Debug,799	})800	if err := os.WriteFile(filepath.Join(dropInDir, "boot.yaml"), []byte(content), 0o644); err != nil {801		return role, err802	}803	for line := range strings.SplitSeq(strings.TrimSpace(content), "\n") {804		if !strings.HasPrefix(line, "#") {805			fmt.Printf("liken: k3s config: %s\n", line)806		}807	}808809	// The kubelet's settings echo like the drop-in's lines, one per810	// line, and the facts tree carries the same values to811	// status.runtime.kubelet. A policy that only the serial port812	// reports is invisible to anyone operating the machine remotely.813	// The apiVersion and kind lines stay out of the echo, because they814	// are the document's envelope and carry no decision.815	for _, setting := range settings {816		fmt.Printf("liken: kubelet config: %s\n", setting)817	}818819	// containerd's level echoes for the same reason, and it names the820	// level rather than the drop-in, because the level is the821	// decision and the drop-in is only where it lands. A cluster that822	// names no level prints nothing, so the console does not claim a823	// setting that no file carries.824	if containerdLevel != "" {825		fmt.Printf("liken: containerd config: log level %s\n", containerdLevel)826	}827828	// The Go runtime discipline is set alongside the configuration. It829	// is derived from the same cluster document, and re-deriving it here830	// means an applied restart re-resolves the environment on the same831	// restart that reconfigures k3s. This code echoes it like the config832	// lines, because an invisible environment variable is where a memory833	// problem is hard to find. An unset spec.runtime.k3s section yields834	// an empty environment, and the echo line is then empty too.835	var si unix.Sysinfo_t836	if err := unix.Sysinfo(&si); err == nil {837		k3sMemoryDiscipline = k3sRuntimeEnv(clusterDoc.RuntimeSpec(), uint64(si.Totalram)*uint64(si.Unit))838		fmt.Printf("liken: k3s env: %s\n", strings.Join(k3sMemoryDiscipline, " "))839	}840	return role, nil841}842843// clusterStateStaging is the private mountpoint where clusterState's844// filesystem sits while its seed files are layered in, before the845// mount moves to its real path. It is a shared constant because two846// files must agree on it exactly: mountAndSeedClusterState stages the847// mount here, and teardownStorage (storage.go) unmounts this same848// path when a failed reconcile leaves it behind.849const clusterStateStaging = "/.liken-claim"850851// mountAndSeedClusterState mounts clusterState's filesystem with the852// image's seed files layered in. The image bakes the seeds (the853// pre-generated CAs, the operator's manifests, and the container854// image) underneath clusterState's own mountpoint, and mounting over855// them would hide all of them. So this function first mounts the856// filesystem to the side, seeds it from the image's copies, and only857// then moves it into place. MS_MOVE re-attaches a live mount858// atomically. This function lives here rather than with the859// partition machinery, because everything it encodes, the seed paths,860// what refreshes, and what persists, belongs to k3s's on-disk861// layout.862func mountAndSeedClusterState(dev, target string) error {863	staging := clusterStateStaging864	if err := os.MkdirAll(staging, 0o755); err != nil {865		return fmt.Errorf("mkdir %s: %w", staging, err)866	}867	if err := unix.Mount(dev, staging, "ext4", 0, ""); err != nil {868		return fmt.Errorf("mounting %s for clusterState: %w", dev, err)869	}870	if err := seedClusterState(staging); err != nil {871		return fmt.Errorf("seeding clusterState: %w", err)872	}873	sweepTornK3sFiles(staging)874	if err := os.MkdirAll(target, 0o755); err != nil {875		return fmt.Errorf("mkdir %s: %w", target, err)876	}877	if err := unix.Mount(staging, target, "", unix.MS_MOVE, ""); err != nil {878		return fmt.Errorf("moving clusterState into place at %s: %w", target, err)879	}880	_ = os.Remove(staging)881	return nil882}883884// sweepTornK3sFiles removes the 0-byte identity files that a power885// cut can leave under k3s's state. ext4 can commit a file's creation886// before its data reaches disk, so a machine that loses power just887// after k3s writes a certificate, key, or credential can reboot to a888// file that exists with no bytes in it. k3s treats these files as889// create-once: it reads and trusts an empty one rather than890// regenerating it, and the result is a control plane that cannot891// start (an apiserver that reads a 0-byte loopback certificate892// reports "failed to find any PEM data" and exits, on every restart,893// forever). None of these files matter on their own: every894// certificate, key, and credential here is re-minted on demand from895// the CAs and token that the image carries, so deleting a torn one896// turns an unbootable machine into an ordinary boot. This sweep is897// deliberately scoped to k3s's identity directories. 0-byte files898// elsewhere (lock files, disabled charts) are not init's concern.899// The node password follows the same pattern with its own handling900// (mintNodePassword), because it lives on machineState, and k3s901// would not re-mint a password that is already registered.902func sweepTornK3sFiles(root string) {903	remove := func(path string, info os.FileInfo) {904		if info.IsDir() || info.Size() > 0 {905			return906		}907		if err := os.Remove(path); err == nil {908			fmt.Printf("liken: swept torn k3s file %s (0 bytes; it will be re-minted)\n", path)909		}910	}911	// The server's tls and cred trees hold nothing but identity912	// material, so every 0-byte file there is torn.913	for _, dir := range []string{"k3s/server/tls", "k3s/server/cred"} {914		_ = filepath.Walk(filepath.Join(root, dir), func(path string, info os.FileInfo, err error) error {915			if err == nil {916				remove(path, info)917			}918			return nil919		})920	}921	// The agent keeps its certificates, keys, and kubeconfigs at the922	// top of its directory, next to subtrees (containerd's store)923	// that are not identity and not walked.924	for _, pattern := range []string{"*.crt", "*.key", "*.kubeconfig"} {925		matches, _ := filepath.Glob(filepath.Join(root, "k3s/agent", pattern))926		for _, path := range matches {927			if info, err := os.Stat(path); err == nil {928				remove(path, info)929			}930		}931	}932}933934// seedClusterState copies the image's seed files into a clusterState935// filesystem. This code copies the TLS material only if the disk has936// none, because those keys are the cluster's identity, and a disk937// that already has an identity keeps it. The manifests and the938// operator image refresh on every boot, because they are pinned to939// the liken version of the running image, and an upgraded image must940// deliver its upgraded operator.941//942// The refresh of the manifests replaces the files of liken's own943// subdirectory, and that is why it reaches no further. k3s stages the944// manifests of its bundled components at the top of the auto-deploy945// directory, and the teardown of a component that a boot disables946// reads that component's file at startup. A wipe of the whole947// directory takes that file away first, and the component then keeps948// running with no file left to retract it.949//950// The refresh has one exception. A CRD manifest that the disk holds951// at a higher liken.sh/schema-revision than the image carries is952// neither removed nor rewritten, because replacing it would downgrade953// a schema the cluster already serves and silently prune stored954// objects. schemarevision.go carries that whole story.955//956// The seeds do not share one failure policy, because they do not957// share one failure cost. The identity seed is fatal: a machine that958// cannot put the cluster's CAs on its disk lets k3s mint its own, and959// the result is an API that no kubeconfig from this image can verify960// and a cluster that no machine holding this image's token can join,961// all while reporting itself healthy. Refusing the boot is the only962// honest outcome there. The content seeds degrade instead: a machine963// whose manifests or images did not refresh runs on what its disk964// already holds, and the cluster can read it, report it, and upgrade965// it. A powered-off machine offers none of that, so a content seed966// must never be the reason PID 1 stops.967func seedClusterState(root string) error {968	for _, seed := range []struct {969		rel      string970		refresh  bool971		identity bool972	}{973		{"k3s/server/tls", false, true},974		{likenManifestsRel, true, false},975		{"k3s/agent/images", true, false},976	} {977		src := filepath.Join(seedSourceDir, seed.rel)978		if _, err := os.Stat(src); err != nil {979			continue // an image without k3s has no seed files980		}981		dst := filepath.Join(root, seed.rel)982		var err error983		if seed.refresh {984			keep := map[string]bool{}985			if seed.rel == likenManifestsRel {986				keep = newerCRDsOnDisk(src, dst)987			}988			err = refreshSeedDir(dst, src, keep)989		} else if _, absent := os.Stat(dst); absent != nil {990			err = copySeedTree(dst, src)991		}992		if err == nil {993			continue994		}995		if seed.identity {996			return fmt.Errorf("%s: %w", seed.rel, err)997		}998		reportSeedFailure(seed.rel, err)999	}1000	sweepLikenManifestsFromTheTop(root)1001	return nil1002}10031004// copySeedTree lays a seed onto a disk that has none of it.1005func copySeedTree(dst, src string) error {1006	if err := os.MkdirAll(filepath.Dir(dst), 0o755); err != nil {1007		return err1008	}1009	return os.CopyFS(dst, os.DirFS(src))1010}10111012// reportSeedFailure states what the boot goes on without. The lines1013// go to stderr, beside the other failures a person reads off the1014// console.1015func reportSeedFailure(rel string, err error) {1016	fmt.Fprintf(os.Stderr, "liken: k3s: seeding %s failed: %v\n", rel, err)1017	fmt.Fprintf(os.Stderr, "liken: k3s: this boot runs on the %s the disk already holds\n", rel)1018}10191020// likenManifestNames lists every file name that liken puts in k3s's1021// auto-deploy directory, from the three places such a name can come1022// from: the manifests the image bakes into the seed, the workload1023// manifests each opt-in feature ships, and the manifests a feature's1024// actuation renders on the machine. The names come off the image1025// rather than out of a list written here, so a release that adds or1026// drops a manifest has one place to change and not two.1027func likenManifestNames() []string {1028	names := map[string]bool{}1029	seeded, _ := os.ReadDir(filepath.Join(seedSourceDir, likenManifestsRel))1030	for _, entry := range seeded {1031		names[entry.Name()] = true1032	}1033	staged, _ := filepath.Glob(filepath.Join(featuresDir, "*", "manifests", "*.yaml"))1034	for _, manifest := range staged {1035		names[filepath.Base(manifest)] = true1036	}1037	for _, rendered := range renderedFeatureManifests {1038		for _, name := range rendered {1039			names[name] = true1040		}1041	}1042	return slices.Sorted(maps.Keys(names))1043}10441045// sweepLikenManifestsFromTheTop removes liken's manifests from the1046// top of k3s's auto-deploy directory, where a machine that upgrades1047// from a release that wrote them there still carries them. Those1048// files declare the same objects as the copies in liken's1049// subdirectory, so leaving them in place declares every one of those1050// objects in two addons, and k3s removes an addon only when its own1051// file names a disabled component. The sweep names only liken's own1052// files, so a k3s file that sits beside them stays.1053func sweepLikenManifestsFromTheTop(root string) {1054	for _, name := range likenManifestNames() {1055		path := filepath.Join(root, k3sManifestsRel, name)1056		if err := os.Remove(path); err == nil {1057			fmt.Printf("liken: swept %s from the top of k3s's auto-deploy directory; liken's copy is under %s\n",1058				name, likenManifestsRel)1059		}1060	}1061}
init/liveload.go 93.4%
1package main23// Loading a staged spec's modules into the running kernel.4//5// Most machine-spec changes converge by reboot, because most of what6// the spec declares (storage above all) can only be actuated by a7// boot. Adding a module is the exception. Loading is live-capable:8// the kernel binds a resident driver to already-plugged hardware on9// its own, in either order. So an additive spec.modules edit10// converges here, in place, with nothing drained and nothing11// restarted. An added spec.serio entry converges the same way, in the12// same load: the serio watch attaches it as soon as this load13// declares it (serio.go). The operator stages the manifest as it14// would for a reboot (durability: the next boot must load the same15// list) and writes a modules intent. This file is init's response.16//17// The intent is only a signal. The staged store is the truth about18// what to apply, and init re-derives the manifest's live-applicability19// for itself, with the same shared drift functions that the operator20// used: identical storage, identical network, and no module21// retracted. This re-derivation is what makes a stale, duplicate, or22// malicious intent harmless.23// Anything that would need a boot is refused and left staged for a24// boot.25//26// Promotion comes after the loads, deliberately. A module that27// panics the kernel on load takes the machine down in the middle of28// applying, and the manifest it came from must still be staged when29// the machine comes back. The next boot tries it once, fails the30// same way, and the rejection machinery quarantines it (staging.go).31// Promoting first would enshrine a kernel-crashing spec as proven. A32// load that merely fails (the kernel refuses the module) is not that33// case. The outcome is recorded, and the spec still promotes, exactly34// as a boot would treat it, because retrying the same load changes35// nothing, and the ModulesLoaded condition carries the record.3637import (38	"fmt"39	"maps"40	"os"41	"slices"42	"strings"4344	"github.com/liken-sh/liken/liken/machine"45)4647// moduleLoader owns the subtrees a live load rewrites: modules/,48// boot/modules, and boot/manifest. It holds the boot's values so a load49// judges the staged spec against what actually booted, without reading50// anything back from the tree. The struct is the module loader's whole51// state, seeded from the boot and updated by each successful load.52type moduleLoader struct {53	tree        machine.FactsTree54	bootStorage machine.StorageSpec55	bootNetwork machine.NetworkSpec56	bootModules []string57	// bootParameters is the parameter map this boot loaded with, the58	// reference a staged spec's parameters are judged against. Init59	// re-derives the judgment from its own record rather than trust60	// the intent, so a stale or forged intent still applies nothing61	// that a reboot would apply differently.62	bootParameters map[string]string63	// bootSerio is the spec.serio list the machine counts as declared,64	// and serio is the registry a load declares an added entry to.65	bootSerio []machine.SerioAttachment66	serio     *serioRegistry67	statuses  []machine.ModuleStatus68}6970// apply performs one live load. It reads the staged manifest, refuses71// it unless it is boot-equivalent plus added modules, loads the72// additions, promotes the manifest, and republishes the facts so the73// operator sees convergence. Every refusal is only a console line. The74// manifest stays staged for the reboot that can actually apply it.75func (l *moduleLoader) apply(intent machine.ModulesIntent, store machine.ManifestStore, moduleBase string) {76	raw, err := store.LoadStaged()77	if err != nil {78		fmt.Fprintf(os.Stderr, "liken: modules: reading the staged spec: %v\n", err)79		return80	}81	if raw == nil {82		// A stale signal: the staged spec is already promoted, or83		// withdrawn. The operator's next pass sees the truth.84		return85	}86	doc, err := machine.Parse(raw)87	if err != nil {88		fmt.Fprintf(os.Stderr, "liken: modules: parsing the staged spec: %v\n", err)89		return90	}91	hash := machine.ManifestHash(raw)92	if intent.ManifestHash != "" && intent.ManifestHash != hash {93		// The store moved on since the intent was written. The newer94		// staged bytes are what the operator requests now, so this95		// code judges those bytes. The hash difference is only worth96		// reporting.97		fmt.Printf("liken: modules: the staged spec (%.12s) is newer than the intent (%.12s); applying the store's copy\n",98			hash, intent.ManifestHash)99	}100101	if diffs := machine.StorageDrift(doc.Spec.Storage, l.bootStorage); len(diffs) != 0 {102		fmt.Printf("liken: modules: the staged spec (%.12s) changes storage (%s); it needs a boot, not a load\n",103			hash, strings.Join(diffs, "; "))104		return105	}106	if diffs := machine.NetworkDrift(doc.Spec.Network, &l.bootNetwork); len(diffs) != 0 {107		fmt.Printf("liken: modules: the staged spec (%.12s) changes the network (%s); it needs a boot, not a load\n",108			hash, strings.Join(diffs, "; "))109		return110	}111	added, retracted := machine.ModuleSetDiff(doc.Spec.Modules, l.bootModules)112	if len(retracted) != 0 {113		fmt.Printf("liken: modules: the staged spec (%.12s) retracts %s; loading is one-way, so it needs a boot\n",114			hash, strings.Join(retracted, ", "))115		return116	}117118	// A loaded module never reads its parameters again, so a changed119	// parameter on one is a reboot-class edit however it arrives.120	// The drift function only writes lines for modules in both sets,121	// which is what lets an added module bring its parameters along122	// and load with them below.123	if diffs := machine.ModuleParameterDrift(doc.Spec.Modules, l.bootModules,124		doc.Spec.ModuleParameters, l.bootParameters); len(diffs) != 0 {125		fmt.Printf("liken: modules: the staged spec (%.12s) changes the parameters of a loaded module (%s); it needs a boot, not a load\n",126			hash, strings.Join(diffs, "; "))127		return128	}129130	// A retracted serio entry waits for a boot, the same as a131	// retracted module: its holder keeps the port for the pods that132	// hold the devices the port created.133	addedSerio, retractedSerio := machine.SerioSetDiff(doc.Spec.Serio, l.bootSerio)134	if len(retractedSerio) != 0 {135		fmt.Printf("liken: modules: the staged spec (%.12s) retracts a serio entry (%s); it needs a boot, not a load\n",136			hash, strings.Join(machine.SerioDrift(doc.Spec.Serio, l.bootSerio), "; "))137		return138	}139140	// The diff's added set comes back sorted, and a reboot loads141	// spec.modules in the manifest's own order. The live load must142	// match the reboot it claims equivalence with, because the order143	// decides which module arrives first as a dependency and so144	// cannot take its parameters when its own turn comes.145	load := declaredOrder(doc.Spec.Modules, added)146147	outcomes := loadDeclaredModulesFrom(moduleBase, load, doc.Spec.ModuleParameters)148149	if err := store.Promote(); err != nil {150		// The loads happened, but the record did not move. The151		// operator re-requests, the loads re-run as no-ops (an152		// already-loaded module returns EEXIST, which counts as153		// loaded), and promotion gets another try.154		fmt.Fprintf(os.Stderr, "liken: modules: promoting the applied spec: %v\n", err)155		return156	}157158	// The record appends what this load added. It does not take the159	// manifest's list, because the modules the boot loaded keep their160	// places in the running kernel, and only the additions loaded161	// now, at the end. A record in the manifest's order would claim162	// an order this machine never loaded, and the operator would then163	// report a declared reorder as converged.164	l.bootModules = slices.Concat(l.bootModules, load)165	l.bootParameters = deliveredParameters(doc.Spec.ModuleParameters, outcomes)166	l.bootSerio = slices.Concat(l.bootSerio, addedSerio)167	l.statuses = mergeModuleStatuses(l.statuses, outcomes)168	// The write order is the commit protocol. boot/modules and modules/169	// land first, and boot/manifest lands last, because the manifest170	// record is the commit point the operator's convergence judges171	// (machine-operator/converge.go). The operator must never read a172	// promoted manifest hash before the module facts that explain it.173	l.tree.WriteBootModules(l.bootModules)174	l.tree.WriteBootModuleParameters(l.bootParameters)175	l.tree.WriteBootSerio(l.bootSerio)176	l.tree.WriteModules(l.statuses)177	l.tree.WriteBootManifest(machine.ManifestSourceProven, hash)178	// The declaration comes after the commit point. The serio watch179	// reports each attachment in status.serio on its own, and the180	// operator's convergence judges only the boot record above.181	if l.serio != nil && len(addedSerio) != 0 {182		l.serio.declare(l.bootSerio)183	}184	fmt.Printf("liken: spec %.12s applied in place without a reboot: %s\n",185		hash, appliedInPlace(load, addedSerio))186}187188// appliedInPlace names what a live load applied, for its console line.189func appliedInPlace(loaded []string, attached []machine.SerioAttachment) string {190	var parts []string191	if len(loaded) != 0 {192		parts = append(parts, strings.Join(loaded, ", ")+" loaded")193	}194	for _, a := range attached {195		parts = append(parts, a.String()+" declared for attachment")196	}197	if len(parts) == 0 {198		return "nothing to load"199	}200	return strings.Join(parts, "; ")201}202203// deliveredParameters is the record a live load writes: the staged204// spec's map, less every key whose module was already in the kernel205// when this load reached it.206//207// A resident module's parameters were not delivered: finit_module208// returned EEXIST and dropped the string. Leaving those keys out of209// the record makes them drift, and parameter drift on a loaded210// module is reboot-class, so the operator stages the manifest and211// rebootPolicy takes the normal turn. The reboot loads in manifest212// order and delivers what this load could not. Keys for Missing and213// Failed modules stay recorded, because a reboot with the same214// image changes nothing for them; their conditions carry the story.215func deliveredParameters(declared map[string]string, outcomes []machine.ModuleStatus) map[string]string {216	delivered := maps.Clone(declared)217	for _, outcome := range outcomes {218		if !outcome.AlreadyResident {219			continue220		}221		for _, parameter := range machine.ModuleParameterNames(outcome.Name, declared) {222			delete(delivered, outcome.Name+"."+parameter)223		}224	}225	return delivered226}227228// declaredOrder filters the manifest's module list down to the added229// set, keeping the manifest's order. Which module is resident when a230// load runs decides who keeps their parameters, so the order a load231// runs in is part of what the manifest declares.232func declaredOrder(declared, added []string) []string {233	want := map[string]bool{}234	for _, name := range added {235		want[name] = true236	}237	order := make([]string, 0, len(added))238	for _, name := range declared {239		if want[name] {240			order = append(order, name)241		}242	}243	return order244}245246// mergeModuleStatuses folds a load's outcomes into the standing247// per-module report. A fresh outcome replaces its module's old248// entry, and everything else stays unchanged. The result stays249// sorted by name, so the facts are stable across publishes.250func mergeModuleStatuses(standing, fresh []machine.ModuleStatus) []machine.ModuleStatus {251	byName := map[string]machine.ModuleStatus{}252	for _, s := range standing {253		byName[s.Name] = s254	}255	for _, s := range fresh {256		byName[s.Name] = s257	}258	names := slices.Sorted(maps.Keys(byName))259	merged := make([]machine.ModuleStatus, 0, len(names))260	for _, name := range names {261		merged = append(merged, byName[name])262	}263	return merged264}
init/loadoption.go 95.7%
1package main23// Boot entries: the firmware's menu, one binary record per entry.4//5// A UEFI machine keeps its boot menu in firmware variables named6// Boot0000, Boot0001, and so on, each holding one EFI_LOAD_OPTION:7// the record for one bootable thing. Each record carries a8// human-readable name, the location of the executable, and the9// arguments to hand it. Three companion variables give the list its10// meaning: BootOrder (the durable preference list), BootNext (the11// entry to try on the next boot, once; the firmware erases it after12// use, and that one-shot behavior is what makes the blue-green13// fallback work), and BootCurrent (read-only: the entry actually14// used this boot).15//16// The record is a packed binary format much like the GPT's, with the17// same Microsoft heritage: strings are UTF-16LE, and the location of18// the executable is a "device path", a chain of variable-length19// nodes that narrows from hardware to file. liken's entries need20// exactly two nodes: a hard-drive node that pins one GPT partition21// by its unique GUID (position-independent, like liken's own22// recognition by name), and a file-path node that names the23// executable inside it, backslashes and all. Whatever follows the24// path list is "optional data", which for a Linux EFI-stub kernel is25// simply the kernel command line.26//27// The encoder writes only the entries that liken itself creates. The28// parser must handle more, because real firmware fills these29// variables with vendor nodes of every kind. This code skips unknown30// node types by their declared lengths and otherwise ignores them.3132import (33	"encoding/binary"34	"fmt"35	"unicode/utf16"36)3738// loadOptionActive marks an entry the boot manager may use on its39// own; without it an entry is listed but never chosen automatically.40const loadOptionActive = 0x000000014142// A loadOption is one Boot#### variable, decoded. Fields that liken43// does not model (vendor device-path nodes, unusual attributes)44// survive a decode only to the extent that each field's comment45// describes.46type loadOption struct {47	attributes  uint3248	description string4950	// hardDrive and filePath are the two device-path nodes that liken51	// cares about. They are nil or empty when an entry does not52	// carry them (a PXE entry, a vendor recovery tool).53	hardDrive *hardDriveNode54	filePath  string5556	// optionalData is everything after the device paths: for a57	// kernel booted by its EFI stub, the command line.58	optionalData []byte59}6061// A hardDriveNode identifies one partition three redundant ways: by62// index, by extent, and by GUID. The GUID is the one that matters.63// It is the partition's unique GUID from the GPT itself, so the64// entry survives disks being reordered, exactly like liken's65// recognition by partition name.66type hardDriveNode struct {67	partitionNumber uint3268	firstLBA        uint6469	sectors         uint6470	partitionGUID   [16]byte71}7273// Device-path node types, from the specification's tables.74const (75	dpTypeMedia        = 0x0476	dpSubtypeHardDrive = 0x0177	dpSubtypeFilePath  = 0x0478	dpTypeEnd          = 0x7F79)8081// parseLoadOption decodes one Boot#### variable's payload (the82// efivarfs attribute word already stripped).83func parseLoadOption(b []byte) (loadOption, error) {84	if len(b) < 6 {85		return loadOption{}, fmt.Errorf("load option is %d bytes; even an empty one needs 6", len(b))86	}87	o := loadOption{attributes: binary.LittleEndian.Uint32(b[0:4])}88	pathListLen := int(binary.LittleEndian.Uint16(b[4:6]))8990	description, rest, err := decodeUTF16Z(b[6:])91	if err != nil {92		return loadOption{}, fmt.Errorf("description: %w", err)93	}94	o.description = description9596	if len(rest) < pathListLen {97		return loadOption{}, fmt.Errorf("device path list claims %d bytes but only %d remain", pathListLen, len(rest))98	}99	o.optionalData = rest[pathListLen:]100101	// This walks the device-path nodes: type, subtype, a102	// little-endian length that includes the 4-byte header, then103	// that node's data. The walk trusts each node's declared length104	// and nothing else.105	paths := rest[:pathListLen]106	for len(paths) > 0 {107		if len(paths) < 4 {108			return loadOption{}, fmt.Errorf("device path node truncated: %d bytes left", len(paths))109		}110		nodeType, subType := paths[0], paths[1]111		nodeLen := int(binary.LittleEndian.Uint16(paths[2:4]))112		if nodeLen < 4 || nodeLen > len(paths) {113			return loadOption{}, fmt.Errorf("device path node claims %d bytes of %d", nodeLen, len(paths))114		}115		data := paths[4:nodeLen]116		switch {117		case nodeType == dpTypeEnd:118			paths = nil119			continue120		case nodeType == dpTypeMedia && subType == dpSubtypeHardDrive:121			if len(data) < 38 {122				return loadOption{}, fmt.Errorf("hard-drive node is %d bytes, want 38", len(data))123			}124			hd := &hardDriveNode{125				partitionNumber: binary.LittleEndian.Uint32(data[0:4]),126				firstLBA:        binary.LittleEndian.Uint64(data[4:12]),127				sectors:         binary.LittleEndian.Uint64(data[12:20]),128			}129			copy(hd.partitionGUID[:], data[20:36])130			o.hardDrive = hd131		case nodeType == dpTypeMedia && subType == dpSubtypeFilePath:132			path, _, err := decodeUTF16Z(data)133			if err != nil {134				return loadOption{}, fmt.Errorf("file path: %w", err)135			}136			o.filePath = path137		}138		paths = paths[nodeLen:]139	}140	return o, nil141}142143// encodeLoadOption is the inverse of parseLoadOption. It produces144// the exact bytes that a Boot#### variable holds (again minus the145// efivarfs attribute word, which belongs to the write, not the146// record).147func encodeLoadOption(o loadOption) []byte {148	var paths []byte149	if o.hardDrive != nil {150		// 42 bytes exactly: the 4-byte header plus 38 of payload. The151		// length field counts the header too, which is the easiest152		// detail to get wrong in this format.153		node := make([]byte, 42)154		node[0], node[1] = dpTypeMedia, dpSubtypeHardDrive155		binary.LittleEndian.PutUint16(node[2:4], 42)156		binary.LittleEndian.PutUint32(node[4:8], o.hardDrive.partitionNumber)157		binary.LittleEndian.PutUint64(node[8:16], o.hardDrive.firstLBA)158		binary.LittleEndian.PutUint64(node[16:24], o.hardDrive.sectors)159		copy(node[24:40], o.hardDrive.partitionGUID[:])160		node[40] = 0x02 // the partition table is GPT161		node[41] = 0x02 // so the signature above is a GUID162		paths = append(paths, node...)163	}164	if o.filePath != "" {165		encoded := encodeUTF16Z(o.filePath)166		node := make([]byte, 4, 4+len(encoded))167		node[0], node[1] = dpTypeMedia, dpSubtypeFilePath168		binary.LittleEndian.PutUint16(node[2:4], uint16(4+len(encoded)))169		paths = append(paths, append(node, encoded...)...)170	}171	paths = append(paths, dpTypeEnd, 0xFF, 0x04, 0x00)172173	b := make([]byte, 6, 6+len(paths))174	binary.LittleEndian.PutUint32(b[0:4], o.attributes)175	binary.LittleEndian.PutUint16(b[4:6], uint16(len(paths)))176	b = append(b, encodeUTF16Z(o.description)...)177	b = append(b, paths...)178	return append(b, o.optionalData...)179}180181// bootEntryID renders an entry number the way its variable is named:182// four uppercase hex digits, so Boot0001 and Boot2001 read exactly as183// the firmware spells them.184func bootEntryID(n uint16) string {185	return fmt.Sprintf("Boot%04X", n)186}187188// decodeUTF16Z reads a NUL-terminated UTF-16LE string and returns189// what follows it. UTF-16 because these formats come from Microsoft.190// The terminator carries real information: the description's length191// is recorded nowhere else in the record.192func decodeUTF16Z(b []byte) (string, []byte, error) {193	var units []uint16194	for i := 0; ; i += 2 {195		if i+2 > len(b) {196			return "", nil, fmt.Errorf("UTF-16 string never reached its terminator")197		}198		u := binary.LittleEndian.Uint16(b[i : i+2])199		if u == 0 {200			return string(utf16.Decode(units)), b[i+2:], nil201		}202		units = append(units, u)203	}204}205206// encodeUTF16Z writes a string as NUL-terminated UTF-16LE.207func encodeUTF16Z(s string) []byte {208	units := utf16.Encode([]rune(s))209	b := make([]byte, (len(units)+1)*2)210	for i, u := range units {211		binary.LittleEndian.PutUint16(b[i*2:], u)212	}213	return b214}
init/logrotate.go 88.5%
1package main23// Log rotation, in two forms, both owned by init because init is the4// only program in a position to do them safely.5//6// The first form is rotate-at-boot: before k3s starts, the previous7// boot's k3s and containerd logs shift aside (k3s.log becomes8// k3s.log.1, and so on, keeping a few generations). Boot is the one9// moment when nothing holds either file open, because every process10// from the previous boot is gone, so a rename cannot strand a11// writer. That constraint is containerd's whole story: k3s reopens12// containerd.log each time it starts containerd, but init does not13// own that file's descriptor, so renaming it mid-run would leave14// containerd writing to the renamed generation forever, while the15// fresh path stayed empty. Rotating only at boot also gives the16// files a useful shape: each generation is one boot, a small17// journald-style boot index, and a boot that died leaves its log on18// disk to read afterward.19//20// The second form is the in-boot size cap on k3s.log, which init can21// enforce because it owns that file's writer (supervisor.go tees22// k3s's output through it). These logs live on clusterState, the23// same filesystem as etcd's data, and a filesystem that fills up24// corrupts more than logs. A bounded worst case (the cap times the25// kept generations) is the difference between a chatty k3s and a26// machine that eats its own datastore. containerd.log has no27// equivalent cap, because init does not hold its descriptor. Its28// volume is a small fraction of k3s's, and this file accepts and29// records that residual risk.3031import (32	"errors"33	"fmt"34	"io/fs"35	"os"36)3738const (39	// logGenerations is how many previous boots (or overflowed caps)40	// each log keeps on disk.41	logGenerations = 34243	// k3sLogCap bounds one boot's k3s.log. With rotation, the worst44	// case on disk per family is logGenerations+1 times this.45	k3sLogCap = 64 << 2046)4748// rotateBootLogs shifts the previous boot's logs aside. It runs in49// the k3s branch of main, after storage has settled (so clusterState50// is mounted and these paths land on the persistent disk when the51// machine has one) and before k3s starts (so nothing holds the52// files open). On a memory-backed machine, the same paths sit on the53// root tmpfs, and rotation works the same way. There is just nothing54// left to rotate after a reboot.55func rotateBootLogs() {56	if err := os.MkdirAll(likenLogDir, 0o755); err != nil {57		fmt.Fprintf(os.Stderr, "liken: creating %s: %v\n", likenLogDir, err)58	}59	rotateGenerations(k3sLog, logGenerations)60	rotateGenerations(containerdLog, logGenerations)61}6263// rotateGenerations does the numbered shift. It deletes the oldest64// generation, moves each survivor down by one, and turns the live65// file into .1. Missing files are normal (a first boot has no logs66// at all, and containerd's directory does not exist until k3s has67// run once), so this function stays silent about absence. Any other68// failure is worth a console line but never worth stopping a boot69// over.70func rotateGenerations(path string, keep int) {71	if err := os.Remove(fmt.Sprintf("%s.%d", path, keep)); err != nil && !errors.Is(err, fs.ErrNotExist) {72		fmt.Fprintf(os.Stderr, "liken: rotating %s: %v\n", path, err)73	}74	for n := keep - 1; n >= 1; n-- {75		shiftLog(fmt.Sprintf("%s.%d", path, n), fmt.Sprintf("%s.%d", path, n+1))76	}77	shiftLog(path, path+".1")78}7980func shiftLog(from, to string) {81	if err := os.Rename(from, to); err != nil && !errors.Is(err, fs.ErrNotExist) {82		fmt.Fprintf(os.Stderr, "liken: rotating %s: %v\n", from, err)83	}84}8586// cappedLogFile is the k3s.log writer. It is append-only, and when a87// boot's log passes the cap, it shifts the generations and reopens a88// fresh file. Two details matter for the log relay that tails this89// file. First, rotation happens only at a line boundary, so the90// renamed generation always ends with a complete line (the tailer91// ships a trailing fragment as a whole line, which would garble a92// line split mid-write). Second, rotation happens by rename, which93// is exactly the identity change that the tailer's inode check94// watches for.95//96// Write never returns an error. This writer sits inside the97// io.MultiWriter that carries k3s's output to the console, and98// io.MultiWriter stops at the first writer that fails. An error99// here (for example, a full disk) would silence the console echo100// too, on a machine whose console is the last resort for reading101// anything. So this code reports a failure once, and file logging102// goes quiet while the console copy keeps flowing.103type cappedLogFile struct {104	path        string105	limit       int64106	f           *os.File107	size        int64108	atLineStart bool109	broken      bool110}111112// openCappedLog opens, or continues, the log at path. Appending to113// an existing file matters within a boot: k3s restarts under the114// supervisor's backoff reopen this same log, and each restart must115// extend the boot's record, not truncate it.116func openCappedLog(path string, limit int64) (*cappedLogFile, error) {117	f, err := os.OpenFile(path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o600)118	if err != nil {119		return nil, err120	}121	st, err := f.Stat()122	if err != nil {123		f.Close()124		return nil, err125	}126	return &cappedLogFile{127		path:        path,128		limit:       limit,129		f:           f,130		size:        st.Size(),131		atLineStart: true,132	}, nil133}134135func (c *cappedLogFile) Write(p []byte) (int, error) {136	if c.broken {137		return len(p), nil138	}139	if c.size >= c.limit && c.atLineStart {140		c.rotate()141		if c.broken {142			return len(p), nil143		}144	}145	n, err := c.f.Write(p)146	c.size += int64(n)147	if n > 0 {148		c.atLineStart = p[n-1] == '\n'149	}150	if err != nil {151		c.fail(fmt.Sprintf("writing %s: %v", c.path, err))152	}153	return len(p), nil154}155156func (c *cappedLogFile) rotate() {157	if err := c.f.Close(); err != nil {158		fmt.Fprintf(os.Stderr, "liken: closing %s to rotate: %v\n", c.path, err)159	}160	rotateGenerations(c.path, logGenerations)161	f, err := os.OpenFile(c.path, os.O_APPEND|os.O_CREATE|os.O_WRONLY, 0o600)162	if err != nil {163		c.fail(fmt.Sprintf("reopening %s after rotation: %v", c.path, err))164		return165	}166	c.f = f167	c.size = 0168	c.atLineStart = true169}170171// fail reports the failure once and stops file logging. The console172// copy of k3s's output is not affected.173func (c *cappedLogFile) fail(reason string) {174	fmt.Fprintf(os.Stderr, "liken: %s; k3s file logging stops here (console continues)\n", reason)175	c.broken = true176	if c.f != nil {177		c.f.Close()178		c.f = nil179	}180}181182func (c *cappedLogFile) Close() error {183	if c.f == nil {184		return nil185	}186	err := c.f.Close()187	c.f = nil188	return err189}
init/main.go 0.0%
1// liken is the first and only program that the kernel starts.2//3// The kernel finishes its own boot, then unpacks the initramfs into an4// in-memory root filesystem. The kernel executes one program as5// process ID 1. We name our program liken. We point the kernel at it6// with rdinit=/liken on the kernel command line. That exec is the7// entire handoff from kernel space: a bare environment, any boot8// parameters that the kernel did not recognize (passed as arguments),9// no other processes, and almost no filesystem. Init must set up10// everything else itself.11//12// PID 1 is special to the kernel in three ways. Each way shapes this13// program:14//15//   - PID 1 cannot exit. If PID 1 exits for any reason, the kernel16//     panics. There is no fallback.17//18//   - PID 1 inherits every orphan process. When a process dies, the19//     kernel re-parents its children to PID 1. PID 1 must collect20//     ("reap") their exit statuses, or they stay in the process table21//     forever as zombies.22//23//   - PID 1 does not receive signals by default. The kernel delivers a24//     signal to PID 1 only when init has installed a handler for that25//     signal. This is a safety measure: it stops a stray kill -9 from26//     panicking the machine.27//28// A traditional init grows from this point into a service manager.29// liken's init does not grow this way. Its whole job is to set up the30// minimum environment that k3s needs, start k3s, and keep k3s running.31// Kubernetes is the service manager. Init runs a few loops for itself.32// These loops are called the machine plane. They are goroutines33// registered in components.go, which states the rule for what may run34// in the machine plane and what must run in the cluster instead. When35// the image carries no k3s, a boot is a self-test: mount the essential36// filesystems, read the Machine manifest, join the network, prove the37// connection with a DNS lookup, and power off.38package main3940import (41	"context"42	"errors"43	"fmt"44	"os"45	"path/filepath"46	"slices"47	"time"4849	"golang.org/x/sys/unix"5051	"github.com/liken-sh/liken/liken/api"52	"github.com/liken-sh/liken/liken/cluster"53	"github.com/liken-sh/liken/liken/machine"54)5556func main() {57	// The kernel opened /dev/console before the exec. File descriptors58	// 0, 1, and 2 already point to a real device. With console=ttyS059	// on the kernel command line, that device is the serial port.60	// Ordinary prints reach it directly.61	fmt.Println("liken: hello from userspace")6263	// The panic fault activates immediately, before the program mounts64	// or supervises anything. When PID 1 dies, the kernel panics.65	// panic=10 reboots the kernel. The firmware consumes BootNext and66	// starts the machine from its proven slot. No liken code takes67	// part in this recovery (fault.go).68	if fault == "panic" {69		panic("liken: fault injection: this release panics at startup")70	}7172	// Refuse to run as an ordinary process. Everything below this point73	// assumes the authority and duties of PID 1. Running it from a74	// shell on a development machine would try to mount filesystems at75	// real system paths.76	if os.Getpid() != 1 {77		fmt.Fprintln(os.Stderr, "liken is an init and must run as PID 1; refusing")78		os.Exit(1)79	}8081	// Before any mount or any child process, replace the kernel's root82	// filesystem with a real root filesystem. On success, this re-execs83	// the program from the new root, and main starts over. This is why84	// the console shows the hello message twice. switchroot.go explains85	// why and how.86	maybeSwitchRoot()8788	mountEssentials()8990	// With /proc and /dev mounted, init sends its own output to91	// /dev/kmsg instead (console.go). This happens before the first92	// machine-plane component starts. At this point, main reassigns93	// the os.Stdout and os.Stderr variables while it is still the only94	// goroutine that reads them.95	redirectToKmsg()9697	// Reaping starts before any child process exists. From the moment98	// init spawns a process, or inherits an orphan, only init collects99	// its exit status. Nothing above this line forks.100	plane.start("the reaper", reap)101102	// The firmware's variable store, when the machine has one, holds103	// the boot entries and the boot order. Both the world report and104	// the facts tree read this store (efi.go).105	mountEFIVars()106107	// The OS's own kernel modules load before storage settles, because108	// some roles need their filesystems to arrive as modules. The109	// system slots use FAT32, and mounting vfat pulls in its default110	// character-encoding table (nls_iso8859-1), which Ubuntu's config111	// builds as a module. Everything else on the fixed list is for112	// k3s, which starts much later. Modules load in exactly two113	// passes: this pass, and the declared extras below. The declared114	// extras must wait until storage has settled which manifest this115	// boot runs under.116	loadModules()117118	// A report boot changes nothing on the machine. It mounts the119	// payload's module tree, loads the drivers this hardware names,120	// observes which disks and interfaces appear, writes a proposed121	// manifest to the installation stick, and reboots. It runs here,122	// before storage settles, because it must never claim, format, or123	// write to any of the machine's own disks (report.go).124	if reporting() {125		runHardwareReport() // never returns; it reboots126	}127128	// An install boot reads its cluster document here, before it129	// touches a disk, and refuses the whole install if the document130	// does not parse. The install bakes that document into both slots,131	// and a machine cannot derive its role without it, so a document132	// that fails here fails every boot from the disk afterward. That is133	// the one boot nobody is watching: it powers off a few seconds in,134	// and the reason goes to a console that nothing records.135	//136	// The check sits above the two steps below, and the order is the137	// point. A reinstall blanks the machine's disks, and settleStorage138	// claims and formats them. A document typo must not cost a machine139	// its data before anybody reads the complaint. failBoot holds the140	// console on an install boot, so the person who picked the entry141	// gets the reason while every disk is still as they left it.142	if installing() {143		if err := checkSeedCluster(cluster.ClusterManifestPath); err != nil {144			failBoot(err)145		}146	}147148	// A reinstall boot blanks the disks its manifest declares before149	// storage settles, so a machine that already carries a liken150	// install can be installed over with no external tool. It runs151	// after loadModules (the device nodes exist) and before152	// settleStorage (which then finds the disks blank and claims153	// them). It does not power off; the boot goes on to install154	// (reinstall.go).155	if bootParam(reinstallParam) {156		reclaimManifestDisks()157	}158159	// Storage settles first. This also settles which manifest this160	// boot runs under: the staged manifest awaiting its proving boot,161	// the proven last-known-good manifest, or, on the first boot only,162	// the seed manifest baked into the image. When the image carries163	// manifests for many machines, liken.machine= selects among them.164	// manifests.go explains the full selection. Everything after this165	// line configures the machine from the manifest that this166	// selection chose, never from a manifest that this selection167	// rejected. This call is also one of the two places that can stop168	// a boot. failBoot's rationale explains both places.169	choice, storage, boot, err := settleStorage()170	if err != nil {171		failBoot(err)172	}173	m := choice.m174175	// The installer sets liken.slot= into each boot entry's command176	// line. This parameter names which half of the blue-green pair the177	// boot runs from, and every from-disk boot has it. The boot record178	// carries this fact to the cluster. The operator uses the fact to179	// direct downloads to the other slot.180	if slot := bootParamValue("liken.slot"); slot != "" {181		boot.Slot = slot182		fmt.Printf("liken: firmware: running from system slot %s\n", slot)183	}184185	// An install boot does exactly one job and stops: it puts this186	// running version on the machine's own disk (install.go). This187	// check runs early because nothing after it, such as network,188	// time, or k3s, is relevant to an install boot. An install boot189	// must power off rather than reboot. The install medium is still190	// first in the boot order, so a reboot would run the installer191	// again.192	//193	// Both endings report through holdInstallerConsole, which is also194	// where an attended install waits for the person who picked the195	// entry (attended.go). An install nobody picked prints the same196	// message and goes straight to the power-off below.197	if installing() {198		slot, err := installToDisk(m.Metadata.Name)199		if err != nil {200			fmt.Fprintf(os.Stderr, "liken: %v\n", err)201			fmt.Fprintln(os.Stderr, "liken: install failed; powering off (installs are idempotent: fix the cause and boot the installer again)")202			holdInstallerConsole("liken: press Enter to power off", true)203		} else {204			holdInstallerConsole(fmt.Sprintf("liken: installed to slot %s; remove the stick, then press Enter to power off; the next power-on boots from the disk.", slot), false)205		}206		powerOff()207		for {208			time.Sleep(time.Hour) // PID 1 must not exit, even here209		}210	}211212	// A crash in an earlier boot left its evidence in the platform213	// store. This step reads it, preserves it under machineState, and214	// derives the one-line summary that becomes status.lastCrash215	// (crash.go). It sits here because it needs settled storage: the216	// machineState mount to preserve into, and the backing verdict,217	// which reports whether preserving is possible at all. An install218	// boot never reaches this line, which is correct: the store219	// belongs to the installed machine's history, not to the install220	// medium's.221	mountPstore()222	lastCrash := settleCrashRecords(machine.MachineStateDir,223		storage.MachineState.Backing == machine.BackingPartition)224225	// A crash is one way an earlier boot ended badly. A refusal is the226	// other, and it is the quieter one, because init powered the227	// machine off on purpose and the reason went to a console that228	// nothing recorded. This step reads that reason back off229	// machineState and reports it (failstop.go). It sits beside the230	// crash step because both need settled storage, and, like the crash231	// step, it never clears what it reads.232	lastFailStop := reportFailStop(machine.MachineStateDir)233234	// This block settles where this boot sits in the system release235	// lifecycle. This boot may be the trial of a staged release, in236	// which case the staged record comes back to arm the proving237	// watch. This boot may be the fallback from a staged release. Or238	// this boot may be an ordinary boot, whose only job here is to239	// keep the firmware's boot preference in agreement with the store240	// (proving.go). The actuator is the firmware dialect that these241	// conversations use. main chooses the actuator once, for the242	// whole boot (actuator.go).243	actuator := chooseBootActuator()244	trial := settleSystemRelease(actuator, machine.MachineStateDir, boot.Slot,245		storage.MachineState.Backing == machine.BackingPartition, &boot)246247	// The cluster document says which machines are leaders. This248	// machine derives its own role from that document. The cluster249	// document goes through the same staged, proven, and seed250	// lifecycle as the Machine manifest (cluster.go). Reading the251	// cluster document can stop the boot: if a machine's only cluster252	// document does not parse, the machine cannot determine its role, and a253	// machine that cannot tell its role must not guess.254	clusterDoc, clusterRaw, err := chooseCluster(machine.MachineStateDir, cluster.ClusterManifestPath,255		storage.MachineState.Backing == machine.BackingPartition, &boot)256	if err != nil {257		failBoot(fmt.Errorf("%w: %v", errIdentity, err))258	}259260	// The registry credentials follow their own document lifecycle261	// (registries.go). The operator stages credentials from the262	// registry-credentials Secret. main chooses credentials here, and263	// renders them into k3s's registries.yaml below. This step is264	// never fatal. A machine without credentials pulls images265	// anonymously.266	creds := chooseRegistryCredentials(machine.MachineStateDir,267		storage.MachineState.Backing == machine.BackingPartition, &boot)268269	// The declared modules load in the second of the two module270	// passes. This pass is possible only now that main has the271	// chosen manifest. The boot record keeps the request (the drift272	// reference: rebooting with the same image would request the same273	// modules). The statuses keep the results, bound for274	// status.modules through the facts tree.275	// The record keeps the order this pass loads the modules in, which276	// is the manifest's order. The order decides which driver claims a277	// device, so the operator compares it against the spec278	// (machine-operator/converge.go). A sorted record would report an279	// order no machine ever loaded.280	boot.Modules = slices.Clone(m.Spec.Modules)281	boot.ModuleParameters = m.Spec.ModuleParameters282	moduleStatuses := loadDeclaredModules(m.Spec.Modules, m.Spec.ModuleParameters)283	// The serio entries depend on the modules above, so they are284	// declared after the loads. The boot record keeps the request, the285	// drift reference, the same way it keeps the module list. The286	// serio watch attaches them once the facts tree exists (serio.go).287	boot.Serio = slices.Clone(m.Spec.Serio)288	serioAttachments.declare(m.Spec.Serio)289290	// The cluster's opt-in features are actuated and reported per291	// machine, the same way as declared modules (features.go). Bundled292	// components take effect in the k3s drop-in rendered below.293	// Vendored payloads load their modules, write their boot files,294	// and seed their workloads at this point. There is no boot record295	// entry for features, because features drift by the cluster296	// document's whole-document hash, not field by field.297	featureStatuses := actuateFeatures(clusterDoc, m.Metadata.Name)298299	if name := m.Metadata.Name; name != "" {300		// Sethostname is one syscall. It needs no hostnamectl command301		// and no daemon. The kernel only keeps a string.302		if err := unix.Sethostname([]byte(name)); err != nil {303			fmt.Fprintf(os.Stderr, "liken: sethostname %q: %v\n", name, err)304		} else {305			fmt.Printf("liken: hostname is %s\n", name)306		}307	}308309	// The settings every liken machine holds apply first. The Machine310	// spec's sysctls apply after them, so a deployment that disagrees311	// with one of them overwrites it. Both sets apply before k3s312	// starts, so every value is set by the time k3s reads it, and both313	// apply before the network comes up below, which is what puts the314	// default queueing discipline on this machine's own interfaces.315	// The operator applies both sets again once the cluster is up.316	// This is what makes a live kubectl edit take effect without a317	// reboot, and what returns a parameter something else changed.318	applySysctls(machine.OSSysctls)319	applySysctls(m.Spec.Sysctls)320321	// Resource limits follow the same defaults-then-spec order, and322	// they differ from sysctls in one way that decides where they can323	// run. A sysctl is a file the kernel re-reads, so the operator324	// reconciles one live. A limit is fixed when a process forks, so325	// the only way to give k3s a limit is to hold it here, before326	// starting k3s, and let inheritance carry it down. Nothing can327	// raise the ceiling of a running process, which is why an edit to328	// spec.rlimits waits for a reboot.329	applyRlimits(machine.OSRlimits)330	applyRlimits(m.Spec.Rlimits)331	// The boot record keeps the request, not the outcome, exactly as332	// it does for modules above. It is the drift reference: booting333	// again under this manifest would ask for these same limits. What334	// the kernel actually holds goes to status.rlimits, read back in335	// publishBootFacts.336	boot.Rlimits = m.Spec.Rlimits337338	worldReport()339340	// The cluster document reaches the network step because two341	// decisions there read it: the park asks whether any settled342	// interface gives a route toward the endpoint, and the pass343	// split asks whether pass one already holds the address k3s344	// starts with. This boot's document is already chosen by this345	// line.346	conns, radios, err := bringUpNetwork(m.Spec.Network, clusterDoc)347	if err != nil {348		fmt.Fprintf(os.Stderr, "liken: network: %v\n", err)349	}350	for _, conn := range conns {351		conn.report()352	}353354	// If the image carries k3s, liken does its main job: it sets up355	// the rest of the environment that Kubernetes expects, then356	// supervises k3s for as long as the machine runs. A machine357	// without k3s (the image's minimal form) only proves that it can358	// boot, then powers off.359	if _, err := os.Stat(k3sBinary); err == nil {360		clusterLife(choice, storage, boot, clusterDoc, clusterRaw, creds,361			conns, radios, moduleStatuses, featureStatuses, actuator, trial, lastCrash, lastFailStop) // never returns362	}363364	// With no k3s to supervise, a boot is complete once the report is365	// out. Powering off, rather than exiting (see above), gives QEMU366	// a clean shutdown signal. This is what lets `make run` also work367	// as a test harness.368	fmt.Println("liken: boot complete, powering off")369	powerOff()370}371372// clusterLife runs the k3s half of a boot, everything after the point373// where the machine itself is settled. It covers the container374// store's trust decision, the role and its configuration, the clock,375// the facts, the machine plane's long-running components, and finally376// the supervisor, which runs for as long as the machine runs.377// clusterLife never returns. Every path out of it is a reboot or a378// power-off.379func clusterLife(choice *manifestChoice, storage machine.StorageStatus, boot machine.BootStatus,380	clusterDoc *cluster.Cluster, clusterRaw []byte, creds *machine.RegistryCredentials,381	conns []*connection, radios *radioPass, moduleStatuses []machine.ModuleStatus,382	featureStatuses []machine.FeatureStatus, actuator bootActuator, trial *machine.SystemRelease,383	lastCrash *machine.CrashStatus, lastFailStop *machine.FailStop) {384	m := choice.m385386	// Before k3s can touch its container store, this call determines387	// whether this boot can trust that store. If a store's last388	// imports were never proven, main discards the store rather389	// than trust it (imports.go). The tarballs that this boot390	// carries are staged as a trial for the operator to prove.391	settleImageImports(machine.MachineStateDir,392		storage.MachineState.Backing == machine.BackingPartition,393		storage.ClusterState.Backing == machine.BackingPartition, &boot)394	// The machine's role and k3s's boot-derived configuration395	// come from the cluster manifest (k3s.go). A failure here is396	// also an identity problem. A follower with no address for397	// its cluster must not start up as if it had one.398	role, err := writeK3sBootConfig(clusterDoc, m, conns)399	if err != nil {400		failBoot(fmt.Errorf("%w: %v", errIdentity, err))401	}402	// This call sets how this machine pulls container images:403	// mirrors and the embedded registry from the cluster document,404	// and credentials from their own document. It renders these405	// into the registries.yaml file that k3s reads at start406	// (registries.go). Writing this file promotes staged407	// credentials. The write is the credentials' whole actuation.408	registries := writeRegistriesConfig(clusterDoc, creds,409		machine.RegistryCredentialsStore(machine.MachineStateDir), boot.CredentialsSource)410	// The node password that k3s creates on first join must411	// outlive this boot. Otherwise, the machine could never rejoin412	// its own cluster (k3s.go explains).413	persistNodePassword(storage)414	// A proven demotion also removes the datastore left over from415	// when this machine was a leader. etcd does not let a removed416	// member rejoin over its old data, so this call must delete417	// the datastore before k3s starts (k3s.go).418	purgeLeaderLeftovers(role, boot.ClusterManifestSource, k3sServerDB)419	// The clock is corrected before k3s starts, because a wrong420	// clock makes TLS fail: every certificate that the CA issued421	// looks like it comes from the future. This is the only422	// moment when liken steps the clock. After this moment, liken423	// only slews the clock (time.go explains both kinds of424	// correction).425	clk := newClock(timeSources(clusterDoc, role, machine.MachineManifestDir))426	firstSync := stepClockAtBoot(clk.sources)427	clk.record(firstSync)428	// This call saves a successful boot measurement at once. If429	// the machine later loses power without a clean shutdown, it430	// still boots with roughly the right time (time.go describes431	// the two moments when the code writes the RTC).432	if firstSync != nil {433		writeRTC()434	}435	hostname, _ := os.Hostname()436	configureNameResolution(hostname, m.Spec.Network.HostEntries)437	prepareForK3s()438	// Pod logs land on podEphemeral, and still appear at the path439	// Kubernetes uses, /var/log/pods. This call comes after440	// prepareForK3s, which creates /var/log and sets the mount441	// propagation that the bind inherits, and before k3s starts, so the442	// mount is in place before kubelet opens its first log file443	// (podlogs.go).444	bindPodLogs(storage)445	// The previous boot's k3s and containerd logs move aside446	// before k3s starts to write this boot's logs. This is a447	// plain function call rather than a machine-plane component,448	// because it must finish before k3s opens the log files. Boot449	// is the only safe moment to rename these files, because450	// nothing holds them open at boot (logrotate.go).451	rotateBootLogs()452	// The hardware walk also waits until this point. Every453	// declared module has loaded and bound whatever it drives, so454	// any device that is still undriven now is a real gap, not a455	// result of a race with the rest of boot.456	catalog := loadHardwareCatalog()457	unclaimed := discoverUnclaimed(catalog)458	blockDevices := discoverBlockDevices()459	initialTime := timeStatus(firstSync, clk.sources)460	// The facts step waits until this point so the tree appears461	// once, complete. Each boot step held its discovered facts462	// locally and hands them here, and the operator's first read463	// sees a whole tree rather than one growing in pieces.464	publishBootFacts(factsTree, bootFacts{465		clusterDoc:   clusterDoc,466		role:         role,467		conns:        conns,468		storage:      storage,469		boot:         boot,470		modules:      moduleStatuses,471		features:     featureStatuses,472		registries:   registries,473		time:         initialTime,474		blockDevices: blockDevices,475		unclaimed:    unclaimed,476		lastCrash:    lastCrash,477		lastFailStop: lastFailStop,478	})479	publishBootManifest(choice)480	publishBootClusterManifest(clusterRaw)481	// A background radio's verdicts land through this component. It starts482	// here, after publishBootFacts, because publishBootFacts writes the483	// network subtree itself: a verdict written before it would be484	// overwritten with the pending state it replaced. The radio's address485	// and its status arrive late, and both beat a boot that waited486	// (plans/completed/64-the-boot-does-not-wait-for-radios.md).487	if radios != nil {488		plane.start("the wireless verdicts", publishRadioVerdicts(radios, factsTree, clusterDoc))489	}490	// The hardware watch keeps that walk correct for the whole491	// life of the machine. Hot-plugged devices arrive as uevents,492	// and the report follows these events (hardware.go). An image493	// without the catalog cannot judge devices, so it does not run494	// this watch either.495	if catalog != nil {496		for _, line := range hardwareTransitions(nil, unclaimed, nil) {497			fmt.Println(line)498		}499		plane.start("the hardware watch", watchHardware(catalog, factsTree, unclaimed, blockDevices))500	}501	// The disk links watch publishes /dev/disk/by-path, /dev/disk/by-id,502	// and /dev/disk/by-uuid, the stable names a CSI driver, mount503	// tooling, and an operator reading the console all resolve a disk504	// or a filesystem through (disklinks.go). It runs on every machine,505	// and outside the catalog's gate above: naming a disk needs no506	// device database, and a machine gains its first iSCSI session long507	// after this boot, whenever a workload asks for a volume.508	plane.start("the disk links", watchDiskLinks)509	// The serio watch holds the attachments spec.serio declares510	// (serio.go). It starts here, before k3s, so an adapter that is511	// plugged in at boot has its devices before the machine operator512	// publishes its first slice. It runs on every machine, because a513	// live load can declare the first entry at any time.514	plane.start("the serio watch", watchSerio(serioAttachments, factsTree))515	// A machine with time sources keeps disciplining its clock for516	// as long as it runs. A free-running machine has no source to517	// follow, and its status already states this.518	if len(clk.sources) > 0 {519		plane.start("the clock", disciplineClock(clk, factsTree, initialTime))520	}521	// Only leaders serve time, because only leaders receive time522	// requests. Followers sync from the leaders themselves.523	// Serving still works when free-running: a fleet with no524	// upstream time source still keeps one consistent time across525	// its machines (responder.go).526	if role == api.RoleLeader {527		plane.start("the time responder", serveTime(clk))528	}529	// The intent channel works like this: init creates the530	// directory and sets its permissions, the operator writes531	// into the directory, and the watcher carries requests to the532	// supervisor (reboot.go). Over the life of a boot, this533	// channel carries at most one reboot request and any number534	// of k3s restart requests.535	if err := os.MkdirAll(machine.OperatorRunDir, 0o755); err != nil {536		fmt.Fprintf(os.Stderr, "liken: creating %s: %v\n", machine.OperatorRunDir, err)537	}538	rebootRequests := make(chan machine.RebootIntent, 1)539	restartRequests := make(chan machine.RestartIntent, 1)540	loadRequests := make(chan machine.ModulesIntent, 1)541	plane.start("the intent watch", func(ctx context.Context) error {542		return watchForOperatorIntents(ctx, machine.OperatorRunDir, rebootRequests, restartRequests, loadRequests)543	})544	// The module loader handles the lightest intent: an additive545	// spec.modules edit that applies to the running kernel, with546	// no reboot and no k3s restart involved (liveload.go). It runs547	// beside the supervisor rather than inside it, because, unlike548	// the other two intents, this intent does not involve k3s.549	loader := &moduleLoader{550		tree:           factsTree,551		bootStorage:    boot.Storage,552		bootNetwork:    m.Spec.Network,553		bootModules:    boot.Modules,554		bootParameters: boot.ModuleParameters,555		bootSerio:      boot.Serio,556		serio:          serioAttachments,557		statuses:       moduleStatuses,558	}559	plane.start("the module loader", func(ctx context.Context) error {560		store := machine.MachineManifests(machine.MachineStateDir)561		moduleBase := filepath.Join("/lib/modules", kernelRelease())562		for {563			select {564			case <-ctx.Done():565				return nil566			case intent := <-loadRequests:567				loader.apply(intent, store, moduleBase)568			}569		}570	})571	// The restart path gathers everything that a k3s restart may572	// re-render, while that data is available (restart.go).573	restarter := newRestartState(machine.MachineStateDir, m, conns, factsTree,574		clusterDoc, clusterRaw, creds, boot.CredentialsSource)575	// A proving boot watches for its own promotion. When the576	// operator's first reconcile proves the staged release, init577	// sets the firmware's boot preference to the newly proven slot578	// (proving.go).579	if trial != nil {580		plane.start("the proving watch", provingWatch(actuator, *trial))581	}582	// Only a leader can report cluster state, because the admin583	// kubeconfig is a control-plane artifact, and followers hold584	// no credentials of their own. A follower's join appears on585	// the leader's console, and in the follower's own k3s log586	// lines.587	if role == api.RoleLeader {588		plane.start("the node report", reportWhenReady)589	}590	// The wedge fault boots everything except k3s. The node never591	// joins, the operator never runs, and no promotion ever592	// happens. This is exactly the failure that the proving593	// watchdog exists to catch (fault.go). The machine plane keeps594	// running, so the watchdog's reboot still works.595	if fault == "wedge-k3s" {596		fmt.Println("liken: fault injection: wedging instead of starting k3s")597		select {}598	}599	superviseK3s(role, rebootRequests, restartRequests, restarter.apply,600		restarter.removeOfflineRetractions) // never returns601}602603// powerOff shuts the machine down cleanly. It makes every disk604// read-only, which is what records that the machine stopped properly605// (quiesce.go), then sync flushes what is left, and the reboot syscall606// powers the machine off. Both steps are no-ops on a machine with no607// writable disk, and both are essential the moment the machine has608// one: the installer reaches this function with the slots it just609// wrote still mounted. PID 1 must not simply exit. This is the only610// correct way for init to stop.611func powerOff() {612	stopSupplicants()613	quiesceDisks()614	syncLogs()615	unix.Sync()616	if err := unix.Reboot(unix.LINUX_REBOOT_CMD_POWER_OFF); err != nil {617		fmt.Fprintf(os.Stderr, "liken: power off failed: %v\n", err)618	}619}620621// failBoot is the fail-stop function. It prints the problem and why622// the problem warrants a power-off, then it powers the machine off.623// failBoot lives here rather than in storage.go or manifests.go624// because it is boot policy, not domain logic. Each domain reports625// what it could not do, and main determines which failures a machine626// must not run through. There are two such failures:627//628//   - identity: the machine cannot tell which manifest, or which629//     role, is its own. A guess could join the wrong cluster, start a630//     rival control plane, or claim another machine's disks.631//   - storage: a declared role cannot be satisfied, and a machine632//     declared to have persistent state must not start up with no633//     persistent state.634//635// The reasoning is the same in both cases. A machine that is down can636// be fixed and booted again. A machine running with the wrong637// configuration can do damage that a reboot will not undo.638func failBoot(err error) {639	rationale := "storage: a declared role can't be satisfied, and a machine declared to have persistent state must not come up ephemeral; powering off"640	if errors.Is(err, errIdentity) {641		rationale = "identity: one image boots many machines, and a machine that can't tell which configuration is its own must not guess; powering off"642	}643	fmt.Fprintf(os.Stderr, "liken: %v\n", err)644	fmt.Fprintf(os.Stderr, "liken: %s\n", rationale)645	// The two lines above reach a console that nothing records.646	// syncLogs is a 50 ms sleep, and no log relay runs on a boot that647	// never starts k3s, so the reason goes onto machineState here for648	// the next boot to report (failstop.go).649	recordFailStop(machine.MachineStateDir, err.Error(), time.Now().UTC())650	// An install boot's refusal is one of the install menu's terminal651	// states, so it reports the disk inventory as its evidence and, if a652	// person picked this boot, holds for them (attended.go). A boot from653	// disk never comes here for a person's benefit, and an install that654	// nobody picked must reach the power-off below rather than stop at a655	// prompt that nobody can answer.656	if installing() {657		holdInstallerConsole("liken: press Enter to power off", true)658	}659	powerOff()660	// powerOff returns only if the reboot syscall failed. PID 1 still661	// must not exit, so this loop keeps the machine here for a person662	// to investigate.663	for {664		time.Sleep(time.Hour)665	}666}
init/manifests.go 89.5%
1package main23// Choosing the manifest a boot runs under.4//5// The machine's most important input is a file on a disk that it has6// not set up yet. The way out of that circle is the same recognition7// that drives all of storage: this code finds the machineState8// partition by the name written on it, which needs no spec at all.9// Init peeks at it first: mount read-only, read the staged and10// proven manifests, unmount. The unmount matters. The same disk may11// need its partition table rewritten minutes later (a grow), and the12// kernel refuses to re-read the table of a disk that is in use.13// Because the peek leaves nothing mounted, machineState's own disk14// stays growable.15//16// Then the attempt order applies: a staged manifest is tried first,17// and if its storage cannot be reconciled, the boot does not stop.18// This code quarantines the staged manifest durably, with the19// reason, tears storage back down, and reconciles the proven manifest20// (the last spec that actually booted) instead. The machine comes up21// degraded but present, which beats a machine that is off.22// Power-off remains the answer only when even the proven spec fails,23// because at that point the machine has no configuration it can24// trust.25//26// The image's baked-in manifest takes part only when the machine has27// no machineState partition, or that partition is empty. It seeds28// the very first boot, and that boot's success writes it down as the29// first proven manifest. From then on, the file in the image plays no30// further role.31//32// A first boot runs end to end like this: no machineState partition33// exists, so this code chooses the seed. Reconciliation claims the34// disks (machineState is first in canonical order, so it becomes35// partition 1). The role mounts at its path. The seed's bytes become36// proven.yaml. Every crash point along the way resumes correctly: a37// claim that died before mkfs completes through recognition (the38// name goes on first), and an empty manifests directory simply39// re-selects the seed.4041import (42	"errors"43	"fmt"44	"io/fs"45	"os"46	"path/filepath"47	"strings"48	"time"4950	"golang.org/x/sys/unix"5152	"github.com/liken-sh/liken/liken/machine"53)5455// manifestPeekPoint is the private mountpoint for the early look at56// machineState, used and released before storage reconciliation runs.57var manifestPeekPoint = "/.liken-machine-state"5859// A manifestChoice is one candidate manifest and its identity: the60// hash travels into facts (and rejections) so the operator can tell61// exactly which bytes this boot ran.62type manifestChoice struct {63	m      *machine.Machine64	raw    []byte65	source machine.ManifestSource66	hash   string67}6869// manifestCandidates is everything the peek learned.70type manifestCandidates struct {71	part      *partition // the machineState partition; nil on first boot72	staged    *manifestChoice73	proven    *manifestChoice74	rejection *machine.Rejection // the standing quarantine record, if any75}7677// findMachineStatePartition scans for the one partition named78// liken:machineState. A missing partition means a first boot, not an79// error. Two of them is the same cloned-disk ambiguity that80// matchRoles refuses, and this function refuses it for the same81// reason.82func findMachineStatePartition() (*partition, error) {83	var found *partition84	for _, p := range discoverPartitions() {85		if p.partName != machine.PartitionPrefix+"machineState" {86			continue87		}88		if found != nil {89			return nil, fmt.Errorf("two partitions claim to be %smachineState (%s and %s); refusing to guess",90				machine.PartitionPrefix, found.name, p.name)91		}92		found = &p93	}94	return found, nil95}9697// loadManifestCandidates performs the peek. A machineState partition98// with no filesystem yet (a boot that died between claim and mkfs)99// yields no candidates. The seed carries that boot, and mountRole's100// resumable claiming makes the filesystem.101func loadManifestCandidates() (manifestCandidates, error) {102	var c manifestCandidates103	part, err := findMachineStatePartition()104	if err != nil {105		return c, err106	}107	c.part = part108	if part == nil {109		return c, nil110	}111	dev := devRoot + "/" + part.name112	if !hasExt4(dev) {113		return c, nil114	}115116	if err := os.MkdirAll(manifestPeekPoint, 0o755); err != nil {117		return c, err118	}119	// Read-only means "this code writes nothing", not "the device is120	// untouched". Mounting ext4 replays its journal if the last boot121	// died mid-write, which is the filesystem's recovery mechanism122	// working exactly as intended.123	if err := mountFilesystem(dev, manifestPeekPoint, "ext4", unix.MS_RDONLY, ""); err != nil {124		return c, fmt.Errorf("peeking at machineState on %s: %w", dev, err)125	}126	// The store returns bytes. Whether they parse as a Machine is127	// this caller's question, asked below after the peek unmounts.128	store := machine.MachineManifests(manifestPeekPoint)129	stagedRaw, stagedErr := store.LoadStaged()130	provenRaw, provenErr := store.LoadProven()131	c.rejection, _ = store.LoadRejection()132	if err := unmountFilesystem(manifestPeekPoint, 0); err != nil {133		// Loud but not fatal: if this mount lingers, a later rewrite134		// of this disk's table fails with EBUSY and names the problem.135		fmt.Fprintf(os.Stderr, "liken: storage: unmounting the manifest peek: %v\n", err)136	}137	_ = os.Remove(manifestPeekPoint)138139	if provenErr != nil {140		fmt.Fprintf(os.Stderr, "liken: storage: the proven manifest is unreadable: %v\n", provenErr)141	} else if provenRaw != nil {142		c.proven = provenCandidate(provenRaw)143	}144145	// A staged manifest is checked before it even becomes a146	// candidate. One that does not parse, or that does not declare147	// the machineState role its own lifecycle lives on, would fail148	// every future boot the same way, so this code rejects it without149	// trying it.150	if stagedErr != nil {151		c.rejection = rejectStaged(part, nil, fmt.Sprintf("the staged manifest is unreadable: %v", stagedErr))152	} else if stagedRaw != nil {153		staged, err := machine.Parse(stagedRaw)154		switch {155		case err != nil:156			c.rejection = rejectStaged(part, stagedRaw, fmt.Sprintf("the staged manifest does not parse: %v", err))157		case staged.Spec.Storage.MachineState == nil:158			c.rejection = rejectStaged(part, stagedRaw, "the staged manifest does not declare the machineState role its own lifecycle lives on")159		default:160			c.staged = &manifestChoice{m: staged, raw: stagedRaw, source: machine.ManifestSourceStaged, hash: machine.ManifestHash(stagedRaw)}161		}162	}163	return c, nil164}165166// provenCandidate parses the proven manifest. The parse skips the167// fields this release does not know (machine.ParseKnown), because a168// rollback boots an older release under a manifest that a newer one169// proved, and that manifest still holds this machine's storage and170// network. The console names each field the boot leaves out. A proven171// manifest that does not parse even then is a corrupted172// last-known-good record. This code reports it and continues without173// one, rather than fail over a file whose whole job is recovery.174func provenCandidate(raw []byte) *manifestChoice {175	proven, ignored, err := machine.ParseKnown(raw)176	if err != nil {177		fmt.Fprintf(os.Stderr, "liken: storage: the proven manifest is unreadable: %v\n", err)178		return nil179	}180	if len(ignored) > 0 {181		fmt.Fprintf(os.Stderr, "liken: storage: the proven manifest names fields this release does not know, and this boot leaves them out: %s\n",182			strings.Join(ignored, ", "))183	}184	return &manifestChoice{m: proven, raw: raw, source: machine.ManifestSourceProven, hash: machine.ManifestHash(raw)}185}186187// rejectStagedDocument renders the verdict that every staged188// lifecycle shares: announce the reason, quarantine the document189// durably (the store moves the bytes aside, so the same document is190// never tried twice), and return the rejection for this boot's191// facts. The document lifecycles differ in what they stage, such as192// the cluster document, registry credentials, a system release, or193// the Machine manifest itself, but they all reject in the same way.194// The domain and noun exist only to keep each console line specific195// to its owner. The record step is a parameter because most stores196// can write immediately (their filesystem is already mounted), while197// the Machine manifest's rejection must mount machineState around198// the write (rejectStaged below).199//200// The rejection outlasts the boot that rendered it. Each store keeps201// it standing, and because the facts rebuild from scratch every202// boot, each chooser republishes the standing rejection into the203// boot record before it consults anything staged.204func rejectStagedDocument(domain, what string, record func(machine.Rejection) error,205	raw []byte, reason string) *machine.Rejection {206	fmt.Fprintf(os.Stderr, "liken: %s: rejecting the staged %s: %s\n", domain, what, reason)207	rejection := machine.NewRejection(raw, reason, time.Now().UTC())208	if err := record(rejection); err != nil {209		fmt.Fprintf(os.Stderr, "liken: %s: recording the rejection: %v\n", domain, err)210	}211	return &rejection212}213214// rejectStaged quarantines the staged Machine manifest with its215// reason. It runs during the peek, before the machineState role is216// properly mounted, so the record step mounts the partition itself.217func rejectStaged(part *partition, raw []byte, reason string) *machine.Rejection {218	return rejectStagedDocument("storage", "manifest", func(rejection machine.Rejection) error {219		return touchMachineState(*part, func(root string) error {220			return machine.MachineManifests(root).Reject(rejection)221		})222	}, raw, reason)223}224225// touchMachineState mounts the machineState partition read-write at226// the private point, runs fn against it, and unmounts. This is the227// narrow window in which init allows itself to write manifests228// before, or after, the role is properly mounted.229func touchMachineState(p partition, fn func(root string) error) error {230	if err := os.MkdirAll(manifestPeekPoint, 0o755); err != nil {231		return err232	}233	if err := mountFilesystem(devRoot+"/"+p.name, manifestPeekPoint, "ext4", 0, ""); err != nil {234		return err235	}236	ferr := fn(manifestPeekPoint)237	if err := unmountFilesystem(manifestPeekPoint, 0); err != nil {238		fmt.Fprintf(os.Stderr, "liken: storage: unmounting %s: %v\n", manifestPeekPoint, err)239	}240	_ = os.Remove(manifestPeekPoint)241	return ferr242}243244// errIdentity marks the failures where a machine could not tell245// which manifest is its own. main treats these differently from246// storage failures when it explains the power-off on the console.247var errIdentity = errors.New("machine identity")248249// seedPath selects this machine's own file from the manifests250// directory. One image boots many machines, so the image carries a251// manifest per machine, and each boot selects its own: explicitly, by252// the liken.machine=<name> kernel parameter (the one parameter the253// bootloader already sets), or implicitly when there is exactly one254// manifest to choose. This function refuses anything else rather than255// guess. A name that matches no manifest is a typo that someone must256// see, and a directory of manifests with no name given is ambiguity,257// the same situation as two disks that claim the same partition258// name. An empty or absent directory is not an error: a machine with259// no manifest is still a valid machine.260func seedPath(dir, requested string) (string, error) {261	if requested != "" {262		path := filepath.Join(dir, requested+".yaml")263		if _, err := os.Stat(path); err != nil {264			return "", fmt.Errorf("%w: liken.machine=%s names no manifest at %s", errIdentity, requested, path)265		}266		return path, nil267	}268	entries, err := os.ReadDir(dir)269	if errors.Is(err, fs.ErrNotExist) {270		return "", nil271	}272	if err != nil {273		return "", err274	}275	var names []string276	for _, entry := range entries {277		if strings.HasSuffix(entry.Name(), ".yaml") {278			names = append(names, entry.Name())279		}280	}281	switch len(names) {282	case 0:283		return "", nil284	case 1:285		return filepath.Join(dir, names[0]), nil286	default:287		return "", fmt.Errorf("%w: %d manifests in %s and no liken.machine=<name> on the kernel command line to choose among %s; refusing to guess",288			errIdentity, len(names), dir, strings.Join(names, ", "))289	}290}291292// loadSeed reads the image's baked-in manifest for this machine,293// selecting by the requested name within the given directory. A seed294// that cannot be selected or parsed is an error rather than a silent295// default. loadSeed runs only on a boot with no proven manifest to296// fall back to, and a first boot under the wrong or empty identity297// could join the wrong cluster or claim the wrong disks. A machine298// that stays down can be fixed. A machine running under the wrong299// identity can do real damage.300func loadSeed(dir, requested string) (*manifestChoice, error) {301	choice := &manifestChoice{m: &machine.Machine{}, source: machine.ManifestSourceSeed}302	path, err := seedPath(dir, requested)303	if err != nil {304		return nil, err305	}306	if path == "" {307		return choice, nil308	}309	raw, err := os.ReadFile(path)310	if err != nil {311		return nil, fmt.Errorf("%w: reading %s: %v", errIdentity, path, err)312	}313	m, err := machine.Parse(raw)314	if err != nil {315		return nil, fmt.Errorf("%w: %s: %v", errIdentity, path, err)316	}317	fmt.Printf("liken: this is %s (manifest %s)\n", m.Metadata.Name, path)318	choice.m, choice.raw, choice.hash = m, raw, machine.ManifestHash(raw)319	return choice, nil320}321322// attemptOrder is the whole preference policy in one place: staged323// before proven. The seed is deliberately not among the candidates.324// It is a first-boot input, not a fallback (a machine that has ever325// proven a manifest never consults the image again), so settleStorage326// loads the seed only when this function returns an empty list.327func attemptOrder(c manifestCandidates) []*manifestChoice {328	switch {329	case c.staged != nil && c.proven != nil:330		return []*manifestChoice{c.staged, c.proven}331	case c.staged != nil:332		return []*manifestChoice{c.staged}333	case c.proven != nil:334		return []*manifestChoice{c.proven}335	default:336		return nil337	}338}339340// settleStorage actuates storage under the best available manifest341// and reports which one won. An error means that even the last342// manifest in the attempt order failed (or, wrapped as errIdentity,343// that a first boot could not tell which manifest is its own), and344// the caller stops the boot. The winning choice comes back whole,345// raw bytes and all, because init later publishes those exact bytes346// for the operator.347func settleStorage() (*manifestChoice, machine.StorageStatus, machine.BootStatus, error) {348	status := machine.AllRolesInMemory()349	boot := machine.BootStatus{}350351	candidates, err := loadManifestCandidates()352	if err != nil {353		return nil, status, boot, err354	}355	boot.Rejection = candidates.rejection356357	attempts := attemptOrder(candidates)358	if len(attempts) == 0 {359		// A machine with no durable manifests is on its first boot.360		// Only now does the image's seed matter, along with the361		// question of which seed belongs to this machine.362		seed, err := loadSeed(machine.MachineManifestDir, bootParamValue("liken.machine"))363		if err != nil {364			return nil, status, boot, err365		}366		attempts = []*manifestChoice{seed}367	}368	for i, choice := range attempts {369		if choice.source != machine.ManifestSourceSeed {370			fmt.Printf("liken: storage: booting under the %s manifest (%.12s)\n", choice.source, choice.hash)371		}372		status, err = reconcileStorage(choice.m.Spec.Storage)373		if err == nil {374			boot.ManifestSource = choice.source375			boot.ManifestHash = choice.hash376			boot.Storage = choice.m.Spec.Storage377			// The network is recorded here, where the manifest wins,378			// and not later, where the links come up. The record says379			// what this boot ran under, which is the question the380			// operator's drift detection asks. What each interface381			// actually got is a different question, and status.network382			// answers that one.383			network := choice.m.Spec.Network384			boot.Network = &network385			settleManifests(machine.MachineManifests(machine.MachineStateDir), choice, status, &boot)386			return choice, status, boot, nil387		}388		if choice.source == machine.ManifestSourceStaged && i+1 < len(attempts) {389			fmt.Fprintf(os.Stderr, "liken: storage: the staged manifest failed: %v\n", err)390			fmt.Fprintln(os.Stderr, "liken: storage: falling back to the proven manifest")391			teardownStorage()392			rejection := machine.NewRejection(choice.raw, err.Error(), time.Now().UTC())393			if terr := touchMachineState(*candidates.part, func(root string) error {394				return machine.MachineManifests(root).Reject(rejection)395			}); terr != nil {396				fmt.Fprintf(os.Stderr, "liken: storage: recording the rejection: %v\n", terr)397			}398			boot.Rejection = &rejection399			continue400		}401		return choice, status, boot, err402	}403	// attemptOrder never returns an empty list, so the loop always returns.404	panic("unreachable")405}406407// settleManifests finishes the lifecycle bookkeeping after a408// successful reconcile. A staged manifest that just proved itself is409// promoted, and a seed's first success becomes the first proven410// manifest. Failures here are loud but not fatal. The machine is up,411// and the next boot simply repeats the step (a staged manifest that412// boots once boots again).413func settleManifests(store machine.ManifestStore, choice *manifestChoice, status machine.StorageStatus, boot *machine.BootStatus) {414	if status.MachineState.Backing != machine.BackingPartition {415		return // nothing durable to keep manifests on416	}417	switch choice.source {418	case machine.ManifestSourceStaged:419		if err := store.Promote(); err != nil {420			fmt.Fprintf(os.Stderr, "liken: storage: promoting the staged manifest: %v\n", err)421			return422		}423		fmt.Printf("liken: storage: the staged manifest is now proven (%.12s)\n", choice.hash)424		boot.ManifestSource = machine.ManifestSourceProven425		boot.Rejection = nil // a success supersedes old history426	case machine.ManifestSourceSeed:427		if len(choice.raw) == 0 {428			return429		}430		if err := store.WriteProven(choice.raw); err != nil {431			fmt.Fprintf(os.Stderr, "liken: storage: recording the seed as proven: %v\n", err)432			return433		}434		fmt.Printf("liken: storage: the seed manifest is now proven (%.12s)\n", choice.hash)435		boot.ManifestSource = machine.ManifestSourceProven436	}437}
init/moduleparams.go 100.0%
1package main23// How a module gets its parameters on a machine with no modprobe and4// no modprobe.d: init builds the parameter string from the spec and5// hands it to finit_module in the one call that loads the module.6// The driver reads it at probe and never again, which is why a7// parameter change on a loaded module needs a boot. The only place8// the result can be read afterward is /sys/module, and the readback9// here is what status.modules[].parameters reports.1011import (12	"fmt"13	"os"14	"path/filepath"15	"strings"1617	"github.com/liken-sh/liken/liken/machine"18)1920// sysModuleDir is the kernel's own registry of what is loaded,21// builtins included. It is a variable so a test can fabricate one22// out of directories.23var sysModuleDir = "/sys/module"2425// The CRD caps a readback value at 1024 bytes, and apiextensions26// refuses the whole status write when one value runs over, so the27// clamp below is what keeps one chatty parameter from freezing a28// machine's entire status.29const moduleParameterValueMax = 10243031// The longest key admission accepts: a 64-byte module name, the dot,32// and a 64-byte parameter name. A longer key would fail as a file33// name in the boot record, partway through the write.34const moduleParameterKeyMax = 1293536// usableModuleParameters drops the declarations the machine cannot37// act on: an empty value, which the facts tree would treat as a38// removed file, and an over-long key, which would abort the boot39// record with ENAMETOOLONG. Admission refuses both, but a manifest40// carried in on a stick never met admission, so this is the last41// guard before the kernel. Each drop prints its reason.42func usableModuleParameters(parameters map[string]string) map[string]string {43	usable := map[string]string{}44	for key, value := range parameters {45		switch {46		case value == "":47			fmt.Fprintf(os.Stderr, "liken: modules: %s declares no value; a parameter needs one\n", key)48		case len(key) > moduleParameterKeyMax:49			fmt.Fprintf(os.Stderr, "liken: modules: %.64s... is longer than %d bytes; skipping it\n",50				key, moduleParameterKeyMax)51		default:52			usable[key] = value53		}54	}55	if len(usable) == 0 {56		return nil57	}58	return usable59}6061// sysModuleName is the spelling /sys/module uses: underscores,62// whatever the module's file name used. The kernel treats - and _ in63// a module name interchangeably and normalizes to _ when it64// registers the module.65func sysModuleName(name string) string {66	return strings.ReplaceAll(name, "-", "_")67}6869// moduleIsResident answers whether the kernel already holds a70// module. The declared pass asks before it loads, because asking71// after answers too late: finit_module returns EEXIST for a resident72// module and drops the parameter string without a trace, and the73// status must be able to say that happened.74func moduleIsResident(dir, name string) bool {75	info, err := os.Stat(filepath.Join(dir, sysModuleName(name)))76	return err == nil && info.IsDir()77}7879// readModuleParameters reads back the declared parameter names for80// one module, and only those; the kernel's directory also lists81// every parameter the declaration never mentioned. A declared name82// whose file cannot be read is left absent. Absence usually means83// the name is wrong, because the kernel accepts an unknown parameter84// name and warns only in its log; it can also mean the parameter is85// real but offers nothing to read, since a driver may register a86// parameter with no sysfs file at all (libata.force does) or with a87// write-only mode. The value kept is the kernel's own rendering, not88// the string that was passed: a bool comes back as Y or N, an array89// with its own separators.90func readModuleParameters(dir, name string, declared map[string]string) map[string]string {91	names := machine.ModuleParameterNames(name, declared)92	if len(names) == 0 {93		return nil94	}95	base := filepath.Join(dir, sysModuleName(name), "parameters")96	held := map[string]string{}97	for _, parameter := range names {98		raw, err := os.ReadFile(filepath.Join(base, parameter))99		if err != nil {100			continue101		}102		value := strings.TrimSuffix(string(raw), "\n")103		// Some parameters print a whole legend on read;104		// acpi.debug_level answers 1313 bytes. The schema takes105		// 1024, and one oversize value would fail the whole status106		// write, so the value is cut and the cut is silent.107		if len(value) > moduleParameterValueMax {108			value = value[:moduleParameterValueMax]109		}110		held[parameter] = value111	}112	if len(held) == 0 {113		return nil114	}115	return held116}
init/modules.go 90.3%
1package main23// Loading kernel modules, without modprobe.4//5// The kernel cannot use a driver that was built as a module until6// something feeds that module back to it, and the kernel itself does7// not locate modules on disk; that is userspace's job. The usual8// program for this is modprobe, from the kmod project. liken ships9// no modprobe, so init does the two things that modprobe would have10// done:11//12//  1. Resolve dependencies. Modules depend on other modules (overlay13//     needs nothing; iptable_nat pulls in a chain of netfilter14//     pieces). Nothing scans the module tree at runtime to work this15//     out. At image build time, depmod wrote an index, modules.dep,16//     that maps every module to the full list of modules it needs,17//     already ordered so that loading right-to-left satisfies every18//     dependency.19//20//  2. Ask the kernel to load each file. This is the finit_module21//     syscall: "here is an open file descriptor, it is a kernel22//     module, trust it." liken's modules are zstd-compressed23//     (.ko.zst) exactly as Ubuntu shipped them. The24//     MODULE_INIT_COMPRESSED_FILE flag tells the kernel to decompress25//     the module itself (liken's vendored config sets26//     CONFIG_MODULE_DECOMPRESS=y), so init never touches the bytes.27//28// Which modules to load comes from two lists, loaded in two passes.29// The first list is /etc/liken/modules.conf, a plain list baked into30// the image, the same list that the image build used to select which31// module files to ship: the OS's own needs, loaded up front because32// the alternative (on-demand autoloading) works by the kernel33// exec'ing /sbin/modprobe itself, and a fixed, reviewable list is34// better than a hidden runtime dependency. The second list is the35// Machine spec's declared extras (spec.modules), the drivers for36// whatever hardware this machine's workloads use, which cannot load37// until the boot settles which manifest won. loadDeclaredModules below38// explains what each outcome means.3940import (41	"errors"42	"fmt"43	"io/fs"44	"os"45	"path/filepath"46	"strings"4748	"golang.org/x/sys/unix"4950	"github.com/liken-sh/liken/liken/machine"51)5253// modulesConf is the image's fixed module list. A package variable54// rather than a constant so tests can point the first pass at a list55// of their own making.56var modulesConf = "/etc/liken/modules.conf"5758// finitModule is the syscall behind every load, a variable so a test59// can watch exactly which file and which parameter string the kernel60// would have received, without a test run being able to load61// anything real.62var finitModule = unix.FinitModule6364func loadModules() {65	release := kernelRelease()66	base := filepath.Join("/lib/modules", release)6768	names, err := readModuleList(modulesConf)69	if err != nil {70		fmt.Fprintf(os.Stderr, "liken: modules: %v\n", err)71		return72	}73	if len(names) == 0 {74		return75	}7677	deps, err := readModulesDep(filepath.Join(base, "modules.dep"))78	if err != nil {79		fmt.Fprintf(os.Stderr, "liken: modules: %v\n", err)80		return81	}8283	loaded := map[string]bool{}84	count := 085	for _, name := range names {86		n, err := loadModule(base, name, "", deps, loaded)87		count += n88		if err != nil {89			fmt.Fprintf(os.Stderr, "liken: modules: %s: %v\n", name, err)90		}91	}92	fmt.Printf("liken: loaded %d kernel modules for %s\n", count, strings.Join(names, ", "))93}9495// loadDeclaredModules loads the extra modules that the winning96// Machine manifest declared, and reports each name's outcome. Unlike97// the fixed list, whose failures are printed and forgotten (the image98// build ships every module on that list), a declared module is a99// deployment's request, and the answer must reach the cluster.100// These outcomes travel through the facts tree into status.modules.101// Nothing here can stop the boot. A machine missing a workload's102// driver is degraded, not down.103func loadDeclaredModules(names []string, parameters map[string]string) []machine.ModuleStatus {104	return loadDeclaredModulesFrom(filepath.Join("/lib/modules", kernelRelease()), names, parameters)105}106107// loadDeclaredModulesFrom is the same pass with the module tree as a108// parameter, so tests can point it at a fabricated tree. Only the109// kernel's own tree ever has real modules to load.110//111// The parameters map is spec.moduleParameters, and only this pass112// carries one. The fixed list and a feature's list ship with the113// release, and a machine that needs to modify how the OS loads its own114// drivers has a bug for liken to fix, not a knob to turn115// (plans/completed/55-kernel-module-parameters.md).116func loadDeclaredModulesFrom(base string, names []string, parameters map[string]string) []machine.ModuleStatus {117	if len(names) == 0 {118		return nil119	}120	// The last guard before the kernel. Admission refuses these121	// shapes, but a manifest carried in on a stick never met122	// admission (moduleparams.go explains each drop).123	parameters = usableModuleParameters(parameters)124125	// Both indexes are depmod's work, shipped beside the modules126	// themselves. modules.dep maps every shipped module to its127	// dependency chain. modules.builtin names what is compiled into128	// vmlinuz, which is what lets a declared name that matches no129	// file mean "already there" instead of "missing".130	deps, err := readModulesDep(filepath.Join(base, "modules.dep"))131	if err != nil {132		fmt.Fprintf(os.Stderr, "liken: modules: %v\n", err)133		deps = map[string][]string{}134	}135	builtin, err := readModulesBuiltin(filepath.Join(base, "modules.builtin"))136	if err != nil {137		fmt.Fprintf(os.Stderr, "liken: modules: %v\n", err)138	}139140	loaded := map[string]bool{}141	statuses := declaredModuleOutcomes(names, deps, builtin, parameters, declaredPass{142		resident: func(name string) bool { return moduleIsResident(sysModuleDir, name) },143		load: func(name, params string) error {144			_, err := loadModule(base, name, params, deps, loaded)145			return err146		},147		readback: func(name string) map[string]string {148			return readModuleParameters(sysModuleDir, name, parameters)149		},150	})151	for _, s := range statuses {152		fmt.Println(declaredModuleLine(s, parameters))153	}154	return statuses155}156157// declaredModuleLine renders one declared module's console line. A158// load that carried parameters names the exact string the kernel159// got, so the console and the log record what this boot passed, the160// same record status.boot.moduleParameters keeps for the API.161func declaredModuleLine(s machine.ModuleStatus, parameters map[string]string) string {162	line := fmt.Sprintf("liken: modules: %s: %s", s.Name, strings.ToLower(string(s.State)))163	if params := passedParameters(s, parameters); params != "" {164		line += " (" + params + ")"165	}166	if s.Message != "" {167		line += ": " + s.Message168	}169	// A resident module's line, and a builtin's line, would otherwise170	// read exactly like a clean load, and the console is the live171	// record on a machine with no shell; the dropped string has to be172	// said here, not only in status.173	if declared := machine.ModuleParameterString(s.Name, parameters); declared != "" {174		switch {175		case s.AlreadyResident:176			line += ": already in the kernel, so " + declared + " did not reach it"177		case s.State == machine.ModuleBuiltin:178			line += ": the kernel builds it in, so " + declared + " did not reach it"179		}180	}181	return line182}183184// passedParameters is the string to show beside a load, and only a185// load that actually delivered one: a resident module never received186// the string, and a refused load already names it in its message.187func passedParameters(s machine.ModuleStatus, parameters map[string]string) string {188	if s.AlreadyResident || s.State != machine.ModuleLoaded {189		return ""190	}191	return machine.ModuleParameterString(s.Name, parameters)192}193194// declaredPass is the three questions a declared pass asks the195// machine: is this module already in the kernel, load it with this196// string, and what does /sys hold now. They are fields so a test can197// answer each one without a kernel.198type declaredPass struct {199	resident func(name string) bool200	load     func(name, parameters string) error201	readback func(name string) map[string]string202}203204// declaredModuleOutcomes classifies each declared name before it205// asks the kernel for anything, so the verdicts do not depend on the206// order that failures happen to occur in. The vocabulary belongs to207// machine.ModuleState: Loaded and Builtin are healthy. Missing means208// the image's kernel build has no module by this name: the image209// carries the kernel's whole module tree, so a missing name is a210// misspelled name, or a driver from outside this kernel build.211// Failed means the kernel refused a module that the image does ship.212func declaredModuleOutcomes(names []string, deps map[string][]string,213	builtin map[string]bool, parameters map[string]string, pass declaredPass) []machine.ModuleStatus {214	statuses := make([]machine.ModuleStatus, 0, len(names))215	for _, name := range names {216		status := machine.ModuleStatus{Name: name}217		key := strings.ReplaceAll(name, "-", "_")218		params := machine.ModuleParameterString(name, parameters)219		switch {220		case deps[key] != nil:221			// Residency is read before the load, because the load222			// cannot report it: finit_module returns EEXIST for a223			// module the kernel already holds and drops the224			// parameter string without a trace. The string is225			// cleared here so the record shows what was delivered,226			// not what was hoped.227			status.AlreadyResident = pass.resident(name)228			if status.AlreadyResident {229				params = ""230			}231			if err := pass.load(name, params); err != nil {232				status.State = machine.ModuleFailed233				status.Message = err.Error()234				if params != "" {235					status.Message += fmt.Sprintf(" (parameters %s)", params)236				}237			} else {238				status.State = machine.ModuleLoaded239				status.Parameters = pass.readback(name)240			}241		case builtin[key]:242			status.State = machine.ModuleBuiltin243			status.Parameters = pass.readback(name)244		default:245			status.State = machine.ModuleMissing246			status.Message = "not a module in this image's kernel build; check the spelling against the machine's unclaimed-hardware report"247		}248		statuses = append(statuses, status)249	}250	return statuses251}252253// readModuleList reads the requested module names: one per line, with254// blank lines and # comments allowed, since the file doubles as255// documentation.256func readModuleList(path string) ([]string, error) {257	raw, err := os.ReadFile(path)258	if errors.Is(err, fs.ErrNotExist) {259		return nil, nil260	}261	if err != nil {262		return nil, err263	}264	var names []string265	for line := range strings.SplitSeq(string(raw), "\n") {266		line = strings.TrimSpace(line)267		if line == "" || strings.HasPrefix(line, "#") {268			continue269		}270		names = append(names, line)271	}272	return names, nil273}274275// readModulesDep parses depmod's index. Each line looks like this:276//277//	kernel/fs/overlayfs/overlay.ko.zst: kernel/a.ko.zst kernel/b.ko.zst278//279// That is, a module's path, then every module it transitively needs.280// This function keys the map by module name (the filename minus281// extensions), because module names use "_" and "-" interchangeably.282func readModulesDep(path string) (map[string][]string, error) {283	raw, err := os.ReadFile(path)284	if err != nil {285		return nil, err286	}287	deps := map[string][]string{}288	for line := range strings.SplitSeq(string(raw), "\n") {289		path, needs, found := strings.Cut(line, ":")290		if !found {291			continue292		}293		entry := append([]string{path}, strings.Fields(needs)...)294		deps[moduleName(path)] = entry295	}296	return deps, nil297}298299// readModulesBuiltin parses depmod's record of what is compiled into300// vmlinuz itself: one module path per line, for example:301//302//	kernel/fs/binfmt_misc.ko303//304// A name found here needs no loading, because the kernel already305// contains it. This set is keyed the same way as modules.dep, with306// names normalized to "_".307func readModulesBuiltin(path string) (map[string]bool, error) {308	raw, err := os.ReadFile(path)309	if err != nil {310		return nil, err311	}312	builtin := map[string]bool{}313	for line := range strings.SplitSeq(string(raw), "\n") {314		line = strings.TrimSpace(line)315		if line == "" {316			continue317		}318		builtin[moduleName(line)] = true319	}320	return builtin, nil321}322323func moduleName(path string) string {324	name := filepath.Base(path)325	name = strings.TrimSuffix(name, ".zst")326	name = strings.TrimSuffix(name, ".ko")327	return strings.ReplaceAll(name, "-", "_")328}329330// loadModule feeds one module and its dependencies to the kernel,331// dependencies first (modules.dep lists them ready to load332// right-to-left), and returns how many files the kernel actually333// loaded. An already-loaded module, whether loaded by this call or334// by an earlier dependency chain, returns EEXIST, which counts as335// success but not toward the count.336//337// The parameter string goes to the module's own file alone, which is338// entry zero of its modules.dep line; every dependency loads with an339// empty string, exactly as modprobe would load it. A parameter meant340// for a dependency belongs to that module's own declaration.341func loadModule(base, name, parameters string, deps map[string][]string, loaded map[string]bool) (int, error) {342	entry, ok := deps[strings.ReplaceAll(name, "-", "_")]343	if !ok {344		return 0, fmt.Errorf("not in modules.dep (is it built into the kernel?)")345	}346	count := 0347	for i := len(entry) - 1; i >= 0; i-- {348		file := entry[i]349		if loaded[file] {350			continue351		}352		f, err := os.Open(filepath.Join(base, file))353		if err != nil {354			return count, err355		}356		params := ""357		if i == 0 {358			params = parameters359		}360		err = finitModule(int(f.Fd()), params, unix.MODULE_INIT_COMPRESSED_FILE)361		f.Close()362		if err != nil && !errors.Is(err, unix.EEXIST) {363			return count, fmt.Errorf("finit_module %s: %w", file, err)364		}365		if err == nil {366			count++367		}368		loaded[file] = true369	}370	return count, nil371}372373func kernelRelease() string {374	var u unix.Utsname375	if err := unix.Uname(&u); err != nil {376		return "unknown"377	}378	return unix.ByteSliceToString(u.Release[:])379}
init/network.go 87.8%
1package main23// Network setup: from no configuration to a routed interface, in4// userspace Go.5//6// This file uses two different kernel interfaces. Neither is a7// classic syscall. Interface configuration (links up, addresses,8// routes) happens over netlink, a socket-based protocol that the9// kernel implements. The `ip` command uses netlink too. The10// vishvananda/netlink library builds the netlink messages.11//12// Getting an address by DHCP works differently. DHCP is a network13// protocol, not a kernel feature, so liken must implement the client14// side itself: broadcast DISCOVER, receive OFFER, send REQUEST,15// receive ACK. The insomniacslk/dhcp library does this work (the16// same library Talos uses to boot). It sends over a raw AF_PACKET17// socket. This socket type lets a machine send UDP before it has an18// IP address to send from.19//20// A static address is the simplest case: it uses no protocol, only21// the netlink calls that apply an address someone already chose.22// Static addressing exists because clustering needs it. A machine's23// peers must have its address before it boots, so the manifest must24// declare the address instead of negotiating it on the wire. The lab25// also needs static addressing, for a different reason: the network26// segment that joins the QEMU guests has no DHCP server on it.2728import (29	"context"30	"fmt"31	"net"32	"os"33	"path/filepath"34	"strings"35	"time"3637	"github.com/insomniacslk/dhcp/dhcpv4/nclient4"38	"github.com/vishvananda/netlink"3940	"github.com/liken-sh/liken/liken/cluster"41	"github.com/liken-sh/liken/liken/machine"42)4344// connection holds the facts that the code gathers while it brings up45// one interface. These facts are enough to print a report of how the46// machine connects to the network, and to publish the same facts to47// the Machine's status.48type connection struct {49	ifname string50	mac    net.HardwareAddr5152	// addr may be nil. A radio that did not join leaves an interface53	// that exists and has no address, and that state is exactly what54	// the status must report, so the boot keeps the connection and55	// every reader checks this field before it reads one.56	addr        *net.IPNet57	method      machine.AddressMethod // how the code obtained the address: DHCP or Static58	gateway     net.IP59	nameservers []net.IP60	leaseTime   time.Duration61	server      net.IP6263	// leaseExpires is fixed at the moment the DHCP ACK lands. The64	// facts render from this absolute time, so a rewrite hours later65	// reports the same expiry; deriving it from the clock at render66	// time would move every lease forward with no DHCP exchange.67	leaseExpires time.Time6869	// radio is the 802.11 session behind this interface, for an70	// interface the spec gave a wireless entry. Nil means the71	// interface is wired.72	radio *radio73}7475// bringUpNetwork configures every interface that the spec names.76// When the spec names no interface, it uses the zero-configuration77// default: DHCP on the first interface that looks like real78// hardware. If one interface fails, the function still configures79// the others. A machine with its uplink down but its cluster segment80// up is degraded, not absent, and the console report shows which81// interface is which. The function returns an error only when no82// interface comes up.83//84// The bring-up runs in two passes (radios.go): wired interfaces85// settle here, and radios settle behind the boot when the machine86// already has a route toward its cluster. The second return value is87// that background pass, nil when nothing runs behind; the caller88// hands it to the component that lands each radio's verdict in the89// facts tree after the boot's own facts publish.90//91// The whole cluster document comes in rather than the endpoint92// alone, because the pass split reads two of its facts: the endpoint93// for the route question, and the nodeCIDR for the node-address94// question (radios.go, backgroundable).95func bringUpNetwork(spec machine.NetworkSpec, clusterDoc *cluster.Cluster) ([]*connection, *radioPass, error) {96	// The code brings up loopback first. Nearly all networked97	// software assumes that 127.0.0.1 exists. The kernel creates the98	// loopback interface; the code only needs to raise it.99	if lo, err := linkByName("lo"); err == nil {100		if err := linkSetUp(lo); err != nil {101			return nil, nil, fmt.Errorf("raising lo: %w", err)102		}103	}104105	// A spec that cannot be right is refused before any link changes.106	// The same check runs in the cluster, where the operator refuses107	// to stage such a spec, but the boot cannot rely on that: init108	// also reads manifests that were written by hand and carried in109	// on a stick, which no API server ever saw.110	if err := spec.Validate(); err != nil {111		return nil, nil, err112	}113114	// The kernel's link list is read once and used for every115	// interface below. Reading it once also means that every error116	// message can list the same set of ports, which is the list a117	// person needs to correct the manifest.118	links, err := listLinks()119	if err != nil {120		return nil, nil, fmt.Errorf("listing interfaces: %w", err)121	}122	present := presentInterfaces(links)123124	interfaces := spec.Interfaces125	if len(interfaces) == 0 {126		name, err := pickInterface(present)127		if err != nil {128			return nil, nil, err129		}130		interfaces = []machine.InterfaceSpec{{Name: name}}131	}132133	conns, pass := bootPasses(present).run(interfaces, clusterDoc)134135	if !anyAddressed(conns) {136		return conns, pass, fmt.Errorf("no interface came up")137	}138	if err := writeResolvConf(conns); err != nil {139		return conns, pass, err140	}141	return conns, pass, nil142}143144// resolvConfPath is where the resolver file goes. It is a variable145// rather than a constant so a test can write into a file of its own,146// the way wirelessRunDir and parkConsole let tests stand in for the147// real machine.148var resolvConfPath = "/etc/resolv.conf"149150// The code builds one resolv.conf file for the whole machine,151// gathered from every interface. The resolvConf function below152// explains which nameservers it keeps. The file is an ordinary153// file. Resolvers, including Go's own resolver, read it by154// convention.155//156// The boot writes the file once, from the pass one connections. A157// radio that settles late renders it again through this same158// function, so its nameservers join the file under the same order159// and the same cap as every other interface's. That late rewrite160// happens while k3s and the kubelet run, and a pod sandbox created161// mid-write copies whatever the file holds, which is why the write162// below is atomic.163func writeResolvConf(conns []*connection) error {164	content := resolvConf(conns)165	if content == "" {166		return nil167	}168	return writeResolvConfAtomic(resolvConfPath, []byte(content))169}170171// writeResolvConfAtomic writes one file through a temp file in the172// same directory and a rename, the pattern machine/staging.go's173// writeAtomic uses for the facts tree.174//175// A rename inside one filesystem replaces the file in a single step,176// so a reader opens either the old content or the new, never an177// empty or partial file. A plain WriteFile truncates first, and a178// reader in that window gets an empty resolv.conf for the life of179// its copy.180func writeResolvConfAtomic(path string, raw []byte) error {181	tmp, err := os.CreateTemp(filepath.Dir(path), ".liken-*")182	if err != nil {183		return fmt.Errorf("writing %s: %w", path, err)184	}185	defer os.Remove(tmp.Name())186	if _, err := tmp.Write(raw); err != nil {187		tmp.Close()188		return fmt.Errorf("writing %s: %w", path, err)189	}190	if err := tmp.Close(); err != nil {191		return fmt.Errorf("writing %s: %w", path, err)192	}193	// CreateTemp makes the file owner-only, and every resolver on194	// the machine reads this file, so the mode is set before the195	// rename publishes it.196	if err := os.Chmod(tmp.Name(), 0o644); err != nil {197		return fmt.Errorf("writing %s: %w", path, err)198	}199	return os.Rename(tmp.Name(), path)200}201202// resolvConf renders the machine's resolv.conf file from its203// connections' nameservers. It includes nameservers from DHCP leases204// and from manifest declarations, in interface order. It removes205// duplicates and keeps at most three nameservers.206//207// Three is the oldest hard limit in the resolver world. Since the208// 1980s, glibc has read at most MAXNS=3 nameservers, and other libc209// stacks follow the same limit. Kubernetes also truncates every210// pod's nameserver list to three, and it logs a warning at each sync211// when a node's file offers more.212//213// Some networks do offer more than three nameservers. For example,214// Linode's DHCP service hands out its whole regional fleet: eighteen215// resolvers in one lease. Writing all eighteen would not change how216// names resolve, and it would cost a warning every minute, forever.217// Interface order is priority order, so the cap of three keeps the218// same resolvers that the machine would consult anyway.219func resolvConf(conns []*connection) string {220	const maxNameservers = 3221	var b strings.Builder222	seen := map[string]bool{}223	for _, conn := range conns {224		for _, ns := range conn.nameservers {225			if len(seen) == maxNameservers || seen[ns.String()] {226				continue227			}228			seen[ns.String()] = true229			fmt.Fprintf(&b, "nameserver %s\n", ns)230		}231	}232	return b.String()233}234235// The netlink calls the passes make, as variables so a test can run236// the whole bring-up with no kernel, including a raise that never237// returns. The link list comes through listLinks (report.go), the238// same seam the hardware report reads.239var (240	linkByName = netlink.LinkByName241	linkSetUp  = netlink.LinkSetUp242	addrAdd    = netlink.AddrAdd243	routeAdd   = netlink.RouteAdd244)245246// requestLease is the DHCP exchange, a variable holding acquireLease247// so a test can hand the addressing a lease with no wire and no248// server.249var requestLease = acquireLease250251// bringUpInterface raises one link and gives it an address, using252// the method that the interface spec chose. Only pass one calls it:253// a wired raise returns in milliseconds, so it runs with no deadline254// machinery, and the wireless half of what this function once did255// lives in bringUpRadio (radios.go).256func bringUpInterface(ifc machine.InterfaceSpec, present []interfaceIdentity) (*connection, error) {257	if err := requirePort(ifc.Name, present); err != nil {258		return nil, err259	}260	link, err := linkByName(ifc.Name)261	if err != nil {262		return nil, fmt.Errorf("opening interface %q: %w", ifc.Name, err)263	}264	fmt.Printf("liken: bringing up %s\n", ifc.Name)265	if err := linkSetUp(link); err != nil {266		return nil, fmt.Errorf("raising %s: %w", ifc.Name, err)267	}268	return addressInterface(link, ifc)269}270271// addressInterface gives one raised link its address, by the method the272// interface spec chose. It is the half of the bring-up that a wireless273// interface reaches only after its radio associates.274func addressInterface(link netlink.Link, ifc machine.InterfaceSpec) (*connection, error) {275	if ifc.Address != "" {276		return applyStatic(link, ifc)277	}278	fmt.Printf("liken: negotiating DHCP on %s\n", ifc.Name)279	lease, err := requestLease(ifc.Name)280	if err != nil {281		return nil, err282	}283	return applyLease(link, lease, ifc)284}285286// anyAddressed reports whether any interface came up with an address.287// An interface that exists with no address is a report, not a path.288func anyAddressed(conns []*connection) bool {289	for _, conn := range conns {290		if conn.addr != nil {291			return true292		}293	}294	return false295}296297// readdressRadio addresses the interface behind a radio that joined298// after the park released the boot. It replaces the addressless299// connection that the failed join left behind.300func readdressRadio(conns []*connection, interfaces []machine.InterfaceSpec, r *radio) []*connection {301	if r.state != machine.WirelessConnected {302		return conns303	}304	var spec machine.InterfaceSpec305	for _, ifc := range interfaces {306		if ifc.Name == r.ifname {307			spec = ifc308		}309	}310	link, err := linkByName(r.ifname)311	if err != nil {312		fmt.Fprintf(os.Stderr, "liken: network: %s: %v\n", r.ifname, err)313		return conns314	}315	conn, err := addressInterface(link, spec)316	if err != nil {317		fmt.Fprintf(os.Stderr, "liken: network: %s: %v\n", r.ifname, err)318		return conns319	}320	conn.radio = r321	for i, existing := range conns {322		if existing.ifname == r.ifname {323			conns[i] = conn324			return conns325		}326	}327	return append(conns, conn)328}329330// interfaceIdentity is one link reduced to the two facts the boot331// needs about it: the name the kernel gave it, and whether the link332// carries a hardware address, which is how a real card is told from a333// virtual device. The functions below take these instead of netlink334// links, so the rules a manifest meets are pure functions that a test335// can drive on any machine.336type interfaceIdentity struct {337	name string338	mac  net.HardwareAddr339}340341// presentInterfaces reduces the kernel's link list to the ports a342// manifest can configure. Loopback is dropped: it is not hardware,343// the boot has already raised it, and leaving it out keeps it out of344// the error messages that list what a machine has.345func presentInterfaces(links []netlink.Link) []interfaceIdentity {346	var present []interfaceIdentity347	for _, link := range links {348		attrs := link.Attrs()349		if attrs.Flags&net.FlagLoopback != 0 {350			continue351		}352		present = append(present, interfaceIdentity{name: attrs.Name, mac: attrs.HardwareAddr})353	}354	return present355}356357// requirePort checks that the machine has the port a manifest names,358// and reports what the machine does have when it does not.359//360// netlink would refuse the name on its own, with the kernel's own361// words. Those words say that a link is missing and stop there, and362// the whole value of this check is the sentence after that: nobody363// can see this machine, which has no shell and no SSH daemon, so the364// console message is the entire diagnosis. A message that names the365// ports the machine really has turns a drive to the site into an edit366// of the manifest.367func requirePort(name string, present []interfaceIdentity) error {368	for _, port := range present {369		if port.name == name {370			return nil371		}372	}373	return fmt.Errorf("no interface is named %s; this machine has %s", name, describePorts(present))374}375376// describePorts lists the ports a person could have named, for the377// errors that report a name no port answers to.378func describePorts(present []interfaceIdentity) string {379	if len(present) == 0 {380		return "no network interface at all"381	}382	described := make([]string, len(present))383	for i, port := range present {384		described[i] = port.name385	}386	return strings.Join(described, ", ")387}388389// pickInterface finds the hardware to configure when the manifest390// names no interface. The rule is simple: the code picks the first391// port that has a MAC address. Such a link looks like a real network392// card. On the hardware that this default serves, one machine with393// one port, there is nothing to choose between. When there is more394// than one interface, the manifest must say which ports to configure.395func pickInterface(present []interfaceIdentity) (string, error) {396	for _, port := range present {397		if len(port.mac) > 0 {398			return port.name, nil399		}400	}401	return "", fmt.Errorf("no network interface found: none of the %d links outside loopback carries a hardware address", len(present))402}403404// applyStatic sets a declared address in the kernel. This produces405// the same kernel state that a DHCP ACK produces, without the406// negotiation. The prefix length inside the CIDR tells the kernel407// which destinations are neighbors on this link, and which408// destinations are beyond the gateway.409func applyStatic(link netlink.Link, ifc machine.InterfaceSpec) (*connection, error) {410	addr, err := netlink.ParseAddr(ifc.Address)411	if err != nil {412		return nil, fmt.Errorf("address %q: %w", ifc.Address, err)413	}414	if err := addrAdd(link, addr); err != nil {415		return nil, fmt.Errorf("assigning %s: %w", ifc.Address, err)416	}417418	conn := &connection{419		ifname: link.Attrs().Name,420		mac:    link.Attrs().HardwareAddr,421		addr:   addr.IPNet,422		method: machine.MethodStatic,423	}424425	if ifc.Gateway != "" {426		gw := net.ParseIP(ifc.Gateway)427		if gw == nil {428			return nil, fmt.Errorf("gateway %q is not an IP address", ifc.Gateway)429		}430		conn.gateway = gw431		route := &netlink.Route{LinkIndex: link.Attrs().Index, Gw: gw}432		if err := routeAdd(route); err != nil {433			return nil, fmt.Errorf("default route via %s: %w", gw, err)434		}435	}436437	for _, raw := range ifc.Nameservers {438		ns := net.ParseIP(raw)439		if ns == nil {440			return nil, fmt.Errorf("nameserver %q is not an IP address", raw)441		}442		conn.nameservers = append(conn.nameservers, ns)443	}444	return conn, nil445}446447// acquireLease runs the DHCP exchange. The library handles448// DISCOVER, OFFER, REQUEST, and ACK, and it handles their retries.449// The code bounds the whole exchange with a deadline. A boot that450// hangs forever on a dead network is worse than a boot that reports451// failure and continues to the console.452//453// The summary logger prints each packet of the exchange to the454// console. On a machine with no shell, the boot log is the only455// record of these packets.456//457// A note on entropy: the DHCP client draws a random transaction ID458// with getrandom(2). This call blocks until the kernel's random459// number generator has initialized, and the block is uninterruptible:460// the context deadline cannot stop it. On hardware with no entropy461// source, such as QEMU's default CPU model, which lacks RDRAND, the462// call blocks forever. The host must supply entropy: RDRAND,463// virtio-rng, or enough time for the kernel's jitter collector to464// gather entropy on its own.465func acquireLease(ifname string) (*nclient4.Lease, error) {466	client, err := nclient4.New(ifname, nclient4.WithSummaryLogger())467	if err != nil {468		return nil, fmt.Errorf("opening DHCP socket on %s: %w", ifname, err)469	}470	defer client.Close()471472	ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)473	defer cancel()474475	lease, err := client.Request(ctx)476	if err != nil {477		return nil, fmt.Errorf("DHCP on %s: %w", ifname, err)478	}479	return lease, nil480}481482// applyLease turns the DHCP ACK into kernel state. The code adds the483// address to the link and sets the router as the default route.484// Nameservers come from the lease, plus any nameservers that the485// manifest adds. They reach /etc/resolv.conf together with every486// other interface's nameservers, in bringUpNetwork.487func applyLease(link netlink.Link, lease *nclient4.Lease, ifc machine.InterfaceSpec) (*connection, error) {488	ack := lease.ACK489490	addr := &net.IPNet{IP: ack.YourIPAddr, Mask: ack.SubnetMask()}491	if err := addrAdd(link, &netlink.Addr{IPNet: addr}); err != nil {492		return nil, fmt.Errorf("assigning %s: %w", addr, err)493	}494495	leaseTime := ack.IPAddressLeaseTime(0)496	conn := &connection{497		ifname:      link.Attrs().Name,498		mac:         link.Attrs().HardwareAddr,499		addr:        addr,500		method:      machine.MethodDHCP,501		nameservers: ack.DNS(),502		leaseTime:   leaseTime,503		// The clock is read once, here at the ACK; see the field's504		// comment for why the expiry must not be derived later.505		leaseExpires: time.Now().Add(leaseTime),506		server:       ack.ServerIdentifier(),507	}508	for _, raw := range ifc.Nameservers {509		if ns := net.ParseIP(raw); ns != nil {510			conn.nameservers = append(conn.nameservers, ns)511		}512	}513514	if routers := ack.Router(); len(routers) > 0 {515		conn.gateway = routers[0]516		route := &netlink.Route{517			LinkIndex: link.Attrs().Index,518			Gw:        conn.gateway,519		}520		if err := routeAdd(route); err != nil {521			return nil, fmt.Errorf("default route via %s: %w", conn.gateway, err)522		}523	}524525	return conn, nil526}527528func (c *connection) report() {529	if c.radio != nil {530		fmt.Printf("liken: %s (%s) is on %s (%s)\n", c.ifname, c.mac, c.radio.ssid, c.radio.state)531		if c.radio.message != "" {532			fmt.Printf("liken:   %s\n", c.radio.message)533		}534	}535	if c.addr == nil {536		fmt.Printf("liken: %s (%s) has no address\n", c.ifname, c.mac)537		return538	}539	fmt.Printf("liken: %s (%s) is %s (%s)\n", c.ifname, c.mac, c.addr, strings.ToLower(string(c.method)))540	if c.method == machine.MethodDHCP {541		fmt.Printf("liken:   gateway %s, dhcp server %s, lease %s\n",542			c.gateway, c.server, c.leaseTime)543	} else if c.gateway != nil {544		fmt.Printf("liken:   gateway %s\n", c.gateway)545	}546	if len(c.nameservers) > 0 {547		fmt.Printf("liken:   nameservers %s\n", joinIPs(c.nameservers))548	}549}550551func joinIPs(ips []net.IP) string {552	strs := make([]string, len(ips))553	for i, ip := range ips {554		strs[i] = ip.String()555	}556	return strings.Join(strs, ", ")557}
init/podlogs.go 89.5%
1package main23// Pod logs, and the filesystem their bytes land on.4//5// kubelet writes every container's stdout and stderr to a file under6// /var/log/pods: one directory per pod, one file per container. That7// path is a convention that a great deal outside liken depends on. A8// log collector mounts it by hostPath, the symlinks in9// /var/log/containers resolve into it, and every runbook names it.10// kubelet does take a podLogsDir setting, and liken deliberately does11// not use it. A node whose logs are somewhere else is a node that12// every tool and every operator has to be taught about.13//14// So the path stays, and the filesystem under it changes.15//16// liken's root is a read-only squashfs with a small tmpfs upper layer17// for writes. switchroot.go states that budget and why it must stay18// small: the runtime's writes under / are few and fixed, and anything19// that grows with use belongs to a disk role. Pod logs are exactly the20// opposite kind of writer. They append for as long as the pods run.21// kubelet bounds one container's logs with containerLogMaxSize and22// containerLogMaxFiles, 10Mi and 5 files by default, and it bounds23// nothing about their sum. A few talkative containers fill the root's24// whole write budget, and on a machine with no shell there is no way25// to clean that up.26//27// A bind mount moves the bytes without moving the path. The28// directory that holds them lives on podEphemeral, the role that29// already holds kubelet's working space, and it appears at30// /var/log/pods. Every reader still finds the logs where it expects31// them. This also puts the logs on the filesystem that kubelet32// already measures: kubelet's nodefs is the filesystem of its root33// directory, so the space these logs consume is space that kubelet's34// own disk-pressure eviction can see and act on. liken sets no35// eviction thresholds of its own, and kubelet's defaults apply.3637import (38	"fmt"39	"os"40	"path/filepath"4142	"golang.org/x/sys/unix"4344	"github.com/liken-sh/liken/liken/machine"45)4647// podLogsDir is where kubelet writes container logs. This is the48// canonical Kubernetes path, and liken keeps it. It is a variable only49// so that a test can bind into a temporary directory.50var podLogsDir = "/var/log/pods"5152// bindPodLogDir is the bind mount itself. It goes through53// mountFilesystem, so a test can check what bindPodLogs binds without54// the privilege to mount.55func bindPodLogDir(source, target string) error {56	return mountFilesystem(source, target, "", unix.MS_BIND, "")57}5859// podLogsSubdir is where the logs live on the podEphemeral60// filesystem. kubelet owns every other directory under its root, so61// the logs sit under a directory named for liken, which nothing can62// mistake for something kubelet manages. The k3s and containerd logs63// sit in such a directory on clusterState for the same reason64// (supervisor.go).65var podLogsSubdir = filepath.Join("liken", "pod-logs")6667// podLogBind is one bind mount: a directory on a role's filesystem,68// and the canonical path it appears at.69type podLogBind struct {70	source string71	target string72}7374// planPodLogBind determines whether this machine binds its pod logs75// onto a disk, and says which directory it binds.76//77// A machine that does not declare podEphemeral gets no bind. Its78// /var/lib/kubelet is on the root overlay as well, so a bind would79// carry bytes from one overlay directory to another and add a mount80// that claims a separation the machine does not have. A mount table81// that describes something untrue is worse than no mount, so this82// decision is to skip, and the caller says so on the console.83//84// The source path comes from the role's own mount translation rather85// than from a second copy of that path, so a role that moves takes86// its pod logs with it.87func planPodLogBind(storage machine.StorageStatus) *podLogBind {88	if storage.PodEphemeral.Backing != machine.BackingPartition {89		return nil90	}91	return &podLogBind{92		source: filepath.Join(roleMounts[machine.PodEphemeralRole].path, podLogsSubdir),93		target: podLogsDir,94	}95}9697// bindPodLogs actuates that decision. It runs after prepareForK3s,98// which creates /var/log and makes / rshared, and before k3s starts,99// so the mount is in place before kubelet opens its first log file.100// The bind inherits the shared propagation that prepareForK3s set, so101// a collector's hostPath mount of /var/log/pods sees this filesystem102// and not the empty directory underneath it.103//104// A failure here is reported and never fatal. Pod logs on the root105// overlay are a machine with a bounded write budget for them, not a106// machine that must refuse to run.107func bindPodLogs(storage machine.StorageStatus) {108	bind := planPodLogBind(storage)109	if bind == nil {110		fmt.Printf("liken: pod logs: this machine declares no podEphemeral role, so %s stays on the root filesystem\n", podLogsDir)111		return112	}113	if err := os.MkdirAll(bind.source, 0o755); err != nil {114		fmt.Fprintf(os.Stderr, "liken: pod logs: creating %s: %v\n", bind.source, err)115		return116	}117	if err := os.MkdirAll(bind.target, 0o755); err != nil {118		fmt.Fprintf(os.Stderr, "liken: pod logs: creating %s: %v\n", bind.target, err)119		return120	}121	if err := bindPodLogDir(bind.source, bind.target); err != nil {122		fmt.Fprintf(os.Stderr, "liken: pod logs: binding %s onto %s: %v\n", bind.source, bind.target, err)123		return124	}125	fmt.Printf("liken: pod logs: %s is %s, on podEphemeral\n", bind.target, bind.source)126}127128// unmountPodLogs detaches the bind. It runs before the roles129// themselves come down, because the bind holds a second reference to130// podEphemeral's filesystem: unmounting /var/lib/kubelet with the131// bind still in place detaches that mount point and leaves the disk132// in use. Detaching the bind first lets the role's own unmount133// release the filesystem.134//135// The caller passes the same flags and reporting choice it gives the136// role unmounts, so both shutdown paths treat this mount exactly as137// they treat the disks under it (storage.go explains the two paths).138func unmountPodLogs(flags int, reportErrors bool) {139	detachMount(podLogsDir, flags, reportErrors)140}
init/proving.go 77.6%
1package main23// The proving boot: how a downloaded release becomes the running4// release.5//6// The operator stages a SystemRelease record (machine/systemrelease.go)7// when a verified release is waiting on the inactive slot. From that8// point, this file carries the upgrade across the reboot. Everything9// in this file is firmware-neutral. The three actions that touch the10// machine's actual boot mechanism go through a bootActuator11// (actuator.go), which carries the firmware's own dialect.12//13//   - Before the reboot (armProvingBoot, called from init's reboot14//     path), init writes the attempted marker and arms the15//     actuator's one-shot trial at the staged slot. The one-shot16//     mechanism is what makes blue-green upgrades safe. The boot17//     that the arming triggers consumes the arming, so the trial18//     gets exactly one chance. Any reset after that chance, such as19//     a kernel panic, a watchdog reset, or a power cut, falls back20//     to the standing preference, which still prefers the proven21//     slot.22//23//   - After the reboot (settleSystemRelease), init reads the files24//     and determines what happened. If this boot came from the staged25//     slot, this boot is the trial. The operator's first reconcile26//     shows that the new kernel, init, k3s, and the cluster join all27//     work, and this is what promotes the release. If this boot came28//     from the proven slot with the attempted marker set, the trial29//     ran and the firmware fell back. In this case the record is30//     rejected durably, and the operator will not stage that exact31//     record again.32//33//   - After promotion (provingWatch, a machine-plane component that34//     runs only on proving boots), init asserts the standing35//     preference so the newly proven slot leads. Init owns this36//     write for the same reason it owns every firmware write:37//     talking to the firmware is the machine plane's job, and the38//     store on disk is the authority. assertProvenSlot re-asserts39//     the proven record's slot on every boot. So, if a power cut40//     happens between promotion and the assertion, the machine costs41//     one extra boot of the old version, and never keeps a wrong42//     preference.4344import (45	"context"46	"crypto/sha256"47	"encoding/hex"48	"fmt"49	"os"50	"path/filepath"51	"time"5253	"github.com/liken-sh/liken/liken/machine"54)5556// settleSystemRelease reads the system store at boot and determines57// the verdict that only init can determine: is this boot a trial, or58// is it the fallback from a trial? When this boot proves a staged59// release, the function returns that release's record. This record is60// what arms the proving watch. Every other verdict returns nil. This61// function runs after storage settles (the store lives on62// machineState) and after main resolves the boot's slot.63func settleSystemRelease(act bootActuator, stateRoot, bootSlot string, durable bool, boot *machine.BootStatus) *machine.SystemRelease {64	if !durable {65		return nil66	}67	store := machine.SystemReleases(stateRoot)6869	// The code republishes the standing rejection into the boot70	// record on every boot. rejectStagedDocument explains why.71	boot.SystemRejection, _ = store.LoadRejection()7273	staged, err := store.LoadStaged()74	if err != nil {75		fmt.Fprintf(os.Stderr, "liken: system: the staged release record is unreadable: %v\n", err)76	}77	if staged == nil {78		assertProvenSlot(act, stateRoot)79		return nil80	}8182	record, perr := machine.ParseSystemRelease(staged)83	if perr != nil {84		boot.SystemRejection = rejectStagedDocument("system", "release", store.Reject,85			staged, fmt.Sprintf("the staged release record does not parse: %v", perr))86		assertProvenSlot(act, stateRoot)87		return nil88	}8990	// A boot that did not come from a slot cannot judge a slot91	// trial. In this case, external media is running, not either92	// half of the blue-green pair.93	if bootSlot == "" {94		fmt.Printf("liken: system: release %s is staged for slot %s, but this boot came from external media; leaving it for a from-disk boot\n",95			record.Version, record.Slot)96		return nil97	}9899	if record.Slot == bootSlot {100		// This boot is the trial. The previous boot armed BootNext at101		// this slot, and this boot runs the staged release's own102		// kernel and init. The code writes nothing here, because the103		// attempted marker already stands. init also does not judge104		// the proof. The operator's first reconcile is what shows105		// that the machine actually serves its cluster on the new106		// release.107		fmt.Printf("liken: system: proving boot: release %s on slot %s; the operator's first pass is the proof\n",108			record.Version, record.Slot)109		return record110	}111112	hash := machine.ManifestHash(staged)113	if attempted, _ := store.LoadAttempted(); attempted == hash {114		// The trial ran, and the firmware brought the machine back115		// to the proven slot. The release used its one chance and116		// did not get promoted. This is the fallback verdict. The117		// code records it durably so that no later boot re-arms the118		// same trial.119		boot.SystemRejection = rejectStagedDocument("system", "release", store.Reject,120			staged, fmt.Sprintf("the machine booted slot %s to prove release %s and fell back to slot %s; the release never proved out",121				record.Slot, record.Version, bootSlot))122		assertProvenSlot(act, stateRoot)123		return nil124	}125126	fmt.Printf("liken: system: release %s is staged for slot %s, awaiting its proving reboot\n",127		record.Version, record.Slot)128	assertProvenSlot(act, stateRoot)129	return nil130}131132// armProvingBoot runs on the reboot path. If a release is staged for133// the other slot, the function marks it attempted and arms the134// actuator's one-shot trial. Only the next boot tries the staged135// release.136//137// The code writes the attempted marker first, deliberately. If a138// crash happens between the two writes, the boot record reads as139// "tried and fell back". This wrongly rejects a release that never140// ran. But the opposite write order could arm a trial with no141// marker. A failing release would then re-arm itself and reboot142// forever. An operator can fix a false rejection by editing the143// store, but nothing can fix a reboot loop. So the write order144// favors the false rejection over the reboot loop.145//146// canArmTrial runs before either write, for the same reason: the147// code must find anything that is knowably wrong while refusing the148// trial still costs nothing.149func armProvingBoot(act bootActuator, stateRoot, runningSlot string) {150	store := machine.SystemReleases(stateRoot)151	staged, err := store.LoadStaged()152	if staged == nil || err != nil {153		return154	}155	record, err := machine.ParseSystemRelease(staged)156	if err != nil {157		return // boot-time vetting owns this verdict158	}159	if record.Slot == runningSlot || record.Slot == "" {160		return161	}162163	// The running slot and the proven slot are not always the same164	// one. A person can pick the other slot from the firmware's own165	// menu, and a firmware that cannot load the proven slot's entry166	// falls through to the other. Either way, this boot runs a slot167	// that the store does not call proven, and the check above, which168	// compares against the running slot, is satisfied by a trial169	// staged for the proven slot itself. Such a trial must be refused:170	// it would put an unproven release on the one slot the fallback171	// has to come from, and the machine would then have a fallback172	// nowhere.173	if proven := provenRecord(stateRoot); proven != nil && record.Slot == proven.Slot {174		fmt.Fprintf(os.Stderr, "liken: system: release %s is staged for slot %s, which is the proven slot; rebooting without arming the trial\n",175			record.Version, record.Slot)176		return177	}178179	if err := checkStagedSlot(slotMountPath(record.Slot), record); err != nil {180		fmt.Fprintf(os.Stderr, "liken: system: slot %s does not hold release %s as staged: %v; rebooting without arming the trial\n",181			record.Slot, record.Version, err)182		return183	}184185	if err := act.canArmTrial(record.Slot); err != nil {186		fmt.Fprintf(os.Stderr, "liken: system: %v; rebooting without arming the trial\n", err)187		return188	}189190	// The code asserts the fallback, and verifies it, before it arms191	// the trial. A trial is only safe when every reset after its one192	// chance lands on a proven slot. The standing preference is what193	// guarantees this. If the firmware does not hold the preference,194	// or if the assertion quietly fails, the fallback does not195	// exist. Arming a trial in that case could make the machine196	// permanently unusable, through its own upgrade. So the code197	// refuses the trial, visibly, and the machine stays on the198	// version that works, where an operator can still reach it.199	if !fallbackInPlace(act, stateRoot) {200		fmt.Fprintln(os.Stderr, "liken: system: the proven fallback could not be verified; refusing to arm the trial")201		return202	}203204	if err := store.WriteAttempted(machine.ManifestHash(staged)); err != nil {205		fmt.Fprintf(os.Stderr, "liken: system: marking the trial attempted: %v; rebooting without arming it\n", err)206		return207	}208	armed, err := act.armTrial(record.Slot)209	if err != nil {210		fmt.Fprintf(os.Stderr, "liken: system: %v\n", err)211		return212	}213	fmt.Printf("liken: system: %s; the next boot tries release %s on slot %s, once\n",214		armed, record.Version, record.Slot)215}216217// checkStagedSlot reports whether a slot holds exactly the release218// that a staged record names. The record is written once, when the219// download verifies, and the slot can change after that: a new220// target starts a download of another release onto the same slot,221// and that download can stop halfway. A trial of such a slot boots222// the files of two releases under the record of one. So the reboot223// path reads the slot again before it arms: the document's bytes224// must hash to the record's digest, and every artifact must match225// the document. The check reads every artifact, a few hundred226// megabytes, which costs seconds on the way down, once for each227// trial.228func checkStagedSlot(mount string, record *machine.SystemRelease) error {229	if mount == "" {230		return fmt.Errorf("the slot is not mounted")231	}232	raw, err := os.ReadFile(filepath.Join(mount, "release.yaml"))233	if err != nil {234		return fmt.Errorf("the slot carries no release document: %w", err)235	}236	sum := sha256.Sum256(raw)237	if digest := "sha256:" + hex.EncodeToString(sum[:]); digest != record.ReleaseDigest {238		return fmt.Errorf("the slot's release document has digest %s, and the record names %s", digest, record.ReleaseDigest)239	}240	return verifySlotContents(mount)241}242243// fallbackInPlace asserts the standing preference at the proven244// slot. Then it checks only what the actuator can read back. A245// write that appears to succeed is not the same as proof that the246// firmware actually holds it.247func fallbackInPlace(act bootActuator, stateRoot string) bool {248	record := provenRecord(stateRoot)249	if record == nil {250		return false251	}252	act.assertProven(record.Slot)253	return act.fallbackLeads(record.Slot)254}255256// provingPatience is the length of time a proving boot may run257// unpromoted before the watchdog reboots it. This is the same ten258// minutes that the rollout conductor waits before it calls a reboot259// stalled. A machine that has not joined its cluster after ten260// minutes is stuck, not merely slow.261const provingPatience = 10 * time.Minute262263// provingWatch builds the machine-plane component that a proving264// boot runs. This component closes over the record that this boot265// is trying. The watch compares the store's proven record against266// the trial's own record. The trial is only over when the code has267// promoted this exact record. A staged file that merely disappears268// was withdrawn, not promoted. Treating its absence as promotion269// would flip BootOrder for a promotion that never happened.270//271// The loop combines two fallback paths. The first is the ordinary272// path: the code polls the store until the operator promotes the273// staged record (the record's disappearance is the signal), then it274// asserts the standing preference so the newly proven slot leads.275// The second path is a watchdog: when a trial reaches276// provingPatience unpromoted, the code forces a deliberate reboot.277// The one-shot arming is already consumed, so the machine lands on278// the proven slot. There, the attempted marker sets the verdict279// (RejectedLastBoot) and prevents any reboot loop. The watchdog280// covers a failure that the one-shot cannot detect: a kernel that281// boots fine into a system that never serves.282func provingWatch(act bootActuator, trial machine.SystemRelease) func(context.Context) error {283	return func(ctx context.Context) error {284		deadline := time.After(provingPatience)285		for {286			select {287			case <-ctx.Done():288				return nil289			case <-deadline:290				fmt.Fprintf(os.Stderr, "liken: system: the proving boot never settled within %s; rebooting onto the proven slot\n", provingPatience)291				rebootMachine(machine.RebootIntent{292					Reason: "the proving boot never settled; the consumed one-shot falls back to the proven slot",293				}) // never returns294			case <-time.After(5 * time.Second):295			}296			store := machine.SystemReleases(machine.MachineStateDir)297			staged, err := store.LoadStaged()298			if err != nil || staged != nil {299				continue300			}301			// The staged file is gone, but gone is not the same as302			// promoted. Only a proven record that matches this303			// boot's own trial counts as the verdict. Anything else304			// means that someone withdrew the record while the305			// trial was still running, most likely because of a306			// retargeted cluster. In that case, the firmware's boot307			// order must not change for a promotion that never308			// happened.309			proven, err := store.LoadProven()310			if err != nil || proven == nil {311				continue312			}313			record, err := machine.ParseSystemRelease(proven)314			if err != nil || record.Version != trial.Version || record.Slot != trial.Slot {315				fmt.Printf("liken: system: the trial of release %s was withdrawn without promotion; leaving BootOrder alone\n",316					trial.Version)317				return nil318			}319			fmt.Printf("liken: system: release %s was promoted; asserting the proven slot from the store\n", record.Version)320			assertProvenSlotUnderLock(act, machine.MachineStateDir)321			return nil322		}323	}324}325326// assertProvenSlotUnderLock is the proving watch's way to the327// firmware. The watch runs beside the rest of the boot until the328// trial settles, so its assertion is the one write that can collide329// with the reboot path, and the comment on firmwareWrites in330// actuator.go says what that collision costs. The shutdown test331// happens under the lock, not before it: a watch that waited out the332// reboot path's turn must see the flag that turn set, and testing333// before the lock would let it pass the test, wait, and then undo the334// trial the reboot path armed.335func assertProvenSlotUnderLock(act bootActuator, stateRoot string) {336	firmwareWrites.Lock()337	defer firmwareWrites.Unlock()338	if shuttingDown.Load() {339		return340	}341	assertProvenSlot(act, stateRoot)342}343344// assertAndArmForReboot is the reboot path's whole firmware turn,345// taken in one held lock. The two calls belong together: a trial is346// only safe when the fallback under it was asserted first, so nothing347// may write between them. The unlock is deferred because this runs348// inside PID 1 on the way down, where a panic is recovered rather349// than fatal; a panic that escaped with the lock held would leave350// every later reboot blocked on it (firmwareWrites in actuator.go).351func assertAndArmForReboot(act bootActuator, stateRoot, runningSlot string) {352	firmwareWrites.Lock()353	defer firmwareWrites.Unlock()354	assertProvenSlot(act, stateRoot)355	armProvingBoot(act, stateRoot, runningSlot)356}357358// assertProvenSlot sets the firmware to match the store: the proven359// record's slot leads the standing boot preference. This function360// runs on every boot and after every promotion. So the store stays361// the authority, and the firmware only ever holds a copy of it. The362// code corrects anything that drifted, such as a dead NVRAM battery363// or someone editing the setup menu, on the next boot. Reading the364// record is plain work that this function does directly. Making the365// firmware match the record is the actuator's job.366func assertProvenSlot(act bootActuator, stateRoot string) {367	record := provenRecord(stateRoot)368	if record == nil {369		return // no record yet; leave the firmware's preference as it is370	}371	act.assertProven(record.Slot)372}373374// provenRecord loads and parses the store's proven release record.375// It reports the ways that a record can fail to exist. An unreadable376// or unparseable record prints a console line, because a machine377// with a damaged proven record has lost its fallback. An absent378// record is ordinary; it means the machine has never upgraded.379func provenRecord(stateRoot string) *machine.SystemRelease {380	proven, err := machine.SystemReleases(stateRoot).LoadProven()381	if err != nil {382		fmt.Fprintf(os.Stderr, "liken: system: the proven release record is unreadable: %v\n", err)383		return nil384	}385	if proven == nil {386		return nil387	}388	record, err := machine.ParseSystemRelease(proven)389	if err != nil {390		fmt.Fprintf(os.Stderr, "liken: system: the proven release record does not parse: %v\n", err)391		return nil392	}393	return record394}
init/quiesce.go 75.7%
1package main23// Leaving the disks in a finished state.4//5// A filesystem records whether it was shut down properly, and it6// records it at one moment: when the kernel releases its superblock.7// vfat clears the dirty bit in the boot sector there. ext4 flushes the8// journal and clears the flag that tells the next mount to replay it.9// Until that release happens, every disk on the machine still says10// that it is in use, and the next boot reads that as a crash.11//12// Unmounting is the ordinary way to reach that release, and two disks13// here are never free to be unmounted. The running root is a squashfs14// image that lives as a file on the booting slot, attached through a15// loop device, so that slot stays busy for as long as the machine16// runs. clusterState stays busy because containerd's overlay mounts17// name directories inside it, and those outlive the containers they18// belonged to. The shutdown detaches both lazily and reboots, which19// takes the mount out of the table and leaves the busy superblock20// unreleased and its record unwritten.21//22// A read-only remount is the operation that fits a busy filesystem. It23// waits for no reference to drop: it flushes the filesystem and writes24// the clean record while the mount stays where it is. So the shutdown25// remounts every writable disk read-only first, and unmounts26// afterwards. After the remount, a disk is correct on its own, and27// whether its unmount can finish no longer determines what the next28// boot reads.29//30// This runs after the last write. On the reboot path that means after31// the boot actuator has asserted the proven slot and armed any trial,32// because both of those write to the slots and to machineState.33//34// Getting this wrong once on a FAT filesystem is permanent. When vfat35// mounts a volume whose dirty bit is already set, it stops managing36// that bit for the life of the mount: it does not clear the bit at37// unmount, and a read-only remount does not clear it either. The bit38// is meant to survive until a person runs fsck, so one unclean stop39// leaves a slot reporting a crash on every boot after it, however40// cleanly the machine stops from then on. A volume that has never been41// stopped uncleanly stays clean, which is why this runs on every path42// that ends a boot rather than only on the one that reboots.4344import (45	"errors"46	"fmt"47	"io/fs"48	"os"49	"strconv"50	"strings"5152	"golang.org/x/sys/unix"53)5455// diskMount is one filesystem that quiesceDisks must finish: where it56// is mounted, and the flags to carry across the remount.57type diskMount struct {58	target string59	flags  uintptr60}6162// mountOptionFlags maps the option names the kernel prints in63// /proc/self/mounts back to the flags that set them. A remount64// replaces the whole flag word rather than adding to it, so an option65// that is not named again is switched off. Only the flags that the66// kernel reports per mount belong here; the rest are filesystem67// options, which a remount keeps on its own.68var mountOptionFlags = map[string]uintptr{69	"nosuid":     unix.MS_NOSUID,70	"nodev":      unix.MS_NODEV,71	"noexec":     unix.MS_NOEXEC,72	"noatime":    unix.MS_NOATIME,73	"nodiratime": unix.MS_NODIRATIME,74	"relatime":   unix.MS_RELATIME,75	"sync":       unix.MS_SYNCHRONOUS,76	"dirsync":    unix.MS_DIRSYNC,77	"lazytime":   unix.MS_LAZYTIME,78	"mand":       unix.MS_MANDLOCK,79}8081// writableDiskMounts reads a mount table and returns the mounts that82// have a clean record to write, newest first.83//84// A mount qualifies when it is backed by a block device and mounted85// read-write. Everything else is skipped for a reason: a tmpfs, an86// overlay, or one of the kernel's own filesystems has no disk behind87// it, and a filesystem that is already read-only wrote its clean88// record when it became read-only.89//90// The order is the reverse of the table, so a mount that covers91// another is handled before the one underneath it. A filesystem that92// answers to more than one path appears more than once, which is what93// the booting slot needs: it is mounted once for the early boot and94// once as its role, and either path reaches the same superblock.95func writableDiskMounts(mountTable string) []diskMount {96	lines := strings.Split(strings.TrimSpace(mountTable), "\n")97	mounts := make([]diskMount, 0, len(lines))98	for i := len(lines) - 1; i >= 0; i-- {99		fields := strings.Fields(lines[i])100		if len(fields) < 4 || !strings.HasPrefix(fields[0], "/dev/") {101			continue102		}103		options := strings.Split(fields[3], ",")104		var flags uintptr105		var writable bool106		for _, option := range options {107			if option == "rw" {108				writable = true109			}110			flags |= mountOptionFlags[option]111		}112		if !writable {113			continue114		}115		mounts = append(mounts, diskMount{target: unescapeMountField(fields[1]), flags: flags})116	}117	return mounts118}119120// unescapeMountField decodes the octal escapes that the kernel writes121// into a mount table. A space, a tab, a newline, and a backslash in a122// path each appear as a backslash and three octal digits, because the123// table itself is separated by spaces and newlines.124func unescapeMountField(field string) string {125	if !strings.Contains(field, `\`) {126		return field127	}128	var out strings.Builder129	for i := 0; i < len(field); i++ {130		if field[i] == '\\' && i+3 < len(field) {131			if b, err := strconv.ParseUint(field[i+1:i+4], 8, 8); err == nil {132				out.WriteByte(byte(b))133				i += 3134				continue135			}136		}137		out.WriteByte(field[i])138	}139	return out.String()140}141142// quiesceDisks makes every writable disk on the machine read-only, so143// that each one carries a clean record before the machine stops. It144// reports what it could not finish and returns either way: a shutdown145// that stops here would leave the machine running with nothing left to146// run it.147func quiesceDisks() {148	table, err := os.ReadFile("/proc/self/mounts")149	if err != nil {150		fmt.Fprintf(os.Stderr, "liken: storage: reading the mount table: %v\n", err)151		return152	}153	for _, m := range writableDiskMounts(string(table)) {154		err := unix.Mount("", m.target, "", unix.MS_REMOUNT|unix.MS_RDONLY|m.flags, "")155		switch {156		case err == nil:157			fmt.Printf("liken: storage: %s is read-only\n", m.target)158		case errors.Is(err, fs.ErrNotExist):159			// A remount reaches a filesystem through its path, and a160			// mount that another mount covers has no path left to name161			// it by. No liken mount covers another, so this is a162			// filesystem that something outside liken stacked on. It is163			// skipped rather than reported: there is no path to remount164			// it by, and the mount underneath may well be one of the165			// entries this loop has already finished.166		default:167			fmt.Fprintf(os.Stderr, "liken: storage: %s stays writable: %v\n", m.target, err)168		}169	}170}
init/radios.go 92.3%
1package main23// Interface bring-up runs in two passes: wired interfaces settle in4// line, and radios settle behind the boot5// (plans/completed/64-the-boot-does-not-wait-for-radios.md). The split6// exists because of one hard fact about the kernel: raising a link7// holds the rtnl lock while the driver's open routine runs, and a8// wedged driver never returns, holds the lock forever, and cannot be9// killed. On 2026-08-26 exactly that took a machine down. Its wired10// path was healthy, but the boot waited on the radio, so nothing that11// could act ever started, and every following boot repeated the wait.12// No timeout can contain a stuck kernel thread. What the boot controls13// is what it risks before the machine can act, so the radios go last,14// and on a machine that does not need them, they go in the background.1516import (17	"context"18	"errors"19	"fmt"20	"os"21	"sync/atomic"22	"time"2324	"github.com/vishvananda/netlink"2526	"github.com/liken-sh/liken/liken/cluster"27	"github.com/liken-sh/liken/liken/machine"28)2930// How long the boot waits for a raise to return. A healthy radio31// raises in under two seconds, most of it firmware load. A raise32// still out at ten seconds is a kernel thread that is not coming33// back, and the wait exists only so the boot can say so; waiting34// longer would report the same thing later.35const raisePatience = 10 * time.Second3637// interfacePasses is the pass split with its moving parts as38// function values, so a test can run the whole split, the route39// question included, with no kernel, no radio, and no supplicant.40// The seams follow startSupplicant and parkConsole: the real boot41// binds the real functions once, in bootPasses below.42type interfacePasses struct {43	wired     func(machine.InterfaceSpec) (*connection, error)44	radio     func(machine.InterfaceSpec) (*connection, error)45	route     routeLookup46	ciphers   func()47	readdress func([]*connection, []machine.InterfaceSpec, *radio) []*connection48}4950// bootPasses binds the passes a real boot runs.51func bootPasses(present []interfaceIdentity) interfacePasses {52	return interfacePasses{53		wired: func(ifc machine.InterfaceSpec) (*connection, error) {54			return bringUpInterface(ifc, present)55		},56		radio: func(ifc machine.InterfaceSpec) (*connection, error) {57			return bringUpRadio(ifc, present, raisePatience)58		},59		route:     routeVia,60		ciphers:   loadWirelessCiphers,61		readdress: readdressRadio,62	}63}6465// radioPass is what pass two hands the rest of the boot: the66// connection list as status should report it now, and a channel that67// delivers each radio's settled connection as it lands.68type radioPass struct {69	conns   []*connection70	settled chan *connection7172	// pending counts the radios that have not answered yet. Every73	// radio sends exactly one verdict, so the handler that drains the74	// channel knows when its work is over.75	pending int76}7778// run is the whole bring-up: pass one in spec order, then the gate79// below decides where pass two goes. A machine that can already80// reach its cluster backgrounds the radios and boots; a machine81// whose only declared path is a radio waits for it in the82// foreground, park included, because it has nothing to do without83// it.84func (p interfacePasses) run(interfaces []machine.InterfaceSpec, clusterDoc *cluster.Cluster) ([]*connection, *radioPass) {85	var conns []*connection86	var radios []machine.InterfaceSpec87	for _, ifc := range interfaces {88		if ifc.Wireless != nil {89			radios = append(radios, ifc)90			continue91		}92		conn, err := p.wired(ifc)93		if err != nil {94			fmt.Fprintf(os.Stderr, "liken: network: %s: %v\n", ifc.Name, err)95			continue96		}97		conns = append(conns, conn)98	}99	if len(radios) == 0 {100		return conns, nil101	}102103	// The ciphers load before any supplicant starts, and only on a104	// boot that a declared radio brought this far (ciphers.go).105	p.ciphers()106107	if p.backgroundable(conns, clusterDoc) {108		return p.background(interfaces, conns, radios)109	}110	return p.foreground(interfaces, conns, radios, clusterDoc.EndpointOrEmpty()), nil111}112113// backgroundable answers whether the boot may go on without its114// radios. Two things must hold: pass one gives a route toward the115// cluster's endpoint, and pass one holds the address k3s will start116// with. nodeAddress (k3s.go) picks that address, so the gate asks it117// rather than restating the rule.118//119// The second condition exists because k3s registers with the node120// address at start and never re-reads it. A machine whose nodeCIDR121// only the radio can answer must wait for the radio, or k3s starts122// with no node IP and the Node and the status disagree about the123// machine's address. A cluster that declares no nodeCIDR has no124// address rule to check, so the route question decides alone.125func (p interfacePasses) backgroundable(conns []*connection, clusterDoc *cluster.Cluster) bool {126	if routed, _ := routeToward(conns, clusterDoc.EndpointOrEmpty(), p.route); !routed {127		return false128	}129	if !declaresNodeCIDR(clusterDoc) {130		return true131	}132	ip, _ := nodeAddress(clusterDoc, conns)133	return ip != ""134}135136// background starts pass two behind the boot. The machine can137// already act, so nothing after this line waits on a radio, and the138// worst a wedged driver can do is what it does anyway; the operator139// it cannot stop is the recovery path.140func (p interfacePasses) background(interfaces []machine.InterfaceSpec, conns []*connection,141	radios []machine.InterfaceSpec) ([]*connection, *radioPass) {142	for _, ifc := range radios {143		conns = append(conns, pendingRadio(ifc))144	}145	conns = inSpecOrder(interfaces, conns)146147	pass := &radioPass{settled: make(chan *connection, len(radios)), pending: len(radios)}148	// The pass copies the list before the boot goes on with it. The149	// verdict handler replaces entries in its own copy, so no reader150	// of the boot's list ever sees a connection change under it.151	pass.conns = append([]*connection(nil), conns...)152153	plane.start("the wireless bring-up", p.joinRadios(pass, radios))154	return conns, pass155}156157// joinRadios is the component behind the boot: every declared radio,158// one at a time, each joined at most once.159//160// The radios are serial because a raise holds the kernel's rtnl lock161// while the driver runs. Two raises at once means the second waits on162// the first inside the kernel, where no deadline reaches it, so the163// first wedge would take every other radio's report with it. After a164// wedge the rest are not attempted at all, and each says so.165//166// The claim flags exist because the machine plane restarts a167// component whose function returns an error, and a panic inside a168// join becomes exactly that. Each radio is claimed before its join169// runs, so a restarted component joins none of them again, starts no170// second supplicant, and sends no second verdict.171func (p interfacePasses) joinRadios(pass *radioPass, radios []machine.InterfaceSpec) func(context.Context) error {172	claimed := make([]atomic.Bool, len(radios))173	return func(context.Context) error {174		stuck := ""175		for i, ifc := range radios {176			if !claimed[i].CompareAndSwap(false, true) {177				continue178			}179			if stuck != "" {180				pass.settled <- notAttemptedRadio(ifc, stuck)181				continue182			}183			fmt.Printf("liken: wireless: %s joins %s in the background\n", ifc.Name, ifc.Wireless.SSID)184			conn := p.joinOne(ifc)185			if conn.radio != nil && conn.radio.state == machine.WirelessNotRaised {186				stuck = ifc.Name187			}188			pass.settled <- conn189		}190		return nil191	}192}193194// foreground runs pass two in the boot path, for the machine whose195// only declared path is its radio. This is the one machine the radio196// is allowed to hold, and the park rule is unchanged from plan 62.197func (p interfacePasses) foreground(interfaces []machine.InterfaceSpec, conns []*connection,198	radios []machine.InterfaceSpec, endpoint string) []*connection {199	for _, ifc := range radios {200		conns = append(conns, p.joinOne(ifc))201	}202	conns = inSpecOrder(interfaces, conns)203204	// The park (plans/completed/62-wifi.md). Every interface has settled by205	// this line, so the decision has everything it needs. The hold206	// ends on a join event rather than a keypress, and the207	// addressing that follows is the same addressing a radio that208	// joined on the first try would have taken.209	if failed, reason := parkDecision(conns, endpoint, p.route); failed != nil {210		park(failed, reason)211		conns = p.readdress(conns, radios, failed)212	}213	return conns214}215216// inSpecOrder puts the connections back into the order the spec217// declared, whichever pass produced each one. The order is not218// cosmetic: resolv.conf keeps the first three nameservers in219// interface order, and the facts summarize the first addressed220// interface, so a radio the spec named first must appear first.221//222// A wired interface that failed its bring-up has no connection, so223// it simply has no place in the result; the map lookup skips it.224func inSpecOrder(interfaces []machine.InterfaceSpec, conns []*connection) []*connection {225	byName := make(map[string]*connection, len(conns))226	for _, conn := range conns {227		byName[conn.ifname] = conn228	}229	ordered := make([]*connection, 0, len(conns))230	for _, ifc := range interfaces {231		if conn, ok := byName[ifc.Name]; ok {232			ordered = append(ordered, conn)233		}234	}235	return ordered236}237238// joinOne is one radio's whole bring-up. It returns a connection in239// every case, because a radio that failed is a report the status240// must carry, never a missing interface.241func (p interfacePasses) joinOne(ifc machine.InterfaceSpec) *connection {242	conn, err := p.radio(ifc)243	if err != nil {244		fmt.Fprintf(os.Stderr, "liken: network: %s: %v\n", ifc.Name, err)245		return failedRadio(ifc, err)246	}247	return conn248}249250// pendingRadio is the connection status reports while pass two still251// works on the radio: associating, no address yet. The verdict252// replaces it.253func pendingRadio(ifc machine.InterfaceSpec) *connection {254	return &connection{ifname: ifc.Name, radio: &radio{255		ifname: ifc.Name, ssid: ifc.Wireless.SSID, state: machine.WirelessAssociating,256	}}257}258259// failedRadio is the connection for a radio whose bring-up errored260// before the join could say anything better, for example a port the261// machine does not have. The interface still owes status a reason,262// and NoCarrier with the error's text is the closest true statement.263// A radio that joined and then failed its addressing never comes264// here; it keeps its Connected verdict with the error as its265// message.266func failedRadio(ifc machine.InterfaceSpec, err error) *connection {267	return &connection{ifname: ifc.Name, radio: &radio{268		ifname: ifc.Name, ssid: ifc.Wireless.SSID,269		state: machine.WirelessNoCarrier, message: err.Error(),270	}}271}272273// notAttemptedRadio is the verdict for a radio the pass never tried,274// because an earlier radio's raise did not return. The state is275// NoCarrier, not NotRaised: NotRaised means this radio's own raise276// is still out, and a radio nothing touched must not claim that.277// The message names the interface actually at fault.278func notAttemptedRadio(ifc machine.InterfaceSpec, stuck string) *connection {279	return &connection{ifname: ifc.Name, radio: &radio{280		ifname: ifc.Name, ssid: ifc.Wireless.SSID,281		state: machine.WirelessNoCarrier,282		message: fmt.Sprintf("not attempted; raising %s is stuck holding the netlink lock every interface needs",283			stuck),284	}}285}286287// joinRadio is the 802.11 session, a variable holding the real join288// below it, so a test can drive the addressing that follows a join289// with no radio and no supplicant. It is the seam startSupplicant290// and parkConsole already are.291var joinRadio = joinWireless292293// bringUpRadio is the wireless half of bringUpInterface, moved here294// so the raise can run under a deadline: raise, join, then address295// exactly as a wired port would.296func bringUpRadio(ifc machine.InterfaceSpec, present []interfaceIdentity, patience time.Duration) (*connection, error) {297	link, err := raiseUnderDeadline(ifc, present, patience)298	if err != nil {299		var stuck raiseStuck300		if !errors.As(err, &stuck) {301			return nil, err302		}303		fmt.Fprintf(os.Stderr, "liken: wireless: %s: %v\n", ifc.Name, err)304		return &connection{ifname: ifc.Name, radio: &radio{305			ifname: ifc.Name, ssid: ifc.Wireless.SSID,306			state: machine.WirelessNotRaised, message: err.Error(),307		}}, nil308	}309310	// The join comes before the addressing because an unassociated311	// radio carries no frames: a DHCP exchange on it would only wait312	// out its own deadline. Once the radio associates, the interface313	// behaves exactly like an ethernet port, which is why the same314	// two addressing paths run below with nothing added.315	r := joinRadio(ifc, machine.MachineStateDir)316	if r.state != machine.WirelessConnected {317		// A failed join still yields a connection. The interface318		// exists, the status must carry the reason it has no319		// address, and the park decision reads these same320		// connections to learn what settled.321		return &connection{ifname: ifc.Name, mac: link.Attrs().HardwareAddr, radio: r}, nil322	}323	conn, err := addressInterface(link, ifc)324	if err != nil {325		// The join happened, so Connected stands; reporting326		// NoCarrier here would blame the radio for a DHCP failure.327		// The address error is the message, and the missing address328		// on the interface says the rest.329		fmt.Fprintf(os.Stderr, "liken: network: %s: %v\n", ifc.Name, err)330		r.message = err.Error()331		return &connection{ifname: ifc.Name, mac: link.Attrs().HardwareAddr, radio: r}, nil332	}333	conn.radio = r334	return conn, nil335}336337// raiseStuck is a raise that never returned, kept apart from a338// refusal the kernel gave. A refusal is an answer; this is the339// absence of one, and it gets its own wireless state because no340// supplicant event will ever follow it.341type raiseStuck struct {342	ifname   string343	patience time.Duration344}345346// The message states only what init observed: the call has not come347// back. Which lock the stuck thread holds is not observable from348// here, and a guess would land in status as a fact.349func (s raiseStuck) Error() string {350	return fmt.Sprintf("raising %s did not return in %s; the netlink call is still out and nothing can cancel it",351		s.ifname, s.patience)352}353354// raiseUnderDeadline opens one link and raises it, and stops waiting355// after the patience runs out. The goroutine is abandoned then, not356// cancelled, because nothing can cancel a thread that is stuck inside357// the kernel; the report is the whole remedy. A raise that returns358// after the deadline lands its answer in a buffered channel that359// nothing reads again, which is the cheapest true accounting of it.360//361// The link lookup runs inside the deadline too, because it is also a362// netlink call: once an earlier raise wedges the kernel's rtnl lock,363// the lookup blocks the same way the raise does, and a deadline that364// covered only the raise would hang before reaching it.365func raiseUnderDeadline(ifc machine.InterfaceSpec, present []interfaceIdentity,366	patience time.Duration) (netlink.Link, error) {367	type raised struct {368		link netlink.Link369		err  error370	}371	done := make(chan raised, 1)372	// The seams bind before the goroutine starts, because the373	// goroutine can outlive this function and a test must not have374	// its stand-ins swapped out mid-raise.375	byName, setUp := linkByName, linkSetUp376	go func() {377		if err := requirePort(ifc.Name, present); err != nil {378			done <- raised{err: err}379			return380		}381		link, err := byName(ifc.Name)382		if err != nil {383			done <- raised{err: fmt.Errorf("opening interface %q: %w", ifc.Name, err)}384			return385		}386		fmt.Printf("liken: bringing up %s\n", ifc.Name)387		if err := setUp(link); err != nil {388			done <- raised{err: fmt.Errorf("raising %s: %w", ifc.Name, err)}389			return390		}391		done <- raised{link: link}392	}()393	select {394	case out := <-done:395		return out.link, out.err396	case <-time.After(patience):397		return nil, raiseStuck{ifname: ifc.Name, patience: patience}398	}399}400401// publishRadioVerdicts is the machine-plane component that carries402// pass two's verdicts into the world. It starts after403// publishBootFacts, because publishBootFacts writes the network404// subtree itself and would overwrite an earlier verdict with the405// pending state it replaced (main.go places the start).406func publishRadioVerdicts(pass *radioPass, tree machine.FactsTree, clusterDoc *cluster.Cluster) func(context.Context) error {407	return func(ctx context.Context) error {408		// The component ends when every declared radio has answered409		// once. The supplicant's supervision owns the session after410		// that; this component only lands the boot's verdicts.411		for settled := 0; settled < pass.pending; {412			select {413			case <-ctx.Done():414				return nil415			case conn := <-pass.settled:416				settled++417				pass.fold(conn)418				conn.report()419				// A failed facts write means the status still shows420				// the pending state; the stderr line is the only421				// other record that the verdict existed.422				if err := tree.WriteNetwork(networkFacts(clusterDoc, pass.conns)); err != nil {423					fmt.Fprintf(os.Stderr, "liken: writing facts: %v\n", err)424				}425				// A late radio's nameservers join resolv.conf the426				// same way an early one's would have: in interface427				// order, under the cap of three.428				if err := writeResolvConf(pass.conns); err != nil {429					fmt.Fprintf(os.Stderr, "liken: network: %v\n", err)430				}431			}432		}433		return nil434	}435}436437// fold replaces the pending entry the pass has been reporting with438// the settled connection. The match is by interface name, and every439// radio has a pending entry, so the append is a guard, not a path.440func (p *radioPass) fold(conn *connection) {441	for i, existing := range p.conns {442		if existing.ifname == conn.ifname {443			p.conns[i] = conn444			return445		}446	}447	p.conns = append(p.conns, conn)448}
init/reboot.go 33.6%
1package main23// Rebooting on request.4//5// Only PID 1 can shut a machine down correctly. For this reason, the6// operator never reboots the machine itself. Instead, the operator7// writes an intent file (machine/reboot.go describes the channel),8// and init does the rest. Init watches the intent directory with9// inotify: it establishes the watch, scans the directory once, and10// then scans again on every event. The watch-then-scan order closes a11// window. An intent that lands between a scan and the watch would12// otherwise wait for the next event.13//14// An event is only a trigger. The scan determines everything: which15// intent stands, and what it carries. A woken watcher reads the16// directory as it is now, not what an event claimed a moment ago.17// writeAtomic renames a finished file into the directory, so18// IN_MOVED_TO is the event that every intent write produces, and the19// watch asks for it.20//21// The machine guarantees its inotify headroom at boot (osSysctls in22// system.go raises the per-uid watch and instance limits), so a watch23// that fails to start is a real fault, not an expected shortage. Init24// runs the watch as a component of the machine plane, so a failed25// watch surfaces on the console and the plane retries it with backoff.26// There is no polling fallback to hide the fault.27//28// The shutdown sequence runs the dependency stack in reverse order.29// First, it signals every process. (k3s was already stopped30// gracefully, but its containers outlive it, so they get their own31// warning here.) Then it waits a fixed grace period, stops the32// machine plane's own components, detaches the role filesystems,33// syncs, and calls the kernel restart syscall. Under QEMU's34// -no-reboot flag, a restart becomes a clean exit instead of an35// actual reboot. A test with a fixed time limit can use this exit36// as its result.3738import (39	"context"40	"fmt"41	"os"42	"time"4344	"golang.org/x/sys/unix"4546	"github.com/liken-sh/liken/liken/machine"47)4849// watchIntentDir watches the intent directory. It is a package50// variable so a test can hand the watch a channel that has closed.51var watchIntentDir = machine.WatchDir5253// watchForOperatorIntents watches the operator's channel and delivers54// each intent that lands there. It establishes an inotify watch on the55// directory, runs an initial scan, and then scans again on every wake.56// The watch comes before the first scan, so an intent that lands57// between the scan and the watch cannot slip past unseen.58//59// A watch that cannot start returns the error to the machine plane,60// which reports it and restarts this component. A returned nil tells61// the plane that a reboot intent was delivered and this component's62// work is complete.63func watchForOperatorIntents(ctx context.Context, dir string,64	reboots chan<- machine.RebootIntent, restarts chan<- machine.RestartIntent,65	loads chan<- machine.ModulesIntent) error {66	wake, err := watchIntentDir(ctx, dir)67	if err != nil {68		return fmt.Errorf("watching %s: %w", dir, err)69	}70	if scanIntents(dir, reboots, restarts, loads) {71		return nil72	}73	for {74		select {75		case <-ctx.Done():76			return nil77		case _, ok := <-wake:78			if !ok {79				return fmt.Errorf("the watch on %s stopped", dir)80			}81		}82		if scanIntents(dir, reboots, restarts, loads) {83			return nil84		}85	}86}8788// scanIntents reads the intent directory once and delivers what it89// finds. It returns true only when it delivered a reboot intent, which90// ends the watch.91//92// The function checks for a reboot intent first. This order is93// deliberate. A reboot re-renders everything that a restart would also94// render. So when both files exist at the same time, the reboot takes95// priority. The restart file is not needed anymore: it disappears with96// the boot's tmpfs, like every reboot intent does.97//98// Delivering a reboot intent ends the watch. A reboot always follows99// it, so there is nothing left to watch for.100//101// A restart intent works differently, because the machine keeps102// running after a restart. The function consumes the restart intent,103// and the watch continues for the rest of the boot. The function104// clears the restart intent file before it delivers the intent to the105// caller. If a crash happens between these two steps, the machine loses106// one restart request. The operator's next pass sends the request107// again. If the function cleared the intent file after delivering it108// instead, a crash between the two steps could cause the machine to109// restart k3s again and again.110//111// The presence of a file is the trigger. The content of a file only112// improves the console message. For this reason, the function honors an113// intent file that it cannot read, instead of ignoring it. (Atomic114// writes make an unreadable file a bug, not a race condition, but a bug115// must not stop the machine from working.)116func scanIntents(dir string, reboots chan<- machine.RebootIntent,117	restarts chan<- machine.RestartIntent, loads chan<- machine.ModulesIntent) bool {118	intent, err := machine.ReadRebootIntent(dir)119	if err != nil {120		fmt.Fprintf(os.Stderr, "liken: reading the reboot intent: %v\n", err)121		intent = &machine.RebootIntent{Reason: "an unreadable reboot intent"}122	}123	if intent != nil {124		reboots <- *intent125		return true126	}127128	restart, err := machine.ReadRestartIntent(dir)129	if err != nil {130		fmt.Fprintf(os.Stderr, "liken: reading the restart intent: %v\n", err)131		restart = &machine.RestartIntent{Reason: "an unreadable restart intent"}132	}133	if restart != nil {134		if err := machine.ClearRestartIntent(dir); err != nil {135			fmt.Fprintf(os.Stderr, "liken: consuming the restart intent: %v\n", err)136		}137		restarts <- *restart138	}139140	// The modules intent behaves like the restart intent: the function141	// consumes it before delivery, and the machine keeps running.142	// Loading modules is the lightest disruption of all: none. The143	// function still honors this intent file even when it cannot read144	// it, because the staged store, not the intent file, holds the145	// truth about what to load. The apply step re-derives everything146	// from the staged store.147	load, err := machine.ReadModulesIntent(dir)148	if err != nil {149		fmt.Fprintf(os.Stderr, "liken: reading the modules intent: %v\n", err)150		load = &machine.ModulesIntent{Reason: "an unreadable modules intent"}151	}152	if load == nil {153		return false154	}155	if err := machine.ClearModulesIntent(dir); err != nil {156		fmt.Fprintf(os.Stderr, "liken: consuming the modules intent: %v\n", err)157	}158	loads <- *load159	return false160}161162// rebootMachine runs init's shutdown sequence: the dependency stack163// in reverse order. The supervisor already stopped k3s gracefully,164// so what remains is its containers, then the machine plane's own165// loops, then the filesystems. Like failBoot, rebootMachine never166// returns, because PID 1 must never exit. If the reboot syscall167// fails, the machine stays in this function until a person168// investigates it.169func rebootMachine(intent machine.RebootIntent) {170	fmt.Printf("liken: rebooting: %s\n", intent.Reason)171	// The flag goes up first and stays up. The proving watch may172	// still be running, and once the firmware turn below is over173	// there is nothing left for it to correct.174	shuttingDown.Store(true)175	// The supplicants stop before the general signal because each176	// one runs under a restart loop that would start it again during177	// the grace period below. A stop through the loop lets the178	// process deauthenticate and put the interface down once, for179	// good.180	stopSupplicants()181	killEverything()182	// The firmware turn happens after every other process has ended,183	// while machineState and the boot path's filesystems are still184	// mounted. It asserts the proven slot and then, when a release is185	// staged for the other slot, checks that the slot holds that186	// release and arms the one-shot trial that this reboot proves, in187	// that order and with nothing between them: assertProven clears a188	// stale one-shot, so the trial armed after it never appears189	// stale, and the trial is only safe over a fallback that was just190	// asserted (assertAndArmForReboot in proving.go). The machine191	// operator writes the slot while it downloads a release, and it192	// is one of the processes that killEverything ended. So no write193	// can change the slot between the check and the boot it arms. On194	// a BIOS machine the assertion also repairs the boot chain on195	// disk. The turn must happen on the way down, because a boot path196	// can become damaged while the machine runs. (Cloud hosts rewrite197	// MBRs under running guests.) A damaged boot path would prevent198	// this reboot from coming back up.199	//200	// The turn's lock is released with the turn: this function is201	// reachable from inside a machine-plane component, and a lock202	// held across the shutdown below would outlive its holder if203	// anything in that shutdown panicked.204	assertAndArmForReboot(chooseBootActuator(), machine.MachineStateDir,205		bootParamValue("liken.slot"))206	// The machine plane stops only after every process ends. The207	// reaper is one of its components, and it must collect exited208	// processes until the very end.209	plane.shutdown(10 * time.Second)210	// Nothing writes to a disk after this point, so every disk can go211	// read-only and record that it was shut down properly. The212	// unmounts below cannot do that for a filesystem that stays busy,213	// and two of them always do (quiesce.go explains which), so the214	// disks are finished here rather than by the unmount.215	quiesceDisks()216	unmountRoleMounts(unix.MNT_DETACH, false)217	syncLogs()218	unix.Sync()219	if err := unix.Reboot(unix.LINUX_REBOOT_CMD_RESTART); err != nil {220		fmt.Fprintf(os.Stderr, "liken: reboot failed: %v\n", err)221	}222	for {223		time.Sleep(time.Hour)224	}225}226227// killEverything sends a signal to every process on the machine.228// kill(-1) from PID 1 reaches all processes except init itself.229// SIGTERM comes first, so containers get the same graceful warning230// that k3s got. A fixed grace period is enough. The function does231// not need to track each remaining process, because SIGKILL follows232// and the reaper collects every process either way.233func killEverything() {234	fmt.Println("liken: stopping every remaining process")235	_ = unix.Kill(-1, unix.SIGTERM)236	time.Sleep(5 * time.Second)237	_ = unix.Kill(-1, unix.SIGKILL)238}
init/registries.go 94.4%
1package main23// Rendering k3s's registries.yaml: how this machine pulls container4// images.5//6// registries.yaml is k3s's file for containerd's registry settings:7// mirror endpoints to pull through, and the credentials to present.8// k3s reads this file once, at process start, when it renders9// containerd's actual configuration. This is why registry changes10// take effect only when k3s restarts (see cluster/changes.go), and11// never need a reboot.12//13// Init is the sole author of this file, and it uses two inputs. The14// mirrors and the embedded-registry choice come from the cluster15// document, like every fleet-wide fact. The credentials come from16// their own document (see registries.go in the machine package).17// The operator authors this document from the registry-credentials18// Secret. The document follows the same staged/proven lifecycle as19// every other document, but it has its own store, because20// credentials rotate on their own schedule. A credentials change21// must not wait for a cluster document edit, and must not trigger22// one either.23//24// The credentials lifecycle is simpler than the cluster document's25// lifecycle in one deliberate way: init promotes staged credentials26// at actuation, with no attempted marker and no downstream proof.27// The cluster document needs the operator's existence as its proof,28// because its failure modes appear only after the boot completes (a29// bad endpoint means the machine never joins the cluster). A30// credentials document has no such failure mode. k3s starts fine31// even with a wrong password. The symptom (ImagePullBackOff) is32// visible in the cluster, and the fix is a Secret edit that flows33// through as a new document. Writing the file is the whole34// actuation, and init observes the result directly. Falling back to35// older credentials on the next boot would repair nothing, and it36// would also hide the newest intent.3738import (39	"errors"40	"fmt"41	"io/fs"42	"maps"43	"os"44	"slices"45	"strings"4647	"sigs.k8s.io/yaml"4849	"github.com/liken-sh/liken/liken/cluster"50	"github.com/liken-sh/liken/liken/machine"51)5253// registriesConfigPath is the path where k3s expects the file. It is54// a package variable, so tests can render into a temporary55// directory.56var registriesConfigPath = "/etc/rancher/k3s/registries.yaml"5758// registriesFile is the registries.yaml shape, reduced to the keys59// that liken writes. It renders through sigs.k8s.io/yaml, which uses60// JSON-path marshaling and sorts keys, so the same inputs always61// produce the same bytes.62type registriesFile struct {63	Mirrors map[string]registryMirror `json:"mirrors,omitempty"`64	Configs map[string]registryConfig `json:"configs,omitempty"`65}6667type registryMirror struct {68	// Endpoint is k3s's key name. The order of endpoints is the order69	// of preference.70	Endpoint []string `json:"endpoint,omitempty"`71}7273type registryConfig struct {74	Auth registryAuth `json:"auth"`75}7677type registryAuth struct {78	Username string `json:"username"`79	Password string `json:"password,omitempty"`80}8182// chooseRegistryCredentials selects the credentials document that83// this rendering uses. It prefers the staged (vetted) document over84// the proven document, and it records the choice in the boot record.85// There is no seed document and no image fallback, because the86// operator is the only author of this document. A machine that has87// never had credentials staged simply has none. This is the normal88// anonymous-pulls state, not an error. The function rejects a staged89// document that does not parse immediately, and the rendering falls90// back to the proven document. Unlike the cluster document, this91// document has no attempted marker. (The file comment above explains92// why promotion happens at actuation instead.)93func chooseRegistryCredentials(stateRoot string, durable bool, boot *machine.BootStatus) *machine.RegistryCredentials {94	if !durable {95		return nil96	}97	store := machine.RegistryCredentialsStore(stateRoot)9899	// The function republishes the standing rejection into the boot100	// record on every boot. (rejectStagedDocument explains why.)101	boot.CredentialsRejection, _ = store.LoadRejection()102103	if raw, err := store.LoadStaged(); err != nil {104		fmt.Fprintf(os.Stderr, "liken: registries: the staged credentials are unreadable: %v\n", err)105	} else if raw != nil {106		c, perr := machine.ParseRegistryCredentials(raw)107		if perr != nil {108			boot.CredentialsRejection = rejectStagedDocument("registries", "credentials", store.Reject,109				raw, fmt.Sprintf("the staged credentials document does not parse: %v", perr))110		} else {111			boot.CredentialsSource = machine.ManifestSourceStaged112			boot.CredentialsHash = machine.ManifestHash(raw)113			return c114		}115	}116117	if raw, err := store.LoadProven(); err != nil {118		fmt.Fprintf(os.Stderr, "liken: registries: the proven credentials are unreadable: %v\n", err)119	} else if raw != nil {120		c, perr := machine.ParseRegistryCredentials(raw)121		if perr != nil {122			// A proven document that does not parse is a corrupted123			// last-known-good copy. The function reports this and pulls124			// images anonymously, instead of stopping because of a125			// credentials problem.126			fmt.Fprintf(os.Stderr, "liken: registries: the proven credentials are unreadable: %v\n", perr)127		} else {128			boot.CredentialsSource = machine.ManifestSourceProven129			boot.CredentialsHash = machine.ManifestHash(raw)130			return c131		}132	}133	return nil134}135136// writeRegistriesConfig renders registries.yaml from the cluster137// document's mirrors and the chosen credentials, and it returns what138// it rendered for the facts. When nothing is declared anywhere, the139// function removes the file instead. (A live retraction must remove140// the old rendering too.) This keeps k3s's default behavior in141// place: every registry is reached directly, and anonymously.142//143// When the embedded registry is on, the spec's mirrors render144// exactly as declared, and a bare "*" entry joins them, unless the145// spec already declares one. With Spegel, a registry participates in146// peer-to-peer sharing only if registries.yaml lists it as a mirror,147// and the wildcard is k3s's way to say "all registries". Turning the148// embedded registry on means "share pulled images across the149// fleet". If the function silently excluded every registry that was150// not also mirrored, it would surprise the deployments that want151// this feature most.152//153// On success, the function promotes the staged credentials, because154// the write was the whole actuation. (The file comment above155// explains why no downstream proof exists to wait for.) A write156// failure leaves the staged credentials in place, so the next boot157// can retry.158func writeRegistriesConfig(clusterDoc *cluster.Cluster, creds *machine.RegistryCredentials,159	store machine.ManifestStore, source machine.ManifestSource) machine.RegistriesStatus {160	status := machine.RegistriesStatus{}161162	// The function always builds both maps. Each map disappears from163	// the output when it is empty (omitempty). The gate below, for164	// the nothing-declared case, reads the size of each map.165	file := registriesFile{166		Mirrors: map[string]registryMirror{},167		Configs: map[string]registryConfig{},168	}169	if clusterDoc != nil {170		for host, endpoints := range clusterDoc.Spec.Registries.Mirrors {171			file.Mirrors[host] = registryMirror{Endpoint: endpoints}172		}173		if clusterDoc.Spec.Registries.Embedded {174			status.Embedded = true175			if _, declared := file.Mirrors["*"]; !declared {176				file.Mirrors["*"] = registryMirror{}177			}178		}179	}180	if creds != nil {181		for _, h := range creds.Hosts {182			file.Configs[h.Host] = registryConfig{Auth: registryAuth{183				Username: h.Username,184				Password: h.Password,185			}}186		}187	}188189	if len(file.Mirrors) == 0 && len(file.Configs) == 0 {190		if err := os.Remove(registriesConfigPath); err != nil && !errors.Is(err, fs.ErrNotExist) {191			fmt.Fprintf(os.Stderr, "liken: registries: removing %s: %v\n", registriesConfigPath, err)192		}193		return status194	}195196	raw, err := yaml.Marshal(&file)197	if err != nil {198		fmt.Fprintf(os.Stderr, "liken: registries: rendering %s: %v\n", registriesConfigPath, err)199		return status200	}201	// 0600, the same permission as the join token, because this file202	// embeds passwords.203	if err := os.WriteFile(registriesConfigPath, raw, 0o600); err != nil {204		fmt.Fprintf(os.Stderr, "liken: registries: writing %s: %v\n", registriesConfigPath, err)205		return status206	}207208	status.Mirrors = slices.Sorted(maps.Keys(file.Mirrors))209	status.CredentialedHosts = slices.Sorted(maps.Keys(file.Configs))210	for _, host := range status.Mirrors {211		if host == "*" {212			fmt.Println("liken: registries: the embedded registry shares every registry's images across the fleet")213			continue214		}215		fmt.Printf("liken: registries: mirroring %s via %s\n",216			host, strings.Join(file.Mirrors[host].Endpoint, ", "))217	}218	if len(status.CredentialedHosts) > 0 {219		fmt.Printf("liken: registries: credentials for %s\n", strings.Join(status.CredentialedHosts, ", "))220	}221222	if source == machine.ManifestSourceStaged {223		if err := store.Promote(); err != nil {224			fmt.Fprintf(os.Stderr, "liken: registries: promoting the staged credentials: %v\n", err)225		} else {226			fmt.Println("liken: registries: the staged credentials are now proven")227		}228	}229	return status230}
init/reinstall.go 12.8%
1package main23// Reinstalling over a disk that already carries a liken install.4//5// The installer only ever claims a blank disk (claim.go), which is6// the right rule: it means liken never destroys data it did not put7// there. The cost of that rule is that a disk liken itself wrote can8// only be reinstalled after something outside liken blanks it, and a9// fresh machine has no such thing: no shell, no second OS, and, until10// its own install succeeds, no network. liken.reinstall is the escape11// hatch. It is liken.install, plus one thing first: it blanks the12// disks this machine's manifest declares, so the claim that follows13// finds them blank.14//15// A person asks for this by picking the stick's "wipe and reinstall16// as <name>" entry, over a manifest that declares these exact disks.17// Picking it is the confirmation, and it is as explicit as a person18// can be at a console with no other tools.19//20// A declared disk may name a stable identity under /dev/disk/by-id/21// or /dev/disk/by-path/ instead of a kernel path, so the wipe22// resolves it through resolveDeclaredDisk, the same computation the23// claim uses. A name that matches two disks wipes neither: guessing24// which one the manifest meant risks the disk the manifest did not25// mean.26//27// Blanking a disk's table is not the whole erasure, and on its own it28// would not be one. Partitions start a megabyte in, so their file29// systems survive the wipe, and a claim that writes the same layout30// back puts them at the same offsets again. The erasure is completed31// by the rule in storage.go: a partition this boot created always32// gets a new file system. Nothing of the previous install survives a33// reinstall, on any disk the manifest declares.34//35// The reclaim runs after loadModules (so a real controller's driver36// has created the device nodes) and before settleStorage (so nothing37// claims or mounts a disk this boot is about to erase). It does not38// power off: the boot goes on to install, the same as a plain39// install boot.4041import (42	"fmt"43	"os"4445	"golang.org/x/sys/unix"4647	"github.com/liken-sh/liken/liken/machine"48)4950const (51	// installParam makes a boot install onto a blank disk.52	installParam = "liken.install"53	// reinstallParam makes a boot reclaim its manifest's disks first,54	// then install.55	reinstallParam = "liken.reinstall"56)5758// installing reports whether this boot's one job is to put liken on59// the machine's own disk, by either door: a plain install onto blank60// disks, or a reinstall that blanks them first.61func installing() bool {62	return bootParam(installParam) || bootParam(reinstallParam)63}6465// wipeRegion is how much is zeroed at each end of a disk. isBlank66// reads only the first 2 KiB (the MBR, the GPT header, and the ext467// magic), and a claim rewrites the whole GPT anyway, so a megabyte at68// the front already reclaims the disk. The same megabyte at the back69// removes the GPT's backup header, which lives in the last sectors,70// so no tool later reports a torn table.71const wipeRegion = 1 << 207273// reclaimManifestDisks blanks every disk the seed manifest declares.74// It reads the seed directly, by the same liken.machine= identity the75// installer uses, because the disks it must reclaim are exactly the76// ones the install that follows will claim. A disk that fails to77// reclaim is left to the ordinary claim, which refuses a disk it78// cannot recognize and stops the boot with its own message; this79// function does not power off on its own.80func reclaimManifestDisks() {81	seed, err := loadSeed(machine.MachineManifestDir, bootParamValue("liken.machine"))82	if err != nil {83		fmt.Fprintf(os.Stderr, "liken: reinstall: cannot read the seed manifest to find its disks: %v\n", err)84		return85	}8687	var devices []string88	seen := map[string]bool{}89	for _, role := range seed.m.Spec.Storage.Roles() {90		if role.Device != "" && !seen[role.Device] {91			seen[role.Device] = true92			devices = append(devices, role.Device)93		}94	}95	if len(devices) == 0 {96		fmt.Fprintln(os.Stderr, "liken: reinstall: the manifest declares no disks to reclaim")97		return98	}99100	for _, device := range devices {101		node, err := awaitDevice(device)102		if err != nil {103			fmt.Fprintf(os.Stderr, "liken: reinstall: %v\n", err)104			continue105		}106		if err := blankDisk(node); err != nil {107			fmt.Fprintf(os.Stderr, "liken: reinstall: %s: %v\n", device, err)108			continue109		}110		fmt.Printf("liken: reinstall: reclaimed %s\n", device)111	}112}113114// awaitDevice waits, boundedly, for a declared device to attach, and115// resolves it to the kernel node that a wipe can open. A reinstall116// hits the same probe race an install does, so it waits the same way117// (resolve.go). A device the manifest names is expected to exist, so118// its continued absence at the deadline is an error, not a silent119// skip. It resolves the declared string rather than opening it120// directly, because a stable name is not a device node a wipe can121// open.122func awaitDevice(declared string) (string, error) {123	disk, err := awaitDeclaredDisk(declared,124		fmt.Sprintf("liken: reinstall: waiting for %s to attach", declared))125	switch {126	case err != nil:127		return "", err128	case disk == nil:129		return "", fmt.Errorf("%s did not attach within %s", declared, declaredDiskDeadline)130	}131	return devicePath(*disk), nil132}133134// blankDisk makes a disk blank in the two ways that matter: on the135// platters and in the kernel's memory of them. It zeros the first and136// last megabyte (the front carries every signature isBlank inspects;137// the back carries the GPT's backup header), then asks the kernel to138// re-read the table. The re-read is not cosmetic. The kernel scanned139// this disk's old table at boot, so its partition nodes and their140// cached GPT labels still name the previous install's roles; without141// the re-read, storage still finds a machineState partition to mount142// and a proven manifest inside it, and reinstalls the machine into143// the boot it was trying to replace. After the re-read of a zeroed144// table, the disk has no partitions, and the claim that follows sees145// a blank disk. A disk shorter than two wipe regions is blanked in146// one pass from the front, which still covers every signature.147// blankDisk takes the resolved kernel node, never the declared148// string: a stable name under /dev/disk/ is not a node this function149// can open.150func blankDisk(device string) error {151	f, err := os.OpenFile(device, os.O_RDWR, 0)152	if err != nil {153		return err154	}155	defer f.Close()156157	zeros := make([]byte, wipeRegion)158	if _, err := f.WriteAt(zeros, 0); err != nil {159		return fmt.Errorf("zeroing the front: %w", err)160	}161162	if d := diskByPath(device); d != nil && d.SizeBytes >= 2*wipeRegion {163		if _, err := f.WriteAt(zeros, int64(d.SizeBytes)-wipeRegion); err != nil {164			return fmt.Errorf("zeroing the back: %w", err)165		}166	}167168	if err := f.Sync(); err != nil {169		return fmt.Errorf("flushing: %w", err)170	}171	if _, err := unix.IoctlRetInt(int(f.Fd()), unix.BLKRRPART); err != nil {172		return fmt.Errorf("re-reading the partition table: %w", err)173	}174	unix.Sync()175	return nil176}
init/report.go 94.1%
1package main23// The hardware report boot: a boot that changes nothing.4//5// A new machine asks a question the lab cannot answer: which drivers,6// which interface names, which disk paths does this hardware need in7// its manifest? The lab cannot answer it because the vendored kernel8// builds the paravirtual drivers in, so a lab guest never loads the9// storage or network module that every real controller needs. The10// report boot answers the question on the real machine, before the11// first install.12//13// A person picks "liken hardware report" from the installation stick's14// menu. The boot carries liken.report on its command line, and no15// liken.machine= identity, because it describes the hardware, not a16// machine in the deployment. It does the smallest amount of work that17// produces a real answer:18//19//  1. It mounts the payload's system image to reach the full module20//     tree. The install medium carries the whole OS as liken.sqfs, and21//     that image holds every driver, the alias table, and the softdep22//     information. The report reads all three from there.23//  2. It loads the drivers this hardware names, from that tree, for24//     the storage and network devices only. Loading a module changes25//     only RAM, so the report keeps its promise to change nothing on26//     any disk. The names it needs are real only after the drivers27//     bind: eth0 does not exist until r8169 loads.28//  3. It observes what appeared: every disk that can hold a role, and29//     every interface with its link brought up long enough to see the30//     carrier.31//  4. It writes a proposed manifest to the stick, prints it, states32//     what would stop that manifest from installing, and reboots when33//     the person presses Enter.34//35// The proposal makes a promise, and the promise is the whole point of36// the boot: a person renames the machine, checks the sizes, and37// installs. So the report never proposes something that cannot work.38// It sizes the roles against the disks it measured. It leaves a39// storage driver out of spec.modules, because the declared modules40// load long after the boot claims disks, and says what image would41// carry the driver instead. It declares the ports with a cable in42// them, and leaves the dark ones as comments, because every declared43// port costs the boot a DHCP wait.44//45// The report never claims, formats, or writes to any of the machine's46// own disks. It runs before storage settles for exactly this reason:47// nothing downstream of it, not the storage reconciliation and not the48// installer, must ever run on a report boot.4950import (51	"context"52	"fmt"53	"net"54	"os"55	"path/filepath"56	"strings"57	"time"5859	"github.com/vishvananda/netlink"60	"golang.org/x/sys/unix"6162	"github.com/liken-sh/liken/liken/hardware"63)6465// stickCeiling bounds the one wait the report makes for the66// installation stick's disk to appear. It is generous because the67// whole report depends on the answer: the stick is where the proposal68// goes, and it is the one disk no role may land on.69const stickCeiling = 15 * time.Second7071// reportParam makes a boot describe the hardware and change nothing. It72// follows installParam and reinstallParam (reinstall.go), but it is its73// own word because it is the one menu entry that never touches a disk.74const reportParam = "liken.report"7576// stickInstallPartition is the GPT partition name that image/stick.go77// writes on the installation stick. The report finds the stick by this78// name to write its proposal there, the same way storage roles are79// found by the names written into their partitions.80const stickInstallPartition = "liken:install"8182// hardwareReportName is the proposal's file name at the root of the83// stick's filesystem. A person pulls the stick, reads this file, edits84// it, and uses it as the machine's manifest.85const hardwareReportName = "hardware-report.yaml"8687// reportImageMount is where the report loop-mounts the payload's system88// image to reach its module tree. reportStickMount is where it mounts89// the stick's filesystem to write the proposal. They are variables so a90// test can point the report at directories of its own making.91var (92	reportImageMount = "/liken-report-image"93	reportStickMount = "/liken-report-stick"94)9596// reporting reports whether this boot's one job is to describe the97// hardware and reboot.98func reporting() bool {99	return bootParam(reportParam)100}101102// runHardwareReport is the whole report boot. It gathers the hardware,103// composes the proposal, writes it to the stick, prints it, and reboots104// when the person acknowledges. Like every install-menu terminal state,105// it ends at a held console, because a person picked this entry and is106// present by construction. It never returns.107func runHardwareReport() {108	report, stick := gatherHardwareReport()109	proposal := composeHardwareReport(report)110111	writeErr := writeReportToStick(stick, proposal)112113	// The console print is the proposal's second copy, and its only copy114	// when the stick write fails. It goes out after the write attempt,115	// so the held message below can tell the person the truth about the116	// file.117	fmt.Println(proposal)118119	// The warnings come after the proposal, immediately above the120	// prompt, because they are the lines a person must act on and the121	// proposal above them is long. A machine with nothing wrong prints122	// none of them.123	for _, line := range reportWarnings(report) {124		fmt.Println(line)125	}126127	if writeErr != nil {128		fmt.Fprintf(os.Stderr, "liken: report: writing to the stick: %v\n", writeErr)129	}130	holdInstallerConsole(reportPrompt(writeErr), false)131	endReport()132}133134// endReport ends the report boot. It is a package variable so a test135// can run the whole report boot and see it end, because the real136// ending reboots the machine and never returns.137var endReport = rebootAfterReport138139// reportPrompt is the held console's last line. It tells the person140// whether the proposal reached the stick, because a person who pulls141// the stick on a failed write leaves with no copy of the report.142func reportPrompt(writeErr error) string {143	if writeErr != nil {144		return fmt.Sprintf(145			"liken: writing %s to the stick FAILED; the text above is the only copy; press Enter to reboot.",146			hardwareReportName)147	}148	return fmt.Sprintf(149		"liken: this report was written to the stick as %s; press Enter to reboot.",150		hardwareReportName)151}152153// gatherHardwareReport does the observation: it mounts the module tree,154// loads the drivers this hardware names, and reads back the disks and155// interfaces that appeared. It returns the stick it resolved along156// with the report, because the write of the proposal must go to that157// same stick and no other.158//159// It degrades rather than fails. A machine whose payload will not160// mount still gets a report of the disks and interfaces the boot-path161// drivers bound, which is more use to a person than a blank screen.162// The degraded path settles and waits exactly as the full path does:163// the hardware takes the same seconds to appear whether or not the164// report has a module tree to read, and a walk that runs a second into165// the boot sees a machine with no SATA disks on it.166func gatherHardwareReport() (hardwareReport, installStick) {167	uefi := firmwareIsUEFI()168169	// Let the boot-path buses finish probing before the first walk. On170	// real hardware, storage settles in the middle of a SATA link's171	// training or a USB device's negotiation, and a disk that has not172	// appeared yet is indistinguishable from one that is not there.173	quiesceHardware()174175	var recommendations, claimable []moduleRecommendation176	base, pciIDs, unmount, err := mountPayloadModules()177	if err != nil {178		fmt.Fprintf(os.Stderr, "liken: report: %v\n", err)179	} else {180		defer unmount()181		catalog, err := hardware.LoadCatalog(base, pciIDs)182		if err != nil {183			fmt.Fprintf(os.Stderr, "liken: report: no hardware catalog, so no driver recommendations: %v\n", err)184		} else {185			recommendations = recommendModules(catalog, base)186			// The claimable list is read from the same catalog before187			// any load, and it is never loaded. It describes the188			// machine as the person found it.189			claimable = recommendClaimable(catalog, base)190			// Load the full ordered chains from the payload's tree. The191			// declared-module loader loads each name in order and prints192			// the outcome, exactly as a from-disk boot loads spec.modules.193			// The interfaces and disks these drivers create are real only194			// after this returns and the probe settles again.195			loadDeclaredModulesFrom(base, dedupChains(recommendations), nil)196			quiesceHardware()197		}198	}199200	// The installation stick is itself a disk, and it must not appear201	// in the proposal: it leaves the machine with the person, so a role202	// laid onto it would vanish with them, and a later "wipe and203	// reinstall" would blank it. The stick is findable only here, after204	// the loads above, because its own controller driver (usb-storage,205	// or uas) is usually among them: before they load, the stick has no206	// block device to find. For the same reason, the driver that exists207	// only to reach the stick must not be recommended into the208	// manifest, though it stayed loaded so the report can write its209	// file.210	stick := awaitInstallStick()211	recommendations = withoutStickRecommendations(recommendations, stick.Disk)212213	// Which disks the loads above brought into existence is knowable214	// only here, with the recommendations and the disks in hand215	// together. It is the fact that determines whether this machine can216	// install from a stock image at all.217	measured, cards := readReportDisks(stick)218	disks := markDisksBehindDrivers(measured, recommendations)219220	return hardwareReport{221		UEFI:            uefi,222		StickPath:       stick.Path,223		Recommendations: recommendations,224		Claimable:       claimable,225		Disks:           disks,226		Cards:           cards,227		Interfaces:      observeInterfaces(),228	}, stick229}230231// mountPayloadModules loop-mounts the payload's system image and returns232// the paths into it that the report reads: the kernel's module tree and233// the PCI naming database. The install medium carries the whole OS as234// liken.sqfs beside the release document, and that image holds the full235// module tree that the boot archive deliberately does not. The mount is236// read-only, so it too changes nothing.237func mountPayloadModules() (base, pciIDs string, unmount func(), err error) {238	image := filepath.Join(releasePayloadDir, slotImageName)239	if err := mountReportImage(image, reportImageMount); err != nil {240		return "", "", nil, fmt.Errorf("mounting the payload's system image %s: %w", image, err)241	}242	base = filepath.Join(reportImageMount, "lib/modules", kernelRelease())243	pciIDs = filepath.Join(reportImageMount, "usr/share/hwdata/pci.ids")244	unmount = func() { _ = unix.Unmount(reportImageMount, unix.MNT_DETACH) }245	return base, pciIDs, unmount, nil246}247248// mountReportImage loop-mounts the payload's system image. It is a249// package variable so a test can stand in for the mount, because a250// loop device needs privileges a test does not have.251var mountReportImage = loopMount252253// recommendModules turns the machine's unclaimed devices into ordered254// driver recommendations. For each undriven device it takes the kernel255// build's preferred candidate and expands its soft dependencies, so a256// NIC that needs its PHY library first reads as [realtek, r8169]. The257// evidence it keeps beside each chain is the device in words and its258// modalias fingerprint, so the proposal can say what each driver259// claims. It also keeps each device's place in sysfs, so the report260// can later tell a recommendation that serves the machine from one261// that serves only the installation stick, and tell which disks exist262// only because of a load.263//264// Unclaimed reports one entry per device, not one per fingerprint, so265// a board with two identical NICs offers the same recommendation266// twice. One chain covers both cards, and the recommendation keeps267// both sysfs paths, so the fingerprint is what this loop keeps once.268func recommendModules(catalog *hardware.Catalog, base string) []moduleRecommendation {269	return recommendFor(catalog, base, reportableClass)270}271272// recommendClaimable is the same walk over the same undriven devices,273// for the kinds the report will not load: the hardware a workload274// could claim once a module drives it. A GPU is the case that named275// this. The install does not need it, so the report neither loads it276// nor declares it, but a person who does not know to ask for it never277// learns their machine has it.278func recommendClaimable(catalog *hardware.Catalog, base string) []moduleRecommendation {279	return recommendFor(catalog, base, claimableClass)280}281282// recommendFor turns the undriven devices of the kinds a caller283// selects into ordered driver recommendations.284func recommendFor(catalog *hardware.Catalog, base string, want func(string) bool) []moduleRecommendation {285	devices := hardware.DiscoverDevices(sysfsRoot, catalog.PCI)286	seen := map[string]bool{}287	var recommendations []moduleRecommendation288	for _, u := range catalog.Unclaimed(devices, nil) {289		if len(u.Candidates) == 0 || seen[u.Modalias] || !want(u.Class) {290			continue291		}292		seen[u.Modalias] = true293		rec := moduleRecommendation{294			Device: describeUnclaimed(u) + " (modalias " + u.Modalias + ")",295			Class:  u.Class,296			Chain:  driverChain(base, u.Candidates[0]),297		}298		for _, d := range devices {299			if d.Modalias == u.Modalias {300				rec.SysfsDirs = append(rec.SysfsDirs,301					filepath.Join(sysfsRoot, "bus", d.Bus, "devices", d.Address))302			}303		}304		recommendations = append(recommendations, rec)305	}306	return recommendations307}308309// reportableClass selects which undriven devices the report loads a310// driver for. The answer is storage and network, and nothing else:311// PCI base class 01, PCI subclass 0805 (the SD and eMMC host312// controllers, which the class word already reads as storage), base313// class 02, and the USB mass-storage class, which is how the report314// reaches the installation stick it writes to. These are the devices315// a machine must have to install itself and to join a316// cluster.317//318// The limit is not tidiness. Loading a driver changes the running319// machine, and some drivers change the one part of it the person is320// using. A display driver takes over the framebuffer console, so the321// proposal and the prompt that follows it would scroll away on a322// screen that just switched modes. A driver for an SMBus bridge or a323// sound card changes nothing a manifest cares about, and advice to324// declare one on a headless node is advice to load a driver for no325// reason. The devices this filter drops still appear in the running326// node's unclaimed-hardware report, where they are information rather327// than instruction.328func reportableClass(class string) bool {329	switch class {330	case "storage", "network", "mass-storage":331		return true332	}333	return false334}335336// claimableClass selects which undriven devices the report names as337// hardware a workload could claim. This list loads nothing, so it can338// be generous where reportableClass cannot.339//340// It leaves out the two kinds the section above already covers, and341// the kinds that never hand a device node to a pod: a bridge, a memory342// controller, and the system devices are the machine's own plumbing,343// and a claim on one would deliver nothing. Everything else that a344// module could drive stays in, because whether it is worth driving is345// the operator's judgement and not the report's.346func claimableClass(class string) bool {347	switch class {348	case "", "storage", "network", "mass-storage", "bridge", "memory", "system":349		return false350	}351	return true352}353354// observeInterfaces brings every real interface admin-up and reads back355// its link state. Admin-up is required before the kernel trains the link356// and detects the carrier, and it changes only kernel state in RAM, so357// the report keeps its promise here too. The report waits a few seconds358// after raising the links, because copper autonegotiation takes that359// long, and a carrier read before the link trains would report every360// port dark.361func observeInterfaces() []reportInterface {362	links, err := listLinks()363	if err != nil {364		fmt.Fprintf(os.Stderr, "liken: report: listing interfaces: %v\n", err)365		return nil366	}367368	var raised []netlink.Link369	for _, link := range links {370		attrs := link.Attrs()371		// Loopback is not hardware, and a link with no MAC is a virtual372		// device the report has nothing to say about.373		if attrs.Flags&net.FlagLoopback != 0 || len(attrs.HardwareAddr) == 0 {374			continue375		}376		if err := raiseLink(link); err != nil {377			fmt.Fprintf(os.Stderr, "liken: report: raising %s: %v\n", attrs.Name, err)378		}379		raised = append(raised, link)380	}381	if len(raised) > 0 {382		time.Sleep(3 * time.Second)383	}384385	var interfaces []reportInterface386	for _, link := range raised {387		attrs := link.Attrs()388		interfaces = append(interfaces, reportInterface{389			Name: attrs.Name,390			MAC:  attrs.HardwareAddr.String(),391			Link: linkState(attrs.Name),392		})393	}394	return interfaces395}396397// listLinks and raiseLink list the machine's interfaces and bring one398// admin-up. They are package variables so a test can hand the report399// interfaces of its own, because raising a real link needs privileges a400// test does not have.401var (402	listLinks = netlink.LinkList403	raiseLink = netlink.LinkSetUp404)405406// linkState reads the kernel's word for an interface's link: up, down,407// or unknown. operstate is the kernel's own summary of the carrier, in408// the form a person reads most easily.409func linkState(name string) string {410	if state := sysfsString(filepath.Join(sysfsRoot, "class/net", name), "operstate"); state != "" {411		return state412	}413	return "unknown"414}415416// quiesceHardware waits for the bus probe to go quiet, so a walk of417// sysfs reads a settled machine rather than one mid-enumeration. It418// reuses the boot's own settle helper over the kernel's uevent socket:419// each device that arrives or binds a driver announces itself, and the420// wait ends once a full second passes with no such announcement, or a421// ceiling passes either way. Without the socket it falls back to a fixed422// pause, so a probe still has a moment to finish, and so does a listener423// that stops during the wait, because settle returns at once when its424// channel closes.425func quiesceHardware() {426	ctx, cancel := context.WithCancel(context.Background())427	defer cancel()428	uevents, err := listenForUevents(ctx)429	if err != nil {430		fmt.Fprintf(os.Stderr, "liken: report: no uevent socket, pausing instead: %v\n", err)431		time.Sleep(quiescePause)432		return433	}434	hardware.Settle(ctx, uevents, time.Second, 10*time.Second)435	if listenerStopped(uevents) {436		fmt.Fprintln(os.Stderr, "liken: report: the uevent listener stopped, pausing instead")437		time.Sleep(quiescePause)438	}439}440441// quiescePause is how long the report waits for a probe to finish when442// it cannot hear the probe's uevents.443const quiescePause = 3 * time.Second444445// listenerStopped answers whether a uevent channel is closed. A wake446// still pending on an open channel is consumed, which costs nothing,447// because the caller is done waiting.448func listenerStopped(uevents <-chan struct{}) bool {449	select {450	case _, ok := <-uevents:451		return !ok452	default:453		return false454	}455}456457// writeReportToStick writes the proposal to the root of the458// installation stick's filesystem. The stick is the one the report459// already resolved, so the file lands on the same disk the proposal460// says it excluded. The stick carries a FAT volume; the report mounts461// it read-write only for this write, writes durably (FAT has no462// journal, so the sync before the rename is what makes the file463// whole), syncs, and unmounts, so the filesystem is clean when the464// person pulls the stick.465func writeReportToStick(stick installStick, proposal string) error {466	device := stick.Partition467	if device == "" {468		if stick.ambiguous() {469			return fmt.Errorf("more than one disk carries %s (%s); refusing to guess which is the stick",470				stickInstallPartition, strings.Join(stick.Candidates, ", "))471		}472		return fmt.Errorf("no installation stick found (no partition named %s)", stickInstallPartition)473	}474475	if err := os.MkdirAll(reportStickMount, 0o755); err != nil {476		return err477	}478	if err := mountStick(device, reportStickMount); err != nil {479		return fmt.Errorf("mounting the stick %s: %w", device, err)480	}481482	writeErr := writeFileDurably(filepath.Join(reportStickMount, hardwareReportName), []byte(proposal))483	unix.Sync()484485	// A plain unmount flushes and detaches the filesystem cleanly. If it486	// is busy, a lazy detach at least releases it, so a later boot does487	// not find a stale mount.488	if err := unmountStick(reportStickMount, 0); err != nil {489		_ = unmountStick(reportStickMount, unix.MNT_DETACH)490	}491	return writeErr492}493494// mountStick and unmountStick mount and unmount the stick's FAT volume.495// They are package variables so a test can stand in for the mount,496// because a mount needs privileges a test does not have.497var (498	mountStick = func(device, target string) error {499		return unix.Mount(device, target, "vfat", 0, "")500	}501	unmountStick = unix.Unmount502)503504// rebootAfterReport restarts the machine. A report boot has no k3s and505// no role mounts to tear down, so this is the plain restart syscall506// after a sync, not the supervisor's full shutdown. The report reboots507// rather than powers off, because the machine's next boot is the person508// walking the install once they have edited the manifest. Like every509// terminal path in init, it never lets PID 1 exit.510func rebootAfterReport() {511	syncLogs()512	unix.Sync()513	if err := unix.Reboot(unix.LINUX_REBOOT_CMD_RESTART); err != nil {514		fmt.Fprintf(os.Stderr, "liken: report: reboot failed: %v\n", err)515	}516	for {517		time.Sleep(time.Hour)518	}519}
init/reportdisks.go 97.7%
1package main23// The report boot's view of the machine's block devices.4//5// Before the report can propose a manifest it has to answer three6// questions about the disks in front of it. Which device is the7// installation stick, so that no role lands on the one disk that8// leaves with the person? Which of the rest can hold a role at all,9// as against an optical drive or an empty card reader that the kernel10// also presents as a block device? And which of those can an install11// reach, as against a disk that exists only because this boot loaded12// a driver? Every answer is a small read of sysfs, the way the rest13// of liken discovers storage: the kernel publishes these facts14// already, so nothing here has to derive them.1516import (17	"path/filepath"18	"slices"19	"strings"20	"time"2122	"github.com/liken-sh/liken/liken/machine"23)2425// installStick is the report's one answer about the installation26// stick, resolved once and threaded through everything that needs it:27// the disk walk that leaves the stick out, the proposal that places no28// role on it, and the write of the proposal onto it. Two independent29// lookups could disagree, and a disagreement here has teeth. A stick30// that one lookup missed would be proposed as a data disk, and the31// "wipe and reinstall" entry blanks every disk the manifest names.32type installStick struct {33	// Disk is the kernel name of the disk that carries the stick, and34	// Path is that disk's device node. Both are empty when the report35	// cannot name one disk with confidence.36	Disk string37	Path string38	// Partition is the device node of the FAT volume to mount to write39	// the proposal.40	Partition string41	// Candidates names every disk that carries a partition called42	// liken:install. One is the ordinary case. More than one is an43	// ambiguity the report refuses to guess through, and every disk in44	// the list must then stay out of the proposal.45	Candidates []string46}4748// ambiguous reports whether more than one disk claims to be the49// stick. The report can neither write to such a stick nor place a50// role on any disk that might be it.51func (s installStick) ambiguous() bool {52	return s.Disk == "" && len(s.Candidates) > 053}5455// resolveInstallStick finds the installation stick the same way liken56// recognizes everything else on a disk: by the GPT partition name that57// image/stick.go writes, not by a device path, which changes with58// enumeration order.59func resolveInstallStick() installStick {60	var stick installStick61	var partitions []string62	for _, p := range discoverPartitions() {63		if p.partName != stickInstallPartition {64			continue65		}66		partitions = append(partitions, p.name)67		if !slices.Contains(stick.Candidates, p.disk) {68			stick.Candidates = append(stick.Candidates, p.disk)69		}70	}71	if len(partitions) != 1 {72		return stick73	}74	stick.Disk = stick.Candidates[0]75	stick.Path = devRoot + "/" + stick.Disk76	stick.Partition = devRoot + "/" + partitions[0]77	return stick78}7980// awaitInstallStick polls for the installation stick until it appears81// or the ceiling passes, ending the moment anything claims to be one.82// The poll is on the partition table itself, not on uevents, so it83// also covers the moment between the disk's arrival and its84// partitions'.85//86// The wait exists because usb-storage delays its scan a full second87// after the device attaches (its delay_use), and USB enumeration runs88// on past the settle the report does before this. An immediate walk89// can therefore run before the stick's disk exists. A report boot90// expects a stick by construction: a person picked the entry from one.91// A boot with no stick at all (a hand-typed liken.report) pays the92// ceiling once, and its report stays on the console.93func awaitInstallStick() installStick {94	deadline := time.Now().Add(stickCeiling)95	for {96		stick := resolveInstallStick()97		if len(stick.Candidates) > 0 || time.Now().After(deadline) {98			return stick99		}100		time.Sleep(200 * time.Millisecond)101	}102}103104// readReportDisks lists the disks that could carry a storage role, in105// the form the proposal records. It reuses the same sysfs walk the106// world report uses, and adds the transport, which a person reads to107// recognize a disk by how it attaches. The installation stick is left108// out entirely, because it belongs to the person and not to the109// machine. A disk that only might be the stick stays in the list as110// evidence, marked, so the proposal can account for it and still place111// no role on it.112func readReportDisks(stick installStick) (disks, cards []reportDisk) {113	for _, d := range discoverBlockDevices() {114		if d.Name == stick.Disk || !canHoldARole(d) {115			continue116		}117		measured := reportDisk{118			Name:       d.Name,119			Path:       devicePath(d),120			SizeBytes:  d.SizeBytes,121			Model:      d.Model,122			Transport:  diskTransport(d.Name),123			StableName: firstByIDName(d.StableNames),124			MaybeStick: stick.ambiguous() && slices.Contains(stick.Candidates, d.Name),125		}126		if cardInASlot(d.Name) {127			cards = append(cards, measured)128			continue129		}130		disks = append(disks, measured)131	}132	return disks, cards133}134135// cardInASlot reports whether a block device is a memory card a136// person put in a slot, as against an eMMC module soldered to the137// board. A card leaves the machine with the person, like the install138// stick, so the report names it and proposes nothing onto it. The mmc139// bus publishes what the card is on the card device: MMC for an eMMC140// module, SD for an SD card, SDcombo for an SD card with SDIO on it.141// Only the mmc bus writes a word here at all; a SCSI disk writes its142// peripheral device type as a number in the same attribute, which is143// what canHoldARole reads.144//145// This is about the report's proposal alone. A spec that names an SD146// card by hand still installs onto it, because a person who declares a147// card has decided something the report may not decide for them.148func cardInASlot(name string) bool {149	switch sysfsString(filepath.Join(sysBlock, name), "device/type") {150	case "SD", "SDcombo":151		return true152	}153	return false154}155156// firstByIDName returns the first by-id name among a disk's stable157// names, or "" when the disk offers none. stableNames orders by-id158// names before the by-path name, so the first entry under159// stableDiskByID is the disk's own identity: a by-path name that160// might precede it in some other order would name the disk's port161// instead, a different fact that the proposal must not present as the162// same thing.163func firstByIDName(names []string) string {164	for _, name := range names {165		if strings.HasPrefix(name, stableDiskByID) {166			return name167		}168	}169	return ""170}171172// canHoldARole rejects the block devices that a storage role cannot173// live on. /sys/block lists every device with a bus parent, and a174// repurposed desktop offers several that no partition table belongs175// on: an optical drive, a card reader with no card in it, a176// write-protected medium. A role proposed onto one of those is a177// layout that cannot be claimed.178//179// The three checks are deliberately narrow, because the mistake in the180// other direction is worse. A small disk is still a disk: a SATA DOM181// of a few gigabytes is exactly the boot device some industrial boards182// ship with, and a memory card holding a role is a legitimate, if183// slow, machine. So the filter tests only for the absence of a medium,184// a read-only device, and the SCSI device types that name optical185// drives.186func canHoldARole(d machine.BlockDevice) bool {187	if d.SizeBytes == 0 {188		return false189	}190	dir := filepath.Join(sysBlock, d.Name)191	if sysfsString(dir, "ro") == "1" {192		return false193	}194	// The SCSI peripheral device type is the kernel's own word for195	// what a device is. Type 5 is a CD or DVD drive and type 7 is196	// optical memory; an ordinary disk is type 0. Every disk that197	// arrives through the SCSI layer publishes this attribute, which198	// includes SATA and USB disks, because libata and usb-storage both199	// present their devices to that layer.200	switch sysfsString(dir, "device/type") {201	case "5", "7":202		return false203	}204	return true205}206207// diskTransport names the bus that carries a disk, read from the disk's208// place in the sysfs device tree. The tree's path names every bus the209// device hangs off, so a SATA disk's path passes through an ata node, an210// NVMe disk's through nvme, and so on. This is the same information udev211// derives its ID_BUS from, read straight from the one file the kernel212// already maintains.213func diskTransport(name string) string {214	real, err := filepath.EvalSymlinks(filepath.Join(sysBlock, name))215	if err != nil {216		return ""217	}218	// The order matters only where a path could match more than one219	// word. A disk reached through libata shows both an ata node and the220	// SCSI layer that libata presents it through, and "sata" is the221	// answer a person wants.222	for _, bus := range []struct{ marker, transport string }{223		{"/nvme", "nvme"},224		{"/ata", "sata"},225		{"/usb", "usb"},226		{"/virtio", "virtio"},227		// An mmc card's sysfs path passes through its host228		// controller's mmc_host node, so that segment names the bus229		// for an eMMC module or an SD card.230		{"/mmc", "mmc"},231	} {232		if strings.Contains(real, bus.marker) {233			return bus.transport234		}235	}236	return ""237}238239// markDisksBehindDrivers finds the disks that exist only because this240// report loaded a driver, and records which chain reached them.241//242// A device is unclaimed exactly because the boot path had no driver243// for it. So a disk that sits below such a device is a disk the boot244// path cannot see, and the install path is the boot path: it claims245// and mounts the manifest's disks before it loads a single module the246// manifest declares. Naming the driver in spec.modules would not247// help, and would read as though it did. The proposal says so instead,248// and the machine needs an image that carries the driver in its boot249// modules.250func markDisksBehindDrivers(disks []reportDisk, recs []moduleRecommendation) []reportDisk {251	for i, d := range disks {252		for _, rec := range recs {253			for _, dir := range rec.SysfsDirs {254				if blockDeviceUnder(d.Name, dir) {255					disks[i].BehindModules = appendNew(d.BehindModules, rec.Chain...)256					break257				}258			}259		}260	}261	return disks262}263264// blockDeviceUnder reports whether a disk sits below one device in the265// sysfs device tree. Both paths resolve through their symlinks first,266// because /sys/block holds links into /sys/devices, where the real267// parent-and-child structure is.268func blockDeviceUnder(diskName, deviceDir string) bool {269	if diskName == "" {270		return false271	}272	disk, err := filepath.EvalSymlinks(filepath.Join(sysBlock, diskName))273	if err != nil {274		return false275	}276	device, err := filepath.EvalSymlinks(deviceDir)277	if err != nil {278		return false279	}280	return strings.HasPrefix(disk+"/", device+"/")281}282283// withoutStickRecommendations drops the recommendations that exist284// only to reach the installation stick. The stick's own controller285// requires a driver like any unclaimed device, and the report rightly286// loaded it, because without it there is no stick to write the287// proposal to. But the stick leaves the machine with the person, so288// its driver does not belong in the machine's manifest. A device289// serves the stick when the stick's disk sits below it in the sysfs290// device tree; a recommendation is dropped only when every device291// that named it serves the stick, so a second, permanent disk on the292// same kind of controller keeps its driver recommended.293func withoutStickRecommendations(recs []moduleRecommendation, stick string) []moduleRecommendation {294	if stick == "" {295		return recs296	}297	var kept []moduleRecommendation298	for _, rec := range recs {299		serves := len(rec.SysfsDirs) > 0300		for _, dir := range rec.SysfsDirs {301			if !blockDeviceUnder(stick, dir) {302				serves = false303				break304			}305		}306		if !serves {307			kept = append(kept, rec)308		}309	}310	return kept311}312313// appendNew adds the names that are not in the list already, keeping314// the order they arrive in. Two controllers of the same kind reach315// their disks through the same chain, and a disk names each chain316// that reached it once.317func appendNew(existing []string, names ...string) []string {318	for _, name := range names {319		if !slices.Contains(existing, name) {320			existing = append(existing, name)321		}322	}323	return existing324}
init/reportlayout.go 97.7%
1package main23// Fitting liken's storage roles onto the disks the report measured.4//5// A proposal that names sizes the disks cannot hold is worse than no6// proposal: it reads as a manifest a person can install from, and the7// install refuses it at the first disk it claims. The installer lays8// each disk out with exact allocations (claim.go), so every byte a9// role asks for must exist on the disk that role names. This file is10// the arithmetic that keeps that promise. It takes the disks the11// report measured and returns the roles those disks can carry, at12// sizes those disks can hold.13//14// The planner is pure. It reads no sysfs and touches no disk, so the15// tests drive it with disks made up by hand, and check the result16// against the same partition math the install runs.1718import (19	"fmt"2021	"github.com/liken-sh/liken/liken/machine"22)2324// These are the sizes the layout starts from. The boot and system25// roles are fixed, because their contents are fixed: a slot holds one26// whole OS image, and the GRUB roles hold structures whose sizes the27// firmware and GRUB set.28//29// The three data roles are not equal, and the difference determines30// which one gives up space on a small disk. clusterState is the31// operating system's own working set: containerd unpacks every image32// the node runs underneath it. podStorage and podEphemeral are the33// workloads' space, and a workload that runs out of it fails one34// workload.35//36// Every size here is a whole number of mebibytes, which is also the37// alignment the installer gives each partition, so a role never38// occupies more of a disk than its size says.39const (40	biosBootBytes         = 1 << 2041	bootHomeBytes         = 64 << 2042	systemSlotBytes       = 1 << 3043	machineStateBytes     = 64 << 2044	machineEphemeralBytes = 512 << 204546	// clusterStateBytes is what clusterState asks for when a disk can47	// hold it. The role mounts /var/lib/rancher, which holds k3s's48	// database, its TLS material, and containerd's image store, so its49	// size follows what the node runs and not what a person prefers.50	// The images add up faster than the number suggests: liken's own51	// bundled images, k3s's packaged ones (coredns, traefik,52	// metrics-server, the local-path provisioner), and every workload53	// image the operator deploys, and an upgrade holds the old set and54	// the new set at the same time. 6Gi is what liken's own public node55	// runs on, which is the one place a person has had to choose this56	// number against real work.57	clusterStateBytes = 6 << 305859	// clusterStateFloorBytes is the smallest clusterState this layout60	// will name. The lab's hardware-parity guest converges with the61	// whole control-plane pod set in this much, so it is the smallest62	// size with evidence behind it. Under the floor the layout names no63	// clusterState at all, for the reason belowFloorNote gives.64	clusterStateFloorBytes = 2 << 306566	// clusterStateShare is the fraction of a roomy disk that67	// clusterState takes, and clusterStateCeilingBytes is where that68	// growth stops. The size is permanent: podStorage sits directly69	// behind clusterState in the canonical order, and a partition can70	// only grow into free space that follows it, so the number this71	// proposal names is the number the machine keeps. A disk with room72	// to spare should spend it here rather than on scratch space,73	// because this is the role that cannot be corrected later. The74	// ceiling exists because an image store is large, not unbounded,75	// and a person who needs more than this is a person who should76	// choose the number themselves.77	clusterStateShare        = 878	clusterStateCeilingBytes = 64 << 307980	// podEphemeralCeilingBytes bounds what kubelet's scratch space81	// takes from a roomy disk. podEphemeral is last in the canonical82	// order, so it is the role that takes the rest, and on a large disk83	// that leaves the volumes a workload claims with almost nothing.84	// Above this size, the space goes to podStorage instead. This role85	// also carries the pod logs (podlogs.go), which is one more reason86	// a proposal names it whenever the disk can hold it.87	podEphemeralCeilingBytes = 64 << 308889	// podStorageBytes is what podStorage asks for. This role is the90	// local-path provisioner's pool: the space pods claim by name. It91	// is the operator's to size against their own workloads, so it is92	// also the first space the layout takes back when a disk is small.93	podStorageBytes = 4 << 309495	// dataRoleFloor is the smallest podStorage or podEphemeral may be96	// before the layout leaves the role out. A role below this size97	// holds too little to be worth the space it takes from the roles98	// beside it, and a role that is absent from the spec is not an99	// error: its directory stays on the machine's RAM root.100	dataRoleFloor = 256 << 20101102	// dataShareUnit rounds a scaled data role down to a size a person103	// reads without effort. An exact remainder of a disk is a number104	// nobody wants to see in a manifest.105	dataShareUnit = 64 << 20106107	// gptOverhead is what the partition table itself costs. The108	// primary header and its entry array sit in the first sectors, and109	// a mirror of both sits in the last 33; the installer then starts110	// the first partition on the 1MiB boundary that every partitioner111	// aligns to. One mebibyte at each end covers both ends with room112	// to spare.113	gptOverhead = 2 << 20114)115116// plannedRole is one role placed on one disk at one size. An empty117// Size means the role takes the rest of its disk, which the spec118// allows for one role per disk. Comment is the reason a person reads119// beside the number.120type plannedRole struct {121	Name    machine.StorageRoleName122	Device  string123	Size    string124	Comment string125}126127// storageLayout is the plan for a whole machine: the roles, in the128// canonical order the installer lays them down, and the notes that129// explain what the plan could not do. A layout with no roles is a130// real answer: it means no disk this report saw can carry a liken131// install, and the notes say what would.132type storageLayout struct {133	Roles []plannedRole134	Notes []string135}136137// planStorageLayout fits the roles onto the disks. It puts the138// machine's own roles on one disk and the cluster's data on another139// when the machine has one to spare, so the cluster's state survives a140// reinstall that replaces the system disk. It never plans a role onto141// a disk that might be the installation stick, and it prefers a disk142// the install can reach over one that only this boot can see.143func planStorageLayout(measured []reportDisk, uefi bool) storageLayout {144	candidates := placeableDisks(measured)145	if len(candidates) == 0 {146		return storageLayout{Notes: []string{147			"This report saw no disk that can carry a storage role."}}148	}149150	system, ok := pickSystemDisk(candidates, uefi)151	if !ok {152		grubRoles := ""153		if !uefi {154			grubRoles = fmt.Sprintf(", %s for GRUB's core image, and %s for GRUB's config",155				machine.SizeText(biosBootBytes), machine.SizeText(bootHomeBytes))156		}157		return storageLayout{Notes: []string{fmt.Sprintf(158			"No disk here can hold liken's own roles. They need %s on one disk: two %s system slots, %s of machine state, %s of /tmp%s. The largest disk this report saw offers %s. Attach a larger disk, then run this report again.",159			gib(systemRoleBytes(uefi)), machine.SizeText(systemSlotBytes), machine.SizeText(machineStateBytes),160			machine.SizeText(machineEphemeralBytes), grubRoles, gib(largestDisk(candidates).SizeBytes))}}161	}162163	layout := storageLayout{Roles: systemRoles(system.deviceName(), uefi)}164	data, available := pickDataDisk(candidates, system, uefi)165	if data.Path == system.Path {166		layout.Notes = append(layout.Notes, fmt.Sprintf(167			"Every role lives on %s. No other disk here gives the cluster's data more room than this disk has left over. A reinstall replaces the system slots and the data roles together.", system.Path))168	} else {169		layout.Notes = append(layout.Notes, fmt.Sprintf(170			"The durable roles live on %s, so the cluster's state and its volumes survive an install onto a replaced system disk. A wipe and reinstall is the other case: it erases every disk this manifest declares, including this one.", data.Path))171	}172173	roles, notes := dataRoles(data.deviceName(), available)174	layout.Roles = append(layout.Roles, roles...)175	layout.Notes = append(layout.Notes, notes...)176	return layout177}178179// placeableDisks is the disks a role may land on, best first. A disk180// that might be the installation stick is out entirely: the stick181// leaves the machine with the person, and a role on it would vanish182// with them. A disk that appeared only after this boot loaded a driver183// comes last, because an install claims its disks before it loads any184// module a manifest declares, so it never sees that disk at all.185func placeableDisks(measured []reportDisk) []reportDisk {186	var reachable, behind []reportDisk187	for _, d := range measured {188		switch {189		case d.MaybeStick:190			continue191		case len(d.BehindModules) > 0:192			behind = append(behind, d)193		default:194			reachable = append(reachable, d)195		}196	}197	return append(reachable, behind...)198}199200// usableBytes is what one disk offers to roles: its size, less what201// the partition table takes at each end.202func usableBytes(d reportDisk) uint64 {203	if d.SizeBytes <= gptOverhead {204		return 0205	}206	return d.SizeBytes - gptOverhead207}208209// systemRoleBytes is the space the machine's own roles need: the two210// system slots, the machine's state, its /tmp, and, on a BIOS machine,211// the two roles that UEFI firmware would otherwise supply.212func systemRoleBytes(uefi bool) uint64 {213	total := uint64(2*systemSlotBytes + machineStateBytes + machineEphemeralBytes + gptOverhead)214	if !uefi {215		total += biosBootBytes + bootHomeBytes216	}217	return total218}219220// pickSystemDisk chooses the disk that boots the machine: the first221// placeable disk with room for the machine's own roles. Order matters222// more than size here. The disks come in the kernel's enumeration223// order, and the first disk is the one a person points at when they224// say "the system disk".225func pickSystemDisk(candidates []reportDisk, uefi bool) (reportDisk, bool) {226	for _, d := range candidates {227		if usableBytes(d)+gptOverhead >= systemRoleBytes(uefi) {228			return d, true229		}230	}231	return reportDisk{}, false232}233234// pickDataDisk chooses where the cluster's data lives, and says how235// much room it has there. A second disk is the better home, because it236// survives a reinstall of the system disk, but only when it holds more237// than the system disk has left over. A 64Mi flash card is a second238// disk and not a home for a cluster's state.239func pickDataDisk(candidates []reportDisk, system reportDisk, uefi bool) (reportDisk, uint64) {240	leftover := usableBytes(system) - (systemRoleBytes(uefi) - gptOverhead)241	best, bestAvailable := system, leftover242	for _, d := range candidates {243		if d.Path == system.Path {244			continue245		}246		if available := usableBytes(d); available > bestAvailable {247			best, bestAvailable = d, available248		}249	}250	return best, bestAvailable251}252253// systemRoles places the machine's own roles, at their fixed sizes.254func systemRoles(device string, uefi bool) []plannedRole {255	var roles []plannedRole256	if !uefi {257		roles = append(roles,258			plannedRole{machine.BIOSBootRole, device, machine.SizeText(biosBootBytes), "# GRUB core image; a tiny raw partition"},259			plannedRole{machine.BootHomeRole, device, machine.SizeText(bootHomeBytes), "# GRUB config and environment block"})260	}261	return append(roles,262		plannedRole{machine.SystemARole, device, machine.SizeText(systemSlotBytes), "# one OS slot; the blue-green pair"},263		plannedRole{machine.SystemBRole, device, machine.SizeText(systemSlotBytes), ""},264		plannedRole{machine.MachineStateRole, device, machine.SizeText(machineStateBytes), "# staged and proven manifests"},265		plannedRole{machine.MachineEphemeralRole, device, machine.SizeText(machineEphemeralBytes), "# the OS's /tmp"})266}267268// roleNote is the sentence every proposal that plans the durable roles269// carries. It teaches the one distinction a person needs before they270// edit a size: which number is theirs to choose, and which number the271// machine chooses for them.272const roleNote = "clusterState holds k3s's database, its TLS material, and containerd's image store, so this node's images decide its size. Raise it if this node runs many images or large ones. This size is permanent: podStorage follows clusterState on the disk, so clusterState cannot grow later. podStorage is the local-path provisioner's pool, which is yours to size for the volumes your workloads claim."273274// dataRoles fits the cluster's roles into the space that is left, in275// the order that keeps the node able to run at all.276//277// The order follows from what each role holds. A node with too little278// podStorage refuses one workload's volume claim. A node with too279// little clusterState cannot unpack the images it is told to run, so280// it fails when an operator deploys anything, weeks after the install281// that sized it. So podStorage shrinks and then goes, then podEphemeral282// falls toward its floor, and clusterState gives up space last and283// never below the floor a node is known to converge in. A role that284// does not fit is left out rather285// than shrunk to nothing, and every departure from the conventional286// layout gets a note that says what it costs.287func dataRoles(device string, available uint64) ([]plannedRole, []string) {288	cluster := plannedRole{machine.ClusterStateRole, device, "", "# k3s's database, TLS material, and containerd's images"}289	pods := plannedRole{machine.PodStorageRole, device, "", "# size to your workloads' volumes"}290	ephemeral := plannedRole{machine.PodEphemeralRole, device, "", "# takes the rest of this disk"}291292	// A disk with room to spare. The conventional sizes were measured293	// against the smallest machines liken runs, and naming them on a294	// large disk spends that disk on scratch space: podEphemeral takes295	// the rest, so a 477Gi disk would carry 6Gi of images, 4Gi of296	// volumes, and 466Gi of /var/lib/kubelet. The roles that hold297	// something a person cares about take the space instead.298	if scaled := scaledClusterState(available); scaled > clusterStateBytes {299		left := available - scaled300		scratch := min(uint64(podEphemeralCeilingBytes), left/2)301		cluster.Size = machine.SizeText(scaled)302		pods.Size = machine.SizeText((left - scratch) / (1 << 30) * (1 << 30))303		return []plannedRole{cluster, pods, ephemeral}, []string{roleNote, fmt.Sprintf(304			"%s has room beyond the conventional sizes, so clusterState takes %s and podStorage takes %s. A larger clusterState is worth more here than a larger scratch space, because clusterState cannot grow after the install and podEphemeral takes whatever is left.",305			device, cluster.Size, pods.Size)}306	}307	if available >= clusterStateBytes+podStorageBytes+dataRoleFloor {308		cluster.Size, pods.Size = machine.SizeText(clusterStateBytes), machine.SizeText(podStorageBytes)309		return []plannedRole{cluster, pods, ephemeral}, []string{roleNote}310	}311	if share := spareAfter(available, clusterStateBytes+dataRoleFloor); share >= dataRoleFloor {312		cluster.Size, pods.Size = machine.SizeText(clusterStateBytes), machine.SizeText(share)313		return []plannedRole{cluster, pods, ephemeral}, []string{roleNote, fmt.Sprintf(314			"%s cannot hold the conventional %s of podStorage beside clusterState, so podStorage takes %s. clusterState keeps its %s, because a node that cannot unpack an image cannot run the pod that requires it.",315			device, machine.SizeText(podStorageBytes), machine.SizeText(share), machine.SizeText(clusterStateBytes))}316	}317	if available >= clusterStateBytes+dataRoleFloor {318		cluster.Size = machine.SizeText(clusterStateBytes)319		return []plannedRole{cluster, ephemeral}, []string{roleNote, fmt.Sprintf(320			"%s has room for clusterState and kubelet's scratch space only, so podStorage is left out. A pod that claims a volume gets one from the machine's RAM root, and loses it at the next reboot. clusterState keeps its %s, because the images this node runs live there.",321			device, machine.SizeText(clusterStateBytes))}322	}323	if share := spareAfter(available, dataRoleFloor); share >= clusterStateFloorBytes {324		cluster.Size = machine.SizeText(share)325		return []plannedRole{cluster, ephemeral}, []string{roleNote, fmt.Sprintf(326			"%s cannot hold a %s clusterState, so podStorage is left out and clusterState takes %s. That is enough for the images liken and k3s bring, and little more: a pull of a large image can fail with no space left, and an upgrade that holds the old images and the new ones at the same time may not fit. Attach a larger disk before this machine runs many images.",327			device, machine.SizeText(clusterStateBytes), machine.SizeText(share))}328	}329	if available >= dataRoleFloor {330		return []plannedRole{ephemeral}, []string{fmt.Sprintf(331			"%s offers %s to the cluster's roles, and a liken node's image store needs %s at the least, so this proposal declares no clusterState and no podStorage. A size under that floor installs without complaint and then fails on an image the node pulls weeks later, which is worse than no role at all. kubelet's scratch space takes this disk. The cluster's state and its volumes stay on the machine's RAM root, so the node imports every image again after every reboot. Attach a larger disk for this machine's durable roles.",332			device, gib(available), gib(clusterStateFloorBytes))}333	}334	return nil, []string{fmt.Sprintf(335		"%s has no room for any data role. The cluster's state, its volumes, and kubelet's scratch space all stay on the machine's RAM root. Attach a larger disk before this machine runs real workloads.", device)}336}337338// scaledClusterState is what clusterState asks for on a disk of a339// given size: a share of the disk, in whole gibibytes, up to the340// ceiling. The result is rounded down to a gibibyte because a person341// reads this number and decides whether to keep it, and an exact342// fraction of a disk reads as arithmetic rather than as a choice.343func scaledClusterState(available uint64) uint64 {344	scaled := available / clusterStateShare / (1 << 30) * (1 << 30)345	return min(scaled, uint64(clusterStateCeilingBytes))346}347348// spareAfter is what a disk has left once a reservation is taken out349// of it, rounded down so the number reads well in a manifest. The350// arithmetic is unsigned, so a reservation larger than the disk has to351// return zero rather than wrap to an enormous size.352func spareAfter(available, reserved uint64) uint64 {353	if available <= reserved {354		return 0355	}356	return (available - reserved) / dataShareUnit * dataShareUnit357}358359// largestDisk is the biggest of a set, for the message that says what360// the machine offers against what liken needs.361func largestDisk(disks []reportDisk) reportDisk {362	largest := reportDisk{}363	for _, d := range disks {364		if d.SizeBytes > largest.SizeBytes {365			largest = d366		}367	}368	return largest369}
init/reportproposal.go 97.8%
1package main23// Composing the hardware report's proposal.4//5// The report boot answers the question that comes before a new6// machine's first install: what does this hardware need in its7// manifest? It answers with a whole Machine manifest, ready to edit,8// with the evidence for every line written beside it as a comment.9// This file turns the facts the boot gathered into that document.10//11// The proposal makes a promise: a person renames the machine, checks12// the sizes, and installs. Everything here serves that promise. A13// driver that the install cannot load in time is not declared. The14// proposal names it in a warning instead. A size that a disk cannot15// hold is not written at all (reportlayout.go does that arithmetic).16// A port with no cable is evidence, not a declaration.17//18// The composition is a pure function. It takes the enumerated hardware19// as a plain value and returns the proposal text. Nothing here touches20// sysfs, netlink, or a disk, so a test drives the whole document from21// fabricated hardware, and the boot flow in report.go owns the messy22// half: mounting the module tree, loading drivers, and observing what23// appears.2425import (26	"fmt"27	"strings"28)2930// reportDisk is one disk as the report observed it: its kernel name31// and device path this boot, its size, its model when the bus32// publishes one, and the transport that carries it (sata, nvme, usb,33// virtio). The transport is evidence a person reads to recognize the34// disk. StableName is the disk's own by-id name, when it has one; the35// proposal declares a role by this name in preference to the device36// path, because the path is only a hint that matters on the boot that37// claims the disk, and the next boot can hand that same path to a38// different disk.39//40// The last two fields carry what the report learned about the disk's41// place in an install, rather than about the disk itself.42// BehindModules names the drivers that had to load before this disk43// existed, which means an install from a stock image never sees it.44// MaybeStick marks a disk that could be the installation stick, on a45// machine where the report could not tell which one is.46type reportDisk struct {47	Name          string48	Path          string49	SizeBytes     uint6450	Model         string51	Transport     string52	StableName    string53	BehindModules []string54	MaybeStick    bool55}5657// deviceName is the value a proposal declares for this disk: its58// stable name when it has one, its kernel device path otherwise. Only59// a by-id name qualifies as a stable name here; a by-path name60// belongs to the disk's port, not the disk, so it is never proposed61// as the disk's identity.62func (d reportDisk) deviceName() string {63	if d.StableName != "" {64		return d.StableName65	}66	return d.Path67}6869// reportInterface is one network interface as the report observed it,70// after it loaded the recommended drivers and brought the link71// admin-up. The name is real only because a driver bound the card:72// eth0 does not exist until its driver registers it. Link is the73// kernel's own word for the carrier, which a person reads to tell a74// connected port from a dark one.75type reportInterface struct {76	Name string77	MAC  string78	Link string79}8081// moduleRecommendation pairs one unclaimed device with the ordered82// driver chain that would bind it. Chain is the softdep-expanded list,83// in load order, so a NIC that needs its PHY library first reads as84// [realtek, r8169], not r8169 alone. Device names the hardware in85// words, so the proposal's comment can say which device each driver86// claims. Class is the device's kind, which determines whether the chain87// can be declared at all: a network driver loads in time to serve the88// machine, and a storage driver does not. SysfsDirs is where the89// devices behind this fingerprint sit in sysfs; the composition90// ignores it, and the report boot uses it to tell a driver that91// serves the machine from one that serves only the installation92// stick, and to tell which disks appeared only because of a load.93type moduleRecommendation struct {94	Device    string95	Class     string96	Chain     []string97	SysfsDirs []string98}99100// storageClass reports whether this recommendation drives storage.101// Such a chain never belongs in spec.modules. The declared modules102// load after storage settles, and storage settles by claiming and103// mounting the disks the manifest names, so a disk behind one of104// these drivers does not exist yet at the only moment it matters.105// The fix is an image that carries the driver in its boot modules,106// not a line in the manifest.107func (r moduleRecommendation) storageClass() bool {108	return r.Class == "storage" || r.Class == "mass-storage"109}110111// hardwareReport is everything the report enumerated, in the form the112// composition consumes. It is a plain value on purpose: the boot fills113// it from real hardware, and a test fills it by hand, and both reach114// the same document through composeHardwareReport.115type hardwareReport struct {116	// UEFI records whether UEFI firmware booted this machine. A UEFI117	// machine keeps its boot entries in firmware and needs no118	// biosBoot/bootHome roles; a BIOS machine needs liken to supply119	// those roles, so the proposal declares them.120	UEFI bool121	// StickPath is the installation stick's own device path, when the122	// report found it among the disks. The stick leaves the machine123	// with the person, so it never appears in Disks and no role may124	// land on it; the proposal names it so a person counting their125	// disks finds every one accounted for.126	StickPath       string127	Recommendations []moduleRecommendation128	// Claimable is the hardware this machine has and does not drive,129	// beyond the disks and network ports the install needs. The report130	// never loads these drivers, and the proposal never declares them:131	// it names them in comments, because whether this machine should132	// drive its GPU is the operator's decision, and they cannot make it133	// if nothing tells them the GPU is there.134	Claimable []moduleRecommendation135	Disks     []reportDisk136	// Cards is every memory card the report found in a slot. A card137	// leaves the machine with the person, so it never appears in138	// Disks and no role may land on it. The proposal still names it,139	// so a person counting their disks finds every one accounted140	// for.141	Cards      []reportDisk142	Interfaces []reportInterface143}144145const reportHeader = `# A proposed Machine manifest, written by the liken hardware report.146#147# This machine booted the report entry from the installation stick. The148# report loaded the drivers this hardware names, watched which disks and149# interfaces appeared, and wrote its findings here. It changed nothing150# on the machine.151#152# Read this file, edit the parts marked CHANGE-ME or "size to ...", and153# use it as the machine's manifest in your deployment layer. Then boot154# the install entry for this machine. The comments beside each field are155# the evidence the report gathered; delete them once you have read them.156`157158// composeHardwareReport renders the whole proposal from the enumerated159// hardware. The result is a valid Machine manifest whose roles fit the160// disks this machine has, so a person can install from it after only161// renaming the machine and checking the sizes.162func composeHardwareReport(r hardwareReport) string {163	var b strings.Builder164	b.WriteString(reportHeader)165	b.WriteString("apiVersion: liken.sh/v1alpha1\n")166	b.WriteString("kind: Machine\n")167	b.WriteString("metadata:\n")168	b.WriteString("  # The name this machine has in your deployment. Boot entries\n")169	b.WriteString("  # carry it as liken.machine=, so it must match a name the\n")170	b.WriteString("  # stick's menu offers.\n")171	b.WriteString("  name: CHANGE-ME\n")172	b.WriteString("spec:\n")173	composeModules(&b, r.Recommendations, r.Claimable)174	composeNetwork(&b, r.Interfaces)175	composeStorage(&b, r)176	return b.String()177}178179// composeModules writes the spec.modules section: the drivers the180// report recommends, in load order, with one comment for each device181// that named them. The declared list is the deduplicated union of the182// chains a manifest can actually load, because a module named twice183// would ask the loader to load it twice. Storage chains are held back184// from the list and stated as a warning instead, for the reason185// storageClass explains.186func composeModules(b *strings.Builder, recs, claimable []moduleRecommendation) {187	b.WriteString("  # Extra kernel modules this machine's hardware requires, beyond\n")188	b.WriteString("  # the drivers the OS already loads. Each comment names a device\n")189	b.WriteString("  # with no driver bound and the modules that would bind it, in\n")190	b.WriteString("  # the order to load them. The report looks at storage and\n")191	b.WriteString("  # network devices only: a machine needs those two kinds to\n")192	b.WriteString("  # install itself and join a cluster, and loading anything else\n")193	b.WriteString("  # would change the machine a person is standing in front of.\n")194	for _, rec := range recs {195		fmt.Fprintf(b, "  #   %s: %s\n", rec.Device, strings.Join(rec.Chain, ", then "))196	}197198	var declarable []moduleRecommendation199	var bootPath []moduleRecommendation200	for _, rec := range recs {201		if rec.storageClass() {202			bootPath = append(bootPath, rec)203		} else {204			declarable = append(declarable, rec)205		}206	}207	for _, rec := range bootPath {208		writeComment(b, "  ", fmt.Sprintf(209			"WARNING: %s is a storage controller, and spec.modules cannot supply its driver. The declared modules load after storage settles, so a disk behind %s does not exist on the boot that claims disks. Build a liken image whose image/boot-modules.conf names %s, and install this machine from that image.",210			rec.Device, strings.Join(rec.Chain, " and "), strings.Join(rec.Chain, " and ")))211	}212213	if len(declarable) == 0 {214		b.WriteString("  # This report found no unclaimed storage or network device\n")215		b.WriteString("  # whose driver a manifest can load. If an interface you\n")216		b.WriteString("  # expect is missing below, its controller may need a driver\n")217		b.WriteString("  # this image does not carry.\n")218		b.WriteString("  modules: []\n")219		composeClaimable(b, claimable)220		return221	}222	b.WriteString("  modules:\n")223	for _, name := range dedupChains(declarable) {224		fmt.Fprintf(b, "    - %s\n", name)225	}226	composeClaimable(b, claimable)227}228229// composeClaimable writes the hardware this machine has and does not230// drive, as lines a person uncomments.231//232// The report loads storage and network drivers and no others, because233// loading a driver changes the machine while a person stands in front234// of it, and a display driver changes the very screen they are235// reading. That limit is about loading, not about knowing. This236// machine's GPU is often the reason someone bought it, and a proposal237// that says nothing about it asks a person to already know what to ask238// for.239//240// The lines stay commented, so the proposal installs exactly as it241// reads, and driving this hardware stays a decision somebody makes.242func composeClaimable(b *strings.Builder, claimable []moduleRecommendation) {243	if len(claimable) == 0 {244		return245	}246	b.WriteString("  # This machine also has hardware that nothing drives. The\n")247	b.WriteString("  # install does not need it, so the report left it alone.\n")248	b.WriteString("  # Uncomment a line to load its driver, and the machine\n")249	b.WriteString("  # publishes the device for workloads to claim. A device with\n")250	b.WriteString("  # no driver is not claimable, and appears in the node's\n")251	b.WriteString("  # unclaimed hardware instead.\n")252	for _, rec := range claimable {253		writeComment(b, "    ", rec.Device)254		for _, name := range rec.Chain {255			fmt.Fprintf(b, "    #- %s\n", name)256		}257	}258}259260// composeNetwork writes the spec.network section from the interfaces261// the report observed. The names are the real kernel names, because the262// report loaded the drivers before it read them.263//264// Only the ports with a carrier are declared, and the dark ports are265// written as commented-out entries with the link state the report266// read. The reason is the cost of a declaration: init configures each267// declared interface in turn, and waits up to thirty seconds for a268// DHCP lease on each one, so a declared port with no cable in it269// delays every boot of the machine by that much. The evidence for270// every port is still here, so a person who moves a cable uncomments271// one line instead of running the report again.272func composeNetwork(b *strings.Builder, ifaces []reportInterface) {273	b.WriteString("  # The network interfaces this report saw once the drivers above\n")274	b.WriteString("  # were loaded and each link was brought up. A name is real only\n")275	b.WriteString("  # after a driver binds the card, so these are the true names.\n")276	if len(ifaces) == 0 {277		b.WriteString("  # This report saw no interface. Its controller may need a\n")278		b.WriteString("  # driver this report could not name.\n")279		b.WriteString("  network: {}\n")280		return281	}282283	var connected, dark []reportInterface284	for _, ifc := range ifaces {285		// The kernel reports "down" only when a driver tracks the286		// carrier and the carrier is absent. A driver that does not287		// track the carrier reports "unknown", and a port the report288		// cannot judge is declared rather than hidden.289		if ifc.Link == "down" || ifc.Link == "lowerlayerdown" {290			dark = append(dark, ifc)291			continue292		}293		connected = append(connected, ifc)294	}295296	if len(connected) == 0 {297		writeComment(b, "  ", "No port had a carrier when this report ran. This proposal declares no interface, so liken configures the first port it finds. Connect the cable you will use and run the report again, or declare the port yourself from the names below.")298		writeDarkInterfaces(b, dark)299		b.WriteString("  network: {}\n")300		return301	}302303	if len(dark) > 0 {304		writeComment(b, "  ", "Only the ports with a carrier are declared here. liken waits up to thirty seconds for a DHCP lease on each declared interface, one after another, so a declared port with no cable delays every boot by that much. The dark ports follow the declared ones as comments; uncomment a port after you connect its cable.")305	}306	b.WriteString("  network:\n")307	b.WriteString("    interfaces:\n")308	for i, ifc := range connected {309		fmt.Fprintf(b, "      # MAC %s, link %s\n", ifc.MAC, ifc.Link)310		if i == 0 {311			b.WriteString("      # This first interface uses DHCP. For a cluster segment\n")312			b.WriteString("      # with a fixed address, add an \"address: 10.0.0.N/24\" line.\n")313		}314		fmt.Fprintf(b, "      - name: %s\n", ifc.Name)315	}316	for _, ifc := range dark {317		fmt.Fprintf(b, "      # MAC %s, link %s\n", ifc.MAC, ifc.Link)318		fmt.Fprintf(b, "      #- name: %s\n", ifc.Name)319	}320}321322// writeDarkInterfaces lists the ports a person could declare, as323// comments, for the machine where no port had a carrier at all.324func writeDarkInterfaces(b *strings.Builder, dark []reportInterface) {325	for _, ifc := range dark {326		fmt.Fprintf(b, "  #   - name: %s   # MAC %s, link %s\n", ifc.Name, ifc.MAC, ifc.Link)327	}328}329330// composeStorage writes the spec.storage section. It first lists every331// disk the report saw as comments, the evidence a person matches332// against the machine in front of them, then writes the layout the333// planner fitted onto those disks, with the planner's notes above it.334// Those notes carry what the sizes alone cannot say: which role the335// machine sizes for itself because containerd's images live in it,336// which role is the operator's to choose, and what the planner had to337// give up on a small disk.338func composeStorage(b *strings.Builder, r hardwareReport) {339	b.WriteString("  # The disks this report saw. A role below names a disk by its\n")340	b.WriteString("  # stable name when it has one, because that name belongs to\n")341	b.WriteString("  # the disk itself and survives a port move; the kernel path\n")342	b.WriteString("  # appears only when the disk offers no such identity, and the\n")343	b.WriteString("  # next boot can hand that same path to a different disk.\n")344	b.WriteString("  # Either way, the value matters only on the boot that claims\n")345	b.WriteString("  # the disk; after that, liken finds each role by the name it\n")346	b.WriteString("  # writes into the GPT.\n")347	for _, d := range r.Disks {348		fmt.Fprintf(b, "  #   %s  %s  %s  (%s)%s%s\n",349			d.Path, gib(d.SizeBytes), orUnknown(d.Model), orUnknown(d.Transport), stableNameField(d), diskCaveat(d))350	}351	if r.StickPath != "" {352		fmt.Fprintf(b, "  #   (%s is the installation stick; it leaves with you,\n", r.StickPath)353		b.WriteString("  #   so it is not listed and no role may live on it)\n")354	}355	// A card in a slot leaves with the person the same way the stick356	// does, so the report says it saw the card and says why it357	// proposes nothing onto it.358	for _, c := range r.Cards {359		fmt.Fprintf(b, "  #   (%s is a %s memory card in a slot; a card leaves with\n", c.Path, gib(c.SizeBytes))360		b.WriteString("  #   you, so no role is proposed on it. Declare it yourself if\n")361		b.WriteString("  #   this card stays in this machine.)\n")362	}363	for _, warning := range storageWarnings(r) {364		writeComment(b, "  ", warning)365	}366367	layout := planStorageLayout(r.Disks, r.UEFI)368	for _, note := range layout.Notes {369		writeComment(b, "  ", note)370	}371	if len(layout.Roles) == 0 {372		b.WriteString("  storage: {}\n")373		return374	}375376	b.WriteString("  storage:\n")377	if r.UEFI {378		b.WriteString("    # This machine booted UEFI, so the firmware holds its boot\n")379		b.WriteString("    # entries and it needs no biosBoot or bootHome role.\n")380	} else {381		b.WriteString("    # This machine booted BIOS, so liken supplies the boot\n")382		b.WriteString("    # bookkeeping that UEFI firmware would otherwise hold.\n")383	}384	for _, role := range layout.Roles {385		writeRole(b, string(role.Name), role.Device, role.Size, role.Comment)386	}387}388389// stableNameField is the fragment a disk's evidence line adds when390// the disk offers a stable name: the same name a role's device: value391// uses, placed beside the model so a person can match the name in the392// roles below to the disk they are holding.393func stableNameField(d reportDisk) string {394	if d.StableName == "" {395		return ""396	}397	return "  " + d.StableName398}399400// diskCaveat is the short note that the report appends to a disk's401// evidence line when the disk is not an ordinary target for a role.402func diskCaveat(d reportDisk) string {403	switch {404	case d.MaybeStick:405		return "  <- this disk may be the installation stick"406	case len(d.BehindModules) > 0:407		return "  <- needs " + strings.Join(d.BehindModules, " and ") + " in the image's boot modules"408	}409	return ""410}411412// storageWarnings states, in the proposal itself, the two facts that413// make a proposal uninstallable as written: a disk this boot could414// see but an install cannot, and a stick the report could not415// identify. Both are the operator's to resolve, and neither is416// visible from the manifest alone.417func storageWarnings(r hardwareReport) []string {418	var warnings []string419	var behind, maybe []string420	for _, d := range r.Disks {421		if len(d.BehindModules) > 0 {422			behind = append(behind, fmt.Sprintf("%s (%s)", d.Path, strings.Join(d.BehindModules, " and ")))423		}424		if d.MaybeStick {425			maybe = append(maybe, d.Path)426		}427	}428	if len(behind) > 0 {429		warnings = append(warnings, fmt.Sprintf(430			"WARNING: these disks appeared only after this report loaded a driver: %s. An install claims its disks before it loads any module a manifest declares, so an install from a stock image never sees them. Build a liken image whose image/boot-modules.conf names those modules, and install this machine from that image.",431			strings.Join(behind, ", ")))432	}433	if len(maybe) > 0 {434		warnings = append(warnings, fmt.Sprintf(435			"WARNING: more than one disk carries an installation partition, so this report cannot tell which disk is the stick you booted from. It placed no role on %s. Remove the other stick and run this report again, or name the disks yourself.",436			strings.Join(maybe, " or ")))437	}438	return warnings439}440441// reportWarnings is the same news for the console. A person reads the442// proposal on screen once and takes the file away, so the facts that443// make the file uninstallable have to reach them at the machine, not444// only in the text they scroll past.445func reportWarnings(r hardwareReport) []string {446	var lines []string447	for _, warning := range storageWarnings(r) {448		lines = append(lines, "liken: report: "+warning)449	}450	for _, rec := range r.Recommendations {451		if rec.storageClass() {452			lines = append(lines, fmt.Sprintf(453				"liken: report: WARNING: %s needs %s, and spec.modules cannot supply a storage driver; build an image whose image/boot-modules.conf names it.",454				rec.Device, strings.Join(rec.Chain, " and ")))455		}456	}457	if len(planStorageLayout(r.Disks, r.UEFI).Roles) == 0 {458		lines = append(lines, "liken: report: WARNING: no disk here can hold a liken install; the proposal declares no storage.")459	}460	return lines461}462463// writeRole renders one storage role. An empty size means the role464// takes the rest of its disk, which the spec allows for one role per465// disk. A comment, when present, follows the value on the same line,466// so a person reads the reason beside the number.467func writeRole(b *strings.Builder, name, device, size, comment string) {468	fmt.Fprintf(b, "    %s:\n", name)469	fmt.Fprintf(b, "      device: %s\n", device)470	switch {471	case size != "" && comment != "":472		fmt.Fprintf(b, "      size: %s  %s\n", size, comment)473	case size != "":474		fmt.Fprintf(b, "      size: %s\n", size)475	case comment != "":476		fmt.Fprintf(b, "      %s\n", comment)477	}478}479480// writeComment renders one sentence of prose as YAML comment lines,481// wrapped so the file reads on a console as well as in an editor.482func writeComment(b *strings.Builder, indent, text string) {483	const width = 64484	line := ""485	for _, word := range strings.Fields(text) {486		switch {487		case line == "":488			line = word489		case len(line)+1+len(word) <= width:490			line += " " + word491		default:492			fmt.Fprintf(b, "%s# %s\n", indent, line)493			line = word494		}495	}496	if line != "" {497		fmt.Fprintf(b, "%s# %s\n", indent, line)498	}499}500501// dedupChains flattens the recommendations into one ordered module502// list, each name once, in first-seen order. This is the list a person503// writes into spec.modules: the loader loads it in order, and a504// duplicate would only ask it to load the same file twice.505func dedupChains(recs []moduleRecommendation) []string {506	seen := map[string]bool{}507	var out []string508	for _, rec := range recs {509		for _, name := range rec.Chain {510			if seen[name] {511				continue512			}513			seen[name] = true514			out = append(out, name)515		}516	}517	return out518}519520// orUnknown returns the value, or "unknown" when the bus published521// nothing. A blank in the evidence would read as a mistake in the522// report rather than as an absent attribute.523func orUnknown(value string) string {524	if value == "" {525		return "unknown"526	}527	return value528}
init/resolve.go 96.9%
1package main23// Resolving a declared disk name to the disk it names, for the claim4// path.5//6// spec.storage.<role>.device may hold a kernel device path (/dev/vda)7// or a stable name under /dev/disk/by-id/ or /dev/disk/by-path/8// (machine/storage.go explains why a manifest may use either form).9// Claiming a blank disk is the only place that reads this field, so10// it is the only place that has to turn a stable name back into a11// disk.12//13// The resolution never reads /dev/disk. That tree is built by14// disklinks.go and its siblings, code that runs once storage has15// already settled, so the very first boot that claims a disk under a16// stable name has no tree there yet to read. Instead,17// resolveDeclaredDisk recomputes every attached disk's own names18// straight from sysfs, the same computation disklinks.go uses to19// build the tree, and compares the declared name against that.2021import (22	"fmt"23	"slices"24	"strings"25	"time"2627	"github.com/liken-sh/liken/liken/machine"28)2930// The shape of the declared-disk wait. The deadline is long enough31// that reaching it means the disk is not coming, not that it is32// slow.33const (34	declaredDiskPoll     = 500 * time.Millisecond35	declaredDiskDeadline = 30 * time.Second36)3738// awaitDeclaredDisk waits, boundedly, for a declared device to39// attach. Every path that claims or wipes a disk hits the same probe40// race: the controller's driver has loaded, but the SATA link, the41// USB device, or the mmc card finishes attaching a moment later, and42// mmc card detection runs on a workqueue up to a second behind. A43// device the manifest names is expected to exist, so the caller44// reports its continued absence at the deadline. Only the not-found45// case waits: an ambiguity is two disks however long the code waits,46// and a disk that attaches later cannot un-name them.47func awaitDeclaredDisk(declared string, notice string) (*machine.BlockDevice, error) {48	disk, err := resolveDeclaredDisk(declared)49	if err != nil || disk != nil {50		return disk, err51	}52	fmt.Println(notice)53	for begin := time.Now(); time.Since(begin) < declaredDiskDeadline; {54		time.Sleep(declaredDiskPoll)55		disk, err = resolveDeclaredDisk(declared)56		if err != nil || disk != nil {57			return disk, err58		}59	}60	return nil, nil61}6263// resolveDeclaredDisk turns one declared device string into the disk64// it names. A plain device path, such as /dev/vda, names whichever65// disk currently answers to that kernel name; it returns nil when no66// attached disk does. A stable name under /dev/disk/by-id/ or67// /dev/disk/by-path/ is matched against every attached disk's own68// computed names instead of read back from a link.69//70// A declared name that matches no disk is the ordinary missing-disk71// case: the disk has not attached yet, or the spec names one that72// does not exist at all. A name that matches two disks is different.73// Guessing which one the spec meant could partition the wrong disk74// and destroy whatever it holds, so resolveDeclaredDisk refuses75// instead, the same way matchRoles refuses two partitions that answer76// to one role's name.77func resolveDeclaredDisk(declared string) (*machine.BlockDevice, error) {78	suffix, ok := strings.CutPrefix(declared, "/dev/disk/")79	if !ok {80		return diskByPath(declared), nil81	}82	tree, name, _ := strings.Cut(suffix, "/")83	switch tree {84	case "by-id":85		return resolveStableName(declared, name, diskIDNames)86	case "by-path":87		return resolveStableName(declared, name, func(disk string) []string {88			if path := diskPathName(disk); path != "" {89				return []string{path}90			}91			return nil92		})93	}94	// Validate refuses every other /dev/disk tree before a spec ever95	// reaches init: by-uuid names a filesystem, and a role claims a96	// blank disk, so no other tree is legal at all97	// (machine/storage.go's Validate). The claim path checks again98	// anyway, because the code about to partition a disk must not99	// depend on an earlier gate having run.100	return nil, fmt.Errorf("declared device %s names a /dev/disk/%s tree; only by-id and by-path names resolve to a disk", declared, tree)101}102103// resolveStableName matches one stable name against every attached104// disk's own computed names, calling namesFor once per disk to get105// the names that disk answers to under the tree the declared name106// came from.107func resolveStableName(declared, name string, namesFor func(disk string) []string) (*machine.BlockDevice, error) {108	var matches []machine.BlockDevice109	for _, d := range discoverBlockDevices() {110		if slices.Contains(namesFor(d.Name), name) {111			matches = append(matches, d)112		}113	}114	switch len(matches) {115	case 0:116		return nil, nil117	case 1:118		return &matches[0], nil119	}120	var disks []string121	for _, d := range matches {122		disks = append(disks, d.Name)123	}124	return nil, fmt.Errorf("declared device %s matches %d disks (%s); refusing to guess",125		declared, len(matches), strings.Join(disks, ", "))126}
init/responder.go 90.4%
1package main23// Serving time to the fleet.4//5// This file implements the other half of the time hierarchy: leaders6// answer the followers' SNTP queries from their own disciplined7// clocks. It is a responder, not a proxy, so it never forwards a8// query upstream. Each stratum serves the clock it keeps, and it9// advertises which reference that clock descends from. This is how10// the real NTP hierarchy works, and it makes a leader self-sufficient11// once synced: followers can boot, join, and stay disciplined with no12// internet access at all.13//14// The packet is 48 bytes long, with a fixed layout, in big-endian15// order. Like the GPT header, it is small enough to build by hand,16// and it is documented well enough (RFC 5905) to build correctly.17// The reply carries four timestamps: the client's own transmit time,18// echoed back (t1, so the client can match replies to requests); the19// time the request arrived here (t2); and the time the reply left20// (t3). The client supplies its own receipt time (t4) and does the21// offset algebra that time.go describes.22//23// Only leaders run this component, and only leaders ever receive24// queries: followers sync from the leaders. A responder on a25// follower would be a listener with no caller, and a shell-less OS26// should have no port open unless something needs to connect to it.27// This component is also the machine plane's named candidate for28// promotion to a separate process (see components.go), because it29// parses unauthenticated network input as PID 1. A fixed-size read30// in a memory-safe language is about the smallest attack surface31// that a network service can have, but it is still an attack32// surface.3334import (35	"context"36	"encoding/binary"37	"fmt"38	"net"39	"time"40)4142// The mode field is three bits of byte zero, and it states what kind43// of packet this is. A client asks with mode 3, and a server answers44// with mode 4. Every other mode belongs to NTP's symmetric and45// broadcast functions, which liken does not implement. Those packets46// get no reply.47const (48	modeClient = 349	modeServer = 450)5152// The leap indicator is two bits of byte zero. Its main purpose is53// to announce leap seconds, but it can also carry the alarm value 3,54// "unsynchronized". This value tells a client not to trust anything55// else in the packet.56const (57	leapNone  = 058	leapAlarm = 359)6061// ntpEpochOffset converts between the two epochs in use. NTP counts62// seconds from 1900-01-01, and Unix counts seconds from 1970-01-01.63// This constant is the number of seconds between the two epochs.64// (NTP's 32-bit seconds field wraps in 2036, the start of era 1. The65// protocol handles this rollover by convention.)66const ntpEpochOffset = 2_208_988_8006768// ntpTimestamp encodes a moment in NTP's 64-bit format: 32 bits of69// seconds since 1900, then 32 bits of binary fraction. Each unit of70// the fraction is 1/2^32 of a second, about 233 picoseconds. This is71// far finer than anything this code measures.72func ntpTimestamp(t time.Time) uint64 {73	seconds := uint64(t.Unix() + ntpEpochOffset)74	fraction := uint64(t.Nanosecond()) << 32 / 1_000_000_00075	return seconds<<32 | fraction76}7778// advertise derives what this machine reports on the wire about its79// clock, from the same state that status.time reports. The wire's80// vocabulary differs from status's vocabulary in one spot. A server81// that has not yet received time answers with stratum 0 and the82// "INIT" kiss code (the protocol's way of saying "do not use me, ask83// again later"), where status reports stratum 16 for the same state.84// Both values mean unsynchronized. One describes this machine, and85// the other instructs a client.86func (c *clock) advertise() (stratum, leap byte, lastSync *timeSync) {87	c.mu.Lock()88	defer c.mu.Unlock()89	switch {90	case len(c.sources) == 0:91		// This is deliberately free-running: the function serves at92		// the local-clock stratum without the alarm. A fleet with no93		// upstream time source still needs one shared reference, and94		// this machine's clock provides it.95		return stratumFreeRunning, leapNone, nil96	case c.last == nil:97		return 0, leapAlarm, nil98	default:99		return byte(c.last.stratum + 1), leapNone, c.last100	}101}102103// respondTo builds the 48-byte answer to one query, or returns nil104// for anything that is not a plausible client request. Garbage input105// gets no reply, not an error. Port 123 on an open network receives106// many kinds of packets, and the cheapest response is no response.107func respondTo(request []byte, clk *clock, now time.Time) []byte {108	if len(request) < 48 {109		return nil110	}111	version := request[0] >> 3 & 0x07112	mode := request[0] & 0x07113	if mode != modeClient || version < 1 || version > 4 {114		return nil115	}116117	stratum, leap, lastSync := clk.advertise()118119	response := make([]byte, 48)120	// Byte zero packs three fields: leap(2) | version(3) | mode(3).121	// The version is the client's own version, echoed back, so an122	// SNTPv3 client gets an SNTPv3 answer.123	response[0] = leap<<6 | version<<3 | modeServer124	response[1] = stratum125	response[2] = request[2] // the client's poll interval, echoed back by convention126	// Precision is a signed log2 value. -20 claims about 1127	// microsecond, a realistic claim for a clock reading taken from128	// the kernel.129	response[3] = byte(0x100 - 20)130131	// Root delay and root dispersion ([4:8] and [8:12]) stay zero.132	// These fields accumulate path error across strata, and tracking133	// them correctly is full-NTP bookkeeping that this responder134	// does not do.135136	// The reference ID names the clock that this answer descends137	// from. It holds the source's IPv4 address when the clock138	// follows a source, "LOCL" for the local clock, or, for stratum139	// 0, the kiss code: an ASCII instruction to the client.140	switch {141	case leap == leapAlarm:142		copy(response[12:16], "INIT")143	case lastSync == nil:144		copy(response[12:16], "LOCL")145	default:146		if ip := net.ParseIP(lastSync.source); ip != nil && ip.To4() != nil {147			copy(response[12:16], ip.To4())148		}149	}150151	// The reference timestamp is the time when this clock last agreed152	// with its own source. A free-running clock is continuously its153	// own reference.154	reference := now155	if lastSync != nil {156		reference = lastSync.at157	}158	if leap == leapAlarm {159		reference = time.Time{}160	}161	if !reference.IsZero() {162		binary.BigEndian.PutUint64(response[16:24], ntpTimestamp(reference))163	}164165	// This is the four-timestamp exchange. The client's transmit time166	// comes back exactly as the originate timestamp (t1). The client167	// uses t1 to match replies to requests and to reject spoofed168	// packets. The timestamps t2 and t3 mark the start and end of169	// this function's handling. One clock reading serves both t2 and170	// t3, because the nanoseconds that this function takes are small171	// compared to the network's microseconds.172	copy(response[24:32], request[40:48])173	binary.BigEndian.PutUint64(response[32:40], ntpTimestamp(now))174	binary.BigEndian.PutUint64(response[40:48], ntpTimestamp(now))175	return response176}177178// answerTimeQueries reads queries and writes answers until the179// context ends or the socket fails. A goroutine watches the context180// and closes the socket to unblock the read, because a blocked181// ReadFrom call has no other way to detect a cancellation.182func answerTimeQueries(ctx context.Context, pc net.PacketConn, clk *clock) error {183	go func() {184		<-ctx.Done()185		pc.Close()186	}()187	buffer := make([]byte, 64)188	for {189		n, addr, err := pc.ReadFrom(buffer)190		if err != nil {191			if ctx.Err() != nil {192				return nil193			}194			return fmt.Errorf("reading a time query: %w", err)195		}196		if response := respondTo(buffer[:n], clk, time.Now()); response != nil {197			// If a reply is lost, the client must retry. UDP does not198			// guarantee delivery, and neither does this responder.199			_, _ = pc.WriteTo(response, addr)200		}201	}202}203204// serveTime runs the responder as a machine-plane component: it205// binds the NTP port and answers queries without stopping. If it206// returns an error (the port is unavailable, or the socket fails),207// the machine plane's backoff logic handles the retry, instead of208// this function looping on its own.209func serveTime(clk *clock) func(context.Context) error {210	return func(ctx context.Context) error {211		pc, err := net.ListenPacket("udp", ":123")212		if err != nil {213			return fmt.Errorf("binding the NTP port: %w", err)214		}215		fmt.Println("liken: time: serving SNTP on port 123")216		return answerTimeQueries(ctx, pc, clk)217	}218}
init/restart.go 90.2%
1package main23// The restart path applies staged restart-class changes without a4// reboot.5//6// A reboot applies staged documents by re-running the whole boot7// process. The restart path does the same work, but only for the8// parts k3s reads at process start: the boot drop-in, registries.yaml,9// and the feature actuation. Init re-renders these from the staged10// documents while k3s still runs. Only after that does init restart11// the child process (see supervisor.go). Downtime is therefore one12// graceful stop and start, and every container stays running under13// its shim.14//15// The staged stores determine what to apply. The intent file only16// signals that new work exists. So a duplicate intent is harmless:17// if a pass finds nothing new to apply, it returns false and does18// not disturb k3s. Both the boot path and the restart path use the19// same classifier (see cluster/changes.go). If a staged document's20// changes need a reboot, the restart path leaves it staged for the21// reboot path. It never applies part of that document here.22//23// Promotion needs no extra step. The proof that a cluster document24// works has always been the operator seeing the machine serve25// correctly under it. The restart path writes the attempted marker26// and publishes new facts that name the staged document. This is the27// same state a proving boot leaves. The operator's next check28// promotes the document. If k3s does not come back, the next real29// boot finds the attempted marker matching the staged document and30// rejects it with a fallback. This is the one-trial rule, and it31// applies the same way here. Credentials promote at actuation time,32// the same as at boot (see registries.go).3334import (35	"fmt"36	"os"37	"path/filepath"3839	"github.com/liken-sh/liken/liken/api"40	"github.com/liken-sh/liken/liken/cluster"41	"github.com/liken-sh/liken/liken/machine"42)4344// restartState holds everything the restart path needs. main gathers45// most of this during the boot. The struct also holds the current46// documents, which each successful apply updates. The function47// fields are seams for tests: tests use them to check the decisions48// and file effects, without a kernel to load modules into and49// without a network to get addresses from.50type restartState struct {51	root  string52	m     *machine.Machine53	conns []*connection54	tree  machine.FactsTree5556	// restarts counts the in-place k3s restarts this boot performed. The57	// restart path is the only writer of boot/restarts, so it holds the58	// count itself and starts it at zero: a fresh boot has done none.59	restarts int6061	// What k3s runs now: the choices from the boot, updated by each62	// applied restart.63	clusterDoc  *cluster.Cluster64	clusterRaw  []byte65	creds       *machine.RegistryCredentials66	credsSource machine.ManifestSource6768	// The seeded files of retracted janitor-teardown features, queued69	// by retractFeatureManifests for removal in the window where k3s70	// is down (removeOfflineRetractions).71	offlineRetractions []string7273	writeBootConfig  func(*cluster.Cluster, *machine.Machine, []*connection) (api.Role, error)74	actuateFeatures  func(*cluster.Cluster, string) []machine.FeatureStatus75	renderRegistries func(*cluster.Cluster, *machine.RegistryCredentials, machine.ManifestStore, machine.ManifestSource) machine.RegistriesStatus76}7778// newRestartState creates a restartState with the real79// implementations. Tests build the struct directly, with seams of80// their own.81func newRestartState(root string, m *machine.Machine, conns []*connection, tree machine.FactsTree,82	clusterDoc *cluster.Cluster, clusterRaw []byte, creds *machine.RegistryCredentials,83	credsSource machine.ManifestSource) *restartState {84	return &restartState{85		root: root, m: m, conns: conns, tree: tree,86		clusterDoc: clusterDoc, clusterRaw: clusterRaw, creds: creds, credsSource: credsSource,87		writeBootConfig:  writeK3sBootConfig,88		actuateFeatures:  actuateFeatures,89		renderRegistries: writeRegistriesConfig,90	}91}9293// apply is the supervisor's applyRestart callback. It loads whatever94// is staged, checks it, runs the restart-class rendering, and95// reports whether the restart is worth doing. Everything here runs96// while k3s still serves.97func (s *restartState) apply(intent machine.RestartIntent) bool {98	fmt.Printf("liken: restart requested: %s\n", intent.Reason)99100	stagedCluster, stagedRaw, clusterHash := s.stagedClusterDocument()101	stagedCreds, stagedCredsRaw := s.stagedCredentials()102	if stagedCluster == nil && stagedCreds == nil {103		fmt.Println("liken: restart: nothing staged that a restart could apply; k3s keeps running")104		return false105	}106107	// The cluster document part: re-render the boot drop-in and108	// re-run feature actuation under the staged document.109	clusterDoc, clusterRaw := s.clusterDoc, s.clusterRaw110	applyingCluster := stagedCluster != nil111	if applyingCluster {112		if _, err := s.writeBootConfig(stagedCluster, s.m, s.conns); err != nil {113			// A document that fails to render would also fail the114			// next boot. Quarantine it now, and keep serving the115			// current document.116			rejectStagedDocument("cluster", "document", machine.ClusterManifests(s.root).Reject,117				stagedRaw, fmt.Sprintf("the staged cluster document does not render a k3s configuration: %v", err))118			applyingCluster = false119		} else {120			if err := machine.ClusterManifests(s.root).WriteAttempted(clusterHash); err != nil {121				fmt.Fprintf(os.Stderr, "liken: restart: marking the staged document attempted: %v\n", err)122			}123			clusterDoc, clusterRaw = stagedCluster, stagedRaw124		}125	}126127	// The credentials part runs whether or not the cluster document128	// changed. writeRegistriesConfig promotes staged credentials129	// after it writes the file.130	creds, credsSource := s.creds, s.credsSource131	if stagedCreds != nil {132		creds, credsSource = stagedCreds, machine.ManifestSourceStaged133	}134	if !applyingCluster && stagedCreds == nil {135		return false136	}137138	featureStatuses := s.actuateFeatures(clusterDoc, s.m.Metadata.Name)139	if applyingCluster {140		s.retractFeatureManifests(s.clusterDoc, clusterDoc)141	}142	registries := s.renderRegistries(clusterDoc, creds, machine.RegistryCredentialsStore(s.root), credsSource)143144	// The facts update before the restart. They name the staged145	// documents, so the operator reads what this restart applied. The146	// write order is a commit protocol: features/ and registries/ land147	// first, then the restart counter, and boot/clusterManifest lands148	// last, because the operator's promotion keys on that record149	// (machine-operator/cluster.go). The boot cluster manifest150	// publication carries the exact bytes the operator compares against.151	s.restarts++152	s.tree.WriteFeatures(featureStatuses)153	s.tree.WriteRegistries(registries)154	s.tree.WriteRuntime(runtimeFacts(clusterDoc, machineMemoryBytes()))155	if stagedCreds != nil {156		s.tree.WriteBootCredentials(machine.ManifestSourceStaged, machine.ManifestHash(stagedCredsRaw))157	}158	s.tree.WriteBootRestarts(s.restarts)159	if applyingCluster {160		s.tree.WriteBootClusterManifest(machine.ManifestSourceStaged, clusterHash)161		publishBootClusterManifest(clusterRaw)162	}163164	// The applied documents are now current. A duplicate intent finds165	// nothing staged, because credentials are promoted, or finds an166	// attempted marker that matches staged, because the operator has167	// not yet promoted the cluster document. Either way, the168	// duplicate intent applies nothing.169	s.clusterDoc, s.clusterRaw = clusterDoc, clusterRaw170	s.creds, s.credsSource = creds, credsSource171	return true172}173174// stagedClusterDocument loads and checks the staged cluster document.175// It returns nil when there is nothing for a restart to apply:176//177//   - There is no staged file.178//   - A document was already attempted, by this restart or by a179//     previous boot. The operator's promotion, or the next boot's180//     rejection, will settle it.181//   - A document fails to parse. This function quarantines it.182//   - A document's changes are reboot-class. This function leaves it183//     staged for the reboot path, because the operator asked for a184//     reboot. This check also stops a racing restart intent from185//     applying part of the document.186func (s *restartState) stagedClusterDocument() (*cluster.Cluster, []byte, string) {187	store := machine.ClusterManifests(s.root)188	raw, err := store.LoadStaged()189	if err != nil || raw == nil {190		return nil, nil, ""191	}192	hash := machine.ManifestHash(raw)193	if attempted, _ := store.LoadAttempted(); attempted == hash {194		return nil, nil, ""195	}196	staged, perr := cluster.ParseCluster(raw)197	if perr != nil {198		rejectStagedDocument("cluster", "document", store.Reject,199			raw, fmt.Sprintf("the staged cluster document does not parse: %v", perr))200		return nil, nil, ""201	}202	if s.clusterDoc == nil {203		return nil, nil, ""204	}205	if !cluster.RestartApplies(s.clusterDoc.Spec, staged.Spec) {206		fmt.Println("liken: restart: the staged cluster document needs a reboot; leaving it for one")207		return nil, nil, ""208	}209	return staged, raw, hash210}211212// stagedCredentials loads and checks the staged credentials document.213// It returns nil when nothing is staged. Credentials promote at214// actuation, so a staged file always means unapplied work. If a215// document fails to parse, this function quarantines it, the same as216// at boot.217func (s *restartState) stagedCredentials() (*machine.RegistryCredentials, []byte) {218	store := machine.RegistryCredentialsStore(s.root)219	raw, err := store.LoadStaged()220	if err != nil || raw == nil {221		return nil, nil222	}223	creds, perr := machine.ParseRegistryCredentials(raw)224	if perr != nil {225		rejectStagedDocument("registries", "credentials", store.Reject,226			raw, fmt.Sprintf("the staged credentials document does not parse: %v", perr))227		return nil, nil228	}229	return creds, raw230}231232// retractFeatureManifests removes the seeded manifests of features233// that the new document no longer declares, so that nothing seeds234// them again. Removing a manifest is not a deletion. k3s's deploy235// controller walks the files that are there and applies them, and236// nothing reconciles an addon against a source file that has gone,237// so the addon and every object it created stay in the cluster. The238// cluster operator's janitor is what deletes a retracted feature's239// workloads, on this path and on the boot path alike240// (cluster-operator/janitor.go).241//242// A janitor-teardown feature's files are queued rather than removed,243// and removeOfflineRetractions removes them once k3s has stopped.244// Given the paragraph above, that ordering guards against nothing245// k3s does: no pass acts on a removed file whether k3s is up or246// down. The queue stays because it costs one list, and the failure247// it rules out reaches the whole fleet. If a deletion did cascade248// from the sync objects while the flux controllers still ran, the249// engine's own deletion finalizer would prune everything the250// repository ever applied, the fleet's own documents included. The251// janitor tears flux down in an order that stops the controllers252// first, and the queue keeps the file removal outside that order.253func (s *restartState) retractFeatureManifests(old, new *cluster.Cluster) {254	declared := map[string]bool{}255	for _, slug := range new.EnabledFeatures() {256		declared[slug] = true257	}258	for _, slug := range old.EnabledFeatures() {259		if declared[slug] {260			continue261		}262		manifests, err := featureManifestPaths(slug)263		if err != nil {264			continue265		}266		var files []string267		for _, manifest := range manifests {268			files = append(files, filepath.Base(manifest))269		}270		files = append(files, renderedFeatureManifests[slug]...)271272		def := cluster.FeatureBySlug(slug)273		if def != nil && def.Teardown == cluster.TeardownJanitor {274			for _, file := range files {275				s.offlineRetractions = append(s.offlineRetractions, filepath.Join(k3sManifestsDir, file))276			}277			continue278		}279		for _, file := range files {280			if err := os.Remove(filepath.Join(k3sManifestsDir, file)); err == nil {281				fmt.Printf("liken: restart: retracted %s; the janitor deletes its workload\n", file)282			}283		}284	}285}286287// removeOfflineRetractions removes the files that288// retractFeatureManifests queued, in the window where k3s is down.289// The supervisor calls it right after each restart's stop. The files290// go while k3s is stopped, so no addon pass runs against the291// removal; the cluster operator's janitor owns that teardown.292func (s *restartState) removeOfflineRetractions() {293	for _, path := range s.offlineRetractions {294		if err := os.Remove(path); err == nil {295			fmt.Printf("liken: restart: retracted %s while k3s is down; the janitor tears its objects down\n", filepath.Base(path))296		}297	}298	s.offlineRetractions = nil299}
init/rlimits.go 100.0%
1package main23// Applying the machine's resource limits.4//5// This is the part of systemd's work that no file can do. A sysctl is6// a file the kernel re-reads, so anything may write one at any time. A7// resource limit is fixed when a process forks, and the only way to8// give k3s a limit is to hold that limit before starting it. So init9// applies these to itself, and every process on the machine inherits10// the result: k3s, containerd, the shims, and the containers below11// them.12//13// The ordering constraint is the whole reason this runs where it does14// in the boot. Init must have read the boot's Machine manifest, so15// that spec.rlimits is known, and it must not yet have started k3s.16// The helpers init ran earlier, mke2fs and modprobe among them, keep17// the kernel's defaults. They are short-lived and open few files, and18// raising a limit for them would mean applying the table before the19// manifest that tunes it.2021import (22	"fmt"23	"maps"24	"os"25	"slices"2627	"github.com/liken-sh/liken/liken/machine"28)2930// applyRlimits applies a set of resource limits to init itself. If one31// fails, applyRlimits reports the failure and skips it, rather than32// treating it as fatal. The reasoning matches applySysctls: a limit33// this machine cannot hold should not cost it a boot, and OSRlimits34// ships with the release, so a bad entry would otherwise take a whole35// fleet down at once.36//37// Init calls this twice, once for the limits every liken machine holds38// and once for the Machine spec's own. Printing on both passes shows a39// reader of the boot log what the OS set, and then which of those40// limits this deployment chose to override.41func applyRlimits(rlimits map[string]string) {42	for _, name := range slices.Sorted(maps.Keys(rlimits)) {43		value := rlimits[name]44		if err := machine.ApplyRlimit(name, value); err != nil {45			fmt.Fprintf(os.Stderr, "liken: %v\n", err)46			continue47		}48		fmt.Printf("liken: rlimit %s = %s\n", name, value)49	}50}5152// readRlimits reads back every limit that either pass tried to set,53// and reports what the kernel actually holds. Reading back rather than54// echoing what was written is the same discipline status.sysctls55// follows: a limit the kernel clamped or refused shows its real value56// here, so this map is the list of limits that hold, not the list57// somebody asked for.58//59// A name that cannot be read at all is left out, which can only happen60// for a name the spec misspelled. That name already produced an error61// on the console during the pass that tried to apply it.62func readRlimits(sets ...map[string]string) map[string]string {63	names := map[string]bool{}64	for _, set := range sets {65		for name := range set {66			names[name] = true67		}68	}69	if len(names) == 0 {70		return nil71	}72	held := map[string]string{}73	for _, name := range slices.Sorted(maps.Keys(names)) {74		value, err := machine.ReadRlimit(name)75		if err != nil {76			continue77		}78		held[name] = value79	}80	if len(held) == 0 {81		return nil82	}83	return held84}
init/rootimage.go 64.0%
1package main23// Finding and mounting the system image: the read-only squashfs file4// that becomes the root filesystem.5//6// The boot loader stages almost nothing in RAM: the kernel, the small7// boot archive that holds this program, and the deployment layer. The8// operating system itself is liken.sqfs. It arrives in one of two9// ways:10//11//   - From a boot slot. On an installed machine, the kernel command12//     line carries liken.slot=A (or B). This is the installer's13//     record of which slot the boot entry belongs to. The slot is a14//     FAT32 partition, recognized by the GPT name written when it15//     was claimed (liken:systemA). The image is a file on that16//     partition.17//18//   - From RAM. A boot with no disk, such as the lab's19//     from-blank-disks drills or QEMU -kernel boots, wraps the image20//     in a cpio archive. The kernel unpacks this archive into rootfs21//     at /liken.sqfs, and init mounts the image from there. The loop22//     device holds the file's memory in place for as long as it23//     stays mounted. This is the RAM cost of having no disk. Only24//     boots that run without a disk pay this cost.25//26// Either way, init loop-mounts the image read-only, with exactly the27// bytes the release published. The running root is the28// digest-verified artifact. Nothing can change it, because of how29// the system builds and mounts it. A bounded tmpfs sits above it for30// the runtime's writes (see switchroot.go for that part).31//32// This file's work happens first, at the start of boot, when almost33// nothing else is mounted. The caller must mount /proc before it34// calls into this file, because the liken.* parameters live only in35// /proc/cmdline. The kernel treats dotted parameter names as module36// parameters and never passes them to init directly.3738import (39	"errors"40	"fmt"41	"os"42	"path/filepath"43	"time"4445	"golang.org/x/sys/unix"4647	"github.com/liken-sh/liken/liken/machine"48)4950// ramImage is where a cpio-wrapped system image lands in rootfs. If51// this file exists, init uses the from-RAM path. This is a package52// variable, not a constant, so tests can point the search at a file53// of their own.54var ramImage = "/liken.sqfs"5556// bootModulesDir is the boot archive's module tree. The directory57// name does not include the kernel release, on purpose. When init58// carries the boot-time files onto the real root, nothing under this59// name can hide the system image's full index at60// /lib/modules/<release>. This is a package variable, for the same61// testing reason as ramImage.62var bootModulesDir = "/lib/modules/boot"6364// slotImageName is the image's file name on a boot slot. It is one65// of the artifacts that the installer copies, and the release66// document names it.67const slotImageName = "liken.sqfs"6869// loadBootModules loads the boot archive's few modules, such as70// overlayfs and vfat's encoding table. It resolves each module71// through the archive's own depmod index. A failure to load a module72// is reported, not fatal. Any mount that needs a missing module73// reports its own failure.74func loadBootModules(names ...string) {75	deps, err := readModulesDep(filepath.Join(bootModulesDir, "modules.dep"))76	if err != nil {77		fmt.Fprintf(os.Stderr, "liken: boot modules: %v\n", err)78		return79	}80	loaded := map[string]bool{}81	for _, name := range names {82		if _, err := loadModule(bootModulesDir, name, "", deps, loaded); err != nil {83			fmt.Fprintf(os.Stderr, "liken: boot modules: %s: %v\n", name, err)84		}85	}86}8788// findSystemImage finds liken.sqfs and returns its path. It mounts89// the boot slot when the image is not already in rootfs. slotMount90// is where the slot mounts when init uses one (use "" for the RAM91// path).92func findSystemImage(slotParam, slotMount string) (imagePath string, err error) {93	if _, err := os.Stat(ramImage); err == nil {94		fmt.Println("liken: system image found in RAM (no disk needed for this boot)")95		return ramImage, nil96	}9798	if slotParam == "" {99		return "", fmt.Errorf("no %s in rootfs and no liken.slot= boot parameter", ramImage)100	}101	device, err := awaitSlotDevice(slotParam)102	if err != nil {103		return "", err104	}105106	if err := os.MkdirAll(slotMount, 0o755); err != nil {107		return "", err108	}109	// This mount is read-write, but not because it writes anything110	// here. The early boot only reads the image. Storage111	// reconciliation later mounts the same partition at its role112	// path, and both mounts share one superblock. A superblock cannot113	// be read-only in one mount and read-write in the other. The slot114	// must stay writable, because the fetcher writes downloaded115	// releases into slots.116	//117	// Mounting for writing is also what sets the volume's mark, so this118	// is the last moment at which the device can be asked what the119	// previous stop left behind (fatstate.go).120	recordBootedSlotStop(device)121	if err := mountFilesystem(device, slotMount, "vfat", 0, ""); err != nil {122		return "", fmt.Errorf("mounting slot %s (%s): %w", slotParam, device, err)123	}124	fmt.Printf("liken: system image on slot %s (%s)\n", slotParam, device)125	return filepath.Join(slotMount, slotImageName), nil126}127128// The shape of the slot wait. The poll is short because the wait129// runs in early boot against a card that attaches within tens of130// milliseconds, and the deadline is long enough that reaching it131// means the disk is not coming, not that it is slow.132const (133	slotPoll     = 50 * time.Millisecond134	slotDeadline = 10 * time.Second135)136137// awaitSlotDevice waits, boundedly, for the slot's partition to138// appear with a device node the mount can open. A controller139// registers before the disk behind it does, and mmc card detection140// runs on a workqueue, so on an eMMC machine the card attaches tens141// of milliseconds after init first looks. Only a boot with142// liken.slot= reaches this function, and that parameter means the143// boot loader has already read this slot, so the partition is on144// this machine and waiting for it is always correct. A boot from145// RAM never gets here at all, and a machine whose disk is already146// visible returns from the first search and never sleeps.147//148// Only the two still-arriving errors wait. Waiting cannot fix an149// ambiguity:150// two partitions already carrying the slot's name stay two however151// long the code waits, and a disk that attaches later cannot un-name152// them. A search that runs out the deadline returns the last153// search's error, so the caller reports the missing slot and the154// boot continues on rootfs, the same as before the wait existed.155// Polling, rather than uevents, keeps this simple: the window is156// short and a walk of /sys/block costs nothing.157func awaitSlotDevice(slotParam string) (string, error) {158	device, err := mountableSlotDevice(slotParam)159	if !stillArriving(err) {160		return device, err161	}162	fmt.Printf("liken: waiting for the disk that carries slot %s\n", slotParam)163	for begin := time.Now(); time.Since(begin) < slotDeadline; {164		time.Sleep(slotPoll)165		device, err = mountableSlotDevice(slotParam)166		if !stillArriving(err) {167			return device, err168		}169	}170	return "", err171}172173// mountableSlotDevice finds the slot's partition and then requires174// its device node, because the kernel publishes a partition's sysfs175// entries before devtmpfs creates its node, and the mount that176// follows this search opens the node. A poll that trusted sysfs177// alone could land in that window and hand the caller a path that178// does not exist yet.179func mountableSlotDevice(slotParam string) (string, error) {180	device, err := slotDevice(discoverPartitions(), slotParam)181	if err != nil {182		return "", err183	}184	if _, err := os.Stat(device); err != nil {185		return "", fmt.Errorf("%w %s", errNoSlotNode, device)186	}187	return device, nil188}189190// stillArriving names the errors a wait can fix. Both mean the191// device is on its way; every other error means waiting changes192// nothing.193func stillArriving(err error) bool {194	return errors.Is(err, errNoSlotPartition) || errors.Is(err, errNoSlotNode)195}196197// errNoSlotPartition is the one error the slot search can wait out:198// a slot that no partition carries yet is what a disk that has not199// attached looks like. The sentinel holds the sentence the error200// reads as, so the message a boot prints is the one it always201// printed.202var errNoSlotPartition = errors.New("no partition carries")203204// errNoSlotNode is the second waitable error: the partition is in205// sysfs and devtmpfs has not created its node yet.206var errNoSlotNode = errors.New("no device node yet for")207208// slotDevice picks the partition that holds the named slot, from the209// machine's discovered partitions. It recognizes the slot the same210// way storage roles are recognized: by the GPT partition name that211// the claim wrote, wherever the disk enumerated during this boot.212// Device paths in specs are only hints from claim time. Names are213// the identity. If two partitions claim one name, slotDevice refuses214// to guess and returns an error.215func slotDevice(parts []partition, slotParam string) (string, error) {216	role := machine.SystemARole217	if slotParam == "B" {218		role = machine.SystemBRole219	}220	want := machine.PartitionPrefix + string(role)221222	var device string223	for _, p := range parts {224		if p.partName == want {225			if device != "" {226				return "", fmt.Errorf("two partitions carry %s; refusing to guess", want)227			}228			device = devRoot + "/" + p.name229		}230	}231	if device == "" {232		return "", fmt.Errorf("%w %s", errNoSlotPartition, want)233	}234	return device, nil235}236237// loopMount attaches path to a free loop device, read-only, and238// mounts it at target as squashfs. The autoclear flag detaches the239// loop device when the mount goes away, so nothing needs to be torn240// down by hand.241func loopMount(path, target string) error {242	ctl, err := os.OpenFile("/dev/loop-control", os.O_RDWR, 0)243	if err != nil {244		return fmt.Errorf("opening loop-control: %w", err)245	}246	defer ctl.Close()247	n, err := unix.IoctlRetInt(int(ctl.Fd()), unix.LOOP_CTL_GET_FREE)248	if err != nil {249		return fmt.Errorf("allocating a loop device: %w", err)250	}251252	backing, err := os.Open(path)253	if err != nil {254		return err255	}256	defer backing.Close()257	loopPath := fmt.Sprintf("/dev/loop%d", n)258	loop, err := os.OpenFile(loopPath, os.O_RDONLY, 0)259	if err != nil {260		return fmt.Errorf("opening %s: %w", loopPath, err)261	}262	defer loop.Close()263	if err := unix.IoctlSetInt(int(loop.Fd()), unix.LOOP_SET_FD, int(backing.Fd())); err != nil {264		return fmt.Errorf("attaching %s to %s: %w", path, loopPath, err)265	}266	status := unix.LoopInfo64{267		Flags: unix.LO_FLAGS_READ_ONLY | unix.LO_FLAGS_AUTOCLEAR,268	}269	copy(status.File_name[:], loopPath)270	if err := unix.IoctlLoopSetStatus64(int(loop.Fd()), &status); err != nil {271		return fmt.Errorf("configuring %s: %w", loopPath, err)272	}273274	if err := os.MkdirAll(target, 0o755); err != nil {275		return err276	}277	if err := unix.Mount(loopPath, target, "squashfs", unix.MS_RDONLY, ""); err != nil {278		return fmt.Errorf("mounting %s at %s: %w", path, target, err)279	}280	return nil281}
init/schemarevision.go 85.6%
1package main23// A boot seeds k3s's manifests from its own image, and the image is4// sometimes older than the cluster it boots: a fallback to the5// previous slot is the designed answer to a release that fails. For6// most manifests an older copy is merely stale. A CRD is different.7// When an older schema replaces a newer one, the API server keeps8// serving the stored objects, but it prunes the fields the older9// schema does not declare, and the next write persists the pruned10// object. No error is reported, and the object's generation does not11// change, so nothing that watches the object learns that data was12// lost.13//14// This file is the guard against that path. Each CRD manifest carries15// a revision annotation, the seed compares the image's copy against16// the copy already on the disk, and the disk's copy stays when it17// carries the higher revision. The guard runs in the image doing the18// seeding, so it protects a cluster only once every slot a machine19// can fall back to carries it.2021import (22	"encoding/json"23	"errors"24	"fmt"25	"io"26	"math"27	"os"28	"path/filepath"29	"regexp"30	"strconv"31	"strings"3233	"sigs.k8s.io/yaml"34)3536// schemaRevisionAnnotation names the integer that orders a CRD's37// schemas across releases. Every change to the schema raises it, and38// the pin test in schemarevision_test.go fails any change that does39// not, so the annotation cannot silently fall behind the schema it40// orders.41const schemaRevisionAnnotation = "liken.sh/schema-revision"4243// manifestSizeLimit caps what the guard will read into memory. The44// reader runs in PID 1 on a machine that can boot with 1GB, and the45// largest seeded CRD is 178KB, so one megabyte is generous headroom46// rather than a constraint anyone meets.47const manifestSizeLimit = 1 << 204849// documentSeparator matches a YAML document separator on its own50// line, with the trailing spaces and the carriage return that a file51// written on another system carries.52var documentSeparator = regexp.MustCompile(`(?m)^---[ \t]*\r?$`)5354// yamlDocuments splits a manifest into its documents and drops the55// empty ones, which a file that opens or closes with a separator56// leaves behind.57func yamlDocuments(raw []byte) [][]byte {58	var documents [][]byte59	for _, doc := range documentSeparator.Split(string(raw), -1) {60		if strings.TrimSpace(doc) == "" {61			continue62		}63		documents = append(documents, []byte(doc))64	}65	return documents66}6768// schemaRevision is what one manifest declares about its schema: this69// is a CustomResourceDefinition, this is the revision it carries,70// and, when the annotation names no revision at all, this is what it71// said instead.72type schemaRevision struct {73	isCRD    bool74	revision int75	bad      string76}7778// crdSchemaRevision reads what a manifest declares about its schema.79// A manifest with no annotation is revision 0, which is what every80// release from before the annotation existed effectively declares.81// Two rules cover a value that does not fit an int. A run of digits82// too long to hold means the highest revision there is: the author's83// intent is unambiguous, and reading it as 0 would be the silent84// downgrade this file exists to stop. Anything else names no revision85// at all; it compares as 0, and the caller reports it, so a corrupted86// annotation is recoverable by the next seed instead of pinning the87// file forever on a machine with no shell.88func crdSchemaRevision(raw []byte) (schemaRevision, error) {89	for _, doc := range yamlDocuments(raw) {90		body, err := yaml.YAMLToJSON(doc)91		if err != nil {92			return schemaRevision{}, fmt.Errorf("parsing a manifest document: %w", err)93		}94		var meta struct {95			Kind     string `json:"kind"`96			Metadata struct {97				Annotations map[string]string `json:"annotations"`98			} `json:"metadata"`99		}100		if err := json.Unmarshal(body, &meta); err != nil {101			return schemaRevision{}, fmt.Errorf("reading a manifest document: %w", err)102		}103		if meta.Kind != "CustomResourceDefinition" {104			continue105		}106		declared := strings.TrimSpace(meta.Metadata.Annotations[schemaRevisionAnnotation])107		if declared == "" {108			return schemaRevision{isCRD: true}, nil109		}110		revision, err := strconv.Atoi(declared)111		if errors.Is(err, strconv.ErrRange) && !strings.HasPrefix(declared, "-") {112			return schemaRevision{isCRD: true, revision: math.MaxInt}, nil113		}114		if err != nil || revision < 0 {115			return schemaRevision{isCRD: true, bad: declared}, nil116		}117		return schemaRevision{isCRD: true, revision: revision}, nil118	}119	return schemaRevision{}, nil120}121122// readManifest reads a file that is a manifest and nothing else.123// init is PID 1: a read of a fifo never returns, and a read of a124// device or of a file the size of the disk takes the memory the125// machine boots with.126func readManifest(path string) ([]byte, error) {127	info, err := os.Lstat(path)128	if err != nil {129		return nil, err130	}131	if !info.Mode().IsRegular() {132		return nil, fmt.Errorf("%s is %s, not a regular file", path, info.Mode().Type())133	}134	if info.Size() > manifestSizeLimit {135		return nil, fmt.Errorf("%s is %d bytes, past the %d-byte limit on a manifest",136			path, info.Size(), manifestSizeLimit)137	}138	return os.ReadFile(path)139}140141// newerCRDsOnDisk names the manifests whose disk copy outranks the142// image's: the files that declare a CRD on both sides, with the143// higher schema revision on the disk. The disk got a higher revision144// from a newer release that ran here before this one, so replacing it145// would downgrade the schema the cluster already serves.146//147// Every failure to read or parse falls through to the ordinary148// refresh, and the image's copy wins. That direction is deliberate: a149// file this function cannot judge must not be able to stop a boot,150// and an unreadable disk copy protects nothing anyway. The console151// line is the record that the comparison did not happen.152func newerCRDsOnDisk(src, dst string) map[string]bool {153	keep := map[string]bool{}154	entries, err := os.ReadDir(src)155	if err != nil {156		fmt.Printf("liken: k3s: reading the image's manifests: %v; seeding them all\n", err)157		return keep158	}159	for _, entry := range entries {160		if entry.IsDir() {161			continue162		}163		name := entry.Name()164		if _, err := os.Lstat(filepath.Join(dst, name)); err != nil {165			continue // the disk holds no copy of this manifest to keep166		}167		onDisk, err := readManifest(filepath.Join(dst, name))168		if err != nil {169			fmt.Printf("liken: k3s: %v; %s cannot be compared, so the image's copy replaces the disk's\n", err, name)170			continue171		}172		incoming, err := readManifest(filepath.Join(src, name))173		if err != nil {174			fmt.Printf("liken: k3s: %v; %s cannot be compared, so the image's copy replaces the disk's\n", err, name)175			continue176		}177		disk, err := crdSchemaRevision(onDisk)178		if err != nil {179			fmt.Printf("liken: k3s: %s on the disk does not parse: %v; the image's copy replaces it\n", name, err)180			continue181		}182		image, err := crdSchemaRevision(incoming)183		if err != nil {184			fmt.Printf("liken: k3s: %s in the image does not parse: %v; the image's copy replaces it\n", name, err)185			continue186		}187		reportBadRevision(name, "on the disk", disk)188		reportBadRevision(name, "in the image", image)189		if !disk.isCRD || !image.isCRD || disk.revision <= image.revision {190			continue191		}192		fmt.Printf("liken: k3s: %s on the disk declares schema revision %d and this image carries %d; keeping the disk's copy\n",193			name, disk.revision, image.revision)194		keep[name] = true195	}196	return keep197}198199// reportBadRevision names a manifest whose annotation names no200// revision. The file then compares as revision 0, which is what a201// manifest with no annotation at all compares as.202func reportBadRevision(name, where string, declared schemaRevision) {203	if declared.bad == "" {204		return205	}206	fmt.Printf("liken: k3s: %s %s declares %s: %q, which is no revision; it compares as 0\n",207		name, where, schemaRevisionAnnotation, declared.bad)208}209210// refreshSeedDir replaces a seed directory's content entry by entry,211// skipping the kept names on both the remove side and the copy side.212// The alternative, a wipe of the directory and a copy with the kept213// files written back, holds the kept content only in memory for a214// moment, and a power cut in that moment leaves the disk with the215// older schema and no record that a newer one existed. Entry by216// entry, the kept manifest is never removed and never rewritten, so217// no instant of the boot leaves the disk without the schema the218// cluster serves.219//220// The refresh is also best effort per entry, and only the failures221// are collected. A refresh that stopped at its first failure would222// let one stuck file decide the fate of every entry after it in223// directory order: files already removed would never be copied back,224// and a directory of six manifests could end the boot holding one.225// Attempted one by one, a file the disk refuses to give up costs226// exactly itself, and it stays as it stood rather than being227// half-replaced, because the copy loop skips what the removal could228// not take away.229func refreshSeedDir(dst, src string, keep map[string]bool) error {230	var failures []string231	existing, err := os.ReadDir(dst)232	if err != nil && !os.IsNotExist(err) {233		return err234	}235	stuck := map[string]bool{}236	for _, entry := range existing {237		if keep[entry.Name()] {238			continue239		}240		if err := os.RemoveAll(filepath.Join(dst, entry.Name())); err != nil {241			stuck[entry.Name()] = true242			failures = append(failures, err.Error())243		}244	}245	if err := os.MkdirAll(dst, 0o755); err != nil {246		return collectSeedFailures(append(failures, err.Error()))247	}248	entries, err := os.ReadDir(src)249	if err != nil {250		return collectSeedFailures(append(failures, err.Error()))251	}252	for _, entry := range entries {253		if keep[entry.Name()] || stuck[entry.Name()] {254			continue255		}256		from, to := filepath.Join(src, entry.Name()), filepath.Join(dst, entry.Name())257		if entry.IsDir() {258			if err := os.CopyFS(to, os.DirFS(from)); err != nil {259				failures = append(failures, err.Error())260			}261			continue262		}263		if err := copySeedFile(to, from); err != nil {264			failures = append(failures, err.Error())265		}266	}267	return collectSeedFailures(failures)268}269270// collectSeedFailures turns the entries that failed into one error.271// It reads as one line, because the console reports one fact per line272// and every line begins with liken:.273func collectSeedFailures(failures []string) error {274	if len(failures) == 0 {275		return nil276	}277	return errors.New(strings.Join(failures, "; "))278}279280// copySeedFile copies one seed file, streaming it rather than holding281// it in memory: the operator images beside the manifests are tens of282// megabytes on a machine that boots with 1GB.283func copySeedFile(dst, src string) error {284	in, err := os.Open(src)285	if err != nil {286		return err287	}288	defer in.Close()289	info, err := in.Stat()290	if err != nil {291		return err292	}293	if !info.Mode().IsRegular() {294		return fmt.Errorf("%s is %s, not a regular file", src, info.Mode().Type())295	}296	out, err := os.OpenFile(dst, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0o666|info.Mode()&0o777)297	if err != nil {298		return err299	}300	if _, err := io.Copy(out, in); err != nil {301		out.Close()302		return err303	}304	return out.Close()305}
init/serio.go 92.9%
1package main23// The serio watch: the machine-plane component that keeps every4// spec.serio attachment in place (machine/serio.go says what an5// attachment is, and serioattach.go how one is held).6//7// The watch has the shape of the disk links watch. It walks sysfs once8// at start, and again after every settled burst of uevents. Each walk9// lists the ttys under /sys/class/tty, reads the USB identity above10// each one, and compares it with the declared entries. A matched tty11// with no holder gets one. An unplug needs no handling of its own: the12// kernel hangs up the tty, serport ends the read, the holder ends, and13// the next walk no longer lists the tty. The plug that follows sends14// uevents, and the walk after them starts a new holder. A refusal is15// kept for the USB port it happened on and retried after a bounded16// backoff (serioretry.go), so an adapter that resets comes back by17// itself.18//19// The attachment belongs on the machine plane for the reason module20// loading does. It is what makes the hardware appear, and no pod can21// do it: the machine operator's DRA driver delivers only the nodes22// that exist when a container starts, so a pod that attached the line23// would never receive the nodes it created, and TIOCSETD to serport24// needs CAP_SYS_ADMIN.25//26// The holders live in a registry that outlives the component, the same27// as the reaper's. A restart of the component starts a new uevent28// listener and a new walk, and the holders stay in their reads29// through the restart. The holders do not stop at shutdown either.30// They write to no disk, so the quiesce has nothing to wait for, and31// the reboot system call ends them. The port stays in place for as32// long as any pod runs.3334import (35	"context"36	"fmt"37	"slices"38	"sync"39	"time"4041	"github.com/liken-sh/liken/liken/hardware"42	"github.com/liken-sh/liken/liken/machine"43)4445// serioQuiet is how long a walk waits for a burst of uevents to stop.46// An adapter's plug announces the USB device, its interfaces, and the47// tty, and an attach announces the port, the CEC adapter, and the48// input device. A pod holding the adapter waits for the result, so the49// wait is short, like the disk links watch's.50const serioQuiet = 250 * time.Millisecond5152// serioSettleTimeout bounds the wait for a new holder's four attach53// calls. On a healthy adapter they take microseconds, but the kernel54// lets them wait longer: cdc_acm's control transfers each time out55// after 5 seconds, and TIOCSETD waits up to 5 seconds for the line56// discipline's lock. The bound sits above the sum of those waits, so57// a slow adapter settles inside it. The walk holds no lock while it58// waits, so the hardware watch and the module loader never wait59// behind it.60const serioSettleTimeout = 15 * time.Second6162// serioRegistry is the state that outlives the component: the63// declared entries, the holders, and what the last walk reported.64// Only the walk and the declaration touch the maps, both under mu. A65// holder never touches the registry. It reports through its own66// channels (serioattach.go).67type serioRegistry struct {68	mu       sync.Mutex69	declared []machine.SerioAttachment70	holders  map[string]*serioHolder71	refusals map[string]serioRefusal72	unbound  map[string]serioUnbound73	nudge    chan struct{}74	open     openLine75	now      func() time.Time7677	// printed is the report the console last showed, and published is78	// the report the facts tree last took. They differ only after a79	// failed write, which the next walk retries. Only the component's80	// goroutine touches them, and the plane never runs two of it.81	printed, published []machine.SerioStatus82}8384func newSerioRegistry(open openLine) *serioRegistry {85	return &serioRegistry{86		holders:  map[string]*serioHolder{},87		refusals: map[string]serioRefusal{},88		unbound:  map[string]serioUnbound{},89		nudge:    make(chan struct{}, 1),90		open:     open,91		now:      time.Now,92	}93}9495// serioAttachments is the boot's one registry, package-level for the96// same reason the machine plane is.97var serioAttachments = newSerioRegistry(openTTY)9899// declare replaces the declared list and wakes the walk. The boot100// declares the manifest's list, and the module loader declares the101// list a live load applied.102func (r *serioRegistry) declare(entries []machine.SerioAttachment) {103	r.mu.Lock()104	r.declared = slices.Clone(entries)105	r.refusals = map[string]serioRefusal{}106	r.mu.Unlock()107	select {108	case r.nudge <- struct{}{}:109	default:110	}111}112113// declaredEntries returns a copy of the declared list, for the114// hardware watch's unclaimed report.115func (r *serioRegistry) declaredEntries() []machine.SerioAttachment {116	r.mu.Lock()117	defer r.mu.Unlock()118	return slices.Clone(r.declared)119}120121// watchSerio is the component. It publishes the first walk at once,122// so an adapter that is plugged in at boot attaches without waiting123// for a uevent.124func watchSerio(r *serioRegistry, tree machine.FactsTree) func(context.Context) error {125	return func(ctx context.Context) error {126		uevents, err := listenForUevents(ctx)127		if err != nil {128			return err129		}130		for {131			r.publish(tree)132			var retry <-chan time.Time133			var timer *time.Timer134			if wait, ok := r.nextRetry(); ok {135				timer = time.NewTimer(wait)136				retry = timer.C137			}138			listening := waitForSerioWork(ctx, uevents, r.nudge, retry)139			if timer != nil {140				timer.Stop()141			}142			if ctx.Err() != nil {143				return nil144			}145			if !listening {146				return errUeventsStopped147			}148		}149	}150}151152// waitForSerioWork returns when the next walk is due: after a burst of153// uevents settles, after a holder's nudge and the uevents that follow154// it settle, or when a refusal's backoff runs out. A holder ends when155// the kernel hangs up its tty, which comes before the kernel removes156// the tty, so a walk at the nudge itself would read a tty that is157// about to leave and report a refusal for one walk. It answers false158// when the uevent listener stopped.159func waitForSerioWork(ctx context.Context, uevents, nudge <-chan struct{}, retry <-chan time.Time) bool {160	select {161	case <-ctx.Done():162	case <-nudge:163		hardware.Settle(ctx, uevents, serioQuiet, 5*time.Second)164	case _, ok := <-uevents:165		if !ok {166			return false167		}168		hardware.Settle(ctx, uevents, serioQuiet, 5*time.Second)169	case <-retry:170	}171	return true172}173174// publish runs one walk, prints each status that changed, and175// rewrites serio/ in the facts tree when the report changed.176func (r *serioRegistry) publish(tree machine.FactsTree) {177	statuses := r.walk()178	for _, s := range statuses {179		if !slices.ContainsFunc(r.printed, func(p machine.SerioStatus) bool { return serioStatusEqual(p, s) }) {180			fmt.Println(describeSerio(s))181		}182	}183	r.printed = statuses184	if slices.EqualFunc(statuses, r.published, serioStatusEqual) {185		return186	}187	if tree.WriteSerio(statuses) == nil {188		r.published = statuses189	}190}191192// walk compares the declared entries with the serial lines sysfs193// shows now, starts a holder for each matched line that has none, and194// returns the report. It waits for the holders it started without the195// registry's lock, and takes the lock again to record their outcomes.196func (r *serioRegistry) walk() []machine.SerioStatus {197	statuses, started := r.plan()198	if len(started) == 0 {199		return statuses200	}201	deadline := time.NewTimer(serioSettleTimeout)202	defer deadline.Stop()203	for _, s := range started {204		select {205		case <-s.holder.settled:206		case <-deadline.C:207		}208	}209	r.mu.Lock()210	defer r.mu.Unlock()211	for _, s := range started {212		statuses[s.index] = r.settledStatus(statuses[s.index], s.protocol, s.line, s.holder)213	}214	return statuses215}216217// startedHolder is a holder a walk started, and the place in the218// report its outcome fills.219type startedHolder struct {220	index    int221	protocol machine.SerioProtocol222	line     serialLineInfo223	holder   *serioHolder224}225226// plan is the part of a walk that runs under the lock: it reads the227// lines, prunes the holders, and reports every entry, starting a228// holder where one is due.229func (r *serioRegistry) plan() ([]machine.SerioStatus, []startedHolder) {230	r.mu.Lock()231	defer r.mu.Unlock()232	if len(r.declared) == 0 && len(r.holders) == 0 {233		return nil, nil234	}235	lines := discoverSerialLines()236	r.prune()237	r.pruneUnbound(lines)238	// The most specific entry wins. Entries that name a serial take239	// their lines first, so an entry without one, declared earlier,240	// cannot hold a line that a specific entry names. The report keeps241	// the declaration order.242	taken := map[string]bool{}243	byEntry := make([][]machine.SerioStatus, len(r.declared))244	started := make([][]startedHolder, len(r.declared))245	for _, specific := range []bool{true, false} {246		for i, entry := range r.declared {247			if (entry.USB.Serial != "") == specific {248				byEntry[i], started[i] = r.entryStatuses(entry, lines, taken)249			}250		}251	}252	// Each entry numbered its started holders within its own report,253	// so the offsets move them to their places in the whole report.254	var all []startedHolder255	offset := 0256	for i := range byEntry {257		for _, s := range started[i] {258			s.index += offset259			all = append(all, s)260		}261		offset += len(byEntry[i])262	}263	return slices.Concat(byEntry...), all264}
init/serioattach.go 96.7%
1package main23// Holding one serio attachment: the five system calls that attach a4// serial line to the kernel's serio layer, and the goroutine that5// keeps the last of them blocked for the life of the boot.6//7// The calls are the ones inputattach runs, in its order. The program8// opens the tty, sets the line to the device's speed in raw mode,9// sets serport's line discipline with TIOCSETD, states the serio type10// with SPIOCSTYPE, and reads. serport registers the serio port inside11// that read, and unregisters it when the read returns12// (drivers/input/serio/serport.c, serport_ldisc_read). So the port,13// and every device its driver creates, exists exactly as long as one14// read blocks.15//16// PID 1 makes that read harder to hold than it is for inputattach.17// Init reaps every orphaned process on the machine, so SIGCHLD arrives18// often, and a signal that lands on the reading thread wakes the19// kernel's wait. serport's read then unregisters the port and returns20// zero, so a signal ends the port the same way an unplug does, and the21// CEC adapter disappears under every pod that holds it. A read that22// ran again would register a new port, not keep the old one. So each23// holder locks its goroutine to one OS thread and24// blocks signals on that thread before the first call. The kernel25// delivers a process-directed signal to a thread that does not block26// it, so the rest of init still receives SIGCHLD.27//28// A holder must never take PID 1 down. The machine plane's recover29// does not cover a goroutine a component starts, so the holder30// recovers its own panics and reports them as a status fact. It31// shares no map with any other goroutine: it writes only its own32// fields, each before it closes the channel that publishes them. Every33// error from the five calls is a status fact, never a failure of init.3435import (36	"fmt"37	"runtime"38	"time"39	"unsafe"4041	"golang.org/x/sys/unix"4243	"github.com/liken-sh/liken/liken/machine"44)4546// serialLine is the syscall boundary of one attachment: the four calls47// made on an open tty. seriotty.go is the real one, and the tests48// supply a fake, because an attachment needs a serial line, serport,49// and CAP_SYS_ADMIN, and a test host has none of the three.50type serialLine interface {51	setTermios(t *unix.Termios) error52	setLineDiscipline(discipline int) error53	setSerioType(serioType uint64) error54	read() error55	close() error56}5758// openLine opens a tty for an attachment.59type openLine func(path string) (serialLine, error)6061// nMouse is the line discipline number that serport registers. The62// kernel's name for it is N_MOUSE, from the serial mice that serport63// was first written for.64const nMouse = 26566// termiosSpeed returns the termios speed bits for a table's baud rate.67// The table names 9600 only, because both USB-CEC adapters speak it.68// It is a switch and not a map, so a holder reads no map at all.69func termiosSpeed(baud int) (uint32, bool) {70	switch baud {71	case 9600:72		return unix.B9600, true73	}74	return 0, false75}7677// serioTermios builds the line settings inputattach's setline writes:78// 8 data bits, the receiver on, a hangup on the last close, no modem79// control, no break and no parity errors in the input, no output or80// local processing, and a read that returns after one byte. The speed81// goes in both directions.82func serioTermios(baud int) (*unix.Termios, error) {83	speed, ok := termiosSpeed(baud)84	if !ok {85		return nil, fmt.Errorf("TCSETS: no termios speed for %d baud", baud)86	}87	t := &unix.Termios{88		Cflag:  unix.CS8 | unix.CREAD | unix.HUPCL | unix.CLOCAL | speed,89		Iflag:  unix.IGNBRK | unix.IGNPAR,90		Ispeed: speed,91		Ospeed: speed,92	}93	t.Cc[unix.VMIN] = 194	t.Cc[unix.VTIME] = 095	return t, nil96}9798// attachSerio runs the first four calls: it opens the tty, sets the99// line, sets the discipline, and states the serio type. It returns the100// open line, ready for the read that creates the port. Each error101// names its call and carries the kernel's text word for word, and a102// line that fails after it opened is closed again.103func attachSerio(open openLine, path string, p machine.SerioProtocol) (serialLine, error) {104	line, err := open(path)105	if err != nil {106		return nil, fmt.Errorf("open %s: %w", path, err)107	}108	if err := setUpLine(line, p); err != nil {109		line.close()110		return nil, err111	}112	return line, nil113}114115func setUpLine(line serialLine, p machine.SerioProtocol) error {116	t, err := serioTermios(p.Baud)117	if err != nil {118		return err119	}120	if err := line.setTermios(t); err != nil {121		return fmt.Errorf("TCSETS: %w", err)122	}123	if err := line.setLineDiscipline(nMouse); err != nil {124		return fmt.Errorf("TIOCSETD: %w", err)125	}126	// SPIOCSTYPE carries the type in the low byte, then an id byte and127	// an extra byte. The CEC adapters use neither, so both are zero.128	if err := line.setSerioType(uint64(p.Type)); err != nil {129		return fmt.Errorf("SPIOCSTYPE: %w", err)130	}131	return nil132}133134// holdRead is the fifth call. serport's read blocks until the port135// ends and then returns zero bytes, so a nil return means the port is136// gone. inputattach repeats the read on EINTR and EAGAIN, and so does137// this loop. The loop allocates nothing of its own and compares errno138// values directly.139func holdRead(line serialLine) error {140	for {141		err := line.read()142		if err == unix.EINTR || err == unix.EAGAIN {143			continue144		}145		return err146	}147}148149// faultSignals are the signals the kernel raises on the thread that150// faults. The holder leaves them open. When a thread faults with its151// fault signal blocked, the kernel resets the signal to its default152// action, which ends the process, and for PID 1 that is a kernel153// panic. With the signal open, Go's handler turns a nil dereference154// into a panic that the holder's recover catches.155var faultSignals = []unix.Signal{156	unix.SIGSEGV, unix.SIGBUS, unix.SIGFPE, unix.SIGILL, unix.SIGTRAP, unix.SIGSYS,157}158159// holderSignalMask is every signal except the fault signals. The mask160// includes the real-time signal that the Go runtime sends to every161// thread for a setuid-family call (runtime.doAllThreadsSyscall). That162// call stops the world, signals each thread, and spins until each one163// answers, and a holder's thread never answers, so the call would164// freeze all of PID 1. The rule is that init never makes a165// setuid-family call: syscall.Setuid, Setgid, Setgroups, and their166// relatives. Init runs as root for its whole life and needs none of167// them.168func holderSignalMask() unix.Sigset_t {169	var set unix.Sigset_t170	for i := range set.Val {171		set.Val[i] = ^set.Val[i]172	}173	for _, sig := range faultSignals {174		width := int(unsafe.Sizeof(set.Val[0])) * 8175		bit := int(sig) - 1176		set.Val[bit/width] &^= 1 << (bit % width)177	}178	return set179}180181// sigsetHas reports whether a signal is in a set.182func sigsetHas(set *unix.Sigset_t, sig unix.Signal) bool {183	width := int(unsafe.Sizeof(set.Val[0])) * 8184	bit := int(sig) - 1185	return set.Val[bit/width]&(1<<(bit%width)) != 0186}187188// serioHolder is one goroutine that holds one attachment. The walk189// that starts it reads its outcome through two channels. settled190// closes when the four attach calls finish, after attachErr is191// written, and a nil attachErr means the holder is in its read. done192// closes when the goroutine ends, after endErr is written.193//194// The registry keeps the last three fields beside the holder, under195// its own lock. The goroutine never reads or writes them.196type serioHolder struct {197	settled   chan struct{}198	attachErr error199	done      chan struct{}200	endErr    error201202	// identity is the inode of the tty's sysfs directory when the203	// holder started, usbPath is the sysfs path of the USB device204	// above the tty, started is when it started, and failures is the205	// count of failures in a row on that USB port before it206	// (serioretry.go).207	identity uint64208	usbPath  string209	started  time.Time210	failures int211}212213// startSerioHolder starts the goroutine that attaches the tty at path214// and holds the read. nudge is the walk's wake channel: the holder215// sends on it when it ends, so the walk reports the end without216// waiting for the next uevent.217func startSerioHolder(open openLine, path string, p machine.SerioProtocol, nudge chan<- struct{}) *serioHolder {218	h := &serioHolder{settled: make(chan struct{}), done: make(chan struct{})}219	go h.run(open, path, p, nudge)220	return h221}222223func (h *serioHolder) run(open openLine, path string, p machine.SerioProtocol, nudge chan<- struct{}) {224	attached := false225	var line serialLine226	defer func() {227		if r := recover(); r != nil {228			err := fmt.Errorf("the holder panicked: %v", r)229			if attached {230				h.endErr = err231				// A panic in the read leaves the tty open with serport's232				// discipline on it. Closing it lets a later holder open233				// the line again.234				closeAfterPanic(line)235			} else {236				h.attachErr = err237			}238		}239		if !attached {240			close(h.settled)241		}242		close(h.done)243		select {244		case nudge <- struct{}{}:245		default:246		}247	}()248249	// The goroutine never unlocks its thread. When a goroutine ends250	// while it holds its lock, the Go runtime ends the thread with it,251	// so the blocked mask leaves with the thread and never reaches a252	// goroutine that expects signals. On the process's main thread the253	// runtime parks the thread for good instead, which keeps the mask254	// out of every other goroutine the same way.255	runtime.LockOSThread()256	mask := holderSignalMask()257	if err := unix.PthreadSigmask(unix.SIG_SETMASK, &mask, nil); err != nil {258		h.attachErr = fmt.Errorf("blocking signals on the holder's thread: %w", err)259		return260	}261	opened, err := attachSerio(open, path, p)262	if err != nil {263		h.attachErr = err264		return265	}266	line = opened267	close(h.settled)268	attached = true269	h.endErr = holdRead(line)270	// The recover path closes line when it is set, so the final close271	// clears it first and a panic in the close cannot close twice.272	last := line273	line = nil274	last.close()275}276277// closeAfterPanic closes a line inside the holder's recovery. A second278// panic there would escape the recover that is already running and279// end PID 1, so the close carries a recover of its own.280func closeAfterPanic(line serialLine) {281	defer func() { _ = recover() }()282	if line != nil {283		line.close()284	}285}
init/seriolines.go 82.1%
1package main23// Finding the serial lines a spec.serio entry can match.4//5// Every tty the kernel registers has an entry under /sys/class/tty,6// and the entry resolves to the tty's directory in the device tree.7// A tty that belongs to hardware carries a device link to its parent.8// For a CDC ACM adapter the parent is the USB interface that cdc_acm9// bound, and the USB device above the interface holds the identity:10// idVendor, idProduct, and serial. A usb-serial converter puts one11// more level between them, so the search climbs until a directory12// holds the identity. The console's virtual terminals and the13// pseudo-terminals have no device link, and the walk passes over them.1415import (16	"os"17	"path/filepath"18	"strings"1920	"golang.org/x/sys/unix"2122	"github.com/liken-sh/liken/liken/machine"23)2425// serialLineInfo is one tty with a USB device above it. identity is26// the inode of the tty's sysfs directory, which is new each time the27// kernel registers the tty, so a tty that registers again under the28// same name reads as new hardware. usbPath is the sysfs path of the29// USB device, which names the port the adapter is plugged into and30// stays the same when the adapter enumerates again.31type serialLineInfo struct {32	tty                     string33	dir                     string34	identity                uint6435	usbPath                 string36	vendor, product, serial string37}3839// usbIdentityDepth bounds the climb from a tty's parent to the USB40// device: the interface's parent for CDC ACM, and one level more for a41// usb-serial port.42const usbIdentityDepth = 34344// discoverSerialLines lists the ttys whose hardware is a USB device.45// The list follows the class directory's order, which is sorted by46// name, so ttyACM0 comes before ttyACM1.47func discoverSerialLines() []serialLineInfo {48	classDir := filepath.Join(sysfsRoot, "class", "tty")49	entries, err := os.ReadDir(classDir)50	if err != nil {51		return nil52	}53	var lines []serialLineInfo54	for _, entry := range entries {55		dir, err := filepath.EvalSymlinks(filepath.Join(classDir, entry.Name()))56		if err != nil {57			continue58		}59		parent, err := filepath.EvalSymlinks(filepath.Join(dir, "device"))60		if err != nil {61			continue62		}63		usb := usbDeviceAbove(parent)64		if usb == "" {65			continue66		}67		var stat unix.Stat_t68		if err := unix.Stat(dir, &stat); err != nil {69			continue70		}71		lines = append(lines, serialLineInfo{72			tty:      entry.Name(),73			dir:      dir,74			identity: stat.Ino,75			usbPath:  usb,76			vendor:   sysfsString(usb, "idVendor"),77			product:  sysfsString(usb, "idProduct"),78			serial:   sysfsString(usb, "serial"),79		})80	}81	return lines82}8384// usbDeviceAbove climbs from a directory to the first one that holds a85// USB device's identity, or returns "" when none does within the86// bound.87func usbDeviceAbove(dir string) string {88	for range usbIdentityDepth {89		if sysfsString(dir, "idVendor") != "" {90			return dir91		}92		up := filepath.Dir(dir)93		if up == dir {94			return ""95		}96		dir = up97	}98	return ""99}100101// usbDevicePresent reports whether a USB device with an entry's102// identity is plugged in, with or without a serial line. It reads the103// bus's own list, where a device's directory name has no colon and an104// interface's has one.105func usbDevicePresent(entry machine.SerioAttachment) bool {106	busDir := filepath.Join(sysfsRoot, "bus", "usb", "devices")107	entries, err := os.ReadDir(busDir)108	if err != nil {109		return false110	}111	for _, e := range entries {112		if strings.Contains(e.Name(), ":") {113			continue114		}115		dir := filepath.Join(busDir, e.Name())116		if entry.Matches(sysfsString(dir, "idVendor"), sysfsString(dir, "idProduct"), sysfsString(dir, "serial")) {117			return true118		}119	}120	return false121}
init/serioretry.go 97.3%
1package main23// How the serio walk retries an attachment that failed.4//5// An attachment fails in two ways. The kernel refuses one of the6// attach calls, or the read ends while the tty stays. The second is7// what cdc_acm does on a USB reset-resume: acm_reset_resume hangs up8// the tty and keeps it, and serport's hangup ends the read. A plain9// USB reset goes further. cdc_acm has no post_reset, so the USB core10// unbinds it and probes it again, and a new tty takes the old name11// within milliseconds, often inside one settle of the uevent burst.12//13// So a refusal is kept for the USB port the adapter is plugged into,14// the sysfs path of its USB device, with the tty it happened on,15// identified by the inode of the tty's sysfs directory. The kernel16// gives a tty that registers again a new directory with a new inode.17// A refusal on the same tty retries after a backoff that doubles from18// one second to five minutes. Opening the line on every uevent instead19// would toggle the adapter's modem lines to get the same refusal20// again, and uevents arrive often on a machine that runs pods.21//22// A new tty on the port attaches at once after the first failure, so23// an adapter that resets once comes back at once. The failure count24// belongs to the port, not to the tty, so an adapter that enumerates25// again on every attach still backs off from its second failure on.26// A refusal stays while its port is empty, so an adapter that is27// plugged in again keeps its count: a long hold before the unplug28// already started the count over.2930import (31	"time"32)3334const (35	// serioRetryBase is the first backoff, and serioRetryMax the bound36	// it doubles to.37	serioRetryBase = time.Second38	serioRetryMax  = 5 * time.Minute3940	// serioStableHold is how long a holder must hold its port for its41	// end to start the backoff over. An adapter that resets once a day42	// then waits one second to come back, not five minutes.43	serioStableHold = time.Minute44)4546// serioRefusal is one refused USB port: the identity of the tty the47// refusal happened on, the message the status carries, the count of48// failures in a row on the port, and the time of the next attempt.49type serioRefusal struct {50	identity uint6451	message  string52	failures int53	retryAt  time.Time54}5556// serioBackoff is the wait after a count of failures in a row.57func serioBackoff(failures int) time.Duration {58	wait := serioRetryBase59	for range failures - 1 {60		wait *= 261		if wait >= serioRetryMax {62			return serioRetryMax63		}64	}65	return wait66}6768// nextFailures counts one more failure, and starts the count over69// after a holder that held its port for serioStableHold or longer.70func nextFailures(prior int, held time.Duration) int {71	if held >= serioStableHold {72		return 173	}74	return prior + 175}7677// refuse records a failure on a holder's USB port.78func (r *serioRegistry) refuse(h *serioHolder, message string, held time.Duration) {79	failures := nextFailures(h.failures, held)80	r.refusals[h.usbPath] = serioRefusal{81		identity: h.identity,82		message:  message,83		failures: failures,84		retryAt:  r.now().Add(serioBackoff(failures)),85	}86}8788// prune removes the holders that ended, and records why each ended on89// its USB port. A holder whose tty registered again while it held the90// read ends the same way, and replace records it (seriostatus.go).91func (r *serioRegistry) prune() {92	for tty, h := range r.holders {93		select {94		case <-h.done:95		default:96			continue97		}98		delete(r.holders, tty)99		r.recordEnd(h)100	}101}102103// recordEnd records the failure a holder's end stands for.104func (r *serioRegistry) recordEnd(h *serioHolder) {105	if h.attachErr != nil {106		r.refuse(h, h.attachErr.Error(), 0)107		return108	}109	r.refuse(h, endMessage(h.endErr), r.now().Sub(h.started))110}111112// endMessage words a holder's end for the status.113func endMessage(err error) string {114	if err != nil {115		return "read: " + err.Error()116	}117	return "read: the kernel ended the attachment while the tty stayed"118}119120// nextRetry returns the wait until the earliest moment a walk is due121// with no uevent to announce it: a refusal's next attempt, or the end122// of an unbound port's grace (seriostatus.go). A moment that has123// passed is not due again. The walk at that moment either acted on it124// or moved it forward, and a refusal whose port is empty has nothing125// to act on, so counting past moments would wake the walk in a loop.126func (r *serioRegistry) nextRetry() (time.Duration, bool) {127	r.mu.Lock()128	defer r.mu.Unlock()129	now := r.now()130	var earliest time.Time131	due := func(at time.Time) {132		if at.After(now) && (earliest.IsZero() || at.Before(earliest)) {133			earliest = at134		}135	}136	for _, refusal := range r.refusals {137		due(refusal.retryAt)138	}139	for _, u := range r.unbound {140		due(u.since.Add(serioBindGrace))141	}142	if earliest.IsZero() {143		return 0, false144	}145	return earliest.Sub(now), true146}
init/seriostatus.go 98.5%
1package main23// What the serio walk reports for each declared entry.45import (6	"fmt"7	"os"8	"path/filepath"9	"slices"10	"strings"11	"time"1213	"github.com/liken-sh/liken/liken/hardware"14	"github.com/liken-sh/liken/liken/machine"15)1617// entryStatuses reports one declared entry: one status for each18// serial line it matches, or one Missing status when it matches none.19// taken keeps a line that another entry took from a second holder, and20// a Missing entry whose lines another entry took says so. It also21// returns the holders it started, numbered by their places in its22// report, for the walk to wait on.23func (r *serioRegistry) entryStatuses(entry machine.SerioAttachment, lines []serialLineInfo, taken map[string]bool) ([]machine.SerioStatus, []startedHolder) {24	base := machine.SerioStatus{Protocol: entry.Protocol, USB: entry.USB}25	p, ok := machine.LookupSerioProtocol(entry.Protocol)26	if !ok {27		base.State = machine.SerioRefused28		base.Message = fmt.Sprintf("this release has no protocol %q; the protocols are %v", entry.Protocol, machine.SerioProtocolNames())29		return []machine.SerioStatus{base}, nil30	}31	var statuses []machine.SerioStatus32	var started []startedHolder33	heldElsewhere := false34	for _, line := range lines {35		if !entry.Matches(line.vendor, line.product, line.serial) {36			continue37		}38		if taken[line.tty] {39			heldElsewhere = true40			continue41		}42		taken[line.tty] = true43		s := base44		s.TTY = line.tty45		status, h := r.lineStatus(s, p, line)46		if h != nil {47			started = append(started, startedHolder{index: len(statuses), protocol: p, line: line, holder: h})48		}49		statuses = append(statuses, status)50	}51	if len(statuses) == 0 {52		base.State = machine.SerioMissing53		base.Message = missingMessage(entry, p)54		if heldElsewhere {55			base.Message = fmt.Sprintf("every serial line of USB device %s:%s is held for another spec.serio entry",56				entry.USB.Vendor, entry.USB.Product)57		}58		return []machine.SerioStatus{base}, nil59	}60	return statuses, started61}6263// serioBindGrace is how long a port may stay unbound before the status64// blames its driver. pulse8_connect exchanges several commands with the65// adapter before the driver binds, each with a timeout of its own, so66// an unbound port is the ordinary state of a probe for a moment.67const serioBindGrace = 5 * time.Second6869// serioUnbound is a port a walk saw with no driver: the tty it is on,70// the port, and when a walk first saw it unbound.71type serioUnbound struct {72	identity uint6473	port     string74	since    time.Time75}7677// lineStatus reports one matched line. It starts the line's holder78// when the line has none and nothing stands in the way, and returns79// that holder for the walk to wait on outside the lock.80func (r *serioRegistry) lineStatus(s machine.SerioStatus, p machine.SerioProtocol, line serialLineInfo) (machine.SerioStatus, *serioHolder) {81	if h, held := r.holders[line.tty]; held {82		if h.identity == line.identity {83			return r.settledStatus(s, p, line, h), nil84		}85		// The tty registered again while this holder held the old one,86		// so the holder describes a tty that is gone. Its goroutine87		// ends when the kernel hangs up the old tty, and the registry88		// records the failure now, so an adapter that enumerates again89		// on every attach still counts toward the backoff.90		delete(r.holders, line.tty)91		r.refuse(h, "the tty registered again while the machine held it", r.now().Sub(h.started))92	}93	failures := 094	refusal, refused := r.refusals[line.usbPath]95	if refused {96		failures = refusal.failures97		// A new tty on the port attaches at once after one failure, and98		// waits for the backoff after more (serioretry.go).99		waits := refusal.identity == line.identity || failures > 1100		if waits && r.now().Before(refusal.retryAt) {101			s.State, s.Message = machine.SerioRefused, refusal.message102			return s, nil103		}104	}105	// The modules are checked before the open, so a machine that is106	// missing one never touches the line. The load that adds a module107	// sends a uevent, and the walk after it tries again. A refusal108	// whose backoff ran out waits one more backoff, so the walk does109	// not come due again at once.110	if missing := missingSerioModules(p); len(missing) > 0 {111		if refused {112			refusal.retryAt = r.now().Add(serioBackoff(failures))113			r.refusals[line.usbPath] = refusal114		}115		s.State = machine.SerioRefused116		s.Message = "declare " + strings.Join(missing, " and ") + " in spec.modules"117		return s, nil118	}119	delete(r.refusals, line.usbPath)120	h := startSerioHolder(r.open, filepath.Join(devRoot, line.tty), p, r.nudge)121	h.identity, h.usbPath, h.started, h.failures = line.identity, line.usbPath, r.now(), failures122	r.holders[line.tty] = h123	s.State = machine.SerioRefused124	s.Message = fmt.Sprintf("the attach calls did not return within %s", serioSettleTimeout)125	return s, h126}127128// settledStatus reports a line from its holder, without waiting. A129// holder whose calls have not returned reports the wait. A holder whose130// attach failed leaves the registry, and its port is refused. A holder131// that ended reports its end, and the next walk's prune records it.132func (r *serioRegistry) settledStatus(s machine.SerioStatus, p machine.SerioProtocol, line serialLineInfo, h *serioHolder) machine.SerioStatus {133	select {134	case <-h.settled:135	default:136		s.State = machine.SerioRefused137		s.Message = fmt.Sprintf("the attach calls did not return within %s", serioSettleTimeout)138		return s139	}140	if h.attachErr != nil {141		if r.holders[line.tty] == h {142			delete(r.holders, line.tty)143			r.refuse(h, h.attachErr.Error(), 0)144		}145		s.State, s.Message = machine.SerioRefused, h.attachErr.Error()146		return s147	}148	select {149	case <-h.done:150		s.State, s.Message = machine.SerioRefused, endMessage(h.endErr)151		return s152	default:153	}154	port, nodes, bound := serioPort(line.dir)155	s.Port, s.Nodes, s.Message = port, nodes, ""156	switch {157	case port == "":158		// The holder is in its read, and the kernel registers the port159		// a moment later. The registration's uevents bring the next160		// walk.161		delete(r.unbound, line.tty)162		s.State = machine.SerioRefused163		s.Message = fmt.Sprintf("attaching: the kernel has not registered a serio port on %s yet", line.tty)164	case !bound:165		// A port that stays unbound past the grace is a probe that166		// failed, for example when the adapter did not answer the167		// driver's first commands, and it has no CEC device.168		u, seen := r.unbound[line.tty]169		if !seen || u.identity != line.identity || u.port != port {170			u = serioUnbound{identity: line.identity, port: port, since: r.now()}171			r.unbound[line.tty] = u172		}173		s.State = machine.SerioRefused174		s.Message = fmt.Sprintf("attaching: waiting for %s to bind %s", p.Driver, port)175		if r.now().Sub(u.since) >= serioBindGrace {176			s.Message = fmt.Sprintf("%s on %s has no driver: %s did not bind it; the kernel log names the cause",177				port, line.tty, p.Driver)178		}179	default:180		delete(r.unbound, line.tty)181		s.State = machine.SerioAttached182	}183	return s184}185186// pruneUnbound forgets the unbound ports of ttys that left or187// registered again.188func (r *serioRegistry) pruneUnbound(lines []serialLineInfo) {189	for tty, u := range r.unbound {190		if !slices.ContainsFunc(lines, func(l serialLineInfo) bool { return l.tty == tty && l.identity == u.identity }) {191			delete(r.unbound, tty)192		}193	}194}195196// missingSerioModules names the modules an attach needs that the197// kernel does not hold: serport, and the protocol's driver. The line198// driver is not checked here, because a line that exists proves it.199func missingSerioModules(p machine.SerioProtocol) []string {200	var missing []string201	for _, name := range []string{p.Discipline, p.Driver} {202		if !moduleIsResident(sysModuleDir, name) {203			missing = append(missing, name)204		}205	}206	return missing207}208209// missingMessage says why an entry matches no line. A USB device with210// the entry's identity and no tty is an adapter whose line driver is211// not loaded, and the fix is a module. No such device is an adapter212// that is unplugged.213func missingMessage(entry machine.SerioAttachment, p machine.SerioProtocol) string {214	identity := entry.USB.Vendor + ":" + entry.USB.Product215	if entry.USB.Serial != "" {216		identity += " with serial " + entry.USB.Serial217	}218	if usbDevicePresent(entry) {219		return fmt.Sprintf("USB device %s has no serial line; declare %s in spec.modules", identity, p.LineDriver)220	}221	return fmt.Sprintf("no USB device %s is plugged in", identity)222}223224// serioPort reads the serio port that serport registered under a tty,225// the device nodes its driver created under the port, and whether a226// driver is bound to the port.227func serioPort(ttyDir string) (string, []string, bool) {228	entries, err := os.ReadDir(ttyDir)229	if err != nil {230		return "", nil, false231	}232	for _, entry := range entries {233		if !entry.IsDir() || !strings.HasPrefix(entry.Name(), "serio") {234			continue235		}236		dir := filepath.Join(ttyDir, entry.Name())237		var nodes []string238		for _, node := range hardware.SubtreeNodes(dir) {239			nodes = append(nodes, node.Path)240		}241		_, err := os.Lstat(filepath.Join(dir, "driver"))242		return entry.Name(), nodes, err == nil243	}244	return "", nil, false245}246247// describeSerio renders one status as a console line.248func describeSerio(s machine.SerioStatus) string {249	entry := s.Attachment().String()250	switch s.State {251	case machine.SerioAttached:252		line := fmt.Sprintf("liken: serio: %s attached on %s", entry, s.TTY)253		if s.Port != "" {254			line += " as " + s.Port255		}256		if len(s.Nodes) > 0 {257			line += " (" + strings.Join(s.Nodes, ", ") + ")"258		}259		return line260	case machine.SerioMissing:261		return fmt.Sprintf("liken: serio: %s is missing: %s", entry, s.Message)262	}263	if s.TTY == "" {264		return fmt.Sprintf("liken: serio: %s refused: %s", entry, s.Message)265	}266	return fmt.Sprintf("liken: serio: %s on %s refused: %s", entry, s.TTY, s.Message)267}268269func serioStatusEqual(a, b machine.SerioStatus) bool {270	return a.Protocol == b.Protocol && a.USB == b.USB && a.TTY == b.TTY && a.Port == b.Port &&271		a.State == b.State && a.Message == b.Message && slices.Equal(a.Nodes, b.Nodes)272}
init/seriotty.go 0.0%
1package main23// The real serial line behind serialLine: an open tty and the system4// calls serioattach.go makes on it.56import (7	"unsafe"89	"golang.org/x/sys/unix"10)1112// ttyLine is one open tty.13type ttyLine struct {14	fd int15}1617// openTTY opens a tty the way inputattach does. O_NOCTTY keeps the18// line from becoming init's controlling terminal, and O_NONBLOCK lets19// the open return before the modem lines say the device is ready.20// O_CLOEXEC is liken's addition: init starts k3s, and the child must21// not inherit a descriptor that holds the port open.22func openTTY(path string) (serialLine, error) {23	fd, err := unix.Open(path, unix.O_RDWR|unix.O_NOCTTY|unix.O_NONBLOCK|unix.O_CLOEXEC, 0)24	if err != nil {25		return nil, err26	}27	return ttyLine{fd: fd}, nil28}2930// setTermios is tcsetattr with TCSANOW, which is the TCSETS ioctl.31func (l ttyLine) setTermios(t *unix.Termios) error {32	return unix.IoctlSetTermios(l.fd, unix.TCSETS, t)33}3435// setLineDiscipline is TIOCSETD. The kernel takes the discipline36// number through a pointer to an int.37func (l ttyLine) setLineDiscipline(discipline int) error {38	return unix.IoctlSetPointerInt(l.fd, unix.TIOCSETD, discipline)39}4041// spiocstype is SPIOCSTYPE from the kernel's include/uapi/linux/serio.h,42// _IOW('q', 0x01, unsigned long): the write direction in bits 30 and43// 31, the argument's size in bits 16 to 29, the type character in bits44// 8 to 15, and the number in bits 0 to 7. x/sys/unix does not carry it.45const spiocstype = 1<<30 | uintptr(unsafe.Sizeof(uintptr(0)))<<16 | uintptr('q')<<8 | 0x014647// setSerioType is SPIOCSTYPE. serport reads the argument as an48// unsigned long through a pointer, so the value goes in a word of that49// size.50func (l ttyLine) setSerioType(serioType uint64) error {51	value := uintptr(serioType)52	_, _, errno := unix.Syscall(unix.SYS_IOCTL, uintptr(l.fd), spiocstype, uintptr(unsafe.Pointer(&value)))53	if errno != 0 {54		return errno55	}56	return nil57}5859// read is read(fd, NULL, 0), which inputattach calls. serport's read60// ignores the buffer, and a zero-length slice passes none.61func (l ttyLine) read() error {62	_, err := unix.Read(l.fd, nil)63	return err64}6566func (l ttyLine) close() error {67	return unix.Close(l.fd)68}
init/slotloader.go 85.4%
1package main23// The loader that a firmware with no boot preferences finds by itself.4//5// bootentries.go writes this machine's boot menu into NVRAM, and that6// menu is the only place a UEFI machine's kernel command line lives.7// Some firmware resets its variables to defaults: an update does it, a8// dead NVRAM battery does it, and the setup menu's own "load defaults"9// does it. That erases every entry and every command line at once, and10// leaves a complete installed disk that nothing can start.11//12// The UEFI specification answers this with one fallback. A firmware13// with nothing to boot searches each device for one fixed path,14// \EFI\BOOT\BOOTX64.EFI, and runs whatever it finds. This is how an15// installation stick boots a machine that has never seen it. So liken16// puts a loader at that path on the slot it has proven, with a Boot17// Loader Specification entry beside it that carries the same command18// line the firmware's own entry would have carried. A machine that19// loses its variables then boots the proven release, and that boot20// writes the entries again.21//22// The loader lives on the proven slot alone. A firmware at its defaults23// takes the first answer it finds, and the other slot holds either an24// older release or nothing at all. One answer means such a firmware25// cannot boot the wrong half of the pair.26//27// The program itself is not vendored twice. systemd-bootx64.efi is a28// release artifact, so every slot already carries the menu that an29// installation stick boots (the systemd-boot domain explains why liken30// ships one). Writing the fallback is a copy inside one slot.3132import (33	"bytes"34	"fmt"35	"os"36	"path/filepath"37	"strings"3839	"golang.org/x/sys/unix"40)4142// bootMenuName is the boot menu program as a release names it. Both an43// installation stick and a slot's fallback loader are copies of this44// one file.45const bootMenuName = "systemd-bootx64.efi"4647// defaultLoaderPath is the one path a firmware with no boot48// preferences looks for. The UEFI specification fixes this name per49// architecture, so it is not a choice liken makes.50func defaultLoaderPath(mount string) string {51	return filepath.Join(mount, "EFI", "BOOT", "BOOTX64.EFI")52}5354// otherSlot names the other half of the blue-green pair.55func otherSlot(slot string) string {56	if slot == "B" {57		return "A"58	}59	return "B"60}6162// writeSlotLoader puts the fallback loader on one slot: the loader63// program at the firmware's default path, the loader's one setting, and64// one entry that boots this slot. Each file is compared before it is65// written, so a slot that already agrees costs no writes. A slot is FAT66// on a real disk, and a rewrite on every boot buys nothing.67//68// A slot with no menu program cannot carry a loader. This returns an69// error in that case and writes nothing at all, because the caller70// removes the other slot's loader only after this one succeeds.71func writeSlotLoader(mount, slot, machineName string) error {72	program, err := os.ReadFile(filepath.Join(mount, bootMenuName))73	if err != nil {74		return fmt.Errorf("slot %s carries no %s, so it cannot answer the firmware's default boot path: %w",75			slot, bootMenuName, err)76	}7778	files := map[string][]byte{79		defaultLoaderPath(mount):                                         program,80		filepath.Join(mount, "loader", "loader.conf"):                    slotLoaderConfText(),81		filepath.Join(mount, "loader", "entries", "liken-"+slot+".conf"): slotLoaderEntryText(mount, slot, machineName),82	}83	wrote := false84	for path, want := range files {85		if current, err := os.ReadFile(path); err == nil && bytes.Equal(current, want) {86			continue87		}88		if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {89			return err90		}91		if err := writeFileDurably(path, want); err != nil {92			return fmt.Errorf("writing %s: %w", path, err)93		}94		wrote = true95	}96	if wrote {97		if err := flushSlot(mount); err != nil {98			return err99		}100		fmt.Printf("liken: system: slot %s now answers the firmware's default boot path\n", slot)101	}102	return nil103}104105// flushSlot forces a slot's directory updates all the way out to the106// disk. Two calls, because they do different halves of the job, and the107// installer's power-off path needs the same pair for the same reason.108//109// Each file above became durable through a rename, and on FAT a rename110// is the only record of a file's name, size, and first cluster. The111// per-file fsync inside writeFileDurably covers the file's own bytes.112// It does not cover the directory entry that the rename wrote, and an113// fsync of the directory does not reach it either: a FAT directory's114// entries live in buffers attached to the block device, not in the115// directory's own pages. A machine cut off in that state holds a loader116// entry with the right name, no size, and no data, which is a fallback117// that a firmware will start and then fail to boot from.118//119// unix.Sync walks every mounted filesystem and writes its dirty state120// back to the drivers. syncDirectory then asks this slot's drive to121// empty its own write cache.122func flushSlot(mount string) error {123	unix.Sync()124	if err := syncDirectory(mount); err != nil {125		return fmt.Errorf("flushing slot %s to the disk: %w", mount, err)126	}127	return nil128}129130// removeSlotLoader takes the fallback loader off a slot. Only the131// loader program goes. Its entry stays, because the program is what a132// firmware looks for, and an entry with no program beside it boots133// nothing.134func removeSlotLoader(mount string) {135	path := defaultLoaderPath(mount)136	if _, err := os.Stat(path); err != nil {137		return138	}139	if err := os.Remove(path); err != nil {140		fmt.Fprintf(os.Stderr, "liken: system: removing the fallback loader at %s: %v\n", path, err)141		return142	}143	if err := flushSlot(mount); err != nil {144		fmt.Fprintf(os.Stderr, "liken: system: %v\n", err)145	}146	fmt.Printf("liken: system: %s no longer answers the firmware's default boot path\n", path)147}148149// slotLoaderConfText is the loader's one setting. A boot that reaches150// this loader reached it because the firmware had no preference left to151// read, and no person is standing at the machine. So it must never152// wait.153func slotLoaderConfText() []byte {154	return []byte(`# liken wrote this loader on the slot it has proven, for one purpose:155# a firmware that lost its boot variables finds \EFI\BOOT\BOOTX64.EFI156# here and boots this slot. The entries directory holds one entry, and157# no person is here to pick it.158timeout 0159`)160}161162// slotLoaderEntryText renders the entry that boots this slot, in163// systemd-boot's Boot Loader Specification form. The options line is164// the command line the firmware's own entry carries, minus the initrd=165// parameters: this loader stages the archives from its own initrd166// lines, so naming them twice would load each one twice.167//168// The entry names only the archives that this slot holds. systemd-boot169// refuses an entry whose initrd is missing, and a release older than an170// archive does not carry it. Such an entry would be worse than no171// fallback, because it fails after the firmware has already committed172// to it.173func slotLoaderEntryText(mount, slot, machineName string) []byte {174	var b strings.Builder175	fmt.Fprintf(&b, "title    %s\n", slotEntryDescription(slot))176	fmt.Fprint(&b, "linux    /vmlinuz\n")177	for _, name := range slotInitrds() {178		if _, err := os.Stat(filepath.Join(mount, name)); err != nil {179			continue180		}181		fmt.Fprintf(&b, "initrd   /%s\n", name)182	}183	fmt.Fprintf(&b, "options  %s\n", strings.Join(slotArgs(machineName, slot, nil), " "))184	return []byte(b.String())185}
init/slots.go 76.9%
1package main23// The role-specific half of slot formatting. The FAT32 format itself4// lives in the disks package, shared with the install-media builder.5// This file holds init's policy: which label a slot carries, and the6// identity field a machine creates at format time.78import (9	"crypto/rand"10	"encoding/binary"11	"os"1213	"github.com/liken-sh/liken/liken/disks"14	"github.com/liken-sh/liken/liken/machine"15)1617// formatSlot formats one of the FAT32 roles' partitions. It labels18// the partition for the role it serves, so the volume shows its role19// in any directory listing. For these roles, the label also lets20// GRUB find the partition, because the label is what GRUB's search21// command uses. The volume ID is FAT's only identity field; FAT3222// has no UUIDs. formatSlot draws the volume ID from crypto/rand,23// the same source a claim already uses for each partition's GPT24// unique GUID, because a claim formats bootHome and both system25// slots inside one second, and volumes formatted in the same second26// must still carry ids that tell them apart.27func formatSlot(devPath string, sizeBytes uint64, role machine.StorageRoleName) error {28	f, err := os.OpenFile(devPath, os.O_RDWR, 0)29	if err != nil {30		return err31	}32	defer f.Close()33	label := "LIKEN-SYS-A"34	switch role {35	case machine.SystemBRole:36		label = "LIKEN-SYS-B"37	case machine.BootHomeRole:38		label = "LIKEN-BOOT"39	}40	return disks.FormatFAT32(f, sizeBytes, label, randomVolumeID())41}4243// randomVolumeID draws a 32-bit FAT volume id from crypto/rand. A44// failure here means the kernel has no randomness source at all, the45// same fault that RandomGUID treats as fatal for a partition's GPT46// unique GUID, so formatSlot stops rather than mint an id two slots47// might end up sharing.48func randomVolumeID() uint32 {49	var b [4]byte50	if _, err := rand.Read(b[:]); err != nil {51		panic(err)52	}53	return binary.LittleEndian.Uint32(b[:])54}
init/softdeps.go 88.3%
1package main23import (4	"bytes"5	"debug/elf"6	"fmt"7	"os"8	"path/filepath"9	"slices"10	"strings"1112	"github.com/klauspost/compress/zstd"13)1415// Reading a module's soft dependencies.16//17// A hard dependency is a symbol one module needs from another, and18// depmod records every one of them in modules.dep, ordered so that19// loading right-to-left satisfies them. A soft dependency is a weaker20// thing: a hint that one module names another to load before it, so21// that a device probes to the right driver. The lab's Realtek NIC is22// the case that taught this. The r8169 driver asks, through a "pre"23// soft dependency, for the realtek PHY library to load ahead of it.24// Without realtek the NIC still binds, but to a generic PHY, and the25// link does not come up the same way.26//27// depmod does not index soft dependencies. modules.dep, modules.alias,28// and the rest say nothing about them. Each module carries its own29// soft dependencies inside its compiled object, in the .modinfo30// section, as strings of the form "softdep=pre: realtek". modprobe31// reads them there, and so does liken. (modprobe also reads softdep32// lines from modprobe.d files, which liken does not ship, so a33// module's own .modinfo is the whole source here.)34//35// This knowledge never reaches the module loader. The loader stays36// explicit: it loads what a manifest declares, in the declared order,37// and a soft dependency the manifest did not name does not load. liken38// has no udev for the same reason. The manifest is the whole truth39// about what a machine runs. Soft dependencies inform the advice a40// person reads before they write the manifest, so a recommendation can41// name realtek and r8169 together, but they never change what the42// loader does with the manifest it is handed.43//44// Only "pre" soft dependencies matter here. liken loads a driver so45// that a device appears, and a "pre" dependency changes which driver46// binds, so it changes that outcome. A "post" dependency loads after47// the module and does not change whether the device appears, so the48// recommendation ignores it.4950// busCompanions names the modules that a driver needs beside it, which51// no index in the module tree can name.52//53// A recommendation starts from modules.alias, which maps a device's54// fingerprint to the drivers that claim that fingerprint. Some drivers55// claim nothing there, because they bind over a bus that another56// driver creates: usbhid turns a USB interface into a HID device, and57// hid-generic then binds that HID device. No modalias points at58// hid-generic, and no soft dependency names it, so a report that reads59// only the index recommends a keyboard's transport and stops one60// module short of a working keyboard.61//62// The table is the place to record each of these pairs as hardware63// proves them. It stays short on purpose: an entry here is a claim64// that the index cannot express, not a shortcut around reading it.65var busCompanions = map[string][]string{66	"usbhid": {"hid_generic"},67}6869// driverChain is the full ordered list a person declares to make one70// driver work: its soft dependencies, the driver, and any companion71// that binds over the bus the driver creates. The companions come last72// because they bind to what the driver produces.73func driverChain(base, name string) []string {74	chain := softdepChain(base, name)75	for _, module := range chain {76		for _, companion := range busCompanions[strings.ReplaceAll(module, "-", "_")] {77			if !slices.Contains(chain, companion) {78				chain = append(chain, companion)79			}80		}81	}82	return chain83}8485// softdepChain expands one module name into the ordered list of86// modules to declare so the kernel binds the intended driver: the87// module's "pre" soft dependencies first, resolved recursively because88// a soft dependency can carry its own, then the module itself. The89// report boot and the unclaimed-hardware advice both call this to turn90// a single recommended driver into the full list a person writes into91// spec.modules. A tree with no readable index yields the name alone,92// which is the honest answer when nothing can be resolved.93func softdepChain(base, name string) []string {94	deps, err := readModulesDep(filepath.Join(base, "modules.dep"))95	if err != nil {96		return []string{name}97	}98	return expandSoftdeps(name, func(n string) []string {99		return modulePreSoftdeps(base, deps, n)100	})101}102103// expandSoftdeps walks the "pre" soft-dependency graph depth-first and104// returns the names in load order, each name once. The seen set is the105// cycle guard. Soft dependencies are only a hint, and nothing stops a106// module from naming one that names it back, so the walk records every107// name it enters and never enters one twice. A name already placed by108// an earlier branch is not placed again.109func expandSoftdeps(name string, pre func(string) []string) []string {110	return appendSoftdeps(name, pre, map[string]bool{}, nil)111}112113func appendSoftdeps(name string, pre func(string) []string, seen map[string]bool, out []string) []string {114	key := strings.ReplaceAll(name, "-", "_")115	if seen[key] {116		return out117	}118	seen[key] = true119	for _, p := range pre(name) {120		out = appendSoftdeps(p, pre, seen, out)121	}122	return append(out, name)123}124125// modulePreSoftdeps returns the "pre" soft dependencies that a single126// module names in its .modinfo, without recursing. It finds the127// module's file through modules.dep, whose first field on each line is128// the module's own path, then reads the strings from that file.129func modulePreSoftdeps(base string, deps map[string][]string, name string) []string {130	entry, ok := deps[strings.ReplaceAll(name, "-", "_")]131	if !ok || len(entry) == 0 {132		return nil133	}134	modinfo, err := readModinfo(filepath.Join(base, entry[0]))135	if err != nil {136		return nil137	}138	return parseSoftdepPre(modinfo)139}140141// readModinfo returns the raw bytes of a module's .modinfo section.142// The module files are compiled objects, so the soft-dependency143// strings live in an ELF section, not in any index. liken's modules144// are zstd-compressed exactly as Ubuntu shipped them, and the kernel145// normally decompresses them itself during finit_module. Reading the146// section back in userspace has no such help, so this decompresses the147// file first when its name says it carries a zstd frame.148func readModinfo(path string) ([]byte, error) {149	raw, err := os.ReadFile(path)150	if err != nil {151		return nil, err152	}153	if strings.HasSuffix(path, ".zst") {154		raw, err = zstdDecode(raw)155		if err != nil {156			return nil, err157		}158	}159	return elfSection(raw, ".modinfo")160}161162// elfSection returns one named section's bytes from an ELF image held163// in memory.164func elfSection(image []byte, name string) ([]byte, error) {165	f, err := elf.NewFile(bytes.NewReader(image))166	if err != nil {167		return nil, err168	}169	defer f.Close()170	section := f.Section(name)171	if section == nil {172		return nil, fmt.Errorf("no %s section", name)173	}174	return section.Data()175}176177// zstdDecode decompresses a whole zstd frame held in memory.178func zstdDecode(data []byte) ([]byte, error) {179	reader, err := zstd.NewReader(nil)180	if err != nil {181		return nil, err182	}183	defer reader.Close()184	return reader.DecodeAll(data, nil)185}186187// parseSoftdepPre reads the "pre" soft dependencies out of a .modinfo188// section. The section is a run of NUL-separated "key=value" strings.189// A soft dependency is a "softdep" key whose value lists names under190// "pre:" and "post:" markers, for example "pre: realtek post: foo".191// This collects the names under "pre:", in order, across every softdep192// string, and drops the "post:" names for the reason stated above.193func parseSoftdepPre(modinfo []byte) []string {194	var pre []string195	for entry := range bytes.SplitSeq(modinfo, []byte{0}) {196		key, value, ok := bytes.Cut(entry, []byte{'='})197		if !ok || string(key) != "softdep" {198			continue199		}200		bucket := ""201		for _, token := range strings.Fields(string(value)) {202			switch token {203			case "pre:":204				bucket = "pre"205			case "post:":206				bucket = "post"207			default:208				if bucket == "pre" {209					pre = append(pre, token)210				}211			}212		}213	}214	return pre215}
init/storage.go 86.9%
1package main23// This file actuates spec.storage.4//5// machine/storage.go documents the API side of the contract. On every6// boot, each declared role is either recognized or claimed. A role is7// recognized when a partition carries the GPT partition name that was8// written at claim time. Claiming happens exactly once. When a disk9// is blank, claiming writes a partition table to it, adds fresh10// filesystems, and writes the roles' names into it. Two rules make it11// safe to run this process on every machine, on every boot:12//13//   - Reconciling never destroys data. The process may only claim a14//     disk with no partition table and no filesystem: a disk that15//     neither liken nor anything else has written to before. It16//     refuses any disk it does not recognize, and it prints the17//     reason to the console.18//19//   - An unsatisfiable role stops the boot. If the process cannot20//     recognize or claim a declared role, the machine powers off. It21//     does not start k3s with that state only in RAM. A person can22//     repair a powered-off machine and boot it again. A cluster's23//     state can end up only in memory without warning, and that24//     state is lost at the next power cycle. (This rule does not25//     apply to roles that are absent from the spec. Their26//     directories stay on the RAM root.)27//28// This file owns the part of the process that runs on every boot:29// recognition, orchestration, and mounting. claim.go describes how30// the process claims a blank disk. grow.go describes how it grows a31// recognized partition.3233import (34	"errors"35	"fmt"36	"io/fs"37	"os"38	"path/filepath"39	"slices"40	"strconv"41	"strings"42	"time"4344	"golang.org/x/sys/unix"4546	"github.com/liken-sh/liken/liken/disks"47	"github.com/liken-sh/liken/liken/machine"48)4950// This comment lists where each role lands in the filesystem. liken51// defines this translation; the Machine API does not include it.52//53//	biosBoot         Not mounted. It is a raw partition that holds54//	                 GRUB's core image. The MBR's boot code reads55//	                 that image before any filesystem exists. liken56//	                 writes to it through the partition's device57//	                 node, never through a mount.58//	bootHome         /var/lib/liken/boot. It holds GRUB's config and59//	                 its environment block, the values a BIOS60//	                 machine uses in place of boot variables. It is61//	                 FAT32 because GRUB reads it with the same62//	                 driver that reads the slots.63//	systemA/systemB  /var/lib/liken/system/{a,b}. These are the64//	                 OS's own boot slots. They are FAT32 because the65//	                 firmware reads them. They allow no suid bit, no66//	                 device files, and no executables, because67//	                 nothing runs from a slot. The firmware only68//	                 loads what is in a slot.69//	machineState     /var/lib/liken/machine. It holds the machine's70//	                 own durable data, chiefly the staged and proven71//	                 manifests.72//	machineEphemeral /tmp, the OS's own scratch space. It disallows73//	                 suid and device files. It is world-writable74//	                 with the sticky bit set, the standard Unix75//	                 setup for /tmp.76//	clusterState     /var/lib/rancher. It holds all of k3s's state:77//	                 its database, its TLS material, and78//	                 containerd's images.79//	podStorage       The local-path provisioner's root directory.80//	                 The same path also appears in the image's81//	                 k3s config.yaml.82//	podEphemeral     kubelet's root directory: emptyDirs and pod83//	                 scratch space. The pod logs live here too, bound84//	                 onto /var/log/pods so that readers still find85//	                 them at the path Kubernetes uses (podlogs.go).86type roleMount struct {87	path   string88	flags  uintptr89	mode   os.FileMode // mode for the mounted root; 0 means the default mode90	fstype string      // "" means ext4, the default file system for data roles91}9293const slotMountFlags = unix.MS_NOSUID | unix.MS_NODEV | unix.MS_NOEXEC9495// bootHomeDir is the mount point for the bootHome role. It holds96// GRUB's config and environment block. init writes to this block97// when it arms or settles a slot trial on a BIOS machine. No role may98// mount on bootMountsDir or above it, because that tree holds the99// mounts this boot came from (switchroot.go).100const bootHomeDir = "/var/lib/liken/boot"101102// machineStateWritable says whether machineState is mounted for103// writing at this moment in the boot. failBoot consults it before it104// writes the fail-stop record, because a record written to the RAM105// root would go out with the power (failstop.go).106//107// The mount sets it, rather than a successful return from108// settleStorage, so that a boot which stops over some other role still109// records why. Roles mount in the canonical order, so a failure after110// machineState lands on disk and a failure before it does not.111var machineStateWritable bool112113var roleMounts = map[machine.StorageRoleName]roleMount{114	machine.BootHomeRole:         {path: bootHomeDir, flags: slotMountFlags, fstype: "vfat"},115	machine.SystemARole:          {path: machine.SystemSlotDir("A"), flags: slotMountFlags, fstype: "vfat"},116	machine.SystemBRole:          {path: machine.SystemSlotDir("B"), flags: slotMountFlags, fstype: "vfat"},117	machine.MachineStateRole:     {path: machine.MachineStateDir},118	machine.MachineEphemeralRole: {path: "/tmp", flags: unix.MS_NOSUID | unix.MS_NODEV, mode: tmpMode},119	machine.ClusterStateRole:     {path: "/var/lib/rancher"},120	machine.PodStorageRole:       {path: "/var/lib/liken/pod-storage"},121	machine.PodEphemeralRole:     {path: "/var/lib/kubelet"},122}123124// mountFilesystem and unmountFilesystem are the mount(2) and umount(2)125// calls that storage reconciliation, its teardown, the slot check, and126// the early slot mount make. They are package variables so a test can127// record which device lands on which path with which flags. A test128// process has no privilege to mount anything, and a mount that only129// fails would leave every decision after it untested.130var (131	mountFilesystem   = unix.Mount132	unmountFilesystem = unix.Unmount133)134135// isSystemSlot reports whether a role is one of the two system slots136// that the firmware reads. These two roles use FAT32. Each one is137// typed as an EFI system partition, so the firmware can find it. Each138// one has a fixed size from the day it is claimed, because FAT cannot139// grow in place.140func isSystemSlot(name machine.StorageRoleName) bool {141	return name == machine.SystemARole || name == machine.SystemBRole142}143144// isRawRole reports whether a role is a bare partition. The process145// recognizes, claims, and reports a bare partition like any other146// role, but it never formats or mounts one. biosBoot is the only147// bare-partition role. The MBR's boot code reads GRUB's core image148// long before any filesystem driver exists, so a filesystem there149// would only get in the way. liken writes to this partition through150// its device node.151func isRawRole(name machine.StorageRoleName) bool {152	return name == machine.BIOSBootRole153}154155// isFixedSizeRole reports whether a role's size is set on the day it156// is claimed. The FAT32 roles have a fixed size, because FAT cannot157// grow in place. biosBoot also has a fixed size: GRUB's boot code158// stores the core image's location as literal sector numbers, and the159// boot-sector healing rewrite depends on a layout that never changes.160func isFixedSizeRole(name machine.StorageRoleName) bool {161	return isSystemSlot(name) || name == machine.BootHomeRole || name == machine.BIOSBootRole162}163164// partitionTypeFor selects a role's GPT partition type. The system165// slots are EFI system partitions; the firmware finds them by this166// type GUID. biosBoot uses GRUB's own well-known type. Every other167// role uses the ordinary Linux data type.168func partitionTypeFor(name machine.StorageRoleName) [16]byte {169	switch {170	case isSystemSlot(name):171		return disks.EFISystemPartition172	case name == machine.BIOSBootRole:173		return disks.BIOSBootPartition174	}175	return disks.LinuxFilesystemData176}177178// teardownStorage unmounts everything that reconciliation may have179// mounted, in reverse canonical order. This returns the machine to180// the state that a fresh reconcile expects, so the process can try a181// different spec. teardownStorage works from the mount table, not182// from a status record, because a reconcile that failed partway183// through may have mounted a role that it never reported. When a path184// is not a mount point, unmounting it returns EINVAL; here that only185// means there was nothing to unmount. Nothing else runs this early in186// boot, so nothing can keep these mounts busy.187func teardownStorage() {188	// mountAndSeedClusterState (in k3s.go) mounts clusterState's file189	// system at the staging point for a short time while it seeds it.190	// If this step fails partway through, the file system stays191	// mounted there.192	_ = unmountFilesystem(clusterStateStaging, 0)193	unmountRoleMounts(0, true)194	// machineState is among the file systems that just came off, so195	// there is nowhere durable to write again until a later attempt196	// mounts it.197	machineStateWritable = false198}199200// unmountRoleMounts detaches every role file system in reverse201// canonical order, and anything stacked on one of them first. Two202// different shutdown paths share this function. Boot-time teardown203// unmounts each file system directly and reports any failure, because204// nothing else runs this early in boot, and a failed unmount here is205// useful information. The reboot path (reboot.go) passes MNT_DETACH206// and ignores errors: a container that was just killed can still hold207// its mount namespace open for a moment, which can pin a file system208// in place, and lazy detachment lets the kernel finish the unmount as209// those references clear, after the sync has already made the data210// safe. When a path is not a mount point, unmounting it returns211// EINVAL, which only means there was nothing to unmount.212func unmountRoleMounts(flags int, reportErrors bool) {213	// The pod-log bind is the one mount that is not a role and still214	// holds a role's file system open (podlogs.go). It comes off first,215	// so that podEphemeral's own unmount below releases the disk rather216	// than detaching a mount point that something still references.217	unmountPodLogs(flags, reportErrors)218	for _, name := range slices.Backward(machine.StorageRoleNames) {219		target := roleMounts[name].path220		if target == "" {221			continue // raw roles are never mounted222		}223		detachMount(target, flags, reportErrors)224	}225}226227// detachMount unmounts one path and reports the outcome the way the228// caller asked for. Every unmount on the way down goes through this229// function, so the console tells one story about what came apart.230func detachMount(target string, flags int, reportErrors bool) {231	err := unmountFilesystem(target, flags)232	switch {233	case err == nil:234		fmt.Printf("liken: storage: unmounted %s\n", target)235	case reportErrors && !errors.Is(err, unix.EINVAL) && !errors.Is(err, fs.ErrNotExist):236		fmt.Fprintf(os.Stderr, "liken: storage: unmounting %s: %v\n", target, err)237	}238}239240// partition is a partition as sysfs presents it: a subdirectory of241// its disk's /sys/block entry. The kernel parses the GPT and writes242// the name it reads into the partition's uevent file. Recognition243// reads that name from uevent, so it never has to re-read any244// partition table itself.245type partition struct {246	name      string // the kernel's node name, for example vda1 or nvme0n1p2247	disk      string // the parent disk's node name, for example vda or nvme0n1248	partName  string // the GPT partition name; "" if the table has no names249	sizeBytes uint64250}251252func discoverPartitions() []partition {253	var parts []partition254	for _, disk := range discoverBlockDevices() {255		dir := filepath.Join(sysBlock, disk.Name)256		entries, err := os.ReadDir(dir)257		if err != nil {258			continue259		}260		for _, entry := range entries {261			// A disk's partitions appear as subdirectories named after262			// their device; for example, vda gives vda1. The `partition`263			// file inside each one distinguishes it from the disk's264			// other attribute directories.265			if _, err := os.Stat(filepath.Join(dir, entry.Name(), "partition")); err != nil {266				continue267			}268			p := partition{name: entry.Name(), disk: disk.Name}269			if raw, err := os.ReadFile(filepath.Join(dir, entry.Name(), "size")); err == nil {270				if sectors, err := strconv.ParseUint(strings.TrimSpace(string(raw)), 10, 64); err == nil {271					p.sizeBytes = sectors * disks.SectorSize272				}273			}274			// The uevent file holds KEY=value lines. PARTNAME275			// appears only in tables that carry names, and GPT is276			// such a table.277			if raw, err := os.ReadFile(filepath.Join(dir, entry.Name(), "uevent")); err == nil {278				for line := range strings.SplitSeq(strings.TrimSpace(string(raw)), "\n") {279					if v, ok := strings.CutPrefix(line, "PARTNAME="); ok {280						p.partName = v281					}282				}283			}284			parts = append(parts, p)285		}286	}287	return parts288}289290// reconcileStorage actuates the storage spec. On success, every291// declared role becomes a file system mounted at its role's path:292// ext4 for the data roles, FAT32 for the system slots that the293// firmware reads. The returned status records where each role294// landed. The status carries the same facts that the process prints295// to the console, and it goes into the Machine's status, because296// anything reported only to the serial port stays invisible to297// anyone who operates the machine remotely. An error means the298// process cannot satisfy a declared role. The caller (main, the only299// place with the authority to do so) then stops the boot instead of300// letting k3s start with that state only in RAM.301//302// The function plans everything, then applies the plan. It computes303// every claim's layout and every growth's table edit before it304// writes the first byte to any disk. A spec that will fail must fail305// before it changes anything on disk, because the boot may go on to306// try a different spec (for example, the proven manifest, after the307// process rejects a staged one). Partitions half-created under the308// failed spec would break that later attempt too. Planning cannot309// prevent one problem: a genuine I/O failure partway through a write.310// A disk claimed by a failed attempt stays claimed. If that leaves311// two partitions carrying the same role's name, recognition refuses312// to guess between them, and the boot stops.313func reconcileStorage(spec machine.StorageSpec) (machine.StorageStatus, error) {314	status := machine.AllRolesInMemory()315	roles := spec.Roles()316	if len(roles) == 0 {317		return status, nil318	}319	if err := spec.Validate(); err != nil {320		return status, err321	}322323	// Storage settles less than a second into boot, and on real324	// hardware that is the middle of the bus probe: a SATA link325	// trains and a USB device negotiates for seconds after their326	// drivers load. A declared disk that has not appeared yet is327	// indistinguishable from one that does not exist, so the spec328	// gets a bounded window to become satisfiable before any of it329	// is judged.330	awaitStorageDevices(roles)331332	// Recognition finds each declared role by the name written on its333	// partition. It does not consult the device listed in the spec. A334	// disk that has moved to a different controller since it was335	// claimed is still the same disk.336	found, err := recognizeRoles(roles)337	if err != nil {338		return status, err339	}340341	// Plan the claims. Any role that is still missing must point at a342	// blank disk. A role that is already found still matters here: if343	// its device names a disk some other role still needs, that disk344	// is not blank, and planAllClaims refuses it.345	claims, err := planAllClaims(roles, found)346	if err != nil {347		return status, err348	}349350	// Plan the growth. A recognized partition may be smaller than the351	// spec now declares, or its disk may have grown underneath it.352	// grow.go explains the rules for growth.353	grows, err := planAllGrowth(roles, found)354	if err != nil {355		return status, err356	}357358	// Apply the plans. The process has already validated every plan,359	// so any failure from this point on is a real I/O problem. The360	// roles a claim creates are remembered, because a partition this361	// boot created always gets a new filesystem.362	created := map[machine.StorageRoleName]bool{}363	for _, plan := range claims {364		if err := applyClaim(plan); err != nil {365			return status, err366		}367		for _, role := range plan.roles {368			created[role.Name] = true369		}370	}371	for _, plan := range grows {372		if err := applyGrowth(plan); err != nil {373			return status, err374		}375	}376	if len(claims) > 0 || len(grows) > 0 {377		if found, err = recognizeRoles(roles); err != nil {378			return status, err379		}380		for _, role := range roles {381			if _, ok := found[role.Name]; !ok {382				return status, fmt.Errorf("role %s: partition %s did not appear after claiming %s",383					role.Name, role.PartitionName(), role.Device)384			}385		}386	}387388	for _, role := range roles {389		p := found[role.Name]390		// A FAT role is asked about its last stop before it is391		// mounted. Mounting a FAT volume for writing sets the mark392		// that answers the question, so after the mount there is393		// nothing left to read (fatstate.go).394		var unclean bool395		if roleMounts[role.Name].fstype == "vfat" {396			unclean = readFATStop(role.Name, devRoot+"/"+p.name, created[role.Name])397		}398		// A raw role needs nothing more once it is recognized: there399		// is no file system to make and no mount point to serve. The400		// status still records where the partition landed, because a401		// fact such as which device holds the boot code must not402		// live only on a serial console.403		if isRawRole(role.Name) {404			fmt.Printf("liken: storage: %s is %s/%s (%s), raw\n",405				role.Name, devRoot, p.name, p.partName)406		} else if err := mountRole(role, p, created[role.Name]); err != nil {407			return status, err408		}409		*status.Role(role.Name) = machine.StorageRoleStatus{410			Backing:         machine.BackingPartition,411			Device:          p.name,412			Partition:       p.partName,413			CapacityBytes:   p.sizeBytes,414			LastStopUnclean: unclean,415		}416	}417	return status, nil418}419420// recognizeRoles matches the declared roles against the machine's421// partitions, as sysfs reports them.422func recognizeRoles(roles []machine.DeclaredRole) (map[machine.StorageRoleName]partition, error) {423	return matchRoles(roles, discoverPartitions())424}425426// matchRoles matches declared roles to partitions by name. When two427// partitions carry the same role name, a disk was usually cloned or428// moved from another machine. A wrong guess about which partition429// holds the real cluster would destroy data, so this ambiguity is an430// error, not a choice to make.431func matchRoles(roles []machine.DeclaredRole, parts []partition) (map[machine.StorageRoleName]partition, error) {432	found := map[machine.StorageRoleName]partition{}433	for _, role := range roles {434		for _, p := range parts {435			if p.partName != role.PartitionName() {436				continue437			}438			if existing, ok := found[role.Name]; ok {439				return nil, fmt.Errorf("two partitions claim to be %s (%s and %s); refusing to guess",440					role.PartitionName(), existing.name, p.name)441			}442			found[role.Name] = p443		}444	}445	return found, nil446}447448func diskByPath(device string) *machine.BlockDevice {449	for _, d := range discoverBlockDevices() {450		if devicePath(d) == device {451			return &d452		}453	}454	return nil455}456457// awaitStorageDevices waits, boundedly, for the spec to become458// satisfiable. Recognition finds a role by the name written on its459// partition, never by the device the spec declares, so this only has460// to resolve the device of a role recognition has not already461// satisfied. An already-recognized role's own device can be stale, or462// even ambiguous, without meaning anything, because nothing is about463// to be claimed for it. With a stable name, the wait is for the named464// disk itself to attach, not for a kernel letter to appear: the465// letter a disk happens to get this boot plays no part in resolving466// the name.467//468// An error from either recognition or resolution ends the wait at469// once, rather than running out the deadline: waiting cannot fix an470// ambiguity, and more disks attaching can only make one worse. Two471// partitions already carrying one role's name stay two partitions472// carrying that name however long the code waits, the same as a473// declared name that already matches two disks stays matching two474// disks: a third disk attaching cannot un-match the two that already475// do. Either error names a role that recognition has not yet476// satisfied; reconcileStorage's own call to recognizeRoles, or477// planAllClaims's own call to resolveDeclaredDisk, meets the same478// ambiguity moments later, so this function only has to stop waiting479// and say why, not resolve the ambiguity itself. A spec that is still480// unsatisfiable at the deadline proceeds to judgment, and the481// ordinary errors report what is missing. Polling, rather than482// uevents, keeps this simple: the window is short, boots are rare,483// and a walk of /sys/block costs nothing.484func awaitStorageDevices(roles []machine.DeclaredRole) {485	const (486		poll     = 500 * time.Millisecond487		deadline = 30 * time.Second488	)489	ready := func() (bool, error) {490		found, err := recognizeRoles(roles)491		if err != nil {492			return false, err493		}494		for _, role := range roles {495			if _, ok := found[role.Name]; ok {496				continue497			}498			disk, err := resolveDeclaredDisk(role.Device)499			if err != nil {500				return false, err501			}502			if disk == nil {503				return false, nil504			}505		}506		return true, nil507	}508509	ok, err := ready()510	if err != nil {511		fmt.Fprintf(os.Stderr, "liken: storage: %v\n", err)512		return513	}514	if ok {515		return516	}517	fmt.Println("liken: storage: waiting for the declared disks to attach")518	for begin := time.Now(); time.Since(begin) < deadline; {519		time.Sleep(poll)520		ok, err := ready()521		if err != nil {522			fmt.Fprintf(os.Stderr, "liken: storage: %v\n", err)523			return524		}525		if ok {526			fmt.Println("liken: storage: the declared disks attached")527			return528		}529	}530	fmt.Fprintf(os.Stderr, "liken: storage: the declared disks did not all attach within %s\n", deadline)531}532533// needsFilesystem reports whether a partition gets a new file system534// before it is mounted. Two conditions call for one, and they are535// different questions.536//537// A partition this boot created is always made fresh. Blanking a538// disk's partition table does not touch the partitions, and a539// reinstall writes the same layout back at the same offsets, so the540// old file systems are still there, superblock and all. Keeping them541// would carry the previous install's etcd database, its proven542// manifest, and its node password into the install that was meant to543// replace them. A partition liken created seconds ago holds nothing544// worth keeping, whatever the bytes under it say.545//546// A partition liken recognized, carrying no file system, is the other547// condition: a boot that died between partitioning and mkfs. Claiming548// writes the role's name first, so the next boot finishes the job.549func needsFilesystem(dev, fstype string, created bool) bool {550	if created {551		return true552	}553	if fstype == "vfat" {554		return !disks.HasFAT32(dev)555	}556	return !hasExt4(dev)557}558559// mountRole mounts a role's file system at the role's path, and makes560// the file system first when the partition needs one. The created561// argument says whether this boot claimed the partition.562func mountRole(role machine.DeclaredRole, p partition, created bool) error {563	// The process looks up the role's translation to a mount before564	// anything touches the partition. It must refuse a role with no565	// mount translation, before mke2fs writes a file system566	// onto it.567	rm, ok := roleMounts[role.Name]568	if !ok {569		return fmt.Errorf("role %s has no mount translation; liken and its manifest disagree about the role vocabulary", role.Name)570	}571	target := rm.path572573	// Each file system type has its own maker. The system slots get574	// FAT32 from liken's own formatter (fat32.go), because the575	// firmware reads them and FAT is the only file system it reads.576	// Every other role gets ext4 from the vendored static mke2fs.577	dev := devRoot + "/" + p.name578	if needsFilesystem(dev, rm.fstype, created) {579		if rm.fstype == "vfat" {580			fmt.Printf("liken: storage: making a FAT32 filesystem on %s for %s\n", dev, role.Name)581			if err := formatSlot(dev, p.sizeBytes, role.Name); err != nil {582				return fmt.Errorf("formatting %s for %s: %w", dev, role.Name, err)583			}584		} else {585			fmt.Printf("liken: storage: making an ext4 filesystem on %s for %s\n", dev, role.Name)586			if !runNarrated("mke2fs | ", "/sbin/mke2fs", "-t", "ext4", dev) {587				return fmt.Errorf("mke2fs on %s for %s failed", dev, role.Name)588			}589		}590	}591592	// clusterState mounts through its own path. The image bakes k3s's593	// seed files in underneath its mount point. Layering those seed594	// files in belongs to k3s, not to partition mechanics; k3s.go595	// handles it.596	if role.Name == machine.ClusterStateRole {597		if err := mountAndSeedClusterState(dev, target); err != nil {598			return err599		}600	} else {601		fstype := rm.fstype602		if fstype == "" {603			fstype = "ext4"604		}605		if err := os.MkdirAll(target, 0o755); err != nil {606			return fmt.Errorf("mkdir %s: %w", target, err)607		}608		if err := mountFilesystem(dev, target, fstype, rm.flags, ""); err != nil {609			return fmt.Errorf("mounting %s for %s: %w", dev, role.Name, err)610		}611	}612	// A partition that grew this boot still carries a file system613	// sized for its old extent. Now that the file system is mounted,614	// ext4 can grow online to fill it (ext4.go explains why a mounted615	// file system is the easy case). FAT cannot grow in place, and616	// that is expected: slots have a fixed size by design, and617	// planAllGrowth refuses to grow them at the planning stage.618	if rm.fstype == "" {619		if err := maybeGrowFilesystem(role, p, target); err != nil {620			return err621		}622	}623624	// A freshly made ext4 root has mode 0755, root-only. Roles such as625	// /tmp need their conventional permissions applied to the mounted626	// root.627	if rm.mode != 0 {628		if err := os.Chmod(target, rm.mode); err != nil {629			fmt.Fprintf(os.Stderr, "liken: storage: chmod %s: %v\n", target, err)630		}631	}632	if role.Name == machine.MachineStateRole {633		machineStateWritable = true634	}635	fmt.Printf("liken: storage: %s is %s (%s) on %s\n",636		role.Name, dev, p.partName, target)637	return nil638}
init/supervisor.go 55.6%
1package main23// Supervising k3s.4//5// This file is the complete service manager for this OS. It starts6// one process, and starts that process again when it dies. Kubernetes7// manages everything a traditional init manages above that level:8// order, dependencies, sockets, and timers. Kubernetes is the process9// under supervision here.10//11// One problem here is subtle and worth understanding, because it is12// the classic bug in every homemade PID 1. As PID 1, this program13// must reap every dead process on the machine, because orphan14// processes reparent to PID 1. So a loop somewhere calls wait(-1),15// which collects the status of any exited child. But the supervisor16// also requires the exit status of the specific child it started.17// wait(-1) in one goroutine races wait(pid) in another goroutine. The18// call that collects the status first consumes it. The other call19// gets an error instead. The fix is a single authority. Only the20// reaper calls wait. Every other part of the code subscribes to the21// reaper. The reaper posts every exit status it collects. When22// nobody has claimed an exit yet, the status stays parked, because a23// child can die before its parent asks for it. The registry's await24// method reads a parked status, or it waits until a status arrives.2526import (27	"bytes"28	"context"29	"fmt"30	"io"31	"os"32	"os/exec"33	"os/signal"34	"path/filepath"35	"strings"36	"sync"37	"time"3839	"golang.org/x/sys/unix"4041	"github.com/liken-sh/liken/liken/api"42	"github.com/liken-sh/liken/liken/cluster"43	"github.com/liken-sh/liken/liken/machine"44)4546// lineWriter forwards each complete line it receives to its47// destination, with a prefix added. Child processes write output in48// chunks of arbitrary size. Buffering the output to line boundaries49// stops the child's output and liken's own messages from50// interleaving in the middle of a line. The destination is the raw51// console, not init's kmsg-routed stdout. k3s produces enough output52// to fill the kernel's small ring buffer within seconds. k3s's lines53// already reach the cluster through the log file that the k3s log54// relay tails, so buffering the echo there too would send everything55// twice.56type lineWriter struct {57	dest   io.Writer58	prefix string59	buf    bytes.Buffer60}6162// teeOutput sends a command's output to its log file and to the63// console. Both streams share one writer, because os/exec copies two64// different writers from two goroutines, and the capped log has no65// lock: a rotation under one copier would close the file under the66// other. With one comparable writer, os/exec gives both streams one67// pipe and one copier.68func teeOutput(cmd *exec.Cmd, logf, console io.Writer) {69	out := io.MultiWriter(logf, &lineWriter{dest: console, prefix: "k3s | "})70	cmd.Stdout = out71	cmd.Stderr = out72}7374func (w *lineWriter) Write(p []byte) (int, error) {75	w.buf.Write(p)76	for {77		line, err := w.buf.ReadString('\n')78		if err != nil {79			// No newline has arrived yet. Put the partial line back in80			// the buffer and wait.81			w.buf.WriteString(line)82			break83		}84		fmt.Fprintf(w.dest, "%s%s", w.prefix, line)85	}86	return len(p), nil87}8889// reap collects the exit status of any child process for as long as90// the machine plane runs. The plane stops only at shutdown, after91// every process it might collect has already stopped. SIGCHLD92// arrives whenever a child dies. Signal coalescing can fold many93// deaths into one delivery, so each wakeup collects every exited94// child, not just one. This loop is the only place in liken that95// calls wait. It posts every exit status it collects to the death96// registry below, where whoever started the process can claim the97// status.98//99// The loop subscribes to SIGCHLD first and then collects once before100// it waits for a signal. A child that exits before the subscription101// sends its SIGCHLD to nobody, and that first collection is what102// reaps it.103// (Go note: signal.Notify registers a handler with the runtime, and104// forwards deliveries onto a channel. This turns an asynchronous105// interrupt into an ordinary receive loop, and satisfies the "PID 1106// must install handlers" rule from main.go's header comment.)107func reap(ctx context.Context) error {108	sigchld := make(chan os.Signal, 1)109	signal.Notify(sigchld, unix.SIGCHLD)110	defer signal.Stop(sigchld)111	for {112		for {113			// -1 means "any child". WNOHANG means "do not block if none114			// have exited"; in that case, Wait4 returns pid 0.115			var status unix.WaitStatus116			pid, err := unix.Wait4(-1, &status, unix.WNOHANG, nil)117			if pid <= 0 || err != nil {118				break119			}120			deaths.record(pid, status)121		}122		select {123		case <-ctx.Done():124			return nil125		case <-sigchld:126		}127	}128}129130// deathRegistry connects the reaper (the only place wait() happens)131// to every part of the code that awaits an exit. The reaper records132// each death, and a waiter either reads a status already parked, or133// leaves a channel open to be filled later. deathRegistry is one134// value with methods, the same pattern machinePlane uses to135// encapsulate the other half of init's shared state.136//137// The registry parks a status only for a child that init started138// through start and has not yet awaited. As PID 1, init also adopts139// every orphan on the machine, such as a containerd shim, and the140// reaper collects those too. Nobody awaits an orphan, so a parked141// status for one would stay forever, and a later child that reuses142// the pid would read it at once as its own death.143type deathRegistry struct {144	mu        sync.Mutex145	waiters   map[int]chan unix.WaitStatus146	unclaimed map[int]unix.WaitStatus147	expected  map[int]bool148}149150var deaths = &deathRegistry{151	waiters:   map[int]chan unix.WaitStatus{},152	unclaimed: map[int]unix.WaitStatus{},153	expected:  map[int]bool{},154}155156// start starts a command and marks its pid as one whose death to157// keep. It holds the lock across the start, because a child can exit158// and be reaped before cmd.Start returns: the reaper's record then159// waits for the lock, and finds the pid already expected.160func (d *deathRegistry) start(cmd *exec.Cmd) error {161	d.mu.Lock()162	defer d.mu.Unlock()163	if err := cmd.Start(); err != nil {164		return err165	}166	d.expected[cmd.Process.Pid] = true167	return nil168}169170// expect marks a pid as one whose death to keep, for a process that171// start did not start.172func (d *deathRegistry) expect(pid int) {173	d.mu.Lock()174	defer d.mu.Unlock()175	d.expected[pid] = true176}177178// record stores the exit status that the reaper collects for each179// process: it wakes the waiter, parks the status for a child that180// init started, and drops the status of an adopted orphan.181func (d *deathRegistry) record(pid int, status unix.WaitStatus) {182	d.mu.Lock()183	defer d.mu.Unlock()184	if ch, ok := d.waiters[pid]; ok {185		ch <- status186		delete(d.waiters, pid)187		delete(d.expected, pid)188	} else if d.expected[pid] {189		d.unclaimed[pid] = status190	}191}192193// await waits until the reaper has collected the given pid.194func (d *deathRegistry) await(pid int) unix.WaitStatus {195	d.mu.Lock()196	if status, ok := d.unclaimed[pid]; ok {197		delete(d.unclaimed, pid)198		delete(d.expected, pid)199		d.mu.Unlock()200		return status201	}202	ch := make(chan unix.WaitStatus, 1)203	d.waiters[pid] = ch204	d.mu.Unlock()205	return <-ch206}207208const (209	k3sBinary = "/bin/k3s"210211	// The k3s log lives on clusterState, in a directory that clearly212	// belongs to liken, so nothing can mistake it for a file that k3s213	// manages. When the machine has a persistent disk, storing the214	// log there lets the log survive the boot that wrote it, and215	// gives the log relay's mount a stable path. containerd chooses216	// its own log path on the same filesystem. init touches that log217	// only at rotation time (logrotate.go).218	likenLogDir   = "/var/lib/rancher/k3s/liken"219	k3sLog        = likenLogDir + "/k3s.log"220	containerdLog = "/var/lib/rancher/k3s/agent/containerd/containerd.log"221)222223// postMortem runs at the end of a one-shot boot. There is no shell to224// investigate from, so postMortem prints the facts an investigator225// would need: the environment that child processes inherited, and226// whether the tools they need resolve and run correctly.227func postMortem() {228	fmt.Printf("liken: post-mortem: init PATH=%s\n", os.Getenv("PATH"))229	resolved, err := filepath.EvalSymlinks("/sbin/iptables")230	if err != nil {231		fmt.Printf("liken: post-mortem: /sbin/iptables: %v\n", err)232	} else {233		fmt.Printf("liken: post-mortem: /sbin/iptables -> %s\n", resolved)234	}235	if out, ok := run("iptables", "-V"); ok {236		fmt.Printf("liken: post-mortem: iptables -V: %s\n", out)237	} else {238		fmt.Printf("liken: post-mortem: iptables -V failed: %q\n", out)239	}240}241242// superviseK3s runs k3s forever, and honors two kinds of243// interruption: a reboot request from the operator, and a restart244// request. A restart request is the lighter disruption: it applies245// staged restart-class changes (cluster/changes.go) by bouncing the246// k3s child process in place. superviseK3s never returns for any247// other reason. Whenever k3s exits on its own, superviseK3s restarts248// it, with backoff, so a fast crash loop does not flood the console.249//250// The code selects on the intent channels in *both* states of the251// supervisor: while k3s runs, and during the backoff sleep between252// restarts. That second select stops a request from racing the253// restart decision. Both selects are alternatives within one select254// statement in one goroutine, so an intent that arrives while k3s is255// crash-looping does not wait out the sleep, and does not collide256// with a restart.257//258// A deliberate bounce is not a crash. applyRestart runs while k3s259// still serves traffic, so all the re-rendering happens before any260// downtime, and applyRestart reports whether it actually applied261// anything. Only then does superviseK3s stop k3s gracefully and262// start it again immediately. This skips the oneshot check and the263// backoff entirely, because both apply to k3s failures, not to liken264// decisions. Stopping k3s does not stop its containers, because the265// containerd shims hold them. This is the entire basis for the266// restart tier: the machine and its pods stay up while k3s reloads267// the configuration that k3s only reads at start.268//269// The liken.oneshot boot parameter disables the crash restart. k3s270// runs once, and its exit powers the machine down. This makes a k3s271// failure visible from outside the machine: QEMU exits, and the272// console holds a complete record. Debugging and automated test runs273// need this visibility on a machine with no shell. superviseK3s still274// honors a reboot intent in oneshot mode. Under QEMU's -no-reboot275// flag, the restart is a clean exit, exactly what a bounded harness276// run needs.277// afterStop runs each time a restart has stopped k3s, before the278// next start. It exists for the retractions that must not happen279// while k3s runs: a janitor-teardown feature's seeded manifests are280// removed here, in the window where k3s is down, so no addon pass281// runs against the removal (retractFeatureManifests explains why a282// deletion by k3s would be dangerous for flux).283func superviseK3s(role api.Role, reboot <-chan machine.RebootIntent,284	restarts <-chan machine.RestartIntent, applyRestart func(machine.RestartIntent) bool,285	afterStop func()) {286	backoff := time.Second287	for {288		started := time.Now()289		bounced := false290		cmd, logf, err := startK3s(role)291		if err != nil {292			fmt.Fprintf(os.Stderr, "liken: k3s: %v\n", err)293		} else {294			// The reaper is the only waiter (see the file comment).295			// This goroutine only carries the reaper's answer into296			// the select statement.297			died := make(chan unix.WaitStatus, 1)298			go func() { died <- deaths.await(cmd.Process.Pid) }()299300		running:301			for {302				select {303				case status := <-died:304					_ = cmd.Process.Release()305					logf.Close()306					fmt.Printf("liken: k3s exited (%s)\n", describeExit(status))307					break running308				case intent := <-reboot:309					stopK3s(cmd.Process, died)310					_ = cmd.Process.Release()311					// This close is part of the shutdown order, not312					// cleanup. The log lives on clusterState, which313					// rebootMachine is about to unmount. An open file314					// handle there would make that unmount fail as busy.315					logf.Close()316					rebootMachine(intent) // never returns317				case intent := <-restarts:318					// A stale or duplicate intent applies nothing. A319					// running k3s is not disturbed because of it.320					if !applyRestart(intent) {321						continue running322					}323					fmt.Println("liken: restarting k3s to apply the staged changes")324					stopK3s(cmd.Process, died)325					afterStop()326					_ = cmd.Process.Release()327					logf.Close()328					bounced = true329					break running330				}331			}332		}333		if bounced {334			continue // goes straight back to startK3s: a bounce is not a crash335		}336		if bootParam("liken.oneshot") {337			postMortem()338			fmt.Println("liken: one-shot boot, not restarting k3s; powering off")339			powerOff()340			return341		}342343		// A k3s that ran for a while resets the backoff. One that died344		// immediately doubles the backoff. The backoff is capped, so a345		// truly broken configuration still retries every half minute.346		if time.Since(started) > time.Minute {347			backoff = time.Second348		} else if backoff < 30*time.Second {349			backoff *= 2350		}351		delay := withJitter(backoff)352		fmt.Printf("liken: restarting k3s in %s\n", delay.Round(time.Millisecond))353		select {354		case <-time.After(delay):355		case intent := <-reboot:356			rebootMachine(intent) // k3s is already dead; nothing to stop357		case intent := <-restarts:358			// k3s is already down. Apply the staged changes now, and359			// skip the rest of the delay, because the next start360			// reads the changes either way.361			_ = applyRestart(intent)362			afterStop()363		}364	}365}366367// k3sRuntimeEnv is the Go runtime discipline that init hands the k3s368// process. The cluster document's spec.runtime.k3s section sets it,369// and init resolves that section against this machine's memory370// (cluster/runtime.go). The section is an opt-in: an unset field adds371// no variable, so the returned environment carries only what the372// cluster names, and an unset section returns nothing at all. k3s then373// runs on Go's own defaults for whatever the cluster left alone.374//375// GOMEMLIMIT is a soft ceiling on everything the runtime manages: heap,376// stacks, and its own metadata. As memory use approaches the ceiling,377// the collector runs harder instead of letting the heap grow. Past the378// ceiling, the runtime caps collection at half the process's CPU and379// lets the heap grow anyway, so a genuine memory spike degrades into380// slowness rather than a heap-exhaustion crash. GOGC sets the everyday381// pace under the ceiling, as a percent of heap growth between382// collections.383//384// init sets the environment when it launches the process, because Go385// reads both variables only at startup. containerd and the shims k3s386// starts inherit this environment, because k3s is their parent. This is387// deliberate and cheap: they are Go programs far smaller than the388// ceiling, so it never constrains them, and the collector pace keeps389// them lean too. Workload processes inherit nothing; their environments390// come from their pod specs.391func k3sRuntimeEnv(spec cluster.K3sRuntimeSpec, memoryBytes uint64) []string {392	var env []string393	if limit, off, err := spec.GoMemoryLimitBytes(memoryBytes); err == nil && !off {394		env = append(env, fmt.Sprintf("GOMEMLIMIT=%dMiB", limit/(1<<20)))395	}396	if gc, ok := spec.GoGCPercent(); ok {397		env = append(env, fmt.Sprintf("GOGC=%d", gc))398	}399	return env400}401402// k3sMemoryDiscipline is the runtime environment that startK3s gives403// every k3s process it launches. writeK3sBootConfig derives this404// environment beside the boot drop-in, at boot and again on every405// applied restart. So a restart that edits spec.runtime.k3s re-resolves406// the environment on the same bounce that reconfigures k3s. An unset407// section leaves the list empty, and startK3s then launches k3s with408// init's own environment untouched.409var k3sMemoryDiscipline []string410411// startK3s launches k3s in the machine's role, and returns the412// running command and its log file. The log file must stay open as413// long as the process writes to it. The console copy flows through414// an in-process pipe.415func startK3s(role api.Role) (*exec.Cmd, io.Closer, error) {416	// k3s's output goes to two places: a file, and the console. On417	// the console, the output arrives live, line-buffered, and418	// prefixed, so a reader can tell it apart from liken's own419	// messages. On a machine with no shell, the console is the only420	// way to read a log. The file is what the k3s log relay tails421	// into the cluster. The file writer caps a boot's log, so a k3s422	// that writes a lot of output cannot fill the filesystem it423	// shares with etcd (logrotate.go).424	logf, err := openCappedLog(k3sLog, k3sLogCap)425	if err != nil {426		return nil, nil, err427	}428429	// This is the one place where liken's role vocabulary meets430	// k3s's own vocabulary: a leader runs `k3s server`, and a431	// follower runs `k3s agent` (cluster/cluster.go). Configuration432	// lives in files, not in flags. A leader's k3s reads433	// /etc/rancher/k3s/config.yaml on its own. A follower's k3s is434	// pointed at its own file, because the leader-only config file435	// would otherwise be misread as unknown flags. k3s.go joins both436	// config files with this boot's derived drop-in before this code437	// runs.438	args := []string{"server"}439	if role == api.RoleFollower {440		args = []string{"agent", "--config", k3sAgentConfig}441	}442	cmd := exec.Command(k3sBinary, args...)443	if len(k3sMemoryDiscipline) > 0 {444		cmd.Env = append(os.Environ(), k3sMemoryDiscipline...)445	}446	teeOutput(cmd, logf, console)447	if err := deaths.start(cmd); err != nil {448		logf.Close()449		return nil, nil, fmt.Errorf("starting k3s: %w", err)450	}451	fmt.Printf("liken: k3s %s started (pid %d), logs in %s\n", role, cmd.Process.Pid, k3sLog)452	return cmd, logf, nil453}454455// stopK3s asks k3s to exit, and waits for the reaper to confirm the456// exit. If k3s takes too long to exit, stopK3s escalates to SIGKILL.457// stopK3s only sends the signal and receives the confirmation. The458// reaper stays the sole authority on calling wait (the file comment's459// one rule).460//461// The signals go through the os.Process, not through the pid. The462// Process holds a pidfd, and Go signals through pidfd_send_signal, so463// a signal to a process that the reaper already collected returns464// os.ErrProcessDone. A pid can be reused once the reaper collects465// it, and a pidfd cannot.466func stopK3s(p *os.Process, died <-chan unix.WaitStatus) {467	fmt.Printf("liken: stopping k3s (pid %d)\n", p.Pid)468	_ = p.Signal(unix.SIGTERM)469	select {470	case status := <-died:471		fmt.Printf("liken: k3s exited (%s)\n", describeExit(status))472	case <-time.After(30 * time.Second):473		fmt.Fprintln(os.Stderr, "liken: k3s ignored SIGTERM for 30s; killing it")474		_ = p.Kill()475		fmt.Printf("liken: k3s exited (%s)\n", describeExit(<-died))476	}477}478479func describeExit(status unix.WaitStatus) string {480	switch {481	case status.Exited():482		return fmt.Sprintf("status %d", status.ExitStatus())483	case status.Signaled():484		return fmt.Sprintf("signal %s", status.Signal())485	default:486		return fmt.Sprintf("wait status %#x", uint32(status))487	}488}489490// reportWhenReady watches for the moment the machine becomes a491// working Kubernetes node. It polls `k3s kubectl get nodes`, which492// reads the admin kubeconfig that k3s writes once its API starts493// serving. reportWhenReady prints the node's status as the status494// changes: registering, NotReady, and finally Ready. reportWhenReady495// is a machine-plane component. Its work completes once it prints496// the report, so every exit path returns nil.497//498// This reporter, and reportPods below, is the one resident of the499// machine plane that k3s does not depend on. The two-planes rule500// (components.go) would normally send a component like this into the501// cluster. It stays in the machine plane by deliberate exception. Its502// entire subject is the window before the cluster can report on503// itself: while k3s is starting, and before the operator pod exists.504// On a machine with no shell, the console is the only place that can505// show this information. The operator takes over reporting the506// moment the operator runs. reportWhenReady is the bridge to that507// moment.508func reportWhenReady(ctx context.Context) error {509	fetch := func(timeout time.Duration) (string, bool) {510		return runWithin(timeout, k3sBinary, kubectlGet(timeout, "nodes")...)511	}512	if pollAndReport(ctx, 3*time.Second, 5*time.Minute, "node", fetch, containsReady) {513		fmt.Println("liken: kubernetes is up")514		reportPods(ctx)515	} else if ctx.Err() == nil {516		fmt.Println("liken: gave up waiting for the node to be Ready (k3s may still get there)")517	}518	return nil519}520521// reportPods prints the system pods as they start, after the node522// goes Ready. It stops printing once every pod reaches Running or523// Completed, or when five minutes pass. This is the console524// equivalent of watching `kubectl get pods -A` until the output525// settles.526func reportPods(ctx context.Context) {527	fetch := func(timeout time.Duration) (string, bool) {528		return runWithin(timeout, k3sBinary, kubectlGet(timeout, "pods", "-A")...)529	}530	if pollAndReport(ctx, 5*time.Second, 5*time.Minute, "pod", fetch, podsSettled) {531		fmt.Println("liken: all system pods are settled")532	} else if ctx.Err() == nil {533		fmt.Println("liken: system pods have not settled; see the pod status lines above")534	}535}536537// kubectlCallTimeout bounds one kubectl call from the boot538// reporters. An API server that accepts the connection and never539// answers would otherwise hold kubectl, and the reporter with it,540// past its patience, because kubectl sets no request timeout of its541// own. A starting k3s answers a node or pod list in well under a542// second, so ten seconds is room for a slow start, and a server that543// does not answer in ten seconds costs one skipped table, not the544// whole wait.545const kubectlCallTimeout = 10 * time.Second546547// kubectlGet builds the arguments for one `k3s kubectl get` with a548// request timeout. The flag bounds each HTTP request, not the whole549// run: kubectl with no discovery cache retries its discovery request550// several times, and each retry waits the full timeout. runWithin551// bounds the run. The flag still ends a hung request early, so a552// retry can reach a server that has started to answer. The timeout553// is always positive, because kubectl reads a zero --request-timeout554// as no timeout at all.555func kubectlGet(timeout time.Duration, args ...string) []string {556	timeout = max(timeout.Round(time.Millisecond), time.Millisecond)557	out := []string{"kubectl", "get"}558	out = append(out, args...)559	return append(out, "--no-headers", "--request-timeout="+timeout.String())560}561562// pollAndReport is the pattern both reporters share: fetch a kubectl563// table on an interval, print the table under the prefix whenever it564// changes, and return true the moment the table satisfies settled.565// pollAndReport returns false when patience runs out, or when the566// plane shuts down. Printing only the changes keeps the console567// readable: a table that stays unchanged for a minute produces no568// lines at all.569//570// Each fetch gets kubectlCallTimeout or the patience that is left,571// whichever is shorter, and the fetch kills kubectl when that time572// passes (runWithin). So a server that never answers holds the loop573// no longer than its patience plus one interval, the last sleep.574func pollAndReport(ctx context.Context, interval, patience time.Duration, prefix string,575	fetch func(timeout time.Duration) (string, bool), settled func(string) bool) bool {576	last := ""577	deadline := time.Now().Add(patience)578	for time.Now().Before(deadline) {579		if !sleepUnlessCancelled(ctx, interval) {580			return false581		}582		left := time.Until(deadline)583		if left <= 0 {584			return false585		}586		out, ok := fetch(min(kubectlCallTimeout, left))587		if !ok || out == "" {588			continue589		}590		if out != last {591			last = out592			for line := range strings.SplitSeq(out, "\n") {593				fmt.Printf("liken: %s: %s\n", prefix, line)594			}595		}596		if settled(out) {597			return true598		}599	}600	return false601}602603// containsReady looks for the word Ready as a whole status field.604// This means NotReady does not match.605func containsReady(out string) bool {606	for line := range strings.SplitSeq(out, "\n") {607		fields := strings.Fields(line)608		if len(fields) >= 2 && fields[1] == "Ready" {609			return true610		}611	}612	return false613}614615// podsSettled reports whether every pod in the table has reached616// Running or Completed. The status field is kubectl's fourth column.617func podsSettled(out string) bool {618	for line := range strings.SplitSeq(out, "\n") {619		fields := strings.Fields(line)620		if len(fields) >= 4 && fields[3] != "Running" && fields[3] != "Completed" {621			return false622		}623	}624	return true625}626627// runWithin executes a command like run, and kills it with SIGKILL628// when the timeout passes. The reaper stays the only caller of wait:629// runWithin waits for the death registry's report, the same way630// stopK3s does. The kill goes through the os.Process and its pidfd,631// so a kill after the reaper collected the process returns632// os.ErrProcessDone and cannot reach a later process with the same633// pid. The output is read on its own goroutine,634// because a command that fills the pipe blocks until someone reads it,635// and that command would then never exit on its own.636func runWithin(timeout time.Duration, path string, args ...string) (string, bool) {637	cmd := exec.Command(path, args...)638	out, err := cmd.StdoutPipe()639	if err != nil {640		return "", false641	}642	cmd.Stderr = nil643	if err := deaths.start(cmd); err != nil {644		return "", false645	}646	pid := cmd.Process.Pid647	died := make(chan unix.WaitStatus, 1)648	go func() { died <- deaths.await(pid) }()649	output := make(chan []byte, 1)650	go func() {651		buf, _ := io.ReadAll(out)652		_ = out.Close()653		output <- buf654	}()655656	timer := time.NewTimer(timeout)657	defer timer.Stop()658	var status unix.WaitStatus659	select {660	case status = <-died:661	case <-timer.C:662		if cmd.Process.Kill() == nil {663			fmt.Fprintf(os.Stderr, "liken: %s did not finish within %s; killed it\n", path, timeout)664		}665		status = <-died666	}667	_ = cmd.Process.Release()668	buf := <-output669	return strings.TrimRight(string(buf), "\r\n"), status.Exited() && status.ExitStatus() == 0670}671672// runNarrated executes a command, echoes its output live to the673// console with a prefix like k3s's output, and reports whether the674// command exited cleanly. Use runNarrated for commands whose output675// matters to someone watching a boot, such as mke2fs reporting on the676// filesystem it creates. run's captured-output shape would hide that677// output instead.678func runNarrated(prefix, path string, args ...string) bool {679	cmd := exec.Command(path, args...)680	w := &lineWriter{dest: console, prefix: prefix}681	cmd.Stdout = w682	cmd.Stderr = w683	if err := deaths.start(cmd); err != nil {684		fmt.Fprintf(os.Stderr, "liken: starting %s: %v\n", path, err)685		return false686	}687	status := deaths.await(cmd.Process.Pid)688	_ = cmd.Process.Release()689	return status.Exited() && status.ExitStatus() == 0690}691692// run executes a command and returns its output. It waits for the693// command through the reaper (see the file comment: nobody but the694// reaper calls wait). Reading the pipe to EOF shows that the process695// finished writing. The reaper reports how the process died.696func run(path string, args ...string) (string, bool) {697	cmd := exec.Command(path, args...)698	out, err := cmd.StdoutPipe()699	if err != nil {700		return "", false701	}702	cmd.Stderr = nil703	if err := deaths.start(cmd); err != nil {704		return "", false705	}706	// Reading the pipe to EOF shows that the process finished707	// writing. The reaper reports how the process died.708	// The reaper collects the process, so cmd.Wait never runs to709	// close the pipe, and run closes it itself.710	buf, _ := io.ReadAll(out)711	_ = out.Close()712	status := deaths.await(cmd.Process.Pid)713	_ = cmd.Process.Release()714	return strings.TrimRight(string(buf), "\r\n"), status.Exited() && status.ExitStatus() == 0715}
init/switchroot.go 43.0%
1package main23// This file replaces the kernel's rootfs with the real root: the4// read-only system image, with a bounded overlay for the runtime's5// writes.6//7// When the kernel unpacks the initramfs, the kernel does not mount a8// filesystem to hold the files. The files land in rootfs: the9// filesystem instance that the kernel creates as the root of the10// mount tree, before any userspace process exists. rootfs is the11// bottom of the mount stack by construction. The kernel can never12// unmount rootfs, and pivot_root refuses to operate on it, because13// either action would leave the mount tree with no root at all.14// rootfs is also RAM with no bound and no accounting: it appears as15// device "rootfs" in the mount table, and nothing can measure its16// size. So an OS that stays on rootfs pays for itself in memory17// forever, and kubelet's node ephemeral-storage accounting reports18// nothing for it.19//20// liken's root lives elsewhere. The system image, which rootimage.go21// finds and loop-mounts, is the lower, read-only layer of an overlay22// filesystem. A small, fixed-size tmpfs is the upper, writable layer.23// The OS costs page cache instead of a permanent copy of itself.24// Everything that grows with use lives on a disk role, not under /.25//26// rootfs holds only what the boot loader delivered beyond the OS27// itself: the boot archive (this program and the early kernel28// modules) and the deployment layer. Before the switch, this code29// carries the layer's files onto the overlay: manifests, identity,30// declared modules, and their index. This preserves the initramfs31// contract that later archives override earlier ones, so the layer's32// files win over the image's files.33//34// The procedure takes its name from the util-linux tool that35// performs it on conventional systems (switch_root):36//37//  1. mount the system image and the overlay above it,38//  2. carry rootfs's extra files (the layer) onto the overlay,39//  3. delete the originals, and return their RAM to the kernel,40//  4. move the mounts into the new root, move the new root onto /,41//     and chroot into it,42//  5. re-exec init from the new root.43//44// The re-exec is required. The running program is the last thing45// pinning the old root in memory: the kernel maps its executable46// from rootfs, so those pages cannot be freed until a different47// program runs in their place. exec replaces the process image while48// it keeps PID 1. The second "hello from userspace" message on the49// console is the same program, restarting from the new root.50//51// An install boot (liken.install) skips this entire procedure. The52// installer runs from rootfs, copies the release payload it carried53// to disk, and powers off. None of this work needs the overlay, and54// the payload is far bigger than the overlay's size bound.5556import (57	"fmt"58	"io"59	"io/fs"60	"os"61	"path/filepath"62	"slices"63	"strings"6465	"golang.org/x/sys/unix"66)6768// switchedMarker is how init tells its post-switch run apart from69// its first run: the re-exec passes switchedMarker as an argument.70// The kernel also hands unrecognized non-key=value boot parameters to71// init as arguments, but nothing on liken's command line starts with72// dashes.73const switchedMarker = "--switched"7475const newRoot = "/newroot"7677// stagingDir is where the boot's mounts assemble before the switch.78// It is a tmpfs holding the overlay's upper and work directories,79// with the system image mounted beneath it, and the slot too, when80// this boot is a slot boot. One MS_MOVE operation carries the whole81// subtree into the new root at bootMountsDir, so the running system82// can see what it booted from.83const stagingDir = "/liken-boot"8485// bootMountsDir is the staging tree's location inside the new root.86// It is deliberately not /var/lib/liken/boot, which is where the87// bootHome role mounts on a BIOS machine. A mount there would cover88// this tree and leave the running system unable to name the slot it89// booted from or the image it is running.90const bootMountsDir = "/var/lib/liken/boot-mounts"9192// writesSize bounds the overlay's upper layer, which is the root93// filesystem's entire write budget. The runtime's writes under / are94// small and fixed: k3s config drop-ins, resolv.conf, and the layer's95// seed files. Everything that grows with use belongs to a disk role.96// This budget size is deliberate. If something fills the budget, that97// is a bug report about something writing to / that should not write98// there, not a reason to raise the budget.99const writesSize = "128m"100101// maybeSwitchRoot runs once, before init touches anything else. The102// switch must happen while the mount tree is empty, because moving103// one mount is simple and moving many is not, and it must happen104// before any child process exists. The liken.* boot parameters that105// maybeSwitchRoot needs live in /proc/cmdline and nowhere else. The106// kernel treats any parameter with a dot in its name as a module107// parameter, and passes it to init neither as an argument nor in the108// environment. So this function mounts /proc here, before anything109// else, and detaches /proc again before the switch.110func maybeSwitchRoot() {111	if slices.Contains(os.Args[1:], switchedMarker) {112		return113	}114	if err := os.MkdirAll("/proc", 0o755); err == nil {115		if err := unix.Mount("proc", "/proc", "proc", unix.MS_NOSUID|unix.MS_NOEXEC|unix.MS_NODEV, ""); err != nil {116			fmt.Fprintf(os.Stderr, "liken: mounting /proc for the switch: %v\n", err)117		}118	}119120	// The boot archive's few modules load on every first run, not121	// only on runs that switch root. An install boot stays on rootfs,122	// but still mounts FAT slots, and vfat's encoding table is one of123	// these modules.124	if names, err := readModuleList(filepath.Join(bootModulesDir, "boot-modules.conf")); err == nil {125		loadBootModules(names...)126	} else {127		fmt.Fprintf(os.Stderr, "liken: boot modules: %v\n", err)128	}129130	if bootParam(installParam) || bootParam(reportParam) {131		// The installer and the hardware report both run from rootfs.132		// The installer copies its payload to disk and powers off; the133		// report reads hardware, mounts the payload's module tree to134		// load drivers, and reboots. Neither needs the overlay, and the135		// payload is far bigger than the overlay's size bound.136		fmt.Println("liken: install or report boot; staying on rootfs")137		_ = unix.Unmount("/proc", unix.MNT_DETACH)138		return139	}140	if err := switchRoot(); err != nil {141		// A machine still on rootfs is degraded but alive, and the142		// console works either way. Continue, and let the report show143		// the state of the machine.144		fmt.Fprintf(os.Stderr, "liken: switch_root: %v (continuing on rootfs)\n", err)145		_ = unix.Unmount("/proc", unix.MNT_DETACH)146	}147}148149func switchRoot() error {150	// Mount devices first. The loop device and the slot's partition151	// both live in devtmpfs, the kernel's own device catalog. The new152	// root gets a fresh mount of the same catalog later153	// (mountEssentials); this mount is detached before the switch.154	if err := os.MkdirAll("/dev", 0o755); err != nil {155		return err156	}157	if err := unix.Mount("devtmpfs", "/dev", "devtmpfs", unix.MS_NOSUID, ""); err != nil {158		return fmt.Errorf("mounting devtmpfs: %w", err)159	}160161	// Slot recognition walks /sys/block for the GPT names that the162	// kernel read, so sysfs joins the early mounts. Like /dev and163	// /proc, this mount is detached before the switch, and remounted164	// fresh in the new root.165	if err := os.MkdirAll("/sys", 0o755); err != nil {166		return err167	}168	if err := unix.Mount("sysfs", "/sys", "sysfs", unix.MS_NOSUID|unix.MS_NOEXEC|unix.MS_NODEV, ""); err != nil {169		return fmt.Errorf("mounting sysfs: %w", err)170	}171172	// This is the staging tmpfs: the overlay's writable layer, and the173	// parent of every mount that this boot assembles. Its size is the174	// bound on what the running system can ever write under /.175	if err := os.MkdirAll(stagingDir, 0o755); err != nil {176		return err177	}178	if err := unix.Mount("tmpfs", stagingDir, "tmpfs",179		unix.MS_NOSUID, "mode=0755,size="+writesSize); err != nil {180		return fmt.Errorf("mounting the writes tmpfs: %w", err)181	}182	for _, dir := range []string{"upper", "work", "system"} {183		if err := os.MkdirAll(filepath.Join(stagingDir, dir), 0o755); err != nil {184			return err185		}186	}187188	imagePath, err := findSystemImage(bootParamValue("liken.slot"), filepath.Join(stagingDir, "slot"))189	if err != nil {190		return err191	}192	if err := loopMount(imagePath, filepath.Join(stagingDir, "system")); err != nil {193		return err194	}195196	if err := os.MkdirAll(newRoot, 0o755); err != nil {197		return err198	}199	overlay := fmt.Sprintf("lowerdir=%s,upperdir=%s,workdir=%s",200		filepath.Join(stagingDir, "system"),201		filepath.Join(stagingDir, "upper"),202		filepath.Join(stagingDir, "work"))203	if err := unix.Mount("overlay", newRoot, "overlay", 0, overlay); err != nil {204		return fmt.Errorf("mounting the overlay: %w", err)205	}206207	// Carry the boot loader's extra files — the deployment layer —208	// onto the overlay. This excludes everything that the boot archive209	// itself brought: the system image already carries init, and the210	// boot module tree stays off the real root deliberately, so its211	// partial index can never override the image's complete index.212	if err := carryTree("/", newRoot, []string{213		newRoot, "/dev", "/proc", "/sys", stagingDir,214		"/liken", ramImage, bootModulesDir,215	}); err != nil {216		return fmt.Errorf("carrying the layer: %w", err)217	}218219	// Delete the originals while they are still reachable by a path.220	// After the move below, the old root has no path at all. This221	// step is what actually reclaims the RAM. Skipping this step222	// would leave the machine carrying the layer twice. A223	// RAM-delivered system image is the one exception: the loop224	// device pins it in memory, deleted or not, for as long as the225	// image stays mounted.226	entries, err := os.ReadDir("/")227	if err != nil {228		return err229	}230	for _, entry := range entries {231		path := "/" + entry.Name()232		if path == newRoot || path == "/dev" || path == "/proc" ||233			path == "/sys" || path == stagingDir {234			continue235		}236		if err := os.RemoveAll(path); err != nil {237			return fmt.Errorf("clearing old root: %w", err)238		}239	}240241	// The staging tree moves into the new root in one piece. The242	// system image and slot mounts move along with it as child243	// mounts, so the running system can see the mounts it booted from244	// at a real path.245	target := filepath.Join(newRoot, bootMountsDir)246	if err := os.MkdirAll(target, 0o755); err != nil {247		return err248	}249	if err := unix.Mount(stagingDir, target, "", unix.MS_MOVE, ""); err != nil {250		return fmt.Errorf("moving boot mounts into the new root: %w", err)251	}252253	// Whatever the kernel put at /dev belongs to the old root. The254	// /proc mount, used only for reading boot parameters, is no255	// longer needed. Detach both mounts, so nothing pins the old tree256	// once it moves. The new root gets fresh mounts of /dev and /proc257	// (mountEssentials). The console keeps working throughout,258	// because this program opened its stdio descriptors before any of259	// this.260	_ = unix.Unmount("/dev", unix.MNT_DETACH)261	_ = unix.Unmount("/proc", unix.MNT_DETACH)262	_ = unix.Unmount("/sys", unix.MNT_DETACH)263264	// Now the move and the chroot. MS_MOVE attaches the overlay onto265	// /. This is legal where pivot_root is not, because nothing has266	// to be unmounted. chroot(".") changes what this process treats267	// as "/". Order matters here: chdir runs first, so "." names the268	// new root throughout.269	if err := os.Chdir(newRoot); err != nil {270		return err271	}272	if err := unix.Mount(".", "/", "", unix.MS_MOVE, ""); err != nil {273		return fmt.Errorf("moving the overlay onto /: %w", err)274	}275	if err := unix.Chroot("."); err != nil {276		return fmt.Errorf("chroot: %w", err)277	}278	if err := os.Chdir("/"); err != nil {279		return err280	}281282	fmt.Println("liken: re-executing from the system image")283	return unix.Exec("/liken", []string{"/liken", switchedMarker}, os.Environ())284}285286// carryTree replicates the filesystem tree at src into dst. It287// copies directories, symlinks, and regular files, and keeps their288// permissions intact, while it skips the subtrees named in skip.289// Directories, symlinks, and regular files are the complete contents290// of an initramfs that liken's image build produces. Any other file291// type is unexpected, and fails the carry. carryTree overwrites292// existing files in dst, because dst is the overlay, whose lower293// layer already carries the system's own copy of anything the294// deployment layer overrides. The depmod index is the most important295// example.296func carryTree(src, dst string, skip []string) error {297	return filepath.WalkDir(src, func(path string, d fs.DirEntry, err error) error {298		if err != nil {299			return err300		}301		if path == src {302			return nil303		}304		if slices.Contains(skip, path) {305			if d.IsDir() {306				return fs.SkipDir307			}308			return nil309		}310		target := filepath.Join(dst, strings.TrimPrefix(path, src))311		info, err := d.Info()312		if err != nil {313			return err314		}315		switch {316		case d.IsDir():317			// The overlay's lower layer already has most directories.318			// Creating a directory again must not fail. An existing319			// directory keeps the image's permissions.320			if err := os.Mkdir(target, 0o755); err != nil {321				if os.IsExist(err) {322					return nil323				}324				return err325			}326			return os.Chmod(target, info.Mode().Perm())327		case d.Type()&fs.ModeSymlink != 0:328			link, err := os.Readlink(path)329			if err != nil {330				return err331			}332			if err := os.Symlink(link, target); err != nil && !os.IsExist(err) {333				return err334			}335			return nil336		case d.Type().IsRegular():337			return copyFile(path, target, info.Mode().Perm())338		default:339			return fmt.Errorf("%s: unexpected file type %s", path, d.Type())340		}341	})342}343344func copyFile(src, dst string, perm fs.FileMode) error {345	in, err := os.Open(src)346	if err != nil {347		return err348	}349	defer in.Close()350	out, err := os.OpenFile(dst, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, perm)351	if err != nil {352		return err353	}354	if _, err := io.Copy(out, in); err != nil {355		out.Close()356		return err357	}358	if err := out.Chmod(perm); err != nil {359		out.Close()360		return err361	}362	return out.Close()363}
init/system.go 19.5%
1package main23// The rest of the environment that k3s expects.4//5// The essential mounts in main.go make a machine usable at a basic6// level. Kubernetes has a longer list of assumptions, built up over7// years of running on full distributions. Each function here8// recreates one of those assumptions directly: it does the part of9// systemd's setup work that this machine needs.1011import (12	"fmt"13	"maps"14	"os"15	"slices"16	"strings"1718	"golang.org/x/sys/unix"1920	"github.com/liken-sh/liken/liken/machine"21)2223// sysctlDir is the tree applySysctls writes into. It is a package24// variable rather than the constant itself, so a test can point it at25// a directory of its own instead of writing real kernel parameters on26// the machine running the test.27var sysctlDir = machine.SysctlDir2829// applySysctls applies a set of kernel parameters. If one fails,30// applySysctls reports the failure and skips it, rather than treating31// it as fatal, because a mistyped parameter should not cost the32// machine its boot. It applies the keys in sorted order, so the33// console shows the same order every time.34//35// Init calls this twice: once for the settings every liken machine36// holds, in machine.OSSysctls, and once for the Machine spec's own37// sysctls. Printing each parameter on both passes is deliberate. A38// reader of the boot log sees what the OS set, and then sees which of39// those values this deployment chose to override.40func applySysctls(sysctls map[string]string) {41	for _, name := range slices.Sorted(maps.Keys(sysctls)) {42		value := sysctls[name]43		if err := machine.ApplySysctl(sysctlDir, name, value); err != nil {44			fmt.Fprintf(os.Stderr, "liken: %v\n", err)45			continue46		}47		fmt.Printf("liken: sysctl %s = %s\n", name, value)48	}49}5051// k3sMounts lists the filesystems that Kubernetes assumes beyond the52// essential mounts. It uses the same table-driven form as the53// essentials list in main.go.54var k3sMounts = []mount{55	// Kubernetes uses cgroup2 to measure and limit every container.56	// cgroup2 is one hierarchy under /sys/fs/cgroup. Under it, the57	// kernel accounts for and caps each cgroup's CPU use, memory use,58	// and process count. kubelet does not run without this mount.59	// The kernel builds every controller in, so this mount is the60	// whole setup this step needs.61	{"cgroup2", "/sys/fs/cgroup", "cgroup2", unix.MS_NOSUID | unix.MS_NOEXEC | unix.MS_NODEV},6263	// Pseudo-terminals. kubectl exec is the only interactive access64	// on an OS with no shell, and it allocates its terminals here.65	{"devpts", "/dev/pts", "devpts", unix.MS_NOSUID | unix.MS_NOEXEC},6667	// POSIX shared memory. Each pod gets its own /dev/shm per68	// container, but the container runtime sometimes needs the69	// host's /dev/shm to exist too.70	{"tmpfs", "/dev/shm", "tmpfs", unix.MS_NOSUID | unix.MS_NODEV},71}7273func prepareForK3s() {74	for _, m := range k3sMounts {75		if err := os.MkdirAll(m.target, 0o755); err != nil {76			fmt.Fprintf(os.Stderr, "liken: mkdir %s: %v\n", m.target, err)77			continue78		}79		if err := unix.Mount(m.source, m.target, m.fstype, m.flags, ""); err != nil {80			fmt.Fprintf(os.Stderr, "liken: mount %s on %s: %v\n", m.fstype, m.target, err)81		}82	}8384	// kubelet requires that mounts made under / propagate into the85	// mount namespaces of its containers. This propagation mode is86	// called rshared. It lets a volume that is mounted after a pod87	// starts still appear inside the pod. A plain root mount88	// defaults to private propagation. This command changes89	// propagation for the whole tree, recursively, to shared.90	if err := unix.Mount("", "/", "", unix.MS_REC|unix.MS_SHARED, ""); err != nil {91		fmt.Fprintf(os.Stderr, "liken: making / rshared: %v\n", err)92	}9394	// /etc/machine-id is the systemd convention for a stable, unique95	// identifier for the installation. Enough software reads it,96	// including k3s, that a machine needs one. The kernel generates97	// a fresh UUID on every read of the /proc file98	// /proc/sys/kernel/random/uuid; machine-id is that UUID, without99	// the dashes. This machine's ID is random on every boot, because100	// a machine with no writable disk keeps nothing across boots.101	if raw, err := os.ReadFile("/proc/sys/kernel/random/uuid"); err == nil {102		id := strings.NewReplacer("-", "", "\n", "").Replace(string(raw))103		if err := os.WriteFile("/etc/machine-id", []byte(id+"\n"), 0o444); err != nil {104			fmt.Fprintf(os.Stderr, "liken: machine-id: %v\n", err)105		}106	}107108	// k3s reads $HOME and $PATH like any Unix program. PID 1 gets109	// neither variable from the kernel, so this code sets them. The110	// four conventional directories are enough, because k3s adds its111	// own unpacked userland to the front of PATH when it builds PATH112	// for the child processes it starts.113	os.Setenv("HOME", "/root")114	os.Setenv("PATH", "/sbin:/bin:/usr/sbin:/usr/bin")115116	_ = os.MkdirAll("/root", 0o700)117	_ = os.MkdirAll("/var/log", 0o755)118119	// /tmp exists on every machine. The container runtime stages120	// kubectl exec sessions there. On a machine that declares the121	// machineEphemeral storage role, a disk partition is already122	// mounted at /tmp, and this sets the same mode on it again. On123	// every other machine, /tmp is RAM, like the rest of the root124	// filesystem.125	openTmp("/tmp")126}127128// tmpMode is the mode of /tmp: writable by every user, with the sticky129// bit, so a process can delete or rename only the files it owns. Without130// the sticky bit, one user can replace another's file in /tmp, or plant131// a symlink where another expects to create a file. Go keeps the sticky132// bit in its own os.ModeSticky flag, not in the octal 01000 bit, so a133// mode written as 0o1777 drops it.134const tmpMode = os.ModeSticky | 0o777135136// openTmp makes dir with tmpMode. It sets the mode with chmod after137// the mkdir, because MkdirAll applies the umask to the mode it is138// given, and leaves an existing directory's mode as it is.139func openTmp(dir string) {140	_ = os.MkdirAll(dir, tmpMode)141	_ = os.Chmod(dir, tmpMode)142}143144// configureNameResolution writes /etc/hosts and /etc/nsswitch.conf, the145// two files that decide how a name on this machine resolves to an146// address.147func configureNameResolution(hostname string, entries []machine.HostEntry) {148	// /etc/hosts is the only place "localhost" and this machine's own149	// hostname resolve, and it is also where spec.network.hostEntries150	// lands: one line below the three fixed lines, for each entry the151	// manifest declares. The fixed lines come first, so a resolver's152	// first match always wins and no entry can override localhost or153	// this machine's own name.154	if err := os.WriteFile("/etc/hosts", []byte(machine.HostsFile(hostname, entries)), 0o644); err != nil {155		fmt.Fprintf(os.Stderr, "liken: /etc/hosts: %v\n", err)156	}157	for _, entry := range entries {158		fmt.Printf("liken: /etc/hosts: %s %s\n", entry.Address, strings.Join(entry.Names, " "))159	}160161	// nsswitch.conf tells a glibc resolver the order to try its162	// lookup sources in. musl ignores this file, and Go's own163	// resolver already defaults to this order, so on this machine's164	// binaries today the file changes nothing. It exists in advance165	// of that: a glibc binary that lands on the node later, in a166	// vendored component or an add-on image, would otherwise query167	// DNS before it read /etc/hosts, and a static entry would lose to168	// DNS silently. Writing "files dns" makes the hosts file always169	// win, on any resolver that reads this file.170	if err := os.WriteFile("/etc/nsswitch.conf", []byte("hosts: files dns\n"), 0o644); err != nil {171		fmt.Fprintf(os.Stderr, "liken: /etc/nsswitch.conf: %v\n", err)172	}173}
init/time.go 82.5%
1package main23// Disciplining the clock.4//5// A computer's clock drifts. Cheap oscillators gain or lose seconds6// each day. Kubernetes assumes that no clock drifts. TLS7// certificates carry notBefore and notAfter instants, leases carry8// renew deadlines, and, on multi-leader clusters, etcd orders events9// by time. If a machine's clock is wrong enough, the machine cannot10// even join a cluster, because every certificate that the CA issued11// appears to be from the future. This is why time is a machine-plane12// concern, and why the first correction happens before k3s starts:13// the cluster cannot fix a clock that is keeping the machine out of14// the cluster.15//16// The protocol is SNTP, the stateless subset of NTP. Each exchange17// sends one 48-byte request and receives one 48-byte reply, with18// four timestamps between them. The client records when it sent the19// request (t1) and when the reply arrived (t4). The server records20// when the request arrived (t2) and when the reply left (t3). The21// value (t2-t1) is the outbound trip time plus the clock error. The22// value (t3-t4) is the return trip time minus the clock error.23// Averaging the two values cancels the travel time when the path is24// symmetric, and leaves only the error. This calculation is the core25// of the protocol.26//27// The vendored client, github.com/beevik/ntp, is the same library28// that Talos uses. It implements this calculation and the protocol's29// checks: leap-second flags, kiss-of-death codes, and stratum30// bounds. As with the DHCP client, liken uses this established31// library for the wire format, and keeps the decisions in this file:32// which source to ask, when to step the clock, and how hard to slew33// it.34//35// The source hierarchy follows liken's usual pattern: explicit36// inputs, not discovery. Leaders ask the upstreams declared on the37// Cluster. Followers ask the leaders directly, resolved from the38// fleet's Machine manifests, with the endpoint's host as the39// fallback. A leader answers from its own disciplined clock (see40// responder.go). A cluster with no upstreams free-runs. A41// free-running cluster is internally consistent, but it is correct42// only if the hardware clocks happen to be correct, and the43// machine's status reports this free-running state.44//45// Correction comes in two strengths. The choice depends on whether46// anything is running that could notice a sudden change in time. At47// boot, before k3s starts, the clock steps to the measured time,48// using clock_settime. Nothing is running yet that would be49// affected, and a machine must not join the cluster with a wrong50// clock. After boot, the clock only slews, using adjtimex. The51// kernel trims the clock's rate so the clock drifts gradually onto52// the correct time, with seconds always moving forward at close to53// one per second. Stepping the clock on a running node would change54// the time underneath lease renewals and container logs. A slew55// removes the drift without any visible jump in time.5657import (58	"context"59	"fmt"60	"net"61	"net/url"62	"os"63	"path/filepath"64	"slices"65	"sync"66	"time"6768	"github.com/beevik/ntp"69	"golang.org/x/sys/unix"7071	"github.com/liken-sh/liken/liken/api"72	"github.com/liken-sh/liken/liken/cluster"73	"github.com/liken-sh/liken/liken/machine"74)7576// timePollInterval sets how often the discipline loop measures the77// clock. 64 seconds is NTP's classic starting interval (minpoll 6,78// or 2^6 seconds). This interval is frequent enough to catch drift79// measured in parts per million, and infrequent enough that it does80// not burden an upstream time source.81const timePollInterval = 64 * time.Second8283// stepThreshold is the offset below which a boot does not step the84// clock. Below this offset, the slew absorbs the error faster than85// the step's disruption is worth. 128ms is the same line that ntpd86// uses between slewing and stepping.87const stepThreshold = 128 * time.Millisecond8889// The stratum values that status reports. NTP counts distance from a90// reference clock: stratum 1 is attached to a reference clock91// directly, and each hop away adds one. Stratum 10 is the common92// convention for a clock that is deliberately local. Stratum 1693// means unsynchronized: the machine has time sources, but has not94// reached one yet.95const (96	stratumFreeRunning    = 1097	stratumUnsynchronized = 1698)99100// timeSync records one successful measurement: which source101// answered, its stratum, how far off this machine's clock was, and102// when the measurement happened.103type timeSync struct {104	source  string105	stratum int106	offset  time.Duration107	at      time.Time108}109110// clock holds the machine's timekeeping state. Two components share111// it: the discipline loop records each measurement, and, on a112// leader, the responder reads the latest measurement for what to113// advertise. Running the machine plane in one process makes this114// possible with a mutex. Two separate daemons would need a socket115// between them; two goroutines only need a mutex.116type clock struct {117	mu      sync.Mutex118	sources []string119	last    *timeSync120}121122func newClock(sources []string) *clock {123	return &clock{sources: sources}124}125126func (c *clock) record(measured *timeSync) {127	c.mu.Lock()128	defer c.mu.Unlock()129	c.last = measured130}131132// lastSyncAt answers when the last good measurement was taken, or the133// zero time before the first one.134func (c *clock) lastSyncAt() time.Time {135	c.mu.Lock()136	defer c.mu.Unlock()137	if c.last == nil {138		return time.Time{}139	}140	return c.last.at141}142143// timeSources works out where this machine gets its time. It uses144// declared inputs by role, the same way liken works out other145// machine state. Leaders ask the Cluster's upstreams. Followers ask146// every leader. timeSources resolves each leader's address from the147// Machine manifests that the image already carries; one boot medium148// carries manifests for the whole fleet. Each leader is identified149// by its declared address on the node network. timeSources appends150// the endpoint's host as a fallback for any leader it could not151// resolve, for example a DHCP-addressed leader that declares no152// address to find. Every source therefore comes from inputs the153// machine already needed to find its cluster, so no new information154// is required. timeSources returns nil for a leader with no155// upstreams declared, which means the leader free-runs.156func timeSources(clusterDoc *cluster.Cluster, role api.Role, manifestDir string) []string {157	if clusterDoc == nil {158		return nil159	}160	if role == api.RoleLeader {161		return clusterDoc.Spec.Time.Upstreams162	}163	sources := leaderAddresses(clusterDoc, manifestDir)164	endpoint, err := url.Parse(clusterDoc.Spec.Endpoint)165	if err == nil && endpoint.Hostname() != "" && !slices.Contains(sources, endpoint.Hostname()) {166		sources = append(sources, endpoint.Hostname())167	}168	return sources169}170171// leaderAddresses resolves each machine named in spec.leaders to its172// static address on the node network. It reads each machine's173// manifest from the image. leaderAddresses skips a leader it cannot174// resolve, for example one with no manifest or no address inside175// nodeCIDR. The endpoint fallback covers a skipped leader, and the176// source list is a preference order in which missing entries are177// acceptable.178func leaderAddresses(clusterDoc *cluster.Cluster, manifestDir string) []string {179	var addresses []string180	for _, name := range clusterDoc.Spec.Leaders {181		if addr := declaredNodeAddress(clusterDoc, manifestDir, name); addr != "" {182			addresses = append(addresses, addr)183		}184	}185	return addresses186}187188// declaredNodeAddress resolves one machine's declared address on the189// node network from its manifest in the image. It returns "" when190// the machine declares no address, for example a DHCP machine, which191// has no address to find. This is how machines find each other192// before any of them has started: through the fleet's declared193// inputs, not through discovery.194func declaredNodeAddress(clusterDoc *cluster.Cluster, manifestDir, name string) string {195	_, subnet, err := net.ParseCIDR(clusterDoc.Spec.Network.NodeCIDR)196	if err != nil {197		return ""198	}199	m, err := machine.Load(filepath.Join(manifestDir, name+".yaml"))200	if err != nil {201		return ""202	}203	for _, iface := range m.Spec.Network.Interfaces {204		ip, _, err := net.ParseCIDR(iface.Address)205		if err == nil && subnet.Contains(ip) {206			return ip.String()207		}208	}209	return ""210}211212// timeStatus reports the clock's state as machine status. It reports213// the same facts that the console prints, in a form other systems214// can query. A machine that has sources but has not synced yet is215// Unsynchronized, at stratum 16. A machine with no sources at all is216// deliberately FreeRunning, at stratum 10. Neither state is reported217// as Synchronized, because that word is reserved for a clock that is218// currently following a source that is itself synchronized.219func timeStatus(sync *timeSync, sources []string) machine.TimeStatus {220	if sync == nil {221		if len(sources) == 0 {222			return machine.TimeStatus{State: machine.TimeFreeRunning, Stratum: stratumFreeRunning}223		}224		return machine.TimeStatus{State: machine.TimeUnsynchronized, Stratum: stratumUnsynchronized}225	}226	return machine.TimeStatus{227		State:   machine.TimeSynchronized,228		Source:  sync.source,229		Stratum: sync.stratum + 1,230		Offset:  sync.offset.Round(10 * time.Microsecond).String(),231	}232}233234// querySources asks each source in turn and uses the first valid235// answer. A full NTP daemon polls every source, scores each one by236// delay and dispersion, and combines the results that pass its237// checks. SNTP's simpler approach, trusting the first valid reply,238// works for a machine whose sources were each chosen directly by an239// operator, rather than drawn from a public pool.240func querySources(sources []string) (*timeSync, error) {241	var lastErr error242	for _, source := range sources {243		response, err := ntp.QueryWithOptions(source, ntp.QueryOptions{})244		if err == nil {245			err = response.Validate()246		}247		if err != nil {248			lastErr = fmt.Errorf("%s: %w", source, err)249			continue250		}251		return &timeSync{252			source:  source,253			stratum: int(response.Stratum),254			offset:  response.ClockOffset,255			at:      time.Now(),256		}, nil257	}258	return nil, lastErr259}260261// stepClockAtBoot is the one moment when liken jumps the clock. k3s262// has not started yet, so no lease, log, or watch depends on the263// current time. The number of attempts is limited, because a264// machine's boot must not depend on the internet being available. If265// no source answers, the boot continues on the hardware clock, and266// the discipline loop keeps trying without limit. stepClockAtBoot267// returns the first successful sync so that status can report it.268func stepClockAtBoot(sources []string) *timeSync {269	if len(sources) == 0 {270		fmt.Println("liken: time: no sources declared; free-running on the hardware clock")271		return nil272	}273	for attempt := range 3 {274		sync, err := queryTimeSources(sources)275		if err != nil {276			fmt.Fprintf(os.Stderr, "liken: time: measuring: %v\n", err)277			time.Sleep(time.Duration(attempt+1) * time.Second)278			continue279		}280		if abs := sync.offset.Abs(); abs < stepThreshold {281			fmt.Printf("liken: time: clock is %s from %s (stratum %d); close enough to slew\n",282				sync.offset.Round(10*time.Microsecond), sync.source, sync.stratum)283			return sync284		}285		corrected := time.Now().Add(sync.offset)286		if err := setSystemClock(corrected); err != nil {287			fmt.Fprintf(os.Stderr, "liken: time: stepping the clock: %v\n", err)288			return sync289		}290		fmt.Printf("liken: time: stepped the clock %s from %s (stratum %d); it is %s\n",291			sync.offset.Round(time.Millisecond), sync.source, sync.stratum,292			corrected.Format(time.RFC3339))293		sync.at = time.Now()294		return sync295	}296	fmt.Fprintln(os.Stderr, "liken: time: no source answered; booting on the hardware clock (the discipline loop keeps trying)")297	return nil298}299300// The clock's actions are package variables so that a test can run301// the decisions above and below against fakes. A real query needs a302// socket, which stops a synctest bubble's clock, and the real clock303// syscalls need PID 1's privileges. Production code never replaces304// them.305var (306	queryTimeSources  = querySources307	setSystemClock    = clockSettime308	slewSystemClock   = slewClock309	saveHardwareClock = writeRTC310)311312// clockSettime steps the system clock to t with clock_settime.313func clockSettime(t time.Time) error {314	ts := unix.NsecToTimespec(t.UnixNano())315	return unix.ClockSettime(unix.CLOCK_REALTIME, &ts)316}317318// slewAmount limits how much correction one adjtimex call requests.319// The kernel's old-style singleshot adjustment works best within320// half a second. slewAmount corrects a larger error across several321// polls, instead of requesting it all at once. This clamp owns the322// limit, not the code that calls slewAmount.323func slewAmount(offset time.Duration) time.Duration {324	return min(max(offset, -500*time.Millisecond), 500*time.Millisecond)325}326327// slewClock asks the kernel to absorb the offset gradually. The328// old-style adjtime interface, ADJ_OFFSET_SINGLESHOT, trims the329// clock's tick rate by about 0.5ms per second until it has absorbed330// the requested offset, then resumes normal ticking. Time never331// jumps and never runs backward. This is the reason to slew the332// clock instead of stepping it.333func slewClock(offset time.Duration) error {334	tx := &unix.Timex{335		Modes:  unix.ADJ_OFFSET_SINGLESHOT,336		Offset: slewAmount(offset).Microseconds(),337	}338	_, err := unix.Adjtimex(tx)339	return err340}341342// syncStaleAfter is how long the loop keeps reporting synchronized343// after its last good measurement. Three missed polls means the344// source is gone, not just busy, and status must stop reporting345// synchronized at that point.346const syncStaleAfter = 3 * timePollInterval347348// worthRepublishing reports whether a fresh measurement changes the349// published time facts. A change in state, source, or stratum must350// always be reported. The offset must be reported only when it has351// moved past offsetPublishThreshold since the last publish, so the352// published offset is within that threshold of the measured one. SNTP353// measurements wobble by microseconds on every poll, and each354// republished fact has a cost: the machine publishes a status update355// whenever the facts change, and each such write causes a raft round356// and an fsync on every one of the cluster's leaders. A fleet whose357// clocks are working correctly should cost etcd nothing extra.358//359// Time passing is not news. The facts hold no time of the last sync,360// because that time would change at every poll, and a machine whose361// polls stop answering reports Unsynchronized after syncStaleAfter.362// So Synchronized means a measurement within the last three polls.363const offsetPublishThreshold = 25 * time.Millisecond364365func worthRepublishing(published, fresh machine.TimeStatus, drift time.Duration) bool {366	if published.State != fresh.State || published.Source != fresh.Source || published.Stratum != fresh.Stratum {367		return true368	}369	return drift.Abs() >= offsetPublishThreshold370}371372// writeRTC copies the system clock into the hardware clock. Linux373// never does this by itself. On a traditional distribution, a374// shutdown script does this job; here, init does it. The RTC is the375// clock that the machine starts its next boot from. Writing the RTC376// after a sync means that even a machine that loses power boots with377// roughly correct time. Writing the RTC at a clean shutdown carries378// the best final time estimate into the next boot. These are the379// only two moments when init writes the RTC; between them, the RTC380// keeps time on its own battery. The value written is UTC. Storing381// local time in the RTC is a desktop-PC convention from the past,382// and a fleet spanning time zones needs its hardware clocks to share383// one convention.384func writeRTC() {385	f, err := os.OpenFile("/dev/rtc0", os.O_WRONLY, 0)386	if err != nil {387		fmt.Fprintf(os.Stderr, "liken: time: opening the RTC: %v\n", err)388		return389	}390	defer f.Close()391	now := time.Now().UTC()392	// The RTC interface takes a broken-down calendar time, in the393	// style of C's struct tm. Months count from zero and years count394	// from 1900. The ioctl inherits these conventions from four395	// decades of C programming.396	rt := unix.RTCTime{397		Sec:  int32(now.Second()),398		Min:  int32(now.Minute()),399		Hour: int32(now.Hour()),400		Mday: int32(now.Day()),401		Mon:  int32(now.Month() - 1),402		Year: int32(now.Year() - 1900),403	}404	if err := unix.IoctlSetRTCTime(int(f.Fd()), &rt); err != nil {405		fmt.Fprintf(os.Stderr, "liken: time: writing the RTC: %v\n", err)406		return407	}408	fmt.Printf("liken: time: hardware clock set to %s\n", now.Format(time.RFC3339))409}410411// disciplineClock builds the machine plane's time component. It412// measures, slews, publishes, sleeps, and repeats. The clock loop is413// the only writer of the time/ subtree, so it keeps a local copy of the414// clock's state and needs no lock: no other component ever writes those415// files. disciplineClock prints transitions rather than every poll: a416// sync gained, a sync lost, and an offset only when the offset exceeds417// the step threshold, which means drift is outrunning the slew. The418// facts follow the same rule. The loop rewrites the time/ files only419// when worthRepublishing reports that the measurement matters, and never420// for microsecond wobble; between writes it holds the latest value in421// its local copy.422func disciplineClock(clk *clock, tree machine.FactsTree, initial machine.TimeStatus) func(context.Context) error {423	return func(ctx context.Context) error {424		// current holds the clock's state. This loop is the only writer425		// of the time/ subtree, so a local copy lets the loop read a426		// consistent value as it makes its decisions. It starts from the427		// seed the boot step published.428		current := initial429430		lastGood := clk.lastSyncAt()431		// published holds what the time/ subtree currently says. It is432		// the baseline that every worthRepublishing check compares433		// against. The boot step published this value, moments ago.434		published := current435		publishedOffset, _ := time.ParseDuration(current.Offset)436		// The boot step, or the lack of one, determined whether the RTC437		// has been written yet. If the boot came up on a wrong438		// hardware clock, this loop corrects the RTC at the first439		// sync it achieves.440		rtcWritten := current.State == machine.TimeSynchronized441		for {442			if !sleepUnlessCancelled(ctx, timePollInterval) {443				// Clean shutdown. Leave the hardware clock holding444				// the best time estimate this machine ever had.445				if !lastGood.IsZero() {446					saveHardwareClock()447				}448				return nil449			}450			sync, err := queryTimeSources(clk.sources)451			if err != nil {452				// A failed poll is worth reporting only when it453				// changes the machine's state. Past the staleness454				// window, the machine stops reporting a synchronized455				// clock.456				if current.State == machine.TimeSynchronized && time.Since(lastGood) > syncStaleAfter {457					fmt.Fprintf(os.Stderr, "liken: time: lost every source (%v); the clock is on its own\n", err)458					current.State = machine.TimeUnsynchronized459					current.Stratum = stratumUnsynchronized460					tree.WriteTime(current)461					published = current462				}463				continue464			}465466			if err := slewSystemClock(sync.offset); err != nil {467				fmt.Fprintf(os.Stderr, "liken: time: slewing the clock: %v\n", err)468			}469			if current.State != machine.TimeSynchronized {470				fmt.Printf("liken: time: synchronized to %s (stratum %d), offset %s\n",471					sync.source, sync.stratum, sync.offset.Round(10*time.Microsecond))472			} else if sync.offset.Abs() >= stepThreshold {473				fmt.Printf("liken: time: offset %s from %s exceeds the slew's pace; correcting over several polls\n",474					sync.offset.Round(time.Millisecond), sync.source)475			}476			lastGood = sync.at477			clk.record(sync)478			if !rtcWritten {479				saveHardwareClock()480				rtcWritten = true481			}482			// This drift check compares against the offset that was483			// last published, not the offset from the last poll.484			// Small wobbles accumulate toward the threshold this485			// way, instead of resetting every 64 seconds.486			current = timeStatus(sync, clk.sources)487			if worthRepublishing(published, current, sync.offset-publishedOffset) {488				tree.WriteTime(current)489				published, publishedOffset = current, sync.offset490			}491		}492	}493}
init/versions.go 90.3%
1package main23// The on-board components record.4//5// A release document lists every outside component, and the6// upstream version of it that shipped (see machine/release.go). This7// document lives on the channel, and a running machine should8// report its own composition with no request to the channel. So the9// image build writes the same record, from the same VERSION pins,10// into the image itself. This file folds that record into the11// version facts that the Machine publishes.12//13// The fold works in one direction only, and it does not override14// a value the machine observed itself. For a component that the15// running machine already has an observed value for, such as the16// kernel via uname, the netfilter userspace via iptables, or liken17// via its own build stamp, this fold keeps that observed value, in18// the running software's own vocabulary. The record fills in only19// what nothing else can observe: the boot artifacts, the bundled20// images, and the data files. If the record is missing, this fold is21// silent about it, because dev boots that predate the record, and22// lab images built by hand, are still valid machines. Their version23// facts are simply sparser.2425import (26	"os"27	"strings"2829	"sigs.k8s.io/yaml"3031	"github.com/liken-sh/liken/liken/machine"32)3334// componentsPath is where the image build stages the record. It is35// under /usr/share, owned by the squashfs, and deliberately outside36// /etc/liken, where the deployment layer's files live. The record is37// a fact about the image, not about any deployment of it. This is a38// variable so tests can override it.39var componentsPath = "/usr/share/liken/components.yaml"4041// cpuinfoPath is where the kernel reports each CPU's state. This is42// a variable so tests can override it.43var cpuinfoPath = "/proc/cpuinfo"4445// microcodeRevision reads the running CPUs' microcode revision, for46// example "0x28". Every core reports the same value once the early47// loader has run, so the first microcode line answers for the48// machine. An architecture or hypervisor that reports none leaves49// the fact empty, and the status omits it.50func microcodeRevision(path string) string {51	raw, err := os.ReadFile(path)52	if err != nil {53		return ""54	}55	for line := range strings.SplitSeq(string(raw), "\n") {56		key, value, found := strings.Cut(line, ":")57		if found && strings.TrimSpace(key) == "microcode" {58			return strings.TrimSpace(value)59		}60	}61	return ""62}6364// observedAtRuntime names the components in the record that this fold65// deliberately skips, because the running machine reports them itself,66// in the running software's own vocabulary. The versions guard test67// reads this list, so a component that belongs here is declared here68// rather than left to a reader to infer from the switch below.69var observedAtRuntime = []string{"liken", "kernel", "xtables"}7071// applyComponentFacts folds the record into the version block. It72// fills only the fields that no runtime probe owns. Every other name73// the image build writes into the record must appear in the switch74// below, and the versions guard test is what enforces that.75func applyComponentFacts(v *machine.VersionStatus) {76	raw, err := os.ReadFile(componentsPath)77	if err != nil {78		return79	}80	var record struct {81		Components []machine.ReleaseComponent `json:"components"`82	}83	if err := yaml.UnmarshalStrict(raw, &record); err != nil {84		return85	}86	for _, c := range record.Components {87		switch c.Name {88		case "k3s":89			v.K3s = c.Version90		case "trust":91			v.Trust = c.Version92		case "e2fsprogs":93			v.E2fsprogs = c.Version94		case "open-iscsi":95			v.OpenISCSI = c.Version96		case "nfs-utils":97			v.NFSUtils = c.Version98		case "wpa-supplicant":99			v.WPASupplicant = c.Version100		case "systemd-boot":101			v.SystemdBoot = c.Version102		case "grub":103			v.Grub = c.Version104		case "hwdata":105			v.Hwdata = c.Version106		case "tzdata":107			v.Tzdata = c.Version108		case "linux-firmware":109			v.LinuxFirmware = c.Version110		case "wireless-regdb":111			v.WirelessRegdb = c.Version112		case "microcode":113			v.Microcode = c.Version114		}115	}116}
init/wireless.go 91.4%
1package main23// Everything the boot does for an interface that declares a radio.4// The split of the work is deliberate: the supplicant owns the 802.115// session, the association, the keys, and the rekeys, and nothing6// else. Init keeps its own addressing, so once a radio associates the7// interface takes DHCP or a static address on the same code a wired8// port does. The order is the supplicant, then the join, then the9// address. The events on the supplicant's control socket10// matter because the kernel cannot tell a wrong passphrase apart from11// an access point that is switched off; only the supplicant can say12// which one it is (plans/completed/62-wifi.md).1314import (15	"context"16	"fmt"17	"net"18	"net/url"19	"os"20	"os/exec"21	"path/filepath"22	"sync"23	"time"2425	"github.com/vishvananda/netlink"26	"golang.org/x/sys/unix"2728	"github.com/liken-sh/liken/liken/machine"29)3031// supplicantBinary is the program the image carries for the 802.1132// session. wpa-supplicant/fetch.sh builds it static, with the nl8021133// driver, the control interface, and SAE, and nothing else.34const supplicantBinary = "/sbin/wpa_supplicant"3536// The whole wireless runtime lives under /run because /run is a37// tmpfs, the generated configuration holds the passphrase, and a38// passphrase must never reach a disk that outlives the boot. /run is39// mounted with the essential filesystems rather than with k3s's,40// because a radio comes up long before k3s does (worldreport.go).41//42// It is a variable rather than a constant so tests can write into a43// directory of their own. A real boot never points it anywhere but /run.44var wirelessRunDir = "/run/liken/wireless"4546// How long a boot waits for a radio to associate. A scan of both47// bands plus a four-way handshake finishes in a few seconds on a48// healthy network, and the DHCP exchange beside this wait already49// bounds itself at 30 seconds. A boot still waiting at 45 seconds is50// waiting on something that waiting will not fix. The wait ends early51// on the first settling event either way, so the bound costs nothing52// when the join works.53const associationPatience = 45 * time.Second5455// radio holds everything init keeps about one wireless interface: what56// the manifest asked for, what the join did, and the two things that57// stay live for the rest of the boot, the supervised process and the58// event stream it publishes.59type radio struct {60	ifname string61	ssid   string6263	// state and message are the join's verdict, bound for the facts64	// tree and, through it, for the Machine's status.65	state   machine.WirelessState66	message string6768	// control is the attached client on the supplicant's control69	// socket. The park below keeps reading it, which is what lets a fix70	// on the network side resume a parked boot with no power cycle.71	control *wpaControl72}7374// deterministic says which failures no amount of waiting corrects. A75// wrong passphrase, a missing passphrase file, and a passphrase with76// no safe rendering are all decisions the machine can state now; all77// three carry the WrongKey state. Every other failure looks exactly78// like an access point that has not answered yet.79func (r *radio) deterministic() bool {80	return r.state == machine.WirelessWrongKey81}8283// wirelessStatus renders the radio for the facts tree.84func (r *radio) wirelessStatus() *machine.WirelessStatus {85	return &machine.WirelessStatus{SSID: r.ssid, State: r.state, Message: r.message}86}8788// joinWireless runs the wireless half of one interface's bring-up: it89// generates the supplicant's configuration, starts the supplicant90// under supervision, and waits for the association. It returns a91// radio in every case, because a radio that failed to join is exactly92// what the status must report.93//94// liken does not touch rfkill. The kernel starts radios unblocked, a95// soft block does not survive a reboot, and nothing in liken writes96// one, so there is no block to clear. A hardware kill switch shows up97// as a radio that never associates, and no software unblock can clear98// that either.99func joinWireless(ifc machine.InterfaceSpec, stateRoot string) *radio {100	w := *ifc.Wireless101	r := &radio{ifname: ifc.Name, ssid: w.SSID, state: machine.WirelessAssociating}102103	config, err := wirelessConfig(w, stateRoot, controlSocketDir(ifc.Name))104	if err != nil {105		r.state = machine.WirelessWrongKey106		r.message = err.Error()107		fmt.Fprintf(os.Stderr, "liken: wireless: %s: %v\n", ifc.Name, err)108		return r109	}110111	path, err := writeWirelessConfig(ifc.Name, config)112	if err != nil {113		r.state = machine.WirelessNoCarrier114		r.message = err.Error()115		fmt.Fprintf(os.Stderr, "liken: wireless: %s: %v\n", ifc.Name, err)116		return r117	}118119	fmt.Printf("liken: wireless: %s joins %s\n", ifc.Name, w.SSID)120	control, err := superviseSupplicant(ifc.Name, path)121	if err != nil {122		r.state = machine.WirelessNoCarrier123		r.message = err.Error()124		fmt.Fprintf(os.Stderr, "liken: wireless: %s: %v\n", ifc.Name, err)125		return r126	}127	r.control = control128129	awaitAssociation(r)130	return r131}132133// awaitAssociation reads the supplicant's events until one of them134// settles the join, or until associationPatience runs out. It writes135// the verdict onto the radio.136//137// A scan that finds nothing is not a verdict. An access point that is138// off, rebooting, or out of range produces exactly these events, and139// the plan's rule is that absence never parks a boot. The console140// line still goes out for every event, because a person watching the141// boot needs to know what the radio is doing.142func awaitAssociation(r *radio) {143	deadline := time.After(associationPatience)144	for {145		select {146		case event, ok := <-r.control.events():147			if !ok {148				r.state = machine.WirelessNoCarrier149				r.message = "the supplicant's control socket closed before the radio associated"150				return151			}152			if line := describeWirelessEvent(r.ifname, event); line != "" {153				fmt.Println(line)154			}155			if state, message, settled := judgeWirelessEvent(event); settled {156				r.state, r.message = state, message157				return158			}159		case <-deadline:160			r.state = machine.WirelessNoCarrier161			r.message = fmt.Sprintf("no access point answered for %s; the supplicant keeps trying", associationPatience)162			fmt.Fprintf(os.Stderr, "liken: wireless: %s: %s\n", r.ifname, r.message)163			return164		}165	}166}167168// controlSocketDir is where the supplicant puts one interface's control169// socket. Each interface gets its own directory, so the socket's name170// inside it is always the interface's name and nothing has to be171// cleaned up between interfaces.172func controlSocketDir(ifname string) string {173	return filepath.Join(wirelessRunDir, ifname, "ctrl")174}175176// writeWirelessConfig puts one interface's generated configuration on177// the tmpfs and reports where. The mode is owner-only, because the file178// holds the network's passphrase.179func writeWirelessConfig(ifname, config string) (string, error) {180	dir := filepath.Join(wirelessRunDir, ifname)181	if err := os.MkdirAll(dir, 0o700); err != nil {182		return "", fmt.Errorf("preparing %s: %w", dir, err)183	}184	path := filepath.Join(dir, "wpa_supplicant.conf")185	if err := os.WriteFile(path, []byte(config), 0o600); err != nil {186		return "", fmt.Errorf("writing %s: %w", path, err)187	}188	return path, nil189}190191// supplicants holds every supplicant this boot started, so the shutdown192// can stop them before it signals the rest of the machine. It is a193// package variable for the same reason the death registry is: init is194// one program, and these are its children.195//196// The lock exists because background radios register their197// supplicants from the radio component's goroutine while the boot198// and the shutdown run on others. The stopping flag latches when the199// shutdown runs, and no supplicant starts after it: a radio that200// settles late during a reboot must not leave a process the shutdown201// already finished stopping.202var (203	supplicantsMu       sync.Mutex204	supplicants         []*supplicantProcess205	supplicantsStopping bool206)207208// registerSupplicant adds one supervised supplicant to the list the209// shutdown stops, and reports false once the shutdown has run. The210// check and the append happen under one hold of the lock, so a211// supplicant either makes the list the shutdown will stop, or is212// refused; there is no window between the two.213func registerSupplicant(p *supplicantProcess) bool {214	supplicantsMu.Lock()215	defer supplicantsMu.Unlock()216	if supplicantsStopping {217		return false218	}219	supplicants = append(supplicants, p)220	return true221}222223// supplicantProcess is one supervised supplicant: the loop that keeps224// it running, and the channels that stop it.225type supplicantProcess struct {226	ifname string227	stop   chan struct{}228	done   chan struct{}229230	// The restart loop holds the control client because a supplicant231	// keeps its list of attached clients in memory. The process that232	// dies takes liken's attachment with it, and the new process233	// reports to nobody until it is asked again.234	control *wpaControl235236	// Restart pacing, in the machine plane's own form: a start that237	// keeps failing waits twice as long each time, up to a cap, and a238	// process that ran for a while before dying starts the pacing over.239	// They are fields rather than constants so a test can run the loop240	// without waiting out real seconds.241	backoff    time.Duration242	maxBackoff time.Duration243}244245// superviseSupplicant starts the supplicant for one interface, keeps it246// running for the life of the boot, and returns a client attached to247// its control socket.248//249// This is the k3s supervisor's second resident, on the supervisor's250// own terms: the reaper stays the only caller of wait, and this loop251// subscribes to the death registry exactly as superviseK3s does. The252// association dies with the process, so a restart is the only repair.253// The address stays on the interface across the restart, which is254// what makes the outage seconds rather than a reboot.255func superviseSupplicant(ifname, config string) (*wpaControl, error) {256	dir := controlSocketDir(ifname)257	if err := os.MkdirAll(dir, 0o700); err != nil {258		return nil, fmt.Errorf("preparing %s: %w", dir, err)259	}260261	proc, err := startSupplicant(ifname, config)262	if err != nil {263		return nil, err264	}265	control, err := attachWPAControl(filepath.Join(dir, ifname))266	if err != nil {267		// A supplicant that never attached is not supervised, so it is268		// stopped here and its death awaited. The start marked its pid269		// as expected, and a death that nothing awaits stays parked in270		// the registry. A later child that reuses the pid, k3s in the271		// worst case, would then read that status at once as its own272		// death.273		endSupplicant(ifname, proc)274		return nil, fmt.Errorf("attaching to the supplicant on %s: %w", ifname, err)275	}276277	p := &supplicantProcess{278		ifname: ifname, control: control,279		stop: make(chan struct{}), done: make(chan struct{}),280		backoff: time.Second, maxBackoff: 30 * time.Second,281	}282	// A radio that settles after the shutdown began gets its283	// supplicant stopped, not supervised. The process just started,284	// so it is stopped by the same deliberate path the shutdown285	// uses, and the caller reports the refusal.286	if !registerSupplicant(p) {287		control.close()288		endSupplicant(ifname, proc)289		return nil, fmt.Errorf("the machine is stopping its supplicants; the one on %s was stopped again", ifname)290	}291	plane.start("the supplicant on "+ifname, func(ctx context.Context) error {292		p.run(ctx, proc, config)293		return nil294	})295	return control, nil296}297298// endSupplicant stops a supplicant that no supervision loop holds,299// and awaits its death, so the registry keeps nothing for its pid.300func endSupplicant(ifname string, proc runningSupplicant) {301	died := make(chan unix.WaitStatus, 1)302	go func() { died <- deaths.await(proc.pid) }()303	stopSupplicant(ifname, proc, died)304}305306// supplicantOutcome is why one wait inside the supervision loop ended.307type supplicantOutcome int308309const (310	supplicantDied   supplicantOutcome = iota // the process exited on its own311	supplicantEnded                           // the loop was asked to finish312	supplicantWaited                          // the delay elapsed and nothing else happened313)314315// await waits for whichever comes first: the process exiting, the loop316// being asked to finish, or a delay elapsing. A nil died channel says317// there is no process to wait on, and a nil delay says wait as long as318// it takes.319//320// The stop step is conditional because between a death and the next321// successful start there is no process, so a stop arriving in that322// window has nothing to signal. Stopping a process whose exit the323// reaper already collected would wait on an exit that can never be324// reported a second time.325func (p *supplicantProcess) await(ctx context.Context, proc runningSupplicant,326	died <-chan unix.WaitStatus, delay <-chan time.Time) supplicantOutcome {327	select {328	case status := <-died:329		fmt.Printf("liken: wireless: the supplicant on %s exited (%s)\n", p.ifname, describeExit(status))330		return supplicantDied331	case <-p.stop:332		if died != nil {333			stopSupplicant(p.ifname, proc, died)334		}335		return supplicantEnded336	case <-ctx.Done():337		if died != nil {338			stopSupplicant(p.ifname, proc, died)339		}340		return supplicantEnded341	case <-delay:342		return supplicantWaited343	}344}345346// next doubles the restart delay, up to the cap. A start that keeps347// failing then retries at a steady pace rather than faster and faster.348func (p *supplicantProcess) next(backoff time.Duration) time.Duration {349	if backoff >= p.maxBackoff {350		return p.maxBackoff351	}352	return backoff * 2353}354355// run keeps one supplicant running for the life of the boot, modelled on356// superviseK3s: the reaper reports each death, the loop reports it on the357// console, and the next start waits out a backoff that doubles while the358// failures stay fast.359//360// The loop has exactly two states, and they are separate functions361// because they wait on different things. While a process runs, watch362// holds it. Between processes, restart holds it, and there is no363// process to signal or to wait on until restart hands back a live one.364func (p *supplicantProcess) run(ctx context.Context, proc runningSupplicant, config string) {365	defer close(p.done)366	backoff := p.backoff367	attached := true368	for {369		started := time.Now()370		died := make(chan unix.WaitStatus, 1)371		// The pid is passed in rather than captured, because the loop372		// gives the variable the next process's pid before this373		// goroutine is guaranteed to have read it.374		go func(pid int) { died <- deaths.await(pid) }(proc.pid)375376		if p.watch(ctx, proc, died, attached) == supplicantEnded {377			return378		}379		// A supplicant that ran for a while and then died is a fresh380		// failure, not a continuing one, so its restart starts the381		// pacing over.382		if time.Since(started) > time.Minute {383			backoff = p.backoff384		}385386		next, ok := p.restart(ctx, config, &backoff)387		if !ok {388			return389		}390		// A new process arrives unattached. The list of attached391		// clients lives in the process that died, so the new one392		// reports to nobody until it is asked, and a parked boot is393		// waiting on exactly those reports.394		proc, attached = next, false395	}396}397398// watch holds one running supplicant. It returns supplicantDied when the399// process exited on its own, and supplicantEnded when the loop was asked400// to finish, in which case it has already stopped the process.401//402// While the event stream is detached, watch keeps asking for it, on the403// same backoff the restarts use. A single failed attach would otherwise404// leave the supplicant running and reporting to nobody, which reads on405// the console and in the status exactly like a radio that went quiet.406func (p *supplicantProcess) watch(ctx context.Context, proc runningSupplicant,407	died <-chan unix.WaitStatus, attached bool) supplicantOutcome {408	backoff := p.backoff409	for {410		if attached {411			return p.await(ctx, proc, died, nil)412		}413		// The attach runs in a goroutine so that waiting for the414		// supplicant's socket to appear cannot delay a stop.415		attempt := make(chan error, 1)416		go func() { attempt <- p.control.attach() }()417		select {418		case err := <-attempt:419			if err == nil {420				fmt.Printf("liken: wireless: the event stream on %s is attached again\n", p.ifname)421				attached = true422				continue423			}424			fmt.Fprintf(os.Stderr, "liken: wireless: %s: %v\n", p.ifname, err)425		case status := <-died:426			fmt.Printf("liken: wireless: the supplicant on %s exited (%s)\n", p.ifname, describeExit(status))427			return supplicantDied428		case <-p.stop:429			stopSupplicant(p.ifname, proc, died)430			return supplicantEnded431		case <-ctx.Done():432			stopSupplicant(p.ifname, proc, died)433			return supplicantEnded434		}435		backoff = p.next(backoff)436		if outcome := p.await(ctx, proc, died, time.After(withJitter(backoff))); outcome != supplicantWaited {437			return outcome438		}439	}440}441442// restart gets a supplicant running again, however many attempts that443// takes. It returns a live process, or false when the loop444// was asked to finish. It never returns after a failed start, because a445// start that failed leaves the machine with no supplicant at all, and446// the radio then has nothing keeping its session alive.447func (p *supplicantProcess) restart(ctx context.Context, config string, backoff *time.Duration) (runningSupplicant, bool) {448	for {449		*backoff = p.next(*backoff)450		delay := withJitter(*backoff)451		fmt.Printf("liken: wireless: restarting the supplicant on %s in %s\n", p.ifname, delay.Round(time.Millisecond))452		if p.await(ctx, runningSupplicant{}, nil, time.After(delay)) == supplicantEnded {453			return runningSupplicant{}, false454		}455		proc, err := startSupplicant(p.ifname, config)456		if err == nil {457			return proc, true458		}459		fmt.Fprintf(os.Stderr, "liken: wireless: %v\n", err)460	}461}462463// runningSupplicant is one supplicant process: its pid, for the death464// registry and the console, and the way to signal it. The signal goes465// through the os.Process that started the process. The Process holds a466// pidfd, and Go signals through pidfd_send_signal, so a signal to a467// process that the reaper already collected returns os.ErrProcessDone468// and cannot reach a later process that reuses the pid. signal is a469// function so that a test can record the signals the loop sends470// without any real process being signalled.471type runningSupplicant struct {472	pid    int473	signal func(os.Signal) error474}475476// supplicantHandle wraps the os.Process that started a supplicant.477func supplicantHandle(p *os.Process) runningSupplicant {478	return runningSupplicant{pid: p.Pid, signal: p.Signal}479}480481// startSupplicant launches the supplicant for one interface and reports482// the running process. It is a variable holding the real launcher below, so a483// test can script the outcomes of the restart loop without a radio, a484// binary, or a reaper, the same way parkConsole and wirelessRunDir let a485// test stand in for a device and a tmpfs.486var startSupplicant = execSupplicant487488// execSupplicant runs the vendored program.489//490// -i names the interface and -c names the generated file. -D names491// the driver outright rather than letting the program try each one it492// was built with; nl80211 is the only driver in liken's build anyway.493// The program stays in the foreground because the supervisor above is494// what keeps it running, and a daemonized process would exit495// immediately and put its own child out of the death registry's496// reach.497func execSupplicant(ifname, config string) (runningSupplicant, error) {498	cmd := exec.Command(supplicantBinary, "-i", ifname, "-c", config, "-D", "nl80211")499	cmd.Stdout = &lineWriter{dest: console, prefix: "wpa | "}500	cmd.Stderr = &lineWriter{dest: console, prefix: "wpa | "}501	if err := deaths.start(cmd); err != nil {502		return runningSupplicant{}, fmt.Errorf("starting the supplicant on %s: %w", ifname, err)503	}504	fmt.Printf("liken: wireless: the supplicant on %s started (pid %d)\n", ifname, cmd.Process.Pid)505	return supplicantHandle(cmd.Process), nil506}507508// stopSupplicant asks one supplicant to exit and waits for the reaper509// to confirm it, escalating when it does not.510//511// The supplicant is stopped deliberately rather than left to the512// general kill. On SIGTERM it deauthenticates from the access point513// and takes the interface down, which leaves the radio in a state the514// next boot can start from. And the shutdown's kill(-1) would race515// the restart loop above into starting a supplicant it is about to516// kill.517func stopSupplicant(ifname string, proc runningSupplicant, died <-chan unix.WaitStatus) {518	fmt.Printf("liken: wireless: stopping the supplicant on %s (pid %d)\n", ifname, proc.pid)519	_ = proc.signal(unix.SIGTERM)520	select {521	case <-died:522	case <-time.After(5 * time.Second):523		fmt.Fprintf(os.Stderr, "liken: wireless: the supplicant on %s ignored SIGTERM for 5s; killing it\n", ifname)524		_ = proc.signal(unix.SIGKILL)525		<-died526	}527}528529// stopSupplicants ends every supervised supplicant. The shutdown calls530// it before it signals the rest of the machine, so the restart loops are531// finished before kill(-1) reaches anything.532func stopSupplicants() {533	// The latch and the list move under one hold of the lock: after534	// this block, every registered supplicant is in running, and535	// every later register is refused.536	supplicantsMu.Lock()537	supplicantsStopping = true538	running := supplicants539	supplicants = nil540	supplicantsMu.Unlock()541542	for _, p := range running {543		close(p.stop)544	}545	for _, p := range running {546		select {547		case <-p.done:548		case <-time.After(10 * time.Second):549			fmt.Fprintf(os.Stderr, "liken: wireless: the supplicant on %s did not stop; going on without it\n", p.ifname)550		}551	}552}553554// routeLookup answers which interface a packet toward one address would555// leave by. It is a function value so that the park decision below is a556// pure function that a test can drive with no netlink and no kernel.557type routeLookup func(net.IP) (string, error)558559// routeVia asks the kernel's routing table the same question the560// kernel would ask when a packet is sent.561func routeVia(dst net.IP) (string, error) {562	routes, err := netlink.RouteGet(dst)563	if err != nil {564		return "", err565	}566	if len(routes) == 0 {567		return "", fmt.Errorf("no route toward %s", dst)568	}569	link, err := netlink.LinkByIndex(routes[0].LinkIndex)570	if err != nil {571		return "", err572	}573	return link.Attrs().Name, nil574}575576// endpointAddress reads the literal address out of a cluster endpoint,577// for example https://10.10.0.1:6443. It reports false for an endpoint578// that names no address, which covers both an empty endpoint and a579// name that only DNS could resolve.580func endpointAddress(endpoint string) (net.IP, bool) {581	if endpoint == "" {582		return nil, false583	}584	u, err := url.Parse(endpoint)585	if err != nil {586		return nil, false587	}588	ip := net.ParseIP(u.Hostname())589	return ip, ip != nil590}591592// parkDecision answers the plan's question: does this boot stop and593// wait, or does it go on degraded? It returns the radio to wait on and594// the reason to print, or nil when the boot goes on.595//596// The rule is the plan's: only a deterministic failure qualifies, and597// the boot goes on whenever any settled interface gives a route598// toward the cluster's endpoint. A machine whose only path was the599// failed radio waits, because a machine that cannot reach its cluster600// has nothing to do anyway.601//602// An endpoint that names no address never parks. A DNS name needs603// DNS, and DNS needs the network that is in question, so nothing604// reliable can be decided from it. A machine alone is its own cluster605// and declares no endpoint at all. The plan's bias in both cases is606// to boot rather than to wait.607//608// A route that leaves by loopback counts as reaching the endpoint. A609// leader's endpoint is its own address, so the kernel answers610// loopback, and a leader must never park on a radio it does not need.611func parkDecision(conns []*connection, endpoint string, route routeLookup) (*radio, string) {612	var failed *radio613	for _, conn := range conns {614		if failed == nil && conn.radio != nil && conn.radio.deterministic() {615			failed = conn.radio616		}617	}618	if failed == nil {619		return nil, ""620	}621	routed, why := routeToward(conns, endpoint, route)622	if routed {623		return nil, ""624	}625	reason := fmt.Sprintf("liken: wireless: %s cannot join %s: %s", failed.ifname, failed.ssid, failed.message)626	return failed, fmt.Sprintf("%s; %s, so this boot waits here", reason, why)627}628629// routeToward answers the one question both the park and the pass630// split ask: can this machine, on the interfaces that settled, send631// a packet toward its cluster's endpoint? The false answers carry632// the reason, in the words the park has always printed.633//634// An endpoint that names no address counts as routed. A DNS name635// needs DNS, and DNS needs the network in question, so nothing636// reliable can be decided from it; a machine alone declares no637// endpoint at all. The bias in both cases is the plan's: boot rather638// than wait. A route that leaves by loopback also counts, because a639// leader's endpoint is its own address.640func routeToward(conns []*connection, endpoint string, route routeLookup) (bool, string) {641	settled := map[string]bool{}642	for _, conn := range conns {643		if conn.addr != nil {644			settled[conn.ifname] = true645		}646	}647	if len(settled) == 0 {648		return false, "no other interface has an address"649	}650	dst, ok := endpointAddress(endpoint)651	if !ok {652		return true, ""653	}654	ifname, err := route(dst)655	if err != nil {656		return false, fmt.Sprintf("nothing on this machine has a route toward %s", dst)657	}658	if ifname == "lo" || settled[ifname] {659		return true, ""660	}661	return false, fmt.Sprintf("the route toward %s leaves by %s, which never came up", dst, ifname)662}663664// parkConsole is the device a park writes its reason to. It is a665// variable so a test can point the hold at a file of its own.666var parkConsole = consoleDevice667668// park holds the boot on a radio that cannot join, and releases it the669// moment the supplicant reports that the radio did.670//671// The message goes out twice: the log copy travels the kmsg pipeline672// to every console the command line named, and the direct copy673// reaches the one device a person is looking at. Unlike the674// installer's holds, this hold reads no key. The thing that ends it675// is an event on the socket rather than a keypress, so a fix on the676// network side resumes the boot with nobody at the machine.677func park(r *radio, reason string) {678	fmt.Fprintln(os.Stderr, reason)679	fmt.Fprintln(os.Stderr, "liken: wireless: the supplicant keeps trying; this boot goes on the moment the radio joins")680	if device, err := os.OpenFile(parkConsole, os.O_RDWR, 0); err == nil {681		fmt.Fprintln(device, reason)682		fmt.Fprintln(device, "liken: wireless: the supplicant keeps trying; this boot goes on the moment the radio joins")683		device.Close()684	}685686	// A park with no supplicant waits anyway. The configuration687	// itself was refused, so no process is running to report a688	// repair, and nothing this machine can do would produce one; the689	// fix is the install media. PID 1 must not exit, which is why690	// this waits rather than returning.691	if r.control == nil {692		fmt.Fprintln(os.Stderr, "liken: wireless: no supplicant is running for this interface; the passphrase on the install media is the fix")693		for {694			time.Sleep(time.Hour)695		}696	}697698	for event := range r.control.events() {699		if line := describeWirelessEvent(r.ifname, event); line != "" {700			fmt.Println(line)701		}702		state, message, settled := judgeWirelessEvent(event)703		if !settled {704			continue705		}706		r.state, r.message = state, message707		if state == machine.WirelessConnected {708			return709		}710	}711	// A closed stream ends the hold. The events are the only thing712	// that can release this wait, and a boot stopped forever with no713	// way to report why is worse than a boot that goes on degraded.714	fmt.Fprintf(os.Stderr, "liken: wireless: %s: the supplicant's control socket closed; going on without the radio\n", r.ifname)715}
init/worldreport.go 71.4%
1package main23// The essential mounts and the world report: the first things init4// sets up, and how init reports what it did.56import (7	"fmt"8	"os"9	"strings"1011	"golang.org/x/sys/unix"12)1314// This code declares the essential filesystems as a table, rather15// than as a sequence of calls. The table is data, and the world16// report below prints what actually got mounted, so a reader can17// compare the two easily.18type mount struct {19	source string20	target string21	fstype string22	flags  uintptr23}2425var essentials = []mount{26	// /proc is two things at once: a directory per running process,27	// and the kernel's interface for settings and information, such28	// as /proc/sys and /proc/cmdline. Almost every tool that29	// inspects the system reads /proc, including the world report30	// below, and k3s does not start without it.31	{"proc", "/proc", "proc", unix.MS_NOSUID | unix.MS_NOEXEC | unix.MS_NODEV},3233	// /sys exposes the kernel's object model as a filesystem: every34	// device, driver, and bus. cgroup2 mounts beneath /sys later;35	// Kubernetes uses cgroup2 to account for and limit every36	// container.37	{"sysfs", "/sys", "sysfs", unix.MS_NOSUID | unix.MS_NOEXEC | unix.MS_NODEV},3839	// devtmpfs is the kernel's own device catalog. When this mount40	// happens, a device node appears for every device the kernel has41	// registered, and the kernel maintains these nodes itself. On a42	// machine with known hardware, devtmpfs replaces the entire udev43	// system.44	{"devtmpfs", "/dev", "devtmpfs", unix.MS_NOSUID},4546	// /run is the standard location for runtime state: sockets, PIDs,47	// and locks. containerd and k3s use hardcoded paths under it, and48	// so do the facts tree and the supplicant's generated49	// configuration.50	//51	// This mount is an essential rather than one of the mounts that52	// prepareForK3s makes. A wireless interface comes up long before53	// k3s does, and its generated configuration and control socket54	// live under /run. A tmpfs mounted over /run later would hide55	// both from the supplicant restart that has to read them again.56	{"tmpfs", "/run", "tmpfs", unix.MS_NOSUID | unix.MS_NODEV},57}5859func mountEssentials() {60	for _, m := range essentials {61		// The initramfs root is plain RAM and is freely writable, so62		// init creates its own mountpoints. The image does not need63		// to ship empty directories.64		if err := os.MkdirAll(m.target, 0o755); err != nil {65			fmt.Fprintf(os.Stderr, "liken: mkdir %s: %v\n", m.target, err)66			continue67		}68		// init reports failures but does not treat them as fatal. A69		// partial environment that can still print its report is far70		// more useful than a kernel panic, because the console is71		// where debugging happens.72		if err := unix.Mount(m.source, m.target, m.fstype, m.flags, ""); err != nil {73			fmt.Fprintf(os.Stderr, "liken: mount %s on %s: %v\n", m.fstype, m.target, err)74		}75	}76}7778// The world report is liken's substitute for an interactive shell.79// init answers, on the console at every boot, the same questions an80// operator would otherwise answer by exploring a shell prompt. When81// something goes wrong, the usual first step is to extend the report82// to answer the new question.83func worldReport() {84	// Uname fills fixed-size byte arrays, rather than returning85	// strings, because it is a thin wrapper over the raw syscall,86	// and the kernel ABI uses fixed-size buffers. ByteSliceToString87	// finds the NUL terminator in each array.88	var u unix.Utsname89	if err := unix.Uname(&u); err == nil {90		fmt.Printf("liken: kernel %s (%s)\n",91			unix.ByteSliceToString(u.Release[:]),92			unix.ByteSliceToString(u.Machine[:]))93	}9495	reportFirmware()9697	// The kernel command line is how the outside world sets98	// parameters for a boot. It is where rdinit= points at init, and99	// it is the way to pass any fact a machine needs before it100	// has a filesystem.101	if cmdline, err := os.ReadFile(cmdlinePath); err == nil {102		fmt.Printf("liken: cmdline: %s\n", strings.TrimSpace(string(cmdline)))103	}104105	// /proc/self/mounts is the kernel's authoritative mount table.106	// If mountEssentials succeeded, its results appear here.107	if mounts, err := os.ReadFile("/proc/self/mounts"); err == nil {108		fmt.Print("liken: mounts:\n")109		for line := range strings.SplitSeq(strings.TrimSpace(string(mounts)), "\n") {110			fmt.Printf("liken:   %s\n", line)111		}112	}113114	// A populated /dev shows that devtmpfs created the device nodes,115	// with no udev involved.116	if entries, err := os.ReadDir("/dev"); err == nil {117		fmt.Printf("liken: /dev has %d entries\n", len(entries))118	}119120	reportBlockDevices()121}
init/wpaconfig.go 97.7%
1package main23// Where the passphrase comes from, and how a network's name and its4// passphrase reach the supplicant without either one being able to5// write a configuration directive of its own.6//7// The passphrase has two homes. The image carries one file for each8// network under /etc/liken/psk, beside the join token, because a9// passphrase is the same class of fact: a cluster credential the10// machine needs before the cluster can give it anything. machineState11// may hold a staged copy, and the read order is staged then image,12// the same order the cluster document resolves in. Nothing writes the13// staged copy today; rotation, when it arrives, becomes a writer of14// that copy and changes nothing here (plans/completed/62-wifi.md).15//16// The generated file uses only value forms that carry no syntax. The17// supplicant's parser reads a value as hex whenever the value does18// not start with a quote, and an unknown line makes the whole file19// fail to parse rather than being skipped. A value that could write a20// line of its own could therefore disable the network block or turn21// the encryption off. Every value liken writes is hex, which holds no22// quote, no space, and no comment character.2324import (25	"crypto/pbkdf2"26	"crypto/sha1"27	"encoding/hex"28	"errors"29	"fmt"30	"io/fs"31	"os"32	"path/filepath"33	"strings"3435	"github.com/liken-sh/liken/liken/machine"36)3738// imagePassphraseDir is where the image carries one passphrase file for39// each network. It is a variable so tests can point it at a directory40// of their own.41var imagePassphraseDir = "/etc/liken/psk"4243// stagedPassphraseDir is the passphrase's staged home on machineState.44func stagedPassphraseDir(stateRoot string) string {45	return filepath.Join(stateRoot, "psk")46}4748// The passphrase length WPA2 itself states. liken enforces the range49// even though the hex form it writes has no length rule, because the50// access point derives its own key from the same bytes under the same51// rule. A passphrase outside this range is one no access point could52// hold, and refusing it here puts the reason on the console instead53// of into a failed handshake nobody can read.54const (55	minPassphraseBytes = 856	maxPassphraseBytes = 6357)5859// readPassphrase resolves one network's passphrase: the staged copy on60// machineState first, then the image's file. It reports where the61// passphrase came from, so the console can say which one this boot used.62//63// A missing file is a decision, not a wait. The manifest declared a64// network that needs a key, no file on this machine holds one, and no65// amount of retrying produces one, so this failure joins WRONG_KEY66// under the park rule.67func readPassphrase(stateRoot, ssid string) (passphrase, source string, err error) {68	staged := filepath.Join(stagedPassphraseDir(stateRoot), ssid)69	image := filepath.Join(imagePassphraseDir, ssid)70	for _, path := range []string{staged, image} {71		raw, err := os.ReadFile(path)72		if errors.Is(err, fs.ErrNotExist) {73			continue74		}75		if err != nil {76			return "", "", fmt.Errorf("reading the passphrase at %s: %w", path, err)77		}78		return trimPassphraseFile(string(raw)), path, nil79	}80	return "", "", fmt.Errorf("no passphrase for %q; the image carries one file for each network at %s/<ssid>, and this machine has none",81		ssid, imagePassphraseDir)82}8384// trimPassphraseFile removes the line ending that a text editor85// leaves on the last line of a file. Exactly one line ending goes and86// nothing else does: a trailing space is a legal character in a WPA287// passphrase, so trimming all trailing whitespace would silently88// change the key, while the one final newline was never part of the89// passphrase at all.90func trimPassphraseFile(raw string) string {91	trimmed := strings.TrimSuffix(raw, "\n")92	return strings.TrimSuffix(trimmed, "\r")93}9495// checkPassphrase refuses a passphrase the generator has no safe96// rendering for, and says which rule refused it. The range is WPA2's97// own. A byte outside printable ASCII in a passphrase file is nearly98// always a stray line ending or an encoding accident, not a key any99// access point holds.100func checkPassphrase(ssid, passphrase string) error {101	if n := len(passphrase); n < minPassphraseBytes || n > maxPassphraseBytes {102		return fmt.Errorf("the passphrase for %q is %d bytes; WPA2 passphrases are %d to %d printable characters",103			ssid, n, minPassphraseBytes, maxPassphraseBytes)104	}105	for i := 0; i < len(passphrase); i++ {106		if b := passphrase[i]; b < 0x20 || b > 0x7e {107			return fmt.Errorf("the passphrase for %q holds the byte %#02x at position %d; a WPA2 passphrase holds printable characters only",108				ssid, b, i)109		}110	}111	return nil112}113114// derivePMK derives the key the WPA2 four-way handshake actually115// uses, so the supplicant never needs the passphrase for that half.116// The 802.11 derivation is PBKDF2 over HMAC-SHA1, with the network's117// name as the salt and 4096 rounds. The supplicant accepts the result118// as 64 hex digits, a form no passphrase can break out of. The salt119// binds the result to the network's name, which is why a change to120// either one re-derives it.121func derivePMK(ssid, passphrase string) string {122	key, err := pbkdf2.Key(sha1.New, passphrase, []byte(ssid), 4096, 32)123	if err != nil {124		// The parameters are constants, so the only error this call125		// can report is one this code itself would have to126		// introduce. An empty result fails the supplicant's parse,127		// which fails closed.128		return ""129	}130	return hex.EncodeToString(key)131}132133// wirelessConfig renders the supplicant's whole configuration for one134// interface. It refuses rather than render a value it cannot write135// safely, because a machine with no shell cannot be asked afterwards.136func wirelessConfig(w machine.WirelessSpec, stateRoot, ctrlDir string) (string, error) {137	var b strings.Builder138139	// The one global line. The control directory lives in the file140	// rather than on the command line because the -C flag overrides141	// this line, upstream's own help text describes that override142	// backwards, and one source of truth is worth more than a flag.143	fmt.Fprintf(&b, "ctrl_interface=%s\n", ctrlDir)144	b.WriteString("network={\n")145146	// The name is written as hex because 802.11 lets a network's147	// name carry any octet. The parser reads an unquoted value as148	// hex, and hex cannot hold the quote, the space, or the comment149	// character that would otherwise end the line early.150	fmt.Fprintf(&b, "\tssid=%s\n", hex.EncodeToString([]byte(w.SSID)))151152	if w.SecurityOrDefault() == machine.WirelessOpen {153		// NONE is the supplicant's word for a network with no154		// encryption at all.155		b.WriteString("\tkey_mgmt=NONE\n")156		b.WriteString("}\n")157		return b.String(), nil158	}159160	passphrase, source, err := readPassphrase(stateRoot, w.SSID)161	if err != nil {162		return "", err163	}164	if err := checkPassphrase(w.SSID, passphrase); err != nil {165		return "", err166	}167	fmt.Printf("liken: wireless: the passphrase for %s comes from %s\n", w.SSID, source)168169	// All three key managements are named at once because a home170	// access point today commonly offers WPA2 and WPA3 together. The171	// supplicant picks the strongest one the access point172	// advertises. Naming only one would refuse half the networks a173	// machine may meet.174	b.WriteString("\tkey_mgmt=WPA-PSK WPA-PSK-SHA256 SAE\n")175176	// Protected management frames are optional (1) rather than off177	// or required. WPA3 requires them and WPA2 does not: required178	// (2) refuses every WPA2-only access point, off (0) breaks SAE179	// against an access point that requires them, and optional lets180	// the supplicant negotiate whichever each access point offers.181	b.WriteString("\tieee80211w=1\n")182183	// Two credential lines carry one passphrase. WPA2 uses the184	// derived key, and WPA3's SAE uses the passphrase itself. The185	// derived form leaves the supplicant no passphrase to hand SAE,186	// so a configuration with only psk= would join a WPA2 access187	// point and fail against a WPA3 one.188	fmt.Fprintf(&b, "\tpsk=%s\n", derivePMK(w.SSID, passphrase))189	fmt.Fprintf(&b, "\tsae_password=%s\n", hex.EncodeToString([]byte(passphrase)))190191	b.WriteString("}\n")192	return b.String(), nil193}
init/wpactrl.go 91.5%
1package main23// The supplicant's control interface is plain text over one UNIX4// datagram socket. liken carries this small client rather than the5// wpa_cli program because the whole of what init needs is ATTACH and6// the events that follow. Those events are the only thing on the7// machine that can tell a refused passphrase apart from an access8// point that never answered.910import (11	"fmt"12	"net"13	"os"14	"path/filepath"15	"strings"16	"sync"17	"time"1819	"github.com/liken-sh/liken/liken/machine"20)2122// The size of one message. Upstream's own client reads into a buffer23// of this size. An event is a short line, but a datagram longer than24// the buffer loses its tail rather than splitting, which is why the25// buffer is generous.26const wpaMessageMax = 40962728// How long the attach waits for the socket to appear. The supplicant29// creates its socket after it starts, so the process id comes back30// before the path exists. This is a race with a starting process on31// this machine, not a network wait, which is why the allowance is32// short.33const wpaSocketPatience = 10 * time.Second3435// wpaEvent is one message the supplicant pushed to an attached client.36type wpaEvent struct {37	// level is the priority the supplicant stamped on the message,38	// from its own debug levels. A monitor receives everything at39	// info and above by default, and info is the level every join40	// event uses.41	level int4243	// name is the event's own word, for example CTRL-EVENT-CONNECTED.44	name string4546	// fields are the key=value pairs the message carries.47	fields map[string]string4849	// text is the whole message after the priority, for the console.50	text string51}5253// parseWPAEvent reads one message off the socket.54//55// The wire format, exactly: the supplicant prefixes each message with56// its priority in angle brackets, and the rest is the event's name57// and then key=value pairs. The global socket adds an IFNAME= prefix58// ahead of the priority; the per-interface socket liken uses does59// not. A value may be quoted, which is how an SSID with a space in it60// arrives.61func parseWPAEvent(message string) (wpaEvent, bool) {62	text := strings.TrimRight(message, "\r\n")63	if rest, found := strings.CutPrefix(text, "IFNAME="); found {64		_, text, _ = strings.Cut(rest, " ")65	}66	level := -167	if strings.HasPrefix(text, "<") {68		if end := strings.Index(text, ">"); end > 0 {69			if _, err := fmt.Sscanf(text[1:end], "%d", &level); err != nil {70				level = -171			}72			text = text[end+1:]73		}74	}75	if text == "" {76		return wpaEvent{}, false77	}78	name, _, _ := strings.Cut(text, " ")79	return wpaEvent{level: level, name: name, fields: parseWPAFields(text), text: text}, true80}8182// parseWPAFields collects the key=value pairs out of one message. A83// value in double quotes may hold spaces, which is how an SSID arrives,84// so the scan joins the tokens of such a value back together.85func parseWPAFields(text string) map[string]string {86	fields := map[string]string{}87	tokens := strings.Fields(text)88	for i := 0; i < len(tokens); i++ {89		key, value, found := strings.Cut(tokens[i], "=")90		// The connected event wraps its pairs in square brackets and91		// every other event does not. A key parsed with a bracket92		// still on it would answer to nothing, so the brackets come93		// off here.94		key = strings.TrimPrefix(key, "[")95		value = strings.TrimSuffix(value, "]")96		if !found || key == "" {97			continue98		}99		if strings.HasPrefix(value, `"`) {100			for !strings.HasSuffix(value, `"`) || len(value) < 2 {101				if i+1 >= len(tokens) {102					break103				}104				i++105				value += " " + tokens[i]106			}107			value = strings.TrimSuffix(strings.TrimPrefix(value, `"`), `"`)108		}109		fields[key] = value110	}111	return fields112}113114// The reasons the supplicant gives for refusing a network, and which115// of them this boot treats as final. The supplicant emits exactly116// four. WRONG_KEY is the access point rejecting the handshake, and117// NO_PSK_AVAILABLE is a configuration with no key in it at all; both118// are decisions. The other two describe an association that failed,119// which the next attempt may well complete, so they stay transient.120var deterministicWPAReasons = map[string]string{121	"WRONG_KEY": "the access point refused the passphrase (WRONG_KEY); " +122		"until rotation exists, a wrong passphrase is fixed by the install media, not in place",123	"NO_PSK_AVAILABLE": "the supplicant has no key for this network (NO_PSK_AVAILABLE)",124}125126// judgeWirelessEvent decides what one event means for the join. It127// reports the verdict and whether the event settled the question at128// all. Most events settle nothing: a scan that found nothing and an129// access point that went away are both states that the next attempt may130// leave behind.131func judgeWirelessEvent(event wpaEvent) (machine.WirelessState, string, bool) {132	switch event.name {133	case "CTRL-EVENT-CONNECTED":134		return machine.WirelessConnected, "", true135	case "CTRL-EVENT-SSID-TEMP-DISABLED":136		if message, final := deterministicWPAReasons[event.fields["reason"]]; final {137			return machine.WirelessWrongKey, message, true138		}139	}140	return "", "", false141}142143// describeWirelessEvent renders one event as a console line, or the144// empty string for an event this boot has nothing to say about. On a145// machine with no shell, these lines are the whole account of what the146// radio did.147func describeWirelessEvent(ifname string, event wpaEvent) string {148	switch event.name {149	case "CTRL-EVENT-CONNECTED":150		return fmt.Sprintf("liken: wireless: %s associated", ifname)151	case "CTRL-EVENT-DISCONNECTED":152		return fmt.Sprintf("liken: wireless: %s lost the access point (reason %s)",153			ifname, orUnknown(event.fields["reason"]))154	case "CTRL-EVENT-SSID-TEMP-DISABLED":155		return fmt.Sprintf("liken: wireless: %s: the access point refused the join (%s), retrying in %ss",156			ifname, orUnknown(event.fields["reason"]), orUnknown(event.fields["duration"]))157	case "CTRL-EVENT-SCAN-RESULTS":158		return fmt.Sprintf("liken: wireless: %s finished a scan", ifname)159	case "CTRL-EVENT-NETWORK-NOT-FOUND":160		return fmt.Sprintf("liken: wireless: %s found no access point for its network", ifname)161	case "CTRL-EVENT-ASSOC-REJECT", "CTRL-EVENT-AUTH-REJECT":162		return fmt.Sprintf("liken: wireless: %s: %s", ifname, event.text)163	}164	return ""165}166167// wpaControl is an attached client on one supplicant's control socket.168//169// wpaControl is an attached client on one supplicant's control170// socket. The client binds a path of its own because the socket is a171// datagram socket: the supplicant answers by sending to the address172// the request arrived from, and an unbound socket has no address to173// answer to.174type wpaControl struct {175	socket string176	local  string177	out    chan wpaEvent178179	// patience is how long attach waits for the socket to appear. It is180	// a field rather than the constant so a test can drive the restart181	// loop's re-attach without waiting out a real boot's allowance.182	patience time.Duration183184	mu     sync.Mutex185	conn   *net.UnixConn186	closed bool187}188189// attachWPAControl opens the supplicant's control socket and asks for190// the stream of unsolicited events.191func attachWPAControl(socket string) (*wpaControl, error) {192	c := &wpaControl{193		socket: socket,194		local:  filepath.Join(filepath.Dir(filepath.Dir(socket)), "client"),195		// The channel is buffered and a full channel drops. Nothing196		// reads these events once the boot goes on, but the197		// supplicant keeps reporting for the life of the machine,198		// and a blocking send would stall the reader in init for199		// good.200		out:      make(chan wpaEvent, 64),201		patience: wpaSocketPatience,202	}203	if err := c.attach(); err != nil {204		return nil, err205	}206	return c, nil207}208209// waitForWPASocket waits for the supplicant to create its socket.210func waitForWPASocket(socket string, patience time.Duration) error {211	deadline := time.Now().Add(patience)212	for {213		if _, err := os.Stat(socket); err == nil {214			return nil215		}216		if time.Now().After(deadline) {217			return fmt.Errorf("the supplicant did not create %s within %s", socket, patience)218		}219		time.Sleep(50 * time.Millisecond)220	}221}222223// attach dials the socket, registers this client as a monitor, and224// starts reading. It is also the repair after a restart: a supplicant225// that died took its list of monitors with it, so the new process must226// be asked again.227func (c *wpaControl) attach() error {228	c.mu.Lock()229	closed := c.closed230	c.mu.Unlock()231	if closed {232		return fmt.Errorf("the control client for %s is closed", c.socket)233	}234	if err := waitForWPASocket(c.socket, c.patience); err != nil {235		return err236	}237	_ = os.Remove(c.local)238	conn, err := net.DialUnix("unixgram",239		&net.UnixAddr{Name: c.local, Net: "unixgram"},240		&net.UnixAddr{Name: c.socket, Net: "unixgram"})241	if err != nil {242		return fmt.Errorf("dialing %s: %w", c.socket, err)243	}244	if err := attachRequest(conn); err != nil {245		conn.Close()246		_ = os.Remove(c.local)247		return err248	}249250	c.mu.Lock()251	old := c.conn252	c.conn = conn253	c.mu.Unlock()254	if old != nil {255		old.Close()256	}257	go c.read(conn)258	return nil259}260261// attachRequest sends ATTACH and reads the answer. The supplicant262// answers OK or FAIL, each on one line.263func attachRequest(conn *net.UnixConn) error {264	if _, err := conn.Write([]byte("ATTACH")); err != nil {265		return fmt.Errorf("sending ATTACH: %w", err)266	}267	if err := conn.SetReadDeadline(time.Now().Add(5 * time.Second)); err != nil {268		return err269	}270	// The answer is looked for rather than simply read. The same271	// socket carries the answers and the events, and this is the one272	// place where the two could arrive in either order. An answer273	// never carries the priority prefix and an event always does,274	// which is how the loop tells them apart.275	buf := make([]byte, wpaMessageMax)276	for {277		n, err := conn.Read(buf)278		if err != nil {279			return fmt.Errorf("reading the answer to ATTACH: %w", err)280		}281		message := string(buf[:n])282		if strings.HasPrefix(message, "<") || strings.HasPrefix(message, "IFNAME=") {283			continue284		}285		if answer := strings.TrimSpace(message); answer != "OK" {286			return fmt.Errorf("the supplicant refused ATTACH: %s", answer)287		}288		return conn.SetReadDeadline(time.Time{})289	}290}291292// read carries one connection's messages onto the event channel, and293// ends when that connection closes.294func (c *wpaControl) read(conn *net.UnixConn) {295	buf := make([]byte, wpaMessageMax)296	for {297		n, err := conn.Read(buf)298		if err != nil {299			return300		}301		event, ok := parseWPAEvent(string(buf[:n]))302		if !ok {303			continue304		}305		select {306		case c.out <- event:307		default:308		}309	}310}311312// events is the stream of messages the supplicant pushes.313func (c *wpaControl) events() <-chan wpaEvent {314	return c.out315}316317// close ends the attachment and the stream with it.318func (c *wpaControl) close() {319	c.mu.Lock()320	defer c.mu.Unlock()321	if c.closed {322		return323	}324	c.closed = true325	if c.conn != nil {326		c.conn.Close()327	}328	_ = os.Remove(c.local)329	close(c.out)330}
kubernetes/apiclient.go 90.0%
1// Package kubernetes lets liken's controllers communicate with the2// Kubernetes API. It provides access to liken's own resources, the3// heartbeat-lease protocol, and pod eviction. The client that sends4// each request is the apiclient package of the shared module5// kubernetes/, at the top of the repository, which the other operators6// use too. The watches are in that module's informer package.7//8// Two programs use this package. The machine operator is a9// privileged DaemonSet that manages the machine it runs on. The10// cluster operator is an unprivileged Deployment that watches the11// fleet.12//13// All code in this package communicates with the API using only14// net/http and encoding/json. It does not use client-go,15// controller-runtime, or code generation. Those libraries hide a16// fact that this package shows: the Kubernetes API is only HTTPS that17// serves JSON, and anything kubectl can do, curl can also do. The one18// part of client-go that liken uses is its reflector, which keeps a19// watch open and recovers it when the stream drops (the shared20// informer package says why). The liken CLI imports this package and21// watches nothing, so this package imports no watch, and the CLI22// links no client-go.23package kubernetes2425// From the pod's credentials, the API is plain REST. Every object26// lives at a predictable URL (/apis/<group>/<version>/<plural>/<name>).27// A GET request reads it. A POST request creates it. A PUT request28// replaces it. Authentication uses one bearer-token header. The29// command kubectl -v=9 prints these exact requests, and the shared30// client sends the same requests directly.3132import (33	"context"34	"errors"35	"fmt"36	"math/rand/v2"37	"net/http"38	"time"3940	"github.com/liken-sh/liken/kubernetes/apiclient"41	"github.com/liken-sh/liken/liken/api"42)4344// serviceAccountDir names the path where kubelet mounts each45// container's API credentials. It is a variable so tests can point46// it at a directory they control, the same seam init's disk code47// leaves open with sysBlock and devRoot.48var serviceAccountDir = apiclient.ServiceAccountDir4950// MachinesPath and ClustersPath name the URLs where our CRDs' objects51// live. Every resource in Kubernetes uses the same URL structure.52// Built-in resources use the legacy /api/v1 root instead of53// /apis/<group>. Both Machines and Clusters are cluster-scoped kinds,54// so their URLs have no /namespaces/<ns>/ segment.55const (56	MachinesPath = "/apis/" + api.APIVersion + "/machines"57	ClustersPath = "/apis/" + api.APIVersion + "/clusters"58)5960// InClusterClient builds an operator's client from the pod's61// ServiceAccount. server, when it is not empty, replaces the address62// that the environment names. The environment names the API63// service's virtual IP. That virtual IP uses iptables NAT: iptables64// pins every new connection to one API server, and a client cannot65// choose which server it uses. Ordinary pods accept this limit. But a66// hostNetwork pod running on a machine that also runs an API server67// (or k3s's health-checked local load balancer over all API servers)68// has a better address to use: its own loopback address, where a dead69// remote server can never strand a connection. The credentials stay70// the same in both cases.71//72// The client answers a 429 at once, with no wait. Each operator's loop73// sets a retry from the 429's Retry-After, and that loop is its retry.74// A pass that waited out a 429 would stretch past what the timing below75// assumes: a Lost verdict or a reboot grant is written from what the76// sweep read when it started.77func InClusterClient(server string) (*apiclient.Client, error) {78	c, err := apiclient.InCluster(apiclient.InClusterOptions{79		ServiceAccountDir: serviceAccountDir,80		Server:            server,81		Timeout:           requestTimeout,82	})83	if err != nil {84		return nil, err85	}86	return c.WithWaitContext(noWait), nil87}8889// Within answers c with every request bound to ctx, for one pass that90// must end by a deadline. apiclient's WithContext also ends the wait91// after a 429 with ctx, which would make each request of the pass wait92// out a 429 for up to ten seconds, so Within keeps the wait that has93// already ended, and the 429 reaches the pass at once.94func Within(c *apiclient.Client, ctx context.Context) *apiclient.Client {95	return c.WithContext(ctx).WithWaitContext(noWait)96}9798// noWait is a context that has already ended, for a client whose wait99// after a 429 must end at once.100var noWait = func() context.Context {101	ctx, cancel := context.WithCancel(context.Background())102	cancel()103	return ctx104}()105106// requestTimeout bounds one request of an in-cluster client, from the107// dial to the last byte of the answer. The shared client's timeouts108// on the dial, the answer's headers, and an idle connection each109// limit a server that stops answering without a FIN or an RST, and110// this bound limits the whole request. Fifteen seconds keeps an111// unlucky pass that hits several dead connections in a row well112// inside the forty-second heartbeat window (see heartbeat.go): a113// client that stalls on another machine's failure must never make114// this machine appear dead. The bound matters to the cluster operator115// too: this client gives up on a write that its leader election116// allowed before a new leader can take the Lease. The API server can117// still commit that write later, so the grant ledger, not this bound,118// keeps a late reboot grant inside the budget119// (cluster-operator/rollout.go). The package120// kubernetes/election gives the numbers as election.RequestTimeout,121// which this package does not import because the machine operator122// must not link client-go's leader election. A test checks that this123// bound stays within it.124const requestTimeout = 15 * time.Second125126// RetryPause sleeps for about five seconds, increased at random by up127// to half again. The random increase matters because a fleet reboots128// together: every machine's operator meets the same not-yet-served129// CRDs and dropped watches at the same moments, and identical retry130// delays would keep every operator retrying at the same moments.131// Randomizing the delay spreads that load over time.132func RetryPause() {133	base := 5 * time.Second134	time.Sleep(base + rand.N(base/2))135}136137// PatchJSON applies a JSON merge patch (RFC 7386). The caller sends138// only the fields to change, and the server merges them into the139// object (a null value deletes a key). This method skips optimistic140// concurrency on purpose: a merge patch carries no resourceVersion,141// so it cannot conflict. Use this method when the caller owns the142// specific fields it changes, such as a cordon flag or an143// annotation, and does not need to check the rest of the object.144//145// The shared client answers a 404 and a 409 as bare errors, with no146// path, because most callers handle them as states. A patch of an147// object that does not exist, or one that conflicts, such as a taints148// patch that states the version it read, answers apiclient.ErrNotFound149// or apiclient.ErrConflict wrapped with the path, so a log line names150// the object. Every other failure already carries the method, the path,151// and the API server's own text.152func PatchJSON(c *apiclient.Client, path string, patch []byte) error {153	err := c.Request(http.MethodPatch, path, "application/merge-patch+json", patch, nil)154	if errors.Is(err, apiclient.ErrNotFound) || errors.Is(err, apiclient.ErrConflict) {155		return fmt.Errorf("PATCH %s: %w", path, err)156	}157	return err158}159160// List sends a GET request for a collection, and unwraps the envelope161// that every Kubernetes list response uses: a <Kind>List object whose162// items field holds the collection. Callers get the items themselves163// and never see the wrapping object.164func List[T any](c *apiclient.Client, path string) ([]T, error) {165	var list struct {166		Items []T `json:"items"`167	}168	if err := c.RequestJSON(http.MethodGet, path, nil, &list); err != nil {169		return nil, err170	}171	return list.Items, nil172}
kubernetes/clusters.go 81.8%
1package kubernetes23// This file reads and reports on Clusters. The machine operator reads4// the one Cluster that its manifest names. The cluster operator lists5// all Clusters, because the list names the Cluster it operates. The6// cluster operator needs no configuration at all: a fleet has7// exactly one Cluster to find.89import (10	"encoding/json"11	"errors"12	"io"13	"net/http"1415	"github.com/liken-sh/liken/kubernetes/apiclient"16	"github.com/liken-sh/liken/liken/cluster"17)1819func GetCluster(c *apiclient.Client, name string) (*cluster.Cluster, error) {20	return apiclient.Get[cluster.Cluster](c, ClustersPath+"/"+name)21}2223// PublishClusterStatus writes through the Cluster's status24// subresource. This is a separate endpoint (…/clusters/<name>/status)25// that updates only the status half of the object. Because of this,26// the single writer of the Cluster's status can never accidentally27// rewrite the spec it acts on, and RBAC can grant access to the two28// halves separately. The write is a PUT request that carries the29// object's resourceVersion. If anything else changed the object in30// the meantime, the server answers with 409 Conflict instead of31// applying the stale copy. The caller then reads the object again on32// its next pass and tries again. This pattern is optimistic33// concurrency, the same contract that PublishStatus uses for34// Machines.35//36// It answers the Cluster that the API server stored, not only its37// resourceVersion. The API server prunes a status field that the38// served CRD does not declare, and answers the write without the field39// and with no error. The cluster operator grants a reboot turn only40// after this answer shows the ledger entry it wrote41// (cluster-operator/rollout.go), so it needs the stored status itself.42// An answer with no body answers no Cluster, and the write still43// succeeded.44func PublishClusterStatus(c *apiclient.Client, clusterDoc *cluster.Cluster) (*cluster.Cluster, error) {45	body, err := json.Marshal(clusterDoc)46	if err != nil {47		return nil, err48	}49	stored := &cluster.Cluster{}50	err = c.RequestJSON(http.MethodPut, ClustersPath+"/"+clusterDoc.Metadata.Name+"/status", body, stored)51	if errors.Is(err, io.EOF) {52		return nil, nil53	}54	if err != nil {55		return nil, err56	}57	return stored, nil58}
kubernetes/fakeapi/fakeapi.go 95.2%
1// Package fakeapi is a small Kubernetes API server for the operators'2// tests. It holds collections of objects by their URL path, and it3// answers the requests an operator's pass and its watches send: a4// list, a streaming list, a read of one object by name, a create, an5// update, and a watch that receives each write. It records every6// request except the watches, so a test can count what a pass sends.7//8// A test serves it with apiservertest.Start, over in-memory9// connections, so the test can run in a synctest bubble and wait out10// the reflector's backoff on the fake clock.11//12// It is not a model of the API server. It ignores selectors and answers13// every list with every object of the collection. By default it checks14// no resourceVersion on an update, so a test of a pass need not keep15// its copies current. A test of optimistic concurrency calls16// CheckVersions, and the fake then answers an update the way the API17// server does: 409 for a stale copy, and no new version for a write18// that changes nothing.19package fakeapi2021import (22	"encoding/json"23	"fmt"24	"maps"25	"net/http"26	"reflect"27	"slices"28	"strconv"29	"strings"30	"sync"31)3233// Collection is one collection's kind and its objects. The kind goes34// in the bookmark that ends a streaming list.35type Collection struct {36	APIVersion string37	Kind       string38	Items      []map[string]any39}4041// Server holds the collections by URL path.42type Server struct {43	mu          sync.Mutex44	collections map[string]*Collection45	requests    []string46	version     int47	watchers    map[string][]chan string4849	// held, while it is true, queues each write's event in queued50	// instead of sending it, so a test can run a pass whose copies lag51	// a write.52	held   bool53	queued []queuedEvent5455	// checkVersions, while it is true, refuses an update from a stale56	// copy and keeps the version of an update that changes nothing.57	checkVersions bool58}5960type queuedEvent struct {61	collection string62	line       string63}6465// Hold queues the event of each write from now on, instead of sending66// it to the watchers.67func (s *Server) Hold() {68	s.mu.Lock()69	defer s.mu.Unlock()70	s.held = true71}7273// Release sends every queued event, in order, and sends each later74// write's event at once again.75func (s *Server) Release() {76	s.mu.Lock()77	defer s.mu.Unlock()78	s.held = false79	for _, e := range s.queued {80		s.send(e.collection, e.line)81	}82	s.queued = nil83}8485// CheckVersions makes each update from now on behave as the API86// server's does. An update that names another resourceVersion than the87// stored object's gets 409 Conflict. An update that names the stored88// version and changes nothing stores nothing, sends no event, and89// answers the stored object at its old version. A test that replays a90// race between two writers needs both: the first is the91// compare-and-swap, and the second means that a write which changes92// nothing does not move the version that other writers compare with.93func (s *Server) CheckVersions() {94	s.mu.Lock()95	defer s.mu.Unlock()96	s.checkVersions = true97}9899// New returns a server that holds the given collections.100func New(collections map[string]*Collection) *Server {101	return &Server{collections: collections, version: 10, watchers: map[string][]chan string{}}102}103104// Object builds an object with the metadata every kind carries.105func Object(apiVersion, kind, namespace, name string, fields map[string]any) map[string]any {106	metadata := map[string]any{"name": name, "resourceVersion": "5", "uid": "uid-" + name}107	if namespace != "" {108		metadata["namespace"] = namespace109	}110	o := map[string]any{"apiVersion": apiVersion, "kind": kind, "metadata": metadata}111	for k, v := range fields {112		o[k] = v113	}114	return o115}116117// Requests answers each request since the last Forget, as118// "METHOD path", in the order they arrived.119func (s *Server) Requests() []string {120	s.mu.Lock()121	defer s.mu.Unlock()122	return slices.Clone(s.requests)123}124125// Forget clears the record of requests.126func (s *Server) Forget() {127	s.mu.Lock()128	defer s.mu.Unlock()129	s.requests = nil130}131132// ResourceVersion answers the resourceVersion of one object, or "" when133// the collection holds no object of that name.134func (s *Server) ResourceVersion(collection, name string) string {135	s.mu.Lock()136	defer s.mu.Unlock()137	for _, item := range s.collections[collection].Items {138		if metadata := item["metadata"].(map[string]any); metadata["name"] == name {139			version, _ := metadata["resourceVersion"].(string)140			return version141		}142	}143	return ""144}145146func (s *Server) ServeHTTP(w http.ResponseWriter, r *http.Request) {147	w.Header().Set("Content-Type", "application/json")148	if r.URL.Query().Get("watch") == "true" {149		s.stream(w, r)150		return151	}152	s.mu.Lock()153	defer s.mu.Unlock()154	s.requests = append(s.requests, r.Method+" "+r.URL.Path)155	path := strings.TrimSuffix(r.URL.Path, "/status")156	if c, ok := s.collections[path]; ok {157		switch r.Method {158		case http.MethodGet:159			_ = json.NewEncoder(w).Encode(map[string]any{160				"apiVersion": c.APIVersion, "kind": c.Kind + "List",161				"metadata": map[string]any{"resourceVersion": strconv.Itoa(s.version)},162				"items":    c.Items,163			})164		case http.MethodPost:165			var created map[string]any166			_ = json.NewDecoder(r.Body).Decode(&created)167			s.store(path, "ADDED", created)168			c.Items = append(c.Items, created)169			w.WriteHeader(http.StatusCreated)170			_ = json.NewEncoder(w).Encode(created)171		}172		return173	}174	collection, name := path[:strings.LastIndex(path, "/")], path[strings.LastIndex(path, "/")+1:]175	if c, ok := s.collections[collection]; ok {176		for i, item := range c.Items {177			if item["metadata"].(map[string]any)["name"] != name {178				continue179			}180			if r.Method == http.MethodDelete {181				s.store(collection, "DELETED", item)182				c.Items = append(c.Items[:i], c.Items[i+1:]...)183				_ = json.NewEncoder(w).Encode(item)184				return185			}186			if r.Method == http.MethodPut {187				var written map[string]any188				_ = json.NewDecoder(r.Body).Decode(&written)189				if s.checkVersions {190					if version(written) != version(item) {191						w.WriteHeader(http.StatusConflict)192						_, _ = w.Write([]byte(`{"kind":"Status","apiVersion":"v1","status":"Failure","reason":"Conflict","code":409}`))193						return194					}195					if sameObject(written, item) {196						_ = json.NewEncoder(w).Encode(item)197						return198					}199				}200				s.store(collection, "MODIFIED", written)201				c.Items[i] = written202				item = written203			}204			_ = json.NewEncoder(w).Encode(item)205			return206		}207	}208	w.WriteHeader(http.StatusNotFound)209	_, _ = w.Write([]byte(`{"kind":"Status","apiVersion":"v1","status":"Failure","reason":"NotFound","code":404}`))210}211212// version answers the resourceVersion an object names.213func version(object map[string]any) string {214	metadata, _ := object["metadata"].(map[string]any)215	v, _ := metadata["resourceVersion"].(string)216	return v217}218219// sameObject reports whether two copies of an object differ only in220// their resourceVersion. The API server compares a written object with221// the stored one the same way before it decides to store anything.222func sameObject(a, b map[string]any) bool {223	strip := func(object map[string]any) map[string]any {224		copied := maps.Clone(object)225		metadata, _ := object["metadata"].(map[string]any)226		metadata = maps.Clone(metadata)227		delete(metadata, "resourceVersion")228		copied["metadata"] = metadata229		return copied230	}231	return reflect.DeepEqual(strip(a), strip(b))232}233234// store gives a written object the next resourceVersion and sends it to235// the collection's watchers. The caller holds mu.236func (s *Server) store(collection, event string, object map[string]any) {237	s.version++238	object["metadata"].(map[string]any)["resourceVersion"] = strconv.Itoa(s.version)239	line, _ := json.Marshal(map[string]any{"type": event, "object": object})240	if s.held {241		s.queued = append(s.queued, queuedEvent{collection, string(line)})242		return243	}244	s.send(collection, string(line))245}246247// send gives one event line to each watcher of a collection. The248// caller holds mu.249func (s *Server) send(collection, line string) {250	for _, watcher := range s.watchers[collection] {251		watcher <- line252	}253}254255// stream answers a streaming list: an ADDED event for each object of256// the collection, the bookmark that ends the initial events, and then257// each write to the collection until the watcher hangs up.258func (s *Server) stream(w http.ResponseWriter, r *http.Request) {259	s.mu.Lock()260	c, ok := s.collections[r.URL.Path]261	if !ok {262		s.mu.Unlock()263		w.WriteHeader(http.StatusNotFound)264		return265	}266	items := slices.Clone(c.Items)267	version := s.version268	writes := make(chan string, 64)269	s.watchers[r.URL.Path] = append(s.watchers[r.URL.Path], writes)270	s.mu.Unlock()271	defer func() {272		s.mu.Lock()273		defer s.mu.Unlock()274		s.watchers[r.URL.Path] = slices.DeleteFunc(s.watchers[r.URL.Path], func(c chan string) bool { return c == writes })275	}()276	for _, item := range items {277		object, _ := json.Marshal(item)278		fmt.Fprintf(w, `{"type":"ADDED","object":%s}`+"\n", object)279	}280	fmt.Fprintf(w, `{"type":"BOOKMARK","object":{"apiVersion":%q,"kind":%q,"metadata":{"resourceVersion":%q,"annotations":{"k8s.io/initial-events-end":"true"}}}}`+"\n",281		c.APIVersion, c.Kind, strconv.Itoa(version))282	w.(http.Flusher).Flush()283	for {284		select {285		case line := <-writes:286			fmt.Fprintln(w, line)287			w.(http.Flusher).Flush()288		case <-r.Context().Done():289			return290		}291	}292}
kubernetes/heartbeat.go 100.0%
1package kubernetes23// This file implements the machine heartbeat protocol: how a machine4// proves it is alive, and how the fleet's observer reads that proof.5//6// A machine's status is only as current as the last update from that7// machine. A dead machine cannot report that it is dead. Its last8// written status stays in the API showing Ready forever, which is9// worse than showing no status. Kubernetes has this same problem10// with kubelets, and solves it with heartbeats. The kubelet renews a11// lease every few seconds, and the node controller turns a silent12// lease into a NotReady Node. liken's machines get the same13// treatment. Each machine's operator renews a coordination.k8s.io14// Lease named for its machine. The cluster operator lists those15// leases to judge the fleet's liveness.16//17// This mechanism comes from kube-node-lease, and liken adopts it for18// the same reasons. A heartbeat must renew on a schedule forever, so19// it should be the cheapest write the API server offers. A Lease is20// a few dozen bytes with no watchers. A timestamp inside Machine21// status would instead rewrite the whole object (hardware inventory,22// boot record, conditions), and would wake every watcher on every23// renewal of every machine. Kubernetes moved the kubelet's24// heartbeats out of Node status and into kube-node-lease to avoid25// that cost. liken has used a lease for its heartbeats from the26// start, for the same reason. The leases live in the liken-system27// namespace, not in a copy of kube-node-lease's dedicated namespace.28// All objects that liken coordinates through this API use the29// namespace that the OS owns. The command `kubectl get leases -n30// liken-system` shows the fleet's complete liveness state.3132import (33	"encoding/json"34	"errors"35	"fmt"36	"net/http"37	"time"3839	"github.com/liken-sh/liken/kubernetes/apiclient"40)4142const heartbeatDir = "/apis/coordination.k8s.io/v1/namespaces/liken-system/leases"4344// HeartbeatRenewAfter sets how old the heartbeat must be before the45// machine's own operator renews it. The operator's renewal timer fires46// at half this period, so a timer that fires a moment early still47// renews on the next firing, and the lease is renewed every 8 to 1248// seconds.49// HeartbeatStaleAfter sets how long a machine may then stay silent50// before the cluster operator marks it Lost. A single missed51// renewal may only mean a busy moment. Several missed renewals mean52// the machine is down.53//54// These numbers come from kube-node-lease. The kubelet renews its55// lease every ten seconds, and the node controller waits forty56// seconds for a silent kubelet before its Node goes NotReady. A dead57// machine stops renewing both leases at the same moment, so matching58// the two thresholds means both systems report the loss together. This way,59// `kubectl get nodes` never disagrees with `kubectl get machines` for60// a minute about a machine that just died.61const (62	HeartbeatRenewAfter = 8 * time.Second63	HeartbeatStaleAfter = 40 * time.Second64)6566// microTime is the time layout that coordination.k8s.io uses for its67// timestamps (metav1.MicroTime): RFC 3339 with microseconds. This is68// a finer grain than most of the API uses, because leases exist to69// compare instants that are close together in time.70const microTime = "2006-01-02T15:04:05.000000Z07:00"7172// Lease is the part of a coordination.k8s.io Lease that the heartbeat73// writes and the cluster operator reads.74type Lease struct {75	APIVersion string    `json:"apiVersion"`76	Kind       string    `json:"kind"`77	Metadata   LeaseMeta `json:"metadata"`78	Spec       struct {79		HolderIdentity       string `json:"holderIdentity,omitempty"`80		LeaseDurationSeconds int    `json:"leaseDurationSeconds,omitempty"`81		AcquireTime          string `json:"acquireTime,omitempty"`82		RenewTime            string `json:"renewTime,omitempty"`83	} `json:"spec"`84}8586// LeaseMeta is api.ObjectMeta with the owner reference that a87// heartbeat lease carries.88//89// A machine's lease names its Machine as its owner. The Machine is90// cluster-scoped and the lease is in liken-system, and Kubernetes91// allows a namespaced object to have a cluster-scoped owner. When the92// Machine of a machine that left the fleet is deleted, the93// garbage collector deletes the lease too. Without the owner the lease94// stays in liken-system with its last renewal, and nothing deletes it.95//96// A renewal replaces the whole lease, so LeaseMeta carries the labels97// and annotations too, and a renewal writes back the ones it read.98type LeaseMeta struct {99	Name            string            `json:"name"`100	ResourceVersion string            `json:"resourceVersion,omitempty"`101	Labels          map[string]string `json:"labels,omitempty"`102	Annotations     map[string]string `json:"annotations,omitempty"`103	OwnerReferences []OwnerReference  `json:"ownerReferences,omitempty"`104}105106// owners is the owner list a lease is written with. The API server107// refuses an owner reference with no UID, and a refused write is a108// missed heartbeat, so an owner with no UID writes a lease with no109// owner instead.110func owners(owner OwnerReference) []OwnerReference {111	if owner.UID == "" {112		return nil113	}114	return []OwnerReference{owner}115}116117// newLease creates a new claim, held by holder as of the given time.118func newLease(name, holder string, owner OwnerReference, duration time.Duration, now time.Time) *Lease {119	l := &Lease{APIVersion: "coordination.k8s.io/v1", Kind: "Lease"}120	l.Metadata.Name = name121	l.Metadata.OwnerReferences = owners(owner)122	l.Spec.HolderIdentity = holder123	l.Spec.LeaseDurationSeconds = int(duration.Seconds())124	l.Spec.AcquireTime = now.UTC().Format(microTime)125	l.Spec.RenewTime = l.Spec.AcquireTime126	return l127}128129// Heartbeat keeps a machine's own lease current. Each machine is the130// only writer of its own lease, the same way a kubelet is the only131// writer of its own node lease, so there is no election here, and every132// failure only means "try again on the next renewal."133//134// A Heartbeat holds the lease as this process last wrote it. The135// resourceVersion in that copy is what a renewal needs, so a steady136// renewal is one update and no read. The machine's operator creates one137// Heartbeat for the life of the process, the same way it keeps one138// release fetcher.139type Heartbeat struct {140	name string141	held *Lease142143	// wrote is the time of the last renewal this process wrote, as144	// the caller's clock gave it. A time from time.Now carries the145	// monotonic reading, so a step of the wall clock does not change146	// the age of the last renewal.147	wrote time.Time148}149150// NewHeartbeat returns the heartbeat of the named machine. It holds no151// lease until its first renewal reads one.152func NewHeartbeat(name string) *Heartbeat {153	return &Heartbeat{name: name}154}155156// Renew renews the lease once the last renewal this process wrote has157// aged past HeartbeatRenewAfter, and creates it when it does not exist.158// A call that finds the last renewal fresh sends nothing. A new process159// renews once whatever the lease's renewTime says: a renewTime that a160// clock wrote before it stepped back reads as a time in the future, so161// a skip that compared it with the clock would skip every renewal until162// the clock caught up.163//164// The first renewal reads the lease, because the process has no copy165// yet. After that, the renewal writes from the copy it holds. The166// update carries the copy's resourceVersion, so a lease that something167// else wrote since then makes the API server answer 409 Conflict, and168// a lease that somebody deleted answers 404. Either answer drops the169// copy, and the same call reads the lease again and renews from what170// it read.171//172// owner is this machine's Machine. Every create and every renewal173// writes it as the lease's one owner, so a lease created before the174// lease had an owner, or owned by an earlier Machine of the same name,175// names the current Machine after its next renewal.176//177// A nil Heartbeat renews nothing.178func (h *Heartbeat) Renew(c *apiclient.Client, owner OwnerReference, now time.Time) {179	if h == nil {180		return181	}182	path := heartbeatDir + "/" + h.name183	if h.held == nil && !h.read(c, owner, now) {184		return185	}186	if !h.wrote.IsZero() && now.Sub(h.wrote) < HeartbeatRenewAfter {187		return188	}189	err := h.write(c, path, owner, now)190	if errors.Is(err, apiclient.ErrConflict) || errors.Is(err, apiclient.ErrNotFound) {191		h.held = nil192		if !h.read(c, owner, now) {193			return194		}195		err = h.write(c, path, owner, now)196	}197	if err != nil {198		h.held = nil199		fmt.Printf("renewing the heartbeat lease: %v\n", err)200	}201}202203// read reads the lease into the held copy, or creates the lease when204// it does not exist. It answers false when the caller has nothing left205// to do: the read failed, or the create already renewed the lease.206func (h *Heartbeat) read(c *apiclient.Client, owner OwnerReference, now time.Time) bool {207	l := &Lease{}208	err := c.RequestJSON(http.MethodGet, heartbeatDir+"/"+h.name, nil, l)209	if errors.Is(err, apiclient.ErrNotFound) {210		// A lease is a struct of strings and ints. Marshaling it cannot fail.211		body, _ := json.Marshal(newLease(h.name, h.name, owner, HeartbeatStaleAfter, now))212		created := &Lease{}213		if err := c.RequestJSON(http.MethodPost, heartbeatDir, body, created); err != nil {214			if !errors.Is(err, apiclient.ErrConflict) {215				fmt.Printf("creating the heartbeat lease: %v\n", err)216			}217			return false218		}219		h.held, h.wrote = created, now220		return false221	}222	if err != nil {223		fmt.Printf("reading the heartbeat lease: %v\n", err)224		return false225	}226	h.held = l227	return true228}229230// write sends the renewal from the held copy, and keeps the lease the231// API server answers with, which carries the new resourceVersion.232func (h *Heartbeat) write(c *apiclient.Client, path string, owner OwnerReference, now time.Time) error {233	renewal := *h.held234	renewal.Metadata.OwnerReferences = owners(owner)235	renewal.Spec.HolderIdentity = h.name236	renewal.Spec.LeaseDurationSeconds = int(HeartbeatStaleAfter.Seconds())237	renewal.Spec.RenewTime = now.UTC().Format(microTime)238	// A lease is a struct of strings and ints. Marshaling it cannot fail.239	body, _ := json.Marshal(&renewal)240	written := &Lease{}241	if err := c.RequestJSON(http.MethodPut, path, body, written); err != nil {242		return err243	}244	h.held, h.wrote = written, now245	return nil246}247248// ListHeartbeats reads every machine's last renewal for the cluster249// operator's sweep. One cheap list request yields the fleet's250// liveness. Renewals says what the answer holds.251func ListHeartbeats(c *apiclient.Client) (map[string]time.Time, error) {252	leases, err := List[Lease](c, heartbeatDir)253	if err != nil {254		return nil, err255	}256	return Renewals(leases), nil257}258259// Renewals maps each lease's name to the moment of its last renewal.260// A lease that is not some machine's heartbeat, such as the cluster261// operator's own leader election Lease, causes no harm in this map:262// the sweep looks up renewals by machine name and never iterates over263// the map, so a stray key can never be read as a machine. A lease264// whose renewal does not parse carries no liveness claim, and it is265// left out.266func Renewals(leases []Lease) map[string]time.Time {267	renewals := map[string]time.Time{}268	for _, l := range leases {269		if renewed, err := time.Parse(microTime, l.Spec.RenewTime); err == nil {270			renewals[l.Metadata.Name] = renewed271		}272	}273	return renewals274}
kubernetes/helmcharts.go 100.0%
1package kubernetes23// This file reads HelmCharts, the resource k3s uses to install a4// Helm release.5//6// The kind belongs to k3s, not to Kubernetes: the Helm controller7// embedded in the k3s server watches HelmChart resources and renders8// each one into an installed release. k3s deploys its own bundled9// components this way, Traefik among them. The controller puts a10// removal finalizer on every chart it manages, so deleting a11// HelmChart takes two steps: the delete request marks the object,12// and the controller uninstalls the release and then clears the13// mark. No other component clears it, so the machine operator counts14// charts before the helm feature may stop15// (machine-operator/retraction.go).1617import (18	"errors"1920	"github.com/liken-sh/liken/kubernetes/apiclient"21)2223// HelmChartsPath names the collection, across all namespaces. A chart24// is namespaced, and the ones k3s creates for its own components live25// in kube-system.26const HelmChartsPath = "/apis/helm.cattle.io/v1/helmcharts"2728// HelmChart holds the part of the resource that liken reads: where29// the chart lives. What a chart installs stays out of this type. The30// count and the names are enough for a person to act on, and reading31// more would mean holding knowledge of what a Traefik release32// contains, which goes stale the first time another project renames33// something.34type HelmChart struct {35	Metadata struct {36		Name      string `json:"name"`37		Namespace string `json:"namespace"`38	} `json:"metadata"`39}4041// ListHelmCharts reads every HelmChart in the cluster. An API server42// that serves no such kind answers 404 for the whole collection, and43// that answer carries the same meaning as an empty list: the cluster44// holds no charts. So a missing kind returns no charts and no error.45// Every other failure stays an error, because a failed read must46// never reach a caller as an empty cluster.47func ListHelmCharts(c *apiclient.Client) ([]HelmChart, error) {48	charts, err := List[HelmChart](c, HelmChartsPath)49	if errors.Is(err, apiclient.ErrNotFound) {50		return nil, nil51	}52	return charts, err53}
kubernetes/kubeconfig.go 71.4%
1package kubernetes23// A client for the operator's workstation, built from the admin4// kubeconfig that the identity package computes. The in-cluster5// client (InClusterClient) authenticates with a ServiceAccount bearer6// token that kubelet mounts and refreshes. A workstation has no7// kubelet and no token. Its kubeconfig embeds a client certificate8// instead, and the TLS handshake itself carries the identity: the9// API server reads the certificate's CN as the username and each O10// as a group, with no user database behind it. So this client sends11// no Authorization header at all.1213import (14	"crypto/tls"15	"crypto/x509"16	"fmt"17	"net"18	"net/http"19	"os"20	"time"2122	"github.com/liken-sh/liken/kubernetes/apiclient"23	"sigs.k8s.io/yaml"24)2526// kubeconfigFile is the part of a kubeconfig this client reads: the27// first cluster's address and CA, and the first user's certificate.28// liken writes single-entry kubeconfigs (identity/kubeconfig.go),29// and this parser holds no opinion about richer files beyond taking30// their first entries. The []byte fields decode from base64 on31// their own: sigs.k8s.io/yaml routes through encoding/json, which32// defines []byte that way, and kubeconfig chose base64 for the same33// reason.34type kubeconfigFile struct {35	Clusters []struct {36		Cluster struct {37			Server                   string `json:"server"`38			CertificateAuthorityData []byte `json:"certificate-authority-data"`39		} `json:"cluster"`40	} `json:"clusters"`41	Users []struct {42		User struct {43			ClientCertificateData []byte `json:"client-certificate-data"`44			ClientKeyData         []byte `json:"client-key-data"`45		} `json:"user"`46	} `json:"users"`47}4849// KubeconfigClient builds a client from a kubeconfig file. The50// client trusts only the embedded CA, not the system trust store,51// so it accepts only the cluster's own API server. The timeouts52// match the in-cluster client's reasoning (the shared module's53// kubernetes/apiclient/client.go): every one of them limits a server that stops responding without sending any54// signal.55func KubeconfigClient(path string) (*apiclient.Client, error) {56	raw, err := os.ReadFile(path)57	if err != nil {58		return nil, err59	}60	var kc kubeconfigFile61	if err := yaml.Unmarshal(raw, &kc); err != nil {62		return nil, fmt.Errorf("parsing %s: %w", path, err)63	}64	switch {65	case len(kc.Clusters) == 0 && len(kc.Users) == 0:66		return nil, fmt.Errorf("%s names no cluster and no user", path)67	case len(kc.Clusters) == 0:68		return nil, fmt.Errorf("%s names no cluster", path)69	case len(kc.Users) == 0:70		return nil, fmt.Errorf("%s names no user", path)71	}72	clusterEntry, userEntry := kc.Clusters[0].Cluster, kc.Users[0].User7374	cert, err := tls.X509KeyPair(userEntry.ClientCertificateData, userEntry.ClientKeyData)75	if err != nil {76		return nil, fmt.Errorf("reading the client certificate in %s: %w", path, err)77	}78	roots := x509.NewCertPool()79	if !roots.AppendCertsFromPEM(clusterEntry.CertificateAuthorityData) {80		return nil, fmt.Errorf("%s carries no certificate authority", path)81	}8283	return apiclient.New(clusterEntry.Server, &http.Client{84		Transport: &http.Transport{85			TLSClientConfig: &tls.Config{86				RootCAs:      roots,87				Certificates: []tls.Certificate{cert},88			},89			DialContext: (&net.Dialer{90				Timeout:   5 * time.Second,91				KeepAlive: 10 * time.Second,92			}).DialContext,93			ResponseHeaderTimeout: 10 * time.Second,94			IdleConnTimeout:       30 * time.Second,95		},96	}, ""), nil97}
kubernetes/machines.go 93.3%
1package kubernetes23// This file reads and reports on Machines: the operations both4// operators share. The machine operator reads and writes its own5// Machine. The cluster operator reads every Machine. The watches are6// in the shared informer package, and their handlers and reads in the7// watch package.89import (10	"encoding/json"11	"errors"12	"io"13	"net/http"1415	"github.com/liken-sh/liken/kubernetes/apiclient"16	"github.com/liken-sh/liken/liken/machine"17)1819func GetMachine(c *apiclient.Client, name string) (*machine.Machine, error) {20	return apiclient.Get[machine.Machine](c, MachinesPath+"/"+name)21}2223// PublishStatus writes through the status subresource. This is a24// separate endpoint (…/machines/<name>/status) that updates only the25// status half of the object. Because of this, a controller can never26// accidentally rewrite the spec it acts on, and RBAC can grant access27// to the two halves separately. The write is a PUT request that28// carries the object's resourceVersion. If anything else changed the29// object in the meantime, the server answers with 409 Conflict30// instead of applying the stale copy. The caller then reads the31// object again on its next pass and tries again. This pattern is32// optimistic concurrency, and every Kubernetes controller uses it to33// handle contention.34//35// It answers the resourceVersion the API server gave the write. A36// caller that reads a watch's copy notes it in the memo of its writes37// (the shared memo package), so the copy answers only once it holds38// the write.39func PublishStatus(c *apiclient.Client, m *machine.Machine, status *machine.MachineStatus) (string, error) {40	updated := *m41	updated.Status = *status42	body, err := json.Marshal(&updated)43	if err != nil {44		return "", err45	}46	return putStatus(c, MachinesPath+"/"+m.Metadata.Name+"/status", body)47}4849// putStatus writes a status subresource and answers the50// resourceVersion of the object the API server stored. An answer with51// no body carries no version, and the write still succeeded, so the52// version is empty.53func putStatus(c *apiclient.Client, path string, body []byte) (string, error) {54	var written struct {55		Metadata struct {56			ResourceVersion string `json:"resourceVersion"`57		} `json:"metadata"`58	}59	err := c.RequestJSON(http.MethodPut, path, body, &written)60	if errors.Is(err, io.EOF) {61		return "", nil62	}63	return written.Metadata.ResourceVersion, err64}
kubernetes/ownwrite.go 100.0%
1package kubernetes23import "sync"45// An OwnWrite remembers the resourceVersion that the API server6// answered for a writer's last write to one object, so the writer's7// watch can tell the echo of that write from another writer's change.8// A watch that woke the loop for its own echo would run a second pass9// for each write, and a pass whose write never compares equal would10// write as fast as the API server answers.11//12// The watch can deliver the echo before the write's answer reaches the13// writer, because the two travel on different connections. So Send14// holds the writer's turn until it has recorded the answer, and Wrote15// waits for the turn. The turn is a channel, not a mutex, so a watch16// handler that waits for it is durably blocked inside a synctest17// bubble, and a test can hold a write at that point.18//19// The zero OwnWrite remembers no write.20type OwnWrite struct {21	once    sync.Once22	turn    chan struct{}23	version string24}2526func (o *OwnWrite) lock() {27	o.once.Do(func() { o.turn = make(chan struct{}, 1) })28	o.turn <- struct{}{}29}3031func (o *OwnWrite) unlock() { <-o.turn }3233// Send runs one write and remembers the version it answers. A write34// that fails remembers no version, because a request that timed out35// can still have landed, and its echo must then wake the loop.36func (o *OwnWrite) Send(write func() (version string, err error)) error {37	o.lock()38	defer o.unlock()39	version, err := write()40	if err != nil {41		version = ""42	}43	o.version = version44	return err45}4647// Forget drops the remembered write, for a write that leaves no48// version, such as a delete.49func (o *OwnWrite) Forget() {50	o.lock()51	defer o.unlock()52	o.version = ""53}5455// Wrote answers whether version is the one the API server answered56// for the last write. It waits for a write in flight to record its57// answer.58func (o *OwnWrite) Wrote(version string) bool {59	o.lock()60	defer o.unlock()61	return version != "" && version == o.version62}
kubernetes/pods.go 92.3%
1package kubernetes23// This file implements pod eviction. Both operators use it because4// both ask pods to leave: the machine operator drains its own node5// ahead of a granted reboot, and the cluster operator refreshes the6// OS's own pods after an upgrade.78import (9	"encoding/json"10	"net/http"1112	"github.com/liken-sh/liken/kubernetes/apiclient"13)1415type PodMetadata struct {16	Name              string            `json:"name"`17	Namespace         string            `json:"namespace"`18	UID               string            `json:"uid"`19	Annotations       map[string]string `json:"annotations"`20	OwnerReferences   []OwnerReference  `json:"ownerReferences"`21	DeletionTimestamp string            `json:"deletionTimestamp,omitempty"`22}2324// HostPathVolume is a directory or file the pod takes from the25// machine it runs on. The path is the whole of it: what a hostPath26// mount reaches on the host is what a drain reads to tell a DRA27// driver's pod from an ordinary workload.28type HostPathVolume struct {29	Path string `json:"path"`30}3132// PodVolume carries the one volume kind liken reads. Every other33// kind decodes with a nil HostPath, which is the answer the readers34// want anyway.35type PodVolume struct {36	Name     string          `json:"name"`37	HostPath *HostPathVolume `json:"hostPath"`38}3940// PodResourceClaim is one entry in spec.resourceClaims: a DRA claim41// the pod's containers may reference by Name. The claim itself is42// either named directly or made from a template, so exactly one of43// the last two fields carries a value.44type PodResourceClaim struct {45	Name                      string `json:"name"`46	ResourceClaimName         string `json:"resourceClaimName"`47	ResourceClaimTemplateName string `json:"resourceClaimTemplateName"`48}4950type PodSpec struct {51	NodeName       string             `json:"nodeName"`52	Volumes        []PodVolume        `json:"volumes"`53	ResourceClaims []PodResourceClaim `json:"resourceClaims"`54}5556// ContainerStatus holds the part of a container's status that liken57// needs: which image it runs, and whether it is currently serving.58// Ready is the kubelet's own verdict. It covers every way a59// container can fail to serve, from a crash loop to an image whose60// binary fails to exec.61type ContainerStatus struct {62	Name  string `json:"name"`63	Image string `json:"image"`64	Ready bool   `json:"ready"`65}6667type PodStatus struct {68	Phase             string            `json:"phase"`69	ContainerStatuses []ContainerStatus `json:"containerStatuses"`70}7172// Pod holds the part of a Kubernetes Pod that liken needs: identity,73// where it runs, who owns it, what it takes from the host and from74// DRA, and whether it is still running.75type Pod struct {76	Metadata PodMetadata `json:"metadata"`77	Spec     PodSpec     `json:"spec"`78	Status   PodStatus   `json:"status"`79}8081// Completed reports whether the pod has finished running. A completed82// pod's containers are correctly not ready, and never will be ready83// again. Because of this, drains do not evict them, and health checks84// do not count them.85func (p *Pod) Completed() bool {86	return p.Status.Phase == "Succeeded" || p.Status.Phase == "Failed"87}8889// Terminating reports whether the pod is already leaving: somebody90// deleted or evicted it, and the kubelet is still stopping it. The pod91// stays in every listing until the kubelet finishes, so a pass that92// asks each listed pod to leave must skip this one, or it asks again93// on every pass until the pod is gone.94func (p *Pod) Terminating() bool {95	return p.Metadata.DeletionTimestamp != ""96}9798// IsDaemon reports whether a DaemonSet owns the pod. Drains skip99// daemon pods, because the DaemonSet controller ignores cordons and100// would only recreate them. The pod steward refreshes only these101// pods.102func (p *Pod) IsDaemon() bool {103	for _, owner := range p.Metadata.OwnerReferences {104		if owner.Kind == "DaemonSet" {105			return true106		}107	}108	return false109}110111// ListPodsOnNode reads every pod running on one node, across all112// namespaces. /api/v1/pods is the whole cluster's pod collection. The113// fieldSelector asks the server to filter this collection by114// spec.nodeName, so only that node's pods ever transfer over the115// network. This is the starting view for a drain: everything that116// might still need to move off a machine before the machine may117// reboot.118func ListPodsOnNode(c *apiclient.Client, nodeName string) ([]Pod, error) {119	return List[Pod](c, "/api/v1/pods?fieldSelector=spec.nodeName%3D"+nodeName)120}121122// EvictPod asks a pod to leave through the eviction subresource. The123// Eviction API is what separates a polite request from plain124// deletion. The server refuses the request while removing the pod125// would violate its PodDisruptionBudget, and the caller then asks126// again later.127//128// The refusal is a 429, the same status that asks a client to wait129// and send a request again. For an eviction it is a verdict, not a130// request to wait: the budget stays short until another pod becomes131// ready, which takes longer than the shared client waits. So the132// eviction goes out on a client whose wait after a 429 has already133// ended, whatever client the caller holds, and the refusal reaches the134// caller at once.135func EvictPod(c *apiclient.Client, p Pod) error {136	c = c.WithWaitContext(noWait)137	body, err := json.Marshal(map[string]any{138		"apiVersion": "policy/v1",139		"kind":       "Eviction",140		"metadata":   map[string]string{"name": p.Metadata.Name, "namespace": p.Metadata.Namespace},141	})142	if err != nil {143		return err144	}145	path := "/api/v1/namespaces/" + p.Metadata.Namespace + "/pods/" + p.Metadata.Name + "/eviction"146	return c.RequestJSON(http.MethodPost, path, body, nil)147}
kubernetes/resourceclaims.go 80.0%
1package kubernetes23// This file reads a ResourceClaim's allocation.4//5// When the kubelet asks the DRA driver to prepare a claim, the6// request names the claim, but not what was allocated to it. The7// allocation lives in the claim's status, written by the scheduler,8// and the driver must read it back from the API server. Kubernetes9// designed it this way on purpose: the claim object is the one10// source of truth for what a pod was granted, so a stale or replayed11// prepare call can never deliver anything except what the scheduler12// actually allocated.13//14// This file covers only the read path: liken never writes claims.15// Workloads create claims, and the scheduler allocates them. Because16// of this, these types carry only the fields that the driver reads.1718import (19	"net/http"2021	"github.com/liken-sh/liken/kubernetes/apiclient"22)2324// ResourceClaim holds the part of the claim that the driver needs:25// which devices were allocated from which driver's pools.26type ResourceClaim struct {27	Metadata struct {28		Name      string `json:"name"`29		Namespace string `json:"namespace"`30		UID       string `json:"uid"`31	} `json:"metadata"`32	Status struct {33		Allocation *struct {34			Devices struct {35				Results []AllocatedDevice `json:"results"`36			} `json:"devices"`37		} `json:"allocation"`38	} `json:"status"`39}4041// AllocatedDevice is one allocation result. The scheduler chose42// Device from Pool, published by Driver, to satisfy the claim's43// named Request. Pool and Device correspond to what the inventory44// published (see resourceslices.go). Driver names whose inventory45// this is, which matters because one claim can mix devices from46// several drivers.47type AllocatedDevice struct {48	Request string `json:"request"`49	Driver  string `json:"driver"`50	Pool    string `json:"pool"`51	Device  string `json:"device"`52}5354// GetResourceClaim reads one claim. Claims are namespaced: each claim55// belongs to the workload that created it. Because of this, the path56// carries the namespace, unlike every other resource this package57// touches.58func GetResourceClaim(c *apiclient.Client, namespace, name string) (*ResourceClaim, error) {59	path := "/apis/resource.k8s.io/v1/namespaces/" + namespace + "/resourceclaims/" + name60	claim := &ResourceClaim{}61	if err := c.RequestJSON(http.MethodGet, path, nil, claim); err != nil {62		return nil, err63	}64	return claim, nil65}
kubernetes/resourceslices.go 94.4%
1package kubernetes23// This file publishes device inventory as ResourceSlices.4//5// Dynamic resource allocation (the resource.k8s.io API group) is how6// workloads reach hardware. A per-node driver publishes each usable7// device in a ResourceSlice. DeviceClasses select over the devices'8// attributes, and pods claim from classes. This file implements the9// publishing side: liken's machine operator is the driver, and the10// slice it maintains is the one Kubernetes-native inventory of what11// this machine's hardware can actually do. Contrast this with12// Machine status.hardware.unclaimed, which carries only what does13// not work and what would fix it. The record of working devices14// lives here, in the API built for that purpose.15//16// Like every type in this package, these structs carry only the part17// of the upstream API that liken uses: the fields liken writes,18// nothing more. The full ResourceSlice can describe partitionable19// devices, shared counters, and per-device node selection, machinery20// built for GPUs split many ways. None of that changes what a whole21// PCI or USB device on one node needs: a name, some attributes, and22// the node's identity.23//24// A slice belongs to a pool, and the pool's generation tells readers25// which slices are current. The scheduler distrusts any slice whose26// generation lags behind the newest generation it can see. This27// protects the scheduler from acting on a multi-slice inventory that28// is only partly updated. liken publishes one slice per node,29// because the whole inventory fits in one slice. Because of this,30// the protocol reduces to a version counter: bump the counter on31// every change, and one slice is always a consistent snapshot.3233import (34	"encoding/json"35	"net/http"36	"reflect"37	"slices"3839	"github.com/liken-sh/liken/kubernetes/apiclient"40)4142// ResourceSlicesPath names the URL where the DRA inventory lives.43// Slices are cluster-scoped, like Nodes. A namespace marks a44// workload boundary, and hardware inventory belongs to the machine,45// not to any tenant.46const ResourceSlicesPath = "/apis/resource.k8s.io/v1/resourceslices"4748// DriverName identifies liken as a DRA driver. By convention, driver49// names are DNS domains, so vendors cannot collide with each other;50// liken owns liken.sh. Every slice this operator publishes carries51// this name. Every DeviceClass that a deployment writes selects on52// this name. The kubelet routes prepare calls for claims allocated53// from these slices to the plugin registered under this name.54const DriverName = "liken.sh"5556type ResourceSlice struct {57	APIVersion string            `json:"apiVersion"`58	Kind       string            `json:"kind"`59	Metadata   ResourceSliceMeta `json:"metadata"`60	Spec       ResourceSliceSpec `json:"spec"`61}6263// ResourceSliceMeta carries the one piece of metadata that64// api.ObjectMeta does not: an owner reference. Owning a slice does65// necessary work; it is not decoration. See WriteResourceSlice.66type ResourceSliceMeta struct {67	Name            string           `json:"name"`68	ResourceVersion string           `json:"resourceVersion,omitempty"`69	OwnerReferences []OwnerReference `json:"ownerReferences,omitempty"`70}7172// OwnerReference ties one object's lifetime to another object's73// lifetime. When the owner is deleted, the garbage collector deletes74// the owned object. The UID matters: a reference names one specific75// instance of the owner, so a Node that is deleted and registered76// again under the same name does not inherit the old node's slices.77// This type is shared across the package. Slices are written with78// all four fields, while the drain (pods.go) only ever reads Kind to79// recognize DaemonSet pods.80type OwnerReference struct {81	APIVersion string `json:"apiVersion,omitempty"`82	Kind       string `json:"kind"`83	Name       string `json:"name,omitempty"`84	UID        string `json:"uid,omitempty"`85}8687type ResourceSliceSpec struct {88	Driver   string        `json:"driver"`89	Pool     ResourcePool  `json:"pool"`90	NodeName string        `json:"nodeName,omitempty"`91	Devices  []SliceDevice `json:"devices,omitempty"`92}9394type ResourcePool struct {95	Name               string `json:"name"`96	Generation         int64  `json:"generation"`97	ResourceSliceCount int64  `json:"resourceSliceCount"`98}99100// SliceDevice is one claimable device. The name must be a DNS label,101// unique within the pool. The attributes are the values that102// DeviceClass CEL selectors match against. An attribute name left103// unqualified belongs to the publishing driver's domain. A selector104// reads these as device.attributes["liken.sh"].driver, and so on.105// AllowMultipleAllocations lets more than one claim allocate the same106// device. Without it the API allocates a device once, which is the107// safe default and the right one for a device that one process must108// hold. Only the driver that publishes a device can set this: a109// DeviceClass or a claim can select a device, but neither can say110// that the hardware divides. The field is a pointer so that a device111// that does not divide publishes nothing at all, rather than an112// explicit false that reads as a claim about the hardware.113//114// The API server honors the field when its DRAConsumableCapacity115// feature is on, which is the default in the k3s that liken ships. An116// API server with the feature off drops the field on write, and every117// device allocates once, which is the safe direction to fail in.118type SliceDevice struct {119	Name                     string                     `json:"name"`120	Attributes               map[string]DeviceAttribute `json:"attributes,omitempty"`121	AllowMultipleAllocations *bool                      `json:"allowMultipleAllocations,omitempty"`122}123124// DeviceAttribute holds exactly one of four typed values. The API125// keeps these types separate so that selectors can compare numbers126// as numbers, and versions by version rules, instead of treating127// everything as a string.128type DeviceAttribute struct {129	Bool    *bool   `json:"bool,omitempty"`130	Int     *int64  `json:"int,omitempty"`131	String  *string `json:"string,omitempty"`132	Version *string `json:"version,omitempty"`133}134135// EnsureResourceSlice makes one node's published slice match its136// actual inventory: it reads the slice (GetResourceSlice) and then137// writes what differs (WriteResourceSlice). A caller that already138// holds a copy of the slice, from a watch, calls WriteResourceSlice139// with that copy and sends no read.140func EnsureResourceSlice(c *apiclient.Client, nodeName string, owner OwnerReference, devices []SliceDevice) error {141	current, err := GetResourceSlice(c, nodeName)142	if err != nil {143		return err144	}145	return WriteResourceSlice(c, nodeName, current, owner, devices)146}147148// ResourceSliceName is the name of one node's slice. Each node gets149// one predictable name, with the driver name added as a suffix. This150// keeps other DRA drivers on the same node from colliding with ours.151// Slices are cluster-scoped, and nothing stops a deployment from152// adding a GPU vendor's driver.153func ResourceSliceName(nodeName string) string {154	return nodeName + "-" + DriverName155}156157// GetResourceSlice reads one node's slice. An absent slice returns158// nil, nil, because a node with no devices has none.159func GetResourceSlice(c *apiclient.Client, nodeName string) (*ResourceSlice, error) {160	current, err := apiclient.Get[ResourceSlice](c, ResourceSlicesPath+"/"+ResourceSliceName(nodeName))161	if err == apiclient.ErrNotFound {162		return nil, nil163	}164	return current, err165}166167// WriteResourceSlice makes one node's published slice match its168// actual inventory, given the slice as it is now (nil when it does169// not exist). It is SliceWriter.Write with no memory of an earlier170// write, for a caller that writes once.171func WriteResourceSlice(c *apiclient.Client, nodeName string, current *ResourceSlice, owner OwnerReference, devices []SliceDevice) error {172	return (&SliceWriter{}).Write(c, nodeName, current, owner, devices)173}174175// A SliceWriter writes one node's slice and remembers its own last176// write: the resourceVersion the API server answered, and the devices177// the write sent.178//179// The memory answers two questions. The first is whether a slice still180// holds this writer's last write. The API server can return a slice181// that differs from what was sent: one with DRAConsumableCapacity off182// drops allowMultipleAllocations from each device. A comparison of the183// desired devices with the returned ones would then differ on every184// pass, and each pass would write again. A slice at the version of the185// writer's own last write, with the same devices desired, is current,186// whatever the server kept of them. The upstream DRA slice controller187// meets the same problem (k8s.io/dynamic-resource-allocation/188// resourceslice), and copies the dropped fields back before it compares.189//190// The second is whether a change that a watch delivers is this191// writer's own echo (Wrote, through OwnWrite).192//193// One goroutine writes through a SliceWriter. The watch's handler194// calls Wrote from another.195type SliceWriter struct {196	own  OwnWrite197	sent []SliceDevice198}199200// Wrote answers whether version is the resourceVersion the API server201// answered for this writer's last write. It waits for a write in202// flight to record its answer.203func (w *SliceWriter) Wrote(version string) bool {204	return w.own.Wrote(version)205}206207// Write creates the slice when the node first has devices, replaces the208// slice when the inventory changed, deletes the slice when the last209// device is gone, and changes nothing when nothing moved. This is the210// same compare-then-write pattern as every other liken reconcile, so a211// steady machine sends no request here.212//213// The Node owns the slice. Neither the Machine nor the operator pod214// owns it. The inventory is a claim about what is ready to use on215// this node. If the node leaves the cluster, the claim must be216// deleted with it: a slice that remains after its node is gone would217// offer the scheduler hardware that nobody can deliver. Owner-based218// garbage collection also cleans up after this operator crashes or219// exits abruptly, when no code runs to delete anything.220//221// The write carries the resourceVersion of the copy it compared. If a222// conflicting writer changed the object in the meantime, or the copy223// from a watch is behind this operator's own last write, this update224// returns apiclient.ErrConflict instead of overwriting that change. A225// create from a copy that says the slice is absent returns226// apiclient.ErrConflict too, when the slice exists. The next pass227// compares against a newer copy and tries again.228func (w *SliceWriter) Write(c *apiclient.Client, nodeName string, current *ResourceSlice, owner OwnerReference, devices []SliceDevice) error {229	name := ResourceSliceName(nodeName)230	path := ResourceSlicesPath + "/" + name231232	if current == nil {233		if len(devices) == 0 {234			return nil235		}236		slice := &ResourceSlice{237			APIVersion: "resource.k8s.io/v1",238			Kind:       "ResourceSlice",239			Metadata: ResourceSliceMeta{240				Name:            name,241				OwnerReferences: []OwnerReference{owner},242			},243			Spec: ResourceSliceSpec{244				Driver:   DriverName,245				NodeName: nodeName,246				Pool:     ResourcePool{Name: nodeName, Generation: 1, ResourceSliceCount: 1},247				Devices:  devices,248			},249		}250		return w.send(c, http.MethodPost, ResourceSlicesPath, slice, devices)251	}252253	if len(devices) == 0 {254		w.own.Forget()255		w.sent = nil256		return c.RequestJSON(http.MethodDelete, path, nil, nil)257	}258	if reflect.DeepEqual(current.Spec.Devices, devices) || w.holds(current, devices) {259		return nil260	}261262	updated := *current263	updated.Spec.NodeName = nodeName264	updated.Spec.Driver = DriverName265	updated.Spec.Pool = ResourcePool{266		Name:               nodeName,267		Generation:         current.Spec.Pool.Generation + 1,268		ResourceSliceCount: 1,269	}270	updated.Spec.Devices = devices271	return w.send(c, http.MethodPut, path, &updated, devices)272}273274// holds answers whether current is this writer's last write, and the275// write sent the devices desired now.276func (w *SliceWriter) holds(current *ResourceSlice, devices []SliceDevice) bool {277	return w.own.Wrote(current.Metadata.ResourceVersion) && reflect.DeepEqual(w.sent, devices)278}279280// send writes the slice and remembers the version the API server281// answered and the devices sent. A write that fails remembers nothing,282// because a request that timed out can still have landed, and the next283// pass then compares the devices themselves.284func (w *SliceWriter) send(c *apiclient.Client, method, path string, slice *ResourceSlice, devices []SliceDevice) error {285	body, err := json.Marshal(slice)286	if err != nil {287		return err288	}289	w.sent = nil290	return w.own.Send(func() (string, error) {291		var answer ResourceSlice292		if err := c.RequestJSON(method, path, body, &answer); err != nil {293			return "", err294		}295		w.sent = slices.Clone(devices)296		return answer.Metadata.ResourceVersion, nil297	})298}299300// AttrString builds a string-typed attribute value without repeating301// pointer syntax at every call site.302func AttrString(s string) DeviceAttribute { return DeviceAttribute{String: &s} }303304// AttrBool builds a boolean attribute value. A selector reads it as a305// boolean, so a DeviceClass asks device.attributes["liken.sh"].x306// rather than comparing it against the string "true".307func AttrBool(b bool) DeviceAttribute { return DeviceAttribute{Bool: &b} }
kubernetes/secrets.go 100.0%
1package kubernetes23// This file reads liken's one Secret: the fleet's registry4// credentials.5//6// Secrets are the API's structure for confidential material. The7// machine operator reads exactly one Secret, by its exact name,8// under RBAC that grants get on that name and nothing else (the9// operator's manifest carries the Role). The well-known name is what10// makes this narrow grant possible: no configuration points at an11// arbitrary Secret, so the operator never needs a wider permission.1213import (14	"errors"15	"net/http"1617	"github.com/liken-sh/liken/kubernetes/apiclient"18)1920// RegistryCredentialsSecretPath names the URL where the fleet's21// registry credentials live: a kubernetes.io/dockerconfigjson Secret22// named registry-credentials in liken-system, in the shape that23// `kubectl create secret docker-registry` produces. The URL carries24// a namespace segment because Secrets live inside a namespace,25// unlike liken's own cluster-scoped CRDs.26const RegistryCredentialsSecretPath = "/api/v1/namespaces/liken-system/secrets/" + RegistryCredentialsSecret2728// RegistryCredentialsSecret is the Secret's name, which the machine29// operator's watch selects by.30const RegistryCredentialsSecret = "registry-credentials"3132// Secret holds the part of a Kubernetes Secret that liken reads: its33// type, which says what the data means, and the data itself. The API34// serves data values base64-encoded. Using []byte tells encoding/json35// to decode that value automatically.36type Secret struct {37	Type string            `json:"type"`38	Data map[string][]byte `json:"data"`39}4041// GetRegistryCredentialsSecret reads the fleet's registry42// credentials. An absent Secret returns nil, nil. A fleet with43// anonymous mirrors, or with no registries at all, is a normal44// state, not an error.45func GetRegistryCredentialsSecret(c *apiclient.Client) (*Secret, error) {46	secret := &Secret{}47	err := c.RequestJSON(http.MethodGet, RegistryCredentialsSecretPath, nil, secret)48	if errors.Is(err, apiclient.ErrNotFound) {49		return nil, nil50	}51	if err != nil {52		return nil, err53	}54	return secret, nil55}
kubernetes/services.go 100.0%
1package kubernetes23import "github.com/liken-sh/liken/kubernetes/apiclient"45// This file reads Services for one purpose: to count the Services of6// type LoadBalancer that the cluster still holds.7//8// A LoadBalancer Service is served by a cloud controller, which9// assigns the address and holds the Service's10// service.kubernetes.io/load-balancer-cleanup finalizer. The11// finalizer keeps the Service in place until the controller releases12// the address and clears the finalizer, and no other component13// clears it. Where the controller has stopped, a deleted14// LoadBalancer Service therefore never finishes deleting. The15// machine operator counts these Services before the feature that16// carries the controller may stop (machine-operator/retraction.go).1718// Service holds the part of a Kubernetes Service that liken reads:19// where the Service lives, and which kind of Service it is. Ports and20// selectors stay out of this type. The type alone states whether21// anything outside the cluster's own networking serves the Service.22type Service struct {23	Metadata struct {24		Name      string `json:"name"`25		Namespace string `json:"namespace"`26	} `json:"metadata"`27	Spec struct {28		Type string `json:"type"`29	} `json:"spec"`30}3132// ListLoadBalancerServices reads every Service in the cluster and33// keeps the ones of type LoadBalancer. The filter runs here rather34// than at the server, because a field selector can name only the35// fields the API server indexes, and spec.type is not one of them.36// The whole collection is cheap to read anyway: a Service is a37// per-workload object, not a per-pod one.38func ListLoadBalancerServices(c *apiclient.Client) ([]Service, error) {39	services, err := List[Service](c, "/api/v1/services")40	if err != nil {41		return nil, err42	}43	var balanced []Service44	for _, s := range services {45		if s.Spec.Type == "LoadBalancer" {46			balanced = append(balanced, s)47		}48	}49	return balanced, nil50}
kubernetes/watch/reads.go 96.7%
1// Package watch holds what liken's two operators add to the shared2// watches in kubernetes/informer: the handlers that wake a pass3// (wake.go), and the reads a pass makes from a watch's store.4//5// A pass reads the store instead of the API server, so a settled pass6// sends no read of a watched kind. The shared informer package answers7// a read only while the store is ready: it holds the whole first read,8// and the API server accepted a watch and forbade none since. Until9// then, the pass reads the API server.10//11// liken's watches each select exactly the objects a pass reads, often12// one object by name. So a ready store that does not hold an object13// answers that the object does not exist, and the pass sends no14// request to learn it. That is the one rule these reads add to the15// shared ones.16//17// The CLI imports the kubernetes package and watches nothing. This is18// a package of its own, apart from that one, so the CLI links no19// client-go.20package watch2122import (23	"errors"24	"slices"25	"strings"2627	"k8s.io/apimachinery/pkg/apis/meta/v1/unstructured"28	"k8s.io/client-go/tools/cache"2930	"github.com/liken-sh/liken/kubernetes/apiclient"31	"github.com/liken-sh/liken/kubernetes/informer"32)3334// Get reads one object from a store by its key: the name of a35// cluster-scoped object, or namespace/name. ok is false when the store36// cannot answer: it is not ready, or the object does not convert37// (which is logged). The caller then reads the API server. found is38// false when a ready store holds no such object.39func Get[T any](view informer.View, key string) (item *T, found, ok bool) {40	if !view.Ready() {41		return nil, false, false42	}43	object, exists, err := view.Store.GetByKey(key)44	if err != nil {45		return nil, false, false46	}47	if !exists {48		return nil, false, true49	}50	out, err := informer.Convert[T](object)51	if err != nil {52		informer.Report("the cached "+key, err)53		return nil, false, false54	}55	return &out, true, true56}5758// List reads every object in a store, in the order of their keys. ok59// is false when the store cannot answer, and the caller then reads the60// API server. An object that does not convert makes the whole answer61// unusable, because a pass that judges a collection must not judge it62// with an object missing: a heartbeat Lease left out would read as a63// machine that stopped renewing.64func List[T any](view informer.View) (items []T, ok bool) {65	if !view.Ready() {66		return nil, false67	}68	return convertAll[T](view.Store.List())69}7071// ByIndex reads the objects of one index value, the same way List72// reads the whole store.73func ByIndex[T any](view informer.View, index, value string) (items []T, ok bool) {74	if !view.Ready() {75		return nil, false76	}77	indexer, isIndexer := view.Store.(cache.Indexer)78	if !isIndexer {79		return nil, false80	}81	objects, err := indexer.ByIndex(index, value)82	if err != nil {83		return nil, false84	}85	return convertAll[T](objects)86}8788func convertAll[T any](objects []any) ([]T, bool) {89	slices.SortFunc(objects, func(a, b any) int { return strings.Compare(objectKey(a), objectKey(b)) })90	items := make([]T, 0, len(objects))91	for _, object := range objects {92		item, err := informer.Convert[T](object)93		if err != nil {94			informer.Report("the cached objects", err)95			return nil, false96		}97		items = append(items, item)98	}99	return items, true100}101102// objectKey is the key a store holds an object under.103func objectKey(object any) string {104	key, _ := cache.MetaNamespaceKeyFunc(object)105	return key106}107108// LabelIndex is an index of the objects by the value of one label.109func LabelIndex(label string) cache.IndexFunc {110	return func(object any) ([]string, error) {111		item, ok := object.(*unstructured.Unstructured)112		if !ok {113			return nil, nil114		}115		value, set := item.GetLabels()[label]116		if !set {117			return nil, nil118		}119		return []string{value}, nil120	}121}122123// ReadOne reads one object of a kind this operator writes, through the124// memo of its writes (the shared memo package). The store's copy125// answers when it is at the version of the operator's last write or126// read, and the API server answers otherwise, once, and then the store127// answers again. When the store holds the noted version, the memo128// drops its record (Settle).129//130// A ready store that holds no such object, and whose memo holds no131// record of it, answers apiclient.ErrNotFound with no request. An132// object the memo noted, such as one whose write failed, is read from133// the API server until the store holds it or the API server answers134// that it is gone. Then the memo forgets it, so the next read answers135// from the store and sends nothing.136func ReadOne[T any, P informer.Object[T]](c *apiclient.Client, held informer.Held, key, path string) (*T, error) {137	if held.View.Ready() {138		if _, stored, _ := held.View.Store.GetByKey(key); !stored && !held.Versions.Noted(key) {139			return nil, apiclient.ErrNotFound140		}141	}142	found, err := informer.ReadOne[T, P](c, held, key, path)143	if errors.Is(err, apiclient.ErrNotFound) {144		held.Versions.Forget(key)145	}146	Settle(held, key)147	return found, err148}149150// Settle drops the memo's record of each key whose copy in a ready151// store is at the version the memo noted. From then on the store holds152// this operator's write, and each later copy is newer, so the record153// guards nothing. A record that stayed would make each later write from154// another writer, such as a machine operator's status write to a155// Machine the cluster operator granted a turn, cost one read from the156// API server.157//158// Settle, like the Forget in ReadOne, drops the memo's request lock of159// the key (memo.Versions.Forget says what that allows). Each of liken's160// operators sends its writes of an object and reads it from the one161// goroutine of its pass loop, so no request about the key runs while162// Settle forgets it.163func Settle(held informer.Held, keys ...string) {164	if held.Versions == nil || !held.View.Ready() {165		return166	}167	for _, key := range keys {168		object, exists, err := held.View.Store.GetByKey(key)169		if err != nil || !exists {170			continue171		}172		if item, ok := object.(*unstructured.Unstructured); ok {173			held.Versions.ForgetAt(key, item.GetResourceVersion())174		}175	}176}
kubernetes/watch/wake.go 100.0%
1package watch23// A liken operator runs one full pass for each wake, and the pass reads4// every object it judges again, from the copies. So a watch's handler5// does one thing: it decides whether a change needs a pass, and wakes6// the loop when it does. The loop's channel has one slot, so a burst of7// changes makes one wake.8//9// Each handler converts the object into the operator's struct before10// it wakes the loop, and logs an object that does not convert. The pass11// that follows cannot use that object either, and the log line is the12// only place a person learns why.1314import (15	"maps"16	"reflect"1718	"k8s.io/apimachinery/pkg/apis/meta/v1/unstructured"19	"k8s.io/client-go/tools/cache"2021	"github.com/liken-sh/liken/kubernetes/informer"22)2324// Signal returns a wake function for a channel with one slot. A send25// that finds the slot full is dropped, because the wake already waiting26// starts a pass that reads everything this change did.27func Signal(wake chan<- struct{}) func() {28	return func() {29		select {30		case wake <- struct{}{}:31		default:32		}33	}34}3536// WakeOnChange wakes the loop for every change to an object: a new37// object, a removed object, and any write to an object. It is the38// handler for a collection whose status the pass reads, such as a39// Machine that another writer reports on. An update that the informer40// delivers after it reads the collection again, for an object whose41// resourceVersion did not move, is no change and does not wake.42func WakeOnChange[T any](source informer.Source, wake func()) cache.ResourceEventHandler {43	h := wakeHandler[T]{source: source, wake: wake, changed: func(before, after *unstructured.Unstructured) bool {44		return before.GetResourceVersion() != after.GetResourceVersion()45	}}46	return h.handler()47}4849// WakeOnAnotherWritersChange is WakeOnChange for an object that the50// operator writes itself. An update at a version that own answers as51// the operator's own last write is the echo of that write, and wakes52// nothing. Every other version wakes the loop, so a change that53// another writer makes after the operator's write still wakes a pass.54// It is the handler for an operator's own Machine, whose status the55// operator writes on a pass, and the cluster operator writes too.56func WakeOnAnotherWritersChange[T any](source informer.Source, wake func(), own func(version string) bool) cache.ResourceEventHandler {57	h := wakeHandler[T]{source: source, wake: wake, changed: func(before, after *unstructured.Unstructured) bool {58		return before.GetResourceVersion() != after.GetResourceVersion() && !own(after.GetResourceVersion())59	}}60	return h.handler()61}6263// WakeOnEdit wakes the loop only for an edit: a new object, a removed64// object, and a write that changed the spec, the deletion mark, or the65// object's identity. It is the handler for a collection whose spec the66// pass acts on, where the writes to its status, such as this67// operator's own, must not wake a pass.68//69// metadata.generation is the API server's count of spec changes, and a70// write to the status subresource does not raise it. The deletion mark71// is compared as well, so a deletion request wakes the loop whether or72// not the API server raised the generation when it set73// deletionTimestamp. The UID is compared because, after a gap in the74// watch, an object that somebody deleted and created again with the75// same name reaches the handler as an update, and the new object can76// have the generation the old one had.77func WakeOnEdit[T any](source informer.Source, wake func()) cache.ResourceEventHandler {78	h := wakeHandler[T]{source: source, wake: wake, changed: func(before, after *unstructured.Unstructured) bool {79		return before.GetUID() != after.GetUID() ||80			before.GetGeneration() != after.GetGeneration() ||81			(before.GetDeletionTimestamp() == nil) != (after.GetDeletionTimestamp() == nil)82	}}83	return h.handler()84}8586// WakeOnContent wakes the loop when an object's content changes: a new87// object, a removed object, and a write that changed anything but the88// resourceVersion and the managedFields. It is the handler for a89// collection whose informer trims each object to the fields the pass90// reads, or whose writers rewrite fields the pass does not read. The91// kubelet writes a pod's status every few seconds while a container92// restarts or a probe runs, and each write moves the resourceVersion.93// After the trim, a write that changed no field the pass reads leaves94// the two copies equal except for the version, and wakes no pass.95//96// Each ignore function removes more fields from a copy of each object97// before the comparison, for a field that a writer rewrites on a timer,98// such as the heartbeat times in a Node's conditions.99func WakeOnContent[T any](source informer.Source, wake func(), ignore ...func(fields map[string]any)) cache.ResourceEventHandler {100	h := wakeHandler[T]{source: source, wake: wake, changed: func(before, after *unstructured.Unstructured) bool {101		return !reflect.DeepEqual(contentOf(before, ignore), contentOf(after, ignore))102	}}103	return h.handler()104}105106// contentOf answers a copy of an object with no resourceVersion and no107// managedFields, and with the fields that each ignore function removes.108// The informer's copy stays as it is.109func contentOf(object *unstructured.Unstructured, ignore []func(map[string]any)) map[string]any {110	var fields map[string]any111	if len(ignore) > 0 {112		fields = object.DeepCopy().Object113	} else {114		fields = maps.Clone(object.Object)115	}116	if metadata, ok := fields["metadata"].(map[string]any); ok {117		metadata = maps.Clone(metadata)118		delete(metadata, "resourceVersion")119		delete(metadata, "managedFields")120		fields["metadata"] = metadata121	}122	for _, drop := range ignore {123		drop(fields)124	}125	return fields126}127128// wakeHandler is the shared half of the four handlers above. changed129// compares the copy the informer held with the new copy. The informer130// hands an update both copies, so the handler keeps no copy of its own.131// After a gap in the watch, the informer reads the collection again132// and reports each difference from what it held as an addition, an133// update, or a deletion, so a change made during the gap reaches the134// handler too.135type wakeHandler[T any] struct {136	source  informer.Source137	wake    func()138	changed func(before, after *unstructured.Unstructured) bool139}140141func (h wakeHandler[T]) handler() cache.ResourceEventHandler {142	return cache.ResourceEventHandlerFuncs{143		AddFunc:    h.added,144		UpdateFunc: h.updated,145		DeleteFunc: h.removed,146	}147}148149func (h wakeHandler[T]) added(object any) {150	if _, err := informer.Convert[T](object); err != nil {151		informer.Report(h.source.String(), err)152		return153	}154	h.wake()155}156157// removed wakes the loop even for an object that does not convert. A158// tombstone can hold no copy of the object at all, and one extra pass159// costs less than a removal the pass never judged.160func (h wakeHandler[T]) removed(object any) {161	if _, err := informer.Convert[T](object); err != nil {162		informer.Report(h.source.String(), err)163	}164	h.wake()165}166167// updated wakes the loop when changed says so. A held copy that is not168// an object says nothing about what changed, so the update counts as a169// change.170func (h wakeHandler[T]) updated(before, after any) {171	if _, err := informer.Convert[T](after); err != nil {172		informer.Report(h.source.String(), err)173		return174	}175	held, heldOK := before.(*unstructured.Unstructured)176	fresh, freshOK := after.(*unstructured.Unstructured)177	if !heldOK || !freshOK || h.changed(held, fresh) {178		h.wake()179	}180}
kubernetes/workloads.go 96.3%
1package kubernetes23// This file reads the workloads that ship a workstation CLI. An4// operator runs as a Deployment, a DaemonSet, or a StatefulSet, and5// the one that carries a CLI marks itself with the cli.liken.sh/plugin6// label, whose value names the domain (audio, display, media,7// bluetooth, library). The base binary's plugins commands read this8// list, take each workload's container image and its version tag, and9// pull the matching CLI image from the same registry at the same10// version.11//12// The three kinds share one pod-template shape, so one type reads all13// of them. bluetooth-operator ships only a DaemonSet, so a selector14// over Deployments alone would miss it.1516import (17	"net/url"18	"strings"1920	"github.com/liken-sh/liken/kubernetes/apiclient"21)2223// PluginLabel is the label a workload carries to declare that it ships24// a workstation CLI. Its value is the domain.25const PluginLabel = "cli.liken.sh/plugin"2627// WorkloadContainer holds the one field the plugins commands read from28// a container: the image reference, whose tag names the version every29// CLI image shares with its operator.30type WorkloadContainer struct {31	Name  string `json:"name"`32	Image string `json:"image"`33}3435// Workload holds the part of a Deployment, DaemonSet, or StatefulSet36// the plugins commands read: which domain the label names, and the37// images its pod template runs. The three kinds carry these fields at38// the same paths, so one type decodes all of them.39type Workload struct {40	Metadata struct {41		Name      string            `json:"name"`42		Namespace string            `json:"namespace"`43		Labels    map[string]string `json:"labels"`44	} `json:"metadata"`45	Spec struct {46		Template struct {47			Spec struct {48				Containers []WorkloadContainer `json:"containers"`49			} `json:"spec"`50		} `json:"template"`51	} `json:"spec"`52}5354// PluginDomain reports the domain the workload's plugin label names,55// or the empty string when the label is absent.56func (w *Workload) PluginDomain() string {57	return w.Metadata.Labels[PluginLabel]58}5960// OperatorImage reports the first container's image, the reference61// whose tag names the operator's version. An operator's pod runs its62// operator as its first container.63func (w *Workload) OperatorImage() string {64	if len(w.Spec.Template.Spec.Containers) == 0 {65		return ""66	}67	return w.Spec.Template.Spec.Containers[0].Image68}6970// pluginWorkloadKinds names the plural resources the selector reads.71// All three live under the apps/v1 group and carry the pod template at72// the same path.73var pluginWorkloadKinds = []string{"deployments", "daemonsets", "statefulsets"}7475// ListPluginWorkloads reads every Deployment, DaemonSet, and76// StatefulSet across the cluster that carries the plugin label, then77// keeps one workload per domain. The label selector names only the78// key, so a workload with any value for it is returned. The collection79// with no namespace segment spans every namespace.80func ListPluginWorkloads(c *apiclient.Client) ([]Workload, error) {81	var all []Workload82	for _, kind := range pluginWorkloadKinds {83		path := "/apis/apps/v1/" + kind + "?labelSelector=" + url.QueryEscape(PluginLabel)84		workloads, err := List[Workload](c, path)85		if err != nil {86			return nil, err87		}88		all = append(all, workloads...)89	}90	return dedupeByDomain(all), nil91}9293// dedupeByDomain keeps one workload per domain. Two workloads may94// carry the same domain label, so the choice prefers the one whose95// image path names the domain's operator, and otherwise keeps the96// first seen. The read order (deployments, daemonsets, statefulsets)97// makes the fallback stable.98func dedupeByDomain(workloads []Workload) []Workload {99	index := map[string]int{}100	result := []Workload{}101	for _, w := range workloads {102		domain, image := w.PluginDomain(), w.OperatorImage()103		if domain == "" || image == "" {104			continue105		}106		i, seen := index[domain]107		if !seen {108			index[domain] = len(result)109			result = append(result, w)110			continue111		}112		if imageNamesOperator(image, domain) && !imageNamesOperator(result[i].OperatorImage(), domain) {113			result[i] = w114		}115	}116	return result117}118119// imageNamesOperator reports whether an image path names the domain's120// operator, the signal that picks between two workloads that share a121// domain label. An operator image is named <domain>-operator.122func imageNamesOperator(image, domain string) bool {123	return strings.Contains(image, domain+"-operator")124}
logs/cursor.go 90.9%
1package main23// The cursor lets a restarted relay resume from its last position4// instead of resending everything it can see. It lives in the pod's5// emptyDir, which survives container restarts, the common failure (a6// crash or an OOM kill). It also survives a reboot: the Pod object7// outlives the machine's restart, so kubelet reuses the same8// emptyDir on the durable pod storage. A reboot restarts the9// kernel's sequence numbers at zero, so a cursor read after one10// names records this kernel never printed, and the kmsg relay dates11// its cursor against the machine's uptime to detect exactly that12// (kmsg.go, earlierBoot). The seq field in every envelope is what13// lets a consumer remove duplicate records from any replay.14//15// This file considered durable stores and rejected them.16// Checkpointing through the cluster API would push node-local state17// that changes on every batch through etcd, where each server pays18// for a consensus round and a disk write on every update. A cursor19// file on the host would give a read-only relay a write access it20// otherwise never needs.2122import (23	"encoding/json"24	"os"25	"path/filepath"26	"time"27)2829const cursorFile = "cursor.json"3031// checkpointInterval limits how often the relay writes the cursor.32// Being at most a second behind means the relay resends at most a33// second of records after a container restart, which costs less34// than a disk write for every record. This is a package variable,35// not a constant, so tests can checkpoint on every record.36var checkpointInterval = time.Second3738// loadCursor reads the cursor into the given struct, and reports39// whether a usable cursor existed. A missing or corrupt cursor means40// a fresh start, never an error. The worst outcome of losing a41// cursor is one replay that the consumer can deduplicate.42func loadCursor(dir string, into any) bool {43	data, err := os.ReadFile(filepath.Join(dir, cursorFile))44	if err != nil {45		return false46	}47	return json.Unmarshal(data, into) == nil48}4950// saveCursor writes the cursor using a temp-file-and-rename. If a51// crash happens mid-write, the previous cursor stays intact instead52// of becoming a torn cursor. The rename is atomic because both names53// live in the same emptyDir. This function does not call fsync on54// purpose: the emptyDir does not survive the pod, so durability55// across a crash of the whole machine gives no benefit.56func saveCursor(dir string, v any) error {57	data, err := json.Marshal(v)58	if err != nil {59		return err60	}61	tmp := filepath.Join(dir, cursorFile+".tmp")62	if err := os.WriteFile(tmp, data, 0o600); err != nil {63		return err64	}65	return os.Rename(tmp, filepath.Join(dir, cursorFile))66}
logs/envelope.go 92.9%
1package main23// The envelope is the relay's entire output contract: one JSON object4// per line on stdout, a thin structured wrapper around a verbatim5// message. The rule for what moves into fields is strict: only what6// the source format itself defines as a header (kmsg's priority and7// sequence, logrus's time= and level=). The relay never parses or8// rewrites the message body. Making sense of the body is the job of9// whatever log stack someone chooses to run, not the job of the OS.10//11// Event time travels inside the envelope, because the container12// runtime stamps log lines at the moment the relay wrote them, and a13// relay that replays a boot's worth of records writes them all at14// once. The runtime's timestamp answers "when was this relayed". The15// envelope's time field answers "when did this happen".1617import (18	"bytes"19	"encoding/json"20	"io"21	"time"22)2324// envelope is one relayed log line. The field order here sets the25// field order in the output, because encoding/json writes struct26// fields in declaration order.27type envelope struct {28	// Time is the event time in RFC3339 with nanoseconds, UTC.29	Time string `json:"time"`3031	// Severity is a syslog severity word (emerg through debug). It32	// comes from the source's header, or defaults to info.33	Severity string `json:"severity"`3435	// Facility appears only on the kmsg relays. There, it is the36	// syslog facility that separates the kernel's records (0) from37	// userspace's records (1). It is a pointer so that the kernel's38	// zero value serializes, instead of disappearing behind39	// omitempty.40	Facility *int `json:"facility,omitempty"`4142	// Seq orders and deduplicates records within one source. For43	// kmsg, this is the kernel's own sequence number. For tailed44	// files, this is the line's starting byte offset. A consumer45	// that sees the same (source, seq) twice is seeing a replay, not46	// a new event.47	Seq uint64 `json:"seq"`4849	// Message is the original line, verbatim.50	Message string `json:"message"`51}5253// envelopeWriter encodes envelopes one per line, and delivers each54// line to the underlying writer in a single Write call. That single55// call matters: the container runtime treats each write to the pod's56// stdout pipe as one unit. Writing a whole line per call is what57// keeps envelopes from interleaving in the pod's log file.58type envelopeWriter struct {59	w   io.Writer60	buf bytes.Buffer61	enc *json.Encoder62}6364func newEnvelopeWriter(w io.Writer) *envelopeWriter {65	ew := &envelopeWriter{w: w}66	ew.enc = json.NewEncoder(&ew.buf)67	// Log lines are full of < and > characters, from Go struct dumps68	// and YAML snippets. Escaping them for HTML would corrupt the69	// verbatim body.70	ew.enc.SetEscapeHTML(false)71	return ew72}7374// emit writes one envelope as one line. Encode appends the newline.75func (ew *envelopeWriter) emit(e envelope) error {76	ew.buf.Reset()77	if err := ew.enc.Encode(e); err != nil {78		return err79	}80	_, err := ew.w.Write(ew.buf.Bytes())81	return err82}8384// notice emits an envelope about the relay itself, such as a85// lost-records warning or a startup marker. Notices follow the same86// contract as every other envelope, so a consumer never needs a87// second parser. The liken-logs: prefix marks these envelopes as88// coming from the relay, not from the source it reads.89func (ew *envelopeWriter) notice(now time.Time, severity string, seq uint64, facility *int, message string) error {90	return ew.emit(envelope{91		Time:     now.UTC().Format(time.RFC3339Nano),92		Severity: severity,93		Facility: facility,94		Seq:      seq,95		Message:  "liken-logs: " + message,96	})97}
logs/kmsg.go 87.0%
1package main23// The kernel's log buffer has a readable device, /dev/kmsg. Its4// format is the one format that the kernel relays parse. Each5// read(2) call returns exactly one record:6//7//	priority,sequence,timestamp,flags[,caller];message8//	 SUBSYSTEM=...9//	 DEVICE=...10//11// The priority byte packs two syslog facts: facility<<3 | severity.12// Facility separates the machine's two streams that share this13// buffer. Records that the kernel printed carry facility 0. Records14// that userspace wrote through /dev/kmsg, such as init's own lines,15// carry facility 1. Each relay sends one facility, so the pod that a16// line came from answers "which program wrote this", instead of17// matching strings.18//19// The sequence number is the buffer's own ordering. It counts every20// record, regardless of facility. Gaps within one relay's output are21// therefore normal, because the other facility's records consumed22// those numbers. Actual loss is detectable: a reader that has fallen23// behind a wrapping buffer gets EPIPE from read(2), and the relay24// writes a notice into the stream. The timestamp is microseconds25// since boot, on the kernel's approximately monotonic printk clock.26// This file converts the timestamp to wall time by anchoring it27// against the current clocks.28//29// The device blocks a reader that has caught up until the next30// record arrives, so the read loop is also the follow mechanism;31// there is no polling. A new reader starts at the oldest record the32// buffer still holds. This is what makes a fresh pod replay the boot33// from its start. The device cannot seek to a sequence number, so34// resuming from a cursor means reading from the oldest record and35// skipping records until the reader passes the cursor.3637import (38	"bytes"39	"errors"40	"fmt"41	"os"42	"strconv"43	"strings"44	"syscall"45	"time"4647	"golang.org/x/sys/unix"4849	"github.com/liken-sh/liken/liken/machine"50)5152const kmsgPath = "/dev/kmsg"5354// kmsgRecord is one parsed record: the header's facts, plus the55// message line, verbatim. Continuation lines, the SUBSYSTEM=/DEVICE=56// dictionary that some records append, are device metadata, not part57// of the message. This code drops them.58type kmsgRecord struct {59	Facility int60	Severity int61	Seq      uint6462	Stamp    time.Duration // since boot63	Message  string64}6566func parseKmsgRecord(raw []byte) (kmsgRecord, error) {67	semi := bytes.IndexByte(raw, ';')68	if semi < 0 {69		return kmsgRecord{}, fmt.Errorf("no header separator in %q", raw)70	}71	fields := strings.Split(string(raw[:semi]), ",")72	if len(fields) < 4 {73		return kmsgRecord{}, fmt.Errorf("header %q has %d fields, want at least 4", raw[:semi], len(fields))74	}75	prio, err := strconv.Atoi(fields[0])76	if err != nil || prio < 0 {77		return kmsgRecord{}, fmt.Errorf("bad priority in header %q", raw[:semi])78	}79	seq, err := strconv.ParseUint(fields[1], 10, 64)80	if err != nil {81		return kmsgRecord{}, fmt.Errorf("bad sequence in header %q", raw[:semi])82	}83	us, err := strconv.ParseInt(fields[2], 10, 64)84	if err != nil || us < 0 {85		return kmsgRecord{}, fmt.Errorf("bad timestamp in header %q", raw[:semi])86	}87	message := raw[semi+1:]88	if nl := bytes.IndexByte(message, '\n'); nl >= 0 {89		message = message[:nl]90	}91	return kmsgRecord{92		Facility: prio >> 3,93		Severity: prio & 7,94		Seq:      seq,95		Stamp:    time.Duration(us) * time.Microsecond,96		Message:  string(message),97	}, nil98}99100// kmsgCursor is the resume point: the last sequence number this101// relay has read. This is the last number read, not the last number102// sent, so a restart does not scan the other facility's records103// again either.104//105// The stamp is the monotonic timestamp of the last record read, and106// it dates the cursor against the boot that wrote it: a stamp larger107// than the current uptime can only have come from an earlier boot.108type kmsgCursor struct {109	Seq   uint64        `json:"seq"`110	Stamp time.Duration `json:"stamp"`111}112113// The uptime and the record stamps come from clocks that sample at114// different moments, so the comparison below carries one second of115// slack rather than demanding exact agreement.116const kmsgClockSlack = time.Second117118// earlierBoot reports whether the cursor was written by a boot119// before this one. The kernel restarts kmsg sequence numbers at zero120// on every boot, so a cursor that survives a reboot names records121// this kernel never printed, and comparing sequences would silently122// drop the whole new boot's log until the numbers caught up. The123// buffer's records cannot settle it either, because a resumed reader124// starts at the oldest surviving record, whose stamp is legitimately125// far below the cursor's. The uptime can: no record of this boot can126// carry a stamp beyond the time this boot has been running, so a127// cursor stamped past it is from an earlier boot.128func (c kmsgCursor) earlierBoot(uptime time.Duration) bool {129	return c.Stamp > uptime+kmsgClockSlack130}131132// kmsgRelay follows one facility of the kernel buffer. The read133// function reads from /dev/kmsg in production, and from a scripted134// fixture in tests. anchor reports the wall-clock moment of boot,135// based on how the clocks currently stand.136type kmsgRelay struct {137	read      func([]byte) (int, error)138	facility  int139	out       *envelopeWriter140	cursorDir string141	anchor    func() time.Time142	now       func() time.Time143}144145func (r *kmsgRelay) run() error {146	var cur kmsgCursor147	resuming := loadCursor(r.cursorDir, &cur)148	// A cursor from an earlier boot is discarded, with a notice, so149	// this boot relays from its first record instead of losing its150	// log to the old boot's sequence numbers (earlierBoot above).151	if resuming && cur.earlierBoot(r.now().Sub(r.anchor())) {152		_ = r.out.notice(r.now(), "info", cur.Seq, &r.facility,153			fmt.Sprintf("the cursor holds sequence %d from an earlier boot; relaying this boot from its first record", cur.Seq))154		cur = kmsgCursor{}155		resuming = false156	}157	if resuming {158		_ = r.out.notice(r.now(), "info", cur.Seq, &r.facility,159			fmt.Sprintf("resuming after sequence %d", cur.Seq))160	}161162	var lastCheckpoint time.Time163	// first marks the first record parsed after a resume. This is164	// the only moment when a jump past the cursor reveals expired165	// records.166	first := true167	buf := make([]byte, 8192)168	for {169		n, err := r.read(buf)170		if err != nil {171			// EPIPE is the kernel's overrun signal: the buffer172			// wrapped past this reader's position. The kernel has173			// already moved the read position to the oldest174			// surviving record, so the only job here is to record175			// that a gap happened.176			if errors.Is(err, syscall.EPIPE) {177				_ = r.out.notice(r.now(), "warning", cur.Seq, &r.facility,178					"records were lost to a ring buffer overrun")179				continue180			}181			return err182		}183		rec, err := parseKmsgRecord(buf[:n])184		if err != nil {185			_ = r.out.notice(r.now(), "warning", cur.Seq, &r.facility,186				"unparseable record: "+err.Error())187			continue188		}189190		if resuming {191			// The buffer may have discarded records past the cursor192			// while the relay was down. This is the same kind of193			// loss as an overrun, and gets the same notice.194			if first && rec.Seq > cur.Seq+1 {195				_ = r.out.notice(r.now(), "warning", cur.Seq, &r.facility,196					"records expired while the relay was down")197			}198			first = false199			if rec.Seq <= cur.Seq {200				continue201			}202			resuming = false203		}204205		cur.Seq, cur.Stamp = rec.Seq, rec.Stamp206		if rec.Facility == r.facility {207			if err := r.out.emit(envelope{208				Time:     r.anchor().Add(rec.Stamp).UTC().Format(time.RFC3339Nano),209				Severity: machine.SeverityNames[rec.Severity],210				Facility: &r.facility,211				Seq:      rec.Seq,212				Message:  rec.Message,213			}); err != nil {214				return err215			}216		}217218		if now := r.now(); now.Sub(lastCheckpoint) >= checkpointInterval {219			lastCheckpoint = now220			if err := saveCursor(r.cursorDir, cur); err != nil {221				return err222			}223		}224	}225}226227// wallAnchor computes the wall-clock time of boot from the current228// clocks: CLOCK_REALTIME minus CLOCK_MONOTONIC. A record's wall time229// is then anchor plus its since-boot stamp. This file samples the230// clocks for every record. The two reads are vDSO calls, so they231// cost almost nothing. Sampling on every record means the conversion232// always uses the clock as it stands now. Because of this, even233// records from before init's one boot-time clock step land on the234// corrected timeline: the step moved CLOCK_REALTIME, and the stamps235// stay monotonic.236func wallAnchor() time.Time {237	var ts unix.Timespec238	if err := unix.ClockGettime(unix.CLOCK_MONOTONIC, &ts); err != nil {239		// Without the monotonic clock, which cannot really happen on240		// Linux, record times fall back to relay time.241		return time.Now()242	}243	return time.Now().Add(-(time.Duration(ts.Sec)*time.Second + time.Duration(ts.Nsec)))244}245246// relayKmsg follows /dev/kmsg, and sends one facility. Opening the247// device requires real privilege. The kernel demands CAP_SYSLOG,248// because CONFIG_SECURITY_DMESG_RESTRICT is set, and the container249// runtime's devices cgroup separately controls the open call. This250// is why the two kmsg containers run privileged (see251// logs/manifests/logs.yaml for that explanation).252func relayKmsg(facility int) error {253	f, err := os.Open(kmsgPath)254	if err != nil {255		return err256	}257	defer f.Close()258	relay := &kmsgRelay{259		read:      f.Read,260		facility:  facility,261		out:       newEnvelopeWriter(os.Stdout),262		cursorDir: cursorDir,263		anchor:    wallAnchor,264		now:       time.Now,265	}266	return relay.run()267}
logs/lift.go 100.0%
1package main23// Lifting is the narrow part of parsing that the relay is allowed to4// do. It recognizes the timestamp-and-level header that a log format5// puts at the front of every line, copies those two facts into the6// envelope, and leaves the line itself untouched. The k3s and7// containerd logs mix two such formats. k3s itself and containerd8// log through logrus (time="..." level=... msg=...). The Kubernetes9// components embedded in k3s log through klog, whose header is a10// single severity letter, the month and day, and a wall-clock time11// (I0707 13:51:16.123456 ...). A line that matches neither format12// ships with the relay's own observation time and info severity.13// This is exact for a line just written, and wrong only for14// unliftable lines replayed long after the fact.15//16// This file uses hand-written scanners instead of regular17// expressions, because each format is a fixed prefix. A handful of18// index checks is clearer to read than a pattern, runs on every19// single log line, and cannot backtrack.2021import (22	"strings"23	"time"24)2526// lift extracts the event time and severity word from a line's27// header. It tries logrus first, because k3s's own lines make up28// most of the volume, then tries klog, then falls back to the29// observation time.30func lift(line string, now time.Time) (time.Time, string) {31	if when, severity, ok := liftLogrus(line); ok {32		return when, severity33	}34	if when, severity, ok := liftKlog(line, now); ok {35		return when, severity36	}37	return now, "info"38}3940// logrusSeverities maps logrus's level words onto syslog severity41// words. Syslog has no trace, so trace joins debug.42var logrusSeverities = map[string]string{43	"panic":   "emerg",44	"fatal":   "crit",45	"error":   "err",46	"warning": "warning",47	"warn":    "warning",48	"info":    "info",49	"debug":   "debug",50	"trace":   "debug",51}5253// liftLogrus recognizes logrus's text format, which always starts54// with time="<RFC3339>" level=<word>. Any byte out of place means55// this line is not a logrus line, and the function abandons the56// whole lift instead of applying it halfway.57func liftLogrus(line string) (time.Time, string, bool) {58	const timePrefix = `time="`59	if len(line) < len(timePrefix) || line[:len(timePrefix)] != timePrefix {60		return time.Time{}, "", false61	}62	rest := line[len(timePrefix):]63	quote := strings.IndexByte(rest, '"')64	if quote < 0 {65		return time.Time{}, "", false66	}67	when, err := time.Parse(time.RFC3339, rest[:quote])68	if err != nil {69		return time.Time{}, "", false70	}7172	const levelPrefix = ` level=`73	rest = rest[quote+1:]74	if len(rest) < len(levelPrefix) || rest[:len(levelPrefix)] != levelPrefix {75		return time.Time{}, "", false76	}77	rest = rest[len(levelPrefix):]78	end := strings.IndexByte(rest, ' ')79	if end < 0 {80		end = len(rest)81	}82	severity, ok := logrusSeverities[rest[:end]]83	if !ok {84		return time.Time{}, "", false85	}86	return when, severity, true87}8889// klogSeverities maps klog's single-letter severities (Info, Warning,90// Error, Fatal) onto syslog words.91var klogSeverities = map[byte]string{92	'I': "info",93	'W': "warning",94	'E': "err",95	'F': "crit",96}9798// liftKlog recognizes klog's header: Lmmdd hh:mm:ss.uuuuuu, fixed99// width, with no year. The function takes the year from the100// observation clock, with one correction. A December line read in101// January would otherwise land eleven months in the future, so102// anything more than a day ahead of now is moved back a year. Log103// lines from the future are otherwise impossible. Log lines from104// months ago are simply from an old file.105func liftKlog(line string, now time.Time) (time.Time, string, bool) {106	// Lmmdd hh:mm:ss.uuuuuu: 21 bytes before the rest of the header.107	const headerLen = 21108	if len(line) < headerLen {109		return time.Time{}, "", false110	}111	severity, ok := klogSeverities[line[0]]112	if !ok {113		return time.Time{}, "", false114	}115	for _, i := range []int{1, 2, 3, 4, 6, 7, 9, 10, 12, 13, 15, 16, 17, 18, 19, 20} {116		if line[i] < '0' || line[i] > '9' {117			return time.Time{}, "", false118		}119	}120	if line[5] != ' ' || line[8] != ':' || line[11] != ':' || line[14] != '.' {121		return time.Time{}, "", false122	}123124	digits := func(from, to int) int {125		n := 0126		for _, c := range []byte(line[from:to]) {127			n = n*10 + int(c-'0')128		}129		return n130	}131	month := digits(1, 3)132	day := digits(3, 5)133	if month < 1 || month > 12 || day < 1 || day > 31 {134		return time.Time{}, "", false135	}136	when := time.Date(now.Year(), time.Month(month), day,137		digits(6, 8), digits(9, 11), digits(12, 14), digits(15, 21)*1_000,138		time.UTC)139	if when.After(now.Add(24 * time.Hour)) {140		when = when.AddDate(-1, 0, 0)141	}142	return when, severity, true143}
logs/main.go 0.0%
1// liken-logs relays the machine's host-level log streams into2// Kubernetes, one relay per stream, so that everything below the3// cluster becomes readable through the cluster. liken's kernel,4// init, k3s, and containerd write their logs to the serial console5// and to files on the machine. None of that is visible to `kubectl6// logs` or to any log stack someone might run, because a collector7// cannot tail a serial port. Each relay reads one stream at its8// source and sends it out again, line by line, on its own stdout.9// Because each relay runs as a pod, its stdout is a pod log: the10// Kubernetes API serves it, RBAC controls access to it, and any11// in-cluster collector reads it with no host privileges. The OS's12// log interface is `kubectl logs`, the same API as everything else.13//14// One binary carries all four relays, chosen by an argument verb15// (the multi-call pattern, the same pattern init uses to re-exec16// itself). The machine-logs DaemonSet runs all four as containers of17// one pod, each differing only in its verb and in what it mounts:18//19//	liken-logs kernel      /dev/kmsg, records with syslog facility 020//	liken-logs liken       /dev/kmsg, records with syslog facility 121//	liken-logs k3s         the k3s log on clusterState22//	liken-logs containerd  containerd's log on clusterState23//24// The kernel and init share /dev/kmsg. init writes its lines there so25// they interleave with the kernel's lines in true order. The26// facility field is what splits them back apart, so each pod carries27// exactly one program's records.28//29// A relay is crash-only. Any unexpected error exits nonzero, the30// kubelet restarts the container, and the cursor in the pod's31// emptyDir resumes the stream. There is no retry logic layered on32// top of this; restarting from a durable position is the retry33// logic.34package main3536import (37	"fmt"38	"os"3940	"github.com/liken-sh/liken/liken/machine"41)4243// These are the host paths each verb reads, and the emptyDir where44// cursors live. These paths are fixed facts of the DaemonSet's45// mounts. Tests point a relay at a temporary directory through its46// parameters instead.47const (48	cursorDir         = "/cursor"49	k3sLogPath        = "/var/lib/rancher/k3s/liken/k3s.log"50	containerdLogPath = "/var/lib/rancher/k3s/agent/containerd/containerd.log"51)5253func main() {54	// The version goes to stderr, because stdout carries only data:55	// the relay never writes anything to stdout except an envelope.56	fmt.Fprintln(os.Stderr, "liken-logs", machine.Version)5758	if len(os.Args) != 2 {59		usage()60		os.Exit(2)61	}62	var err error63	switch os.Args[1] {64	case "kernel":65		err = relayKmsg(machine.FacilityKernel)66	case "liken":67		err = relayKmsg(machine.FacilityUser)68	case "k3s":69		err = relayFile(k3sLogPath)70	case "containerd":71		err = relayFile(containerdLogPath)72	default:73		usage()74		os.Exit(2)75	}76	if err != nil {77		fmt.Fprintln(os.Stderr, "liken-logs:", err)78		os.Exit(1)79	}80}8182func usage() {83	fmt.Fprintln(os.Stderr, "usage: liken-logs kernel|liken|k3s|containerd")84}
logs/tail.go 90.3%
1package main23// The file relays implement a hand-written version of tail -F:4// follow a growing file, and when the file at the path suddenly5// becomes a different file, finish the old one and start the new6// one. This rename sequence is exactly what init's rotate-at-boot7// produces (along with the in-boot size cap on k3s.log). This is8// also why the DaemonSets mount the parent directory instead of the9// file: a bind mount of the file itself would keep the old inode10// pinned forever, while a lookup through the directory sees each new11// generation.12//13// Rotation is detected by identity, not by name. The tailer holds14// the inode it opened, and when the path's (device, inode) pair no15// longer matches, the held file is the renamed previous generation.16// The tailer drains that file to its final EOF, because the writer17// may have added a few last lines before the rename, and then18// reopens the path from the top. Rotated generations that sit on19// disk (.1, .2, ...) are never read. They exist for boot forensics,20// and shipping previous boots is left unsolved here on purpose.21//22// The follow mechanism is inotify on the parent directory. The kernel23// raises an event on the directory for each append and each rename, and24// the tailer waits on that event instead of a timer, so an idle log25// costs almost no wakeups and a growing one ships within a fraction of a26// millisecond. The watch is on the directory, not the file, for the27// same reason the mount is: the directory's inode is stable across the28// rename that rotation makes, while a watch on the file would go dead29// the moment the file is renamed away. The writer holds one descriptor30// open for the whole boot and only appends, so the growth event is31// IN_MODIFY; the descriptor never closes, so IN_CLOSE_WRITE never fires.32//33// An event is per-inode, so it crosses the DaemonSet's read-only bind34// mount: the writer appends on the host side and the tailer wakes on the35// container side. The two filesystems that hold liken's logs, ext4 on36// disk machines and tmpfs on memory machines, both raise IN_MODIFY on37// every append. A burst of appends coalesces into one wake, which is38// harmless, because the tailer reads to EOF on any wake and ships39// everything the burst wrote. An event is only a trigger. On a wake the40// tailer re-reads the file and re-stats the path to get the current41// state, and never trusts what the event claimed. A slow backstop timer42// bounds the damage if an event is ever lost.4344import (45	"bytes"46	"context"47	"errors"48	"fmt"49	"io"50	"io/fs"51	"os"52	"path/filepath"53	"syscall"54	"time"5556	"github.com/liken-sh/liken/liken/machine"57	"golang.org/x/sys/unix"58)5960// tailMask is the event set for the watch on a log's parent directory.61// IN_MODIFY is the growth event, because the writer only ever appends to62// a descriptor it holds open. IN_CREATE fires when a new generation63// appears, and also when containerd.log is first created. IN_MOVED_FROM64// fires when rotation renames the old generation away. Each is only a65// trigger to re-read and re-stat; the tailer never reads the event's66// content.67const tailMask = unix.IN_MODIFY | unix.IN_CREATE | unix.IN_MOVED_FROM6869// backstopInterval bounds how long a lost event can hide growth or a70// rotation. inotify raises an event on every append and every rename, so71// a healthy tailer wakes on the event and never waits this long. The72// timer only covers two faults that liken does not currently have: a73// watch whose reader stopped after a poll or read error, which never74// wakes the tailer again, and a filesystem that does not raise75// IN_MODIFY. A queue that overflows is not one of them, because the76// reader treats IN_Q_OVERFLOW as a wake. On either fault the tailer77// still catches up within one interval. A long interval keeps the idle78// cost near zero.79var backstopInterval = 30 * time.Second8081// tailCursor is the resume point: which file, identified by identity82// rather than name, and the offset of the first byte of the next83// line. Offsets always point at the start of a line. Because of84// this, a resumed relay re-reads a partially sent line as a whole85// line, instead of sending a fragment.86type tailCursor struct {87	Dev    uint64 `json:"dev"`88	Ino    uint64 `json:"ino"`89	Offset int64  `json:"offset"`90}9192// fileIdentity reads the (device, inode) pair that identifies a file93// independent of its path.94func fileIdentity(info fs.FileInfo) (uint64, uint64) {95	st := info.Sys().(*syscall.Stat_t)96	return uint64(st.Dev), st.Ino97}9899// awaitFile opens the path, and waits for the file to exist.100// containerd.log is not created until k3s brings containerd up.101// After a rotation, there is also a moment before the writer creates102// the file again.103func awaitFile(ctx context.Context, path string, wake <-chan struct{}) (*os.File, error) {104	for {105		f, err := os.Open(path)106		if err == nil {107			return f, nil108		}109		if !errors.Is(err, fs.ErrNotExist) {110			return nil, err111		}112		select {113		case <-ctx.Done():114			return nil, ctx.Err()115		case _, ok := <-wake:116			if !ok {117				return nil, errWatchStopped118			}119		case <-time.After(backstopInterval):120		}121	}122}123124// errWatchStopped ends the tailer when its directory watch fails. The125// relay exits nonzero and the kubelet starts it again, for the same126// reason a watch that cannot start ends it: a tailer that kept going127// on the backstop alone would hide the fault.128var errWatchStopped = errors.New("the watch on the log directory stopped")129130// tailFile follows one log file forever, until the context ends. It131// sends an envelope for each line, using the line's starting byte132// offset as its sequence number. Each pass of the loop handles one133// generation of the file: open it, determine where to start reading,134// and follow it until a rotation replaces it.135func tailFile(ctx context.Context, path string, out *envelopeWriter, curDir string, now func() time.Time) error {136	var cur tailCursor137	resuming := loadCursor(curDir, &cur)138139	// Establish the watch on the parent directory before the first open,140	// so a file that is created between the open and the watch still141	// raises an event this tailer will see. A watch that cannot start is142	// a real fault: the relay exits nonzero and the kubelet restarts it,143	// rather than falling back to a poll that would hide the fault.144	wake, err := machine.WatchDirMask(ctx, filepath.Dir(path), tailMask)145	if err != nil {146		return err147	}148149	for {150		f, err := awaitFile(ctx, path, wake)151		if err != nil {152			return err153		}154		info, err := f.Stat()155		if err != nil {156			f.Close()157			return err158		}159		dev, ino := fileIdentity(info)160161		// Resume only into the same file that the cursor described,162		// and only within its current size. A shrunken file means163		// something truncated it in place, and nothing on a liken164		// machine should do that. Report this and start over, instead165		// of reading from the middle of unrelated bytes.166		offset := int64(0)167		if resuming && dev == cur.Dev && ino == cur.Ino {168			if cur.Offset <= info.Size() {169				offset = cur.Offset170				_ = out.notice(now(), "info", uint64(offset), nil,171					fmt.Sprintf("resuming %s at offset %d", path, offset))172			} else {173				_ = out.notice(now(), "warning", uint64(cur.Offset), nil,174					fmt.Sprintf("%s shrank below the cursor; replaying from the head", path))175			}176		}177		resuming = false178179		if err := followGeneration(ctx, f, path, offset, dev, ino, out, curDir, now, wake); err != nil {180			return err181		}182	}183}184185// followGeneration reads one generation of the file. It seeks to the186// starting offset, follows the file as it grows, and when a rotation187// replaces the file at the path, drains the renamed generation to188// its final EOF. This function owns the open file, and closes it on189// every return. A nil return means the generation rotated, and the190// caller should reopen the path.191func followGeneration(ctx context.Context, f *os.File, path string, offset int64, dev, ino uint64, out *envelopeWriter, curDir string, now func() time.Time, wake <-chan struct{}) error {192	defer f.Close()193194	if offset > 0 {195		if _, err := f.Seek(offset, io.SeekStart); err != nil {196			return err197		}198	}199200	// This is the line assembler. partial carries an incomplete line201	// across reads. lineStart is the file offset of that line's202	// first byte, and also the seq of the next line sent. pos is the203	// offset of the next unread byte.204	var partial []byte205	lineStart, pos := offset, offset206207	// ship sends one complete line as an envelope, using the line's208	// starting byte offset as its sequence number.209	ship := func(line string) error {210		when, severity := lift(line, now())211		return out.emit(envelope{212			Time:     when.UTC().Format(time.RFC3339Nano),213			Severity: severity,214			Seq:      uint64(lineStart),215			Message:  line,216		})217	}218219	buf := make([]byte, 32*1024)220	consume := func(data []byte) error {221		for len(data) > 0 {222			nl := bytes.IndexByte(data, '\n')223			if nl < 0 {224				partial = append(partial, data...)225				pos += int64(len(data))226				return nil227			}228			line := string(append(partial, data[:nl]...))229			partial = partial[:0]230			pos += int64(nl + 1)231			if err := ship(line); err != nil {232				return err233			}234			lineStart = pos235			data = data[nl+1:]236		}237		return nil238	}239240	var lastCheckpoint time.Time241	rotated := false242	for !rotated {243		n, err := f.Read(buf)244		if n > 0 {245			if err := consume(buf[:n]); err != nil {246				return err247			}248		}249		if err == nil {250			continue251		}252		if !errors.Is(err, io.EOF) {253			return err254		}255256		// The tailer has caught up. Checkpoint (at a limited rate), wait257		// for a wake, then check whether the path still names the file258		// this loop holds. The wake arrives on an append or a rename;259		// the backstop timer covers a lost event. Both are pure260		// triggers, so after either the tailer re-reads to EOF and261		// re-stats the path, and acts on what it finds now.262		if t := now(); t.Sub(lastCheckpoint) >= checkpointInterval {263			lastCheckpoint = t264			if err := saveCursor(curDir, tailCursor{Dev: dev, Ino: ino, Offset: lineStart}); err != nil {265				return err266			}267		}268		select {269		case <-ctx.Done():270			return ctx.Err()271		case _, ok := <-wake:272			if !ok {273				return errWatchStopped274			}275		case <-time.After(backstopInterval):276		}277		st, err := os.Stat(path)278		if err != nil {279			if !errors.Is(err, fs.ErrNotExist) {280				return err281			}282			rotated = true283		} else if d, i := fileIdentity(st); d != dev || i != ino {284			rotated = true285		}286	}287288	// The held file is the renamed previous generation. Drain289	// whatever the writer appended before the rename. Send a290	// trailing unterminated line as it is: a crash mid-write is the291	// only way such a line can exist, and it will never be292	// completed. Then move on to the new file at the path.293	for {294		n, err := f.Read(buf)295		if n > 0 {296			if err := consume(buf[:n]); err != nil {297				return err298			}299		}300		if errors.Is(err, io.EOF) {301			break302		}303		if err != nil {304			return err305		}306	}307	if len(partial) > 0 {308		return ship(string(partial))309	}310	return nil311}312313// relayFile implements the k3s and containerd verbs.314func relayFile(path string) error {315	return tailFile(context.Background(), path, newEnvelopeWriter(os.Stdout), cursorDir, time.Now)316}
machine-operator/backstop.go 100.0%
1package main23// The two timers of a settled machine: the check of the sysctls, and4// the backstop pass.5//6// The sysctls are the one state a pass writes that sends this pod no7// event when another process changes it (machineevents.go). So every8// sysctlCheckEvery the loop reads each parameter that the last pass9// applied, and runs a pass only when one reads other than the value the10// pass read back after its write. The comparison is against what the11// kernel reported, not against the value the pass wrote, because the12// kernel reports some values in another form: 0x10 reads back as 16. A13// settled machine pays a few small reads, and no pass and no request.14//15// The backstop covers a wake that the operator's own code forgot to16// send. Every event source the operator uses either delivers each17// change or reports that it lost some: a Kubernetes watch resumes from18// its last version and lists again on a 410, the uevent socket reports19// ENOBUFS and inotify IN_Q_OVERFLOW, and both readers wake a pass on20// it, and a reader that stops ends the process. So a lost change comes21// from a path in this program that changes state and sends no wake. A22// pass runs after backstopEvery with no other pass, and on a correct23// machine it writes nothing. A backstop pass that writes anything found24// that bug, and the operator reports each write: a log line, the25// backstop_repairs_total counter by step, and a BackstopRepaired26// Warning Event on the Machine. The fix for a repair is the missing27// wake, not a shorter backstop.2829import (30	"fmt"31	"math/rand/v2"32	"slices"33	"strings"34	"time"3536	"github.com/liken-sh/liken/liken/machine"37)3839const (40	sysctlCheckEvery = 10 * time.Second41	backstopEvery    = 5 * time.Minute42)4344// sysctlCheck is what the check of the sysctls reads: what the last45// pass read back for each parameter it applied, and the parameters it46// could not apply because their file did not exist.47type sysctlCheck struct {48	applied map[string]string49	missing []string50}5152// drifted answers whether a parameter that the last pass applied now53// reads other than the value the pass read back, or a parameter whose54// file did not exist now has one. A parameter under a network interface55// gets its file when the interface appears, and an interface under56// /devices/virtual sends no uevent this operator hears (hardware's57// InventoryEvent), so the check is what applies it. A parameter that58// does not read at all counts as drifted, and the pass it starts records59// the failure.60func (c sysctlCheck) drifted(dir string) bool {61	for name, want := range c.applied {62		got, err := machine.ReadSysctl(dir, name)63		if err != nil || !sameSysctlValue(got, want) {64			return true65		}66	}67	for _, name := range c.missing {68		if _, err := machine.ReadSysctl(dir, name); err == nil {69			return true70		}71	}72	return false73}7475// backstopDelay answers backstopEvery with up to a tenth more at76// random, so the machines of a fleet that started together do not run77// their backstop passes at the same moments. jitter answers a number in78// [0, 1), and nil means math/rand.79func backstopDelay(jitter func() float64) time.Duration {80	j := rand.Float6481	if jitter != nil {82		j = jitter83	}84	return backstopEvery + time.Duration(j()*float64(backstopEvery)/10)85}8687// reportRepairs reports each write of a backstop pass.88func reportRepairs(writes []string, notes machineEvents, mm *machineMetrics) {89	if len(writes) == 0 {90		return91	}92	steps := slices.Compact(slices.Sorted(slices.Values(writes)))93	fmt.Printf("the backstop pass repaired state that no wake reported: %s\n", strings.Join(steps, "; "))94	for _, step := range steps {95		mm.backstopRepaired(step)96	}97	notes.warning(reasonBackstopRepaired, fmt.Sprintf(98		"a pass that no event started changed this machine, so a wake is missing: %s", strings.Join(steps, "; ")))99}
machine-operator/cdi.go 96.6%
1package main23// Writing CDI specs: how a prepared claim becomes device nodes in a4// container.5//6// The Container Device Interface connects two things: which device to7// use, and what appears inside the container. A JSON file in a8// well-known directory describes named devices and the edits that9// grant one device to a container. Here, those edits are device10// nodes only; the CDI spec format also allows mounts and environment11// variables for drivers that need them, but liken does not use those.12// The DRA driver answers the kubelet's prepare call with CDI device13// IDs. Each ID has the form kind=name. The kubelet passes the ID14// through the CRI, and containerd resolves it against these files15// when it creates the container. No privilege is involved anywhere:16// the pod gets exactly the nodes the spec names, with the default17// cgroup device rules to match.18//19// Each claim gets one spec file, named by the claim's UID, not by its20// namespace and name. This is deliberate. When a claim is deleted and21// recreated under the same name, it is a different grant, and its22// file must not collide with a stale one. The specs live under23// /var/run, which is the machine's runtime tmpfs at /run under its24// older name (the image build explains the symlink). The kubelet25// re-prepares every claim after a reboot, so each file only needs to26// last one boot, and a tmpfs directory removes the files27// automatically at that point.28//29// A file also has to stay correct for the whole boot. The kubelet30// prepares a claim once and reuses the answer for every later pod31// that names the same claim, so nothing re-prepares a claim while one32// of its pods runs. Meanwhile the nodes a device delivers can move33// under it: a USB device that is unplugged and plugged back in34// enumerates again with a new device number, and its usbfs node moves35// with it. The reconcile pass rewrites every prepared claim's file36// from the same sysfs walk that publishes the inventory, and the37// uevent of the replug wakes that pass.3839import (40	"encoding/json"41	"errors"42	"fmt"43	"io/fs"44	"os"45	"path/filepath"46	"slices"47	"strings"48	"sync"4950	"github.com/liken-sh/liken/liken/hardware"51)5253// cdiWrites serializes the writes to these files. The kubelet's54// prepare calls and the reconcile pass both write them, and both55// stage a write through the same temporary path.56var cdiWrites sync.Mutex5758// cdiDir is the directory where containerd looks for the CDI specs59// that liken writes while the system runs. It is a variable so the60// tests can change it.61var cdiDir = "/var/run/cdi"6263// cdiSpec holds the part of the CDI spec schema that liken writes.64// liken delivers device nodes only, so the struct omits the fields65// for mounts and environment variables.66type cdiSpec struct {67	Version string      `json:"cdiVersion"`68	Kind    string      `json:"kind"`69	Devices []cdiDevice `json:"devices"`70}7172type cdiDevice struct {73	Name           string   `json:"name"`74	ContainerEdits cdiEdits `json:"containerEdits"`75}7677type cdiEdits struct {78	DeviceNodes []cdiDeviceNode `json:"deviceNodes"`79}8081// cdiDeviceNode is one node the runtime injects. A node named by82// path alone is one the runtime reads from the host: it stats the83// path to learn the node's kind and numbers. Type, Major, and Minor84// state those facts instead, and a node that states them needs no85// host node at all.86type cdiDeviceNode struct {87	Path     string `json:"path"`88	Type     string `json:"type,omitempty"`89	Major    int    `json:"major,omitempty"`90	Minor    int    `json:"minor,omitempty"`91	FileMode *int   `json:"fileMode,omitempty"`92}9394// deviceNodes turns the paths one published device delivers into the95// container edits that grant them. A path whose numbers the kernel96// fixes carries those numbers, because the node it names can be97// absent when the kubelet prepares the claim. The runtime then98// creates the node with mknod and writes the matching cgroup rule,99// and the container can open the device the moment the kernel100// registers it.101func deviceNodes(paths []string) []cdiDeviceNode {102	nodes := make([]cdiDeviceNode, 0, len(paths))103	for _, path := range paths {104		node := cdiDeviceNode{Path: path}105		if major, minor, ok := hardware.EvdevNumbers(path); ok {106			mode := evdevFileMode107			node.Type, node.Major, node.Minor, node.FileMode = "c", major, minor, &mode108		}109		nodes = append(nodes, node)110	}111	return nodes112}113114// sameDeviceNodes answers whether two lists grant the same nodes. A115// node's FileMode is a pointer, so the comparison reads the mode, not116// the pointer: a spec read back from its file and the nodes built again117// for the same devices hold equal modes behind different pointers.118func sameDeviceNodes(a, b []cdiDeviceNode) bool {119	return slices.EqualFunc(a, b, func(x, y cdiDeviceNode) bool {120		sameMode := (x.FileMode == nil) == (y.FileMode == nil) && (x.FileMode == nil || *x.FileMode == *y.FileMode)121		return x.Path == y.Path && x.Type == y.Type && x.Major == y.Major && x.Minor == y.Minor && sameMode122	})123}124125// evdevFileMode is the mode the runtime gives a node it creates with126// mknod. The runtime copies the mode of a host node it can stat, but a127// node in the evdev range can be absent when the container starts, and128// a node created with no mode is openable only by a process that holds129// CAP_DAC_OVERRIDE. The owner is the container's own user, so owner130// read and write is what the program needs.131const evdevFileMode = 0o600132133// cdiKind identifies liken's CDI devices, the same way the driver134// name identifies liken's slices. A CDI device ID has the form135// "<kind>=<name>".136const cdiKind = "liken.sh/device"137138// writeCDISpec writes one claim's devices to a file where the139// runtime can find them.140func writeCDISpec(claimUID string, devices []cdiDevice) error {141	cdiWrites.Lock()142	defer cdiWrites.Unlock()143	return writeSpecFile(claimUID, devices)144}145146// writeSpecFile is the write itself, with the lock already held. It147// is atomic. containerd may list the directory at any moment, and a148// half-written spec would fail every container creation that reads it149// at that moment.150func writeSpecFile(claimUID string, devices []cdiDevice) error {151	if err := os.MkdirAll(cdiDir, 0o755); err != nil {152		return err153	}154	spec := cdiSpec{Version: "0.6.0", Kind: cdiKind, Devices: devices}155	raw, err := json.Marshal(&spec)156	if err != nil {157		return err158	}159	path := cdiSpecPath(claimUID)160	tmp := path + ".tmp"161	if err := os.WriteFile(tmp, raw, 0o644); err != nil {162		return err163	}164	return os.Rename(tmp, path)165}166167// removeCDISpec deletes a claim's spec file. If the spec is already168// gone, this counts as success, because unprepare must be169// idempotent: the kubelet retries it whenever it is not sure the170// call succeeded.171func removeCDISpec(claimUID string) error {172	cdiWrites.Lock()173	defer cdiWrites.Unlock()174	err := os.Remove(cdiSpecPath(claimUID))175	if os.IsNotExist(err) {176		return nil177	}178	return err179}180181func cdiSpecPath(claimUID string) string {182	return filepath.Join(cdiDir, "liken.sh-"+claimUID+".json")183}184185// refreshCDISpecs rewrites each prepared claim's spec with the nodes186// its devices deliver now. It resolves each device the same way187// prepare does, from one walk of sysfs, so a spec written by a188// refresh and a spec written by a prepare always agree.189//190// This cannot repair a container that is already running. The runtime191// injects the nodes at container creation, and a node that moves192// under a running container stays wrong until the pod restarts. What193// it prevents is a stale file that every later pod would receive.194func refreshCDISpecs(sysRoot string, out *passOutcome) {195	entries, err := os.ReadDir(cdiDir)196	if err != nil {197		// No directory means no claim has been prepared on this boot.198		return199	}200	var byName map[string]hardware.Device201	for _, entry := range entries {202		claimUID, ok := claimUIDFromSpecName(entry.Name())203		if !ok {204			continue205		}206		if byName == nil {207			byName = map[string]hardware.Device{}208			for _, d := range hardware.DiscoverInventory(sysRoot, draNaming()) {209				byName[deviceName(d)] = d210			}211		}212		if err := refreshCDISpec(sysRoot, claimUID, byName, out); err != nil {213			fmt.Fprintf(os.Stderr, "device inventory: refreshing claim %s: %v\n", claimUID, err)214			out.fail("refreshing the CDI specification of claim "+claimUID, err)215		}216	}217}218219// refreshCDISpec rewrites one claim's spec, and writes nothing when220// every device still delivers what the file says.221//222// A device that this machine no longer publishes names223// deviceAbsentNode. The claim names hardware that left, and unprepare224// ends the claim when its pods do. Its old nodes would hand the next225// container whatever device the kernel gave those names since, and an226// empty edit list would start that container with no device and no227// error. When the hardware returns at the same address, the refresh228// writes its nodes back.229//230// A CEC adapter's devices name their own absent node, serioAbsentNode, so231// a container fails to start rather than open a node whose number232// another device may hold. So does a claim allocated to the adapter's233// interface before spec.serio declared it: its old nodes are the tty234// and the usbfs node, which end the attachment (serioFailsClosed).235//236// A disk that the protection withholds names withheldNode, for the237// same reason: the next container fails to start rather than open a238// disk that may back the machine's own filesystems (protection.go).239// When the facts read again and the disk backs no role, the refresh240// writes its nodes back.241//242// A device that is present with no driver is a different case: the243// program under this claim detached the kernel driver, and the244// kernel driver's nodes went with it. The publish policy resolves245// that shape to the bus node alone, so the refresh rewrites the spec246// to the one node the program uses. Without this rewrite the spec247// keeps a node the program deleted, and the claim's container can248// never restart: the runtime injects the spec's nodes at every249// container creation, and a stat on the deleted node fails it.250//251// A spec that does not decode is wrapped in fs.ErrInvalid: nothing but252// a new prepare replaces it, so the pass's outcome retries it slowly.253func refreshCDISpec(sysRoot, claimUID string, byName map[string]hardware.Device, out *passOutcome) error {254	cdiWrites.Lock()255	defer cdiWrites.Unlock()256257	raw, err := os.ReadFile(cdiSpecPath(claimUID))258	if os.IsNotExist(err) {259		// Unprepare removed the claim between the directory listing260		// and this read.261		return nil262	}263	if err != nil {264		return err265	}266	var spec cdiSpec267	if err := json.Unmarshal(raw, &spec); err != nil {268		return fmt.Errorf("decoding %s: %w: %w", cdiSpecPath(claimUID), err, fs.ErrInvalid)269	}270	changed := false271	for i, device := range spec.Devices {272		// prepare names each CDI device for the claim and the273		// allocated device together, so the allocated name is in the274		// file, and the refresh needs no call to the API server.275		allocated, ok := strings.CutPrefix(device.Name, claimUID+"-")276		if !ok {277			continue278		}279		serio := declaredSerio()280		published, err := resolveAllocated(allocated, sysRoot, byName, serio, currentProtection())281		switch {282		case errors.Is(err, errWithheld):283			published = publishedDevice{Nodes: []string{withheldNode}}284		case err != nil && serioFailsClosed(allocated, byName, serio):285			published = publishedDevice{Nodes: []string{serioAbsentNode}}286		case err != nil:287			published = publishedDevice{Nodes: []string{deviceAbsentNode}}288		}289		nodes := deviceNodes(published.Nodes)290		if sameDeviceNodes(nodes, device.ContainerEdits.DeviceNodes) {291			continue292		}293		spec.Devices[i].ContainerEdits.DeviceNodes = nodes294		changed = true295	}296	if !changed {297		return nil298	}299	if err := writeSpecFile(claimUID, spec.Devices); err != nil {300		return err301	}302	out.wrote("refreshing the CDI specification of claim " + claimUID)303	return nil304}305306// deviceAbsentNode is the node a prepared claim's spec names while its307// hardware is absent. No such node exists, so the runtime fails to308// create the next container that holds the claim, and the kubelet309// retries it under its restart backoff. A container that started310// before the hardware left keeps the nodes it received.311const deviceAbsentNode = "/dev/liken.sh/device-absent"312313// claimUIDFromSpecName reads a claim's UID back out of its spec file314// name. A name that does not fit the pattern belongs to another315// writer, or is a temporary file mid-rename, and the refresh leaves316// it alone.317func claimUIDFromSpecName(name string) (string, bool) {318	uid, ok := strings.CutPrefix(name, "liken.sh-")319	if !ok {320		return "", false321	}322	return strings.CutSuffix(uid, ".json")323}
machine-operator/cluster.go 91.6%
1package main23// The operator's half of the cluster document lifecycle: promotion4// and convergence.5//6// Init boots a staged cluster document without proof that the7// document works. Init cannot prove it, because problems with the8// document show up only after the boot. For example, a bad endpoint9// only means the machine never joins its cluster, and init sees only10// that k3s never settles. The operator provides the proof. It runs11// as a pod, so if this code executes, then containerd, the kubelet,12// and the machine's registration with its cluster all work under the13// document this boot ran. At that point the staged document is14// proven, and the operator is the right component to record this,15// because it already has the read-write machineState mount it uses16// to stage documents.17//18// The same authority records a first boot's seed as the first proven19// copy. This closes the loop for a machine that has never had a20// staged document: from that point on, the durable store carries the21// cluster document forward, not the image.22//23// Convergence applies the Machine's decision table to a different24// document. The operator reads the in-cluster Cluster resource,25// renders it in its canonical form, compares the result against what26// this boot ran, and stages the difference for the next boot. The27// one structural difference is scope. The Cluster is one document,28// but the convergence machinery runs per machine. Each machine29// stages its own copy on its own schedule, so machines can run30// different cluster documents for a time. Each Machine's31// ClusterConverged condition shows this disagreement when it32// happens. This file contains no fleet orchestration by design. A33// Cluster edit causes drift on every machine at once, and this code34// only stages the change and asks for whatever disruption the change35// needs, which for some edits is none at all. On a cluster member36// with rebootPolicy Auto, asking means waiting for a turn from the37// cluster operator's rollout conductor. The conductor grants38// reboots to one machine at a time, so a fleet-wide edit rolls39// through the fleet instead of rebooting every machine together.40// Manual, the default policy, leaves each machine's pending reboot41// visible and waiting for a person to act.4243import (44	"fmt"45	"os"4647	"sigs.k8s.io/yaml"4849	"github.com/liken-sh/liken/liken/api"50	"github.com/liken-sh/liken/liken/cluster"51	"github.com/liken-sh/liken/liken/machine"52)5354// renderCluster produces the canonical bytes to stage: a Cluster55// document with no status. The rendering is deterministic for the56// same reason renderManifest's rendering is deterministic: yaml57// marshals through JSON with sorted keys. So the hash of these bytes58// is the document's identity everywhere.59//60// The rendering excludes the release feed (spec.version and61// spec.releases) before it runs. The operator reads those fields62// live: it rereads them from the in-cluster resource on every pass,63// so they take effect without a reboot. The drift comparison here64// hashes the whole document. If the rendering included those65// fields, every catalog append and every retargeting would change66// the hash. Each change would then read as drift on every machine at67// once and stage a fleet-wide reboot, even though the only work68// needed is a download. (The Machine can carry sysctls in its69// staged manifest, because its drift check, storageDrift, checks70// individual fields rather than hashing the whole document.)71func renderCluster(name string, spec cluster.ClusterSpec) ([]byte, string, error) {72	spec.Version = ""73	spec.Releases = cluster.ClusterReleasesSpec{}74	doc := cluster.Cluster{75		APIVersion: api.APIVersion,76		Kind:       "Cluster",77		Metadata:   api.ObjectMeta{Name: name},78		Spec:       spec,79	}80	body, err := yaml.Marshal(&doc)81	if err != nil {82		return nil, "", err83	}84	return body, machine.ManifestHash(body), nil85}8687// decideClusterConvergence follows the same order of checks as88// decideConvergence, applied to the cluster document. With no facts,89// the verdict is Unknown. When the boot ran the current document,90// the machine is converged. This case also withdraws a stale staged91// copy and clears a spent rejection. When init rejected a document92// at the last boot, this function holds rather than staging it93// again. When there is nowhere durable to stage, the condition94// states this. Otherwise the code stages the drift. The disruption95// then follows the Machine's rebootPolicy, the one setting that96// governs both kinds of staging, and the machine's turn with the97// rollout conductor. A cluster document edit causes drift on every98// machine at once, which is the case the conductor is built to99// sequence.100//101// The classifier in cluster/changes.go determines the kind of102// disruption, and whether the document needs one at all. When the103// only difference is the origin or the endpoint, nothing running104// reads the change, so the document is staged and the next boot105// applies it. When every differing domain is read only when k3s106// starts its process (features, registries, the runtime107// environment), a k3s restart applies the document, and the machine108// and its pods stay up. Any other difference requires a reboot. An109// unreadable boot document also requires a reboot, because a reboot110// is the one action that always works.111//112// bootDoc and bootHash describe the document this boot ran (see113// bootClusterDocument below). The hash is always the canonical114// hash, never facts.Boot.ClusterManifestHash. The facts field115// hashes raw bytes, and a hand-written seed and the operator's116// rendering of the same spec produce different bytes with the same117// meaning. Drift is a difference in meaning. A difference in118// formatting alone must never disrupt a fleet.119//120// reduced is the document this machine stages, which is not always121// the document the deployment wrote: the retraction barrier122// (retraction.go) puts back every feature whose precondition the123// cluster does not satisfy, and held names those features. The whole124// decision below runs on the reduced document, so a held feature125// keeps running through the disruption this pass asks for. When the126// reduction produces exactly the document this boot ran, no127// disruption can advance the edit, and the verdict names the feature128// that stays and the objects that keep it.129func decideClusterConvergence(reduced *cluster.Cluster, held []featureHold, m *machine.Machine, facts *machine.MachineStatus, rejection *machine.Rejection, bootDoc *cluster.Cluster, bootHash, stagedHash string, t turn) convergence {130	if facts == nil || facts.Boot.ManifestSource == "" {131		return factsIncomplete("ClusterConverged")132	}133	if facts.Boot.ClusterManifestSource != "" && bootHash == "" {134		return convergence{condition: convergenceUnknown("ClusterConverged", "FactsIncomplete",135			"the boot ran a cluster document but its publication is unreadable")}136	}137138	manifest, hash, err := renderCluster(reduced.Metadata.Name, reduced.Spec)139	if err != nil {140		return convergence{condition: notConverged("ClusterConverged", "StagingFailed", err.Error())}141	}142143	if hash == bootHash {144		// This boot already runs the reduced document, so no staging145		// and no disruption moves the machine closer to the146		// deployment's document. Only a change to the cluster's147		// objects satisfies the precondition, and until that happens148		// every pass reaches this same verdict. A stale staged copy149		// still goes, because those bytes are not the document the150		// next boot must run.151		if len(held) > 0 {152			return convergence{153				condition: notConverged("ClusterConverged", "RetractionBlocked", holdMessage(held)),154				withdraw:  stagedHash != "",155			}156		}157		return convergedWithCleanup(158			converged("ClusterConverged", "Converged", "this boot ran the current cluster document"),159			stagedHash, rejection)160	}161162	// The rejection comes from the durable record, not from facts.163	// Facts are a snapshot taken at boot, and they do not change164	// while the machine runs. But when a revert clears a rejection,165	// a retry must work again within the same boot.166	if rejection != nil && rejection.Hash == hash {167		return convergence{condition: notConverged("ClusterConverged", "RejectedLastBoot",168			fmt.Sprintf("init rejected this exact cluster document at boot: %s; edit the cluster to something different", rejection.Reason))}169	}170	if facts.Storage.MachineState.Backing != machine.BackingPartition {171		return machineStateEphemeral("ClusterConverged", "the cluster document")172	}173174	// The next-boot tier comes before the restart and reboot gate,175	// because it is the lightest answer the classifier can give. Only176	// a boot reads the fields this edit touches, so staging the177	// document is the whole of the work, and the machine adopts it at178	// its next boot. No turn is requested, so a fleet-wide edit of179	// these fields never enters the rollout conductor's queue.180	// cluster/changes.go carries the reasoning, down to the one value181	// a running follower still holds: the endpoint's host, in its time182	// sources. The message states what happens, and leaves that183	// argument to the classifier and to the CRD's field descriptions.184	if bootDoc != nil && cluster.NextBootApplies(bootDoc.Spec, reduced.Spec) {185		return convergence{186			manifest: manifest,187			hash:     hash,188			stage:    stagedHash != hash,189			condition: notConverged("ClusterConverged", "StagedForNextBoot",190				fmt.Sprintf("cluster document staged (%.12s); this machine applies it at its next boot, and asks for no turn", hash)),191		}192	}193194	restart := bootDoc != nil && cluster.RestartApplies(bootDoc.Spec, reduced.Spec)195196	c := convergence{197		manifest: manifest,198		hash:     hash,199		stage:    stagedHash != hash,200	}201	// On Manual, a person applies the change with a reboot or with202	// the approve-disruption grant. The boot path applies staged203	// documents too, and nobody can restart k3s by hand on a machine204	// with no shell, but an approved restart-class change converges205	// through a granted k3s restart in place. The messages used206	// after a grant name what init will actually do.207	pending := fmt.Sprintf("cluster document staged (%.12s); rebootPolicy is Manual, so reboot the machine to apply (or set rebootPolicy: Auto)", hash)208	if restart {209		pending = fmt.Sprintf("cluster document staged (%.12s); rebootPolicy is Manual, so reboot the machine to apply (or set rebootPolicy: Auto, which would apply it with just a k3s restart)", hash)210	}211	apply := "a reboot"212	if restart {213		apply = "a k3s restart"214	}215	gateDisruption(&c, "ClusterConverged", m.Spec.RebootPolicyOrDefault(), t, restart,216		m.Metadata.Annotations[machine.ApproveDisruptionAnnotation],217		"the cluster document",218		pending,219		fmt.Sprintf("cluster document staged (%.12s); waiting for the cluster to grant a turn to apply it by %s", hash, apply),220		fmt.Sprintf("%s requested to apply the staged cluster document (%.12s)", apply, hash))221	return c222}223224// convergeClusterDocument runs the cluster document's part of one225// reconcile pass. It reads the live Cluster resource, loads this226// machine's durable rejection and staged copy from the store, and227// makes the convergence decision. The decision is where the fleet's228// temporary disagreement about the Cluster becomes visible: each229// machine stages its own copy and reboots on its own policy, and230// this condition reports where this one machine stands. The231// function returns the live Cluster alongside the decision, because232// version convergence (release.go) reads its release feed live. A233// nil Cluster means the read failed, and the verdict already234// reports that.235//236// This is also where the retraction barrier reaches the live cluster.237// A precondition is a statement about objects that no document238// describes, such as the HelmCharts that still exist, so the239// reduction needs the API client that the reader holds. It reads240// nothing unless the edit stops a feature, so an ordinary pass reads241// only the Cluster's copy. The HelmCharts and the LoadBalancer242// Services are watched only while a retraction waits on them243// (waits.go), and the copy of the Services keeps each one's name and244// type alone.245func convergeClusterDocument(r *reader, store machine.ManifestStore, clusterName string, m *machine.Machine, facts *machine.MachineStatus, t turn) (convergence, *cluster.Cluster) {246	liveCluster, err := r.cluster(clusterName)247	if err != nil {248		return convergence{condition: convergenceUnknown("ClusterConverged", "ClusterUnavailable",249			fmt.Sprintf("reading cluster %s: %v", clusterName, err))}, nil250	}251	rejection, _ := store.LoadRejection()252	bootDoc, bootHash := bootClusterDocument(cluster.BootClusterManifestPath)253	reduced, held := reduceRetraction(bootDoc, liveCluster, func(p cluster.Precondition) (bool, string, error) {254		return evaluatePrecondition(r, p)255	})256	return decideClusterConvergence(reduced, held, m, facts, rejection,257		bootDoc, bootHash, readStagedHash(store), t), liveCluster258}259260// bootClusterDocument describes the document this boot ran. It261// returns the parsed document, so the classifier can compare it262// against the desired spec, and the document's canonical hash.263// Canonical means the function parses the bytes init published and264// re-renders them the way the operator renders every document. Both265// sides of the drift comparison then pass through the same266// rendering, so any remaining difference is a difference in267// content, not formatting. A nil document and an empty hash mean the268// published document is missing or unreadable.269func bootClusterDocument(path string) (*cluster.Cluster, string) {270	raw, err := os.ReadFile(path)271	if err != nil {272		return nil, ""273	}274	c, err := cluster.ParseCluster(raw)275	if err != nil {276		return nil, ""277	}278	_, hash, err := renderCluster(c.Metadata.Name, c.Spec)279	if err != nil {280		return nil, ""281	}282	return c, hash283}284285// settleClusterLifecycle promotes whatever this boot proved. It runs286// on every reconcile pass and is idempotent. Once the operator287// promotes a document, or once a newer document is staged for its288// own proving boot, there is nothing left to do. The facts identify289// exactly which bytes this boot ran, and the operator promotes only290// those bytes.291func settleClusterLifecycle(root, seedPath string, facts *machine.MachineStatus, out *passOutcome) {292	if facts == nil || facts.Storage.MachineState.Backing != machine.BackingPartition {293		return // nothing durable to settle294	}295	store := machine.ClusterManifests(root)296297	switch facts.Boot.ClusterManifestSource {298	case machine.ManifestSourceStaged:299		raw, err := store.LoadStaged()300		if err != nil || raw == nil {301			return // already promoted, or nothing staged302		}303		if machine.ManifestHash(raw) != facts.Boot.ClusterManifestHash {304			// A newer document arrived after this boot started. It305			// has not had its own proving boot yet, and promoting it306			// now would skip that trial.307			return308		}309		if err := store.Promote(); err != nil {310			fmt.Fprintf(os.Stderr, "promoting the cluster document: %v\n", err)311			out.fail("promoting the cluster document", err)312			return313		}314		out.wrote("promoting the cluster document")315		fmt.Printf("the cluster document proved out; %.12s is now proven\n", facts.Boot.ClusterManifestHash)316317	case machine.ManifestSourceSeed:318		if proven, err := store.LoadProven(); proven != nil || err != nil {319			return320		}321		raw, err := os.ReadFile(seedPath)322		if err != nil {323			return324		}325		if machine.ManifestHash(raw) != facts.Boot.ClusterManifestHash {326			// The seed file changed after this machine booted.327			// Recording it now would mark as proven bytes that328			// nobody ran.329			return330		}331		if err := store.WriteProven(raw); err != nil {332			fmt.Fprintf(os.Stderr, "recording the seed cluster document as proven: %v\n", err)333			out.fail("recording the seed cluster document as proven", err)334			return335		}336		out.wrote("recording the seed cluster document as proven")337		fmt.Printf("the seed cluster document is now proven (%.12s)\n", facts.Boot.ClusterManifestHash)338	}339}
machine-operator/conditions.go 100.0%
1package main23// The condition constructors that reconcile publishes on each pass.4// Each one checks one aspect of the machine: the facts, the sysctls,5// the storage, the modules, the features, or the Node's health. Each6// one reports its check as a standard Kubernetes condition.78import (9	"errors"10	"fmt"11	"io/fs"12	"maps"13	"slices"14	"strings"1516	"github.com/liken-sh/liken/liken/api"17	"github.com/liken-sh/liken/liken/machine"18)1920func factsCondition(err error) api.Condition {21	if err != nil {22		return api.Condition{23			Type: "FactsPublished", Status: api.ConditionFalse,24			Reason:  "FactsUnreadable",25			Message: err.Error() + "; the device inventory offers no disk until the facts read",26		}27	}28	return api.Condition{Type: "FactsPublished", Status: api.ConditionTrue, Reason: "FactsRead"}29}3031// sysctlsCondition reports both halves of the sysctl pass, and only32// the spec's half can make the condition False.33//34// The reason is who wrote the failing parameter. A value from35// spec.sysctls belongs to this machine: a person asked for it here,36// nowhere else, and a machine that cannot honour its own spec is37// degraded. A value from machine.OSSysctls ships with the release, so38// every machine running that release applies the same table. A single39// bad entry there would take an entire fleet to Degraded in the same40// pass, which is the moment a per-machine health signal stops carrying41// any information and starts hiding the one machine with a real42// problem. So a failing default reports DefaultsIncomplete and leaves43// the machine Ready.44//45// A failing default is still visible twice. It names itself in this46// message, and its parameter is missing from status.sysctls, because47// applySysctls never reads back a value it could not write. That48// absence is what makes status.sysctls a list of the parameters that49// currently hold rather than the parameters somebody wanted.50func sysctlsCondition(defaultsErr, specErr error) api.Condition {51	if specErr != nil {52		message := specErr.Error()53		if defaultsErr != nil {54			message += "; " + defaultsErr.Error()55		}56		return api.Condition{57			Type: "SysctlsApplied", Status: api.ConditionFalse,58			Reason: "ApplyFailed", Message: message,59		}60	}61	if defaultsErr != nil {62		return api.Condition{63			Type: "SysctlsApplied", Status: api.ConditionTrue,64			Reason: "DefaultsIncomplete", Message: defaultsErr.Error(),65		}66	}67	return api.Condition{Type: "SysctlsApplied", Status: api.ConditionTrue, Reason: "Applied"}68}6970// hostEntriesCondition reports the outcome of applyHostEntries, on71// the same terms as storageCondition and modulesCondition above:72// True and Applied on an ordinary pass, True and NothingDeclared when73// the spec declares no host entry at all, False and ApplyFailed when74// a read, a render, or a write failed, and False and75// AwaitingPodRefresh when that same failure is the pod-freshness76// guard's concern instead (awaitingPodRefresh, below). podStale is77// this pass's verdict from staleness.go.78func hostEntriesCondition(desired []machine.HostEntry, err error, podStale bool) api.Condition {79	if err != nil {80		if awaitingPodRefresh(podStale, err) {81			return api.Condition{82				Type: "HostEntriesApplied", Status: api.ConditionFalse,83				Reason: "AwaitingPodRefresh",84				Message: "the pod's template predates the release this machine runs; " +85					"the pod steward replaces the pod after a leader boots that release: " + err.Error(),86			}87		}88		return api.Condition{89			Type: "HostEntriesApplied", Status: api.ConditionFalse,90			Reason: "ApplyFailed", Message: err.Error(),91		}92	}93	if len(desired) == 0 {94		return api.Condition{95			Type: "HostEntriesApplied", Status: api.ConditionTrue,96			Reason: "NothingDeclared", Message: "no host entries declared",97		}98	}99	return api.Condition{Type: "HostEntriesApplied", Status: api.ConditionTrue, Reason: "Applied"}100}101102// awaitingPodRefresh judges whether an actuation failure is the103// template lag itself, rather than a fault the machine actually has.104// System pods run the stable :installed tag and their DaemonSets105// update on OnDelete (cluster-operator/steward.go), so a reboot106// restarts a machine's own operator into a new binary without107// touching the pod spec around it. Only a leader's boot rewrites the108// AddOn manifests that produce a fresh template, so a follower that109// reboots first runs the new binary inside the old pod spec for a110// while. A path that does not exist inside that stale pod means a111// mount the old template lacks, whatever the mount is, so this rule112// covers every mount a future release may add without naming any of113// them by name.114//115// Two precedents already treat a release-wide condition as something116// other than one machine's own fault. The DRA plugin tolerates a117// mount its own stale pod lacks (main.go), because dying there would118// kill the very status publishing the pod steward waits on.119// sysctlsCondition reports a bad default as DefaultsIncomplete rather120// than ApplyFailed, because a fault every machine on the release121// carries at once tells a person nothing about which machine needs122// attention. This rule follows the same reasoning for a stale pod's123// missing mount.124//125// The reason this rule reports must not be AwaitingTurn. The rollout126// conductor scans a machine's conditions for that exact reason127// (cluster-operator/rollout.go, wantsTurn) to learn that the machine128// has a staged change ready for a disruption. AwaitingPodRefresh129// names a wait on the pod steward instead, so the conductor never130// reads this guard as a change the machine is asking permission to131// make.132func awaitingPodRefresh(podStale bool, err error) bool {133	return podStale && errors.Is(err, fs.ErrNotExist)134}135136// applySysctls writes both sets of kernel parameters to the host's137// /proc/sys (dir): the settings every liken machine holds, and then138// the Machine spec's own. The pod runs privileged in the host's139// namespaces, so it reaches /proc/sys directly.140//141// spec.sysctls is an override: a name in both sets is applied with the142// spec's value alone, and the two spellings of one name, dots and143// slashes, count as one. init applies the two sets in order at boot,144// default first, and the operator skips the default instead. Each pass145// compares and writes, so applying both in order would write the146// default and then the spec's value on every pass of a converged147// machine, and the kernel would hold the default for a moment each148// time.149//150// After both sets, every parameter is read once more, and that read is151// what the function answers and what mem keeps. Two names can write152// one kernel variable: net.ipv4.ip_forward and153// net.ipv4.conf.all.forwarding are the same switch, and a write of154// vm.dirty_bytes zeroes vm.dirty_ratio. A read taken right after each155// write would hold a value that a later write in the same pass changed,156// and the check of the sysctls (backstop.go) would find it drifted on157// every check.158//159// One failure never stops the function from applying the rest of the160// parameters. The two errors stay apart because the condition treats161// them differently, and each joins every failure in its own set,162// because a message that names one bad parameter, when three are163// failing, would send a person through this loop three times. missing164// names each parameter whose file does not exist, for the check.165func applySysctls(dir string, defaults, desired map[string]string, out *passOutcome, mem *sysctlMemory) (map[string]string, []string, error, error) {166	observed, defaultsMissing, defaultsErr := applySysctlSet(dir, withoutKeys(defaults, desired), out, mem)167	fromSpec, specMissing, specErr := applySysctlSet(dir, desired, out, mem)168	maps.Copy(observed, fromSpec)169	for name := range observed {170		if value, err := machine.ReadSysctl(dir, name); err == nil {171			observed[name] = value172		}173	}174	mem.readBack(dir, observed)175	return observed, append(defaultsMissing, specMissing...), defaultsErr, specErr176}177178// withoutKeys answers the entries of m whose names are not in drop. A179// name matches in either spelling, net.ipv4.ip_forward or180// net/ipv4/ip_forward, because both name one file.181func withoutKeys(m, drop map[string]string) map[string]string {182	dropped := map[string]bool{}183	for name := range drop {184		dropped[sysctlFile(name)] = true185	}186	kept := maps.Clone(m)187	maps.DeleteFunc(kept, func(name, _ string) bool { return dropped[sysctlFile(name)] })188	return kept189}190191// sysctlFile answers the path of a parameter under /proc/sys, with the192// rule machine.ApplySysctl uses: a name with a slash is a path already,193// and a name without one has dots for slashes.194func sysctlFile(name string) string {195	if strings.Contains(name, "/") {196		return name197	}198	return strings.ReplaceAll(name, ".", "/")199}200201// applySysctl writes one parameter. It is a variable so a test can play202// a kernel that stores a value in another form, or that changes a second203// parameter with the first.204var applySysctl = machine.ApplySysctl205206// sysctlMemory keeps what the operator last wrote to each parameter, and207// what the kernel reported after. The kernel stores some values in208// another form than the one written: 0x10 reads back as 16, a write of209// one value to kernel.printk reads back as four, and vm.nr_hugepages210// reads back as many pages as the kernel could allocate. Compared with211// the spec's value, such a parameter differs on every pass, and every212// pass writes it again, which a backstop pass reports as a repair. So a213// parameter that still reads what the kernel reported after the214// operator's last write of the same value is current. A write-only215// parameter, such as vm.drop_caches, refuses every read, so it is216// written once for each value the spec gives it. Only the loop's217// goroutine applies sysctls, so the memory has no lock. A nil memory218// remembers nothing.219type sysctlMemory struct {220	written map[string]sysctlWrite221}222223type sysctlWrite struct {224	value    string225	readBack string226	// writeOnly is true for a parameter whose read the kernel refused227	// after the write.228	writeOnly bool229}230231func newSysctlMemory() *sysctlMemory {232	return &sysctlMemory{written: map[string]sysctlWrite{}}233}234235// holds answers whether the parameter is as the operator's last write236// of value left it.237func (m *sysctlMemory) holds(name, value, kernel string, readErr error) bool {238	if m == nil {239		return false240	}241	w, ok := m.written[name]242	if !ok || w.value != value {243		return false244	}245	if readErr != nil {246		return w.writeOnly && errors.Is(readErr, fs.ErrPermission)247	}248	return !w.writeOnly && sameSysctlValue(kernel, w.readBack)249}250251// wrote records a write of value. readBack completes the record.252func (m *sysctlMemory) wrote(name, value string) {253	if m != nil {254		m.written[name] = sysctlWrite{value: value, writeOnly: true}255	}256}257258// readBack records what the kernel reports for each parameter written259// this pass. A parameter the kernel refused to read stays write-only.260func (m *sysctlMemory) readBack(dir string, observed map[string]string) {261	if m == nil {262		return263	}264	for name, w := range m.written {265		if value, ok := observed[name]; ok {266			w.readBack, w.writeOnly = value, false267			m.written[name] = w268		}269	}270}271272// applySysctlSet reconciles one set of parameters against the kernel,273// under the same write-on-divergence rule as applyHostEntries274// (hosts.go): read a parameter first, and write it only when the275// kernel's reported value differs from the desired one, and from what276// the kernel reported after the last write of it (sysctlMemory). A277// converged parameter costs one read and no write, which is the common278// case on every pass after the first.279//280// The comparison ignores how the values are spaced. A parameter that281// holds several values, such as net.ipv4.ip_local_port_range, is282// written as "1024 65535", and the kernel reports it as "1024\t65535".283// Without this, every such parameter differs from its spec, and every284// pass writes it again.285//286// The returned map holds what the kernel now reports, not what the287// function wrote. If another process resets a value, the next pass288// finds the divergence and writes it again. The pass's outcome records289// each write, and each failure for the retry, by the parameter's name.290func applySysctlSet(dir string, desired map[string]string, out *passOutcome, mem *sysctlMemory) (map[string]string, []string, error) {291	var errs []error292	var missing []string293	observed := map[string]string{}294	for _, name := range slices.Sorted(maps.Keys(desired)) {295		value := desired[name]296		current, readErr := machine.ReadSysctl(dir, name)297		if mem.holds(name, value, current, readErr) || readErr == nil && sameSysctlValue(current, value) {298			if readErr == nil {299				observed[name] = current300			}301			continue302		}303		if err := applySysctl(dir, name, value); err != nil {304			errs = append(errs, err)305			out.fail("writing the sysctl "+name, err)306			if errors.Is(err, fs.ErrNotExist) {307				missing = append(missing, name)308			}309			continue310		}311		out.wrote("writing the sysctl " + name)312		mem.wrote(name, value)313		if value, err := machine.ReadSysctl(dir, name); err == nil {314			observed[name] = value315		}316	}317	return observed, missing, errors.Join(errs...)318}319320// sameSysctlValue reports whether two spellings of a parameter's value321// hold the same values in the same order. The kernel separates the322// values of a parameter with tabs, and a spec separates them with323// spaces.324func sameSysctlValue(a, b string) bool {325	return slices.Equal(strings.Fields(a), strings.Fields(b))326}327328// storageCondition summarizes storage as one standard Kubernetes329// condition. It compares what the spec declared against where the330// system actually backs each role. True means every declared role331// sits on its partition. False should not happen on a running332// machine, because init powers off instead of booting with a333// declared role left unsatisfied. But a condition must be able to334// report every state it names, and a future, softer failure mode may335// need this one.336func storageCondition(spec machine.StorageSpec, status machine.StorageStatus) api.Condition {337	var placed, inMemory []string338	for _, role := range spec.Roles() {339		rs := status.Role(role.Name)340		if rs != nil && rs.Backing == machine.BackingPartition {341			placed = append(placed, fmt.Sprintf("%s on %s", role.Name, rs.Device))342		} else {343			inMemory = append(inMemory, string(role.Name))344		}345	}346	switch {347	case len(inMemory) > 0:348		return api.Condition{349			Type: "StorageReady", Status: api.ConditionFalse, Reason: "RolesInMemory",350			Message: fmt.Sprintf("declared roles backed by memory: %s", strings.Join(inMemory, ", ")),351		}352	case len(placed) > 0:353		return api.Condition{354			Type: "StorageReady", Status: api.ConditionTrue, Reason: "AllRolesPlaced",355			Message: strings.Join(placed, ", "),356		}357	default:358		return api.Condition{359			Type: "StorageReady", Status: api.ConditionTrue, Reason: "NothingDeclared",360			Message: "no storage declared; all roles backed by memory",361		}362	}363}364365// outcomesCondition reduces a boot's outcomes for individual items366// (modules, features) to one condition. Any problem makes the367// condition False and carries every item's message. When every item368// is healthy, the condition is True with a summary. When nothing is369// declared, the condition is also True, with its own message.370func outcomesCondition(condType string, observed int, problems []string, failedReason, healthyReason, healthyMessage, noneMessage string) api.Condition {371	switch {372	case len(problems) > 0:373		return api.Condition{374			Type: condType, Status: api.ConditionFalse, Reason: failedReason,375			Message: strings.Join(problems, "; "),376		}377	case observed > 0:378		return api.Condition{379			Type: condType, Status: api.ConditionTrue, Reason: healthyReason,380			Message: healthyMessage,381		}382	default:383		return api.Condition{384			Type: condType, Status: api.ConditionTrue, Reason: "NothingDeclared",385			Message: noneMessage,386		}387	}388}389390// modulesCondition summarizes the boot's outcomes for declared391// modules as one condition. Loaded and Builtin are both healthy392// states. Any other state carries init's message, which names the393// fix: a rebuilt image for a Missing module, or the hardware's error394// for a Failed one. A status that names the fix is more useful than395// one that only names the problem.396func modulesCondition(observed []machine.ModuleStatus) api.Condition {397	var problems []string398	for _, s := range observed {399		if s.State == machine.ModuleLoaded || s.State == machine.ModuleBuiltin {400			continue401		}402		problems = append(problems, fmt.Sprintf("%s: %s", s.Name, s.Message))403	}404	return outcomesCondition("ModulesLoaded", len(observed), problems,405		"ModulesNotLoaded", "AllLoaded",406		fmt.Sprintf("all %d declared modules are in the kernel", len(observed)),407		"no extra modules declared")408}409410// moduleParametersCondition reports the two cases where a declared411// parameter structurally cannot have reached the kernel: the module412// is built in, or it was already resident when the declared pass got413// to it. Both are facts about the load, not about values. The414// declared string is never compared against the /sys readback,415// because the kernel renders a bool as Y or N and an array with its416// own separators, so a machine comparison would report false drift417// on the most common parameter types; a person compares the two418// status fields that sit beside each other. Each problem message419// names its own fix, the way every other outcome message does.420func moduleParametersCondition(declared map[string]string, observed []machine.ModuleStatus) api.Condition {421	byName := map[string]machine.ModuleStatus{}422	for _, s := range observed {423		byName[s.Name] = s424	}425	var problems []string426	// Only a module the boot observed can say whether its load427	// carried the parameters. A module declared since the last boot428	// has no load to report on, so it counts toward nothing here;429	// the convergence machinery already carries it to the reboot.430	loaded := 0431	for _, name := range machine.ModuleParameterModules(declared) {432		s, seen := byName[name]433		if !seen {434			continue435		}436		// Only a load that succeeded can have carried the string, so437		// only Loaded modules count toward the healthy message. A438		// Failed or Missing module is ModulesLoaded's problem, and439		// claiming its parameters "reached the kernel" would be440		// false.441		switch {442		case s.State == machine.ModuleBuiltin:443			problems = append(problems, fmt.Sprintf(444				"%s: the kernel builds %s in, so no load carried %s; set it on the kernel command line",445				name, name, machine.ModuleParameterString(name, declared)))446		case s.AlreadyResident:447			problems = append(problems, fmt.Sprintf(448				"%s: %s was already in the kernel when the declared modules loaded, so no load carried %s; "+449					"it comes from the image's fixed list, a cluster feature, or an earlier declared module's dependencies",450				name, name, machine.ModuleParameterString(name, declared)))451		case s.State == machine.ModuleLoaded:452			loaded++453		}454	}455	// Parameters were declared even when no load succeeded, so the456	// message must not claim nothing was declared; it says no load457	// carried one, which is the fact.458	none := "no module parameters declared"459	if len(declared) != 0 {460		none = "no load this boot carried a declared parameter"461	}462	return outcomesCondition("ModuleParametersApplied", loaded, problems,463		"ParametersNotApplied", "Applied",464		fmt.Sprintf("every parameter declared for %d modules reached the kernel at the load", loaded),465		none)466}467468// serioAttachedCondition is the type of the condition serioCondition469// builds. The phase and the Ready roll-up skip it by this name470// (phase.go).471const serioAttachedCondition = "SerioAttached"472473// serioCondition summarizes status.serio. It is True when every entry474// is Attached. Otherwise its reason is the first unattached entry's475// state, Missing or Refused, and its message names that entry and476// carries init's message, which gives the kernel's error text or the477// module to declare. Unlike the modules, the attachments change while478// the machine runs, because an adapter can be unplugged, so this479// condition moves between passes without a boot.480func serioCondition(observed []machine.SerioStatus) api.Condition {481	for _, s := range observed {482		if s.State == machine.SerioAttached {483			continue484		}485		reason := string(machine.SerioRefused)486		if s.State == machine.SerioMissing {487			reason = string(machine.SerioMissing)488		}489		entry := s.Attachment().String()490		if s.TTY != "" {491			entry += " on " + s.TTY492		}493		return api.Condition{494			Type: serioAttachedCondition, Status: api.ConditionFalse,495			Reason: reason, Message: entry + ": " + s.Message,496		}497	}498	return outcomesCondition(serioAttachedCondition, len(observed), nil, "", "AllAttached",499		fmt.Sprintf("all %d serio attachments hold", len(observed)),500		"no serio entries declared")501}502503// featuresCondition summarizes the boot's feature outcomes as one504// condition, in the same form as modulesCondition. Any state other505// than Active carries init's message, which names the fix. For a506// Missing feature, the fix is a release whose image carries the507// needed payload, because enabling a feature never rebuilds anything508// by itself.509func featuresCondition(observed []machine.FeatureStatus) api.Condition {510	var problems []string511	for _, s := range observed {512		if s.State == machine.FeatureActive {513			continue514		}515		problems = append(problems, fmt.Sprintf("%s: %s", s.Name, s.Message))516	}517	return outcomesCondition("FeaturesReady", len(observed), problems,518		"FeaturesNotReady", "AllActive",519		fmt.Sprintf("all %d enabled features are active on this machine", len(observed)),520		"the cluster enables no features")521}522523// wirelessCondition summarizes every radio the boot was asked to524// join as one condition, in the same form as modulesCondition and525// featuresCondition. Only Connected is a healthy state. A machine526// with no wireless entry declares nothing and stays Ready. The527// message carries the supplicant's own reason, the one fact that528// tells a wrong passphrase apart from an access point that is529// switched off.530//531// A radio still associating is work in progress, not a failure: the532// boot handed it to the background on purpose, and the verdict533// arrives in seconds. The Joining reason marks that window so the534// phase mapping can leave the machine Ready while it lasts.535func wirelessCondition(interfaces []machine.InterfaceStatus) api.Condition {536	declared, joining := 0, 0537	var problems []string538	for _, iface := range interfaces {539		w := iface.Wireless540		if w == nil {541			continue542		}543		declared++544		if w.State == machine.WirelessConnected {545			continue546		}547		if w.State == machine.WirelessAssociating {548			joining++549		}550		problems = append(problems, fmt.Sprintf("%s (%s): %s", iface.Name, w.SSID, wirelessReason(*w)))551	}552	// One settled failure makes the reason NotJoined whatever the553	// other radios are doing, because a wrong key or a stuck raise554	// must not hide behind a neighbor that is merely slow.555	reason := "NotJoined"556	if joining == len(problems) {557		reason = "Joining"558	}559	return outcomesCondition("WirelessJoined", declared, problems,560		reason, "AllJoined",561		fmt.Sprintf("all %d declared wireless networks are joined", declared),562		"no wireless network declared")563}564565// wirelessReason names why one radio is not joined. Init writes a566// message for every failure it has words for. The state is the567// fallback, for a radio that is still associating and has said568// nothing yet.569func wirelessReason(w machine.WirelessStatus) string {570	if w.Message != "" {571		return w.Message572	}573	return string(w.State)574}575576// nodeHealthyCondition translates the Node's Ready condition into the577// Machine's own condition. When the Node carries no Ready condition,578// this function reports the machine as unhealthy: a kubelet that has579// never reported in cannot be assumed to be serving.580func nodeHealthyCondition(node *nodeObject) api.Condition {581	for _, c := range node.Status.Conditions {582		if c.Type != "Ready" {583			continue584		}585		if c.Status == api.ConditionTrue {586			return api.Condition{Type: "NodeHealthy", Status: api.ConditionTrue, Reason: "KubeletReady",587				Message: "the Node reports Ready; the kubelet is serving this machine to the cluster"}588		}589		return api.Condition{Type: "NodeHealthy", Status: api.ConditionFalse, Reason: "NodeNotReady",590			Message: fmt.Sprintf("the Node reports Ready=%s: %s", c.Status, c.Message)}591	}592	return api.Condition{Type: "NodeHealthy", Status: api.ConditionFalse, Reason: "NodeNotReady",593		Message: "the Node carries no Ready condition; the kubelet has never reported in"}594}
machine-operator/converge.go 94.5%
1package main23// Convergence: keeping the cluster's spec and the machine's boot in4// agreement.5//6// Sysctls reconcile live. Storage cannot, because the system cannot7// swap a filesystem under a running cluster. The network cannot8// either, because the cluster reaches this machine over the very9// addresses an edit changes: re-addressing a running machine would10// cut the connection that carries the next instruction, on the one11// kind of machine that has no shell to repair it from. The declared12// module list cannot reconcile live either, because loading a module13// is one-way: the kernel offers no safe way to remove a driver while14// something is using it. Storage, network, and modules therefore15// converge through a reboot. The operator stages the desired16// manifest onto the machineState filesystem, where the next boot17// finds it, tries it, and promotes or rejects it (machine/staging.go18// covers that side).19// This file covers the operator's half of that work: notice drift,20// refuse what the machine cannot satisfy, stage what it can, and21// either request the reboot (rebootPolicy: Auto) or report that a22// reboot is pending (Manual, the default policy).23//24// Every decision in this file is a pure function over the cluster's25// Machine and the boot's facts. reconcile() supplies the few lines of26// I/O. init's storage code uses the same split between decisions and27// actions, and this split makes the whole feature testable with28// tables, without a cluster or a disk.2930import (31	"fmt"32	"slices"33	"strings"3435	"sigs.k8s.io/yaml"3637	"github.com/liken-sh/liken/liken/api"38	"github.com/liken-sh/liken/liken/machine"39)4041// validateStaging checks everything that admission cannot check,42// because these checks need the actual machine. CEL rules in the CRD43// compare the spec against the last boot's published status, but44// only the facts record what partitions exist and what disks are45// attached.46func validateStaging(spec machine.StorageSpec, facts *machine.MachineStatus) error {47	if err := spec.Validate(); err != nil {48		return err49	}50	for _, role := range spec.Roles() {51		placed := facts.Storage.Role(role.Name)52		if placed != nil && placed.Backing == machine.BackingPartition {53			// The role has a partition, so its declared size may only54			// grow. This also catches the case of a remainder role55			// given a fixed size smaller than the space it already56			// occupies.57			if role.Size == "" {58				continue59			}60			declared, err := machine.ParseSize(role.Size)61			if err != nil {62				return err63			}64			if declared < placed.CapacityBytes {65				// The message carries the remedy, because the person who66				// reads it is usually meeting this for the first time,67				// after a reinstall laid the disk out differently from68				// the Machine document that outlived it. The document is69				// authoritative, so the fix is always to edit the70				// document, and the size it must name is a size they can71				// paste.72				return fmt.Errorf("%s: the spec declares %s, and this machine's partition holds %s; storage roles are grow-only, so declare %s or more. A machine reinstalled with a different layout needs its Machine document edited to the layout it now carries",73					role.Name, role.Size, machine.SizeText(placed.CapacityBytes), machine.SizeText(placed.CapacityBytes))74			}75			continue76		}77		// A new role must name a disk this machine actually has. The78		// device path only matters at claim time, which is the boot79		// that this staging prepares for.80		if !deviceAttached(role.Device, facts.Hardware.BlockDevices) {81			return fmt.Errorf("%s: device %s is not among this machine's block devices (%s)",82				role.Name, role.Device, deviceNames(facts.Hardware.BlockDevices))83		}84	}85	return nil86}8788func deviceAttached(device string, disks []machine.BlockDevice) bool {89	for _, d := range disks {90		if "/dev/"+d.Name == device {91			return true92		}93	}94	return false95}9697func deviceNames(disks []machine.BlockDevice) string {98	var names []string99	for _, d := range disks {100		names = append(names, d.Name)101	}102	if len(names) == 0 {103		return "none attached"104	}105	return strings.Join(names, ", ")106}107108// renderManifest produces the canonical bytes to stage: a complete109// Machine document with no status. The document carries the whole110// spec, including the sysctls that need no reboot at all, so the111// reboot converges everything at once. The rendering is112// deterministic: sigs.k8s.io/yaml marshals through JSON with sorted113// keys, so the same spec always produces the same bytes. The hash of114// those bytes is the spec's identity everywhere: in staging115// idempotence, in rejections, and in the facts.116func renderManifest(name string, spec machine.MachineSpec) ([]byte, string, error) {117	doc := machine.Machine{118		APIVersion: api.APIVersion,119		Kind:       "Machine",120		Metadata:   api.ObjectMeta{Name: name},121		Spec:       spec,122	}123	body, err := yaml.Marshal(&doc)124	if err != nil {125		return nil, "", err126	}127	return body, machine.ManifestHash(body), nil128}129130// A turn is the machine's standing with the rollout conductor131// (rollout.go): whether the machine may reboot right now. A machine132// with no cluster document has no conductor, so it reboots whenever133// it needs to. A cluster member waits until the conductor writes a134// RebootApproved condition onto it.135// rebootPolicy: Auto checks this, and so does a Manual machine that136// a person has approved through the approve-disruption annotation:137// the approval moves the machine onto the same turn-taking path.138type turn int139140const (141	turnStandalone turn = iota // no cluster: reboots whenever it needs to142	turnAwaiting               // cluster member, waiting for a grant143	turnGranted                // cluster member, already granted144)145146// A convergence is one reconcile pass's decision: the condition to147// publish and which side effects to perform. decideConvergence148// makes the decision; reconcile() acts.149type convergence struct {150	condition      api.Condition151	stage          bool                       // write the manifest to the machineState filesystem152	requestReboot  bool                       // write the reboot intent for init153	requestRestart bool                       // write the restart intent; a k3s restart applies it154	requestLoad    bool                       // write the modules intent; init loads the staged additions while the system runs155	withdraw       bool                       // remove the staged manifest; the spec no longer names it156	clearRejection bool                       // remove the rejection record; the spec it blocks is gone157	manifest       []byte                     // the bytes to stage158	hash           string                     // the bytes' identity159	pending        *machine.PendingDisruption // the status.pending entry for this document, when one waits160}161162// The condition constructors for every convergence verdict. Three163// documents converge through this machinery, each under its own164// condition type (SpecConverged, ClusterConverged, VersionConverged).165// All three share one set of reasons, so the constructors take the166// type as a parameter instead of hard-coding it.167func converged(condType, reason, message string) api.Condition {168	return api.Condition{Type: condType, Status: api.ConditionTrue, Reason: reason, Message: message}169}170171func notConverged(condType, reason, message string) api.Condition {172	return api.Condition{Type: condType, Status: api.ConditionFalse, Reason: reason, Message: message}173}174175func convergenceUnknown(condType, reason, message string) api.Condition {176	return api.Condition{Type: condType, Status: api.ConditionUnknown, Reason: reason, Message: message}177}178179// The convergence constructors for the verdicts that every document's180// decision table shares. The decision tables mirror one another by181// design: they use the same guards in the same order, so a reader182// who has followed one document's convergence can follow them all.183// These constructors keep that mirroring exact, rather than184// accidental.185186// factsIncomplete is the guard every decision starts with. With no187// facts, or with facts that carry no boot record (an older init, or188// a machine in the middle of an upgrade), the verdict is Unknown.189// Guessing here could reboot a machine because of a misreading.190func factsIncomplete(condType string) convergence {191	return convergence{condition: convergenceUnknown(condType, "FactsIncomplete",192		"the machine's facts carry no boot record yet")}193}194195// machineStateEphemeral is the verdict for when there is nowhere196// durable to stage a document. The machineState role is backed by197// memory, so anything staged would disappear at the next reboot,198// which is exactly when it would be needed. what names the document199// that has nowhere to go.200func machineStateEphemeral(condType, what string) convergence {201	return convergence{condition: notConverged(condType, "MachineStateEphemeral",202		fmt.Sprintf("machineState is backed by memory; there is no durable filesystem to stage %s into; declare machineState in the machine's manifest", what))}203}204205// convergedWithCleanup wraps a True verdict with the cleanup that206// every document performs on convergence. When a manifest is still207// staged for a spec the cluster no longer requests, this function208// withdraws it, because the next boot would otherwise apply it. This209// function also clears a standing rejection for the same reason: the210// spec it blocks is no longer requested, so the record no longer211// blocks anything.212func convergedWithCleanup(cond api.Condition, stagedHash string, rejection *machine.Rejection) convergence {213	return convergence{214		condition:      cond,215		withdraw:       stagedHash != "",216		clearRejection: rejection != nil,217	}218}219220// gateDisruption finishes a staged document's convergence. The221// staged bytes are already in the convergence, and what remains is222// whether this machine may take its disruption right now. The223// decision table is the same for every staged document, and it is224// the safety core of the rollout design, so this function holds it225// in one place. Manual policy waits for a person, and the226// approve-disruption annotation is how the person answers: an227// approval naming the staged hash moves the machine onto the same228// path Auto takes, through the conductor's turn and the drain,229// which is safer than the state it replaces, where an operator230// following the machine's own advice cuts power and no budget231// applies at all. An approval naming some other hash is reported,232// not ignored: the pending message carries both values, so a wrong233// paste is visible where the person is already looking. A cluster234// member on Auto (or approved Manual) waits for the conductor's235// turn (AwaitingTurn is the same reason for both kinds of236// disruption, which lets the conductor sequence them without237// knowing the difference between them). Only a standalone machine,238// or a machine that has been granted a turn, asks init to act. The239// restart flag picks the kind of disruption: a k3s restart for240// changes that k3s reads only when its process starts, and a241// machine reboot for everything else. A leader's restart still242// restarts the embedded datastore. This is the same exposure to a243// lost quorum that a reboot has, so restarts wait for the same244// turns as reboots do. Every branch also records the document in245// status.pending, because a staged document waits for its246// disruption until the disruption runs, whatever it waits on. The247// messages differ for each document, but the reasons and their248// order of precedence must not.249func gateDisruption(c *convergence, condType string, policy machine.RebootPolicy, t turn, restart bool, approval, summary, pending, awaiting, requested string) {250	kind := machine.DisruptionReboot251	pendingReason, requestedReason := "RebootPending", "RebootRequested"252	if restart {253		kind = machine.DisruptionRestart254		pendingReason, requestedReason = "RestartPending", "RestartRequested"255	}256	c.pending = &machine.PendingDisruption{257		Condition: condType, Kind: kind, Hash: c.hash, Summary: summary,258	}259	switch {260	case policy != machine.RebootAuto && !machine.ApprovalGrants(approval, c.hash):261		message := pending262		if approval != "" {263			message = fmt.Sprintf("%s; the %s annotation names %s, and the staged document is %.12s",264				pending, machine.ApproveDisruptionAnnotation, approval, c.hash)265		}266		c.condition = notConverged(condType, pendingReason, message)267	case t == turnAwaiting:268		c.condition = notConverged(condType, "AwaitingTurn", awaiting)269	default:270		c.requestReboot = !restart271		c.requestRestart = restart272		c.condition = notConverged(condType, requestedReason, requested)273	}274}275276// decideConvergence makes the whole convergence decision in one pure277// function. The cases run in this order, and each one stops the278// function as soon as it applies:279//280//  1. No facts, or facts with no boot record (an older init, or a281//     machine in the middle of an upgrade): the verdict is Unknown.282//     Guessing here could reboot a machine because of a misreading.283//  2. No drift: the verdict is converged. This case also cleans up284//     after an edit that was reverted. When a manifest is still285//     staged for a spec the cluster no longer requests, this case286//     withdraws it, because the next boot would otherwise apply it.287//     This case also clears a standing rejection for the same288//     reason: the spec it blocks is no longer requested, so the289//     record no longer blocks anything.290//  3. The desired spec is the one init rejected: the function291//     refuses to stage it again. The rejection parameter comes from292//     the durable quarantine record on machineState, not from293//     facts. Facts are a snapshot taken at boot, and they do not294//     change while the machine runs. But when an edit is reverted295//     and then retried within one boot, the clearing of the296//     rejection must take effect right away, not at the next297//     reboot. Only a genuinely different edit, or clearing the298//     record through convergence, unblocks the hash.299//  4. The facts claim this exact manifest was actuated, yet drift300//     still computes: this is a contradiction, and it can only mean301//     a liken bug, with one exception. Parameter-only drift under a302//     matching hash is a state a live load produces by design: it303//     promotes the manifest and records only the parameters it304//     delivered, so an undelivered one drifts toward the reboot305//     that can deliver it (init/liveload.go). That shape takes the306//     staged path below. Every other drift shape under a matching307//     hash holds in the stuck condition, because holding is better308//     than rebooting the machine on every reconcile pass.309//  5. machineState is backed by memory: there is nowhere durable to310//     stage a document.311//  6. The spec fails validation against the machine's reality.312//  7. Valid drift: the function stages the manifest, unless these313//     exact bytes are already staged. Then, following rebootPolicy314//     and the machine's turn, it requests the reboot, waits for the315//     cluster's grant, or reports that a reboot is pending.316func decideConvergence(m *machine.Machine, facts *machine.MachineStatus, rejection *machine.Rejection, stagedHash string, t turn) convergence {317	if facts == nil || facts.Boot.ManifestSource == "" {318		return factsIncomplete("SpecConverged")319	}320321	storageDiffs := machine.StorageDrift(m.Spec.Storage, facts.Boot.Storage)322	networkDiffs := machine.NetworkDrift(m.Spec.Network, facts.Boot.Network)323	added, retracted := machine.ModuleSetDiff(m.Spec.Modules, facts.Boot.Modules)324	parameterDiffs := machine.ModuleParameterDrift(m.Spec.Modules, facts.Boot.Modules,325		m.Spec.ModuleParameters, facts.Boot.ModuleParameters)326	serioAdded, serioRetracted := machine.SerioSetDiff(m.Spec.Serio, facts.Boot.Serio)327	drift := slices.Concat(storageDiffs, networkDiffs,328		machine.ModulesDrift(m.Spec.Modules, facts.Boot.Modules,329			m.Spec.ModuleParameters, facts.Boot.ModuleParameters),330		machine.SerioDrift(m.Spec.Serio, facts.Boot.Serio),331		machine.RlimitDrift(m.Spec.Rlimits, facts.Boot.Rlimits))332	// The order difference stays apart from drift and never joins it.333	// The lines in drift are the count that decides the live-load334	// tier below, and every line there names something a load or a335	// reboot can actuate. Only a boot actuates a reorder, and a336	// reorder is worth no boot of its own, so it takes neither tier.337	// It stages, and any later boot loads the list in the order the338	// spec now gives.339	orderDiffs := machine.ModuleOrderDrift(m.Spec.Modules, facts.Boot.Modules)340	if len(drift) == 0 && len(orderDiffs) == 0 {341		return convergedWithCleanup(342			converged("SpecConverged", "Converged", "this boot actuated the current spec"),343			stagedHash, rejection)344	}345	diffs := strings.Join(slices.Concat(drift, orderDiffs), "; ")346347	manifest, hash, err := renderManifest(m.Metadata.Name, m.Spec)348	if err != nil {349		return convergence{condition: notConverged("SpecConverged", "StagingFailed", err.Error())}350	}351352	if rejection != nil && rejection.Hash == hash {353		return convergence{condition: notConverged("SpecConverged", "RejectedLastBoot",354			fmt.Sprintf("init rejected this exact spec at boot: %s; edit the spec to something different", rejection.Reason))}355	}356	// A matching hash may legitimately carry two shapes, and both are357	// what a live load leaves behind. The first is a parameter the358	// load could not deliver (list item 4 above): every drift line is359	// a module parameter on an unchanged module set. The second is an360	// order the load could not change, because the modules the boot361	// loaded keep their places in the running kernel. An order362	// difference lives outside drift, so it passes this test with no363	// term of its own. The test is structural, a count over the364	// drift's parts, so no drift text is ever matched and any other365	// shape stays a contradiction.366	parametersOnly := len(storageDiffs) == 0 && len(networkDiffs) == 0 &&367		len(added) == 0 && len(retracted) == 0 && len(parameterDiffs) == len(drift)368	if facts.Boot.ManifestHash == hash && !parametersOnly {369		return convergence{condition: notConverged("SpecConverged", "BootMismatch",370			fmt.Sprintf("facts claim this spec was actuated, yet it differs from the boot's record (%s); refusing to reboot over a contradiction; this is a liken bug", diffs))}371	}372	if facts.Storage.MachineState.Backing != machine.BackingPartition {373		return machineStateEphemeral("SpecConverged", "a manifest")374	}375	// The network spec is checked here as well as at boot. A spec376	// that init would refuse is a spec that costs a reboot to find377	// out about, and the machine would come back on its proven378	// manifest with a rejection record instead of the network the379	// person asked for. The check is the manifest's own consistency380	// only: whether this machine really has a port with a declared381	// name is a question only the boot can settle.382	if err := m.Spec.Network.Validate(); err != nil {383		return convergence{condition: notConverged("SpecConverged", "StagingRejected", err.Error())}384	}385	// Resource limits are checked for the same reason. Init would skip386	// a limit it cannot apply rather than refuse the boot, so a typo387	// here costs a reboot and then applies nothing at all, with only a388	// console line to say why. Refusing to stage it says so in a389	// condition instead.390	if err := machine.ValidateRlimits(m.Spec.Rlimits); err != nil {391		return convergence{condition: notConverged("SpecConverged", "StagingRejected", err.Error())}392	}393	// A serio entry that fails here could never match a device, and394	// the API server refuses the same entries, so only a manifest395	// that no API server admitted reaches this check.396	if err := machine.ValidateSerio(m.Spec.Serio); err != nil {397		return convergence{condition: notConverged("SpecConverged", "StagingRejected", err.Error())}398	}399	if err := validateStaging(m.Spec.Storage, facts); err != nil {400		return convergence{condition: notConverged("SpecConverged", "StagingRejected", err.Error())}401	}402403	c := convergence{404		manifest: manifest,405		hash:     hash,406		stage:    stagedHash != hash, // idempotence: skip the write when these exact bytes are already staged407	}408409	// The next-boot tier comes first, because a difference in nothing410	// but the declared order is the lightest answer here. Nothing can411	// load a resident module again in a new place, so no load applies412	// a reorder. Nothing on the machine is wrong while the reorder413	// waits, so no reboot is worth asking for. Staging is the whole of414	// the work, and the next boot, whatever its cause, loads the list415	// in the new order. The cluster document has the same tier for the416	// same reason (cluster.go).417	if len(drift) == 0 {418		c.condition = notConverged("SpecConverged", "StagedForNextBoot",419			fmt.Sprintf("spec staged for the next boot (%.12s); the machine loads the modules in the new order at its next boot and asks for no turn: %s", hash, diffs))420		return c421	}422423	// Adding modules is the one machine-spec change that needs no424	// disruption. Loading can happen while the system runs: the425	// kernel binds a resident driver to hardware that is already426	// plugged in, on its own. So when every difference is an added427	// module, the manifest stages for durability, and init loads the428	// additions into the running kernel. This case needs no policy429	// gate and no reboot turn, the same as the sysctls the operator430	// reconciles live: the gates exist for disruptions, and this is431	// not one. (Removing a module still needs a reboot, because432	// loading is one-way. The kernel offers no safe way to remove a433	// driver while something is using it.)434	//435	// The test counts rather than naming the fields that must be436	// unchanged. ModulesDrift writes exactly one line for each added437	// module, so the counts match only when the added modules are the438	// whole of the drift. This makes the safety property structural,439	// the same way RestartApplies does it for the cluster document: a440	// new spec field is reboot-class from the day it is added,441	// because its diffs land in drift and never in added. Naming the442	// fields instead would make a forgotten field silently live-class,443	// and init's live loader promotes the staged manifest to proven444	// whether or not it applied anything. A machine would then report445	// itself converged on a spec it never actuated.446	//447	// A declared reorder rides along with the additions. The load448	// applies the additions in the order the manifest lists them, and449	// the reorder of what the boot already loaded stays staged for the450	// next boot.451	//452	// An added serio entry is live-class on the same terms. Init453	// attaches it the moment the load declares it, and SerioDrift454	// writes exactly one line for each added entry, so the count455	// extends without naming a field. A retracted entry is456	// reboot-class, like a retracted module: its holder keeps the port457	// for the pods that hold the devices the port created.458	if len(retracted) == 0 && len(serioRetracted) == 0 && len(drift) == len(added)+len(serioAdded) {459		c.requestLoad = true460		c.condition = notConverged("SpecConverged", "LoadRequested",461			fmt.Sprintf("module load requested to apply the staged spec (%.12s) in place: %s", hash, diffs))462		return c463	}464465	gateDisruption(&c, "SpecConverged", m.Spec.RebootPolicyOrDefault(), t, false,466		m.Metadata.Annotations[machine.ApproveDisruptionAnnotation],467		"the machine spec: "+diffs,468		fmt.Sprintf("spec staged for the next boot (%.12s); rebootPolicy is Manual, so reboot the machine (or set rebootPolicy: Auto) to apply: %s", hash, diffs),469		fmt.Sprintf("spec staged for the next boot (%.12s); waiting for the cluster to grant a reboot turn: %s", hash, diffs),470		fmt.Sprintf("reboot requested to apply the staged spec (%.12s): %s", hash, diffs))471	return c472}
machine-operator/demotion.go 100.0%
1package main23// Demotion cleanup finishes what a role change starts.4//5// Promotion completes on its own. A follower rebooted into the6// leader role starts a control plane, and k3s labels the Node and7// registers the etcd member without help. Demotion does not8// complete on its own. A leader rebooted into the follower role9// runs `k3s agent`, but the Kubernetes Node object it reattaches to10// still claims control-plane and etcd. Worse, its etcd membership11// stays registered. A registered member that never votes still12// counts toward the quorum size, so it breaks the majority math the13// next time an actual leader reboots.14//15// The demoted machine's own operator holds everything needed to16// finish this job. The facts state what this machine is (a17// follower), and the Node object states what the cluster still18// records it as. When these disagree, the operator requests a reboot19// through the intent channel it already owns, then deletes its own20// Node object. Deleting the Node triggers k3s's etcd21// member-removal controller. The operator writes the intent first,22// deliberately. Deleting the Node kills this same pod, because pods23// bound to a deleted Node are garbage-collected, and a machine whose24// Node is gone cannot re-register without a reboot. So the reboot25// must already be in progress before the delete happens. If the26// delete itself fails, the next boot detects the same mismatch and27// tries again. Each retry costs a reboot, but the state converges.28//29// The reboot policy gates all of this, the same as every other30// staged change. Under Manual, the operator only reports the state31// (DemotionPending), because deleting the Node without a reboot32// already under way would strand a working machine.3334import (35	"fmt"3637	"github.com/liken-sh/liken/kubernetes/apiclient"38	"github.com/liken-sh/liken/liken/api"39	"github.com/liken-sh/liken/liken/machine"40)4142// The role labels that k3s applies to a Node when it runs a control43// plane. Their presence on a follower's Node is what a demotion44// leaves behind.45var leaderNodeLabels = []string{46	"node-role.kubernetes.io/control-plane",47	"node-role.kubernetes.io/etcd",48}4950// A demotion is the cleanup decision: whether to act, and the51// NodeCurrent condition to publish either way.52type demotion struct {53	cleanup   bool54	condition api.Condition55}5657// decideDemotion compares what this machine is, its derived role,58// against what its Node object claims. Only one mismatch is the59// operator's job to fix: a follower whose Node still says60// control-plane. The other direction, a leader whose Node lacks the61// labels, only means a control plane still starting up, and k3s62// finishes that on its own.63//64// The demotion's reboot waits its turn like any other reboot. A65// demotion always follows a Cluster edit, so other machines are66// converging on the same edit at the same time. This is exactly the67// traffic the rollout conductor is built to sequence.68func decideDemotion(role api.Role, nodeLabels map[string]string, rebootPolicy machine.RebootPolicy, t turn) demotion {69	nodeCurrent := func(status api.ConditionStatus, reason, message string) api.Condition {70		return api.Condition{Type: "NodeCurrent", Status: status, Reason: reason, Message: message}71	}7273	if role != api.RoleFollower {74		return demotion{condition: nodeCurrent("True", "NodeMatchesRole", "the Node object matches this machine's role")}75	}76	stale := false77	for _, label := range leaderNodeLabels {78		if _, ok := nodeLabels[label]; ok {79			stale = true80		}81	}82	if !stale {83		return demotion{condition: nodeCurrent("True", "NodeMatchesRole", "the Node object matches this machine's role")}84	}8586	if rebootPolicy != machine.RebootAuto {87		return demotion{condition: nodeCurrent("False", "DemotionPending",88			"this machine was demoted to follower but its Node object still claims control-plane; set rebootPolicy: Auto to let the operator delete the Node and reboot, completing the demotion")}89	}90	if t == turnAwaiting {91		return demotion{condition: nodeCurrent("False", "AwaitingTurn",92			"this machine was demoted to follower; waiting for the cluster to grant a reboot turn to complete the demotion")}93	}94	return demotion{95		cleanup: true,96		condition: nodeCurrent("False", "DemotionRebooting",97			"completing the demotion: deleting the stale control-plane Node object and rebooting to re-register as a follower"),98	}99}100101// carryOutDemotion performs the cleanup. It writes the reboot intent102// first, because deleting the Node kills this pod, so the reboot103// must already be in progress. Then it deletes the Node instance the104// pass read, by its UID, which triggers etcd member removal.105func carryOutDemotion(c *apiclient.Client, runDir string, node *nodeObject, d demotion, out *passOutcome) api.Condition {106	if !d.cleanup {107		return d.condition108	}109	intent := &machine.RebootIntent{Reason: "completing the demotion to follower"}110	if err := machine.WriteRebootIntent(runDir, intent); err != nil {111		out.fail("writing the demotion's reboot intent", err)112		return api.Condition{Type: "NodeCurrent", Status: api.ConditionFalse, Reason: "DemotionFailed",113			Message: fmt.Sprintf("writing the reboot intent: %v", err)}114	}115	out.wrote("writing the demotion's reboot intent")116	name := node.Metadata.Name117	if err := deleteNode(c, name, node.Metadata.UID); err != nil {118		// The reboot is already in progress. The next boot detects119		// the mismatch again and retries the delete.120		fmt.Printf("deleting the stale Node %s: %v\n", name, err)121	} else {122		fmt.Printf("deleted the stale control-plane Node %s; rebooting to re-register as a follower\n", name)123	}124	return d.condition125}
machine-operator/disruptions.go 100.0%
1package main23import (4	"time"5)67// disruptions is one pass's running record of what has already8// started: whether some document requested the reboot, and whether9// a drain is holding one back. The documents pass through the gate10// in a fixed order: the Machine's spec, the cluster document, the11// system release, the registry credentials, and finally the12// demotion. The restart suppression in gate depends on this order.13// A reboot requested by an earlier document silences a later14// document's restart, never the reverse.15type disruptions struct {16	draining  bool17	rebooting bool1819	// events posts the drain's cordon about this Machine.20	events machineEvents2122	// out records what the drain did not finish, and when the drain's23	// deadline wants a pass.24	out *passOutcome25}2627// gate intercepts one document's convergence decision on its way to28// its side effects. A reboot already requested this pass covers any29// restart: the boot path re-renders everything a restart would have30// applied, so a second intent would only add noise. (Init also31// prefers the reboot file when both exist, so this guard is not32// strictly needed, but it does no harm.) A granted reboot goes33// through the drain first (drain.go): the node is cordoned and34// emptied before the intent is written, so workloads move to other35// nodes instead of being killed by the reboot. A pass whose Node36// read failed skips the drain, because during a demotion there is37// no Node to cordon, and the reboot must still happen. The Node's38// copy stops answering after a failed watch (watches.go), so while the39// API server is down the Node read fails and the drain is skipped the40// same way. A Node that reads but whose pods do not list holds the41// reboot (gateThroughDrain): a slow API server must not let a reboot42// kill pods past their disruption budgets.43func (d *disruptions) gate(r *reader, node *nodeObject, nodeErr error, t turn, now time.Time, conv convergence) convergence {44	conv.requestRestart = conv.requestRestart && !d.rebooting45	if conv.requestReboot && t == turnGranted && nodeErr == nil {46		conv = gateThroughDrain(r, node, conv, now, d.events, d.out)47		d.draining = d.draining || !conv.requestReboot48	}49	d.rebooting = d.rebooting || conv.requestReboot50	return conv51}
machine-operator/download.go 85.8%
1package main23// The download: what one run of the fetcher writes to the inactive4// slot (fetch.go runs it).5//6// A complete fetch leaves a bootable slot, and this takes more than7// downloading the release. The public artifacts are downloaded and8// verified against the document. Then the machine's own deployment9// layer is carried over from the slot it is running on (carryLayer),10// because the layer never travels the network, and no release can11// supply it.12//13// Downloads resume through re-verification, not through byte14// ranges. Each run first verifies whatever the slot already holds15// against the release document, and fetches only what fails16// verification. A torn download, from a power cut or a killed17// server, leaves either a .partial file, which no verification ever18// counts, or a final file that either verifies or does not. The next19// run converges either way. FAT has no journal, so every file lands20// the way the installer's copies do: temp file, fsync, rename. The21// function re-reads and verifies the file after writing it, because22// bytes sitting in the page cache are not durable until they are23// synced and read back.2425import (26	"bytes"27	"context"28	"crypto/sha256"29	"encoding/hex"30	"errors"31	"fmt"32	"io"33	"net/http"34	"os"35	"path/filepath"36	"strings"3738	"golang.org/x/sys/unix"3940	"github.com/liken-sh/liken/liken/machine"41	"github.com/liken-sh/liken/liken/releases"42)4344// fetchRelease runs one complete pass. It fetches and checks the45// release document, removes the slot's old document, verifies or46// fetches each artifact, and writes the new document to the slot47// last. This order means a slot carrying release.yaml is a slot whose48// artifacts were complete when the document was written, and that no49// writer has changed since. fetchRelease returns how many50// artifacts it actually downloaded, and how many bytes those51// artifacts hold. Zero is the idempotent case, where everything was52// already verified in place. The byte total counts an artifact only53// after the artifact lands and verifies, so a torn file adds nothing54// until the run that completes it.55func fetchRelease(ctx context.Context, client *http.Client, ask fetchAsk) (int, int64, error) {56	base := strings.TrimSuffix(ask.source, "/") + "/" + ask.version5758	raw, err := fetchBytes(ctx, client, base+"/release.yaml")59	if err != nil {60		return 0, 0, fmt.Errorf("fetching the release document: %w", err)61	}6263	// The first check in the trust chain: the document's bytes must64	// hash to exactly what the catalog promised. Until that check65	// passes, nothing the document says can be trusted.66	sum := sha256.Sum256(raw)67	if digest := "sha256:" + hex.EncodeToString(sum[:]); digest != ask.digest {68		return 0, 0, fmt.Errorf("the release document's digest %s does not match the catalog's %s: %w", digest, ask.digest, errCorrupt)69	}70	release, err := machine.ParseRelease(raw)71	if err != nil {72		return 0, 0, fmt.Errorf("the release document does not parse: %v: %w", err, errCorrupt)73	}74	if release.Metadata.Name != ask.version {75		return 0, 0, fmt.Errorf("the release document names version %s, not %s: %w", release.Metadata.Name, ask.version, errCorrupt)76	}7778	if err := withdrawSlotDocument(ask.slotDir, raw); err != nil {79		return 0, 0, fmt.Errorf("removing the slot's previous release document: %w", err)80	}8182	fetched := 083	downloaded := int64(0)84	for _, artifact := range release.Artifacts {85		dest := filepath.Join(ask.slotDir, artifact.Name)86		if verifySlotFile(artifact, dest) == nil {87			continue // already here from an earlier, interrupted run88		}89		if err := fetchArtifact(ctx, client, base, artifact, dest); err != nil {90			return fetched, downloaded, err91		}92		fetched++93		downloaded += artifact.Size94	}9596	// The deployment layer is the one file the release cannot97	// supply. It belongs to this cluster alone, so the machine98	// carries it forward from the slot it is running on. This step99	// runs between the artifacts and the document deliberately: a100	// slot with release.yaml is bootable, and a slot without its101	// layer is not.102	if err := carryLayer(ask); err != nil {103		return fetched, downloaded, err104	}105106	// The document lands after the artifacts it describes, written107	// durably. This makes the slot self-describing: it records108	// which release it holds, byte for byte, without asking the109	// network.110	if err := writeDurably(filepath.Join(ask.slotDir, "release.yaml"), raw); err != nil {111		return fetched, downloaded, fmt.Errorf("writing the release document to the slot: %w", err)112	}113	return fetched, downloaded, nil114}115116// withdrawSlotDocument removes the slot's release document unless it117// is already the document of this release. The slot then claims no118// release while its files change, so init cannot arm a trial of a119// slot that holds part of one release and part of another120// (armProvingBoot in init/proving.go checks the document and every121// artifact it names). The removal reaches the disk before the first122// artifact is written, so a power cut in the middle of the download123// leaves a slot with no document, not one with the old document.124//125// A slot is FAT, where a file's directory entry lives in buffers of126// the block device, and an fsync of the directory does not write them127// (flushSlot in init/slotloader.go). So syncfs writes the slot's128// filesystem back, buffers included, and the fsync of the directory129// then empties the drive's write cache.130func withdrawSlotDocument(slotDir string, raw []byte) error {131	path := filepath.Join(slotDir, "release.yaml")132	existing, err := os.ReadFile(path)133	if errors.Is(err, os.ErrNotExist) || bytes.Equal(existing, raw) {134		return nil135	}136	if err := os.Remove(path); err != nil {137		return err138	}139	dir, err := os.Open(slotDir)140	if err != nil {141		return err142	}143	defer dir.Close()144	if err := unix.Syncfs(int(dir.Fd())); err != nil {145		return err146	}147	return dir.Sync()148}149150// carryLayer copies the running slot's deployment layer and151// sidecar to the inactive slot. The active slot is the source of152// truth. Its sidecar was written from verified bytes at install, or153// by the carry that filled it. So a layer that fails to verify154// against the active sidecar means the running slot itself is155// damaged, a condition that no retry and no download can repair.156// This is why the fetcher holds it the way it holds corruption. The157// remedy belongs to a person: repair or reinstall the machine.158func carryLayer(ask fetchAsk) error {159	sidecar, err := os.ReadFile(filepath.Join(ask.activeSlotDir, machine.LayerSidecarName))160	if err != nil {161		return fmt.Errorf("the running slot's deployment layer cannot be vouched for (%v); repair or reinstall this machine: %w", err, errLayer)162	}163	digest, err := machine.ParseLayerSidecar(sidecar)164	if err != nil {165		return fmt.Errorf("the running slot's layer sidecar is damaged (%v); repair or reinstall this machine: %w", err, errLayer)166	}167	verify := func(path string) error {168		f, err := os.Open(path)169		if err != nil {170			return err171		}172		defer f.Close()173		return machine.VerifyLayer(digest, f)174	}175	source := filepath.Join(ask.activeSlotDir, machine.LayerName)176	if err := verify(source); err != nil {177		return fmt.Errorf("the running slot's deployment layer does not verify (%v); repair or reinstall this machine: %w", err, errLayer)178	}179180	// This resumes the same way the artifacts do. A layer already181	// carried, with a sidecar matching the active one, needs nothing182	// more. A layer from some older install fails this check and183	// gets replaced. A carry that died between writing the layer and184	// writing its sidecar resumes by rewriting only the sidecar.185	dest := filepath.Join(ask.slotDir, machine.LayerName)186	destSidecar := filepath.Join(ask.slotDir, machine.LayerSidecarName)187	if verify(dest) != nil {188		f, err := os.Open(source)189		if err != nil {190			return fmt.Errorf("reading the running slot's layer: %w", err)191		}192		tmp, err := spillDurably(dest, f)193		f.Close()194		if err != nil {195			return fmt.Errorf("carrying %s: %w", machine.LayerName, err)196		}197		if err := verify(tmp); err != nil {198			os.Remove(tmp)199			return fmt.Errorf("the carried layer does not verify: %v: %w", err, errLayer)200		}201		if err := os.Rename(tmp, dest); err != nil {202			return err203		}204	}205206	// The sidecar lands last, written durably. A slot whose sidecar207	// matches its layer is a slot whose carry completed.208	if existing, err := os.ReadFile(destSidecar); err != nil || !bytes.Equal(existing, sidecar) {209		if err := writeDurably(destSidecar, sidecar); err != nil {210			return fmt.Errorf("carrying %s: %w", machine.LayerSidecarName, err)211		}212	}213	return nil214}215216// fetchArtifact streams one artifact onto the slot: temp file,217// fsync, verify the durable bytes by re-reading them, then rename218// into place. Verifying before renaming means a final-looking file219// name never points at unverified bytes.220func fetchArtifact(ctx context.Context, client *http.Client, base string, artifact machine.ReleaseArtifact, dest string) error {221	resp, err := releases.Get(ctx, client, base+"/"+artifact.Name)222	if err != nil {223		return fmt.Errorf("fetching %s: %w", artifact.Name, err)224	}225	defer resp.Body.Close()226227	// The size cap protects the slot. An artifact that runs past its228	// declared size is already wrong, and there is no reason to229	// fill a 512Mi filesystem with the rest of it before finding230	// that out.231	tmp, err := spillDurably(dest, io.LimitReader(resp.Body, artifact.Size+1))232	if err != nil {233		return fmt.Errorf("writing %s: %w", artifact.Name, err)234	}235236	if err := verifySlotFile(artifact, tmp); err != nil {237		os.Remove(tmp)238		return fmt.Errorf("%s from the server does not verify: %v: %w", artifact.Name, err, errCorrupt)239	}240	return os.Rename(tmp, dest)241}242243// verifySlotFile checks one file on the slot against its244// artifact's digest and size. It returns an error for any reason245// the file fails, including that the file does not exist, which is246// the common case on a first run.247func verifySlotFile(artifact machine.ReleaseArtifact, path string) error {248	f, err := os.Open(path)249	if err != nil {250		return err251	}252	defer f.Close()253	return artifact.Verify(f)254}255256// fetchBytes reads a small document whole with an HTTP GET. The257// 1MiB limit is far larger than any reasonable release.yaml, and258// small enough to read into memory without concern.259func fetchBytes(ctx context.Context, client *http.Client, url string) ([]byte, error) {260	resp, err := releases.Get(ctx, client, url)261	if err != nil {262		return nil, err263	}264	defer resp.Body.Close()265	return io.ReadAll(io.LimitReader(resp.Body, 1<<20))266}267268// spillDurably writes a stream next to its destination with the269// same steps the installer applies to file copies: write to a270// .partial temp file, fsync, close. On any failure, the function271// removes the temp file, and nothing further sees it. On success,272// the function returns the temp path for the caller to finish: the273// caller either verifies first and then renames (fetchArtifact), or274// renames immediately (writeDurably).275func spillDurably(dest string, r io.Reader) (string, error) {276	tmp := dest + ".partial"277	f, err := os.Create(tmp)278	if err != nil {279		return "", err280	}281	_, err = io.Copy(f, r)282	if err == nil {283		err = f.Sync()284	}285	if closeErr := f.Close(); err == nil {286		err = closeErr287	}288	if err != nil {289		os.Remove(tmp)290		return "", err291	}292	return tmp, nil293}294295// writeDurably writes bytes already in memory with the same296// steps: temp file, fsync, rename.297func writeDurably(dest string, contents []byte) error {298	tmp, err := spillDurably(dest, bytes.NewReader(contents))299	if err != nil {300		return err301	}302	return os.Rename(tmp, dest)303}
machine-operator/dra.go 96.2%
1package main23// The DRA driver's inventory half: publishing this machine's driven4// devices as a ResourceSlice.5//6// liken's answer to the question of how workloads reach hardware is7// dynamic resource allocation, and the driver is this operator: one8// more job in a process that already runs on every node, not a9// second daemon, because the memory envelope has no room for one.10// The driver symlink divides the milestone's two halves. A device11// with no driver is init's unclaimed report in Machine status, aimed12// at the person who can declare a module. A device with a driver is13// working equipment, and working equipment belongs in the cluster's14// own inventory API, where DeviceClasses can select it and pods can15// claim it. spec.modules is the gate between the two: declaring a16// driver is what moves a device from one report to the other.17//18// The operator walks sysfs itself, rather than reading init's19// facts. This is deliberate. The facts tree carries what init20// observes for the Machine status, and inventory is not status: it21// is a separate report, to a separate audience, with a separate22// lifetime. Slices end with the Node; status lives with the Machine.23// The walk uses the same shared package init uses, so the two24// reports can never disagree about what a device is. The walk runs25// on each pass, and a uevent for a device that arrives, leaves, or26// changes its driver wakes a pass (machineevents.go), so a device27// reaches the slice about a second after the kernel reports it.2829import (30	"fmt"31	"os"32	"slices"33	"strings"34	"sync"35	"sync/atomic"3637	"github.com/liken-sh/liken/liken/hardware"38	"github.com/liken-sh/liken/liken/kubernetes"39	"github.com/liken-sh/liken/liken/machine"40)4142// The operator reads the host's sysfs directly, because this pod43// runs in the host's namespaces, and names PCI devices from the44// database the image ships. These are variables so tests can45// substitute both, the same seam init's hardware watcher leaves.46var (47	draSysfsRoot  = "/sys"48	draPCIIDsPath = "/usr/share/hwdata/pci.ids"49)5051// draNaming loads the PCI naming database once per process. The52// file is part of the image, so it cannot change while the operator53// runs. A missing database degrades the device names, but never the54// inventory itself.55var draNaming = sync.OnceValue(func() *hardware.PCIIDs {56	naming, err := hardware.LoadPCIIDs(draPCIIDsPath)57	if err != nil {58		fmt.Fprintf(os.Stderr, "device inventory: no PCI naming database: %v\n", err)59		return nil60	}61	return naming62})6364// maxSliceDevices is the API's limit on devices in one65// ResourceSlice. One slice per node is enough: a physical device66// publishes a primary and a few companions, four in the widest shape67// the fleet has met, so a busy server's dozens of physical devices68// stay far under the limit of 128. If a machine ever69// exceeds this limit, the operator drops the overflow and reports70// it, rather than splitting devices across slices. This will change71// if real hardware needs the multi-slice pool protocol.72const maxSliceDevices = 1287374// publishDeviceInventory converges this node's ResourceSlice with75// what sysfs shows right now. The function logs failures and answers76// them, and the pass's outcome sets the retry, instead of reporting77// them as a condition. Inventory is a report about hardware, and a failure to78// write it is a problem in the operator's own machinery, not a fact79// about the machine.80func publishDeviceInventory(r *reader, node *nodeObject, facts *machine.MachineStatus, serio []machine.SerioAttachment, mm *machineMetrics) error {81	held := heldNodes(draSysfsRoot)82	devices := inventoryDevices(83		hardware.DiscoverInventory(draSysfsRoot, draNaming()),84		func(d hardware.Device) hardware.Delivery {85			return withoutHeld(hardware.InspectDelivery(draSysfsRoot, d), held)86		},87		protectionOf(facts), serio)88	if len(devices) > maxSliceDevices {89		fmt.Fprintf(os.Stderr, "device inventory: %d devices exceed one slice's capacity of %d; dropping the overflow\n",90			len(devices), maxSliceDevices)91		devices = devices[:maxSliceDevices]92	}93	// The device metric counts the very list the slice carries, so94	// the count comes from this pass's one sysfs walk and never from95	// a second one (metrics.go).96	mm.observeDevices(devices)9798	owner := kubernetes.OwnerReference{99		APIVersion: "v1",100		Kind:       "Node",101		Name:       node.Metadata.Name,102		UID:        node.Metadata.UID,103	}104	current, err := r.resourceSlice(node.Metadata.Name)105	if err == nil {106		err = r.sliceWriter().Write(r.client, node.Metadata.Name, current, owner, devices)107	}108	if err != nil {109		fmt.Fprintf(os.Stderr, "device inventory: %v\n", err)110	}111	return err112}113114// serioInEffect is the spec.serio list the machine holds attachments115// for: the spec's entries, and the boot record's. A retracted entry116// stays in the boot record until the next boot, and its holder keeps117// the port until then, so its tty must stay withheld until then too.118// An entry added since the boot is in the spec before init reports it,119// so its tty is withheld from the first pass after the edit.120func serioInEffect(spec []machine.SerioAttachment, facts *machine.MachineStatus) []machine.SerioAttachment {121	if facts == nil {122		return slices.Clone(spec)123	}124	return slices.Concat(spec, facts.Boot.Serio)125}126127// serioDeclared carries serioInEffect to the DRA plugin, which128// prepares claims on the kubelet's schedule, apart from the reconcile129// pass. main seeds it from the boot manifest and the boot's facts130// before the plugin serves, and every pass sets it again from the131// live spec, so the plugin never resolves a claim to a serial line132// that init holds.133var serioDeclared atomic.Pointer[[]machine.SerioAttachment]134135func setDeclaredSerio(entries []machine.SerioAttachment) {136	serioDeclared.Store(&entries)137}138139func declaredSerio() []machine.SerioAttachment {140	if entries := serioDeclared.Load(); entries != nil {141		return *entries142	}143	return nil144}145146// inventoryDevices applies the publish rule to the devices147// hardware.DiscoverInventory finds: the pci and usb devices, and the148// devices on the board that own a node, such as a firmware TPM or a149// laptop's keyboard. A device is offered to workloads when it passes150// all three tests:151//152//  1. The device has a driver and is not part of the bus structure153//     itself. Undriven hardware belongs in the unclaimed report154//     instead. usbcore's device nodes, hubs, and PCIe ports are the155//     structure that the peripherals connect to, not peripherals156//     themselves. A USB device that no driver binds at all is the157//     exception: a program in userspace drives it, and the device158//     publishes whole (userspace.go).159//  2. Claiming the device would deliver something: its subtree160//     carries device nodes that a pod could receive. A NIC or a161//     bare controller fails this test, because it is real hardware162//     with nothing to hand to a pod. A Bluetooth adapter is the one163//     device that passes on its driver instead. hci is a socket164//     interface, so the radio registers no node, and a working165//     adapter has an empty subtree. This test alone would refuse166//     hardware that workloads do claim. publishing.go carries the167//     rest of that story.168//  3. The machine does not depend on the device: nothing in its169//     subtree backs a storage role, and while the facts do not read,170//     nothing in its subtree is a disk (protection.go). The console171//     and the clock that init writes leave the delivery before this test,172//     so a device whose only nodes they are has nothing to deliver173//     (held.go).174//175// A serial line that init holds a serio attachment for passes the176// three tests with its tty node, and the policy then publishes the177// devices the attached driver created and never the tty178// (publishingserio.go). serio names those lines.179//180// A ResourceSlice is an offer, not a full record of the hardware.181// The scheduler can only allocate what a slice lists, so publishing182// the slice is itself the enforcement, ahead of whatever checks a183// deployment's DeviceClasses perform.184func inventoryDevices(discovered []hardware.Device,185	inspect func(hardware.Device) hardware.Delivery, platform protection,186	serio []machine.SerioAttachment) []kubernetes.SliceDevice {187	plumbing := map[string]bool{"usb": true, "hub": true, "pcieport": true}188	var out []kubernetes.SliceDevice189	for _, d := range discovered {190		if userspace, ok := userspaceDevice(d, discovered); ok {191			d = userspace192		} else if d.Driver == "" || plumbing[d.Driver] {193			continue194		}195		delivery := inspect(d)196		if len(delivery.DevNodes()) == 0 && !bluetoothAdapter(d) {197			continue198		}199		if platform.withholds(delivery) {200			continue201		}202		// One physical device can publish more than one slice203		// device: the loop publishes one slice device for each device204		// the policy derived from this delivery. Its suffix names the205		// slice device, joined to the physical device's own name, so206		// the primary keeps the bare name and an allocation made207		// before a split stays valid.208		for _, p := range publishFor(d, delivery, serio) {209			attrs := map[string]kubernetes.DeviceAttribute{}210			// Attribute names are unqualified, so the Kubernetes API211			// places them under the driver's own domain: a DeviceClass212			// selector reads them as device.attributes["liken.sh"].driver213			// and similar names. The code omits absent facts instead of214			// publishing them empty, so a selector like215			// `has(device.attributes["liken.sh"].serial)` means what it216			// says. Every published device carries its physical parent's217			// identifying attributes: the companion is the same silicon,218			// the same driver, and the same address as its primary.219			//220			// The address is also what pairs one physical device's221			// published devices back together. On a machine with two222			// GPUs, a claim that asks for a render node and a card node223			// constrains its two requests with matchAttribute, and224			// matchAttribute reads an attribute, never a name. The225			// address is equal across the devices of one card and226			// different across cards, so it is the fact that constraint227			// needs.228			for name, value := range map[string]string{229				"bus":       d.Bus,230				"address":   d.Address,231				"driver":    d.Driver,232				"class":     d.Class,233				"classCode": d.ClassCode,234				"subsystem": p.Subsystem,235				"name":      attributeString(d.Name),236				"modalias":  d.Modalias,237				"serial":    attributeString(d.Serial),238				"vendor":    d.Vendor,239				"product":   d.Product,240			} {241				if value != "" {242					attrs[name] = kubernetes.AttrString(value)243				}244			}245			// A render node is the fact a workload actually selects on. A246			// person who deploys a transcoder asks for any GPU that can247			// encode, and asking for a vendor and a product ID instead248			// names one machine's hardware in a document meant for a249			// fleet. A DeviceClass that selects on it can never allocate250			// the monitor buses by mistake.251			if p.RenderNode {252				attrs["renderNode"] = kubernetes.AttrBool(true)253			}254			// A display node is the other fact a workload selects on: the255			// card node that carries modesetting authority, which a256			// player or a kiosk needs and a transcoder never does. The two257			// facts are exclusive of each other, so a DeviceClass that258			// asks for one can never allocate the other by mistake.259			if p.DisplayNode {260				attrs["displayNode"] = kubernetes.AttrBool(true)261			}262			// The sound attribute states that a sound server can run263			// against this device. It is qualified where every other264			// attribute here is bare, because sound.liken.sh belongs to265			// no single driver: any driver may stamp the attribute, and266			// a DeviceClass that selects it names no driver, so a device267			// that supports a sound server joins that class by stamping268			// this one field. monitor.liken.sh/id takes the same form269			// for pairing a monitor's outputs.270			if p.Subsystem == "sound" {271				attrs["sound.liken.sh/supportsSound"] = kubernetes.AttrBool(true)272			}273			// The bare address pairs the devices of one card only274			// inside this driver, because a bare name belongs to the275			// driver that published it. Another driver's device for276			// the same card, such as media-operator's statement of277			// what a GPU decodes, pairs with these devices through278			// the attribute Kubernetes defines for a PCI device's279			// address. A sysfs PCI address is already the extended280			// BDF that the attribute's format names.281			if d.Bus == "pci" {282				attrs[pciBusIDAttribute] = kubernetes.AttrString(d.Address)283			}284			device := kubernetes.SliceDevice{285				Name:       deviceName(d) + p.Suffix,286				Attributes: attrs,287			}288			if p.Shareable {289				shared := true290				device.AllowMultipleAllocations = &shared291			}292			out = append(out, device)293		}294	}295	// The list is sorted, so the same hardware always publishes the296	// same slice. This lets the change detection in297	// WriteResourceSlice see actual inventory changes, and nothing298	// caused only by the order of the walk.299	slices.SortFunc(out, func(a, b kubernetes.SliceDevice) int {300		return strings.Compare(a.Name, b.Name)301	})302	return out303}304305// deviceName turns a sysfs address into the DNS label the API306// requires. The function prefixes the label with the bus, because307// addresses are unique only within a bus, lowercases it, and308// replaces the address punctuation (PCI's colons and dots, USB's309// dots and colons) with dashes. The address is the right identity310// for a slice device, because it names the slot, not the individual311// unit. Replacing a failed dongle in the same port produces the312// same device name, which is the behavior a claim against "the UPS313// on this wall" needs. When the hardware carries a serial number,314// the serial attribute is what identifies the individual unit.315//316// A device on the board takes its name from the firmware, such as317// MSFT0101:00 for a TPM or acpi.video_bus.0 for the display's318// brightness keys, so every character a DNS label cannot hold319// becomes a dash, and the name stops at the label's 63 characters.320func deviceName(d hardware.Device) string {321	name := []byte(strings.ToLower(d.Bus + "-" + d.Address))322	for i, c := range name {323		if (c < 'a' || c > 'z') && (c < '0' || c > '9') {324			name[i] = '-'325		}326	}327	return strings.Trim(string(name[:min(len(name), 63)]), "-")328}329330// attributeString limits a free-text value to the API's331// 64-character limit on attribute strings. Identifiers, such as332// addresses, hex IDs, and modaliases on these buses, always fit333// within this limit. Only the human-readable names that the334// hardware or pci.ids provides can run longer, and a truncated name335// still identifies the device.336func attributeString(s string) string {337	if len(s) <= 64 {338		return s339	}340	return s[:64]341}342343// pciBusIDAttribute is the standard attribute that Kubernetes defines344// for the address of a PCI device, in the `resource.kubernetes.io`345// domain that belongs to no single driver. Its value is the extended346// BDF notation, Domain:Bus:Device.Function, so it identifies one347// device on a node. The reference is348// https://kubernetes.io/docs/reference/node/dra-standard-device-attributes/.349const pciBusIDAttribute = "resource.kubernetes.io/pciBusID"
machine-operator/drain.go 100.0%
1package main23// Draining empties a node of workloads before its granted reboot.4//5// An unannounced reboot stops every pod on the machine at once, in6// the middle of whatever it was doing. Kubernetes answers this7// problem with `kubectl drain`: mark the node unschedulable so8// nothing new lands on it (a cordon), then ask each pod to leave9// through the Eviction API. Asking, rather than deleting directly,10// is the point. The API server refuses an eviction (429) when it11// would violate the workload's own PodDisruptionBudget. That12// refusal is the entire benefit over plain deletion: the workload's13// availability promise holds while the machine empties. This file14// implements that procedure. The machine's own operator runs it on15// itself when the rollout conductor grants its turn (rollout.go).16//17// The drain runs in steps, by design. One reconcile pass cordons18// the node and asks pods to leave; the next pass checks what19// remains and asks again. Blocking a pass until the node empties20// would stop the operator's heartbeat, and a machine that stops21// sending its heartbeat gets declared Lost. So the drain must never22// make a healthy machine look dead. The drain's state lives on the23// Node itself, in annotations, so a restarted operator resumes where24// it left off.2526import (27	"encoding/json"28	"fmt"29	"net/http"30	"strings"31	"time"3233	"github.com/liken-sh/liken/kubernetes/apiclient"34	"github.com/liken-sh/liken/liken/api"35	"github.com/liken-sh/liken/liken/kubernetes"36)3738const (39	// cordonedAnnotation marks a cordon as one that liken applied.40	// Uncordoning checks this annotation: a node that was already41	// unschedulable when the drain started was cordoned by a person,42	// and the operator must not remove their cordon.43	cordonedAnnotation = "liken.sh/cordoned"4445	// drainingSinceAnnotation records when the drain's deadline46	// started. It lives on the Node rather than in memory, so the47	// deadline keeps running across operator restarts.48	drainingSinceAnnotation = "liken.sh/draining-since"4950	// mirrorPodAnnotation is how the kubelet marks a static pod's51	// reflection in the API server. The kubelet recreates mirror pods52	// from disk, so the operator cannot evict them, and the drain53	// skips them.54	mirrorPodAnnotation = "kubernetes.io/config.mirror"5556	// drainDeadline limits how long workloads may delay the reboot. A57	// pod that has not moved by this deadline (because a58	// PodDisruptionBudget can never be satisfied, or a workload has59	// nowhere to go) stays running through the reboot instead. A60	// machine that can never apply its staged change is worse than a61	// pod that has to restart.62	drainDeadline = 5 * time.Minute63)6465// evictablePods returns the pods a drain actually has to move. It66// excludes DaemonSet pods, because the DaemonSet controller ignores67// the cordon and would just recreate them (this operator is itself68// a DaemonSet pod). It also excludes mirror pods, because the69// kubelet recreates those from disk, and pods that have already run70// to completion.71//72// A completed pod also stops counting as a claim holder, and that is73// safe for its DRA driver. The kubelet unprepares a pod's claims after74// its containers stop and before it reports a terminal phase, and its75// status manager holds the phase back while the DRA manager still76// holds a claim for the pod. So a pod that the listing shows as77// Succeeded or Failed needs no unprepare call from its driver.78func evictablePods(pods []kubernetes.Pod) []kubernetes.Pod {79	var evictable []kubernetes.Pod80	for _, p := range pods {81		if p.Completed() {82			continue83		}84		if _, ok := p.Metadata.Annotations[mirrorPodAnnotation]; ok {85			continue86		}87		if p.IsDaemon() {88			continue89		}90		evictable = append(evictable, p)91	}92	return evictable93}9495// draPluginPaths are the kubelet directories that a DRA driver's96// pod mounts from the host. The kubelet discovers a driver by97// watching plugins_registry for the driver's registration socket,98// and it makes its prepare and unprepare calls on a second socket,99// which the driver serves from its own directory under plugins. A100// pod that mounts neither of these cannot be serving a driver.101var draPluginPaths = []string{102	"/var/lib/kubelet/plugins_registry",103	"/var/lib/kubelet/plugins",104}105106// servesDRAPlugin reports whether the pod serves a DRA driver on107// this node. The test reads the pod's own hostPath mounts, which the108// drain already has in hand, and it needs no read of the node's109// ResourceSlices. That second route does not reach a pod anyway:110// a slice names its driver, and nothing in the API links a driver111// name back to the pod that serves it.112//113// The test names no driver, so it also matches pods that mount these114// directories for another reason: a CSI driver registers through the115// same directory, and so would a pod that mounts it to look at it.116// The cost of a wrong match is ordering alone. Such a pod leaves in117// the second phase, a few seconds later than it could have.118//119// liken's own driver, liken.sh, is this operator. It matches the120// test and never reaches it, because it runs as a DaemonSet pod and121// evictablePods drops those first. So nothing ever sequences against122// liken's driver, which is correct: the kubelet keeps calling it123// until the machine reboots.124func servesDRAPlugin(p kubernetes.Pod) bool {125	for _, v := range p.Spec.Volumes {126		if v.HostPath == nil {127			continue128		}129		for _, path := range draPluginPaths {130			if v.HostPath.Path == path || strings.HasPrefix(v.HostPath.Path, path+"/") {131				return true132			}133		}134	}135	return false136}137138// holdsResourceClaim reports whether the pod consumes DRA devices.139// spec.resourceClaims is the whole signal. A pod that names a claim140// directly and a pod that gets one from a template both carry an141// entry there, and the kubelet calls NodeUnprepareResources for142// either one. What the entry points at does not change the order the143// pods leave in, so the drain reads no further.144func holdsResourceClaim(p kubernetes.Pod) bool {145	return len(p.Spec.ResourceClaims) > 0146}147148// drainOrder returns the pods to ask to leave on this pass, so that149// a claim holder always leaves before the driver that prepared its150// claim.151//152// The kubelet calls NodeUnprepareResources while a pod terminates,153// on the driver's own socket. A driver whose pod is already gone154// answers nothing, and the kubelet retries the unprepare with no155// bound, so the pod never finishes terminating. The drain then sits156// until drainDeadline and the machine reboots with the pod still on157// it, which is the outcome the drain exists to avoid.158//159// The first phase asks every other pod to leave, claim holders160// included. The second phase asks the driver pods, and it starts161// only on a pass where no claim holder is left on the node. The162// Eviction API returns as soon as the API server accepts the163// request, long before the pod is gone, so the pod listing is what164// reports the termination: a terminating pod stays in the listing165// until the kubelet finishes with it. The node's pod watch wakes a pass166// when a pod leaves, and each pass runs this split again, so the wait167// blocks nothing, and drainDeadline bounds it. Past the deadline decideDrainStep168// clears the drain and the reboot proceeds.169//170// A driver pod that holds a claim of its own belongs in the second171// phase, and the driver test runs first to put it there. A device172// operator claims its raw hardware from liken.sh, whose driver is173// this operator, which no drain evicts. So a driver pod's own claim174// has no pod to wait for, and the first phase would only make the175// pod wait for itself.176//177// A pod that is already terminating is not asked again, because an178// earlier pass asked it and the API server accepted. This holds for a179// driver pod too. A terminating claim holder still counts as a holder180// until it completes or leaves the listing, because the kubelet181// unprepares its claim while it terminates, and the driver must stay182// until that call is done.183func drainOrder(evictable []kubernetes.Pod) []kubernetes.Pod {184	var first, drivers []kubernetes.Pod185	holders := 0186	for _, p := range evictable {187		if servesDRAPlugin(p) {188			if !p.Terminating() {189				drivers = append(drivers, p)190			}191			continue192		}193		if holdsResourceClaim(p) {194			holders++195		}196		if !p.Terminating() {197			first = append(first, p)198		}199	}200	if holders > 0 {201		return first202	}203	return append(first, drivers...)204}205206// A drainStep is one pass's worth of drain: the Node patch to apply207// (nil when the cordon and deadline record are already in place),208// the pods to ask to leave on this pass, how many pods still have to209// move, and whether the node is clear, meaning nothing is left210// delaying the reboot. The two pod counts differ while the drain211// holds a driver back for its claim holders.212type drainStep struct {213	patch     []byte214	evict     []kubernetes.Pod215	remaining int216	clear     bool217218	// deadline is when the drain stops waiting: draining-since plus219	// drainDeadline. The annotation holds whole seconds, so the220	// deadline comes from the parsed annotation, not from the time in221	// memory.222	deadline time.Time223}224225// decideDrainStep is the drain's decision for one pass. The226// deadline runs from the draining-since annotation. A node without227// this annotation gets it recorded now, along with the cordon, if228// the node is not already unschedulable. The pass that reaches the229// deadline lets the reboot proceed, the same rule Cluster API uses for230// its drain timeout, so the pass that the deadline's own wake starts231// does not hold.232func decideDrainStep(node *nodeObject, pods []kubernetes.Pod, now time.Time) drainStep {233	var step drainStep234235	since, err := time.Parse(time.RFC3339, node.Metadata.Annotations[drainingSinceAnnotation])236	if !node.Spec.Unschedulable || err != nil {237		annotations := map[string]string{drainingSinceAnnotation: now.Format(time.RFC3339)}238		patch := map[string]any{"metadata": map[string]any{"annotations": annotations}}239		if !node.Spec.Unschedulable {240			annotations[cordonedAnnotation] = "true"241			patch["spec"] = map[string]any{"unschedulable": true}242		}243		step.patch, _ = json.Marshal(patch)244		since, _ = time.Parse(time.RFC3339, now.Format(time.RFC3339))245	}246247	step.deadline = since.Add(drainDeadline)248	if !now.Before(step.deadline) {249		step.clear = true // the reboot proceeds; whatever remains stays running through it250		return step251	}252	evictable := evictablePods(pods)253	step.remaining = len(evictable)254	step.evict = drainOrder(evictable)255	step.clear = step.remaining == 0256	return step257}258259// gateThroughDrain intercepts a convergence that requires a reboot260// and releases it only once this machine's node is clear. It cordons261// the node and evicts pods. Until nothing evictable remains, or the262// deadline passes, it holds the reboot and reports the drain's263// progress on the same condition. The cordon posts one Event about264// the Machine, because it is the start of the drain and the Node265// carries no record of who cordoned it or why.266//267// Three things wake the pass that looks again. The node's pod watch268// wakes it when a pod leaves, the budget watch wakes it when a budget269// allows an eviction it refused (waits.go), and the outcome asks for a270// pass at the deadline.271func gateThroughDrain(r *reader, node *nodeObject, conv convergence, now time.Time, notes machineEvents, out *passOutcome) convergence {272	r.watchBudgets()273	pods, err := r.nodePods(node.Metadata.Name)274	if err != nil {275		fmt.Printf("listing pods for the drain: %v\n", err)276		return holdForDrain(conv, "listing this node's pods failed; retrying")277	}278	step := decideDrainStep(node, pods, now)279	if step.patch != nil {280		if err := kubernetes.PatchJSON(r.client, nodesPath+"/"+node.Metadata.Name, step.patch); err != nil {281			fmt.Printf("cordoning %s: %v\n", node.Metadata.Name, err)282			return holdForDrain(conv, "cordoning this node failed; retrying")283		}284		fmt.Printf("cordoned %s ahead of its reboot\n", node.Metadata.Name)285		notes.normal(reasonCordoned, fmt.Sprintf("cordoned the Node %s ahead of the reboot; %d pods to move", node.Metadata.Name, step.remaining))286	}287	evict(r, step.evict, now, out)288	if step.clear {289		return conv290	}291	out.wakeBy(step.deadline)292	return holdForDrain(conv, fmt.Sprintf("draining this node ahead of the reboot; %d pods still to move", step.remaining))293}294295// evict asks each pod to leave, and sorts the answers the way296// `kubectl drain` does (k8s.io/kubectl/pkg/drain). A 404 means the pod297// is gone already, which happens when the copy is a moment behind. A298// 429 means a PodDisruptionBudget refused it, and the budget is working299// as intended: when the answer states a wait, the pass asks for a wake300// then, and otherwise the budget watch wakes the pass when a budget301// allows an eviction. Neither is a failure of the pass. Every other302// refusal, such as a 403, a 5xx, or no answer, reaches the outcome303// through the pass's observer, for milestone 76's retry.304func evict(r *reader, pods []kubernetes.Pod, now time.Time, out *passOutcome) {305	if len(pods) == 0 {306		return307	}308	client := r.client.WithObserver(func(answer apiclient.Outcome) {309		if answer.Status == http.StatusNotFound || answer.Status == http.StatusTooManyRequests {310			return311		}312		out.observe(answer)313	})314	for _, p := range pods {315		err := kubernetes.EvictPod(client, p)316		if err == nil {317			continue318		}319		fmt.Printf("evicting %s/%s: %v\n", p.Metadata.Namespace, p.Metadata.Name, err)320		if seconds := apiclient.RetryAfterSeconds(err); seconds > 0 {321			out.wakeBy(now.Add(time.Duration(seconds) * time.Second))322		}323	}324}325326// holdForDrain keeps reporting the convergence on its own condition327// while the drain works. It keeps the same condition type, withholds328// the reboot, and sets the reason to Draining.329func holdForDrain(conv convergence, message string) convergence {330	conv.requestReboot = false331	conv.condition = api.Condition{332		Type: conv.condition.Type, Status: api.ConditionFalse, Reason: "Draining", Message: message,333	}334	return conv335}336337// decideUncordon reports whether a node carries a cordon that this338// operator applied. It returns true only for the operator's own339// cordon: the annotation records that the operator set it, and a340// cordon without the annotation belongs to a person.341func decideUncordon(node *nodeObject) bool {342	return node.Spec.Unschedulable && node.Metadata.Annotations[cordonedAnnotation] == "true"343}344345// uncordonPatch releases the node back to the scheduler and removes346// the drain's recorded state. In a merge patch, a null value deletes347// the key.348func uncordonPatch() []byte {349	patch, _ := json.Marshal(map[string]any{350		"spec": map[string]any{"unschedulable": false},351		"metadata": map[string]any{"annotations": map[string]any{352			cordonedAnnotation:      nil,353			drainingSinceAnnotation: nil,354		}},355	})356	return patch357}
machine-operator/draplugin.go 87.3%
1package main23// The DRA driver's kubelet half: the plugin the kubelet calls before4// it starts a pod that holds a claim.5//6// The wire arrangement is the opposite of what the word "plugin"7// suggests. The driver runs two gRPC servers, and the kubelet is8// the only client of both. The first service is registration: the9// kubelet watches a well-known directory for sockets, dials each10// one, and calls GetInfo to read what is there. The second service11// is the DRA plugin API itself, on a socket of the driver's own,12// whose path GetInfo announces. Both sockets live under the13// kubelet's own state directory. Unix sockets are the entire14// transport: nothing here touches the network, and file permissions15// on the kubelet's directories provide the authentication.16//17// The prepare protocol deliberately tells the driver almost18// nothing: a claim's namespace, name, and UID. What was allocated19// lives on the claim's status in the API server, so the driver reads20// that back (kubernetes/resourceclaims.go), walks its own inventory21// again, and hands the claim the published device's nodes. This22// follows the same rule liken applies everywhere: a call only23// signals that something happened, and the driver acts on the24// shared, durable record, not on data carried in the call itself.25//26// Failures are per-claim strings inside the response, not gRPC27// errors. The kubelet holds the affected pod in ContainerCreating28// and retries. This is the right behavior for every temporary29// cause, such as a device still being enumerated or an API hiccup,30// and it is also the honest behavior for a permanent one, such as31// hardware that has been removed. The pod waits, visibly, for32// hardware the cluster said it could have, and a describe of the33// pod shows why.3435import (36	"context"37	"errors"38	"fmt"39	"net"40	"os"41	"path/filepath"42	"strings"43	"sync"4445	"google.golang.org/grpc"46	healthv1alpha1 "k8s.io/kubelet/pkg/apis/dra-health/v1alpha1"47	drav1 "k8s.io/kubelet/pkg/apis/dra/v1"48	regv1 "k8s.io/kubelet/pkg/apis/pluginregistration/v1"4950	"github.com/liken-sh/liken/kubernetes/apiclient"51	"github.com/liken-sh/liken/liken/hardware"52	"github.com/liken-sh/liken/liken/kubernetes"53	"github.com/liken-sh/liken/liken/machine"54)5556// The kubelet's plugin directories. The registry is where the57// kubelet discovers plugins. The plugin's own directory holds the58// socket that does the real work. These are variables so the tests59// can substitute them.60var (61	draRegistryDir = "/var/lib/kubelet/plugins_registry"62	draPluginDir   = "/var/lib/kubelet/plugins/liken.sh"63)6465// draPlugin answers the kubelet's DRA calls. The API client is its66// only state. It derives everything else again on each call, from67// the claim and from sysfs.68type draPlugin struct {69	drav1.UnimplementedDRAPluginServer70	client *apiclient.Client71}7273// draRegistrar answers the kubelet's plugin-watcher handshake.74type draRegistrar struct {75	regv1.UnimplementedRegistrationServer76	endpoint string77}7879func (r *draRegistrar) GetInfo(ctx context.Context, req *regv1.InfoRequest) (*regv1.PluginInfo, error) {80	return &regv1.PluginInfo{81		Type:     regv1.DRAPlugin,82		Name:     kubernetes.DriverName,83		Endpoint: r.endpoint,84		// These strings name gRPC services, not semantic versions. The85		// kubelet picks the newest version that it also supports.86		// liken serves exactly the v1 API. The Kubernetes version87		// liken ships is never older than its own OS components, so88		// there is no version gap to bridge with a beta shim.89		SupportedVersions: []string{drav1.DRAPluginService},90	}, nil91}9293func (r *draRegistrar) NotifyRegistrationStatus(ctx context.Context, status *regv1.RegistrationStatus) (*regv1.RegistrationStatusResponse, error) {94	if !status.PluginRegistered {95		fmt.Fprintf(os.Stderr, "dra: the kubelet rejected the plugin registration: %s\n", status.Error)96	}97	return &regv1.RegistrationStatusResponse{}, nil98}99100// serveDRAPlugin starts both servers and blocks until the context101// ends or a server fails. The order matters: the plugin socket must102// already be listening before the registration socket exists,103// because the kubelet dials the announced endpoint as soon as it104// sees the registration. The function removes stale sockets from a105// previous operator first. A bind to an orphaned socket file fails106// even when nothing is listening on it.107func serveDRAPlugin(ctx context.Context, client *apiclient.Client) error {108	if err := os.MkdirAll(draPluginDir, 0o755); err != nil {109		return err110	}111	pluginSocket := filepath.Join(draPluginDir, "dra.sock")112	_ = os.Remove(pluginSocket)113	pluginListener, err := net.Listen("unix", pluginSocket)114	if err != nil {115		return fmt.Errorf("the plugin socket: %w", err)116	}117	pluginServer := grpc.NewServer()118	drav1.RegisterDRAPluginServer(pluginServer, &draPlugin{client: client})119	healthv1alpha1.RegisterDRAResourceHealthServer(pluginServer, &draHealth{})120121	registrationSocket := filepath.Join(draRegistryDir, kubernetes.DriverName+"-reg.sock")122	_ = os.Remove(registrationSocket)123	registrationListener, err := net.Listen("unix", registrationSocket)124	if err != nil {125		return fmt.Errorf("the registration socket: %w", err)126	}127	registrationServer := grpc.NewServer()128	regv1.RegisterRegistrationServer(registrationServer, &draRegistrar{endpoint: pluginSocket})129130	errs := make(chan error, 2)131	go func() { errs <- pluginServer.Serve(pluginListener) }()132	go func() { errs <- registrationServer.Serve(registrationListener) }()133	select {134	case <-ctx.Done():135		registrationServer.Stop()136		pluginServer.Stop()137		return nil138	case err := <-errs:139		return err140	}141}142143// NodePrepareResources prepares every claim in the request. The144// response must carry one entry for each claim, because the kubelet145// treats a missing entry as a failure to retry. Each entry stands146// on its own, so trouble with one claim never blocks another147// claim's pod.148func (p *draPlugin) NodePrepareResources(ctx context.Context, req *drav1.NodePrepareResourcesRequest) (*drav1.NodePrepareResourcesResponse, error) {149	resp := &drav1.NodePrepareResourcesResponse{Claims: map[string]*drav1.NodePrepareResourceResponse{}}150	for _, claim := range req.Claims {151		resp.Claims[claim.Uid] = p.prepareClaim(claim)152	}153	return resp, nil154}155156// resolveAllocated maps an allocated device name to the published set157// that the name denotes. A bare name is the primary. Any other name158// must equal a bare name plus a published suffix, which the policy159// recomputes here from a fresh walk, so the answer is always what160// this claim would receive now. The bare names are tried first,161// exactly: one device's bare name can begin with another's, so a162// prefix alone identifies nothing.163//164// A device that the protection withholds answers errWithheld, and a165// name that no present device publishes answers errNotPublished.166func resolveAllocated(name string, sysRoot string, byName map[string]hardware.Device, serio []machine.SerioAttachment, platform protection) (publishedDevice, error) {167	if device, ok := byName[name]; ok {168		delivery := claimDelivery(sysRoot, device)169		if platform.withholds(delivery) {170			return publishedDevice{}, errWithheld171		}172		for _, p := range publishFor(device, delivery, serio) {173			if p.Suffix == "" {174				return p, nil175			}176		}177		return publishedDevice{}, errNotPublished178	}179	for bare, device := range byName {180		if !strings.HasPrefix(name, bare+"-") {181			continue182		}183		delivery := claimDelivery(sysRoot, device)184		for _, p := range publishFor(device, delivery, serio) {185			if bare+p.Suffix != name {186				continue187			}188			if platform.withholds(delivery) {189				return publishedDevice{}, errWithheld190			}191			return p, nil192		}193	}194	return publishedDevice{}, errNotPublished195}196197// errNotPublished is an allocated device name that no present device198// publishes.199var errNotPublished = errors.New("it is not in this machine's inventory now")200201func (p *draPlugin) prepareClaim(claim *drav1.Claim) *drav1.NodePrepareResourceResponse {202	fail := func(format string, args ...any) *drav1.NodePrepareResourceResponse {203		message := fmt.Sprintf(format, args...)204		fmt.Fprintf(os.Stderr, "dra: preparing claim %s/%s: %s\n", claim.Namespace, claim.Name, message)205		return &drav1.NodePrepareResourceResponse{Error: message}206	}207208	allocated, err := kubernetes.GetResourceClaim(p.client, claim.Namespace, claim.Name)209	if err != nil {210		return fail("reading the claim: %v", err)211	}212	if allocated.Metadata.UID != claim.Uid {213		// The named claim was deleted and recreated after the214		// kubelet asked. Whatever this new claim holds, it is not215		// the grant this pod was scheduled against.216		return fail("the claim's UID changed (%s became %s)", claim.Uid, allocated.Metadata.UID)217	}218	if allocated.Status.Allocation == nil {219		return fail("the claim has no allocation yet")220	}221222	// One walk maps allocated device names back to hardware. It uses223	// the same walk and the same naming that published the224	// inventory, so the two can never disagree about which device a225	// name identifies.226	byName := map[string]hardware.Device{}227	for _, d := range hardware.DiscoverInventory(draSysfsRoot, draNaming()) {228		byName[deviceName(d)] = d229	}230231	var specDevices []cdiDevice232	var devices []*drav1.Device233	for _, result := range allocated.Status.Allocation.Devices.Results {234		if result.Driver != kubernetes.DriverName {235			// This is another driver's allocation in the same claim.236			// That driver's own plugin prepares it.237			continue238		}239		published, err := resolveAllocated(result.Device, draSysfsRoot, byName, declaredSerio(), currentProtection())240		if err != nil {241			return fail("allocated device %s: %v", result.Device, err)242		}243		nodes := deviceNodes(published.Nodes)244		name := claim.Uid + "-" + result.Device245		specDevices = append(specDevices, cdiDevice{246			Name:           name,247			ContainerEdits: cdiEdits{DeviceNodes: nodes},248		})249		devices = append(devices, &drav1.Device{250			PoolName:     result.Pool,251			DeviceName:   result.Device,252			RequestNames: []string{result.Request},253			CdiDeviceIds: []string{cdiKind + "=" + name},254		})255	}256	if len(specDevices) > 0 {257		if err := writeCDISpec(claim.Uid, specDevices); err != nil {258			return fail("writing the CDI spec: %v", err)259		}260	}261	return &drav1.NodePrepareResourceResponse{Devices: devices}262}263264// draHealth is the device-health stream. The driver keeps it open265// and sends nothing on it. The service is optional in the DRA266// protocol, but the kubelet does not treat it that way in practice:267// an unregistered service produces an Unimplemented error and a268// retry every few seconds, forever, in the k3s log. Accepting the269// stream and reporting nothing states the truth, because liken270// makes no health claims about devices yet. This is the same stream271// that real health reports will use once the uevent watcher starts272// feeding it (see the plan doc's device-health note).273type draHealth struct {274	healthv1alpha1.UnimplementedDRAResourceHealthServer275}276277func (h *draHealth) NodeWatchResources(req *healthv1alpha1.NodeWatchResourcesRequest, stream grpc.ServerStreamingServer[healthv1alpha1.NodeWatchResourcesResponse]) error {278	<-stream.Context().Done()279	return nil280}281282// NodeUnprepareResources gives each claim's hardware back to the283// machine and removes the claim's CDI spec. As with prepare, every284// claim gets an answer, and failures stay specific to each claim.285//286// The repair runs before the removal, because the spec file is the287// only record of what the claim held (rebinding.go explains the288// repair and why it never fails the call). One sysfs walk serves289// every claim in the request, and the walk runs only when a claim has290// a spec file, so an unprepare the kubelet repeats costs nothing.291func (p *draPlugin) NodeUnprepareResources(ctx context.Context, req *drav1.NodeUnprepareResourcesRequest) (*drav1.NodeUnprepareResourcesResponse, error) {292	devices := sync.OnceValue(func() map[string]hardware.Device {293		byName := map[string]hardware.Device{}294		for _, d := range hardware.DiscoverInventory(draSysfsRoot, draNaming()) {295			byName[deviceName(d)] = d296		}297		return byName298	})299	resp := &drav1.NodeUnprepareResourcesResponse{Claims: map[string]*drav1.NodeUnprepareResourceResponse{}}300	for _, claim := range req.Claims {301		rebindClaimDevices(claim.Uid, devices)302		if err := removeCDISpec(claim.Uid); err != nil {303			resp.Claims[claim.Uid] = &drav1.NodeUnprepareResourceResponse{Error: err.Error()}304			continue305		}306		resp.Claims[claim.Uid] = &drav1.NodeUnprepareResourceResponse{}307	}308	return resp, nil309}
machine-operator/endpoint.go 100.0%
1package main23import (4	"slices"56	"github.com/liken-sh/liken/liken/cluster"7)89// localAPIEndpoint returns the machine's own path to the API10// server, chosen instead of the service VIP that the pod environment11// offers. The VIP works through iptables NAT: each connection is12// pinned to one API server. When that server's machine dies13// silently, everything pinned to it stalls on timeouts, for long14// enough that healthy machines' heartbeats lapse and read as Lost.15// This pod runs on the host's network, so localhost reaches the16// machine, and the machine always has a better path. A leader runs17// an API server of its own on 6443, and a follower runs k3s's agent18// load balancer on 6444, which checks the health of every server and19// fails over between them. These are the same endpoints the20// machine's own kubelet uses, so the operator's view of the API can21// never be worse than the kubelet's. A machine with no cluster22// document falls back to the environment's endpoint ("").23func localAPIEndpoint(clusterDoc *cluster.Cluster, name string) string {24	if clusterDoc == nil {25		return ""26	}27	if slices.Contains(clusterDoc.Spec.Leaders, name) {28		return "https://127.0.0.1:6443"29	}30	return "https://127.0.0.1:6444"31}
machine-operator/events.go 100.0%
1package main23// The Events the machine operator posts about its own Machine.4// `kubectl describe machine` lists them under the status, so a person5// reads what happened to a machine in the last hour with no log to6// open. kubernetes/events writes them, and root plan 78 gives the rule7// for a condition, an Event, and a log line.8//9// A Machine is cluster-scoped, so its Events are in the namespace10// default. `kubectl describe` finds them there, and `kubectl events11// --for machine/<name>` finds them only with -n default or -A.12//13//   - Each condition transition posts one Event, with the condition's14//     reason, after the status write lands (postStatusEvents). The15//     transitions of Ready are the transitions of the phase, because16//     Ready's reason is the phase word when it is False. A spec that17//     the operator refuses to stage posts StagingRejected, and a18//     release or a spec that fell back at boot posts RejectedLastBoot.19//     A reboot requested to apply a document posts RebootRequested.20//     RebootApproved posts nothing here: the cluster operator writes21//     it and posts its own Events.22//   - A new kernel crash record and a new refused boot each post one23//     Event, because both arrive as status fields and not as24//     conditions.25//   - The operator's own actions on the Node post one Event each:26//     the cordon before a drain and the uncordon after the reboot.27//   - The creation of the Machine from the boot manifest posts one28//     Event.29//30// Nothing that repeats on each pass posts an Event: a refused31// eviction, a sysctl written back, or a heartbeat renewal.3233import (34	"fmt"35	"time"3637	"github.com/liken-sh/liken/kubernetes/conditions"38	"github.com/liken-sh/liken/kubernetes/events"39	"github.com/liken-sh/liken/liken/api"40	"github.com/liken-sh/liken/liken/machine"41)4243// The reasons of the Events that are not condition transitions. A44// transition's Event takes the condition's own reason.45const (46	reasonMachineJoined = "MachineJoined"47	reasonKernelCrashed = "KernelCrashed"48	reasonBootRefused   = "BootRefused"49	reasonCordoned      = "Cordoned"50	reasonUncordoned    = "Uncordoned"5152	// reasonBackstopRepaired marks a write by a pass that only the53	// backstop started, which names a wake the code does not send54	// (backstop.go).55	reasonBackstopRepaired = "BackstopRepaired"56)5758// machineReference answers the object reference of a Machine, for an59// Event about it.60func machineReference(m *machine.Machine) events.ObjectReference {61	return events.ObjectReference{62		APIVersion: api.APIVersion, Kind: machineKind,63		Name: m.Metadata.Name, UID: m.Metadata.UID,64	}65}6667// machineEvents posts the Events about one Machine. The zero value68// posts nothing, because a nil recorder posts nothing, so a test of a69// decision that posts no Event passes machineEvents{}.70type machineEvents struct {71	recorder *events.Recorder72	machine  events.ObjectReference73}7475func (e machineEvents) normal(reason, message string) {76	e.recorder.Normal(e.machine, reason, message)77}7879func (e machineEvents) warning(reason, message string) {80	e.recorder.Warning(e.machine, reason, message)81}8283// postStatusEvents posts the Events of one status write that landed:84// one for each condition that transitioned from stored, and one for85// each new crash or refused boot that the facts carried in.86func postStatusEvents(e machineEvents, stored, written *machine.MachineStatus) {87	for _, c := range api.Transitions(stored.Conditions, written.Conditions) {88		if c.Type == machine.RebootApprovedCondition {89			continue90		}91		e.recorder.Transition(e.machine, transitionEvent(c), badStatus(c))92	}93	if crash := written.LastCrash; crash != nil && crash.Time != nil && !sameCrash(stored.LastCrash, crash) {94		e.recorder.Warning(e.machine, reasonKernelCrashed, fmt.Sprintf(95			"the kernel recorded a %s at %s: %s; status.lastCrash names the records",96			crash.Reason, crash.Time.UTC().Format(time.RFC3339), crash.Message))97	}98	if stop := written.LastFailStop; stop != nil && (stored.LastFailStop == nil || !stored.LastFailStop.Time.Equal(stop.Time)) {99		e.recorder.Warning(e.machine, reasonBootRefused, fmt.Sprintf(100			"init refused the boot at %s and powered the machine off: %s",101			stop.Time.UTC().Format(time.RFC3339), stop.Reason))102	}103}104105// sameCrash reports whether the stored status already names the crash106// at the same time.107func sameCrash(stored, crash *machine.CrashStatus) bool {108	return stored != nil && stored.Time != nil && stored.Time.Equal(*crash.Time)109}110111// transitionEvent answers the condition an Event reports. Five112// convergence conditions share one set of reasons, and Converged113// alone says nothing about which document converged, so the message114// starts with the condition's type and status.115func transitionEvent(c api.Condition) api.Condition {116	prefix := c.Type + " is " + string(c.Status)117	if c.Message == "" {118		c.Message = prefix119	} else {120		c.Message = prefix + ": " + c.Message121	}122	return c123}124125// badStatus answers the status of a condition that needs a person,126// for events.Recorder.Transition. A condition needs a person when the127// phase it argues for is Blocked, Degraded, or Unknown. A False that128// argues for Downloading, Updating, UpdatePending, or Booting is a129// change in progress, which needs no one. Ready's reason is the phase130// word.131func badStatus(c api.Condition) conditions.Status {132	phase := conditionPhase(c)133	if c.Type == "Ready" && c.Status != api.ConditionTrue {134		phase = api.Phase(c.Reason)135	}136	switch phase {137	case api.PhaseBlocked, api.PhaseDegraded, api.PhaseUnknown:138		return c.Status139	}140	return ""141}
machine-operator/fetch.go 98.5%
1package main23// The release fetcher runs one background download at a time and4// never blocks the reconcile loop.5//6// One download at a time is a rule about the slot, not about the7// network. The download writes the inactive slot, so a second writer8// for another release would interleave its files with the first one's.9// When the ask changes while a download runs, Ensure cancels that10// download and starts the new one only after the old goroutine11// returns. The cancelled download removes its partial file on the12// way out, and the next run verifies whatever already landed.13//14// Reconcile passes never download anything themselves. They ask the15// fetcher instead. Ensure records what the machine currently needs16// (an ask: version, digest, source, and destination slot), starts17// the download on its own goroutine if one is not already running,18// and returns immediately with the current state. The pass that19// starts a download and the pass that finds it verified are20// different passes, minutes apart, and every pass in between keeps21// the heartbeat fresh. The lease must never wait on a socket. The22// download wakes the loop when it ends (fetcher.wake), so the pass23// that reads the result runs at once.24//25// Failure comes in two kinds, and the distinction matters26// throughout this file. A transient failure means the server is27// down or the network dropped. The fetcher retries a transient28// failure forever, after a backoff that starts at ten seconds and29// doubles up to two minutes (fetchRetryLimit). A release server that30// fails usually stays down for minutes, and every pass would31// otherwise download again the moment it saw the failure. A corrupt32// failure means the bytes do not match the digests the catalog33// promised. The fetcher holds a corrupt failure, without retrying,34// until the ask itself changes, because refetching cannot change what35// the server publishes.36// Corruption is the reason the chain of checks in download.go37// exists: the API names the document, the document names the38// artifacts, and a mismatch anywhere means someone's bytes are wrong. The fetcher39// abandons a corrupt release rather than patching it. The recovery40// is to publish a corrected release under a new version and point41// the catalog at it.4243import (44	"context"45	"errors"46	"fmt"47	"math/rand/v2"48	"net/http"49	"sync"50	"time"51)5253// A fetchAsk is one reconcile decision's request: fetch this54// version, confirmed by this digest, from this source, onto this55// slot. Asks compare by value, so a changed catalog digest produces56// a different ask, and a different ask is what clears a corruption57// hold.58type fetchAsk struct {59	version       string60	digest        string // the catalog's "sha256:<hex>" over release.yaml61	source        string // the base URL releases are served under62	slot          string // "A" or "B", for the humans reading conditions63	slotDir       string // the slot's mounted filesystem64	activeSlotDir string // the running slot's, which lends its layer65}6667type fetchState string6869const (70	fetchIdle     fetchState = "Idle"     // nothing started yet71	fetchRunning  fetchState = "Running"  // a goroutine is downloading72	fetchVerified fetchState = "Verified" // every artifact on the slot checks out73	fetchFailed   fetchState = "Failed"   // transient; retried at retryAt74	fetchRejected fetchState = "Rejected" // corrupt; held until the ask changes75)7677// A fetchSnapshot is what a reconcile pass sees: the ask the state78// describes, the state, a human sentence for condition messages, and,79// for a Failed state, when the retry starts.80type fetchSnapshot struct {81	ask     fetchAsk82	state   fetchState83	detail  string84	retryAt time.Time85}8687const (88	fetchFirstRetry = 10 * time.Second89	fetchRetryLimit = 2 * time.Minute90)9192type fetcher struct {93	mu   sync.Mutex94	snap fetchSnapshot95	busy bool9697	// cancel ends the running download. It is nil while no download98	// runs.99	cancel context.CancelCauseFunc100101	// client sends the downloads. A nil client is102	// http.DefaultClient; a test gives one that reaches its server.103	client *http.Client104105	// retryDelay is the backoff after the ask's last transient106	// failure, and zero before the first. A new ask resets it, and so107	// does a verified download.108	retryDelay time.Duration109110	// wake wakes the reconcile loop when a download ends, whatever111	// the end: verified, failed, rejected, or stopped because the ask112	// changed. The pass it starts reads the result, or starts the113	// download the new ask needs. A failed download is safe to wake on,114	// because its backoff (retryAt) gives the next attempt its time; a115	// download that fails in milliseconds, such as one that meets a116	// 404, would otherwise start again at once. Nil wakes nothing.117	wake func()118119	// jitter answers a number in [0, 1), for the tenth of the delay120	// that each retry adds at random, so a fleet that failed together121	// does not retry together. Nil means math/rand.122	jitter func() float64123124	// The fetcher's running totals, kept beside the snapshot because125	// they outlive every ask. The snapshot describes one release; a126	// total describes what this machine has spent on downloads since127	// the operator started, which is what a graph of staging progress128	// and of a download that keeps failing is drawn from129	// (metrics.go).130	downloaded int64131	failures   int132}133134// DownloadedBytes is how many release artifact bytes this machine135// has downloaded and verified onto a slot. Bytes count when the136// artifact lands, so a torn download adds nothing until its retry137// completes the file.138func (f *fetcher) DownloadedBytes() int64 {139	f.mu.Lock()140	defer f.mu.Unlock()141	return f.downloaded142}143144// DownloadFailures is how many downloads ended without a complete145// slot, whether the network dropped or the bytes did not match the146// catalog's digests.147func (f *fetcher) DownloadFailures() int {148	f.mu.Lock()149	defer f.mu.Unlock()150	return f.failures151}152153// errCorrupt distinguishes verification failures from transport154// failures. The code wraps it into any error that should stop the155// retries, and an error carrying it holds the fetcher at Rejected.156var errCorrupt = errors.New("the bytes do not match the release's digests")157158// errLayer marks a failure in the machine's own deployment layer.159// It holds the fetcher the same way corruption does, because no160// retry can repair the slot the machine is running on. But the161// remedy is local: repair or reinstall this machine. So the162// condition must not send a person off to republish a release that163// was never the problem.164var errLayer = errors.New("this machine's deployment layer is unusable")165166// errSuperseded is the cause of a download that Ensure cancelled167// because the ask changed. It is not a failure of the download.168var errSuperseded = errors.New("the machine no longer asks for this release")169170// Ensure records the ask, starts a download when one is needed and171// none is running, and returns the current state. It never blocks:172// the heaviest thing it does is start a goroutine.173func (f *fetcher) Ensure(ask fetchAsk) fetchSnapshot {174	f.mu.Lock()175	defer f.mu.Unlock()176177	if f.snap.ask != ask {178		// This is a different release, digest, or destination.179		// Everything known so far, including a Rejected hold, was180		// about the old ask. A changed catalog is exactly the event181		// that should clear the hold, so the state resets with the182		// ask.183		f.snap = fetchSnapshot{ask: ask, state: fetchIdle, detail: "waiting to start"}184		f.retryDelay = 0185		if f.busy {186			// The running download is for the old ask, and it187			// writes the same slot this one will. It stops at188			// once, and this ask waits until its goroutine189			// returns, so that two writers never overlap.190			f.cancel(errSuperseded)191			f.snap.detail = "waiting for the previous download to stop"192		}193	}194	if f.busy || f.snap.state == fetchVerified || f.snap.state == fetchRejected {195		return f.snap196	}197	if f.snap.state == fetchFailed && time.Now().Before(f.snap.retryAt) {198		return f.snap199	}200201	// A restart after a transient failure keeps the failure's reason,202	// so a condition written while the retry runs still says what the203	// last attempt met.204	detail := "starting"205	if f.snap.state == fetchFailed {206		detail = "retrying after: " + f.snap.detail207	}208	ctx, cancel := context.WithCancelCause(context.Background())209	f.busy, f.cancel = true, cancel210	f.snap.state = fetchRunning211	f.snap.detail = detail212	go f.run(ctx, ask)213	return f.snap214}215216// run is the goroutine. It runs the fetch, then records the217// verdict. If the ask changed while the fetch ran, the verdict218// describes a release the machine no longer needs, so the function219// discards it.220func (f *fetcher) run(ctx context.Context, ask fetchAsk) {221	client := f.client222	if client == nil {223		client = http.DefaultClient224	}225	fetched, downloaded, err := fetchRelease(ctx, client, ask)226227	// The wake runs after the unlock below, because deferred calls run228	// in reverse, so the pass it starts finds the fetcher idle.229	if f.wake != nil {230		defer f.wake()231	}232	f.mu.Lock()233	defer f.mu.Unlock()234	f.cancel(nil)235	f.busy, f.cancel = false, nil236	// The totals count what this run did, even when the ask has237	// moved on. The bytes reached the slot and the failure happened,238	// whatever the machine wants now. A download that Ensure239	// cancelled did not fail.240	f.downloaded += downloaded241	if err != nil && !errors.Is(context.Cause(ctx), errSuperseded) {242		f.failures++243	}244	// A download that Ensure cancelled describes no verdict, even when245	// the ask has changed back to it since: Ensure reset the state to246	// Idle for the new ask, and the next pass starts the download.247	if f.snap.ask != ask || errors.Is(context.Cause(ctx), errSuperseded) {248		return249	}250	switch {251	case err == nil:252		f.snap.state = fetchVerified253		f.snap.detail = fmt.Sprintf("%d artifacts fetched, the rest already verified in place", fetched)254		f.retryDelay = 0255	case errors.Is(err, errLayer):256		f.snap.state = fetchRejected257		f.snap.detail = err.Error()258	case errors.Is(err, errCorrupt):259		f.snap.state = fetchRejected260		f.snap.detail = fmt.Sprintf("release %s at %s is corrupt (%v); publish a corrected release under a new version", ask.version, ask.source, err)261	default:262		f.snap.state = fetchFailed263		f.snap.detail = err.Error()264		f.retryDelay = grow(f.retryDelay, true, fetchFirstRetry, fetchRetryLimit)265		f.snap.retryAt = time.Now().Add(f.retryDelay + time.Duration(float64(f.retryDelay)*f.random()/10))266	}267}268269func (f *fetcher) random() float64 {270	if f.jitter == nil {271		return rand.Float64()272	}273	return f.jitter()274}
machine-operator/held.go 100.0%
1package main23// The nodes the machine itself holds, which no claim may receive.4//5// The storage test in inventoryDevices keeps the machine's own disks6// out of the slice. Two more nodes belong to the machine in the same7// way, and the board walk reaches both. The console is where init8// writes the boot and where the kernel writes its messages, and a pod9// that held it could read and write over both. init writes the10// system clock into /dev/rtc0 each time the clock is set, and the11// kernel lets one process at a time open an RTC, so a pod that held it12// would make that write fail.13//14// The TPM is the machine's too. Its authorization on most boards is an15// empty password for the owner and lockout hierarchies, so a pod that16// held /dev/tpm0 or /dev/tpmrm0, with no privilege, could set those17// passwords, clear the TPM, extend its PCRs until the next boot, or fill18// its storage, and each of those reaches every later user of the TPM.19// The TPM's nodes are held by their kernel subsystem, whatever their20// numbers are.21//22// The test removes the node, not the device. A board's serial23// controller can carry the console on one port and a UPS on another,24// and the UPS port stays claimable. A device left with no node is25// not published, by the same test that skips a device with nothing26// to deliver.2728import (29	"os"30	"path/filepath"31	"slices"32	"strings"3334	"github.com/liken-sh/liken/liken/hardware"35)3637// machineClock is the RTC that init writes the system clock into38// (init/time.go).39const machineClock = "/dev/rtc0"4041// heldNodes reads the nodes the machine holds now. The kernel lists42// each console it writes to in /sys/class/tty/console/active, by the43// tty's name under /dev.44func heldNodes(sysRoot string) map[string]bool {45	held := map[string]bool{machineClock: true}46	raw, err := os.ReadFile(filepath.Join(sysRoot, "class", "tty", "console", "active"))47	if err != nil {48		return held49	}50	for _, name := range strings.Fields(string(raw)) {51		held["/dev/"+name] = true52	}53	return held54}5556// heldSubsystems are the kernel subsystems whose every node the machine57// holds: the TPM's raw device, and its resource manager.58var heldSubsystems = map[string]bool{"tpm": true, "tpmrm": true}5960// withoutHeld removes the held nodes from one delivery.61func withoutHeld(delivery hardware.Delivery, held map[string]bool) hardware.Delivery {62	delivery.Nodes = slices.DeleteFunc(delivery.Nodes, func(n hardware.DeliveredNode) bool {63		return held[n.Path] || heldSubsystems[n.Subsystem]64	})65	return delivery66}
machine-operator/hosts.go 87.5%
1package main23// Host entries, reconciled live from the Machine spec's4// spec.network.hostEntries, alongside sysctls in the same pass.5//6// Init writes /etc/hosts once, at boot, so the entries hold before7// k3s starts and every boot proves the cold-start order on its own.8// This file is the second writer: the operator reconciles the same9// file on every pass, so a later edit lands within one reconcile10// pass, with no reboot. When another process rewrites the file, the11// inotify watch on it wakes a pass, and the pass writes the entries12// back (machineevents.go). The pass's own write wakes one more pass,13// which finds the file converged and writes nothing. The two writers14// share one renderer, machine.HostsFile, so they can only ever produce15// one shape of the file.1617import (18	"errors"19	"fmt"20	"io/fs"21	"os"22	"path/filepath"23	"strings"2425	"github.com/liken-sh/liken/liken/machine"26)2728// hostsPath is the file this operator reconciles: the host's29// /etc/hosts, reached through the /host/etc hostPath mount30// (manifests/machine-operator.yaml explains why the mount is the31// directory rather than the file). It is a package variable, the32// same pattern init's sysctlDir uses, so a test can point it at a33// tempdir's file instead of a real path under /host.34var hostsPath = "/host/etc/hosts"3536// applyHostEntries reconciles /etc/hosts against the desired entries.37// This is the write-on-divergence rule this milestone applies to38// every live reconciliation: render the desired file with the shared39// renderer, read the actual file, and write only when the bytes40// differ. A converged machine is the common case, and skipping the41// write leaves no false modification signal, an mtime bump or an42// inotify event, for whatever else watches the file. The same rule is43// what gives the file its healing property: a file that an outside44// edit changed differs from the render, so the next pass rewrites it.45//46// A missing file reads as maximally divergent rather than as an47// error, because the first pass on a machine, or a pass right after48// an unrelated file loss, must still converge the file instead of49// giving up. Any other read failure comes back to the caller, the50// same as a write failure does.51//52// The returned entries come from a fresh read of the file after this53// pass, not from the desired list, so the caller's status report is54// observed, not assumed. The pass's outcome records the write, and a55// failure for the retry.56func applyHostEntries(path, hostname string, desired []machine.HostEntry, out *passOutcome) ([]machine.HostEntry, error) {57	entries, err := reconcileHostsFile(path, hostname, desired, out)58	out.fail("writing the host entries", err)59	return entries, err60}6162func reconcileHostsFile(path, hostname string, desired []machine.HostEntry, out *passOutcome) ([]machine.HostEntry, error) {63	render := machine.HostsFile(hostname, desired)6465	actual, err := os.ReadFile(path)66	if err != nil && !errors.Is(err, fs.ErrNotExist) {67		return nil, fmt.Errorf("reading %s: %w", path, err)68	}6970	if string(actual) != render {71		if err := writeFileAtomically(path, render); err != nil {72			return nil, fmt.Errorf("writing %s: %w", path, err)73		}74		out.wrote("writing the host entries")75	}7677	holds, err := os.ReadFile(path)78	if err != nil {79		return nil, fmt.Errorf("reading %s: %w", path, err)80	}81	return parseHostEntries(holds), nil82}8384// writeFileAtomically replaces path's contents without ever exposing85// a reader to a partial write. It creates a temporary file beside the86// target, on the same filesystem, so the rename that follows is one87// atomic directory-entry swap rather than a copy. A reader that opens88// path mid-write always sees either the old bytes or the new ones,89// never a mix. The rename also shapes the operator's mount: a bind90// mount of the file itself would hold the inode the rename discards,91// so the hostPath mount covers the /etc directory instead92// (manifests/machine-operator.yaml).93func writeFileAtomically(path, content string) error {94	tmp, err := os.CreateTemp(filepath.Dir(path), filepath.Base(path)+".tmp-*")95	if err != nil {96		return err97	}98	defer os.Remove(tmp.Name()) // no-op once the rename below succeeds99	if _, err := tmp.WriteString(content); err != nil {100		tmp.Close()101		return err102	}103	if err := tmp.Close(); err != nil {104		return err105	}106	if err := os.Chmod(tmp.Name(), 0o644); err != nil {107		return err108	}109	return os.Rename(tmp.Name(), path)110}111112// parseHostEntries recovers the host entries from a rendered hosts113// file: the lines below the three fixed lines that machine.HostsFile114// always writes first. It is the inverse of HostsFile, close enough115// for status, because applyHostEntries only ever calls it against a116// file this pass already brought into the shared renderer's shape.117func parseHostEntries(raw []byte) []machine.HostEntry {118	lines := strings.Split(strings.TrimRight(string(raw), "\n"), "\n")119	if len(lines) <= 3 {120		return nil121	}122	var entries []machine.HostEntry123	for _, line := range lines[3:] {124		fields := strings.Fields(line)125		if len(fields) < 2 {126			continue127		}128		entries = append(entries, machine.HostEntry{Address: fields[0], Names: fields[1:]})129	}130	return entries131}
machine-operator/imports.go 96.8%
1package main23// The operator's half of crash-safe image imports: proving a trial.4//5// Init stages an imported-images record before k3s first sees new6// tarballs, and discards the container store when it finds a record7// still staged from a boot that died (init's imports.go describes8// that half). This file provides the proof that finishes the work.9// The record cannot prove itself. Only something that watches10// containers actually run from the imported images can confirm the11// unpacks worked, and this operator does exactly that: its own pod12// runs from the tarball most worth proving.13//14// The proof rests on two observations and one barrier. First, every15// OS container on this node, meaning every container running a16// liken.sh/ image, must be Ready. This is the kubelet's own verdict,17// and it fails for a torn image the same way it fails for a crash18// loop, so a half-unpacked logs relay holds back the whole19// promotion. Second, the operator runs syncfs on the container20// store's filesystem. The OS pods only prove the images that run on21// this node, and a tarball whose image never schedules here (this22// includes the cluster operator, on most machines) could still carry23// a latent tear, until its dirty pages are written to disk. At that24// point no tear is possible at all. Only then does the record25// promote. A promotion that never happens is itself the signal: the26// condition stays False, the phase shows it, and the next reboot27// discards the store and tries again.2829import (30	"fmt"31	"os"32	"strings"3334	"golang.org/x/sys/unix"3536	"github.com/liken-sh/liken/liken/api"37	"github.com/liken-sh/liken/liken/kubernetes"38	"github.com/liken-sh/liken/liken/machine"39)4041// osImagePrefix marks the container images that arrive by tarball.42// Everything liken builds is named under the project's domain, and43// nothing else is.44const osImagePrefix = "liken.sh/"4546// importsInputs holds everything settleImportsLifecycle observed, so47// the decision itself stays a pure function.48type importsInputs struct {49	stagedHash string // identity of the staged record, "" when none50	provenHash string // identity of the proven record, "" when none51	storeErr   error  // reading the store failed52	pods       []kubernetes.Pod53	podsErr    error54}5556// importsVerdict is the decision: whether to promote now, and the57// ImportsConverged condition to publish either way.58type importsVerdict struct {59	promote   bool60	condition api.Condition61}6263// decideImportsPromotion judges one pass of the imports lifecycle,64// based on what was observed. The order of checks mirrors the other65// convergence decisions: no facts, nothing tracked, already settled,66// and then the proof itself.67func decideImportsPromotion(in importsInputs, facts *machine.MachineStatus) importsVerdict {68	condType := "ImportsConverged"69	if facts == nil {70		return importsVerdict{condition: convergenceUnknown(condType, "FactsIncomplete",71			"no facts published yet; the boot record's imports entry names what to prove")}72	}73	switch facts.Boot.ImportsSource {74	case "":75		// Init did not run the lifecycle. An ephemeral machineState76		// has nowhere to remember a trial. An ephemeral container77		// store resets with every boot and cannot get stuck. An78		// image from before the record existed reports the same way.79		return importsVerdict{condition: converged(condType, "NotTracked",80			"this boot tracks no imports; ephemeral state cannot get stuck and needs no proof")}81	case machine.ManifestSourceProven:82		return importsVerdict{condition: converged(condType, "Converged",83			fmt.Sprintf("the container store serves the proven imports (%.12s)", facts.Boot.ImportsHash))}84	}8586	// A trial is in progress. The store is read fresh on each pass,87	// because the facts cannot change after boot, but the store can.88	// This operator's own earlier pass may have already promoted it.89	if in.storeErr != nil {90		return importsVerdict{condition: convergenceUnknown(condType, "MachineStateUnavailable",91			fmt.Sprintf("reading the imports store: %v", in.storeErr))}92	}93	if in.stagedHash == "" {94		if in.provenHash == facts.Boot.ImportsHash {95			return importsVerdict{condition: converged(condType, "Converged",96				fmt.Sprintf("this boot's imports were proven (%.12s)", facts.Boot.ImportsHash))}97		}98		return importsVerdict{condition: convergenceUnknown(condType, "MachineStateUnavailable",99			"the boot record names a trial the store no longer holds; the next boot re-stages it")}100	}101	if in.stagedHash != facts.Boot.ImportsHash {102		return importsVerdict{condition: convergenceUnknown(condType, "FactsIncomplete",103			"the staged record is not the one this boot ran; waiting for fresh facts")}104	}105106	// The proof: every OS container on this node must be serving.107	// The kubelet's Ready condition covers every way a container can108	// fail, including the one this lifecycle exists for: a torn109	// image whose binary will not run.110	if in.podsErr != nil {111		return importsVerdict{condition: convergenceUnknown(condType, "ClusterUnavailable",112			fmt.Sprintf("listing this node's pods: %v", in.podsErr))}113	}114	observed := 0115	var waiting []string116	for _, pod := range in.pods {117		if pod.Completed() {118			continue119		}120		for _, container := range pod.Status.ContainerStatuses {121			if !strings.HasPrefix(container.Image, osImagePrefix) {122				continue123			}124			observed++125			if !container.Ready {126				waiting = append(waiting, fmt.Sprintf("%s/%s", pod.Metadata.Name, container.Name))127			}128		}129	}130	if observed == 0 {131		return importsVerdict{condition: notConverged(condType, "Proving",132			"no OS containers observed on this node yet; the trial is still proving")}133	}134	if len(waiting) > 0 {135		return importsVerdict{condition: notConverged(condType, "Proving",136			fmt.Sprintf("waiting for OS containers to serve the trialed imports: %s", strings.Join(waiting, ", ")))}137	}138	return importsVerdict{promote: true, condition: converged(condType, "Converged",139		fmt.Sprintf("%d OS containers serve the trialed imports (%.12s); proven", observed, facts.Boot.ImportsHash))}140}141142// settleImportsLifecycle observes, judges, and promotes when the143// proof succeeds. The syncfs barrier runs before the promotion144// write. If the promotion write ran first, a badly-timed power cut145// could prove a store whose latent unpacks are still dirty, which is146// the exact false claim this lifecycle exists to prevent.147func settleImportsLifecycle(r *reader, root, nodeName string, facts *machine.MachineStatus, out *passOutcome) api.Condition {148	store := machine.ImportedImagesStore(root)149	in := importsInputs{}150	if facts != nil && facts.Boot.ImportsSource == machine.ManifestSourceStaged {151		staged, err := store.LoadStaged()152		in.storeErr = err153		switch {154		case staged != nil:155			in.stagedHash = machine.ManifestHash(staged)156			if in.storeErr == nil && in.stagedHash == facts.Boot.ImportsHash {157				in.pods, in.podsErr = r.nodePods(nodeName)158			}159		case in.storeErr == nil:160			// Nothing staged under a Staged boot usually means an161			// earlier pass already promoted it. The proven record's162			// identity confirms this.163			proven, err := store.LoadProven()164			in.storeErr = err165			if proven != nil {166				in.provenHash = machine.ManifestHash(proven)167			}168		}169	}170	out.fail("reading the imports record", in.storeErr)171	v := decideImportsPromotion(in, facts)172	if !v.promote {173		return v.condition174	}175	if err := syncContainerStore(); err != nil {176		out.fail("syncing the container store", err)177		return convergenceUnknown(v.condition.Type, "PromotionFailed",178			fmt.Sprintf("syncing the container store before promotion: %v", err))179	}180	if err := store.Promote(); err != nil {181		out.fail("promoting the imports record", err)182		return convergenceUnknown(v.condition.Type, "PromotionFailed",183			fmt.Sprintf("promoting the imports record: %v", err))184	}185	out.wrote("promoting the imports record")186	fmt.Printf("proved this boot's imports (%.12s); the container store is trusted\n", facts.Boot.ImportsHash)187	return v.condition188}189190// containerStoreDir is the container store's tree. It is a package191// variable for the same reason as sysctlRoot: a test of a promotion192// points it at a tempdir, so the test syncs no host filesystem.193var containerStoreDir = machine.K3sAgentDir194195// syncContainerStore flushes everything on the container store's196// filesystem to disk. The store is reachable inside this pod as a197// read-only hostPath of the same tree init discards198// (machine.K3sAgentDir names it for both halves). Here, its only use199// is as a handle for the syncfs call. One syscall turns the fact200// that the OS pods we can see are serving into the fact that every201// byte the imports wrote is durable. After it returns, no image on202// this store, including images whose pods never schedule here, can203// be torn by a crash.204func syncContainerStore() error {205	f, err := os.Open(containerStoreDir)206	if err != nil {207		return err208	}209	defer f.Close()210	return unix.Syncfs(int(f.Fd()))211}
machine-operator/labels.go 100.0%
1package main23// Node labels, reconciled live from the Machine spec.4//5// Workloads schedule based on node labels, so spec.nodeLabels sets a6// machine's scheduling identity, and it reaches the Node in two7// ways. Init renders the labels into the k3s boot drop-in, so the8// node registers with them already applied, which covers a9// machine's first moments. But registration only adds labels: the10// kubelet applies labels and never removes one, so a label taken out11// of the spec would stay on the Node forever. Live reconciliation12// belongs here, in the same pass that re-applies sysctls.13//14// Removing a label needs a record. Nothing about a label on a Node15// says who put it there, and the operator must never remove one that16// a person or another controller applied. The record is an17// annotation on the Node that lists exactly the keys this operator18// manages. When a key is in the annotation but not in the spec, that19// difference tells the operator to remove it. This is the same20// method the drain uses to tell its own cordon apart from a21// person's.2223import (24	"encoding/json"25	"fmt"26	"maps"27	"slices"28	"strings"2930	"github.com/liken-sh/liken/kubernetes/apiclient"31	"github.com/liken-sh/liken/liken/api"32	"github.com/liken-sh/liken/liken/kubernetes"33)3435// ownedLabelsAnnotation records, on the Node itself, which label36// keys liken manages. Its value is the spec's keys, sorted and37// joined with commas. It lives on the Node rather than in the38// Machine's status, so the record and the labels it describes can39// never drift apart across operator restarts or Machine rewrites.40const ownedLabelsAnnotation = "liken.sh/node-labels"4142// A labelStep is one pass's worth of label reconciliation: the Node43// patch to apply (nil when the Node already matches the spec) and44// the condition to publish once the patch lands.45type labelStep struct {46	patch     []byte47	condition api.Condition48}4950// decideNodeLabels compares the spec's labels against the Node and51// produces the merge patch that closes the gap. It reapplies52// missing and changed labels, deletes removed ones (a null value in53// a merge patch deletes the key), and keeps the ownership annotation54// matching the spec. Labels outside both the spec and the55// annotation belong to someone else, and the function never touches56// them.57func decideNodeLabels(desired map[string]string, node *nodeObject) labelStep {58	labels := map[string]any{}59	for key, value := range desired {60		if node.Metadata.Labels[key] != value {61			labels[key] = value62		}63	}64	for owned := range strings.SplitSeq(node.Metadata.Annotations[ownedLabelsAnnotation], ",") {65		if owned == "" {66			continue67		}68		if _, still := desired[owned]; still {69			continue70		}71		if _, present := node.Metadata.Labels[owned]; present {72			labels[owned] = nil73		}74	}7576	annotations := map[string]any{}77	ownedNow := strings.Join(slices.Sorted(maps.Keys(desired)), ",")78	if ownedNow != node.Metadata.Annotations[ownedLabelsAnnotation] {79		if ownedNow == "" {80			annotations[ownedLabelsAnnotation] = nil81		} else {82			annotations[ownedLabelsAnnotation] = ownedNow83		}84	}8586	condition := api.Condition{Type: "NodeLabelsApplied", Status: api.ConditionTrue, Reason: "Applied",87		Message: fmt.Sprintf("the Node carries all %d declared labels", len(desired))}88	if len(desired) == 0 {89		condition = api.Condition{Type: "NodeLabelsApplied", Status: api.ConditionTrue, Reason: "NothingDeclared",90			Message: "no node labels declared"}91	}9293	if len(labels) == 0 && len(annotations) == 0 {94		return labelStep{condition: condition}95	}96	metadata := map[string]any{}97	if len(labels) > 0 {98		metadata["labels"] = labels99	}100	if len(annotations) > 0 {101		metadata["annotations"] = annotations102	}103	patch, _ := json.Marshal(map[string]any{"metadata": metadata})104	return labelStep{patch: patch, condition: condition}105}106107// carryOutNodeLabels applies the step's patch. It downgrades the108// condition when the API server refuses the patch. The next pass109// reads the Node again, builds the step again, and patches again.110func carryOutNodeLabels(c *apiclient.Client, name string, step labelStep) api.Condition {111	if step.patch == nil {112		return step.condition113	}114	if err := kubernetes.PatchJSON(c, nodesPath+"/"+name, step.patch); err != nil {115		return api.Condition{Type: "NodeLabelsApplied", Status: api.ConditionFalse, Reason: "ApplyFailed",116			Message: fmt.Sprintf("patching the Node's labels: %v", err)}117	}118	return step.condition119}
machine-operator/liveness.go 96.9%
1package main23// The heartbeat's clock, and how a stuck loop stops it.4//5// cluster-operator marks a machine Lost when its heartbeat lease goes6// unrenewed for 40 seconds (kubernetes/heartbeat.go). The lease must7// prove that this operator does its job, not only that the process8// runs, so a renewal must stop when the reconcile loop is stuck: a call9// that never answers, or a lock that never releases. A loop that waits10// in its select for a wake is healthy, because it acts the moment a11// wake comes. So the loop marks itself busy from the moment its select12// returns until it enters the select again, and a timer of its own13// renews the lease every few seconds unless the loop has been busy for14// longer than stuckAfter.15//16// stuckAfter must be longer than any pass that is not stuck, so the17// pass's requests run under passDeadline (loop.go), and the margin18// covers the work that sends no request, such as a walk of sysfs or a19// syncfs. The gauge of the longest pass shows how close a machine comes20// to the limit.21//22// A loop that is stuck stops the renewals, and the same timer ends the23// process, the crash-only rule at the head of main.go: the kubelet24// starts the container again, and its first pass reads the state as it25// is. A process that cannot restart its loop keeps restarting, and the26// kubelet's growing backoff lets the lease age past 40 seconds, so27// cluster-operator marks the machine Lost. The timer is in the process,28// not a kubelet liveness probe, because a probe also fails while the29// listener is not open: during the setup before the loop, when port30// 9200 is taken or moved, and for a binary older than its pod31// template. /healthz answers the same check, for a person.3233import (34	"context"35	"errors"36	"fmt"37	"os"38	"sync/atomic"39	"time"4041	"github.com/liken-sh/liken/kubernetes/apiclient"42	"github.com/liken-sh/liken/liken/api"43	"github.com/liken-sh/liken/liken/kubernetes"44	"github.com/liken-sh/liken/liken/machine"45)4647const (48	// passDeadline ends every request a pass sends that is still open49	// this long after the pass began.50	passDeadline = 45 * time.Second5152	// stuckAfter is how long the loop may stay busy before the53	// renewals stop and /healthz fails.54	stuckAfter = 60 * time.Second5556	// renewEvery is half of HeartbeatRenewAfter, so a timer that fires57	// a moment early, and finds the last renewal a moment short of the58	// age that Renew wants, renews on its next firing.59	renewEvery = kubernetes.HeartbeatRenewAfter / 260)6162// liveness is what the loop and the renewal timer share. Each value is63// atomic, because the timer runs on its own goroutine.64type liveness struct {65	// start is the reference for busySince. A duration since start66	// reads the monotonic clock, so a step of the wall clock does not67	// make the loop look busy or idle.68	start time.Time6970	// busySince is the time since start at which the loop's select71	// returned, plus one, or zero while the loop waits in its select.72	busySince atomic.Int647374	// owner is this machine's Machine, the owner of the lease. gone is75	// true while the Machine does not exist: a renewal would create76	// the lease again, owned by a Machine that does not exist, for the77	// garbage collector to delete again.78	owner atomic.Pointer[kubernetes.OwnerReference]79	gone  atomic.Bool8081	// exit ends the process. main passes os.Exit, and a test passes a82	// function that records the call.83	exit func(code int)84}8586func newLiveness() *liveness {87	return &liveness{start: time.Now(), exit: os.Exit}88}8990// markBusy marks the loop busy from now, unless it is busy already. A91// nil liveness marks nothing.92func (l *liveness) markBusy() {93	if l != nil {94		l.busySince.CompareAndSwap(0, int64(time.Since(l.start))+1)95	}96}9798// markIdle marks the loop as waiting for a wake.99func (l *liveness) markIdle() {100	if l != nil {101		l.busySince.Store(0)102	}103}104105// busyFor answers how long the loop has been busy, or zero while it106// waits.107func (l *liveness) busyFor() time.Duration {108	since := l.busySince.Load()109	if since == 0 {110		return 0111	}112	return time.Since(l.start) - time.Duration(since-1)113}114115// errStuck is the answer of a check of a loop that is stuck.116var errStuck = errors.New("the reconcile loop is stuck")117118// check answers an error when the loop has been busy past stuckAfter.119// It is /healthz.120func (l *liveness) check() error {121	if busy := l.busyFor(); busy > stuckAfter {122		return fmt.Errorf("%w: busy for %s", errStuck, busy.Round(time.Second))123	}124	return nil125}126127// sawMachineOf records the Machine a pass read, and sawNoMachine a128// Machine that is gone.129func (l *liveness) sawMachineOf(m *machine.Machine) {130	if l == nil {131		return132	}133	l.owner.Store(&kubernetes.OwnerReference{APIVersion: api.APIVersion, Kind: machineKind, Name: m.Metadata.Name, UID: m.Metadata.UID})134	l.gone.Store(false)135}136137func (l *liveness) sawNoMachine() {138	if l != nil {139		l.gone.Store(true)140	}141}142143// renew renews the lease once, unless the Machine is gone or the loop144// is stuck. Only the renewal timer calls it after the loop starts,145// because a Heartbeat has no lock.146func (l *liveness) renew(hb *kubernetes.Heartbeat, c *apiclient.Client) {147	owner := l.owner.Load()148	if owner == nil || l.gone.Load() || l.check() != nil {149		return150	}151	hb.Renew(c, *owner, time.Now())152}153154// renewUntil renews the lease every renewEvery until ctx ends, and155// ends the process once the loop is stuck. The caller renews once156// before it starts the loop's first pass, so a machine that boots into157// a fleet that already declared it Lost announces itself before its158// first status write.159func (l *liveness) renewUntil(ctx context.Context, hb *kubernetes.Heartbeat, c *apiclient.Client) {160	ticker := time.NewTicker(renewEvery)161	defer ticker.Stop()162	for {163		select {164		case <-ctx.Done():165			return166		case <-ticker.C:167			if err := l.check(); err != nil {168				fmt.Fprintf(os.Stderr, "%v; ending the process so the kubelet starts it again\n", err)169				l.exit(1)170				return171			}172			l.renew(hb, c)173		}174	}175}
machine-operator/loop.go 96.6%
1package main23// The reconcile loop: what wakes a pass, and what the loop keeps from4// one pass to the next.5//6// The core of every operator is a level-triggered loop. Every pass7// reconciles from the current state as it is, never from the event8// that woke it, so missing one wake can never matter. Four things wake9// it, and two timers check what no event reports. The Kubernetes10// watches wake the loop when an object this machine11// acts on changes, so a conductor's grant or a person's edit is acted12// on at once (watches.go names each watch and what wakes it). The facts13// watch wakes the loop when init publishes a change under14// /run/liken/facts, so a fresh fact like a time sync reaches status15// without waiting on a timer. The uevent listener and the hosts watch16// wake the loop when a device or /etc/hosts changes on the machine17// (machineevents.go). The retry timer wakes the loop when the last18// pass left a step unfinished, or when a step asked to run again at a19// set time (retry.go). The check of the sysctls runs a pass when a20// parameter that the last pass applied reads another value, and the21// backstop runs a pass after five minutes with no other pass22// (backstop.go). A settled machine runs no pass until something23// changes. The heartbeat lease renews on a timer of its own, which24// stops when the loop is stuck (liveness.go).25//26// The wake channel has one slot, so a burst of changes (the27// conductor's grant, the sweeper's verdict, this operator's own28// publishes echoing back) makes one wake, and one pass over the29// newest state answers the whole burst. That is what level-triggered30// means, and it is the same merging an informer's work queue does.3132import (33	"context"34	"errors"35	"fmt"36	"os"37	"strings"38	"time"3940	"github.com/liken-sh/liken/kubernetes/apiclient"41	"github.com/liken-sh/liken/liken/kubernetes"42	"github.com/liken-sh/liken/liken/machine"43	"github.com/liken-sh/liken/liken/metrics"44)4546// loop holds what the reconcile loop keeps across passes, and the47// channels that wake it. main builds one, and a test builds one with48// its own channels.49type loop struct {50	objects     *reader51	name        string52	clusterName string53	fetcher     *fetcher54	heartbeat   *kubernetes.Heartbeat55	operator    *metrics.Operator56	layer       *machineMetrics5758	// wakes carries the watches' wakes.59	wakes <-chan struct{}6061	// uevents and hosts are the machine's readers (machineevents.go),62	// opened before the loop starts. A nil channel is a reader that63	// did not open, and the loop runs without it.64	uevents <-chan struct{}65	hosts   <-chan struct{}6667	// live is the busy mark that the renewal timer reads68	// (liveness.go). Nil renews no lease.69	live *liveness7071	// backstopJitter answers a number in [0, 1) for the backstop's72	// delay. Nil means math/rand.73	backstopJitter func() float647475	// watchFactsTree opens the facts watch. main passes76	// machine.WatchFactsTree, and a test passes a watch it controls.77	watchFactsTree func(ctx context.Context) (*factsWatch, error)7879	retry retrySchedule80}8182// run runs passes until ctx ends, starting from current, the Machine83// that main read or created. It answers an error when a reader stops84// or the facts watch cannot open, and main ends the process on it85// (machineevents.go gives the reason).86func (l *loop) run(ctx context.Context, current *machine.Machine) error {87	// The facts watch turns init's writes into wakes. inotify does not88	// recurse, so the watch reconciles its set with the tree before89	// every read (Sync, below). init publishes the facts before it90	// writes /run/liken/machine.yaml and before it starts k3s, and91	// main exits when machine.yaml is missing, so the tree exists by92	// now, and a watch that cannot open means something is wrong with93	// the machine.94	factsWatch, err := l.watchFactsTree(ctx)95	if err != nil {96		return fmt.Errorf("watching the facts tree: %w", err)97	}9899	// The relays turn the machine's events into wakes on a channel of100	// their own, which the select below reads beside the watches'.101	ctx, cancel := context.WithCancel(ctx)102	defer cancel()103	machineWakes := make(chan struct{}, 1)104	stopped := make(chan error, 2)105	if l.uevents != nil {106		go func() {107			stopped <- relayStopped("the uevent listener", relay(ctx, l.uevents, machineWakes, settleEvents))108		}()109	}110	if l.hosts != nil {111		go func() { stopped <- relayStopped("the hosts watch", relay(ctx, l.hosts, machineWakes, settleEvents)) }()112	}113114	// The lease renews once before the first pass, with the owner from115	// the Machine that main read, so a machine that boots into a fleet116	// that already declared it Lost announces itself before its first117	// status write. After that only the timer renews it.118	if l.live != nil {119		l.live.sawMachineOf(current)120		l.live.renew(l.heartbeat, l.objects.client)121		go l.live.renewUntil(ctx, l.heartbeat, l.objects.client)122	}123124	// The loop is busy until it first waits for a wake. The first125	// renewal above runs before the mark, because its requests run under126	// no pass's deadline (liveness.go).127	l.live.markBusy()128	woke := time.Now()129130	var retry *time.Timer131	sysctlChecks := time.NewTicker(sysctlCheckEvery)132	defer sysctlChecks.Stop()133	backstop := time.NewTimer(backstopDelay(l.backstopJitter))134	defer backstop.Stop()135	cause := causeStart136	var sysctls sysctlCheck137	for {138		// Sync before the read closes the window between a new subtree139		// and the watch on it: a directory that init created since the140		// last pass gets a watch now, before this pass reads the tree.141		//142		// A watch whose channel closed has a closed descriptor, and a143		// Sync on it would add watches to a descriptor number the144		// process may have given to something else since.145		if watchStopped(factsWatch) {146			return errFactsStopped147		}148		// A Sync that fails leaves a new subtree with no watch, so its149		// writes would wake nothing. The failure goes to the outcome, and150		// the retry runs Sync again soon.151		out := &passOutcome{}152		if err := factsWatch.sync(); err != nil {153			fmt.Fprintf(os.Stderr, "syncing the facts watch: %v\n", err)154			out.failSoon("syncing the facts watch", err)155		}156		l.layer.observeWake(cause)157158		// Every request of the pass ends at passDeadline, so a pass that159		// is slow but not stuck finishes inside stuckAfter, and only a160		// pass that is stuck stops the heartbeat (liveness.go).161		passCtx, endPass := context.WithTimeout(ctx, passDeadline)162		objects := l.objects.throughAPIOnce().within(passCtx)163		// Each pass starts from the newest copy of this machine's164		// object. Status writes change resourceVersion, and165		// reconciling against a stale copy would make every status166		// update a conflict. A read that fails keeps the copy the last167		// pass had; the publish conflict retry handles a stale one.168		//169		// A Machine that is gone gets no heartbeat. The lease names the170		// Machine as its owner, so the garbage collector deletes the171		// lease with it, and a renewal would create the lease again,172		// owned by a Machine that does not exist, for the collector to173		// delete again.174		if fresh, err := objects.observedBy(out).machine(l.name); err == nil {175			current = fresh176			l.live.sawMachineOf(current)177		} else if errors.Is(err, apiclient.ErrNotFound) {178			l.live.sawNoMachine()179		}180		started := time.Now()181		err := reconcile(objects, current, l.clusterName, l.fetcher, l.layer, out)182		took := time.Since(started)183		endPass()184		l.operator.ObserveReconcile(machineKind, took, err)185		// The gauge times the whole busy window that the liveness check186		// judges, from the wake to the end of the pass.187		l.layer.observePass(time.Since(woke))188		// A wait's watch runs while the passes read it (waits.go).189		l.objects.waits.endPass()190		if out.sysctls != nil {191			sysctls = sysctlCheck{applied: out.sysctls, missing: out.sysctlsMissing}192		}193		if cause == causeBackstop {194			reportRepairs(out.writes, machineEvents{recorder: l.objects.recorder, machine: machineReference(current)}, l.layer)195		}196197		// One timer serves the retry and every step's wake, and each198		// pass sets it again from its own outcome, so a pass that199		// finished everything stops it.200		if retry != nil {201			retry.Stop()202		}203		var retryC <-chan time.Time204		if at, ok := l.retry.next(out, time.Now()); ok {205			retry = time.NewTimer(time.Until(at))206			retryC = retry.C207			if len(out.failures) > 0 {208				fmt.Printf("the pass left %s; a retry is due in %s\n",209					describeUnfinished(out.failures), max(0, time.Until(at)).Round(10*time.Millisecond))210			}211		}212		// The backstop counts from the end of the last pass, so a213		// machine that runs passes for other reasons runs no backstop.214		// While a retry is due, the retry's pass does the backstop's work,215		// and a write it makes is the retry's, not a missed wake.216		backstop.Stop()217		if retryC == nil {218			backstop.Reset(backstopDelay(l.backstopJitter))219		}220221		var stop bool222		cause, stop, err = l.wait(ctx, waitSources{stopped: stopped, machineWakes: machineWakes, retry: retryC,223			facts: factsWatch, sysctlChecks: sysctlChecks.C, backstop: backstop.C}, sysctls)224		if stop {225			if retry != nil {226				retry.Stop()227			}228			return err229		}230		woke = time.Now()231	}232}233234// The causes of a pass, for the passes_total counter, so each pass on235// a settled machine can be explained.236const (237	causeStart    = "start"238	causeWatch    = "watch"239	causeMachine  = "machine event"240	causeRetry    = "retry"241	causeFacts    = "facts"242	causeSysctls  = "sysctl drift"243	causeBackstop = "backstop"244)245246// waitSources are the channels the loop's select reads.247type waitSources struct {248	stopped      <-chan error249	machineWakes <-chan struct{}250	retry        <-chan time.Time251	facts        *factsWatch252	sysctlChecks <-chan time.Time253	backstop     <-chan time.Time254}255256// wait waits in the loop's select until something asks for a pass, and257// answers its cause. It marks the loop busy from the moment the select258// returns. A check of the sysctls that finds every parameter as the259// last pass left it goes back to the select with no pass. It answers260// true, with the error for run to return, when the loop must end.261//262// A backstop that fires while another cause is ready gives the pass to263// that cause, so its writes are not reported as a missed wake: a wake264// already waiting, and a sysctl that drifted, explain the pass better.265func (l *loop) wait(ctx context.Context, from waitSources, sysctls sysctlCheck) (string, bool, error) {266	for {267		l.live.markIdle()268		select {269		case <-ctx.Done():270			return "", true, nil271		case err := <-from.stopped:272			l.live.markBusy()273			if err != nil {274				return "", true, err275			}276			return causeMachine, false, nil277		case <-l.wakes:278			l.live.markBusy()279			return causeWatch, false, nil280		case <-from.machineWakes:281			l.live.markBusy()282			return causeMachine, false, nil283		case <-from.retry:284			l.live.markBusy()285			return causeRetry, false, nil286		case _, ok := <-from.facts.wake:287			l.live.markBusy()288			// A closed channel means the watch died, and init's writes289			// since then reach nobody.290			if !ok {291				return "", true, errFactsStopped292			}293			return causeFacts, false, nil294		case <-from.backstop:295			l.live.markBusy()296			return l.backstopCause(from, sysctls), false, nil297		case <-from.sysctlChecks:298			l.live.markBusy()299			if sysctls.drifted(sysctlRoot) {300				return causeSysctls, false, nil301			}302		}303	}304}305306// backstopCause answers the cause of a pass that the backstop's timer307// started: another cause that is ready at the same moment, or the308// backstop itself.309func (l *loop) backstopCause(from waitSources, sysctls sysctlCheck) string {310	select {311	case <-l.wakes:312		return causeWatch313	case <-from.machineWakes:314		return causeMachine315	case <-from.retry:316		return causeRetry317	default:318	}319	if sysctls.drifted(sysctlRoot) {320		return causeSysctls321	}322	return causeBackstop323}324325// factsWatch is the part of the facts watch the loop uses: the wake326// channel and the Sync of a *machine.TreeWatch.327type factsWatch struct {328	wake <-chan struct{}329	sync func() error330}331332// errFactsStopped ends the loop when the facts watch's channel333// closes.334var errFactsStopped = errors.New("the facts watch stopped")335336// relayStopped names the reader in a relay's error, and passes the nil337// of a relay whose context ended.338func relayStopped(reader string, err error) error {339	if err == nil {340		return nil341	}342	return fmt.Errorf("%s: %w", reader, err)343}344345// watchStopped answers whether the facts watch's channel is closed. A346// wake still pending on an open channel is consumed, which costs347// nothing, because the pass that follows reads the whole tree.348func watchStopped(w *factsWatch) bool {349	select {350	case _, ok := <-w.wake:351		return !ok352	default:353		return false354	}355}356357// describeUnfinished names the steps a pass did not finish, with the358// error each met, for the one log line a pass with failures writes. It359// names the first three and counts the rest, because a machine whose360// sysctl tree is missing fails every parameter at once, and the step361// itself already logged each one.362func describeUnfinished(failures []passFailure) string {363	const named = 3364	var parts []string365	for _, f := range failures[:min(len(failures), named)] {366		parts = append(parts, fmt.Sprintf("%s unfinished (%s: %v)", f.step, f.kind, f.err))367	}368	if len(failures) > named {369		parts = append(parts, fmt.Sprintf("and %d more", len(failures)-named))370	}371	return strings.Join(parts, "; ")372}
machine-operator/machineevents.go 100.0%
1package main23// The kernel's events for the state on the machine that a pass reads.4//5// A pass reads two kinds of state from the machine itself that the6// kernel announces, beside the facts that init publishes. The device7// inventory and the claims' CDI specifications come from a walk of8// sysfs (dra.go, cdi.go), and the host entries come from the host's9// /etc/hosts (hosts.go). The kernel announces a change to either one:10// a uevent for a device that arrives, leaves, or changes its driver,11// and an inotify event for a file. So two readers turn those events12// into wakes of the loop.13//14// The sysctls have no such event. A write through /proc/sys does send15// inotify events, but only to a watch on the same mount of procfs,16// because each mount has inodes of its own. Each container mounts its17// own /proc, and the host has its own, so a watch in this pod sees18// this pod's writes alone. The loop checks them on a timer instead19// (backstop.go).20//21// The readers open before the first pass, so a change during the first22// walk or the first read of a file still sends a wake after it.23//24// A reader that stops after it opened ends the operator. A reader stops25// only on a poll error or a descriptor that is no longer open, which26// nothing in the operator can repair, and a reader that stopped misses27// every change after it. The kubelet starts the container again with28// its backoff, and the new process opens new readers before its first29// pass, so a change made while nothing listened is still read. This is30// the crash-only rule at the head of main.go.3132import (33	"context"34	"errors"35	"io/fs"36	"path/filepath"37	"time"3839	"github.com/liken-sh/liken/liken/hardware"40	"github.com/liken-sh/liken/liken/kubernetes/watch"41	"github.com/liken-sh/liken/liken/machine"42)4344// listenForUevents opens the uevent listener, filtered to the events45// that can change what the inventory reads (hardware.InventoryEvent).46// The veth pairs that pods add and remove wake nothing. It is a47// variable so a test can hand the loop a listener whose wakes the test48// sends.49var listenForUevents = func(ctx context.Context) (<-chan struct{}, error) {50	return hardware.ListenForUeventsMatching(ctx, hardware.InventoryEvent)51}5253// watchHostsFile watches the name hosts in the host's /etc, the54// directory the pod mounts (hosts.go). It is a variable so a test can55// hand the loop a watch whose wakes the test sends.56var watchHostsFile = func(ctx context.Context) (<-chan struct{}, error) {57	return machine.WatchName(ctx, filepath.Dir(hostsPath), filepath.Base(hostsPath))58}5960// hostsWatchFailure answers the error of a hosts watch that failed to61// open, unless its directory does not exist. A pod from a template62// older than the /host/etc mount has no directory, and the operator63// runs without the watch while hostEntriesCondition reports the missing64// mount. Any other failure, such as EMFILE or ENOSPC from inotify,65// would leave the hosts file with no wake for the life of the process,66// so it ends the process, the same as a reader that stops.67func hostsWatchFailure(err error) error {68	if err == nil || errors.Is(err, fs.ErrNotExist) {69		return nil70	}71	return err72}7374// errReaderStopped is what relay answers when its reader closes the75// channel.76var errReaderStopped = errors.New("the reader stopped")7778// settleEvents gathers a burst of events into one wake, with init's79// intervals for its own hardware watch: one second of quiet, and five80// seconds at most. One USB device that enumerates sends a dozen events81// in a few milliseconds, and the pass after the burst reads every one82// of them. The hosts file needs it too. The pass's own write to83// /etc/hosts sends an event, and a process that keeps writing another84// file would otherwise trade writes with the pass as fast as both can85// write. With the settle, that costs a pass every five seconds at most.86// The settle runs in the relay's goroutine, not in the loop, so a87// stream that holds it at the ceiling never delays a wake from a watch.88func settleEvents(ctx context.Context, events <-chan struct{}) {89	hardware.Settle(ctx, events, time.Second, 5*time.Second)90}9192// relay turns a reader's events into wakes of the loop until the93// context ends, and answers errReaderStopped when the reader closes its94// channel. settle, when it is not nil, waits out a burst after the95// first event of it.96func relay(ctx context.Context, events <-chan struct{}, wakes chan<- struct{}, settle func(context.Context, <-chan struct{})) error {97	wake := watch.Signal(wakes)98	for {99		select {100		case <-ctx.Done():101			return nil102		case _, ok := <-events:103			if !ok {104				return errReaderStopped105			}106			if settle != nil {107				settle(ctx, events)108			}109			wake()110		}111	}112}
machine-operator/main.go 0.0%
1// liken-machine-operator: the program that makes the Kubernetes API2// the machine API.3//4// An operator is not a special kind of software. It is an ordinary5// program that runs in a pod, reads the state of the world, compares6// it to a declared spec, and acts until they agree. Then it keeps7// watching, forever. Kubernetes itself is built out of these loops8// (kube-controller-manager alone runs dozens of them). This one9// reconciles the machine underneath the cluster, instead of10// something inside it.11//12// liken's OS runs two programs, split by their scope. This one is13// node-local: it runs privileged on every machine (a DaemonSet),14// reads the facts init published, actuates the spec against the15// machine itself, and reports the result as its Machine's status.16// Its counterpart, liken-cluster-operator, is an ordinary17// unprivileged workload that watches the whole fleet and writes the18// verdicts no single machine can make: which machines are Lost, the19// Cluster's headcount, and whose turn it is to reboot. The20// connection between them is the Machine status this program writes21// and the heartbeat lease it renews.22//23// This program divides the work with init (see the machine24// package) as follows. Init observes the boot and writes facts to25// /run/liken. This operator reads them through a hostPath mount,26// adds what it can observe itself, and publishes the result as the27// Machine's status. In the other direction, it actuates the spec.28// Today that means sysctls, written straight to /proc/sys, which29// belongs to the host, because this pod runs privileged in the30// host's namespaces (see manifests/machine-operator.yaml for what31// that means and why it is justified here).32package main3334import (35	"context"36	"flag"37	"fmt"38	"os"39	"strings"4041	"github.com/liken-sh/liken/kubernetes/events"42	"github.com/liken-sh/liken/kubernetes/informer"43	"github.com/liken-sh/liken/liken/cluster"44	"github.com/liken-sh/liken/liken/kubernetes"45	"github.com/liken-sh/liken/liken/kubernetes/watch"46	"github.com/liken-sh/liken/liken/machine"47)4849// component is this program's name in every metric that carries one.50const component = "liken-machine-operator"5152// metricsAddress is where this operator answers a Prometheus scrape.53// The default is the port that54// `plans/completed/65-prometheus-metrics.md` at the top of the55// repository gives the machine operator, so the binary carries the56// contract and the pod template only has to name the port it exposes.57// The address is an argument, and not a constant, for two reasons. An58// empty value turns the listener off, for an owner who runs no59// Prometheus and wants the port back. This pod also runs on the host's60// network, so the port belongs to the whole machine, and an owner who61// already serves 9200 there needs a way to move liken.62var metricsAddress = flag.String("metrics-address", ":9200",63	"the address to serve /metrics on; empty serves no metrics")6465func main() {66	flag.Parse()67	fmt.Println(component, machine.Version)6869	// The boot manifest tells the operator which Machine it manages.70	// These are the exact bytes init booted under: the proven or71	// staged copy from machineState, or the image's seed on a first72	// boot. Init publishes them to /run/liken, and the operator reads73	// them here through a hostPath mount. One image carries manifests74	// for a whole fleet, and init already selected this machine's.75	// The operator trusts that selection instead of repeating the76	// work.77	// A rollback can boot this release under a proven manifest that a78	// newer release wrote, so the read skips the fields this release79	// does not know, as init's did (machine.ParseKnown).80	m, ignored, err := machine.LoadKnown(machine.BootManifestPath)81	if err != nil {82		fatal("boot manifest: %v", err)83	}84	if len(ignored) > 0 {85		fmt.Printf("the boot manifest names fields this release does not know: %s\n", strings.Join(ignored, ", "))86	}87	name := m.Metadata.Name88	if name == "" {89		fatal("the boot manifest names no machine; nothing to operate")90	}9192	// This reads the cluster document that this boot ran, before93	// building the client, because the document names which API94	// endpoint this machine should use (localAPIEndpoint,95	// endpoint.go).96	clusterDoc, err := cluster.LoadCluster(cluster.ClusterManifestPath)97	if err != nil {98		fatal("cluster manifest: %v", err)99	}100101	// Failures during setup end the process deliberately. This code102	// has no retry logic, because the kubelet already provides it: a103	// pod that exits nonzero is restarted with backoff, and the104	// failure is visible in `kubectl get pods` instead of hidden in a105	// log. This is the crash-only style most Kubernetes components106	// use.107	client, err := kubernetes.InClusterClient(localAPIEndpoint(clusterDoc, name))108	if err != nil {109		fatal("in-cluster config: %v", err)110	}111112	// The DRA plugin serves the kubelet for the life of the process113	// (draplugin.go). Its failure is loud but not fatal, deliberately.114	// Right after an upgrade, this binary can run in a pod created115	// from the previous release's template: OnDelete keeps the old116	// pod, and the stable image tag resolves to the new build. If117	// that template lacks a mount this plugin needs, dying here would118	// kill the whole operator, including the status publishing that119	// the pod steward is waiting on to refresh this same pod. The120	// machine must keep operating without device claims. The121	// refreshed pod brings the plugin up.122	// The plugin can prepare a claim before the first reconcile pass123	// reads the live spec, so the boot manifest's serio entries, and124	// the entries a live load added since the boot, seed the list of125	// serial lines it withholds (dra.go). Facts that do not read leave126	// the boot manifest's entries alone.127	bootFacts, _ := factsTree.Read()128	setDeclaredSerio(serioInEffect(m.Spec.Serio, bootFacts))129	setPlatformProtection(protectionOf(bootFacts))130	go func() {131		if err := serveDRAPlugin(context.Background(), client); err != nil {132			fmt.Fprintf(os.Stderr, "the DRA plugin is not serving: %v\n", err)133		}134	}()135136	// The file seeds the cluster. If no Machine object exists yet,137	// the manifest's spec becomes the first version of it. From then138	// on, the cluster's copy is authoritative: a kubectl edit wins139	// over the file until someone rebuilds the image. (The flux140	// feature closes this loop for good: a deployment that declares141	// it hands the in-cluster copy to its git repository, and the142	// two sides converge on every commit.)143	// The recorder posts the Events about this Machine (events.go). It144	// writes from its own goroutine, so a pass never waits on an Event.145	recorder := events.New(context.Background(), client, component, events.Options{})146147	current, err := ensureMachine(client, m, recorder)148	if err != nil {149		fatal("ensuring machine %s exists: %v", name, err)150	}151	fmt.Printf("operating machine %s\n", name)152153	// The Cluster resource gets the same treatment as the Machine:154	// the image's cluster.yaml seeds it if it does not exist. Every155	// machine's operator tries this, because every image carries the156	// manifest, so most of them lose the race and find the object157	// already there. That still counts as success. What matters is158	// that the cluster's topology can be read, not which machine159	// published it. (Seeding happens here, rather than in the160	// cluster operator, because this program has the image's manifest161	// available to it. The cluster operator has no mounts at all.)162	if clusterDoc != nil {163		if err := ensureCluster(client, clusterDoc); err != nil {164			fatal("ensuring cluster %s exists: %v", clusterDoc.Metadata.Name, err)165		}166	}167168	// The cluster's name is what the operator uses to read the live169	// Cluster resource on each pass (cluster convergence). A machine170	// with no cluster manifest has no document to converge.171	clusterName := ""172	if clusterDoc != nil {173		clusterName = clusterDoc.Metadata.Name174	}175176	// The release fetcher outlives any one pass: downloads take177	// minutes, passes take milliseconds, and the fetcher is the one178	// piece of state that connects them (fetch.go).179	f := &fetcher{}180181	// The metrics registry outlives every pass too, because a182	// counter's whole value is that it accumulates (metrics.go).183	// The liveness check reads the loop's busy mark, so the kubelet's184	// liveness probe restarts a loop that is stuck (liveness.go).185	live := newLiveness()186	operatorMetrics, machineLayer := serveMetrics(*metricsAddress, f, live.check)187188	// The loop's wake channel has one slot, so a burst of changes makes189	// one wake (loop.go).190	watcher, err := informer.InClusterAt(localAPIEndpoint(clusterDoc, name))191	if err != nil {192		fatal("in-cluster config for the watches: %v", err)193	}194	wakes := make(chan struct{}, 1)195	f.wake = watch.Signal(wakes)196	objects := watchThisMachine(context.Background(), watcher, client, name, clusterName,197		watch.Signal(wakes), operatorMetrics.WatchRestarted)198	objects.recorder = recorder199200	// The machine's readers open before the first pass, so a change201	// during that pass still sends a wake after it (machineevents.go).202	// A pod from a template older than the /host/etc mount has no203	// directory to watch. The operator then runs without the hosts204	// watch, and hostEntriesCondition reports the missing mount.205	uevents, err := listenForUevents(context.Background())206	if err != nil {207		fatal("listening for uevents: %v", err)208	}209	hosts, err := watchHostsFile(context.Background())210	if err := hostsWatchFailure(err); err != nil {211		fatal("watching %s: %v", hostsPath, err)212	} else if hosts == nil {213		fmt.Fprintf(os.Stderr, "watching %s: the directory does not exist\n", hostsPath)214	}215216	l := &loop{217		objects:     objects,218		name:        name,219		clusterName: clusterName,220		fetcher:     f,221		// The heartbeat outlives every renewal, because it holds the222		// lease it last wrote, and a renewal from that copy needs no223		// read (kubernetes/heartbeat.go).224		heartbeat: kubernetes.NewHeartbeat(name),225		operator:  operatorMetrics,226		layer:     machineLayer,227		wakes:     wakes,228		uevents:   uevents,229		hosts:     hosts,230		live:      live,231		watchFactsTree: func(ctx context.Context) (*factsWatch, error) {232			w, err := machine.WatchFactsTree(ctx, machine.FactsDir)233			if err != nil {234				return nil, err235			}236			return &factsWatch{wake: w.Wake, sync: w.Sync}, nil237		},238	}239	if err := l.run(context.Background(), current); err != nil {240		fatal("%v", err)241	}242}243244func fatal(format string, args ...any) {245	fmt.Fprintf(os.Stderr, format+"\n", args...)246	os.Exit(1)247}
machine-operator/metrics.go 94.1%
1package main23// Layer 3 for the machine operator: the metrics that make this4// program what it is.5//6// Most of these are Machine status fields, and that is the design.7// The status says what the machine is now, and the metric is the8// same fact over time. Console parity says that what init prints9// must also reach Machine status. This file extends that one step10// further, to Prometheus, under one rule: every metric here reads11// the fact that the status writer reads, and never a second source.12// So observeStatus takes the very status value that reconcile is13// about to publish, and observeDevices takes the very device list14// that the ResourceSlice is about to carry.15//16// Three metrics from the plan's table are missing, because the17// machine holds no fact for them yet:18//19//   - liken_upgrade_duration_seconds. Measuring an upgrade from20//     staged to booted needs a staging time that survives the21//     reboot, and the system release record carries no timestamp.22//   - liken_firmware_info and liken_firmware_update_pending. Status23//     reports the firmware's boot entries and nothing about its24//     version. Reading a BIOS version, and updating one, is25//     plans/33-firmware-updates.md.2627import (28	"fmt"29	"os"30	"strings"31	"time"3233	"github.com/liken-sh/liken/liken/api"34	"github.com/liken-sh/liken/liken/kubernetes"35	"github.com/liken-sh/liken/liken/machine"36	"github.com/liken-sh/liken/liken/metrics"37	"github.com/prometheus/client_golang/prometheus"38)3940// serveMetrics builds this operator's whole registry, the contract's41// layer 1 and layer 2 beside this machine's layer 3, and starts the42// listener that answers a scrape. A listener that cannot bind43// reports itself and nothing more: the machine must keep operating44// whether or not anybody watches it. This pod runs on the host45// network, so a port another program already holds is the ordinary46// way that bind fails. /healthz answers health, for the liveness47// probe.48func serveMetrics(address string, f *fetcher, health func() error) (*metrics.Operator, *machineMetrics) {49	o := metrics.NewOperator(component, machine.Version,50		[]string{machineKind}, watchKinds)51	o.SetHealth(health)52	layer := newMachineMetrics(o, f)53	if addr, err := o.Serve(address); err != nil {54		fmt.Fprintf(os.Stderr, "the metrics listener is not serving: %v\n", err)55	} else if addr != nil {56		fmt.Printf("serving metrics on %s/metrics\n", addr)57	}58	return o, layer59}6061// machineMetrics holds this operator's layer 3. One instance lives62// for the life of the process, and the reconcile pass hands it the63// facts it has already gathered.64type machineMetrics struct {65	release       *prometheus.GaugeVec66	bootTimestamp optionalGauge67	changePending *prometheus.GaugeVec68	converged     optionalGauge69	devices       *prometheus.GaugeVec70	lastCrash     optionalGauge71	longestPass   prometheus.Gauge72	longest       float6473	repairs       *prometheus.CounterVec74	passes        *prometheus.CounterVec7576	// classes remembers every device class this machine has77	// published. A class whose devices all disappear is set to zero78	// rather than removed, because the plan asks for a drop that a79	// graph can show, and a series that stops reporting draws a gap80	// instead of a fall.81	classes map[string]bool82}8384// An optionalGauge is a gauge that a pass may leave unpublished. A85// machine with no crash on record must publish no crash timestamp,86// because zero reads as 1970 on a graph, while an absent series87// reads as "nothing to report", which is the fact. A GaugeVec with88// no labels gives exactly that: one sample while a pass sets it, and89// no metric family at all while nothing does.90type optionalGauge struct{ vec *prometheus.GaugeVec }9192func newOptionalGauge(name, help string) optionalGauge {93	return optionalGauge{vec: prometheus.NewGaugeVec(94		prometheus.GaugeOpts{Name: name, Help: help}, nil)}95}9697func (g optionalGauge) set(value float64) { g.vec.WithLabelValues().Set(value) }9899func (g optionalGauge) clear() { g.vec.Reset() }100101// tierOf turns a staged disruption's kind into the tier word the102// contract publishes. liken has four tiers of convergence103// (cluster/changes.go), and only two of them ever wait: an in-place104// change applies on the pass that finds it, and a next-boot change105// waits for no disruption of its own.106func tierOf(kind machine.DisruptionKind) string {107	return strings.ToLower(string(kind))108}109110// pendingTiers are the tiers that can hold a change back. Both are111// published on every pass, so a fleet that is waiting for nothing112// reports zero rather than nothing at all.113var pendingTiers = []machine.DisruptionKind{machine.DisruptionReboot, machine.DisruptionRestart}114115// newMachineMetrics registers layer 3 on the operator's registry.116// The fetcher is a parameter because two of these metrics are117// counters that only the fetcher can total: it is the one thing that118// downloads a release, and it keeps those totals across the passes119// that ask it for its state (fetch.go).120func newMachineMetrics(o *metrics.Operator, f *fetcher) *machineMetrics {121	m := &machineMetrics{122		release: prometheus.NewGaugeVec(prometheus.GaugeOpts{123			Name: metrics.Prefix + "release_info",124			Help: "The release this machine runs and the slot it booted from, as labels on a gauge that is always 1.",125		}, []string{"version", "slot"}),126		bootTimestamp: newOptionalGauge(metrics.Prefix+"machine_boot_timestamp_seconds",127			"When this machine booted, in seconds since the epoch."),128		changePending: prometheus.NewGaugeVec(prometheus.GaugeOpts{129			Name: metrics.Prefix + "machine_change_pending",130			Help: "Staged changes that wait for a disruption, by the tier that applies them.",131		}, []string{"tier"}),132		converged: newOptionalGauge(metrics.Prefix+"machine_converged",133			"1 while the machine runs the spec the cluster holds, and 0 while it does not."),134		devices: prometheus.NewGaugeVec(prometheus.GaugeOpts{135			Name: metrics.Prefix + "devices",136			Help: "Devices this node offers to workloads, by class.",137		}, []string{"class"}),138		lastCrash: newOptionalGauge(metrics.Prefix+"last_crash_timestamp_seconds",139			"When the newest kernel crash this machine still holds records for happened, in seconds since the epoch."),140		longestPass: prometheus.NewGauge(prometheus.GaugeOpts{141			Name: metrics.Prefix + "machine_longest_pass_seconds",142			Help: "The longest reconcile pass since the operator started. The heartbeat stops once the loop is busy for 60 seconds.",143		}),144		repairs: prometheus.NewCounterVec(prometheus.CounterOpts{145			Name: metrics.Prefix + "machine_backstop_repairs_total",146			Help: "Writes by a pass that only the backstop started, by step. Each one names a wake the operator does not send.",147		}, []string{"step"}),148		passes: prometheus.NewCounterVec(prometheus.CounterOpts{149			Name: metrics.Prefix + "machine_passes_total",150			Help: "Reconcile passes, by what started each one: start, watch, machine event, facts, retry, sysctl drift, or backstop.",151		}, []string{"cause"}),152		classes: map[string]bool{},153	}154155	// The two download counters read the fetcher's own totals at156	// scrape time. A counter function suits them because the fetcher157	// already keeps these numbers for its own use, and copying them158	// into a second place on every pass would let the two disagree.159	// Reading them costs one mutex and no I/O, so a scrape still160	// touches nothing but memory.161	downloadBytes := prometheus.NewCounterFunc(prometheus.CounterOpts{162		Name: metrics.Prefix + "release_download_bytes_total",163		Help: "Release artifact bytes this machine has downloaded and verified onto a slot.",164	}, func() float64 { return float64(f.DownloadedBytes()) })165	downloadFailures := prometheus.NewCounterFunc(prometheus.CounterOpts{166		Name: metrics.Prefix + "release_download_failures_total",167		Help: "Release downloads that ended without a complete slot.",168	}, func() float64 { return float64(f.DownloadFailures()) })169170	o.Registry().MustRegister(m.release, m.bootTimestamp.vec, m.changePending,171		m.converged.vec, m.devices, m.lastCrash.vec, downloadBytes, downloadFailures,172		m.longestPass, m.repairs, m.passes)173174	for _, kind := range pendingTiers {175		m.changePending.WithLabelValues(tierOf(kind)).Set(0)176	}177	return m178}179180// observeStatus reads the status that this pass is about to publish.181// Every value here comes from that one struct, so the graph and the182// `kubectl get machine -o yaml` output can never disagree.183// observePass raises the longest pass when took is longer. A nil184// machineMetrics observes nothing.185func (m *machineMetrics) observePass(took time.Duration) {186	if m == nil {187		return188	}189	if took.Seconds() > m.longest {190		m.longest = took.Seconds()191		m.longestPass.Set(m.longest)192	}193}194195// observeWake counts one pass by its cause.196func (m *machineMetrics) observeWake(cause string) {197	if m == nil {198		return199	}200	m.passes.WithLabelValues(cause).Inc()201}202203// backstopRepaired counts one write of a backstop pass.204func (m *machineMetrics) backstopRepaired(step string) {205	if m == nil {206		return207	}208	m.repairs.WithLabelValues(step).Inc()209}210211func (m *machineMetrics) observeStatus(status *machine.MachineStatus) {212	// Reset before the set, because a machine that upgrades or falls213	// back changes both labels. Without the reset, the old release214	// would keep reporting 1 beside the new one, and a fleet panel215	// would count the machine twice.216	m.release.Reset()217	if status.Version.Liken != "" {218		m.release.WithLabelValues(status.Version.Liken, status.Boot.Slot).Set(1)219	}220221	m.bootTimestamp.clear()222	if status.Boot.Time != nil && !status.Boot.Time.IsZero() {223		m.bootTimestamp.set(float64(status.Boot.Time.Unix()))224	}225226	pending := map[string]float64{}227	for _, p := range status.Pending {228		pending[tierOf(p.Kind)]++229	}230	for _, kind := range pendingTiers {231		tier := tierOf(kind)232		m.changePending.WithLabelValues(tier).Set(pending[tier])233	}234235	// SpecConverged is the condition that answers "does this machine236	// run the spec the cluster holds". A pass that could not judge it237	// publishes nothing, because Unknown is not 0.238	m.converged.clear()239	if c := api.FindCondition(status.Conditions, "SpecConverged"); c != nil {240		switch c.Status {241		case api.ConditionTrue:242			m.converged.set(1)243		case api.ConditionFalse:244			m.converged.set(0)245		}246	}247248	m.lastCrash.clear()249	if status.LastCrash != nil && status.LastCrash.Time != nil && !status.LastCrash.Time.IsZero() {250		m.lastCrash.set(float64(status.LastCrash.Time.Unix()))251	}252}253254// observeDevices counts the devices this node offers to workloads,255// by class. It takes the list that the ResourceSlice carries, so the256// count and the offer come from the one sysfs walk the pass made.257func (m *machineMetrics) observeDevices(devices []kubernetes.SliceDevice) {258	counts := map[string]float64{}259	for _, d := range devices {260		counts[deviceClass(d)]++261	}262	for class := range m.classes {263		if _, held := counts[class]; !held {264			counts[class] = 0265		}266	}267	for class, count := range counts {268		m.classes[class] = true269		m.devices.WithLabelValues(class).Set(count)270	}271}272273// deviceClass reads the class word a slice device publishes. The274// buses name a class in a table with a fixed set of words275// (hardware/names.go), so this label stays bounded. A device on a276// bus whose table has no word for it counts as unknown, rather than277// leaving the totals short of the slice.278func deviceClass(d kubernetes.SliceDevice) string {279	if attr, held := d.Attributes["class"]; held && attr.String != nil && *attr.String != "" {280		return *attr.String281	}282	return "unknown"283}284285// machineKind is the resource kind this operator's loop reconciles,286// and the label every layer 2 series carries.287const machineKind = "Machine"
machine-operator/node.go 75.0%
1package main23// The operator's access to its own Kubernetes Node object. The Node4// is the kubelet's record of this machine, and the operator reads5// and writes it for several jobs: mirroring its health onto the6// Machine (conditions.go), reconciling its labels (labels.go),7// reconciling its taints (taints.go), cordoning and draining it ahead8// of a reboot (drain.go), and deleting it to finish a demotion9// (demotion.go).1011import (12	"encoding/json"13	"net/http"1415	"github.com/liken-sh/liken/kubernetes/apiclient"16	"github.com/liken-sh/liken/liken/api"17)1819// nodesPath is the core API's home for Node objects: no group, just20// a version, which is what "core" means in the URL scheme.21const nodesPath = "/api/v1/nodes"2223// nodeObject holds the small part of a Kubernetes Node that the24// operator needs: the labels, where a demoted machine's old role25// still shows; the conditions, where the kubelet's health shows26// (reconcile.go mirrors the Node's Ready condition onto the27// Machine); the cordon state, meaning the unschedulable flag plus28// the annotations that record whether liken set it (drain.go); the29// taints, which the operator rewrites as a whole list under the30// resourceVersion it read (taints.go); and the UID, which ties the31// device inventory's owner reference to this instance of the node32// (dra.go).33type nodeObject struct {34	Metadata struct {35		Name            string            `json:"name"`36		UID             string            `json:"uid"`37		ResourceVersion string            `json:"resourceVersion"`38		Labels          map[string]string `json:"labels"`39		Annotations     map[string]string `json:"annotations"`40	} `json:"metadata"`41	Spec struct {42		Unschedulable bool        `json:"unschedulable"`43		Taints        []nodeTaint `json:"taints"`44	} `json:"spec"`45	Status struct {46		Conditions []api.Condition `json:"conditions"`47	} `json:"status"`48}4950// nodeTaint is one entry of the Node's spec.taints, in the Node's own51// shape rather than the Machine spec's. The value carries omitempty,52// because a taint with no value stores no value field at all, and a53// written empty string would be a value the API server keeps.54type nodeTaint struct {55	Key    string `json:"key"`56	Value  string `json:"value,omitempty"`57	Effect string `json:"effect"`58}5960func getNode(c *apiclient.Client, name string) (*nodeObject, error) {61	n := &nodeObject{}62	if err := c.RequestJSON(http.MethodGet, nodesPath+"/"+name, nil, n); err != nil {63		return nil, err64	}65	return n, nil66}6768// deleteNode deletes the Node instance whose UID the caller read. The69// UID is a precondition of the delete: a Node that k3s registered again70// under the same name after the read has another UID, and the API71// server answers 409 and keeps it. Without the precondition, a pass72// that acted on a stale copy could delete the Node a machine just73// registered.74func deleteNode(c *apiclient.Client, name, uid string) error {75	body, err := json.Marshal(map[string]any{76		"apiVersion":    "v1",77		"kind":          "DeleteOptions",78		"preconditions": map[string]string{"uid": uid},79	})80	if err != nil {81		return err82	}83	return c.RequestJSON(http.MethodDelete, nodesPath+"/"+name, body, nil)84}
machine-operator/outcome.go 97.6%
1package main23// A pass's outcome: what it could not finish, what it wrote, and when4// it asks to run again.5//6// A pass carries on past a failure. A write to the API server that7// fails is logged, a store write that fails becomes a condition, and8// the rest of the pass still runs, because one broken step must not9// hide what the others observed. The pass still needs to try the step10// again. The outcome collects each failure in one place, so the loop11// can set one timer for the retry (retry.go), and a step that waits for12// a deadline can ask the loop to wake it then.13//14// The API helpers do not report into the outcome one by one. The15// pass's client carries an observer (apiclient.WithObserver), which16// hears the final answer to every request the pass sends, including a17// failure that the step that sent it only logs. So no request can fail18// without a record in the outcome. A failure on the machine itself, such as19// a store write or a sysctl, comes from a call that the client never20// sees, and its step reports it with fail.2122import (23	"errors"24	"io/fs"25	"net/http"26	"slices"27	"time"2829	"golang.org/x/sys/unix"3031	"github.com/liken-sh/liken/kubernetes/apiclient"32)3334// failureKind sorts a failure by whether a retry can clear it.35type failureKind int3637const (38	// transient is a failure that can clear by itself: a 5xx, a 429, a39	// timeout, a connection that failed, a busy device. The next try40	// comes soon.41	transient failureKind = iota4243	// lasting is a failure that fails the same way on every try until44	// somebody changes something: a 403, a 422, a sysctl name the45	// kernel does not have. The next try comes later. A fix to the46	// Machine's spec sends a watch event that runs a pass at once, and47	// a fix the operator does not watch, such as an RBAC grant or a48	// webhook, waits for the next try, up to five minutes.49	lasting50)5152func (k failureKind) String() string {53	if k == lasting {54		return "lasting"55	}56	return "transient"57}5859// passFailure is one step that did not finish.60type passFailure struct {61	step string62	kind failureKind63	err  error64}6566// passOutcome is one pass's record. A nil outcome records nothing, so67// a test that drives one step alone can pass nil.68type passOutcome struct {69	failures []passFailure7071	// writes names each write the pass made, to the API server and to72	// the machine.73	writes []string7475	// sysctls is what the kernel reported for each parameter the pass76	// applied, for the loop's check of the sysctls (backstop.go). Nil77	// means the pass did not reach the step.78	sysctls map[string]string7980	// sysctlsMissing names the parameters the pass could not apply81	// because their file does not exist, for the same check.82	sysctlsMissing []string8384	// wake is the earliest time a step asked the loop to run a pass85	// again, or zero.86	wake time.Time8788	// notBefore is the earliest time the API server allows the next89	// try, from the Retry-After of a 429, or zero. liken's client90	// answers a 429 at once, with no wait (kubernetes/apiclient.go), so91	// the wait it asked for falls to the retry.92	notBefore time.Time93}9495// failureCount answers how many failures the pass has recorded so far.96// A nil outcome records none.97func (o *passOutcome) failureCount() int {98	if o == nil {99		return 0100	}101	return len(o.failures)102}103104// fail records a failure on the machine itself. The kind comes from the105// errno, when the error carries one (lastingErrno), or from the io/fs106// error that a check of the code's own answers: a missing file, or a107// name that the code refuses before it reaches the kernel.108func (o *passOutcome) fail(step string, err error) {109	if o == nil || err == nil {110		return111	}112	kind := transient113	var errno unix.Errno114	if errors.As(err, &errno) && lastingErrno(errno) ||115		errors.Is(err, fs.ErrNotExist) || errors.Is(err, fs.ErrPermission) || errors.Is(err, fs.ErrInvalid) {116		kind = lasting117	}118	o.failures = append(o.failures, passFailure{step: step, kind: kind, err: err})119}120121// lastingErrno answers whether a failed call fails the same way on each122// try. A path that does not exist, a value or a name the kernel123// refuses, a sysctl name that runs under a parameter or names a124// directory, a permission, and a read-only file system stay that way125// until somebody changes the machine or the spec. Anything else, such126// as EIO or EBUSY, can clear, and so can an error with no errno at all,127// which retries at the pace it always had.128func lastingErrno(errno unix.Errno) bool {129	switch errno {130	case unix.ENOENT, unix.EINVAL, unix.ENOTDIR, unix.EISDIR, unix.EACCES, unix.EPERM, unix.EROFS:131		return true132	}133	return false134}135136// failSoon records a failure that a retry can clear whatever its error137// says. A read of a file that another process writes, such as init's138// facts, fails with ENOENT until that process writes it, and that139// failure must not wait the five minutes that a missing kernel140// parameter waits.141func (o *passOutcome) failSoon(step string, err error) {142	if o == nil || err == nil {143		return144	}145	o.failures = append(o.failures, passFailure{step: step, kind: transient, err: err})146}147148// wrote records a write to the machine.149func (o *passOutcome) wrote(step string) {150	if o == nil {151		return152	}153	o.writes = append(o.writes, step)154}155156// wakeBy asks the loop to run a pass no later than at.157func (o *passOutcome) wakeBy(at time.Time) {158	if o == nil {159		return160	}161	if o.wake.IsZero() || at.Before(o.wake) {162		o.wake = at163	}164}165166// observe is the pass client's observer. A 404 is not a failure: an167// absent object is a state the pass reads, such as a lease to create.168// A 2xx to a write is a write, even when its answer did not decode,169// because the API server stored it.170//171// A request that succeeds withdraws an earlier 409 on the same method172// and path, and nothing else. The status write and the heartbeat each173// answer a 409 by reading the object again and writing once more, and174// when that second write lands, nothing is left to retry. Any other175// failure stays, because several writes share one path: the taints,176// the labels, the cordon, and the uncordon each patch the Node, and177// one of them that lands says nothing about another that failed.178func (o *passOutcome) observe(answer apiclient.Outcome) {179	if o == nil {180		return181	}182	step := answer.Method + " " + answer.Path183	if answer.Status >= 200 && answer.Status <= 299 && answer.Method != http.MethodGet {184		o.writes = append(o.writes, step)185	}186	switch {187	case answer.Status == http.StatusNotFound:188	case answer.Err == nil:189		o.failures = slices.DeleteFunc(o.failures, func(f passFailure) bool {190			return f.step == step && errors.Is(f.err, apiclient.ErrConflict)191		})192	default:193		o.failures = append(o.failures, passFailure{step: step, kind: statusKind(answer.Status), err: answer.Err})194		if seconds := apiclient.RetryAfterSeconds(answer.Err); seconds > 0 {195			if at := time.Now().Add(time.Duration(seconds) * time.Second); at.After(o.notBefore) {196				o.notBefore = at197			}198		}199	}200}201202// statusKind sorts a failed answer by its status. A 2xx whose body did203// not decode met a connection that failed partway. A 409 means the204// pass wrote from a copy another writer changed, and the next pass205// writes from a fresh one. A 401 means the token on disk expired, and206// the kubelet refreshes it. A 408, a 429, a 5xx, and no answer at all207// are the API server's or the network's trouble. Every other 4xx208// refuses the request itself, and the same request gets the same209// answer until somebody changes the spec or the RBAC.210func statusKind(status int) failureKind {211	switch {212	case status == 0, status >= 500, status >= 200 && status <= 299,213		status == http.StatusConflict, status == http.StatusUnauthorized,214		status == http.StatusRequestTimeout, status == http.StatusTooManyRequests:215		return transient216	}217	return lasting218}
machine-operator/ownedconditions.go 100.0%
1package main23import (4	"github.com/liken-sh/liken/liken/api"5	"github.com/liken-sh/liken/liken/machine"6)78// The condition types this operator writes. A pass starts from the9// conditions it owns and drops every other type, so a condition that10// a newer release wrote does not outlive a rollback. The older phase11// table reads a reason it does not know as Degraded, and a Degraded12// machine holds every other machine's reboot turn, so a stale False13// such as an unplugged adapter's would stall the fleet until a person14// edited the status by hand.15//16// RebootApproved is the one condition another program writes: the17// cluster operator grants a turn with it, and this operator carries18// it unchanged. A type that a pass writes but this list misses would19// vanish at the start of each pass and return at its end, and a test20// (ownedconditions_test.go) fails on that.21var ownedConditionTypes = map[string]bool{22	"Ready":                   true,23	"FactsPublished":          true,24	"SysctlsApplied":          true,25	"HostEntriesApplied":      true,26	"WirelessJoined":          true,27	"StorageReady":            true,28	"ModulesLoaded":           true,29	"ModuleParametersApplied": true,30	"FeaturesReady":           true,31	serioAttachedCondition:    true,32	"NodeHealthy":             true,33	"NodeTaintsApplied":       true,34	"NodeLabelsApplied":       true,35	"NodeCurrent":             true,36	"SpecConverged":           true,37	"ClusterConverged":        true,38	"VersionConverged":        true,39	"CredentialsConverged":    true,40	"ImportsConverged":        true,41	"RebootRequestHonored":    true,42}4344// ownsCondition reports whether a pass keeps a condition of the type.45func ownsCondition(conditionType string) bool {46	return ownedConditionTypes[conditionType] || conditionType == machine.RebootApprovedCondition47}4849// ownedConditions answers the conditions of the types a pass keeps,50// in a new slice.51func ownedConditions(conditions []api.Condition) []api.Condition {52	var kept []api.Condition53	for _, c := range conditions {54		if ownsCondition(c.Type) {55			kept = append(kept, c)56		}57	}58	return kept59}
machine-operator/ownstatus.go 92.3%
1package main23import (4	"bytes"5	"encoding/json"6	"errors"7	"slices"89	"github.com/liken-sh/liken/kubernetes/apiclient"10	"github.com/liken-sh/liken/liken/api"11	"github.com/liken-sh/liken/liken/machine"12)1314// publishOwnStatus is kubernetes.PublishStatus for the machine15// writing about itself, which is the one writer entitled to resolve16// a conflict, rather than give up on the write. A Machine's status17// has exactly two other writers: the rollout conductor, granting and18// reclaiming reboot turns, and the fleet sweep, marking silent19// machines Lost. If one of them wrote between this pass's read and20// its write, this machine's observations are still the freshest21// thing anyone has, because it observes the hardware directly. So22// the answer is to retry against a fresh read, rather than discard23// the pass. The merge honors each condition's owner. The24// conductor's grant carries over from the fresh copy exactly as25// written, present or absent, with its transition time untouched26// (the rollout's stall clock measures from that time), and every27// other field is this pass's own observation. A Lost verdict needs28// no special handling: overwriting it is exactly how a machine29// announces that it is back.30//31// before is the status the object carried when the pass began,32// rendered as the JSON a write would send. When this pass observed33// exactly that, the function writes nothing at all. A settled34// machine's report is the same on every pass, and sending it35// anyway would make the API server, and every etcd leader behind36// it, process a write that changes nothing. The kubelet applies the37// same restraint to Node status, and the machine's liveness does38// not depend on this write anyway, because that is the heartbeat39// lease's job. Skipping against a stale working copy is safe for40// the same reason every skipped event is safe: whatever made the41// server's copy differ arrives on the watch, and the pass it42// triggers sees the difference and writes.43//44// One retry is enough. A second conflict means the object is45// changing faster than this pass can read it, and the write that46// won the race is already queued on the watch, so the pass it47// triggers will publish moments from now.48func publishOwnStatus(r *reader, m *machine.Machine, status *machine.MachineStatus, before []byte) error {49	after, err := json.Marshal(status)50	if err == nil && bytes.Equal(before, after) {51		return nil52	}5354	err = r.publishStatus(m, status)55	if !errors.Is(err, apiclient.ErrConflict) {56		return err57	}58	// This read goes to the API server, not to the watch's copy. It59	// follows a write that lost to another writer, and the copy can60	// still lag behind the write that won.61	fresh, gerr := r.freshMachine(m.Metadata.Name)62	if gerr != nil {63		return err64	}65	status.Conditions = api.RemoveCondition(slices.Clone(status.Conditions), machine.RebootApprovedCondition)66	if grant := api.FindCondition(fresh.Status.Conditions, machine.RebootApprovedCondition); grant != nil {67		status.Conditions = append(status.Conditions, *grant)68	}69	return r.publishStatus(fresh, status)70}
machine-operator/phase.go 100.0%
1package main23import (4	"github.com/liken-sh/liken/liken/api"5	"github.com/liken-sh/liken/liken/machine"6)78// The machine's phase: the whole set of conditions summarized in9// one word.10//11// Conditions are the machine's full account of itself, typed,12// reasoned, and timestamped, and programs consume them (kubectl13// wait, controllers, the Ready roll-up). But a person scanning a14// fleet listing does not want five columns of True and False. They15// want one word per machine that says whether it needs attention.16// Pods solved this with status.phase (Pending, Running, Succeeded,17// Failed), and the Machine borrows this idea exactly. The operator18// derives the phase from the conditions on every pass, and never19// stores it, so it can never disagree with them.20//21// One phase is deliberately missing from this table: Lost. A22// machine cannot derive its own death from its own conditions,23// because if this code is running, the machine is not lost. The24// cluster operator writes Lost on a silent machine's behalf, and25// the machine's own operator overwrites it the moment the machine26// reports again.2728// phasePrecedence orders the phases from most severe to least29// severe. When several conditions point at different phases, the30// machine reports the most severe one. A machine that is both31// waiting on a Manual reboot and failing a sysctl is UpdatePending32// and Degraded at the same time. The listing should show the one33// that needs a person soonest. Downloading comes last because a34// download finishes on its own: a machine that downloads a release35// while a change waits on a reboot, or while something fails, shows36// that instead.37var phasePrecedence = []api.Phase{38	api.PhaseUnknown,39	api.PhaseBooting,40	api.PhaseBlocked,41	api.PhaseUpdating,42	api.PhaseUpdatePending,43	api.PhaseDegraded,44	api.PhaseDownloading,45}4647// conditionPhase maps one condition to the phase it indicates. It48// returns "" when the condition indicates nothing: when it is True,49// or when it is the Ready roll-up, which summarizes the others and50// would double-count. The mapping works by reason, because the51// reasons already distinguish what the boolean status cannot.52// RebootPending and RejectedLastBoot are both "not converged," but53// one resolves with a reboot and the other never will without a54// different edit.55//56// SerioAttached indicates nothing either. An adapter that is unplugged57// or refused leaves the machine working, and its workloads that do not58// use the adapter run as before, so the condition reports the adapter59// and leaves the machine's health to the others.60func conditionPhase(c api.Condition) api.Phase {61	if c.Type == "Ready" || c.Type == serioAttachedCondition || c.Status == api.ConditionTrue {62		return ""63	}64	switch c.Reason {65	case "Joining":66		// A radio the boot handed to the background is still67		// associating. The condition stays False, because it reports68		// what the boot actuated and the join has not finished, but69		// the machine is not degraded: the boot went on without the70		// radio on purpose, and a join takes seconds. The fleet71		// listing must not show a fault for that window: a Degraded72		// that appears on every boot and clears itself teaches73		// people to ignore Degraded.74		return ""75	case "FactsUnreadable":76		// The operator is running but cannot read the facts, so it77		// has no information about the machine it runs on.78		return api.PhaseUnknown79	case "FactsIncomplete":80		// Facts exist but carry no boot record yet. init is still81		// starting up.82		return api.PhaseBooting83	case "RejectedLastBoot", "StagingRejected", "BootMismatch", "MachineStateEphemeral",84		"NoSystemSlots", "NotInstalled", "NoReleaseSource", "VersionNotInCatalog", "DigestMismatch",85		"CredentialsInvalid", "RetractionBlocked":86		// Drift exists, but liken refuses to stage it, or cannot.87		// Time will not fix these; a different edit will. A blocked88		// retraction takes an edit to the cluster's objects rather89		// than to a document: a feature stays enabled while objects90		// only its controller can remove still exist. The91		// version target can get stuck in several ways: no slots to92		// hold a release, a boot that did not come from a slot (so93		// no boot entry could ever run the download), a catalog with94		// no source or without the target, or a download whose bytes95		// do not match the catalog's digest. That last case is96		// corrupt at the source, where refetching cannot change what97		// the server publishes. A malformed credentials Secret has98		// the same shape: only a corrected Secret fixes it.99		return api.PhaseBlocked100	case "Downloading":101		// A release downloads to the inactive slot in the background,102		// and the machine serves its workloads until the reboot that103		// proves the release. The cluster operator counts a machine104		// that downloads as available, so a download takes no slot105		// of the disruption budget, and a download that never ends106		// holds no other machine's turn.107		return api.PhaseDownloading108	case "RebootRequested", "RestartRequested", "DemotionRebooting", "Draining",109		"Proving", "LoadRequested":110		// A disruption is in progress; the machine is in the middle111		// of a change. Draining is the first step of a reboot: the112		// node is cordoned and its workloads are being evicted113		// before the machine goes down (a k3s restart skips this114		// step, and pods survive). So is Proving: a boot's imports115		// stay on trial until the OS pods prove them, which116		// ordinarily takes seconds. LoadRequested is the live load's117		// moment in flight, usually under a second, and a machine118		// applying its spec in place is updating, not degraded.119		return api.PhaseUpdating120	case "RebootPending", "RestartPending", "DemotionPending", "AwaitingTurn",121		"StagedForNextBoot", "AwaitingPodRefresh":122		// A change is staged and waiting, either on a Manual reboot123		// or on the cluster granting this machine its turn. A124		// verified release waiting for its proving reboot reads the125		// same way, because it is waiting on exactly the same things.126		// A change staged for the next boot needs no disruption of its127		// own, so nothing is scheduled to apply it and any later boot128		// does. The listing still shows the machine as pending,129		// because it does not yet run the document the deployment130		// wrote. AwaitingPodRefresh waits the same way: the machine131		// already runs the new release, and only the pod steward's132		// refresh, after a leader boots that release, is left.133		return api.PhaseUpdatePending134	}135	// Anything unrecognized reads as Degraded, deliberately. A136	// reason missing from this table fails visibly in the fleet137	// listing, instead of passing silently as Ready.138	return api.PhaseDegraded139}140141// readyCondition is the Ready roll-up: True exactly when every other142// condition is True. The scan skips any prior Ready, so the previous143// pass's value cannot affect this one. It also skips the conductor's144// grant, because the grant is a permission, not an observation about145// this machine's health, and SerioAttached, for the reason146// conditionPhase gives.147//148// When something is False, the reason is the phase word the149// conditions argue for (decidePhase), so the roll-up and the phase150// can never disagree about how bad things are. A machine waiting on151// its reboot turn reads Ready=False with reason UpdatePending, and152// Degraded appears on the roll-up exactly when the phase says153// Degraded. The message still names a failing condition, which is154// where the detail lives.155func readyCondition(conditions []api.Condition) api.Condition {156	ready := api.Condition{Type: "Ready", Status: api.ConditionTrue, Reason: "Reconciled"}157	for _, condition := range conditions {158		if condition.Type == "Ready" || condition.Type == machine.RebootApprovedCondition ||159			condition.Type == serioAttachedCondition {160			continue161		}162		if condition.Status != api.ConditionTrue {163			ready = api.Condition{164				Type: "Ready", Status: api.ConditionFalse,165				Reason:  string(decidePhase(conditions)),166				Message: condition.Type + " is " + string(condition.Status),167			}168		}169	}170	return ready171}172173// decidePhase reduces the conditions to the single most severe174// phase, or Ready when nothing indicates otherwise.175func decidePhase(conditions []api.Condition) api.Phase {176	argued := map[api.Phase]bool{}177	for _, c := range conditions {178		if phase := conditionPhase(c); phase != "" {179			argued[phase] = true180		}181	}182	for _, phase := range phasePrecedence {183		if argued[phase] {184			return phase185		}186	}187	return api.PhaseReady188}
machine-operator/protection.go 100.0%
1package main23// Storage protection: the disks a workload must never claim.4//5// A disk belongs either to the machine, as a storage role, or to the6// workloads, through DRA, never both. A claim on the system disk would7// hand an unprivileged pod the machine's own root filesystem. The8// facts that init publishes are the only record of which partitions9// back a role, because init claimed the disks before this cluster10// existed. So when the facts do not read, the operator cannot tell a11// system disk from a spare one, and it treats every disk as the12// machine's own. A device with no block node holds nothing a role can13// use, so a GPU or a radio stays on offer while the facts are away.14//15// Three places check the same protection: the inventory that offers a16// device, the prepare call that delivers an allocation, and the17// refresh that rewrites a prepared claim. Each of them checks, because18// an allocation can outlive the offer it came from: a slice write can19// fail, and the scheduler can allocate from a slice the pass has not20// yet withdrawn.2122import (23	"errors"24	"slices"25	"sync/atomic"2627	"github.com/liken-sh/liken/liken/hardware"28	"github.com/liken-sh/liken/liken/machine"29)3031// protection is the set of block devices that back a storage role. Its32// zero value is the protection of a machine whose facts did not read,33// which protects every block device.34type protection struct {35	known  bool36	blocks map[string]bool37}3839// protectionOf reads the protection from the facts. facts.Read returns40// a whole record or none, so a non-nil record names every role, and a41// role in memory names no device. A record whose roles name no disk is42// a complete answer, and it protects nothing.43func protectionOf(facts *machine.MachineStatus) protection {44	if facts == nil {45		return protection{}46	}47	p := protection{known: true, blocks: map[string]bool{}}48	for _, name := range machine.StorageRoleNames {49		if role := facts.Storage.Role(name); role != nil && role.Device != "" {50			p.blocks[role.Device] = true51		}52	}53	return p54}5556// protects answers whether a block device must stay with the machine.57func (p protection) protects(block string) bool {58	return !p.known || p.blocks[block]59}6061// withholds answers whether a delivery carries a block device that62// must stay with the machine.63func (p protection) withholds(d hardware.Delivery) bool {64	return slices.ContainsFunc(d.Blocks(), p.protects)65}6667// platformProtection carries the protection to the DRA plugin, which68// prepares claims on the kubelet's schedule, apart from the reconcile69// pass. main seeds it from the boot's facts before the plugin serves,70// and every pass sets it again from the facts it read. Before the71// seed, it protects every disk.72var platformProtection atomic.Pointer[protection]7374func setPlatformProtection(p protection) {75	platformProtection.Store(&p)76}7778func currentProtection() protection {79	if p := platformProtection.Load(); p != nil {80		return *p81	}82	return protection{}83}8485// errWithheld is an allocated device that is present but delivers a86// block device the machine uses, or may use.87var errWithheld = errors.New("it delivers a disk that backs a storage role, or the facts do not say which disks do")8889// withheldNode is the node a prepared claim's spec names while its90// device is withheld. No such node exists, so the runtime fails to91// create the next container that holds the claim, and the claim's92// pod shows the failure. A pod that started before the device was93// withheld keeps the nodes it received.94const withheldNode = "/dev/liken.sh/device-withheld"
machine-operator/publishing.go 100.0%
1package main23// The publish policy: how one physical device becomes the devices a4// ResourceSlice offers.5//6// A device's sysfs subtree can deliver nodes of more than one kernel7// subsystem. i915 is the case that forced the question: beside its8// drm nodes, the driver registers an i2c bus for every display9// output, the wire that carries DDC and monitor control. A claim10// must not receive both, because a transcoder holds no business on a11// monitor's wire, and the sharing rule must not judge both, because12// the drm nodes share and the raw wires do not.13//14// So the policy publishes one device per subsystem the delivery15// holds, and the allocated device's own name states which nodes a16// claim receives. One subsystem splits further: drm registers the17// card node and the render node for the same silicon, and the two18// carry different authority. Modesetting on the card node belongs to19// one process, and work on the render node divides between many, so20// each publishes as its own device with its own sharing rule.21//22// An audio controller goes the other way. Its jack nodes belong to23// the input subsystem, and they stay with the card, because a jack24// reports the state of an output the same claim plays through. The25// question each time is what a claim on this hardware is, and the26// answer is a device, not a subsystem.27//28// A Bluetooth adapter is the third examined shape. It delivers no29// node of its own, because a program reaches a radio through a30// socket, so the policy names this shape by its driver rather than by31// the kinds it delivers. The device it publishes carries the usbfs32// node.33// On a machine that has /dev/uhid, the device also carries that node,34// because a Bluetooth stack is what it exists for. A machine that has35// /dev/uinput adds that node and the legacy evdev range beside it.36// claimDelivery states the reasoning.37//38// A serio attachment is the fourth examined shape, and the one that39// needs the Machine's own spec. A USB-CEC adapter's serial line40// delivers a tty node alone until init attaches the line41// (init/serio.go). Then the kernel registers the serio port under the42// tty, the CEC adapter under the port, and the remote-control input43// device under the adapter, and the same interface delivers tty, cec,44// and input nodes. publishingserio.go states the policy for them.45//46// The mechanism is generic; the tables below are deliberately not.47// Only a shape somebody examined splits. Everything else publishes48// whole and exclusive, so hardware liken has not met keeps the49// default that milestone 38 set: a device wrongly published as50// shareable hands the same hardware to two workloads and nothing can51// take that back, while a device wrongly published as exclusive costs52// a claim that waits, where a person can see it waiting.5354import (55	"slices"56	"strings"5758	"github.com/liken-sh/liken/liken/hardware"59)6061// publishedDevice is one slice device derived from one physical62// device: the nodes it delivers, the one kind they are, and the facts63// the slice publishes about them. The suffix joins the physical64// device's name to make the published name; the primary keeps the65// bare name, so an allocation that predates a split stays valid.66type publishedDevice struct {67	Suffix      string68	Subsystem   string69	Nodes       []string70	RenderNode  bool71	DisplayNode bool72	Shareable   bool73}7475// graphicsSubsystems are the kernel subsystems a graphics device76// delivers nodes through. drm is the modern interface, and graphics77// is the legacy framebuffer that the kernel's fbdev emulation creates78// for the same hardware. A GPU delivers both, so a rule that accepted79// drm alone would never fire on a real machine.80var graphicsSubsystems = map[string]bool{"drm": true, "graphics": true}8182// companionSubsystems are the non-graphics kinds a graphics device is83// known to deliver, each published as its own exclusive device. i2c-dev84// is the monitor-control bus i915 registers per display output: raw85// wire access, one writer, no arbitration contract. drm_dp_aux_dev is86// the DisplayPort AUX channel, the wire that carries DPCD register87// access and EDID reads to a monitor over a DisplayPort output. The88// kernel registers it only while a display is connected, and like the89// i2c buses it is raw wire access with one writer, so it publishes as90// its own exclusive companion, never shared, and never delivered with91// the GPU's own claim.92var companionSubsystems = map[string]bool{"i2c-dev": true, "drm_dp_aux_dev": true}9394// audioSubsystems are the kinds an audio controller delivers. sound95// is ALSA's own nodes: the control node, the hardware-dependent96// nodes, and one node for each PCM subdevice. input is jack97// detection: ALSA registers an input device for every jack it can98// sense, the kernel puts that device under the card, and each one99// carries an event node. An HDA controller with an HDMI output has100// one jack for each display pin, so a real machine delivers both101// kinds, and a rule that accepted sound alone would fire only on a102// USB audio interface.103var audioSubsystems = map[string]bool{"sound": true, "input": true}104105// displaySuffix names the published device that carries a graphics106// device's card node.107const displaySuffix = "-display"108109// claimDelivery is the delivery walk plus the nodes a claim receives110// from outside the device's own subtree. A Bluetooth adapter's claim111// also receives /dev/uhid when the machine has that node. /dev/uhid112// is the kernel's inlet for a HID stack that runs in userspace:113// BlueZ carries HID over GATT there and writes this node to present114// a BLE peripheral as an input device, while classic HID rides hidp115// inside the kernel and needs no node. So the node exists for a116// Bluetooth stack's use, and the adapter's claim is what a Bluetooth117// stack holds. The node exists only while the machine declares the118// uhid module, and a machine that declares none delivers what it119// delivered before.120//121// The adapter's claim also receives /dev/uinput and the 32 legacy122// evdev nodes. A Bluetooth stack relays a controller's events from123// the evdev node the radio link creates, which appears and vanishes124// with the link, into a uinput virtual device whose minor stays fixed125// while the stack holds it. The relay needs both ends, and its126// container is unprivileged, so both come through the adapter claim it127// already holds. The evdev nodes are stated by number, because the128// kernel registers each one only while a controller is connected, and129// the container must already hold the node a controller registers130// later. uinput is built into the liken kernel, so every machine has131// the node, and a sysfs root without it delivers what it delivered132// before.133func claimDelivery(sysRoot string, d hardware.Device) hardware.Delivery {134	delivery := withoutHeld(hardware.InspectDelivery(sysRoot, d), heldNodes(sysRoot))135	if !bluetoothAdapter(d) {136		return delivery137	}138	if node := hardware.MiscNode(sysRoot, "uhid"); node != "" {139		delivery.Nodes = append(delivery.Nodes, hardware.DeliveredNode{Path: node, Subsystem: "misc"})140	}141	if node := hardware.MiscNode(sysRoot, "uinput"); node != "" {142		delivery.Nodes = append(delivery.Nodes, hardware.DeliveredNode{Path: node, Subsystem: "misc"})143		delivery.Nodes = append(delivery.Nodes, hardware.EvdevNodes()...)144	}145	return delivery146}147148// publishDevices applies the policy to one device and what it149// delivers. Most of the policy reads the delivery alone, because the150// nodes state what the hardware is. A Bluetooth adapter is the shape151// that needs the device itself: its driver is the only fact that152// distinguishes it, because it delivers no node of its own.153//154// The primary device is always first, the display companion follows155// it, and the wire companions follow in sorted subsystem order, so156// the same hardware always publishes the same devices.157//158// The primary alone carries the delivery's bus node. A claim on the159// primary is a claim on the hardware, and a userspace driver needs160// that node to reach the device at all. A companion is one wire161// beside the primary. The bus node opens the whole device, so a162// companion must not carry it.163//164// A delivery with a bus node and nothing else publishes the bus node165// alone. This is the state a userspace driver leaves an interface in166// while it runs: libusb detaches the kernel driver to claim the167// interface, and the kernel driver's nodes leave with it. The bus168// node is the one node such a program opens, and the one node it169// cannot destroy, so a spec refreshed to this shape survives a170// container restart under the same prepared claim. For a device with171// a driver, publishing here does not add it to any slice: the172// inventory gates on the subtree's nodes before it applies this173// policy, so only resolution for a claim that already holds the174// hardware sees this shape. A Bluetooth adapter is the exception the175// inventory makes to that gate.176func publishDevices(d hardware.Device, delivery hardware.Delivery) []publishedDevice {177	if bluetoothAdapter(d) {178		return publishBluetooth(delivery)179	}180	published := splitBySubsystem(delivery)181	if delivery.BusNode == "" {182		return published183	}184	if len(published) == 0 {185		return []publishedDevice{{Nodes: []string{delivery.BusNode}}}186	}187	published[0].Nodes = append(published[0].Nodes, delivery.BusNode)188	return published189}190191// splitBySubsystem sorts one delivery's nodes into the devices the192// policy publishes for them.193func splitBySubsystem(delivery hardware.Delivery) []publishedDevice {194	if len(delivery.Nodes) == 0 {195		return nil196	}197	byKind := map[string][]string{}198	for _, node := range delivery.Nodes {199		byKind[node.Subsystem] = append(byKind[node.Subsystem], node.Path)200	}201	kinds := delivery.Subsystems()202	render := hasRenderNode(delivery)203204	// A node the kernel did not categorize (sysfs's subsystem readlink205	// failed) is hardware nobody has examined. Route the whole delivery206	// to the unknown branch, which publishes it whole and exclusive,207	// applying the milestone 38 default to any hardware liken has not208	// met.209	if slices.ContainsFunc(delivery.Nodes, func(n hardware.DeliveredNode) bool {210		return n.Subsystem == ""211	}) {212		return []publishedDevice{{213			Nodes:      delivery.DevNodes(),214			RenderNode: render,215		}}216	}217218	// The two examined shapes that sort nodes come before the219	// single-kind branch. The third, a Bluetooth adapter, delivers no220	// node to sort, so publishDevices answers it before this. A221	// delivery of drm nodes alone, or of sound nodes alone, answers222	// that branch too, and the examined answer is the specific one: it223	// separates the card node from the render node, and it states what224	// the kernel arbitrates. The single-kind branch below is what is225	// left, and it says nothing about sharing.226	if render && graphicsDevice(kinds) {227		return publishGraphics(byKind, kinds)228	}229230	if slices.Contains(kinds, "sound") && audioDevice(kinds) {231		return publishAudio(delivery)232	}233234	if len(kinds) == 1 {235		return []publishedDevice{{236			Subsystem:  kinds[0],237			Nodes:      delivery.DevNodes(),238			RenderNode: render,239		}}240	}241242	return []publishedDevice{{243		Nodes:      delivery.DevNodes(),244		RenderNode: render,245	}}246}247248// publishGraphics splits one graphics device into the devices the249// policy publishes for it: the render node, the card node, and each250// companion wire.251//252// The render node is the primary. It keeps the bare name, and it is253// the half that shares, because the driver arbitrates concurrent254// clients on it.255//256// The card node publishes as an exclusive companion, because DRM257// master is one per card. The kernel gives modesetting authority to258// one open card node, and a second display program on the same card259// fails when it starts, after the scheduler has already placed its260// pod. An exclusive companion moves that refusal to the scheduler,261// where the second claim waits and a person can see it waiting.262//263// The companion delivers the card node alone. Every device this264// policy publishes carries a disjoint part of the delivery, so the265// allocated name states exactly which nodes the claim receives, and266// the render node keeps one sharing rule rather than one rule per267// device that carries it. A workload that modesets and renders asks268// for both devices, in two requests of one claim.269//270// The framebuffer node is delivered nowhere. It is the kernel's271// legacy console interface, holding it grants display takeover, and272// no workload claims a bare framebuffer.273func publishGraphics(byKind map[string][]string, kinds []string) []publishedDevice {274	var render, cards []string275	for _, node := range byKind["drm"] {276		if isRenderNode(node) {277			render = append(render, node)278			continue279		}280		cards = append(cards, node)281	}282283	published := []publishedDevice{{284		Subsystem:  "drm",285		Nodes:      render,286		RenderNode: true,287		Shareable:  true,288	}}289	if len(cards) > 0 {290		published = append(published, publishedDevice{291			Suffix:      displaySuffix,292			Subsystem:   "drm",293			Nodes:       cards,294			DisplayNode: true,295		})296	}297	for _, kind := range kinds {298		if !companionSubsystems[kind] {299			continue300		}301		published = append(published, publishedDevice{302			Suffix:    "-" + strings.ReplaceAll(kind, "_", "-"),303			Subsystem: kind,304			Nodes:     byKind[kind],305		})306	}307	return published308}309310// publishAudio publishes one audio controller as one exclusive311// device that delivers its whole subtree.312//313// The card is exclusive for the same reason the display node is: in314// practice one program owns it. A sound server such as PipeWire opens315// the card's PCMs and mixes every stream through them, and an316// exclusive device makes a second claimant wait in the scheduler,317// where a person can see it wait. ALSA itself would tolerate sharing,318// because the core gives each PCM subdevice to one opener and refuses319// a second open with EBUSY. But two claims sharing a card have no320// contract over which claim gets which output, so the EBUSY arrives321// at play time instead of at scheduling time, on whichever pod opened322// second.323//324// The jack nodes are delivered with the card, not split off. An i2c325// bus is a wire to separate hardware, and the split withholds it. A326// jack belongs to the outputs the same claim already plays through,327// and reading its state tells a player whether a display is328// connected.329//330// The device names sound as its subsystem, because that is the kind a331// deployment selects an audio controller by. The jack nodes ride332// along under that name.333func publishAudio(delivery hardware.Delivery) []publishedDevice {334	return []publishedDevice{{335		Subsystem: "sound",336		Nodes:     delivery.DevNodes(),337	}}338}339340// publishBluetooth publishes one Bluetooth adapter as one exclusive341// device that delivers the adapter's usbfs node.342//343// An adapter has no device node of its own. hci is a socket344// interface: a Bluetooth stack opens an AF_BLUETOOTH socket and binds345// it to an adapter by index, and sysfs registers no `dev` file346// anywhere the adapter owns. The nodes that do appear under the347// adapter belong to the peripherals connected to it, and the walk348// stops at the bluetooth subtree that holds them. So the test the349// rest of the inventory applies, that the subtree carries device350// nodes, does not see an adapter that works correctly, and no other351// branch of this policy publishes one.352//353// The claim is worth publishing without a node to hand over. It354// states which workload owns the radio, and it holds that workload on355// the machine the radio is plugged into, which is what a Bluetooth356// service in a container needs. The usbfs node is what a userspace357// stack opens if it drives the hardware itself.358//359// Any node left in the subtree belongs to the adapter, because the360// walk already removed the peripherals' nodes, so the delivery361// carries it beside the usbfs node. An adapter with no usbfs node362// publishes nothing, because such a claim would deliver no node at363// all.364//365// The delivery can also carry /dev/uhid, /dev/uinput, and the legacy366// evdev nodes, which claimDelivery adds from outside the subtree, and367// this device is where those nodes reach a claim.368//369// The device is exclusive. The kernel arbitrates nothing between two370// Bluetooth stacks on one radio: both scan, both pair, and each one371// reads a state the other wrote.372func publishBluetooth(delivery hardware.Delivery) []publishedDevice {373	if delivery.BusNode == "" {374		return nil375	}376	return []publishedDevice{{Nodes: append(delivery.DevNodes(), delivery.BusNode)}}377}378379// hasRenderNode reports whether the device delivers a DRM render380// node. The kernel names these /dev/dri/renderD<n>, and they are the381// nodes that do GPU work without display authority: a container that382// holds one can encode, decode, and compute. The render node is also383// what makes sharing safe: the driver arbitrates concurrent clients384// on it, and the lab measured twelve encoders dividing one integrated385// GPU evenly.386func hasRenderNode(delivery hardware.Delivery) bool {387	return slices.ContainsFunc(delivery.DevNodes(), isRenderNode)388}389390// isRenderNode reports whether one drm node is a render node. Every391// other node the drm subsystem registers for a card carries display392// authority with it, so the inverse of this test is the card half.393func isRenderNode(node string) bool {394	return strings.HasPrefix(node, "/dev/dri/renderD")395}396397// graphicsDevice reports whether every kind belongs to the graphics398// stack or to the companions a graphics device is known to deliver. A399// kind outside both is hardware nobody has examined, and such a400// delivery keeps the whole and exclusive default.401func graphicsDevice(kinds []string) bool {402	for _, kind := range kinds {403		if !graphicsSubsystems[kind] && !companionSubsystems[kind] {404			return false405		}406	}407	return true408}409410// audioDevice reports whether every kind belongs to an audio411// controller. A kind outside the table is something else beside the412// card, and such a delivery keeps the whole and exclusive default413// until somebody examines it.414func audioDevice(kinds []string) bool {415	for _, kind := range kinds {416		if !audioSubsystems[kind] {417			return false418		}419	}420	return true421}422423// bluetoothAdapter reports whether a device is a Bluetooth host424// controller on the USB bus. This is the one shape the policy names425// by its driver instead of by the kinds it delivers, because it426// delivers no kind at all. btusb drives every adapter that follows427// the standard USB Bluetooth interface, which covers the dongles and428// the radios built into a board.429func bluetoothAdapter(d hardware.Device) bool {430	return d.Bus == "usb" && d.Driver == "btusb"431}
machine-operator/publishingserio.go 100.0%
1package main23// The fourth examined shape of the publish policy: a serial line that4// init holds a serio attachment for (publishing.go describes the5// policy, and init/serio.go the attachment).67import (8	"slices"9	"strings"1011	"github.com/liken-sh/liken/liken/hardware"12	"github.com/liken-sh/liken/liken/machine"13)1415// publishFor is the policy's entry point. A device that carries a16// serial line a spec.serio entry matches takes the serio shape, and17// every other device takes publishDevices. serio is the list the18// machine holds attachments for (serioInEffect, dra.go).19//20// Every other interface of a matched adapter keeps its own devices,21// but loses the usbfs node. That node opens the whole USB device, so22// a claimant of the Pulse-Eight's HID interface could reset the23// adapter through it with USBDEVFS_RESET, and the reset ends the24// attachment under every CEC claim.25func publishFor(d hardware.Device, delivery hardware.Delivery, serio []machine.SerioAttachment) []publishedDevice {26	if !serioAdapter(d, serio) {27		return publishDevices(d, delivery)28	}29	if slices.Contains(delivery.Subsystems(), "tty") {30		return publishSerio(delivery)31	}32	delivery.BusNode = ""33	return publishDevices(d, delivery)34}3536// serioAdapter reports whether a device is an interface of a USB37// device that a spec.serio entry matches. An interface carries its38// USB device's identity, so every interface of the adapter matches:39// the one with the serial line, and the others beside it.40func serioAdapter(d hardware.Device, serio []machine.SerioAttachment) bool {41	if d.Bus != "usb" {42		return false43	}44	return slices.ContainsFunc(serio, func(a machine.SerioAttachment) bool {45		return a.Matches(d.Vendor, d.Product, d.Serial)46	})47}4849// The suffixes of the two devices an attached CEC adapter publishes.50// Neither device keeps the interface's bare name. Before spec.serio51// declared the adapter, the bare name was the tty's device, and a52// claim allocated to it then must resolve to nothing now, not to the53// CEC node.54const (55	serioCECSuffix   = "-cec"56	serioInputSuffix = "-input"57)5859// serioAbsentNode is the node a claim's spec names while its serio60// device is absent: unplugged, not yet attached, or unbound. No such61// node exists, so the runtime fails to create the container, and the62// kubelet retries it under its restart backoff. The old nodes would be63// worse. An event node carries a fixed device number, and in the64// meantime another input device can take that number, so a restarted65// container would open that device instead.66const serioAbsentNode = "/dev/liken.sh/serio-device-absent"6768// serioFailsClosed reports whether a claim's device that no longer69// resolves must name serioAbsentNode instead of keeping its old nodes.70// It must when the name is one of the two a CEC adapter publishes, and71// when the name is an interface of a matched adapter. The second is a72// claim allocated before spec.serio declared the adapter, to the tty's73// device or to another interface that delivered the usbfs node. Its74// old nodes are the tty and the usbfs node, and a restarted container75// that received them could set another line discipline or reset the76// adapter, and either one ends the attachment under every CEC claim.77func serioFailsClosed(name string, byName map[string]hardware.Device, serio []machine.SerioAttachment) bool {78	if strings.HasSuffix(name, serioCECSuffix) || strings.HasSuffix(name, serioInputSuffix) {79		return true80	}81	device, ok := byName[name]82	return ok && serioAdapter(device, serio)83}8485// publishSerio publishes one attached serial line as the devices its86// driver created.87//88// The tty is never published, before or after the attachment. A pod89// that received it could set another line discipline or write to the90// line, and either one takes the port down under every other claim.91// This is the same rule the inventory applies to a disk that holds a92// storage role. So before the attachment, the line publishes nothing.93//94// The CEC node is one device, with the -cec suffix. It is the bus: a95// claimant sends and receives CEC messages through it. The CEC96// core lets several processes open one adapter, but only one can be97// the exclusive initiator or follower, and two programs that configure98// logical addresses on one adapter undo each other, so it is99// exclusive.100//101// The input node is a second device with the -input suffix. It is the102// TV remote's key presses, which one reader must own, so it is103// exclusive too. It is a device of its own, not part of the CEC104// device, because it answers a different claim: a remote control claims it105// the way it claims a Bluetooth remote. This is the opposite of the106// audio shape, where a jack reports on the output the same claim107// plays through. Both devices carry the adapter's attributes, so a108// claim can pair them with matchAttribute on address.109//110// The usbfs node is delivered nowhere. Through it, a claimant could111// detach cdc_acm from the interface, which hangs up the tty and ends112// the port under the other claim. Any other kind of node under the113// line is hardware nobody has examined, and it is withheld with the114// tty. The CEC core registers no LIRC node for the input device, so115// the delivery holds none to withhold.116func publishSerio(delivery hardware.Delivery) []publishedDevice {117	var cec, input []string118	for _, node := range delivery.Nodes {119		switch node.Subsystem {120		case "cec":121			cec = append(cec, node.Path)122		case "input":123			input = append(input, node.Path)124		}125	}126	if len(cec) == 0 {127		return nil128	}129	published := []publishedDevice{{Suffix: serioCECSuffix, Subsystem: "cec", Nodes: cec}}130	if len(input) > 0 {131		published = append(published, publishedDevice{Suffix: serioInputSuffix, Subsystem: "input", Nodes: input})132	}133	return published134}
machine-operator/rebinding.go 92.9%
1package main23// Binding the kernel driver again after a claim ends.4//5// A program that communicates with a USB device through libusb6// cannot share an interface with a kernel driver. The kernel binds7// one driver to an interface at a time, so libusb detaches the kernel8// driver first and claims the interface for itself. Network UPS Tools does this to9// a UPS that usbhid binds. When the program stops, the interface10// keeps no driver, because nothing in the kernel binds one again.11//12// An interface with no driver leaves the node's ResourceSlice, since13// liken publishes driven hardware only (dra.go gives the rule). The14// driver's nodes leave with it, so the next pod's prepare fails: the15// allocated name matches nothing this machine publishes now. Without16// a repair, the device stays out of the inventory until somebody17// unplugs it and plugs it in again, or reboots the machine.18//19// Unprepare is the one safe place for the repair. While a pod runs,20// an interface with no driver under a prepared claim is the correct21// state, and it is the state the program in the pod needs. A22// reconcile pass that bound a driver to every interface with none23// would take the interface away from a running program. The kubelet24// calls unprepare after the last pod that holds the claim stops, so25// the grant is over at that moment and the hardware belongs to the26// machine again.27//28// The repair writes the interface's address to the USB bus's29// drivers_probe attribute, for example "3-4:1.0". A write to one30// driver's bind attribute would need that driver's name, and the walk31// reports no driver here, because the interface has none. Sysfs keeps32// no record of the driver an interface had. drivers_probe runs the33// kernel's own match over every registered USB driver, the same match34// that runs at enumeration, so the interface ends in the state a35// fresh boot gives it.36//37// Every failure here prints one line, and the repair stops. An error38// in the unprepare answer makes the kubelet call unprepare again, and39// the pod stays in Terminating until a call succeeds. A device that40// keeps no driver is a problem for the next pod that claims it, and41// it must never block the deletion of a pod that is already finished.4243import (44	"encoding/json"45	"fmt"46	"os"47	"path/filepath"48	"strings"4950	"github.com/liken-sh/liken/liken/hardware"51)5253// rebindClaimDevices binds a kernel driver again to each USB54// interface that one claim held.55//56// The kubelet's unprepare call carries the claim's UID and nothing57// else, and the claim can be deleted before the call arrives, so a58// read from the API server can answer nothing. The claim's own CDI59// spec file is the durable record. Prepare names each CDI device60// "<claim UID>-<allocated name>", so the file holds every device name61// the claim was allocated. A file that is already gone belongs to an62// unprepare that succeeded, and the retry needs no repair.63//64// devices supplies the sysfs walk. The caller shares one walk across65// every claim in a request, and builds it only when a claim has a66// spec file to act on.67func rebindClaimDevices(claimUID string, devices func() map[string]hardware.Device) {68	spec, err := claimSpec(claimUID)69	if os.IsNotExist(err) {70		return71	}72	if err != nil {73		fmt.Fprintf(os.Stderr, "dra: reading claim %s to bind its drivers again: %v\n", claimUID, err)74		return75	}76	for _, granted := range spec.Devices {77		allocated, ok := strings.CutPrefix(granted.Name, claimUID+"-")78		if !ok {79			continue80		}81		device, ok := resolveHardware(allocated, devices())82		if !ok {83			fmt.Fprintf(os.Stderr, "dra: claim %s held device %s, which this machine does not publish now\n",84				claimUID, allocated)85			continue86		}87		if err := rebindInterface(device); err != nil {88			fmt.Fprintf(os.Stderr, "dra: binding a driver to %s again: %v\n", device.Address, err)89		}90	}91}9293// claimSpec reads one claim's CDI spec, under the lock that every94// writer of these files takes.95func claimSpec(claimUID string) (cdiSpec, error) {96	cdiWrites.Lock()97	defer cdiWrites.Unlock()98	raw, err := os.ReadFile(cdiSpecPath(claimUID))99	if err != nil {100		return cdiSpec{}, err101	}102	var spec cdiSpec103	err = json.Unmarshal(raw, &spec)104	return spec, err105}106107// resolveHardware maps an allocated device name back to the physical108// device it names. The rules are the ones resolveAllocated applies: a109// bare name is a device's own name, and every other name is a110// companion, which is a bare name plus a suffix. A companion is one111// wire of its parent, on the same physical device, so both names112// answer with the same hardware. The longest bare name that fits113// wins, because one device's name can be the start of another's.114func resolveHardware(name string, byName map[string]hardware.Device) (hardware.Device, bool) {115	if device, ok := byName[name]; ok {116		return device, true117	}118	var match string119	var found hardware.Device120	for bare, device := range byName {121		if !strings.HasPrefix(name, bare+"-") || len(bare) <= len(match) {122			continue123		}124		match, found = bare, device125	}126	return found, match != ""127}128129// rebindInterface asks the kernel to match a driver to one USB130// interface again.131//132// It writes nothing in three cases. A device on another bus has no133// drivers_probe of this kind. An address with no colon names a whole134// USB device rather than one of its interfaces, and drivers bind to135// interfaces. A device that reports a driver has one bound now, which136// is what a pod that never detached the driver leaves behind.137func rebindInterface(device hardware.Device) error {138	if device.Bus != "usb" || !strings.Contains(device.Address, ":") || device.Driver != "" {139		return nil140	}141	// The attribute exists for as long as the bus does, so the open142	// creates nothing. A tree with no drivers_probe is a kernel143	// without USB support, and there is no interface to repair.144	probe, err := os.OpenFile(filepath.Join(draSysfsRoot, "bus", "usb", "drivers_probe"), os.O_WRONLY, 0)145	if err != nil {146		return err147	}148	if _, err := probe.WriteString(device.Address); err != nil {149		probe.Close()150		return err151	}152	return probe.Close()153}
machine-operator/rebootrequest.go 92.6%
1package main23// The reboot a person asks for, with nothing to apply.4//5// Every other reboot in liken is the tail of a staged document: the6// operator finds drift, stages the new bytes, and the boot actuates7// them (converge.go). That leaves no way to reboot a machine that8// has no drift at all, and a machine with no shell and no SSH server9// has no other way to get one. Two cases need one anyway. A kernel10// driver that bound the wrong device releases it only at boot, and a11// machine used as a testbed needs a clean boot between experiments.12// The only other way to start that boot is the power button, which13// takes no turn from the cluster, cordons nothing, and drains14// nothing.15//16// The request is an annotation on the Machine17// (machine.RequestRebootAnnotation), valued with the identity of the18// boot that is running now. That value is what makes the request19// one-shot: the boot that comes back has a different identity, so20// the annotation no longer names the running boot, and nobody has to21// clear it for a later request to work. The Cluster takes an22// immediate release check the same way, through the23// liken.sh/check-releases annotation: an annotation asks for one24// action, where a spec field declares a standing state.25//26// From the gate onward this is an ordinary disruption. The decision27// below builds the same convergence value the documents build and28// hands it to gateDisruption, so rebootPolicy, the rollout29// conductor's turn, and the drain all apply unchanged, and30// status.pending carries the request where liken approve-reboot can31// find it.32//33// What the request never does is stage anything. A boot's slot34// bookkeeping reads the staged system release on machineState, not35// the reboot intent (armProvingBoot in init/proving.go), so a boot36// that no download preceded arms no trial and promotes no slot. The37// machine comes back on the documents it already ran.3839import (40	"fmt"41	"time"4243	"github.com/liken-sh/liken/liken/api"44	"github.com/liken-sh/liken/liken/machine"45)4647// rebootRequestCondition reports whether a requested reboot is still48// outstanding. True covers both settled states: no request at all,49// and a request whose reboot has happened.50const rebootRequestCondition = "RebootRequestHonored"5152// decideRebootRequest is the whole decision, as a pure function over53// the annotation and the boot record. The cases run in this order:54//55//  1. No annotation: nothing was asked for. The message still names56//     this boot's identity, because that is the value an annotation57//     has to carry, and kubectl describe machine is where a person58//     goes to find it. Without it, asking for a reboot would need59//     the liken CLI and nothing else would do.60//  2. An annotation, but no boot record, or a boot record with no61//     time: the machine publishes no identity for an annotation to62//     name yet, so the verdict is Unknown rather than a guess.63//  3. The annotation names some other boot. This is either a spent64//     request, which is what every fulfilled request looks like65//     afterward, or a value a person mistyped. The message reports66//     both values, so a mistyped one is visible where the person is67//     already looking.68//  4. The annotation names this boot: gate the reboot on the policy69//     and the turn.70func decideRebootRequest(m *machine.Machine, facts *machine.MachineStatus, t turn) convergence {71	request := m.Metadata.Annotations[machine.RequestRebootAnnotation]72	bootID := ""73	if facts != nil {74		bootID = machine.BootID(facts.Boot)75	}76	if request == "" {77		message := "no reboot has been requested"78		if bootID != "" {79			message = fmt.Sprintf("no reboot has been requested; to ask for one, set the %s annotation to %.12s, this boot's identity",80				machine.RequestRebootAnnotation, bootID)81		}82		return convergence{condition: converged(rebootRequestCondition, "NothingRequested", message)}83	}84	if bootID == "" {85		return factsIncomplete(rebootRequestCondition)86	}87	if !machine.RebootRequestNames(request, bootID) {88		return convergence{condition: converged(rebootRequestCondition, "RequestSpent",89			fmt.Sprintf("the %s annotation names %s, and this boot is %.12s; the request applies to a boot that is no longer running",90				machine.RequestRebootAnnotation, request, bootID))}91	}9293	c := convergence{hash: bootID}94	gateDisruption(&c, rebootRequestCondition, m.Spec.RebootPolicyOrDefault(), t, false,95		m.Metadata.Annotations[machine.ApproveDisruptionAnnotation],96		"a reboot of this machine, with no change to apply",97		fmt.Sprintf("a reboot is requested for this boot (%.12s); rebootPolicy is Manual, so approve the reboot (or set rebootPolicy: Auto) to let it run", bootID),98		fmt.Sprintf("a reboot is requested for this boot (%.12s); waiting for the cluster to grant a reboot turn", bootID),99		fmt.Sprintf("reboot requested for this boot (%.12s); nothing is staged, so the machine comes back on the documents it runs now", bootID))100	return c101}102103// carryOutRebootRequest performs the decision's one side effect. The104// documents go through carryOutConvergence, which writes a store and105// then the intent. This request has no store, and giving it one would106// open a path to the write a plain reboot must never make. dir is the107// operator's intent channel to init, named rather than assumed so a108// test can read what lands in it.109//110// The intent deliberately carries no manifest hash. That field names111// the staged document a reboot applies, and this reboot applies112// none.113func carryOutRebootRequest(dir string, conv convergence, now time.Time, out *passOutcome) api.Condition {114	if !conv.requestReboot {115		return conv.condition116	}117	intent := &machine.RebootIntent{118		Reason:      fmt.Sprintf("a reboot was requested for boot %.12s", conv.hash),119		RequestedAt: now,120	}121	if err := machine.WriteRebootIntent(dir, intent); err != nil {122		out.fail("writing the requested reboot's intent", err)123		return api.Condition{Type: rebootRequestCondition, Status: api.ConditionFalse,124			Reason: "RequestFailed", Message: err.Error()}125	}126	out.wrote("writing the requested reboot's intent")127	fmt.Printf("requested a reboot that a person asked for; nothing is staged to apply (boot %.12s)\n", conv.hash)128	return conv.condition129}
machine-operator/reconcile.go 98.3%
1package main23// This file is the working half of the reconcile loop. Each pass4// observes the machine, acts on the spec, and reports status.56import (7	"encoding/json"8	"fmt"9	"slices"10	"time"1112	"github.com/liken-sh/liken/liken/api"13	"github.com/liken-sh/liken/liken/cluster"14	"github.com/liken-sh/liken/liken/kubernetes"15	"github.com/liken-sh/liken/liken/machine"16)1718// factsTree is the facts init publishes, read through the operator's19// read-only /run/liken hostPath. It is a package variable so a test can20// point it at a tempdir instead of the machine's real /run.21var factsTree = machine.FactsTree{Dir: machine.FactsDir}2223// sysctlRoot is the kernel's tuning interface the pass writes. It is a24// package variable for the same reason as factsTree: a test of a whole25// pass points it at a tempdir, so the test never writes the host's26// kernel parameters.27var sysctlRoot = machine.SysctlDir2829// reconcile is one full pass of the operator's job, always30// starting from the current state: read the facts init left, apply31// the spec's sysctls, read back what actually holds, and publish32// all of it as status. It deliberately keeps no memory between33// passes. Every value in the status it writes was observed moments34// ago, which is what the Kubernetes convention means by status35// being reconstructible.36//37// The pass returns the status write's error. Everything above that38// write reports itself as a condition, which is a fact about the39// machine. A failed status write is different: it is a fault in the40// operator, and it means nobody outside this pod can see what the pass41// observed. That is what the layer 2 error counter counts42// (metrics.go).43//44// The pass also records into out each step it did not finish, each45// write it made, and the earliest time a step asks to run again46// (outcome.go). The pass's client reports every answer from the API47// server there, and each step on the machine reports its own failures,48// so the loop can retry the pass when something failed (retry.go).49func reconcile(r *reader, m *machine.Machine, clusterName string, f *fetcher, mm *machineMetrics, out *passOutcome) error {50	now := time.Now()51	r = r.observedBy(out)52	c := r.client5354	// This records what the object held before this pass touched55	// anything. It must be captured now, because it cannot be56	// captured later: SetCondition edits the slice it is given, so57	// the conditions this pass builds share their backing array with58	// m.Status, and by publish time the two are the same list. This59	// snapshot is what lets the publish step below skip a write that60	// would change nothing.61	before, _ := json.Marshal(&m.Status)6263	// The Events of this pass compare the status it writes with the64	// stored one (events.go), so the stored conditions need their own65	// copy for the same reason.66	stored := m.Status67	stored.Conditions = slices.Clone(m.Status.Conditions)68	notes := machineEvents{r.recorder, machineReference(m)}6970	status := &machine.MachineStatus{}7172	facts, err := factsTree.Read()73	if err == nil {74		*status = *facts75	}76	out.failSoon("reading the facts", err)77	// The pass starts from the conditions this release owns, so one78	// that a newer release wrote drops here (ownedconditions.go).79	status.Conditions = api.SetCondition(ownedConditions(m.Status.Conditions), factsCondition(err), now)8081	// The operator's own existence is the evidence that promotes a82	// staged cluster document. If this line runs, the machine joined83	// its cluster under whatever document this boot ran (cluster.go).84	// The same evidence, together with the version this boot85	// reported in the facts, promotes a system release's proving86	// boot (release.go).87	settleClusterLifecycle(machine.MachineStateDir, cluster.ClusterManifestPath, facts, out)88	settleSystemReleaseLifecycle(machine.MachineStateDir, facts, out)8990	// The imports lifecycle settles on its own evidence. This is not91	// this operator's existence, but the Ready condition of every OS92	// container on this node, because the trial covers every93	// tarball the boot imported, not only the one this pod runs from94	// (imports.go).95	status.Conditions = api.SetCondition(status.Conditions,96		settleImportsLifecycle(r, machine.MachineStateDir, m.Metadata.Name, facts, out), now)9798	// Both sets of kernel parameters, on every pass. Applying the99	// settings every liken machine holds is what returns a parameter100	// that something else on the machine changed, within one pass and101	// without a reboot. status.sysctls reports the two together, so an102	// operator sees every parameter liken sets and its actual value in103	// one place.104	sysctls, missing, defaultsErr, specErr := applySysctls(sysctlRoot, machine.OSSysctls, m.Spec.Sysctls, out, r.sysctls)105	status.Sysctls = sysctls106	if out != nil {107		out.sysctls, out.sysctlsMissing = sysctls, missing108	}109	status.Conditions = api.SetCondition(status.Conditions, sysctlsCondition(defaultsErr, specErr), now)110111	// podStale answers whether this pod's own template predates the112	// release it is running (staleness.go). A follower that reboots113	// first always runs its new binary inside the old pod spec for a114	// while, because the OS DaemonSets update on OnDelete and only a115	// leader's boot rewrites the AddOn manifests that produce a fresh116	// template. hostEntriesCondition reads this verdict below to judge117	// a missing mount as that ordinary lag instead of a fault.118	podStale := ownPodIsStale(r, m.Metadata.Name, status.Version.Liken)119120	// Host entries reconcile live too, under the same write-on-121	// divergence rule (hosts.go). The hostname is the Machine's own122	// name rather than a read of the host's hostname, because init123	// derives the kernel's hostname from this same field124	// (unix.Sethostname(m.Metadata.Name) in init/main.go) and a125	// Kubernetes object's name never changes once it exists, so the126	// two can never disagree the way they could if this program127	// depended on the pod's network namespace carrying the host's UTS128	// namespace along with it.129	hostEntries, hostsErr := applyHostEntries(hostsPath, m.Metadata.Name, m.Spec.Network.HostEntries, out)130	status.HostEntries = hostEntries131	status.Conditions = api.SetCondition(status.Conditions,132		hostEntriesCondition(m.Spec.Network.HostEntries, hostsErr, podStale), now)133134	// Modules judge what the boot reported, not what the spec asks135	// for now. A freshly declared module has no outcome yet; it136	// stays SpecConverged's concern until a reboot loads it. This137	// condition is the other half of that split: SpecConverged can138	// be True, meaning the boot ran the manifest, while this139	// condition is False, because a spec the boot honored can still140	// name modules the booted image never carried.141	status.Conditions = api.SetCondition(status.Conditions,142		modulesCondition(status.Modules), now)143144	// ModulesLoaded keeps its meaning: a module that loaded is145	// loaded, whatever happened to its parameters. The parameters146	// report through their own condition, so each answers one147	// question and a person reads two plain answers instead of one148	// mixed one.149	status.Conditions = api.SetCondition(status.Conditions,150		moduleParametersCondition(m.Spec.ModuleParameters, status.Modules), now)151152	// The serio attachments report what init holds now, not what the153	// boot did: init's serio watch rewrites status.serio when an154	// adapter is plugged in or unplugged. An entry declared since the155	// last pass has no report yet, and SpecConverged carries it until156	// the live load declares it to init.157	status.Conditions = api.SetCondition(status.Conditions,158		serioCondition(status.Serio), now)159160	// The unclaimed-hardware report deliberately has no condition,161	// even though it looks like modules and features at first162	// glance. The difference is that those judge requests: a163	// declared module that did not load is a broken promise.164	// Unclaimed devices are hardware that nobody has asked anything165	// about, and staying undriven is a normal, permanent state.166	// Every QEMU guest carries a VGA adapter that no server image167	// drives, and a headless machine with a GPU leaves it undriven168	// by design. A condition would read every one of those machines169	// as Degraded forever. So the report follows the same pattern as170	// the undeclared-disk report instead: inventory in the status171	// (hardware.unclaimed arrives live from the facts, because172	// init's uevent watcher republishes on every hot-plug), loud on173	// the console, and judged by nobody until a person declares the174	// driver. At that point, status.modules judges the request.175176	// Features judge what the boot reported, on the same terms as177	// modules. The split from ClusterConverged is the point of this178	// design: the cluster document's hash proves this boot ran the179	// document that enables a feature, and this condition proves the180	// booted image could carry it out. In the middle of a rollout,181	// the fleet runs mixed releases, so the answers legitimately182	// differ from machine to machine.183	status.Conditions = api.SetCondition(status.Conditions,184		featuresCondition(status.Features), now)185186	// The radios judge the boot's report, not the spec. A join187	// happens once, at boot, on the same terms as a module load, so188	// a freshly declared wireless entry is SpecConverged's concern189	// until a reboot joins it. A machine that joined nothing and190	// still reached this line reached it over some other interface,191	// which is the degraded case plans/completed/62-wifi.md describes.192	status.Conditions = api.SetCondition(status.Conditions,193		wirelessCondition(status.Network.Interfaces), now)194195	// Storage compares the spec's declared roles against the facts'196	// report of where each is actually backed. The operator cannot197	// observe the disks directly, because claiming happened before198	// this cluster existed, so init's facts are the only source, and199	// this condition checks them against the spec.200	status.Conditions = api.SetCondition(status.Conditions,201		storageCondition(m.Spec.Storage, status.Storage), now)202203	// t is this machine's standing with the rollout conductor. A204	// standalone machine reboots whenever it needs to. A cluster205	// member reboots only on a granted turn. The grant is a206	// condition the conductor wrote onto this Machine (rollout.go).207	// This operator reads it, carries it along in its own status208	// writes, and never sets or clears it.209	t := turnStandalone210	if clusterName != "" {211		t = turnAwaiting212		if g := api.FindCondition(m.Status.Conditions, machine.RebootApprovedCondition); g != nil && g.Status == api.ConditionTrue {213			t = turnGranted214		}215	}216217	// This reads the machine's own Node once. The read serves three218	// purposes: the NodeHealthy condition, demotion cleanup, and the219	// cordon state the drain works through. The read can fail220	// without being a problem, because during a demotion the Node is221	// deleted and not yet re-registered, and while the API server is222	// down the Node's copy does not answer (watches.go). A pass where223	// the read fails simply skips all three, and the next pass224	// settles them.225	node, nodeErr := r.node(m.Metadata.Name)226227	// The device inventory converges on the same cadence as228	// everything else: one sysfs walk per pass, published as this229	// node's ResourceSlice (dra.go). It waits on the Node read,230	// because the slice is owned by the Node's UID. A pass without a231	// Node, during a demotion, has no owner to attach inventory to,232	// and skipping is correct: the old slice is being233	// garbage-collected along with the old Node. The facts supply234	// the storage roles, which is what keeps the machine's own disks235	// out of the offer.236	//237	// The serio list is the spec's and the boot record's together238	// (dra.go), set here for the DRA plugin as well, so the inventory239	// and the claims it prepares withhold the same serial lines.240	//241	// The facts set the protection the same way: the disks that back242	// a storage role, or every disk when the facts did not read243	// (protection.go).244	serio := serioInEffect(m.Spec.Serio, facts)245	setDeclaredSerio(serio)246	setPlatformProtection(protectionOf(facts))247	if nodeErr == nil {248		_ = publishDeviceInventory(r, node, facts, serio, mm)249	}250251	// The claims the kubelet already prepared get the same treatment,252	// because a device that enumerates again moves the nodes a claim253	// delivers (cdi.go). This runs without a Node, because a prepared254	// claim is a file on this machine, and containerd reads that file at255	// every container creation.256	refreshCDISpecs(draSysfsRoot, out)257258	// Convergence checks whether the cluster's copy of each document259	// matches what this boot actuated. If not, it stages the260	// difference for the next boot (converge.go for the Machine,261	// cluster.go for the Cluster, release.go for the version262	// target, registries.go for the credentials). The decisions are263	// pure functions. carryOutConvergence performs their side264	// effects against each document's own store. The rejection265	// records come from the durable store, not from facts, because the266	// store is the rejections' authority. A rejection cleared in the267	// middle of a boot, by an edit that reverted, must unblock a retry268	// the moment it lands, and the store carries that change at once.269	// Every decision passes270	// through the disruption gate on its way to its side effects,271	// and the gate depends on the order in which the documents272	// converge (see disruptions).273	disr := &disruptions{events: notes, out: out}274	machineStore := machine.MachineManifests(machine.MachineStateDir)275	machineRejection, _ := machineStore.LoadRejection()276	conv := disr.gate(r, node, nodeErr, t, now,277		decideConvergence(m, facts, machineRejection, readStagedHash(machineStore), t))278	status.Conditions = api.SetCondition(status.Conditions,279		carryOutConvergence(conv, machineStore, machine.OperatorRunDir, "spec", now, out), now)280	// Each gated document also reports itself in status.pending, so281	// a client that needs the staged hash has a field instead of a282	// condition message to read. The list rebuilds on every pass,283	// like the rest of status.284	if conv.pending != nil {285		status.Pending = append(status.Pending, *conv.pending)286	}287288	// The cluster document converges through the same machinery, per289	// machine. This machine stages its own copy and reboots on its290	// own policy, and this condition is where the fleet's temporary291	// disagreement about the Cluster becomes visible. A machine with292	// no cluster document carries no operator-authored documents at293	// all, so the version target and the registry credentials also294	// converge only on a cluster member.295	var liveCluster *cluster.Cluster296	if clusterName != "" {297		clusterStore := machine.ClusterManifests(machine.MachineStateDir)298		var cconv convergence299		cconv, liveCluster = convergeClusterDocument(r, clusterStore, clusterName, m, facts, t)300		cconv = disr.gate(r, node, nodeErr, t, now, cconv)301		status.Conditions = api.SetCondition(status.Conditions,302			carryOutConvergence(cconv, clusterStore, machine.OperatorRunDir, "cluster document", now, out), now)303		if cconv.pending != nil {304			status.Pending = append(status.Pending, *cconv.pending)305		}306307		// The version target reads the live Cluster's release feed,308		// so it can converge only on a pass that read the Cluster.309		if liveCluster != nil {310			systemStore := machine.SystemReleases(machine.MachineStateDir)311			vconv := disr.gate(r, node, nodeErr, t, now,312				convergeSystemRelease(systemStore, liveCluster, m, facts, f, t, out))313			status.Conditions = api.SetCondition(status.Conditions,314				carryOutConvergence(vconv, systemStore, machine.OperatorRunDir, "system release", now, out), now)315			if vconv.pending != nil {316				status.Pending = append(status.Pending, *vconv.pending)317			}318		}319320		credentialsStore := machine.RegistryCredentialsStore(machine.MachineStateDir)321		rconv := disr.gate(r, node, nodeErr, t, now,322			convergeRegistryCredentials(r, credentialsStore, m, facts, t))323		status.Conditions = api.SetCondition(status.Conditions,324			carryOutConvergence(rconv, credentialsStore, machine.OperatorRunDir, "registry credentials", now, out), now)325		if rconv.pending != nil {326			status.Pending = append(status.Pending, *rconv.pending)327		}328	}329330	// A reboot a person asked for, which no document requires331	// (rebootrequest.go). It goes through the same gate as every332	// staged document, so it takes its turn, cordons, and drains333	// like the rest. It goes last because the gate's order decides334	// which entry liken approve-reboot offers first, and a staged335	// document is the more useful answer: approving it reboots the336	// machine and satisfies the request along the way. This runs337	// outside the cluster block above, because a standalone machine338	// can be asked to reboot too.339	rreq := disr.gate(r, node, nodeErr, t, now, decideRebootRequest(m, facts, t))340	status.Conditions = api.SetCondition(status.Conditions,341		carryOutRebootRequest(machine.OperatorRunDir, rreq, now, out), now)342	if rreq.pending != nil {343		status.Pending = append(status.Pending, *rreq.pending)344	}345346	if nodeErr == nil {347		// NodeHealthy mirrors the Node's Ready condition onto the348		// Machine. This catches the one failure the heartbeat349		// cannot: this operator runs on the host's network and talks350		// to the API directly, so it can keep reporting a351		// healthy-looking machine while the kubelet under it is352		// dead. The kubelet's own heartbeat, its node lease, which353		// the node controller turns into the Node's Ready condition,354		// is the evidence that the machine is actually serving the355		// cluster, not merely reachable.356		status.Conditions = api.SetCondition(status.Conditions, nodeHealthyCondition(node), now)357358		// Node taints reconcile live, and go before the labels359		// (taints.go). The taints patch names the resourceVersion this360		// pass read, and the labels patch names none. Sending the361		// taints patch first spends that precondition while the362		// version it names is still current. The labels patch cannot363		// conflict with the version bump the taints patch causes,364		// because it states no version to conflict with.365		status.Conditions = api.SetCondition(status.Conditions,366			carryOutNodeTaints(c, m.Metadata.Name, decideNodeTaints(m.Spec.NodeTaints, node)), now)367368		// Node labels reconcile live, like sysctls, but against the369		// Node object instead of the kernel (labels.go). This370		// reapplies what the spec declares, and removes what it took371		// out, which the kubelet never does on its own.372		status.Conditions = api.SetCondition(status.Conditions,373			carryOutNodeLabels(c, m.Metadata.Name, decideNodeLabels(m.Spec.NodeLabels, node)), now)374375		// Demotion cleanup (demotion.go). A follower whose Node376		// object still claims control-plane was just demoted. That377		// stale Node carries a registered etcd membership, so the378		// operator must delete it.379		d := decideDemotion(status.Role, node.Metadata.Labels, m.Spec.RebootPolicyOrDefault(), t)380		condition := carryOutDemotion(c, machine.OperatorRunDir, node, d, out)381		status.Conditions = api.SetCondition(status.Conditions, condition, now)382		disr.rebooting = disr.rebooting || d.cleanup383384		// When this operator set a cordon and no longer needs it,385		// because the reboot happened and the machine converged, the386		// node goes back to the scheduler. This applies only to387		// cordons the operator set itself: decideUncordon leaves a388		// person's cordon in place.389		if !disr.rebooting && !disr.draining && decideUncordon(node) {390			if err := kubernetes.PatchJSON(c, nodesPath+"/"+node.Metadata.Name, uncordonPatch()); err != nil {391				fmt.Printf("uncordoning %s: %v\n", node.Metadata.Name, err)392			} else {393				fmt.Printf("uncordoned %s; its reboot is complete\n", node.Metadata.Name)394				notes.normal(reasonUncordoned, "uncordoned the Node "+node.Metadata.Name+"; its reboot is complete")395			}396		}397	}398399	// Ready is the roll-up: True exactly when every other condition400	// is True, with a reason that agrees with the phase (phase.go).401	status.Conditions = api.SetCondition(status.Conditions, readyCondition(status.Conditions), now)402403	// Every condition this pass publishes judged the spec at this404	// generation. The API server increases metadata.generation only405	// on spec writes, so recording it here lets a consumer tell a406	// verdict on the current spec apart from a verdict on a spec407	// that has since been edited. The conductor's grant keeps its408	// own generation stamp, because it is the conductor's verdict,409	// and this writer must not overwrite it. The status carries the410	// same stamp at its top, where clients that only ask "has the411	// operator seen my edit yet" expect to find it.412	status.ObservedGeneration = m.Metadata.Generation413	for i := range status.Conditions {414		if status.Conditions[i].Type == machine.RebootApprovedCondition {415			continue416		}417		status.Conditions[i].ObservedGeneration = m.Metadata.Generation418	}419420	// The phase compresses the conditions into the one word a fleet421	// listing shows (phase.go).422	status.Phase = decidePhase(status.Conditions)423424	// The metrics read the very status this pass is about to425	// publish, so a graph and a `kubectl get machine -o yaml` always426	// answer from the same observation (metrics.go).427	mm.observeStatus(status)428429	err = publishOwnStatus(r, m, status, before)430	if err != nil {431		fmt.Printf("publishing status: %v\n", err)432		return err433	}434	postStatusEvents(notes, &stored, status)435	return nil436}
machine-operator/registries.go 97.0%
1package main23// The operator's half of the registry-credentials lifecycle.4//5// Credentials enter the cluster as a kubernetes.io/dockerconfigjson6// Secret at a well-known name (the kubernetes package), because that7// is the shape the whole ecosystem's tooling already produces.8// `kubectl create secret docker-registry` writes one, `docker login`9// writes the same JSON to disk, and imagePullSecrets consume it. The10// operator reads that Secret on each pass, renders it into liken's11// canonical credentials document, and stages the rendering onto12// machineState whenever it differs from what the boot actuated.13// This is the same drift-and-stage loop every other document uses.14//15// This whole pipeline belongs to the restart class of changes.16// Credentials land in registries.yaml, which k3s reads only when17// its process starts, so the staged document converges by18// restarting k3s in place, never by rebooting the machine. And19// unlike the cluster document, no canonicalization pass is needed20// before comparing hashes. The operator is this document's only21// author (no image carries a seed, and no person hand-writes one),22// so the facts' hash and the desired rendering are always outputs23// of the same function.2425import (26	"encoding/base64"27	"encoding/json"28	"fmt"29	"net/url"30	"strings"3132	"github.com/liken-sh/liken/liken/kubernetes"33	"github.com/liken-sh/liken/liken/machine"34)3536// dockerConfig is the .dockerconfigjson payload: registry host37// mapped to login. This is Docker's own format, reduced to the38// fields liken reads.39type dockerConfig struct {40	Auths map[string]dockerAuth `json:"auths"`41}4243// dockerAuth is one registry's login. Tools write it two ways:44// explicit username/password fields, or an auth field holding45// base64("username:password"). `docker login` writes the auth form;46// `kubectl create secret docker-registry` writes both.47type dockerAuth struct {48	Username string `json:"username"`49	Password string `json:"password"`50	Auth     string `json:"auth"`51}5253// desiredRegistryCredentials renders the Secret into the canonical54// credential list: one username/password pair per registry host,55// sorted later by RenderRegistryCredentials. A nil result means no56// credentials exist anywhere: the Secret is absent, or it names no57// hosts. This is the ordinary "nothing declared" state, not an58// error. A malformed Secret is an error, and the message names what59// to create, because the fix is always a corrected Secret.60func desiredRegistryCredentials(secret *kubernetes.Secret) ([]machine.RegistryCredential, error) {61	if secret == nil {62		return nil, nil63	}64	if secret.Type != "kubernetes.io/dockerconfigjson" {65		return nil, fmt.Errorf("the registry-credentials Secret has type %q; create it with `kubectl create secret docker-registry registry-credentials -n liken-system ...` so it carries type kubernetes.io/dockerconfigjson", secret.Type)66	}67	raw, ok := secret.Data[".dockerconfigjson"]68	if !ok {69		return nil, fmt.Errorf("the registry-credentials Secret carries no .dockerconfigjson key; create it with `kubectl create secret docker-registry`")70	}71	var config dockerConfig72	if err := json.Unmarshal(raw, &config); err != nil {73		return nil, fmt.Errorf("the registry-credentials Secret's .dockerconfigjson is not valid JSON: %v", err)74	}7576	var hosts []machine.RegistryCredential77	for key, auth := range config.Auths {78		username, password := auth.Username, auth.Password79		if username == "" && auth.Auth != "" {80			decoded, err := base64.StdEncoding.DecodeString(auth.Auth)81			if err != nil {82				return nil, fmt.Errorf("the auth for %s is not valid base64: %v", key, err)83			}84			var found bool85			username, password, found = strings.Cut(string(decoded), ":")86			if !found {87				return nil, fmt.Errorf("the auth for %s does not decode to username:password", key)88			}89		}90		if username == "" {91			return nil, fmt.Errorf("the entry for %s carries no username, in neither the username field nor the auth field", key)92		}93		hosts = append(hosts, machine.RegistryCredential{94			Host:     registryHost(key),95			Username: username,96			Password: password,97		})98	}99	if len(hosts) == 0 {100		return nil, nil101	}102	return hosts, nil103}104105// registryHost reduces an auths key to the host that containerd106// matches against. Keys are usually bare hosts already, but `docker107// login` records Docker Hub as the URL https://index.docker.io/v1/,108// so the function reduces a key carrying a scheme to its host, and109// maps index.docker.io to docker.io, the name that image110// references, and therefore registries.yaml, use for the Hub. This111// is the one normalization worth doing, because it matches the112// shape the ecosystem's own tool produces.113func registryHost(key string) string {114	host := key115	if strings.Contains(key, "://") {116		if u, err := url.Parse(key); err == nil && u.Host != "" {117			host = u.Host118		}119	}120	if host == "index.docker.io" {121		return "docker.io"122	}123	return host124}125126// registriesInputs is one pass's view of the credentials Secret,127// gathered by convergeRegistryCredentials so the decision below128// stays pure.129type registriesInputs struct {130	fetchErr error                        // the API read failed (not absence: absence is desired == nil)131	parseErr error                        // the Secret exists but is malformed132	desired  []machine.RegistryCredential // nil when absent or empty133}134135// convergeRegistryCredentials runs the credentials document's part136// of one reconcile pass. It reads the registry-credentials Secret,137// loads this machine's durable rejection and staged copy from the138// store, and makes the convergence decision. Credentials converge139// through the same machinery as the other documents, from a140// different source: the Secret, rather than a CRD. Only a cluster141// member carries credentials at all. A machine with no cluster142// document has no operator-authored documents of any kind.143func convergeRegistryCredentials(r *reader, store machine.ManifestStore, m *machine.Machine, facts *machine.MachineStatus, t turn) convergence {144	rejection, _ := store.LoadRejection()145	in := registriesInputs{}146	if secret, fetchErr := r.registryCredentials(); fetchErr != nil {147		in.fetchErr = fetchErr148	} else {149		in.desired, in.parseErr = desiredRegistryCredentials(secret)150	}151	return decideRegistriesConvergence(in, m, facts, rejection, readStagedHash(store), t)152}153154// decideRegistriesConvergence is the credentials document's155// convergence decision, following the same order of checks as the156// other documents. The cases:157//158//  1. No facts, or facts with no boot record: Unknown.159//  2. The Secret is unreadable, an API failure rather than absence:160//     Unknown, the same verdict the ClusterUnavailable case gets.161//  3. The Secret is malformed: CredentialsInvalid, and nothing is162//     staged. The last good rendering keeps running, and the163//     message names the Secret to fix. This reports the problem164//     instead of getting stuck.165//  4. Nothing declared anywhere, meaning no Secret and this boot166//     rendered no credentials: converged. This case keeps a machine167//     that has never had credentials from staging an empty document168//     and restarting once for nothing. It also withdraws a stale169//     staged copy and clears a spent rejection, like every document.170//  5. The boot rendered exactly the desired credentials: converged.171//  6. The desired rendering is the one init rejected: hold.172//  7. machineState is backed by memory: there is nowhere durable to173//     stage.174//  8. Drift: stage the document (a deleted Secret stages the empty175//     document, the retraction rendering) and gate the k3s restart176//     through policy and the conductor's turn. This is always a177//     restart, never a reboot, because credentials only touch178//     registries.yaml.179func decideRegistriesConvergence(in registriesInputs, m *machine.Machine, facts *machine.MachineStatus, rejection *machine.Rejection, stagedHash string, t turn) convergence {180	if facts == nil || facts.Boot.ManifestSource == "" {181		return factsIncomplete("CredentialsConverged")182	}183	if in.fetchErr != nil {184		return convergence{condition: convergenceUnknown("CredentialsConverged", "SecretUnavailable",185			fmt.Sprintf("reading the registry-credentials Secret: %v", in.fetchErr))}186	}187	if in.parseErr != nil {188		return convergence{condition: notConverged("CredentialsConverged", "CredentialsInvalid",189			fmt.Sprintf("the registry-credentials Secret is malformed, so the machine keeps its last good credentials: %v", in.parseErr))}190	}191192	if in.desired == nil && facts.Boot.CredentialsHash == "" {193		return convergedWithCleanup(194			converged("CredentialsConverged", "NothingDeclared", "no registry credentials are declared"),195			stagedHash, rejection)196	}197198	manifest, hash, err := machine.RenderRegistryCredentials(in.desired)199	if err != nil {200		return convergence{condition: notConverged("CredentialsConverged", "StagingFailed", err.Error())}201	}202203	if hash == facts.Boot.CredentialsHash {204		return convergedWithCleanup(205			converged("CredentialsConverged", "Converged", fmt.Sprintf("this machine runs the current registry credentials (%d hosts)", len(in.desired))),206			stagedHash, rejection)207	}208209	if rejection != nil && rejection.Hash == hash {210		return convergence{condition: notConverged("CredentialsConverged", "RejectedLastBoot",211			fmt.Sprintf("init rejected this exact credentials document: %s; edit the Secret to something different", rejection.Reason))}212	}213	if facts.Storage.MachineState.Backing != machine.BackingPartition {214		return machineStateEphemeral("CredentialsConverged", "credentials")215	}216217	c := convergence{218		manifest: manifest,219		hash:     hash,220		stage:    stagedHash != hash,221	}222	what := fmt.Sprintf("registry credentials for %d hosts", len(in.desired))223	if in.desired == nil {224		what = "the retraction of the registry credentials"225	}226	gateDisruption(&c, "CredentialsConverged", m.Spec.RebootPolicyOrDefault(), t, true,227		m.Metadata.Annotations[machine.ApproveDisruptionAnnotation],228		what,229		fmt.Sprintf("%s staged (%.12s); rebootPolicy is Manual, so reboot the machine to apply (or set rebootPolicy: Auto, which would apply them with just a k3s restart)", what, hash),230		fmt.Sprintf("%s staged (%.12s); waiting for the cluster to grant a turn to apply them by k3s restart", what, hash),231		fmt.Sprintf("k3s restart requested to apply %s (%.12s)", what, hash))232	return c233}
machine-operator/release.go 97.7%
1package main23// Version convergence: moving a machine toward the cluster's target4// release.5//6// The Cluster declares one target version (spec.version) and the7// catalog that confirms it. Each machine's operator compares the8// version its boot reported against that target, live, on every9// pass. There is nothing to compare on the Machine's spec, because10// machines carry no version field. This is what makes an upgrade11// one edit, instead of one edit per machine.12//13// Convergence here is a download, not a reboot. The operator brings14// the release's artifacts onto the machine's inactive system slot15// and verifies every byte against the catalog's digest chain. The16// reboot that makes the downloaded release the running one comes17// later (the proving reboot). This file's job ends with verified18// bytes sitting on the other slot.19//20// The decisions are pure functions, reconcile supplies the I/O, and21// the fetch itself runs on its own goroutine (fetch.go). A blocking22// 100MB GET inside a reconcile pass would starve the heartbeat23// lease, and the fleet would read the silence as a dead machine.2425import (26	"fmt"27	"os"28	"time"2930	"github.com/liken-sh/liken/liken/api"31	"github.com/liken-sh/liken/liken/cluster"32	"github.com/liken-sh/liken/liken/machine"33)3435// versionAsk reports whether this machine should be downloading a36// release, and which one. The ask is the request that the fetcher37// carries out. When the answer is no (converged, no target, or a38// machine that cannot take a release), the returned condition is39// the whole verdict, and ok is false.40//41// The order of checks mirrors decideConvergence's: the permanently42// blocked cases are checked before the in-progress ones, so a43// machine that can never comply reports that, instead of reporting44// an attempt that could not succeed.45func versionAsk(clusterDoc *cluster.Cluster, facts *machine.MachineStatus) (fetchAsk, api.Condition, bool) {46	none := fetchAsk{}47	if facts == nil {48		return none, convergenceUnknown("VersionConverged", "FactsIncomplete",49			"the machine's facts haven't been read yet"), false50	}5152	target := clusterDoc.Spec.Version53	if target == "" {54		return none, converged("VersionConverged", "NoTarget",55			"the cluster declares no target version"), false56	}57	if facts.Version.Liken == target {58		return none, converged("VersionConverged", "Converged",59			fmt.Sprintf("this machine runs the cluster's target version %s", target)), false60	}6162	// The catalog admission rule makes this unreachable through the63	// API, but the operator checks anyway. The lookup can fail, so64	// the failure has to be handled, no matter what the API server65	// promised.66	entry := clusterDoc.Spec.Releases.Entry(target)67	if entry == nil {68		return none, notConverged("VersionConverged", "VersionNotInCatalog",69			fmt.Sprintf("the target version %s is not in the release catalog", target)), false70	}71	if clusterDoc.Spec.Releases.Source == "" {72		return none, notConverged("VersionConverged", "NoReleaseSource",73			"the catalog names releases but spec.releases.source gives nowhere to fetch them from"), false74	}7576	// A release needs somewhere to land: both slots claimed and77	// backed by a partition. It also needs the machine to be running78	// from a slot, because a machine that did not boot from a slot79	// has no boot entries to reboot through, so a download could80	// never become a boot.81	if facts.Storage.SystemA.Backing != machine.BackingPartition ||82		facts.Storage.SystemB.Backing != machine.BackingPartition {83		return none, notConverged("VersionConverged", "NoSystemSlots",84			"this machine has no system slots to hold a release; declare systemA and systemB in its manifest"), false85	}86	slot := machine.InactiveSlot(facts.Boot.Slot)87	if slot == "" {88		return none, notConverged("VersionConverged", "NotInstalled",89			"this boot didn't come from a system slot; install the machine (make install) before it can take releases"), false90	}9192	return fetchAsk{93		version: target,94		digest:  entry.Digest,95		source:  clusterDoc.Spec.Releases.Source,96		slot:    slot,97		slotDir: machine.SystemSlotDir(slot),98		// The running slot lends its deployment layer to the99		// download. The release supplies the OS; the machine100		// supplies itself.101		activeSlotDir: machine.SystemSlotDir(facts.Boot.Slot),102	}, api.Condition{}, true103}104105// versionCondition turns the fetcher's answer about an ask into106// the VersionConverged condition, for every state short of107// verified. (decideSystemStaging owns the verified state, because a108// verified download is reported through its staged record.) Every109// state here means "not converged yet". What differs is whether110// time will fix it. A failed fetch deliberately reads as111// Downloading: a down release server is transient by definition,112// the fetcher retries after its backoff, and the condition's message113// says what failed and when the retry starts. The message names the114// retry's time, not the wait left, so it stays the same from one pass115// to the next and the status is not written again for it. A digest116// mismatch is the opposite. Refetching cannot change what the server117// publishes, so the machine holds at DigestMismatch (phase Blocked)118// until the catalog names different bytes, and nothing is ever119// staged.120func versionCondition(ask fetchAsk, snap fetchSnapshot) api.Condition {121	switch snap.state {122	case fetchRejected:123		return notConverged("VersionConverged", "DigestMismatch", snap.detail)124	case fetchFailed:125		return notConverged("VersionConverged", "Downloading",126			fmt.Sprintf("downloading release %s to slot %s; will retry at %s: %s",127				ask.version, ask.slot, snap.retryAt.UTC().Format(time.RFC3339), snap.detail))128	}129	return notConverged("VersionConverged", "Downloading",130		fmt.Sprintf("downloading release %s to slot %s: %s", ask.version, ask.slot, snap.detail))131}132133// versionConvergence wraps a short-circuit verdict from versionAsk134// into a convergence. A True verdict, meaning converged or no135// target at all, also cleans up. A staged record left behind would136// reboot the machine into an upgrade nobody is asking for anymore,137// and a standing rejection no longer blocks anything.138func versionConvergence(cond api.Condition, stagedHash string, rejection *machine.Rejection) convergence {139	if cond.Status == api.ConditionTrue {140		return convergedWithCleanup(cond, stagedHash, rejection)141	}142	return convergence{condition: cond}143}144145// convergeSystemRelease runs the version target's part of one146// reconcile pass. It loads this machine's durable rejection and147// staged record from the store, asks the fetcher where the download148// stands, and makes the decision. The version target converges149// through its own machinery: a download aimed at the inactive slot.150// The download runs on the fetcher's goroutine, so that no151// reconcile pass, and no heartbeat, ever waits on a socket152// (versionAsk forms the ask, fetch.go moves the bytes). Once the153// download verifies, the rest works like the other documents: a154// staged SystemRelease record, the reboot chain, the drain gate, and155// the same carryOutConvergence.156func convergeSystemRelease(store machine.ManifestStore, liveCluster *cluster.Cluster, m *machine.Machine, facts *machine.MachineStatus, f *fetcher, t turn, out *passOutcome) convergence {157	rejection, _ := store.LoadRejection()158	stagedHash := readStagedHash(store)159	ask, cond, ok := versionAsk(liveCluster, facts)160	if !ok {161		return versionConvergence(cond, stagedHash, rejection)162	}163	stagedHash, err := withdrawOtherStage(store, ask, stagedHash)164	if err != nil {165		out.fail("withdrawing another release's staged record", err)166		return convergence{condition: notConverged("VersionConverged", "StagingFailed",167			fmt.Sprintf("release %s waits to download onto slot %s, because the staged record of another release could not be withdrawn: %v",168				ask.version, ask.slot, err))}169	}170	snap := f.Ensure(ask)171	if snap.state == fetchFailed {172		out.wakeBy(snap.retryAt)173	}174	return decideSystemStaging(ask, snap, m, rejection, stagedHash, t)175}176177// withdrawOtherStage withdraws a staged record that names another178// release, another slot, or another digest than the ask, and answers179// the staged hash that remains. It runs before the fetcher starts,180// because the download writes the slot that the old record names.181// While the record stands, any reboot that init manages arms a trial182// of that slot, and the slot would then hold files of two releases.183// A withdrawal that fails stops the download, so the slot keeps the184// release that the record names.185func withdrawOtherStage(store machine.ManifestStore, ask fetchAsk, stagedHash string) (string, error) {186	if stagedHash == "" {187		return "", nil188	}189	if _, hash, err := machine.RenderSystemRelease(ask.version, ask.slot, ask.digest); err == nil && hash == stagedHash {190		return stagedHash, nil191	}192	if err := store.WithdrawStaged(); err != nil {193		return stagedHash, err194	}195	fmt.Printf("withdrew the staged system release %.12s; the cluster now asks for release %s\n", stagedHash, ask.version)196	return "", nil197}198199// decideSystemStaging finishes version convergence. A verified200// download becomes a staged SystemRelease record, and that record201// goes through exactly the reboot machinery every other staged202// document uses. Manual policy reports RebootPending until a person203// approves the staged release through the approve-disruption204// annotation. Once approved, or when the policy is Auto from the205// start, a cluster member waits for its turn from the rollout206// conductor. A granted turn requests the reboot, gated through the207// drain like all the rest. The proving boot is the reboot itself:208// init sets the firmware's BootNext to the staged slot on the way209// down, and the operator that comes up running the new release is210// the proof that promotes the record.211func decideSystemStaging(ask fetchAsk, snap fetchSnapshot, m *machine.Machine, rejection *machine.Rejection, stagedHash string, t turn) convergence {212	if snap.state != fetchVerified {213		return convergence{condition: versionCondition(ask, snap)}214	}215216	record, hash, err := machine.RenderSystemRelease(ask.version, ask.slot, ask.digest)217	if err != nil {218		return convergence{condition: notConverged("VersionConverged", "StagingFailed", err.Error())}219	}220	// The rejection is a durable record of a trial that fell back.221	// The machine booted the staged slot, and the firmware returned222	// it to the proven one. Refusing to stage the identical decision223	// again is what breaks the reboot loop. A new version or a224	// republished digest is a different decision, and it passes.225	if rejection != nil && rejection.Hash == hash {226		return convergence{condition: notConverged("VersionConverged", "RejectedLastBoot",227			fmt.Sprintf("the machine tried release %s on slot %s and fell back: %s; publish a corrected release under a new version",228				ask.version, ask.slot, rejection.Reason))}229	}230231	c := convergence{manifest: record, hash: hash, stage: stagedHash != hash}232	gateDisruption(&c, "VersionConverged", m.Spec.RebootPolicyOrDefault(), t, false,233		m.Metadata.Annotations[machine.ApproveDisruptionAnnotation],234		fmt.Sprintf("release %s on slot %s", ask.version, ask.slot),235		fmt.Sprintf("release %s is verified on slot %s and staged (%.12s); rebootPolicy is Manual, so reboot the machine (or set rebootPolicy: Auto) to prove it", ask.version, ask.slot, hash),236		fmt.Sprintf("release %s is verified on slot %s and staged (%.12s); waiting for the cluster to grant a reboot turn", ask.version, ask.slot, hash),237		fmt.Sprintf("reboot requested to prove release %s on slot %s (%.12s)", ask.version, ask.slot, hash))238	return c239}240241// settleSystemReleaseLifecycle promotes what this boot proved, the242// way settleClusterLifecycle does for the cluster document: the243// operator's own existence is the evidence. If this pass is244// running, then the kernel, init, k3s, and the machine's place in245// its cluster all work. And if the version this boot reported246// matches the staged record for the same slot it came from, then247// the trial release is what is doing that work, so the record gets248// promoted.249//250// The comparison runs against the facts, meaning init's version251// stamp for the OS actually running, and deliberately not this252// binary's own stamp (machine.Version). The operator pod comes from253// whatever image the cluster's DaemonSet pins, and in a mixed fleet254// that pin lags behind the OS: the proving boot of a new release255// runs the old operator until the leaders themselves upgrade.256// Judging by the pod's stamp would therefore block every promotion257// a mixed fleet attempts. Worse, the unpromoted staged record would258// look stale to a machine that considers itself converged. The259// machine would withdraw the record and re-stage it on the next260// pass, so it would re-upgrade itself on every boot, forever. The261// pod plays no part in the trial; the machine is what is being262// proved.263//264// A machine running from a slot with no record at all, because its265// install predates any catalog, writes its current standing down as266// the first proven record, so init's every-boot BootOrder repair267// has an authority to enforce from the start.268func settleSystemReleaseLifecycle(root string, facts *machine.MachineStatus, out *passOutcome) {269	if facts == nil || facts.Storage.MachineState.Backing != machine.BackingPartition ||270		facts.Boot.Slot == "" || facts.Version.Liken == "" {271		return272	}273	store := machine.SystemReleases(root)274275	if staged, _ := store.LoadStaged(); staged != nil {276		record, err := machine.ParseSystemRelease(staged)277		if err != nil {278			return // init checks staged records at boot; the operator does not judge them279		}280		if record.Slot != facts.Boot.Slot || record.Version != facts.Version.Liken {281			return // not this boot's trial; nothing proved282		}283		if err := store.Promote(); err != nil {284			fmt.Fprintf(os.Stderr, "promoting the system release: %v\n", err)285			out.fail("promoting the system release", err)286			return287		}288		out.wrote("promoting the system release")289		fmt.Printf("release %s proved out on slot %s; the store now names it proven\n",290			record.Version, record.Slot)291		return292	}293294	if proven, err := store.LoadProven(); proven != nil || err != nil {295		return296	}297	raw, _, err := machine.RenderSystemRelease(facts.Version.Liken, facts.Boot.Slot, "")298	if err != nil {299		return300	}301	if err := store.WriteProven(raw); err != nil {302		fmt.Fprintf(os.Stderr, "recording the running release as proven: %v\n", err)303		out.fail("recording the running release as proven", err)304		return305	}306	out.wrote("recording the running release as proven")307	fmt.Printf("recorded the running release %s on slot %s as proven\n", facts.Version.Liken, facts.Boot.Slot)308}
machine-operator/retraction.go 94.1%
1package main23// The retraction barrier: what must be true before a feature stops.4//5// A feature's controller runs because the cluster document declares6// the feature. When the document stops declaring it, init stops7// rendering its configuration and the controller does not start at8// the next boot. What that controller programmed stays, and some of9// it can only be removed by the controller itself.10//11// Two features carry the consequence. k3s deploys Traefik through a12// HelmChart, and the Helm controller puts a removal finalizer on13// every chart it manages. Nothing but traefik requires helm, so an14// edit that removes traefik removes helm with it. k3s deletes the15// HelmChart once the disable list names traefik, and if the Helm16// controller has already stopped, the deletion never completes: the17// finalizer stays, and the release keeps running. servicelb has the18// same problem, over objects a deployment owns. The cloud controller19// inside it holds the cleanup finalizer on every LoadBalancer20// Service, so a Service deleted after that controller stops also21// never finishes deleting.22//23// So a retraction waits for the cluster instead of applying halfway.24// On every pass the operator derives the document it stages: the25// cluster document as the deployment wrote it, with every feature put26// back whose precondition the cluster does not satisfy yet. That27// reduced document is what the operator hashes, stages, and boots28// under. A machine that reboots in the middle of a retraction comes29// back up running the held feature, so the controller runs for as30// long as its programming is in force.31//32// The order between dependent features needs no rule of its own. An33// edit that removes traefik retracts helm too, and the HelmCharts34// still exist at that moment, so the pass holds helm and stops35// traefik alone. k3s deletes the charts, the Helm controller36// uninstalls the releases, and the next pass reads no charts and37// stops helm. Nothing here consults the dependency graph.38//39// This is the machine operator's work because the machine operator40// selects the document each machine stages. There is no handshake41// between operators and no field in the cluster document for a held42// feature. Every machine reads the same objects, evaluates the same43// preconditions, and derives the same reduced document.4445import (46	"fmt"47	"maps"48	"strings"4950	"github.com/liken-sh/liken/liken/cluster"51)5253// A featureHold is one feature that keeps running although the54// document no longer declares it, because the cluster still holds55// objects that only this feature's controller can remove. Blocker56// names those objects, in a sentence a person can act on.57type featureHold struct {58	slug    string59	blocker string60}6162// preconditionEvaluator answers one feature's precondition against63// the live cluster: whether it holds, and when it does not, what64// still stands in the way. The reduction takes this as a parameter65// rather than calling the API itself, so the decision stays a pure66// function over its inputs, the way every other decision in this67// operator is.68type preconditionEvaluator func(cluster.Precondition) (holds bool, blocker string, err error)6970// reduceRetraction derives the document this machine stages from the71// document the deployment wrote. Every feature whose precondition the72// cluster does not satisfy goes back into spec.features, and the73// caller receives those features and the reason for each.74//75// An edit that stops no feature evaluates nothing, so the ordinary76// pass, where the document is unchanged or changes something else,77// costs no API calls.78//79// A held feature goes back with the declaration it runs under now,80// which withFeature takes from the boot document.81func reduceRetraction(bootDoc, desired *cluster.Cluster, evaluate preconditionEvaluator) (*cluster.Cluster, []featureHold) {82	if bootDoc == nil {83		// A retraction is a difference between two documents, and84		// without the document this boot ran there is no difference to85		// measure. The convergence decision applies an unreadable boot86		// document by rebooting (cluster.go), and a reboot is also the87		// safest way to stop a controller.88		return desired, nil89	}90	reduced := desired91	var held []featureHold92	for _, slug := range cluster.RetractedFeatures(bootDoc, desired) {93		def := cluster.FeatureBySlug(slug)94		if def == nil || def.Retraction.Precondition == cluster.PreconditionNone {95			continue96		}97		holds, blocker, err := evaluate(def.Retraction.Precondition)98		if err != nil {99			// A failed read is not a satisfied precondition. A feature100			// held on a failed read costs a delayed retraction, and the101			// next pass evaluates it again. A feature stopped on a102			// failed read costs the stranded objects this barrier103			// prevents.104			blocker = fmt.Sprintf("the check for what still depends on it could not run: %v", err)105		} else if holds {106			continue107		}108		reduced = withFeature(reduced, slug, bootDoc.Spec.Features[slug])109		held = append(held, featureHold{slug: slug, blocker: blocker})110	}111	return reduced, held112}113114// withFeature returns a copy of the document with one more feature115// declared. It copies rather than edits, because the document it116// receives is the live Cluster that the rest of the pass reads: the117// release feed comes from that same object, and the reduction must118// change nothing that anything else reads.119//120// The configuration comes from the document this boot ran under, so a121// held feature keeps running exactly as it has been running. A122// parameterized feature declared with {} instead would come back123// stripped of its parameters, which is a different feature from the124// one the machine is holding. A requirement that nobody declared has125// no configuration of its own, and {} is right for it.126func withFeature(doc *cluster.Cluster, slug string, cfg *cluster.FeatureConfig) *cluster.Cluster {127	reduced := *doc128	reduced.Spec.Features = maps.Clone(doc.Spec.Features)129	if reduced.Spec.Features == nil {130		reduced.Spec.Features = map[string]*cluster.FeatureConfig{}131	}132	if cfg == nil {133		cfg = &cluster.FeatureConfig{}134	}135	reduced.Spec.Features[slug] = cfg136	return &reduced137}138139// evaluatePrecondition answers one precondition against the live140// cluster. Every precondition is a count of objects that only the141// feature's own controller can finish removing, so an unsatisfied142// precondition always carries the names of those objects.143func evaluatePrecondition(r *reader, p cluster.Precondition) (bool, string, error) {144	switch p {145	case cluster.NoHelmCharts:146		charts, err := r.helmCharts()147		if err != nil {148			return false, "", err149		}150		if len(charts) == 0 {151			return true, "", nil152		}153		names := make([]string, len(charts))154		for i, chart := range charts {155			names[i] = chart.Metadata.Namespace + "/" + chart.Metadata.Name156		}157		return false, fmt.Sprintf(158			"the cluster still holds HelmCharts (%s); they must be deleted, and their releases uninstalled, before the Helm controller stops",159			nameList(names)), nil160161	case cluster.NoLoadBalancerServices:162		services, err := r.loadBalancerServices()163		if err != nil {164			return false, "", err165		}166		if len(services) == 0 {167			return true, "", nil168		}169		names := make([]string, len(services))170		for i, s := range services {171			names[i] = s.Metadata.Namespace + "/" + s.Metadata.Name172		}173		return false, fmt.Sprintf(174			"the cluster still holds Services of type LoadBalancer (%s); they must be deleted before the controller that cleans them up stops",175			nameList(names)), nil176	}177	// The vocabulary is compiled in, so a precondition this switch178	// does not cover is one the feature table states and this file179	// has no case for. Reading it as unsatisfied is the safe180	// direction, and the feature keeps running.181	return false, fmt.Sprintf("this build cannot evaluate the %s precondition", p), nil182}183184// nameListLimit is how many object names a blocker carries. A185// condition message is written for a person, who acts on the first186// few names, and every name after them costs message length without187// changing the correction.188const nameListLimit = 3189190func nameList(names []string) string {191	if len(names) <= nameListLimit {192		return strings.Join(names, ", ")193	}194	return fmt.Sprintf("%s and %d more", strings.Join(names[:nameListLimit], ", "), len(names)-nameListLimit)195}196197// holdMessage is the condition's message for a retraction that198// cannot proceed: each feature that keeps running, and the objects199// the cluster must lose before that feature may stop.200func holdMessage(held []featureHold) string {201	sentences := make([]string, len(held))202	for i, h := range held {203		sentences[i] = fmt.Sprintf("%s cannot stop yet: %s", h.slug, h.blocker)204	}205	return strings.Join(sentences, "; ")206}
machine-operator/retry.go 96.7%
1package main23// When the loop runs the next pass after one that did not finish.4//5// A pass that leaves a failure behind asks for a retry, and the loop6// keeps one timer for it. The delay depends on the kind of failure7// (outcome.go). A failure that a retry can clear starts at one second8// and doubles up to ten, so a write that failed because the API server9// restarted lands within a second or two of the server's return, and10// a server that stays away is asked every ten seconds. A11// failure that will not clear by itself starts at ten seconds and12// doubles up to five minutes, so a misconfigured machine does not send13// the same refused request every ten seconds forever. A person who14// fixes the spec sends a watch event, and that pass tries at once.15//16// No pass comes sooner than a 429's Retry-After asked, not even one a17// step asked for, because that pass would send the same requests to a18// throttled API server.19//20// Each delay gets up to a tenth more at random. When the API server21// fails, every machine in the fleet fails at the same moment, and22// without the jitter they would all retry at the same moments too.2324import (25	"math/rand/v2"26	"time"27)2829const (30	transientFirstRetry = time.Second31	transientRetryLimit = 10 * time.Second32	lastingFirstRetry   = 10 * time.Second33	lastingRetryLimit   = 5 * time.Minute34)3536// retrySchedule holds the current delay for each kind of failure, from37// one pass to the next. A zero delay means the last pass had no failure38// of that kind.39type retrySchedule struct {40	transient time.Duration41	lasting   time.Duration4243	// jitter answers a number in [0, 1). Nil means math/rand.44	jitter func() float6445}4647// next answers when the loop should run a pass after one that ended at48// now with this outcome, and false when nothing asks for one.49func (s *retrySchedule) next(out *passOutcome, now time.Time) (time.Time, bool) {50	var hasTransient, hasLasting bool51	for _, f := range out.failures {52		switch f.kind {53		case transient:54			hasTransient = true55		case lasting:56			hasLasting = true57		}58	}59	s.transient = grow(s.transient, hasTransient, transientFirstRetry, transientRetryLimit)60	s.lasting = grow(s.lasting, hasLasting, lastingFirstRetry, lastingRetryLimit)6162	var delay time.Duration63	switch {64	case hasTransient:65		delay = s.transient66	case hasLasting:67		delay = s.lasting68	}6970	at, ok := time.Time{}, false71	if delay > 0 {72		at, ok = now.Add(delay+time.Duration(float64(delay)*s.random()/10)), true73	}74	if !out.wake.IsZero() && (!ok || out.wake.Before(at)) {75		at, ok = out.wake, true76	}77	if ok && at.Before(out.notBefore) {78		at = out.notBefore79	}80	return at, ok81}8283// grow answers the next delay for one kind of failure: the first delay84// after a pass without one, double the last after a pass with one, and85// zero after a pass without one.86func grow(last time.Duration, failed bool, first, limit time.Duration) time.Duration {87	switch {88	case !failed:89		return 090	case last == 0:91		return first92	}93	return min(2*last, limit)94}9596func (s *retrySchedule) random() float64 {97	if s.jitter == nil {98		return rand.Float64()99	}100	return s.jitter()101}
machine-operator/seeding.go 88.6%
1package main23// The seeding loops: making the boot manifest's Machine and the4// image's Cluster real in the API at startup. Seeding tolerates the5// races and not-yet-served CRDs of a fleet booting together. It runs6// once, before the reconcile loop starts. From then on, the7// cluster's copies are authoritative.89import (10	"encoding/json"11	"errors"12	"fmt"13	"net/http"1415	"github.com/liken-sh/liken/kubernetes/apiclient"16	"github.com/liken-sh/liken/kubernetes/events"17	"github.com/liken-sh/liken/liken/api"18	"github.com/liken-sh/liken/liken/cluster"19	"github.com/liken-sh/liken/liken/kubernetes"20	"github.com/liken-sh/liken/liken/machine"21)2223// ensureMachine makes the manifest's Machine real in the cluster.24// The retry-forever loop covers the operator's first minutes. k3s25// applies the Machine CRD from its manifests directory around the26// same time it starts this pod, and until the API server serves27// that CRD, the operator's requests get a 404. The loop waits28// instead of crashing, because that 404 is expected during startup,29// not a sign that something is wrong.30//31// A Machine this function creates posts MachineJoined, once the32// server's copy reads back with its UID: the machine's first boot in33// the cluster, or its first boot after a person deleted the Machine.34func ensureMachine(c *apiclient.Client, seed *machine.Machine, recorder *events.Recorder) (*machine.Machine, error) {35	created := false36	for {37		current, err := kubernetes.GetMachine(c, seed.Metadata.Name)38		if err == nil {39			if created {40				recorder.Normal(machineReference(current), reasonMachineJoined,41					"created the Machine from the boot manifest, because the cluster held none")42			}43			return current, nil44		}45		if !errors.Is(err, apiclient.ErrNotFound) {46			return nil, err47		}4849		body, err := json.Marshal(&machine.Machine{50			APIVersion: api.APIVersion,51			Kind:       "Machine",52			Metadata:   api.ObjectMeta{Name: seed.Metadata.Name},53			Spec:       seed.Spec,54		})55		if err != nil {56			return nil, err57		}58		err = c.RequestJSON(http.MethodPost, kubernetes.MachinesPath, body, nil)59		if err == nil {60			fmt.Printf("created machine %s from %s\n", seed.Metadata.Name, machine.BootManifestPath)61			created = true62			continue // re-read so the function returns the server's copy, resourceVersion and all63		}64		if errors.Is(err, apiclient.ErrNotFound) {65			fmt.Println("machine API not served yet; waiting")66			kubernetes.RetryPause()67			continue68		}69		return nil, err70	}71}7273// ensureCluster makes the manifest's Cluster real in the cluster.74// It waits out an unserved CRD the same way ensureMachine does, and75// it tolerates one extra answer: 409 Conflict. Every machine's76// operator races to create the same object at boot, so all but one77// of those POSTs will conflict. That conflict causes no harm: the78// loop's next GET confirms the object exists, which is the only79// outcome that matters.80func ensureCluster(c *apiclient.Client, seed *cluster.Cluster) error {81	for {82		if _, err := kubernetes.GetCluster(c, seed.Metadata.Name); err == nil {83			return nil84		} else if !errors.Is(err, apiclient.ErrNotFound) {85			return err86		}8788		body, err := json.Marshal(&cluster.Cluster{89			APIVersion: api.APIVersion,90			Kind:       "Cluster",91			Metadata:   api.ObjectMeta{Name: seed.Metadata.Name},92			Spec:       seed.Spec,93		})94		if err != nil {95			return err96		}97		switch err := c.RequestJSON(http.MethodPost, kubernetes.ClustersPath, body, nil); {98		case err == nil:99			fmt.Printf("created cluster %s from %s\n", seed.Metadata.Name, cluster.ClusterManifestPath)100		case errors.Is(err, apiclient.ErrNotFound):101			fmt.Println("cluster API not served yet; waiting")102			kubernetes.RetryPause()103		case errors.Is(err, apiclient.ErrConflict):104			// Another machine's operator got there first.105		default:106			return err107		}108	}109}
machine-operator/staging.go 100.0%
1package main23// Staging: the side effects of one convergence decision. The decisions4// themselves are pure (converge.go), so a test can judge them without5// a disk. This file is where a decision touches the machine: the6// staged document, the rejection record, and the intent files that7// init acts on.89import (10	"fmt"11	"time"1213	"github.com/liken-sh/liken/liken/api"14	"github.com/liken-sh/liken/liken/machine"15)1617// carryOutConvergence performs one convergence decision's side18// effects against one document's store and init's intent directory,19// runDir, and returns the condition to publish. An I/O failure20// downgrades the condition to StagingFailed on the same condition21// type, so the report stays attached to the right document.22func carryOutConvergence(conv convergence, store machine.ManifestStore, runDir, what string, now time.Time, out *passOutcome) api.Condition {23	step := "carrying out the " + what + "'s convergence"24	failed := func(err error) api.Condition {25		out.fail(step, err)26		return api.Condition{Type: conv.condition.Type, Status: api.ConditionFalse, Reason: "StagingFailed", Message: err.Error()}27	}28	if conv.withdraw {29		if err := store.WithdrawStaged(); err != nil {30			fmt.Printf("withdrawing the staged %s: %v\n", what, err)31			out.fail("withdrawing the staged "+what, err)32		} else {33			fmt.Printf("withdrew the staged %s; the cluster's copy matches this boot again\n", what)34			out.wrote("withdrawing the staged " + what)35		}36	}37	if conv.clearRejection {38		if err := store.ClearRejection(); err != nil {39			fmt.Printf("clearing the %s rejection record: %v\n", what, err)40			out.fail("clearing the "+what+" rejection record", err)41		} else {42			out.wrote("clearing the " + what + " rejection record")43		}44	}45	if conv.stage {46		if err := store.WriteStaged(conv.manifest); err != nil {47			return failed(err)48		}49		out.wrote(step)50		fmt.Printf("staged %s %.12s for the next boot\n", what, conv.hash)51	}52	if conv.requestReboot {53		intent := &machine.RebootIntent{54			Reason:       "applying the staged " + what,55			ManifestHash: conv.hash,56			RequestedAt:  now,57		}58		if err := machine.WriteRebootIntent(runDir, intent); err != nil {59			return failed(err)60		}61		out.wrote(step)62		fmt.Printf("requested a reboot to apply %s %.12s\n", what, conv.hash)63	}64	if conv.requestRestart {65		intent := &machine.RestartIntent{66			Reason:      "applying the staged " + what,67			RequestedAt: now,68		}69		if err := machine.WriteRestartIntent(runDir, intent); err != nil {70			return failed(err)71		}72		out.wrote(step)73		fmt.Printf("requested a k3s restart to apply %s %.12s\n", what, conv.hash)74	}75	if conv.requestLoad {76		intent := &machine.ModulesIntent{77			Reason:       "loading the staged " + what + "'s added modules",78			ManifestHash: conv.hash,79			RequestedAt:  now,80		}81		if err := machine.WriteModulesIntent(runDir, intent); err != nil {82			return failed(err)83		}84		out.wrote(step)85		fmt.Printf("requested a live module load to apply %s %.12s\n", what, conv.hash)86	}87	return conv.condition88}8990// readStagedHash returns the hash of the document currently staged91// in the store, or "" when nothing is staged. The function hashes92// staged bytes even when they fail to parse, because the idempotence93// check compares bytes, not parsed meaning.94func readStagedHash(store machine.ManifestStore) string {95	raw, _ := store.LoadStaged()96	if raw == nil {97		return ""98	}99	return machine.ManifestHash(raw)100}
machine-operator/staleness.go 100.0%
1package main23// The pod-freshness guard: whether this operator's own pod predates4// the release it is running.5//6// System pods run the stable image tag :installed, and their7// DaemonSets update on OnDelete rather than a rolling update8// (cluster-operator/steward.go explains why). A reboot restarts the9// container into a new binary without touching the pod spec around10// it. Only a leader's boot rewrites the AddOn manifests that produce11// a fresh template, so a follower that reboots first runs the new12// binary inside the old pod spec until the pod steward refreshes the13// pod, seconds after the first leader boots that release.14// conditions.go reads15// the verdict this file computes to judge a missing-mount failure as16// that ordinary lag instead of a fault on the machine.1718import "github.com/liken-sh/liken/liken/kubernetes"1920// osVersionAnnotation names the release a pod's template shipped21// with. image/build.sh stamps this annotation onto the22// machine-operator DaemonSet and its pod template23// (manifests/machine-operator.yaml), and cluster-operator/steward.go24// reads the same name off the DaemonSet to know which pods to25// refresh.26const osVersionAnnotation = "liken.sh/os-version"2728// ownPodPath asks the API for this node's own machine-operator pod:29// liken-system's pods labeled app=liken-machine-operator, filtered30// again by spec.nodeName so the server never sends any other node's31// pod over the wire. kubernetes.Pod carries no labels, only32// annotations, so the label filtering happens in the query string33// here rather than in a second pass over the response, the same way34// cluster-operator/steward.go filters a DaemonSet's pods by its app35// label.36func ownPodPath(nodeName string) string {37	return "/api/v1/namespaces/liken-system/pods?labelSelector=app%3Dliken-machine-operator&fieldSelector=spec.nodeName%3D" + nodeName38}3940// decidePodStale judges whether this operator's own pod predates the41// release it is running. pods is that one pod's listing, already42// narrowed to this node and this DaemonSet's label by ownPodPath.43// runningVersion is the release this boot's facts report44// (status.Version.Liken in reconcile.go). A pod's os-version45// annotation names the template it was created from, and the two46// disagreeing is exactly the state this guard exists for.47//48// No pod, or a pod carrying no annotation, both read as current. A49// pod that predates this annotation, from before this design existed,50// must not be judged stale forever, and a machine with no pod to find51// yet, early in a boot, has nothing to judge as stale either. An52// empty runningVersion also reads as current: facts that carry no53// version cannot support a verdict, and a wrong "stale" here would54// hide a real fault behind AwaitingPodRefresh.55func decidePodStale(pods []kubernetes.Pod, runningVersion string) bool {56	if runningVersion == "" {57		return false58	}59	for _, p := range pods {60		if v, ok := p.Metadata.Annotations[osVersionAnnotation]; ok {61			return v != runningVersion62		}63	}64	return false65}6667// ownPodIsStale is the thin read around decidePodStale. A list that68// fails, for example because the API server is briefly unreachable,69// reads the same as a pod with no annotation: current. The guard this70// feeds must never manufacture a fault where none exists, so an API71// error here can only ever hide a real fault behind ApplyFailed's72// ordinary reporting, never behind a false AwaitingPodRefresh.73func ownPodIsStale(r *reader, nodeName, runningVersion string) bool {74	pods, err := r.operatorPods(nodeName)75	if err != nil {76		return false77	}78	return decidePodStale(pods, runningVersion)79}
machine-operator/taints.go 100.0%
1package main23// Node taints, reconciled live from the Machine spec.4//5// A label attracts a workload and a taint repels one, so6// spec.nodeTaints is the other half of the scheduling identity that7// labels.go applies. It reaches the Node the same two ways. Init8// renders the taints into the k3s boot drop-in, so a node registers9// already repelling, which is what keeps a fresh node from accepting,10// in its first minutes, the pods the taint exists to keep out. But11// the kubelet applies registration taints only when it creates the12// Node object, on a first boot or after a reinstall. On every later13// boot the Node already exists and the setting does nothing, so live14// reconciliation is the only mechanism from then on. It runs here, in15// the same pass that reconciles labels.16//17// Removing a taint needs a record, for the reason labels.go gives:18// nothing about a taint on a Node says who applied it, and the19// operator must never remove one that a person or another controller20// applied.2122import (23	"encoding/json"24	"fmt"25	"maps"26	"slices"27	"strings"2829	"github.com/liken-sh/liken/kubernetes/apiclient"30	"github.com/liken-sh/liken/liken/api"31	"github.com/liken-sh/liken/liken/kubernetes"32	"github.com/liken-sh/liken/liken/machine"33)3435// ownedTaintsAnnotation records, on the Node itself, which taints36// liken manages. Its value is the pairs it owns, each written37// key:Effect, sorted and joined with commas. It lives on the Node38// rather than in the Machine's status, so the record and the taints39// it describes can never drift apart across operator restarts or40// Machine rewrites.41//42// The pair is the unit of ownership, where labels own a bare key. A43// taint's identity is its key together with its effect: one key can44// carry NoSchedule and NoExecute at the same time, and those are two45// separate taints that a pod tolerates separately. A record keyed on46// the key alone would give the operator permission to remove a taint47// it never applied, whenever someone added a second effect under a48// declared key.49const ownedTaintsAnnotation = "liken.sh/node-taints"5051// A taintStep is one pass's worth of taint reconciliation: the Node52// patch to apply (nil when the Node already matches the spec) and the53// condition to publish once the patch lands.54type taintStep struct {55	patch     []byte56	condition api.Condition57}5859// taintPair renders a taint's identity, its key and its effect, in60// the form the ownership annotation stores.61func taintPair(key, effect string) string {62	return key + ":" + effect63}6465// decideNodeTaints compares the spec's taints against the Node and66// produces the merge patch that closes the gap. It upserts every67// declared taint, drops every pair the annotation owns that the spec68// no longer declares, and keeps every other taint exactly as it69// found it. A taint whose pair is in neither the spec nor the70// annotation belongs to someone else, and this function never71// changes it.72func decideNodeTaints(desired []machine.NodeTaint, node *nodeObject) taintStep {73	wanted := map[string]nodeTaint{}74	for _, taint := range desired {75		pair := taintPair(taint.Key, string(taint.Effect))76		wanted[pair] = nodeTaint{Key: taint.Key, Value: taint.Value, Effect: string(taint.Effect)}77	}78	owned := map[string]bool{}79	for pair := range strings.SplitSeq(node.Metadata.Annotations[ownedTaintsAnnotation], ",") {80		if pair != "" {81			owned[pair] = true82		}83	}8485	// The merged list is what the Node must hold after the patch. A86	// taint the Node already carries keeps the position it has, and a87	// newly declared one goes on the end in sorted order. Position88	// means nothing to the scheduler, so the only requirement is that89	// the same inputs always produce the same list. A list whose90	// order moved from pass to pass would patch the Node forever.91	merged := []nodeTaint{}92	placed := map[string]bool{}93	for _, on := range node.Spec.Taints {94		pair := taintPair(on.Key, on.Effect)95		if declared, is := wanted[pair]; is {96			merged = append(merged, declared)97			placed[pair] = true98			continue99		}100		if owned[pair] {101			continue102		}103		merged = append(merged, on)104	}105	for _, pair := range slices.Sorted(maps.Keys(wanted)) {106		if !placed[pair] {107			merged = append(merged, wanted[pair])108		}109	}110111	annotations := map[string]any{}112	ownedNow := strings.Join(slices.Sorted(maps.Keys(wanted)), ",")113	if ownedNow != node.Metadata.Annotations[ownedTaintsAnnotation] {114		if ownedNow == "" {115			annotations[ownedTaintsAnnotation] = nil116		} else {117			annotations[ownedTaintsAnnotation] = ownedNow118		}119	}120121	condition := api.Condition{Type: "NodeTaintsApplied", Status: api.ConditionTrue, Reason: "Applied",122		Message: fmt.Sprintf("the Node carries all %d declared taints", len(desired))}123	if len(desired) == 0 {124		condition = api.Condition{Type: "NodeTaintsApplied", Status: api.ConditionTrue, Reason: "NothingDeclared",125			Message: "no node taints declared"}126	}127128	rewrite := !slices.Equal(merged, node.Spec.Taints)129	if !rewrite && len(annotations) == 0 {130		return taintStep{condition: condition}131	}132133	patch := map[string]any{}134	metadata := map[string]any{}135	if len(annotations) > 0 {136		metadata["annotations"] = annotations137	}138	if rewrite {139		// A JSON merge patch replaces an array whole, where it merges140		// a map key by key. So the taints write must carry the full141		// list, foreign entries included, and a write that starts142		// from a stale read erases whatever landed in between. The143		// node lifecycle controller writes this same array: it adds144		// node.kubernetes.io/not-ready and unreachable when a node145		// stops reporting, and removes them when it recovers. Naming146		// the resourceVersion this pass read turns that race into a147		// 409 conflict, so the API server refuses the write instead148		// of dropping the controller's taint. The watch of the Node149		// delivers the write that won and wakes the next pass, which150		// patches again against the new version.151		//152		// The labels patch needs no such precondition, because a153		// merge patch on a map touches only the keys it names, and a154		// concurrent writer's key is not one of them.155		metadata["resourceVersion"] = node.Metadata.ResourceVersion156		patch["spec"] = map[string]any{"taints": merged}157	}158	patch["metadata"] = metadata159	encoded, _ := json.Marshal(patch)160	return taintStep{patch: encoded, condition: condition}161}162163// carryOutNodeTaints applies the step's patch. It downgrades the164// condition when the API server refuses the patch. A refusal here is165// often the resourceVersion conflict, and the recovery is the same as166// for any other failure: the next pass reads the Node's copy again,167// builds the step again, and patches again.168func carryOutNodeTaints(c *apiclient.Client, name string, step taintStep) api.Condition {169	if step.patch == nil {170		return step.condition171	}172	if err := kubernetes.PatchJSON(c, nodesPath+"/"+name, step.patch); err != nil {173		return api.Condition{Type: "NodeTaintsApplied", Status: api.ConditionFalse, Reason: "ApplyFailed",174			Message: fmt.Sprintf("patching the Node's taints: %v", err)}175	}176	return step.condition177}
machine-operator/userspace.go 100.0%
1package main23// USB devices that a program in userspace drives.4//5// Some USB devices never get a kernel driver, because the vendor's6// driver is a program that reaches the device through libusb. ZWO's7// astronomy cameras, smart card readers that pcscd serves, and many8// software-defined radios are this kind. The kernel enumerates such9// a device, creates its interfaces, binds none of them, and gives the10// device its usbfs node, /dev/bus/usb/<bus>/<device>. The program11// reads sysfs to find the device and opens that node, and it needs12// nothing more.13//14// The inventory publishes such a device whole, as one exclusive slice15// device that delivers the usbfs node, and names it by the device's16// own port path, usb-1-2, not by an interface's. The usbfs node opens17// every interface of the device, so one claim on the device is the18// only claim that can hand it over.19//20// The rule asks that no interface has a driver at all. A device that21// the kernel drives in part already publishes its driven interfaces,22// each with the usbfs node beside its own nodes, and a second claim on23// the whole device would hand the same hardware to a second workload.24// A USB audio adapter whose HID interface no driver bound is the case25// the lab's testbed showed: a player holds the audio interface, and a26// claim on the whole adapter could reset it under the player. The27// rule also refuses an interface that usbfs holds. A program bound to28// an interface while it runs, and a device that a program holds this29// way leaves the slice until the program lets go, so a second claim30// cannot arrive while the first one is using the device.3132import (33	"strings"3435	"github.com/liken-sh/liken/liken/hardware"36)3738// userspaceDevice reports whether d is a USB device whose every39// interface has no driver, and returns it with the first interface's40// class, so that a DeviceClass can select a camera or a card reader by41// what the device says it is. A USB device's own class is most often42// 00, which means that each interface states its own.43func userspaceDevice(d hardware.Device, discovered []hardware.Device) (hardware.Device, bool) {44	if d.Bus != "usb" || d.Driver != "usb" || strings.Contains(d.Address, ":") {45		return d, false46	}47	var interfaces []hardware.Device48	for _, other := range discovered {49		if other.Bus == "usb" && strings.HasPrefix(other.Address, d.Address+":") {50			interfaces = append(interfaces, other)51		}52	}53	if len(interfaces) == 0 {54		return d, false55	}56	for _, i := range interfaces {57		if i.Driver != "" {58			return d, false59		}60	}61	d.Class, d.ClassCode = interfaces[0].Class, interfaces[0].ClassCode62	return d, true63}
machine-operator/waits.go 97.6%
1package main23// The watches that run only while a pass waits on objects that other4// programs change.5//6// Three features wait that way. The image proof waits for every OS7// container on the node to become Ready (imports.go). The drain waits8// for the node's pods to leave, and for a PodDisruptionBudget to allow9// an eviction it refused (drain.go). A feature removal waits for the10// cluster's HelmCharts or LoadBalancer Services to be deleted11// (retraction.go). Each of those reads lists objects that change while12// the wait lasts, so each wait gets a watch, and the watch wakes the13// pass that reads it.14//15// A watch runs only while its wait does. A pass that reads a kind16// starts its watch, and the loop stops every watch that the last pass17// did not read (endPass). A wait begins and ends on a pass, and each18// pass works out from the state it reads whether the wait goes on, so19// an operator that restarts in the middle of a drain starts the20// watches it needs on its first pass. Outside a wait, a busy node's pod21// writes wake no pass, and no copy of them costs memory.22//23// A copy that is not ready yet answers nothing, and the read goes to24// the API server, as every read of the reader does (watches.go). The25// first pass of a wait reads the API server and starts the watch, and26// the watch's first sync wakes the next pass, which reads the copy.2728import (29	"context"30	"slices"31	"strings"32	"sync"3334	"github.com/liken-sh/liken/kubernetes/informer"35	"github.com/liken-sh/liken/liken/kubernetes"36	"github.com/liken-sh/liken/liken/kubernetes/watch"37	apierrors "k8s.io/apimachinery/pkg/api/errors"38	"k8s.io/apimachinery/pkg/apis/meta/v1/unstructured"39	"k8s.io/apimachinery/pkg/runtime/schema"40	"k8s.io/client-go/dynamic"41	"k8s.io/client-go/tools/cache"42)4344// The kinds a wait can watch.45const (46	waitPods     = "node pods"47	waitBudgets  = "PodDisruptionBudgets"48	waitCharts   = "HelmCharts"49	waitServices = "Services"50)5152var (53	budgetResource  = schema.GroupVersionResource{Group: "policy", Version: "v1", Resource: "poddisruptionbudgets"}54	chartResource   = schema.GroupVersionResource{Group: "helm.cattle.io", Version: "v1", Resource: "helmcharts"}55	serviceResource = schema.GroupVersionResource{Version: "v1", Resource: "services"}56)5758// waits starts and stops the watches of the waits. A nil *waits starts59// nothing, and every read goes to the API server.60type waits struct {61	ctx     context.Context62	watcher dynamic.Interface63	wake    func()64	node    string6566	mu      sync.Mutex67	running map[string]*waitWatch68	used    map[string]bool69}7071// waitWatch is one running watch and the cancel that stops it.72type waitWatch struct {73	copy   *informer.Collection74	cancel context.CancelFunc75}7677func newWaits(ctx context.Context, watcher dynamic.Interface, wake func(), node string) *waits {78	return &waits{ctx: ctx, watcher: watcher, wake: wake, node: node,79		running: map[string]*waitWatch{}, used: map[string]bool{}}80}8182// use answers the copy of one kind, and starts its watch when it is not83// running. The pass that calls it keeps the watch running for one more84// pass.85func (w *waits) use(kind string) *informer.Collection {86	if w == nil {87		return nil88	}89	w.mu.Lock()90	defer w.mu.Unlock()91	w.used[kind] = true92	if running, ok := w.running[kind]; ok {93		return running.copy94	}95	ctx, cancel := context.WithCancel(w.ctx)96	source, options := w.watchOf(kind)97	options.Synced = w.wake98	options.UnreadyOnWatchError = true99	running := &waitWatch{copy: informer.Start(ctx, w.watcher, source, options), cancel: cancel}100	w.running[kind] = running101	return running.copy102}103104// endPass stops every watch that the pass did not use, and starts the105// count again for the next pass.106func (w *waits) endPass() {107	if w == nil {108		return109	}110	w.mu.Lock()111	defer w.mu.Unlock()112	for kind, running := range w.running {113		if !w.used[kind] {114			running.cancel()115			delete(w.running, kind)116		}117	}118	clear(w.used)119}120121// watchOf names the source of one kind and what wakes the loop.122func (w *waits) watchOf(kind string) (informer.Source, informer.Options) {123	switch kind {124	case waitPods:125		pods := informer.Source{Resource: podResource, FieldSelector: "spec.nodeName=" + w.node}126		return pods, informer.Options{127			Handler:   watch.WakeOnContent[kubernetes.Pod](pods, w.wake),128			Transform: trimPod,129		}130	case waitBudgets:131		return informer.Source{Resource: budgetResource}, informer.Options{132			Handler: budgetHandler(w.wake),133		}134	case waitCharts:135		return informer.Source{Resource: chartResource}, informer.Options{136			Handler: presenceHandler(w.wake),137			// The HelmChart kind exists only while k3s's Helm controller138			// runs. An absent kind holds no chart.139			Absent: apierrors.IsNotFound,140		}141	default:142		return informer.Source{Resource: serviceResource}, informer.Options{143			Handler:   serviceHandler(w.wake),144			Transform: trimService,145		}146	}147}148149// budgetHandler wakes the loop when a budget could allow an eviction it150// refused: a budget whose status.disruptionsAllowed rose, and a budget151// that was deleted, which guards nothing any more. A new budget152// guards more, so it wakes nothing.153func budgetHandler(wake func()) cache.ResourceEventHandler {154	return cache.ResourceEventHandlerFuncs{155		UpdateFunc: func(before, after any) {156			old, okOld := before.(*unstructured.Unstructured)157			now, okNew := after.(*unstructured.Unstructured)158			if !okOld || !okNew {159				return160			}161			was, _, _ := unstructured.NestedInt64(old.Object, "status", "disruptionsAllowed")162			is, _, _ := unstructured.NestedInt64(now.Object, "status", "disruptionsAllowed")163			if is > was {164				wake()165			}166		},167		DeleteFunc: func(any) { wake() },168	}169}170171// presenceHandler wakes the loop when an object appears or leaves. A172// removal waits only on whether the objects exist, and the Helm173// controller writes each chart's status as it works, which changes174// nothing the wait reads.175func presenceHandler(wake func()) cache.ResourceEventHandler {176	return cache.ResourceEventHandlerFuncs{177		AddFunc:    func(any) { wake() },178		DeleteFunc: func(any) { wake() },179	}180}181182// serviceHandler wakes the loop when a Service appears or leaves, or183// changes its type, which can make it a LoadBalancer or stop it being184// one.185func serviceHandler(wake func()) cache.ResourceEventHandler {186	return cache.ResourceEventHandlerFuncs{187		AddFunc: func(any) { wake() },188		UpdateFunc: func(before, after any) {189			old, okOld := before.(*unstructured.Unstructured)190			now, okNew := after.(*unstructured.Unstructured)191			if !okOld || !okNew {192				return193			}194			was, _, _ := unstructured.NestedString(old.Object, "spec", "type")195			is, _, _ := unstructured.NestedString(now.Object, "spec", "type")196			if was != is {197				wake()198			}199		},200		DeleteFunc: func(any) { wake() },201	}202}203204// trimPod keeps only what the proof and the drain read from a pod205// (kubernetes.Pod), so a copy of a node's pods costs under a kilobyte206// for each pod rather than tens of kilobytes. A field the trim drops207// converts to its zero value, so the trim must keep every field the208// code reads. Without the owner references, every DaemonSet pod looks209// evictable, and without the mirror annotation, the drain evicts210// mirror pods that the kubelet creates again.211func trimPod(object *unstructured.Unstructured) {212	kept := map[string]any{213		"apiVersion": object.GetAPIVersion(),214		"kind":       object.GetKind(),215	}216	metadata := map[string]any{}217	for _, key := range []string{"name", "namespace", "uid", "resourceVersion", "ownerReferences", "deletionTimestamp"} {218		if value, ok := object.Object["metadata"].(map[string]any)[key]; ok {219			metadata[key] = value220		}221	}222	annotations := map[string]any{}223	for key, value := range object.GetAnnotations() {224		if key == mirrorPodAnnotation || key == osVersionAnnotation {225			annotations[key] = value226		}227	}228	if len(annotations) > 0 {229		metadata["annotations"] = annotations230	}231	kept["metadata"] = metadata232233	spec := map[string]any{}234	if name, ok, _ := unstructured.NestedString(object.Object, "spec", "nodeName"); ok {235		spec["nodeName"] = name236	}237	if claims, ok, _ := unstructured.NestedSlice(object.Object, "spec", "resourceClaims"); ok {238		spec["resourceClaims"] = claims239	}240	if volumes, ok, _ := unstructured.NestedSlice(object.Object, "spec", "volumes"); ok {241		var kept []any242		for _, v := range volumes {243			volume, _ := v.(map[string]any)244			trimmed := map[string]any{"name": volume["name"]}245			if hostPath, ok := volume["hostPath"].(map[string]any); ok {246				trimmed["hostPath"] = map[string]any{"path": hostPath["path"]}247			}248			kept = append(kept, trimmed)249		}250		spec["volumes"] = kept251	}252	kept["spec"] = spec253254	status := map[string]any{}255	if phase, ok, _ := unstructured.NestedString(object.Object, "status", "phase"); ok {256		status["phase"] = phase257	}258	if containers, ok, _ := unstructured.NestedSlice(object.Object, "status", "containerStatuses"); ok {259		var trimmed []any260		for _, c := range containers {261			container, _ := c.(map[string]any)262			trimmed = append(trimmed, map[string]any{263				"name": container["name"], "image": container["image"], "ready": container["ready"],264			})265		}266		status["containerStatuses"] = trimmed267	}268	kept["status"] = status269	object.Object = kept270}271272// trimService keeps a Service's identity and its type, which is all a273// removal reads, so a copy of every Service in the cluster stays small.274func trimService(object *unstructured.Unstructured) {275	kind, _, _ := unstructured.NestedString(object.Object, "spec", "type")276	object.Object = map[string]any{277		"apiVersion": object.GetAPIVersion(),278		"kind":       object.GetKind(),279		"metadata": map[string]any{280			"name":            object.GetName(),281			"namespace":       object.GetNamespace(),282			"uid":             string(object.GetUID()),283			"resourceVersion": object.GetResourceVersion(),284		},285		"spec": map[string]any{"type": kind},286	}287}288289// nodePods reads every pod on this node, from the wait's copy when it290// is ready, and from the API server otherwise.291func (r *reader) nodePods(node string) ([]kubernetes.Pod, error) {292	copy := r.waits.use(waitPods)293	if pods, ok := watch.List[kubernetes.Pod](r.view(copy)); ok {294		// The store answers in no order, and the API server lists by295		// namespace and name, so the drain asks pods to leave in the296		// same order from either.297		slices.SortFunc(pods, func(a, b kubernetes.Pod) int {298			return strings.Compare(a.Metadata.Namespace+"/"+a.Metadata.Name, b.Metadata.Namespace+"/"+b.Metadata.Name)299		})300		return pods, nil301	}302	return kubernetes.ListPodsOnNode(r.client, node)303}304305// watchBudgets keeps the budget watch running for one more pass, so a306// budget that allows a refused eviction wakes the drain.307func (r *reader) watchBudgets() {308	r.waits.use(waitBudgets)309}310311// helmCharts reads every HelmChart in the cluster.312func (r *reader) helmCharts() ([]kubernetes.HelmChart, error) {313	copy := r.waits.use(waitCharts)314	if charts, ok := watch.List[kubernetes.HelmChart](r.view(copy)); ok {315		return charts, nil316	}317	return kubernetes.ListHelmCharts(r.client)318}319320// loadBalancerServices reads every Service of type LoadBalancer in the321// cluster.322func (r *reader) loadBalancerServices() ([]kubernetes.Service, error) {323	copy := r.waits.use(waitServices)324	services, ok := watch.List[kubernetes.Service](r.view(copy))325	if !ok {326		return kubernetes.ListLoadBalancerServices(r.client)327	}328	var balanced []kubernetes.Service329	for _, s := range services {330		if s.Spec.Type == "LoadBalancer" {331			balanced = append(balanced, s)332		}333	}334	return balanced, nil335}
machine-operator/watches.go 95.0%
1package main23// The machine operator's watches, and the reads a pass makes through4// them.5//6// A pass judges six API objects: this machine's Machine, its Node, the7// Cluster, the registry credentials Secret, this node's own operator8// pod, and this node's ResourceSlice. A pass that read each object from9// the API server would send six requests on every pass from every10// machine, and on most passes find nothing new. So the operator watches11// each object, keeps a copy in memory, and the pass reads the copies,12// and a change to a copy is what wakes most passes.13//14// Each watch is scoped to the one object, or the few objects, this15// machine reads, by field selector and label selector, so no other16// machine's writes reach this pod. A five-hundred-machine fleet must17// not cost every machine five hundred wakes.18//19// A watch also decides whether a change wakes the loop at once:20//21//   - The Machine wakes the loop on every change, because the22//     conductor's grant and the sweep's Lost verdict are status writes23//     the pass must act on, and they arrive among this operator's own.24//   - The Cluster wakes the loop only on an edit to its spec. The25//     cluster operator writes the Cluster's status after every change26//     in the fleet, and none of that concerns one machine.27//   - The Node wakes the loop on every change except the heartbeat28//     times in its conditions (withoutHeartbeats). The kubelet writes29//     the Node's status every five minutes even when nothing changed,30//     and a pass for each of those writes would also hide every repair31//     from the backstop (backstop.go). A cordon, a label, or a taint32//     that somebody set by hand is reverted at once.33//   - The Secret and the operator pod wake the loop on every change,34//     and both change rarely.35//   - The ResourceSlice wakes the loop when it is deleted, or updated36//     by another writer (sliceHandler). The kubelet deletes a driver's37//     slices when it starts, so a restart of k3s deletes this node's38//     slice, and the pass writes it again. Without this wake a deleted39//     slice would stay deleted until the next pass for another reason.40//41// Three waits watch more than these six objects, and only while they42// wait: the image proof and the drain watch the node's pods, the drain43// watches the PodDisruptionBudgets, and a feature removal watches the44// HelmCharts or the Services (waits.go).45//46// A watch that the API server accepts again after an outage wakes the47// loop too, and the pass it starts reads every kind from the API48// server (reader.throughAPIOnce), because a pass whose writes failed49// during the outage has nothing else to start it.50//51// Every watch wakes the loop once when its first read is done. A pass52// that ran before then read the API server, and a change made between53// that read and the watch's first read is in no event.54//55// A copy that cannot answer, because its watch has not synced yet, is56// never read as the truth. The pass reads the API server instead, the57// way it did before the watch existed. That covers the first seconds58// of a start, and it covers a release skew: right after an upgrade,59// this binary can run under the previous release's RBAC, which grants60// no list or watch on some of these kinds. The watch then keeps61// retrying with a backoff, and every pass reads the API server until62// the new RBAC lands.63//64// A copy also stops answering after any watch that fails, such as each65// watch while k3s restarts (informer.Options.UnreadyOnWatchError), until66// the API server accepts a watch again. The reflector waits out a67// backoff of up to a minute before that, and a write made in that time,68// such as a withdrawn Cluster rollback or a removed reboot approval, is69// in no copy. A pass that acted on the copy could stage the old target70// and write the reboot intent before any 409 stopped it. So the pass71// reads the API server, which fails while it is down: the Node read72// then fails and a granted reboot skips the drain, and the Machine read73// keeps the last copy for the status the pass publishes.7475import (76	"context"77	"strings"78	"sync/atomic"7980	"github.com/liken-sh/liken/kubernetes/apiclient"81	"github.com/liken-sh/liken/kubernetes/events"82	"github.com/liken-sh/liken/kubernetes/informer"83	"github.com/liken-sh/liken/kubernetes/memo"84	"github.com/liken-sh/liken/liken/api"85	"github.com/liken-sh/liken/liken/cluster"86	"github.com/liken-sh/liken/liken/kubernetes"87	"github.com/liken-sh/liken/liken/kubernetes/watch"88	"github.com/liken-sh/liken/liken/machine"89	"k8s.io/apimachinery/pkg/apis/meta/v1/unstructured"90	"k8s.io/apimachinery/pkg/runtime/schema"91	"k8s.io/client-go/dynamic"92	"k8s.io/client-go/tools/cache"93)9495// The kinds this operator watches, as the dynamic client names them.96var (97	likenVersion     = schema.GroupVersion{Group: "liken.sh", Version: strings.TrimPrefix(api.APIVersion, "liken.sh/")}98	machineResource  = likenVersion.WithResource("machines")99	clusterResource  = likenVersion.WithResource("clusters")100	nodeResource     = schema.GroupVersionResource{Version: "v1", Resource: "nodes"}101	secretResource   = schema.GroupVersionResource{Version: "v1", Resource: "secrets"}102	podResource      = schema.GroupVersionResource{Version: "v1", Resource: "pods"}103	sliceResource    = schema.GroupVersionResource{Group: "resource.k8s.io", Version: "v1", Resource: "resourceslices"}104	credentialsKey   = "liken-system/" + kubernetes.RegistryCredentialsSecret105	operatorPodLabel = "app=liken-machine-operator"106)107108// The kind labels of liken_watch_restarts_total, one for each watch.109const (110	nodeKind          = "Node"111	clusterWatchKind  = "Cluster"112	secretKind        = "Secret"113	podKind           = "Pod"114	resourceSliceKind = "ResourceSlice"115)116117// watchKinds lists every kind label above, with the Machine's, so the118// counter reports a zero for each watch before its first restart.119var watchKinds = []string{machineKind, nodeKind, clusterWatchKind, secretKind, podKind, resourceSliceKind}120121// reader reads each API object a pass judges: from the copy a watch122// keeps, or from the API server when the copy cannot answer. A reader123// with no copies at all reads the API server every time, which is what124// the tests of a pass use.125type reader struct {126	client *apiclient.Client127128	// recorder posts the Events of a pass about its Machine (events.go).129	// It is here because every part of a pass that acts reads through130	// the reader. A nil recorder posts nothing.131	recorder *events.Recorder132133	machines    *informer.Collection134	nodes       *informer.Collection135	clusters    *informer.Collection136	credentials *informer.Collection137	ownPods     *informer.Collection138	slices      *informer.Collection139140	// machineVersions is the memo of this operator's own status writes141	// to its Machine, and of its reads of it from the API server142	// (kubernetes/memo). The watch delivers a write a moment after the143	// API server answers it, so without the memo the next pass could144	// start from a copy older than the status it just wrote. The pass145	// also writes its Node's taints and its ResourceSlice with the146	// version it read, and a copy older than its own write ends in a147	// 409 that the next pass repairs, so neither kind has a memo. Its148	// other Node writes are merge patches that a second pass sends149	// again with no harm.150	machineVersions *memo.Versions151152	// statusWritten remembers this operator's own last status write to153	// its Machine (kubernetes.OwnWrite). The Machine watch reads it to154	// tell the conductor's grant or the sweep's verdict from the echo155	// of the write, which would otherwise wake a second pass for each156	// write.157	statusWritten *kubernetes.OwnWrite158159	// slicesWritten remembers this operator's own last write of its160	// ResourceSlice (kubernetes.SliceWriter). The slice watch reads it161	// to tell another writer's update from this operator's echo.162	slicesWritten *kubernetes.SliceWriter163164	// recovered is set by a watch that the API server accepts again165	// after a watch on the same kind failed, and the loop clears it at166	// the start of the next pass. A copy is marked ready the moment the167	// resumed watch is accepted, before the events missed during the168	// outage arrive, so that pass reads every kind from the API server169	// (throughAPI).170	recovered *atomic.Bool171172	// throughAPI makes every read of this pass go to the API server.173	throughAPI bool174175	// sysctls keeps what the operator last wrote to each kernel176	// parameter (conditions.go). A nil memory remembers nothing.177	sysctls *sysctlMemory178179	// waits runs the watches of the waits (waits.go). A nil waits180	// watches nothing, and those reads go to the API server.181	waits *waits182}183184// withoutHeartbeats removes the times that the kubelet rewrites in a185// Node's conditions on each status report, so a report that changed no186// condition wakes no pass.187func withoutHeartbeats(fields map[string]any) {188	conditions, _, _ := unstructured.NestedSlice(fields, "status", "conditions")189	for _, c := range conditions {190		if condition, ok := c.(map[string]any); ok {191			delete(condition, "lastHeartbeatTime")192		}193	}194	if conditions != nil {195		_ = unstructured.SetNestedSlice(fields, conditions, "status", "conditions")196	}197}198199// within answers a reader for one pass whose requests end with ctx.200func (r *reader) within(ctx context.Context) *reader {201	pass := *r202	pass.client = kubernetes.Within(r.client, ctx)203	return &pass204}205206// observedBy answers a reader for one pass, whose client reports the207// final answer to each request into the pass's outcome. It shares208// every copy, memo, and recorder with r. A nil outcome answers r.209func (r *reader) observedBy(out *passOutcome) *reader {210	if out == nil {211		return r212	}213	pass := *r214	pass.client = r.client.WithObserver(out.observe)215	return &pass216}217218// watchThisMachine opens the watches and returns the reader over their219// copies. Each change that needs a pass calls wake. clusterName is220// empty on a machine with no cluster document, which reads no Cluster221// and no Secret, so it opens neither watch.222func watchThisMachine(ctx context.Context, watcher dynamic.Interface, client *apiclient.Client,223	name, clusterName string, wake func(), restarted func(kind string)) *reader {224	r := &reader{client: client, machineVersions: memo.New(),225		statusWritten: &kubernetes.OwnWrite{}, slicesWritten: &kubernetes.SliceWriter{}, recovered: &atomic.Bool{},226		waits: newWaits(ctx, watcher, wake, name), sysctls: newSysctlMemory()}227	start := func(kind string, source informer.Source, handler cache.ResourceEventHandler) *informer.Collection {228		return informer.Start(ctx, watcher, source, informer.Options{229			Handler:  handler,230			Synced:   wake,231			Reopened: func() { restarted(kind) },232			Recovered: func() {233				r.recovered.Store(true)234				wake()235			},236			// A copy stops answering after any failed watch, as the237			// head of this file says.238			UnreadyOnWatchError: true,239		})240	}241	named := func(n string) string { return "metadata.name=" + n }242243	machines := informer.Source{Resource: machineResource, FieldSelector: named(name)}244	nodes := informer.Source{Resource: nodeResource, FieldSelector: named(name)}245	pods := informer.Source{Resource: podResource, Namespace: "liken-system",246		LabelSelector: operatorPodLabel, FieldSelector: "spec.nodeName=" + name}247	slices := informer.Source{Resource: sliceResource, FieldSelector: named(kubernetes.ResourceSliceName(name))}248249	r.machines = start(machineKind, machines, watch.WakeOnAnotherWritersChange[machine.Machine](machines, wake, r.statusWritten.Wrote))250	r.nodes = start(nodeKind, nodes, watch.WakeOnContent[nodeObject](nodes, wake, withoutHeartbeats))251	r.ownPods = start(podKind, pods, watch.WakeOnChange[kubernetes.Pod](pods, wake))252	r.slices = start(resourceSliceKind, slices, sliceHandler(r.slicesWritten, wake))253	if clusterName != "" {254		clusters := informer.Source{Resource: clusterResource, FieldSelector: named(clusterName)}255		secrets := informer.Source{Resource: secretResource, Namespace: "liken-system",256			FieldSelector: named(kubernetes.RegistryCredentialsSecret)}257		r.clusters = start(clusterWatchKind, clusters, watch.WakeOnEdit[cluster.Cluster](clusters, wake))258		r.credentials = start(secretKind, secrets, watch.WakeOnChange[kubernetes.Secret](secrets, wake))259	}260	return r261}262263// sliceHandler wakes the loop when the slice is deleted, or updated by264// a writer other than this operator. The kubelet deletes a driver's265// slices when it starts, so a restart of k3s deletes this node's slice,266// and the pass writes it again. The operator's own write reaches the267// watch as an update at the version the write answered, and wakes268// nothing (kubernetes.SliceWriter). An update that the informer269// delivers after it reads the collection again, at a resourceVersion270// that did not move, is no change.271func sliceHandler(written *kubernetes.SliceWriter, wake func()) cache.ResourceEventHandler {272	return cache.ResourceEventHandlerFuncs{273		UpdateFunc: func(before, after any) {274			old, okOld := before.(*unstructured.Unstructured)275			now, okNew := after.(*unstructured.Unstructured)276			if !okOld || !okNew || old.GetResourceVersion() == now.GetResourceVersion() || written.Wrote(now.GetResourceVersion()) {277				return278			}279			wake()280		},281		DeleteFunc: func(any) { wake() },282	}283}284285// throughAPIOnce answers the reader for one pass: one that reads every286// kind from the API server when a watch recovered since the last pass,287// and r otherwise. The flag is set before the recovery's wake, so the288// pass that clears it reads after the recovery, whichever pass that is.289func (r *reader) throughAPIOnce() *reader {290	if r.recovered == nil || !r.recovered.Swap(false) {291		return r292	}293	pass := *r294	pass.throughAPI = true295	return &pass296}297298// machine reads this machine's own Machine. The copy answers only at299// the version of this operator's own last status write or read, and300// the pass otherwise reads the API server once, so a pass never starts301// from a copy older than the status it just wrote.302func (r *reader) machine(name string) (*machine.Machine, error) {303	if r.throughAPI {304		return r.freshMachine(name)305	}306	return watch.ReadOne[machine.Machine](r.client,307		informer.Held{View: r.machines.View(), Versions: r.machineVersions}, name, kubernetes.MachinesPath+"/"+name)308}309310// freshMachine reads this machine's Machine from the API server, and311// notes the version for the Machine's copy.312func (r *reader) freshMachine(name string) (*machine.Machine, error) {313	return informer.ReadFresh[machine.Machine](r.client, r.machineVersions, name, kubernetes.MachinesPath+"/"+name)314}315316// publishStatus writes this machine's status, and notes the version the317// API server answered for the Machine's copy. A write that fails notes318// that this operator holds no current copy, because a request that319// timed out can still have landed, so the next read goes to the API320// server.321//322// The write is also this operator's own, so the Machine watch does not323// wake the loop for its echo.324func (r *reader) publishStatus(m *machine.Machine, status *machine.MachineStatus) error {325	return r.ownStatusWrites().Send(func() (string, error) {326		var version string327		err := r.machineVersions.Send(m.Metadata.Name, func() (string, error) {328			written, err := kubernetes.PublishStatus(r.client, m, status)329			version = written330			return written, err331		})332		return version, err333	})334}335336// ownStatusWrites answers the reader's memory of its status writes, or337// one that remembers nothing for a reader with no watches.338func (r *reader) ownStatusWrites() *kubernetes.OwnWrite {339	if r.statusWritten == nil {340		return &kubernetes.OwnWrite{}341	}342	return r.statusWritten343}344345// node reads this machine's Node.346func (r *reader) node(name string) (*nodeObject, error) {347	if n, found, ok := watch.Get[nodeObject](r.view(r.nodes), name); ok {348		return orNotFound(n, found)349	}350	return getNode(r.client, name)351}352353// cluster reads the Cluster this machine belongs to.354func (r *reader) cluster(name string) (*cluster.Cluster, error) {355	if c, found, ok := watch.Get[cluster.Cluster](r.view(r.clusters), name); ok {356		return orNotFound(c, found)357	}358	return kubernetes.GetCluster(r.client, name)359}360361// registryCredentials reads the fleet's registry credentials. An362// absent Secret returns nil, nil, as GetRegistryCredentialsSecret363// does.364func (r *reader) registryCredentials() (*kubernetes.Secret, error) {365	if s, found, ok := watch.Get[kubernetes.Secret](r.view(r.credentials), credentialsKey); ok {366		if !found {367			return nil, nil368		}369		return s, nil370	}371	return kubernetes.GetRegistryCredentialsSecret(r.client)372}373374// operatorPods reads this node's own machine-operator pod, as a list375// of one, or none early in a boot.376func (r *reader) operatorPods(nodeName string) ([]kubernetes.Pod, error) {377	if pods, ok := watch.List[kubernetes.Pod](r.view(r.ownPods)); ok {378		return pods, nil379	}380	return kubernetes.List[kubernetes.Pod](r.client, ownPodPath(nodeName))381}382383// resourceSlice reads this node's ResourceSlice, nil when it does not384// exist.385func (r *reader) resourceSlice(nodeName string) (*kubernetes.ResourceSlice, error) {386	if s, found, ok := watch.Get[kubernetes.ResourceSlice](r.view(r.slices), kubernetes.ResourceSliceName(nodeName)); ok {387		if !found {388			return nil, nil389		}390		return s, nil391	}392	return kubernetes.GetResourceSlice(r.client, nodeName)393}394395// view answers a copy's view, or the view of no copy on a pass that396// reads through the API server, which answers nothing, so the read397// goes to the API server.398func (r *reader) view(c *informer.Collection) informer.View {399	if r.throughAPI {400		var none *informer.Collection401		return none.View()402	}403	return c.View()404}405406// orNotFound turns a ready store's answer into the answer a direct read407// gives: the object, or apiclient.ErrNotFound.408func orNotFound[T any](item *T, found bool) (*T, error) {409	if !found {410		return nil, apiclient.ErrNotFound411	}412	return item, nil413}414415// sliceWriter answers the reader's slice writer, or a writer with no416// memory for a reader that has none.417func (r *reader) sliceWriter() *kubernetes.SliceWriter {418	if r.slicesWritten == nil {419		return &kubernetes.SliceWriter{}420	}421	return r.slicesWritten422}
machine/channel.go 90.0%
1package machine23// The channel document shows the newest release in a release channel.4//5// liken has one linear channel. Releases only move forward. So the6// document needs only one fact: the latest version. The document is7// at the root of the channel, one URL above the releases. It names8// the newest published version.9//10// The document is advisory. It stays outside the trust chain. A11// cluster can poll the document to find a newer release, and show12// that release next to the version it runs. Adopting a release still13// needs the digest-pinned catalog entry in the Cluster document. A14// tampered channel document can show wrong information. It cannot15// change what a machine installs.1617import (18	"fmt"1920	"github.com/liken-sh/liken/liken/api"21	"sigs.k8s.io/yaml"22)2324type Channel struct {25	APIVersion string         `json:"apiVersion"`26	Kind       string         `json:"kind"`27	Metadata   api.ObjectMeta `json:"metadata"`28	Latest     string         `json:"latest"`29}3031// ParseChannel validates a channel document when it reads the32// document. ParseRelease validates a release document the same way.33// If a document is not exactly what it claims to be, ParseChannel34// rejects it with the reason. It never accepts a document partially.35func ParseChannel(raw []byte) (*Channel, error) {36	c := &Channel{}37	if err := yaml.UnmarshalStrict(raw, c); err != nil {38		return nil, err39	}40	if c.Kind != "Channel" {41		return nil, fmt.Errorf("expected kind Channel, got %q", c.Kind)42	}43	if c.Metadata.Name == "" {44		return nil, fmt.Errorf("a channel must have a name")45	}46	if err := api.ValidVersion(c.Latest); err != nil {47		return nil, fmt.Errorf("channel %s: latest: %w", c.Metadata.Name, err)48	}49	return c, nil50}
machine/drift.go 100.0%
1package machine23// Drift compares what a Machine spec declares against what a boot4// actuated. This package holds shared logic, not logic specific to5// the operator, because two programs must agree on the comparison6// exactly. The operator uses the comparison to determine whether to7// converge, and how. init uses the comparison to check that a staged8// manifest is really applicable without a reboot, before init9// applies the manifest live. If the operator and init used two10// different implementations of "is this the same storage?", the two11// implementations would eventually disagree. The operator would then12// request an action forever that init keeps refusing.1314import (15	"fmt"16	"maps"17	"slices"18	"strings"19)2021// StorageDrift compares the declared storage against what the boot22// actuated. It compares role by role and normalizes sizes, so23// 2048Mi and 2Gi declare the same thing. StorageDrift writes the24// returned diffs for people to read. The diffs appear verbatim in25// condition messages.26func StorageDrift(desired, actuated StorageSpec) []string {27	var diffs []string28	desiredRoles := rolesByName(desired)29	actuatedRoles := rolesByName(actuated)30	for _, name := range StorageRoleNames {31		d, dok := desiredRoles[name]32		a, aok := actuatedRoles[name]33		switch {34		case dok && !aok:35			diffs = append(diffs, fmt.Sprintf("%s: declared but not actuated", name))36		case !dok && aok:37			diffs = append(diffs, fmt.Sprintf("%s: actuated but no longer declared", name))38		case dok && aok:39			if d.Device != a.Device {40				diffs = append(diffs, fmt.Sprintf("%s: device %s declared, %s actuated", name, d.Device, a.Device))41			}42			if !sameSize(d.Size, a.Size) {43				diffs = append(diffs, fmt.Sprintf("%s: size %s declared, %s actuated", name, orRemainder(d.Size), orRemainder(a.Size)))44			}45		}46	}47	return diffs48}4950// NetworkDrift compares the declared network against what the boot51// actuated. Unlike storage, a network spec has no grow-only rule and52// no shape a machine must keep: any spec may replace any other, and53// the only question is whether the running machine matches the one54// the cluster asks for now.55//56// A nil actuated spec means the boot recorded no network. There is57// nothing to compare against, so there is no drift to report. The58// alternative would be to read every declared interface as drift, and59// stage a manifest and ask for a reboot on every machine whose facts60// happen to lack this record.61//62// HostEntries takes no part in this comparison. Unlike an interface,63// a host entry reconciles live (milestone 53): the machine operator64// applies spec.network.hostEntries on every pass, so an edit never65// waits for a reboot and never belongs in the set of differences66// that asks for one.67func NetworkDrift(desired NetworkSpec, actuated *NetworkSpec) []string {68	if actuated == nil {69		return nil70	}71	var diffs []string72	for i := range max(len(desired.Interfaces), len(actuated.Interfaces)) {73		switch {74		case i >= len(actuated.Interfaces):75			diffs = append(diffs, fmt.Sprintf("network: %s declared but not actuated", desired.Interfaces[i].Name))76		case i >= len(desired.Interfaces):77			diffs = append(diffs, fmt.Sprintf("network: %s actuated but no longer declared", actuated.Interfaces[i].Name))78		default:79			diffs = append(diffs, interfaceDrift(i, desired.Interfaces[i], actuated.Interfaces[i])...)80		}81	}82	return diffs83}8485// interfaceDrift compares one position of the two lists. When the two86// entries name different ports, that one difference is the whole87// report: the addressing under it belongs to another port, so88// reporting each field as well would say the same thing several times89// over.90func interfaceDrift(position int, desired, actuated InterfaceSpec) []string {91	if desired.Name != actuated.Name {92		return []string{fmt.Sprintf("network: interface %d: %s declared, %s actuated",93			position+1, desired.Name, actuated.Name)}94	}95	var diffs []string96	if desired.Address != actuated.Address {97		diffs = append(diffs, fmt.Sprintf("network: %s: address %s declared, %s actuated",98			desired.Name, orDHCP(desired.Address), orDHCP(actuated.Address)))99	}100	if desired.Gateway != actuated.Gateway {101		diffs = append(diffs, fmt.Sprintf("network: %s: gateway %s declared, %s actuated",102			desired.Name, orNone(desired.Gateway), orNone(actuated.Gateway)))103	}104	if !slices.Equal(desired.Nameservers, actuated.Nameservers) {105		diffs = append(diffs, fmt.Sprintf("network: %s: nameservers %s declared, %s actuated",106			desired.Name,107			orNone(strings.Join(desired.Nameservers, ", ")),108			orNone(strings.Join(actuated.Nameservers, ", "))))109	}110	if wirelessSummary(desired.Wireless) != wirelessSummary(actuated.Wireless) {111		diffs = append(diffs, fmt.Sprintf("network: %s: wireless %s declared, %s actuated",112			desired.Name,113			orNone(wirelessSummary(desired.Wireless)),114			orNone(wirelessSummary(actuated.Wireless))))115	}116	return diffs117}118119// wirelessSummary renders one wireless entry as the pair of facts120// that decide whether a rejoin is needed. The security is resolved121// through its default first, so a spec that left the field unset and122// a record that holds the default read as the same request instead123// of as drift that never settles.124func wirelessSummary(w *WirelessSpec) string {125	if w == nil {126		return ""127	}128	return fmt.Sprintf("%s (%s)", w.SSID, w.SecurityOrDefault())129}130131// RlimitDrift compares the declared resource limits against what the132// boot actuated. Both sides are the spec's own map, so this is a133// straight comparison of two requests, the way ModulesDrift compares134// two module lists.135//136// The limits every liken machine holds take no part in this. They ship137// with the release rather than with the spec, so a change to the table138// arrives with a new system image and the reboot that installs it.139// Comparing them here would read as drift on every machine in a fleet140// the moment the table changed.141//142// A value is compared as written, not as parsed. "1048576" and143// "1048576:1048576" set the same pair of numbers, and this reports144// them as drift. That costs one reboot on a machine whose operator145// rewrote a value without changing it, and the alternative costs a146// parse in the one comparison that decides whether to reboot a147// machine. A drift that reboots once too often is a smaller fault than148// one that never reboots at all.149func RlimitDrift(desired, actuated map[string]string) []string {150	var diffs []string151	names := map[string]bool{}152	for name := range desired {153		names[name] = true154	}155	for name := range actuated {156		names[name] = true157	}158	for _, name := range slices.Sorted(maps.Keys(names)) {159		d, dok := desired[name]160		a, aok := actuated[name]161		switch {162		case dok && !aok:163			diffs = append(diffs, fmt.Sprintf("rlimit %s: %s declared but not actuated", name, d))164		case !dok && aok:165			diffs = append(diffs, fmt.Sprintf("rlimit %s: %s actuated but no longer declared", name, a))166		case d != a:167			diffs = append(diffs, fmt.Sprintf("rlimit %s: %s declared, %s actuated", name, d, a))168		}169	}170	return diffs171}172173// ModuleSetDiff compares two module lists as sets. Order and174// repetition carry no meaning in these lists. ModuleSetDiff reports175// both directions separately, because the two directions converge in176// different ways. The system can load an added module into the177// running kernel. But a retracted module can only leave the system178// at a reboot. The kernel has no safe way to remove a driver while179// something else uses it.180func ModuleSetDiff(desired, actuated []string) (added, retracted []string) {181	want := map[string]bool{}182	for _, name := range desired {183		want[name] = true184	}185	have := map[string]bool{}186	for _, name := range actuated {187		have[name] = true188	}189	for _, name := range slices.Sorted(maps.Keys(want)) {190		if !have[name] {191			added = append(added, name)192		}193	}194	for _, name := range slices.Sorted(maps.Keys(have)) {195		if !want[name] {196			retracted = append(retracted, name)197		}198	}199	return added, retracted200}201202// ModulesDrift writes ModuleSetDiff results for people to read, the203// same way StorageDrift does. The actuated side is the boot record's204// copy of the request, not the result of loading modules. This205// design is deliberate. A declared module that the image lacked206// still counts as actuated, because rebooting again with the same207// image would change nothing. The ModulesLoaded condition reports208// that problem instead. The fix for that problem is a new image, not209// a reboot.210//211// Parameter drift rides in this same list rather than in a list of212// its own, because the live-load rule counts: a staged spec applies213// live only when added modules are the whole of the drift214// (converge.go). A parameter line in the count makes any parameter215// change on a loaded module miss that test and take the reboot path,216// with no new rule written anywhere.217func ModulesDrift(desired, actuated []string, desiredParameters, actuatedParameters map[string]string) []string {218	added, retracted := ModuleSetDiff(desired, actuated)219	var diffs []string220	for _, name := range added {221		diffs = append(diffs, fmt.Sprintf("modules: %s declared but this boot ran without it", name))222	}223	for _, name := range retracted {224		diffs = append(diffs, fmt.Sprintf("modules: %s no longer declared but this boot ran with it", name))225	}226	return append(diffs, ModuleParameterDrift(desired, actuated, desiredParameters, actuatedParameters)...)227}228229// ModuleOrderDrift writes one line when the declared order of the230// modules differs from the order the boot loaded them in.231//232// Order is drift because the position of a name decides which driver233// claims the hardware. When a sound controller loads before the234// codec's own driver, the kernel binds the codec to the generic235// driver, and the codec's outputs never appear. Only a boot can load236// the list again in a new order, so a reorder stages the manifest237// and asks for no disruption (machine-operator/converge.go).238//239// The comparison is its own function beside the set diff above, and240// it uses only the modules both lists hold. An addition or a241// retraction is then never read as a reorder. It reports the earliest242// pair that changed places, because one pair names the change.243func ModuleOrderDrift(desired, actuated []string) []string {244	shared := sharedModules(desired, actuated)245	declared := sharedOrder(desired, shared)246	loaded := sharedOrder(actuated, shared)247	for i := range declared {248		if declared[i] != loaded[i] {249			return []string{fmt.Sprintf("modules: the declared order changed: %s now loads before %s", declared[i], loaded[i])}250		}251	}252	return nil253}254255// sharedOrder reduces a list to the shared modules alone, in the256// list's own order and with repetition dropped. The two reduced lists257// hold the same names, so they compare position by position.258func sharedOrder(names []string, shared map[string]bool) []string {259	var order []string260	for _, name := range names {261		if shared[name] && !slices.Contains(order, name) {262			order = append(order, name)263		}264	}265	return order266}267268// ModuleParameterDrift writes a line only for a module that both the269// desired and the actuated sets declare. A parameter that arrives270// with a module the boot never loaded is part of that module's own271// added line: the live loader passes the string at the load, because272// a module loading for the first time takes its parameters normally.273// A parameter on a module the boot already loaded writes its own274// line here, and that line is what forces the reboot path, because a275// loaded module never reads its parameters again.276func ModuleParameterDrift(desiredModules, actuatedModules []string, desired, actuated map[string]string) []string {277	shared := sharedModules(desiredModules, actuatedModules)278	keys := map[string]bool{}279	for key := range desired {280		keys[key] = true281	}282	for key := range actuated {283		keys[key] = true284	}285	var diffs []string286	for _, key := range slices.Sorted(maps.Keys(keys)) {287		module, _, ok := splitModuleParameterKey(key)288		if !ok || !shared[module] {289			continue290		}291		d, dok := desired[key]292		a, aok := actuated[key]293		switch {294		case dok && !aok:295			diffs = append(diffs, fmt.Sprintf("module parameter %s: %s declared but not actuated", key, d))296		case !dok && aok:297			diffs = append(diffs, fmt.Sprintf("module parameter %s: %s actuated but no longer declared", key, a))298		case d != a:299			diffs = append(diffs, fmt.Sprintf("module parameter %s: %s declared, %s actuated", key, d, a))300		}301	}302	return diffs303}304305// sharedModules is the set of modules present in both lists, the306// only modules a parameter drift line may name.307func sharedModules(desired, actuated []string) map[string]bool {308	have := map[string]bool{}309	for _, name := range actuated {310		have[name] = true311	}312	shared := map[string]bool{}313	for _, name := range desired {314		if have[name] {315			shared[name] = true316		}317	}318	return shared319}320321func rolesByName(spec StorageSpec) map[StorageRoleName]DeclaredRole {322	byName := map[StorageRoleName]DeclaredRole{}323	for _, role := range spec.Roles() {324		byName[role.Name] = role325	}326	return byName327}328329// sameSize compares two size declarations by the number of bytes330// each one describes, not by the spelling. If a size cannot be331// parsed, sameSize falls back to a string comparison instead of a332// panic. Validation refuses an unparseable size anyway.333func sameSize(a, b string) bool {334	if a == "" || b == "" {335		return a == b336	}337	aBytes, aErr := ParseSize(a)338	bBytes, bErr := ParseSize(b)339	if aErr != nil || bErr != nil {340		return a == b341	}342	return aBytes == bBytes343}344345func orRemainder(size string) string {346	if size == "" {347		return "(remainder)"348	}349	return size350}351352// orDHCP names the empty address for a reader. An interface with no353// declared address asks for DHCP, so the diff says so rather than354// leaving a gap where a value should be.355func orDHCP(address string) string {356	if address == "" {357		return "(DHCP)"358	}359	return address360}361362// orNone names an empty optional field for a reader, for the same363// reason orDHCP does.364func orNone(value string) string {365	if value == "" {366		return "(none)"367	}368	return value369}370371// SerioSetDiff compares two spec.serio lists as sets of whole entries.372// Order carries no meaning in the list. The two directions converge373// differently, the same way modules do: init attaches an added entry374// while the machine runs, and a retracted entry's holder keeps the375// port until the next boot, because a pod may hold the devices the376// port created.377func SerioSetDiff(desired, actuated []SerioAttachment) (added, retracted []SerioAttachment) {378	for _, a := range desired {379		if !slices.Contains(actuated, a) && !slices.Contains(added, a) {380			added = append(added, a)381		}382	}383	for _, a := range actuated {384		if !slices.Contains(desired, a) && !slices.Contains(retracted, a) {385			retracted = append(retracted, a)386		}387	}388	return added, retracted389}390391// SerioDrift writes SerioSetDiff for people to read, one line per392// entry. The line count is part of the live-load rule393// (machine-operator/converge.go): a spec applies live only when added394// modules and added serio entries are the whole of the drift.395func SerioDrift(desired, actuated []SerioAttachment) []string {396	added, retracted := SerioSetDiff(desired, actuated)397	var diffs []string398	for _, a := range added {399		diffs = append(diffs, fmt.Sprintf("serio: %s declared but this boot ran without it", a))400	}401	for _, a := range retracted {402		diffs = append(diffs, fmt.Sprintf("serio: %s no longer declared but this boot ran with it", a))403	}404	return diffs405}
machine/factsread.go 77.1%
1package machine23// The read side of the facts tree. Only the operator reads the tree,4// and only the operator assembles a MachineStatus from it. Read walks5// every subtree that a writer fills, and returns the status that those6// files describe.7//8// Read leaves four fields at their zero value: phase,9// observedGeneration, sysctls, and conditions. The operator owns these10// fields and overlays them on every pass, because they are the11// operator's own observation, not init's.1213import (14	"errors"15	"fmt"16	"io/fs"17	"os"18	"path/filepath"19	"sort"20	"strings"2122	"github.com/liken-sh/liken/liken/api"23)2425// Read assembles a MachineStatus from the facts tree. A missing root is26// an error that wraps fs.ErrNotExist: the facts describe a boot, and27// their absence on a running machine is a fact the operator must report,28// not a default it may invent. Read ignores a file or directory it29// does not recognize, so a tree can carry more than this reader30// consumes. A file that does not parse is an error that names its31// path.32func (t FactsTree) Read() (*MachineStatus, error) {33	if _, err := os.Stat(t.Dir); err != nil {34		return nil, fmt.Errorf("reading facts tree %s: %w", t.Dir, err)35	}3637	s := &MachineStatus{}38	var err error3940	if s.Role, err = t.readRole(); err != nil {41		return nil, err42	}43	if s.LastCrash, err = t.readLastCrash(); err != nil {44		return nil, err45	}46	if s.LastFailStop, err = t.readLastFailStop(); err != nil {47		return nil, err48	}49	if s.Version, err = t.readVersion(); err != nil {50		return nil, err51	}52	if s.Network, err = t.readNetwork(); err != nil {53		return nil, err54	}55	if s.Time, err = t.readTimeStatus(); err != nil {56		return nil, err57	}58	if s.Hardware, err = t.readHardware(); err != nil {59		return nil, err60	}61	if s.Firmware, err = t.readFirmware(); err != nil {62		return nil, err63	}64	// Rlimits comes from the tree, unlike sysctls above. The operator65	// can read a sysctl back itself at any time, but a resource limit66	// belongs to a process, and the process that matters is init. Only67	// init can report it.68	if s.Rlimits, err = t.readKeyedScalars("rlimits"); err != nil {69		return nil, err70	}71	if s.Storage, err = t.readStorage(); err != nil {72		return nil, err73	}74	if s.Modules, err = t.readModules(); err != nil {75		return nil, err76	}77	if s.Serio, err = t.readSerio(); err != nil {78		return nil, err79	}80	if s.Features, err = t.readFeatures(); err != nil {81		return nil, err82	}83	if s.Registries, err = t.readRegistries(); err != nil {84		return nil, err85	}86	if s.Runtime, err = t.readRuntime(); err != nil {87		return nil, err88	}89	if s.Boot, err = t.readBoot(); err != nil {90		return nil, err91	}92	return s, nil93}9495// readKeyedScalars reads a directory of scalar files back into a map,96// the reader for what writeKeyedScalars publishes. An absent directory97// reads as no map at all, which is rule 2's omitempty semantics.98func (t FactsTree) readKeyedScalars(rel string) (map[string]string, error) {99	entries, err := os.ReadDir(filepath.Join(t.Dir, rel))100	if errors.Is(err, fs.ErrNotExist) {101		return nil, nil102	}103	if err != nil {104		return nil, err105	}106	values := map[string]string{}107	for _, entry := range entries {108		if entry.IsDir() {109			continue110		}111		// writeAtomic stages its temp file inside this directory as112		// .liken-*, and a read can race the rename; a dotfile is113		// never a key.114		if strings.HasPrefix(entry.Name(), ".") {115			continue116		}117		value, err := t.readFact(filepath.Join(rel, entry.Name()))118		if err != nil {119			return nil, err120		}121		values[entry.Name()] = value122	}123	if len(values) == 0 {124		return nil, nil125	}126	return values, nil127}128129func (t FactsTree) readRuntime() (RuntimeStatus, error) {130	r := RuntimeStatus{}131	var err error132	if r.K3s.GoMemoryLimit, err = t.readFact("runtime/k3s/goMemoryLimit"); err != nil {133		return RuntimeStatus{}, err134	}135	if r.K3s.GoGC, err = t.readInt("runtime/k3s/goGC"); err != nil {136		return RuntimeStatus{}, err137	}138	debug, err := t.readFact("runtime/k3s/debug")139	if err != nil {140		return RuntimeStatus{}, err141	}142	r.K3s.Debug = debug == "true"143	if r.Containerd.LogLevel, err = t.readFact("runtime/containerd/logLevel"); err != nil {144		return RuntimeStatus{}, err145	}146	gc := &r.Kubelet.ImageGC147	if gc.HighThresholdPercent, err = t.readInt("runtime/kubelet/imageGC/highThresholdPercent"); err != nil {148		return RuntimeStatus{}, err149	}150	if gc.LowThresholdPercent, err = t.readInt("runtime/kubelet/imageGC/lowThresholdPercent"); err != nil {151		return RuntimeStatus{}, err152	}153	if gc.MaximumAge, err = t.readFact("runtime/kubelet/imageGC/maximumAge"); err != nil {154		return RuntimeStatus{}, err155	}156	if gc.MinimumAge, err = t.readFact("runtime/kubelet/imageGC/minimumAge"); err != nil {157		return RuntimeStatus{}, err158	}159	return r, nil160}161162func (t FactsTree) readRole() (api.Role, error) {163	value, err := t.readFact("role")164	return api.Role(value), err165}166167func (t FactsTree) readLastCrash() (*CrashStatus, error) {168	if _, err := os.Stat(filepath.Join(t.Dir, "lastCrash")); err != nil {169		if errors.Is(err, fs.ErrNotExist) {170			return nil, nil171		}172		return nil, err173	}174	c := &CrashStatus{}175	var err error176	if c.Time, err = t.readTime("lastCrash/time"); err != nil {177		return nil, err178	}179	reason, err := t.readFact("lastCrash/reason")180	if err != nil {181		return nil, err182	}183	c.Reason = CrashReason(reason)184	if c.Message, err = t.readFact("lastCrash/message"); err != nil {185		return nil, err186	}187	if c.Records, err = t.readFact("lastCrash/records"); err != nil {188		return nil, err189	}190	return c, nil191}192193func (t FactsTree) readLastFailStop() (*FailStop, error) {194	if _, err := os.Stat(filepath.Join(t.Dir, "lastFailStop")); err != nil {195		if errors.Is(err, fs.ErrNotExist) {196			return nil, nil197		}198		return nil, err199	}200	f := &FailStop{}201	when, err := t.readTime("lastFailStop/time")202	if err != nil {203		return nil, err204	}205	if when != nil {206		f.Time = *when207	}208	if f.Reason, err = t.readFact("lastFailStop/reason"); err != nil {209		return nil, err210	}211	return f, nil212}213214func (t FactsTree) readVersion() (VersionStatus, error) {215	v := VersionStatus{}216	fields := []struct {217		rel string218		dst *string219	}{220		{"version/liken", &v.Liken},221		{"version/kernel", &v.Kernel},222		{"version/xtables", &v.Xtables},223		{"version/k3s", &v.K3s},224		{"version/trust", &v.Trust},225		{"version/e2fsprogs", &v.E2fsprogs},226		{"version/openIscsi", &v.OpenISCSI},227		{"version/nfsUtils", &v.NFSUtils},228		{"version/wpaSupplicant", &v.WPASupplicant},229		{"version/systemdBoot", &v.SystemdBoot},230		{"version/grub", &v.Grub},231		{"version/hwdata", &v.Hwdata},232		{"version/tzdata", &v.Tzdata},233		{"version/linuxFirmware", &v.LinuxFirmware},234		{"version/wirelessRegdb", &v.WirelessRegdb},235		{"version/microcode", &v.Microcode},236		{"version/microcodeRevision", &v.MicrocodeRevision},237	}238	for _, f := range fields {239		value, err := t.readFact(f.rel)240		if err != nil {241			return VersionStatus{}, err242		}243		*f.dst = value244	}245	return v, nil246}247248func (t FactsTree) readNetwork() (NetworkStatus, error) {249	n := NetworkStatus{}250	var err error251	if n.Interface, err = t.readFact("network/interface"); err != nil {252		return NetworkStatus{}, err253	}254	if n.MAC, err = t.readFact("network/mac"); err != nil {255		return NetworkStatus{}, err256	}257	if n.Gateway, err = t.readFact("network/gateway"); err != nil {258		return NetworkStatus{}, err259	}260	if n.LeaseExpires, err = t.readTime("network/leaseExpires"); err != nil {261		return NetworkStatus{}, err262	}263	if n.Addresses, err = t.readListFact("network/addresses"); err != nil {264		return NetworkStatus{}, err265	}266	if n.Nameservers, err = t.readListFact("network/nameservers"); err != nil {267		return NetworkStatus{}, err268	}269	names, err := t.entryKeys("network/interfaces")270	if err != nil {271		return NetworkStatus{}, err272	}273	for _, name := range names {274		iface := InterfaceStatus{Name: name}275		base := filepath.Join("network", "interfaces", name)276		if iface.MAC, err = t.readFact(filepath.Join(base, "mac")); err != nil {277			return NetworkStatus{}, err278		}279		if iface.Address, err = t.readFact(filepath.Join(base, "address")); err != nil {280			return NetworkStatus{}, err281		}282		method, err := t.readFact(filepath.Join(base, "method"))283		if err != nil {284			return NetworkStatus{}, err285		}286		iface.Method = AddressMethod(method)287		if iface.Gateway, err = t.readFact(filepath.Join(base, "gateway")); err != nil {288			return NetworkStatus{}, err289		}290		if iface.LeaseExpires, err = t.readTime(filepath.Join(base, "leaseExpires")); err != nil {291			return NetworkStatus{}, err292		}293		if iface.Nameservers, err = t.readListFact(filepath.Join(base, "nameservers")); err != nil {294			return NetworkStatus{}, err295		}296		if iface.Wireless, err = t.readInterfaceWireless(base); err != nil {297			return NetworkStatus{}, err298		}299		n.Interfaces = append(n.Interfaces, iface)300	}301	return n, nil302}303304// readInterfaceWireless reads the radio's standing for one305// interface. The ssid file is the record's presence: a wired306// interface writes none, so an absent or empty ssid reads back as no307// status at all rather than an empty one.308func (t FactsTree) readInterfaceWireless(base string) (*WirelessStatus, error) {309	dir := filepath.Join(base, "wireless")310	ssid, err := t.readFact(filepath.Join(dir, "ssid"))311	if err != nil || ssid == "" {312		return nil, err313	}314	w := &WirelessStatus{SSID: ssid}315	state, err := t.readFact(filepath.Join(dir, "state"))316	if err != nil {317		return nil, err318	}319	w.State = WirelessState(state)320	if w.Message, err = t.readFact(filepath.Join(dir, "message")); err != nil {321		return nil, err322	}323	return w, nil324}325326func (t FactsTree) readTimeStatus() (TimeStatus, error) {327	ts := TimeStatus{}328	state, err := t.readFact("time/state")329	if err != nil {330		return TimeStatus{}, err331	}332	ts.State = TimeState(state)333	if ts.Source, err = t.readFact("time/source"); err != nil {334		return TimeStatus{}, err335	}336	if ts.Stratum, err = t.readInt("time/stratum"); err != nil {337		return TimeStatus{}, err338	}339	if ts.Offset, err = t.readFact("time/offset"); err != nil {340		return TimeStatus{}, err341	}342	return ts, nil343}344345func (t FactsTree) readHardware() (HardwareStatus, error) {346	h := HardwareStatus{}347	var err error348	if h.CPUs, err = t.readInt("hardware/cpus"); err != nil {349		return HardwareStatus{}, err350	}351	if h.MemoryBytes, err = t.readUint("hardware/memoryBytes"); err != nil {352		return HardwareStatus{}, err353	}354	if h.BlockDevices, err = t.readBlockDevices(); err != nil {355		return HardwareStatus{}, err356	}357	if h.Unclaimed, err = t.readUnclaimed(); err != nil {358		return HardwareStatus{}, err359	}360	return h, nil361}362363func (t FactsTree) readBlockDevices() ([]BlockDevice, error) {364	names, err := t.entryKeys("hardware/blockDevices")365	if err != nil {366		return nil, err367	}368	var devices []BlockDevice369	for _, name := range names {370		d := BlockDevice{Name: name}371		base := filepath.Join("hardware", "blockDevices", name)372		if d.SizeBytes, err = t.readUint(filepath.Join(base, "sizeBytes")); err != nil {373			return nil, err374		}375		if d.Model, err = t.readFact(filepath.Join(base, "model")); err != nil {376			return nil, err377		}378		if d.Serial, err = t.readFact(filepath.Join(base, "serial")); err != nil {379			return nil, err380		}381		if d.StableNames, err = t.readListFact(filepath.Join(base, "stableNames")); err != nil {382			return nil, err383		}384		devices = append(devices, d)385	}386	return devices, nil387}388389func (t FactsTree) readUnclaimed() ([]UnclaimedDevice, error) {390	keys, err := t.entryKeys("hardware/unclaimed")391	if err != nil {392		return nil, err393	}394	var devices []UnclaimedDevice395	for _, key := range keys {396		d := UnclaimedDevice{}397		base := filepath.Join("hardware", "unclaimed", key)398		if d.Modalias, err = t.readFact(filepath.Join(base, "modalias")); err != nil {399			return nil, err400		}401		if d.Bus, err = t.readFact(filepath.Join(base, "bus")); err != nil {402			return nil, err403		}404		if d.Name, err = t.readFact(filepath.Join(base, "name")); err != nil {405			return nil, err406		}407		if d.Class, err = t.readFact(filepath.Join(base, "class")); err != nil {408			return nil, err409		}410		if d.Message, err = t.readFact(filepath.Join(base, "message")); err != nil {411			return nil, err412		}413		if d.Candidates, err = t.readListFact(filepath.Join(base, "candidates")); err != nil {414			return nil, err415		}416		devices = append(devices, d)417	}418	return devices, nil419}420421func (t FactsTree) readFirmware() (FirmwareStatus, error) {422	f := FirmwareStatus{}423	mode, err := t.readFact("firmware/mode")424	if err != nil {425		return FirmwareStatus{}, err426	}427	f.Mode = FirmwareMode(mode)428	if f.BootCurrent, err = t.readFact("firmware/bootCurrent"); err != nil {429		return FirmwareStatus{}, err430	}431	if f.BootNext, err = t.readFact("firmware/bootNext"); err != nil {432		return FirmwareStatus{}, err433	}434	if f.BootOrder, err = t.readListFact("firmware/bootOrder"); err != nil {435		return FirmwareStatus{}, err436	}437	return f, nil438}439440func (t FactsTree) readStorage() (StorageStatus, error) {441	s := StorageStatus{}442	for _, name := range StorageRoleNames {443		role := s.Role(name)444		base := filepath.Join("storage", string(name))445		backing, err := t.readFact(filepath.Join(base, "backing"))446		if err != nil {447			return StorageStatus{}, err448		}449		role.Backing = Backing(backing)450		if role.Device, err = t.readFact(filepath.Join(base, "device")); err != nil {451			return StorageStatus{}, err452		}453		if role.Partition, err = t.readFact(filepath.Join(base, "partition")); err != nil {454			return StorageStatus{}, err455		}456		if role.CapacityBytes, err = t.readUint(filepath.Join(base, "capacityBytes")); err != nil {457			return StorageStatus{}, err458		}459		unclean, err := t.readFact(filepath.Join(base, "lastStopUnclean"))460		if err != nil {461			return StorageStatus{}, err462		}463		role.LastStopUnclean = unclean == "true"464	}465	return s, nil466}467468func (t FactsTree) readModules() ([]ModuleStatus, error) {469	names, err := t.entryKeys("modules")470	if err != nil {471		return nil, err472	}473	var modules []ModuleStatus474	for _, name := range names {475		m := ModuleStatus{Name: name}476		base := filepath.Join("modules", name)477		state, err := t.readFact(filepath.Join(base, "state"))478		if err != nil {479			return nil, err480		}481		m.State = ModuleState(state)482		if m.Message, err = t.readFact(filepath.Join(base, "message")); err != nil {483			return nil, err484		}485		resident, err := t.readFact(filepath.Join(base, "alreadyResident"))486		if err != nil {487			return nil, err488		}489		m.AlreadyResident = resident == "true"490		if m.Parameters, err = t.readKeyedScalars(filepath.Join(base, "parameters")); err != nil {491			return nil, err492		}493		modules = append(modules, m)494	}495	return modules, nil496}497498func (t FactsTree) readFeatures() ([]FeatureStatus, error) {499	names, err := t.entryKeys("features")500	if err != nil {501		return nil, err502	}503	var features []FeatureStatus504	for _, name := range names {505		f := FeatureStatus{Name: name}506		base := filepath.Join("features", name)507		state, err := t.readFact(filepath.Join(base, "state"))508		if err != nil {509			return nil, err510		}511		f.State = FeatureState(state)512		if f.Message, err = t.readFact(filepath.Join(base, "message")); err != nil {513			return nil, err514		}515		features = append(features, f)516	}517	return features, nil518}519520func (t FactsTree) readRegistries() (RegistriesStatus, error) {521	r := RegistriesStatus{}522	var err error523	if r.Mirrors, err = t.readListFact("registries/mirrors"); err != nil {524		return RegistriesStatus{}, err525	}526	if r.CredentialedHosts, err = t.readListFact("registries/credentialedHosts"); err != nil {527		return RegistriesStatus{}, err528	}529	embedded, err := t.readFact("registries/embedded")530	if err != nil {531		return RegistriesStatus{}, err532	}533	r.Embedded = embedded == "true"534	return r, nil535}536537// entryKeys returns the sorted names of the subdirectories under a538// collection directory. It is how the reader discovers a collection's539// elements. It ignores a plain file, so a stray file left beside the540// element directories does not become a phantom element. A directory541// that does not exist is an empty collection. The order is sorted, so542// a round trip through the tree returns the elements in one stable543// order.544func (t FactsTree) entryKeys(rel string) ([]string, error) {545	entries, err := os.ReadDir(filepath.Join(t.Dir, rel))546	if errors.Is(err, fs.ErrNotExist) {547		return nil, nil548	}549	if err != nil {550		return nil, err551	}552	var keys []string553	for _, entry := range entries {554		if entry.IsDir() {555			keys = append(keys, entry.Name())556		}557	}558	sort.Strings(keys)559	return keys, nil560}
machine/factsreadboot.go 80.7%
1package machine23// The read side of the boot subtree. Boot holds the configuration this4// boot ran under: the four manifest records, the actuated storage and5// network, and the four standing rejections. It is the half of drift6// detection that only init can supply, so the operator reads it here7// and compares it against the cluster's copies.89import (10	"errors"11	"io/fs"12	"os"13	"path/filepath"14	"strconv"15)1617func (t FactsTree) readBoot() (BootStatus, error) {18	b := BootStatus{}1920	var err error21	if b.Time, err = t.readTime("boot/time"); err != nil {22		return BootStatus{}, err23	}2425	manifest, err := t.readRecordFact("boot/manifest")26	if err != nil {27		return BootStatus{}, err28	}29	b.ManifestSource = ManifestSource(manifest["source"])30	b.ManifestHash = manifest["hash"]3132	cluster, err := t.readRecordFact("boot/clusterManifest")33	if err != nil {34		return BootStatus{}, err35	}36	b.ClusterManifestSource = ManifestSource(cluster["source"])37	b.ClusterManifestHash = cluster["hash"]3839	credentials, err := t.readRecordFact("boot/credentials")40	if err != nil {41		return BootStatus{}, err42	}43	b.CredentialsSource = ManifestSource(credentials["source"])44	b.CredentialsHash = credentials["hash"]4546	imports, err := t.readRecordFact("boot/imports")47	if err != nil {48		return BootStatus{}, err49	}50	b.ImportsSource = ManifestSource(imports["source"])51	b.ImportsHash = imports["hash"]52	b.ImportsDiscarded = imports["discarded"] == "true"5354	if b.Slot, err = t.readFact("boot/slot"); err != nil {55		return BootStatus{}, err56	}57	if b.CommandLine, err = t.readFact("boot/commandLine"); err != nil {58		return BootStatus{}, err59	}60	if b.Restarts, err = t.readInt("boot/restarts"); err != nil {61		return BootStatus{}, err62	}63	if b.Modules, err = t.readListFact("boot/modules"); err != nil {64		return BootStatus{}, err65	}66	if b.Rlimits, err = t.readKeyedScalars("boot/rlimits"); err != nil {67		return BootStatus{}, err68	}69	if b.ModuleParameters, err = t.readKeyedScalars("boot/moduleParameters"); err != nil {70		return BootStatus{}, err71	}72	if b.Serio, err = t.readBootSerio(); err != nil {73		return BootStatus{}, err74	}75	if b.Storage, err = t.readBootStorage(); err != nil {76		return BootStatus{}, err77	}78	if b.Network, err = t.readBootNetwork(); err != nil {79		return BootStatus{}, err80	}81	if b.Rejection, err = t.readRejection(RejectMachine); err != nil {82		return BootStatus{}, err83	}84	if b.ClusterRejection, err = t.readRejection(RejectCluster); err != nil {85		return BootStatus{}, err86	}87	if b.SystemRejection, err = t.readRejection(RejectSystem); err != nil {88		return BootStatus{}, err89	}90	if b.CredentialsRejection, err = t.readRejection(RejectCredentials); err != nil {91		return BootStatus{}, err92	}93	return b, nil94}9596// readBootStorage reads the actuated storage spec, one directory for97// each declared role. A role directory that is absent means the spec98// did not declare that role, so its pointer stays nil.99func (t FactsTree) readBootStorage() (StorageSpec, error) {100	spec := StorageSpec{}101	roleFields := map[StorageRoleName]**StorageRole{102		BIOSBootRole:         &spec.BIOSBoot,103		BootHomeRole:         &spec.BootHome,104		SystemARole:          &spec.SystemA,105		SystemBRole:          &spec.SystemB,106		MachineStateRole:     &spec.MachineState,107		MachineEphemeralRole: &spec.MachineEphemeral,108		ClusterStateRole:     &spec.ClusterState,109		PodStorageRole:       &spec.PodStorage,110		PodEphemeralRole:     &spec.PodEphemeral,111	}112	for _, name := range StorageRoleNames {113		base := filepath.Join("boot", "storage", string(name))114		if _, err := os.Stat(filepath.Join(t.Dir, base)); err != nil {115			if errors.Is(err, fs.ErrNotExist) {116				continue117			}118			return StorageSpec{}, err119		}120		role := &StorageRole{}121		var err error122		if role.Device, err = t.readFact(filepath.Join(base, "device")); err != nil {123			return StorageSpec{}, err124		}125		if role.Size, err = t.readFact(filepath.Join(base, "size")); err != nil {126			return StorageSpec{}, err127		}128		*roleFields[name] = role129	}130	return spec, nil131}132133// readBootNetwork reads the network spec the boot actuated. The134// boot/network directory is the record's presence, so a tree without135// it comes from a boot that reported nothing about its network, and136// the pointer stays nil. A record that holds no interface directory137// is a spec that declared no interface, which is a different fact.138//139// The interfaces are read by position, in order, until a position is140// absent. Reading them this way keeps the declared order, which141// sorted directory names would lose as soon as a machine declared ten142// of them.143func (t FactsTree) readBootNetwork() (*NetworkSpec, error) {144	base := filepath.Join("boot", "network")145	if _, err := os.Stat(filepath.Join(t.Dir, base)); err != nil {146		if errors.Is(err, fs.ErrNotExist) {147			return nil, nil148		}149		return nil, err150	}151	spec := &NetworkSpec{}152	for i := 0; ; i++ {153		dir := filepath.Join(base, "interfaces", strconv.Itoa(i))154		if _, err := os.Stat(filepath.Join(t.Dir, dir)); err != nil {155			if errors.Is(err, fs.ErrNotExist) {156				return spec, nil157			}158			return nil, err159		}160		ifc := InterfaceSpec{}161		var err error162		if ifc.Name, err = t.readFact(filepath.Join(dir, "name")); err != nil {163			return nil, err164		}165		if ifc.Address, err = t.readFact(filepath.Join(dir, "address")); err != nil {166			return nil, err167		}168		if ifc.Gateway, err = t.readFact(filepath.Join(dir, "gateway")); err != nil {169			return nil, err170		}171		if ifc.Nameservers, err = t.readListFact(filepath.Join(dir, "nameservers")); err != nil {172			return nil, err173		}174		if ifc.Wireless, err = t.readBootWireless(dir); err != nil {175			return nil, err176		}177		spec.Interfaces = append(spec.Interfaces, ifc)178	}179}180181// readBootWireless reads the wireless half of one interface's boot182// record. The ssid file is the record's presence, so an interface183// that came up wired reads back as a nil pointer, not as an empty184// spec.185func (t FactsTree) readBootWireless(dir string) (*WirelessSpec, error) {186	base := filepath.Join(dir, "wireless")187	ssid, err := t.readFact(filepath.Join(base, "ssid"))188	if err != nil || ssid == "" {189		return nil, err190	}191	security, err := t.readFact(filepath.Join(base, "security"))192	if err != nil {193		return nil, err194	}195	return &WirelessSpec{SSID: ssid, Security: WirelessSecurity(security)}, nil196}197198// readRejection reads one standing quarantine record. A record199// directory that is absent means the document has no standing200// rejection, so the pointer stays nil.201func (t FactsTree) readRejection(kind RejectionKind) (*Rejection, error) {202	base := filepath.Join("boot", string(kind))203	if _, err := os.Stat(filepath.Join(t.Dir, base)); err != nil {204		if errors.Is(err, fs.ErrNotExist) {205			return nil, nil206		}207		return nil, err208	}209	r := &Rejection{}210	var err error211	if r.Hash, err = t.readFact(filepath.Join(base, "hash")); err != nil {212		return nil, err213	}214	if r.Reason, err = t.readFact(filepath.Join(base, "reason")); err != nil {215		return nil, err216	}217	at, err := t.readTime(filepath.Join(base, "rejectedAt"))218	if err != nil {219		return nil, err220	}221	if at != nil {222		r.RejectedAt = *at223	}224	return r, nil225}
machine/factsserio.go 90.5%
1package machine23// The serio subtrees of the facts tree: serio/, the standing of each4// attachment, and boot/serio/, the entries the boot declared.5//6// Both are lists keyed by position, counted from zero, the way7// boot/network/interfaces is. A status list mixes matched lines with8// entries that match none, so no single field is a natural key, and9// the position keeps the order init reported them in.10//11// Each element is one record file, by rule 5, because both lists12// change while the machine runs: the serio watch rewrites serio/ when13// an adapter is plugged in or unplugged, and the module loader14// rewrites boot/serio/ when a live load adds an entry. A rename15// replaces one element's record in one step, so a reader never reads16// a record that names one tty and another tty's state.1718import (19	"path/filepath"20	"strconv"21	"strings"22)2324// WriteSerio publishes the standing of every attachment. The serio25// watch owns this subtree (init/serio.go).26func (t FactsTree) WriteSerio(statuses []SerioStatus) error {27	records := make([][][2]string, 0, len(statuses))28	for _, s := range statuses {29		records = append(records, [][2]string{30			{"protocol", s.Protocol},31			{"vendor", s.USB.Vendor},32			{"product", s.USB.Product},33			{"serial", s.USB.Serial},34			{"tty", s.TTY},35			{"port", s.Port},36			{"state", string(s.State)},37			{"message", s.Message},38			// A node path comes from a DEVNAME, which holds no space,39			// so one space separates the list on its one line.40			{"nodes", strings.Join(s.Nodes, " ")},41		})42	}43	return t.report(t.writePositionalRecords("serio", records))44}4546// WriteBootSerio publishes the spec.serio list the boot declared. The47// boot writes it, and the module loader rewrites it when a live load48// adds an entry.49func (t FactsTree) WriteBootSerio(entries []SerioAttachment) error {50	records := make([][][2]string, 0, len(entries))51	for _, a := range entries {52		records = append(records, [][2]string{53			{"protocol", a.Protocol},54			{"vendor", a.USB.Vendor},55			{"product", a.USB.Product},56			{"serial", a.USB.Serial},57		})58	}59	return t.report(t.writePositionalRecords(filepath.Join("boot", "serio"), records))60}6162// writePositionalRecords writes one record file for each element,63// named by its position, and removes the files of the positions the64// list no longer reaches.65func (t FactsTree) writePositionalRecords(dir string, records [][][2]string) error {66	want := map[string]bool{}67	for i, record := range records {68		key := strconv.Itoa(i)69		want[key] = true70		if err := t.writeRecordFact(filepath.Join(dir, key), record); err != nil {71			return err72		}73	}74	return syncEntryFiles(filepath.Join(t.Dir, dir), want)75}7677// readSerio reads serio/ back into the status list, in position78// order.79func (t FactsTree) readSerio() ([]SerioStatus, error) {80	records, err := t.readPositionalRecords("serio")81	if err != nil {82		return nil, err83	}84	var statuses []SerioStatus85	for _, r := range records {86		s := SerioStatus{87			Protocol: r["protocol"],88			USB:      SerioUSB{Vendor: r["vendor"], Product: r["product"], Serial: r["serial"]},89			TTY:      r["tty"],90			Port:     r["port"],91			State:    SerioState(r["state"]),92			Message:  r["message"],93		}94		if r["nodes"] != "" {95			s.Nodes = strings.Split(r["nodes"], " ")96		}97		statuses = append(statuses, s)98	}99	return statuses, nil100}101102// readBootSerio reads boot/serio/ back into the declared list.103func (t FactsTree) readBootSerio() ([]SerioAttachment, error) {104	records, err := t.readPositionalRecords(filepath.Join("boot", "serio"))105	if err != nil {106		return nil, err107	}108	var entries []SerioAttachment109	for _, r := range records {110		entries = append(entries, SerioAttachment{111			Protocol: r["protocol"],112			USB:      SerioUSB{Vendor: r["vendor"], Product: r["product"], Serial: r["serial"]},113		})114	}115	return entries, nil116}117118// readPositionalRecords reads the records at positions 0, 1, 2, and on,119// until a position has no file. Reading by position keeps the list's120// order, which sorted file names would lose at the tenth element.121func (t FactsTree) readPositionalRecords(dir string) ([]map[string]string, error) {122	var records []map[string]string123	for i := 0; ; i++ {124		record, err := t.readRecordFact(filepath.Join(dir, strconv.Itoa(i)))125		if err != nil {126			return nil, err127		}128		if record == nil {129			return records, nil130		}131		records = append(records, record)132	}133}
machine/factstree.go 89.4%
1package machine23// The facts tree carries what init observed to the operator, as a4// directory of small files under /run/liken/facts.5//6// Init gathers facts that no program inside the cluster can observe7// directly: the DHCP exchange, the moment of boot, the hardware as the8// kernel first showed it. The operator reads these facts, adds what it9// observes itself, and publishes the result to the Machine's status.10//11// The tree is one rendering of the same contract that MachineStatus and12// the CRD describe. Each path segment is a JSON field name, so the13// three renderings read alike. A person can explore the tree with ls,14// cat, and grep, and read one fact without parsing the whole status.15//16// One value lives in one file. This is the reason the tree exists as a17// tree at all. A single status file forces one writer, because a write18// serializes the whole struct, and two writers would race to rewrite19// the same bytes. A tree gives each fact its own file, so each init20// component writes its own subtree with no shared lock. The ownership21// map has no file with two owners:22//23//	time/                          the clock loop24//	hardware/blockDevices/         the hardware watch25//	hardware/unclaimed/            the hardware watch26//	boot/manifest, boot/modules    the module loader27//	boot/serio/                    the module loader28//	modules/                       the module loader29//	serio/                         the serio watch30//	boot/clusterManifest           the restart path31//	boot/credentials               the restart path32//	boot/restarts                  the restart path33//	features/, registries/         the restart path34//	runtime/                       the restart path35//	everything else                the boot step that discovers it36//37// A boot step writes its subtree once, at the point where it discovers38// the fact, and prints its console line from the same place. The tree39// fills in as the boot runs. A partial tree is safe, because the40// operator runs under k3s, and k3s starts only after the boot steps41// finish.42//43// The grammar has five rules:44//45//  1. A scalar fact is one file: the value plus one trailing newline.46//     Strings are raw, integers decimal, booleans the word true,47//     timestamps RFC3339Nano in UTC, and enums their API word.48//  2. An absent fact has no file, and a zero value has no file. A49//     reader treats a missing file as the zero value, which is50//     omitempty semantics. A writer removes the file for a fact that51//     disappears.52//  3. A scalar list is one file, one item per line, in list order. An53//     empty list has no file.54//  4. An identity-keyed collection is a directory for each element,55//     named by the element's natural key.56//  5. A group of facts that change together mid-run is one record file57//     of key=value lines.58//59// Rule 5 applies to four boot records, boot/manifest,60// boot/clusterManifest, boot/credentials, and boot/imports, and to61// each element of the two serio lists, serio/ and boot/serio/. Each of62// these is rewritten while the machine runs. A rename replaces one63// inode in one step, so a rename of one file is the only write that a64// concurrent reader can never see half done. The paired fields of each65// record must land together, so each pair is one file that one rename66// replaces. The rest of the tree writes once at boot, before the67// operator exists, so those facts cannot tear and stay one file each.68//69// Rule 2 has one exception, boot/network, and the exception is what70// the fact is for. A machine that declared no interface and a machine71// whose boot recorded no network are different facts, and the72// operator must act differently on each, so the boot/network73// directory exists whenever the boot recorded anything, even with no74// file under it. Nowhere else does an empty directory mean anything.75//76// The tree lives on tmpfs, so a write needs no fsync. Every write goes77// through writeAtomic: a temp file in the same directory, then a78// rename. A reader that polls or wakes on its own schedule sees either79// the old file or the new file, never a torn write.8081import (82	"crypto/sha256"83	"encoding/hex"84	"errors"85	"fmt"86	"io/fs"87	"os"88	"path/filepath"89	"regexp"90	"strconv"91	"strings"92	"time"93)9495// FactsDir is the root of the facts tree. It sits on /run, which is96// tmpfs, because the facts describe one boot and a reboot must start97// them empty.98const FactsDir = "/run/liken/facts"99100// FactsTree is one facts tree, rooted at Dir. Production code roots it101// at FactsDir. Tests root it at a temporary directory, so a test never102// touches the machine's real /run.103//104// A fact write must never stop the machine: a lost fact is a reporting105// gap the operator fills with a zero value, not a reason to halt PID 1.106// The program that owns the tree states that policy once, in Report,107// rather than at every call site. Each writer routes its result through108// report, which hands a failure to Report and returns the error109// unchanged. The reporter is additive: the error still reaches the110// caller, so a test with a nil Report asserts it directly.111type FactsTree struct {112	Dir string113	// Report receives each failed write. A nil Report leaves the error114	// with the caller.115	Report func(err error)116}117118// report hands a failed write to the tree's reporter and returns the119// error unchanged, so the caller still sees it.120func (t FactsTree) report(err error) error {121	if err != nil && t.Report != nil {122		t.Report(err)123	}124	return err125}126127// keyPattern is the set of characters that a collection key may use128// directly, for interfaces, disks, modules, and features. These keys129// come from the kernel and from the spec, and they already read as130// plain names, so the writer asserts the pattern instead of rewriting131// the key. A key outside the pattern is a programming error, not a132// value to sanitize.133var keyPattern = regexp.MustCompile(`^[A-Za-z0-9._:-]+$`)134135// assertKey checks that a collection key is safe to use as a directory136// name. It rejects a key that could escape its parent or collide with137// the tree's own structure.138func assertKey(kind, key string) error {139	if !keyPattern.MatchString(key) {140		return fmt.Errorf("facts %s key %q is not a plain name", kind, key)141	}142	return nil143}144145// safeKey turns a modalias into a directory name. A modalias is the146// kernel's own fingerprint string, and it carries bytes that a path147// segment cannot hold, such as a slash or a space. safeKey replaces148// every byte outside [A-Za-z0-9._:+-] with an underscore. If it149// changed any byte, or the result runs past 200 bytes, it truncates to150// 64 bytes and appends a dash and the first twelve hex digits of the151// modalias's sha256. The hash makes the key unique, so two aliases152// that differ only in an unsafe byte never collide. The exact modalias153// lives in the modalias file inside the directory, so the key itself154// needs no way back to the original.155func safeKey(modalias string) string {156	var b strings.Builder157	changed := false158	for i := range len(modalias) {159		c := modalias[i]160		switch {161		case c >= 'A' && c <= 'Z', c >= 'a' && c <= 'z', c >= '0' && c <= '9':162			b.WriteByte(c)163		case c == '.' || c == '_' || c == ':' || c == '+' || c == '-':164			b.WriteByte(c)165		default:166			b.WriteByte('_')167			changed = true168		}169	}170	key := b.String()171	if !changed && len(key) <= 200 {172		return key173	}174	sum := sha256.Sum256([]byte(modalias))175	suffix := hex.EncodeToString(sum[:])[:12]176	if len(key) > 64 {177		key = key[:64]178	}179	return key + "-" + suffix180}181182// writeFact writes one scalar file. An empty value means the fact is183// absent, so writeFact removes the file instead. A reader treats the184// missing file as the zero value.185func (t FactsTree) writeFact(rel, value string) error {186	path := filepath.Join(t.Dir, rel)187	if value == "" {188		return removeFile(path)189	}190	if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {191		return err192	}193	return writeAtomic(path, []byte(value+"\n"))194}195196// writeListFact writes one scalar list, one item per line, in order.197// An empty list means the fact is absent, so writeListFact removes the198// file.199func (t FactsTree) writeListFact(rel string, items []string) error {200	path := filepath.Join(t.Dir, rel)201	if len(items) == 0 {202		return removeFile(path)203	}204	if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {205		return err206	}207	return writeAtomic(path, []byte(strings.Join(items, "\n")+"\n"))208}209210// writeRecordFact writes one record file of key=value lines, in the211// given order. It skips a pair whose value is empty, so an absent212// field of the record leaves no line. If no pair has a value, the whole213// record is absent, so writeRecordFact removes the file. A value cannot214// hold a newline or a control byte, because those would break the line215// grammar, so writeRecordFact replaces each with a space.216func (t FactsTree) writeRecordFact(rel string, pairs [][2]string) error {217	var b strings.Builder218	for _, p := range pairs {219		if p[1] == "" {220			continue221		}222		b.WriteString(p[0])223		b.WriteByte('=')224		b.WriteString(sanitizeValue(p[1]))225		b.WriteByte('\n')226	}227	path := filepath.Join(t.Dir, rel)228	if b.Len() == 0 {229		return removeFile(path)230	}231	if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {232		return err233	}234	return writeAtomic(path, []byte(b.String()))235}236237// sanitizeValue replaces every newline and control byte with a space,238// so a record value stays on one line. A record file's grammar is one239// key=value pair per line, and a raw newline in a value would forge a240// second line.241func sanitizeValue(s string) string {242	return strings.Map(func(r rune) rune {243		if r < 0x20 || r == 0x7f {244			return ' '245		}246		return r247	}, s)248}249250// readFact reads one scalar file. A missing file is the zero value, so251// readFact returns an empty string with no error. It strips the single252// trailing newline that writeFact added.253func (t FactsTree) readFact(rel string) (string, error) {254	raw, err := os.ReadFile(filepath.Join(t.Dir, rel))255	if errors.Is(err, fs.ErrNotExist) {256		return "", nil257	}258	if err != nil {259		return "", err260	}261	return strings.TrimSuffix(string(raw), "\n"), nil262}263264// readListFact reads one scalar list into its lines, in order. A265// missing file is an empty list.266func (t FactsTree) readListFact(rel string) ([]string, error) {267	raw, err := os.ReadFile(filepath.Join(t.Dir, rel))268	if errors.Is(err, fs.ErrNotExist) {269		return nil, nil270	}271	if err != nil {272		return nil, err273	}274	text := strings.TrimSuffix(string(raw), "\n")275	if text == "" {276		return nil, nil277	}278	return strings.Split(text, "\n"), nil279}280281// readRecordFact reads one record file into a map of its key=value282// lines. A missing file is an empty record. It reads each key on its283// own, so it accepts a record that carries an unknown key, and a284// caller reads only the keys it needs.285func (t FactsTree) readRecordFact(rel string) (map[string]string, error) {286	raw, err := os.ReadFile(filepath.Join(t.Dir, rel))287	if errors.Is(err, fs.ErrNotExist) {288		return nil, nil289	}290	if err != nil {291		return nil, err292	}293	record := map[string]string{}294	for line := range strings.SplitSeq(strings.TrimSuffix(string(raw), "\n"), "\n") {295		if line == "" {296			continue297		}298		key, value, ok := strings.Cut(line, "=")299		if !ok {300			continue301		}302		record[key] = value303	}304	return record, nil305}306307// readInt reads a scalar file as a decimal integer. A missing file is308// zero. A file that does not hold an integer is an error that names its309// path, so an operator can find the bad file.310func (t FactsTree) readInt(rel string) (int, error) {311	value, err := t.readFact(rel)312	if err != nil || value == "" {313		return 0, err314	}315	n, err := strconv.Atoi(value)316	if err != nil {317		return 0, fmt.Errorf("%s: %w", filepath.Join(t.Dir, rel), err)318	}319	return n, nil320}321322// readUint reads a scalar file as a decimal unsigned integer, the shape323// of every byte count in the tree. A missing file is zero, and a file324// that does not hold an unsigned integer is an error that names its325// path.326func (t FactsTree) readUint(rel string) (uint64, error) {327	value, err := t.readFact(rel)328	if err != nil || value == "" {329		return 0, err330	}331	n, err := strconv.ParseUint(value, 10, 64)332	if err != nil {333		return 0, fmt.Errorf("%s: %w", filepath.Join(t.Dir, rel), err)334	}335	return n, nil336}337338// readTime reads a scalar file as an RFC3339Nano timestamp in UTC. A339// missing file is a nil time, the absent value for every timestamp in340// the tree. A file that does not hold a timestamp is an error that341// names its path.342func (t FactsTree) readTime(rel string) (*time.Time, error) {343	value, err := t.readFact(rel)344	if err != nil || value == "" {345		return nil, err346	}347	parsed, err := time.Parse(time.RFC3339Nano, value)348	if err != nil {349		return nil, fmt.Errorf("%s: %w", filepath.Join(t.Dir, rel), err)350	}351	parsed = parsed.UTC()352	return &parsed, nil353}354355// formatTime renders a timestamp for the tree: RFC3339Nano in UTC. A356// nil or zero time renders empty, so its file is absent.357func formatTime(t *time.Time) string {358	if t == nil || t.IsZero() {359		return ""360	}361	return t.UTC().Format(time.RFC3339Nano)362}363364// formatInt renders an integer for the tree. Zero renders empty, so a365// zero value leaves no file.366func formatInt(n int) string {367	if n == 0 {368		return ""369	}370	return strconv.Itoa(n)371}372373// formatUint renders an unsigned integer for the tree. Zero renders374// empty, so a zero value leaves no file.375func formatUint(n uint64) string {376	if n == 0 {377		return ""378	}379	return strconv.FormatUint(n, 10)380}381382// formatBool renders a boolean for the tree. Only a true value has a383// file, and its content is the word true. A false value leaves no file,384// which the reader treats as false.385func formatBool(b bool) string {386	if b {387		return "true"388	}389	return ""390}391392// removeFile deletes a file that a fact no longer needs. A file that is393// already gone is not an error, because the goal is only its absence.394func removeFile(path string) error {395	err := os.Remove(path)396	if errors.Is(err, fs.ErrNotExist) {397		return nil398	}399	return err400}401402// syncEntryDirs removes the element directories under parent whose keys403// are not in want. It is how every collection writer honors rule 2: an404// element that leaves the set loses its directory. It removes405// directories only, so a temp file that another write left behind stays406// untouched. A parent that does not exist yet has nothing to remove.407func syncEntryDirs(parent string, want map[string]bool) error {408	entries, err := os.ReadDir(parent)409	if errors.Is(err, fs.ErrNotExist) {410		return nil411	}412	if err != nil {413		return err414	}415	for _, entry := range entries {416		if !entry.IsDir() || want[entry.Name()] {417			continue418		}419		if err := os.RemoveAll(filepath.Join(parent, entry.Name())); err != nil {420			return err421		}422	}423	return nil424}425426// syncEntryFiles is syncEntryDirs for a collection whose elements are427// scalars rather than records: it removes the file of every key the428// writer no longer names. A collection of scalars is a directory of429// files, not a directory of directories, so the two cannot share one430// walk.431func syncEntryFiles(parent string, want map[string]bool) error {432	entries, err := os.ReadDir(parent)433	if errors.Is(err, fs.ErrNotExist) {434		return nil435	}436	if err != nil {437		return err438	}439	for _, entry := range entries {440		if entry.IsDir() || want[entry.Name()] {441			continue442		}443		if err := removeFile(filepath.Join(parent, entry.Name())); err != nil {444			return err445		}446	}447	return nil448}
machine/factswrite.go 83.9%
1package machine23// The per-subtree writers. Each writer covers one subtree of the facts4// tree, and one init component calls it at the point where it discovers5// those facts. There is no writer for the whole tree, on purpose. A6// whole-tree writer would be a second owner of every file, and the7// tree exists so that each fact has one owner and needs no lock.8//9// A collection writer reconciles. It writes the element directories10// that the new set holds, then removes the element directories that the11// new set dropped. This one step gives a collection the same omitempty12// semantics that a scalar file gets from writeFact: what is no longer13// true loses its file.1415import (16	"os"17	"path/filepath"18	"strconv"19	"time"2021	"github.com/liken-sh/liken/liken/api"22)2324// RejectionKind names one of the four standing quarantine records under25// boot/. Each names the document whose rejection it holds. The value is26// the record's directory name in the tree, which matches the field name27// in BootStatus.28type RejectionKind string2930const (31	RejectMachine     RejectionKind = "rejection"32	RejectCluster     RejectionKind = "clusterRejection"33	RejectSystem      RejectionKind = "systemRejection"34	RejectCredentials RejectionKind = "credentialsRejection"35)3637// WriteRole publishes the machine's role in its cluster: leader or38// follower. Boot derives it from the Cluster manifest's leaders list.39func (t FactsTree) WriteRole(role api.Role) error {40	return t.report(t.writeFact("role", string(role)))41}4243// WriteBootTime publishes the moment the machine booted, derived from44// the kernel's uptime counter. It belongs to the boot record because it45// shares the record's lifetime: a reboot moves it, an in-place k3s46// restart does not.47func (t FactsTree) WriteBootTime(bootedAt *time.Time) error {48	return t.report(t.writeFact("boot/time", formatTime(bootedAt)))49}5051// WriteLastCrash publishes the newest kernel crash the machine still52// holds records for. A nil crash means the machine holds no crash on53// record, so the whole subtree is removed.54func (t FactsTree) WriteLastCrash(c *CrashStatus) error {55	if c == nil {56		return t.report(os.RemoveAll(filepath.Join(t.Dir, "lastCrash")))57	}58	return t.report(firstError(59		t.writeFact("lastCrash/time", formatTime(c.Time)),60		t.writeFact("lastCrash/reason", string(c.Reason)),61		t.writeFact("lastCrash/message", c.Message),62		t.writeFact("lastCrash/records", c.Records),63	))64}6566// WriteLastFailStop publishes the last boot this machine refused to67// run. A nil record means no boot has ever been refused, so the whole68// subtree is removed.69func (t FactsTree) WriteLastFailStop(f *FailStop) error {70	if f == nil {71		return t.report(os.RemoveAll(filepath.Join(t.Dir, "lastFailStop")))72	}73	return t.report(firstError(74		t.writeFact("lastFailStop/time", formatTime(&f.Time)),75		t.writeFact("lastFailStop/reason", f.Reason),76	))77}7879// WriteVersion publishes the machine's version inventory: liken's own80// version, and every outside component the image carries.81func (t FactsTree) WriteVersion(v VersionStatus) error {82	return t.report(firstError(83		t.writeFact("version/liken", v.Liken),84		t.writeFact("version/kernel", v.Kernel),85		t.writeFact("version/xtables", v.Xtables),86		t.writeFact("version/k3s", v.K3s),87		t.writeFact("version/trust", v.Trust),88		t.writeFact("version/e2fsprogs", v.E2fsprogs),89		t.writeFact("version/openIscsi", v.OpenISCSI),90		t.writeFact("version/nfsUtils", v.NFSUtils),91		t.writeFact("version/wpaSupplicant", v.WPASupplicant),92		t.writeFact("version/systemdBoot", v.SystemdBoot),93		t.writeFact("version/grub", v.Grub),94		t.writeFact("version/hwdata", v.Hwdata),95		t.writeFact("version/tzdata", v.Tzdata),96		t.writeFact("version/linuxFirmware", v.LinuxFirmware),97		t.writeFact("version/wirelessRegdb", v.WirelessRegdb),98		t.writeFact("version/microcode", v.Microcode),99		t.writeFact("version/microcodeRevision", v.MicrocodeRevision),100	))101}102103// WriteNetwork publishes the outcome of the boot's networking. The104// top-level fields summarize the primary interface, and the interfaces105// collection carries the detail for each interface.106func (t FactsTree) WriteNetwork(n NetworkStatus) error {107	if err := firstError(108		t.writeFact("network/interface", n.Interface),109		t.writeFact("network/mac", n.MAC),110		t.writeFact("network/gateway", n.Gateway),111		t.writeFact("network/leaseExpires", formatTime(n.LeaseExpires)),112		t.writeListFact("network/addresses", n.Addresses),113		t.writeListFact("network/nameservers", n.Nameservers),114	); err != nil {115		return t.report(err)116	}117	want := map[string]bool{}118	for _, iface := range n.Interfaces {119		if err := assertKey("interface", iface.Name); err != nil {120			return t.report(err)121		}122		want[iface.Name] = true123		base := filepath.Join("network", "interfaces", iface.Name)124		if err := os.MkdirAll(filepath.Join(t.Dir, base), 0o755); err != nil {125			return t.report(err)126		}127		if err := firstError(128			t.writeFact(filepath.Join(base, "mac"), iface.MAC),129			t.writeFact(filepath.Join(base, "address"), iface.Address),130			t.writeFact(filepath.Join(base, "method"), string(iface.Method)),131			t.writeFact(filepath.Join(base, "gateway"), iface.Gateway),132			t.writeFact(filepath.Join(base, "leaseExpires"), formatTime(iface.LeaseExpires)),133			t.writeListFact(filepath.Join(base, "nameservers"), iface.Nameservers),134		); err != nil {135			return t.report(err)136		}137		if err := t.writeInterfaceWireless(base, iface.Wireless); err != nil {138			return t.report(err)139		}140	}141	return t.report(syncEntryDirs(filepath.Join(t.Dir, "network", "interfaces"), want))142}143144// writeInterfaceWireless publishes the radio's standing for one145// interface. The ssid file is the record's presence, for the same146// reason it is in the boot record: a wireless entry always names a147// network, so a wired interface leaves no wireless directory at all.148func (t FactsTree) writeInterfaceWireless(base string, w *WirelessStatus) error {149	dir := filepath.Join(base, "wireless")150	if w == nil {151		return os.RemoveAll(filepath.Join(t.Dir, dir))152	}153	return firstError(154		t.writeFact(filepath.Join(dir, "ssid"), w.SSID),155		t.writeFact(filepath.Join(dir, "state"), string(w.State)),156		t.writeFact(filepath.Join(dir, "message"), w.Message),157	)158}159160// WriteTime publishes the state of the machine's clock. The clock loop161// owns this subtree and rewrites it as it disciplines the clock.162func (t FactsTree) WriteTime(ts TimeStatus) error {163	return t.report(firstError(164		t.writeFact("time/state", string(ts.State)),165		t.writeFact("time/source", ts.Source),166		t.writeFact("time/stratum", formatInt(ts.Stratum)),167		t.writeFact("time/offset", ts.Offset),168	))169}170171// WriteHardwareBasics publishes the machine's processor count and total172// memory. The disks and the unclaimed devices have their own writers,173// because the hardware watch keeps them current after boot.174func (t FactsTree) WriteHardwareBasics(cpus int, memoryBytes uint64) error {175	return t.report(firstError(176		t.writeFact("hardware/cpus", formatInt(cpus)),177		t.writeFact("hardware/memoryBytes", formatUint(memoryBytes)),178	))179}180181// WriteBlockDevices publishes the machine's storage inventory: every182// real disk the kernel found. The directory name is the kernel's name183// for the disk this boot.184func (t FactsTree) WriteBlockDevices(devices []BlockDevice) error {185	want := map[string]bool{}186	for _, d := range devices {187		if err := assertKey("blockDevice", d.Name); err != nil {188			return t.report(err)189		}190		want[d.Name] = true191		base := filepath.Join("hardware", "blockDevices", d.Name)192		if err := os.MkdirAll(filepath.Join(t.Dir, base), 0o755); err != nil {193			return t.report(err)194		}195		if err := firstError(196			t.writeFact(filepath.Join(base, "sizeBytes"), formatUint(d.SizeBytes)),197			t.writeFact(filepath.Join(base, "model"), d.Model),198			t.writeFact(filepath.Join(base, "serial"), d.Serial),199			t.writeListFact(filepath.Join(base, "stableNames"), d.StableNames),200		); err != nil {201			return t.report(err)202		}203	}204	return t.report(syncEntryDirs(filepath.Join(t.Dir, "hardware", "blockDevices"), want))205}206207// WriteUnclaimed publishes every device the kernel enumerated but that208// nothing drives. The directory name is safeKey(modalias), and the209// exact modalias lives in the modalias file inside. The reconcile210// removes a device's directory once a module claims it.211func (t FactsTree) WriteUnclaimed(devices []UnclaimedDevice) error {212	want := map[string]bool{}213	for _, d := range devices {214		key := safeKey(d.Modalias)215		want[key] = true216		base := filepath.Join("hardware", "unclaimed", key)217		if err := os.MkdirAll(filepath.Join(t.Dir, base), 0o755); err != nil {218			return t.report(err)219		}220		if err := firstError(221			t.writeFact(filepath.Join(base, "modalias"), d.Modalias),222			t.writeFact(filepath.Join(base, "bus"), d.Bus),223			t.writeFact(filepath.Join(base, "name"), d.Name),224			t.writeFact(filepath.Join(base, "class"), d.Class),225			t.writeFact(filepath.Join(base, "message"), d.Message),226			t.writeListFact(filepath.Join(base, "candidates"), d.Candidates),227		); err != nil {228			return t.report(err)229		}230	}231	return t.report(syncEntryDirs(filepath.Join(t.Dir, "hardware", "unclaimed"), want))232}233234// WriteFirmware publishes the machine's standing firmware state: the235// mode it boots in, and the boot menu from its non-volatile store.236func (t FactsTree) WriteFirmware(f FirmwareStatus) error {237	return t.report(firstError(238		t.writeFact("firmware/mode", string(f.Mode)),239		t.writeFact("firmware/bootCurrent", f.BootCurrent),240		t.writeFact("firmware/bootNext", f.BootNext),241		t.writeListFact("firmware/bootOrder", f.BootOrder),242	))243}244245// WriteStorage publishes where each storage role is actually backed246// this boot. The nine roles are a fixed set, so this writer needs no247// reconcile: a role that moves from a partition back to memory loses248// its device, partition, and capacity files through writeFact.249func (t FactsTree) WriteStorage(s StorageStatus) error {250	for _, name := range StorageRoleNames {251		role := s.Role(name)252		base := filepath.Join("storage", string(name))253		if err := firstError(254			t.writeFact(filepath.Join(base, "backing"), string(role.Backing)),255			t.writeFact(filepath.Join(base, "device"), role.Device),256			t.writeFact(filepath.Join(base, "partition"), role.Partition),257			t.writeFact(filepath.Join(base, "capacityBytes"), formatUint(role.CapacityBytes)),258			t.writeFact(filepath.Join(base, "lastStopUnclean"), formatBool(role.LastStopUnclean)),259		); err != nil {260			return t.report(err)261		}262	}263	return nil264}265266// WriteModules publishes the outcome of every module named in267// spec.modules. The module loader owns this subtree.268func (t FactsTree) WriteModules(modules []ModuleStatus) error {269	want := map[string]bool{}270	for _, m := range modules {271		if err := assertKey("module", m.Name); err != nil {272			return t.report(err)273		}274		want[m.Name] = true275		base := filepath.Join("modules", m.Name)276		if err := os.MkdirAll(filepath.Join(t.Dir, base), 0o755); err != nil {277			return t.report(err)278		}279		if err := firstError(280			t.writeFact(filepath.Join(base, "state"), string(m.State)),281			t.writeFact(filepath.Join(base, "message"), m.Message),282			t.writeFact(filepath.Join(base, "alreadyResident"), formatBool(m.AlreadyResident)),283			t.writeKeyedScalars(filepath.Join(base, "parameters"), m.Parameters),284		); err != nil {285			return t.report(err)286		}287	}288	return t.report(syncEntryDirs(filepath.Join(t.Dir, "modules"), want))289}290291// WriteRlimits publishes the resource limits init holds, read back292// from the kernel. Each resource is one file holding the limit in the293// spec's syntax, because a limit is a scalar and rule 4's directory294// per element would wrap each one around a single value.295func (t FactsTree) WriteRlimits(rlimits map[string]string) error {296	return t.report(t.writeKeyedScalars("rlimits", rlimits))297}298299// WriteBootRlimits publishes the resource limits the winning manifest300// declared. This is the drift reference, so it carries the request301// rather than the outcome, the way boot/modules and boot/storage do.302func (t FactsTree) WriteBootRlimits(rlimits map[string]string) error {303	return t.report(t.writeKeyedScalars("boot/rlimits", rlimits))304}305306// writeKeyedScalars publishes a map of scalars as a directory of307// files, one for each key. A key that disappears takes its file with308// it, and an empty map leaves no directory at all, which is rule 2's309// omitempty semantics.310func (t FactsTree) writeKeyedScalars(dir string, values map[string]string) error {311	want := map[string]bool{}312	for key := range values {313		if err := assertKey(dir, key); err != nil {314			return err315		}316		want[key] = true317	}318	if len(want) > 0 {319		if err := os.MkdirAll(filepath.Join(t.Dir, dir), 0o755); err != nil {320			return err321		}322	}323	for key, value := range values {324		if err := t.writeFact(filepath.Join(dir, key), value); err != nil {325			return err326		}327	}328	return syncEntryFiles(filepath.Join(t.Dir, dir), want)329}330331// WriteFeatures publishes this machine's standing on every feature the332// cluster document enables. The restart path owns this subtree, because333// a k3s restart can change a feature's outcome without a reboot.334func (t FactsTree) WriteFeatures(features []FeatureStatus) error {335	want := map[string]bool{}336	for _, f := range features {337		if err := assertKey("feature", f.Name); err != nil {338			return t.report(err)339		}340		want[f.Name] = true341		base := filepath.Join("features", f.Name)342		if err := os.MkdirAll(filepath.Join(t.Dir, base), 0o755); err != nil {343			return t.report(err)344		}345		if err := firstError(346			t.writeFact(filepath.Join(base, "state"), string(f.State)),347			t.writeFact(filepath.Join(base, "message"), f.Message),348		); err != nil {349			return t.report(err)350		}351	}352	return t.report(syncEntryDirs(filepath.Join(t.Dir, "features"), want))353}354355// WriteRegistries publishes what this boot rendered into k3s's356// registries.yaml: the mirrored hosts, the credentialed hosts, and357// whether the embedded registry is on. It never carries credential358// material.359func (t FactsTree) WriteRegistries(r RegistriesStatus) error {360	return t.report(firstError(361		t.writeListFact("registries/mirrors", r.Mirrors),362		t.writeListFact("registries/credentialedHosts", r.CredentialedHosts),363		t.writeFact("registries/embedded", formatBool(r.Embedded)),364	))365}366367// WriteRuntime publishes the runtime discipline init imposed this368// boot: the Go environment and log level it gave the k3s process, the369// configuration it wrote for the kubelet inside it, and the level it370// gave containerd beside it. These are the resolved values, not the371// spec's strings. The restart path owns this subtree, because a k3s372// restart can change all of them without a reboot. A setting the373// cluster left unset writes no file, so its absence reads as the374// default that the reader supplies for itself.375func (t FactsTree) WriteRuntime(r RuntimeStatus) error {376	gc := r.Kubelet.ImageGC377	return t.report(firstError(378		t.writeFact("runtime/k3s/goMemoryLimit", r.K3s.GoMemoryLimit),379		t.writeFact("runtime/k3s/goGC", formatInt(r.K3s.GoGC)),380		t.writeFact("runtime/k3s/debug", formatBool(r.K3s.Debug)),381		t.writeFact("runtime/containerd/logLevel", r.Containerd.LogLevel),382		t.writeFact("runtime/kubelet/imageGC/highThresholdPercent", formatInt(gc.HighThresholdPercent)),383		t.writeFact("runtime/kubelet/imageGC/lowThresholdPercent", formatInt(gc.LowThresholdPercent)),384		t.writeFact("runtime/kubelet/imageGC/maximumAge", gc.MaximumAge),385		t.writeFact("runtime/kubelet/imageGC/minimumAge", gc.MinimumAge),386	))387}388389// WriteBootSlot publishes the system slot this boot came from, A or B.390// It is empty when the boot did not come from a slot, as in a391// direct-kernel boot.392func (t FactsTree) WriteBootSlot(slot string) error {393	return t.report(t.writeFact("boot/slot", slot))394}395396// WriteBootCommandLine publishes the kernel command line this boot ran397// with, whole. init prints the same line to the console, and this is398// what carries it to a reader who has no console.399func (t FactsTree) WriteBootCommandLine(cmdline string) error {400	return t.report(t.writeFact("boot/commandLine", cmdline))401}402403// WriteBootRestarts publishes the count of in-place k3s restarts this404// boot has performed. The restart path owns it, and it returns to zero405// on the next reboot.406func (t FactsTree) WriteBootRestarts(n int) error {407	return t.report(t.writeFact("boot/restarts", formatInt(n)))408}409410// WriteBootModules publishes the module list the winning manifest411// declared, recorded as actuated regardless of each load's outcome.412// The list keeps the order the machine loaded the modules in, and the413// operator compares that order against the declared one.414func (t FactsTree) WriteBootModules(modules []string) error {415	return t.report(t.writeListFact("boot/modules", modules))416}417418// WriteBootModuleParameters publishes the parameter map this machine419// counts as actuated. A boot records the whole declared map, the420// same way boot/modules records its list. A live load records only421// the keys it delivered, so an undelivered parameter drifts and the422// reboot that can deliver it gets scheduled (init/liveload.go). The423// per-module readback under modules/ reports what the kernel holds.424func (t FactsTree) WriteBootModuleParameters(parameters map[string]string) error {425	return t.report(t.writeKeyedScalars("boot/moduleParameters", parameters))426}427428// WriteBootManifest publishes the Machine manifest this boot ran under.429// The module loader writes it last, because it is the commit point the430// operator's convergence judges.431func (t FactsTree) WriteBootManifest(src ManifestSource, hash string) error {432	return t.report(t.writeRecordFact("boot/manifest", [][2]string{433		{"source", string(src)},434		{"hash", hash},435	}))436}437438// WriteBootClusterManifest publishes the Cluster manifest this boot ran439// under. The restart path writes it last, because promotion keys on it.440func (t FactsTree) WriteBootClusterManifest(src ManifestSource, hash string) error {441	return t.report(t.writeRecordFact("boot/clusterManifest", [][2]string{442		{"source", string(src)},443		{"hash", hash},444	}))445}446447// WriteBootCredentials publishes the registry-credentials document this448// boot, or the latest k3s restart, rendered into registries.yaml.449func (t FactsTree) WriteBootCredentials(src ManifestSource, hash string) error {450	return t.report(t.writeRecordFact("boot/credentials", [][2]string{451		{"source", string(src)},452		{"hash", hash},453	}))454}455456// WriteBootImports publishes the imported-images record this boot ran457// under. The discarded field records that this boot found a trial still458// standing from a boot that died unproven, and threw the container459// store away.460func (t FactsTree) WriteBootImports(src ManifestSource, hash string, discarded bool) error {461	return t.report(t.writeRecordFact("boot/imports", [][2]string{462		{"source", string(src)},463		{"hash", hash},464		{"discarded", formatBool(discarded)},465	}))466}467468// WriteBootStorage publishes the storage the boot actuated, one469// directory for each declared role. Only declared roles appear, so the470// reconcile removes a role's directory once the spec drops it.471func (t FactsTree) WriteBootStorage(spec StorageSpec) error {472	want := map[string]bool{}473	for _, role := range spec.Roles() {474		name := string(role.Name)475		want[name] = true476		base := filepath.Join("boot", "storage", name)477		if err := os.MkdirAll(filepath.Join(t.Dir, base), 0o755); err != nil {478			return t.report(err)479		}480		if err := firstError(481			t.writeFact(filepath.Join(base, "device"), role.Device),482			t.writeFact(filepath.Join(base, "size"), role.Size),483		); err != nil {484			return t.report(err)485		}486	}487	return t.report(syncEntryDirs(filepath.Join(t.Dir, "boot", "storage"), want))488}489490// WriteBootNetwork publishes the network the boot actuated: one491// directory for each declared interface. A nil spec means this boot492// recorded nothing about its network, so the whole subtree goes away.493//494// The boot/network directory is the record itself, and it exists even495// when the spec declares no interface at all. This is the one place496// in the tree where an empty directory carries meaning, and it has497// to: a machine that declares no interface and a machine that498// recorded nothing are different facts, and only the second one is499// beyond judging (machine/drift.go explains what each one means for500// drift).501//502// Each interface's key is its position in the declared list, counted503// from zero, and not the name of its port. The order of this list is504// part of what it asks for, since it is the order in which each505// interface's nameservers reach resolv.conf, and position is also506// what the comparison uses.507//508// Host entries carry no boot record. A boot record answers what one509// boot actuated, and spec.network.hostEntries reconciles live: the510// machine operator may rewrite /etc/hosts on any later pass, so a511// file it may have already changed is not one boot's fact.512// status.hostEntries is the current view, the same as status.sysctls513// is for sysctls, which also keeps no boot record.514func (t FactsTree) WriteBootNetwork(spec *NetworkSpec) error {515	base := filepath.Join("boot", "network")516	if spec == nil {517		return t.report(os.RemoveAll(filepath.Join(t.Dir, base)))518	}519	interfaces := filepath.Join(base, "interfaces")520	if err := os.MkdirAll(filepath.Join(t.Dir, interfaces), 0o755); err != nil {521		return t.report(err)522	}523	want := map[string]bool{}524	for i, ifc := range spec.Interfaces {525		key := strconv.Itoa(i)526		want[key] = true527		dir := filepath.Join(interfaces, key)528		if err := os.MkdirAll(filepath.Join(t.Dir, dir), 0o755); err != nil {529			return t.report(err)530		}531		if err := firstError(532			t.writeFact(filepath.Join(dir, "name"), ifc.Name),533			t.writeFact(filepath.Join(dir, "address"), ifc.Address),534			t.writeFact(filepath.Join(dir, "gateway"), ifc.Gateway),535			t.writeListFact(filepath.Join(dir, "nameservers"), ifc.Nameservers),536		); err != nil {537			return t.report(err)538		}539		if err := t.writeBootWireless(dir, ifc.Wireless); err != nil {540			return t.report(err)541		}542	}543	return t.report(syncEntryDirs(filepath.Join(t.Dir, interfaces), want))544}545546// writeBootWireless publishes the wireless half of one interface's547// boot record. The ssid file is the record's presence, because a548// wireless entry always names a network. An interface with no radio549// needs no file here, and a respec that drops the radio removes the550// directory whole, so no stale SSID outlives the spec that named it.551func (t FactsTree) writeBootWireless(dir string, w *WirelessSpec) error {552	base := filepath.Join(dir, "wireless")553	if w == nil {554		return os.RemoveAll(filepath.Join(t.Dir, base))555	}556	return firstError(557		t.writeFact(filepath.Join(base, "ssid"), w.SSID),558		t.writeFact(filepath.Join(base, "security"), string(w.Security)),559	)560}561562// WriteRejection publishes one of the four standing quarantine records.563// A nil rejection means the document has no standing rejection, so the564// whole record directory is removed.565func (t FactsTree) WriteRejection(kind RejectionKind, r *Rejection) error {566	base := filepath.Join("boot", string(kind))567	if r == nil {568		return t.report(os.RemoveAll(filepath.Join(t.Dir, base)))569	}570	if err := os.MkdirAll(filepath.Join(t.Dir, base), 0o755); err != nil {571		return t.report(err)572	}573	rejectedAt := r.RejectedAt574	return t.report(firstError(575		t.writeFact(filepath.Join(base, "hash"), r.Hash),576		t.writeFact(filepath.Join(base, "reason"), r.Reason),577		t.writeFact(filepath.Join(base, "rejectedAt"), formatTime(&rejectedAt)),578	))579}580581// firstError returns the first non-nil error of a sequence of writes,582// or nil when all succeed. Go evaluates the arguments before the call,583// so every write in the sequence runs even when an early one fails.584// The writes are independent files, so the later writes stay correct,585// and the caller receives the first failure.586func firstError(errs ...error) error {587	for _, err := range errs {588		if err != nil {589			return err590		}591	}592	return nil593}
machine/failstop.go 94.1%
1package machine23// The record a machine leaves behind when it refuses to boot.4//5// Init stops a boot for two failures, identity and storage6// (init/main.go's failBoot states both). The stop prints its reason7// and powers the machine off. The power-off is what makes that reason8// hard to keep: it ends the boot long before k3s starts, so no log9// relay carries the console anywhere, and a powered-off machine has no10// console to read. A person who was not sitting at the serial port at11// that moment has nothing to go on.12//13// So the reason goes into one small file on machineState before the14// power-off. The next boot reads the file, prints it, and publishes it15// as status.lastFailStop. The question "why did this machine refuse"16// then takes one kubectl get, rather than a source read of init.17//18// The record is never cleared. It answers one question, the last time19// this machine refused to boot and why, and that answer stays true20// until the next refusal replaces it. A machine that boots cleanly for21// months still reports the refusal it had before that, the same way22// status.lastCrash keeps reporting an old panic.2324import (25	"errors"26	"io/fs"27	"os"28	"path/filepath"29	"time"3031	"sigs.k8s.io/yaml"32)3334// failStopRecord is the record's file name under the machineState35// root. It sits at the root rather than in a directory of its own,36// because there is only ever one of these: each refusal replaces the37// last.38const failStopRecord = "failstop.yaml"3940// FailStop is one refused boot: what init would not run with, and41// when. One type serves as both the file's schema and the status42// field, so the console message, the record on disk, and the cluster's43// status all carry the same words.44type FailStop struct {45	// Reason is failBoot's own message, the text the console printed46	// before the machine powered off.47	Reason string `json:"reason"`4849	// Time is the machine's clock at the refusal. A boot stops well50	// before its first clock synchronization, so this is the hardware51	// clock's reading, reported as recorded.52	Time time.Time `json:"time"`53}5455// WriteFailStop records a refusal under the machineState root. The56// write is durable and not merely atomic, in the same way as every57// other machineState write (staging.go explains the four steps). The58// whole purpose of this record is to survive the power-off that59// follows it, and a rename that never reached the disk would leave60// nothing behind.61func WriteFailStop(root string, f FailStop) error {62	raw, err := yaml.Marshal(f)63	if err != nil {64		return err65	}66	if err := os.MkdirAll(root, 0o755); err != nil {67		return err68	}69	return WriteDurable(filepath.Join(root, failStopRecord), raw)70}7172// ReadFailStop reads the standing record. A root with no record is the73// ordinary case, a machine that has never refused a boot, so it74// returns nothing and no error.75func ReadFailStop(root string) (*FailStop, error) {76	raw, err := os.ReadFile(filepath.Join(root, failStopRecord))77	if errors.Is(err, fs.ErrNotExist) {78		return nil, nil79	}80	if err != nil {81		return nil, err82	}83	f := &FailStop{}84	if err := yaml.UnmarshalStrict(raw, f); err != nil {85		return nil, err86	}87	// A record with no reason is not a refusal. An empty file88	// unmarshals into an empty struct and reports no error, so without89	// this test a truncated record would read back as a refusal that90	// happened in the year one and gave no reason. The write is durable91	// precisely so this does not happen, and a machine can still lose92	// the file's contents to hardware that acknowledges a flush it did93	// not make. Nothing beats no news here, because the field's whole94	// job is to say something a person can act on.95	if f.Reason == "" {96		return nil, nil97	}98	return f, nil99}
machine/hosts.go 100.0%
1package machine23// HostsFile renders /etc/hosts. It lives here, rather than in init or4// the operator, because both programs write this file and must never5// disagree about its shape (drift.go states the same rule for the6// rest of this package). Init calls it at every boot, and the machine7// operator calls it on every reconcile pass to check the file for8// drift and to heal it. One function, called from two places, is what9// keeps the file to one shape no matter which of the two wrote it10// last.1112import (13	"fmt"14	"strings"15)1617// HostsFile renders the three fixed lines that resolve this18// machine's own identity, localhost, its IPv6 form, and the19// machine's own name at 127.0.1.1, followed by one line for each host20// entry, in the order the caller passes them. Entry order has no21// effect on how a name resolves, since each address answers on its22// own line, but keeping the caller's order is what lets a person23// compare the file against the manifest that produced it.24//25// The fixed lines always come first. A resolver takes its first26// match, so an entry can add a name but can never override localhost27// or the machine's own name.28func HostsFile(hostname string, entries []HostEntry) string {29	var b strings.Builder30	fmt.Fprintf(&b, "127.0.0.1 localhost\n::1 localhost\n127.0.1.1 %s\n", hostname)31	for _, entry := range entries {32		fmt.Fprintf(&b, "%s %s\n", entry.Address, strings.Join(entry.Names, " "))33	}34	return b.String()35}
machine/imports.go 95.2%
1package machine23// The imported-images record stores which OS image tarballs a4// machine's container store has proven it can serve.5//6// At startup, k3s's embedded containerd imports every OCI tarball in7// its agent/images directory. Content and metadata land in its8// database, and containerd extracts each layer into a snapshot9// directory on clusterState. Those writes do not happen in a10// crash-safe order. A machine that dies at the wrong moment can be11// left with a database that says a layer is unpacked, while the12// extracted files are torn. containerd trusts its own record, so it13// never unpacks the same digest again. Every container from that14// image then fails with "exec format error", on every later boot,15// forever. If the torn image is the machine operator's image, the16// machine has lost the program that would have reported the17// problem.18//19// The fix does not sit inside containerd. containerd's unpack cannot20// become transactional from outside containerd. Instead, the imports21// go through the same staged and proven lifecycle as every liken22// document (staging.go), with this record as the document. The23// tarballs' digests are staged before k3s first sees them, and24// proven once the operator observes the images actually serving25// containers. If a boot finds a staged record still present, the26// store took writes that were never proven. The boot discards the27// record completely, instead of trusting it. init's imports.go28// carries that half of the work.2930import (31	"crypto/sha256"32	"encoding/hex"33	"errors"34	"io"35	"io/fs"36	"os"37	"path/filepath"38	"strings"3940	"github.com/liken-sh/liken/liken/api"41)4243// K3sAgentDir names the tree this record covers: k3s's agent state,44// where containerd keeps its store and the tarballs arrive (in its45// images/ subdirectory). Both halves of the protocol share this one46// spelling, because their safety depends on agreement. init discards47// this tree when a trial died unproven. The operator's promotion48// barrier flushes exactly this filesystem.49const K3sAgentDir = "/var/lib/rancher/k3s/agent"5051// ImportedImages lists the OS image tarballs that one boot handed to52// containerd, each keyed by the sha256 of its bytes. The digests53// cover the tarballs, not the OCI images inside them. The record54// exists to notice that the boot brought different bytes to import,55// and the tarball is the unit that arrives.56type ImportedImages struct {57	APIVersion string `json:"apiVersion"`58	Kind       string `json:"kind"`5960	// Images maps each tarball's basename to the sha256 of its61	// contents. Images is empty on machines whose image carries no62	// tarballs.63	Images map[string]string `json:"images,omitempty"`64}6566// ImportedImagesStore returns the record's lifecycle store under the67// given machineState root, beside the other documents' stores.68func ImportedImagesStore(root string) ManifestStore {69	return ManifestStore{dir: filepath.Join(root, "imports")}70}7172// RenderImportedImages produces the record's canonical bytes and73// their hash. A boot compares this hash against the proven record's74// hash, to determine whether the boot brings different tarballs to75// import.76func RenderImportedImages(images map[string]string) ([]byte, string, error) {77	return renderDocument(ImportedImages{78		APIVersion: api.APIVersion,79		Kind:       "ImportedImages",80		Images:     images,81	})82}8384// HashImageTarballs digests every .tar file in a directory, keyed by85// basename. A missing directory means no tarballs are present, as86// happens with an image that has no k3s. HashImageTarballs treats a87// missing directory as a normal answer, not an error.88func HashImageTarballs(dir string) (map[string]string, error) {89	entries, err := os.ReadDir(dir)90	if errors.Is(err, fs.ErrNotExist) {91		return nil, nil92	}93	if err != nil {94		return nil, err95	}96	digests := map[string]string{}97	for _, entry := range entries {98		if entry.IsDir() || !strings.HasSuffix(entry.Name(), ".tar") {99			continue100		}101		f, err := os.Open(filepath.Join(dir, entry.Name()))102		if err != nil {103			return nil, err104		}105		sum := sha256.New()106		_, err = io.Copy(sum, f)107		f.Close()108		if err != nil {109			return nil, err110		}111		digests[entry.Name()] = hex.EncodeToString(sum.Sum(nil))112	}113	return digests, nil114}
machine/inotify.go 88.3%
1package machine23// The facts tree carries state between init and the operator, and both4// sides must see a change the moment it lands, not on the next5// tick of a timer. inotify is how the kernel tells a program that a6// directory changed. These helpers wrap it for one use only: a wake.7//8// An event is a trigger, never a message. The kernel reports what9// changed and where, but this package throws that content away. A woken10// reader re-reads the current state of the tree, so it always acts on11// what the filesystem holds now, not on what an event claimed a moment12// ago. This is the reason a burst of events collapses into one wake:13// the reader would read the same final state whether it woke once or a14// hundred times. The wake channel has room for one pending wake, so15// every extra event during a burst is dropped, and the reader does one16// read for the whole burst.17//18// inotify watches an inode, not a path. This matters twice. First, the19// operator reads the tree through a read-only hostPath, which is a bind20// mount of the same inodes that init writes on the host. An event fires21// on the inode, so it crosses the bind mount, and the operator sees a22// change that init made on the other side. Second, a watch on a file23// goes stale the moment writeAtomic renames a new file into place,24// because the rename installs a new inode and the old watched inode is25// gone. So these helpers watch directories, never files. A directory's26// inode is stable across the renames of the files inside it, and a27// rename into the directory is itself an event on the directory.28//29// The watch must exist before the first read, or a change between the30// read and the watch is lost forever. So every constructor here31// establishes the watch, and only then returns to the caller for its32// first scan. A caller that reads before the watch exists has opened a33// window; a caller that watches first has closed it.34//35// The event mask asks for the writes that liken makes and the writes a36// person makes. writeAtomic only ever renames, so IN_MOVED_TO is the37// event that every normal fact write produces. IN_CLOSE_WRITE is here38// for the other writer: a person who, during an incident, echoes a39// value straight into a fact file at the console. That write ends with40// a close, not a rename, and the operator should still wake and read41// it. A watch that ignored the console would make the tree feel dead to42// the one person most likely to be poking at it.43//44// The kernel's event queue can overflow if a reader falls behind. The45// kernel then delivers one IN_Q_OVERFLOW and drops the events it could46// not hold. This is safe here, because an overflow is just another47// wake: the reader re-reads the whole state and misses nothing. The48// content it would have parsed from the lost events is content it never49// trusted.50//51// A watch ends when its directory leaves the path that the caller52// named: a remove, a rename of another directory over it, a rename of53// it away, or an unmount. The kernel then drops the watch, or keeps it54// on an inode that the path no longer names, and no later change at55// the path reaches the reader. So the reader stops and closes its56// channel, the same way it does when a read fails, and the caller57// watches the path again.58//59// Cancellation cannot rely on closing the inotify descriptor, because a60// close does not wake a thread already blocked in a read on that61// descriptor. So the reader never blocks in the read itself. The62// descriptor is non-blocking, and the reader blocks in poll over two63// descriptors: the inotify descriptor and the read end of a cancel64// pipe. When the context is done, a second goroutine closes the write65// end of the pipe, the poll wakes on the hangup, and the reader66// returns. No goroutine outlives its context.6768import (69	"bytes"70	"context"71	"encoding/binary"72	"errors"73	"io/fs"74	"path/filepath"75	"sync"7677	"golang.org/x/sys/unix"78)7980// errWatchClosed is what Sync answers once the reader has stopped and81// closed the inotify descriptor.82var errWatchClosed = errors.New("the inotify watch is closed")8384// dirMask is the event set for a single watched directory. IN_MOVED_TO85// catches every write that goes through writeAtomic, because that write86// ends in a rename into the directory. IN_CLOSE_WRITE catches a87// person's direct write at the console. IN_ONLYDIR makes the kernel88// refuse the watch if the path is not a directory, which turns a caller89// that passes a file by mistake into an error instead of a stale watch.90const dirMask = unix.IN_MOVED_TO | unix.IN_CLOSE_WRITE | unix.IN_ONLYDIR9192// selfMask asks for the events on the watched directory itself, so the93// reader learns that the directory left its path. Every single94// directory watch adds it to the caller's mask. The kernel sends95// IN_IGNORED and IN_UNMOUNT without a request.96const selfMask = unix.IN_DELETE_SELF | unix.IN_MOVE_SELF9798// goneMask is the set of events on the watched directory that end the99// watch. IN_IGNORED arrives every time the kernel drops a watch, after100// IN_DELETE_SELF for a remove or a rename over the directory, and101// after IN_UNMOUNT for an unmount. The reader stops on the first of102// them, so it does not send a wake for a directory that is already103// gone. IN_MOVE_SELF leaves the watch in place, but on a directory104// that now has another path, so it ends the watch too.105const goneMask = unix.IN_IGNORED | unix.IN_UNMOUNT | unix.IN_DELETE_SELF | unix.IN_MOVE_SELF106107// treeMask is the event set for a directory inside a recursive tree108// watch. It adds the events that announce a subdirectory. IN_CREATE and109// IN_MOVED_TO fire when a new subdirectory appears, so the next Sync110// finds it and adds a watch. IN_DELETE_SELF fires when a watched111// directory is removed; the kernel then also delivers IN_IGNORED and112// drops the watch on its own, so the next Sync only has to forget the113// bookkeeping. IN_MOVE_SELF is for the root, which ends the watch when114// it leaves its path (goneMask). On a subdirectory it is one more115// wake.116const treeMask = unix.IN_MOVED_TO | unix.IN_CLOSE_WRITE |117	unix.IN_CREATE | unix.IN_DELETE_SELF | unix.IN_MOVE_SELF | unix.IN_ONLYDIR118119// watch holds the descriptors and the wake channel for one inotify120// instance. The inotify descriptor is non-blocking, so the reader can121// wait for it in poll alongside the cancel pipe instead of blocking in122// a read that a close cannot interrupt.123type watch struct {124	fd      int125	cancelR int126	cancelW int127	wake    chan struct{}128129	// name, when it is set, limits the wakes to events on that one130	// name in the watched directory. An overflow still wakes, because131	// the events it dropped may have named it.132	name string133134	// dir is the watch descriptor of the directory the caller named:135	// the one directory of a single watch, or the root of a tree. The136	// reader ends the watch on a goneMask event for it. It is set137	// before the reader starts and never changes, so the reader reads138	// it without a lock. Zero matches no record, because the kernel139	// numbers watch descriptors from one.140	dir int32141142	// fdMu makes the reader's close of fd and a TreeWatch's Sync143	// exclusive. The reader closes the wake channel before it closes144	// fd, so a caller can see an open channel, call Sync, and add a145	// watch to a descriptor number that the process has since given to146	// another file. fdClosed tells Sync that the reader has stopped.147	fdMu     sync.Mutex148	fdClosed bool149}150151// closeFd closes the inotify descriptor once, under fdMu.152func (w *watch) closeFd() {153	w.fdMu.Lock()154	defer w.fdMu.Unlock()155	if !w.fdClosed {156		w.fdClosed = true157		unix.Close(w.fd)158	}159}160161// newWatch creates a non-blocking inotify instance and the cancel pipe162// that stops its reader. It adds no watches; the caller adds them163// before it starts the reader, so the watch exists before the first164// scan.165func newWatch() (*watch, error) {166	fd, err := unix.InotifyInit1(unix.IN_NONBLOCK | unix.IN_CLOEXEC)167	if err != nil {168		return nil, err169	}170	var pipe [2]int171	if err := unix.Pipe2(pipe[:], unix.O_CLOEXEC); err != nil {172		unix.Close(fd)173		return nil, err174	}175	return &watch{176		fd:      fd,177		cancelR: pipe[0],178		cancelW: pipe[1],179		wake:    make(chan struct{}, 1),180	}, nil181}182183// closeFds releases every descriptor. A caller uses it only on the184// setup path, before the reader starts. Once the reader runs, the185// reader owns the inotify descriptor and the cancel pipe's read end,186// and the cancel goroutine owns the write end.187func (w *watch) closeFds() {188	w.closeFd()189	unix.Close(w.cancelR)190	unix.Close(w.cancelW)191}192193// start launches the reader and the goroutine that cancels it. The194// cancel goroutine waits for the context and then closes the pipe's195// write end, which wakes the reader's poll with a hangup. This split of196// ownership means no descriptor is closed twice: the cancel goroutine197// closes the write end, and the reader closes the rest as it returns.198//199// The cancel goroutine also ends when the reader stops on its own, so a200// watch that failed leaves no goroutine and no descriptor behind for201// the life of the context.202func (w *watch) start(ctx context.Context) {203	done := make(chan struct{})204	go func() {205		select {206		case <-ctx.Done():207		case <-done:208		}209		unix.Close(w.cancelW)210	}()211	go func() {212		w.run()213		close(done)214	}()215}216217// run is the reader loop. It blocks in poll over the inotify descriptor218// and the cancel pipe. A ready inotify descriptor means events to219// drain; a ready cancel pipe means the context is done and the loop220// returns. A poll error or a drain error also ends the loop, for good:221// the reader does not retry. It closes the wake channel on that way222// out, and only on that way out, so a caller can tell a watch that223// died from one whose context ended. A watch that died misses every224// change after it, so the caller watches again and reads the whole225// state again. Only this goroutine sends on the channel, so the close226// cannot race a send. It closes the descriptors it owns as it leaves.227func (w *watch) run() {228	defer w.closeFd()229	defer unix.Close(w.cancelR)230	fds := []unix.PollFd{231		{Fd: int32(w.fd), Events: unix.POLLIN},232		{Fd: int32(w.cancelR), Events: unix.POLLIN},233	}234	buf := make([]byte, 64*1024)235	for {236		_, err := unix.Poll(fds, -1)237		if errors.Is(err, unix.EINTR) {238			continue239		}240		if err != nil {241			close(w.wake)242			return243		}244		if fds[1].Revents != 0 {245			return246		}247		if !w.drain(buf) {248			close(w.wake)249			return250		}251	}252}253254// drain reads every queued event and wakes the channel for each one. It255// reads until the kernel reports EAGAIN, which means the queue is256// empty, because the descriptor is non-blocking. It returns whether the257// reader should keep running: true after a normal drain, false after an258// error or a goneMask event on the watched directory, which end the259// watch. The check for the directory comes before the check for the260// name, because IN_IGNORED and IN_UNMOUNT carry no name.261func (w *watch) drain(buf []byte) bool {262	for {263		n, err := unix.Read(w.fd, buf)264		switch {265		case errors.Is(err, unix.EINTR):266			continue267		case errors.Is(err, unix.EAGAIN):268			return true269		case err != nil:270			return false271		}272		gone := false273		parseInotifyEvents(buf[:n], func(wd int32, mask uint32, name string) {274			switch {275			case gone:276			case wd == w.dir && mask&goneMask != 0:277				gone = true278			case w.name == "" || name == w.name || mask&unix.IN_Q_OVERFLOW != 0:279				w.signal()280			}281		})282		if gone {283			return false284		}285	}286}287288// signal does one non-blocking send on the wake channel. The channel289// holds one wake, so the first event during a burst fills it and every290// event after that is dropped until the reader drains it. This is the291// coalescing: a burst becomes one wake, and the reader does one read.292func (w *watch) signal() {293	select {294	case w.wake <- struct{}{}:295	default:296	}297}298299// parseInotifyEvents walks the records in one inotify read and calls300// visit for each. It reads the fixed header with the host's byte order,301// because the kernel writes these records in native form. The header is302// SizeofInotifyEvent bytes: the watch descriptor, the mask, a cookie,303// and the length of the name that follows. The name is padded with null304// bytes to align the next record, so the reader trims at the first null.305// A record whose name runs past the buffer is a truncated read, so the306// walk stops. An IN_Q_OVERFLOW record carries a descriptor of -1 and no307// name, and it walks like any other, because a reader treats it as one308// more wake.309func parseInotifyEvents(buf []byte, visit func(wd int32, mask uint32, name string)) {310	for len(buf) >= unix.SizeofInotifyEvent {311		wd := int32(binary.NativeEndian.Uint32(buf[0:4]))312		mask := binary.NativeEndian.Uint32(buf[4:8])313		nameLen := int(binary.NativeEndian.Uint32(buf[12:16]))314		end := unix.SizeofInotifyEvent + nameLen315		if end > len(buf) {316			return317		}318		name := ""319		if nameLen > 0 {320			raw := buf[unix.SizeofInotifyEvent:end]321			if i := bytes.IndexByte(raw, 0); i >= 0 {322				raw = raw[:i]323			}324			name = string(raw)325		}326		visit(wd, mask, name)327		buf = buf[end:]328	}329}330331// WatchDirMask watches one directory for the events in mask and332// coalesces every event into a wake channel of capacity one. It is the333// seam under WatchDir: a caller whose writer produces a different set of334// events than the fact writers do passes its own mask. The tailer, for335// example, follows a file that a held-open descriptor keeps appending336// to, so it asks for IN_MODIFY rather than the rename and close that a337// fact write ends with. IN_ONLYDIR is always added to the mask, so the338// kernel refuses a watch on a path that is not a directory and a caller339// cannot end up with a stale watch on a file. selfMask is always added340// too, so the watch ends when the directory leaves its path. The watch exists before341// the function returns, so a caller that scans right after the call342// cannot miss a change that lands between the watch and the scan. The343// context ends the watch, and a watch that fails closes the channel344// (run). A directory that does not exist is an error.345func WatchDirMask(ctx context.Context, dir string, mask uint32) (<-chan struct{}, error) {346	w, err := newWatch()347	if err != nil {348		return nil, err349	}350	wd, err := unix.InotifyAddWatch(w.fd, dir, mask|selfMask|unix.IN_ONLYDIR)351	if err != nil {352		w.closeFds()353		return nil, err354	}355	w.dir = int32(wd)356	w.start(ctx)357	return w.wake, nil358}359360// nameMask is the event set for one name in a directory: the name361// appears, by a create or a rename onto it, it is written and closed,362// or it leaves, by a delete or a rename away.363const nameMask = unix.IN_CREATE | unix.IN_MOVED_TO | unix.IN_CLOSE_WRITE |364	unix.IN_DELETE | unix.IN_MOVED_FROM365366// WatchName watches one name in a directory and coalesces the events367// on that name into a wake channel of capacity one. The watch is on368// the directory, because a writer that replaces the file by a rename369// installs a new inode, and a watch on the old inode would go quiet at370// the first rename. The events on every other name in the directory371// wake nothing. The machine operator watches the host's /etc for the372// name hosts this way, and init's writes of resolv.conf beside it, and373// the operator's own temporary file, do not wake a pass. The watch374// exists before the function returns, the context ends it, and a watch375// that fails or whose directory leaves its path closes the channel376// (run). A directory that does not exist is an error.377func WatchName(ctx context.Context, dir, name string) (<-chan struct{}, error) {378	w, err := newWatch()379	if err != nil {380		return nil, err381	}382	w.name = name383	wd, err := unix.InotifyAddWatch(w.fd, dir, nameMask|selfMask|unix.IN_ONLYDIR)384	if err != nil {385		w.closeFds()386		return nil, err387	}388	w.dir = int32(wd)389	w.start(ctx)390	return w.wake, nil391}392393// WatchDir watches one directory for the fact writers' events and394// coalesces every event into a wake channel of capacity one. The watch395// exists before the function returns, so a caller that scans right after396// the call cannot miss a change that lands between the watch and the397// scan. The context ends the watch: when it is done, the reader returns398// and the channel goes quiet. A watch that fails closes the channel399// instead (run). A directory that does not exist is an error.400func WatchDir(ctx context.Context, dir string) (<-chan struct{}, error) {401	return WatchDirMask(ctx, dir, dirMask)402}403404// TreeWatch is the recursive form of WatchDir, for the facts tree.405// inotify does not recurse, so one watch covers one directory only. A406// TreeWatch holds a watch for every directory in the tree and adds new407// ones as the tree grows. A caller calls Sync before every read, which408// reconciles the watch set with the tree on disk and closes the window409// between a new subdirectory and the watch on it. Wake fires for a410// change anywhere in the tree, and closes when the watch fails or the411// root leaves its path (run). A subdirectory that leaves is only a412// wake, because the next Sync forgets it.413type TreeWatch struct {414	Wake <-chan struct{}415416	w    *watch417	root string418	wds  map[string]int419}420421// WatchFactsTree establishes a recursive watch over the tree at root.422// It watches every directory that exists now, then returns, so the423// caller's first read sees a tree that is already watched. A root that424// does not exist is an error, and machine-operator ends the process on425// it, so the kubelet starts it again and the new process watches again.426// The watch ends when the root it found427// leaves its path, and the caller watches again, so a new root at the428// path gets a new watch.429func WatchFactsTree(ctx context.Context, root string) (*TreeWatch, error) {430	w, err := newWatch()431	if err != nil {432		return nil, err433	}434	t := &TreeWatch{435		Wake: w.wake,436		w:    w,437		root: root,438		wds:  map[string]int{},439	}440	if err := t.Sync(); err != nil {441		w.closeFds()442		return nil, err443	}444	w.dir = int32(t.wds[root])445	w.start(ctx)446	return t, nil447}448449// Sync reconciles the watch set with the directories under root. It450// walks the tree and adds a watch on every directory it finds. A watch451// on a directory that already has one is idempotent, because the kernel452// returns the same descriptor for the same inode. It then drops the453// bookkeeping for a directory that is gone; the kernel already removed454// that watch when the directory vanished, so the drop only forgets the455// descriptor. A missing root is an error the caller may retry. A456// directory that vanishes mid-walk is not an error, because the walk is457// a snapshot of a tree that another process is writing.458func (t *TreeWatch) Sync() error {459	t.w.fdMu.Lock()460	defer t.w.fdMu.Unlock()461	if t.w.fdClosed {462		return errWatchClosed463	}464	seen := map[string]bool{}465	err := filepath.WalkDir(t.root, func(path string, d fs.DirEntry, err error) error {466		if err != nil {467			if path == t.root {468				return err469			}470			if errors.Is(err, fs.ErrNotExist) {471				return nil472			}473			return err474		}475		if !d.IsDir() {476			return nil477		}478		wd, err := unix.InotifyAddWatch(t.w.fd, path, treeMask)479		if err != nil {480			if errors.Is(err, fs.ErrNotExist) {481				return nil482			}483			return err484		}485		t.wds[path] = wd486		seen[path] = true487		return nil488	})489	if err != nil {490		return err491	}492	for path, wd := range t.wds {493		if seen[path] {494			continue495		}496		unix.InotifyRmWatch(t.w.fd, uint32(wd))497		delete(t.wds, path)498	}499	return nil500}
machine/known.go 95.7%
1package machine23// A manifest that this machine already proved can name fields that4// this release does not know. A person sets a field that a newer5// release added, the field applies, and the manifest that holds it6// becomes the proven manifest. A rollback then boots the older7// release, whose strict parse cannot tell that field from a typo.8// Without the proven manifest, init falls back to the seed from the9// install: the machine boots its install-time storage and network,10// or, when the seed names the newer field too, it powers off.11//12// So the readers of a manifest that a boot already proved parse it13// with ParseKnown. The strict parse runs first. When it fails only14// because of fields this release does not know, ParseKnown parses the15// manifest again without them and names each field it skipped, so the16// console says what the rollback leaves out. A staged manifest and a17// seed still parse strictly, because those are where a typo must fail18// before anything boots it.1920import (21	"encoding"22	"encoding/json"23	"errors"24	"fmt"25	"io/fs"26	"os"27	"reflect"28	"slices"29	"strings"3031	"sigs.k8s.io/yaml"32)3334// ParseKnown reads a Machine manifest that a boot already proved. It35// answers the manifest and the path of each field it skipped, sorted.36// A manifest that fails the strict parse for any other reason, such37// as a known field with the wrong type, fails here too.38func ParseKnown(raw []byte) (*Machine, []string, error) {39	m, strictErr := Parse(raw)40	if strictErr == nil {41		return m, nil, nil42	}43	var doc any44	if err := yaml.Unmarshal(raw, &doc); err != nil {45		return nil, nil, strictErr46	}47	unknown := unknownFields(reflect.TypeFor[Machine](), doc, "")48	if len(unknown) == 0 {49		return nil, nil, strictErr50	}51	m = &Machine{}52	if err := yaml.Unmarshal(raw, m); err != nil {53		return nil, nil, strictErr54	}55	if m.Kind != "Machine" {56		return nil, nil, fmt.Errorf("expected kind Machine, got %q", m.Kind)57	}58	slices.Sort(unknown)59	return m, unknown, nil60}6162// LoadKnown reads a manifest file that a boot already proved, with63// ParseKnown. A missing file is a machine with every field at its64// default, as it is for Load.65func LoadKnown(path string) (*Machine, []string, error) {66	raw, err := os.ReadFile(path)67	if errors.Is(err, fs.ErrNotExist) {68		return &Machine{}, nil, nil69	}70	if err != nil {71		return nil, nil, err72	}73	m, ignored, err := ParseKnown(raw)74	if err != nil {75		return nil, nil, fmt.Errorf("%s: %w", path, err)76	}77	return m, ignored, nil78}7980// unknownFields walks a decoded document beside the Go type it81// decodes into, and answers the path of each key that the type has82// no field for. It matches names the way encoding/json does, without83// regard to case, so it agrees with the strict parse about which84// fields are known.85func unknownFields(t reflect.Type, value any, path string) []string {86	for t.Kind() == reflect.Pointer {87		t = t.Elem()88	}89	if decodesItself(t) {90		return nil91	}92	var unknown []string93	switch v := value.(type) {94	case map[string]any:95		switch t.Kind() {96		case reflect.Struct:97			for key, child := range v {98				field, ok := jsonField(t, key)99				if !ok {100					unknown = append(unknown, join(path, key))101					continue102				}103				unknown = append(unknown, unknownFields(field.Type, child, join(path, key))...)104			}105		case reflect.Map:106			for key, child := range v {107				unknown = append(unknown, unknownFields(t.Elem(), child, join(path, key))...)108			}109		}110	case []any:111		if t.Kind() == reflect.Slice || t.Kind() == reflect.Array {112			for i, child := range v {113				unknown = append(unknown, unknownFields(t.Elem(), child, fmt.Sprintf("%s[%d]", path, i))...)114			}115		}116	}117	return unknown118}119120// decodesItself reports whether a type decodes its own JSON, so its121// keys are its own business.122func decodesItself(t reflect.Type) bool {123	pointer := reflect.PointerTo(t)124	return pointer.Implements(reflect.TypeFor[json.Unmarshaler]()) ||125		pointer.Implements(reflect.TypeFor[encoding.TextUnmarshaler]())126}127128// jsonField finds the field of a struct that a JSON key decodes into,129// including the fields that an embedded struct promotes.130func jsonField(t reflect.Type, key string) (reflect.StructField, bool) {131	for i := range t.NumField() {132		field := t.Field(i)133		if !field.IsExported() && !field.Anonymous {134			continue135		}136		tag, _, _ := strings.Cut(field.Tag.Get("json"), ",")137		if tag == "-" {138			continue139		}140		if field.Anonymous && tag == "" {141			embedded := field.Type142			for embedded.Kind() == reflect.Pointer {143				embedded = embedded.Elem()144			}145			if embedded.Kind() == reflect.Struct {146				if promoted, ok := jsonField(embedded, key); ok {147					return promoted, true148				}149				continue150			}151		}152		name := tag153		if name == "" {154			name = field.Name155		}156		if strings.EqualFold(name, key) {157			return field, true158		}159	}160	return reflect.StructField{}, false161}162163func join(path, key string) string {164	if path == "" {165		return key166	}167	return path + "." + key168}
machine/layer.go 85.0%
1package machine23// This file covers the deployment layer and its sidecar file. The4// deployment layer is the part of the OS that belongs to one5// cluster. The sidecar file lets a machine confirm the layer's6// integrity.7//8// A boot slot carries the release's public artifacts. The release9// document names these artifacts, and the catalog's digest chain10// carries their integrity. A boot slot also carries one file that11// belongs to no release: the deployment layer (deployment.cpio). The12// deployment layer is private to one cluster, and it never travels13// over the network. So the release document cannot name it. Instead,14// a machine carries its own deployment layer forward from slot to15// slot at every upgrade.16//17// The sidecar file makes that carry safe. It is a one-line file that18// names the layer's sha256 digest. The machine writes the sidecar19// file durably at install time, from bytes it has just verified. The20// machine checks the sidecar file again before it trusts any copy of21// the layer.22//23// The sidecar's format matches sha256sum's own format:24// "<digest>  <name>". So a person at a rescue prompt can check a25// slot with a tool that exists everywhere: `sha256sum -c26// deployment.cpio.sha256`.2728import (29	"crypto/sha256"30	"encoding/hex"31	"fmt"32	"io"33)3435// LayerName is the deployment layer's file name, on install media36// and on every boot slot. Boot entries name LayerName as their37// second initrd= parameter, after the generic archive. The kernel38// unpacks initrd parameters in order, and each later entry overrides39// the entries before it.40const LayerName = "deployment.cpio"4142// LayerSidecarName is the sidecar's file name. The sidecar file sits43// beside the layer, wherever the layer is.44const LayerSidecarName = LayerName + ".sha256"4546// DigestLayer streams a layer through sha256 and returns the hex47// digest. A caller must digest the bytes that actually landed on48// disk, by rereading them. A caller must never digest the bytes it49// meant to write instead.50func DigestLayer(r io.Reader) (string, error) {51	h := sha256.New()52	if _, err := io.Copy(h, r); err != nil {53		return "", fmt.Errorf("reading the layer: %w", err)54	}55	return hex.EncodeToString(h.Sum(nil)), nil56}5758// FormatLayerSidecar writes the sidecar's single line for a layer59// digest.60func FormatLayerSidecar(digest string) []byte {61	return []byte(digest + "  " + LayerName + "\n")62}6364// ParseLayerSidecar reads a sidecar file back, with strict checks.65// The file must hold exactly one well-formed line that names the66// layer. Anything else counts as damage: truncation, a torn write,67// or the wrong file. When ParseLayerSidecar finds damage, the caller68// must treat the layer as unverifiable. The caller must not guess at69// the layer's digest.70func ParseLayerSidecar(raw []byte) (string, error) {71	want := FormatLayerSidecar("")72	if len(raw) != 64+len(want) {73		return "", fmt.Errorf("the layer sidecar is %d bytes, want %d", len(raw), 64+len(want))74	}75	digest := string(raw[:64])76	if _, err := hex.DecodeString(digest); err != nil {77		return "", fmt.Errorf("the layer sidecar's digest is not hex: %w", err)78	}79	if string(raw[64:]) != string(want) {80		return "", fmt.Errorf("the layer sidecar names %q, want %q", string(raw[64:]), string(want))81	}82	return digest, nil83}8485// VerifyLayer streams a layer through sha256 and compares the result86// against the digest that a sidecar file named.87func VerifyLayer(digest string, r io.Reader) error {88	got, err := DigestLayer(r)89	if err != nil {90		return err91	}92	if got != digest {93		return fmt.Errorf("%s digest mismatch: got %s, want %s", LayerName, got, digest)94	}95	return nil96}
machine/machine.go 100.0%
1// Package machine is the Machine API. It defines liken's one2// configuration document as Go types.3//4// A Machine has the same shape as a Kubernetes resource. Kubelet's5// KubeletConfiguration, k0s's config, and Talos's machine config also6// use this shape. This design lets one schema-validated document7// reach the machine in two ways. At boot, init reads the document8// from a file baked into the image, because no API server exists yet9// at that point. After the cluster starts, the same document exists10// in the cluster as a custom resource. There, the liken operator11// publishes the machine's live facts into the document's status. The12// file that a person writes by hand and the object that the command13// `kubectl get machine -o yaml` returns are the same document.14//15// Two programs use this API, and this package is what they share.16// Init reads the spec and applies the network settings and the17// sysctls at boot. Init also produces the facts. The operator reads18// the facts and reconciles the spec for as long as the machine runs.19// This division is deliberate. Init never talks to Kubernetes. The20// operator never touches boot-time state. The facts tree under21// `/run/liken/facts` is the one-way channel between the two programs22// (see factstree.go, which defines the tree and its grammar).23//24// The api package defines the document's shape: the group and25// version it declares, its metadata, and the condition and phase26// vocabulary that its status uses. Every other liken document shares27// this same shape.28//29// A note on naming: the names `machine.MachineSpec` and30// `machine.MachineStatus` repeat the package name. Go's naming advice31// warns against this repetition, but this package repeats it on32// purpose. The types mirror the CRD kind, `Machine`, and Kubernetes'33// `XxxSpec`/`XxxStatus` convention. Matching what the command34// `kubectl explain machine.spec` shows is worth more than avoiding35// this repetition.36package machine3738import (39	"errors"40	"fmt"41	"io/fs"42	"net"43	"os"44	"slices"4546	"github.com/liken-sh/liken/liken/api"47	"sigs.k8s.io/yaml"48)4950const (51	// MachineManifestDir is the directory where the image carries52	// Machine manifests, one file for each machine (<name>.yaml). One53	// image boots a whole fleet of machines, so the image carries54	// every machine's manifest. Each boot selects its own manifest by55	// the liken.machine=<name> kernel parameter. On a machine with56	// exactly one manifest, that manifest is the only choice, and the57	// boot uses it automatically.58	MachineManifestDir = "/etc/liken/machines"5960	// BootManifestPath is the file where init publishes the manifest61	// that this boot actually ran under. This manifest is the staged62	// or proven copy from machineState, or, on a first boot, the63	// image's seed manifest. The operator reads this file through a64	// hostPath mount. The operator uses the file to identify the65	// Machine it manages, and to seed the in-cluster Machine on the first66	// boot. Like the facts tree, this file lives under /run because it67	// describes only the current boot. It stays one whole file, so the68	// operator reads the manifest's exact bytes, not a rendering.69	BootManifestPath = "/run/liken/machine.yaml"7071	// SysctlDir is the kernel's tuning interface: one file for each72	// parameter. The sysctl helper functions take the directory as a73	// parameter, so tests can point them at a small copy of the74	// directory. Real callers pass this constant.75	SysctlDir = "/proc/sys"76)7778// Version is the liken version that this binary was built as. The79// build process stamps this value using -ldflags -X. When the80// releases domain builds the binary, the value is a release name. For81// a development build, the value is the git-described commit82// (liken/version.mk explains this mechanism). This value83// reaches the cluster as status.version.liken. The operator compares84// this value against the Cluster's spec.version target to determine85// whether this machine needs an upgrade.86var Version = "dev"8788// The struct tags are json, not yaml, because parsing goes through89// sigs.k8s.io/yaml, the same converter that Kubernetes tooling uses.90// This converter turns YAML into JSON before it unmarshals the data.91// This step gives Kubernetes documents their camelCase convention. It92// also means these structs serialize the same way, whether the data93// comes from a file or from the API server.94type Machine struct {95	APIVersion string         `json:"apiVersion"`96	Kind       string         `json:"kind"`97	Metadata   api.ObjectMeta `json:"metadata"`98	Spec       MachineSpec    `json:"spec,omitzero"`99	Status     MachineStatus  `json:"status,omitzero"`100}101102// GetObjectMeta answers the metadata that a watch's store and the103// operators' memo read (api.Meta).104func (m *Machine) GetObjectMeta() api.Meta { return m.Metadata }105106// MachineSpec is the declared half of a Machine. It states what a107// person asks this machine to be. A git repository can also declare108// this, through the cluster's flux feature. Each109// field notes who acts on it and when, because the actuators differ.110// Some state can only be set while the machine is being built. Some111// state can be reconciled live.112type MachineSpec struct {113	// Network is applied by init at boot. The cluster cannot reconcile114	// most of this field live, because the cluster reaches this115	// machine over the addresses that an edit changes: re-addressing a116	// running machine would cut the connection that carries the next117	// instruction. So an edit converges the same way a storage edit118	// does. The operator stages it to the machineState filesystem, and119	// the next boot applies it. RebootPolicy says who starts that120	// boot. HostEntries is the one exception: it names no address of121	// this machine's own, so an edit to it cannot cut the connection,122	// and it reconciles live instead (see its own comment below).123	Network NetworkSpec `json:"network,omitzero"`124125	// Sysctls is kernel tuning: it maps a parameter name to its126	// desired value, for example "vm.overcommit_memory": "1". The127	// system applies sysctls twice, by design. Init sets them at boot,128	// so they hold their values before k3s starts. The operator then129	// reconciles them live, so a kubectl edit takes effect without a130	// reboot.131	Sysctls map[string]string `json:"sysctls,omitempty"`132133	// Rlimits is per-process resource tuning: it maps a resource to134	// its desired limit, for example "nofile": "1048576". Init applies135	// the limits every liken machine holds (machine.OSRlimits) and136	// then these, to itself, before it starts k3s. Every process on137	// the machine inherits the result.138	//139	// Unlike sysctls, these cannot reconcile live. The kernel fixes a140	// process's limits when it forks, so no edit can reach a k3s that141	// is already running. An edit stages to the machineState142	// filesystem and applies at the next boot, the way storage and143	// network do. RebootPolicy says who starts that boot.144	Rlimits map[string]string `json:"rlimits,omitempty"`145146	// Modules names extra kernel modules that this machine loads at147	// boot, beyond the fixed list that the OS itself needs (the148	// image's modules.conf). These extra modules are the drivers for149	// whatever hardware this machine's workloads use. Init loads them150	// only after it reads the boot's manifest, so these modules cannot151	// serve the boot path itself. A driver that the boot depends on152	// must belong in the fixed list instead.153	//154	// The image carries the kernel build's whole module tree, so any155	// name that kernel has is loadable here, whatever manifests156	// existed when the image was built. status.modules reports the157	// outcome for each name, including the name this kernel has no158	// module for.159	//160	// A name added here converges without a reboot: loading a driver161	// is live-capable, and the kernel binds a resident driver to162	// hardware that is already plugged in. Removing a name needs a163	// boot, because unloading is not part of that path.164	Modules []string `json:"modules,omitempty"`165166	// ModuleParameters sets how a declared module loads: a flat map167	// from "<module>.<parameter>" to the value, in the kernel's own168	// spelling, the same form the kernel command line uses and169	// /sys/module mirrors. Each key must name a module that Modules170	// declares, spelled the same way. A parameter applies once, at171	// the load, because a loaded module never reads its parameters172	// again; so a value added with its module converges live, and a173	// change on a module this boot already loaded stages for the174	// next boot like every other reboot-class edit.175	ModuleParameters map[string]string `json:"moduleParameters,omitempty"`176177	// Serio names the serial-line devices whose kernel driver binds178	// only after a program attaches the line to the serio layer, such179	// as a USB-CEC adapter (serio.go). Init holds each attachment for180	// the life of the boot. An entry loads no module for itself: the181	// line driver, serport, and the protocol's driver belong in182	// Modules, the same as every other driver.183	//184	// The field converges on the terms Modules does. An added entry185	// attaches without a reboot, through the same live load that186	// loads an added module. A removed entry stages for the next187	// boot, because the holder keeps the port for any pod that holds188	// the devices the port created.189	Serio []SerioAttachment `json:"serio,omitempty"`190191	// NodeLabels is this machine's scheduling identity: the labels192	// that its Kubernetes Node object carries. Workloads select193	// machines using these labels, for example to find which machine194	// has the GPU, or which machine runs on battery-backed power. The195	// system applies node labels twice, like sysctls. Init renders the196	// labels into the k3s boot drop-in, so the node already carries197	// them when it registers. The operator then reconciles the labels198	// live afterward. The operator also removes a label that this spec199	// once declared but no longer declares. The kubelet applies200	// registration labels but never removes stale ones, so without201	// this removal step, a retracted label would remain until someone202	// noticed it. The operator never touches labels applied outside203	// this spec, such as labels set by kubectl label or by other204	// controllers.205	NodeLabels map[string]string `json:"nodeLabels,omitempty"`206207	// NodeTaints is the repelling half of that same scheduling208	// identity. A label draws a workload in; a taint keeps out every209	// pod that does not tolerate it, which is how a machine dedicated210	// to one job stays dedicated to it. A taint's identity is its key211	// together with its effect, and one key can carry two effects at212	// once, so this is a list where NodeLabels is a map.213	//214	// Init renders these taints into the k3s boot drop-in, so a node215	// that registers for the first time is born repelling and never216	// accepts, in its first minutes, the pods that the taint exists to217	// keep out. The kubelet applies registration taints only when it218	// creates the Node object, on a first boot or after a reinstall. On219	// every later boot the Node object already exists and the setting220	// does nothing, so the operator's live reconciliation is the only221	// mechanism from then on. The operator also removes a taint that222	// this spec once declared and no longer declares, and it never223	// touches a taint applied outside this spec, such as one set by224	// kubectl taint or by another controller.225	NodeTaints []NodeTaint `json:"nodeTaints,omitempty"`226227	// Storage assigns storage roles to disks (see storage.go). Init228	// applies this field at boot, before k3s starts, because a229	// filesystem cannot be swapped while a cluster is running. Edits230	// are staged to the machineState filesystem and take effect at the231	// next boot. RebootPolicy says who starts that boot.232	Storage StorageSpec `json:"storage,omitzero"`233234	// RebootPolicy states what the operator may do when applying the235	// spec requires a reboot. A storage change requires one, a network236	// change requires one, and so does retracting a module. Manual,237	// the default, stages the change and reports it;238	// the next boot, whenever it happens, applies the change. Auto239	// lets the operator reboot the machine itself. Manual is the240	// default because a reboot on a single-node cluster is a total241	// outage, and a mistyped edit should never reboot the machine242	// automatically.243	RebootPolicy RebootPolicy `json:"rebootPolicy,omitempty"`244}245246// NodeTaint is one taint on this machine's Node object. A pod247// tolerates a taint by naming its key and its effect, so those two248// fields together are the taint's identity. The value is optional: a249// taint often needs none, because the key and the effect alone already250// say to stay off this machine.251type NodeTaint struct {252	Key    string      `json:"key"`253	Value  string      `json:"value,omitempty"`254	Effect TaintEffect `json:"effect"`255}256257// TaintEffect states what happens to a pod that does not tolerate the258// taint. NoSchedule keeps a new pod off the node and leaves the pods259// already running there alone. PreferNoSchedule asks the scheduler to260// avoid the node but permits it when no other node fits. NoExecute261// keeps new pods off and evicts the running ones as well.262type TaintEffect string263264const (265	TaintNoSchedule       TaintEffect = "NoSchedule"266	TaintPreferNoSchedule TaintEffect = "PreferNoSchedule"267	TaintNoExecute        TaintEffect = "NoExecute"268)269270// RebootPolicy states who starts the reboot that a staged change271// waits on. The system treats any unrecognized value as Manual, so an272// unrecognized value can never cause an automatic reboot.273type RebootPolicy string274275const (276	RebootAuto   RebootPolicy = "Auto"277	RebootManual RebootPolicy = "Manual"278)279280func (s MachineSpec) RebootPolicyOrDefault() RebootPolicy {281	if s.RebootPolicy == RebootAuto {282		return RebootAuto283	}284	return RebootManual285}286287// NetworkSpec is deliberately almost empty. The default is zero288// configuration: DHCP on the first physical interface, the hostname289// from the manifest, and DNS from the DHCP lease. Fields exist here290// only for machines that need to differ from this default.291type NetworkSpec struct {292	// Interfaces configures the machine's interfaces explicitly, each293	// by name. An empty value means the zero-configuration default294	// described above. A machine in a cluster typically declares two295	// interfaces. One is an uplink that still uses DHCP. The other is296	// the cluster-facing interface, which uses the static address297	// that other machines were configured to use when they contact298	// it.299	Interfaces []InterfaceSpec `json:"interfaces,omitempty"`300301	// HostEntries names addresses that resolve with no DNS lookup.302	// Init writes each one as a line in /etc/hosts, below the three303	// fixed lines that define localhost and this machine's own name,304	// so an entry can add a name but never override those two. Init305	// also writes /etc/nsswitch.conf on every boot, so that the hosts306	// file wins over DNS on every resolver the machine runs. This307	// gives a program that resolves a name from the host's own files,308	// such as the NFS mount helper, an answer that does not depend on309	// cluster DNS being up.310	//311	// Unlike the rest of network, an edit here applies live: init312	// writes the file at every boot, so the entries prove the cold313	// start on their own, and the machine operator then reconciles314	// the same file on every pass, so a later edit lands within one315	// reconcile pass, with no reboot.316	HostEntries []HostEntry `json:"hostEntries,omitempty"`317}318319// Validate checks the spec's internal consistency. It catches the320// errors that a person can fix in the manifest, before the code321// touches any link.322//323// The API server enforces every one of these rules on any spec324// applied through it: the interface list is keyed by name, the host325// entry list is keyed by address, an address matches a pattern, a326// name list holds at least one item, and a CEL rule refuses the name327// localhost. This check exists for the specs that reach a machine328// another way: init also reads manifests that a person wrote by hand329// and carried in on a stick, and no API server ever saw those.330//331// Whether the machine really has a port with a declared name is a332// question only the machine can answer, so init answers that one333// against the links that exist.334func (s NetworkSpec) Validate() error {335	claimed := map[string]int{}336	for i, ifc := range s.Interfaces {337		if ifc.Name == "" {338			return fmt.Errorf("interface %d declares no name; the name is how an entry says which port it means", i)339		}340		// Two entries that name the same port are a manifest bug,341		// because the second entry's addressing would land on a link342		// the first already configured.343		if first, seen := claimed[ifc.Name]; seen {344			return fmt.Errorf("interfaces %d and %d both declare %s; declare each port once", first, i, ifc.Name)345		}346		claimed[ifc.Name] = i347		if ifc.Wireless != nil {348			if err := ifc.Wireless.validate(ifc.Name); err != nil {349				return err350			}351		}352	}353354	claimedAddresses := map[string]int{}355	for i, entry := range s.HostEntries {356		// A hosts file line takes one literal address, not a hostname357		// or a CIDR block. init writes this value straight into358		// /etc/hosts with no lookup step, so a value that does not359		// parse as an address would write a line that resolves360		// nothing.361		if net.ParseIP(entry.Address) == nil {362			return fmt.Errorf("host entry %d declares %q as its address, which is not a literal address", i, entry.Address)363		}364		// One address is one line of the file, and the facts record365		// keys entries by address. Two entries that declare the same366		// address would collide on both, so a manifest declares each367		// address once, with all of its names.368		if first, seen := claimedAddresses[entry.Address]; seen {369			return fmt.Errorf("host entries %d and %d both declare %s; declare each address once", first, i, entry.Address)370		}371		claimedAddresses[entry.Address] = i372		if len(entry.Names) == 0 {373			return fmt.Errorf("host entry %d (%s) declares no names; an entry with no name resolves nothing", i, entry.Address)374		}375		// The fixed lines that init writes first already define376		// localhost. An entry that names it again could only shadow377		// or contradict a fixed line, never add to it.378		if slices.Contains(entry.Names, "localhost") {379			return fmt.Errorf("host entry %d (%s) names localhost; the fixed lines already define localhost, ahead of any entry", i, entry.Address)380		}381	}382	return nil383}384385// InterfaceSpec configures one interface. Beyond Name, the zero value386// means DHCP. Static addressing is the deviation from the default, so387// a person must spell it out explicitly.388type InterfaceSpec struct {389	// Name is the interface to configure (for example, "eth1"), using390	// the name that the kernel gives it. Because no udev process391	// renames interfaces, kernel names follow the hardware enumeration392	// order, which stays stable for fixed hardware.393	Name string `json:"name"`394395	// Address is a static address in CIDR form (for example,396	// "10.10.0.1/24"). The prefix length tells the kernel the subnet,397	// so the prefix length is not optional. An empty value means DHCP398	// on this interface.399	Address string `json:"address,omitempty"`400401	// Gateway makes this interface the default route. This field is402	// optional, even for static addresses. A cluster segment with403	// nothing to route to declares no gateway, and the uplink's DHCP404	// lease supplies the real default route.405	Gateway string `json:"gateway,omitempty"`406407	// Nameservers lists nameservers to use in addition to any that408	// DHCP leases supply.409	Nameservers []string `json:"nameservers,omitempty"`410411	// Wireless names the network this interface joins before any of412	// the addressing above applies. An absent value means the413	// interface is wired.414	Wireless *WirelessSpec `json:"wireless,omitempty"`415}416417// HostEntry is one static line for /etc/hosts: one address and the418// names that resolve to it. Address is a literal IP address, not a419// CIDR block or a hostname, because a hosts file line answers with420// one address, and Names is the list of names that get that answer.421type HostEntry struct {422	Address string   `json:"address"`423	Names   []string `json:"names"`424}425426// Parse reads a Machine manifest from its bytes. Parsing is strict,427// because a misspelled field name in a manifest should produce an428// error that someone sees. Without strict parsing, a misspelled field429// would become a setting that silently never applies.430func Parse(raw []byte) (*Machine, error) {431	m := &Machine{}432	if err := yaml.UnmarshalStrict(raw, m); err != nil {433		return nil, err434	}435	if m.Kind != "Machine" {436		return nil, fmt.Errorf("expected kind Machine, got %q", m.Kind)437	}438	return m, nil439}440441// Load reads a Machine manifest from a file. A machine with no442// manifest is still a valid machine, because every field defaults.443// But a manifest that exists and does not parse, or that declares444// some other kind, is a configuration error. Load reports this error445// as a configuration error.446func Load(path string) (*Machine, error) {447	raw, err := os.ReadFile(path)448	if errors.Is(err, fs.ErrNotExist) {449		return &Machine{}, nil450	}451	if err != nil {452		return nil, err453	}454	m, err := Parse(raw)455	if err != nil {456		return nil, fmt.Errorf("%s: %w", path, err)457	}458	return m, nil459}
machine/moduleparameters.go 100.0%
1package machine23// A module parameter is a setting a driver reads once, when the4// kernel loads it, and never again. The spec declares one as a5// "<module>.<parameter>" key, which is the kernel's own spelling:6// the same form the kernel command line takes and the same path7// /sys/module/<module>/parameters/<parameter> mirrors, so a person8// who knows the parameter can find the declaration by its name. The9// functions here turn that flat map into what each consumer needs:10// the string finit_module takes, the names to read back, and the11// modules the map mentions.1213import (14	"maps"15	"slices"16	"strings"17)1819// splitModuleParameterKey splits the dotted key into its module and20// parameter halves. Admission's pattern allows exactly one dot, so21// the first dot is the split; a key that reaches here another way,22// for example from a hand-written manifest on a stick, still splits23// or is rejected as no key at all.24func splitModuleParameterKey(key string) (module, parameter string, ok bool) {25	module, parameter, ok = strings.Cut(key, ".")26	if !ok || module == "" || parameter == "" {27		return "", "", false28	}29	return module, parameter, true30}3132// ModuleParameterString builds the string finit_module takes for one33// module: the parameter=value pairs whose keys name it, sorted by34// parameter name and joined with spaces. The sort makes the string35// stable, so the same spec produces the same boot record on every36// boot and the drift comparison stays a string comparison.37func ModuleParameterString(module string, parameters map[string]string) string {38	var pairs []string39	for _, key := range slices.Sorted(maps.Keys(parameters)) {40		name, parameter, ok := splitModuleParameterKey(key)41		if !ok || name != module {42			continue43		}44		pairs = append(pairs, parameter+"="+parameters[key])45	}46	return strings.Join(pairs, " ")47}4849// ModuleParameterNames lists the parameter names declared for one50// module, sorted, which is the list the readback walks under51// /sys/module/<module>/parameters after the load.52func ModuleParameterNames(module string, parameters map[string]string) []string {53	var names []string54	for _, key := range slices.Sorted(maps.Keys(parameters)) {55		name, parameter, ok := splitModuleParameterKey(key)56		if !ok || name != module {57			continue58		}59		names = append(names, parameter)60	}61	return names62}6364// ModuleParameterModules lists the modules the key set mentions,65// sorted, in the keys' own exact spelling. Admission has no66// normalizer, and module names take - and _ interchangeably, so a67// key must spell its module the way spec.modules spells it; the CEL68// rule on the spec states that requirement and this function does69// not soften it.70func ModuleParameterModules(parameters map[string]string) []string {71	names := map[string]bool{}72	for key := range parameters {73		if module, _, ok := splitModuleParameterKey(key); ok {74			names[module] = true75		}76	}77	return slices.Sorted(maps.Keys(names))78}
machine/reboot.go 97.4%
1package machine23// The reboot channel is how the operator asks init to reboot the4// machine.5//6// Only PID 1 can shut a machine down properly, so the operator never7// reboots the machine itself. Instead, the operator writes an intent8// file. Init watches this directory with inotify. When init finds the9// file, init stops k3s cleanly and reboots the machine.10//11// The channel is a directory of its own under /run/liken, because12// the two programs' mounts enforce the two directions of the flow.13// Facts flow from init to the operator through a read-only mount.14// Intents flow from the operator to init through this directory.15// /run is a fresh tmpfs on every boot. So an intent can never survive16// the reboot that it requested, and it can never cause a second17// reboot.1819import (20	"errors"21	"io/fs"22	"os"23	"path/filepath"24	"time"2526	"sigs.k8s.io/yaml"27)2829// OperatorRunDir is the operator's writable channel to init. Init30// creates this directory, along with the rest of /run/liken, before31// it starts k3s.32const OperatorRunDir = "/run/liken/operator"3334const rebootIntentFile = "reboot-intent.yaml"3536// A RebootIntent states why the machine should reboot and which37// staged manifest the reboot is meant to apply. The presence of the38// file is the trigger for the reboot. The content of the file only39// adds detail to what init prints on the console.40type RebootIntent struct {41	Reason       string    `json:"reason"`42	ManifestHash string    `json:"manifestHash,omitempty"`43	RequestedAt  time.Time `json:"requestedAt"`44}4546// writeIntent writes one intent file atomically, using writeAtomic,47// the same function the facts use. writeAtomic renames a finished file48// into the directory, so init reads either a whole intent or no file at49// all, and that same rename is the event that wakes init's watch. The50// channel directory belongs to init to create, so writeIntent reports a51// missing directory as an error. writeIntent never creates the52// directory itself.53func writeIntent(dir, name string, intent any) error {54	raw, err := yaml.Marshal(intent)55	if err != nil {56		return err57	}58	return writeAtomic(filepath.Join(dir, name), raw)59}6061// readIntent reads one intent file into out, and reports whether the62// file was present. A result of false with no error means that no63// intent exists, which is the result for almost every scan.64func readIntent(dir, name string, out any) (bool, error) {65	raw, err := os.ReadFile(filepath.Join(dir, name))66	if errors.Is(err, fs.ErrNotExist) {67		return false, nil68	}69	if err != nil {70		return false, err71	}72	if err := yaml.UnmarshalStrict(raw, out); err != nil {73		return false, err74	}75	return true, nil76}7778// WriteRebootIntent writes a request to reboot the machine.79func WriteRebootIntent(dir string, intent *RebootIntent) error {80	return writeIntent(dir, rebootIntentFile, intent)81}8283// ReadRebootIntent reports the pending intent, or reports nil when no84// reboot has been requested.85func ReadRebootIntent(dir string) (*RebootIntent, error) {86	intent := &RebootIntent{}87	present, err := readIntent(dir, rebootIntentFile, intent)88	if !present || err != nil {89		return nil, err90	}91	return intent, nil92}9394const restartIntentFile = "restart-intent.yaml"9596// A RestartIntent asks init to bounce the k3s child process in97// place. Init uses this bounce for changes that k3s reads only at98// process start; cluster/changes.go names these changes. This intent99// is a sibling file to the reboot intent, and it is deliberately not100// a field on the reboot intent. Init honors an unreadable reboot101// intent by rebooting anyway. Suppose restart information became a102// new field on the reboot intent instead. An older init that predates103// that field would fail to parse the reboot-intent file whenever only104// a restart was requested. Because of that failure, that older init105// would reboot the machine by surprise, instead of only restarting106// k3s. A sibling file avoids this problem. An older init that107// predates the restart-intent file does not look for it. So the file108// stays invisible to that init, and causes no reaction at all.109//110// The two intents also differ in how they are used. Init never111// consumes a reboot intent, because /run is destroyed along with the112// boot that the intent requested. But init must consume a restart113// intent, or the scan that found it would bounce k3s forever. Like114// the reboot intent, the presence of the restart-intent file is the115// trigger. The staged stores on machineState hold the truth about116// what to apply. This means a duplicate intent is harmless: init117// checks the stores, not the intent, to determine what to apply.118type RestartIntent struct {119	Reason      string    `json:"reason"`120	RequestedAt time.Time `json:"requestedAt"`121}122123// WriteRestartIntent writes a request to restart k3s.124func WriteRestartIntent(dir string, intent *RestartIntent) error {125	return writeIntent(dir, restartIntentFile, intent)126}127128// ReadRestartIntent reports the pending intent, or reports nil when129// no restart has been requested.130func ReadRestartIntent(dir string) (*RestartIntent, error) {131	intent := &RestartIntent{}132	present, err := readIntent(dir, restartIntentFile, intent)133	if !present || err != nil {134		return nil, err135	}136	return intent, nil137}138139// ClearRestartIntent consumes the intent. Init clears the intent file140// before it bounces k3s. If a crash happens between these two steps,141// the machine loses one restart request, but the operator's next pass142// re-requests it. This order is the self-healing order. The reverse143// order would bounce k3s forever, because init would find the same144// intent file on every scan after each bounce. An absent file is145// fine; clearing an absent file is idempotent.146func ClearRestartIntent(dir string) error {147	err := os.Remove(filepath.Join(dir, restartIntentFile))148	if errors.Is(err, fs.ErrNotExist) {149		return nil150	}151	return err152}153154const modulesIntentFile = "modules-intent.yaml"155156// A ModulesIntent asks init to load the staged spec's added kernel157// modules into the running kernel. This intent exists for the one158// machine-spec change that needs no disruption at all: module159// loading. Module loading is possible while the machine is live. A160// driver that is already loaded into the kernel can take control of161// already-plugged-in hardware without a reboot. So an additive162// spec.modules edit should not cost a node drain and a reboot.163//164// This file is a third sibling file, for the same reasons that the165// restart intent is a sibling file rather than a field. First, the166// file is invisible to any init that predates it. Second, init must167// consume the file, like the restart intent, because the machine168// keeps running afterward. If init left the file in place, init would169// load the modules again on every later scan. The presence of the170// file is the trigger, and the staged store holds the truth about171// what to load. Init re-derives whether the staged manifest can be172// applied live, on its own, and init refuses to act on anything that173// would need a boot. This means a stale or duplicate intent is174// harmless.175type ModulesIntent struct {176	Reason       string    `json:"reason"`177	ManifestHash string    `json:"manifestHash,omitempty"`178	RequestedAt  time.Time `json:"requestedAt"`179}180181// WriteModulesIntent writes a request for a live module load.182func WriteModulesIntent(dir string, intent *ModulesIntent) error {183	return writeIntent(dir, modulesIntentFile, intent)184}185186// ReadModulesIntent reports the pending intent, or reports nil when187// no load has been requested.188func ReadModulesIntent(dir string) (*ModulesIntent, error) {189	intent := &ModulesIntent{}190	present, err := readIntent(dir, modulesIntentFile, intent)191	if !present || err != nil {192		return nil, err193	}194	return intent, nil195}196197// ClearModulesIntent consumes the intent. It clears the intent file198// before it acts, in the same order that the restart intent uses, and199// for the same reason. If a crash happens between these two steps,200// the machine loses one request, but the operator's next pass201// re-requests it.202func ClearModulesIntent(dir string) error {203	err := os.Remove(filepath.Join(dir, modulesIntentFile))204	if errors.Is(err, fs.ErrNotExist) {205		return nil206	}207	return err208}
machine/registries.go 100.0%
1package machine23// This file holds the machine's half of private registries: the4// credentials document and the status this boot reports. The5// cluster's half lives in the cluster package (cluster/registries.go).6// It holds the mirror and embedded-registry declarations in7// spec.registries.8//9// The credentials document (RegistryCredentials) is deliberately not10// part of the Cluster spec. A spec is public: anyone who can get the11// Cluster can read every field. A password in a spec would also make12// every credential rotation a document edit that a person must write13// by hand. Instead, credentials enter through a Kubernetes Secret14// (kubernetes.io/dockerconfigjson, the shape that15// `kubectl create secret docker-registry` produces). The machine16// operator reads this Secret and renders it into this document. The17// document has canonical bytes with a hash, and it uses the same18// staged/proven lifecycle as every other document, in its own store.19// The operator is the document's only author: no image ever carries a20// seed. A machine that has never had credentials staged simply has21// none. This is an ordinary state (anonymous pulls), not an error.2223import (24	"cmp"25	"fmt"26	"path/filepath"27	"slices"2829	"github.com/liken-sh/liken/liken/api"30	"sigs.k8s.io/yaml"31)3233// RegistryCredential is one registry's login. It holds the auth that34// containerd presents when it pulls from this host, or from a mirror35// endpoint on it.36type RegistryCredential struct {37	Host     string `json:"host"`38	Username string `json:"username"`39	Password string `json:"password,omitempty"`40}4142// RegistryCredentials is the credentials document. It is the43// operator's rendering of the registry-credentials Secret, in44// liken's own canonical shape. An empty document (no hosts) is the45// retraction rendering: a deleted Secret stages this document. The46// document needs a real hash so the lifecycle can tell "credentials47// withdrawn" apart from "never had any".48type RegistryCredentials struct {49	APIVersion string               `json:"apiVersion"`50	Kind       string               `json:"kind"`51	Hosts      []RegistryCredential `json:"hosts,omitempty"`52}5354// RegistryCredentialsStore is the credentials document's lifecycle55// store. It sits under the given machineState root, beside the56// Machine's, the Cluster's, and the system release's stores. The57// files land owner-only: WriteDurable's temp files are 0600, and58// rename preserves that mode. This matters here more than anywhere59// else, because these bytes are passwords.60func RegistryCredentialsStore(root string) ManifestStore {61	return ManifestStore{dir: filepath.Join(root, "registries")}62}6364// RenderRegistryCredentials produces the document's canonical bytes65// and their hash. It sorts the hosts so the same credentials always66// render the same bytes. The hash is the document's identity for67// staging idempotence, so an incidental ordering difference must68// never look like drift. The function copies the input before it69// sorts the hosts; a caller's slice stays the caller's own.70func RenderRegistryCredentials(hosts []RegistryCredential) ([]byte, string, error) {71	sorted := slices.Clone(hosts)72	slices.SortFunc(sorted, func(a, b RegistryCredential) int {73		return cmp.Compare(a.Host, b.Host)74	})75	return renderDocument(RegistryCredentials{76		APIVersion: api.APIVersion,77		Kind:       "RegistryCredentials",78		Hosts:      sorted,79	})80}8182// ParseRegistryCredentials reads the document strictly, like every83// liken document. It rejects a malformed rendering once, visibly,84// instead of retrying it forever. Every entry must name its host and85// a username. A password may be empty, because some registries86// accept an empty password. But an entry with no username could87// never authenticate anywhere, so the function reports it as a88// rendering bug.89func ParseRegistryCredentials(raw []byte) (*RegistryCredentials, error) {90	c := &RegistryCredentials{}91	if err := yaml.UnmarshalStrict(raw, c); err != nil {92		return nil, err93	}94	if c.Kind != "RegistryCredentials" {95		return nil, fmt.Errorf("expected kind RegistryCredentials, got %q", c.Kind)96	}97	for _, h := range c.Hosts {98		if h.Host == "" {99			return nil, fmt.Errorf("a credential entry names no host")100		}101		if h.Username == "" {102			return nil, fmt.Errorf("the credential for %s names no username", h.Host)103		}104	}105	return c, nil106}107108// RegistriesStatus is what this boot rendered into registries.yaml.109// It holds hosts and counts only, never credential material. It110// exists for console parity: it makes the same facts init prints at111// boot queryable on the Machine.112type RegistriesStatus struct {113	// Mirrors lists the registry hosts that registries.yaml carries114	// mirror entries for. The list includes "*" when the embedded115	// registry adds it.116	Mirrors []string `json:"mirrors,omitempty"`117118	// CredentialedHosts lists the hosts that registries.yaml carries119	// auth for. It lists only the hosts, and it never lists the120	// credentials.121	CredentialedHosts []string `json:"credentialedHosts,omitempty"`122123	// Embedded reports whether this boot turned the embedded124	// registry on.125	Embedded bool `json:"embedded,omitempty"`126}
machine/release.go 100.0%
1package machine23// The release document defines what one version of liken is, by4// digest.5//6// This document names the release's artifact files with their7// sha256 digests and sizes. It also records which upstream8// components (the kernel, k3s, and the rest) shipped inside them.9// The Cluster's release catalog pins a version to the digest of this10// document's exact bytes. This document, in turn, pins every11// artifact's exact bytes. A machine that checks both digests has12// proven that what it is about to boot is exactly what the catalog13// named. Nothing here is signed, because liken already trusts the14// Kubernetes API, and the catalog lives in it.15//16// Two consumers read this document. The installer reads a copy17// baked beside the artifacts it describes, and verifies it before it18// copies anything to a slot. The operator's release downloader reads19// a copy fetched from the release server, and verifies it against20// the catalog before it fetches any artifact.2122import (23	"crypto/sha256"24	"encoding/hex"25	"fmt"26	"io"2728	"github.com/liken-sh/liken/liken/api"29	"sigs.k8s.io/yaml"30)3132type Release struct {33	APIVersion string             `json:"apiVersion"`34	Kind       string             `json:"kind"`35	Metadata   api.ObjectMeta     `json:"metadata"`36	Artifacts  []ReleaseArtifact  `json:"artifacts"`37	Components []ReleaseComponent `json:"components,omitempty"`38}3940// A ReleaseArtifact names one file of the release. The size is41// informational: the code uses it for progress reporting and42// validity checks. The digest is the identity.43type ReleaseArtifact struct {44	Name   string `json:"name"`45	SHA256 string `json:"sha256"`46	Size   int64  `json:"size"`47}4849// A ReleaseComponent records one vendored piece of the system, and50// the upstream version of it that shipped: the kernel, k3s, and the51// rest. liken's own version is a calendar date. The date states when52// a release was made, and it says nothing about what is inside the53// release. So this document carries that information instead. The54// versions are informational, each in its own upstream format. The55// artifacts' digests remain the only identity.56type ReleaseComponent struct {57	Name    string `json:"name"`58	Version string `json:"version"`59}6061// ParseRelease validates a release document as it reads the62// document, the same way Parse and ParseCluster validate theirs. If63// a document is not exactly what it claims to be, ParseRelease64// rejects it and states the reason. It never accepts a document65// partially.66func ParseRelease(raw []byte) (*Release, error) {67	r := &Release{}68	if err := yaml.UnmarshalStrict(raw, r); err != nil {69		return nil, err70	}71	if r.Kind != "Release" {72		return nil, fmt.Errorf("expected kind Release, got %q", r.Kind)73	}74	if r.Metadata.Name == "" {75		return nil, fmt.Errorf("a release must name its version")76	}77	if len(r.Artifacts) == 0 {78		return nil, fmt.Errorf("release %s lists no artifacts", r.Metadata.Name)79	}80	for _, a := range r.Artifacts {81		if a.Name == "" {82			return nil, fmt.Errorf("release %s has an unnamed artifact", r.Metadata.Name)83		}84		if len(a.SHA256) != 64 {85			return nil, fmt.Errorf("artifact %s: sha256 must be 64 hex characters, got %d", a.Name, len(a.SHA256))86		}87		if _, err := hex.DecodeString(a.SHA256); err != nil {88			return nil, fmt.Errorf("artifact %s: sha256 is not hex: %w", a.Name, err)89		}90	}91	for _, c := range r.Components {92		if c.Name == "" || c.Version == "" {93			return nil, fmt.Errorf("release %s has a component missing its name or version", r.Metadata.Name)94		}95	}96	return r, nil97}9899// Verify streams a reader through sha256, and compares the result100// against this artifact's declared digest and size. Because Verify101// streams the data, a 100MB artifact never needs to stay in memory102// all at once. Callers verify by reading again the file they wrote.103// This checks the bytes that actually reached the disk, rather than104// the bytes the caller meant to write.105func (a ReleaseArtifact) Verify(r io.Reader) error {106	h := sha256.New()107	n, err := io.Copy(h, r)108	if err != nil {109		return fmt.Errorf("reading %s: %w", a.Name, err)110	}111	if n != a.Size {112		return fmt.Errorf("%s is %d bytes, want %d", a.Name, n, a.Size)113	}114	if got := hex.EncodeToString(h.Sum(nil)); got != a.SHA256 {115		return fmt.Errorf("%s digest mismatch: got %s, want %s", a.Name, got, a.SHA256)116	}117	return nil118}
machine/rlimit.go 98.3%
1package machine23// Resource limits are the kernel's per-process ceilings: how many4// files one process may hold open, how many processes one user may5// run. The kernel gives PID 1 a fixed pair of numbers at boot and6// every process inherits its parent's pair across fork and exec.7// There is no /proc file to write and no way to change a process8// that is already running. Whoever starts a process is the only one9// who can decide its limits, and only before it starts.10//11// That inheritance rule is why this matters on liken. A stock12// distribution sets these per service, in each unit's Limit*13// directives, and systemd applies them when it forks the service.14// liken runs no systemd, so nothing applies them, and k3s, containerd,15// the shims, and every container run under the kernel's own defaults16// of 1024 soft and 4096 hard. rlimitdefaults.go is the table that17// answers this, and init applies it to itself before it starts18// anything, because inheritance is the only transport available.19//20// The value syntax is systemd's, so an operator writes into21// spec.rlimits what a unit file writes in LimitNOFILE. A bare number22// sets both halves, the word "infinity" means no limit, and a23// "soft:hard" pair sets the halves separately. Reusing that grammar24// costs nothing and means the value in a Machine manifest can be25// compared to the unit it replaces without translation.26//27// A resource is named by its ulimit short name, again systemd's28// vocabulary: "nofile" is RLIMIT_NOFILE, "nproc" is RLIMIT_NPROC.29// Only the names in rlimitResources exist. An unknown name is an30// error rather than a silent no-op, for the same reason ApplySysctl31// refuses to create a parameter file: a typo that applies nothing32// must be visible.3334import (35	"fmt"36	"maps"37	"slices"38	"strconv"39	"strings"4041	"golang.org/x/sys/unix"42)4344// RlimitInfinity is the word an operator writes, and the word the45// status reports back, for a limit the kernel does not enforce. The46// kernel's own value is the largest unsigned 64-bit integer, which is47// not a number anybody should have to read or type.48const RlimitInfinity = "infinity"4950// rlimitResources maps liken's spelling of a resource to the kernel's51// constant. The list is deliberately short. Every entry here is a52// limit that liken has a reason to set, or that a deployment has a53// reason to override, and adding a name is how that reason gets54// recorded. "core" appears here although OSRlimits does not set it,55// because a deployment that wants core dumps must be able to ask for56// them; rlimitdefaults.go explains why liken does not ask by default.57var rlimitResources = map[string]int{58	"nofile":  unix.RLIMIT_NOFILE,59	"nproc":   unix.RLIMIT_NPROC,60	"core":    unix.RLIMIT_CORE,61	"memlock": unix.RLIMIT_MEMLOCK,62	"stack":   unix.RLIMIT_STACK,63	"fsize":   unix.RLIMIT_FSIZE,64	"nice":    unix.RLIMIT_NICE,65}6667// RlimitResourceNames lists every resource liken accepts, in sorted68// order. Error messages and the CRD's own description read from this69// list, so the accepted set is stated in one place.70func RlimitResourceNames() []string {71	names := make([]string, 0, len(rlimitResources))72	for name := range rlimitResources {73		names = append(names, name)74	}75	slices.Sort(names)76	return names77}7879// ParseRlimit turns one spec value into the pair of numbers80// setrlimit(2) takes. The three accepted forms are systemd's:81//82//	"1048576"          both halves83//	"infinity"         both halves, unlimited84//	"1024:1048576"     soft, then hard85//86// A soft limit above the hard limit is rejected here rather than at87// the syscall, because the syscall's EINVAL says nothing about which88// half was wrong.89func ParseRlimit(value string) (unix.Rlimit, error) {90	soft, hard, found := strings.Cut(value, ":")91	if !found {92		hard = soft93	}94	softN, err := parseRlimitHalf(soft)95	if err != nil {96		return unix.Rlimit{}, err97	}98	hardN, err := parseRlimitHalf(hard)99	if err != nil {100		return unix.Rlimit{}, err101	}102	if softN > hardN {103		return unix.Rlimit{}, fmt.Errorf(104			"soft limit %s is above hard limit %s; a process may never exceed its hard limit", soft, hard)105	}106	return unix.Rlimit{Cur: softN, Max: hardN}, nil107}108109// parseRlimitHalf reads one half of a value. An empty half is110// rejected, so "1024:" and ":4096" are errors rather than a half111// silently taken as zero. A limit of zero is a real setting, and it112// must be spelled out.113func parseRlimitHalf(half string) (uint64, error) {114	if half == RlimitInfinity {115		return unix.RLIM_INFINITY, nil116	}117	if half == "" {118		return 0, fmt.Errorf("empty limit; write a number, %q, or \"soft:hard\"", RlimitInfinity)119	}120	n, err := strconv.ParseUint(half, 10, 64)121	if err != nil {122		return 0, fmt.Errorf("limit %q is not a number or %q", half, RlimitInfinity)123	}124	// The largest unsigned 64-bit value is what the kernel uses for125	// RLIM_INFINITY. A spec that spells that number out gets the word126	// back from FormatRlimit, so refusing it here would make a value127	// that the status reports impossible to declare.128	return n, nil129}130131// FormatRlimit renders a pair of numbers back into the spec's own132// syntax. The status reports what the kernel holds, and it reports it133// in the grammar the spec is written in, so a person comparing the134// two is comparing like with like. Equal halves collapse to one135// number, the same way a person would write them.136func FormatRlimit(lim unix.Rlimit) string {137	soft := formatRlimitHalf(lim.Cur)138	hard := formatRlimitHalf(lim.Max)139	if soft == hard {140		return soft141	}142	return soft + ":" + hard143}144145func formatRlimitHalf(n uint64) string {146	if n == unix.RLIM_INFINITY {147		return RlimitInfinity148	}149	return strconv.FormatUint(n, 10)150}151152// RlimitResource translates a resource name to the kernel's constant.153func RlimitResource(name string) (int, error) {154	resource, ok := rlimitResources[name]155	if !ok {156		return 0, fmt.Errorf("unknown resource %q; liken accepts %s",157			name, strings.Join(RlimitResourceNames(), ", "))158	}159	return resource, nil160}161162// ValidateRlimits checks that every entry names a resource this163// machine has and carries a value that parses. It catches the errors a164// person can fix in the manifest, before a reboot spends itself165// finding them out.166//167// The API server enforces the same rules on any spec applied through168// it, from the CRD's own schema. This check exists for the specs that169// reach a machine another way: init also reads manifests that a person170// wrote by hand and carried in on a stick, and no API server ever saw171// those. It is the same reasoning NetworkSpec.Validate states.172//173// Init does not call this. A limit it cannot apply is skipped with a174// message, because OSRlimits ships with the release and one bad entry175// must never cost a fleet its boot. The operator calls it, where the176// cost of a wrong answer is a condition rather than an outage.177func ValidateRlimits(rlimits map[string]string) error {178	for _, name := range slices.Sorted(maps.Keys(rlimits)) {179		if _, err := RlimitResource(name); err != nil {180			return fmt.Errorf("rlimit %s: %w", name, err)181		}182		if _, err := ParseRlimit(rlimits[name]); err != nil {183			return fmt.Errorf("rlimit %s = %s: %w", name, rlimits[name], err)184		}185	}186	return nil187}188189// ApplyRlimit sets one resource limit on the calling process, and190// therefore on every process it starts afterward.191//192// The call goes through golang.org/x/sys/unix rather than the raw193// syscall for a reason that is easy to miss. Go's runtime raises its194// own soft RLIMIT_NOFILE to the hard limit at startup, and remembers195// the original so that os/exec can restore it in each child, sparing196// old programs that use select(2) and its fixed-size descriptor set.197// A child that inherits the restored value inherits the limit liken198// was trying to replace. unix.Setrlimit delegates to syscall.Setrlimit,199// which discards that remembered value, so children inherit what this200// function set. A direct syscall would not, and the machine would look201// correct in /proc/1/limits while every process below it kept the old202// ceiling.203func ApplyRlimit(name, value string) error {204	resource, err := RlimitResource(name)205	if err != nil {206		return fmt.Errorf("rlimit %s: %w", name, err)207	}208	lim, err := ParseRlimit(value)209	if err != nil {210		return fmt.Errorf("rlimit %s = %s: %w", name, value, err)211	}212	if err := unix.Setrlimit(resource, &lim); err != nil {213		return fmt.Errorf("rlimit %s = %s: %w", name, value, err)214	}215	return nil216}217218// ReadRlimit reads one resource limit back from the calling process.219// Init builds its report from these reads rather than from what it220// wrote, for the same reason ReadSysctl exists: the report must say221// what the kernel holds, not what somebody asked for. A limit the222// kernel clamped or refused shows its real value here.223func ReadRlimit(name string) (string, error) {224	resource, err := RlimitResource(name)225	if err != nil {226		return "", fmt.Errorf("rlimit %s: %w", name, err)227	}228	var lim unix.Rlimit229	if err := unix.Getrlimit(resource, &lim); err != nil {230		return "", fmt.Errorf("rlimit %s: %w", name, err)231	}232	return FormatRlimit(lim), nil233}
machine/serio.go 97.1%
1package machine23// Serio attachments: the serial-line devices whose kernel driver binds4// only after a program attaches the line to the kernel's serio layer.5//6// Serio is the kernel's layer for input devices that talk over a byte7// stream: old serial mice, touchscreens, and USB-CEC adapters. A serio8// driver binds to a serio port, not to a tty, and a serio port exists9// only while a program holds the serport line discipline on the tty10// and blocks in a read on it. The kernel unregisters the port when11// that read returns. On a general-purpose distribution, udev starts a12// service that runs inputattach to hold the read. liken has neither,13// so init holds it (init/serio.go), for each entry in spec.serio.14//15// This file is the one piece of device knowledge liken keeps for16// these devices: the protocol table. Each name maps to the values17// that inputattach.c uses for the same device, and the serio type18// comes from the kernel's include/uapi/linux/serio.h. A name joins19// the table only when somebody examines that device on a liken20// machine, because a wrong type binds no driver, and a wrong line21// setting can bind one that reads garbage.2223import (24	"fmt"25	"regexp"26	"slices"27)2829// SerioAttachment is one spec.serio entry: a protocol from the table,30// and the USB device whose serial line carries it.31type SerioAttachment struct {32	// Protocol names a row of the protocol table, for example33	// pulse8-cec.34	Protocol string `json:"protocol"`3536	// USB is the identity of the hardware that carries the serial37	// line. The match uses this identity, not a tty name, because the38	// kernel numbers ttyACM0 and ttyACM1 in the order the devices were39	// plugged in.40	USB SerioUSB `json:"usb"`41}4243// SerioUSB is a USB device's identity, in the spelling the DRA driver44// publishes as the vendor and product attributes: four lowercase hex45// digits each, with no prefix. Serial is optional. With it, the entry46// matches only the unit that reports that serial number. Without it,47// the entry matches every device of that model, and the serio walk48// gives it only the lines that no entry with a serial took49// (init/serio.go).50type SerioUSB struct {51	Vendor  string `json:"vendor"`52	Product string `json:"product"`53	Serial  string `json:"serial,omitempty"`54}5556// Matches reports whether a USB device's identity is the one this57// entry declares.58func (a SerioAttachment) Matches(vendor, product, serial string) bool {59	if a.USB.Vendor != vendor || a.USB.Product != product {60		return false61	}62	return a.USB.Serial == "" || a.USB.Serial == serial63}6465// String names the entry for a console line or a status message.66func (a SerioAttachment) String() string {67	s := fmt.Sprintf("%s %s:%s", a.Protocol, a.USB.Vendor, a.USB.Product)68	if a.USB.Serial != "" {69		s += " serial " + a.USB.Serial70	}71	return s72}7374// SerioProtocol is one row of the protocol table: the values init75// passes to the kernel for the attach, and the modules the attach76// depends on.77type SerioProtocol struct {78	Name string7980	// Type is the serio type from the kernel's serio.h, which81	// SPIOCSTYPE hands to serport. The kernel matches serio drivers82	// to ports by this number, so it is what makes pulse8_cec, and no83	// other driver, bind the port.84	Type uint88586	// Baud is the line speed, in both directions. Every row today87	// uses 9600 baud, 8 data bits, and raw mode, which is what the88	// adapters' firmware speaks.89	Baud int9091	// LineDriver is the module that creates the serial line, and92	// Discipline is the module that registers the serport line93	// discipline. Driver is the serio driver that binds the port.94	// The attach needs all three loaded, and spec.modules is where a95	// machine declares them.96	LineDriver string97	Discipline string98	Driver     string99100	// Adapters lists the USB identities known to speak this protocol.101	// The unclaimed report uses the list to name the rest of the fix102	// for an adapter that a line driver binds but no entry attaches.103	Adapters []SerioUSB104}105106// serioProtocols is the table. The two rows are the two USB-CEC107// adapters the kernel supports. inputattach runs both as108// `--pulse8-cec` and `--rainshadow-cec`, at 9600 baud, with the serio109// types SERIO_PULSE8_CEC and SERIO_RAINSHADOW_CEC.110var serioProtocols = []SerioProtocol{111	{112		Name: "pulse8-cec", Type: 0x40, Baud: 9600,113		LineDriver: "cdc_acm", Discipline: "serport", Driver: "pulse8_cec",114		Adapters: []SerioUSB{{Vendor: "2548", Product: "1002"}},115	},116	{117		Name: "rainshadow-cec", Type: 0x41, Baud: 9600,118		LineDriver: "cdc_acm", Discipline: "serport", Driver: "rainshadow_cec",119	},120}121122// LookupSerioProtocol returns the table's row for a protocol name.123func LookupSerioProtocol(name string) (SerioProtocol, bool) {124	for _, p := range serioProtocols {125		if p.Name == name {126			return p, true127		}128	}129	return SerioProtocol{}, false130}131132// SerioProtocolNames lists the table's names, sorted. The CRD's enum133// holds the same list, and a test keeps the two equal.134func SerioProtocolNames() []string {135	names := make([]string, 0, len(serioProtocols))136	for _, p := range serioProtocols {137		names = append(names, p.Name)138	}139	slices.Sort(names)140	return names141}142143// KnownSerioAdapter returns the protocol a USB identity is known to144// speak, or "" for an identity the table does not list.145func KnownSerioAdapter(vendor, product string) string {146	for _, p := range serioProtocols {147		for _, adapter := range p.Adapters {148			if adapter.Vendor == vendor && adapter.Product == product {149				return p.Name150			}151		}152	}153	return ""154}155156// usbIDPattern is the spelling sysfs uses for idVendor and idProduct.157// serialPattern refuses whitespace and control bytes, because the158// serial reaches a record file in the facts tree, whose grammar is one159// line per field.160var (161	usbIDPattern  = regexp.MustCompile(`^[0-9a-f]{4}$`)162	serialPattern = regexp.MustCompile(`^[!-~]{1,126}$`)163)164165// ValidateSerio checks the entries the API server also checks, for166// the manifests that reach a machine without it: a manifest carried167// in on a stick was never admitted. An entry that fails here could168// never match a device, or would match one twice.169func ValidateSerio(entries []SerioAttachment) error {170	for i, a := range entries {171		if _, ok := LookupSerioProtocol(a.Protocol); !ok {172			return fmt.Errorf("serio entry %d names protocol %q; the protocols are %v", i, a.Protocol, SerioProtocolNames())173		}174		if !usbIDPattern.MatchString(a.USB.Vendor) {175			return fmt.Errorf("serio entry %d declares vendor %q; a vendor is four lowercase hex digits, as sysfs spells idVendor", i, a.USB.Vendor)176		}177		if !usbIDPattern.MatchString(a.USB.Product) {178			return fmt.Errorf("serio entry %d declares product %q; a product is four lowercase hex digits, as sysfs spells idProduct", i, a.USB.Product)179		}180		if a.USB.Serial != "" && !serialPattern.MatchString(a.USB.Serial) {181			return fmt.Errorf("serio entry %d declares serial %q; a serial holds no space or control byte", i, a.USB.Serial)182		}183		for j := range i {184			if entries[j] == a {185				return fmt.Errorf("serio entries %d and %d both declare %s; declare each attachment once", j, i, a)186			}187		}188	}189	return nil190}191192// SerioState is one attachment's standing on the machine.193//194// Attached means init holds the read, and the port exists. Missing195// means no serial line matches the entry: the adapter is unplugged, or196// its line driver is not loaded. Refused means a line matches and the197// attach did not complete: a module the attach needs is not loaded,198// or the kernel refused one of the calls.199type SerioState string200201const (202	SerioAttached SerioState = "Attached"203	SerioMissing  SerioState = "Missing"204	SerioRefused  SerioState = "Refused"205)206207// SerioStatus is one attachment that init holds or tried to hold. An208// entry that matches no line reports once, as Missing. An entry that209// matches two lines, because it declares no serial and two identical210// adapters are plugged in, reports once for each line.211type SerioStatus struct {212	Protocol string   `json:"protocol"`213	USB      SerioUSB `json:"usb"`214215	// TTY is the serial line init matched, for example ttyACM0, and216	// Port is the serio port the kernel registered on it, for example217	// serio0.218	TTY  string `json:"tty,omitempty"`219	Port string `json:"port,omitempty"`220221	State SerioState `json:"state"`222223	// Message gives the cause of a Missing or Refused state, with the224	// kernel's error text word for word, or the module to declare.225	Message string `json:"message,omitempty"`226227	// Nodes lists the device nodes the attached driver created under228	// the port, for example /dev/cec0 and the remote's event node.229	Nodes []string `json:"nodes,omitempty"`230}231232// Attachment returns the spec entry a status reports on.233func (s SerioStatus) Attachment() SerioAttachment {234	return SerioAttachment{Protocol: s.Protocol, USB: s.USB}235}
machine/staging.go 87.8%
1package machine23// The manifest lifecycle is how a document edit survives a reboot.4//5// A machine with a machineState storage role keeps its manifests on6// that filesystem, one directory for each document. Several documents7// use this lifecycle. Each one has a constructor below, or in its own8// file (systemrelease.go, registries.go, imports.go) that names its9// directory. Every directory holds the same three files, with the10// same meanings:11//12//	staged.yaml    a document that awaits its first successful boot,13//	               written by the operator when the cluster's copy14//	               differs from what this boot actuated15//	proven.yaml    the last document that fully proved out: the16//	               last-known-good copy that a failed staged17//	               document falls back to18//	rejected.yaml  a staged document that failed its boot, moved19//	               aside (with rejection.yaml stating why) so that20//	               init never tries it again, but never silently21//	               forgets it either22//23// At boot, init prefers the staged document over the proven one.24// Success promotes staged to proven with one rename. Failure25// quarantines the staged document and boots the proven document26// instead. The image's baked-in copy takes part only when none of27// these files exist: it seeds the very first boot, and the first28// success writes it down as the first proven.yaml.29//30// A store deals in bytes, never in parsed documents. The lifecycle's31// whole job is to make the right bytes survive reboots and power32// loss. Which kind of document those bytes contain, a Machine or a33// Cluster, is the caller's business. This is also why rejections34// hash the raw bytes: a document that will not even parse must still35// be identifiable as exactly the bytes that init refused.36//37// Every write here is both atomic and durable: a temp file, an fsync38// of the file, a rename, and an fsync of the directory. A rename39// alone makes a write atomic against a crash of the writer, but not40// against power loss. The directory records the rename, and an41// unsynced directory update can vanish when power fails.42// The facts tree skips the fsyncs, because /run is tmpfs. These lifecycle files43// exist precisely to survive a power loss, so they perform both44// fsyncs every time.4546import (47	"crypto/sha256"48	"encoding/hex"49	"errors"50	"io/fs"51	"os"52	"path/filepath"53	"strings"54	"time"5556	"sigs.k8s.io/yaml"57)5859// MachineStateDir is where the machineState role mounts. It is the60// root that the store constructors take as a parameter, so tests can61// use a temporary directory, and it is the path through which the62// operator reaches the same trees.63const MachineStateDir = "/var/lib/liken/machine"6465const (66	stagedManifest   = "staged.yaml"67	provenManifest   = "proven.yaml"68	rejectedManifest = "rejected.yaml"69	rejectionNote    = "rejection.yaml"70	attemptedMarker  = "attempted"71)7273// A ManifestStore is one document's lifecycle storage: a directory on74// machineState that holds that document's staged, proven, and75// rejected files. Each document's constructor is the only place that76// names its directory, so no two lifecycles can ever collide.77type ManifestStore struct {78	dir string79}8081// MachineManifests is the Machine manifest's store under the given82// machineState root.83func MachineManifests(root string) ManifestStore {84	return ManifestStore{dir: filepath.Join(root, "manifests")}85}8687// ClusterManifests is the Cluster manifest's store under the same88// root. It runs the same lifecycle as the Machine's, in its own89// directory.90func ClusterManifests(root string) ManifestStore {91	return ManifestStore{dir: filepath.Join(root, "cluster")}92}9394// A Rejection records why a staged document was refused. One type95// is both rejection.yaml's schema and the facts entry, so the96// console message, the on-disk record, and the cluster's status all97// carry the same record. The hash identifies exactly which bytes were98// rejected. The operator refuses to re-stage a document that matches99// it, and only a genuinely different edit clears the block.100type Rejection struct {101	Hash       string    `json:"hash,omitempty"`102	Reason     string    `json:"reason"`103	RejectedAt time.Time `json:"rejectedAt"`104}105106// ManifestHash is a document's identity: the sha256 of its exact107// bytes as they sit in the file. "The same document" must mean the108// same thing to the operator that staged it and to the init that109// booted it, or rejected it.110func ManifestHash(raw []byte) string {111	sum := sha256.Sum256(raw)112	return hex.EncodeToString(sum[:])113}114115// renderDocument produces a document's canonical bytes and their116// hash, for the documents that liken itself authors: the117// credentials, the imported-images record, and the system release.118// yaml marshals through JSON with sorted keys, so the same document119// always renders the same bytes. This is what lets a hash comparison120// answer the question "did anything change".121func renderDocument(doc any) ([]byte, string, error) {122	raw, err := yaml.Marshal(doc)123	if err != nil {124		return nil, "", err125	}126	return raw, ManifestHash(raw), nil127}128129// NewRejection builds the record for refusing a staged document:130// exactly these bytes, for this reason, at this moment. It exists131// apart from Reject, because a boot reports a rejection (on the132// console, in the boot's facts) even when the durable write of the133// record fails. The record's existence must not depend on the write134// succeeding.135func NewRejection(raw []byte, reason string, at time.Time) Rejection {136	return Rejection{Hash: ManifestHash(raw), Reason: reason, RejectedAt: at}137}138139// load reads one lifecycle file. A missing file is not an error. It140// returns nil bytes, because most machines have nothing staged most141// of the time.142func (s ManifestStore) load(name string) ([]byte, error) {143	raw, err := os.ReadFile(filepath.Join(s.dir, name))144	if errors.Is(err, fs.ErrNotExist) {145		return nil, nil146	}147	if err != nil {148		return nil, err149	}150	return raw, nil151}152153func (s ManifestStore) LoadStaged() ([]byte, error) {154	return s.load(stagedManifest)155}156157func (s ManifestStore) LoadProven() ([]byte, error) {158	return s.load(provenManifest)159}160161// LoadRejection reads the standing rejection, if any. A rejection162// stands until a later promotion removes it. tmpfs holds the facts163// file, and each reboot loses it, so this file is the durable record164// that lets every boot keep reporting "this document was rejected"165// until something supersedes it.166func (s ManifestStore) LoadRejection() (*Rejection, error) {167	raw, err := s.load(rejectionNote)168	if raw == nil || err != nil {169		return nil, err170	}171	r := &Rejection{}172	if err := yaml.UnmarshalStrict(raw, r); err != nil {173		return nil, err174	}175	return r, nil176}177178// WriteStaged stages a document for the next boot: the operator's one179// write into the lifecycle.180func (s ManifestStore) WriteStaged(raw []byte) error {181	if err := os.MkdirAll(s.dir, 0o755); err != nil {182		return err183	}184	return WriteDurable(filepath.Join(s.dir, stagedManifest), raw)185}186187// WriteProven records a document as the last-known-good copy188// directly. The first successful boot does this with the image's189// seed. This closes the loop for a machine that has never had a190// staged document to promote.191func (s ManifestStore) WriteProven(raw []byte) error {192	if err := os.MkdirAll(s.dir, 0o755); err != nil {193		return err194	}195	return WriteDurable(filepath.Join(s.dir, provenManifest), raw)196}197198// Promote marks the staged document proven with one rename. The199// rename is atomic by the filesystem's own guarantee, and it replaces200// the old proven file in the same step, so there is never a moment201// with neither file present. A success supersedes any old rejection,202// so those files go too. A crash between the rename and that cleanup203// leaves a stale rejection note beside a newer proven file. This is204// harmless: the boot that just succeeded reports no rejection, and205// the next promotion cleans up the stale note.206func (s ManifestStore) Promote() error {207	if err := os.Rename(filepath.Join(s.dir, stagedManifest), filepath.Join(s.dir, provenManifest)); err != nil {208		return err209	}210	_ = os.Remove(filepath.Join(s.dir, rejectedManifest))211	_ = os.Remove(filepath.Join(s.dir, rejectionNote))212	_ = os.Remove(filepath.Join(s.dir, attemptedMarker))213	return syncDir(s.dir)214}215216// Reject quarantines the staged document. Reject writes the note217// durably first, then turns staged into rejected. A crash between the218// two steps leaves staged.yaml in place, and the next boot simply219// retries it. A persistent failure re-records this same rejection and220// finishes the rename. A transient failure, such as an unplugged221// disk, gets a second chance. Either way, the interrupted state222// converges on its own.223func (s ManifestStore) Reject(r Rejection) error {224	note, err := yaml.Marshal(r)225	if err != nil {226		return err227	}228	if err := WriteDurable(filepath.Join(s.dir, rejectionNote), note); err != nil {229		return err230	}231	if err := os.Rename(filepath.Join(s.dir, stagedManifest), filepath.Join(s.dir, rejectedManifest)); err != nil {232		return err233	}234	_ = os.Remove(filepath.Join(s.dir, attemptedMarker))235	return syncDir(s.dir)236}237238// The attempted marker gives a verdict to a trial that could not239// reach one on its own. Some documents cannot be proven by the boot240// that runs them. A cluster document's failure modes show up241// downstream, because a bad endpoint just means the machine never242// joins the cluster. So init marks the staged document attempted when243// it boots the document, and the component that can observe the244// proof promotes the document and clears the marker. For the cluster245// document, that component is the operator, and the operator's246// existence as a pod demonstrates the join. When a boot finds the247// marker still matching the staged document, the last try was never248// proven, so the boot rejects the document and falls back. Each249// staged document gets exactly one proving boot. No retry counters250// are needed, and a crash at any point leaves a state that the next251// boot reads correctly.252253// WriteAttempted marks the staged document, identified by its hash,254// as tried by this boot.255func (s ManifestStore) WriteAttempted(hash string) error {256	if err := os.MkdirAll(s.dir, 0o755); err != nil {257		return err258	}259	return WriteDurable(filepath.Join(s.dir, attemptedMarker), []byte(hash+"\n"))260}261262// LoadAttempted reads the marker. An empty string means no trial is263// underway.264func (s ManifestStore) LoadAttempted() (string, error) {265	raw, err := s.load(attemptedMarker)266	if raw == nil || err != nil {267		return "", err268	}269	return strings.TrimSpace(string(raw)), nil270}271272// WithdrawStaged removes the staged document. The operator calls this273// when someone has edited the cluster's copy back to what the machine274// already runs. If the staged file stayed behind, the next boot would275// apply an edit that the cluster no longer asks for. The attempted276// marker goes with it. The marker described a trial of the withdrawn277// document, and if it stayed behind, it would read as "tried and278// failed" the next time someone stages the identical document. That279// would be a false rejection, because the trial never reached a280// verdict. A missing file is not an error; there is nothing to281// withdraw.282func (s ManifestStore) WithdrawStaged() error {283	if err := os.Remove(filepath.Join(s.dir, stagedManifest)); err != nil {284		if errors.Is(err, fs.ErrNotExist) {285			return nil286		}287		return err288	}289	_ = os.Remove(filepath.Join(s.dir, attemptedMarker))290	return syncDir(s.dir)291}292293// ClearRejection removes the rejected document and its rejection294// note. The rejection's purpose is to stop the operator from staging295// the same failed document again. Once the cluster stops asking for296// that document, the record has no further use. Without this297// cleanup, every future boot would keep republishing the record in298// status.299func (s ManifestStore) ClearRejection() error {300	removed := false301	for _, name := range []string{rejectedManifest, rejectionNote} {302		err := os.Remove(filepath.Join(s.dir, name))303		if err != nil && !errors.Is(err, fs.ErrNotExist) {304			return err305		}306		removed = removed || err == nil307	}308	if !removed {309		return nil310	}311	return syncDir(s.dir)312}313314// writeAtomic performs the atomic write without the durability: a315// temp file in the same directory, because a rename cannot cross316// filesystems, followed by a rename. A rename within one filesystem317// is atomic, so a reader that polls on its own schedule sees either318// the old contents or the new, never a torn write. It is319// WriteDurable's sibling for files on tmpfs, such as the facts and320// the intent channel, where an fsync would have nothing to flush to.321func writeAtomic(path string, raw []byte) error {322	tmp, err := os.CreateTemp(filepath.Dir(path), ".liken-*")323	if err != nil {324		return err325	}326	defer os.Remove(tmp.Name())327	if _, err := tmp.Write(raw); err != nil {328		tmp.Close()329		return err330	}331	if err := tmp.Close(); err != nil {332		return err333	}334	return os.Rename(tmp.Name(), path)335}336337// WriteDurable performs the atomic, power-loss-proof write: a temp338// file in the same directory, because a rename cannot cross339// filesystems; the contents fsynced before the rename makes them340// visible; and the directory fsynced so the rename itself reaches341// disk. It is exported because init needs the same guarantee for the342// identity files that k3s reads, such as the node password.343func WriteDurable(path string, raw []byte) error {344	dir := filepath.Dir(path)345	tmp, err := os.CreateTemp(dir, ".liken-*")346	if err != nil {347		return err348	}349	defer os.Remove(tmp.Name())350	if _, err := tmp.Write(raw); err != nil {351		tmp.Close()352		return err353	}354	if err := tmp.Sync(); err != nil {355		tmp.Close()356		return err357	}358	if err := tmp.Close(); err != nil {359		return err360	}361	if err := os.Rename(tmp.Name(), path); err != nil {362		return err363	}364	return syncDir(dir)365}366367func syncDir(dir string) error {368	d, err := os.Open(dir)369	if err != nil {370		return err371	}372	defer d.Close()373	return d.Sync()374}
machine/status.go 100.0%
1package machine23// This file defines the observed half of the Machine API. Kubernetes4// convention keeps the two halves separate: the spec is what a user5// requested, and the status is what the controllers observed. The6// status must be reconstructible. An operator must be able to erase7// the status and rebuild it only by observing the machine. The types8// in this file do not store anything between passes. Each pass9// re-derives every value from the current observation.1011import (12	"strings"13	"time"1415	"github.com/liken-sh/liken/liken/api"16)1718type MachineStatus struct {19	// Phase summarizes the machine's state in one word. Each pass20	// computes it from the conditions below. No component stores21	// Phase between passes, so it can never go out of date compared22	// to the conditions. Conditions serve programs, such as kubectl23	// wait and other controllers. Phase serves the human who reads a24	// fleet listing. See the api package's Phase constants for the25	// vocabulary.26	Phase api.Phase `json:"phase,omitempty"`2728	// ObservedGeneration is the metadata.generation of the spec that29	// this status judged, stamped by the operator on every pass. The30	// conditions each carry the same stamp, but a client that only31	// asks "has the operator seen my edit yet" should not have to32	// parse conditions for the answer. Kubernetes controllers publish33	// this field at the top of status for exactly that question.34	// Init leaves this field empty in the facts, because init runs35	// before the cluster exists and a generation is the API server's36	// number to hand out.37	ObservedGeneration int64 `json:"observedGeneration,omitempty"`3839	// Version reports what this machine runs. It does not include the40	// k3s and kubelet versions, because Kubernetes already reports41	// them on the built-in Node object. Version covers the layer42	// below the Node object instead.43	Version VersionStatus `json:"version,omitzero"`4445	// Role is what this machine is in its cluster: a leader that runs46	// a control plane, or a follower that runs workloads. Boot derives47	// Role from the Cluster manifest's leaders list. No one declares48	// Role here directly.49	Role api.Role `json:"role,omitempty"`5051	// Network is the outcome of the boot's networking: DHCP leases and52	// static assignments alike. It makes queryable the same facts that53	// init prints to the console.54	Network NetworkStatus `json:"network,omitzero"`5556	// Time reports the state of this machine's clock. It makes57	// queryable the same facts that the time loop prints to the58	// console. Unlike most of status, Time changes for the whole life59	// of the machine, because the clock is disciplined continuously,60	// not configured once at boot.61	Time TimeStatus `json:"time,omitzero"`6263	// Hardware is what the machine found when it started running.64	Hardware HardwareStatus `json:"hardware,omitzero"`6566	// Firmware is the machine's standing firmware state: the mode it67	// boots in, and the boot menu from its non-volatile store. On68	// UEFI machines, the variables are decoded into words. BIOS is69	// the name for any machine without firmware variables to consult,70	// such as a legacy server or a direct-kernel boot under a71	// hypervisor. Firmware sits beside Hardware because it describes72	// the machine, not the boot: BootOrder is a standing preference,73	// and BootNext is about the next boot. Only BootCurrent is a fact74	// about this boot, and it stays here so the firmware fields read75	// together. Contrast this with Boot, the per-boot record below.76	Firmware FirmwareStatus `json:"firmware,omitzero"`7778	// Storage reports where each storage role is actually backed this79	// boot, whether declared or not. The spec says what was80	// requested, and hardware.blockDevices says what is attached.81	// Storage connects the two.82	Storage StorageStatus `json:"storage,omitzero"`8384	// Sysctls echoes the observed value of every parameter liken sets,85	// read back from /proc/sys: the settings every liken machine holds86	// (machine.OSSysctls) and the ones spec.sysctls declares. This87	// puts what was asked for and what the kernel holds side by side88	// in one kubectl get.89	//90	// A parameter liken could not write is absent here, because a91	// failed write is never read back. That is what makes this map the92	// list of parameters that currently hold, rather than the list93	// somebody wanted.94	Sysctls map[string]string `json:"sysctls,omitempty"`9596	// HostEntries echoes the entries /etc/hosts actually holds,97	// observed on the pass that published them: the live view of98	// spec.network.hostEntries, the way Sysctls above is the live99	// view of spec.sysctls. Init writes the file at boot, and the100	// machine operator reconciles it live afterward, so this field101	// changes within one reconcile pass of a spec edit, with no102	// reboot. There is no boot record for it, for the same reason103	// there is none for Sysctls: a file the operator may rewrite on104	// any pass is not one boot's fact.105	HostEntries []HostEntry `json:"hostEntries,omitempty"`106107	// Rlimits echoes the resource limits init holds, read back from108	// the kernel: the limits every liken machine holds109	// (machine.OSRlimits) and the ones spec.rlimits declares. Init is110	// PID 1, so these are the limits every process on the machine111	// inherits, k3s and containerd and the containers below them.112	// Reading /proc/1/limits on the machine gives the same answer in113	// the kernel's own layout.114	//115	// Each value is in the spec's syntax: a number when both halves116	// agree, "infinity" for no limit, and "soft:hard" when the halves117	// differ. So this map and spec.rlimits can be compared without118	// translating either.119	//120	// A limit liken could not set is absent here, because a failed121	// write is never read back. That is what makes this map the list122	// of limits that hold, rather than the list somebody wanted.123	Rlimits map[string]string `json:"rlimits,omitempty"`124125	// Modules reports the outcome of every module named in126	// spec.modules. It makes queryable the same verdicts that init127	// prints to the console. Only the declared extras appear here.128	// The fixed list of modules the OS loads by itself belongs to the129	// image, not to the spec.130	Modules []ModuleStatus `json:"modules,omitempty"`131132	// Serio reports each attachment init holds or tried to hold for133	// spec.serio, and the device nodes each attached driver created.134	// Unlike Modules, it changes while the machine runs: an adapter135	// that is unplugged goes Missing, and it goes Attached again136	// when it is plugged back in.137	Serio []SerioStatus `json:"serio,omitempty"`138139	// Features reports this machine's standing on every feature that140	// the cluster document enables, in the Cluster's spec.features.141	// It makes queryable the same verdicts that init prints at boot.142	// The Cluster declares features for the whole fleet, but this143	// field is the per-machine answer, because honoring a feature144	// depends on what the booted image carries. Machines can run145	// different releases in the middle of a rollout.146	Features []FeatureStatus `json:"features,omitempty"`147148	// Registries reports what this machine rendered into k3s's149	// registries.yaml: which registries are mirrored, which have150	// credentials, and whether the embedded registry is on. It151	// reports the hosts only, never the credential material. This152	// gives console parity, like Features above.153	Registries RegistriesStatus `json:"registries,omitzero"`154155	// Runtime reports the Go runtime environment init actually handed156	// the k3s process this boot: the resolved values, not the spec's157	// strings. It gives console parity for spec.runtime.k3s, so an158	// operator who edits the ceiling can read back what each machine159	// runs. This is what makes the values visible while an experiment160	// runs across a fleet of different sizes.161	Runtime RuntimeStatus `json:"runtime,omitzero"`162163	// Boot is what this boot ran under: which documents, and the164	// storage as actuated. Each boot re-derives it, and it is the165	// record the operator compares against the spec. Lifetime is what166	// separates Boot from Firmware. If you power-cut the machine and167	// boot it unchanged, everything in Boot is freshly re-derived (a168	// staged manifest applies, or a rejection lands), while Firmware169	// reports the standing state that carried across the reboot.170	Boot BootStatus `json:"boot,omitzero"`171172	// LastCrash is the newest kernel crash this machine still holds173	// records for: a panic or an oops that pstore carried across the174	// reboot. It is not necessarily the previous boot's crash. A175	// machine can boot cleanly for years while an old crash stays on176	// record, and the timestamp is what says how old the news is.177	// Every boot re-derives this field from the preserved records on178	// machineState, so an erased status rebuilds it, and it clears179	// only when the records themselves leave the retention window.180	LastCrash *CrashStatus `json:"lastCrash,omitempty"`181182	// LastFailStop is the last boot this machine refused to run, and183	// the reason init gave for the refusal (failstop.go). It is not184	// necessarily the previous boot: a machine that refuses once,185	// gets repaired, and then runs for months keeps reporting the186	// refusal, and the timestamp is what says how old the news is.187	// Every boot re-derives this field from the record on188	// machineState, so an erased status rebuilds it. Nothing clears189	// it; the next refusal overwrites it.190	LastFailStop *FailStop `json:"lastFailStop,omitempty"`191192	// Conditions follow the standard Kubernetes pattern: a set of193	// typed, timestamped observations, such as "Ready" and194	// "SysctlsApplied", that controllers maintain and that humans and195	// tooling read.196	Conditions []api.Condition `json:"conditions,omitempty"`197198	// Pending lists what this machine waits to apply: each staged199	// document that needs a disruption, with the hash a person200	// approves. The conditions carry the same facts in their201	// messages, but a message is prose. This field exists so that202	// nothing has to parse a condition message to find a hash, and203	// so the question "what is this machine waiting for me to204	// allow" has a field that answers it.205	Pending []PendingDisruption `json:"pending,omitempty"`206}207208// VersionStatus is the complete inventory of what this machine209// runs: liken's own version, and every outside component the OS210// carries. Two kinds of fact live here, sourced in different ways.211// For the kernel and the netfilter userspace, the running machine212// itself reports a version (`uname` and `iptables -V`), so those213// fields are observed, in the running software's own vocabulary. The214// rest — boot artifacts, bundled images, data files — cannot report215// anything about themselves. VersionStatus reports those from the216// components record that the image build wrote alongside the bytes217// it staged (/usr/share/liken/components.yaml). This is the same218// record of pins that the release document publishes, so the two can219// never disagree. This block lists k3s even though the Node object220// already reports a kubelet version, because this block answers a221// different question: not "what is running the pods" but "what did222// this OS image carry".223type VersionStatus struct {224	Liken  string `json:"liken,omitempty"`225	Kernel string `json:"kernel,omitempty"`226227	// Xtables is the netfilter userspace version as it reports228	// itself, for example "v1.8.11 (legacy)" from `iptables -V`. It229	// is observed, not echoed from a build pin.230	Xtables string `json:"xtables,omitempty"`231232	K3s       string `json:"k3s,omitempty"`233	Trust     string `json:"trust,omitempty"`234	E2fsprogs string `json:"e2fsprogs,omitempty"`235	OpenISCSI string `json:"openIscsi,omitempty"`236	NFSUtils  string `json:"nfsUtils,omitempty"`237238	// WPASupplicant is the hostap release that supplied the239	// supplicant init runs for a wireless spec.network entry.240	WPASupplicant string `json:"wpaSupplicant,omitempty"`241242	SystemdBoot   string `json:"systemdBoot,omitempty"`243	Grub          string `json:"grub,omitempty"`244	Hwdata        string `json:"hwdata,omitempty"`245	Tzdata        string `json:"tzdata,omitempty"`246	LinuxFirmware string `json:"linuxFirmware,omitempty"`247248	// WirelessRegdb is the regulatory database under /lib/firmware.249	// Its pin lives inside linux-firmware/fetch.sh rather than in a250	// domain of its own, but the channels a radio may use depend on251	// it, so it reports here like any other component.252	WirelessRegdb string `json:"wirelessRegdb,omitempty"`253254	// Microcode is the pin: which early cpio the release carries.255	// MicrocodeRevision is observed from the running CPUs. The two256	// agreeing is the proof that the early cpio applied, and only257	// real hardware can give it; on a virtual machine the revision is258	// the hypervisor's.259	Microcode         string `json:"microcode,omitempty"`260	MicrocodeRevision string `json:"microcodeRevision,omitempty"`261}262263// NetworkStatus reports how the boot attached this machine to the264// network. The top-level fields summarize the primary interface: the265// cluster-facing interface when the Cluster's nodeCIDR identifies it,266// or otherwise the first interface that came up. Interfaces carries267// the full detail for each interface, for a machine with more than268// one.269type NetworkStatus struct {270	Interface    string     `json:"interface,omitempty"`271	MAC          string     `json:"mac,omitempty"`272	Addresses    []string   `json:"addresses,omitempty"`273	Gateway      string     `json:"gateway,omitempty"`274	Nameservers  []string   `json:"nameservers,omitempty"`275	LeaseExpires *time.Time `json:"leaseExpires,omitempty"`276277	Interfaces []InterfaceStatus `json:"interfaces,omitempty"`278}279280// AddressMethod is how an interface got its address: from a DHCP281// lease, or from a static assignment in the Machine spec.282type AddressMethod string283284const (285	MethodDHCP   AddressMethod = "DHCP"286	MethodStatic AddressMethod = "Static"287)288289// InterfaceStatus is one interface, as the boot configured it.290type InterfaceStatus struct {291	Name         string        `json:"name"`292	MAC          string        `json:"mac,omitempty"`293	Address      string        `json:"address,omitempty"`294	Method       AddressMethod `json:"method,omitempty"`295	Gateway      string        `json:"gateway,omitempty"`296	Nameservers  []string      `json:"nameservers,omitempty"`297	LeaseExpires *time.Time    `json:"leaseExpires,omitempty"`298299	// Wireless is the radio's own standing, present only for an300	// interface the spec gave a wireless entry. The addressing301	// fields above describe the same interface after the join, so an302	// interface can carry a wireless failure and no address at all.303	Wireless *WirelessStatus `json:"wireless,omitempty"`304}305306// WirelessState is what a radio is doing at the moment the machine307// reports it. Distinct words exist because the kernel alone cannot308// tell them apart: it reports the same missing carrier for a wrong309// passphrase and for an access point that is switched off. Only the310// supplicant's own events separate WrongKey from NoCarrier, and311// WrongKey is the one value that no amount of waiting corrects.312// NotRaised is the exception that comes from init rather than the313// supplicant: the kernel call that raises the interface never314// returned, so no supplicant ever ran on it.315type WirelessState string316317const (318	WirelessAssociating WirelessState = "Associating"319	WirelessConnected   WirelessState = "Connected"320	WirelessWrongKey    WirelessState = "WrongKey"321	WirelessNoCarrier   WirelessState = "NoCarrier"322	// WirelessNotRaised reports a raise that did not return within323	// init's deadline. A driver that deadlocks in the kernel while324	// powering the radio produces it; the thread cannot be stopped,325	// so this state is a report about the driver, not the network,326	// and only a reboot without the radio in the spec clears it.327	WirelessNotRaised WirelessState = "NotRaised"328)329330// WirelessStatus is the wireless half of one interface's status: the331// network the spec named, what the join did, and the reason behind a332// join that failed. Everything init prints about the radio also333// lands here, because a machine with no shell cannot be asked to334// repeat itself.335type WirelessStatus struct {336	SSID    string        `json:"ssid,omitempty"`337	State   WirelessState `json:"state,omitempty"`338	Message string        `json:"message,omitempty"`339}340341// TimeState is a machine clock's condition. FreeRunning and342// Unsynchronized both mean the clock does not follow any source, but343// for different reasons. A free-running machine never received any344// sources and runs on its hardware clock by design. An unsynchronized345// machine has sources, but currently cannot reach them. The346// distinction matters when you decide whether a fleet listing shows a347// configuration choice or an outage.348type TimeState string349350const (351	TimeSynchronized   TimeState = "Synchronized"352	TimeFreeRunning    TimeState = "FreeRunning"353	TimeUnsynchronized TimeState = "Unsynchronized"354)355356// TimeStatus reports the state of the machine's clock. A fleet with357// no upstreams free-runs and agrees with itself, but still does not358// report Synchronized. Machines that agree with each other make a359// different claim than machines that agree with the rest of the360// world, and certificate validation cares about that difference.361type TimeStatus struct {362	// State reports whether the machine currently disciplines the363	// clock against a source that is itself synchronized. When it364	// does not, State also reports whether that is by design365	// (FreeRunning) or by outage (Unsynchronized).366	State TimeState `json:"state,omitempty"`367368	// Source is who this machine follows: an upstream's name on a369	// leader, or one of the cluster's leaders on a follower.370	Source string `json:"source,omitempty"`371372	// Stratum is the machine's distance from a reference clock, in373	// NTP's own vocabulary: a source at stratum n makes this machine374	// stratum n+1. A leader that free-runs by design reports the375	// local-clock convention (10). A value of 16 means unsynchronized,376	// the value NTP reserves for a clock that no machine should use377	// as a source.378	Stratum int `json:"stratum,omitempty"`379380	// Offset is the clock error that init last published, as a381	// human-readable duration such as "1.28ms". It is positive when382	// this machine was behind its source. Init measures the clock every383	// 64 seconds and publishes a new offset only when the measured one384	// moves 25 ms or more from this value, so the clock error now is385	// within 25 ms of it.386	Offset string `json:"offset,omitempty"`387}388389type HardwareStatus struct {390	CPUs        int    `json:"cpus,omitempty"`391	MemoryBytes uint64 `json:"memoryBytes,omitempty"`392393	// BlockDevices is the machine's storage inventory: every real disk394	// the kernel found, whether or not the spec says anything about395	// it. An attached but undeclared disk shows up here. This is how396	// an operator notices one.397	BlockDevices []BlockDevice `json:"blockDevices,omitempty"`398399	// Unclaimed lists every device the kernel enumerated but that400	// nothing drives: hardware that waits on a module that401	// spec.modules does not declare. It is the gap, never the full402	// count. A machine whose hardware is fully driven reports nothing403	// here, the same way healthy conditions read True and a healthy404	// fleet listing looks uneventful. The full inventory of working405	// devices is deliberately not status material. Workloads reach it406	// through /sys, and claimable devices belong to ResourceSlices.407	Unclaimed []UnclaimedDevice `json:"unclaimed,omitempty"`408}409410// UnclaimedDevice is one enumerated but undriven device, reported411// with everything an operator needs to fix it: the device named in412// words, and the candidate modules whose alias patterns match its413// fingerprint. Only a device that some loadable module could drive414// appears here at all. A device the kernel build has no module for,415// such as a host bridge or a platform stub, is not actionable, and416// reporting it would hide the fixable gaps among noise.417type UnclaimedDevice struct {418	// Modalias is the kernel's fingerprint for this device, the same419	// string it announces in uevents and matches driver patterns420	// against. It identifies the device precisely, in cases where the421	// words above it do not.422	Modalias string `json:"modalias"`423424	// Bus is where the device lives: pci or usb.425	Bus string `json:"bus"`426427	// Name is the device in words: a USB device's own manufacturer428	// and product strings, or a PCI device's names from the pci.ids429	// database. It falls back to numeric vendor:device IDs when no430	// better name exists.431	Name string `json:"name,omitempty"`432433	// Class is the device's coarse kind, such as mass-storage,434	// display, or network, decoded from the bus's class code.435	Class string `json:"class,omitempty"`436437	// Candidates are the loadable modules whose alias patterns match438	// this device, in the kernel build's preference order. More than439	// one candidate is normal; USB storage matches both uas and440	// usb_storage. The choice belongs to whoever edits spec.modules.441	Candidates []string `json:"candidates,omitempty"`442443	// Message says what would fix the gap, like every message in this444	// status: declare a candidate when the image carries one, or get445	// an image that carries one when none is present.446	Message string `json:"message,omitempty"`447}448449// BlockDevice is one disk, as the machine observed it directly from450// sysfs. Name is the kernel's name for this boot, such as vda or451// nvme0n1, assigned in driver probe order. It addresses the device452// within this boot, but it does not identify the device across453// boots. StableNames carries the names that do. Model and serial454// come from the device itself.455type BlockDevice struct {456	Name      string `json:"name"`457	SizeBytes uint64 `json:"sizeBytes,omitempty"`458	Model     string `json:"model,omitempty"`459	Serial    string `json:"serial,omitempty"`460461	// StableNames is every name this disk answers to across boots:462	// each by-id name, built from a value the disk's own controller463	// reports, then the by-path name, built from the port the disk464	// sits on, when the disk has one. spec.storage.<role>.device465	// accepts any name in this list. The first by-id entry is the one466	// to prefer, because a by-id name follows the disk itself, while467	// a by-path name follows the port and stops naming the disk the468	// moment it moves to a different bay. A disk with no by-id name469	// carries only a by-path name, so the first entry overall is not470	// always the one to prefer.471	StableNames []string `json:"stableNames,omitempty"`472}473474// Backing is where a storage role's data actually lives. There are475// exactly two options: a partition claimed for the role, or the476// machine's RAM root, which is the default and needs no setup.477type Backing string478479const (480	BackingPartition Backing = "Partition"481	BackingMemory    Backing = "Memory"482)483484// FirmwareMode is which kind of firmware booted the machine. Like485// every closed vocabulary in this API, it is a named string type, so486// the compiler can catch a value used in the wrong place.487type FirmwareMode string488489const (490	FirmwareUEFI FirmwareMode = "UEFI"491	FirmwareBIOS FirmwareMode = "BIOS"492)493494// FirmwareStatus reports the firmware's boot configuration, read from495// its variable store. Each entry field renders as the variable's own496// name plus the entry's decoded description, such as "Boot0001497// (liken slot A)", because a fleet listing should read in words.498type FirmwareStatus struct {499	Mode FirmwareMode `json:"mode,omitempty"`500501	// BootCurrent is the entry the firmware reports it used this502	// boot. It is empty when the firmware never picked one, as in a503	// direct-kernel boot, or when the machine is not UEFI at all.504	BootCurrent string `json:"bootCurrent,omitempty"`505506	// BootNext, when present, is a one-shot override armed for the507	// next boot. The firmware consumes it at power-on. Seeing a value508	// here means a proving boot is queued but has not yet happened.509	BootNext string `json:"bootNext,omitempty"`510511	// BootOrder is the firmware's standing preference list, with the512	// first choice listed first.513	BootOrder []string `json:"bootOrder,omitempty"`514}515516// StorageStatus lists every role liken defines, whether declared or517// not. An absent role must be visible in one kubectl get, not merely518// implied. The fields mirror the spec's keys exactly, so spec and519// status line up name for name.520type StorageStatus struct {521	BIOSBoot         StorageRoleStatus `json:"biosBoot"`522	BootHome         StorageRoleStatus `json:"bootHome"`523	SystemA          StorageRoleStatus `json:"systemA"`524	SystemB          StorageRoleStatus `json:"systemB"`525	MachineState     StorageRoleStatus `json:"machineState"`526	MachineEphemeral StorageRoleStatus `json:"machineEphemeral"`527	ClusterState     StorageRoleStatus `json:"clusterState"`528	PodStorage       StorageRoleStatus `json:"podStorage"`529	PodEphemeral     StorageRoleStatus `json:"podEphemeral"`530}531532// StorageRoleStatus is where one role is backed. A memory-backed role533// deliberately reports no capacity. All memory-backed roles share the534// one RAM root, and a per-role figure would count that root several535// times over.536type StorageRoleStatus struct {537	Backing       Backing `json:"backing"`538	Device        string  `json:"device,omitempty"`    // the partition's node this boot: vda1539	Partition     string  `json:"partition,omitempty"` // its on-disk name: liken:clusterState540	CapacityBytes uint64  `json:"capacityBytes,omitempty"`541542	// LastStopUnclean says this boot found the role's filesystem543	// still marked as mounted, which means the machine's previous544	// stop did not release it. Only the FAT32 roles report it: the545	// system slots and the boot home carry a mark for exactly this,546	// and the ext4 roles replay a journal instead and need no547	// warning. The field describes the stop before this boot, so it548	// keeps its value for the life of the boot.549	LastStopUnclean bool `json:"lastStopUnclean,omitempty"`550}551552// Role addresses one role's status by its spec name. It returns nil553// for a name outside the vocabulary.554func (s *StorageStatus) Role(name StorageRoleName) *StorageRoleStatus {555	switch name {556	case BIOSBootRole:557		return &s.BIOSBoot558	case BootHomeRole:559		return &s.BootHome560	case SystemARole:561		return &s.SystemA562	case SystemBRole:563		return &s.SystemB564	case MachineStateRole:565		return &s.MachineState566	case MachineEphemeralRole:567		return &s.MachineEphemeral568	case ClusterStateRole:569		return &s.ClusterState570	case PodStorageRole:571		return &s.PodStorage572	case PodEphemeralRole:573		return &s.PodEphemeral574	}575	return nil576}577578// AllRolesInMemory marks every role as backed by the RAM root. This579// is the accurate starting point. Reconciliation upgrades each role,580// one at a time, as it places the role on a partition.581func AllRolesInMemory() StorageStatus {582	s := StorageStatus{}583	for _, name := range StorageRoleNames {584		s.Role(name).Backing = BackingMemory585	}586	return s587}588589// ModuleState is one declared module's outcome. Like every state word590// in this API, it is a closed vocabulary. Two of the four values are591// healthy. Loaded means the kernel took the module, or already had592// it. Builtin means the kernel compiles the name in, so there was593// nothing to load and nothing wrong. Missing means this kernel has no594// module by that name at all. The image carries the kernel's whole595// module tree, so the usual cause is a misspelling, and the fix is596// the name rather than a new image. A name that is spelled right and597// still missing needs a release whose kernel builds it. Failed means598// the kernel has the module and refused it, which usually points to a599// problem in the hardware.600type ModuleState string601602const (603	ModuleLoaded  ModuleState = "Loaded"604	ModuleBuiltin ModuleState = "Builtin"605	ModuleMissing ModuleState = "Missing"606	ModuleFailed  ModuleState = "Failed"607)608609// ModuleStatus is one declared module's outcome, from the boot that610// loaded it or from the load that followed a later edit. Message611// carries the detail for the unhealthy states, phrased to name the612// fix. A status that names the repair is more useful than one that613// only names the problem.614type ModuleStatus struct {615	Name    string      `json:"name"`616	State   ModuleState `json:"state"`617	Message string      `json:"message,omitempty"`618619	// Parameters is what the machine read back from620	// /sys/module/<name>/parameters/ for the declared parameter621	// names, keyed by parameter name, verbatim. A declared name that622	// is absent here usually means the kernel ignored a wrong name,623	// which it says once in its log as "unknown parameter ...624	// ignored". It can also mean the parameter is real but offers625	// nothing to read: a driver may register a parameter with no626	// sysfs file at all, or with a write-only mode. The values are627	// not compared against the declaration, because the kernel628	// prints a bool back as Y or N and an array with its own629	// separators; a person compares the two fields that sit beside630	// each other.631	Parameters map[string]string `json:"parameters,omitempty"`632633	// AlreadyResident reports that the module was in the kernel634	// before the declared pass reached it, loaded by the fixed list,635	// a feature, or an earlier module's dependency chain. It matters636	// because a resident module never receives a parameter string:637	// finit_module returns EEXIST and ignores it, so the residency638	// has to be read before the load to be reported at all.639	AlreadyResident bool `json:"alreadyResident,omitempty"`640}641642// FeatureState is one enabled feature's standing on one machine, a643// closed vocabulary like ModuleState's. Active means this boot could644// honor everything the feature asks of this machine. Missing means645// the booted image predates the feature. The cluster document646// declares the feature, but the image carries no payload for it, so647// the fix is a release that does. Failed means the image carries the648// payload, but actuating it went wrong, for example a module the649// kernel refused or a boot hook that returned an error; the message650// explains what happened.651type FeatureState string652653const (654	FeatureActive  FeatureState = "Active"655	FeatureMissing FeatureState = "Missing"656	FeatureFailed  FeatureState = "Failed"657)658659// FeatureStatus is one enabled feature's outcome on this machine, on660// this boot. Like ModuleStatus, Message carries detail for the661// unhealthy states, phrased to name the fix.662type FeatureStatus struct {663	Name    string       `json:"name"`664	State   FeatureState `json:"state"`665	Message string       `json:"message,omitempty"`666}667668// RuntimeStatus reports the runtime discipline this machine actually669// runs under. It mirrors spec.runtime subsection for subsection, and670// holds the resolved values rather than the spec's strings.671type RuntimeStatus struct {672	K3s        K3sRuntimeStatus        `json:"k3s,omitzero"`673	Kubelet    KubeletRuntimeStatus    `json:"kubelet,omitzero"`674	Containerd ContainerdRuntimeStatus `json:"containerd,omitzero"`675}676677// K3sRuntimeStatus is the discipline init imposed on the k3s process678// this boot. GoMemoryLimit is the resolved ceiling as an absolute679// quantity in MiB, for example "256Mi"; it is empty when the cluster680// left the ceiling unset or turned it off, so the field's absence reads681// as "no ceiling, Go's own default". GoGC is the resolved collector682// pace; it is empty when the cluster set none, so its absence reads as683// "Go's own pace". Debug is true when init rendered k3s's debug key,684// so its absence reads as "k3s logs at info".685type K3sRuntimeStatus struct {686	GoMemoryLimit string `json:"goMemoryLimit,omitempty"`687	GoGC          int    `json:"goGC,omitempty"`688	Debug         bool   `json:"debug,omitempty"`689}690691// ContainerdRuntimeStatus is the containerd configuration init wrote on692// this machine's boot. LogLevel is the level the drop-in gives693// containerd; it is empty when the cluster named none, which reads as694// containerd's own default of info. This is the console parity for695// spec.runtime.containerd, so an operator who turns a fleet down can696// read back which machines carry the change.697type ContainerdRuntimeStatus struct {698	LogLevel string `json:"logLevel,omitempty"`699}700701// KubeletRuntimeStatus is the kubelet configuration init rendered on702// this machine's boot. It is absent when the cluster named none, which703// reads as "the kubelet's own defaults".704type KubeletRuntimeStatus struct {705	ImageGC ImageGCStatus `json:"imageGC,omitzero"`706}707708// ImageGCStatus is the image collection policy init wrote into the709// kubelet's configuration file. Each field is present only when the710// cluster named it, so an absent field reads as the kubelet's own711// default for that field: 85 percent, 80 percent, two minutes, and no712// age ceiling at all. This is the console parity for713// spec.runtime.kubelet.imageGC, so an operator who tunes the policy714// can read back which machines carry it.715type ImageGCStatus struct {716	HighThresholdPercent int    `json:"highThresholdPercent,omitempty"`717	LowThresholdPercent  int    `json:"lowThresholdPercent,omitempty"`718	MaximumAge           string `json:"maximumAge,omitempty"`719	MinimumAge           string `json:"minimumAge,omitempty"`720}721722// ManifestSource is which copy of a document a boot ran under, in723// preference order: a staged manifest awaiting its proving boot, the724// proven last-known-good copy, or the image's seed, for the first725// boot only (see staging.go).726type ManifestSource string727728const (729	ManifestSourceStaged ManifestSource = "Staged"730	ManifestSourceProven ManifestSource = "Proven"731	ManifestSourceSeed   ManifestSource = "Seed"732)733734// BootStatus records the configuration this boot actually used: which735// copy of each document, identified by the hashes of their exact736// bytes, and the storage spec it actuated. This is the half of drift737// detection that only init can supply. The operator compares it738// against the cluster's copies.739type BootStatus struct {740	// Time is the moment the machine booted, derived by init from the741	// kernel's uptime counter. It is a timestamp rather than a742	// duration, because a timestamp never goes out of date in the743	// cluster. `kubectl get machines` renders it as a live elapsed744	// time in the Uptime column. It belongs to the boot record because745	// it shares the record's lifetime: a reboot moves it, an in-place746	// k3s restart does not.747	//748	// This status deliberately has no heartbeat. The machine's749	// liveness signal is a Lease in the liken-system namespace, not a750	// status field, because a heartbeat must renew forever, and every751	// status write rewrites this whole object and wakes every752	// watcher. This is the same reason kube-node-lease exists (see the753	// kubernetes package). The cluster operator reads the leases and754	// marks a silent machine Lost.755	Time *time.Time `json:"time,omitempty"`756757	// The Machine manifest this boot ran under.758	ManifestSource ManifestSource `json:"manifestSource,omitempty"`759	ManifestHash   string         `json:"manifestHash,omitempty"`760	Storage        StorageSpec    `json:"storage,omitzero"`761762	// Network is the network spec the winning manifest declared,763	// recorded as actuated whatever each interface's outcome was. It764	// is the drift reference, like Storage above: booting again under765	// this manifest would ask for the same network, and status.network766	// reports what each interface actually got.767	//768	// The field is a pointer because absence and emptiness say769	// different things here. An empty spec is a machine that declares770	// no interface and takes the zero-configuration default. An absent771	// record is a boot that reported nothing about its network at all,772	// which the operator cannot judge: comparing a real spec against773	// nothing would read as drift on every machine at once and ask a774	// whole fleet to reboot on the strength of a missing file.775	Network *NetworkSpec `json:"network,omitempty"`776777	// Modules is the module list the winning manifest declared,778	// recorded as actuated regardless of each load's outcome. It is779	// the drift reference, like Storage above. Outcomes are a health780	// signal and live in status.modules instead; a module the image781	// lacked still counts as actuated here, because rebooting again782	// with the same image would not change anything.783	//784	// The list is in the order this machine loaded the modules in,785	// because the order decides which driver claims a device. A live786	// load appends what it loaded and does not adopt the manifest's787	// order (init/liveload.go). The operator compares this order788	// against spec.modules and stages a reorder for the next boot.789	Modules []string `json:"modules,omitempty"`790791	// Rlimits is the resource limit map the winning manifest declared,792	// recorded as actuated whatever each limit's outcome was. It is793	// the drift reference, like Storage and Modules above: booting794	// again under this manifest would ask for the same limits, and795	// status.rlimits reports what the kernel actually holds.796	//797	// The limits every liken machine holds are deliberately not part798	// of this record. They ship with the release rather than with the799	// spec, so a change to them arrives with a new system image and a800	// reboot of its own. Comparing them here would read as drift on801	// every machine at once the moment the table changed.802	Rlimits map[string]string `json:"rlimits,omitempty"`803804	// ModuleParameters is the parameter map the winning manifest805	// declared and this boot passed to the loads, the drift806	// reference for spec.moduleParameters the same way Modules is807	// for spec.modules. It records the request; the readback lives808	// in status.modules[].parameters.809	ModuleParameters map[string]string `json:"moduleParameters,omitempty"`810811	// Serio is the spec.serio list this boot declared, the drift812	// reference for that field the same way Modules is for813	// spec.modules. It records the request whatever each attachment's814	// outcome was; status.serio reports the outcomes. A live load815	// adds the entries it applied.816	Serio []SerioAttachment `json:"serio,omitempty"`817818	// Slot is the system slot this boot came from, "A" or "B", read819	// from the liken.slot= parameter in each boot entry's command820	// line. It is empty when the boot did not come from a slot at821	// all, as in a direct-kernel boot or install media. This is how a822	// machine reports which side of the blue-green pair it runs from:823	// releases download to the other slot.824	Slot string `json:"slot,omitempty"`825826	// CommandLine is the kernel command line this boot ran with, whole827	// and unparsed. Slot above is one parameter out of it, and init828	// reads a handful of others, but the line as a whole is the only829	// input a machine has before it reads a single file, and every830	// other field here describes what happened after that point.831	//832	// It is reported and not declared. No spec field sets it: the boot833	// entries liken writes name the console, the machine, and the834	// slot, and nothing else. So this field answers the question a835	// person asks when a machine behaves unlike its siblings, which is836	// what the firmware actually handed the kernel. Reading it needed837	// a serial console before it was here.838	CommandLine string `json:"commandLine,omitempty"`839840	// The Cluster manifest this boot ran under: the same lifecycle,841	// recorded separately, because the two documents stage and prove842	// independently. A machine can be current on one document while843	// drifted on the other.844	ClusterManifestSource ManifestSource `json:"clusterManifestSource,omitempty"`845	ClusterManifestHash   string         `json:"clusterManifestHash,omitempty"`846847	// The registry-credentials document this boot, or the latest k3s848	// restart, rendered into registries.yaml: the same lifecycle849	// again, in its own store. The source is only ever Staged or850	// Proven, because the operator is this document's sole author, so851	// no image carries a seed. Both fields are empty on a machine852	// that has never had credentials, which is an ordinary state, not853	// a gap.854	CredentialsSource ManifestSource `json:"credentialsSource,omitempty"`855	CredentialsHash   string         `json:"credentialsHash,omitempty"`856857	// The imported-images record this boot ran under (imports.go):858	// Staged while the container store serves unpacks that no859	// operator has yet proven, or Proven on the ordinary boot whose860	// tarballs all match the record. It is empty when the lifecycle861	// is not running, either because there is no durable machineState862	// to remember a trial, or because the container store is863	// ephemeral and a reboot resets it anyway. ImportsDiscarded864	// records that this boot found a trial still standing from a boot865	// that died unproven, and threw the container store away rather866	// than trust it.867	ImportsSource    ManifestSource `json:"importsSource,omitempty"`868	ImportsHash      string         `json:"importsHash,omitempty"`869	ImportsDiscarded bool           `json:"importsDiscarded,omitempty"`870871	// Restarts counts the in-place k3s restarts this boot has872	// performed to apply restart-class changes (cluster/changes.go).873	// It lives in the boot record because it shares the boot's874	// lifetime: a reboot re-makes the record, and the count returns875	// to zero. That reset is itself a signal. A change that arrived876	// by restart increments this count without moving the boot877	// record's time, and a change that arrived by reboot does the878	// opposite.879	Restarts int `json:"restarts,omitempty"`880881	// Rejection is the standing quarantine record for the Machine882	// manifest. Every boot republishes it until a promotion clears883	// it, so a rejected spec stays visible in the cluster no matter884	// how many times the machine power-cycles. ClusterRejection is885	// the same record for the Cluster document. SystemRejection is886	// for a system release whose proving boot fell back.887	// CredentialsRejection is for a credentials document that would888	// not parse.889	Rejection            *Rejection `json:"rejection,omitempty"`890	ClusterRejection     *Rejection `json:"clusterRejection,omitempty"`891	SystemRejection      *Rejection `json:"systemRejection,omitempty"`892	CredentialsRejection *Rejection `json:"credentialsRejection,omitempty"`893}894895// CrashReason is the kernel's own word for why it dumped its log:896// the kmsg-dump reason, taken from the first line of each pstore897// record. Panic and Oops are the two words the vendored kernel's898// configuration dumps. The type stays an open string rather than a899// closed vocabulary, because the kernel owns these words and a900// future kernel may say something new. A status write must not fail901// over an unexpected word.902type CrashReason string903904const (905	CrashPanic CrashReason = "Panic"906	CrashOops  CrashReason = "Oops"907)908909// CrashStatus summarizes one kernel crash from the records that910// pstore preserved across the reboot. It is a stub, not the911// evidence: the kernel log tail around a crash runs to kilobytes,912// and status is read on every list and watch, so the full text913// stays on disk and this block carries just enough to say what914// happened and where to read the rest.915type CrashStatus struct {916	// Time is the machine's own clock at the moment of the crash. A917	// crash usually beats the boot's first clock sync, so this is918	// RTC time, reported as recorded.919	Time *time.Time `json:"time,omitempty"`920921	// Reason is the kernel's word for the crash: Panic or Oops.922	Reason CrashReason `json:"reason,omitempty"`923924	// Message is the kernel's own first description of the failure:925	// the "Kernel panic - not syncing:" line, or the oops's BUG926	// line. It is capped and cleaned of control bytes before it927	// leaves the machine.928	Message string `json:"message,omitempty"`929930	// Records is where the full kernel log tail lives on the931	// machine: a directory under machineState's crash store, or932	// /sys/fs/pstore itself on a machine whose machineState fell933	// back to memory and whose records therefore stay in the934	// firmware's own store.935	Records string `json:"records,omitempty"`936}937938// RebootApprovedCondition is the rollout conductor's grant of a939// reboot turn. It is the one condition type on a Machine's status940// that two different programs use: the cluster operator writes and941// removes it, and the machine's own operator carries it along942// unchanged and acts on it. It belongs here because it is shared943// vocabulary, in the same way that PodScheduled is a condition the944// scheduler writes onto Pods that the kubelet owns.945const RebootApprovedCondition = "RebootApproved"946947// ApproveDisruptionAnnotation grants a Manual machine one948// disruption. A person (or the liken CLI's approve-reboot command)949// sets it on the Machine, valued with the hash of the staged950// document it approves. The hash makes the grant one-shot by951// construction: once the change applies, that hash is no longer952// pending, and the next change hashes differently, so a stale953// annotation approves nothing. Nobody has to consume the grant, and954// that matters for RBAC: clearing an annotation would need patch on955// machines, which would let a per-node operator edit any machine's956// spec. A grant that expires on its own needs no new verb.957const ApproveDisruptionAnnotation = "liken.sh/approve-disruption"958959// ApprovalGrants reports whether an approve-disruption annotation960// approves the staged document with this hash.961func ApprovalGrants(annotation, hash string) bool {962	return annotationNames(annotation, hash)963}964965// RequestRebootAnnotation asks a machine to reboot once, with no966// change to apply. Every other reboot liken performs is the tail of967// a staged document, so a machine whose documents all match its boot968// has no way to reboot: a driver bound to the wrong device, and a969// machine used as a testbed, both need one anyway, and the machine970// has no shell to ask from. A person (or the liken CLI's971// request-reboot command) sets this on the Machine, valued with the972// identity of the boot that is running now (BootID).973//974// Naming the running boot is what makes the request one-shot, the975// same construction the approve-disruption grant uses. The boot that976// comes back has a different identity, so the annotation no longer977// names the running boot, and a later request names that boot instead.978// Nobody has to clear it, which matters for the same reason the979// grant is never consumed: clearing an annotation would need patch980// on machines, and the machine operator holds no such verb.981//982// The request skips none of the coordination. It gates on983// rebootPolicy, waits for the rollout conductor's turn, and drains984// the node, exactly as a staged document's reboot does. What it985// skips is the staged document, and only that.986const RequestRebootAnnotation = "liken.sh/request-reboot"987988// RebootRequestNames reports whether a request-reboot annotation989// names the boot that is running now. A boot that published no time990// has no identity, and no annotation names it.991func RebootRequestNames(annotation, bootID string) bool {992	return bootID != "" && annotationNames(annotation, bootID)993}994995// annotationNames is the matching rule both annotations use. The996// annotation may carry the full hash or a prefix of it, because997// every condition message shows a hash in its 12-character short998// form and a person pastes what they can see. A prefix shorter than999// 12 characters never matches: below that length a paste error could1000// match by accident.1001func annotationNames(annotation, hash string) bool {1002	return len(annotation) >= 12 && strings.HasPrefix(hash, annotation)1003}10041005// BootID is the running boot's identity: the sha256 of the moment1006// the machine booted, which is the one value in the boot record that1007// no two boots share. A request-reboot annotation names it, so a1008// request applies to one boot and to no other.1009//1010// It is a hash rather than the timestamp itself so that it reads and1011// truncates like every other identity liken hands a person: 121012// hexadecimal characters in a condition message, in status.pending,1013// and in an annotation. A boot that published no time has no1014// identity, and this returns "".1015func BootID(boot BootStatus) string {1016	if boot.Time == nil || boot.Time.IsZero() {1017		return ""1018	}1019	return ManifestHash([]byte(boot.Time.UTC().Format(time.RFC3339Nano)))1020}10211022// A DisruptionKind names what applies a staged document: a machine1023// reboot, or a k3s restart in place.1024type DisruptionKind string10251026const (1027	DisruptionReboot  DisruptionKind = "Reboot"1028	DisruptionRestart DisruptionKind = "Restart"1029)10301031// A PendingDisruption is one staged document waiting for its1032// disruption. The machine operator publishes one entry per gated1033// document on every pass and stores nothing between passes, like1034// the rest of status. A reboot a person requested is an entry here1035// too, because it waits on the same policy, the same turn, and the1036// same drain; its document is the one thing it lacks.1037type PendingDisruption struct {1038	// Condition names the condition that reported this entry:1039	// SpecConverged, ClusterConverged, VersionConverged,1040	// CredentialsConverged, or RebootRequestHonored.1041	Condition string `json:"condition"`10421043	// Kind says what applies the staged document. A reboot applies1044	// every staged document at once; a restart applies only the1045	// documents k3s reads when its process starts.1046	Kind DisruptionKind `json:"kind"`10471048	// Hash is the staged document's identity, the value an1049	// approve-disruption annotation names to grant this change its1050	// disruption. A requested reboot has no document, so its entry1051	// carries the running boot's identity instead (BootID), which1052	// spends itself the same way.1053	Hash string `json:"hash"`10541055	// Summary says in one line what the document changes.1056	Summary string `json:"summary"`1057}
machine/storage.go 98.5%
1package machine23// This file holds the storage half of the Machine spec.4//5// Storage is declared by purpose, not by mount path. The manifest6// never says "/var/lib/rancher" (a k3s implementation detail that7// liken owns). Instead it says "this disk holds the cluster's8// state", and liken translates that purpose into the actual path.9// The roles are fields rather than a list, on purpose. Each role is10// a singleton: a machine has one cluster state, and one pod-storage11// pool. Making the roles fields in the schema means a duplicate role12// cannot even be expressed. A new role is visibly a change to the13// API.14//15// The device path in each role matters only on the boot that claims16// the disk. A manifest can name the disk by its kernel name17// (/dev/vda), or by a stable name under /dev/disk/by-id/ or18// /dev/disk/by-path/. The kernel assigns its own device names in19// driver probe order, so a kernel name addresses a disk within one20// boot, but does not identify the disk across boots. A stable name21// survives longer: a by-id name belongs to the disk itself, and a22// by-path name belongs to the port the disk is plugged into. None of23// these names has to survive past the claim, because claiming a disk24// writes the role's name onto the partition itself, as its GPT25// partition name. Every boot after that finds the partition by that26// name, wherever the disk enumerates. A /dev/disk/by-uuid/ name is27// never legal here: that tree names a filesystem, and the disk a28// role claims is blank.29//30// init resolves a stable name to a disk by recomputing every attached31// disk's own by-id and by-path names from sysfs, the same computation32// that publishes the /dev/disk link trees, rather than by reading33// those trees back. The component that publishes the trees starts34// once storage has already settled, so the very first boot that35// claims a disk under a stable name has no tree there yet to read.3637import (38	"fmt"39	"strconv"40	"strings"41)4243// PartitionPrefix namespaces liken's GPT partition names. Anyone who44// looks at a partition table can see which partitions belong to45// liken, and which role each partition serves.46const PartitionPrefix = "liken:"4748// StorageRoleName names one of the storage roles. The vocabulary is49// closed: these nine names are the spec's field names, the GPT50// partition names (behind PartitionPrefix), and the status's keys.51// The names are defined once here, and everything else ranges over52// StorageRoleNames instead of spelling them out again.53type StorageRoleName string5455const (56	BIOSBootRole         StorageRoleName = "biosBoot"57	BootHomeRole         StorageRoleName = "bootHome"58	SystemARole          StorageRoleName = "systemA"59	SystemBRole          StorageRoleName = "systemB"60	MachineStateRole     StorageRoleName = "machineState"61	MachineEphemeralRole StorageRoleName = "machineEphemeral"62	ClusterStateRole     StorageRoleName = "clusterState"63	PodStorageRole       StorageRoleName = "podStorage"64	PodEphemeralRole     StorageRoleName = "podEphemeral"65)6667// StorageRoleNames is the canonical order. It is the order that68// liken lays partitions down when roles share a disk. The order is69// fixed here, rather than by YAML map order, because Kubernetes does70// not preserve YAML map order.71//72// The boot roles lead the list, with the earliest reader first. BIOS73// firmware executes the MBR before anything else exists, so the74// partition that its boot code jumps into comes first. GRUB's own75// config home comes next. The system slots come after that: an EFI76// system partition conventionally leads its disk, and here it leads77// everything that the firmware does not read. machineState comes78// next, ahead of all the data roles. It holds the partition that a79// future boot must find, before that boot has read any spec.80//81// liken recognizes a role by its partition name, never by its82// position in this order. The order is a layout convention, not a83// way to discover roles.84var StorageRoleNames = []StorageRoleName{85	BIOSBootRole,86	BootHomeRole,87	SystemARole,88	SystemBRole,89	MachineStateRole,90	MachineEphemeralRole,91	ClusterStateRole,92	PodStorageRole,93	PodEphemeralRole,94}9596type StorageSpec struct {97	// BIOSBoot and BootHome are how a machine declares that it boots98	// through GRUB rather than UEFI firmware. There is no separate99	// firmware field. Declaring the partitions that GRUB needs is100	// itself the declaration. A BIOS machine has no boot variables to101	// hold the blue-green bookkeeping, so liken supplies the pieces102	// that the firmware would otherwise provide. BIOSBoot is a tiny103	// raw partition (about 1Mi, with no filesystem) that holds GRUB's104	// core image, the code that the MBR's 440 boot bytes jump into.105	// BootHome is a small FAT32 partition (about 64Mi) that holds106	// GRUB's config and its environment block, the file that holds107	// the same role as BootNext and BootOrder. Machines that boot108	// UEFI leave both fields out.109	BIOSBoot *StorageRole `json:"biosBoot,omitempty"`110	BootHome *StorageRole `json:"bootHome,omitempty"`111112	// SystemA and SystemB are the operating system's own boot slots.113	// Each slot holds one complete liken version: the kernel and the114	// initramfs that makes up the rest of the OS. The machine runs115	// from one slot while it writes upgrades to the other slot. This116	// is a blue-green deployment at the scale of the whole operating117	// system. The slots are EFI system partitions that carry FAT32,118	// because the firmware itself is their first reader, and FAT is119	// the only filesystem that the firmware promises to understand. A120	// machine without these slots boots from external media forever,121	// for example the lab's -kernel flag or a USB stick. A machine122	// with these slots can be installed once, and upgraded123	// declaratively after that.124	SystemA *StorageRole `json:"systemA,omitempty"`125	SystemB *StorageRole `json:"systemB,omitempty"`126127	// MachineState is the machine's own durable data: the manifests128	// that configure the machine. It holds the staged and proven129	// copies that let a spec edit survive a reboot, and apply at the130	// next reboot. The data is small, but it changes the machine in a131	// fundamental way. With this data, the configuration outlives the132	// image that first delivered it.133	MachineState *StorageRole `json:"machineState,omitempty"`134135	// MachineEphemeral is the operating system's own scratch space:136	// /tmp. It is small, but necessary, because the container runtime137	// stages exec sessions there. On a machine whose root filesystem138	// is RAM, moving /tmp to disk frees that memory for pods.139	MachineEphemeral *StorageRole `json:"machineEphemeral,omitempty"`140141	// ClusterState is k3s's state: the etcd or sqlite database, TLS142	// material, and containerd's images. Persisting this state is143	// what lets a reboot resume the same cluster, instead of starting144	// a new one. This role is named for the cluster, not the machine,145	// because this data belongs to the cluster rather than to any one146	// machine.147	ClusterState *StorageRole `json:"clusterState,omitempty"`148149	// PodStorage is durable storage that pods claim by name: the150	// PersistentVolumeClaim pool. k3s's local-path provisioner serves151	// this pool from this role's filesystem.152	PodStorage *StorageRole `json:"podStorage,omitempty"`153154	// PodEphemeral is kubelet's working space: emptyDir volumes and155	// per-pod scratch space. It is the pool that pods measure with156	// ephemeral-storage requests and limits. The pod logs live here157	// too, bound onto /var/log/pods, because a log that appends for158	// as long as its pod runs must not be written to the root159	// filesystem's small write budget.160	PodEphemeral *StorageRole `json:"podEphemeral,omitempty"`161}162163// StorageRole places one role onto hardware. A role that is missing164// from the spec is not an error. That role's directory simply stays165// on the machine's RAM root.166type StorageRole struct {167	// Device is the disk that this role lives on: a device path168	// (/dev/vda), or a stable name under /dev/disk/by-id/ or169	// /dev/disk/by-path/. The code reads this field only when it170	// claims a blank disk. A by-id name belongs to the disk and171	// survives a port move. A by-path name belongs to the port. After172	// the claim, every boot finds the role's partition by the GPT173	// name written on it, so this field never names a filesystem, and174	// a /dev/disk/by-uuid/ name is refused.175	Device string `json:"device"`176177	// Size is how much of the device this role takes, as a binary178	// quantity ("2Gi"). This is an exact allocation, not a request.179	// If Size is omitted, the role takes the rest of its disk. Only180	// one role per disk may omit Size.181	Size string `json:"size,omitempty"`182}183184// A DeclaredRole is one role that is present in the spec, paired185// with its name. This is the form that the rest of liken works with,186// because the name becomes the partition's on-disk identity.187type DeclaredRole struct {188	Name StorageRoleName189	StorageRole190}191192// PartitionName is the role's on-disk identity: the GPT partition193// name. liken writes this name when it claims the role's disk, and194// matches it on every boot after that.195func (r DeclaredRole) PartitionName() string {196	return PartitionPrefix + string(r.Name)197}198199// systemSlotsDir is where the system slots' filesystems are mounted:200// slot A at system/a, and slot B at system/b. Init mounts the slots201// there, through its roleMounts table. The operator writes202// downloaded releases there too, through a hostPath mount. So this203// path is defined once, in the package that both programs share.204const systemSlotsDir = "/var/lib/liken/system"205206// SystemSlotDir is one slot's mountpoint. Slots are named "A" and207// "B" everywhere a person sees them: boot entries, conditions, and208// the liken.slot= parameter. The directory names use the same209// letters, in lowercase.210func SystemSlotDir(slot string) string {211	return systemSlotsDir + "/" + strings.ToLower(slot)212}213214// InactiveSlot is the slot that a machine is not running from. A215// downloaded release lands on this slot; that placement is the point216// of the blue-green arrangement. InactiveSlot returns "" for a217// machine whose boot did not come from a slot at all. Such a machine218// has no inactive slot, and it should never download a release that219// it could not boot.220func InactiveSlot(running string) string {221	switch running {222	case "A":223		return "B"224	case "B":225		return "A"226	}227	return ""228}229230// Role addresses one role's declaration by name. It returns nil for231// names outside the vocabulary, and for roles that the spec leaves232// out.233func (s *StorageSpec) Role(name StorageRoleName) *StorageRole {234	switch name {235	case BIOSBootRole:236		return s.BIOSBoot237	case BootHomeRole:238		return s.BootHome239	case SystemARole:240		return s.SystemA241	case SystemBRole:242		return s.SystemB243	case MachineStateRole:244		return s.MachineState245	case MachineEphemeralRole:246		return s.MachineEphemeral247	case ClusterStateRole:248		return s.ClusterState249	case PodStorageRole:250		return s.PodStorage251	case PodEphemeralRole:252		return s.PodEphemeral253	}254	return nil255}256257// Roles returns the declared roles in StorageRoleNames' canonical258// order.259func (s StorageSpec) Roles() []DeclaredRole {260	var roles []DeclaredRole261	for _, name := range StorageRoleNames {262		if role := s.Role(name); role != nil {263			roles = append(roles, DeclaredRole{name, *role})264		}265	}266	return roles267}268269// Validate checks the spec's internal consistency. It catches the270// errors that a person can fix in the manifest, before the code271// touches any disk.272func (s StorageSpec) Validate() error {273	remainders := map[string]StorageRoleName{}274	for _, role := range s.Roles() {275		if role.Device == "" {276			return fmt.Errorf("storage role %s: no device", role.Name)277		}278		if suffix, ok := strings.CutPrefix(role.Device, "/dev/disk/"); ok {279			tree, name, _ := strings.Cut(suffix, "/")280			if tree == "by-uuid" {281				return fmt.Errorf(282					"storage role %s: device %s is a by-uuid name; a role claims a whole blank disk, and a blank disk has no filesystem, so no UUID can name it",283					role.Name, role.Device)284			}285			if tree != "by-id" && tree != "by-path" {286				return fmt.Errorf(287					"storage role %s: device %s names a /dev/disk/%s tree; only by-id and by-path names are allowed",288					role.Name, role.Device, tree)289			}290			if name == "" {291				return fmt.Errorf(292					"storage role %s: device %s names the /dev/disk/%s tree, not a disk under it",293					role.Name, role.Device, tree)294			}295		}296		if role.Size == "" {297			if other, ok := remainders[role.Device]; ok {298				return fmt.Errorf(299					"storage roles %s and %s both omit their size on %s; only one role per disk may omit its size",300					other, role.Name, role.Device)301			}302			remainders[role.Device] = role.Name303			continue304		}305		if _, err := ParseSize(role.Size); err != nil {306			return fmt.Errorf("storage role %s: %w", role.Name, err)307		}308	}309	return nil310}311312// ParseSize reads a binary quantity ("2Gi", "512Mi", or a plain313// count of bytes) into bytes. ParseSize accepts only the314// power-of-two suffixes. liken divides disks into partitions using315// the same units that its partition math uses. Accepting "2G"316// (decimal) alongside "2Gi" (binary) would invite subtle mistakes,317// because the two units differ by about 7%.318func ParseSize(s string) (uint64, error) {319	// The units form an ordered list rather than a map. The search320	// stops at the first suffix that matches, so the order of the321	// candidates must stay fixed, and a map's iteration order does322	// not stay fixed. No suffix here is a suffix of another suffix,323	// so any fixed order works. If someone adds a suffix that324	// overlaps another (for example "KiB"), that suffix must come325	// before the shorter suffix that it ends with.326	units := []struct {327		suffix string328		factor uint64329	}{330		{"Ki", 1 << 10},331		{"Mi", 1 << 20},332		{"Gi", 1 << 30},333		{"Ti", 1 << 40},334	}335	digits := s336	var unit uint64 = 1337	for _, u := range units {338		if rest, ok := strings.CutSuffix(s, u.suffix); ok {339			digits, unit = rest, u.factor340			break341		}342	}343	n, err := strconv.ParseUint(digits, 10, 64)344	if err != nil {345		return 0, fmt.Errorf("size %q: expected bytes or a Ki/Mi/Gi/Ti quantity", s)346	}347	// ParseSize rejects zero here, instead of leaving it for the348	// partition math to fail on. A zero-sector partition would have349	// an end before its start, and a zero-byte role is always a350	// mistake in the manifest.351	if n == 0 {352		return 0, fmt.Errorf("size %q: a storage role can't be zero bytes", s)353	}354	return n * unit, nil355}356357// SizeText renders a byte count as the quantity a manifest would carry:358// the largest binary unit that divides it exactly. It is the inverse of359// ParseSize for every size a spec can name, so a number the system360// prints back to a person is a number they can paste into a spec. A361// count that no unit divides exactly renders as plain bytes, which362// ParseSize also accepts.363func SizeText(bytes uint64) string {364	switch {365	case bytes%(1<<40) == 0:366		return fmt.Sprintf("%dTi", bytes>>40)367	case bytes%(1<<30) == 0:368		return fmt.Sprintf("%dGi", bytes>>30)369	case bytes%(1<<20) == 0:370		return fmt.Sprintf("%dMi", bytes>>20)371	case bytes%(1<<10) == 0:372		return fmt.Sprintf("%dKi", bytes>>10)373	}374	return fmt.Sprintf("%d", bytes)375}
machine/sysctl.go 95.7%
1package machine23// Sysctls are the kernel's runtime tuning parameters. The kernel4// exposes each sysctl as one file under /proc/sys. No syscall exists5// for sysctls. Writing "1" to /proc/sys/vm/overcommit_memory is the6// interface itself. So applying a sysctl needs no privileges beyond7// a writable /proc/sys. PID 1 has that access at boot. The operator8// gets that access by running as a privileged process.9//10// Other components manage sysctls too. kubelet sets a few parameters11// itself (vm.overcommit_memory, kernel.panic, and others). Per-pod12// sysctls exist for the namespaced net.* family. spec.sysctls covers13// the machine-level parameters that neither of those two paths14// covers.1516import (17	"fmt"18	"io/fs"19	"os"20	"path/filepath"21	"strings"22)2324// sysctlPath translates a parameter name to its file path. Dots25// become path separators, unless the name already contains slashes.26// A name with slashes uses the kernel's own escape hatch, documented27// in sysctl(8), for path segments with literal dots in them, like an28// interface named eth0.100.29//30// The containment check matters because sysctl names arrive from the31// Machine spec, and the Machine spec is user input. A crafted name32// like "../../etc/passwd" must fail here. It must not become a write33// outside /proc/sys.34func sysctlPath(dir, name string) (string, error) {35	rel := name36	if !strings.Contains(name, "/") {37		rel = strings.ReplaceAll(name, ".", "/")38	}39	// filepath.Join cleans its result. This cleaning would quietly40	// repair a malicious name, by folding "a/../../b" or by rooting an41	// absolute path inside dir. So this function rejects anything42	// absolute or upward-pointing before the join, instead of43	// inspecting the result after.44	if filepath.IsAbs(rel) || rel != filepath.Clean(rel) || strings.HasPrefix(rel, "..") {45		return "", fmt.Errorf("sysctl name %q escapes %s: %w", name, dir, fs.ErrInvalid)46	}47	return filepath.Join(dir, rel), nil48}4950// ApplySysctl sets one parameter. The open call does not create the51// file, by design. A parameter file that the kernel did not put52// there names a parameter this kernel does not have. Creating the53// file would hide a typo silently.54func ApplySysctl(dir, name, value string) error {55	path, err := sysctlPath(dir, name)56	if err != nil {57		return err58	}59	f, err := os.OpenFile(path, os.O_WRONLY|os.O_TRUNC, 0)60	if err != nil {61		return fmt.Errorf("sysctl %s: %w", name, err)62	}63	defer f.Close()64	if _, err := fmt.Fprintln(f, value); err != nil {65		return fmt.Errorf("sysctl %s = %s: %w", name, value, err)66	}67	return nil68}6970// ReadSysctl reads a parameter's current value, trimmed of the71// newline that the kernel appends. The operator uses ReadSysctl to72// build status.sysctls. The operator reads the kernel's current73// values instead of keeping a record of what it wrote.74func ReadSysctl(dir, name string) (string, error) {75	path, err := sysctlPath(dir, name)76	if err != nil {77		return "", err78	}79	raw, err := os.ReadFile(path)80	if err != nil {81		return "", fmt.Errorf("sysctl %s: %w", name, err)82	}83	return strings.TrimSpace(string(raw)), nil84}
machine/systemrelease.go 100.0%
1package machine23// The system release record is the document that carries an OS4// upgrade through its proving reboot.5//6// A verified download is bytes on a slot. This record states the7// intent to boot those bytes. The record goes through the same8// staged, attempted, and proven lifecycle as the Machine and Cluster9// manifests (staging.go), in its own store. The record is staged10// when the operator has verified a release on the inactive slot and11// the machine should reboot into it. The record is attempted when12// init arms the firmware and takes the machine down. The record is13// proven when the operator, now running as the new release, confirms14// that the machine came up and serves its cluster. The proven record15// is the standing answer to the question "which slot is this16// machine's known-good slot?" init sets the firmware's BootOrder17// from the proven record on every boot. The store holds the18// authoritative answer. The firmware holds a cached copy of it.19//20// The record carries no artifact digests, by design. Those digests21// live in the release document on the slot itself, and the trust22// chain already ran when the bytes landed. The record pins the23// decision itself: this version, under this catalog digest, on this24// slot.2526import (27	"fmt"2829	"github.com/liken-sh/liken/liken/api"30	"sigs.k8s.io/yaml"31)3233// SystemRelease names one release installed on one slot.34type SystemRelease struct {35	APIVersion string `json:"apiVersion"`36	Kind       string `json:"kind"`3738	// Version is the release, exactly as the catalog names it.39	Version string `json:"version"`4041	// Slot names where the release's artifacts are: "A" or "B".42	Slot string `json:"slot"`4344	// ReleaseDigest is the catalog's digest over the release45	// document. The record carries this digest so the record's46	// identity changes when the catalog's entry changes. The same47	// version, republished under a different digest, counts as a48	// different decision. ReleaseDigest is empty on records that49	// describe an installed system instead of a catalog download.50	// The installer's release predates any catalog.51	ReleaseDigest string `json:"releaseDigest,omitempty"`52}5354// SystemReleases returns the record's lifecycle store under the55// given machineState root, beside the Machine's store and the56// Cluster's store.57func SystemReleases(root string) ManifestStore {58	return ManifestStore{dir: root + "/system"}59}6061// RenderSystemRelease produces the record's canonical bytes and their62// hash. The hash serves as the record's identity for staging63// idempotence, for the attempted marker, and for rejections.64func RenderSystemRelease(version, slot, releaseDigest string) ([]byte, string, error) {65	return renderDocument(SystemRelease{66		APIVersion:    api.APIVersion,67		Kind:          "SystemRelease",68		Version:       version,69		Slot:          slot,70		ReleaseDigest: releaseDigest,71	})72}7374// ParseSystemRelease reads a record with strict checks and validates75// the record right away. liken validates every document for the same76// reason: if a record would fail every future boot in the same way,77// the system must reject the record once, in a visible way. The78// system must not retry the record forever.79func ParseSystemRelease(raw []byte) (*SystemRelease, error) {80	r := &SystemRelease{}81	if err := yaml.UnmarshalStrict(raw, r); err != nil {82		return nil, err83	}84	if r.Kind != "SystemRelease" {85		return nil, fmt.Errorf("expected kind SystemRelease, got %q", r.Kind)86	}87	if r.Version == "" {88		return nil, fmt.Errorf("the record names no version")89	}90	if r.Slot != "A" && r.Slot != "B" {91		return nil, fmt.Errorf("the record's slot must be A or B, got %q", r.Slot)92	}93	return r, nil94}
machine/wireless.go 100.0%
1package machine23// The wireless half of an interface: what a manifest may say about a4// radio. The spec carries only public facts, the network's name and5// how the machine joins it. The passphrase is deliberately absent. A6// manifest travels on install sticks and in deployment git, so the7// secret lives in a file on the machine instead, and the SSID is the8// key that finds it (plans/completed/62-wifi.md).910import (11	"fmt"12	"strings"13)1415// WirelessSpec names the network a radio joins and how it joins it.16// Both fields are public facts. An access point broadcasts its SSID17// to anyone listening, so writing it in a manifest reveals nothing.18type WirelessSpec struct {19	// SSID is the network's name, as the access point advertises20	// it. The name doubles as the name of the passphrase file at21	// /etc/liken/psk/<ssid>, which is why validate refuses an SSID22	// that cannot name a file.23	SSID string `json:"ssid"`2425	// Security is how the radio joins. An unset value means26	// wpa-psk, the common case. open is the deviation a person27	// spells out.28	Security WirelessSecurity `json:"security,omitempty"`29}3031// WirelessSecurity is the way a machine joins a network. wpa-psk32// covers WPA2 and WPA3 personal with one passphrase, because the33// supplicant negotiates the protocol with each access point.34type WirelessSecurity string3536const (37	WirelessWPAPSK WirelessSecurity = "wpa-psk"38	WirelessOpen   WirelessSecurity = "open"39)4041// SecurityOrDefault names the security an unset field asks for. The42// CRD writes the default into a spec applied through the API server,43// but a manifest carried in on a stick never met the API server, so44// both readers resolve the value here. rebootPolicy takes the same45// two-path treatment.46func (w WirelessSpec) SecurityOrDefault() WirelessSecurity {47	if w.Security == WirelessOpen {48		return WirelessOpen49	}50	return WirelessWPAPSK51}5253// maxSSIDOctets is the largest SSID that 802.11 carries in one54// information element. The limit counts octets, not characters.55const maxSSIDOctets = 325657// isControlByte reports whether a rune is one of the C0 controls or58// DEL. These are the bytes that the SSID's consumers read as59// structure rather than as text: the facts tree writes the SSID as60// one line of a file, and init renders it into a generated61// supplicant configuration.62func isControlByte(r rune) bool {63	return r < 0x20 || r == 0x7f64}6566// validate checks one interface's wireless entry. Each rule states67// its reason at the point where it refuses.68func (w WirelessSpec) validate(name string) error {69	// Init reads the passphrase from /etc/liken/psk/<ssid>, so the70	// SSID has to be a name a path can end in. These values are the71	// ones that are not: the empty name opens the directory itself,72	// the two dot names open a directory beside it, and a separator73	// opens a file somewhere else entirely.74	switch {75	case w.SSID == "":76		return fmt.Errorf("interface %s declares a wireless network with no ssid; the ssid names both the network and the file that holds its passphrase", name)77	case w.SSID == "." || w.SSID == "..":78		return fmt.Errorf("interface %s declares the ssid %q, which names a directory and not a passphrase file", name, w.SSID)79	case strings.Contains(w.SSID, "/"):80		return fmt.Errorf("interface %s declares the ssid %q, which holds a path separator and so names a file outside /etc/liken/psk", name, w.SSID)81	// 802.11 lets an SSID carry any octet, but every consumer of82	// the name here reads lines: an embedded newline would split the83	// facts tree's one-line record and end a generated configuration84	// line early. A NUL falls to the same rule, because it would end85	// the passphrase path before the name does. No manifest has a86	// legitimate reason to spell any of these bytes.87	case strings.ContainsFunc(w.SSID, isControlByte):88		return fmt.Errorf("interface %s declares the ssid %q, which holds a control byte; an ssid holds printable characters only", name, w.SSID)89	}90	if len(w.SSID) > maxSSIDOctets {91		return fmt.Errorf("interface %s declares an ssid of %d octets; 802.11 carries at most %d",92			name, len(w.SSID), maxSSIDOctets)93	}94	// Init renders the security into a supplicant configuration. A95	// word it has no rendering for must be refused here, where the96	// operator can read the answer, because on the machine it would97	// become a failed join with no shell to investigate from.98	switch w.Security {99	case "", WirelessWPAPSK, WirelessOpen:100		return nil101	default:102		return fmt.Errorf("interface %s declares the security %q; the values are %s and %s",103			name, w.Security, WirelessWPAPSK, WirelessOpen)104	}105}
metrics/listener.go 95.7%
1package metrics23// The listener: how a scrape reaches an operator's registry.4//5// A Prometheus server pulls. It opens an HTTP connection to each6// target every few seconds and reads a text document of the current7// numbers. So every process that publishes metrics has to serve a8// port, and the contract gives each liken process a port of its own,9// because some of these pods run on the host network and two of them10// can share a node.11//12// A scrape reads only the in-memory registry. It never walks sysfs,13// never calls the Kubernetes API, and never waits on hardware. This14// is what makes the listener safe to answer at any moment: a scraper15// that arrives during a reboot decision reads the last numbers the16// reconcile pass published, and delays nothing.1718import (19	"fmt"20	"io"21	"net"22	"net/http"23	"os"24	"time"2526	"github.com/prometheus/client_golang/prometheus/promhttp"27)2829// Handler answers a scrape from this operator's registry. It is30// exported so a test can scrape without a socket, and so a program31// can put /metrics on a server of its own.32func (o *Operator) Handler() http.Handler {33	return promhttp.HandlerFor(o.registry, promhttp.HandlerOpts{34		// A collector that fails is reported as a broken scrape,35		// rather than served as a document with a gap in it. A36		// scraper reads the failure and records the target as down,37		// which is the accurate answer.38		ErrorHandling: promhttp.HTTPErrorOnError,39	})40}4142// SetHealth gives the listener a /healthz that answers 200 while check43// answers nil, and 500 with its error otherwise, for a liveness probe.44// Like a scrape, the check must read only memory. Call it before Serve.45func (o *Operator) SetHealth(check func() error) {46	o.health = check47}4849// mux routes the listener's paths.50func (o *Operator) mux() *http.ServeMux {51	mux := http.NewServeMux()52	mux.Handle("/metrics", o.Handler())53	if check := o.health; check != nil {54		mux.HandleFunc("/healthz", func(w http.ResponseWriter, _ *http.Request) {55			if err := check(); err != nil {56				http.Error(w, err.Error(), http.StatusInternalServerError)57				return58			}59			_, _ = io.WriteString(w, "ok\n")60		})61	}62	return mux63}6465// Serve starts the /metrics listener and returns at once. It returns66// the address the listener holds, so a caller can report the real67// port after asking for port 0.68//69// An empty address turns the listener off. The function then returns70// no address and no error, and the process runs with no metrics at71// all. This is what lets an owner who runs no Prometheus give up the72// port, and it is why the address is a setting rather than a73// constant.74//75// The listen call happens here, in the caller's own goroutine, so a76// port already in use is an error the caller can report. Serving77// happens on a goroutine, because the work the process exists for78// must not wait on a scraper. A failure after this point ends the79// listener alone.80func (o *Operator) Serve(address string) (net.Addr, error) {81	if address == "" {82		return nil, nil83	}84	listener, err := net.Listen("tcp", address)85	if err != nil {86		return nil, err87	}8889	server := &http.Server{90		Handler: o.mux(),91		// A client that opens a connection and sends no headers holds92		// a goroutine for as long as it stays. This bound releases93		// that goroutine, and it is the one timeout a metrics94		// endpoint needs: the handler itself only reads memory.95		ReadHeaderTimeout: 10 * time.Second,96	}97	go func() {98		if err := server.Serve(listener); err != nil {99			fmt.Fprintf(os.Stderr, "the metrics listener stopped: %v\n", err)100		}101	}()102	return listener.Addr(), nil103}
metrics/metrics.go 96.0%
1// Package metrics is liken's half of the Prometheus contract.2//3// A Machine's status says what a machine is now. It says nothing about4// the past: how long the last upgrade took, how often a download5// failed, or what the fleet's release skew was an hour ago. Prometheus6// keeps that history, because a scraper reads the same numbers every7// few seconds and stores each reading with its time.8// `plans/completed/65-prometheus-metrics.md` at the top of the9// repository is the contract that every liken repository follows, and10// this package is liken's half of it.11//12// The contract has three layers. Layer 1 is the runtime: the client13// library's go_* and process_* series, and one liken_build_info14// gauge that names the component and the release it runs. Layer 2 is15// the reconcile loop, which every operator has: how long a pass16// takes, how many passes fail, and how often a watch restarts, each17// labeled by the resource kind the loop serves. Layer 3 is the18// domain, and it belongs to each program. The machine operator's19// layer 3 is in machine-operator/metrics.go, and the cluster20// operator's is in cluster-operator/metrics.go.21//22// The plan holds layers 1 and 2 to be about fifty lines that each23// repository writes for itself, because the operators in the other24// repositories import nothing from each other. Inside liken, the two25// operators are two programs of one Go module, so they share this26// package the way they already share api/ and kubernetes/.27//28// Every collector here lives on a registry that this package builds,29// never on the client library's global default. A registry that a30// program builds is a value that a test builds too, so every series31// below is proven by a real scrape of a real registry.32package metrics3334import (35	"time"3637	"github.com/prometheus/client_golang/prometheus"38	"github.com/prometheus/client_golang/prometheus/collectors"39)4041// Prefix is the name liken puts in front of every metric it42// publishes. The contract gives one prefix to each repository, so a43// person reading a dashboard knows which program a series comes44// from, and two repositories can never name two different facts the45// same thing.46const Prefix = "liken_"4748// An Operator is one operator process's registry: layer 1, layer 2,49// and whatever layer 3 the program registers on it. Each operator50// builds exactly one, and hands it to the parts of itself that51// observe something.52type Operator struct {53	registry          *prometheus.Registry54	reconcileDuration *prometheus.HistogramVec55	reconcileErrors   *prometheus.CounterVec56	watchRestarts     *prometheus.CounterVec5758	// health is the program's liveness check, which /healthz answers.59	// Nil serves no /healthz.60	health func() error61}6263// reconcileBuckets are the histogram's boundaries, in seconds. A64// reconcile pass that changes nothing takes about a millisecond, and65// a pass that writes status takes a few. The buckets step by four,66// from one millisecond to sixteen seconds, so the ordinary pass and67// a pass that waits on a slow API server both land inside the range.68// A pass slower than the top bucket still counts, in the +Inf bucket69// that every Prometheus histogram carries.70var reconcileBuckets = prometheus.ExponentialBuckets(0.001, 4, 8)7172// NewOperator builds the registry for one operator process.73// component is the program's own name, and version is the release it74// was built as. reconciles names every resource kind whose loop this75// program runs, and watches names every kind it holds a watch on.76// The two lists differ where an operator reconciles one kind from a77// watch on another: the cluster operator writes the `Cluster`, and78// hears about the fleet on a `Machine` watch.79//80// The kinds matter at construction, not only at observation time.81// Prometheus computes a rate from two readings of the same series,82// so a counter that appears only when it first increases has no83// earlier reading to compare against, and its first error is invisible84// on a graph. Creating the series here publishes each one at zero85// from the first scrape. A kind that a program never serves is left86// out, so a series that is always zero never appears at all.87func NewOperator(component, version string, reconciles, watches []string) *Operator {88	o := &Operator{89		registry: prometheus.NewRegistry(),90		reconcileDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{91			Name:    Prefix + "reconcile_duration_seconds",92			Help:    "How long one reconcile pass takes, by resource kind.",93			Buckets: reconcileBuckets,94		}, []string{"kind"}),95		reconcileErrors: prometheus.NewCounterVec(prometheus.CounterOpts{96			Name: Prefix + "reconcile_errors_total",97			Help: "Reconcile passes that failed, by resource kind.",98		}, []string{"kind"}),99		watchRestarts: prometheus.NewCounterVec(prometheus.CounterOpts{100			Name: Prefix + "watch_restarts_total",101			Help: "Watches that closed and opened again, by resource kind.",102		}, []string{"kind"}),103	}104105	// Layer 1. The runtime collectors report the process itself: its106	// memory, its goroutines, its open files, and its start time. The107	// build info gauge always holds 1, and its labels carry the fact,108	// which is the Prometheus convention for a string. One panel that109	// reads liken_build_info across a cluster shows every component's110	// release, because every liken process publishes this one name.111	buildInfo := prometheus.NewGaugeVec(prometheus.GaugeOpts{112		Name: Prefix + "build_info",113		Help: "The release each liken component runs, as labels on a gauge that is always 1.",114	}, []string{"component", "version"})115	buildInfo.WithLabelValues(component, version).Set(1)116	o.registry.MustRegister(117		collectors.NewGoCollector(),118		collectors.NewProcessCollector(collectors.ProcessCollectorOpts{}),119		buildInfo,120		o.reconcileDuration,121		o.reconcileErrors,122		o.watchRestarts,123	)124125	for _, kind := range reconciles {126		o.reconcileDuration.WithLabelValues(kind)127		o.reconcileErrors.WithLabelValues(kind)128	}129	for _, kind := range watches {130		o.watchRestarts.WithLabelValues(kind)131	}132	return o133}134135// Registry is where a program registers its layer 3. One registry136// serves one process, so the domain metrics and the two layers above137// them answer the same scrape.138func (o *Operator) Registry() *prometheus.Registry {139	return o.registry140}141142// ObserveReconcile records one finished reconcile pass. Every pass143// counts, including a pass that changed nothing, because the rate of144// passes is what says the loop still runs. A pass that returned an145// error counts twice: once in the histogram, because it still took146// time, and once in the error counter.147func (o *Operator) ObserveReconcile(kind string, took time.Duration, err error) {148	o.reconcileDuration.WithLabelValues(kind).Observe(took.Seconds())149	if err != nil {150		o.reconcileErrors.WithLabelValues(kind).Inc()151	}152}153154// WatchRestarted records one watch that the operator opened again. A155// Kubernetes API server ends a watch on its own schedule, so a low156// rate here is normal. A high rate says the stream breaks faster157// than the loop can use it.158func (o *Operator) WatchRestarted(kind string) {159	o.watchRestarts.WithLabelValues(kind).Inc()160}
mount/args.go 77.9%
1package main23// The command line mount(8) accepts, and what liken's mount does4// with it.5//6// This grammar is old and irregular, and it cannot be rebuilt on7// Go's flag package: options repeat and accumulate (-o may appear8// many times), short names cluster (-rv), a value may be attached to9// its name (-tnfs) or follow it, and long names carry an operation10// rather than a value (--make-rshared). The parser is therefore11// written out by hand, and it accepts the forms that callers really12// use rather than every form that has ever existed.13//14// One form is deliberately refused. `mount -a` mounts everything in15// /etc/fstab, and a liken machine has no fstab: every filesystem it16// mounts is either init's work or a pod's volume, and both name what17// they need. Refusing with a message is better than reading a file18// that will never exist and reporting success.1920import (21	"fmt"22	"strings"2324	"golang.org/x/sys/unix"25)2627// request is one mount, as the command line described it.28type request struct {29	// source is what to mount and target is where. A command line30	// with one path names only the target, which is how remounts and31	// propagation changes are written.32	source string33	target string3435	// fstype is the -t argument. It is empty for bind mounts,36	// remounts, and propagation changes, none of which introduce a37	// filesystem.38	fstype string3940	// options are the -o lists, in the order given. They stay41	// unsplit until the last moment, because a mount helper needs42	// them exactly as they arrived.43	options []string4445	// flags are the bits that came from long options rather than46	// from an -o list, such as --bind.47	flags uintptr4849	// These four are mount(8)'s own behavior switches. This program50	// acts on none of them, and forwards them to a helper, which is51	// the only place they can still mean anything.52	verbose bool53	fake    bool54	noMtab  bool55	sloppy  bool56}5758// longFlags are the long options that name an operation. Each is a59// spelling of an option that could also have arrived inside -o, and60// each sets exactly the same bits.61var longFlags = map[string]uintptr{62	"--bind":             unix.MS_BIND,63	"--rbind":            unix.MS_BIND | unix.MS_REC,64	"--move":             unix.MS_MOVE,65	"--make-shared":      unix.MS_SHARED,66	"--make-rshared":     unix.MS_SHARED | unix.MS_REC,67	"--make-slave":       unix.MS_SLAVE,68	"--make-rslave":      unix.MS_SLAVE | unix.MS_REC,69	"--make-private":     unix.MS_PRIVATE,70	"--make-rprivate":    unix.MS_PRIVATE | unix.MS_REC,71	"--make-unbindable":  unix.MS_UNBINDABLE,72	"--make-runbindable": unix.MS_UNBINDABLE | unix.MS_REC,73}7475// parser walks one command line. It carries its position, because76// an option's value may sit in the next argument, and reading it77// there means the walk skips it.78type parser struct {79	argv  []string80	at    int81	req   request82	paths []string83}8485// value takes the value that belongs to an option: the rest of this86// argument when one is attached to the name, and the following87// argument otherwise.88func (p *parser) value(name, attached string) (string, error) {89	if attached != "" {90		return attached, nil91	}92	if p.at+1 >= len(p.argv) {93		return "", fmt.Errorf("%s needs a value", name)94	}95	p.at++96	return p.argv[p.at], nil97}9899// parseArgs reads a mount command line. It returns an error only for100// a command line this program cannot act on, never for an option it101// merely does not understand: an unknown -o entry is the102// filesystem's business, and splitOptions passes it along.103func parseArgs(argv []string) (*request, error) {104	p := &parser{argv: argv}105	for ; p.at < len(p.argv); p.at++ {106		if err := p.argument(p.argv[p.at]); err != nil {107			return nil, err108		}109	}110111	// One path is a target, which is what a remount or a propagation112	// change is written with. Two are a source and a target. The113	// long spellings may have supplied either already.114	switch len(p.paths) {115	case 0:116	case 1:117		p.req.target = p.paths[0]118	case 2:119		p.req.source, p.req.target = p.paths[0], p.paths[1]120	default:121		return nil, fmt.Errorf("too many paths: %s", strings.Join(p.paths, " "))122	}123	if p.req.target == "" {124		return nil, fmt.Errorf("no mount point given")125	}126	return &p.req, nil127}128129// argument handles one argument of the command line.130func (p *parser) argument(arg string) error {131	switch {132	case arg == "--":133		// Everything after this is a path, however it is spelled. A134		// target whose name begins with a dash is why this exists.135		p.paths = append(p.paths, p.argv[p.at+1:]...)136		p.at = len(p.argv)137		return nil138	case arg == "-a" || arg == "--all":139		return fmt.Errorf("-a mounts what /etc/fstab lists, and this system has no fstab; name what to mount")140	case strings.HasPrefix(arg, "--"):141		return p.longOption(arg)142	case strings.HasPrefix(arg, "-") && arg != "-":143		return p.shortOptions(arg)144	default:145		p.paths = append(p.paths, arg)146		return nil147	}148}149150// longOption handles one argument that begins with two dashes.151func (p *parser) longOption(arg string) error {152	switch arg {153	case "--options", "--types", "--source", "--target":154		v, err := p.value(arg, "")155		if err != nil {156			return err157		}158		switch arg {159		case "--options":160			p.req.options = append(p.req.options, v)161		case "--types":162			p.req.fstype = v163		case "--source":164			p.req.source = v165		case "--target":166			p.req.target = v167		}168	case "--read-only":169		p.req.options = append(p.req.options, "ro")170	case "--rw", "--read-write":171		p.req.options = append(p.req.options, "rw")172	case "--verbose":173		p.req.verbose = true174	case "--fake":175		p.req.fake = true176	case "--no-mtab":177		p.req.noMtab = true178	case "--sloppy":179		p.req.sloppy = true180	case "--no-canonicalize", "--internal-only":181		// Both ask mount(8) to skip work this program never does.182		// Paths arrive here as written, and there is no mount table183		// of this program's own to consult.184	default:185		bits, known := longFlags[arg]186		if !known {187			return fmt.Errorf("unknown option %s", arg)188		}189		p.req.flags |= bits190	}191	return nil192}193194// shortOptions handles one argument that begins with a single dash.195// Several names may share the argument, and the last of them may196// carry a value, so -rv, -o bind, and -obind are each one argument.197func (p *parser) shortOptions(arg string) error {198	for pos, name := range arg[1:] {199		attached := arg[1+pos+len(string(name)):]200		switch name {201		case 'o', 't':202			v, err := p.value("-"+string(name), attached)203			if err != nil {204				return err205			}206			if name == 'o' {207				p.req.options = append(p.req.options, v)208			} else {209				p.req.fstype = v210			}211			// A name that took a value ends the argument, because212			// the value ran to the end of it.213			return nil214		case 'r':215			p.req.options = append(p.req.options, "ro")216		case 'w':217			p.req.options = append(p.req.options, "rw")218		case 'v':219			p.req.verbose = true220		case 'f':221			p.req.fake = true222		case 'n':223			p.req.noMtab = true224		case 's':225			p.req.sloppy = true226		case 'B':227			p.req.flags |= unix.MS_BIND228		case 'R':229			p.req.flags |= unix.MS_BIND | unix.MS_REC230		case 'M':231			p.req.flags |= unix.MS_MOVE232		default:233			return fmt.Errorf("unknown option -%c", name)234		}235	}236	return nil237}238239// optionString joins the -o lists back into the single list that240// both mount(2) and a mount helper expect. Rebuilding it in the241// order the options arrived is what keeps the last name for a242// setting the winning one, whether the caller wrote -o ro,rw or243// -o ro -o rw.244func (r *request) optionString() string {245	return strings.Join(r.options, ",")246}
mount/main.go 36.8%
1// Command mount is liken's mount(8): the program Kubernetes runs2// when a pod needs a volume.3//4// Nearly everything else on a liken machine speaks to the kernel5// directly. Init mounts filesystems with the mount system call, and6// so would any Go program here. This one exists because the kubelet7// does not. When a pod declares a volume, the kubelet builds a8// command line and runs a program named `mount`, found on its PATH.9// So the OS has to supply one, and the OS may as well supply one10// that it can explain.11//12// mount(8) does two small jobs. Neither of them is the mount itself.13//14// The first job is translation, and options.go does it. mount(2)15// takes a flag word and an opaque data string, and the kernel's16// split between them is arbitrary: `ro` is a bit the VFS acts on for17// every filesystem, while `vers=4.1` is text that only the NFS18// client understands. Callers write one comma list mixing the two.19//20// The second job is dispatch, and it is why this program had to be21// written. Some filesystems cannot be mounted by the syscall alone.22// NFS is the one that matters here: before the kernel can mount an23// export, something in userspace has to reach the server, agree on a24// protocol version, and turn the result into the options the kernel25// needs. That work lives in a helper program named26// /sbin/mount.<type>, and running it is mount(8)'s job. The kernel27// never runs a helper, and nothing else will do it either.28//29// This is why a plain `nfs:` volume needs no version in its options.30// The helper negotiates the highest version the server offers, and31// asks the kernel for that one. Without a helper, the raw syscall32// asks for NFSv3, which liken does not carry, and the kernel33// refuses. A version written by hand in the mount options is a34// workaround for a missing mount(8), not a requirement of NFS.35//36// This program ships at /sbin/mount so that it wins. k3s carries a37// busybox with a `mount` applet in its own bin directory, and that38// applet has no helper support: every mount it performs is the raw39// syscall. k3s puts /sbin ahead of that directory on the PATHs it40// builds, which is the same seam that makes liken's static iptables41// win over k3s's bundled one.42//43// There is no umount here, for the same reason in reverse. k3s's44// busybox umount sits earlier on the PATH than /sbin, so a umount45// shipped here would never run, and none is needed: unmounting an46// NFSv4 mount is the plain syscall, with nobody to tell.47package main4849import (50	"fmt"51	"os"52	"path/filepath"5354	"golang.org/x/sys/unix"55)5657// helperDir is where mount helpers live. The name /sbin/mount.<type>58// is a contract older than any of the software here, and liken's59// image puts mount.nfs and its mount.nfs4 alias there. It is a60// variable so the tests can point the lookup at a directory of their61// own.62var helperDir = "/sbin"6364// helperExcluded are the operations that never use a helper, because65// none of them introduces a filesystem. A bind mount grafts a tree66// that is already mounted, a move relocates one, a remount edits67// one, and a propagation change only tells the kernel how mount68// events should travel. In every case the filesystem's userspace has69// already done its work, or has nothing to do. Real mount commands70// draw this same line, and it matters here because the kubelet's71// bind mounts carry the volume's own -o list along with `bind`.72const helperExcluded = unix.MS_BIND | unix.MS_MOVE | unix.MS_REMOUNT |73	unix.MS_SHARED | unix.MS_PRIVATE | unix.MS_SLAVE | unix.MS_UNBINDABLE7475// These are mount(8)'s exit codes, and callers read them. The76// kubelet reports the failure of a volume mount by this number and77// the program's output together.78const (79	exitUsage   = 180	exitFailure = 3281)8283func main() {84	req, err := parseArgs(os.Args[1:])85	if err != nil {86		fmt.Fprintf(os.Stderr, "mount: %v\n", err)87		os.Exit(exitUsage)88	}89	if err := req.run(); err != nil {90		fmt.Fprintf(os.Stderr, "mount: %v\n", err)91		os.Exit(exitFailure)92	}93}9495// run performs the mount the command line asked for, by helper when96// the filesystem has one and by syscall otherwise.97func (r *request) run() error {98	flags, data := splitOptions(r.optionString())99	flags |= r.flags100101	if helper := r.helper(flags); helper != "" {102		// A helper replaces this process rather than running under103		// it. Its exit code is then the exit code of the mount, and104		// its output is the output of the mount, with nothing here105		// left to translate or lose.106		argv := r.helperArgs(helper)107		if err := unix.Exec(helper, argv, os.Environ()); err != nil {108			return fmt.Errorf("running %s: %w", helper, err)109		}110	}111112	if err := unix.Mount(r.source, r.target, r.fstype, flags, data); err != nil {113		return fmt.Errorf("mounting %s on %s: %w", r.describeSource(), r.target, err)114	}115	return nil116}117118// helper finds the mount helper for this request's filesystem type,119// and answers with an empty string when the mount must go straight120// to the syscall. A type with no helper installed is the ordinary121// case: ext4, tmpfs, and every other filesystem the kernel mounts122// alone. Falling through to the syscall is therefore not a failure123// path, it is the common one.124func (r *request) helper(flags uintptr) string {125	if r.fstype == "" || flags&helperExcluded != 0 {126		return ""127	}128	path := filepath.Join(helperDir, "mount."+r.fstype)129	info, err := os.Stat(path)130	if err != nil || info.IsDir() || info.Mode().Perm()&0o111 == 0 {131		return ""132	}133	return path134}135136// helperArgs builds the command line a mount helper expects:137//138//	mount.<type> <source> <target> [-sfnv] [-o options]139//140// The options go across whole and unsplit. The helper has its own141// idea of which names it acts on, which names it forwards to the142// kernel, and which names it rewrites on the way, and none of that143// is this program's business. There is no -t argument, because the144// helper is always named for the type it was chosen for, and a145// helper reads its own name to get the filesystem type:146// this is how one binary answers as both mount.nfs and mount.nfs4.147func (r *request) helperArgs(helper string) []string {148	argv := []string{helper, r.source, r.target}149	for _, sw := range []struct {150		set  bool151		name string152	}{153		{r.sloppy, "-s"},154		{r.fake, "-f"},155		{r.noMtab, "-n"},156		{r.verbose, "-v"},157	} {158		if sw.set {159			argv = append(argv, sw.name)160		}161	}162	if opts := r.optionString(); opts != "" {163		argv = append(argv, "-o", opts)164	}165	return argv166}167168// describeSource names what was being mounted, for a failure169// message. A remount or a propagation change has no source to name,170// so the message says what it did have.171func (r *request) describeSource() string {172	if r.source != "" {173		return r.source174	}175	if r.fstype != "" {176		return r.fstype177	}178	return "the mount"179}
mount/options.go 100.0%
1package main23// The translation half of mount(8): turning one comma list into the4// two arguments mount(2) actually takes.5//6// The kernel splits a filesystem's options in two, and the split is7// not principled. Some options are bits in a flag word that the VFS8// itself acts on, the same bits for every filesystem: read-only,9// no-setuid, no-execute, the times policy. The rest is an opaque10// string that the kernel hands to the filesystem driver, which11// parses it however it likes: vers=4.1 means something to the NFS12// client and nothing to ext4.13//14// Nobody writes options that way. A pod spec, a fstab line, and the15// kubelet all write one comma list that mixes the two kinds freely.16// Sorting that list into the flag word and the data string is what17// this file does, and it is most of what mount(8) is.1819import (20	"strings"2122	"golang.org/x/sys/unix"23)2425// kernelFlags are the option names that become bits in mount(2)'s26// flag word. Each entry names the bit and says whether the option27// sets or clears it, because these names come in pairs: nosuid and28// suid are two names for one bit. The parser applies them in the29// order written, so the last name for a bit wins, which is the rule30// every mount command has followed.31//32// None of these names ever takes a value, so the parser matches the33// whole token rather than the part before an equals sign.34var kernelFlags = map[string]struct {35	bit uintptr36	set bool37}{38	"ro":            {unix.MS_RDONLY, true},39	"rw":            {unix.MS_RDONLY, false},40	"suid":          {unix.MS_NOSUID, false},41	"nosuid":        {unix.MS_NOSUID, true},42	"dev":           {unix.MS_NODEV, false},43	"nodev":         {unix.MS_NODEV, true},44	"exec":          {unix.MS_NOEXEC, false},45	"noexec":        {unix.MS_NOEXEC, true},46	"sync":          {unix.MS_SYNCHRONOUS, true},47	"async":         {unix.MS_SYNCHRONOUS, false},48	"dirsync":       {unix.MS_DIRSYNC, true},49	"atime":         {unix.MS_NOATIME, false},50	"noatime":       {unix.MS_NOATIME, true},51	"diratime":      {unix.MS_NODIRATIME, false},52	"nodiratime":    {unix.MS_NODIRATIME, true},53	"relatime":      {unix.MS_RELATIME, true},54	"norelatime":    {unix.MS_RELATIME, false},55	"strictatime":   {unix.MS_STRICTATIME, true},56	"nostrictatime": {unix.MS_STRICTATIME, false},57	"lazytime":      {unix.MS_LAZYTIME, true},58	"nolazytime":    {unix.MS_LAZYTIME, false},59	"mand":          {unix.MS_MANDLOCK, true},60	"nomand":        {unix.MS_MANDLOCK, false},61	"iversion":      {unix.MS_I_VERSION, true},62	"noiversion":    {unix.MS_I_VERSION, false},63	"nosymfollow":   {unix.MS_NOSYMFOLLOW, true},64	"symfollow":     {unix.MS_NOSYMFOLLOW, false},65	"silent":        {unix.MS_SILENT, true},66	"loud":          {unix.MS_SILENT, false},6768	// These last names are not properties of a mount but operations69	// on one. The kernel takes them in the same flag word, so the70	// options list can carry them, and the kubelet's bind mounts do71	// exactly that. Everything that changes propagation is recursive72	// under its r-prefixed name, which is the only form anything73	// here uses.74	"remount":     {unix.MS_REMOUNT, true},75	"bind":        {unix.MS_BIND, true},76	"rbind":       {unix.MS_BIND | unix.MS_REC, true},77	"move":        {unix.MS_MOVE, true},78	"shared":      {unix.MS_SHARED, true},79	"rshared":     {unix.MS_SHARED | unix.MS_REC, true},80	"slave":       {unix.MS_SLAVE, true},81	"rslave":      {unix.MS_SLAVE | unix.MS_REC, true},82	"private":     {unix.MS_PRIVATE, true},83	"rprivate":    {unix.MS_PRIVATE | unix.MS_REC, true},84	"unbindable":  {unix.MS_UNBINDABLE, true},85	"runbindable": {unix.MS_UNBINDABLE | unix.MS_REC, true},86}8788// userspaceOptions are the names that mount commands act on89// themselves and never pass down. They answer questions the kernel90// was never asked: whether a boot should mount this entry, who may91// unmount it, whether the network must be up first. A liken machine92// has no fstab and no unprivileged users, so every one of these is93// inert here. They are recognized anyway, because dropping them is94// the difference between an option a caller may harmlessly include95// and a mount that fails with an unknown option.96//97// The parser also drops any name beginning with x-, the reserved98// prefix for comments that tools keep in the mount table for99// themselves.100var userspaceOptions = map[string]bool{101	"defaults": true,102	"auto":     true,103	"noauto":   true,104	"nofail":   true,105	"_netdev":  true,106	"user":     true,107	"nouser":   true,108	"users":    true,109	"owner":    true,110	"group":    true,111	"comment":  true,112}113114// splitOptions sorts a comma list into the flag word and the data115// string that mount(2) takes. Anything this file does not recognize116// is data, which is the correct default: an option missing from the117// table above is one the filesystem driver defines, and the driver118// is the one that must judge it.119func splitOptions(options string) (uintptr, string) {120	var flags uintptr121	var data []string122	for token := range strings.SplitSeq(options, ",") {123		token = strings.TrimSpace(token)124		if token == "" {125			continue126		}127		if entry, known := kernelFlags[token]; known {128			if entry.set {129				flags |= entry.bit130			} else {131				flags &^= entry.bit132			}133			continue134		}135		// An option may carry a value, and the name is the part136		// before the equals sign. Only the userspace list is137		// consulted by name, because every kernel flag above is a138		// bare word.139		name, _, _ := strings.Cut(token, "=")140		if userspaceOptions[name] || strings.HasPrefix(name, "x-") {141			continue142		}143		data = append(data, token)144	}145	return flags, strings.Join(data, ",")146}
plugins/image.go 86.0%
1package plugins23// This file derives a CLI image from its operator image, pulls it4// from the registry for the workstation's architecture, and extracts5// the single binary the image runs as its entrypoint. The pull uses6// go-containerregistry compiled into the base binary, so the7// workstation needs no docker, oras, or crane.89import (10	"archive/tar"11	"fmt"12	"io"13	"strings"1415	"github.com/google/go-containerregistry/pkg/authn"16	"github.com/google/go-containerregistry/pkg/name"17	v1 "github.com/google/go-containerregistry/pkg/v1"18	"github.com/google/go-containerregistry/pkg/v1/mutate"19	"github.com/google/go-containerregistry/pkg/v1/remote"20)2122// cliImageRef derives the CLI image reference from an operator image23// reference: the repository gains a -cli suffix and the tag stays24// the same, so the CLI comes from the same version as the running25// operator and the two never drift. A reference by digest carries no26// version tag to reuse, so it has no derived CLI.27func cliImageRef(operatorImage string) (string, error) {28	if strings.Contains(operatorImage, "@") {29		return "", fmt.Errorf("image %q is pinned by digest and names no version tag", operatorImage)30	}31	tag, err := name.NewTag(operatorImage)32	if err != nil {33		return "", fmt.Errorf("reading image reference %q: %w", operatorImage, err)34	}35	return tag.Context().Name() + "-cli:" + tag.TagStr(), nil36}3738// ImageVersion reports the version tag an operator image carries, the39// universal source every CLI compares against.40func ImageVersion(operatorImage string) (string, error) {41	if strings.Contains(operatorImage, "@") {42		return "", fmt.Errorf("image %q is pinned by digest and names no version tag", operatorImage)43	}44	tag, err := name.NewTag(operatorImage)45	if err != nil {46		return "", fmt.Errorf("reading image reference %q: %w", operatorImage, err)47	}48	return tag.TagStr(), nil49}5051// entrypoint reports the path of the binary an image runs. The CLI52// image runs its binary as its entrypoint, so the first entrypoint53// element names the file to extract.54func entrypoint(cfg *v1.ConfigFile) (string, error) {55	if cfg == nil || len(cfg.Config.Entrypoint) == 0 {56		return "", fmt.Errorf("image declares no entrypoint")57	}58	return cfg.Config.Entrypoint[0], nil59}6061// binaryFromTar reads one file out of a flattened image filesystem.62// Tar entries carry paths as "./x" or "x" where the entrypoint names63// "/x", so both sides drop a leading slash and "./" before they64// compare.65func binaryFromTar(r io.Reader, path string) ([]byte, error) {66	want := strings.TrimPrefix(path, "/")67	tr := tar.NewReader(r)68	for {69		header, err := tr.Next()70		if err == io.EOF {71			break72		}73		if err != nil {74			return nil, err75		}76		name := strings.TrimPrefix(header.Name, "./")77		name = strings.TrimPrefix(name, "/")78		if name == want {79			return io.ReadAll(tr)80		}81	}82	return nil, fmt.Errorf("no file %q in the image", path)83}8485// pullBinary is the seam the tests replace. It pulls one CLI image86// for one architecture and returns its binary.87var pullBinary = remotePull8889// remotePull pulls the CLI image for the workstation's architecture,90// reads the binary its entrypoint names, and returns it.91func remotePull(ref, arch string) ([]byte, error) {92	tag, err := name.NewTag(ref)93	if err != nil {94		return nil, fmt.Errorf("reading image reference %q: %w", ref, err)95	}96	image, err := remote.Image(tag,97		remote.WithAuthFromKeychain(authn.DefaultKeychain),98		remote.WithPlatform(v1.Platform{OS: "linux", Architecture: arch}))99	if err != nil {100		return nil, fmt.Errorf("pulling %q: %w", ref, err)101	}102	cfg, err := image.ConfigFile()103	if err != nil {104		return nil, fmt.Errorf("reading the config of %q: %w", ref, err)105	}106	path, err := entrypoint(cfg)107	if err != nil {108		return nil, fmt.Errorf("%q: %w", ref, err)109	}110	fs := mutate.Extract(image)111	defer fs.Close()112	return binaryFromTar(fs, path)113}
plugins/list.go 93.6%
1package plugins23// This file reports what is installed against what the cluster runs.4// `kubectl liken plugins list` reads each installed CLI's stamped5// version and the operator version the cluster holds for that domain,6// and marks the ones that differ. It also names the operators that7// run with no local CLI, the cold-start case kubectl reports only as8// a raw "not found".910import (11	"fmt"12	"io"13	"os"14	"sort"15	"strings"16)1718// Entry pairs one domain's installed CLI version with the operator19// version the cluster holds. Either version is empty when its side20// is absent.21type Entry struct {22	Domain          string23	CLIVersion      string24	OperatorVersion string25}2627// Drift reports whether an installed CLI faces an operator of a28// different version.29func (e Entry) Drift() bool {30	return e.CLIVersion != "" && e.OperatorVersion != "" && e.CLIVersion != e.OperatorVersion31}3233// MissingCLI reports whether an operator runs with no local CLI. This34// is the cold-start case that `plugins sync` fixes.35func (e Entry) MissingCLI() bool {36	return e.CLIVersion == "" && e.OperatorVersion != ""37}3839// Installed reads the plugin directory and reports the domains whose40// CLIs are installed. A directory that does not exist yet reads as no41// domains, the state before the first sync.42func Installed(binDir string) ([]string, error) {43	entries, err := os.ReadDir(binDir)44	if os.IsNotExist(err) {45		return nil, nil46	}47	if err != nil {48		return nil, err49	}50	prefix := Name("")51	var domains []string52	for _, entry := range entries {53		if entry.IsDir() {54			continue55		}56		if domain, ok := strings.CutPrefix(entry.Name(), prefix); ok && domain != "" {57			domains = append(domains, domain)58		}59	}60	sort.Strings(domains)61	return domains, nil62}6364// Merge builds the sorted entry list from the installed CLI versions65// and the operator versions, over the union of their domains.66func Merge(installed, operators map[string]string) []Entry {67	seen := map[string]bool{}68	var domains []string69	for domain := range installed {70		if !seen[domain] {71			seen[domain] = true72			domains = append(domains, domain)73		}74	}75	for domain := range operators {76		if !seen[domain] {77			seen[domain] = true78			domains = append(domains, domain)79		}80	}81	sort.Strings(domains)82	entries := make([]Entry, 0, len(domains))83	for _, domain := range domains {84		entries = append(entries, Entry{85			Domain:          domain,86			CLIVersion:      installed[domain],87			OperatorVersion: operators[domain],88		})89	}90	return entries91}9293// Render writes the entry list, one domain per line with its two94// versions and a drift mark, and then names any operators that have95// no local CLI so the person can sync them.96func Render(entries []Entry, out io.Writer) {97	var missing []string98	for _, e := range entries {99		if e.MissingCLI() {100			missing = append(missing, e.Domain)101		}102		cli := e.CLIVersion103		if cli == "" {104			cli = "-"105		}106		operator := e.OperatorVersion107		if operator == "" {108			operator = "-"109		}110		mark := ""111		if e.Drift() {112			mark = "\tdrift"113		}114		fmt.Fprintf(out, "%s\t%s\t%s%s\n", e.Domain, cli, operator, mark)115	}116	if len(missing) > 0 {117		fmt.Fprintf(out, "no local CLI for: %s; run: liken plugins sync\n", strings.Join(missing, ", "))118	}119}
plugins/plugins.go 80.0%
1// Package plugins installs and manages the per-operator workstation2// CLIs that give the base liken binary its two-layer commands. Each3// operator ships one binary named kubectl-liken-<domain>, published4// beside its operator image at the same version. The base binary5// pulls these binaries out of the registry, keeps them in one6// directory, and reports which are installed and whether any has7// drifted from the operator it faces.8//9// The functions here hold no Kubernetes knowledge. The base binary10// reads the operators from the cluster and hands this package a list11// of domains and images. Everything below is the registry pull, the12// filesystem layout, and the comparison, so the package tests with a13// fake registry and a temporary directory and never a cluster.14package plugins1516import (17	"os"18	"path/filepath"19)2021// Operator names one operator that ships a CLI: the domain the22// cluster's plugin label gives it, and the operator image whose tag23// names the version its CLI shares.24type Operator struct {25	Domain string26	Image  string27}2829// Name returns the file name a domain's CLI installs under. kubectl30// dispatches kubectl liken <domain> by finding this name on PATH, so31// every layer of the design agrees on it.32func Name(domain string) string {33	return "kubectl-liken-" + domain34}3536// BinDir returns the directory the plugin CLIs install into, under37// the operator's home. This is the directory the base binary adds to38// PATH so kubectl and the base binary both find the CLIs.39func BinDir() (string, error) {40	home, err := os.UserHomeDir()41	if err != nil {42		return "", err43	}44	return filepath.Join(home, ".liken", "plugins", "bin"), nil45}
plugins/remove.go 100.0%
1package plugins23// Remove deletes one installed CLI. It is the step behind4// `kubectl liken plugins remove <domain>`.56import (7	"os"8	"path/filepath"9)1011// Remove deletes a domain's CLI from the plugin directory. Removing12// one that is not installed is an error that names the path, so the13// person learns the domain was never there.14func Remove(binDir, domain string) error {15	return os.Remove(filepath.Join(binDir, Name(domain)))16}
plugins/sync.go 65.9%
1package plugins23// Sync pulls each operator's CLI out of the registry and writes it4// into the plugin directory, replacing whatever was there. It is the5// step behind `kubectl liken plugins sync`, and it is idempotent:6// re-running installs the same versions over the same files.78import (9	"fmt"10	"io"11	"os"12	"path/filepath"13)1415// The one architecture every -cli image is built for. The operators16// build no other, so a workstation of another architecture has no17// binary to pull.18const builtArch = "amd64"1920// Sync installs one CLI per operator, then prints the one line to add21// the plugin directory to PATH when it is not already there. arch is22// the workstation's architecture, so the pull takes the workstation's23// binary regardless of the node's. Sync refuses an architecture with24// no build before it pulls or writes anything, because the registry's25// answer for a missing platform does not say that no build exists.26func Sync(operators []Operator, arch, binDir, pathEnv string, out io.Writer) error {27	if arch != builtArch {28		return fmt.Errorf("the operator CLIs are built for %s only, and this workstation is %s", builtArch, arch)29	}30	if err := os.MkdirAll(binDir, 0o755); err != nil {31		return err32	}33	for _, op := range operators {34		ref, err := cliImageRef(op.Image)35		if err != nil {36			return fmt.Errorf("%s: %w", op.Domain, err)37		}38		data, err := pullBinary(ref, arch)39		if err != nil {40			return fmt.Errorf("%s: %w", op.Domain, err)41		}42		if err := installBinary(binDir, op.Domain, data); err != nil {43			return err44		}45		fmt.Fprintf(out, "installed %s from %s\n", Name(op.Domain), ref)46	}47	if !onPath(pathEnv, binDir) {48		fmt.Fprintf(out, "export PATH=\"%s:$PATH\"\n", binDir)49	}50	return nil51}5253// installBinary writes one CLI to its path with mode 0755, through a54// temporary file and a rename. The rename replaces the old binary in55// one step, so a re-run never leaves a half-written file where a56// working one was.57func installBinary(binDir, domain string, data []byte) error {58	target := filepath.Join(binDir, Name(domain))59	tmp, err := os.CreateTemp(binDir, Name(domain)+".*")60	if err != nil {61		return err62	}63	tmpPath := tmp.Name()64	if _, err := tmp.Write(data); err != nil {65		tmp.Close()66		os.Remove(tmpPath)67		return err68	}69	if err := tmp.Chmod(0o755); err != nil {70		tmp.Close()71		os.Remove(tmpPath)72		return err73	}74	if err := tmp.Close(); err != nil {75		os.Remove(tmpPath)76		return err77	}78	if err := os.Rename(tmpPath, target); err != nil {79		os.Remove(tmpPath)80		return err81	}82	return nil83}8485// onPath reports whether dir is already one of the directories in a86// PATH value.87func onPath(pathEnv, dir string) bool {88	for _, entry := range filepath.SplitList(pathEnv) {89		if entry == dir {90			return true91		}92	}93	return false94}
releases/bundle.go 84.1%
1package releases23// This file bundles a release of liken: the artifacts and the4// document that every fleet, everywhere, upgrades from.5//6// The bundle holds nine artifacts. vmlinuz, the system image7// (liken.sqfs, the OS as a read-only filesystem that a machine mounts8// as its root), and the boot archive (boot.cpio, the small initramfs9// that the boot loader stages) are the operating system, with no10// deployment inside. microcode.cpio is the CPU microcode early cpio:11// boot entries name it as their first initrd, ahead of everything,12// at the point where the kernel looks for microcode before it13// decompresses anything (the microcode domain explains the format).14// It is vendored with its own pin, so an OS update never recomposes15// it. The liken binary is the toolkit that turns the artifacts into16// a deployment's bootable media without this repo or a build.17// systemd-bootx64.efi is the menu program that media boots (the18// systemd-boot domain explains why a stick needs one when installed19// machines do not). grub-boot.img and grub-core.img are the two20// halves of the bootloader that BIOS machines carry (the grub domain21// explains why they need one when UEFI machines do not). These stay22// inert on a UEFI machine; on a BIOS machine, init writes their bytes23// into the MBR and the biosBoot partition. LICENSES.md is the24// third-party notices that the licensing domain assembles: several of25// the artifacts carry other projects' GPL- and LGPL-licensed26// binaries, and the license terms of those projects require the27// notices to travel with the bytes. So the notices are an artifact28// like any other, and use the same channel, sticks, and slots the29// binaries do. Because nothing here embeds a deployment, every digest30// stays stable for a given source tree. This makes it publishable on31// a release page, and the same for everyone.32//33// That stability is what lets machines upgrade straight from the34// public channel. A deployment pins the release document's digest in35// its Cluster's spec.releases.catalog (the entry this bundle36// prints). Its machines verify the document against the catalog, and37// verify the artifacts against the document. Each machine supplies38// the one thing the release cannot: its own deployment layer,39// carried forward from the slot it is running on. Nothing is40// composed or hosted for one deployment specifically.4142import (43	"crypto/sha256"44	"fmt"45	"io"46	"os"47	"path/filepath"4849	"github.com/liken-sh/liken/liken/api"50	"github.com/liken-sh/liken/liken/machine"51)5253// slotHeadroom is the room a boot slot must keep beyond the release54// artifacts: the deployment layer and its sidecar (carried between55// slots, never part of a release), the release document, and FAT32's56// own tables. 64 MiB is generous for all of them together, so the57// budget check below fails before a real slot would.58const slotHeadroom = 64 << 205960// Bundle lays out a release: it copies the named artifacts into61// <channel>/<version>/, beside a release.yaml that names each one by62// sha256 digest and size, and records which upstream components63// shipped. The version must fit liken's calendar grammar (the api64// package defines it). Bundle enforces this here, where versions are65// authored, so a malformed version never reaches a channel at all.66//67// slotSize declares the boot slot the fleet's machines format (the68// manifests' systemA and systemB roles). Every artifact lands on the69// slot at install and at upgrade, so their sum plus slotHeadroom70// must fit, and a release that outgrows its slots fails right here,71// at build time, rather than surprising someone at install time.72func Bundle(vmlinuz, systemImage, bootArchive, microcode, cli, bootMenu, grubBoot, grubCore, licenses, channelDir, version, slotSize string, components []machine.ReleaseComponent, out io.Writer) error {73	if err := api.ValidVersion(version); err != nil {74		return err75	}76	slotBytes, err := machine.ParseSize(slotSize)77	if err != nil {78		return fmt.Errorf("slot size: %w", err)79	}80	dest := filepath.Join(channelDir, version)81	if err := os.RemoveAll(dest); err != nil {82		return err83	}84	if err := os.MkdirAll(dest, 0o755); err != nil {85		return err86	}8788	// The artifacts keep canonical names in the channel, whatever89	// their build-tree paths were, so the document and the URLs stay90	// the same for every release.91	sources := []struct {92		src, name string93	}{94		{vmlinuz, "vmlinuz"},95		{systemImage, "liken.sqfs"},96		{bootArchive, "boot.cpio"},97		{microcode, "microcode.cpio"},98		{cli, "liken"},99		{bootMenu, "systemd-bootx64.efi"},100		{grubBoot, "grub-boot.img"},101		{grubCore, "grub-core.img"},102		{licenses, "LICENSES.md"},103	}104	document := fmt.Sprintf("apiVersion: liken.sh/v1alpha1\nkind: Release\nmetadata:\n  name: %s\nartifacts:\n", version)105	var slotPayload uint64106	for _, s := range sources {107		dst := filepath.Join(dest, s.name)108		if err := copyFile(s.src, dst); err != nil {109			return err110		}111		digest, err := fileSHA256(dst)112		if err != nil {113			return err114		}115		info, err := os.Stat(dst)116		if err != nil {117			return err118		}119		slotPayload += uint64(info.Size())120		document += fmt.Sprintf("  - name: %s\n    sha256: %s\n    size: %d\n", s.name, digest, info.Size())121	}122123	// The budget check the doc comment describes. It runs after the124	// copies so the message can state real sizes, and it removes the125	// oversized bundle so nothing serves it by accident.126	if slotPayload+slotHeadroom > slotBytes {127		os.RemoveAll(dest)128		return fmt.Errorf(129			"the release's artifacts total %s, and with %s of slot headroom they outgrow the %s boot slots; shrink the image or grow the fleet's slots first",130			humanSize(int64(slotPayload)), humanSize(int64(slotHeadroom)), slotSize)131	}132133	// The components record what shipped inside those artifacts. The134	// version above is a calendar date, and deliberately says nothing135	// about contents. So the document is where a reader learns which136	// kernel or k3s this release carries.137	if len(components) > 0 {138		document += "components:\n"139		for _, c := range components {140			document += fmt.Sprintf("  - name: %s\n    version: %s\n", c.Name, c.Version)141		}142	}143144	// Bundle generates the document from the copies in the channel,145	// the bytes that will actually be served, so the two can never146	// disagree.147	if err := os.WriteFile(filepath.Join(dest, "release.yaml"), []byte(document), 0o644); err != nil {148		return err149	}150151	// The channel document at the root names the newest release152	// present, so a cluster can discover that something newer exists153	// without listing the channel (the machine package explains why154	// the document is advisory, and never trusted).155	latest, err := writeChannelDocument(channelDir)156	if err != nil {157		return err158	}159160	if err := report(dest, version, out); err != nil {161		return err162	}163	fmt.Fprintf(out, "\nchannel.yaml names the latest release: %s\n", latest)164165	// The document's own digest is the root of the trust chain. A166	// person verifies a download against it. A deployment commits it167	// to its Cluster, so the fleet can verify what it fetches.168	digest, err := fileSHA256(filepath.Join(dest, "release.yaml"))169	if err != nil {170		return err171	}172	fmt.Fprintf(out, "\ncatalog entry for a Cluster's spec.releases.catalog:\n")173	fmt.Fprintf(out, "  - version: %s\n", version)174	fmt.Fprintf(out, "    digest: sha256:%s\n", digest)175	return nil176}177178// writeChannelDocument rewrites the channel's root document to name179// the newest release present. It computes the newest from what is180// actually in the channel, not from the version just bundled, so181// re-bundling an old release never moves the channel backwards. The182// zero-padded calendar grammar makes plain string order match the183// version order (releases/versioning.md), so the maximum string is184// the latest version.185func writeChannelDocument(channelDir string) (string, error) {186	entries, err := os.ReadDir(channelDir)187	if err != nil {188		return "", err189	}190	latest := ""191	for _, e := range entries {192		if !e.IsDir() || api.ValidVersion(e.Name()) != nil {193			continue194		}195		if e.Name() > latest {196			latest = e.Name()197		}198	}199	document := fmt.Sprintf("apiVersion: liken.sh/v1alpha1\nkind: Channel\nmetadata:\n  name: liken\nlatest: %s\n", latest)200	return latest, os.WriteFile(filepath.Join(channelDir, "channel.yaml"), []byte(document), 0o644)201}202203// report prints what landed in the channel.204func report(dest, version string, out io.Writer) error {205	entries, err := os.ReadDir(dest)206	if err != nil {207		return err208	}209	fmt.Fprintf(out, "\npublished liken %s to %s:\n", version, dest)210	for _, e := range entries {211		info, err := e.Info()212		if err != nil {213			return err214		}215		fmt.Fprintf(out, "  %8s  %s\n", humanSize(info.Size()), e.Name())216	}217	return nil218}219220func copyFile(src, dst string) error {221	data, err := os.ReadFile(src)222	if err != nil {223		return err224	}225	return os.WriteFile(dst, data, 0o644)226}227228func fileSHA256(path string) (string, error) {229	raw, err := os.ReadFile(path)230	if err != nil {231		return "", err232	}233	return fmt.Sprintf("%x", sha256.Sum256(raw)), nil234}235236// humanSize renders a byte count the way a person scans a listing,237// in the unit that keeps the number small.238func humanSize(n int64) string {239	switch {240	case n >= 1<<30:241		return fmt.Sprintf("%.1fG", float64(n)/(1<<30))242	case n >= 1<<20:243		return fmt.Sprintf("%.0fM", float64(n)/(1<<20))244	case n >= 1<<10:245		return fmt.Sprintf("%.0fK", float64(n)/(1<<10))246	}247	return fmt.Sprintf("%dB", n)248}
releases/download.go 93.8%
1package releases23// This file bounds how long a download from a release channel can4// stall. A machine's operator downloads releases onto a boot slot,5// and a workstation downloads them to build install media. Both read6// the same documents and artifacts through Get.7//8// net/http bounds the dial and the TLS handshake, but nothing bounds9// the wait for the response headers or for the next byte of the body.10// A server that accepts the request and then stops sending keeps the11// connection open, and the download waits on it forever. On a machine12// this holds the operator's one release writer, so every later13// upgrade waits too. So Get cancels a request when no byte arrives14// for StallLimit, from the request to the headers and between any two15// reads of the body.16//17// Get bounds progress, not the whole transfer. An artifact is a few18// hundred megabytes, and a machine on a slow link takes as long as it19// takes. A deadline for the whole transfer would fail a download that20// makes steady progress, and it would retry from the start of the21// artifact each time.2223import (24	"context"25	"errors"26	"fmt"27	"io"28	"net/http"29	"time"30)3132// StallLimit is how long a download can wait for its next byte. A33// minute covers a busy server that is slow to answer and the34// retransmissions of a lossy link. A link that delivers nothing for a35// minute is broken, and a new request is the only way past a server36// that stopped sending.37const StallLimit = time.Minute3839// errStalled is the cause of a request that Get cancelled because no40// byte arrived for StallLimit.41var errStalled = fmt.Errorf("no bytes arrived for %v", StallLimit)4243// Get sends a GET for url through client and returns the response44// when the server answers 200. The request ends when ctx ends, or45// when no byte arrives for StallLimit. The caller closes the body.46func Get(ctx context.Context, client *http.Client, url string) (*http.Response, error) {47	ctx, cancel := context.WithCancelCause(ctx)48	stall := time.AfterFunc(StallLimit, func() { cancel(errStalled) })49	stop := func() {50		stall.Stop()51		cancel(nil)52	}5354	req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)55	if err != nil {56		stop()57		return nil, err58	}59	resp, err := client.Do(req)60	if err != nil {61		stop()62		return nil, causeOf(ctx, err)63	}64	if resp.StatusCode != http.StatusOK {65		resp.Body.Close()66		stop()67		return nil, fmt.Errorf("the server answered %s", resp.Status)68	}69	stall.Reset(StallLimit)70	resp.Body = &progressBody{body: resp.Body, ctx: ctx, stall: stall, stop: stop}71	return resp, nil72}7374// causeOf reports why a request failed. A request that Get or the75// caller cancelled fails with the transport's "context canceled", and76// the cause says which of the two happened.77func causeOf(ctx context.Context, err error) error {78	if cause := context.Cause(ctx); cause != nil && !errors.Is(err, cause) {79		return fmt.Errorf("%w (%v)", cause, err)80	}81	return err82}8384// progressBody restarts the stall timer each time a read returns85// bytes.86type progressBody struct {87	body  io.ReadCloser88	ctx   context.Context89	stall *time.Timer90	stop  func()91}9293func (b *progressBody) Read(p []byte) (int, error) {94	n, err := b.body.Read(p)95	if n > 0 {96		b.stall.Reset(StallLimit)97	}98	if err != nil && err != io.EOF {99		err = causeOf(b.ctx, err)100	}101	return n, err102}103104func (b *progressBody) Close() error {105	b.stop()106	return b.body.Close()107}
releases/fetch.go 87.2%
1package releases23// This file downloads a release from a channel: the workstation half4// of the trust chain the machines already walk.5//6// A machine's operator downloads releases onto a boot slot. This7// fetch downloads the same releases onto an operator's workstation,8// where they become install media. The verification discipline is9// identical, because the threat is identical: the bytes crossed a10// network, and nothing about the transport is trusted. Fetch11// downloads the document first, and, when the caller pins a digest,12// checks it against that digest before a single artifact moves.13// Fetch verifies every artifact against the document before the14// artifact takes its final name. The document itself lands last, so15// a directory holding release.yaml is a directory whose artifacts16// were complete when release.yaml was written. This last property17// also makes the download resumable and safe to cache: a rerun18// verifies whatever already landed and fetches only what fails, so19// an interrupted download converges instead of staying half-finished20// under a name that looks complete.21//22// The digest pin is optional here and required on machines, and the23// difference is who holds the catalog. A machine always has its24// cluster's spec.releases.catalog to vouch for the document. A25// workstation composing a deployment's first media has no cluster26// yet, so the pin is offered (paste it from the release's published27// catalog entry) rather than required. What is never optional is the28// inner chain: no artifact escapes verification against the document29// that names it.3031import (32	"context"33	"crypto/sha256"34	"fmt"35	"io"36	"net/http"37	"os"38	"path/filepath"39	"strings"4041	"github.com/liken-sh/liken/liken/machine"42)4344// Fetch downloads one release from a channel's source URL into45// <channelDir>/<version>/, the same layout that Bundle produces and46// that the media and stick builders consume. The version "latest"47// resolves through the channel's advisory document. digest, when not48// empty, pins the release document's own bytes ("sha256:<hex>", the49// catalog-entry form).50func Fetch(source, version, digest, channelDir string, out io.Writer) error {51	if out == nil {52		out = io.Discard53	}54	source = strings.TrimSuffix(source, "/")5556	if version == "latest" {57		resolved, err := resolveLatest(source)58		if err != nil {59			return err60		}61		fmt.Fprintf(out, "the channel's latest release is %s\n", resolved)62		version = resolved63	}6465	base := source + "/" + version66	raw, err := fetchDocument(base + "/release.yaml")67	if err != nil {68		return fmt.Errorf("fetching the release document: %w", err)69	}7071	// This is the trust chain's first link, when the caller holds one:72	// the document's bytes must hash to exactly what the catalog entry73	// promised. Until that holds true, nothing the document says74	// counts.75	if digest != "" {76		if got := fmt.Sprintf("sha256:%x", sha256.Sum256(raw)); got != digest {77			return fmt.Errorf("the release document's digest %s does not match the pinned %s", got, digest)78		}79	}80	release, err := machine.ParseRelease(raw)81	if err != nil {82		return fmt.Errorf("the release document does not parse: %w", err)83	}84	if release.Metadata.Name != version {85		return fmt.Errorf("the release document names version %s, not %s", release.Metadata.Name, version)86	}8788	dest := filepath.Join(channelDir, version)89	if err := os.MkdirAll(dest, 0o755); err != nil {90		return err91	}92	fetched := 093	for _, artifact := range release.Artifacts {94		path := filepath.Join(dest, artifact.Name)95		if verifyFile(artifact, path) == nil {96			continue // already here, byte for byte, from an earlier run97		}98		fmt.Fprintf(out, "  fetching %s (%s)\n", artifact.Name, humanSize(artifact.Size))99		if err := fetchArtifact(base, artifact, path); err != nil {100			return err101		}102		fetched++103	}104105	// The document lands after every artifact it describes. This makes106	// the directory self-describing: it records which release it107	// holds, byte for byte, without asking the network again.108	if err := os.WriteFile(filepath.Join(dest, "release.yaml"), raw, 0o644); err != nil {109		return err110	}111	fmt.Fprintf(out, "%s: %d fetched, the rest already verified in place\n", version, fetched)112	return nil113}114115// resolveLatest asks the channel's root document what the newest116// published release is. The document is advisory, outside the trust117// chain, which is exactly right here: it only chooses which version118// to fetch, and Fetch still verifies everything fetched against that119// version's own document.120func resolveLatest(source string) (string, error) {121	raw, err := fetchDocument(source + "/channel.yaml")122	if err != nil {123		return "", fmt.Errorf("fetching the channel document: %w", err)124	}125	channel, err := machine.ParseChannel(raw)126	if err != nil {127		return "", fmt.Errorf("the channel document does not parse: %w", err)128	}129	return channel.Latest, nil130}131132// fetchArtifact streams one artifact into the channel directory: it133// downloads to a .partial name, verifies against the document, then134// renames the file. This verify-before-rename order means a135// final-looking name never points at unverified bytes, which is what136// lets reruns trust whatever they find already in place.137func fetchArtifact(base string, artifact machine.ReleaseArtifact, dest string) error {138	resp, err := Get(context.Background(), http.DefaultClient, base+"/"+artifact.Name)139	if err != nil {140		return fmt.Errorf("fetching %s: %w", artifact.Name, err)141	}142	defer resp.Body.Close()143144	tmp := dest + ".partial"145	f, err := os.Create(tmp)146	if err != nil {147		return err148	}149	// The size cap catches an artifact that runs past its declared150	// size, without downloading the rest of it.151	_, err = io.Copy(f, io.LimitReader(resp.Body, artifact.Size+1))152	if closeErr := f.Close(); err == nil {153		err = closeErr154	}155	if err != nil {156		os.Remove(tmp)157		return fmt.Errorf("writing %s: %w", artifact.Name, err)158	}159160	if err := verifyFile(artifact, tmp); err != nil {161		os.Remove(tmp)162		return fmt.Errorf("%s from the server does not verify: %w", artifact.Name, err)163	}164	return os.Rename(tmp, dest)165}166167// verifyFile checks one downloaded file against its artifact's digest168// and size. It returns an error for any reason the check fails,169// including that the file does not exist, the common case on a first170// run.171func verifyFile(artifact machine.ReleaseArtifact, path string) error {172	f, err := os.Open(path)173	if err != nil {174		return err175	}176	defer f.Close()177	return artifact.Verify(f)178}179180// fetchDocument GETs a small document whole. The 1MiB cap is far181// larger than any reasonable release or channel document, and small182// enough to hold in memory without concern.183func fetchDocument(url string) ([]byte, error) {184	resp, err := Get(context.Background(), http.DefaultClient, url)185	if err != nil {186		return nil, err187	}188	defer resp.Body.Close()189	return io.ReadAll(io.LimitReader(resp.Body, 1<<20))190}
releases/index.go 86.0%
1package releases23// This file renders the channel's index pages: a front page listing4// every published release, a page for each release, and a page for5// the source mirror.6//7// Object storage cannot build these pages. The bucket's own ACL is8// private while each object is public-read, so a named path downloads9// anonymously and the root refuses to say what exists. The hostname10// class the channel answers on serves an index document for any11// prefix, and generates no listing of its own. So the pages are12// objects like any other, and something has to write them.13//14// Nothing here is new information. A release document already names15// every artifact with its digest and size, and records the version of16// each component inside. The version list is the set of top-level17// prefixes. Each page is therefore a render of a document the channel18// already serves, which makes the whole tree derived: no digest names19// a page, no machine reads one, and deleting them all would cost the20// channel nothing but its readability. That is what makes a rerun21// safe, and it is why one command both backfills the releases that22// were published before these pages existed and keeps up with each23// new one.24//25// Listing the bucket needs a credential and rendering does not, so26// the two are split at the key list: the caller supplies the keys,27// and this code fetches only public documents. The S3 protocol28// therefore stays out of the toolkit that ships to operators, and a29// test drives the renderer from a list it writes by hand.3031import (32	"crypto/sha256"33	"embed"34	"fmt"35	"html/template"36	"io"37	"maps"38	"os"39	"path/filepath"40	"slices"41	"strings"4243	"github.com/liken-sh/brand"4445	"github.com/liken-sh/liken/liken/api"46	"github.com/liken-sh/liken/liken/machine"47)4849//go:embed page.html.tmpl index.html.tmpl release.html.tmpl sources.html.tmpl50var pageTemplates embed.FS5152// Index renders a channel's pages into outDir. keys are the channel's53// object keys, one per line as its storage reports them, and source54// is the base URL the channel answers on. The pages link from the55// channel's root, so outDir's contents belong at the root of whatever56// serves them.57func Index(source string, keys []string, outDir string, out io.Writer) error {58	if out == nil {59		out = io.Discard60	}61	source = strings.TrimSuffix(source, "/")6263	versions, notes, sources := readKeys(keys)64	if len(versions) == 0 {65		return fmt.Errorf("the key list names no release")66	}6768	// The channel document names which release is marked latest,69	// rather than the highest version in the list. The document is what70	// a polling cluster reads, so a page that disagreed with it would71	// be telling an operator something no machine acts on.72	latest, err := resolveLatest(source)73	if err != nil {74		return err75	}7677	channel := &channelView{page: newPage("liken releases"), Source: source, Sources: sources}78	for _, version := range versions {79		release, err := releaseView(source, version, notes[version])80		if err != nil {81			return err82		}83		release.IsLatest = version == latest84		release.HasSources = len(sources) > 085		channel.Releases = append(channel.Releases, release)86	}8788	pages, err := template.ParseFS(pageTemplates, "*.html.tmpl")89	if err != nil {90		return err91	}92	if err := writePage(pages, "index.html.tmpl", channel, outDir, "index.html"); err != nil {93		return err94	}95	for _, release := range channel.Releases {96		if err := writePage(pages, "release.html.tmpl", release, outDir, release.Version, "index.html"); err != nil {97			return err98		}99	}100	// A channel with no mirror gets no sources page, because an empty101	// page under a license obligation reads as an offer with nothing102	// behind it.103	if len(sources) > 0 {104		mirror := &channelView{page: newPage("liken sources"), Source: source, Sources: sources}105		if err := writePage(pages, "sources.html.tmpl", mirror, outDir, "sources", "index.html"); err != nil {106			return err107		}108	}109110	// The versions document is the front page's machine-readable twin,111	// written from the same releases the pages were rendered from112	// (versions.go).113	document := versionsDocument(latest, channel.Releases)114	if err := os.WriteFile(filepath.Join(outDir, "versions.yaml"), document, 0o644); err != nil {115		return err116	}117118	fmt.Fprintf(out, "%d releases indexed, latest %s\n", len(channel.Releases), latest)119	return nil120}121122// page is what every template needs, whichever page it renders. The123// stylesheet and the mark are inlined rather than linked: a channel124// page must render whole with no request to any other site, so a125// page here fetches nothing (the stylesheet section of126// brand/README.md says more).127type page struct {128	Title      string129	Stylesheet template.CSS130	Mark       template.HTML131}132133func newPage(title string) page {134	return page{135		Title:      title,136		Stylesheet: template.CSS(brand.Stylesheet),137		Mark:       template.HTML(brand.Mark),138	}139}140141// channelView is the front page's and the sources page's input.142type channelView struct {143	page144	Source   string145	Releases []*releaseInfo146	Sources  []*sourceComponent147}148149// releaseInfo is one release's page, and one row of the front page.150type releaseInfo struct {151	page152	Version    string153	Digest     string154	Artifacts  []artifactInfo155	Components []machine.ReleaseComponent156	IsLatest   bool157	HasSources bool158	// Notes is the release's changes list, from the channel's own159	// notes.md beside the release. The notes are announcement prose:160	// no digest pins them and no machine reads them, which is what161	// lets a release published before notes existed gain them later.162	Notes string163}164165// bareTagsFrom is the first release whose git tag is the bare166// version. The tags before it carry a v prefix, and tags never move,167// so both forms exist on GitHub permanently. Every field of a168// version is zero-padded, so CompareVersions places any version on169// the correct side of this boundary.170const bareTagsFrom = "2026.08.18-002"171172// Tag gives the git tag that names this release on GitHub, in173// whichever form GitHub holds it. The pages are rewritten on every174// release, so this link is constructed here, not recorded at publish175// time, and it must stay correct for old releases and new ones176// alike.177func (r *releaseInfo) Tag() string {178	if api.CompareVersions(r.Version, bareTagsFrom) < 0 {179		return "v" + r.Version180	}181	return r.Version182}183184// Component gives one component's version for the front page's185// columns, and an empty string when the release carries no component186// by that name. An older release predates a component that a newer187// one records, and a blank cell is the honest way to show that.188func (r *releaseInfo) Component(name string) string {189	for _, component := range r.Components {190		if component.Name == name {191			return component.Version192		}193	}194	return ""195}196197type artifactInfo struct {198	Name   string199	SHA256 string200	Human  string201}202203type sourceComponent struct {204	Name     string205	Versions []*sourceVersion206}207208type sourceVersion struct {209	Version string210	Files   []sourceFile211}212213type sourceFile struct {214	Name string // the file's own name, for the link's text215	Path string // <component>/<version>/<name>, under /sources/216}217218// readKeys sorts a channel's object keys into the releases it holds,219// the releases that carry notes, and the source mirror. Anything else220// on the channel, the channel document and the pages themselves221// included, belongs to no listing and is skipped. A key whose first222// segment does not parse as a version is not a release, which is what223// keeps a stray prefix from becoming a page.224func readKeys(keys []string) ([]string, map[string]bool, []*sourceComponent) {225	versions := map[string]bool{}226	notes := map[string]bool{}227	components := map[string]map[string][]sourceFile{}228	for _, key := range keys {229		segments := strings.Split(strings.TrimPrefix(key, "/"), "/")230		switch {231		case len(segments) == 2 && api.ValidVersion(segments[0]) == nil:232			versions[segments[0]] = true233			if segments[1] == "notes.md" {234				notes[segments[0]] = true235			}236		case len(segments) == 4 && segments[0] == "sources":237			component, version, name := segments[1], segments[2], segments[3]238			if components[component] == nil {239				components[component] = map[string][]sourceFile{}240			}241			components[component][version] = append(components[component][version],242				sourceFile{Name: name, Path: component + "/" + version + "/" + name})243		}244	}245246	// Newest first: the release an operator wants is nearly always one247	// of the last few, and a rollback needs the one below the top.248	ordered := slices.Collect(maps.Keys(versions))249	slices.SortFunc(ordered, func(a, b string) int { return api.CompareVersions(b, a) })250251	// Everything else sorts by name, so that the same channel renders252	// the same bytes on every run.253	var sources []*sourceComponent254	for _, name := range slices.Sorted(maps.Keys(components)) {255		component := &sourceComponent{Name: name}256		for _, version := range slices.Sorted(maps.Keys(components[name])) {257			files := components[name][version]258			slices.SortFunc(files, func(a, b sourceFile) int { return strings.Compare(a.Name, b.Name) })259			component.Versions = append(component.Versions, &sourceVersion{Version: version, Files: files})260		}261		sources = append(sources, component)262	}263	return ordered, notes, sources264}265266// releaseView fetches one release's document from the channel and267// arranges it for the page. Fetching over the public URL is also a268// check: a release the channel does not serve gets no page, and the269// error names it. withNotes says the key list saw a notes object270// beside this release, so a fetch that fails then is the same271// disagreement between the listing and the channel, not an old272// release from before notes existed.273func releaseView(source, version string, withNotes bool) (*releaseInfo, error) {274	raw, err := fetchDocument(source + "/" + version + "/release.yaml")275	if err != nil {276		return nil, fmt.Errorf("fetching %s's release document: %w", version, err)277	}278	release, err := machine.ParseRelease(raw)279	if err != nil {280		return nil, fmt.Errorf("%s's release document does not parse: %w", version, err)281	}282	if release.Metadata.Name != version {283		return nil, fmt.Errorf("the document under %s names version %s", version, release.Metadata.Name)284	}285286	info := &releaseInfo{287		page: newPage("liken " + version),288		// The digest is computed from the document's exact bytes, the289		// same value the release workflow prints and an operator pins290		// in a catalog entry. It is computed here and never read from291		// the channel, because a digest the channel supplied would292		// vouch for nothing.293		Version:    version,294		Digest:     fmt.Sprintf("sha256:%x", sha256.Sum256(raw)),295		Components: release.Components,296	}297	for _, artifact := range release.Artifacts {298		info.Artifacts = append(info.Artifacts, artifactInfo{299			Name:   artifact.Name,300			SHA256: artifact.SHA256,301			Human:  humanSize(artifact.Size),302		})303	}304	if withNotes {305		raw, err := fetchDocument(source + "/" + version + "/notes.md")306		if err != nil {307			return nil, fmt.Errorf("fetching %s's notes: %w", version, err)308		}309		info.Notes = strings.TrimSpace(string(raw))310	}311	return info, nil312}313314// writePage renders one template to one path under the output315// directory.316func writePage(pages *template.Template, name string, data any, outDir string, parts ...string) error {317	path := filepath.Join(append([]string{outDir}, parts...)...)318	if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {319		return err320	}321	f, err := os.Create(path)322	if err != nil {323		return err324	}325	if err := pages.ExecuteTemplate(f, name, data); err != nil {326		f.Close()327		return err328	}329	return f.Close()330}
releases/serve.go 88.2%
1// Package releases produces and operates the release channel: the2// layout a release server publishes (dist/<version>/ holding the3// artifacts and the release.yaml that names them by digest), and the4// server that exposes it.5//6// There is one kind of channel, and it is public: the generic OS7// (vmlinuz, the generic liken.cpio, the toolkit binary) with no8// deployment inside. Every fleet upgrades from it directly. Each9// machine carries its own deployment layer between its slots, so the10// channel never composes or hosts anything specific to one11// deployment. The server here stands in for the releases on the12// liken.sh website. A deployment's choices live on its machines and13// in its cluster's API, not on a server.14package releases1516// This file implements the release server: a dist/ tree over HTTP,17// with every request logged.18//19// A real release server is nothing more than this: static files20// under stable URLs. The trust chain deliberately asks nothing of21// the transport. A machine verifies every byte it downloads against22// a digest it got from the cluster's API, so the server needs no23// authentication, no TLS, and no logic of its own to be safe to24// upgrade from. (Integrity comes from the digests. Privacy would25// need TLS, but there is nothing private about an OS image.)26//27// The logging is why this file exists instead of something like28// python3 -m http.server. During upgrade drills, a person watches29// this terminal to see machines fetch: which release, which30// artifact, how many bytes. A stalled or repeated download is31// visible the moment it happens.3233import (34	"fmt"35	"log"36	"net"37	"net/http"38	"time"39)4041// Serve exposes a channel directory the way a release server would42// and blocks until the listener fails.43func Serve(dir, addr string) error {44	fmt.Println(banner(dir, addr))45	return http.ListenAndServe(addr, handler(dir))46}4748// handler serves the published releases under /releases/, mirroring49// the source URL the Cluster's spec declares, and logs one line for50// each request. It is separate from Serve so the tests can run the51// server against a throwaway directory.52func handler(dir string) http.Handler {53	files := http.StripPrefix("/releases/", http.FileServer(http.Dir(dir)))54	return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {55		start := time.Now()56		recorder := &responseRecorder{ResponseWriter: w, status: http.StatusOK}57		files.ServeHTTP(recorder, r)58		log.Printf("%s %s -> %d (%d bytes, %s)",59			r.Method, r.URL.Path, recorder.status, recorder.bytes,60			time.Since(start).Round(time.Millisecond))61	})62}6364// responseRecorder wraps a ResponseWriter to remember what was sent.65// The standard library's handlers write directly to the socket, so66// the only way to log a response is to wrap the writer and watch what67// passes through it.68type responseRecorder struct {69	http.ResponseWriter70	status int71	bytes  int6472}7374func (r *responseRecorder) WriteHeader(status int) {75	r.status = status76	r.ResponseWriter.WriteHeader(status)77}7879func (r *responseRecorder) Write(p []byte) (int, error) {80	n, err := r.ResponseWriter.Write(p)81	r.bytes += int64(n)82	return n, err83}8485// banner is the startup line: where the releases come from, where the86// server listens, and how a guest reaches it. QEMU's user-mode87// networking presents the host's loopback to every guest as88// 10.0.2.2, so the hint spells out the exact URL that a machine's89// release source points at, derived from whatever port the server90// was given. An address with no port cannot produce the hint, so91// banner omits it.92func banner(dir, addr string) string {93	_, port, err := net.SplitHostPort(addr)94	if err != nil {95		return fmt.Sprintf("serving releases from %s on %s", dir, addr)96	}97	return fmt.Sprintf("serving releases from %s on %s (guests reach this at http://10.0.2.2:%s/releases)", dir, addr, port)98}
releases/versions.go 100.0%
1package releases23// The versions document is the channel's list of every release it4// holds, in one file that a script can read.5//6// The channel document (machine/channel.go) names one version, the7// newest, and that is all a polling cluster needs. A person often8// wants the other question answered: what else is here? A script9// that lists releases needs the same answer.10// Object storage will not answer it, because the bucket refuses to11// list itself, so the answer has to be published like anything else.12// This document is the machine-readable twin of the front page, and13// both come out of the same index run.14//15// It is a separate document rather than a field on the channel16// document for one reason: every cluster in the fleet fetches17// channel.yaml on a poll, and that document must stay one small,18// fixed size no matter how many releases have ever been published.19//20// The digests here are a convenience and not an authority. A digest21// that the channel served vouches for nothing on its own, which is22// why adopting a release means an operator pins the digest in their23// own Cluster document. This file makes that entry easy to find and24// easy to diff. It does not make the channel trusted.2526import (27	"fmt"28	"strings"2930	"github.com/liken-sh/liken/liken/api"31)3233// A Versions document lists every release a channel serves, newest34// first.35type Versions struct {36	APIVersion string         `json:"apiVersion"`37	Kind       string         `json:"kind"`38	Metadata   api.ObjectMeta `json:"metadata"`39	Latest     string         `json:"latest"`40	Releases   []VersionEntry `json:"releases"`41}4243// A VersionEntry is one release, in the shape of the catalog entry44// that adopts it. The field names match a Cluster's45// spec.releases.catalog, so an entry copies straight across.46type VersionEntry struct {47	Version string `json:"version"`48	Digest  string `json:"digest"`49}5051// versionsDocument arranges the indexed releases as the document the52// channel publishes at /versions.yaml.53//54// The text is written out rather than marshalled, the same way the55// release and channel documents are (bundle.go). A marshaller sorts56// the keys, which would put the digest above the version it belongs57// to. A person reads this file, and the entries are meant to be58// copied straight into a Cluster's spec.releases.catalog, so they59// have to appear in the order that catalog uses.60func versionsDocument(latest string, releases []*releaseInfo) []byte {61	var out strings.Builder62	fmt.Fprintf(&out, "apiVersion: %s\nkind: Versions\nmetadata:\n  name: liken\n", api.APIVersion)63	fmt.Fprintf(&out, "latest: %s\nreleases:\n", latest)64	for _, release := range releases {65		fmt.Fprintf(&out, "  - version: %s\n    digest: %s\n", release.Version, release.Digest)66	}67	return []byte(out.String())68}
scaffold/scaffold.go 80.4%
1// Package scaffold starts a deployment from answers. `liken new` asks2// a short series of plain questions, and writes the deployment3// directory: cluster.yaml and one machine manifest per machine. The4// rest of the toolkit (mint, layer, stick) builds on that directory.5//6// The generated documents are the documentation. Each one carries the7// teaching comments that a person needs to change it later, adapted8// from the dev cluster's manifests. A scaffold that writes an invalid9// document would fail its user at first boot. Because of this, this10// package parses everything it generates back through the same11// strict parsers that machines use, before it writes a single file.12// A failure there is a bug in this package, and the error message13// says so.14package scaffold1516import (17	"bufio"18	"embed"19	"fmt"20	"io"21	"net/netip"22	"os"23	"path/filepath"24	"slices"25	"strings"26	"text/template"2728	"github.com/liken-sh/liken/liken/cluster"29	"github.com/liken-sh/liken/liken/machine"30)3132//go:embed cluster.yaml.tmpl machine.yaml.tmpl33var templates embed.FS3435// answers holds everything the questions collect. It is also the36// templates' input.37type answers struct {38	ClusterName string39	Machines    []machineAnswers40	Leaders     []string41	Endpoint    string42	NodeCIDR    string43	Upstreams   []string44	Features    []string45	Source      string4647	// This is the fleet-wide hardware shape. The interview asks for48	// interface names and the disk layout once, and writes them into49	// every manifest, where a person can edit them per machine50	// afterward.51	UplinkNIC    string // empty means single-NIC machines52	ClusterNIC   string53	Gateway      string // single-NIC only: no DHCP to supply a route54	Nameservers  []string55	Disks        []string // 1 or 3 devices56	RebootPolicy string57}5859type machineAnswers struct {60	Name    string61	Address string // CIDR form, inside NodeCIDR62}6364// New runs the questions against in/out, and writes the deployment65// directory. It refuses a directory that already has a cluster.yaml,66// because scaffolding is for starting a deployment, not overwriting67// one.68func New(dir string, in io.Reader, out io.Writer) error {69	if _, err := os.Stat(filepath.Join(dir, "cluster.yaml")); err == nil {70		return fmt.Errorf("%s already has a cluster.yaml; the scaffold only starts new deployments", dir)71	}7273	a, err := interview(bufio.NewScanner(in), out, filepath.Base(strings.TrimRight(dir, "/")))74	if err != nil {75		return err76	}7778	clusterYAML, err := render("cluster.yaml.tmpl", a)79	if err != nil {80		return err81	}82	if _, err := cluster.ParseCluster(clusterYAML); err != nil {83		return fmt.Errorf("this is a scaffold bug: the generated cluster.yaml does not parse: %w", err)84	}8586	machines := map[string][]byte{}87	for _, m := range a.Machines {88		doc, err := render("machine.yaml.tmpl", struct {89			answers90			Machine machineAnswers91		}{a, m})92		if err != nil {93			return err94		}95		parsed, err := machine.Parse(doc)96		if err != nil {97			return fmt.Errorf("this is a scaffold bug: %s's manifest does not parse: %w", m.Name, err)98		}99		if err := parsed.Spec.Storage.Validate(); err != nil {100			return fmt.Errorf("this is a scaffold bug: %s's storage does not validate: %w", m.Name, err)101		}102		if err := parsed.Spec.Network.Validate(); err != nil {103			return fmt.Errorf("this is a scaffold bug: %s's network does not validate: %w", m.Name, err)104		}105		machines[m.Name] = doc106	}107108	// Nothing was written until everything validated.109	if err := os.MkdirAll(filepath.Join(dir, "machines"), 0o755); err != nil {110		return err111	}112	if err := os.WriteFile(filepath.Join(dir, "cluster.yaml"), clusterYAML, 0o644); err != nil {113		return err114	}115	for name, doc := range machines {116		// The filename must equal the machine's name. The image117		// carries every manifest, and a boot selects its own118		// manifest as machines/<liken.machine>.yaml.119		if err := os.WriteFile(filepath.Join(dir, "machines", name+".yaml"), doc, 0o644); err != nil {120			return err121		}122	}123	if err := os.WriteFile(filepath.Join(dir, ".gitignore"), []byte(gitignore), 0o644); err != nil {124		return err125	}126127	fmt.Fprintf(out, "\nwrote %s/cluster.yaml and %d machine manifest(s)\n", dir, len(machines))128	fmt.Fprintf(out, "next: liken mint %s\n", filepath.Join(dir, "identity"))129	return nil130}131132const gitignore = `# The identity directory holds the cluster's private keys and join133# token: never commit it. The image directory holds built artifacts.134identity/135image/136`137138func render(name string, data any) ([]byte, error) {139	t, err := template.ParseFS(templates, name)140	if err != nil {141		return nil, err142	}143	var buf strings.Builder144	if err := t.Execute(&buf, data); err != nil {145		return nil, err146	}147	return []byte(buf.String()), nil148}149150// interview asks every question, validates each answer as it151// arrives, and asks again until the answer holds. The prompts always152// show the default that an empty answer takes.153func interview(in *bufio.Scanner, out io.Writer, defaultName string) (answers, error) {154	a := answers{}155	ask := func(prompt, deflt string) (string, error) {156		if deflt != "" {157			fmt.Fprintf(out, "%s [%s]: ", prompt, deflt)158		} else {159			fmt.Fprintf(out, "%s: ", prompt)160		}161		if !in.Scan() {162			return "", fmt.Errorf("no more answers (input ended at %q)", prompt)163		}164		answer := strings.TrimSpace(in.Text())165		if answer == "" {166			answer = deflt167		}168		return answer, nil169	}170171	var err error172	if a.ClusterName, err = ask("cluster name", defaultName); err != nil {173		return a, err174	}175176	// This asks for the machines, then for leaders among them. One177	// leader works for a lone machine or a small fleet. Three leaders178	// work once there are three machines to choose from, because179	// etcd's quorum needs an odd count.180	names, err := ask("machine names, space-separated", "machine-1")181	if err != nil {182		return a, err183	}184	machines := strings.Fields(names)185	if len(machines) == 0 || slices.ContainsFunc(machines, func(n string) bool {186		return strings.ContainsAny(n, "/. ") || n == ""187	}) {188		return a, fmt.Errorf("machine names become hostnames and filenames; %q won't do", names)189	}190191	defaultLeaders := machines[:1]192	if len(machines) >= 3 {193		defaultLeaders = machines[:3]194	}195	for {196		leaders, err := ask("leaders (odd count; the first is the founding leader)", strings.Join(defaultLeaders, " "))197		if err != nil {198			return a, err199		}200		a.Leaders = strings.Fields(leaders)201		if len(a.Leaders)%2 == 0 {202			fmt.Fprintln(out, "etcd needs an odd number of leaders (1, 3, 5): a tie has no majority")203			continue204		}205		outsider := slices.IndexFunc(a.Leaders, func(l string) bool { return !slices.Contains(machines, l) })206		if outsider >= 0 {207			fmt.Fprintf(out, "%s is not one of the machines\n", a.Leaders[outsider])208			continue209		}210		break211	}212213	// This asks for the cluster subnet, and each machine's fixed214	// address on it. The founding leader's address becomes the215	// endpoint that every follower uses to make first contact. This216	// is why these addresses are fixed, and not assigned through217	// DHCP.218	var prefix netip.Prefix219	for {220		cidr, err := ask("cluster subnet (the machines' own network)", "10.10.0.0/24")221		if err != nil {222			return a, err223		}224		if prefix, err = netip.ParsePrefix(cidr); err == nil {225			a.NodeCIDR = cidr226			break227		}228		fmt.Fprintln(out, "that isn't a CIDR like 10.10.0.0/24")229	}230	base := prefix.Addr()231	for i, name := range machines {232		// The defaults count up from the subnet's first host address:233		// .1, .2, and so on, in machine order.234		addr := base235		for range i + 1 {236			addr = addr.Next()237		}238		deflt := addr.String()239		for {240			answer, err := ask(fmt.Sprintf("%s's address", name), deflt)241			if err != nil {242				return a, err243			}244			ip, err := netip.ParseAddr(answer)245			if err != nil || !prefix.Contains(ip) {246				fmt.Fprintf(out, "%s must be an address inside %s\n", name, a.NodeCIDR)247				continue248			}249			a.Machines = append(a.Machines, machineAnswers{250				Name:    name,251				Address: fmt.Sprintf("%s/%d", ip, prefix.Bits()),252			})253			if name == a.Leaders[0] {254				a.Endpoint = fmt.Sprintf("https://%s:6443", ip)255			}256			break257		}258	}259260	// This asks for the interfaces once for the whole fleet. Two NICs261	// is the designed shape: an uplink on DHCP, and a fixed address262	// on the cluster segment. A single-NIC machine puts its fixed263	// address on its only interface, and needs a gateway and264	// nameservers spelled out, because there is no DHCP lease to265	// supply them.266	uplink, err := ask(`uplink interface for internet access ("none" if machines have only one interface)`, "eth0")267	if err != nil {268		return a, err269	}270	a.UplinkNIC = strings.TrimSpace(uplink)271	if strings.EqualFold(a.UplinkNIC, "none") {272		a.UplinkNIC = ""273	}274	clusterDefault := "eth1"275	if a.UplinkNIC == "" {276		clusterDefault = "eth0"277	}278	if a.ClusterNIC, err = ask("cluster interface (carries the fixed address)", clusterDefault); err != nil {279		return a, err280	}281	if a.UplinkNIC == "" {282		for {283			gw, err := ask("gateway (single-NIC machines have no DHCP to learn a route from)", "")284			if err != nil {285				return a, err286			}287			if _, err := netip.ParseAddr(gw); err == nil {288				a.Gateway = gw289				break290			}291			fmt.Fprintln(out, "that isn't an IP address")292		}293		ns, err := ask("nameservers, space-separated", "1.1.1.1 9.9.9.9")294		if err != nil {295			return a, err296		}297		a.Nameservers = strings.Fields(ns)298	}299300	// This asks for the disks. One device fits a real machine with301	// one drive, with all seven roles carved from it. Three devices302	// match the dev cluster's shape: state, pods, boot. The roles and303	// sizes follow the same reasoning that the dev cluster's comments304	// teach.305	for {306		disks, err := ask("disk devices, space-separated (1 disk, or 3 for state/pods/boot)", "/dev/sda")307		if err != nil {308			return a, err309		}310		a.Disks = strings.Fields(disks)311		if len(a.Disks) == 1 || len(a.Disks) == 3 {312			break313		}314		fmt.Fprintln(out, "one disk or three: one carries every role, three split state/pods/boot")315	}316317	ntp, err := ask("NTP servers for the leaders (liken deliberately has no default)", "time.cloudflare.com time.nist.gov")318	if err != nil {319		return a, err320	}321	a.Upstreams = strings.Fields(ntp)322323	for {324		policy, err := ask("reboot policy: Manual (a person grants each reboot) or Auto", "Manual")325		if err != nil {326			return a, err327		}328		if policy == "Manual" || policy == "Auto" {329			a.RebootPolicy = policy330			break331		}332		fmt.Fprintln(out, "Manual or Auto, capitalized the way the API spells them")333	}334335	// The interview offers the simple opt-ins, not the whole336	// vocabulary. flux is deliberately absent: it takes parameters,337	// and enabling GitOps ends at the forge, after boot, when the338	// minted deploy key is registered. No interview can finish that339	// handshake, so the manual's GitOps guide owns the flow. The340	// rest of the vocabulary stays reachable the ordinary way, by341	// editing the cluster document this scaffold writes.342	features, err := ask("features to enable, space-separated (traefik servicelb metrics-server iscsi nfs), or none", "")343	if err != nil {344		return a, err345	}346	a.Features = strings.Fields(features)347	for _, f := range a.Features {348		if !slices.Contains([]string{"traefik", "servicelb", "metrics-server", "iscsi", "nfs"}, f) {349			return a, fmt.Errorf("%q is not a feature this interview can enable; the manual's Cluster reference lists the whole vocabulary", f)350		}351	}352353	if a.Source, err = ask("release channel URL for over-the-network updates (blank to decide later)", ""); err != nil {354		return a, err355	}356	return a, nil357}