#!/usr/bin/bash
# The one script busybox init runs at boot, before it starts any shell.
#
# The initramfs has already handed over /proc, /sys, /dev and /run, and the
# root is a writable overlay. What is left is the rest of the standard mount
# set, a hostname, and the two things nothing else will do without a udev:
# loading the drivers for hardware that is present, and refreshing the linker
# cache.

PATH=/usr/local/bin:/usr/bin:/usr/local/sbin:/usr/sbin
export PATH

msg() { echo "  * $*"; }

# Guarded, every one of them. The initramfs moved these across rather than
# leaving them for us, and mounting a fresh tmpfs over /run in particular would
# hide the live medium and the squashfs the root filesystem is made of --
# which does not fail here, it fails later and inexplicably.
mountpoint -q /proc || mount -t proc     proc     /proc
mountpoint -q /sys  || mount -t sysfs    sysfs    /sys
mountpoint -q /dev  || mount -t devtmpfs devtmpfs /dev
mountpoint -q /run  || mount -t tmpfs    tmpfs    /run -o mode=0755

mkdir -p /dev/pts /dev/shm /run/lock /tmp
mountpoint -q /dev/pts || mount -t devpts devpts /dev/pts -o gid=5,mode=620
mountpoint -q /dev/shm || mount -t tmpfs  tmpfs  /dev/shm -o mode=1777
chmod 1777 /tmp /var/tmp 2>/dev/null

# THE tmpfiles.d RULE THIS TREE DOES NOT HAVE, AND THE REASON THE FIRST GREETER
# NEVER DREW A PIXEL.
#
# systemd ships tmpfiles.d/x11.conf, four rules, all 1777:
#
#     D! /tmp/.X11-unix    D! /tmp/.ICE-unix
#     D! /tmp/.XIM-unix    D! /tmp/.font-unix
#
# We run elogind, elogind has no tmpfiles, and nothing else here creates them.
# mutter notices /tmp/.X11-unix is missing and makes it itself -- and then
# REJECTS WHAT IT JUST MADE. meta-xwayland.c calls choose_xdisplay() twice in
# consecutive statements (public connection, then private); each one starts with
# mkdir("/tmp/.X11-unix", 01777) and, ON EEXIST ONLY, checks the mode it finds.
# So the first call creates the directory and never looks at it, and the second
# call finds it and applies (st_mode & 0022) == 0022 -- which the first call
# cannot satisfy, because mkdir(2) is masked by the umask and 01777 lands as
# 01755. The failure is a g_error, and a fatal glib log on aarch64 is
# raise(SIGTRAP): gnome-shell died of signal 5, six spawns running, with the
# reason in a greeter log nothing reads.
#
# WHY mkdir AND chmod ARE SEPARATE COMMANDS, and it is NOT the reason it first
# looks like. `mkdir -p -m 1777` is the obvious one-liner. The tempting story is
# that -m is umask-masked the way mkdir(2) is; that story is FALSE and was
# measured false in GNU coreutils and in busybox alike -- POSIX gives -m chmod
# semantics, so both produce 1777 under umask 022.
# The real trap is the -p: **`mkdir -p -m 1777` applies the mode ONLY when it
# creates the directory, and does nothing at all when one is already there.**
# Measured, both implementations, on a directory pre-set to 0755: the mode after
# the command is still 0755, silently, rc=0. And an already-there directory in
# the WRONG mode is precisely the state this bug leaves behind -- mutter's own
# first call makes 01755 before its second call rejects it. So the one-liner
# would fix a clean boot and quietly fail to repair the exact case that produced
# the failure. The unconditional chmod repairs both.
#
# systemd's x11.conf has a fifth rule that removes stale /tmp/.X[0-9]*-lock
# files. Not carried: this root is a fresh overlay on every boot, so there is
# never a stale one, and a boot script that deletes by glob is a worse thing to
# own than the problem it solves.
for d in .X11-unix .ICE-unix .XIM-unix .font-unix; do
	mkdir -p "/tmp/$d" 2>/dev/null && chmod 1777 "/tmp/$d" 2>/dev/null
done

# Announced rather than assumed, and the two failure modes are told apart on
# purpose: an empty reading means stat is missing and this check is BLIND,
# which is not the same news as a mode that is genuinely wrong.
x11mode=$(stat -c %a /tmp/.X11-unix 2>/dev/null)
if [ "$x11mode" = "1777" ]; then
	msg "X11 socket directories ready (/tmp/.X11-unix 1777)"
elif [ -z "$x11mode" ]; then
	msg "WARNING: cannot read the mode of /tmp/.X11-unix -- no stat, so this check is blind, not passing"
else
	msg "WARNING: /tmp/.X11-unix is $x11mode, not 1777 -- mutter will abort at Xwayland startup and gnome-shell will die of signal 5"
fi

# Written rather than set with hostname(1): uutils' hostname does not implement
# setting one, and this is what the syscall does anyway.
echo duct >/proc/sys/kernel/hostname

# The machine id, generated per boot rather than shipped.
#
# The ISO build leaves /etc/machine-id empty on purpose -- baking one in would
# give every machine that ever boots this ISO the same identity, and D-Bus,
# logging and anything keyed on "which machine is this" would be unable to tell
# two of them apart on the same network.
#
# 16 random bytes as 32 hex digits is the whole format. od rather than a
# dedicated tool because there is no uuidgen in the base set and this needs
# nothing that is not already here.
if [ ! -s /etc/machine-id ]; then
	if id=$(od -An -tx1 -N16 /dev/urandom 2>/dev/null | tr -d ' \n') && [ -n "$id" ]; then
		msg "generating a machine id"
		# The file is 0444 in the squashfs; the overlay lets us replace it,
		# but only after the read-only mode is out of the way.
		rm -f /etc/machine-id
		printf '%s\n' "$id" >/etc/machine-id
		chmod 0444 /etc/machine-id
	fi
fi

# There are no install hooks in tape, so the image build runs ldconfig once
# after installing everything. Running it again here costs a second and covers
# the case where something was installed into the overlay after boot.
if [ -x /usr/sbin/ldconfig ]; then
	/usr/sbin/ldconfig 2>/dev/null || true
fi

# Devices: udev if this medium has one, a modalias sweep if it does not.
#
# THE TWO ARE NOT INTERCHANGEABLE and only one of them runs. The sweep below
# loads drivers, and that is all it does: it creates no symlinks, applies no
# permissions, sets no ACLs and TAGS NOTHING. Tags are how a seat is assembled
# -- elogind decides what belongs to seat0 by reading udev's properties -- so a
# graphical session on a machine that was coldplugged this way finds a seat
# with no master device and fails in a way that reads as a compositor bug.
#
# So: udevd when it is installed, and the sweep only when it is not. A console
# ISO carries no eudev and takes exactly the path it always took.
udev_running=
for udevd in /usr/sbin/udevd /usr/bin/udevd; do
	[ -x "$udevd" ] || continue
	msg "starting udevd"
	"$udevd" --daemon && udev_running=1
	break
done

if [ -n "$udev_running" ]; then
	# Two triggers, subsystems before devices, because that is the order the
	# kernel would have announced them in on a cold boot: a device event whose
	# subsystem has not been seen is a rule set that has not been applied.
	# settle waits for the queue rather than for a fixed sleep -- and does not
	# fail the boot if it times out, because a slow device is not a reason to
	# have no console.
	msg "replaying device events through udev"
	udevadm trigger --type=subsystems --action=add 2>/dev/null || true
	udevadm trigger --type=devices --action=add 2>/dev/null || true
	udevadm settle --timeout=30 2>/dev/null || \
		msg "udev did not settle in 30s; continuing"
elif [ -x /usr/bin/modprobe ] && [ -d /usr/lib/modules/"$(uname -r)" ]; then
	# Coldplug, without udev.
	#
	# Everything needed to reach this point is built into the kernel;
	# everything else -- the GPU, the network card, the sound device -- is a
	# module, and with no udev running there is nothing to notice they exist.
	# Each device that needs a driver advertises a modalias in sysfs, which is
	# precisely the string modprobe matches against, so feeding the whole set
	# to modprobe loads exactly the drivers this machine has hardware for.
	#
	# -a takes many aliases at once, -b honours the blacklist, -q keeps the
	# console clear of one "module not found" line per device that has no
	# driver -- which is most of them, and is not an error.
	msg "loading drivers for detected hardware"
	find /sys/devices -name modalias -type f -print0 2>/dev/null |
		xargs -0 -r cat 2>/dev/null |
		sort -u |
		xargs -r /usr/bin/modprobe -abq 2>/dev/null || true
fi

# NO SYSLOG SINK IS STARTED HERE, AND THAT IS A MEASUREMENT RATHER THAN AN
# OVERSIGHT. Read this before adding one.
#
# The problem is real. gdm routes every g_warning and g_critical through
# syslog(3) and NOTHING ELSE: common/gdm-log.c's handler picks stderr only
# when `is_sd_booted` is true, and in gdm 48.0 that is a `static gboolean =
# FALSE` which is READ TWICE AND ASSIGNED NOWHERE IN THE TARBALL -- so the
# stderr branch is dead code on every system, and syslog is the only channel
# gdm will ever use. A gdm that starts, does nothing and says nothing is
# therefore indistinguishable from a healthy one idling, which is exactly what
# the first gdm boot looked like. Same for sshd, cups, wpa_supplicant and
# polkit, which put anything that matters there too.
#
# THE OBVIOUS FIX DOES NOT WORK. busybox is already PID 1 here, so `busybox
# syslogd -n -O /dev/console` looks free. It was tried and MEASURED, and this
# tree's busybox syslogd BINDS /dev/log AND NEVER READS FROM IT: its own
# startup banner reaches the sink, and not one client message does. Controlled
# three ways against the same rootfs -- util-linux `logger`, busybox `logger`,
# and glibc's own syslog(3) through python -- with both output modes (-O file
# and -C circular buffer read back with logread). Nothing arrives. The same
# three clients, on the same socket path, in the same image, ALL DELIVER to a
# nine-line python listener that binds /dev/log itself, which is the positive
# arm that makes the negative one mean something.
#
# So the sink is not merely useless, it is ACTIVELY MISLEADING: a /dev/log
# that nobody drains makes the system look instrumented while discarding
# exactly the messages someone went looking for. With no /dev/log at all,
# syslog(3) fails and the messages are lost the same way -- minus the false
# impression. Absence is the honest state until busybox syslogd is fixed or a
# real syslog daemon is packaged, and this comment is here so the next person
# starts from the control rather than re-deriving it.
#
# The D-Bus system bus.
#
# Started here rather than activated, because it is what does the activating:
# polkit, elogind, NetworkManager, accountsservice, upower and colord are all
# started by dbus-daemon on the first call to their name, and none of them can
# be reached before it is listening. elogind in particular is D-Bus activated
# and nothing else starts it -- its own service file is
# `Exec=/usr/libexec/elogind --daemon` -- so the bus is the whole of what makes
# a session possible.
#
# AFTER the machine id above, and that ordering is real: dbus-daemon reads
# /etc/machine-id at startup, and a bus that came up before it was written
# would hand out the empty one for the life of the boot.
if [ -x /usr/bin/dbus-daemon ]; then
	msg "starting the D-Bus system bus"
	mkdir -p /run/dbus
	/usr/bin/dbus-daemon --system --fork || msg "the system bus did not start"
fi

# elogind, STARTED HERE AND NOT LEFT TO D-BUS ACTIVATION.
#
# The previous reading of this was correct and incomplete, and the gap is worth
# stating exactly, because it is what kept a greeter off the screen: elogind IS
# D-Bus activated -- its own service file is `Exec=/usr/libexec/elogind
# --daemon` -- so every client that CALLS org.freedesktop.login1 brings it up
# with no help from this script. gdm is not such a client.
#
# GDM ASKS THE FILESYSTEM. gdm-local-display-factory.c gates the entire greeter
# on `sd_seat_can_graphical()` (lines 863 and 1136), and in libelogind that
# function is a read of /run/systemd/seats/ -- the string is in the library, and
# the matching writes are in the elogind binary. No D-Bus method is called, so
# there is nothing for dbus-daemon to activate on. With elogind not running the
# directory does not exist, the call returns an error, and the factory's
# `if (ret < 0) return;` sends gdm quietly back to its main loop. gdm then runs
# forever, holds /run/gdm/gdm.pid, spawns no session worker, and says NOTHING --
# measured, with a process list: gdm alive, no greeter, /run/systemd absent.
#
# THE GENERAL SHAPE: AN ACTIVATION MECHANISM ONLY FIRES FOR CLIENTS THAT USE
# THE ACTIVATING CHANNEL. "It is D-Bus activated" is a fact about the daemon
# and reads like a fact about the system; the first consumer that asks a
# different way finds it absent, and absence returns as "no seats" rather than
# as "not running". A display manager is exactly that consumer -- it needs a
# seat BEFORE anyone has logged in, which is before any activation could have
# happened.
#
# Starting it explicitly does not conflict with the activation path: whoever
# gets there first takes org.freedesktop.login1, and dbus-daemon does not
# activate a name that is already owned. AFTER the bus, because elogind needs
# it to take that name; BEFORE seatd and the greeter, because both are asking
# about seats.
if [ -x /usr/libexec/elogind ]; then
	msg "starting elogind -- gdm reads /run/systemd/seats, which nothing else creates"
	if ! el_err=$(/usr/libexec/elogind --daemon 2>&1); then
		msg "elogind did not start: ${el_err:-it printed nothing}"
	fi
fi

# seatd, if it is installed.
#
# Only some compositors want it: mutter speaks to logind directly, and weston
# reaches a seat through libseat, which tries seatd first, then logind, then a
# root-only builtin path. Starting it when it is present means weston gets a
# real seat without needing a registered login session, and costs a daemon
# nothing else talks to.
#
# -g video, the group duct-filesystem already creates for exactly this: the
# devices a seat hands out are the DRM node and the input devices, and both are
# group-owned rather than world-readable.
for seatd in /usr/sbin/seatd /usr/bin/seatd; do
	[ -x "$seatd" ] || continue
	msg "starting seatd"
	"$seatd" -g video &
	break
done

# bluetoothd.
#
# NOTHING ELSE WILL EVER START IT. bluez installs its D-Bus activation file
# inside `if SYSTEMD` in Makefile.am, so a build with --disable-systemd ships
# no org.bluez.service and there is no other activation path -- the package
# says so in its own post-install and has been waiting for someone to pick it
# up. Without this, the Bluetooth panel loads, finds no org.bluez, and offers
# to turn on an adapter that never appears.
if [ -x /usr/libexec/bluetooth/bluetoothd ]; then
	msg "starting bluetoothd"
	mkdir -p /var/lib/bluetooth
	# Its message, not just its exit status. The first desktop boot printed
	# "bluetoothd did not start" and nothing else, which says a daemon failed
	# and not one thing about why -- and a diagnosis you cannot read is the
	# reason the next person re-runs the whole boot to learn what the console
	# already knew.
	#
	# AND NO --daemon. That message, once it could be read, said "Unknown
	# option --daemon" -- bluetoothd has never had one. Read from the shipped
	# binary rather than from a manual page, its long options are exactly
	# compat, configfile, debug, experimental, nodetach, plugin and version:
	# it DETACHES BY DEFAULT, and --nodetach is the flag that would keep it in
	# the foreground. So the flag asked for the default behaviour under a name
	# the parser does not know, and the daemon this whole block exists to start
	# has never once started. (gdm's --nodaemon in graphical-session was the
	# same mistake in the same boot -- a plausible flag from another
	# distribution's invocation, never checked against the binary.)
	#
	# The command substitution does not hang on a daemon that detaches:
	# bluetoothd forks through daemon(0,0), which points its stdio at
	# /dev/null and closes this pipe. An option-parse failure happens before
	# that fork, which is exactly the failure this capture is here to report.
	if ! bt_err=$(/usr/libexec/bluetooth/bluetoothd 2>&1); then
		msg "bluetoothd did not start: ${bt_err:-it printed nothing}"
	fi
fi

msg "Duct live system ready"
