[RFC PATCH 27/46] oncore: Implement on-core session lifecycle and scheduling loop

Pasha Tatashin pasha.tatashin at soleen.com
Sun Sep 20 12:36:31 PDT 2026


Implement On-Core session management, job submission and cancellation,
isolated address space mapping helpers, and the round-robin time-sliced
scheduling loop for preserved physical CPUs in kernel/liveupdate/oncore.c,
and include the On-Core kernel-doc documentation in
Documentation/liveupdate/cpu_preservation.rst.

Signed-off-by: Pasha Tatashin <pasha.tatashin at soleen.com>
---
 Documentation/liveupdate/cpu_preservation.rst |  14 +
 kernel/liveupdate/Makefile                    |  12 +
 kernel/liveupdate/cpu_preserve.c              |  15 +
 kernel/liveupdate/oncore.c                    | 947 ++++++++++++++++++
 4 files changed, 988 insertions(+)
 create mode 100644 kernel/liveupdate/oncore.c

diff --git a/Documentation/liveupdate/cpu_preservation.rst b/Documentation/liveupdate/cpu_preservation.rst
index a7a534188685..605d808e61c8 100644
--- a/Documentation/liveupdate/cpu_preservation.rst
+++ b/Documentation/liveupdate/cpu_preservation.rst
@@ -18,6 +18,20 @@ Architecture Backend Interface
 
 .. kernel-doc:: include/linux/cpu_preserve.h
 
+On-Core Execution and Scheduling Framework
+==========================================
+
+.. kernel-doc:: kernel/liveupdate/oncore.c
+   :doc: On-Core Execution and Scheduling Framework
+
+On-Core Session & Job API
+=========================
+
+.. kernel-doc:: include/linux/oncore.h
+
+.. kernel-doc:: kernel/liveupdate/oncore.c
+   :identifiers:
+
 CPU Preservation ABI
 ====================
 
diff --git a/kernel/liveupdate/Makefile b/kernel/liveupdate/Makefile
index 486f7854bb75..ab0d44079fb2 100644
--- a/kernel/liveupdate/Makefile
+++ b/kernel/liveupdate/Makefile
@@ -1,6 +1,17 @@
 # SPDX-License-Identifier: GPL-2.0
 
+KASAN_SANITIZE_cpu_preserve.o := n
+KCSAN_SANITIZE_cpu_preserve.o := n
+UBSAN_SANITIZE_cpu_preserve.o := n
+KCOV_INSTRUMENT_cpu_preserve.o := n
+KASAN_SANITIZE_oncore.o := n
+KCSAN_SANITIZE_oncore.o := n
+UBSAN_SANITIZE_oncore.o := n
+KCOV_INSTRUMENT_oncore.o := n
+CFLAGS_REMOVE_cpu_preserve.o = $(CC_FLAGS_FTRACE)
+CFLAGS_REMOVE_oncore.o = $(CC_FLAGS_FTRACE)
 CFLAGS_cpu_preserve.o += $(call cc-option,-mbranch-protection=none) -fno-stack-protector $(call cc-option,-ftrivial-auto-var-init=uninitialized) $(call cc-option,-fno-jump-tables)
+CFLAGS_oncore.o       += $(call cc-option,-mbranch-protection=none) -fno-stack-protector $(call cc-option,-ftrivial-auto-var-init=uninitialized) $(call cc-option,-fno-jump-tables)
 
 luo-y :=								\
 		kho_block.o						\
@@ -14,3 +25,4 @@ obj-$(CONFIG_KEXEC_HANDOVER_DEBUGFS)	+= kexec_handover_debugfs.o
 
 obj-$(CONFIG_LIVEUPDATE)		+= luo.o
 obj-$(CONFIG_LIVEUPDATE_CPU)		+= cpu_preserve.o
+obj-$(CONFIG_LIVEUPDATE_ONCORE)		+= oncore.o
diff --git a/kernel/liveupdate/cpu_preserve.c b/kernel/liveupdate/cpu_preserve.c
index 9a039ba9e912..430afbca860d 100644
--- a/kernel/liveupdate/cpu_preserve.c
+++ b/kernel/liveupdate/cpu_preserve.c
@@ -178,6 +178,7 @@
 #include <linux/liveupdate.h>
 #include <linux/mm.h>
 #include <linux/objtool.h>
+#include <linux/oncore.h>
 #include <linux/reboot.h>
 
 #include <asm/sections.h>
@@ -1552,14 +1553,22 @@ static int cpu_preserve_preserve(struct liveupdate_file_op_args *args)
 	if (ret)
 		return ret;
 
+	ret = oncore_session_add_cpu(args->session, cpu);
+	if (ret) {
+		cpu_unpreserve(cpu);
+		return ret;
+	}
+
 	fser = kho_alloc_preserve(sizeof(*fser));
 	if (IS_ERR(fser)) {
 		cpu_unpreserve(cpu);
+		oncore_session_remove_cpu(args->session, cpu);
 		return PTR_ERR(fser);
 	}
 
 	memset(fser, 0, sizeof(*fser));
 	fser->cpu = cpu;
+	KHOSER_STORE_PTR(fser->oncore, oncore_session_get_ser(args->session));
 
 	scoped_guard(mutex, &cpu_preserved_lock)
 		fser->stack_pa = cpu_preserved_outgoing.pcpus[cpu].stack_pa;
@@ -1580,6 +1589,7 @@ static void cpu_preserve_unpreserve(struct liveupdate_file_op_args *args)
 	cpu = fser->cpu;
 
 	cpu_unpreserve(cpu);
+	oncore_session_remove_cpu(args->session, cpu);
 
 	kho_unpreserve_free(fser);
 }
@@ -1587,8 +1597,12 @@ static void cpu_preserve_unpreserve(struct liveupdate_file_op_args *args)
 static void cpu_preserve_restore_incoming_cpu(struct liveupdate_session *session,
 					      struct cpu_preserved_file_ser *fser)
 {
+	struct oncore_session_ser *oncore = KHOSER_LOAD_PTR(fser->oncore);
 	unsigned int cpu = fser->cpu;
 
+	if (oncore)
+		oncore_session_restore(session, oncore);
+
 	scoped_guard(mutex, &cpu_preserved_lock) {
 		cpumask_set_cpu(cpu, &cpu_preserved_incoming.mask);
 		cpumask_set_cpu(cpu, &cpu_preserved_mask);
@@ -1646,6 +1660,7 @@ static void cpu_preserve_finish(struct liveupdate_file_op_args *args)
 		cpu_preserve_restore_incoming_cpu(args->session, fser);
 
 	cpu_unpreserve(fser->cpu);
+	oncore_session_remove_cpu(args->session, fser->cpu);
 
 	kho_restore_free(fser);
 }
diff --git a/kernel/liveupdate/oncore.c b/kernel/liveupdate/oncore.c
new file mode 100644
index 000000000000..196ed0a67563
--- /dev/null
+++ b/kernel/liveupdate/oncore.c
@@ -0,0 +1,947 @@
+// SPDX-License-Identifier: GPL-2.0-only
+/*
+ * Copyright (c) 2026, Google LLC.
+ * Pasha Tatashin <pasha.tatashin at soleen.com>
+ *
+ * On-Core Session and Scheduler Framework for Live Update
+ */
+
+/**
+ * DOC: On-Core Execution and Scheduling Framework
+ *
+ * The On-Core framework enables latency-sensitive workloads to continue
+ * executing directly on preserved physical CPUs (__cpu_preserved_text /
+ * __cpu_preserved_data) across a kexec Live Update (CONFIG_LIVEUPDATE_CPU)
+ * while the outgoing host kernel shuts down and the incoming kernel boots.
+ *
+ * Each Live Update session (struct liveupdate_session) can create an
+ * oncore_session that manages:
+ *
+ * 1. Isolated Address Space (struct cpu_preserved_as):
+ *    Page tables containing only the preserved runtime text/data sections,
+ *    the session and runqueue metadata, and workload-specific buffers
+ *    explicitly mapped via oncore_session_map_range() or
+ *    oncore_session_map_buffer().
+ *
+ * 2. Preserved Physical CPU Pool:
+ *    One or more physical CPUs isolated from Linux scheduling via
+ *    cpu_preserve and attached to the session via oncore_session_add_cpu().
+ *    Each preserved CPU switches to the session's isolated page tables and
+ *    executes oncore_cpu_schedule_loop().
+ *
+ * 3. Cooperative Time-Sliced Runqueue (struct oncore_runqueue):
+ *    A lockless/atomic round-robin FIFO scheduler supporting M jobs across N
+ *    preserved physical CPUs (including M > N oversubscription).  It provides
+ *    initial-run starvation avoidance, least-loaded CPU assignment, work
+ *    stealing, single-job fast-path continuation, and low-power backoff (WFE on
+ *    arm64, PAUSE on x86) when the queue is idle or a job returns
+ *    ONCORE_EXIT_STALL.
+ *
+ * Integration with KVM Caretaker and Future Kernel Caretaker Workloads
+ * --------------------------------------------------------------------
+ * On-Core is workload- and hypervisor-agnostic:
+ *
+ * - KVM Caretaker: During LUO prepare/freeze, KVM detaches each vCPU into a
+ *   self-contained KHO-preserved Caretaker page (caretaker_x86_page on x86,
+ *   caretaker_arm64_page on arm64), maps the page and guest/virtualization
+ *   structures into the session's address space, and submits an oncore_job
+ *   whose callback enters the guest (VMLAUNCH/VMRESUME, VMRUN, or EL2 world
+ *   switch) until the time-slice deadline (deadline_ticks) expires, the vCPU
+ *   yields on HLT/WFI (ONCORE_EXIT_YIELD_IDLE), hits an exit requiring the
+ *   incoming kernel (ONCORE_EXIT_STALL), or observes the incoming kernel's
+ *   reclaim signal (ONCORE_EXIT_ATTACH_SIGNALED).
+ *
+ * - Future Kernel Caretaker Workloads: Any self-contained kernel subsystem
+ *   compiled into the preserved runtime sections (such as hardware watchdog
+ *   feeders, health/heartbeat responders, or zero-loss network/storage
+ *   polling drivers) can submit an oncore_job_fn(void *data, u64 deadline) to
+ *   share preserved physical cores alongside or independently of KVM vCPUs.
+ *
+ * Architecture Requirements
+ * -------------------------
+ * To support CONFIG_LIVEUPDATE_ONCORE, an architecture must implement:
+ *
+ * - <asm/oncore.h>:
+ *   - arch_oncore_read_counter(): Read a monotonic hardware counter directly
+ *     from __cpu_preserved_text without relying on kernel timekeeping
+ *     (e.g., rdtsc() on x86, __arch_counter_get_cntpct() on arm64).
+ *   - arch_oncore_counter_freq_hz(): Return the hardware counter frequency in
+ *     Hz used to convert the time quantum (oncore.quantum_ms) into ticks.
+ *
+ * - <asm/cpu_preserve.h> (from CONFIG_LIVEUPDATE_CPU):
+ *   - arch_cpu_preserved_switch_pgd(): Switch to the session's isolated PGD.
+ *   - arch_cpu_preserved_park_wait() / arch_cpu_preserved_kick(): Low-power
+ *     wait instruction and cross-CPU wakeup mechanism.
+ *
+ * Public Interfaces (<linux/oncore.h>)
+ * ------------------------------------
+ * - Session & CPU Lifecycle:
+ *   oncore_session_add_cpu(), oncore_session_remove_cpu(),
+ *   oncore_session_get_ser(), oncore_session_restore().
+ * - Session Address Space Mapping:
+ *   oncore_session_map_range(), oncore_session_map_buffer(),
+ *   oncore_session_get_pgd_pa().
+ * - Job Submission & Control:
+ *   oncore_session_submit_job(), oncore_job_set_data(),
+ *   oncore_job_cpu(), oncore_job_session(),
+ *   oncore_session_activate_job(), oncore_session_cancel_job().
+ */
+
+#define pr_fmt(fmt) "oncore: " fmt
+
+#include <linux/cpu_preserve.h>
+#include <linux/cpumask.h>
+#include <linux/delay.h>
+#include <linux/init.h>
+#include <linux/io.h>
+#include <linux/kexec.h>
+#include <linux/kexec_handover.h>
+#include <linux/list.h>
+#include <linux/liveupdate.h>
+#include <linux/mm.h>
+#include <linux/module.h>
+#include <linux/mutex.h>
+#include <linux/oncore.h>
+#include <linux/overflow.h>
+#include <linux/slab.h>
+#include <linux/string.h>
+#include <linux/types.h>
+
+#define ONCORE_DEFAULT_QUANTUM_MS	10
+#define ONCORE_CANCEL_TIMEOUT_US	1000000
+#define ONCORE_CANCEL_STEP_US		100
+
+enum oncore_job_state {
+	ONCORE_JOB_NEW = 0,
+	ONCORE_JOB_RUNNABLE,
+	ONCORE_JOB_RUNNING,
+	ONCORE_JOB_CANCELING,
+	ONCORE_JOB_DEAD,
+};
+
+struct oncore_job {
+	struct list_head		node;
+	struct list_head		sess_node;
+	phys_addr_t			next_job_pa;
+	struct oncore_session		*session;
+	enum oncore_job_state		state;
+	oncore_job_fn			run_fn;
+	void				*data;
+	int				assigned_cpu;
+	int				last_cpu;
+	u64				total_runs;
+};
+
+struct oncore_sched_config {
+	u32				quantum_ms;
+	u64				quantum_ticks;
+	u64				counter_freq_hz;
+};
+
+struct oncore_runqueue {
+	atomic_t			lock;
+	struct list_head		runnable;
+	unsigned int			nr_runnable;
+};
+
+struct oncore_session {
+	struct list_head		node;
+	/* Protects session state and job list */
+	struct mutex			lock;
+	struct list_head		jobs;
+	phys_addr_t			first_job_pa;
+	struct oncore_runqueue		rq;
+	struct oncore_sched_config	sched_config;
+	struct cpu_preserved_as		*as;
+	struct oncore_session_ser	*ser;
+	bool				is_incoming;
+};
+
+static struct oncore_sched_config global_oncore_sched_config __cpu_preserved_data;
+static DEFINE_MUTEX(oncore_sessions_lock);
+static LIST_HEAD(oncore_sessions);
+
+static inline struct cpumask *oncore_session_cpus(struct oncore_session *sess)
+{
+	return to_cpumask((unsigned long *)sess->ser->cpus_bitmap);
+}
+
+static inline void oncore_rq_lock(struct oncore_runqueue *rq)
+{
+	while (atomic_cmpxchg_acquire(&rq->lock, 0, 1) != 0) {
+		while (atomic_read(&rq->lock) != 0)
+			cpu_relax();
+	}
+}
+
+static inline void oncore_rq_unlock(struct oncore_runqueue *rq)
+{
+	atomic_set_release(&rq->lock, 0);
+}
+
+static int __init parse_oncore_quantum(char *arg)
+{
+	u32 val;
+
+	if (kstrtou32(arg, 0, &val) == 0 && val >= 1 && val <= 1000)
+		global_oncore_sched_config.quantum_ms = val;
+	return 0;
+}
+early_param("oncore.quantum_ms", parse_oncore_quantum);
+
+static void oncore_sched_update_ticks(void)
+{
+	u32 ms = global_oncore_sched_config.quantum_ms ? : ONCORE_DEFAULT_QUANTUM_MS;
+	u64 freq = arch_oncore_counter_freq_hz();
+
+	global_oncore_sched_config.counter_freq_hz = freq;
+	global_oncore_sched_config.quantum_ticks = (freq * ms) / 1000ULL;
+}
+
+static void oncore_runqueue_init(struct oncore_runqueue *rq)
+{
+	atomic_set(&rq->lock, 0);
+	INIT_LIST_HEAD(&rq->runnable);
+	rq->nr_runnable = 0;
+}
+
+static int oncore_sched_enqueue(struct oncore_runqueue *rq,
+				struct oncore_job *job)
+{
+	int cpu;
+
+	oncore_rq_lock(rq);
+	job->state = ONCORE_JOB_RUNNABLE;
+	list_add_tail(&job->node, &rq->runnable);
+	rq->nr_runnable++;
+	oncore_rq_unlock(rq);
+
+	for_each_cpu(cpu, oncore_session_cpus(job->session)) {
+		if (cpu_is_preserved(cpu))
+			arch_cpu_preserved_kick(cpu);
+	}
+
+	return 0;
+}
+
+static int oncore_sched_dequeue(struct oncore_runqueue *rq,
+				struct oncore_job *job)
+{
+	oncore_rq_lock(rq);
+	if (!list_empty(&job->node)) {
+		list_del_init(&job->node);
+		rq->nr_runnable--;
+	}
+	if (job->state == ONCORE_JOB_RUNNING)
+		job->state = ONCORE_JOB_CANCELING;
+	else
+		job->state = ONCORE_JOB_DEAD;
+	oncore_rq_unlock(rq);
+	return 0;
+}
+
+static struct oncore_job *__cpu_preserved_text
+oncore_sched_pick_next(struct oncore_runqueue *rq, int cpu)
+{
+	struct oncore_job *job = NULL, *iter;
+
+	if (!rq || READ_ONCE(rq->nr_runnable) == 0)
+		return NULL;
+
+	oncore_rq_lock(rq);
+
+	/* 1. Starvation avoidance: if head job has NEVER run, take it immediately */
+	if (!list_empty(&rq->runnable)) {
+		iter = list_first_entry(&rq->runnable, struct oncore_job, node);
+		if (iter->total_runs == 0) {
+			job = iter;
+			goto found;
+		}
+	}
+
+	/* 2. Prefer job affine to this core */
+	list_for_each_entry(iter, &rq->runnable, node) {
+		if (iter->assigned_cpu == cpu) {
+			job = iter;
+			goto found;
+		}
+	}
+
+	/* 3. Work-stealing fallback: take oldest job from head of queue */
+	if (!list_empty(&rq->runnable))
+		job = list_first_entry(&rq->runnable, struct oncore_job, node);
+
+found:
+	if (job) {
+		list_del_init(&job->node);
+		rq->nr_runnable--;
+	}
+
+	oncore_rq_unlock(rq);
+	return job;
+}
+
+static void __cpu_preserved_text
+oncore_sched_put_prev(struct oncore_runqueue *rq,
+		      struct oncore_job *job)
+{
+	oncore_rq_lock(rq);
+	job->state = ONCORE_JOB_RUNNABLE;
+	list_add_tail(&job->node, &rq->runnable);
+	rq->nr_runnable++;
+	oncore_rq_unlock(rq);
+}
+
+static void __cpu_preserved_text
+oncore_cpu_schedule_loop(int cpu, struct oncore_runqueue *rq,
+			 struct oncore_sched_config *cfg)
+{
+	enum oncore_exit_reason reason;
+	struct oncore_job *curr = NULL;
+	u64 deadline;
+
+	if (!rq || !cfg)
+		return;
+
+	while (!cpu_preserved_should_exit(cpu)) {
+		/* 1. Pick the next runnable job from the FIFO queue */
+		if (!curr) {
+			curr = oncore_sched_pick_next(rq, cpu);
+			if (!curr) {
+				/* No runnable jobs; execute low-power park wait */
+				arch_cpu_preserved_park_wait();
+				continue;
+			}
+		}
+
+		/* 2. Compute quantum deadline */
+		deadline = arch_oncore_read_counter() + cfg->quantum_ticks;
+
+		/* 3. Execute workload on physical silicon */
+		curr->last_cpu = cpu;
+		curr->state = ONCORE_JOB_RUNNING;
+		reason = curr->run_fn(curr->data, deadline);
+
+		/* 4. Update telemetry and accounting */
+		curr->total_runs++;
+
+		/*
+		 * A stalled job cannot make progress until the incoming kernel
+		 * reclaims it, so re-running it immediately would just repeat
+		 * the same unhandled exit as fast as the hardware allows.  Back
+		 * off in the architecture's low-power wait (WFE on arm64, PAUSE
+		 * on x86) first.  arch_cpu_preserved_kick() wakes it, so the
+		 * attach signal is still observed promptly by the checks below
+		 * and at the top of the loop.
+		 */
+		if (reason == ONCORE_EXIT_STALL)
+			arch_cpu_preserved_park_wait();
+
+		/* Fast-path: single runnable job continues uninterrupted */
+		if (READ_ONCE(rq->nr_runnable) == 0 &&
+		    READ_ONCE(curr->state) == ONCORE_JOB_RUNNING &&
+		    reason != ONCORE_EXIT_ATTACH_SIGNALED &&
+		    reason != ONCORE_EXIT_ERROR &&
+		    !cpu_preserved_should_exit(cpu)) {
+			continue;
+		}
+
+		/* 5. Handle exit and return job to queue */
+		oncore_rq_lock(rq);
+		if (curr->state == ONCORE_JOB_CANCELING ||
+		    curr->state == ONCORE_JOB_DEAD ||
+		    reason == ONCORE_EXIT_ATTACH_SIGNALED ||
+		    reason == ONCORE_EXIT_ERROR) {
+			WRITE_ONCE(curr->state, ONCORE_JOB_DEAD);
+			oncore_rq_unlock(rq);
+			curr = NULL;
+			continue;
+		}
+		oncore_rq_unlock(rq);
+
+		oncore_sched_put_prev(rq, curr);
+		curr = NULL;
+	}
+}
+
+static void __cpu_preserved_text oncore_sched_cpu_worker(void *data)
+{
+	struct cpu_preserved_stack_context *sctx = cpu_preserved_get_stack_context();
+	struct oncore_session *sess;
+	int cpu;
+
+	if (!sctx)
+		return;
+
+	sess = sctx->workload_context ?
+		(struct oncore_session *)(uintptr_t)sctx->workload_context :
+		data;
+	if (!sess)
+		return;
+	cpu = sctx->cpu;
+
+	if (sctx->session_pgd_pa)
+		arch_cpu_preserved_switch_pgd(sctx->session_pgd_pa);
+
+	oncore_cpu_schedule_loop(cpu, &sess->rq, &sess->sched_config);
+}
+
+static int __init oncore_sched_init(void)
+{
+	if (!arch_oncore_counter_freq_hz()) {
+		pr_err("Counter frequency is unknown; on-core scheduler disabled\n");
+		return -ENODEV;
+	}
+	if (!global_oncore_sched_config.quantum_ms)
+		global_oncore_sched_config.quantum_ms = ONCORE_DEFAULT_QUANTUM_MS;
+	oncore_sched_update_ticks();
+	pr_info("Round-Robin scheduler initialized (quantum=%u ms, ticks=%llu)\n",
+		global_oncore_sched_config.quantum_ms,
+		global_oncore_sched_config.quantum_ticks);
+	return 0;
+}
+early_initcall(oncore_sched_init);
+
+/**
+ * oncore_session_map_range - Map a physical memory range into the session's address space
+ * @sess: On-Core session (may be %NULL to map into the global preserved address space).
+ * @pa:   Start physical address of the range.
+ * @va:   Target virtual address in the isolated page tables.
+ * @size: Size of the mapping in bytes.
+ * @prot: Page protection flags.
+ *
+ * Maps `[pa, pa + size)` at `@va` in the isolated page tables used by
+ * preserved CPUs executing workloads in @sess across kexec.
+ *
+ * Return: 0 on success, or a negative errno on failure.
+ */
+int oncore_session_map_range(struct oncore_session *sess, phys_addr_t pa,
+			     unsigned long va, size_t size, pgprot_t prot)
+{
+	return cpu_preserved_map_range(pa, va, size, prot);
+}
+EXPORT_SYMBOL_GPL(oncore_session_map_range);
+
+/**
+ * oncore_session_map_buffer - Map a direct-map kernel buffer into the session's address space
+ * @sess: On-Core session.
+ * @va:   Kernel direct-map virtual address of the buffer (no-op if %NULL).
+ * @size: Size of the buffer in bytes (no-op if 0).
+ *
+ * Convenience helper that resolves `virt_to_phys(@va)` and maps the buffer at
+ * its existing kernel virtual address `@va` with %PAGE_KERNEL permissions in
+ * the session's isolated page tables.
+ *
+ * Return: 0 on success, or a negative errno on failure.
+ */
+int oncore_session_map_buffer(struct oncore_session *sess, void *va,
+			      size_t size)
+{
+	if (!va || !size)
+		return 0;
+	return oncore_session_map_range(sess, virt_to_phys(va),
+					(unsigned long)va, size, PAGE_KERNEL);
+}
+EXPORT_SYMBOL_GPL(oncore_session_map_buffer);
+
+/**
+ * oncore_session_get_pgd_pa - Return the root page table physical address of a session
+ * @sess: On-Core session.
+ *
+ * Return: Physical address of the session's isolated PGD, or 0 if @sess or its
+ *         address space is %NULL.
+ */
+phys_addr_t oncore_session_get_pgd_pa(struct oncore_session *sess)
+{
+	return (sess && sess->as) ? sess->as->pgd_pa : 0;
+}
+EXPORT_SYMBOL_GPL(oncore_session_get_pgd_pa);
+
+static struct oncore_session *oncore_find_session_locked(const char *sname)
+{
+	struct oncore_session *sess;
+
+	if (!sname || !sname[0])
+		return NULL;
+
+	list_for_each_entry(sess, &oncore_sessions, node) {
+		if (strcmp(sess->ser->session_name, sname) == 0)
+			return sess;
+	}
+
+	return NULL;
+}
+
+static struct oncore_session *oncore_find_session(struct liveupdate_session *s)
+{
+	guard(mutex)(&oncore_sessions_lock);
+
+	return oncore_find_session_locked(liveupdate_session_name(s));
+}
+
+static struct oncore_session *oncore_create_session_locked(const char *sname)
+{
+	unsigned int nr_ser_words = BITS_TO_U64(nr_cpu_ids);
+	struct oncore_session *sess;
+	size_t ser_sz;
+
+	ser_sz = struct_size(sess->ser, cpus_bitmap, nr_ser_words);
+
+	sess = kho_alloc_preserve(sizeof(*sess));
+	if (IS_ERR(sess))
+		return NULL;
+
+	memset(sess, 0, sizeof(*sess));
+	mutex_init(&sess->lock);
+	INIT_LIST_HEAD(&sess->jobs);
+	oncore_runqueue_init(&sess->rq);
+	oncore_sched_update_ticks();
+	sess->sched_config = global_oncore_sched_config;
+
+	sess->as = cpu_preserved_as_create();
+	if (IS_ERR(sess->as))
+		sess->as = NULL;
+
+	oncore_session_map_range(sess, virt_to_phys(sess),
+				 (unsigned long)sess, sizeof(*sess),
+				 PAGE_KERNEL);
+
+	sess->ser = kho_alloc_preserve(ser_sz);
+	if (IS_ERR(sess->ser)) {
+		cpu_preserved_as_destroy(sess->as);
+		kho_unpreserve_free(sess);
+		return NULL;
+	}
+
+	memset(sess->ser, 0, ser_sz);
+	sess->ser->nr_cpu_words = nr_ser_words;
+	strscpy(sess->ser->session_name, sname, sizeof(sess->ser->session_name));
+	sess->ser->sess_pa = virt_to_phys(sess);
+	KHOSER_STORE_PTR(sess->ser->as, sess->as ? sess->as->ser : NULL);
+
+	list_add_tail(&sess->node, &oncore_sessions);
+	return sess;
+}
+
+static struct oncore_session *oncore_get_or_create_session(struct liveupdate_session *s)
+{
+	const char *sname = liveupdate_session_name(s);
+	struct oncore_session *sess;
+
+	if (!sname || !sname[0])
+		return NULL;
+
+	guard(mutex)(&oncore_sessions_lock);
+
+	sess = oncore_find_session_locked(sname);
+	if (sess)
+		return sess;
+
+	return oncore_create_session_locked(sname);
+}
+
+/**
+ * oncore_session_add_cpu - Attach a preserved physical CPU to an On-Core session
+ * @s:   Live Update session handle.
+ * @cpu: Logical ID of the preserved physical CPU to add.
+ *
+ * Creates the On-Core session for @s if it does not yet exist, adds @cpu to
+ * the session's CPU mask, configures @cpu's preserved stack context with the
+ * session pointer and isolated PGD physical address, and attaches
+ * oncore_sched_cpu_worker() so @cpu begins servicing the session's runqueue.
+ *
+ * Return: 0 on success, or a negative errno on failure.
+ */
+int oncore_session_add_cpu(struct liveupdate_session *s, int cpu)
+{
+	struct oncore_session *sess;
+	int ret = 0;
+
+	if (!arch_oncore_counter_freq_hz()) {
+		pr_err("On-core counter frequency is unknown; refusing to add CPU\n");
+		return -ENODEV;
+	}
+
+	sess = oncore_get_or_create_session(s);
+	if (!sess || cpu < 0 || cpu >= nr_cpu_ids)
+		return -EINVAL;
+
+	guard(mutex)(&sess->lock);
+	cpumask_set_cpu(cpu, oncore_session_cpus(sess));
+
+	cpu_preserved_set_workload_context(cpu, sess,
+					   oncore_session_get_pgd_pa(sess));
+
+	ret = cpu_preserved_attach_workload(cpu,
+					    oncore_sched_cpu_worker,
+					    sess);
+	if (ret) {
+		cpu_preserved_set_workload_context(cpu, NULL, 0);
+		cpumask_clear_cpu(cpu, oncore_session_cpus(sess));
+		return ret;
+	}
+
+	return 0;
+}
+
+static void oncore_sync_jobs_pa(struct oncore_session *sess)
+{
+	phys_addr_t *tail = &sess->first_job_pa;
+	struct oncore_job *j;
+
+	list_for_each_entry(j, &sess->jobs, sess_node) {
+		j->next_job_pa = 0;
+		*tail = virt_to_phys(j);
+		tail = &j->next_job_pa;
+	}
+	*tail = 0;
+}
+
+static void oncore_session_destroy_incoming(struct oncore_session *sess)
+{
+	struct oncore_session *old_sess;
+	struct oncore_job *job;
+	phys_addr_t job_pa;
+
+	if (!sess->ser) {
+		kfree(sess);
+		return;
+	}
+
+	if (sess->ser->sess_pa) {
+		old_sess = phys_to_virt(sess->ser->sess_pa);
+		job_pa = old_sess->first_job_pa;
+		while (job_pa) {
+			job = phys_to_virt(job_pa);
+			job_pa = job->next_job_pa;
+			kho_restore_free(job);
+		}
+		kho_restore_free(old_sess);
+	}
+
+	kho_restore_free(sess->ser);
+	kfree(sess);
+}
+
+static void oncore_session_destroy_outgoing(struct oncore_session *sess)
+{
+	struct oncore_job *job, *tmp;
+
+	list_for_each_entry_safe(job, tmp, &sess->jobs, sess_node) {
+		list_del_init(&job->sess_node);
+		kho_unpreserve_free(job);
+	}
+
+	if (sess->ser)
+		kho_unpreserve_free(sess->ser);
+	kho_unpreserve_free(sess);
+}
+
+static void oncore_session_destroy(struct oncore_session *sess)
+{
+	scoped_guard(mutex, &oncore_sessions_lock)
+		list_del_init(&sess->node);
+
+	cpu_preserved_as_destroy(sess->as);
+
+	if (sess->is_incoming)
+		oncore_session_destroy_incoming(sess);
+	else
+		oncore_session_destroy_outgoing(sess);
+}
+
+/**
+ * oncore_session_remove_cpu - Detach a preserved physical CPU from an On-Core session
+ * @s:   Live Update session handle.
+ * @cpu: Logical ID of the preserved physical CPU to remove.
+ *
+ * Detaches the On-Core scheduler worker from @cpu and clears @cpu from the
+ * session's CPU mask.  When the last CPU in the session is removed, destroys
+ * the session's isolated address space and reclaims all KHO-preserved job and
+ * session structures (via kho_restore_free() in the incoming kernel after
+ * kexec, or kho_unpreserve_free() in the outgoing kernel on cancellation).
+ */
+void oncore_session_remove_cpu(struct liveupdate_session *s, int cpu)
+{
+	struct oncore_session *sess = oncore_find_session(s);
+
+	if (!sess || cpu < 0 || cpu >= nr_cpu_ids)
+		return;
+
+	scoped_guard(mutex, &sess->lock)
+		cpumask_clear_cpu(cpu, oncore_session_cpus(sess));
+
+	cpu_preserved_detach_workload(cpu);
+	cpu_preserved_set_workload_context(cpu, NULL, 0);
+
+	if (cpumask_empty(oncore_session_cpus(sess)))
+		oncore_session_destroy(sess);
+}
+
+/**
+ * oncore_session_get_ser - Retrieve the KHO serialization block for an On-Core session
+ * @s: Live Update session handle.
+ *
+ * Called by the preserved-CPU LUO file handler (cpu_preserve) when serializing
+ * a preserved CPU file descriptor so it can store a KHO pointer
+ * (&cpu_preserved_file_ser.oncore) to the session's metadata across kexec.
+ *
+ * Return: Pointer to the KHO-preserved &struct oncore_session_ser for @s, or
+ *         %NULL if @s has no On-Core session.
+ */
+struct oncore_session_ser *oncore_session_get_ser(struct liveupdate_session *s)
+{
+	struct oncore_session *sess = oncore_find_session(s);
+
+	return sess ? sess->ser : NULL;
+}
+
+/**
+ * oncore_session_restore - Reconstruct an On-Core session in the incoming kernel
+ * @s:   Incoming Live Update session handle.
+ * @ser: KHO-preserved session metadata handed over from the outgoing kernel.
+ *
+ * Adopts the KHO-preserved session metadata and isolated page tables (@ser->as)
+ * in the incoming kernel so that subsequent calls to oncore_session_remove_cpu()
+ * as preserved CPUs reattach to Linux can cleanly release the preserved page
+ * tables, jobs, and session structures.  Idempotent if @s has already been
+ * restored.
+ */
+void oncore_session_restore(struct liveupdate_session *s,
+			    struct oncore_session_ser *ser)
+{
+	struct cpu_preserved_as_ser *as_ser;
+	struct oncore_session *sess;
+	const char *sname;
+
+	if (!ser)
+		return;
+
+	sname = liveupdate_session_name(s);
+	if (!sname || !sname[0])
+		sname = ser->session_name;
+
+	guard(mutex)(&oncore_sessions_lock);
+
+	if (oncore_find_session_locked(sname))
+		return;
+
+	sess = kzalloc_obj(*sess, GFP_KERNEL);
+	if (!sess)
+		return;
+
+	mutex_init(&sess->lock);
+	INIT_LIST_HEAD(&sess->jobs);
+	oncore_runqueue_init(&sess->rq);
+	sess->is_incoming = true;
+	sess->ser = ser;
+
+	as_ser = KHOSER_LOAD_PTR(ser->as);
+	if (as_ser)
+		sess->as = cpu_preserved_as_adopt(as_ser);
+
+	list_add_tail(&sess->node, &oncore_sessions);
+}
+
+static unsigned int oncore_session_cpu_job_count(struct oncore_session *sess,
+						 int cpu)
+{
+	struct oncore_job *j;
+	unsigned int count = 0;
+
+	list_for_each_entry(j, &sess->jobs, sess_node) {
+		if (j->assigned_cpu == cpu)
+			count++;
+	}
+
+	return count;
+}
+
+static int oncore_select_job_cpu(struct oncore_session *sess)
+{
+	struct cpumask *cpus = oncore_session_cpus(sess);
+	unsigned int min_count = UINT_MAX, count;
+	int cpu, min_cpu = -1;
+
+	for_each_cpu(cpu, cpus) {
+		count = oncore_session_cpu_job_count(sess, cpu);
+		if (count < min_count) {
+			min_count = count;
+			min_cpu = cpu;
+		}
+	}
+
+	return min_cpu;
+}
+
+/**
+ * oncore_session_submit_job - Allocate and register a workload job in an On-Core session
+ * @s:      Live Update session handle.
+ * @run_fn: Workload callback executed on a preserved physical CPU.
+ * @data:   Opaque context pointer passed to @run_fn (may be %NULL and set
+ *          later via oncore_job_set_data() before activation).
+ *
+ * Allocates a KHO-preserved &struct oncore_job, maps it into the session's
+ * isolated address space, assigns it to the least-loaded preserved CPU in the
+ * session, and links it into @s.  The job is not placed on the runqueue until
+ * oncore_session_activate_job() is called.
+ *
+ * Return: Pointer to the allocated &struct oncore_job on success, %NULL if @s
+ *         has no preserved CPUs, or an ERR_PTR() on failure.
+ */
+struct oncore_job *oncore_session_submit_job(struct liveupdate_session *s,
+					     oncore_job_fn run_fn,
+					     void *data)
+{
+	struct oncore_session *sess = oncore_find_session(s);
+	struct oncore_job *job;
+
+	if (!run_fn)
+		return ERR_PTR(-EINVAL);
+	if (!sess)
+		return NULL;
+
+	guard(mutex)(&sess->lock);
+	if (cpumask_empty(oncore_session_cpus(sess)))
+		return NULL;
+
+	job = kho_alloc_preserve(sizeof(*job));
+	if (IS_ERR(job))
+		return job;
+
+	memset(job, 0, sizeof(*job));
+	oncore_session_map_buffer(sess, job, sizeof(*job));
+	INIT_LIST_HEAD(&job->node);
+	INIT_LIST_HEAD(&job->sess_node);
+	job->session = sess;
+	job->state = ONCORE_JOB_NEW;
+	job->run_fn = run_fn;
+	job->data = data;
+	job->last_cpu = -1;
+	job->assigned_cpu = oncore_select_job_cpu(sess);
+
+	list_add_tail(&job->sess_node, &sess->jobs);
+	oncore_sync_jobs_pa(sess);
+
+	return job;
+}
+EXPORT_SYMBOL_GPL(oncore_session_submit_job);
+
+/**
+ * oncore_session_activate_job - Map a submitted job's data and enqueue it for execution
+ * @s:   Live Update session handle.
+ * @job: Job previously returned by oncore_session_submit_job().
+ *
+ * Maps the first page of @job->data (if non-%NULL) into the session's isolated
+ * address space, places @job onto the session's round-robin FIFO runqueue, and
+ * kicks the assigned preserved physical CPU so it wakes from low-power park
+ * wait and begins executing @job.
+ *
+ * Return: 0 on success, or -EINVAL if @s or @job is invalid.
+ */
+int oncore_session_activate_job(struct liveupdate_session *s,
+				struct oncore_job *job)
+{
+	struct oncore_session *sess = oncore_find_session(s);
+
+	if (!sess || !job)
+		return -EINVAL;
+
+	guard(mutex)(&sess->lock);
+	if (job->data)
+		oncore_session_map_buffer(sess, job->data, PAGE_SIZE);
+	oncore_sched_enqueue(&sess->rq, job);
+	if (cpu_is_preserved(job->assigned_cpu))
+		arch_cpu_preserved_kick(job->assigned_cpu);
+
+	return 0;
+}
+EXPORT_SYMBOL_GPL(oncore_session_activate_job);
+
+/**
+ * oncore_job_set_data - Set the opaque argument passed to a job's run callback
+ * @job:  Job to update.
+ * @data: Pointer handed to @job's run_fn.  May be %NULL.
+ *
+ * Callers that cannot determine @data at submission time submit with %NULL and
+ * call this once the object exists.  It must be called before
+ * oncore_session_activate_job(), which is what maps @data into the session's
+ * address space.
+ */
+void oncore_job_set_data(struct oncore_job *job, void *data)
+{
+	if (job)
+		job->data = data;
+}
+EXPORT_SYMBOL_GPL(oncore_job_set_data);
+
+/**
+ * oncore_job_cpu - Return the preserved physical CPU assigned to a job
+ * @job: Job to query.
+ *
+ * Return: Logical CPU ID assigned to @job, or -1 if @job is %NULL.
+ */
+int oncore_job_cpu(const struct oncore_job *job)
+{
+	return job ? job->assigned_cpu : -1;
+}
+EXPORT_SYMBOL_GPL(oncore_job_cpu);
+
+/**
+ * oncore_job_session - Return the On-Core session that owns a job
+ * @job: Job to query.
+ *
+ * Return: Pointer to the owning &struct oncore_session, or %NULL if @job is %NULL.
+ */
+struct oncore_session *oncore_job_session(const struct oncore_job *job)
+{
+	return job ? job->session : NULL;
+}
+EXPORT_SYMBOL_GPL(oncore_job_session);
+
+/**
+ * oncore_session_cancel_job - Stop and free a submitted or running On-Core job
+ * @s:   Live Update session handle.
+ * @job: Job to cancel.
+ *
+ * Removes @job from the session and runqueue.  If @job is currently executing
+ * on a preserved physical CPU (%ONCORE_JOB_CANCELING), kicks that CPU and
+ * waits for the current scheduling quantum to finish before unpreserving and
+ * freeing @job.
+ *
+ * Return: 0 on success, or -EINVAL if @s or @job is invalid.
+ */
+int oncore_session_cancel_job(struct liveupdate_session *s,
+			      struct oncore_job *job)
+{
+	struct oncore_session *sess = oncore_find_session(s);
+	int cpu, retries;
+
+	if (!sess || !job)
+		return -EINVAL;
+
+	guard(mutex)(&sess->lock);
+	list_del_init(&job->sess_node);
+	oncore_sync_jobs_pa(sess);
+	if (job->assigned_cpu >= 0)
+		job->assigned_cpu = -1;
+	oncore_sched_dequeue(&sess->rq, job);
+
+	if (READ_ONCE(job->state) == ONCORE_JOB_CANCELING) {
+		cpu = READ_ONCE(job->last_cpu);
+		retries = 0;
+
+		while (READ_ONCE(job->state) == ONCORE_JOB_CANCELING &&
+		       retries < (ONCORE_CANCEL_TIMEOUT_US / ONCORE_CANCEL_STEP_US)) {
+			if ((retries % 50) == 0 && cpu >= 0)
+				arch_cpu_preserved_kick(cpu);
+			udelay(ONCORE_CANCEL_STEP_US);
+			retries++;
+		}
+	}
+
+	kho_unpreserve_free(job);
+
+	return 0;
+}
+EXPORT_SYMBOL_GPL(oncore_session_cancel_job);
-- 
2.55.0.1082.g2b9226bbc0-goog




More information about the kexec mailing list