Files
laptops-kernel/include/linux/memory.h
Gregory PriceandAndrew Morton 23860341bf mm/memory: add memory_block_aligned_range() helper
Patch series "dax/kmem: atomic whole-device hotplug via sysfs", v7.

The dax kmem driver onlines memory during probe using the system default
policy, with no atomic control for the state of an entire region at
runtime - only by toggling individual memory blocks.

Offlining and removing a whole region therefore races with other userland
controllers that interfere between the two steps.

This series adds a sysfs "state" attribute for atomic whole-device hotplug
control, plus the mm and dax plumbing to support it.

Transitions are atomic across every range of the device.  The state names
mirror the per-block memoryX/state ABI with one modification:

  - "unplugged":      memory blocks are not present
  - "online":         online as system RAM, zone chosen by the kernel
  - "online_kernel":  online in ZONE_NORMAL
  - "online_movable": online in ZONE_MOVABLE

"offline" (blocks present but offline) is reportable for backward
compatibility but is not writable because it entices the race condition we
are trying to solve (separate atomic steps for offline and unplug).

'unplugged' (atomic offline+remove of the whole device) is the new
capability provided by the new kmem sysfs attribute.

dax/kmem probe still creates the memory blocks by default when the default
policy is "offline", to preserve backwards compatibility.


This patch (of 10):

Memory hotplug operations require ranges aligned to memory block
boundaries.  This is a generic operation for hotplug.

Add memory_block_aligned_range() as a common helper in <linux/memory.h>
that aligns the start address up and end address down to memory block
boundaries.  Guard against end underflow when the range falls below the
first memory block boundary, returning an empty range instead.

Update dax/kmem to use this helper.

Link: https://lore.kernel.org/20260712154505.3564379-1-gourry@gourry.net
Link: https://lore.kernel.org/20260712154505.3564379-2-gourry@gourry.net
Signed-off-by: Gregory Price <gourry@gourry.net>
Reviewed-by: Dave Jiang <dave.jiang@intel.com>
Reviewed-by: Dan Williams <djbw@kernel.org>
Acked-by: David Hildenbrand (Arm) <david@kernel.org>
Cc: Alison Schofield <alison.schofield@intel.com>
Cc: Danilo Krummrich <dakr@kernel.org>
Cc: Greg Kroah-Hartman <gregkh@linuxfoundation.org>
Cc: Liam R. Howlett <liam@infradead.org>
Cc: Lorenzo Stoakes <ljs@kernel.org>
Cc: Michal Hocko <mhocko@suse.com>
Cc: Mike Rapoport <rppt@kernel.org>
Cc: Oscar Salvador <osalvador@suse.de>
Cc: "Rafael J. Wysocki" <rafael@kernel.org>
Cc: Shuah Khan <shuah@kernel.org>
Cc: Suren Baghdasaryan <surenb@google.com>
Cc: Vishal Verma <vishal.l.verma@intel.com>
Cc: Vlastimil Babka <vbabka@kernel.org>
Cc: Hannes Reinecke <hare@suse.de>
Cc: Pankaj Gupta <pankaj.gupta@amd.com>
Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
2026-07-30 19:48:59 -07:00

242 lines
7.7 KiB
C

/* SPDX-License-Identifier: GPL-2.0 */
/*
* include/linux/memory.h - generic memory definition
*
* This is mainly for topological representation. We define the
* basic "struct memory_block" here, which can be embedded in per-arch
* definitions or NUMA information.
*
* Basic handling of the devices is done in drivers/base/memory.c
* and system devices are handled in drivers/base/sys.c.
*
* Memory block are exported via sysfs in the class/memory/devices/
* directory.
*
*/
#ifndef _LINUX_MEMORY_H_
#define _LINUX_MEMORY_H_
#include <linux/node.h>
#include <linux/compiler.h>
#include <linux/mutex.h>
#include <linux/memory_hotplug.h>
#include <linux/range.h>
#define MIN_MEMORY_BLOCK_SIZE (1UL << SECTION_SIZE_BITS)
/**
* struct memory_group - a logical group of memory blocks
* @nid: The node id for all memory blocks inside the memory group.
* @memory_blocks: List of all memory blocks belonging to this memory group.
* @present_kernel_pages: Present (online) memory outside ZONE_MOVABLE of this
* memory group.
* @present_movable_pages: Present (online) memory in ZONE_MOVABLE of this
* memory group.
* @is_dynamic: The memory group type: static vs. dynamic
* @s.max_pages: Valid with &memory_group.is_dynamic == false. The maximum
* number of pages we'll have in this static memory group.
* @d.unit_pages: Valid with &memory_group.is_dynamic == true. Unit in pages
* in which memory is added/removed in this dynamic memory group.
* This granularity defines the alignment of a unit in physical
* address space; it has to be at least as big as a single
* memory block.
*
* A memory group logically groups memory blocks; each memory block
* belongs to at most one memory group. A memory group corresponds to
* a memory device, such as a DIMM or a NUMA node, which spans multiple
* memory blocks and might even span multiple non-contiguous physical memory
* ranges.
*
* Modification of members after registration is serialized by memory
* hot(un)plug code.
*/
struct memory_group {
int nid;
struct list_head memory_blocks;
unsigned long present_kernel_pages;
unsigned long present_movable_pages;
bool is_dynamic;
union {
struct {
unsigned long max_pages;
} s;
struct {
unsigned long unit_pages;
} d;
};
};
enum memory_block_state {
/* These states are exposed to userspace as text strings in sysfs */
MEM_ONLINE, /* exposed to userspace */
MEM_GOING_OFFLINE, /* exposed to userspace */
MEM_OFFLINE, /* exposed to userspace */
MEM_GOING_ONLINE,
MEM_CANCEL_ONLINE,
MEM_CANCEL_OFFLINE,
};
struct memory_block {
unsigned long start_section_nr;
enum memory_block_state state; /* serialized by the dev->lock */
enum mmop online_type; /* for passing data to online routine */
int nid; /* NID for this memory block */
/*
* The single zone of this memory block if all PFNs of this memory block
* that are System RAM (not a memory hole, not ZONE_DEVICE ranges) are
* managed by a single zone. NULL if multiple zones (including nodes)
* apply.
*/
struct zone *zone;
struct device dev;
struct vmem_altmap *altmap;
struct memory_group *group; /* group (if any) for this block */
struct list_head group_next; /* next block inside memory group */
#if defined(CONFIG_MEMORY_FAILURE) && defined(CONFIG_MEMORY_HOTPLUG)
atomic_long_t nr_hwpoison;
#endif
};
int arch_get_memory_phys_device(unsigned long start_pfn);
unsigned long memory_block_size_bytes(void);
int set_memory_block_size_order(unsigned int order);
/**
* memory_block_aligned_range - align a physical address range to memory blocks
* @range: the input range to align
*
* Aligns the start address up and the end address down to memory block
* boundaries. This is required for memory hotplug operations which must
* operate on memory-block aligned ranges.
*
* Returns the aligned range. Callers should check that the returned
* range is valid (aligned.start < aligned.end) before using it.
*/
static inline struct range memory_block_aligned_range(const struct range *range)
{
struct range aligned;
aligned.start = ALIGN(range->start, memory_block_size_bytes());
aligned.end = ALIGN_DOWN(range->end + 1, memory_block_size_bytes());
/* No whole block fits (e.g. range below the first boundary): empty. */
if (aligned.end <= aligned.start)
aligned.start = aligned.end;
else
aligned.end -= 1;
return aligned;
}
struct memory_notify {
unsigned long start_pfn;
unsigned long nr_pages;
};
struct notifier_block;
struct mem_section;
/*
* Priorities for the hotplug memory callback routines. Invoked from
* high to low. Higher priorities correspond to higher numbers.
*/
#define DEFAULT_CALLBACK_PRI 0
#define SLAB_CALLBACK_PRI 1
#define CXL_CALLBACK_PRI 5
#define HMAT_CALLBACK_PRI 6
#define MM_COMPUTE_BATCH_PRI 10
#define CPUSET_CALLBACK_PRI 10
#define MEMTIER_HOTPLUG_PRI 100
#define KSM_CALLBACK_PRI 100
#ifndef CONFIG_MEMORY_HOTPLUG
static inline void memory_dev_init(void)
{
return;
}
static inline int register_memory_notifier(struct notifier_block *nb)
{
return 0;
}
static inline void unregister_memory_notifier(struct notifier_block *nb)
{
}
static inline int memory_notify(enum memory_block_state state, void *v)
{
return 0;
}
static inline int hotplug_memory_notifier(notifier_fn_t fn, int pri)
{
return 0;
}
static inline int memory_block_advise_max_size(unsigned long size)
{
return -ENODEV;
}
static inline unsigned long memory_block_advised_max_size(void)
{
return 0;
}
#else /* CONFIG_MEMORY_HOTPLUG */
extern int register_memory_notifier(struct notifier_block *nb);
extern void unregister_memory_notifier(struct notifier_block *nb);
int create_memory_block_devices(unsigned long start, unsigned long size,
int nid, struct vmem_altmap *altmap,
struct memory_group *group);
void remove_memory_block_devices(unsigned long start, unsigned long size);
extern void memory_dev_init(void);
extern int memory_notify(enum memory_block_state state, void *v);
struct memory_block *memory_block_get(unsigned long block_id);
static inline void memory_block_put(struct memory_block *mem)
{
put_device(&mem->dev);
}
typedef int (*walk_memory_blocks_func_t)(struct memory_block *, void *);
extern int walk_memory_blocks(unsigned long start, unsigned long size,
void *arg, walk_memory_blocks_func_t func);
extern int for_each_memory_block(void *arg, walk_memory_blocks_func_t func);
extern int memory_group_register_static(int nid, unsigned long max_pages);
extern int memory_group_register_dynamic(int nid, unsigned long unit_pages);
extern int memory_group_unregister(int mgid);
struct memory_group *memory_group_find_by_id(int mgid);
typedef int (*walk_memory_groups_func_t)(struct memory_group *, void *);
int walk_dynamic_memory_groups(int nid, walk_memory_groups_func_t func,
struct memory_group *excluded, void *arg);
#define hotplug_memory_notifier(fn, pri) ({ \
static __meminitdata struct notifier_block fn##_mem_nb =\
{ .notifier_call = fn, .priority = pri };\
register_memory_notifier(&fn##_mem_nb); \
})
extern int sections_per_block;
static inline unsigned long memory_block_id(unsigned long section_nr)
{
return section_nr / sections_per_block;
}
static inline unsigned long pfn_to_block_id(unsigned long pfn)
{
return memory_block_id(pfn_to_section_nr(pfn));
}
static inline unsigned long phys_to_block_id(unsigned long phys)
{
return pfn_to_block_id(PFN_DOWN(phys));
}
#ifdef CONFIG_NUMA
void memory_block_add_nid_early(struct memory_block *mem, int nid);
#endif /* CONFIG_NUMA */
int memory_block_advise_max_size(unsigned long size);
unsigned long memory_block_advised_max_size(void);
#endif /* CONFIG_MEMORY_HOTPLUG */
/*
* Kernel text modification mutex, used for code patching. Users of this lock
* can sleep.
*/
extern struct mutex text_mutex;
#endif /* _LINUX_MEMORY_H_ */