mirror of
https://github.com/linux-msm/laptops-kernel.git
synced 2026-08-13 14:19:53 -07:00
Merge tag 'perf-core-2024-11-18' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip
Pull performance events updates from Ingo Molnar:
"Uprobes:
- Add BPF session support (Jiri Olsa)
- Switch to RCU Tasks Trace flavor for better performance (Andrii
Nakryiko)
- Massively increase uretprobe SMP scalability by SRCU-protecting
the uretprobe lifetime (Andrii Nakryiko)
- Kill xol_area->slot_count (Oleg Nesterov)
Core facilities:
- Implement targeted high-frequency profiling by adding the ability
for an event to "pause" or "resume" AUX area tracing (Adrian
Hunter)
VM profiling/sampling:
- Correct perf sampling with guest VMs (Colton Lewis)
New hardware support:
- x86/intel: Add PMU support for Intel ArrowLake-H CPUs (Dapeng Mi)
Misc fixes and enhancements:
- x86/intel/pt: Fix buffer full but size is 0 case (Adrian Hunter)
- x86/amd: Warn only on new bits set (Breno Leitao)
- x86/amd/uncore: Avoid a false positive warning about snprintf
truncation in amd_uncore_umc_ctx_init (Jean Delvare)
- uprobes: Re-order struct uprobe_task to save some space
(Christophe JAILLET)
- x86/rapl: Move the pmu allocation out of CPU hotplug (Kan Liang)
- x86/rapl: Clean up cpumask and hotplug (Kan Liang)
- uprobes: Deuglify xol_get_insn_slot/xol_free_insn_slot paths (Oleg
Nesterov)"
* tag 'perf-core-2024-11-18' of git://git.kernel.org/pub/scm/linux/kernel/git/tip/tip: (32 commits)
perf/core: Correct perf sampling with guest VMs
perf/x86: Refactor misc flag assignments
perf/powerpc: Use perf_arch_instruction_pointer()
perf/core: Hoist perf_instruction_pointer() and perf_misc_flags()
perf/arm: Drop unused functions
uprobes: Re-order struct uprobe_task to save some space
perf/x86/amd/uncore: Avoid a false positive warning about snprintf truncation in amd_uncore_umc_ctx_init
perf/x86/intel: Do not enable large PEBS for events with aux actions or aux sampling
perf/x86/intel/pt: Add support for pause / resume
perf/core: Add aux_pause, aux_resume, aux_start_paused
perf/x86/intel/pt: Fix buffer full but size is 0 case
uprobes: SRCU-protect uretprobe lifetime (with timeout)
uprobes: allow put_uprobe() from non-sleepable softirq context
perf/x86/rapl: Clean up cpumask and hotplug
perf/x86/rapl: Move the pmu allocation out of CPU hotplug
uprobe: Add support for session consumer
uprobe: Add data pointer to consumer handlers
perf/x86/amd: Warn only on new bits set
uprobes: fold xol_take_insn_slot() into xol_get_insn_slot()
uprobes: kill xol_area->slot_count
...
This commit is contained in:
@@ -135,6 +135,7 @@ config KPROBES_ON_FTRACE
|
||||
config UPROBES
|
||||
def_bool n
|
||||
depends on ARCH_SUPPORTS_UPROBES
|
||||
select TASKS_TRACE_RCU
|
||||
help
|
||||
Uprobes is the user-space counterpart to kprobes: they
|
||||
enable instrumentation applications (such as 'perf probe')
|
||||
|
||||
@@ -8,13 +8,6 @@
|
||||
#ifndef __ARM_PERF_EVENT_H__
|
||||
#define __ARM_PERF_EVENT_H__
|
||||
|
||||
#ifdef CONFIG_PERF_EVENTS
|
||||
struct pt_regs;
|
||||
extern unsigned long perf_instruction_pointer(struct pt_regs *regs);
|
||||
extern unsigned long perf_misc_flags(struct pt_regs *regs);
|
||||
#define perf_misc_flags(regs) perf_misc_flags(regs)
|
||||
#endif
|
||||
|
||||
#define perf_arch_fetch_caller_regs(regs, __ip) { \
|
||||
(regs)->ARM_pc = (__ip); \
|
||||
frame_pointer((regs)) = (unsigned long) __builtin_frame_address(0); \
|
||||
|
||||
@@ -96,20 +96,3 @@ perf_callchain_kernel(struct perf_callchain_entry_ctx *entry, struct pt_regs *re
|
||||
arm_get_current_stackframe(regs, &fr);
|
||||
walk_stackframe(&fr, callchain_trace, entry);
|
||||
}
|
||||
|
||||
unsigned long perf_instruction_pointer(struct pt_regs *regs)
|
||||
{
|
||||
return instruction_pointer(regs);
|
||||
}
|
||||
|
||||
unsigned long perf_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
int misc = 0;
|
||||
|
||||
if (user_mode(regs))
|
||||
misc |= PERF_RECORD_MISC_USER;
|
||||
else
|
||||
misc |= PERF_RECORD_MISC_KERNEL;
|
||||
|
||||
return misc;
|
||||
}
|
||||
|
||||
@@ -10,10 +10,6 @@
|
||||
#include <asm/ptrace.h>
|
||||
|
||||
#ifdef CONFIG_PERF_EVENTS
|
||||
struct pt_regs;
|
||||
extern unsigned long perf_instruction_pointer(struct pt_regs *regs);
|
||||
extern unsigned long perf_misc_flags(struct pt_regs *regs);
|
||||
#define perf_misc_flags(regs) perf_misc_flags(regs)
|
||||
#define perf_arch_bpf_user_pt_regs(regs) ®s->user_regs
|
||||
#endif
|
||||
|
||||
|
||||
@@ -38,31 +38,3 @@ void perf_callchain_kernel(struct perf_callchain_entry_ctx *entry,
|
||||
|
||||
arch_stack_walk(callchain_trace, entry, current, regs);
|
||||
}
|
||||
|
||||
unsigned long perf_instruction_pointer(struct pt_regs *regs)
|
||||
{
|
||||
if (perf_guest_state())
|
||||
return perf_guest_get_ip();
|
||||
|
||||
return instruction_pointer(regs);
|
||||
}
|
||||
|
||||
unsigned long perf_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
unsigned int guest_state = perf_guest_state();
|
||||
int misc = 0;
|
||||
|
||||
if (guest_state) {
|
||||
if (guest_state & PERF_GUEST_USER)
|
||||
misc |= PERF_RECORD_MISC_GUEST_USER;
|
||||
else
|
||||
misc |= PERF_RECORD_MISC_GUEST_KERNEL;
|
||||
} else {
|
||||
if (user_mode(regs))
|
||||
misc |= PERF_RECORD_MISC_USER;
|
||||
else
|
||||
misc |= PERF_RECORD_MISC_KERNEL;
|
||||
}
|
||||
|
||||
return misc;
|
||||
}
|
||||
|
||||
@@ -102,8 +102,8 @@ struct power_pmu {
|
||||
int __init register_power_pmu(struct power_pmu *pmu);
|
||||
|
||||
struct pt_regs;
|
||||
extern unsigned long perf_misc_flags(struct pt_regs *regs);
|
||||
extern unsigned long perf_instruction_pointer(struct pt_regs *regs);
|
||||
extern unsigned long perf_arch_misc_flags(struct pt_regs *regs);
|
||||
extern unsigned long perf_arch_instruction_pointer(struct pt_regs *regs);
|
||||
extern unsigned long int read_bhrb(int n);
|
||||
|
||||
/*
|
||||
@@ -111,7 +111,7 @@ extern unsigned long int read_bhrb(int n);
|
||||
* if we have hardware PMU support.
|
||||
*/
|
||||
#ifdef CONFIG_PPC_PERF_CTRS
|
||||
#define perf_misc_flags(regs) perf_misc_flags(regs)
|
||||
#define perf_arch_misc_flags(regs) perf_arch_misc_flags(regs)
|
||||
#endif
|
||||
|
||||
/*
|
||||
|
||||
@@ -51,7 +51,7 @@ perf_callchain_kernel(struct perf_callchain_entry_ctx *entry, struct pt_regs *re
|
||||
|
||||
lr = regs->link;
|
||||
sp = regs->gpr[1];
|
||||
perf_callchain_store(entry, perf_instruction_pointer(regs));
|
||||
perf_callchain_store(entry, perf_arch_instruction_pointer(regs));
|
||||
|
||||
if (!validate_sp(sp, current))
|
||||
return;
|
||||
|
||||
@@ -139,7 +139,7 @@ void perf_callchain_user_32(struct perf_callchain_entry_ctx *entry,
|
||||
long level = 0;
|
||||
unsigned int __user *fp, *uregs;
|
||||
|
||||
next_ip = perf_instruction_pointer(regs);
|
||||
next_ip = perf_arch_instruction_pointer(regs);
|
||||
lr = regs->link;
|
||||
sp = regs->gpr[1];
|
||||
perf_callchain_store(entry, next_ip);
|
||||
|
||||
@@ -74,7 +74,7 @@ void perf_callchain_user_64(struct perf_callchain_entry_ctx *entry,
|
||||
struct signal_frame_64 __user *sigframe;
|
||||
unsigned long __user *fp, *uregs;
|
||||
|
||||
next_ip = perf_instruction_pointer(regs);
|
||||
next_ip = perf_arch_instruction_pointer(regs);
|
||||
lr = regs->link;
|
||||
sp = regs->gpr[1];
|
||||
perf_callchain_store(entry, next_ip);
|
||||
|
||||
@@ -2332,7 +2332,7 @@ static void record_and_restart(struct perf_event *event, unsigned long val,
|
||||
* Called from generic code to get the misc flags (i.e. processor mode)
|
||||
* for an event_id.
|
||||
*/
|
||||
unsigned long perf_misc_flags(struct pt_regs *regs)
|
||||
unsigned long perf_arch_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
u32 flags = perf_get_misc_flags(regs);
|
||||
|
||||
@@ -2346,7 +2346,7 @@ unsigned long perf_misc_flags(struct pt_regs *regs)
|
||||
* Called from generic code to get the instruction pointer
|
||||
* for an event_id.
|
||||
*/
|
||||
unsigned long perf_instruction_pointer(struct pt_regs *regs)
|
||||
unsigned long perf_arch_instruction_pointer(struct pt_regs *regs)
|
||||
{
|
||||
unsigned long siar = mfspr(SPRN_SIAR);
|
||||
|
||||
|
||||
@@ -37,9 +37,9 @@ extern ssize_t cpumf_events_sysfs_show(struct device *dev,
|
||||
|
||||
/* Perf callbacks */
|
||||
struct pt_regs;
|
||||
extern unsigned long perf_instruction_pointer(struct pt_regs *regs);
|
||||
extern unsigned long perf_misc_flags(struct pt_regs *regs);
|
||||
#define perf_misc_flags(regs) perf_misc_flags(regs)
|
||||
extern unsigned long perf_arch_instruction_pointer(struct pt_regs *regs);
|
||||
extern unsigned long perf_arch_misc_flags(struct pt_regs *regs);
|
||||
#define perf_arch_misc_flags(regs) perf_arch_misc_flags(regs)
|
||||
#define perf_arch_bpf_user_pt_regs(regs) ®s->user_regs
|
||||
|
||||
/* Perf pt_regs extension for sample-data-entry indicators */
|
||||
|
||||
@@ -57,7 +57,7 @@ static unsigned long instruction_pointer_guest(struct pt_regs *regs)
|
||||
return sie_block(regs)->gpsw.addr;
|
||||
}
|
||||
|
||||
unsigned long perf_instruction_pointer(struct pt_regs *regs)
|
||||
unsigned long perf_arch_instruction_pointer(struct pt_regs *regs)
|
||||
{
|
||||
return is_in_guest(regs) ? instruction_pointer_guest(regs)
|
||||
: instruction_pointer(regs);
|
||||
@@ -84,7 +84,7 @@ static unsigned long perf_misc_flags_sf(struct pt_regs *regs)
|
||||
return flags;
|
||||
}
|
||||
|
||||
unsigned long perf_misc_flags(struct pt_regs *regs)
|
||||
unsigned long perf_arch_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
/* Check if the cpum_sf PMU has created the pt_regs structure.
|
||||
* In this case, perf misc flags can be easily extracted. Otherwise,
|
||||
|
||||
@@ -943,11 +943,12 @@ static int amd_pmu_v2_snapshot_branch_stack(struct perf_branch_entry *entries, u
|
||||
static int amd_pmu_v2_handle_irq(struct pt_regs *regs)
|
||||
{
|
||||
struct cpu_hw_events *cpuc = this_cpu_ptr(&cpu_hw_events);
|
||||
static atomic64_t status_warned = ATOMIC64_INIT(0);
|
||||
u64 reserved, status, mask, new_bits, prev_bits;
|
||||
struct perf_sample_data data;
|
||||
struct hw_perf_event *hwc;
|
||||
struct perf_event *event;
|
||||
int handled = 0, idx;
|
||||
u64 reserved, status, mask;
|
||||
bool pmu_enabled;
|
||||
|
||||
/*
|
||||
@@ -1012,7 +1013,12 @@ static int amd_pmu_v2_handle_irq(struct pt_regs *regs)
|
||||
* the corresponding PMCs are expected to be inactive according to the
|
||||
* active_mask
|
||||
*/
|
||||
WARN_ON(status > 0);
|
||||
if (status > 0) {
|
||||
prev_bits = atomic64_fetch_or(status, &status_warned);
|
||||
// A new bit was set for the very first time.
|
||||
new_bits = status & ~prev_bits;
|
||||
WARN(new_bits, "New overflows for inactive PMCs: %llx\n", new_bits);
|
||||
}
|
||||
|
||||
/* Clear overflow and freeze bits */
|
||||
amd_pmu_ack_global_status(~status);
|
||||
|
||||
@@ -916,7 +916,8 @@ int amd_uncore_umc_ctx_init(struct amd_uncore *uncore, unsigned int cpu)
|
||||
u8 group_num_pmcs[UNCORE_GROUP_MAX] = { 0 };
|
||||
union amd_uncore_info info;
|
||||
struct amd_uncore_pmu *pmu;
|
||||
int index = 0, gid, i;
|
||||
int gid, i;
|
||||
u16 index = 0;
|
||||
|
||||
if (pmu_version < 2)
|
||||
return 0;
|
||||
@@ -948,7 +949,7 @@ int amd_uncore_umc_ctx_init(struct amd_uncore *uncore, unsigned int cpu)
|
||||
for_each_set_bit(gid, gmask, UNCORE_GROUP_MAX) {
|
||||
for (i = 0; i < group_num_pmus[gid]; i++) {
|
||||
pmu = &uncore->pmus[index];
|
||||
snprintf(pmu->name, sizeof(pmu->name), "amd_umc_%d", index);
|
||||
snprintf(pmu->name, sizeof(pmu->name), "amd_umc_%hu", index);
|
||||
pmu->num_counters = group_num_pmcs[gid] / group_num_pmus[gid];
|
||||
pmu->msr_base = MSR_F19H_UMC_PERF_CTL + i * pmu->num_counters * 2;
|
||||
pmu->rdpmc_base = -1;
|
||||
|
||||
+44
-22
@@ -3003,35 +3003,57 @@ static unsigned long code_segment_base(struct pt_regs *regs)
|
||||
return 0;
|
||||
}
|
||||
|
||||
unsigned long perf_instruction_pointer(struct pt_regs *regs)
|
||||
unsigned long perf_arch_instruction_pointer(struct pt_regs *regs)
|
||||
{
|
||||
if (perf_guest_state())
|
||||
return perf_guest_get_ip();
|
||||
|
||||
return regs->ip + code_segment_base(regs);
|
||||
}
|
||||
|
||||
unsigned long perf_misc_flags(struct pt_regs *regs)
|
||||
static unsigned long common_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
unsigned int guest_state = perf_guest_state();
|
||||
int misc = 0;
|
||||
|
||||
if (guest_state) {
|
||||
if (guest_state & PERF_GUEST_USER)
|
||||
misc |= PERF_RECORD_MISC_GUEST_USER;
|
||||
else
|
||||
misc |= PERF_RECORD_MISC_GUEST_KERNEL;
|
||||
} else {
|
||||
if (user_mode(regs))
|
||||
misc |= PERF_RECORD_MISC_USER;
|
||||
else
|
||||
misc |= PERF_RECORD_MISC_KERNEL;
|
||||
}
|
||||
|
||||
if (regs->flags & PERF_EFLAGS_EXACT)
|
||||
misc |= PERF_RECORD_MISC_EXACT_IP;
|
||||
return PERF_RECORD_MISC_EXACT_IP;
|
||||
|
||||
return misc;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static unsigned long guest_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
unsigned long guest_state = perf_guest_state();
|
||||
|
||||
if (!(guest_state & PERF_GUEST_ACTIVE))
|
||||
return 0;
|
||||
|
||||
if (guest_state & PERF_GUEST_USER)
|
||||
return PERF_RECORD_MISC_GUEST_USER;
|
||||
else
|
||||
return PERF_RECORD_MISC_GUEST_KERNEL;
|
||||
|
||||
}
|
||||
|
||||
static unsigned long host_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
if (user_mode(regs))
|
||||
return PERF_RECORD_MISC_USER;
|
||||
else
|
||||
return PERF_RECORD_MISC_KERNEL;
|
||||
}
|
||||
|
||||
unsigned long perf_arch_guest_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
unsigned long flags = common_misc_flags(regs);
|
||||
|
||||
flags |= guest_misc_flags(regs);
|
||||
|
||||
return flags;
|
||||
}
|
||||
|
||||
unsigned long perf_arch_misc_flags(struct pt_regs *regs)
|
||||
{
|
||||
unsigned long flags = common_misc_flags(regs);
|
||||
|
||||
flags |= host_misc_flags(regs);
|
||||
|
||||
return flags;
|
||||
}
|
||||
|
||||
void perf_get_x86_pmu_capability(struct x86_pmu_capability *cap)
|
||||
|
||||
+123
-14
@@ -3962,8 +3962,8 @@ static int intel_pmu_hw_config(struct perf_event *event)
|
||||
|
||||
if (!(event->attr.freq || (event->attr.wakeup_events && !event->attr.watermark))) {
|
||||
event->hw.flags |= PERF_X86_EVENT_AUTO_RELOAD;
|
||||
if (!(event->attr.sample_type &
|
||||
~intel_pmu_large_pebs_flags(event))) {
|
||||
if (!(event->attr.sample_type & ~intel_pmu_large_pebs_flags(event)) &&
|
||||
!has_aux_action(event)) {
|
||||
event->hw.flags |= PERF_X86_EVENT_LARGE_PEBS;
|
||||
event->attach_state |= PERF_ATTACH_SCHED_CB;
|
||||
}
|
||||
@@ -4599,6 +4599,28 @@ static inline bool erratum_hsw11(struct perf_event *event)
|
||||
X86_CONFIG(.event=0xc0, .umask=0x01);
|
||||
}
|
||||
|
||||
static struct event_constraint *
|
||||
arl_h_get_event_constraints(struct cpu_hw_events *cpuc, int idx,
|
||||
struct perf_event *event)
|
||||
{
|
||||
struct x86_hybrid_pmu *pmu = hybrid_pmu(event->pmu);
|
||||
|
||||
if (pmu->pmu_type == hybrid_tiny)
|
||||
return cmt_get_event_constraints(cpuc, idx, event);
|
||||
|
||||
return mtl_get_event_constraints(cpuc, idx, event);
|
||||
}
|
||||
|
||||
static int arl_h_hw_config(struct perf_event *event)
|
||||
{
|
||||
struct x86_hybrid_pmu *pmu = hybrid_pmu(event->pmu);
|
||||
|
||||
if (pmu->pmu_type == hybrid_tiny)
|
||||
return intel_pmu_hw_config(event);
|
||||
|
||||
return adl_hw_config(event);
|
||||
}
|
||||
|
||||
/*
|
||||
* The HSW11 requires a period larger than 100 which is the same as the BDM11.
|
||||
* A minimum period of 128 is enforced as well for the INST_RETIRED.ALL.
|
||||
@@ -4924,17 +4946,26 @@ static struct x86_hybrid_pmu *find_hybrid_pmu_for_cpu(void)
|
||||
|
||||
/*
|
||||
* This essentially just maps between the 'hybrid_cpu_type'
|
||||
* and 'hybrid_pmu_type' enums:
|
||||
* and 'hybrid_pmu_type' enums except for ARL-H processor
|
||||
* which needs to compare atom uarch native id since ARL-H
|
||||
* contains two different atom uarchs.
|
||||
*/
|
||||
for (i = 0; i < x86_pmu.num_hybrid_pmus; i++) {
|
||||
enum hybrid_pmu_type pmu_type = x86_pmu.hybrid_pmu[i].pmu_type;
|
||||
u32 native_id;
|
||||
|
||||
if (cpu_type == HYBRID_INTEL_CORE &&
|
||||
pmu_type == hybrid_big)
|
||||
return &x86_pmu.hybrid_pmu[i];
|
||||
if (cpu_type == HYBRID_INTEL_ATOM &&
|
||||
pmu_type == hybrid_small)
|
||||
if (cpu_type == HYBRID_INTEL_CORE && pmu_type == hybrid_big)
|
||||
return &x86_pmu.hybrid_pmu[i];
|
||||
if (cpu_type == HYBRID_INTEL_ATOM) {
|
||||
if (x86_pmu.num_hybrid_pmus == 2 && pmu_type == hybrid_small)
|
||||
return &x86_pmu.hybrid_pmu[i];
|
||||
|
||||
native_id = get_this_hybrid_cpu_native_id();
|
||||
if (native_id == skt_native_id && pmu_type == hybrid_small)
|
||||
return &x86_pmu.hybrid_pmu[i];
|
||||
if (native_id == cmt_native_id && pmu_type == hybrid_tiny)
|
||||
return &x86_pmu.hybrid_pmu[i];
|
||||
}
|
||||
}
|
||||
|
||||
return NULL;
|
||||
@@ -5965,6 +5996,37 @@ static struct attribute *lnl_hybrid_events_attrs[] = {
|
||||
NULL
|
||||
};
|
||||
|
||||
/* The event string must be in PMU IDX order. */
|
||||
EVENT_ATTR_STR_HYBRID(topdown-retiring,
|
||||
td_retiring_arl_h,
|
||||
"event=0xc2,umask=0x02;event=0x00,umask=0x80;event=0xc2,umask=0x0",
|
||||
hybrid_big_small_tiny);
|
||||
EVENT_ATTR_STR_HYBRID(topdown-bad-spec,
|
||||
td_bad_spec_arl_h,
|
||||
"event=0x73,umask=0x0;event=0x00,umask=0x81;event=0x73,umask=0x0",
|
||||
hybrid_big_small_tiny);
|
||||
EVENT_ATTR_STR_HYBRID(topdown-fe-bound,
|
||||
td_fe_bound_arl_h,
|
||||
"event=0x9c,umask=0x01;event=0x00,umask=0x82;event=0x71,umask=0x0",
|
||||
hybrid_big_small_tiny);
|
||||
EVENT_ATTR_STR_HYBRID(topdown-be-bound,
|
||||
td_be_bound_arl_h,
|
||||
"event=0xa4,umask=0x02;event=0x00,umask=0x83;event=0x74,umask=0x0",
|
||||
hybrid_big_small_tiny);
|
||||
|
||||
static struct attribute *arl_h_hybrid_events_attrs[] = {
|
||||
EVENT_PTR(slots_adl),
|
||||
EVENT_PTR(td_retiring_arl_h),
|
||||
EVENT_PTR(td_bad_spec_arl_h),
|
||||
EVENT_PTR(td_fe_bound_arl_h),
|
||||
EVENT_PTR(td_be_bound_arl_h),
|
||||
EVENT_PTR(td_heavy_ops_adl),
|
||||
EVENT_PTR(td_br_mis_adl),
|
||||
EVENT_PTR(td_fetch_lat_adl),
|
||||
EVENT_PTR(td_mem_bound_adl),
|
||||
NULL,
|
||||
};
|
||||
|
||||
/* Must be in IDX order */
|
||||
EVENT_ATTR_STR_HYBRID(mem-loads, mem_ld_adl, "event=0xd0,umask=0x5,ldlat=3;event=0xcd,umask=0x1,ldlat=3", hybrid_big_small);
|
||||
EVENT_ATTR_STR_HYBRID(mem-stores, mem_st_adl, "event=0xd0,umask=0x6;event=0xcd,umask=0x2", hybrid_big_small);
|
||||
@@ -5983,6 +6045,21 @@ static struct attribute *mtl_hybrid_mem_attrs[] = {
|
||||
NULL
|
||||
};
|
||||
|
||||
EVENT_ATTR_STR_HYBRID(mem-loads,
|
||||
mem_ld_arl_h,
|
||||
"event=0xd0,umask=0x5,ldlat=3;event=0xcd,umask=0x1,ldlat=3;event=0xd0,umask=0x5,ldlat=3",
|
||||
hybrid_big_small_tiny);
|
||||
EVENT_ATTR_STR_HYBRID(mem-stores,
|
||||
mem_st_arl_h,
|
||||
"event=0xd0,umask=0x6;event=0xcd,umask=0x2;event=0xd0,umask=0x6",
|
||||
hybrid_big_small_tiny);
|
||||
|
||||
static struct attribute *arl_h_hybrid_mem_attrs[] = {
|
||||
EVENT_PTR(mem_ld_arl_h),
|
||||
EVENT_PTR(mem_st_arl_h),
|
||||
NULL,
|
||||
};
|
||||
|
||||
EVENT_ATTR_STR_HYBRID(tx-start, tx_start_adl, "event=0xc9,umask=0x1", hybrid_big);
|
||||
EVENT_ATTR_STR_HYBRID(tx-commit, tx_commit_adl, "event=0xc9,umask=0x2", hybrid_big);
|
||||
EVENT_ATTR_STR_HYBRID(tx-abort, tx_abort_adl, "event=0xc9,umask=0x4", hybrid_big);
|
||||
@@ -6006,8 +6083,8 @@ static struct attribute *adl_hybrid_tsx_attrs[] = {
|
||||
|
||||
FORMAT_ATTR_HYBRID(in_tx, hybrid_big);
|
||||
FORMAT_ATTR_HYBRID(in_tx_cp, hybrid_big);
|
||||
FORMAT_ATTR_HYBRID(offcore_rsp, hybrid_big_small);
|
||||
FORMAT_ATTR_HYBRID(ldlat, hybrid_big_small);
|
||||
FORMAT_ATTR_HYBRID(offcore_rsp, hybrid_big_small_tiny);
|
||||
FORMAT_ATTR_HYBRID(ldlat, hybrid_big_small_tiny);
|
||||
FORMAT_ATTR_HYBRID(frontend, hybrid_big);
|
||||
|
||||
#define ADL_HYBRID_RTM_FORMAT_ATTR \
|
||||
@@ -6030,7 +6107,7 @@ static struct attribute *adl_hybrid_extra_attr[] = {
|
||||
NULL
|
||||
};
|
||||
|
||||
FORMAT_ATTR_HYBRID(snoop_rsp, hybrid_small);
|
||||
FORMAT_ATTR_HYBRID(snoop_rsp, hybrid_small_tiny);
|
||||
|
||||
static struct attribute *mtl_hybrid_extra_attr_rtm[] = {
|
||||
ADL_HYBRID_RTM_FORMAT_ATTR,
|
||||
@@ -6238,8 +6315,9 @@ static inline int intel_pmu_v6_addr_offset(int index, bool eventsel)
|
||||
}
|
||||
|
||||
static const struct { enum hybrid_pmu_type id; char *name; } intel_hybrid_pmu_type_map[] __initconst = {
|
||||
{ hybrid_small, "cpu_atom" },
|
||||
{ hybrid_big, "cpu_core" },
|
||||
{ hybrid_small, "cpu_atom" },
|
||||
{ hybrid_big, "cpu_core" },
|
||||
{ hybrid_tiny, "cpu_lowpower" },
|
||||
};
|
||||
|
||||
static __always_inline int intel_pmu_init_hybrid(enum hybrid_pmu_type pmus)
|
||||
@@ -6272,7 +6350,7 @@ static __always_inline int intel_pmu_init_hybrid(enum hybrid_pmu_type pmus)
|
||||
0, x86_pmu_num_counters(&pmu->pmu), 0, 0);
|
||||
|
||||
pmu->intel_cap.capabilities = x86_pmu.intel_cap.capabilities;
|
||||
if (pmu->pmu_type & hybrid_small) {
|
||||
if (pmu->pmu_type & hybrid_small_tiny) {
|
||||
pmu->intel_cap.perf_metrics = 0;
|
||||
pmu->intel_cap.pebs_output_pt_available = 1;
|
||||
pmu->mid_ack = true;
|
||||
@@ -7111,6 +7189,37 @@ __init int intel_pmu_init(void)
|
||||
name = "lunarlake_hybrid";
|
||||
break;
|
||||
|
||||
case INTEL_ARROWLAKE_H:
|
||||
intel_pmu_init_hybrid(hybrid_big_small_tiny);
|
||||
|
||||
x86_pmu.pebs_latency_data = arl_h_latency_data;
|
||||
x86_pmu.get_event_constraints = arl_h_get_event_constraints;
|
||||
x86_pmu.hw_config = arl_h_hw_config;
|
||||
|
||||
td_attr = arl_h_hybrid_events_attrs;
|
||||
mem_attr = arl_h_hybrid_mem_attrs;
|
||||
tsx_attr = adl_hybrid_tsx_attrs;
|
||||
extra_attr = boot_cpu_has(X86_FEATURE_RTM) ?
|
||||
mtl_hybrid_extra_attr_rtm : mtl_hybrid_extra_attr;
|
||||
|
||||
/* Initialize big core specific PerfMon capabilities. */
|
||||
pmu = &x86_pmu.hybrid_pmu[X86_HYBRID_PMU_CORE_IDX];
|
||||
intel_pmu_init_lnc(&pmu->pmu);
|
||||
|
||||
/* Initialize Atom core specific PerfMon capabilities. */
|
||||
pmu = &x86_pmu.hybrid_pmu[X86_HYBRID_PMU_ATOM_IDX];
|
||||
intel_pmu_init_skt(&pmu->pmu);
|
||||
|
||||
/* Initialize Lower Power Atom specific PerfMon capabilities. */
|
||||
pmu = &x86_pmu.hybrid_pmu[X86_HYBRID_PMU_TINY_IDX];
|
||||
intel_pmu_init_grt(&pmu->pmu);
|
||||
pmu->extra_regs = intel_cmt_extra_regs;
|
||||
|
||||
intel_pmu_pebs_data_source_arl_h();
|
||||
pr_cont("ArrowLake-H Hybrid events, ");
|
||||
name = "arrowlake_h_hybrid";
|
||||
break;
|
||||
|
||||
default:
|
||||
switch (x86_pmu.version) {
|
||||
case 1:
|
||||
|
||||
@@ -177,6 +177,17 @@ void __init intel_pmu_pebs_data_source_mtl(void)
|
||||
__intel_pmu_pebs_data_source_cmt(data_source);
|
||||
}
|
||||
|
||||
void __init intel_pmu_pebs_data_source_arl_h(void)
|
||||
{
|
||||
u64 *data_source;
|
||||
|
||||
intel_pmu_pebs_data_source_lnl();
|
||||
|
||||
data_source = x86_pmu.hybrid_pmu[X86_HYBRID_PMU_TINY_IDX].pebs_data_source;
|
||||
memcpy(data_source, pebs_data_source, sizeof(pebs_data_source));
|
||||
__intel_pmu_pebs_data_source_cmt(data_source);
|
||||
}
|
||||
|
||||
void __init intel_pmu_pebs_data_source_cmt(void)
|
||||
{
|
||||
__intel_pmu_pebs_data_source_cmt(pebs_data_source);
|
||||
@@ -388,6 +399,16 @@ u64 lnl_latency_data(struct perf_event *event, u64 status)
|
||||
return lnc_latency_data(event, status);
|
||||
}
|
||||
|
||||
u64 arl_h_latency_data(struct perf_event *event, u64 status)
|
||||
{
|
||||
struct x86_hybrid_pmu *pmu = hybrid_pmu(event->pmu);
|
||||
|
||||
if (pmu->pmu_type == hybrid_tiny)
|
||||
return cmt_latency_data(event, status);
|
||||
|
||||
return lnl_latency_data(event, status);
|
||||
}
|
||||
|
||||
static u64 load_latency_data(struct perf_event *event, u64 status)
|
||||
{
|
||||
union intel_x86_pebs_dse dse;
|
||||
|
||||
@@ -418,6 +418,9 @@ static void pt_config_start(struct perf_event *event)
|
||||
struct pt *pt = this_cpu_ptr(&pt_ctx);
|
||||
u64 ctl = event->hw.aux_config;
|
||||
|
||||
if (READ_ONCE(event->hw.aux_paused))
|
||||
return;
|
||||
|
||||
ctl |= RTIT_CTL_TRACEEN;
|
||||
if (READ_ONCE(pt->vmx_on))
|
||||
perf_aux_output_flag(&pt->handle, PERF_AUX_FLAG_PARTIAL);
|
||||
@@ -534,7 +537,24 @@ static void pt_config(struct perf_event *event)
|
||||
reg |= (event->attr.config & PT_CONFIG_MASK);
|
||||
|
||||
event->hw.aux_config = reg;
|
||||
|
||||
/*
|
||||
* Allow resume before starting so as not to overwrite a value set by a
|
||||
* PMI.
|
||||
*/
|
||||
barrier();
|
||||
WRITE_ONCE(pt->resume_allowed, 1);
|
||||
/* Configuration is complete, it is now OK to handle an NMI */
|
||||
barrier();
|
||||
WRITE_ONCE(pt->handle_nmi, 1);
|
||||
barrier();
|
||||
pt_config_start(event);
|
||||
barrier();
|
||||
/*
|
||||
* Allow pause after starting so its pt_config_stop() doesn't race with
|
||||
* pt_config_start().
|
||||
*/
|
||||
WRITE_ONCE(pt->pause_allowed, 1);
|
||||
}
|
||||
|
||||
static void pt_config_stop(struct perf_event *event)
|
||||
@@ -828,11 +848,13 @@ static void pt_buffer_advance(struct pt_buffer *buf)
|
||||
buf->cur_idx++;
|
||||
|
||||
if (buf->cur_idx == buf->cur->last) {
|
||||
if (buf->cur == buf->last)
|
||||
if (buf->cur == buf->last) {
|
||||
buf->cur = buf->first;
|
||||
else
|
||||
buf->wrapped = true;
|
||||
} else {
|
||||
buf->cur = list_entry(buf->cur->list.next, struct topa,
|
||||
list);
|
||||
}
|
||||
buf->cur_idx = 0;
|
||||
}
|
||||
}
|
||||
@@ -846,8 +868,11 @@ static void pt_buffer_advance(struct pt_buffer *buf)
|
||||
static void pt_update_head(struct pt *pt)
|
||||
{
|
||||
struct pt_buffer *buf = perf_get_aux(&pt->handle);
|
||||
bool wrapped = buf->wrapped;
|
||||
u64 topa_idx, base, old;
|
||||
|
||||
buf->wrapped = false;
|
||||
|
||||
if (buf->single) {
|
||||
local_set(&buf->data_size, buf->output_off);
|
||||
return;
|
||||
@@ -865,7 +890,7 @@ static void pt_update_head(struct pt *pt)
|
||||
} else {
|
||||
old = (local64_xchg(&buf->head, base) &
|
||||
((buf->nr_pages << PAGE_SHIFT) - 1));
|
||||
if (base < old)
|
||||
if (base < old || (base == old && wrapped))
|
||||
base += buf->nr_pages << PAGE_SHIFT;
|
||||
|
||||
local_add(base - old, &buf->data_size);
|
||||
@@ -1511,6 +1536,7 @@ void intel_pt_interrupt(void)
|
||||
buf = perf_aux_output_begin(&pt->handle, event);
|
||||
if (!buf) {
|
||||
event->hw.state = PERF_HES_STOPPED;
|
||||
WRITE_ONCE(pt->resume_allowed, 0);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1519,6 +1545,7 @@ void intel_pt_interrupt(void)
|
||||
ret = pt_buffer_reset_markers(buf, &pt->handle);
|
||||
if (ret) {
|
||||
perf_aux_output_end(&pt->handle, 0);
|
||||
WRITE_ONCE(pt->resume_allowed, 0);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1573,6 +1600,26 @@ static void pt_event_start(struct perf_event *event, int mode)
|
||||
struct pt *pt = this_cpu_ptr(&pt_ctx);
|
||||
struct pt_buffer *buf;
|
||||
|
||||
if (mode & PERF_EF_RESUME) {
|
||||
if (READ_ONCE(pt->resume_allowed)) {
|
||||
u64 status;
|
||||
|
||||
/*
|
||||
* Only if the trace is not active and the error and
|
||||
* stopped bits are clear, is it safe to start, but a
|
||||
* PMI might have just cleared these, so resume_allowed
|
||||
* must be checked again also.
|
||||
*/
|
||||
rdmsrl(MSR_IA32_RTIT_STATUS, status);
|
||||
if (!(status & (RTIT_STATUS_TRIGGEREN |
|
||||
RTIT_STATUS_ERROR |
|
||||
RTIT_STATUS_STOPPED)) &&
|
||||
READ_ONCE(pt->resume_allowed))
|
||||
pt_config_start(event);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
buf = perf_aux_output_begin(&pt->handle, event);
|
||||
if (!buf)
|
||||
goto fail_stop;
|
||||
@@ -1583,7 +1630,6 @@ static void pt_event_start(struct perf_event *event, int mode)
|
||||
goto fail_end_stop;
|
||||
}
|
||||
|
||||
WRITE_ONCE(pt->handle_nmi, 1);
|
||||
hwc->state = 0;
|
||||
|
||||
pt_config_buffer(buf);
|
||||
@@ -1601,6 +1647,12 @@ static void pt_event_stop(struct perf_event *event, int mode)
|
||||
{
|
||||
struct pt *pt = this_cpu_ptr(&pt_ctx);
|
||||
|
||||
if (mode & PERF_EF_PAUSE) {
|
||||
if (READ_ONCE(pt->pause_allowed))
|
||||
pt_config_stop(event);
|
||||
return;
|
||||
}
|
||||
|
||||
/*
|
||||
* Protect against the PMI racing with disabling wrmsr,
|
||||
* see comment in intel_pt_interrupt().
|
||||
@@ -1608,6 +1660,15 @@ static void pt_event_stop(struct perf_event *event, int mode)
|
||||
WRITE_ONCE(pt->handle_nmi, 0);
|
||||
barrier();
|
||||
|
||||
/*
|
||||
* Prevent a resume from attempting to restart tracing, or a pause
|
||||
* during a subsequent start. Do this after clearing handle_nmi so that
|
||||
* pt_event_snapshot_aux() will not re-allow them.
|
||||
*/
|
||||
WRITE_ONCE(pt->pause_allowed, 0);
|
||||
WRITE_ONCE(pt->resume_allowed, 0);
|
||||
barrier();
|
||||
|
||||
pt_config_stop(event);
|
||||
|
||||
if (event->hw.state == PERF_HES_STOPPED)
|
||||
@@ -1657,6 +1718,10 @@ static long pt_event_snapshot_aux(struct perf_event *event,
|
||||
if (WARN_ON_ONCE(!buf->snapshot))
|
||||
return 0;
|
||||
|
||||
/* Prevent pause/resume from attempting to start/stop tracing */
|
||||
WRITE_ONCE(pt->pause_allowed, 0);
|
||||
WRITE_ONCE(pt->resume_allowed, 0);
|
||||
barrier();
|
||||
/*
|
||||
* There is no PT interrupt in this mode, so stop the trace and it will
|
||||
* remain stopped while the buffer is copied.
|
||||
@@ -1676,8 +1741,13 @@ static long pt_event_snapshot_aux(struct perf_event *event,
|
||||
* Here, handle_nmi tells us if the tracing was on.
|
||||
* If the tracing was on, restart it.
|
||||
*/
|
||||
if (READ_ONCE(pt->handle_nmi))
|
||||
if (READ_ONCE(pt->handle_nmi)) {
|
||||
WRITE_ONCE(pt->resume_allowed, 1);
|
||||
barrier();
|
||||
pt_config_start(event);
|
||||
barrier();
|
||||
WRITE_ONCE(pt->pause_allowed, 1);
|
||||
}
|
||||
|
||||
return ret;
|
||||
}
|
||||
@@ -1793,7 +1863,9 @@ static __init int pt_init(void)
|
||||
if (!intel_pt_validate_hw_cap(PT_CAP_topa_multiple_entries))
|
||||
pt_pmu.pmu.capabilities = PERF_PMU_CAP_AUX_NO_SG;
|
||||
|
||||
pt_pmu.pmu.capabilities |= PERF_PMU_CAP_EXCLUSIVE | PERF_PMU_CAP_ITRACE;
|
||||
pt_pmu.pmu.capabilities |= PERF_PMU_CAP_EXCLUSIVE |
|
||||
PERF_PMU_CAP_ITRACE |
|
||||
PERF_PMU_CAP_AUX_PAUSE;
|
||||
pt_pmu.pmu.attr_groups = pt_attr_groups;
|
||||
pt_pmu.pmu.task_ctx_nr = perf_sw_context;
|
||||
pt_pmu.pmu.event_init = pt_event_init;
|
||||
|
||||
@@ -65,6 +65,7 @@ struct pt_pmu {
|
||||
* @head: logical write offset inside the buffer
|
||||
* @snapshot: if this is for a snapshot/overwrite counter
|
||||
* @single: use Single Range Output instead of ToPA
|
||||
* @wrapped: buffer advance wrapped back to the first topa table
|
||||
* @stop_pos: STOP topa entry index
|
||||
* @intr_pos: INT topa entry index
|
||||
* @stop_te: STOP topa entry pointer
|
||||
@@ -82,6 +83,7 @@ struct pt_buffer {
|
||||
local64_t head;
|
||||
bool snapshot;
|
||||
bool single;
|
||||
bool wrapped;
|
||||
long stop_pos, intr_pos;
|
||||
struct topa_entry *stop_te, *intr_te;
|
||||
void **data_pages;
|
||||
@@ -117,6 +119,8 @@ struct pt_filters {
|
||||
* @filters: last configured filters
|
||||
* @handle_nmi: do handle PT PMI on this cpu, there's an active event
|
||||
* @vmx_on: 1 if VMX is ON on this cpu
|
||||
* @pause_allowed: PERF_EF_PAUSE is allowed to stop tracing
|
||||
* @resume_allowed: PERF_EF_RESUME is allowed to start tracing
|
||||
* @output_base: cached RTIT_OUTPUT_BASE MSR value
|
||||
* @output_mask: cached RTIT_OUTPUT_MASK MSR value
|
||||
*/
|
||||
@@ -125,6 +129,8 @@ struct pt {
|
||||
struct pt_filters filters;
|
||||
int handle_nmi;
|
||||
int vmx_on;
|
||||
int pause_allowed;
|
||||
int resume_allowed;
|
||||
u64 output_base;
|
||||
u64 output_mask;
|
||||
};
|
||||
|
||||
@@ -668,24 +668,38 @@ enum {
|
||||
#define PERF_PEBS_DATA_SOURCE_GRT_MAX 0x10
|
||||
#define PERF_PEBS_DATA_SOURCE_GRT_MASK (PERF_PEBS_DATA_SOURCE_GRT_MAX - 1)
|
||||
|
||||
/*
|
||||
* CPUID.1AH.EAX[31:0] uniquely identifies the microarchitecture
|
||||
* of the core. Bits 31-24 indicates its core type (Core or Atom)
|
||||
* and Bits [23:0] indicates the native model ID of the core.
|
||||
* Core type and native model ID are defined in below enumerations.
|
||||
*/
|
||||
enum hybrid_cpu_type {
|
||||
HYBRID_INTEL_NONE,
|
||||
HYBRID_INTEL_ATOM = 0x20,
|
||||
HYBRID_INTEL_CORE = 0x40,
|
||||
};
|
||||
|
||||
enum hybrid_pmu_type {
|
||||
not_hybrid,
|
||||
hybrid_small = BIT(0),
|
||||
hybrid_big = BIT(1),
|
||||
|
||||
hybrid_big_small = hybrid_big | hybrid_small, /* only used for matching */
|
||||
};
|
||||
|
||||
#define X86_HYBRID_PMU_ATOM_IDX 0
|
||||
#define X86_HYBRID_PMU_CORE_IDX 1
|
||||
#define X86_HYBRID_PMU_TINY_IDX 2
|
||||
|
||||
#define X86_HYBRID_NUM_PMUS 2
|
||||
enum hybrid_pmu_type {
|
||||
not_hybrid,
|
||||
hybrid_small = BIT(X86_HYBRID_PMU_ATOM_IDX),
|
||||
hybrid_big = BIT(X86_HYBRID_PMU_CORE_IDX),
|
||||
hybrid_tiny = BIT(X86_HYBRID_PMU_TINY_IDX),
|
||||
|
||||
/* The belows are only used for matching */
|
||||
hybrid_big_small = hybrid_big | hybrid_small,
|
||||
hybrid_small_tiny = hybrid_small | hybrid_tiny,
|
||||
hybrid_big_small_tiny = hybrid_big | hybrid_small_tiny,
|
||||
};
|
||||
|
||||
enum atom_native_id {
|
||||
cmt_native_id = 0x2, /* Crestmont */
|
||||
skt_native_id = 0x3, /* Skymont */
|
||||
};
|
||||
|
||||
struct x86_hybrid_pmu {
|
||||
struct pmu pmu;
|
||||
@@ -1578,6 +1592,8 @@ u64 cmt_latency_data(struct perf_event *event, u64 status);
|
||||
|
||||
u64 lnl_latency_data(struct perf_event *event, u64 status);
|
||||
|
||||
u64 arl_h_latency_data(struct perf_event *event, u64 status);
|
||||
|
||||
extern struct event_constraint intel_core2_pebs_event_constraints[];
|
||||
|
||||
extern struct event_constraint intel_atom_pebs_event_constraints[];
|
||||
@@ -1697,6 +1713,8 @@ void intel_pmu_pebs_data_source_grt(void);
|
||||
|
||||
void intel_pmu_pebs_data_source_mtl(void);
|
||||
|
||||
void intel_pmu_pebs_data_source_arl_h(void);
|
||||
|
||||
void intel_pmu_pebs_data_source_cmt(void);
|
||||
|
||||
void intel_pmu_pebs_data_source_lnl(void);
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user