summaryrefslogtreecommitdiff
path: root/tools/perf/builtin-sched.c
diff options
context:
space:
mode:
Diffstat (limited to 'tools/perf/builtin-sched.c')
-rw-r--r--tools/perf/builtin-sched.c987
1 files changed, 724 insertions, 263 deletions
diff --git a/tools/perf/builtin-sched.c b/tools/perf/builtin-sched.c
index 3f509cfdd58c..dd39a4fb6c7a 100644
--- a/tools/perf/builtin-sched.c
+++ b/tools/perf/builtin-sched.c
@@ -36,6 +36,7 @@
#include <linux/zalloc.h>
#include <sys/prctl.h>
#include <sys/resource.h>
+#include <sys/wait.h>
#include <inttypes.h>
#include <errno.h>
@@ -54,9 +55,72 @@
#define COMM_LEN 20
#define SYM_LEN 129
#define MAX_PID 1024000
+#define PID_MAX_LIMIT 4194304 /* kernel limit on 64-bit */
#define MAX_PRIO 140
#define SEP_LEN 100
+#define NUM_LAT_BUCKETS 22
+
+enum hist_mode {
+ HIST_MODE_LOG = 0,
+ HIST_MODE_LINEAR,
+};
+
+static const char *lat_bucket_names[NUM_LAT_BUCKETS] = {
+ "< 1 us",
+ "1 - 2 us",
+ "2 - 4 us",
+ "4 - 8 us",
+ "8 - 16 us",
+ "16 - 32 us",
+ "32 - 64 us",
+ "64 - 128 us",
+ "128 - 256 us",
+ "256 - 512 us",
+ "512 - 1024 us",
+ "1 - 2 ms",
+ "2 - 4 ms",
+ "4 - 8 ms",
+ "8 - 16 ms",
+ "16 - 32 ms",
+ "32 - 64 ms",
+ "64 - 128 ms",
+ "128 - 256 ms",
+ "256 - 512 ms",
+ "512 - 1024 ms",
+ ">= 1.05 s"
+};
+
+static const char *linear_bucket_names[NUM_LAT_BUCKETS] = {
+ "< 100 us",
+ "100 - 200 us",
+ "200 - 300 us",
+ "300 - 400 us",
+ "400 - 500 us",
+ "500 - 600 us",
+ "600 - 700 us",
+ "700 - 800 us",
+ "800 - 900 us",
+ "900 - 1000 us",
+ "1.0 - 1.1 ms",
+ "1.1 - 1.2 ms",
+ "1.2 - 1.3 ms",
+ "1.3 - 1.4 ms",
+ "1.4 - 1.5 ms",
+ "1.5 - 1.6 ms",
+ "1.6 - 1.7 ms",
+ "1.7 - 1.8 ms",
+ "1.8 - 1.9 ms",
+ "1.9 - 2.0 ms",
+ "2.0 - 2.1 ms",
+ ">= 2.1 ms"
+};
+
+struct perf_sched;
+static int latency_bucket(struct perf_sched *sched, u64 delta_ns);
+static void print_latency_histogram(struct perf_sched *sched, u64 *hist,
+ u64 total_count, const char *title);
+
static const char *cpu_list;
static struct perf_cpu_map *user_requested_cpus;
static DECLARE_BITMAP(cpu_bitmap, MAX_NR_CPUS);
@@ -122,6 +186,7 @@ struct work_atoms {
u64 nb_atoms;
u64 total_runtime;
int num_merged;
+ u64 hist[NUM_LAT_BUCKETS];
};
typedef int (*sort_fn_t)(struct work_atoms *, struct work_atoms *);
@@ -129,21 +194,20 @@ typedef int (*sort_fn_t)(struct work_atoms *, struct work_atoms *);
struct perf_sched;
struct trace_sched_handler {
- int (*switch_event)(struct perf_sched *sched, struct evsel *evsel,
- struct perf_sample *sample, struct machine *machine);
+ int (*switch_event)(struct perf_sched *sched, struct perf_sample *sample,
+ struct machine *machine);
- int (*runtime_event)(struct perf_sched *sched, struct evsel *evsel,
- struct perf_sample *sample, struct machine *machine);
+ int (*runtime_event)(struct perf_sched *sched, struct perf_sample *sample,
+ struct machine *machine);
- int (*wakeup_event)(struct perf_sched *sched, struct evsel *evsel,
- struct perf_sample *sample, struct machine *machine);
+ int (*wakeup_event)(struct perf_sched *sched, struct perf_sample *sample,
+ struct machine *machine);
/* PERF_RECORD_FORK event, not sched_process_fork tracepoint */
int (*fork_event)(struct perf_sched *sched, union perf_event *event,
struct machine *machine);
int (*migrate_task_event)(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine);
};
@@ -218,6 +282,10 @@ struct perf_sched {
struct list_head sort_list, cmp_pid;
bool force;
bool skip_merge;
+ bool show_histogram;
+ enum hist_mode hist_mode;
+ const char *hist_mode_str;
+ u64 global_hist[NUM_LAT_BUCKETS];
struct perf_sched_map map;
/* options for timehist command */
@@ -245,6 +313,70 @@ struct perf_sched {
struct perf_data *data;
};
+static int scnprintf_latency_unit(char *buf, size_t size, u64 nsecs)
+{
+ if (nsecs < 1000)
+ return scnprintf(buf, size, "%6" PRIu64 " ns", nsecs);
+ if (nsecs < NSEC_PER_MSEC)
+ return scnprintf(buf, size, "%6.3f us", (double)nsecs / NSEC_PER_USEC);
+ if (nsecs < NSEC_PER_SEC)
+ return scnprintf(buf, size, "%6.3f ms", (double)nsecs / NSEC_PER_MSEC);
+ return scnprintf(buf, size, "%6.3f s ", (double)nsecs / NSEC_PER_SEC);
+}
+
+static int latency_bucket(struct perf_sched *sched, u64 delta_ns)
+{
+ u64 delta_us = delta_ns / NSEC_PER_USEC;
+ u64 b;
+
+ if (sched->hist_mode == HIST_MODE_LINEAR) {
+ b = delta_us / 100;
+ } else {
+ if (delta_us == 0)
+ return 0;
+ b = 64 - __builtin_clzll(delta_us);
+ }
+
+ if (b >= NUM_LAT_BUCKETS - 1)
+ return NUM_LAT_BUCKETS - 1;
+ return b;
+}
+
+static void print_latency_histogram(struct perf_sched *sched, u64 *hist,
+ u64 total_count, const char *title)
+{
+ const char **bucket_names = (sched->hist_mode == HIST_MODE_LINEAR) ?
+ linear_bucket_names : lat_bucket_names;
+ int bar_total = 40;
+ char bar[] = "########################################";
+ int i;
+
+ if (total_count == 0)
+ return;
+
+ printf("\n %s (total samples: %" PRIu64 ")\n", title, total_count);
+ printf(" -------------------------------------------------------------------\n");
+ printf(" %-16s | %10s | %6s | %s\n",
+ "Latency Range", "Count", "Pct", "Histogram Graph");
+ printf(" -------------------------------------------------------------------\n");
+
+ for (i = 0; i < NUM_LAT_BUCKETS; i++) {
+ double pct;
+ int bar_len;
+
+ if (hist[i] == 0)
+ continue;
+ pct = (double)hist[i] * 100.0 / total_count;
+ bar_len = (hist[i] * bar_total) / total_count;
+ if (bar_len == 0 && hist[i] > 0)
+ bar_len = 1;
+ printf(" %-16s | %10" PRIu64 " | %5.1f%% | %.*s\n",
+ bucket_names[i], hist[i], pct,
+ bar_len, bar);
+ }
+ printf(" -------------------------------------------------------------------\n");
+}
+
/* per thread run time data */
struct thread_runtime {
u64 last_time; /* time of previous sched in/out event */
@@ -273,6 +405,7 @@ struct thread_runtime {
u64 migrations;
int prio;
+ bool color;
};
/* per event run time data */
@@ -364,14 +497,25 @@ get_new_event(struct task_desc *task, u64 timestamp)
struct sched_atom *event = zalloc(sizeof(*event));
unsigned long idx = task->nr_events;
size_t size;
+ struct sched_atom **atoms_p;
+
+ if (event == NULL) {
+ pr_err("ERROR: sched: failed to allocate event\n");
+ return NULL;
+ }
event->timestamp = timestamp;
event->nr = idx;
+ size = sizeof(struct sched_atom *) * (task->nr_events + 1);
+ atoms_p = realloc(task->atoms, size);
+ if (!atoms_p) {
+ pr_err("ERROR: sched: failed to grow atoms array\n");
+ free(event);
+ return NULL;
+ }
+ task->atoms = atoms_p;
task->nr_events++;
- size = sizeof(struct sched_atom *) * task->nr_events;
- task->atoms = realloc(task->atoms, size);
- BUG_ON(!task->atoms);
task->atoms[idx] = event;
@@ -402,6 +546,8 @@ static void add_sched_event_run(struct perf_sched *sched, struct task_desc *task
}
event = get_new_event(task, timestamp);
+ if (event == NULL)
+ return;
event->type = SCHED_EVENT_RUN;
event->duration = duration;
@@ -415,6 +561,8 @@ static void add_sched_event_wakeup(struct perf_sched *sched, struct task_desc *t
struct sched_atom *event, *wakee_event;
event = get_new_event(task, timestamp);
+ if (event == NULL)
+ return;
event->type = SCHED_EVENT_WAKEUP;
event->wakee = wakee;
@@ -429,6 +577,10 @@ static void add_sched_event_wakeup(struct perf_sched *sched, struct task_desc *t
}
wakee_event->wait_sem = zalloc(sizeof(*wakee_event->wait_sem));
+ if (!wakee_event->wait_sem) {
+ pr_err("ERROR: sched: failed to allocate semaphore\n");
+ return;
+ }
sem_init(wakee_event->wait_sem, 0, 0);
event->wait_sem = wakee_event->wait_sem;
@@ -440,6 +592,9 @@ static void add_sched_event_sleep(struct perf_sched *sched, struct task_desc *ta
{
struct sched_atom *event = get_new_event(task, timestamp);
+ if (event == NULL)
+ return;
+
event->type = SCHED_EVENT_SLEEP;
sched->nr_sleep_events++;
@@ -448,17 +603,28 @@ static void add_sched_event_sleep(struct perf_sched *sched, struct task_desc *ta
static struct task_desc *register_pid(struct perf_sched *sched,
unsigned long pid, const char *comm)
{
- struct task_desc *task;
+ struct task_desc *task, **tasks_p;
static int pid_max;
+ /* perf.data is untrusted — cap pid to prevent overflow in size calculations */
+ if (pid >= PID_MAX_LIMIT) {
+ pr_err("pid %lu exceeds limit %d, skipping\n", pid, PID_MAX_LIMIT);
+ return NULL;
+ }
+
if (sched->pid_to_task == NULL) {
if (sysctl__read_int("kernel/pid_max", &pid_max) < 0)
pid_max = MAX_PID;
- BUG_ON((sched->pid_to_task = calloc(pid_max, sizeof(struct task_desc *))) == NULL);
+ sched->pid_to_task = calloc(pid_max, sizeof(struct task_desc *));
+ if (sched->pid_to_task == NULL)
+ return NULL;
}
if (pid >= (unsigned long)pid_max) {
- BUG_ON((sched->pid_to_task = realloc(sched->pid_to_task, (pid + 1) *
- sizeof(struct task_desc *))) == NULL);
+ void *p = realloc(sched->pid_to_task, (pid + 1) * sizeof(struct task_desc *));
+
+ if (p == NULL)
+ return NULL;
+ sched->pid_to_task = p;
while (pid >= (unsigned long)pid_max)
sched->pid_to_task[pid_max++] = NULL;
}
@@ -469,9 +635,11 @@ static struct task_desc *register_pid(struct perf_sched *sched,
return task;
task = zalloc(sizeof(*task));
+ if (task == NULL)
+ return NULL;
task->pid = pid;
- task->nr = sched->nr_tasks;
- strcpy(task->comm, comm);
+ if (comm)
+ strlcpy(task->comm, comm, sizeof(task->comm));
/*
* every task starts in sleeping state - this gets ignored
* if there's no wakeup pointing to this sleep state:
@@ -479,10 +647,12 @@ static struct task_desc *register_pid(struct perf_sched *sched,
add_sched_event_sleep(sched, task, 0);
sched->pid_to_task[pid] = task;
- sched->nr_tasks++;
- sched->tasks = realloc(sched->tasks, sched->nr_tasks * sizeof(struct task_desc *));
- BUG_ON(!sched->tasks);
- sched->tasks[task->nr] = task;
+ tasks_p = realloc(sched->tasks, (sched->nr_tasks + 1) * sizeof(struct task_desc *));
+ if (!tasks_p)
+ return NULL;
+ sched->tasks = tasks_p;
+ sched->tasks[sched->nr_tasks] = task;
+ task->nr = sched->nr_tasks++;
if (verbose > 0)
printf("registered task #%ld, PID %ld (%s)\n", sched->nr_tasks, pid, comm);
@@ -826,42 +996,43 @@ static void test_calibrations(struct perf_sched *sched)
static int
replay_wakeup_event(struct perf_sched *sched,
- struct evsel *evsel, struct perf_sample *sample,
+ struct perf_sample *sample,
struct machine *machine __maybe_unused)
{
- const char *comm = evsel__strval(evsel, sample, "comm");
- const u32 pid = evsel__intval(evsel, sample, "pid");
+ const char *comm = perf_sample__strval(sample, "comm");
+ const u32 pid = perf_sample__intval(sample, "pid");
struct task_desc *waker, *wakee;
if (verbose > 0) {
- printf("sched_wakeup event %p\n", evsel);
+ printf("sched_wakeup event %p\n", sample->evsel);
printf(" ... pid %d woke up %s/%d\n", sample->tid, comm, pid);
}
waker = register_pid(sched, sample->tid, "<unknown>");
wakee = register_pid(sched, pid, comm);
+ if (waker == NULL || wakee == NULL)
+ return -1;
add_sched_event_wakeup(sched, waker, sample->time, wakee);
return 0;
}
static int replay_switch_event(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine __maybe_unused)
{
- const char *prev_comm = evsel__strval(evsel, sample, "prev_comm"),
- *next_comm = evsel__strval(evsel, sample, "next_comm");
- const u32 prev_pid = evsel__intval(evsel, sample, "prev_pid"),
- next_pid = evsel__intval(evsel, sample, "next_pid");
+ const char *prev_comm = perf_sample__strval(sample, "prev_comm"),
+ *next_comm = perf_sample__strval(sample, "next_comm");
+ const u32 prev_pid = perf_sample__intval(sample, "prev_pid"),
+ next_pid = perf_sample__intval(sample, "next_pid");
struct task_desc *prev, __maybe_unused *next;
u64 timestamp0, timestamp = sample->time;
int cpu = sample->cpu;
s64 delta;
if (verbose > 0)
- printf("sched_switch event %p\n", evsel);
+ printf("sched_switch event %p\n", sample->evsel);
if (cpu >= MAX_CPUS || cpu < 0)
return 0;
@@ -882,6 +1053,8 @@ static int replay_switch_event(struct perf_sched *sched,
prev = register_pid(sched, prev_pid, prev_comm);
next = register_pid(sched, next_pid, next_comm);
+ if (prev == NULL || next == NULL)
+ return -1;
sched->cpu_last_switched[cpu] = timestamp;
@@ -1055,20 +1228,33 @@ add_sched_out_event(struct work_atoms *atoms,
char run_state,
u64 timestamp)
{
- struct work_atom *atom = zalloc(sizeof(*atom));
+ struct work_atom *atom = NULL;
+
+ if (!list_empty(&atoms->work_list)) {
+ atom = list_entry(atoms->work_list.prev, struct work_atom, list);
+ if (atom->state != THREAD_SCHED_IN)
+ goto reuse;
+ }
+
+ atom = zalloc(sizeof(*atom));
if (!atom) {
pr_err("Non memory at %s", __func__);
return -1;
}
+ list_add_tail(&atom->list, &atoms->work_list);
+
+reuse:
atom->sched_out_time = timestamp;
if (run_state == 'R') {
atom->state = THREAD_WAIT_CPU;
atom->wake_up_time = atom->sched_out_time;
+ } else {
+ atom->state = THREAD_SLEEPING;
+ atom->wake_up_time = 0;
}
- list_add_tail(&atom->list, &atoms->work_list);
return 0;
}
@@ -1087,10 +1273,12 @@ add_runtime_event(struct work_atoms *atoms, u64 delta,
}
static void
-add_sched_in_event(struct work_atoms *atoms, u64 timestamp)
+add_sched_in_event(struct perf_sched *sched, struct work_atoms *atoms,
+ u64 timestamp)
{
struct work_atom *atom;
u64 delta;
+ int b;
if (list_empty(&atoms->work_list))
return;
@@ -1105,6 +1293,9 @@ add_sched_in_event(struct work_atoms *atoms, u64 timestamp)
return;
}
+ if (perf_time__skip_sample(&sched->ptime, timestamp))
+ return;
+
atom->state = THREAD_SCHED_IN;
atom->sched_in_time = timestamp;
@@ -1115,7 +1306,13 @@ add_sched_in_event(struct work_atoms *atoms, u64 timestamp)
atoms->max_lat_start = atom->wake_up_time;
atoms->max_lat_end = timestamp;
}
+
atoms->nb_atoms++;
+
+ b = latency_bucket(sched, delta);
+ atoms->hist[b]++;
+ if (thread__tid(atoms->thread) != 0)
+ sched->global_hist[b]++;
}
static void free_work_atoms(struct work_atoms *atoms)
@@ -1134,20 +1331,24 @@ static void free_work_atoms(struct work_atoms *atoms)
}
static int latency_switch_event(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
- const u32 prev_pid = evsel__intval(evsel, sample, "prev_pid"),
- next_pid = evsel__intval(evsel, sample, "next_pid");
- const char prev_state = evsel__taskstate(evsel, sample, "prev_state");
+ const u32 prev_pid = perf_sample__intval(sample, "prev_pid"),
+ next_pid = perf_sample__intval(sample, "next_pid");
+ const char prev_state = perf_sample__taskstate(sample, "prev_state");
struct work_atoms *out_events, *in_events;
struct thread *sched_out, *sched_in;
u64 timestamp0, timestamp = sample->time;
int cpu = sample->cpu, err = -1;
s64 delta;
- BUG_ON(cpu >= MAX_CPUS || cpu < 0);
+ /* perf.data is untrusted input — CPU may be absent or corrupted */
+ if (cpu >= MAX_CPUS || cpu < 0) {
+ pr_warning("WARNING: at offset %#" PRIx64 ": out-of-bound sample CPU %d, skipping sample\n",
+ sample->file_offset, cpu);
+ return 0;
+ }
timestamp0 = sched->cpu_last_switched[cpu];
sched->cpu_last_switched[cpu] = timestamp;
@@ -1177,7 +1378,7 @@ static int latency_switch_event(struct perf_sched *sched,
}
}
if (add_sched_out_event(out_events, prev_state, timestamp))
- return -1;
+ goto out_put;
in_events = thread_atoms_search(&sched->atom_root, sched_in, &sched->cmp_pid);
if (!in_events) {
@@ -1195,7 +1396,7 @@ static int latency_switch_event(struct perf_sched *sched,
if (add_sched_out_event(in_events, 'R', timestamp))
goto out_put;
}
- add_sched_in_event(in_events, timestamp);
+ add_sched_in_event(sched, in_events, timestamp);
err = 0;
out_put:
thread__put(sched_out);
@@ -1204,21 +1405,32 @@ out_put:
}
static int latency_runtime_event(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
- const u32 pid = evsel__intval(evsel, sample, "pid");
- const u64 runtime = evsel__intval(evsel, sample, "runtime");
- struct thread *thread = machine__findnew_thread(machine, -1, pid);
- struct work_atoms *atoms = thread_atoms_search(&sched->atom_root, thread, &sched->cmp_pid);
+ const u32 pid = perf_sample__intval(sample, "pid");
+ const u64 runtime = perf_sample__intval(sample, "runtime");
+ struct thread *thread;
+ struct work_atoms *atoms;
u64 timestamp = sample->time;
int cpu = sample->cpu, err = -1;
+ if (perf_time__skip_sample(&sched->ptime, timestamp))
+ return 0;
+
+ thread = machine__findnew_thread(machine, -1, pid);
if (thread == NULL)
return -1;
- BUG_ON(cpu >= MAX_CPUS || cpu < 0);
+ atoms = thread_atoms_search(&sched->atom_root, thread, &sched->cmp_pid);
+
+ /* perf.data is untrusted input — CPU may be absent or corrupted */
+ if (cpu >= MAX_CPUS || cpu < 0) {
+ pr_warning("WARNING: at offset %#" PRIx64 ": out-of-bound sample CPU %d, skipping sample\n",
+ sample->file_offset, cpu);
+ err = 0;
+ goto out_put;
+ }
if (!atoms) {
if (thread_atoms_insert(sched, thread))
goto out_put;
@@ -1239,11 +1451,10 @@ out_put:
}
static int latency_wakeup_event(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
- const u32 pid = evsel__intval(evsel, sample, "pid");
+ const u32 pid = perf_sample__intval(sample, "pid");
struct work_atoms *atoms;
struct work_atom *atom;
struct thread *wakee;
@@ -1300,11 +1511,10 @@ out_put:
}
static int latency_migrate_task_event(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
- const u32 pid = evsel__intval(evsel, sample, "pid");
+ const u32 pid = perf_sample__intval(sample, "pid");
u64 timestamp = sample->time;
struct work_atoms *atoms;
struct work_atom *atom;
@@ -1354,6 +1564,8 @@ static void output_lat_thread(struct perf_sched *sched, struct work_atoms *work_
int i;
int ret;
u64 avg;
+ char runtime_lat[32];
+ char avg_lat[32], max_lat[32];
char max_lat_start[32], max_lat_end[32];
if (!work_list->nb_atoms)
@@ -1361,17 +1573,17 @@ static void output_lat_thread(struct perf_sched *sched, struct work_atoms *work_
/*
* Ignore idle threads:
*/
- if (!strcmp(thread__comm_str(work_list->thread), "swapper"))
+ if (thread__tid(work_list->thread) == 0)
return;
sched->all_runtime += work_list->total_runtime;
sched->all_count += work_list->nb_atoms;
if (work_list->num_merged > 1) {
- ret = printf(" %s:(%d) ", thread__comm_str(work_list->thread),
+ ret = printf(" %s:(%d)", thread__comm_str(work_list->thread),
work_list->num_merged);
} else {
- ret = printf(" %s:%d ", thread__comm_str(work_list->thread),
+ ret = printf(" %s:%d", thread__comm_str(work_list->thread),
thread__tid(work_list->thread));
}
@@ -1379,14 +1591,21 @@ static void output_lat_thread(struct perf_sched *sched, struct work_atoms *work_
printf(" ");
avg = work_list->total_lat / work_list->nb_atoms;
+ scnprintf_latency_unit(runtime_lat, sizeof(runtime_lat), work_list->total_runtime);
+ scnprintf_latency_unit(avg_lat, sizeof(avg_lat), avg);
+ scnprintf_latency_unit(max_lat, sizeof(max_lat), work_list->max_lat);
timestamp__scnprintf_usec(work_list->max_lat_start, max_lat_start, sizeof(max_lat_start));
timestamp__scnprintf_usec(work_list->max_lat_end, max_lat_end, sizeof(max_lat_end));
- printf("|%11.3f ms |%9" PRIu64 " | avg:%8.3f ms | max:%8.3f ms | max start: %12s s | max end: %12s s\n",
- (double)work_list->total_runtime / NSEC_PER_MSEC,
- work_list->nb_atoms, (double)avg / NSEC_PER_MSEC,
- (double)work_list->max_lat / NSEC_PER_MSEC,
- max_lat_start, max_lat_end);
+ printf(" |%15s |%9" PRIu64 " |%16s |%16s |%20s s |%20s s |\n",
+ runtime_lat,
+ work_list->nb_atoms, avg_lat, max_lat,
+ max_lat_start, max_lat_end);
+
+ if (sched->show_histogram && verbose > 0)
+ print_latency_histogram(sched, work_list->hist,
+ work_list->nb_atoms,
+ "Task Latency Histogram");
}
static int pid_cmp(struct work_atoms *l, struct work_atoms *r)
@@ -1519,44 +1738,46 @@ again:
}
static int process_sched_wakeup_event(const struct perf_tool *tool,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
struct perf_sched *sched = container_of(tool, struct perf_sched, tool);
if (sched->tp_handler->wakeup_event)
- return sched->tp_handler->wakeup_event(sched, evsel, sample, machine);
+ return sched->tp_handler->wakeup_event(sched, sample, machine);
return 0;
}
-static int process_sched_wakeup_ignore(const struct perf_tool *tool __maybe_unused,
- struct evsel *evsel __maybe_unused,
- struct perf_sample *sample __maybe_unused,
- struct machine *machine __maybe_unused)
-{
- return 0;
-}
static bool thread__has_color(struct thread *thread)
{
- return thread__priv(thread) != NULL;
+ struct thread_runtime *tr = thread__priv(thread);
+
+ return tr != NULL && tr->color;
}
static struct thread*
map__findnew_thread(struct perf_sched *sched, struct machine *machine, pid_t pid, pid_t tid)
{
struct thread *thread = machine__findnew_thread(machine, pid, tid);
- bool color = false;
- if (!sched->map.color_pids || !thread || thread__priv(thread))
+ if (!sched->map.color_pids || !thread)
return thread;
- if (thread_map__has(sched->map.color_pids, tid))
- color = true;
+ /*
+ * Always check the color-pids map, even if thread__priv() is
+ * already set. COMM events processed before the first sched_switch
+ * allocate a thread_runtime via thread__get_runtime(), so priv is
+ * non-NULL before we ever get here. Skipping the check on non-NULL
+ * priv would prevent those threads from being colored.
+ */
+ if (thread_map__has(sched->map.color_pids, tid)) {
+ struct thread_runtime *tr = thread__get_runtime(thread);
- thread__set_priv(thread, color ? ((void*)1) : NULL);
+ if (tr)
+ tr->color = true;
+ }
return thread;
}
@@ -1626,11 +1847,11 @@ static void print_sched_map(struct perf_sched *sched, struct perf_cpu this_cpu,
}
}
-static int map_switch_event(struct perf_sched *sched, struct evsel *evsel,
- struct perf_sample *sample, struct machine *machine)
+static int map_switch_event(struct perf_sched *sched, struct perf_sample *sample,
+ struct machine *machine)
{
- const u32 next_pid = evsel__intval(evsel, sample, "next_pid");
- const u32 prev_pid = evsel__intval(evsel, sample, "prev_pid");
+ const u32 next_pid = perf_sample__intval(sample, "next_pid");
+ const u32 prev_pid = perf_sample__intval(sample, "prev_pid");
struct thread *sched_in, *sched_out;
struct thread_runtime *tr;
int new_shortname;
@@ -1647,7 +1868,12 @@ static int map_switch_event(struct perf_sched *sched, struct evsel *evsel,
const char *str;
int ret = -1;
- BUG_ON(this_cpu.cpu >= MAX_CPUS || this_cpu.cpu < 0);
+ /* perf.data is untrusted input — CPU may be absent or corrupted */
+ if (this_cpu.cpu >= MAX_CPUS || this_cpu.cpu < 0) {
+ pr_warning("WARNING: at offset %#" PRIx64 ": out-of-bound sample CPU %d, skipping sample\n",
+ sample->file_offset, this_cpu.cpu);
+ return 0;
+ }
if (this_cpu.cpu > sched->max_cpu.cpu)
sched->max_cpu = this_cpu;
@@ -1659,7 +1885,7 @@ static int map_switch_event(struct perf_sched *sched, struct evsel *evsel,
new_cpu = true;
}
} else
- cpus_nr = sched->max_cpu.cpu;
+ cpus_nr = sched->max_cpu.cpu + 1;
timestamp0 = sched->cpu_last_switched[this_cpu.cpu];
sched->cpu_last_switched[this_cpu.cpu] = timestamp;
@@ -1769,7 +1995,7 @@ static int map_switch_event(struct perf_sched *sched, struct evsel *evsel,
sched_out:
if (sched->map.task_name) {
tr = thread__get_runtime(sched->curr_out_thread[this_cpu.cpu]);
- if (strcmp(tr->shortname, "") == 0)
+ if (tr == NULL || strcmp(tr->shortname, "") == 0)
goto out;
if (proceed == 1)
@@ -1791,14 +2017,20 @@ out:
}
static int process_sched_switch_event(const struct perf_tool *tool,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
struct perf_sched *sched = container_of(tool, struct perf_sched, tool);
int this_cpu = sample->cpu, err = 0;
- u32 prev_pid = evsel__intval(evsel, sample, "prev_pid"),
- next_pid = evsel__intval(evsel, sample, "next_pid");
+ u32 prev_pid = perf_sample__intval(sample, "prev_pid"),
+ next_pid = perf_sample__intval(sample, "next_pid");
+
+ /* perf.data is untrusted input — CPU may be absent or corrupted */
+ if (this_cpu < 0 || this_cpu >= MAX_CPUS) {
+ pr_warning("WARNING: at offset %#" PRIx64 ": out-of-bound sample CPU %d, skipping sample\n",
+ sample->file_offset, this_cpu);
+ return 0;
+ }
if (sched->curr_pid[this_cpu] != (u32)-1) {
/*
@@ -1810,21 +2042,27 @@ static int process_sched_switch_event(const struct perf_tool *tool,
}
if (sched->tp_handler->switch_event)
- err = sched->tp_handler->switch_event(sched, evsel, sample, machine);
+ err = sched->tp_handler->switch_event(sched, sample, machine);
sched->curr_pid[this_cpu] = next_pid;
return err;
}
static int process_sched_runtime_event(const struct perf_tool *tool,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
struct perf_sched *sched = container_of(tool, struct perf_sched, tool);
+ /* perf.data is untrusted input — CPU may be absent or corrupted */
+ if (sample->cpu >= MAX_CPUS) {
+ pr_warning("WARNING: at offset %#" PRIx64 ": out-of-bound sample CPU %u, skipping sample\n",
+ sample->file_offset, sample->cpu);
+ return 0;
+ }
+
if (sched->tp_handler->runtime_event)
- return sched->tp_handler->runtime_event(sched, evsel, sample, machine);
+ return sched->tp_handler->runtime_event(sched, sample, machine);
return 0;
}
@@ -1847,34 +2085,64 @@ static int perf_sched__process_fork_event(const struct perf_tool *tool,
}
static int process_sched_migrate_task_event(const struct perf_tool *tool,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
struct perf_sched *sched = container_of(tool, struct perf_sched, tool);
if (sched->tp_handler->migrate_task_event)
- return sched->tp_handler->migrate_task_event(sched, evsel, sample, machine);
+ return sched->tp_handler->migrate_task_event(sched, sample, machine);
return 0;
}
typedef int (*tracepoint_handler)(const struct perf_tool *tool,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine);
+static struct evsel_str_handler latency_handlers[] = {
+ { "sched:sched_switch", process_sched_switch_event, },
+ { "sched:sched_stat_runtime", process_sched_runtime_event, },
+ { "sched:sched_wakeup", process_sched_wakeup_event, },
+ { "sched:sched_waking", process_sched_wakeup_event, },
+ { "sched:sched_wakeup_new", process_sched_wakeup_event, },
+ { "sched:sched_migrate_task", process_sched_migrate_task_event, },
+};
+
+static int process_sched_ignore(const struct perf_tool *tool __maybe_unused,
+ struct perf_sample *sample __maybe_unused,
+ struct machine *machine __maybe_unused)
+{
+ return 0;
+}
+
static int perf_sched__process_tracepoint_sample(const struct perf_tool *tool __maybe_unused,
union perf_event *event __maybe_unused,
struct perf_sample *sample,
- struct evsel *evsel,
struct machine *machine)
{
+ struct evsel *evsel = sample->evsel;
int err = 0;
- if (evsel->handler != NULL) {
+ if (evsel->handler == NULL) {
+ evsel->handler = process_sched_ignore;
+ for (size_t i = 0; i < ARRAY_SIZE(latency_handlers); i++) {
+ if (!evsel__name_is(evsel, latency_handlers[i].name))
+ continue;
+
+ if (!strcmp(latency_handlers[i].name, "sched:sched_wakeup") &&
+ sample->evsel->evlist &&
+ evlist__find_tracepoint_by_name(sample->evsel->evlist, "sched:sched_waking"))
+ break;
+
+ evsel->handler = latency_handlers[i].handler;
+ break;
+ }
+ }
+
+ if (evsel->handler != process_sched_ignore) {
tracepoint_handler f = evsel->handler;
- err = f(tool, evsel, sample, machine);
+ err = f(tool, sample, machine);
}
return err;
@@ -1913,21 +2181,13 @@ static int perf_sched__process_comm(const struct perf_tool *tool __maybe_unused,
static int perf_sched__read_events(struct perf_sched *sched)
{
- struct evsel_str_handler handlers[] = {
- { "sched:sched_switch", process_sched_switch_event, },
- { "sched:sched_stat_runtime", process_sched_runtime_event, },
- { "sched:sched_wakeup", process_sched_wakeup_event, },
- { "sched:sched_waking", process_sched_wakeup_event, },
- { "sched:sched_wakeup_new", process_sched_wakeup_event, },
- { "sched:sched_migrate_task", process_sched_migrate_task_event, },
- };
struct perf_session *session;
struct perf_data data = {
.path = input_name,
.mode = PERF_DATA_MODE_READ,
.force = sched->force,
};
- int rc = -1;
+ int rc = -1, err;
session = perf_session__new(&data, &sched->tool);
if (IS_ERR(session)) {
@@ -1937,25 +2197,34 @@ static int perf_sched__read_events(struct perf_sched *sched)
symbol__init(perf_session__env(session));
- /* prefer sched_waking if it is captured */
- if (evlist__find_tracepoint_by_name(session->evlist, "sched:sched_waking"))
- handlers[2].handler = process_sched_wakeup_ignore;
+ if (!perf_data__is_pipe(session->data)) {
+ /* prefer sched_waking if it is captured */
+ if (evlist__find_tracepoint_by_name(session->evlist, "sched:sched_waking"))
+ latency_handlers[2].handler = process_sched_ignore;
- if (perf_session__set_tracepoints_handlers(session, handlers))
+ if (perf_session__set_tracepoints_handlers(session, latency_handlers))
+ goto out_delete;
+ }
+
+ if (!perf_data__is_pipe(session->data) &&
+ !perf_session__has_traces(session, "record -R"))
goto out_delete;
- if (perf_session__has_traces(session, "record -R")) {
- int err = perf_session__process_events(session);
- if (err) {
- pr_err("Failed to process events, error %d", err);
- goto out_delete;
- }
+ err = perf_session__process_events(session);
+ if (err) {
+ pr_err("Failed to process events, error %d", err);
+ goto out_delete;
+ }
- sched->nr_events = session->evlist->stats.nr_events[0];
- sched->nr_lost_events = session->evlist->stats.total_lost;
- sched->nr_lost_chunks = session->evlist->stats.nr_events[PERF_RECORD_LOST];
+ if (perf_data__is_pipe(session->data) &&
+ !perf_session__has_traces(session, "record -R")) {
+ goto out_delete;
}
+ sched->nr_events = evlist__stats(session->evlist)->nr_events[0];
+ sched->nr_lost_events = evlist__stats(session->evlist)->total_lost;
+ sched->nr_lost_chunks = evlist__stats(session->evlist)->nr_events[PERF_RECORD_LOST];
+
rc = 0;
out_delete:
perf_session__delete(session);
@@ -2068,12 +2337,11 @@ static char *timehist_get_commstr(struct thread *thread)
/* prio field format: xxx or xxx->yyy */
#define MAX_PRIO_STR_LEN 8
-static char *timehist_get_priostr(struct evsel *evsel,
- struct thread *thread,
+static char *timehist_get_priostr(struct thread *thread,
struct perf_sample *sample)
{
static char prio_str[16];
- int prev_prio = (int)evsel__intval(evsel, sample, "prev_prio");
+ int prev_prio = (int)perf_sample__intval(sample, "prev_prio");
struct thread_runtime *tr = thread__priv(thread);
if (tr->prio != prev_prio && tr->prio != -1)
@@ -2161,21 +2429,21 @@ static void timehist_header(struct perf_sched *sched)
}
static void timehist_print_sample(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct addr_location *al,
struct thread *thread,
u64 t, const char state)
{
struct thread_runtime *tr = thread__priv(thread);
- const char *next_comm = evsel__strval(evsel, sample, "next_comm");
- const u32 next_pid = evsel__intval(evsel, sample, "next_pid");
+ const char *next_comm = perf_sample__strval(sample, "next_comm");
+ const u32 next_pid = perf_sample__intval(sample, "next_pid");
u32 max_cpus = sched->max_cpu.cpu + 1;
char tstr[64];
char nstr[30];
u64 wait_time;
- if (cpu_list && !test_bit(sample->cpu, cpu_bitmap))
+ if (cpu_list && (sample->cpu >= MAX_NR_CPUS ||
+ !test_bit(sample->cpu, cpu_bitmap)))
return;
timestamp__scnprintf_usec(t, tstr, sizeof(tstr));
@@ -2197,15 +2465,10 @@ static void timehist_print_sample(struct perf_sched *sched,
printf(" ");
}
- if (!thread__comm_set(thread)) {
- const char *prev_comm = evsel__strval(evsel, sample, "prev_comm");
- thread__set_comm(thread, prev_comm, sample->time);
- }
-
printf(" %-*s ", comm_width, timehist_get_commstr(thread));
if (sched->show_prio)
- printf(" %-*s ", MAX_PRIO_STR_LEN, timehist_get_priostr(evsel, thread, sample));
+ printf(" %-*s ", MAX_PRIO_STR_LEN, timehist_get_priostr(thread, sample));
wait_time = tr->dt_sleep + tr->dt_iowait + tr->dt_preempt;
print_sched_time(wait_time, 6);
@@ -2314,19 +2577,17 @@ static void timehist_update_runtime_stats(struct thread_runtime *r,
r->total_pre_mig_time += r->dt_pre_mig;
}
-static bool is_idle_sample(struct perf_sample *sample,
- struct evsel *evsel)
+static bool is_idle_sample(struct perf_sample *sample)
{
/* pid 0 == swapper == idle task */
- if (evsel__name_is(evsel, "sched:sched_switch"))
- return evsel__intval(evsel, sample, "prev_pid") == 0;
+ if (evsel__name_is(sample->evsel, "sched:sched_switch"))
+ return perf_sample__intval(sample, "prev_pid") == 0;
return sample->pid == 0;
}
static void save_task_callchain(struct perf_sched *sched,
struct perf_sample *sample,
- struct evsel *evsel,
struct machine *machine)
{
struct callchain_cursor *cursor;
@@ -2346,7 +2607,7 @@ static void save_task_callchain(struct perf_sched *sched,
cursor = get_tls_callchain_cursor();
- if (thread__resolve_callchain(thread, cursor, evsel, sample,
+ if (thread__resolve_callchain(thread, cursor, sample,
NULL, NULL, sched->max_stack + 2) != 0) {
if (verbose > 0)
pr_err("Failed to resolve callchain. Skipping\n");
@@ -2371,7 +2632,7 @@ static void save_task_callchain(struct perf_sched *sched,
if (!strcmp(sym->name, "schedule") ||
!strcmp(sym->name, "__schedule") ||
!strcmp(sym->name, "preempt_schedule"))
- sym->ignore = 1;
+ symbol__set_ignore(sym, true);
}
callchain_cursor_advance(cursor);
@@ -2405,7 +2666,7 @@ static int init_idle_threads(int ncpu)
{
int i, ret;
- idle_threads = zalloc(ncpu * sizeof(struct thread *));
+ idle_threads = calloc(ncpu, sizeof(struct thread *));
if (!idle_threads)
return -ENOMEM;
@@ -2439,10 +2700,13 @@ static void free_idle_threads(void)
struct idle_thread_runtime *itr;
itr = thread__priv(idle);
- if (itr)
+ if (itr) {
thread__put(itr->last_thread);
+ free_callchain(&itr->callchain);
+ callchain_cursor_cleanup(&itr->cursor);
+ }
- thread__delete(idle);
+ thread__put(idle);
}
}
@@ -2475,8 +2739,11 @@ static struct thread *get_idle_thread(int cpu)
idle_threads[cpu] = thread__new(0, 0);
if (idle_threads[cpu]) {
- if (init_idle_thread(idle_threads[cpu]) < 0)
+ if (init_idle_thread(idle_threads[cpu]) < 0) {
+ /* clean up so next call doesn't find a half-initialized thread */
+ thread__zput(idle_threads[cpu]);
return NULL;
+ }
}
}
@@ -2501,12 +2768,11 @@ static void save_idle_callchain(struct perf_sched *sched,
static struct thread *timehist_get_thread(struct perf_sched *sched,
struct perf_sample *sample,
- struct machine *machine,
- struct evsel *evsel)
+ struct machine *machine)
{
struct thread *thread;
- if (is_idle_sample(sample, evsel)) {
+ if (is_idle_sample(sample)) {
thread = get_idle_thread(sample->cpu);
if (thread == NULL)
pr_err("Failed to get idle thread for cpu %d.\n", sample->cpu);
@@ -2520,7 +2786,7 @@ static struct thread *timehist_get_thread(struct perf_sched *sched,
sample->tid);
}
- save_task_callchain(sched, sample, evsel, machine);
+ save_task_callchain(sched, sample, machine);
if (sched->idle_hist) {
struct thread *idle;
struct idle_thread_runtime *itr;
@@ -2528,19 +2794,25 @@ static struct thread *timehist_get_thread(struct perf_sched *sched,
idle = get_idle_thread(sample->cpu);
if (idle == NULL) {
pr_err("Failed to get idle thread for cpu %d.\n", sample->cpu);
+ thread__put(thread);
return NULL;
}
itr = thread__priv(idle);
- if (itr == NULL)
+ if (itr == NULL) {
+ thread__put(idle);
+ thread__put(thread);
return NULL;
+ }
thread__put(itr->last_thread);
itr->last_thread = thread__get(thread);
/* copy task callchain when entering to idle */
- if (evsel__intval(evsel, sample, "next_pid") == 0)
+ if (perf_sample__intval(sample, "next_pid") == 0)
save_idle_callchain(sched, itr, sample);
+
+ thread__put(idle);
}
}
@@ -2549,7 +2821,6 @@ static struct thread *timehist_get_thread(struct perf_sched *sched,
static bool timehist_skip_sample(struct perf_sched *sched,
struct thread *thread,
- struct evsel *evsel,
struct perf_sample *sample)
{
bool rc = false;
@@ -2571,20 +2842,22 @@ static bool timehist_skip_sample(struct perf_sched *sched,
tr = thread__get_runtime(thread);
if (tr && tr->prio != -1)
prio = tr->prio;
- else if (evsel__name_is(evsel, "sched:sched_switch"))
- prio = evsel__intval(evsel, sample, "prev_prio");
+ else if (evsel__name_is(sample->evsel, "sched:sched_switch"))
+ prio = perf_sample__intval(sample, "prev_prio");
- if (prio != -1 && !test_bit(prio, sched->prio_bitmap)) {
+ /* negative prio means no info; out-of-range prio can't match the filter */
+ if (prio >= 0 &&
+ (prio >= MAX_PRIO || !test_bit(prio, sched->prio_bitmap))) {
rc = true;
sched->skipped_samples++;
}
}
if (sched->idle_hist) {
- if (!evsel__name_is(evsel, "sched:sched_switch"))
+ if (!evsel__name_is(sample->evsel, "sched:sched_switch"))
rc = true;
- else if (evsel__intval(evsel, sample, "prev_pid") != 0 &&
- evsel__intval(evsel, sample, "next_pid") != 0)
+ else if (perf_sample__intval(sample, "prev_pid") != 0 &&
+ perf_sample__intval(sample, "next_pid") != 0)
rc = true;
}
@@ -2592,7 +2865,6 @@ static bool timehist_skip_sample(struct perf_sched *sched,
}
static void timehist_print_wakeup_event(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine,
struct thread *awakened)
@@ -2605,8 +2877,8 @@ static void timehist_print_wakeup_event(struct perf_sched *sched,
return;
/* show wakeup unless both awakee and awaker are filtered */
- if (timehist_skip_sample(sched, thread, evsel, sample) &&
- timehist_skip_sample(sched, awakened, evsel, sample)) {
+ if (timehist_skip_sample(sched, thread, sample) &&
+ timehist_skip_sample(sched, awakened, sample)) {
thread__put(thread);
return;
}
@@ -2630,7 +2902,6 @@ static void timehist_print_wakeup_event(struct perf_sched *sched,
static int timehist_sched_wakeup_ignore(const struct perf_tool *tool __maybe_unused,
union perf_event *event __maybe_unused,
- struct evsel *evsel __maybe_unused,
struct perf_sample *sample __maybe_unused,
struct machine *machine __maybe_unused)
{
@@ -2639,7 +2910,6 @@ static int timehist_sched_wakeup_ignore(const struct perf_tool *tool __maybe_unu
static int timehist_sched_wakeup_event(const struct perf_tool *tool,
union perf_event *event __maybe_unused,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
@@ -2647,7 +2917,7 @@ static int timehist_sched_wakeup_event(const struct perf_tool *tool,
struct thread *thread;
struct thread_runtime *tr = NULL;
/* want pid of awakened task not pid in sample */
- const u32 pid = evsel__intval(evsel, sample, "pid");
+ const u32 pid = perf_sample__intval(sample, "pid");
thread = machine__findnew_thread(machine, 0, pid);
if (thread == NULL)
@@ -2665,14 +2935,13 @@ static int timehist_sched_wakeup_event(const struct perf_tool *tool,
/* show wakeups if requested */
if (sched->show_wakeups &&
!perf_time__skip_sample(&sched->ptime, sample->time))
- timehist_print_wakeup_event(sched, evsel, sample, machine, thread);
+ timehist_print_wakeup_event(sched, sample, machine, thread);
thread__put(thread);
return 0;
}
static void timehist_print_migration_event(struct perf_sched *sched,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine,
struct thread *migrated)
@@ -2686,15 +2955,15 @@ static void timehist_print_migration_event(struct perf_sched *sched,
return;
max_cpus = sched->max_cpu.cpu + 1;
- ocpu = evsel__intval(evsel, sample, "orig_cpu");
- dcpu = evsel__intval(evsel, sample, "dest_cpu");
+ ocpu = perf_sample__intval(sample, "orig_cpu");
+ dcpu = perf_sample__intval(sample, "dest_cpu");
thread = machine__findnew_thread(machine, sample->pid, sample->tid);
if (thread == NULL)
return;
- if (timehist_skip_sample(sched, thread, evsel, sample) &&
- timehist_skip_sample(sched, migrated, evsel, sample)) {
+ if (timehist_skip_sample(sched, thread, sample) &&
+ timehist_skip_sample(sched, migrated, sample)) {
thread__put(thread);
return;
}
@@ -2728,7 +2997,6 @@ static void timehist_print_migration_event(struct perf_sched *sched,
static int timehist_migrate_task_event(const struct perf_tool *tool,
union perf_event *event __maybe_unused,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
@@ -2736,7 +3004,7 @@ static int timehist_migrate_task_event(const struct perf_tool *tool,
struct thread *thread;
struct thread_runtime *tr = NULL;
/* want pid of migrated task not pid in sample */
- const u32 pid = evsel__intval(evsel, sample, "pid");
+ const u32 pid = perf_sample__intval(sample, "pid");
thread = machine__findnew_thread(machine, 0, pid);
if (thread == NULL)
@@ -2753,22 +3021,20 @@ static int timehist_migrate_task_event(const struct perf_tool *tool,
/* show migrations if requested */
if (sched->show_migrations) {
- timehist_print_migration_event(sched, evsel, sample,
- machine, thread);
+ timehist_print_migration_event(sched, sample, machine, thread);
}
thread__put(thread);
return 0;
}
-static void timehist_update_task_prio(struct evsel *evsel,
- struct perf_sample *sample,
+static void timehist_update_task_prio(struct perf_sample *sample,
struct machine *machine)
{
struct thread *thread;
struct thread_runtime *tr = NULL;
- const u32 next_pid = evsel__intval(evsel, sample, "next_pid");
- const u32 next_prio = evsel__intval(evsel, sample, "next_prio");
+ const u32 next_pid = perf_sample__intval(sample, "next_pid");
+ const u32 next_prio = perf_sample__intval(sample, "next_prio");
if (next_pid == 0)
thread = get_idle_thread(sample->cpu);
@@ -2787,7 +3053,6 @@ static void timehist_update_task_prio(struct evsel *evsel,
static int timehist_sched_change_event(const struct perf_tool *tool,
union perf_event *event,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine)
{
@@ -2798,26 +3063,34 @@ static int timehist_sched_change_event(const struct perf_tool *tool,
struct thread_runtime *tr = NULL;
u64 tprev, t = sample->time;
int rc = 0;
- const char state = evsel__taskstate(evsel, sample, "prev_state");
+ const char state = perf_sample__taskstate(sample, "prev_state");
+
+ /* perf.data is untrusted input — CPU may be absent or corrupted */
+ if (sample->cpu >= MAX_CPUS) {
+ pr_warning("WARNING: at offset %#" PRIx64 ": out-of-bound sample CPU %d, skipping sample\n",
+ sample->file_offset, sample->cpu);
+ return 0;
+ }
addr_location__init(&al);
if (machine__resolve(machine, &al, sample) < 0) {
- pr_err("problem processing %d event. skipping it\n",
- event->header.type);
+ pr_err("problem processing %s (%u) event at offset %#" PRIx64 ", skipping it\n",
+ perf_event__name(event->header.type), event->header.type,
+ sample->file_offset);
rc = -1;
goto out;
}
if (sched->show_prio || sched->prio_str)
- timehist_update_task_prio(evsel, sample, machine);
+ timehist_update_task_prio(sample, machine);
- thread = timehist_get_thread(sched, sample, machine, evsel);
+ thread = timehist_get_thread(sched, sample, machine);
if (thread == NULL) {
rc = -1;
goto out;
}
- if (timehist_skip_sample(sched, thread, evsel, sample))
+ if (timehist_skip_sample(sched, thread, sample))
goto out;
tr = thread__get_runtime(thread);
@@ -2826,7 +3099,7 @@ static int timehist_sched_change_event(const struct perf_tool *tool,
goto out;
}
- tprev = evsel__get_time(evsel, sample->cpu);
+ tprev = evsel__get_time(sample->evsel, sample->cpu);
/*
* If start time given:
@@ -2856,8 +3129,15 @@ static int timehist_sched_change_event(const struct perf_tool *tool,
t = ptime->end;
}
- if (!sched->idle_hist || thread__tid(thread) == 0) {
- if (!cpu_list || test_bit(sample->cpu, cpu_bitmap))
+ /*
+ * Use is_idle_sample() not thread__tid() == 0: a crafted perf.data
+ * can set common_pid=0 with prev_pid!=0, giving us a machine thread
+ * whose priv is thread_runtime, not idle_thread_runtime — the cast
+ * below would read past the allocation.
+ */
+ if (!sched->idle_hist || is_idle_sample(sample)) {
+ if (!cpu_list || (sample->cpu < MAX_NR_CPUS &&
+ test_bit(sample->cpu, cpu_bitmap)))
timehist_update_runtime_stats(tr, t, tprev);
if (sched->idle_hist) {
@@ -2887,11 +3167,21 @@ static int timehist_sched_change_event(const struct perf_tool *tool,
if (itr->cursor.nr)
callchain_append(&itr->callchain, &itr->cursor, t - tprev);
- itr->last_thread = NULL;
+ thread__zput(itr->last_thread);
+ }
+
+ /*
+ * If the process name is not set for the thread, use "prev_comm"
+ * to set it. Otherwise the sched summary will have just pid information
+ */
+ if (!thread__comm_set(thread)) {
+ const char *prev_comm = perf_sample__strval(sample, "prev_comm");
+
+ thread__set_comm(thread, prev_comm, sample->time);
}
if (!sched->summary_only)
- timehist_print_sample(sched, evsel, sample, &al, thread, t, state);
+ timehist_print_sample(sched, sample, &al, thread, t, state);
}
out:
@@ -2916,7 +3206,7 @@ out:
tr->migrated = 0;
}
- evsel__save_time(evsel, sample->time, sample->cpu);
+ evsel__save_time(sample->evsel, sample->time, sample->cpu);
thread__put(thread);
addr_location__exit(&al);
@@ -2925,11 +3215,10 @@ out:
static int timehist_sched_switch_event(const struct perf_tool *tool,
union perf_event *event,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine __maybe_unused)
{
- return timehist_sched_change_event(tool, event, evsel, sample, machine);
+ return timehist_sched_change_event(tool, event, sample, machine);
}
static int process_lost(const struct perf_tool *tool __maybe_unused,
@@ -3035,7 +3324,7 @@ static size_t callchain__fprintf_folded(FILE *fp, struct callchain_node *node)
list_for_each_entry(chain, &node->val, list) {
if (chain->ip >= PERF_CONTEXT_MAX)
continue;
- if (chain->ms.sym && chain->ms.sym->ignore)
+ if (chain->ms.sym && symbol__ignore(chain->ms.sym))
continue;
ret += fprintf(fp, "%s%s", first ? "" : sep,
callchain_list__sym_name(chain, bf, sizeof(bf),
@@ -3051,7 +3340,8 @@ static size_t timehist_print_idlehist_callchain(struct rb_root_cached *root)
size_t ret = 0;
FILE *fp = stdout;
struct callchain_node *chain;
- struct rb_node *rb_node = rb_first_cached(root);
+ /* sort() uses rb_insert_color() on rb_root, not rb_root_cached */
+ struct rb_node *rb_node = rb_first(&root->rb_root);
printf(" %16s %8s %s\n", "Idle time (msec)", "Count", "Callchains");
printf(" %.16s %.8s %.50s\n", graph_dotted_line, graph_dotted_line,
@@ -3177,29 +3467,30 @@ static void timehist_print_summary(struct perf_sched *sched,
typedef int (*sched_handler)(const struct perf_tool *tool,
union perf_event *event,
- struct evsel *evsel,
struct perf_sample *sample,
struct machine *machine);
static int perf_timehist__process_sample(const struct perf_tool *tool,
union perf_event *event,
struct perf_sample *sample,
- struct evsel *evsel,
struct machine *machine)
{
struct perf_sched *sched = container_of(tool, struct perf_sched, tool);
+ struct evsel *evsel = sample->evsel;
int err = 0;
struct perf_cpu this_cpu = {
.cpu = sample->cpu,
};
- if (this_cpu.cpu > sched->max_cpu.cpu)
+ /* max_cpu indexes arrays allocated with MAX_CPUS entries */
+ if (this_cpu.cpu >= 0 && this_cpu.cpu < MAX_CPUS &&
+ this_cpu.cpu > sched->max_cpu.cpu)
sched->max_cpu = this_cpu;
if (evsel->handler != NULL) {
sched_handler f = evsel->handler;
- err = f(tool, event, evsel, sample, machine);
+ err = f(tool, event, sample, machine);
}
return err;
@@ -3211,7 +3502,7 @@ static int timehist_check_attr(struct perf_sched *sched,
struct evsel *evsel;
struct evsel_runtime *er;
- list_for_each_entry(evsel, &evlist->core.entries, core.node) {
+ list_for_each_entry(evsel, &evlist__core(evlist)->entries, core.node) {
er = evsel__get_runtime(evsel);
if (er == NULL) {
pr_err("Failed to allocate memory for evsel runtime data\n");
@@ -3299,6 +3590,7 @@ static int perf_sched__timehist(struct perf_sched *sched)
*/
sched->tool.sample = perf_timehist__process_sample;
sched->tool.mmap = perf_event__process_mmap;
+ sched->tool.mmap2 = perf_event__process_mmap2;
sched->tool.comm = perf_event__process_comm;
sched->tool.exit = perf_event__process_exit;
sched->tool.fork = perf_event__process_fork;
@@ -3362,8 +3654,8 @@ static int perf_sched__timehist(struct perf_sched *sched)
perf_session__set_tracepoints_handlers(session, migrate_handlers))
goto out;
- /* pre-allocate struct for per-CPU idle stats */
- sched->max_cpu.cpu = env->nr_cpus_online;
+ /* pre-allocate struct for per-CPU idle stats; cap to array bounds */
+ sched->max_cpu.cpu = min(env->nr_cpus_online, MAX_CPUS);
if (sched->max_cpu.cpu == 0)
sched->max_cpu.cpu = 4;
if (init_idle_threads(sched->max_cpu.cpu))
@@ -3382,9 +3674,9 @@ static int perf_sched__timehist(struct perf_sched *sched)
goto out;
}
- sched->nr_events = evlist->stats.nr_events[0];
- sched->nr_lost_events = evlist->stats.total_lost;
- sched->nr_lost_chunks = evlist->stats.nr_events[PERF_RECORD_LOST];
+ sched->nr_events = evlist__stats(evlist)->nr_events[0];
+ sched->nr_lost_events = evlist__stats(evlist)->total_lost;
+ sched->nr_lost_chunks = evlist__stats(evlist)->nr_events[PERF_RECORD_LOST];
if (sched->summary)
timehist_print_summary(sched, session);
@@ -3450,6 +3742,8 @@ static void __merge_work_atoms(struct rb_root_cached *root, struct work_atoms *d
this->max_lat_start = data->max_lat_start;
this->max_lat_end = data->max_lat_end;
}
+ for (int i = 0; i < NUM_LAT_BUCKETS; i++)
+ this->hist[i] += data->hist[i];
free_work_atoms(data);
return;
}
@@ -3483,7 +3777,7 @@ static int setup_cpus_switch_event(struct perf_sched *sched)
if (!sched->cpu_last_switched)
return -1;
- sched->curr_pid = malloc(MAX_CPUS * sizeof(*(sched->curr_pid)));
+ sched->curr_pid = calloc(MAX_CPUS, sizeof(*(sched->curr_pid)));
if (!sched->curr_pid) {
zfree(&sched->cpu_last_switched);
return -1;
@@ -3505,9 +3799,28 @@ static int perf_sched__lat(struct perf_sched *sched)
{
int rc = -1;
struct rb_node *next;
+ char total_runtime_str[32];
setup_pager();
+ if (sched->hist_mode_str) {
+ sched->show_histogram = true;
+ if (!strcmp(sched->hist_mode_str, "linear"))
+ sched->hist_mode = HIST_MODE_LINEAR;
+ else if (!strcmp(sched->hist_mode_str, "log"))
+ sched->hist_mode = HIST_MODE_LOG;
+ else {
+ pr_err("Invalid --hist-mode '%s', expected 'log' or 'linear'\n",
+ sched->hist_mode_str);
+ return -EINVAL;
+ }
+ }
+
+ if (sched->time_str && perf_time__parse_str(&sched->ptime, sched->time_str) != 0) {
+ pr_err("Invalid time string\n");
+ return -EINVAL;
+ }
+
if (setup_cpus_switch_event(sched))
return rc;
@@ -3517,9 +3830,24 @@ static int perf_sched__lat(struct perf_sched *sched)
perf_sched__merge_lat(sched);
perf_sched__sort_lat(sched);
- printf("\n -------------------------------------------------------------------------------------------------------------------------------------------\n");
- printf(" Task | Runtime ms | Count | Avg delay ms | Max delay ms | Max delay start | Max delay end |\n");
- printf(" -------------------------------------------------------------------------------------------------------------------------------------------\n");
+ next = rb_first_cached(&sched->sorted_atom_root);
+ while (next) {
+ struct work_atoms *work_list = rb_entry(next, struct work_atoms, node);
+
+ if (work_list->nb_atoms && thread__tid(work_list->thread) != 0)
+ break;
+ next = rb_next(next);
+ }
+
+ if (!next) {
+ pr_info("No matching trace samples found.\n");
+ rc = 0;
+ goto out_free_atoms;
+ }
+
+ printf("\n ------------------------------------------------------------------------------------------------------------------------------------------\n");
+ printf(" Task | Runtime | Count | Avg delay | Max delay | Max delay start | Max delay end |\n");
+ printf(" ------------------------------------------------------------------------------------------------------------------------------------------\n");
next = rb_first_cached(&sched->sorted_atom_root);
@@ -3531,17 +3859,23 @@ static int perf_sched__lat(struct perf_sched *sched)
next = rb_next(next);
}
- printf(" -----------------------------------------------------------------------------------------------------------------\n");
- printf(" TOTAL: |%11.3f ms |%9" PRIu64 " |\n",
- (double)sched->all_runtime / NSEC_PER_MSEC, sched->all_count);
+ printf(" ------------------------------------------------------------------------------------------------------------------------------------------\n");
+ scnprintf_latency_unit(total_runtime_str, sizeof(total_runtime_str), sched->all_runtime);
+ printf(" TOTAL: |%15s |%9" PRIu64 " |\n",
+ total_runtime_str, sched->all_count);
- printf(" ---------------------------------------------------\n");
+ printf(" ------------------------------------------------------\n");
print_bad_events(sched);
printf("\n");
+ if (sched->show_histogram)
+ print_latency_histogram(sched, sched->global_hist, sched->all_count,
+ "CPU Wait Latency Distribution Histogram (between snapshots)");
+
rc = 0;
+out_free_atoms:
while ((next = rb_first_cached(&sched->sorted_atom_root))) {
struct work_atoms *data;
@@ -3556,10 +3890,8 @@ out_free_cpus_switch_event:
static int setup_map_cpus(struct perf_sched *sched)
{
- sched->max_cpu.cpu = sysconf(_SC_NPROCESSORS_CONF);
-
if (sched->map.comp) {
- sched->map.comp_cpus = zalloc(sched->max_cpu.cpu * sizeof(int));
+ sched->map.comp_cpus = calloc(MAX_CPUS, sizeof(*sched->map.comp_cpus));
if (!sched->map.comp_cpus)
return -1;
}
@@ -3757,8 +4089,11 @@ static int process_synthesized_schedstat_event(const struct perf_tool *tool,
return 0;
}
+static volatile sig_atomic_t done;
+
static void sighandler(int sig __maybe_unused)
{
+ done = 1;
}
static int enable_sched_schedstats(int *reset)
@@ -3818,6 +4153,7 @@ static int perf_sched__schedstat_record(struct perf_sched *sched,
.mode = PERF_DATA_MODE_WRITE,
};
+ done = 0;
signal(SIGINT, sighandler);
signal(SIGCHLD, sighandler);
signal(SIGTERM, sighandler);
@@ -3829,7 +4165,7 @@ static int perf_sched__schedstat_record(struct perf_sched *sched,
session = perf_session__new(&data, &sched->tool);
if (IS_ERR(session)) {
pr_err("Perf session creation failed.\n");
- evlist__delete(evlist);
+ evlist__put(evlist);
return PTR_ERR(session);
}
@@ -3887,7 +4223,7 @@ static int perf_sched__schedstat_record(struct perf_sched *sched,
if (err < 0)
goto out;
- user_requested_cpus = evlist->core.user_requested_cpus;
+ user_requested_cpus = evlist__core(evlist)->user_requested_cpus;
err = perf_event__synthesize_schedstat(&(sched->tool),
process_synthesized_schedstat_event,
@@ -3902,8 +4238,11 @@ static int perf_sched__schedstat_record(struct perf_sched *sched,
if (argc)
evlist__start_workload(evlist);
- /* wait for signal */
- pause();
+ while (!done) {
+ if (argc && waitpid(evlist__workload_pid(evlist), NULL, WNOHANG) > 0)
+ break;
+ sleep(1);
+ }
if (reset) {
err = disable_sched_schedstat();
@@ -3925,8 +4264,8 @@ out:
else
fprintf(stderr, "[ perf sched stats: Failed !! ]\n");
- evlist__delete(evlist);
- close(fd);
+ perf_session__delete(session);
+ evlist__put(evlist);
return err;
}
@@ -3947,6 +4286,8 @@ static struct schedstat_domain *domain_second_pass;
static bool after_workload_flag;
static bool verbose_field;
+static void free_schedstat(struct list_head *head);
+
static void store_schedstat_cpu_diff(struct schedstat_cpu *after_workload)
{
struct perf_record_schedstat_cpu *before = cpu_second_pass->cpu_data;
@@ -4170,37 +4511,50 @@ static void summarize_schedstat_domain(struct schedstat_domain *summary_domain,
*/
static int get_all_cpu_stats(struct list_head *head)
{
- struct schedstat_cpu *cptr = list_first_entry(head, struct schedstat_cpu, cpu_list);
- struct schedstat_cpu *summary_head = NULL;
- struct perf_record_schedstat_domain *ds;
- struct perf_record_schedstat_cpu *cs;
+ struct schedstat_cpu *cptr, *summary_head = NULL;
struct schedstat_domain *dptr, *tdptr;
bool is_last = false;
int cnt = 1;
int ret = 0;
+ struct list_head tmp_cleanup_list;
- if (cptr) {
- summary_head = zalloc(sizeof(*summary_head));
- if (!summary_head)
- return -ENOMEM;
+ assert(!list_empty(head));
+ cptr = list_first_entry(head, struct schedstat_cpu, cpu_list);
- summary_head->cpu_data = zalloc(sizeof(*cs));
- memcpy(summary_head->cpu_data, cptr->cpu_data, sizeof(*cs));
+ INIT_LIST_HEAD(&tmp_cleanup_list);
- INIT_LIST_HEAD(&summary_head->domain_head);
+ summary_head = zalloc(sizeof(*summary_head));
+ if (!summary_head)
+ return -ENOMEM;
- list_for_each_entry(dptr, &cptr->domain_head, domain_list) {
- tdptr = zalloc(sizeof(*tdptr));
- if (!tdptr)
- return -ENOMEM;
+ INIT_LIST_HEAD(&summary_head->domain_head);
+ INIT_LIST_HEAD(&summary_head->cpu_list);
+ list_add(&summary_head->cpu_list, &tmp_cleanup_list);
- tdptr->domain_data = zalloc(sizeof(*ds));
- if (!tdptr->domain_data)
- return -ENOMEM;
+ summary_head->cpu_data = zalloc(sizeof(*summary_head->cpu_data));
+ if (!summary_head->cpu_data) {
+ ret = -ENOMEM;
+ goto out_cleanup;
+ }
+ memcpy(summary_head->cpu_data, cptr->cpu_data, sizeof(*summary_head->cpu_data));
- memcpy(tdptr->domain_data, dptr->domain_data, sizeof(*ds));
- list_add_tail(&tdptr->domain_list, &summary_head->domain_head);
+ list_for_each_entry(dptr, &cptr->domain_head, domain_list) {
+ tdptr = zalloc(sizeof(*tdptr));
+ if (!tdptr) {
+ ret = -ENOMEM;
+ goto out_cleanup;
}
+ INIT_LIST_HEAD(&tdptr->domain_list);
+
+ tdptr->domain_data = zalloc(sizeof(*tdptr->domain_data));
+ if (!tdptr->domain_data) {
+ free(tdptr);
+ ret = -ENOMEM;
+ goto out_cleanup;
+ }
+
+ memcpy(tdptr->domain_data, dptr->domain_data, sizeof(*tdptr->domain_data));
+ list_add_tail(&tdptr->domain_list, &summary_head->domain_head);
}
list_for_each_entry(cptr, head, cpu_list) {
@@ -4212,32 +4566,52 @@ static int get_all_cpu_stats(struct list_head *head)
cnt++;
summarize_schedstat_cpu(summary_head, cptr, cnt, is_last);
+ if (list_empty(&summary_head->domain_head))
+ continue;
+
tdptr = list_first_entry(&summary_head->domain_head, struct schedstat_domain,
domain_list);
list_for_each_entry(dptr, &cptr->domain_head, domain_list) {
summarize_schedstat_domain(tdptr, dptr, cnt, is_last);
+ if (list_is_last(&tdptr->domain_list, &summary_head->domain_head)) {
+ tdptr = NULL;
+ break;
+ }
tdptr = list_next_entry(tdptr, domain_list);
}
}
+ list_del_init(&summary_head->cpu_list);
list_add(&summary_head->cpu_list, head);
+ return 0;
+
+out_cleanup:
+ free_schedstat(&tmp_cleanup_list);
return ret;
}
-static int show_schedstat_data(struct list_head *head1, struct cpu_domain_map **cd_map1,
- struct list_head *head2, struct cpu_domain_map **cd_map2,
+static int show_schedstat_data(struct list_head *head1, struct cpu_domain_map **cd_map1, int nr1,
+ struct list_head *head2, struct cpu_domain_map **cd_map2, int nr2,
bool summary_only)
{
struct schedstat_cpu *cptr1 = list_first_entry(head1, struct schedstat_cpu, cpu_list);
struct perf_record_schedstat_domain *ds1 = NULL, *ds2 = NULL;
- struct perf_record_schedstat_cpu *cs1 = NULL, *cs2 = NULL;
struct schedstat_domain *dptr1 = NULL, *dptr2 = NULL;
struct schedstat_cpu *cptr2 = NULL;
__u64 jiffies1 = 0, jiffies2 = 0;
bool is_summary = true;
int ret = 0;
+ if (!cd_map1) {
+ pr_err("Error: CPU domain map 1 is missing.\n");
+ return -1;
+ }
+ if (head2 && !cd_map2) {
+ pr_err("Error: CPU domain map 2 is missing.\n");
+ return -1;
+ }
+
printf("Description\n");
print_separator2(SEP_LEN, "", 0);
printf("%-30s-> %s\n", "DESC", "Description of the field");
@@ -4260,21 +4634,47 @@ static int show_schedstat_data(struct list_head *head1, struct cpu_domain_map **
printf("\n");
ret = get_all_cpu_stats(head1);
+ if (ret)
+ return ret;
if (cptr2) {
ret = get_all_cpu_stats(head2);
+ if (ret)
+ return ret;
cptr2 = list_first_entry(head2, struct schedstat_cpu, cpu_list);
}
list_for_each_entry(cptr1, head1, cpu_list) {
struct cpu_domain_map *cd_info1 = NULL, *cd_info2 = NULL;
+ struct perf_record_schedstat_cpu *cs1 = cptr1->cpu_data;
+ struct perf_record_schedstat_cpu *cs2 = NULL;
- cs1 = cptr1->cpu_data;
+ dptr2 = NULL;
+ if (cs1->cpu >= (u32)nr1) {
+ pr_err("Error: CPU %d exceeds domain map size %d\n", cs1->cpu, nr1);
+ return -1;
+ }
cd_info1 = cd_map1[cs1->cpu];
+ if (!cd_info1) {
+ pr_err("Error: CPU %d domain info is missing in map 1.\n",
+ cs1->cpu);
+ return -1;
+ }
if (cptr2) {
cs2 = cptr2->cpu_data;
+ if (cs2->cpu >= (u32)nr2) {
+ pr_err("Error: CPU %d exceeds domain map size %d\n", cs2->cpu, nr2);
+ return -1;
+ }
cd_info2 = cd_map2[cs2->cpu];
- dptr2 = list_first_entry(&cptr2->domain_head, struct schedstat_domain,
- domain_list);
+ if (!cd_info2) {
+ pr_err("Error: CPU %d domain info is missing in map 2.\n",
+ cs2->cpu);
+ return -1;
+ }
+ if (!list_empty(&cptr2->domain_head))
+ dptr2 = list_first_entry(&cptr2->domain_head,
+ struct schedstat_domain,
+ domain_list);
}
if (cs2 && cs1->cpu != cs2->cpu) {
@@ -4302,10 +4702,31 @@ static int show_schedstat_data(struct list_head *head1, struct cpu_domain_map **
struct domain_info *dinfo1 = NULL, *dinfo2 = NULL;
ds1 = dptr1->domain_data;
+ ds2 = NULL;
+ if (ds1->domain >= cd_info1->nr_domains) {
+ pr_err("Error: Domain %d exceeds max domains %d for CPU %d in map 1.\n",
+ ds1->domain, cd_info1->nr_domains, cs1->cpu);
+ return -1;
+ }
dinfo1 = cd_info1->domains[ds1->domain];
+ if (!dinfo1) {
+ pr_err("Error: Domain %d info is missing for CPU %d in map 1.\n",
+ ds1->domain, cs1->cpu);
+ return -1;
+ }
if (dptr2) {
ds2 = dptr2->domain_data;
+ if (ds2->domain >= cd_info2->nr_domains) {
+ pr_err("Error: Domain %d exceeds max domains %d for CPU %d in map 2.\n",
+ ds2->domain, cd_info2->nr_domains, cs2->cpu);
+ return -1;
+ }
dinfo2 = cd_info2->domains[ds2->domain];
+ if (!dinfo2) {
+ pr_err("Error: Domain %d info is missing for CPU %d in map 2.\n",
+ ds2->domain, cs2->cpu);
+ return -1;
+ }
}
if (dinfo2 && dinfo1->domain != dinfo2->domain) {
@@ -4334,14 +4755,22 @@ static int show_schedstat_data(struct list_head *head1, struct cpu_domain_map **
print_domain_stats(ds1, ds2, jiffies1, jiffies2);
print_separator2(SEP_LEN, "", 0);
- if (dptr2)
- dptr2 = list_next_entry(dptr2, domain_list);
+ if (dptr2) {
+ if (list_is_last(&dptr2->domain_list, &cptr2->domain_head))
+ dptr2 = NULL;
+ else
+ dptr2 = list_next_entry(dptr2, domain_list);
+ }
}
if (summary_only)
break;
- if (cptr2)
- cptr2 = list_next_entry(cptr2, cpu_list);
+ if (cptr2) {
+ if (list_is_last(&cptr2->cpu_list, head2))
+ cptr2 = NULL;
+ else
+ cptr2 = list_next_entry(cptr2, cpu_list);
+ }
is_summary = false;
}
@@ -4439,6 +4868,8 @@ static int perf_sched__process_schedstat(const struct perf_tool *tool __maybe_un
domain_second_pass = list_first_entry(&cpu_second_pass->domain_head,
struct schedstat_domain, domain_list);
store_schedstat_cpu_diff(temp);
+ free(temp->cpu_data);
+ free(temp);
}
} else if (event->header.type == PERF_RECORD_SCHEDSTAT_DOMAIN) {
struct schedstat_cpu *cpu_tail;
@@ -4459,6 +4890,8 @@ static int perf_sched__process_schedstat(const struct perf_tool *tool __maybe_un
} else {
store_schedstat_domain_diff(temp);
domain_second_pass = list_next_entry(domain_second_pass, domain_list);
+ free(temp->domain_data);
+ free(temp);
}
}
@@ -4473,9 +4906,11 @@ static void free_schedstat(struct list_head *head)
list_for_each_entry_safe(cptr, n2, head, cpu_list) {
list_for_each_entry_safe(dptr, n1, &cptr->domain_head, domain_list) {
list_del_init(&dptr->domain_list);
+ free(dptr->domain_data);
free(dptr);
}
list_del_init(&cptr->cpu_list);
+ free(cptr->cpu_data);
free(cptr);
}
}
@@ -4509,7 +4944,7 @@ static int perf_sched__schedstat_report(struct perf_sched *sched)
if (err < 0)
goto out;
- user_requested_cpus = session->evlist->core.user_requested_cpus;
+ user_requested_cpus = evlist__core(session->evlist)->user_requested_cpus;
err = perf_session__process_events(session);
@@ -4523,7 +4958,9 @@ static int perf_sched__schedstat_report(struct perf_sched *sched)
}
cd_map = session->header.env.cpu_domain;
- err = show_schedstat_data(&cpu_head, cd_map, NULL, NULL, false);
+ err = show_schedstat_data(&cpu_head, cd_map,
+ session->header.env.nr_cpus_avail,
+ NULL, NULL, 0, false);
}
out:
@@ -4538,7 +4975,7 @@ static int perf_sched__schedstat_diff(struct perf_sched *sched,
struct cpu_domain_map **cd_map0 = NULL, **cd_map1 = NULL;
struct list_head cpu_head_ses0, cpu_head_ses1;
struct perf_session *session[2];
- struct perf_data data[2];
+ struct perf_data data[2] = {0};
int ret = 0, err = 0;
static const char *defaults[] = {
"perf.data.old",
@@ -4573,8 +5010,10 @@ static int perf_sched__schedstat_diff(struct perf_sched *sched,
}
err = perf_session__process_events(session[0]);
- if (err)
+ if (err) {
+ free_schedstat(&cpu_head);
goto out_delete_ses0;
+ }
cd_map0 = session[0]->header.env.cpu_domain;
list_replace_init(&cpu_head, &cpu_head_ses0);
@@ -4590,8 +5029,10 @@ static int perf_sched__schedstat_diff(struct perf_sched *sched,
}
err = perf_session__process_events(session[1]);
- if (err)
+ if (err) {
+ free_schedstat(&cpu_head);
goto out_delete_ses1;
+ }
cd_map1 = session[1]->header.env.cpu_domain;
list_replace_init(&cpu_head, &cpu_head_ses1);
@@ -4607,10 +5048,13 @@ static int perf_sched__schedstat_diff(struct perf_sched *sched,
if (list_empty(&cpu_head_ses0)) {
pr_err("Data is not available\n");
ret = -1;
- goto out_delete_ses0;
+ goto out_delete_ses1;
}
- show_schedstat_data(&cpu_head_ses0, cd_map0, &cpu_head_ses1, cd_map1, true);
+ ret = show_schedstat_data(&cpu_head_ses0, cd_map0, session[0]->header.env.nr_cpus_avail,
+ &cpu_head_ses1, cd_map1, session[1]->header.env.nr_cpus_avail, true);
+ if (ret)
+ goto out_delete_ses1;
out_delete_ses1:
free_schedstat(&cpu_head_ses1);
@@ -4645,6 +5089,7 @@ static int perf_sched__schedstat_live(struct perf_sched *sched,
int reset = 0;
int err = 0;
+ done = 0;
signal(SIGINT, sighandler);
signal(SIGCHLD, sighandler);
signal(SIGTERM, sighandler);
@@ -4675,7 +5120,7 @@ static int perf_sched__schedstat_live(struct perf_sched *sched,
if (err < 0)
goto out;
- user_requested_cpus = evlist->core.user_requested_cpus;
+ user_requested_cpus = evlist__core(evlist)->user_requested_cpus;
err = perf_event__synthesize_schedstat(&(sched->tool),
process_synthesized_event_live,
@@ -4690,8 +5135,11 @@ static int perf_sched__schedstat_live(struct perf_sched *sched,
if (argc)
evlist__start_workload(evlist);
- /* wait for signal */
- pause();
+ while (!done) {
+ if (argc && waitpid(evlist__workload_pid(evlist), NULL, WNOHANG) > 0)
+ break;
+ sleep(1);
+ }
if (reset) {
err = disable_sched_schedstat();
@@ -4720,11 +5168,11 @@ static int perf_sched__schedstat_live(struct perf_sched *sched,
goto out;
}
- show_schedstat_data(&cpu_head, cd_map, NULL, NULL, false);
+ err = show_schedstat_data(&cpu_head, cd_map, nr, NULL, NULL, 0, false);
free_cpu_domain_info(cd_map, sv, nr);
out:
free_schedstat(&cpu_head);
- evlist__delete(evlist);
+ evlist__put(evlist);
return err;
}
@@ -4848,6 +5296,12 @@ int cmd_sched(int argc, const char **argv)
"CPU to profile on"),
OPT_BOOLEAN('p', "pids", &sched.skip_merge,
"latency stats per pid instead of per comm"),
+ OPT_BOOLEAN('H', "histogram", &sched.show_histogram,
+ "show CPU wait latency distribution histogram"),
+ OPT_STRING(0, "hist-mode", &sched.hist_mode_str, "log|linear",
+ "latency bucket mode (log or linear, default: log)"),
+ OPT_STRING(0, "time", &sched.time_str, "str",
+ "Time span for analysis (start,stop)"),
OPT_PARENT(sched_options)
};
const struct option replay_options[] = {
@@ -4879,8 +5333,8 @@ int cmd_sched(int argc, const char **argv)
"Display call chains if present (default on)"),
OPT_UINTEGER(0, "max-stack", &sched.max_stack,
"Maximum number of functions to display backtrace."),
- OPT_STRING(0, "symfs", &symbol_conf.symfs, "directory",
- "Look for files with symbols relative to this directory"),
+ OPT_CALLBACK(0, "symfs", NULL, "directory[,layout]", SYMFS_HELP,
+ symbol__config_symfs),
OPT_BOOLEAN('s', "summary", &sched.summary_only,
"Show only syscall summary with statistics"),
OPT_BOOLEAN('S', "with-summary", &sched.summary,
@@ -4955,6 +5409,7 @@ int cmd_sched(int argc, const char **argv)
.switch_event = replay_switch_event,
.fork_event = replay_fork_event,
};
+ struct trace_sched_handler stats_ops = {};
int ret;
perf_tool__init(&sched.tool, /*ordered_events=*/true);
@@ -4963,6 +5418,10 @@ int cmd_sched(int argc, const char **argv)
sched.tool.namespaces = perf_event__process_namespaces;
sched.tool.lost = perf_event__process_lost;
sched.tool.fork = perf_sched__process_fork_event;
+ sched.tool.attr = perf_event__process_attr;
+ sched.tool.tracing_data = perf_event__process_tracing_data;
+ sched.tool.build_id = perf_event__process_build_id;
+ sched.tool.feature = perf_event__process_feature;
argc = parse_options_subcommand(argc, argv, sched_options, sched_subcommands,
sched_usage, PARSE_OPT_STOP_AT_NON_OPTION);
@@ -5037,6 +5496,7 @@ int cmd_sched(int argc, const char **argv)
} else if (!strcmp(argv[0], "stats")) {
const char *const stats_subcommands[] = {"record", "report", NULL};
+ sched.tp_handler = &stats_ops;
argc = parse_options_subcommand(argc, argv, stats_options,
stats_subcommands,
stats_usage,
@@ -5046,19 +5506,20 @@ int cmd_sched(int argc, const char **argv)
if (argc)
argc = parse_options(argc, argv, stats_options,
stats_usage, 0);
- return perf_sched__schedstat_record(&sched, argc, argv);
+ ret = perf_sched__schedstat_record(&sched, argc, argv);
} else if (argv[0] && !strcmp(argv[0], "report")) {
if (argc)
argc = parse_options(argc, argv, stats_options,
stats_usage, 0);
- return perf_sched__schedstat_report(&sched);
+ ret = perf_sched__schedstat_report(&sched);
} else if (argv[0] && !strcmp(argv[0], "diff")) {
if (argc)
argc = parse_options(argc, argv, stats_options,
stats_usage, 0);
- return perf_sched__schedstat_diff(&sched, argc, argv);
+ ret = perf_sched__schedstat_diff(&sched, argc, argv);
+ } else {
+ ret = perf_sched__schedstat_live(&sched, argc, argv);
}
- return perf_sched__schedstat_live(&sched, argc, argv);
} else {
usage_with_options(sched_usage, sched_options);
}