Re: [PATCH v7 8/8] perf tool: add cgroup identifier entry in perf report
From: Arnaldo Carvalho de Melo
Date: Wed Mar 01 2017 - 16:17:52 EST
Em Tue, Feb 21, 2017 at 07:33:13PM +0530, Hari Bathini escreveu:
> This patch introduces a cgroup identifier entry field in perf report to
> identify or distinguish data of different cgroups. It uses the device
> number and inode number of cgroup namespace, included in perf data with
> the new PERF_RECORD_NAMESPACES event, as cgroup identifier. With the
> assumption that each container is created with it's own cgroup namespace,
> this allows assessment/analysis of multiple containers at once.
Could you try to do this with some real world example? I.e. telling that
systemd creates some cgroups and then how to map the perf report output
you show below with what systemd puts in place.
Doing it with docker would be handy as well, no?
You also forgot to update the documentation with this new sort key,
please add it in the --sort part of
tools/perf/Documentation/perf-report.txt.
Ah, and just another minor request: please format the changeset
summaries as:
perf tools: Add cgroup identifier sort order keyword
First the component, that in this case is the generic one, as it affects
multiple tools (top, report), and then start with a capital letter after
the colon.
A git log --oneline tools/perf/ -20
will show you the pattern used.
Thanks!
- Arnaldo
> Shown below is the output of perf report, sorted based on cgroup id, on
> a system that was running three containers at the time of perf record
> and clearly showing one of the containers' considerable use of kernel
> memory in comparison with others:
>
>
> $ perf report -s cgroup_id,sample --stdio
> #
> # Total Lost Samples: 0
> #
> # Samples: 16K of event 'kmem:kmalloc'
> # Event count (approx.): 16043
> #
> # Overhead cgroup id (dev/inode) Samples
> # ........ ..................... ............
> #
> 96.33% 3/0xf00000d0 15454
> 3.02% 3/0xeffffffb 485
> 0.31% 3/0xf00000ce 49
> 0.29% 3/0xf00000cf 47
> 0.05% 0/0x0 8
>
> While this is a start, there is further scope of improving this. For
> example, instead of cgroup namespace's device and inode numbers, dev
> and inode numbers of some or all namespaces may be used to distinguish
> which processes are running in a given container context. Also, scripts
> to map device and inode info to containers sounds plausible for better
> tracing of containers.
>
> Signed-off-by: Hari Bathini <hbathini@xxxxxxxxxxxxxxxxxx>
> ---
> tools/perf/util/hist.c | 7 +++++++
> tools/perf/util/hist.h | 1 +
> tools/perf/util/sort.c | 41 +++++++++++++++++++++++++++++++++++++++++
> tools/perf/util/sort.h | 7 +++++++
> 4 files changed, 56 insertions(+)
>
> diff --git a/tools/perf/util/hist.c b/tools/perf/util/hist.c
> index 32c6a93..559ea27 100644
> --- a/tools/perf/util/hist.c
> +++ b/tools/perf/util/hist.c
> @@ -3,6 +3,7 @@
> #include "hist.h"
> #include "map.h"
> #include "session.h"
> +#include "namespaces.h"
> #include "sort.h"
> #include "evlist.h"
> #include "evsel.h"
> @@ -169,6 +170,7 @@ void hists__calc_col_len(struct hists *hists, struct hist_entry *h)
> hists__set_unres_dso_col_len(hists, HISTC_MEM_DADDR_DSO);
> }
>
> + hists__new_col_len(hists, HISTC_CGROUP_ID, 20);
> hists__new_col_len(hists, HISTC_CPU, 3);
> hists__new_col_len(hists, HISTC_SOCKET, 6);
> hists__new_col_len(hists, HISTC_MEM_LOCKED, 6);
> @@ -574,9 +576,14 @@ __hists__add_entry(struct hists *hists,
> bool sample_self,
> struct hist_entry_ops *ops)
> {
> + struct namespaces *ns = thread__namespaces(al->thread);
> struct hist_entry entry = {
> .thread = al->thread,
> .comm = thread__comm(al->thread),
> + .cgroup_id = {
> + .dev = ns ? ns->link_info[CGROUP_NS_INDEX].dev : 0,
> + .ino = ns ? ns->link_info[CGROUP_NS_INDEX].ino : 0,
> + },
> .ms = {
> .map = al->map,
> .sym = al->sym,
> diff --git a/tools/perf/util/hist.h b/tools/perf/util/hist.h
> index 28c216e..4c1da48 100644
> --- a/tools/perf/util/hist.h
> +++ b/tools/perf/util/hist.h
> @@ -30,6 +30,7 @@ enum hist_column {
> HISTC_DSO,
> HISTC_THREAD,
> HISTC_COMM,
> + HISTC_CGROUP_ID,
> HISTC_PARENT,
> HISTC_CPU,
> HISTC_SOCKET,
> diff --git a/tools/perf/util/sort.c b/tools/perf/util/sort.c
> index df622f4..9f5f404 100644
> --- a/tools/perf/util/sort.c
> +++ b/tools/perf/util/sort.c
> @@ -536,6 +536,46 @@ struct sort_entry sort_cpu = {
> .se_width_idx = HISTC_CPU,
> };
>
> +/* --sort cgroup_id */
> +
> +static int64_t _sort__cgroup_dev_cmp(u64 left_dev, u64 right_dev)
> +{
> + return (int64_t)(right_dev - left_dev);
> +}
> +
> +static int64_t _sort__cgroup_inode_cmp(u64 left_ino, u64 right_ino)
> +{
> + return (int64_t)(right_ino - left_ino);
> +}
> +
> +static int64_t
> +sort__cgroup_id_cmp(struct hist_entry *left, struct hist_entry *right)
> +{
> + int64_t ret;
> +
> + ret = _sort__cgroup_dev_cmp(right->cgroup_id.dev, left->cgroup_id.dev);
> + if (ret != 0)
> + return ret;
> +
> + return _sort__cgroup_inode_cmp(right->cgroup_id.ino,
> + left->cgroup_id.ino);
> +}
> +
> +static int hist_entry__cgroup_id_snprintf(struct hist_entry *he,
> + char *bf, size_t size,
> + unsigned int width __maybe_unused)
> +{
> + return repsep_snprintf(bf, size, "%lu/0x%lx", he->cgroup_id.dev,
> + he->cgroup_id.ino);
> +}
> +
> +struct sort_entry sort_cgroup_id = {
> + .se_header = "cgroup id (dev/inode)",
> + .se_cmp = sort__cgroup_id_cmp,
> + .se_snprintf = hist_entry__cgroup_id_snprintf,
> + .se_width_idx = HISTC_CGROUP_ID,
> +};
> +
> /* --sort socket */
>
> static int64_t
> @@ -1418,6 +1458,7 @@ static struct sort_dimension common_sort_dimensions[] = {
> DIM(SORT_GLOBAL_WEIGHT, "weight", sort_global_weight),
> DIM(SORT_TRANSACTION, "transaction", sort_transaction),
> DIM(SORT_TRACE, "trace", sort_trace),
> + DIM(SORT_CGROUP_ID, "cgroup_id", sort_cgroup_id),
> };
>
> #undef DIM
> diff --git a/tools/perf/util/sort.h b/tools/perf/util/sort.h
> index 7aff317..68a5abb 100644
> --- a/tools/perf/util/sort.h
> +++ b/tools/perf/util/sort.h
> @@ -54,6 +54,11 @@ struct he_stat {
> u32 nr_events;
> };
>
> +struct namespace_id {
> + u64 dev;
> + u64 ino;
> +};
> +
> struct hist_entry_diff {
> bool computed;
> union {
> @@ -91,6 +96,7 @@ struct hist_entry {
> struct map_symbol ms;
> struct thread *thread;
> struct comm *comm;
> + struct namespace_id cgroup_id;
> u64 ip;
> u64 transaction;
> s32 socket;
> @@ -211,6 +217,7 @@ enum sort_type {
> SORT_GLOBAL_WEIGHT,
> SORT_TRANSACTION,
> SORT_TRACE,
> + SORT_CGROUP_ID,
>
> /* branch stack specific sort keys */
> __SORT_BRANCH_STACK,