From: Don Zickus <dzickus@redhat.com>
To: acme@ghostprotocols.net
Cc: LKML <linux-kernel@vger.kernel.org>,
jolsa@redhat.com, jmario@redhat.com, fowles@inreach.com,
eranian@google.com, Don Zickus <dzickus@redhat.com>
Subject: [PATCH 20/21] perf, c2c: Add selected extreme latencies to output cacheline stats table
Date: Mon, 10 Feb 2014 12:29:15 -0500 [thread overview]
Message-ID: <1392053356-23024-21-git-send-email-dzickus@redhat.com> (raw)
In-Reply-To: <1392053356-23024-1-git-send-email-dzickus@redhat.com>
This just takes the previously calculated extreme latencies and prints them
in a pretty table with the cacheline and its offsets exposed for to help
further understand what they are coming from.
Original work done by Dick Fowles, ported to perf by me.
Suggested-by: Joe Mario <jmario@redhat.com>
Original-by: Dick Fowles <rfowles@redhat.com>
Signed-off-by: Don Zickus <dzickus@redhat.com>
---
tools/perf/builtin-c2c.c | 265 +++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 265 insertions(+)
diff --git a/tools/perf/builtin-c2c.c b/tools/perf/builtin-c2c.c
index b1d4a8b..1fa21b4 100644
--- a/tools/perf/builtin-c2c.c
+++ b/tools/perf/builtin-c2c.c
@@ -76,6 +76,7 @@ struct perf_c2c {
struct c2c_entry {
struct rb_node rb_node;
struct rb_node latency;
+ struct rb_node latency_scratch;
struct list_head scratch; /* scratch list for resorting */
struct thread *thread;
int tid; /* FIXME perf maps broken */
@@ -571,6 +572,62 @@ static int c2c_latency__add_to_list(struct rb_root *root, struct c2c_entry *n)
return 0;
}
+static struct c2c_entry *c2c_latency__add_to_list_physid(struct rb_root *root,
+ struct c2c_entry *entry)
+{
+ struct rb_node **p;
+ struct rb_node *parent = NULL;
+ struct c2c_entry *ce;
+ int64_t cmp;
+
+ p = &root->rb_node;
+
+ while (*p != NULL) {
+ parent = *p;
+ ce = rb_entry(parent, struct c2c_entry, latency_scratch);
+
+ cmp = physid_cmp(ce, entry);
+
+ if (cmp > 0)
+ p = &(*p)->rb_left;
+ else
+ p = &(*p)->rb_right;
+ }
+
+ rb_link_node(&entry->latency_scratch, parent, p);
+ rb_insert_color(&entry->latency_scratch, root);
+
+ return entry;
+}
+
+static int c2c_latency__add_to_list_count(struct rb_root *root,
+ struct c2c_hit *h)
+{
+ struct rb_node **p;
+ struct rb_node *parent = NULL;
+ struct c2c_hit *he;
+ int64_t cmp;
+
+ p = &root->rb_node;
+
+ while (*p != NULL) {
+ parent = *p;
+ he = rb_entry(parent, struct c2c_hit, rb_node);
+
+ cmp = h->stats.stats.n - he->stats.stats.n;
+
+ if (cmp > 0)
+ p = &(*p)->rb_left;
+ else
+ p = &(*p)->rb_right;
+ }
+
+ rb_link_node(&h->rb_node, parent, p);
+ rb_insert_color(&h->rb_node, root);
+
+ return 0;
+}
+
static int perf_c2c__fprintf_header(FILE *fp)
{
int printed = fprintf(fp, "%c %-16s %6s %6s %4s %18s %18s %18s %6s %-10s %-60s %s\n",
@@ -1107,6 +1164,209 @@ cleanup:
}
}
+static void print_latency_select_cacheline_offset(struct c2c_hit *offset,
+ int total)
+{
+ struct stats *s = &offset->stats.stats;
+ struct addr_map_symbol *ams = &offset->mi->iaddr;
+
+ printf("%5s %6s %6s %7.1f%% %14s0x%02lx %#18lx %8ld %7.1f %8ld %7.1f %7.1f%% %-30s %-20s\n",
+ " ",
+ " ",
+ " ",
+ ((double) s->n / (double)total) * 100.0,
+ " ",
+ (cloffset == LVL2) ? (offset->mi->daddr.addr & 0xff) : CLOFFSET(offset->mi->daddr.addr),
+ offset->mi->iaddr.addr,
+ s->min,
+ 0.0,
+ s->max,
+ avg_stats(s),
+ (stddev_stats(s)/avg_stats(s) * 100.0),
+ (ams->sym ? ams->sym->name : "?????"),
+ ams->map->dso->short_name);
+}
+
+static void print_latency_select_header(void)
+{
+#define EXCESS_LATENCY_TITLE "Non Shared Data Loads With Excessive Execution Latency"
+
+ static char delimit[MAXTITLE_SZ];
+ static char title[MAXTITLE_SZ];
+ int pad;
+ int i;
+
+ sprintf(title, "%5s %6s %6s %8s %18s %18s %8s %8s %8s %8s %8s %-30s %-20s",
+ "Num",
+ "%dist",
+ "%cumm",
+ "Count",
+ "Data Address",
+ "Inst Address",
+ "Min",
+ "Median",
+ "Max",
+ "Mean",
+ "CV",
+ "Symbol",
+ "Object");
+
+ memset(delimit, 0, sizeof(delimit));
+ for (i = 0; i < (int)strlen(title); i++) delimit[i] = '=';
+
+ printf("\n\n");
+ printf("%s\n", delimit);
+
+ pad = (strlen(title)/2) - (strlen(EXCESS_LATENCY_TITLE)/2);
+ for (i = 0; i < pad; i++) printf(" ");
+ printf("%s\n", EXCESS_LATENCY_TITLE);
+ printf("\n");
+
+ printf("%5s %6s %6s %8s %18s %18s %44s %-30s %-20s\n",
+ " ",
+ " ",
+ " ",
+ "Load",
+ " ",
+ " ",
+ "------ Load Inst Execute Latency ------",
+ " ",
+ " ");
+
+ printf("%s\n", title);
+ printf("%s\n", delimit);
+}
+
+static void print_latency_select_info(struct rb_root *root,
+ struct c2c_stats *stats)
+{
+#define XLAT_DIST_LIMIT 0.1
+
+ struct rb_node *next = rb_first(root);
+ struct c2c_hit *h, *clo = NULL;
+ struct c2c_entry *entry;
+ double tot_dist, tot_cumm;
+ int idx = 0, j;
+ static char delimit[MAXTITLE_SZ];
+ static char summary[MAXTITLE_SZ];
+
+ print_latency_select_header();
+
+ tot_cumm = 0.0;
+
+ while (next) {
+ h = rb_entry(next, struct c2c_hit, rb_node);
+ next = rb_next(&h->rb_node);
+
+ tot_dist = ((double)h->stats.stats.n / stats->stats.n);
+ tot_cumm += tot_dist;
+
+ /*
+ * don't display lines with insignificant sharing contribution
+ */
+ if (tot_dist*100.0 < XLAT_DIST_LIMIT)
+ break;
+
+ sprintf(summary, "%5d %5.1f%% %5.1f%% %8d %#18lx",
+ idx,
+ tot_dist*100.0,
+ tot_cumm*100.0,
+ (int)h->stats.stats.n,
+ h->cacheline);
+
+ if (delimit[0] != '-') {
+ memset(delimit, 0, sizeof(delimit));
+ for (j = 0; j < (int)strlen(summary); j++) delimit[j] = '-';
+ }
+
+ printf("%s\n", delimit);
+ printf("%s\n", summary);
+ printf("%s\n", delimit);
+
+ list_for_each_entry(entry, &h->list, scratch) {
+
+ if (!clo || !matching_coalescing(clo, entry)) {
+ u64 addr;
+
+ if (clo)
+ print_latency_select_cacheline_offset(clo, h->stats.stats.n);
+
+ free(clo);
+ addr = entry->mi->iaddr.al_addr;
+ clo = c2c_hit__new(addr, entry);
+ }
+ update_stats(&clo->stats.stats, entry->weight);
+ }
+ if (clo) {
+ print_latency_select_cacheline_offset(clo, h->stats.stats.n);
+ free(clo);
+ clo = NULL;
+ }
+
+ idx++;
+ }
+ printf("\n\n");
+}
+
+static void calculate_latency_selected_info(struct rb_root *root,
+ struct rb_node *start,
+ struct c2c_stats *lat_stats)
+{
+ struct rb_node *next = start;
+ struct rb_root lat_tree = RB_ROOT;
+ struct c2c_hit *h = NULL;
+ struct c2c_entry *n;
+ u64 cl;
+
+ /* new sort of 'selected' tree using physid_cmp */
+ while (next) {
+ n = rb_entry(next, struct c2c_entry, latency);
+ next = rb_next(&n->latency);
+
+ c2c_latency__add_to_list_physid(&lat_tree, n);
+ }
+
+ /* resort based on number of entries in each cacheline */
+ next = rb_first(&lat_tree);
+ while (next) {
+ n = rb_entry(next, struct c2c_entry, latency_scratch);
+ next = rb_next(&n->latency_scratch);
+
+ cl = n->mi->daddr.al_addr;
+
+ /* switch cache line objects */
+ /* 'color' forces a boundary change based on the original sort */
+ if (!h || !n->color || (CLADRS(cl) != h->cacheline)) {
+ if (h)
+ c2c_latency__add_to_list_count(root, h);
+
+ h = c2c_hit__new(CLADRS(cl), n);
+ if (!h)
+ goto cleanup;
+ }
+
+ update_stats(&h->stats.stats, n->weight);
+ update_stats(&lat_stats->stats, n->weight);
+
+ /* save the entry for later processing */
+ list_add_tail(&n->scratch, &h->list);
+ }
+ /* last chunk */
+ if (h)
+ c2c_latency__add_to_list_count(root, h);
+ return;
+
+cleanup:
+ next = rb_first(root);
+ while (next) {
+ h = rb_entry(next, struct c2c_hit, rb_node);
+ next = rb_next(&h->rb_node);
+ rb_erase(&h->rb_node, root);
+
+ free(h);
+ }
+}
+
stats_t data[] = {
{ "Samples ", "%20d", &hist_info[OVERALL].cnt, &hist_info[EXTREMES].cnt, &hist_info[ANALYZE].cnt },
{ " ", NULL, NULL, NULL, NULL },
@@ -1471,6 +1731,8 @@ static void c2c_analyze_latency(struct perf_c2c *c2c)
struct c2c_stats lat_stats;
u64 snoop;
struct stats s;
+ int i;
+ struct rb_root lat_select_tree = RB_ROOT;
init_stats(&s);
memset(&lat_stats, 0, sizeof(struct c2c_stats));
@@ -1500,6 +1762,9 @@ static void c2c_analyze_latency(struct perf_c2c *c2c)
calculate_latency_info(&lat_tree, &s, overall, extremes, selected);
print_latency_info();
+ calculate_latency_selected_info(&lat_select_tree, selected->start, &lat_stats);
+ print_latency_select_info(&lat_select_tree, &lat_stats);
+
return;
}
--
1.7.11.7
next prev parent reply other threads:[~2014-02-10 17:32 UTC|newest]
Thread overview: 72+ messages / expand[flat|nested] mbox.gz Atom feed top
2014-02-10 17:28 [PATCH 00/21] perf, c2c: Add new tool to analyze cacheline contention on NUMA systems Don Zickus
2014-02-10 17:28 ` [PATCH 03/21] Revert "perf: Disable PERF_RECORD_MMAP2 support" Don Zickus
2014-02-10 17:28 ` [PATCH 04/21] perf, machine: Use map as success in ip__resolve_ams Don Zickus
2014-02-10 17:29 ` [PATCH 05/21] perf, session: Change header.misc dump from decimal to hex Don Zickus
2014-02-18 12:56 ` Jiri Olsa
2014-02-19 2:40 ` Don Zickus
2014-02-10 17:29 ` [PATCH 06/21] perf, stat: FIXME Stddev calculation is incorrect Don Zickus
2014-02-10 17:29 ` [PATCH 07/21] perf, callchain: Add generic callchain print handler for stdio Don Zickus
2014-02-10 17:29 ` [PATCH 08/21] perf, c2c: Rework setup code to prepare for features Don Zickus
2014-02-18 13:02 ` Jiri Olsa
2014-02-19 2:45 ` Don Zickus
2014-02-10 17:29 ` [PATCH 09/21] perf, c2c: Add rbtree sorted on mmap2 data Don Zickus
2014-02-18 13:04 ` Jiri Olsa
2014-02-19 2:48 ` Don Zickus
2014-02-21 2:45 ` Don Zickus
2014-02-21 16:59 ` Jiri Olsa
2014-02-26 3:12 ` Don Zickus
2014-02-10 17:29 ` [PATCH 10/21] perf, c2c: Add stats to track data source bits and cpu to node maps Don Zickus
2014-02-18 13:05 ` Jiri Olsa
2014-02-19 2:51 ` Don Zickus
2014-02-10 17:29 ` [PATCH 11/21] perf, c2c: Sort based on hottest cache line Don Zickus
2014-02-10 17:29 ` [PATCH 12/21] perf, c2c: Display cacheline HITM analysis to stdout Don Zickus
2014-02-10 17:29 ` [PATCH 13/21] perf, c2c: Add callchain support Don Zickus
2014-02-18 13:07 ` Jiri Olsa
2014-02-19 2:54 ` Don Zickus
2014-02-10 17:29 ` [PATCH 14/21] perf, c2c: Output summary stats Don Zickus
2014-02-10 17:29 ` [PATCH 15/21] perf, c2c: Dump rbtree for debugging Don Zickus
2014-02-10 17:29 ` [PATCH 16/21] perf, c2c: Fixup tid because of perf map is broken Don Zickus
2014-02-10 17:29 ` [PATCH 17/21] perf, c2c: Add symbol count table Don Zickus
2014-02-18 13:09 ` Jiri Olsa
2014-02-19 2:56 ` Don Zickus
2014-02-10 17:29 ` [PATCH 18/21] perf, c2c: Add shared cachline summary table Don Zickus
2014-02-10 17:29 ` [PATCH 19/21] perf, c2c: Add framework to analyze latency and display summary stats Don Zickus
2014-02-10 17:29 ` Don Zickus [this message]
2014-02-10 17:29 ` [PATCH 21/21] perf, c2c: Add summary latency table for various parts of caches Don Zickus
2014-02-10 18:59 ` [PATCH 00/21] perf, c2c: Add new tool to analyze cacheline contention on NUMA systems Davidlohr Bueso
2014-02-10 19:17 ` Don Zickus
2014-02-10 19:18 ` [PATCH 01/21] perf c2c: Shared data analyser Don Zickus
2014-02-10 22:10 ` Davidlohr Bueso
2014-02-11 11:24 ` Jiri Olsa
2014-02-11 11:31 ` Arnaldo Carvalho de Melo
2014-02-11 13:54 ` Don Zickus
2014-02-11 14:36 ` Don Zickus
2014-02-11 15:41 ` Arnaldo Carvalho de Melo
2014-02-10 19:18 ` [PATCH 02/21] perf c2c: Dump raw records, decode data_src bits Don Zickus
2014-02-10 21:18 ` [PATCH 00/21] perf, c2c: Add new tool to analyze cacheline contention on NUMA systems Peter Zijlstra
2014-02-10 22:11 ` Don Zickus
2014-02-10 21:29 ` Peter Zijlstra
2014-02-10 22:20 ` Don Zickus
2014-02-10 22:21 ` Stephane Eranian
2014-02-11 7:14 ` Peter Zijlstra
2014-02-11 10:35 ` Stephane Eranian
2014-02-11 10:52 ` Peter Zijlstra
2014-02-11 10:58 ` Stephane Eranian
2014-02-11 11:02 ` Peter Zijlstra
2014-02-11 11:04 ` Stephane Eranian
2014-02-11 11:08 ` Peter Zijlstra
2014-02-11 11:08 ` Stephane Eranian
2014-02-11 11:14 ` Peter Zijlstra
2014-02-11 11:28 ` Stephane Eranian
2014-02-11 11:31 ` Peter Zijlstra
2014-02-11 11:51 ` Peter Zijlstra
2014-02-11 11:50 ` Arnaldo Carvalho de Melo
2014-02-11 12:09 ` Peter Zijlstra
2014-02-13 13:02 ` Jiri Olsa
2014-02-13 13:10 ` Stephane Eranian
[not found] ` <1392053356-23024-2-git-send-email-dzickus@redhat.com>
2014-02-18 12:52 ` [PATCH 01/21] perf c2c: Shared data analyser Jiri Olsa
2014-02-18 12:56 ` Arnaldo Carvalho de Melo
2014-02-19 2:42 ` Don Zickus
[not found] ` <1392053356-23024-3-git-send-email-dzickus@redhat.com>
2014-02-18 12:53 ` [PATCH 02/21] perf c2c: Dump raw records, decode data_src bits Jiri Olsa
2014-02-18 13:49 ` Arnaldo Carvalho de Melo
2014-02-19 3:04 ` Don Zickus
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=1392053356-23024-21-git-send-email-dzickus@redhat.com \
--to=dzickus@redhat.com \
--cc=acme@ghostprotocols.net \
--cc=eranian@google.com \
--cc=fowles@inreach.com \
--cc=jmario@redhat.com \
--cc=jolsa@redhat.com \
--cc=linux-kernel@vger.kernel.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®