mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Zachary Amsden <zamsden@redhat.com>
To: kvm@vger.kernel.org
Cc: Zachary Amsden <zamsden@redhat.com>, Avi Kivity <avi@redhat.com>,
	Marcelo Tosatti <mtosatti@redhat.com>,
	Joerg Roedel <joerg.roedel@amd.com>,
	linux-kernel@vger.kernel.org, Dor Laor <dlaor@redhat.com>
Subject: [PATCH RFC: kvm tsc virtualization 12/20] Higher accuracy TSC offset computation
Date: Mon, 14 Dec 2009 18:08:39 -1000	[thread overview]
Message-ID: <1260850127-9766-13-git-send-email-zamsden@redhat.com> (raw)
In-Reply-To: <1260850127-9766-12-git-send-email-zamsden@redhat.com>

Use per-cpu delta counters and statistically eliminate outliers which
might happen due to platform anomalies such as NMI.

Signed-off-by: Zachary Amsden <zamsden@redhat.com>
---
 arch/x86/kvm/x86.c |  100 +++++++++++++++++++++++++++++++++++++---------------
 1 files changed, 71 insertions(+), 29 deletions(-)

diff --git a/arch/x86/kvm/x86.c b/arch/x86/kvm/x86.c
index 3c4266f..c66dede 100644
--- a/arch/x86/kvm/x86.c
+++ b/arch/x86/kvm/x86.c
@@ -840,6 +840,11 @@ again:
 EXPORT_SYMBOL_GPL(kvm_get_ref_tsc);
 
 #define SYNC_TRIES 64
+struct cpu_delta_array {
+	s64 delta[SYNC_TRIES];
+};
+static DEFINE_PER_CPU(struct cpu_delta_array, delta_array);
+
 
 /*
  * sync_tsc_helper is a dual-entry coroutine meant to be run by only
@@ -849,25 +854,16 @@ EXPORT_SYMBOL_GPL(kvm_get_ref_tsc);
  *
  * To discount cache latency effects, this routine will be called
  * twice, one with the measure / recording CPUs reversed.  In addition,
- * the first 4 and last 2 results will be discarded to allow branch
- * predicition to become established (and to discount effects from
- * a potentially early predicted loop exit).
+ * the first and last results will be discarded to allow branch predicition
+ * to become established (and to discount effects from a potentially early
+ * predicted loop exit).
  *
- * Because of this, we must be extra careful to guard the entrance
- * and exit against the CPU switch.  I chose to use atomic instructions
- * only at the end of the measure loop and use the same routine for
- * both CPUs, with symmetric comparisons, and a statically placed
- * recording array, hopefully maximizing the branch predicition and
- * cache locality.  The results appear quite good; on known to be
- * synchronized CPUs, I typically get < 10 TSC delta measured, with
- * maximum observed error on the order of 100 cycles.
- *
- * This doesn't account for NUMA cache effects, and could potentially
- * be improved there by moving the delta[] array to the stack of the
- * measuring CPU.  In fact, this modification might be worth trying
- * for non-NUMA systems as well, but this appears to be fine for now.
+ * We allocate the delta array as a per-cpu variable to get local cache
+ * allocation, and also take steps to statistically ignore outliers which
+ * might be caused by SMIs.  It may seem like overkill, but it is very
+ * accurate, which is what we're aiming for.
  */
-static void sync_tsc_helper(int measure_cpu, u64 *delta, atomic_t *ready)
+static void sync_tsc_helper(int measure_cpu, s64 *delta, atomic_t *ready)
 {
 	int tries;
 	static u64 tsc_other;
@@ -915,12 +911,59 @@ static void sync_tsc_helper(int measure_cpu, u64 *delta, atomic_t *ready)
 	mb();
 }
 
+/*
+ * Average and trim the samples of any outliers; we use > 2 x sigma
+ */
+static s64 average_samples(s64 *samples, unsigned num_samples)
+{
+	unsigned i, j;
+	s64 mean;
+	u64 stddev;
+	s64 average;
+
+	/* First get the mean */
+	mean = 0;
+	for (i = 0; i < num_samples; i++)
+		mean += samples[i];
+	mean = mean / num_samples;
+
+	/* Now the deviation */
+	stddev = 0;
+	for (i = 0; i < num_samples; i++) {
+		s64 dev = samples[i] - mean;
+		stddev += dev * dev;
+	}
+	stddev = stddev / (num_samples - 1);
+	stddev = int_sqrt(stddev);
+
+	/* Throw out anything outside 2x stddev */
+	stddev <<= 1;
+	average = 0, j = 0;
+	for (i = 0; i < num_samples; i++) {
+		s64 dev = samples[i] - mean;
+		if (dev < 0)
+			dev = -dev;
+		if (dev <= stddev) {
+			average += samples[i];
+			j++;
+		}
+	}
+	if (j > 0)
+		average = average / j;
+	else
+		printk(KERN_ERR "kvm: TSC samples failed to converge!\n");
+	pr_debug("%s: mean = %lld, stddev = %llu, average = %lld\n",
+		 __func__, mean, stddev, average);
+
+	return average;
+}
+
 static void kvm_sync_tsc(void *cpup)
 {
 	int new_cpu = *(int *)cpup;
 	unsigned long flags;
-	static s64 delta[SYNC_TRIES*2];
-	static atomic_t ready = ATOMIC_INIT(1);
+	s64 *delta1, *delta2;
+	static atomic_t ready ____cacheline_aligned = ATOMIC_INIT(1);
 
 	BUG_ON(tsc_base_cpu == -1);
 	pr_debug("%s: IN, cpu = %d, freq = %ldkHz, tsc_base_cpu = %d\n", __func__, raw_smp_processor_id(), per_cpu(cpu_tsc_khz, raw_smp_processor_id()) , tsc_base_cpu);
@@ -933,27 +976,26 @@ static void kvm_sync_tsc(void *cpup)
 			&per_cpu(cpu_tsc_multiplier, new_cpu),
 			&per_cpu(cpu_tsc_shift, new_cpu));
 	}
-	sync_tsc_helper(tsc_base_cpu, delta, &ready);
-	sync_tsc_helper(new_cpu, &delta[SYNC_TRIES], &ready);
+	delta1 = per_cpu(delta_array, tsc_base_cpu).delta;
+	delta2 = per_cpu(delta_array, new_cpu).delta;
+	sync_tsc_helper(tsc_base_cpu, delta1, &ready);
+	sync_tsc_helper(new_cpu, delta2, &ready);
 	if (raw_smp_processor_id() == new_cpu) {
-		int i;
 		s64 accumulator = 0;
 
 		/*
-		 * accumulate [SYNC_TRIES+4,-2) of tsc{base} - tsc{new}
-		 * subtract   [SYNC_TRIES+4,-2) of tsc{new} - tsc{base}
+		 * accumulate [2,SYNC_TRIES-1) of tsc{base} - tsc{new}
+		 * subtract   [SYNC_TRIES+2,-1) of tsc{new} - tsc{base}
 		 *
 		 * this allows instruction cycle and cache differences to
 		 * cancel each other out and drops warm up/cool down variation
 		 *
 		 * Note the arithmatic must be signed because of the divide
 		 */
+		accumulator += average_samples(&delta1[2], SYNC_TRIES-3);
+		accumulator -= average_samples(&delta2[2], SYNC_TRIES-3);
+		accumulator /= 2;
 
-		for (i = 4; i < SYNC_TRIES - 2; i++)
-			accumulator += delta[i];
-		for (i = 4; i < SYNC_TRIES - 2; i++)
-			accumulator -= delta[i+SYNC_TRIES];
-		accumulator = accumulator / (SYNC_TRIES*2-12);
 		per_cpu(cpu_tsc_offset, new_cpu) = accumulator;
 		++per_cpu(cpu_tsc_generation, new_cpu);
 		atomic_set(&per_cpu(cpu_tsc_synchronized, new_cpu), 1);
-- 
1.6.5.2


  reply	other threads:[~2009-12-15  4:11 UTC|newest]

Thread overview: 24+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
     [not found] <1260850127-9766-1-git-send-email-zamsden@redhat.com>
2009-12-15  4:08 ` [PATCH RFC: kvm tsc virtualization 01/20] Move TSC read to vmx_vcpu_put Zachary Amsden
2009-12-15  4:08   ` [PATCH RFC: kvm tsc virtualization 02/20] Add a hotplug notifier to KVM x86 backend Zachary Amsden
2009-12-15  4:08     ` [PATCH RFC: kvm tsc virtualization 03/20] TSC offset framework Zachary Amsden
2009-12-15  4:08       ` [PATCH RFC: kvm tsc virtualization 04/20] Synchronize TSC when a new CPU comes up Zachary Amsden
2009-12-15  4:08         ` [PATCH RFC: kvm tsc virtualization 05/20] Fix AMD C1 TSC desynchronization Zachary Amsden
2009-12-15  4:08           ` [PATCH RFC: kvm tsc virtualization 06/20] Make TSC reference stable across frequency changes Zachary Amsden
2009-12-15  4:08             ` [PATCH RFC: kvm tsc virtualization 07/20] Basic SVM implementation of RDTSC trapping Zachary Amsden
2009-12-15  4:08               ` [PATCH RFC: kvm tsc virtualization 08/20] Export the reference TSC from KVM module Zachary Amsden
2009-12-15  4:08                 ` [PATCH RFC: kvm tsc virtualization 09/20] Use TSC reference for SVM Zachary Amsden
2009-12-15  4:08                   ` [PATCH RFC: kvm tsc virtualization 10/20] Add a stat counter for RDTSC exits Zachary Amsden
2009-12-15  4:08                     ` [PATCH RFC: kvm tsc virtualization 11/20] Use highest TSC frequency as reference clock Zachary Amsden
2009-12-15  4:08                       ` Zachary Amsden [this message]
2009-12-15  4:08                         ` [PATCH RFC: kvm tsc virtualization 13/20] Combine observed TSC deviation into moving average Zachary Amsden
2009-12-15  4:08                           ` [PATCH RFC: kvm tsc virtualization 14/20] Move TSC cpu vars to a struct Zachary Amsden
2009-12-15  4:08                             ` [PATCH RFC: kvm tsc virtualization 15/20] Fix longstanding races Zachary Amsden
2009-12-15  4:08                               ` [PATCH RFC: kvm tsc virtualization 16/20] Fix 32-bit mult_precise Zachary Amsden
2009-12-15  4:08                                 ` [PATCH RFC: kvm tsc virtualization 17/20] Periodically measure TSC skew Zachary Amsden
2009-12-15  4:08                                   ` [PATCH RFC: kvm tsc virtualization 18/20] Implement variable speed TSC Zachary Amsden
2009-12-15  4:08                                     ` [PATCH RFC: kvm tsc virtualization 19/20] IOCTL for different TSC modes Zachary Amsden
2009-12-15  4:08                                       ` [PATCH RFC: kvm tsc virtualization 20/20] Get passthrough TSC working in SVM again Zachary Amsden
2009-12-15 13:58                               ` [PATCH RFC: kvm tsc virtualization 15/20] Fix longstanding races Andi Kleen
2009-12-15 18:21                               ` Marcelo Tosatti
2009-12-15 21:26                                 ` Zachary Amsden
2009-12-16 14:41                                   ` Marcelo Tosatti

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=1260850127-9766-13-git-send-email-zamsden@redhat.com \
    --to=zamsden@redhat.com \
    --cc=avi@redhat.com \
    --cc=dlaor@redhat.com \
    --cc=joerg.roedel@amd.com \
    --cc=kvm@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=mtosatti@redhat.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®