mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Con Kolivas <kernel@kolivas.org>
To: Nathan Fredrickson <8nrf@qlink.queensu.ca>
Cc: Linux Kernel Mailing List <linux-kernel@vger.kernel.org>
Subject: Re: HT schedulers' performance on single HT processor
Date: Tue, 16 Dec 2003 11:55:02 +1100	[thread overview]
Message-ID: <1071536102.3fde57e670996@vds.kolivas.org> (raw)
In-Reply-To: <1071533802.24673.35.camel@rocky>

[-- Attachment #1: Type: text/plain, Size: 1362 bytes --]

Quoting Nathan Fredrickson <8nrf@qlink.queensu.ca>:
>           X =  1     2     3     4     5     6     7     8     9    16
> 1phys UP      1.00  1.00  1.00  1.00  1.00  1.00  1.00  1.00  1.00  1.00
> 4phys SMP     1.00  0.99  0.51  0.35  0.27  0.27  0.27  0.27  0.27  0.27
> 4phys HT      1.01  1.00  0.55  0.40  0.33  0.29  0.27  0.26  0.25  0.26
> 4phys HT(w26) 1.01  1.01  0.54  0.37  0.31  0.27  0.26  0.26  0.26  0.26
> 4phys HT(C1)  1.01  1.00  0.52  0.36  0.29  0.28  0.27  0.26  0.25  0.26
> 
> Interesting that the overhead due to HT in the X=1 column is only 1%
> with 4 physical processors.  It was 1-3% before with 1 or 2 physical
> processors.
> 
> In the partial load columns where there are less compiler processes than
> logical CPUs (X=3,4,5,6,7), it appears that both patches are doing a
> better job scheduling than the standard scheduler.  At full load (X=>8)
> all three HT test cases perform about equally and beat standard SMP by
> 1-2%.
> 
> Hope these results are helpful.  I'd be happy to run more cases and/or
> other patches.

(cc list stripped)

Well since you asked... I've been looking for someone with more HT cpus to give
a much simpler approach a try. Here's a sample patch for vanilla test11 with
HT. This one actually helps UP HT performance ever so slightly and I'd be
curious to see if it does anything on more cpus.

Con

[-- Attachment #2: patch-test11-ht-3 --]
[-- Type: application/octet-stream, Size: 2612 bytes --]

--- linux-2.6.0-test11-base/kernel/sched.c	2003-11-24 22:18:56.000000000 +1100
+++ linux-2.6.0-test11-ht3/kernel/sched.c	2003-12-15 23:38:33.250059542 +1100
@@ -204,6 +204,7 @@ struct runqueue {
 	struct mm_struct *prev_mm;
 	prio_array_t *active, *expired, arrays[2];
 	int prev_cpu_load[NR_CPUS];
+	unsigned long cpu;
 #ifdef CONFIG_NUMA
 	atomic_t *node_nr_running;
 	int prev_node_load[MAX_NUMNODES];
@@ -221,6 +222,10 @@ static DEFINE_PER_CPU(struct runqueue, r
 #define task_rq(p)		cpu_rq(task_cpu(p))
 #define cpu_curr(cpu)		(cpu_rq(cpu)->curr)
 
+#define ht_active		(cpu_has_ht && smp_num_siblings > 1)
+#define ht_siblings(cpu1, cpu2)	(ht_active && \
+	cpu_sibling_map[(cpu1)] == (cpu2))
+
 /*
  * Default context-switch locking:
  */
@@ -1157,8 +1162,9 @@ can_migrate_task(task_t *tsk, runqueue_t
 {
 	unsigned long delta = sched_clock() - tsk->timestamp;
 
-	if (!idle && (delta <= JIFFIES_TO_NS(cache_decay_ticks)))
-		return 0;
+	if (!idle && (delta <= JIFFIES_TO_NS(cache_decay_ticks)) &&
+		!ht_siblings(this_cpu, task_cpu(tsk)))
+			return 0;
 	if (task_running(rq, tsk))
 		return 0;
 	if (!cpu_isset(this_cpu, tsk->cpus_allowed))
@@ -1193,15 +1199,23 @@ static void load_balance(runqueue_t *thi
 	imbalance /= 2;
 
 	/*
+	 * For hyperthread siblings take tasks from the active array
+	 * to get cache-warm tasks since they share caches.
+	 */
+	if (ht_siblings(this_cpu, busiest->cpu))
+		array = busiest->active;
+	/*
 	 * We first consider expired tasks. Those will likely not be
 	 * executed in the near future, and they are most likely to
 	 * be cache-cold, thus switching CPUs has the least effect
 	 * on them.
 	 */
-	if (busiest->expired->nr_active)
-		array = busiest->expired;
-	else
-		array = busiest->active;
+	else {
+		if (busiest->expired->nr_active)
+			array = busiest->expired;
+		else
+			array = busiest->active;
+	}
 
 new_array:
 	/* Start searching at priority 0: */
@@ -1212,9 +1226,16 @@ skip_bitmap:
 	else
 		idx = find_next_bit(array->bitmap, MAX_PRIO, idx);
 	if (idx >= MAX_PRIO) {
-		if (array == busiest->expired) {
-			array = busiest->active;
-			goto new_array;
+		if (ht_siblings(this_cpu, busiest->cpu)){
+			if (array == busiest->active) {
+				array = busiest->expired;
+				goto new_array;
+			}
+		} else {
+			if (array == busiest->expired) {
+				array = busiest->active;
+				goto new_array;
+			}
 		}
 		goto out_unlock;
 	}
@@ -2812,6 +2833,7 @@ void __init sched_init(void)
 		prio_array_t *array;
 
 		rq = cpu_rq(i);
+		rq->cpu = (unsigned long)(i);
 		rq->active = rq->arrays;
 		rq->expired = rq->arrays + 1;
 		spin_lock_init(&rq->lock);

  reply	other threads:[~2003-12-16  0:40 UTC|newest]

Thread overview: 9+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2003-12-12 14:57 Con Kolivas
2003-12-14 19:49 ` Nathan Fredrickson
2003-12-14 20:35   ` Adam Kropelin
2003-12-14 21:15     ` Nathan Fredrickson
2003-12-15 10:11   ` Con Kolivas
2003-12-16  0:16     ` Nathan Fredrickson
2003-12-16  0:55       ` Con Kolivas [this message]
2003-12-16  3:57         ` Nathan Fredrickson
2004-01-03 17:56 ` Bill Davidsen

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=1071536102.3fde57e670996@vds.kolivas.org \
    --to=kernel@kolivas.org \
    --cc=8nrf@qlink.queensu.ca \
    --cc=linux-kernel@vger.kernel.org \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®