Index: MM-2.6.X/kernel/sched.c =================================================================== --- MM-2.6.X.orig/kernel/sched.c 2006-02-13 10:23:12.000000000 +1100 +++ MM-2.6.X/kernel/sched.c 2006-02-14 09:47:17.000000000 +1100 @@ -2033,9 +2033,11 @@ find_busiest_group(struct sched_domain * struct sched_group *busiest = NULL, *this = NULL, *group = sd->groups; unsigned long max_load, avg_load, total_load, this_load, total_pwr; unsigned long max_pull; + unsigned long avg_load_per_task, busiest_nr_running; int load_idx; max_load = this_load = total_load = total_pwr = 0; + busiest_nr_running = 0; if (idle == NOT_IDLE) load_idx = sd->busy_idx; else if (idle == NEWLY_IDLE) @@ -2047,11 +2049,12 @@ find_busiest_group(struct sched_domain * unsigned long load; int local_group; int i; + unsigned long sum_nr_running; local_group = cpu_isset(this_cpu, group->cpumask); /* Tally up the load of all CPUs in the group */ - avg_load = 0; + sum_nr_running = avg_load = 0; for_each_cpu_mask(i, group->cpumask) { if (*sd_idle && !idle_cpu(i)) @@ -2064,6 +2067,7 @@ find_busiest_group(struct sched_domain * load = source_load(i, load_idx); avg_load += load; + sum_nr_running += cpu_rq(i)->nr_running; } total_load += avg_load; @@ -2078,6 +2082,7 @@ find_busiest_group(struct sched_domain * } else if (avg_load > max_load) { max_load = avg_load; busiest = group; + busiest_nr_running = sum_nr_running; } group = group->next; } while (group != sd->groups); @@ -2111,12 +2116,20 @@ find_busiest_group(struct sched_domain * (avg_load - this_load) * this->cpu_power) / SCHED_LOAD_SCALE; - if (*imbalance < SCHED_LOAD_SCALE) { + /* Don't assume that busiest_nr_running > 0 */ + avg_load_per_task = busiest_nr_running ? max_load / busiest_nr_running : avg_load; + /* + * if *imbalance is less than the average load per runnable task + * there is no gaurantee that any tasks will be moved so we'll have + * a think about bumping its value to force at least one task to be + * moved + */ + if (*imbalance < avg_load_per_task) { unsigned long pwr_now = 0, pwr_move = 0; unsigned long tmp; - if (max_load - this_load >= SCHED_LOAD_SCALE*2) { - *imbalance = NICE_TO_BIAS_PRIO(0); + if (max_load - this_load >= avg_load_per_task*2) { + *imbalance = biased_load(avg_load_per_task); return busiest; } @@ -2146,11 +2159,14 @@ find_busiest_group(struct sched_domain * pwr_move /= SCHED_LOAD_SCALE; /* Move if we gain throughput */ - if (pwr_move <= pwr_now) - goto out_balanced; - - *imbalance = NICE_TO_BIAS_PRIO(0); - return busiest; + if (pwr_move <= pwr_now) { + /* or if there's a reasonable chance that *imbalance + * is big enough to cause a move + */ + if (*imbalance <= avg_load_per_task / 2) + goto out_balanced; + } else + *imbalance = avg_load_per_task; } /*