mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Thomas Gleixner <tglx@linutronix.de>
To: Arun R Bharadwaj <arun@linux.vnet.ibm.com>
Cc: linux-kernel@vger.kernel.org,
	linux-pm@lists.linux-foundation.org, a.p.zijlstra@chello.nl,
	ego@in.ibm.com, mingo@elte.hu, andi@firstfloor.org,
	venkatesh.pallipadi@intel.com, vatsa@linux.vnet.ibm.com,
	arjan@infradead.org, svaidy@linux.vnet.ibm.com
Subject: Re: [v4 RFC PATCH 4/4] timers: logic to move non pinned timers
Date: Fri, 3 Apr 2009 23:52:51 +0200 (CEST)	[thread overview]
Message-ID: <alpine.LFD.2.00.0904032142460.12916@localhost.localdomain> (raw)
In-Reply-To: <20090401113738.GE22478@linux.vnet.ibm.com>

Arun,

On Wed, 1 Apr 2009, Arun R Bharadwaj wrote:
> @@ -627,6 +628,12 @@ __mod_timer(struct timer_list *timer, un
>  
>  	new_base = __get_cpu_var(tvec_bases);
>  
> +	current_cpu = smp_processor_id();
> +	preferred_cpu = get_nohz_load_balancer();
> +	if (get_sysctl_timer_migration() && idle_cpu(current_cpu)
> +			&& !pinned && preferred_cpu != -1)

Can we please check the preliminaries first to avoid the atomic_read ?

    cpu = smp_processor_id();
    if (!pinned && idle_cpu() && get_sysctl_timer_migration()) {
       newcpu = get_nohz_load_balancer();
       if (newcpu >= 0)
       	  cpu = newcpu;   
    }

    new_base = per_cpu(tvec_bases, cpu);

> +		new_base = per_cpu(tvec_bases, preferred_cpu);
> +
>  	if (base != new_base) {
>  		/*
>  		 * We are trying to schedule the timer on the local CPU.
> @@ -635,7 +642,8 @@ __mod_timer(struct timer_list *timer, un
>  		 * handler yet has not finished. This also guarantees that
>  		 * the timer is serialized wrt itself.
>  		 */
> -		if (likely(base->running_timer != timer)) {
> +		if (likely(base->running_timer != timer) ||
> +				get_sysctl_timer_migration()) {

  No, that's wrong. We can not migrate a running timer ever.

>  			/* See the comment in lock_timer_base() */
>  			timer_set_base(timer, NULL);
>  			spin_unlock(&base->lock);
> @@ -1063,10 +1071,9 @@ cascade:
>   * Check, if the next hrtimer event is before the next timer wheel
>   * event:
>   */
> -static unsigned long cmp_next_hrtimer_event(unsigned long now,
> -					    unsigned long expires)
> +static unsigned long __cmp_next_hrtimer_event(unsigned long now,
> +			unsigned long expires, ktime_t hr_delta)
>  {
> -	ktime_t hr_delta = hrtimer_get_next_event();
>  	struct timespec tsdelta;
>  	unsigned long delta;
>  
> @@ -1103,24 +1110,59 @@ static unsigned long cmp_next_hrtimer_ev
>  	return expires;
>  }
>  
> +static unsigned long cmp_next_hrtimer_event(unsigned long now,
> +					unsigned long expires)
> +{
> +	 ktime_t hr_delta = hrtimer_get_next_event();
> +	 return __cmp_next_hrtimer_event(now, expires, hr_delta);
> +}
> +
> +static unsigned long cmp_next_hrtimer_event_on(unsigned long now,
> +					unsigned long expires, int cpu)
> +{
> +	 ktime_t hr_delta = hrtimer_get_next_event_on(cpu);
> +	 return __cmp_next_hrtimer_event(now, expires, hr_delta);
> +}
> +
> +unsigned long __get_next_timer_interrupt(unsigned long now, int cpu)
> +{
> +	struct tvec_base *base = per_cpu(tvec_bases, cpu);
> +	unsigned long expires;
> +
> +	spin_lock(&base->lock);
> +	expires = __next_timer_interrupt(base);
> +	spin_unlock(&base->lock);
> +	return expires;
> +}
> +
>  /**
>   * get_next_timer_interrupt - return the jiffy of the next pending timer
>   * @now: current time (in jiffies)
>   */
>  unsigned long get_next_timer_interrupt(unsigned long now)
>  {
> -	struct tvec_base *base = __get_cpu_var(tvec_bases);
>  	unsigned long expires;
> +	int cpu = smp_processor_id();
>  
> -	spin_lock(&base->lock);
> -	expires = __next_timer_interrupt(base);
> -	spin_unlock(&base->lock);
> +	expires = __get_next_timer_interrupt(now, cpu);
>  
>  	if (time_before_eq(expires, now))
>  		return now;
>  
>  	return cmp_next_hrtimer_event(now, expires);
>  }
> +
> +unsigned long get_next_timer_interrupt_on(unsigned long now, int cpu)
> +{
> +	unsigned long expires;
> +
> +	expires = __get_next_timer_interrupt(now, cpu);
> +
> +	if (time_before_eq(expires, now))
> +		return now;
> +
> +	return cmp_next_hrtimer_event_on(now, expires, cpu);
> +}
>  #endif

   What's the purpose of all those changes ? Just for the latency check ?
   See below.
  
>  /*
> Index: linux.trees.git/kernel/hrtimer.c
> ===================================================================
> --- linux.trees.git.orig/kernel/hrtimer.c
> +++ linux.trees.git/kernel/hrtimer.c
> @@ -43,6 +43,8 @@
>  #include <linux/seq_file.h>
>  #include <linux/err.h>
>  #include <linux/debugobjects.h>
> +#include <linux/sched.h>
> +#include <linux/timer.h>
>  
>  #include <asm/uaccess.h>
>  
> @@ -198,8 +200,16 @@ switch_hrtimer_base(struct hrtimer *time
>  {
>  	struct hrtimer_clock_base *new_base;
>  	struct hrtimer_cpu_base *new_cpu_base;
> +	int current_cpu, preferred_cpu;
> +
> +	current_cpu = smp_processor_id();
> +	preferred_cpu = get_nohz_load_balancer();
> +	if (get_sysctl_timer_migration() && !pinned && preferred_cpu != -1 &&
> +			check_hrtimer_latency(timer, preferred_cpu))

  Comments from timer.c __mod_timer() apply here as well.

  You have a cpu_idle check in timer.c, why not here ?

  This check_hrtimer_latency business is ugly and I think we should
  try to solve this different.

...
  int cpu, tocpu = -1;

  cpu = smp_processor_id();
  if (preliminaries_for_migration) {
     tocpu = get_nohz_load_balancer();
     if (tocpu >= 0)
     	cpu = tocpu;
  }

again:
  new_cpu_base = &per_cpu(hrtimer_bases, cpu);
  new_base = &new_cpu_base->clock_base[base->index];

  if (base != new_base) {

     if (unlikely(hrtimer_callback_running(timer)))
     	return base;

     timer->base = NULL;
     spin_unlock(&base->cpu_base->lock);
     spin_lock(&new_base->cpu_base->lock);
     timer->base = new_base;

     if (cpu == tocpu) {
     	/* Calc clock monotonic expiry time */  	
        ktime_t expires = ktime_sub(hrtimer_get_expires(timer), new_base->offset);

	/*
	 * Get the next event on the target cpu from the clock events layer. This
	 * covers the highres=off nohz=on case as well.
	 */
	ktime_t next = clockevents_get_next_event(cpu);

	ktime_t delta = ktime_sub(expires, next);

	/*
	 * We do not migrate the timer when it is expiring before the next
	 * event on the target cpu because we can not reprogram the target
	 * cpu timer hardware and we would cause it to fire late.
	 */
	if (delta.tv64 < 0) {
	   cpu = smp_processor_id();
	   goto again;
     	}
	
	/*
	 * We might add a check here which does not migrate the timer
     	 * when it's expiry is very close, but that needs to be evaluated
	 * if it's really a problem. Again we can ask the clock events layer
         * here when the next tick timer is due and compare against it to
         * avoid an extra ktime_get() call.
	 * Probably it's not a problem as a possible wakeup of some task will
	 * push that task anyway to the preferred cpu, but we'll see.
         */
     } 
   }

That way we avoid the whole poking in the timer wheel and adding /
modifying functions all over the place for no real value.

The information we are looking for is already there.

Thanks,

	tglx

  parent reply	other threads:[~2009-04-03 21:54 UTC|newest]

Thread overview: 16+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2009-04-01 11:31 [v4 RFC PATCH 0/4] timers: Framework for migration of timers Arun R Bharadwaj
2009-04-01 11:32 ` [v4 RFC PATCH 1/4] timers: Framework for identifying pinned timers Arun R Bharadwaj
2009-04-01 11:41   ` Andi Kleen
2009-04-02  5:09     ` Arun R Bharadwaj
2009-04-01 11:34 ` [v4 RFC PATCH 2/4] timers: Identifying the existing " Arun R Bharadwaj
2009-04-01 11:36 ` [v4 RFC PATCH 3/4] timers: /proc/sys sysctl hook to enable timer migration Arun R Bharadwaj
2009-04-01 11:37 ` [v4 RFC PATCH 4/4] timers: logic to move non pinned timers Arun R Bharadwaj
2009-04-01 11:46   ` Arun R Bharadwaj
2009-04-03 21:52   ` Thomas Gleixner [this message]
2009-04-06  5:16     ` Arun R Bharadwaj
2009-04-06 10:42       ` Arun R Bharadwaj
2009-04-06 10:56         ` Thomas Gleixner
2009-04-06 15:28           ` Arun R Bharadwaj
2009-04-06 15:31             ` Arun R Bharadwaj
2009-04-06 15:35             ` Thomas Gleixner
2009-04-06 16:00               ` Arun R Bharadwaj

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=alpine.LFD.2.00.0904032142460.12916@localhost.localdomain \
    --to=tglx@linutronix.de \
    --cc=a.p.zijlstra@chello.nl \
    --cc=andi@firstfloor.org \
    --cc=arjan@infradead.org \
    --cc=arun@linux.vnet.ibm.com \
    --cc=ego@in.ibm.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=linux-pm@lists.linux-foundation.org \
    --cc=mingo@elte.hu \
    --cc=svaidy@linux.vnet.ibm.com \
    --cc=vatsa@linux.vnet.ibm.com \
    --cc=venkatesh.pallipadi@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®