]> git.hungrycats.org Git - linux/commitdiff
ia64: Avoid intermediate-overflows in sched_clock().
authorDavid Mosberger <davidm@tiger.hpl.hp.com>
Fri, 4 Jun 2004 08:44:11 +0000 (01:44 -0700)
committerDavid Mosberger <davidm@tiger.hpl.hp.com>
Fri, 4 Jun 2004 08:44:11 +0000 (01:44 -0700)
Bug reported by Zoran Menyhart.

arch/ia64/kernel/head.S
arch/ia64/kernel/setup.c
arch/ia64/kernel/smpboot.c
arch/ia64/kernel/time.c

index 6f74b9a1292d0209483e64f1d60fa1282b5acb20..0a5eb48d5de91a432b3b05ba09d14b030c031d28 100644 (file)
@@ -815,6 +815,42 @@ GLOBAL_ENTRY(ia64_delay_loop)
        br.ret.sptk.many rp
 END(ia64_delay_loop)
 
+/*
+ * Return a CPU-local timestamp in nano-seconds.  This timestamp is
+ * NOT synchronized across CPUs its return value must never be
+ * compared against the values returned on another CPU.  The usage in
+ * kernel/sched.c ensures that.
+ *
+ * The return-value of sched_clock() is NOT supposed to wrap-around.
+ * If it did, it would cause some scheduling hiccups (at the worst).
+ * Fortunately, with a 64-bit cycle-counter ticking at 100GHz, even
+ * that would happen only once every 5+ years.
+ *
+ * The code below basically calculates:
+ *
+ *   (ia64_get_itc() * local_cpu_data->nsec_per_cyc) >> IA64_NSEC_PER_CYC_SHIFT
+ *
+ * except that the multiplication and the shift are done with 128-bit
+ * intermediate precision so that we can produce a full 64-bit result.
+ */
+GLOBAL_ENTRY(sched_clock)
+       addl r8=THIS_CPU(cpu_info) + IA64_CPUINFO_NSEC_PER_CYC_OFFSET,r0
+       mov.m r9=ar.itc         // fetch cycle-counter                          (35 cyc)
+       ;;
+       ldf8 f8=[r8]
+       ;;
+       setf.sig f9=r9          // certain to stall, so issue it _after_ ldf8...
+       ;;
+       xmpy.lu f10=f9,f8       // calculate low 64 bits of 128-bit product     (4 cyc)
+       xmpy.hu f11=f9,f8       // calculate high 64 bits of 128-bit product
+       ;;
+       getf.sig r8=f10         //                                              (5 cyc)
+       getf.sig r9=f11
+       ;;
+       shrp r8=r9,r8,IA64_NSEC_PER_CYC_SHIFT
+       br.ret.sptk.many rp
+END(sched_clock)
+
 GLOBAL_ENTRY(start_kernel_thread)
        .prologue
        .save rp, r0                            // this is the end of the call-chain
index 26eb18052e44538d44e5594340e220a6c816b66b..41177a9e9ac06749db9ab450f6999dfbee7a2301 100644 (file)
@@ -635,6 +635,9 @@ cpu_init (void)
        ia32_cpu_init();
 #endif
 
+       /* Clear ITC to eliminiate sched_clock() overflows in human time.  */
+       ia64_set_itc(0);
+
        /* disable all local interrupt sources: */
        ia64_set_itv(1 << 16);
        ia64_set_lrr0(1 << 16);
index 8058fb5881f508ab900bd40f2252569d8ab5b787..97541e8998939f8cd92d1f244a174ba7d92db570 100644 (file)
@@ -202,7 +202,6 @@ ia64_sync_itc (unsigned int master)
 {
        long i, delta, adj, adjust_latency = 0, done = 0;
        unsigned long flags, rt, master_time_stamp, bound;
-       extern void ia64_cpu_local_tick (void);
 #if DEBUG_ITC_SYNC
        struct {
                long rt;        /* roundtrip time */
index 3ebc74a5dc8661d9c7e9048c259145ac86ffa23f..e33bcb6610ef5634144064eec663dd526d99e7f3 100644 (file)
@@ -45,14 +45,6 @@ EXPORT_SYMBOL(last_cli_ip);
 
 #endif
 
-unsigned long long
-sched_clock (void)
-{
-       unsigned long offset = ia64_get_itc();
-
-       return (offset * local_cpu_data->nsec_per_cyc) >> IA64_NSEC_PER_CYC_SHIFT;
-}
-
 static void
 itc_reset (void)
 {