]> www.infradead.org Git - users/hch/dma-mapping.git/commitdiff
x86/tsc: Add option to force frequency recalibration with HW timer
authorFeng Tang <feng.tang@intel.com>
Wed, 4 Jan 2023 08:19:38 +0000 (16:19 +0800)
committerPaul E. McKenney <paulmck@kernel.org>
Thu, 2 Feb 2023 22:22:52 +0000 (14:22 -0800)
The kernel assumes that the TSC frequency which is provided by the
hardware / firmware via MSRs or CPUID(0x15) is correct after applying
a few basic consistency checks. This disables the TSC recalibration
against HPET or PM timer.

As a result there is no mechanism to validate that frequency in cases
where a firmware or hardware defect is suspected. And there was case
that some user used atomic clock to measure the TSC frequency and
reported an inaccuracy issue, which was later fixed in firmware.

Add an option 'recalibrate' for 'tsc' kernel parameter to force the
tsc freq recalibration with HPET or PM timer, and warn if the
deviation from previous value is more than about 500 PPM, which
provides a way to verify the data from hardware / firmware.

There is no functional change to existing work flow.

Recently there was a real-world case: "The 40ms/s divergence between
TSC and HPET was observed on hardware that is quite recent" [1], on
that platform the TSC frequence 1896 MHz was got from CPUID(0x15),
and the force-reclibration with HPET/PMTIMER both calibrated out
value of 1975 MHz, which also matched with check from software
'chronyd', indicating it's a problem of BIOS or firmware.

[Thanks tglx for helping improving the commit log]
[ paulmck: Wordsmith Kconfig help text. ]

[1]. https://lore.kernel.org/lkml/20221117230910.GI4001@paulmck-ThinkPad-P17-Gen-1/
Signed-off-by: Feng Tang <feng.tang@intel.com>
Cc: Thomas Gleixner <tglx@linutronix.de>
Cc: Ingo Molnar <mingo@redhat.com>
Cc: Borislav Petkov <bp@alien8.de>
Cc: Dave Hansen <dave.hansen@linux.intel.com>
Cc: "H. Peter Anvin" <hpa@zytor.com>
Cc: Jonathan Corbet <corbet@lwn.net>
Cc: <x86@kernel.org>
Cc: <linux-doc@vger.kernel.org>
Signed-off-by: Paul E. McKenney <paulmck@kernel.org>
Documentation/admin-guide/kernel-parameters.txt
arch/x86/kernel/tsc.c

index 6cfa6e3996cf75ee6bb1e8b399daa4d5701ab92b..95f0d104c2322d3ea41d0561be95dc48f3e7b7eb 100644 (file)
                        in situations with strict latency requirements (where
                        interruptions from clocksource watchdog are not
                        acceptable).
+                       [x86] recalibrate: force recalibration against a HW timer
+                       (HPET or PM timer) on systems whose TSC frequency was
+                       obtained from HW or FW using either an MSR or CPUID(0x15).
+                       Warn if the difference is more than 500 ppm.
 
        tsc_early_khz=  [X86] Skip early TSC calibration and use the given
                        value instead. Useful when the early TSC frequency discovery
index a78e73da4a74b351bf82407babe079b0b057fbc1..92bbc4a6b3fc326203f78bdb532368e5cc9654eb 100644 (file)
@@ -48,6 +48,8 @@ static DEFINE_STATIC_KEY_FALSE(__use_tsc);
 
 int tsc_clocksource_reliable;
 
+static int __read_mostly tsc_force_recalibrate;
+
 static u32 art_to_tsc_numerator;
 static u32 art_to_tsc_denominator;
 static u64 art_to_tsc_offset;
@@ -303,6 +305,8 @@ static int __init tsc_setup(char *str)
                mark_tsc_unstable("boot parameter");
        if (!strcmp(str, "nowatchdog"))
                no_tsc_watchdog = 1;
+       if (!strcmp(str, "recalibrate"))
+               tsc_force_recalibrate = 1;
        return 1;
 }
 
@@ -1374,6 +1378,25 @@ restart:
        else
                freq = calc_pmtimer_ref(delta, ref_start, ref_stop);
 
+       /* Will hit this only if tsc_force_recalibrate has been set */
+       if (boot_cpu_has(X86_FEATURE_TSC_KNOWN_FREQ)) {
+
+               /* Warn if the deviation exceeds 500 ppm */
+               if (abs(tsc_khz - freq) > (tsc_khz >> 11)) {
+                       pr_warn("Warning: TSC freq calibrated by CPUID/MSR differs from what is calibrated by HW timer, please check with vendor!!\n");
+                       pr_info("Previous calibrated TSC freq:\t %lu.%03lu MHz\n",
+                               (unsigned long)tsc_khz / 1000,
+                               (unsigned long)tsc_khz % 1000);
+               }
+
+               pr_info("TSC freq recalibrated by [%s]:\t %lu.%03lu MHz\n",
+                       hpet ? "HPET" : "PM_TIMER",
+                       (unsigned long)freq / 1000,
+                       (unsigned long)freq % 1000);
+
+               return;
+       }
+
        /* Make sure we're within 1% */
        if (abs(tsc_khz - freq) > tsc_khz/100)
                goto out;
@@ -1407,8 +1430,10 @@ static int __init init_tsc_clocksource(void)
        if (!boot_cpu_has(X86_FEATURE_TSC) || !tsc_khz)
                return 0;
 
-       if (tsc_unstable)
-               goto unreg;
+       if (tsc_unstable) {
+               clocksource_unregister(&clocksource_tsc_early);
+               return 0;
+       }
 
        if (boot_cpu_has(X86_FEATURE_NONSTOP_TSC_S3))
                clocksource_tsc.flags |= CLOCK_SOURCE_SUSPEND_NONSTOP;
@@ -1421,9 +1446,10 @@ static int __init init_tsc_clocksource(void)
                if (boot_cpu_has(X86_FEATURE_ART))
                        art_related_clocksource = &clocksource_tsc;
                clocksource_register_khz(&clocksource_tsc, tsc_khz);
-unreg:
                clocksource_unregister(&clocksource_tsc_early);
-               return 0;
+
+               if (!tsc_force_recalibrate)
+                       return 0;
        }
 
        schedule_delayed_work(&tsc_irqwork, 0);