Commit 453386217a for ffmpeg

commit 453386217a3140f9cfc3ec923390104d5e814151
Author: Niklas Haas <git@haasn.dev>
Date:   Tue Sep 29 18:53:27 2026 +0200

    swscale/rational64: use 64-bit math whenever possible

    The runtime of av_q64() ops are dominated by reduce128()'s 128-bit divisions.
    We can vastly improve upon this in the common case of values that already fit
    into 64 bits, by falling back to a 64-bit reduction fast path.

    On 64-bit:
      time make fate-sws-ops-list # 3.57s -> 2.38s

    On 32-bit:
      time make fate-sws-ops-list # 49.13s -> 6.79s

    Signed-off-by: Niklas Haas <git@haasn.dev>

diff --git a/libswscale/rational64.c b/libswscale/rational64.c
index 6577a66613..d1658e2792 100644
--- a/libswscale/rational64.c
+++ b/libswscale/rational64.c
@@ -28,9 +28,36 @@

 #include <limits.h>

+#include "libavutil/attributes.h"
+#include "libavutil/common.h"
 #include "libavutil/int128.h"
+#include "libavutil/intmath.h"
 #include "rational64.h"

+static av_always_inline uint64_t high64(av_int128 a)
+{
+    return av_from128u(av_shr128(a, 64));
+}
+
+/* unsigned version of av_gcd(); see libavutil/mathematics.c */
+static av_always_inline uint64_t gcd_u64(uint64_t a, uint64_t b)
+{
+    if (!a)
+        return b;
+    if (!b)
+        return a;
+
+    const int k = ff_ctzll(a | b);
+    a >>= ff_ctzll(a);
+    do {
+        b >>= ff_ctzll(b);
+        if (a > b)
+            FFSWAP(uint64_t, a, b);
+        b -= a;
+    } while (b);
+    return a << k;
+}
+
 static av_int128 gcd128(av_int128 a, av_int128 b)
 {
     while (av_test128(b)) {
@@ -52,6 +79,27 @@ static AVRational64 reduce64(av_int128 num, av_int128 den)
     if (den_sign)
         den = av_sub128(zero, den);

+    if (!high64(num) && !high64(den)) {
+        /* Fast path: both values already fit into 64 bits */
+        uint64_t n = av_from128u(num), d = av_from128u(den);
+        const uint64_t gcd = gcd_u64(n, d);
+        if (gcd) {
+            n /= gcd;
+            d /= gcd;
+        }
+
+        if (!((n | d) >> 63)) {
+            AVRational64 res = { n, d };
+            if (num_sign ^ den_sign)
+                res.num = -res.num;
+            return res;
+        }
+
+        /* Fall back to 128-bit arithmetic for values > INT64_MAX */
+        num = av_to128u(n);
+        den = av_to128u(d);
+    }
+
     const av_int128 gcd = gcd128(num, den);
     if (av_test128(gcd)) {
         num = av_div128(num, gcd);
@@ -105,11 +153,47 @@ done:;
     return res;
 }

+static av_always_inline int fits31(int64_t x)
+{
+    return x >= -INT32_MAX && x <= INT32_MAX;
+}
+
+static av_always_inline int fits31_q(AVRational64 b, AVRational64 c)
+{
+    return fits31(b.num) && fits31(b.den) && fits31(c.num) && fits31(c.den);
+}
+
+/* Simplified version of reduce64() for small values */
+static AVRational64 reduce64_fast(int64_t num, int64_t den)
+{
+    const int sign = (num < 0) ^ (den < 0);
+    uint64_t n = num < 0 ? -(uint64_t) num : num;
+    uint64_t d = den < 0 ? -(uint64_t) den : den;
+    const uint64_t gcd = gcd_u64(n, d);
+    if (gcd) {
+        n /= gcd;
+        d /= gcd;
+    }
+
+    AVRational64 res = { n, d };
+    if (sign)
+        res.num = -res.num;
+    return res;
+}
+
+
 int ff_cmp_q64(AVRational64 a, AVRational64 b)
 {
-    const av_int128 p = av_mul128(av_to128i(a.num), av_to128i(b.den));
-    const av_int128 q = av_mul128(av_to128i(b.num), av_to128i(a.den));
-    const int test = av_cmp128(p, q);
+    int test;
+    if (fits31_q(a, b)) {
+        const int64_t p = a.num * b.den;
+        const int64_t q = b.num * a.den;
+        test = (p > q) - (p < q);
+    } else {
+        const av_int128 p = av_mul128(av_to128i(a.num), av_to128i(b.den));
+        const av_int128 q = av_mul128(av_to128i(b.num), av_to128i(a.den));
+        test = av_cmp128(p, q);
+    }

     if (test)
         return (a.den < 0) ^ (b.den < 0) ? -test : test;
@@ -123,6 +207,9 @@ int ff_cmp_q64(AVRational64 a, AVRational64 b)

 AVRational64 ff_mul_q64(AVRational64 b, AVRational64 c)
 {
+    if (fits31_q(b, c))
+        return reduce64_fast(b.num * c.num, b.den * c.den);
+
     return reduce64(av_mul128(av_to128i(b.num), av_to128i(c.num)),
                     av_mul128(av_to128i(b.den), av_to128i(c.den)));
 }
@@ -132,7 +219,11 @@ AVRational64 ff_div_q64(AVRational64 b, AVRational64 c)
     return ff_mul_q64(b, ff_inv_q64(c));
 }

-AVRational64 ff_add_q64(AVRational64 b, AVRational64 c) {
+AVRational64 ff_add_q64(AVRational64 b, AVRational64 c)
+{
+    if (fits31_q(b, c))
+        return reduce64_fast(b.num * c.den + c.num * b.den, b.den * c.den);
+
     return reduce64(av_add128(av_mul128(av_to128i(b.num), av_to128i(c.den)),
                               av_mul128(av_to128i(c.num), av_to128i(b.den))),
                     av_mul128(av_to128i(b.den), av_to128i(c.den)));
@@ -140,6 +231,9 @@ AVRational64 ff_add_q64(AVRational64 b, AVRational64 c) {

 AVRational64 ff_sub_q64(AVRational64 b, AVRational64 c)
 {
+    if (fits31_q(b, c))
+        return reduce64_fast(b.num * c.den - c.num * b.den, b.den * c.den);
+
     return reduce64(av_sub128(av_mul128(av_to128i(b.num), av_to128i(c.den)),
                               av_mul128(av_to128i(c.num), av_to128i(b.den))),
                     av_mul128(av_to128i(b.den), av_to128i(c.den)));