Skip to main content

convert_crt_input

Function convert_crt_input 

Source
fn convert_crt_input<M, N1, N2, N3>(
    t: Vec<MInt<M>>,
    capacity: usize,
) -> (Vec<MInt<N1>>, Vec<MInt<N2>>, Vec<MInt<N3>>)
Examples found in repository?
crates/competitive/src/math/number_theoretic_transform.rs (line 821)
819    fn transform(t: Self::T, len: usize) -> Self::F {
820        let npot = len.max(1).next_power_of_two();
821        let f = convert_crt_input(t, npot);
822        (
823            Convolve::<N1>::transform(f.0, npot),
824            Convolve::<N2>::transform(f.1, npot),
825            Convolve::<N3>::transform(f.2, npot),
826        )
827    }
828    fn inverse_transform(f: Self::F, len: usize) -> Self::T {
829        reconstruct_mint_crt((
830            Convolve::<N1>::inverse_transform(f.0, len),
831            Convolve::<N2>::inverse_transform(f.1, len),
832            Convolve::<N3>::inverse_transform(f.2, len),
833        ))
834    }
835    fn multiply(f: &mut Self::F, g: &Self::F) {
836        Convolve::<N1>::multiply(&mut f.0, &g.0);
837        Convolve::<N2>::multiply(&mut f.1, &g.1);
838        Convolve::<N3>::multiply(&mut f.2, &g.2);
839    }
840    fn convolve(a: Self::T, b: Self::T) -> Self::T {
841        let max_len = Self::length(&a).max(Self::length(&b));
842        let min_len = Self::length(&a).min(Self::length(&b));
843        let (balanced, short) = crate::avx_helper!(@dispatch_avx2_fma (30, 10), (384, 128));
844        if max_len <= balanced || min_len <= short {
845            return convolve_karatsuba(&a, &b);
846        }
847        // Limit coefficient growth to leave headroom for FFT roundoff.
848        let fft_limit = crate::avx_helper!(@dispatch_avx2_fma
849            1usize << ((1u64 << 50) / <M as MIntConvert<u32>>::mod_into() as u64).ilog2().min(20), 0);
850        let convolve = |a: Self::T, b: Self::T| {
851            let fft_len = (a.len() + b.len() - 1).next_power_of_two();
852            if fft_len <= 256 && a.len() * b.len() <= fft_len * 8 {
853                return convolve_karatsuba(&a, &b);
854            }
855            if fft_len <= fft_limit {
856                crate::avx_helper!(@dispatch_avx2_fma return unsafe {
857                    convolve_mint_avx2(a, b)
858                }, ());
859            }
860            convolve_mint_crt::<M, N1, N2, N3>(a, b)
861        };
862        let block_len = min_len.next_power_of_two() * 8 - min_len + 1;
863        let block_len = if min_len <= fft_limit / 2 {
864            block_len.min(fft_limit - min_len + 1)
865        } else {
866            block_len
867        };
868        if max_len <= block_len {
869            return convolve(a, b);
870        }
871        let (a, b) = if a.len() >= b.len() { (a, b) } else { (b, a) };
872        let mut result = vec![MInt::<M>::zero(); a.len() + b.len() - 1];
873        for (i, a) in a.chunks(block_len).enumerate() {
874            let product = convolve(a.to_vec(), b.clone());
875            for (value, product) in result[i * block_len..].iter_mut().zip(product) {
876                *value += product;
877            }
878        }
879        result
880    }
881}
882
883fn convolve_mint_crt<M, N1, N2, N3>(a: MVec<M>, b: MVec<M>) -> MVec<M>
884where
885    M: MIntConvert + MIntConvert<u32>,
886    N1: Montgomery32NttModulus,
887    N2: Montgomery32NttModulus,
888    N3: Montgomery32NttModulus,
889{
890    let convolve = |a: MVec<M>, b: MVec<M>| {
891        let a_len = a.len();
892        let b_len = b.len();
893        let a = convert_crt_input(a, a_len);
894        let b = convert_crt_input(b, b_len);
895        reconstruct_mint_crt((
896            Convolve::<N1>::convolve(a.0, b.0),
897            Convolve::<N2>::convolve(a.1, b.1),
898            Convolve::<N3>::convolve(a.2, b.2),
899        ))
900    };
901    let modulus = <M as MIntConvert<u32>>::mod_into() as u128;
902    let capacity = N1::MOD as u128 * N2::MOD as u128 * N3::MOD as u128;
903    if a.len().min(b.len()) as u128 * (modulus - 1).pow(2) < capacity {
904        return convolve(a, b);
905    }
906    let block_len = ((capacity - 1) / (modulus - 1).pow(2)) as usize;
907    if block_len == 0 {
908        return convolve_naive(&a, &b);
909    }
910    let mut result = vec![MInt::<M>::zero(); a.len() + b.len() - 1];
911    for (i, a) in a.chunks(block_len).enumerate() {
912        for (j, b) in b.chunks(block_len).enumerate() {
913            let product = convolve(a.to_vec(), b.to_vec());
914            for (value, product) in result[(i + j) * block_len..].iter_mut().zip(product) {
915                *value += product;
916            }
917        }
918    }
919    result
920}
921
922impl<N1, N2, N3> ConvolveSteps for Convolve<(u64, (N1, N2, N3))>
923where
924    N1: Montgomery32NttModulus,
925    N2: Montgomery32NttModulus,
926    N3: Montgomery32NttModulus,
927{
928    type T = Vec<u64>;
929    type F = ([MVec<N1>; 3], [MVec<N2>; 3], [MVec<N3>; 3]);
930
931    fn length(t: &Self::T) -> usize {
932        t.len()
933    }
934
935    fn transform(t: Self::T, len: usize) -> Self::F {
936        let npot = len.max(1).next_power_of_two();
937        assert!(npot <= 1usize << N1::RANK.min(N2::RANK).min(N3::RANK));
938        // The 22-bit fallback needs room for three limb products per coefficient.
939        assert!(
940            3 * npot as u128 * ((1u128 << 22) - 1).pow(2)
941                < N1::MOD as u128 * N2::MOD as u128 * N3::MOD as u128
942        );
943        let bits = if 2 * npot as u128 * (u32::MAX as u128).pow(2)
944            < N1::MOD as u128 * N2::MOD as u128 * N3::MOD as u128
945        {
946            32
947        } else {
948            22
949        };
950        let parts = if bits == 32 && t.iter().all(|&value| value <= u32::MAX as u64) {
951            1
952        } else {
953            64usize.div_ceil(bits)
954        };
955        fn split<M: Montgomery32NttModulus>(
956            t: &[u64],
957            len: usize,
958            bits: usize,
959            parts: usize,
960        ) -> [MVec<M>; 3] {
961            std::array::from_fn(|part| {
962                if part >= parts {
963                    return Vec::new();
964                }
965                Convolve::<M>::transform(
966                    t.iter()
967                        .map(|&t| MInt::from((t >> (part * bits)) & ((1u64 << bits) - 1)))
968                        .collect(),
969                    len,
970                )
971            })
972        }
973        (
974            split(&t, npot, bits, parts),
975            split(&t, npot, bits, parts),
976            split(&t, npot, bits, parts),
977        )
978    }
979
980    fn inverse_transform(f: Self::F, len: usize) -> Self::T {
981        let bits = if f.0[2].is_empty() { 32 } else { 22 };
982        let t1 = MInt::<N2>::new(N1::get_mod()).inv();
983        let m1 = N1::get_mod() as u64;
984        let m1_3 = MInt::<N3>::new(N1::get_mod());
985        let t2 = (m1_3 * MInt::<N3>::new(N2::get_mod())).inv();
986        let m2 = m1 * N2::get_mod() as u64;
987        let mut result = vec![0u64; len.min(f.0[0].len())];
988        for (part, ((f1, f2), f3)) in f.0.into_iter().zip(f.1).zip(f.2).enumerate() {
989            if f1.is_empty() {
990                continue;
991            }
992            for (value, ((c1, c2), c3)) in result.iter_mut().zip(
993                Convolve::<N1>::inverse_transform(f1, len)
994                    .into_iter()
995                    .zip(Convolve::<N2>::inverse_transform(f2, len))
996                    .zip(Convolve::<N3>::inverse_transform(f3, len)),
997            ) {
998                let d1 = c1.inner();
999                let d2 = ((c2 - MInt::<N2>::from(d1)) * t1).inner();
1000                let x = MInt::<N3>::new(d1) + MInt::<N3>::new(d2) * m1_3;
1001                let d3 = ((c3 - x) * t2).inner();
1002                let limb = (d1 as u64)
1003                    .wrapping_add((d2 as u64).wrapping_mul(m1))
1004                    .wrapping_add((d3 as u64).wrapping_mul(m2));
1005                *value = value.wrapping_add(limb << (part * bits));
1006            }
1007        }
1008        result
1009    }
1010
1011    fn multiply(f: &mut Self::F, g: &Self::F) {
1012        fn multiply<M: Montgomery32NttModulus>(f: &mut [MVec<M>; 3], g: &[MVec<M>; 3]) {
1013            assert_eq!(f[0].len(), g[0].len());
1014            if f[1].is_empty() || g[1].is_empty() {
1015                if f[1].is_empty() && !g[1].is_empty() {
1016                    f[1] = f[0].clone();
1017                    Convolve::<M>::multiply(&mut f[1], &g[1]);
1018                } else if !f[1].is_empty() {
1019                    Convolve::<M>::multiply(&mut f[1], &g[0]);
1020                }
1021                Convolve::<M>::multiply(&mut f[0], &g[0]);
1022                return;
1023            }
1024            #[cfg(target_arch = "x86_64")]
1025            if use_block_ntt::<M>(f[0].len()) {
1026                for part in (1..if f[2].is_empty() { 2 } else { 3 }).rev() {
1027                    let mut sum = f[0].clone();
1028                    Convolve::<M>::multiply(&mut sum, &g[part]);
1029                    for left in 1..=part {
1030                        let mut product = f[left].clone();
1031                        Convolve::<M>::multiply(&mut product, &g[part - left]);
1032                        for (value, product) in sum.iter_mut().zip(product) {
1033                            // Block products contain lazy Montgomery residues.
1034                            *value = MInt::new(value.inner() + product.inner());
1035                        }
1036                    }
1037                    f[part] = sum;
1038                }
1039                Convolve::<M>::multiply(&mut f[0], &g[0]);
1040                return;
1041            }
1042            if f[2].is_empty() {
1043                for i in 0..f[0].len() {
1044                    f[1][i] = f[0][i] * g[1][i] + f[1][i] * g[0][i];
1045                    f[0][i] *= g[0][i];
1046                }
1047                return;
1048            }
1049            for i in 0..f[0].len() {
1050                f[2][i] = f[0][i] * g[2][i] + f[1][i] * g[1][i] + f[2][i] * g[0][i];
1051                f[1][i] = f[0][i] * g[1][i] + f[1][i] * g[0][i];
1052                f[0][i] *= g[0][i];
1053            }
1054        }
1055        multiply(&mut f.0, &g.0);
1056        multiply(&mut f.1, &g.1);
1057        multiply(&mut f.2, &g.2);
1058    }
1059
1060    fn square(t: Self::T, len: usize) -> Self::T {
1061        let mut f = Self::transform(t, len);
1062        let g = f.clone();
1063        Self::multiply(&mut f, &g);
1064        Self::inverse_transform(f, len)
1065    }
1066
1067    fn convolve(a: Self::T, b: Self::T) -> Self::T {
1068        let max_len = Self::length(&a).max(Self::length(&b));
1069        let min_len = Self::length(&a).min(Self::length(&b));
1070        let (balanced, short) = crate::avx_helper!(@dispatch_avx2_fma (300, 64), (1536, 512));
1071        if max_len <= balanced || min_len <= short {
1072            let a_wrapping: &[Wrapping<u64>] =
1073                unsafe { std::slice::from_raw_parts(a.as_ptr().cast(), a.len()) };
1074            let b_wrapping: &[Wrapping<u64>] =
1075                unsafe { std::slice::from_raw_parts(b.as_ptr().cast(), b.len()) };
1076            let mut c = std::mem::ManuallyDrop::new(if max_len <= 300 || min_len > 60 {
1077                convolve_karatsuba(a_wrapping, b_wrapping)
1078            } else {
1079                convolve_naive(a_wrapping, b_wrapping)
1080            });
1081            return unsafe { Vec::from_raw_parts(c.as_mut_ptr().cast(), c.len(), c.capacity()) };
1082        }
1083        let len = (Self::length(&a) + Self::length(&b)).saturating_sub(1);
1084        let block_len = if min_len >= 1 << 20 {
1085            1 << 20
1086        } else {
1087            (min_len.next_power_of_two() * 8).min(1 << 21) - min_len + 1
1088        };
1089        if max_len <= block_len {
1090            return convolve_u64_fft(a, b);
1091        }
1092        let mut result = vec![0u64; len];
1093        for (i, a) in a.chunks(block_len).enumerate() {
1094            for (j, b) in b.chunks(block_len).enumerate() {
1095                if a.len().min(b.len()) <= 60 {
1096                    for (x, &a) in a.iter().enumerate() {
1097                        for (y, &b) in b.iter().enumerate() {
1098                            let value = &mut result[(i + j) * block_len + x + y];
1099                            *value = value.wrapping_add(a.wrapping_mul(b));
1100                        }
1101                    }
1102                    continue;
1103                }
1104                let product = convolve_u64_fft(a.to_vec(), b.to_vec());
1105                for (value, product) in result[(i + j) * block_len..].iter_mut().zip(product) {
1106                    *value = value.wrapping_add(product);
1107                }
1108            }
1109        }
1110        result
1111    }
1112}
1113
1114fn convolve_u64_fft(a: Vec<u64>, b: Vec<u64>) -> Vec<u64> {
1115    // Keep limb convolutions below 2^47 at the 2^21 FFT limit.
1116    crate::avx_helper!(@dispatch_avx2_fma return unsafe {
1117        convolve_u64_avx2(a, b)
1118    }, ());
1119    convolve_u64_fft_scalar(a, b)
1120}
1121
1122fn convolve_u64_fft_scalar(a: Vec<u64>, b: Vec<u64>) -> Vec<u64> {
1123    fn split(values: &[u64]) -> [Vec<i64>; 5] {
1124        let mut result = std::array::from_fn(|_| Vec::with_capacity(values.len()));
1125        for mut value in values.iter().copied() {
1126            for part in &mut result {
1127                let digit = ((value << 51) as i64) >> 51;
1128                part.push(digit);
1129                value = (value >> 13).wrapping_add(u64::from(digit < 0));
1130            }
1131        }
1132        result
1133    }
1134
1135    let len = a.len() + b.len() - 1;
1136    let transform = |values: &[u64]| {
1137        if values.iter().any(|&value| value > u32::MAX as u64) {
1138            return split(values).map(|part| ConvolveRealFft::transform(part, len));
1139        }
1140        let [a, b, c, _, _] = split(values);
1141        let a = ConvolveRealFft::transform(a, len);
1142        let size = a.len();
1143        [
1144            a,
1145            ConvolveRealFft::transform(b, len),
1146            ConvolveRealFft::transform(c, len),
1147            vec![Zero::zero(); size],
1148            vec![Zero::zero(); size],
1149        ]
1150    };
1151    let fa = transform(&a);
1152    drop(a);
1153    let fb = transform(&b);
1154    drop(b);
1155    let values: [Vec<i64>; 5] = std::array::from_fn(|part| {
1156        let mut sum = fa[0].clone();
1157        ConvolveRealFft::multiply(&mut sum, &fb[part]);
1158        for left in 1..=part {
1159            let mut product = fa[left].clone();
1160            ConvolveRealFft::multiply(&mut product, &fb[part - left]);
1161            for (sum, product) in sum.iter_mut().zip(product) {
1162                *sum += product;
1163            }
1164        }
1165        ConvolveRealFft::inverse_transform(sum, len)
1166    });
1167    (0..len)
1168        .map(|i| {
1169            (values[0][i] as u64)
1170                .wrapping_add((values[1][i] as u64) << 13)
1171                .wrapping_add((values[2][i] as u64) << 26)
1172                .wrapping_add((values[3][i] as u64) << 39)
1173                .wrapping_add((values[4][i] as u64) << 52)
1174        })
1175        .collect()
1176}
1177
1178pub trait NttReuse: ConvolveSteps {
1179    const MULTIPLE: bool = true;
1180
1181    /// Transforms coefficients into the usual NTT frequency order.
1182    fn transform_ntt(t: Self::T, len: usize) -> Self::F {
1183        Self::transform(t, len)
1184    }
1185
1186    /// Inverts a value produced by `transform_ntt`.
1187    fn inverse_transform_ntt(f: Self::F, len: usize) -> Self::T {
1188        Self::inverse_transform(f, len)
1189    }
1190
1191    /// Extends a value produced by `transform_ntt` to twice its length.
1192    /// If `monic`, the input represents a monic degree-`n` polynomial modulo
1193    /// `x^n - 1`, where `n` is the transform length.
1194    fn ntt_doubling(f: Self::F, monic: bool) -> Self::F;
1195
1196    /// Extracts the even coefficients of `a(x) * b(-x)` in the usual NTT frequency order.
1197    fn even_mul_normal_neg(f: &Self::F, g: &Self::F) -> Self::F;
1198
1199    /// Extracts the odd coefficients of `a(x) * b(-x)` in the usual NTT frequency order.
1200    fn odd_mul_normal_neg(f: &Self::F, g: &Self::F) -> Self::F;
1201
1202    /// Multiplies a usual NTT transform by the corresponding prefix of another one.
1203    fn multiply_prefix(f: &mut Self::F, g: &Self::F);
1204
1205    /// Adds the pointwise product of two usual NTT transforms to `sum`.
1206    fn multiply_add(sum: &mut Self::F, f: &Self::F, g: &Self::F);
1207
1208    /// Maximum number of products that can be summed before reconstruction.
1209    /// Both factors must transform canonical coefficients at the supplied transform's length,
1210    /// and each cyclic product must itself be reconstructible.
1211    fn max_product_sum_count(_f: &Self::F) -> usize {
1212        if Self::MULTIPLE { 1 } else { usize::MAX }
1213    }
1214
1215    fn power_projection_step(
1216        p_flat: Self::T,
1217        q_flat: Self::T,
1218        n: usize,
1219        py: usize,
1220        qy: usize,
1221    ) -> (Self::T, Self::T) {
1222        let base = n * 2;
1223        let len_p = base * py;
1224        let len_q = base * qy;
1225        let len = (len_p + len_q - 1).max(len_q + len_q - 1);
1226        let half = len.max(1).next_power_of_two() / 2;
1227
1228        let p_fft = Self::transform_ntt(p_flat, len);
1229        let q_fft = Self::transform_ntt(q_flat, len);
1230        let pr_fft = Self::odd_mul_normal_neg(&p_fft, &q_fft);
1231        let qr_fft = Self::even_mul_normal_neg(&q_fft, &q_fft);
1232        (
1233            Self::inverse_transform_ntt(pr_fft, half),
1234            Self::inverse_transform_ntt(qr_fft, half),
1235        )
1236    }
1237}
1238
1239thread_local!(
1240    static BIT_REVERSE: UnsafeCell<Vec<Vec<usize>>> = const { UnsafeCell::new(vec![]) };
1241);
1242
1243impl<M> NttReuse for Convolve<M>
1244where
1245    M: Montgomery32NttModulus,
1246{
1247    const MULTIPLE: bool = false;
1248
1249    fn transform_ntt(mut t: Self::T, len: usize) -> Self::F {
1250        t.resize_with(len.max(1).next_power_of_two(), Zero::zero);
1251        ntt(&mut t);
1252        t
1253    }
1254
1255    fn inverse_transform_ntt(mut f: Self::F, len: usize) -> Self::T {
1256        intt(&mut f);
1257        f.truncate(len);
1258        f
1259    }
1260
1261    fn ntt_doubling(mut f: Self::F, monic: bool) -> Self::F {
1262        let n = f.len();
1263        let k = n.trailing_zeros() as usize;
1264        let mut a = Self::inverse_transform_ntt(f.clone(), n);
1265        if monic {
1266            a[0] -= MInt::<M>::from(2);
1267        }
1268        let zeta = MInt::<M>::new_unchecked(M::INFO.root[k + 1]);
1269        let zeta2 = zeta * zeta;
1270        let mut rot = [MInt::one(), zeta, zeta2, zeta2 * zeta];
1271        let step = zeta2 * zeta2;
1272        for a in a.chunks_mut(4) {
1273            for (a, rot) in a.iter_mut().zip(&mut rot) {
1274                *a *= *rot;
1275                *rot *= step;
1276            }
1277        }
1278        f.extend(Self::transform_ntt(a, n));
1279        f
1280    }
1281
1282    fn even_mul_normal_neg(f: &Self::F, g: &Self::F) -> Self::F {
1283        assert_eq!(f.len(), g.len());
1284        assert!(f.len().is_power_of_two());
1285        assert!(f.len() >= 2);
1286        if std::ptr::eq(f, g) {
1287            return f.as_chunks::<2>().0.iter().map(|a| a[0] * a[1]).collect();
1288        }
1289        let inv2 = MInt::<M>::from(2).inv();
1290        let n = f.len() / 2;
1291        (0..n)
1292            .map(|i| (f[i << 1] * g[i << 1 | 1] + f[i << 1 | 1] * g[i << 1]) * inv2)
1293            .collect()
1294    }
1295
1296    fn odd_mul_normal_neg(f: &Self::F, g: &Self::F) -> Self::F {
1297        assert_eq!(f.len(), g.len());
1298        assert!(f.len().is_power_of_two());
1299        assert!(f.len() >= 2);
1300        let mut inv2 = MInt::<M>::from(2).inv();
1301        let n = f.len() / 2;
1302        let k = f.len().trailing_zeros() as usize;
1303        let mut h = vec![MInt::<M>::zero(); n];
1304        let w = MInt::<M>::new_unchecked(M::INFO.inv_root[k]);
1305        BIT_REVERSE.with(|br| {
1306            let br = unsafe { &mut *br.get() };
1307            if br.len() < k {
1308                br.resize_with(k, Default::default);
1309            }
1310            let k = k - 1;
1311            if br[k].is_empty() {
1312                let mut v = vec![0; 1 << k];
1313                for i in 0..1 << k {
1314                    v[i] = (v[i >> 1] >> 1) | ((i & 1) << k.saturating_sub(1));
1315                }
1316                br[k] = v;
1317            }
1318            for &i in &br[k] {
1319                h[i] = (f[i << 1] * g[i << 1 | 1] - f[i << 1 | 1] * g[i << 1]) * inv2;
1320                inv2 *= w;
1321            }
1322        });
1323        h
1324    }
1325
1326    fn multiply_prefix(f: &mut Self::F, g: &Self::F) {
1327        pointwise_multiply(f, g);
1328    }
1329
1330    fn multiply_add(sum: &mut Self::F, f: &Self::F, g: &Self::F) {
1331        assert!(sum.len() == f.len() && sum.len() == g.len());
1332        pointwise_multiply_add(sum, f, g);
1333    }
1334
1335    fn power_projection_step(
1336        p_flat: Vec<MInt<M>>,
1337        q_flat: Vec<MInt<M>>,
1338        n: usize,
1339        py: usize,
1340        qy: usize,
1341    ) -> (Vec<MInt<M>>, Vec<MInt<M>>) {
1342        let high_degree = (qy - 1) * 2;
1343        let rows = (py + qy - 1).max(high_degree).next_power_of_two();
1344        let cols = n * 2;
1345        let size = rows * cols;
1346        let mut p = p_flat;
1347        p.resize_with(size, MInt::<M>::zero);
1348        ntt_rows(&mut p, cols);
1349        ntt_batch(&mut p, cols);
1350
1351        let mut q = q_flat;
1352        q.resize_with(size, MInt::<M>::zero);
1353        ntt_rows(&mut q, cols);
1354        let q_high = (rows == high_degree).then(|| q[(qy - 1) * cols..qy * cols].to_vec());
1355        ntt_batch(&mut q, cols);
1356
1357        let half = cols / 2;
1358        let mut odd_factor = vec![MInt::<M>::zero(); half];
1359        let mut factor = MInt::<M>::from(2).inv();
1360        let k = cols.trailing_zeros() as usize;
1361        let w = MInt::<M>::new_unchecked(M::INFO.inv_root[k]);
1362        BIT_REVERSE.with(|br| {
1363            let br = unsafe { &mut *br.get() };
1364            if br.len() < k {
1365                br.resize_with(k, Default::default);
1366            }
1367            let k = k - 1;
1368            if br[k].is_empty() {
1369                let mut v = vec![0; 1 << k];
1370                for i in 0..1 << k {
1371                    v[i] = (v[i >> 1] >> 1) | ((i & 1) << k.saturating_sub(1));
1372                }
1373                br[k] = v;
1374            }
1375            for &i in &br[k] {
1376                odd_factor[i] = factor;
1377                factor *= w;
1378            }
1379        });
1380
1381        let mut pr = vec![MInt::<M>::zero(); rows * half];
1382        let mut qr = vec![MInt::<M>::zero(); rows * half];
1383        for i in 0..pr.len() {
1384            pr[i] = (p[i << 1] * q[i << 1 | 1] - p[i << 1 | 1] * q[i << 1])
1385                * odd_factor[i & (half - 1)];
1386            qr[i] = q[i << 1] * q[i << 1 | 1];
1387        }
1388        intt_batch(&mut pr, half);
1389        intt_rows(&mut pr, half);
1390        intt_batch(&mut qr, half);
1391        intt_rows(&mut qr, half);
1392
1393        if let Some(q_high) = q_high {
1394            let mut q_high_even = vec![MInt::<M>::zero(); half];
1395            for i in 0..half {
1396                q_high_even[i] = q_high[i << 1] * q_high[i << 1 | 1];
1397            }
1398            intt(&mut q_high_even);
1399            for (value, high) in qr.iter_mut().zip(&q_high_even) {
1400                *value -= *high;
1401            }
1402            qr.extend_from_slice(&q_high_even);
1403        }
1404        (pr, qr)
1405    }
1406}
1407
1408impl<M, N1, N2, N3> NttReuse for Convolve<(M, (N1, N2, N3))>
1409where
1410    M: MIntConvert + MIntConvert<u32>,
1411    N1: Montgomery32NttModulus,
1412    N2: Montgomery32NttModulus,
1413    N3: Montgomery32NttModulus,
1414{
1415    fn max_product_sum_count(f: &Self::F) -> usize {
1416        let modulus = <M as MIntConvert<u32>>::mod_into() as u128;
1417        if modulus == 1 {
1418            return usize::MAX;
1419        }
1420        let capacity = N1::MOD as u128 * N2::MOD as u128 * N3::MOD as u128;
1421        ((capacity - 1) / ((modulus - 1) * (modulus - 1)) / f.0.len() as u128)
1422            .clamp(1, usize::MAX as u128) as usize
1423    }
1424
1425    fn transform_ntt(t: Self::T, len: usize) -> Self::F {
1426        let npot = len.max(1).next_power_of_two();
1427        let f = convert_crt_input(t, npot);
1428        (
1429            Convolve::<N1>::transform_ntt(f.0, npot),
1430            Convolve::<N2>::transform_ntt(f.1, npot),
1431            Convolve::<N3>::transform_ntt(f.2, npot),
1432        )
1433    }