Skip to content

large_factorial.hpp

SECTIONMath INCLUDEnoya/large_factorial.hpp

Compute several factorials modulo a fixed prime without a linear table. Split 1..N into blocks of length B about sqrt(N). The product inside one block is the degree-B polynomial P(x)=prod_{i=1}^B(x+i); a product tree constructs P and multipoint evaluation obtains P(0),P(B),P(2B),.... Prefix products answer every full block, followed by at most B direct factors.

Verified by factorial, many_factorials.

\[ \displaystyle n!=\prod_{i=1}^{n}i \]

Implementation

View on GitHub

#ifndef NOYA_LARGE_FACTORIAL_HPP
#define NOYA_LARGE_FACTORIAL_HPP 1

/// @complexity Time: `large_factorials` uses
/// O(M(sqrt N) log N + T sqrt N); `many_factorials`, with B=2^15, uses
/// O(M(B) log B + M(N/B) log(N/B) +
/// sum_b ceil(Q_b/2^b) M(2^b) log(2^b)), where Q_b is the number of queries
/// whose remainder has bit b set.
/// Space: O(sqrt N log N) and O(B log B), respectively.

#include "noya/combinatorial_sequences.hpp"
#include "noya/polynomial_multipoint.hpp"
#include "noya/polynomial_special_points.hpp"

#include <algorithm>
#include <cassert>
#include <cmath>
#include <cstdint>
#include <vector>

namespace noya {

namespace large_factorial_detail {

template <class Mint>
std::vector<Mint> factorial_block_prefix(std::uint64_t maximum, int block) {
  int full_blocks = int(maximum / block);
  std::vector<Mint> prefix(full_blocks + 1, Mint(1));
  if (full_blocks == 0) {
    return prefix;
  }

  std::vector<std::vector<Mint>> factors;
  factors.reserve(block);
  for (int i = 1; i <= block; i++) {
    factors.push_back({Mint(i), Mint(1)});
  }
  std::vector<Mint> block_polynomial =
      polynomial_product_sequence(std::move(factors));
  std::vector<Mint> points(full_blocks);
  for (int i = 0; i < full_blocks; i++) {
    points[i] = Mint(std::uint64_t(i) * block);
  }
  std::vector<Mint> products =
      polynomial_multipoint_evaluation(block_polynomial, points);
  for (int i = 0; i < full_blocks; i++) {
    prefix[i + 1] = prefix[i] * products[i];
  }
  return prefix;
}

} // namespace large_factorial_detail

/// @brief Compute several factorials modulo a fixed prime without a linear
/// table. Split 1..N into blocks of length B about sqrt(N). The product inside
/// one block is the degree-B polynomial P(x)=prod_{i=1}^B(x+i); a product tree
/// constructs P and multipoint evaluation obtains P(0),P(B),P(2B),.... Prefix
/// products answer every full block, followed by at most B direct factors.
template <class Mint>
std::vector<Mint>
large_factorials(const std::vector<std::uint64_t> &queries) {
  if (queries.empty()) {
    return {};
  }
  std::uint64_t maximum =
      *std::max_element(queries.begin(), queries.end());
  assert(maximum < std::uint64_t(Mint::mod()));
  int block = int(std::sqrt(static_cast<long double>(maximum + 1)));
  block = std::max(block, 1);
  while (std::uint64_t(block) * block < maximum + 1) {
    block++;
  }
  std::vector<Mint> prefix =
      large_factorial_detail::factorial_block_prefix<Mint>(maximum, block);

  std::vector<Mint> result;
  result.reserve(queries.size());
  for (std::uint64_t n : queries) {
    std::uint64_t completed = n / block;
    Mint value = prefix[completed];
    for (std::uint64_t i = completed * block + 1; i <= n; i++) {
      value *= Mint(i);
    }
    result.push_back(value);
  }
  return result;
}

/// @brief Compute a large batch of factorials modulo a fixed prime. Boundary
/// values (kB)! are obtained by evaluating the block-product polynomial
/// prod_{i=1}^B(x+i). For a query n=qB+r, split the remaining product
/// (qB+1)...n into power-of-two suffixes. A suffix of length 2^b is the falling
/// factorial polynomial x(x-1)...(x-2^b+1) evaluated at its current right
/// endpoint. Queries sharing b are evaluated in batches of at most 2^b points,
/// so no query performs a linear tail scan.
template <class Mint>
std::vector<Mint>
many_factorials(const std::vector<std::uint32_t> &queries) {
  if (queries.empty()) {
    return {};
  }
  constexpr int log_block = 15;
  constexpr int block = 1 << log_block;
  std::uint32_t maximum =
      *std::max_element(queries.begin(), queries.end());
  assert(maximum < std::uint32_t(Mint::mod()));

  std::vector<Mint> prefix =
      large_factorial_detail::factorial_block_prefix<Mint>(maximum, block);
  std::vector<std::vector<std::pair<Mint, int>>> evaluation_points(log_block);
  std::vector<Mint> result(queries.size());
  for (int query = 0; query < int(queries.size()); query++) {
    std::uint32_t n = queries[query];
    int quotient = int(n / block);
    int remainder = int(n % block);
    result[query] = prefix[quotient];
    std::uint32_t endpoint = n;
    for (int bit = 0; bit < log_block; bit++) {
      if ((remainder >> bit) & 1) {
        evaluation_points[bit].emplace_back(Mint(endpoint), query);
        endpoint -= std::uint32_t(1) << bit;
      }
    }
    assert(endpoint == std::uint32_t(quotient * block));
  }

  for (int bit = 0; bit < log_block; bit++) {
    auto &items = evaluation_points[bit];
    if (items.empty()) {
      continue;
    }
    int length = 1 << bit;
    std::vector<Mint> falling_factorial =
        stirling_first_kind_row<Mint>(length);
    for (int left = 0; left < int(items.size()); left += length) {
      int right = std::min(left + length, int(items.size()));
      std::vector<Mint> points;
      points.reserve(right - left);
      for (int index = left; index < right; index++) {
        points.push_back(items[index].first);
      }
      std::vector<Mint> values = polynomial_multipoint_evaluation(
          falling_factorial, points);
      for (int index = left; index < right; index++) {
        result[items[index].second] *= values[index - left];
      }
    }
  }
  return result;
}

} // namespace noya

#endif // NOYA_LARGE_FACTORIAL_HPP
#include <algorithm>
#include <array>
#include <cassert>
#include <cmath>
#include <cstdint>
#include <functional>
#include <numeric>
#include <optional>
#include <queue>
#include <type_traits>
#include <utility>
#include <vector>

/// @complexity Time: `large_factorials` uses
/// O(M(sqrt N) log N + T sqrt N); `many_factorials`, with B=2^15, uses
/// O(M(B) log B + M(N/B) log(N/B) +
/// sum_b ceil(Q_b/2^b) M(2^b) log(2^b)), where Q_b is the number of queries
/// whose remainder has bit b set.
/// Space: O(sqrt N log N) and O(B log B), respectively.

/// @complexity Time: O(M(n) log n) for each generated sequence, where M(n)
/// is polynomial multiplication time.
/// Space: O(n log n) temporary coefficients.

/// @complexity Time: O(M(n) log n) for inverse/log/exp with convolution cost M(n).
/// Space: O(n log n) temporary coefficients.

/// @complexity Time: O(log^2 p).
/// Space: O(1).

/// @complexity Time: O(log^3 n) primality testing; Pollard-rho factorization is expected about O(n^(1/4)).
/// Space: O(log n) recursion and factors.

namespace noya {
namespace factorize_internal {

using u64 = std::uint64_t;
using u128 = unsigned __int128;

inline u64 multiply_mod(u64 a, u64 b, u64 mod) {
  return u64(u128(a) * b % mod);
}

inline u64 power_mod(u64 a, u64 exponent, u64 mod) {
  u64 result = 1;
  while (exponent > 0) {
    if (exponent & 1) {
      result = multiply_mod(result, a, mod);
    }
    a = multiply_mod(a, a, mod);
    exponent >>= 1;
  }
  return result;
}

inline bool miller_rabin(u64 n) {
  if (n < 2) {
    return false;
  }
  for (u64 p :
       std::array<u64, 12>{2, 3, 5, 7, 11, 13, 17, 19, 23, 29, 31, 37}) {
    if (n % p == 0) {
      return n == p;
    }
  }
  int shift = __builtin_ctzll(n - 1);
  u64 odd = (n - 1) >> shift;
  for (u64 base :
       std::array<u64, 7>{2, 325, 9375, 28178, 450775, 9780504, 1795265022}) {
    if (base % n == 0) {
      continue;
    }
    u64 value = power_mod(base % n, odd, n);
    if (value == 1 || value == n - 1) {
      continue;
    }
    bool composite = true;
    for (int i = 1; i < shift; i++) {
      value = multiply_mod(value, value, n);
      if (value == n - 1) {
        composite = false;
        break;
      }
    }
    if (composite) {
      return false;
    }
  }
  return true;
}

inline u64 splitmix64(u64 &state) {
  u64 z = (state += 0x9e3779b97f4a7c15ULL);
  z = (z ^ (z >> 30)) * 0xbf58476d1ce4e5b9ULL;
  z = (z ^ (z >> 27)) * 0x94d049bb133111ebULL;
  return z ^ (z >> 31);
}

inline u64 pollard_rho(u64 n) {
  if (n % 2 == 0) {
    return 2;
  }
  if (n % 3 == 0) {
    return 3;
  }
  static u64 state = 0x123456789abcdef0ULL;
  while (true) {
    u64 y = splitmix64(state) % (n - 1) + 1;
    u64 c = splitmix64(state) % (n - 1) + 1;
    constexpr u64 block = 128;
    u64 g = 1;
    u64 r = 1;
    u64 q = 1;
    u64 x = 0;
    u64 saved_y = 0;
    auto next = [&](u64 value) {
      return u64((u128(multiply_mod(value, value, n)) + c) % n);
    };
    while (g == 1) {
      x = y;
      for (u64 i = 0; i < r; i++) {
        y = next(y);
      }
      for (u64 offset = 0; offset < r && g == 1; offset += block) {
        saved_y = y;
        for (u64 i = 0; i < std::min(block, r - offset); i++) {
          y = next(y);
          u64 difference = x > y ? x - y : y - x;
          q = multiply_mod(q, difference, n);
        }
        g = std::gcd(q, n);
      }
      r <<= 1;
    }
    if (g == n) {
      do {
        saved_y = next(saved_y);
        u64 difference = x > saved_y ? x - saved_y : saved_y - x;
        g = std::gcd(difference, n);
      } while (g == 1);
    }
    if (g != n) {
      return g;
    }
  }
}

inline void collect_factors(u64 n, std::vector<u64> &result) {
  if (n == 1) {
    return;
  }
  if (miller_rabin(n)) {
    result.push_back(n);
    return;
  }
  u64 factor = pollard_rho(n);
  collect_factors(factor, result);
  collect_factors(n / factor, result);
}

} // namespace factorize_internal

/// @brief Deterministic Miller-Rabin primality test for unsigned 64-bit
/// integers.
inline bool is_prime(std::uint64_t n) {
  return factorize_internal::miller_rabin(n);
}

/// @brief Return the prime factors of n with multiplicity in increasing order.
inline std::vector<std::uint64_t> prime_factors(std::uint64_t n) {
  assert(n >= 1);
  std::vector<std::uint64_t> result;
  factorize_internal::collect_factors(n, result);
  std::sort(result.begin(), result.end());
  return result;
}

/// @brief Return the prime factorization of n as (prime, exponent) pairs.
inline std::vector<std::pair<std::uint64_t, int>> factorize(std::uint64_t n) {
  std::vector<std::pair<std::uint64_t, int>> result;
  for (std::uint64_t p : prime_factors(n)) {
    if (result.empty() || result.back().first != p) {
      result.emplace_back(p, 1);
    } else {
      result.back().second++;
    }
  }
  return result;
}

} // namespace noya

namespace noya {

/// @brief Compute the smaller square root modulo a prime, or nullopt if no
/// square root exists.
inline std::optional<std::uint64_t> mod_sqrt(std::uint64_t value,
                                             std::uint64_t modulus) {
  assert(modulus >= 2 && is_prime(modulus));
  value %= modulus;
  if (modulus == 2 || value == 0) {
    return value;
  }
  using factorize_internal::multiply_mod;
  using factorize_internal::power_mod;
  if (power_mod(value, (modulus - 1) / 2, modulus) != 1) {
    return std::nullopt;
  }
  if (modulus % 4 == 3) {
    std::uint64_t root = power_mod(value, (modulus + 1) / 4, modulus);
    return std::min(root, modulus - root);
  }

  std::uint64_t odd = modulus - 1;
  int exponent = 0;
  while ((odd & 1) == 0) {
    odd >>= 1;
    exponent++;
  }
  std::uint64_t non_residue = 2;
  while (power_mod(non_residue, (modulus - 1) / 2, modulus) != modulus - 1) {
    non_residue++;
  }

  std::uint64_t root = power_mod(value, (odd + 1) / 2, modulus);
  std::uint64_t remainder = power_mod(value, odd, modulus);
  std::uint64_t step = power_mod(non_residue, odd, modulus);
  int remaining = exponent;
  while (remainder != 1) {
    std::uint64_t squared = remainder;
    int shift = 0;
    while (squared != 1 && shift < remaining) {
      squared = multiply_mod(squared, squared, modulus);
      shift++;
    }
    assert(shift < remaining);
    std::uint64_t multiplier =
        power_mod(step, std::uint64_t(1) << (remaining - shift - 1), modulus);
    root = multiply_mod(root, multiplier, modulus);
    step = multiply_mod(multiplier, multiplier, modulus);
    remainder = multiply_mod(remainder, step, modulus);
    remaining = shift;
  }
  return std::min(root, modulus - root);
}

} // namespace noya

/// @complexity Time: O(M(n) log n) inverse/division and O(M(n)) Taylor shift.
/// Space: O(n log n) temporaries.

#ifdef _MSC_VER
#include <intrin.h>
#endif

#if __cplusplus >= 202002L
#include <bit>
#endif

namespace atcoder {

namespace internal {

#if __cplusplus >= 202002L

using std::bit_ceil;

#else

// @return same with std::bit::bit_ceil
unsigned int bit_ceil(unsigned int n) {
    unsigned int x = 1;
    while (x < (unsigned int)(n)) x *= 2;
    return x;
}

#endif

// @param n `1 <= n`
// @return same with std::bit::countr_zero
int countr_zero(unsigned int n) {
#ifdef _MSC_VER
    unsigned long index;
    _BitScanForward(&index, n);
    return index;
#else
    return __builtin_ctz(n);
#endif
}

// @param n `1 <= n`
// @return same with std::bit::countr_zero
constexpr int countr_zero_constexpr(unsigned int n) {
    int x = 0;
    while (!(n & (1 << x))) x++;
    return x;
}

}  // namespace internal

}  // namespace atcoder

#ifdef _MSC_VER
#include <intrin.h>
#endif

#ifdef _MSC_VER
#include <intrin.h>
#endif

namespace atcoder {

namespace internal {

// @param m `1 <= m`
// @return x mod m
constexpr long long safe_mod(long long x, long long m) {
    x %= m;
    if (x < 0) x += m;
    return x;
}

// Fast modular multiplication by barrett reduction
// Reference: https://en.wikipedia.org/wiki/Barrett_reduction
// NOTE: reconsider after Ice Lake
struct barrett {
    unsigned int _m;
    unsigned long long im;

    // @param m `1 <= m`
    explicit barrett(unsigned int m) : _m(m), im((unsigned long long)(-1) / m + 1) {}

    // @return m
    unsigned int umod() const { return _m; }

    // @param a `0 <= a < m`
    // @param b `0 <= b < m`
    // @return `a * b % m`
    unsigned int mul(unsigned int a, unsigned int b) const {
        // [1] m = 1
        // a = b = im = 0, so okay

        // [2] m >= 2
        // im = ceil(2^64 / m)
        // -> im * m = 2^64 + r (0 <= r < m)
        // let z = a*b = c*m + d (0 <= c, d < m)
        // a*b * im = (c*m + d) * im = c*(im*m) + d*im = c*2^64 + c*r + d*im
        // c*r + d*im < m * m + m * im < m * m + 2^64 + m <= 2^64 + m * (m + 1) < 2^64 * 2
        // ((ab * im) >> 64) == c or c + 1
        unsigned long long z = a;
        z *= b;
#ifdef _MSC_VER
        unsigned long long x;
        _umul128(z, im, &x);
#else
        unsigned long long x =
            (unsigned long long)(((unsigned __int128)(z)*im) >> 64);
#endif
        unsigned long long y = x * _m;
        return (unsigned int)(z - y + (z < y ? _m : 0));
    }
};

// @param n `0 <= n`
// @param m `1 <= m`
// @return `(x ** n) % m`
constexpr long long pow_mod_constexpr(long long x, long long n, int m) {
    if (m == 1) return 0;
    unsigned int _m = (unsigned int)(m);
    unsigned long long r = 1;
    unsigned long long y = safe_mod(x, m);
    while (n) {
        if (n & 1) r = (r * y) % _m;
        y = (y * y) % _m;
        n >>= 1;
    }
    return r;
}

// Reference:
// M. Forisek and J. Jancina,
// Fast Primality Testing for Integers That Fit into a Machine Word
// @param n `0 <= n`
constexpr bool is_prime_constexpr(int n) {
    if (n <= 1) return false;
    if (n == 2 || n == 7 || n == 61) return true;
    if (n % 2 == 0) return false;
    long long d = n - 1;
    while (d % 2 == 0) d /= 2;
    constexpr long long bases[3] = {2, 7, 61};
    for (long long a : bases) {
        long long t = d;
        long long y = pow_mod_constexpr(a, t, n);
        while (t != n - 1 && y != 1 && y != n - 1) {
            y = y * y % n;
            t <<= 1;
        }
        if (y != n - 1 && t % 2 == 0) {
            return false;
        }
    }
    return true;
}
template <int n> constexpr bool is_prime = is_prime_constexpr(n);

// @param b `1 <= b`
// @return pair(g, x) s.t. g = gcd(a, b), xa = g (mod b), 0 <= x < b/g
constexpr std::pair<long long, long long> inv_gcd(long long a, long long b) {
    a = safe_mod(a, b);
    if (a == 0) return {b, 0};

    // Contracts:
    // [1] s - m0 * a = 0 (mod b)
    // [2] t - m1 * a = 0 (mod b)
    // [3] s * |m1| + t * |m0| <= b
    long long s = b, t = a;
    long long m0 = 0, m1 = 1;

    while (t) {
        long long u = s / t;
        s -= t * u;
        m0 -= m1 * u;  // |m1 * u| <= |m1| * s <= b

        // [3]:
        // (s - t * u) * |m1| + t * |m0 - m1 * u|
        // <= s * |m1| - t * u * |m1| + t * (|m0| + |m1| * u)
        // = s * |m1| + t * |m0| <= b

        auto tmp = s;
        s = t;
        t = tmp;
        tmp = m0;
        m0 = m1;
        m1 = tmp;
    }
    // by [3]: |m0| <= b/g
    // by g != b: |m0| < b/g
    if (m0 < 0) m0 += b / s;
    return {s, m0};
}

// Compile time primitive root
// @param m must be prime
// @return primitive root (and minimum in now)
constexpr int primitive_root_constexpr(int m) {
    if (m == 2) return 1;
    if (m == 167772161) return 3;
    if (m == 469762049) return 3;
    if (m == 754974721) return 11;
    if (m == 998244353) return 3;
    int divs[20] = {};
    divs[0] = 2;
    int cnt = 1;
    int x = (m - 1) / 2;
    while (x % 2 == 0) x /= 2;
    for (int i = 3; (long long)(i)*i <= x; i += 2) {
        if (x % i == 0) {
            divs[cnt++] = i;
            while (x % i == 0) {
                x /= i;
            }
        }
    }
    if (x > 1) {
        divs[cnt++] = x;
    }
    for (int g = 2;; g++) {
        bool ok = true;
        for (int i = 0; i < cnt; i++) {
            if (pow_mod_constexpr(g, (m - 1) / divs[i], m) == 1) {
                ok = false;
                break;
            }
        }
        if (ok) return g;
    }
}
template <int m> constexpr int primitive_root = primitive_root_constexpr(m);

// @param n `n < 2^32`
// @param m `1 <= m < 2^32`
// @return sum_{i=0}^{n-1} floor((ai + b) / m) (mod 2^64)
unsigned long long floor_sum_unsigned(unsigned long long n,
                                      unsigned long long m,
                                      unsigned long long a,
                                      unsigned long long b) {
    unsigned long long ans = 0;
    while (true) {
        if (a >= m) {
            ans += n * (n - 1) / 2 * (a / m);
            a %= m;
        }
        if (b >= m) {
            ans += n * (b / m);
            b %= m;
        }

        unsigned long long y_max = a * n + b;
        if (y_max < m) break;
        // y_max < m * (n + 1)
        // floor(y_max / m) <= n
        n = (unsigned long long)(y_max / m);
        b = (unsigned long long)(y_max % m);
        std::swap(m, a);
    }
    return ans;
}

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

#ifndef _MSC_VER
template <class T>
using is_signed_int128 =
    typename std::conditional<std::is_same<T, __int128_t>::value ||
                                  std::is_same<T, __int128>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using is_unsigned_int128 =
    typename std::conditional<std::is_same<T, __uint128_t>::value ||
                                  std::is_same<T, unsigned __int128>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using make_unsigned_int128 =
    typename std::conditional<std::is_same<T, __int128_t>::value,
                              __uint128_t,
                              unsigned __int128>;

template <class T>
using is_integral = typename std::conditional<std::is_integral<T>::value ||
                                                  is_signed_int128<T>::value ||
                                                  is_unsigned_int128<T>::value,
                                              std::true_type,
                                              std::false_type>::type;

template <class T>
using is_signed_int = typename std::conditional<(is_integral<T>::value &&
                                                 std::is_signed<T>::value) ||
                                                    is_signed_int128<T>::value,
                                                std::true_type,
                                                std::false_type>::type;

template <class T>
using is_unsigned_int =
    typename std::conditional<(is_integral<T>::value &&
                               std::is_unsigned<T>::value) ||
                                  is_unsigned_int128<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using to_unsigned = typename std::conditional<
    is_signed_int128<T>::value,
    make_unsigned_int128<T>,
    typename std::conditional<std::is_signed<T>::value,
                              std::make_unsigned<T>,
                              std::common_type<T>>::type>::type;

#else

template <class T> using is_integral = typename std::is_integral<T>;

template <class T>
using is_signed_int =
    typename std::conditional<is_integral<T>::value && std::is_signed<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using is_unsigned_int =
    typename std::conditional<is_integral<T>::value &&
                                  std::is_unsigned<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using to_unsigned = typename std::conditional<is_signed_int<T>::value,
                                              std::make_unsigned<T>,
                                              std::common_type<T>>::type;

#endif

template <class T>
using is_signed_int_t = std::enable_if_t<is_signed_int<T>::value>;

template <class T>
using is_unsigned_int_t = std::enable_if_t<is_unsigned_int<T>::value>;

template <class T> using to_unsigned_t = typename to_unsigned<T>::type;

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

struct modint_base {};
struct static_modint_base : modint_base {};

template <class T> using is_modint = std::is_base_of<modint_base, T>;
template <class T> using is_modint_t = std::enable_if_t<is_modint<T>::value>;

}  // namespace internal

template <int m, std::enable_if_t<(1 <= m)>* = nullptr>
struct static_modint : internal::static_modint_base {
    using mint = static_modint;

  public:
    static constexpr int mod() { return m; }
    static mint raw(int v) {
        mint x;
        x._v = v;
        return x;
    }

    static_modint() : _v(0) {}
    template <class T, internal::is_signed_int_t<T>* = nullptr>
    static_modint(T v) {
        long long x = (long long)(v % (long long)(umod()));
        if (x < 0) x += umod();
        _v = (unsigned int)(x);
    }
    template <class T, internal::is_unsigned_int_t<T>* = nullptr>
    static_modint(T v) {
        _v = (unsigned int)(v % umod());
    }

    int val() const { return _v; }

    mint& operator++() {
        _v++;
        if (_v == umod()) _v = 0;
        return *this;
    }
    mint& operator--() {
        if (_v == 0) _v = umod();
        _v--;
        return *this;
    }
    mint operator++(int) {
        mint result = *this;
        ++*this;
        return result;
    }
    mint operator--(int) {
        mint result = *this;
        --*this;
        return result;
    }

    mint& operator+=(const mint& rhs) {
        _v += rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator-=(const mint& rhs) {
        _v -= rhs._v;
        if (_v >= umod()) _v += umod();
        return *this;
    }
    mint& operator*=(const mint& rhs) {
        unsigned long long z = _v;
        z *= rhs._v;
        _v = (unsigned int)(z % umod());
        return *this;
    }
    mint& operator/=(const mint& rhs) { return *this = *this * rhs.inv(); }

    mint operator+() const { return *this; }
    mint operator-() const { return mint() - *this; }

    mint pow(long long n) const {
        assert(0 <= n);
        mint x = *this, r = 1;
        while (n) {
            if (n & 1) r *= x;
            x *= x;
            n >>= 1;
        }
        return r;
    }
    mint inv() const {
        if (prime) {
            assert(_v);
            return pow(umod() - 2);
        } else {
            auto eg = internal::inv_gcd(_v, m);
            assert(eg.first == 1);
            return eg.second;
        }
    }

    friend mint operator+(const mint& lhs, const mint& rhs) {
        return mint(lhs) += rhs;
    }
    friend mint operator-(const mint& lhs, const mint& rhs) {
        return mint(lhs) -= rhs;
    }
    friend mint operator*(const mint& lhs, const mint& rhs) {
        return mint(lhs) *= rhs;
    }
    friend mint operator/(const mint& lhs, const mint& rhs) {
        return mint(lhs) /= rhs;
    }
    friend bool operator==(const mint& lhs, const mint& rhs) {
        return lhs._v == rhs._v;
    }
    friend bool operator!=(const mint& lhs, const mint& rhs) {
        return lhs._v != rhs._v;
    }

  private:
    unsigned int _v;
    static constexpr unsigned int umod() { return m; }
    static constexpr bool prime = internal::is_prime<m>;
};

template <int id> struct dynamic_modint : internal::modint_base {
    using mint = dynamic_modint;

  public:
    static int mod() { return (int)(bt.umod()); }
    static void set_mod(int m) {
        assert(1 <= m);
        bt = internal::barrett(m);
    }
    static mint raw(int v) {
        mint x;
        x._v = v;
        return x;
    }

    dynamic_modint() : _v(0) {}
    template <class T, internal::is_signed_int_t<T>* = nullptr>
    dynamic_modint(T v) {
        long long x = (long long)(v % (long long)(mod()));
        if (x < 0) x += mod();
        _v = (unsigned int)(x);
    }
    template <class T, internal::is_unsigned_int_t<T>* = nullptr>
    dynamic_modint(T v) {
        _v = (unsigned int)(v % mod());
    }

    int val() const { return _v; }

    mint& operator++() {
        _v++;
        if (_v == umod()) _v = 0;
        return *this;
    }
    mint& operator--() {
        if (_v == 0) _v = umod();
        _v--;
        return *this;
    }
    mint operator++(int) {
        mint result = *this;
        ++*this;
        return result;
    }
    mint operator--(int) {
        mint result = *this;
        --*this;
        return result;
    }

    mint& operator+=(const mint& rhs) {
        _v += rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator-=(const mint& rhs) {
        _v += mod() - rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator*=(const mint& rhs) {
        _v = bt.mul(_v, rhs._v);
        return *this;
    }
    mint& operator/=(const mint& rhs) { return *this = *this * rhs.inv(); }

    mint operator+() const { return *this; }
    mint operator-() const { return mint() - *this; }

    mint pow(long long n) const {
        assert(0 <= n);
        mint x = *this, r = 1;
        while (n) {
            if (n & 1) r *= x;
            x *= x;
            n >>= 1;
        }
        return r;
    }
    mint inv() const {
        auto eg = internal::inv_gcd(_v, mod());
        assert(eg.first == 1);
        return eg.second;
    }

    friend mint operator+(const mint& lhs, const mint& rhs) {
        return mint(lhs) += rhs;
    }
    friend mint operator-(const mint& lhs, const mint& rhs) {
        return mint(lhs) -= rhs;
    }
    friend mint operator*(const mint& lhs, const mint& rhs) {
        return mint(lhs) *= rhs;
    }
    friend mint operator/(const mint& lhs, const mint& rhs) {
        return mint(lhs) /= rhs;
    }
    friend bool operator==(const mint& lhs, const mint& rhs) {
        return lhs._v == rhs._v;
    }
    friend bool operator!=(const mint& lhs, const mint& rhs) {
        return lhs._v != rhs._v;
    }

  private:
    unsigned int _v;
    static internal::barrett bt;
    static unsigned int umod() { return bt.umod(); }
};
template <int id> internal::barrett dynamic_modint<id>::bt(998244353);

using modint998244353 = static_modint<998244353>;
using modint1000000007 = static_modint<1000000007>;
using modint = dynamic_modint<-1>;

namespace internal {

template <class T>
using is_static_modint = std::is_base_of<internal::static_modint_base, T>;

template <class T>
using is_static_modint_t = std::enable_if_t<is_static_modint<T>::value>;

template <class> struct is_dynamic_modint : public std::false_type {};
template <int id>
struct is_dynamic_modint<dynamic_modint<id>> : public std::true_type {};

template <class T>
using is_dynamic_modint_t = std::enable_if_t<is_dynamic_modint<T>::value>;

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

template <class mint,
          int g = internal::primitive_root<mint::mod()>,
          internal::is_static_modint_t<mint>* = nullptr>
struct fft_info {
    static constexpr int rank2 = countr_zero_constexpr(mint::mod() - 1);
    std::array<mint, rank2 + 1> root;   // root[i]^(2^i) == 1
    std::array<mint, rank2 + 1> iroot;  // root[i] * iroot[i] == 1

    std::array<mint, std::max(0, rank2 - 2 + 1)> rate2;
    std::array<mint, std::max(0, rank2 - 2 + 1)> irate2;

    std::array<mint, std::max(0, rank2 - 3 + 1)> rate3;
    std::array<mint, std::max(0, rank2 - 3 + 1)> irate3;

    fft_info() {
        root[rank2] = mint(g).pow((mint::mod() - 1) >> rank2);
        iroot[rank2] = root[rank2].inv();
        for (int i = rank2 - 1; i >= 0; i--) {
            root[i] = root[i + 1] * root[i + 1];
            iroot[i] = iroot[i + 1] * iroot[i + 1];
        }

        {
            mint prod = 1, iprod = 1;
            for (int i = 0; i <= rank2 - 2; i++) {
                rate2[i] = root[i + 2] * prod;
                irate2[i] = iroot[i + 2] * iprod;
                prod *= iroot[i + 2];
                iprod *= root[i + 2];
            }
        }
        {
            mint prod = 1, iprod = 1;
            for (int i = 0; i <= rank2 - 3; i++) {
                rate3[i] = root[i + 3] * prod;
                irate3[i] = iroot[i + 3] * iprod;
                prod *= iroot[i + 3];
                iprod *= root[i + 3];
            }
        }
    }
};

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
void butterfly(std::vector<mint>& a) {
    int n = int(a.size());
    int h = internal::countr_zero((unsigned int)n);

    static const fft_info<mint> info;

    int len = 0;  // a[i, i+(n>>len), i+2*(n>>len), ..] is transformed
    while (len < h) {
        if (h - len == 1) {
            int p = 1 << (h - len - 1);
            mint rot = 1;
            for (int s = 0; s < (1 << len); s++) {
                int offset = s << (h - len);
                for (int i = 0; i < p; i++) {
                    auto l = a[i + offset];
                    auto r = a[i + offset + p] * rot;
                    a[i + offset] = l + r;
                    a[i + offset + p] = l - r;
                }
                if (s + 1 != (1 << len))
                    rot *= info.rate2[countr_zero(~(unsigned int)(s))];
            }
            len++;
        } else {
            // 4-base
            int p = 1 << (h - len - 2);
            mint rot = 1, imag = info.root[2];
            for (int s = 0; s < (1 << len); s++) {
                mint rot2 = rot * rot;
                mint rot3 = rot2 * rot;
                int offset = s << (h - len);
                for (int i = 0; i < p; i++) {
                    auto mod2 = 1ULL * mint::mod() * mint::mod();
                    auto a0 = 1ULL * a[i + offset].val();
                    auto a1 = 1ULL * a[i + offset + p].val() * rot.val();
                    auto a2 = 1ULL * a[i + offset + 2 * p].val() * rot2.val();
                    auto a3 = 1ULL * a[i + offset + 3 * p].val() * rot3.val();
                    auto a1na3imag =
                        1ULL * mint(a1 + mod2 - a3).val() * imag.val();
                    auto na2 = mod2 - a2;
                    a[i + offset] = a0 + a2 + a1 + a3;
                    a[i + offset + 1 * p] = a0 + a2 + (2 * mod2 - (a1 + a3));
                    a[i + offset + 2 * p] = a0 + na2 + a1na3imag;
                    a[i + offset + 3 * p] = a0 + na2 + (mod2 - a1na3imag);
                }
                if (s + 1 != (1 << len))
                    rot *= info.rate3[countr_zero(~(unsigned int)(s))];
            }
            len += 2;
        }
    }
}

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
void butterfly_inv(std::vector<mint>& a) {
    int n = int(a.size());
    int h = internal::countr_zero((unsigned int)n);

    static const fft_info<mint> info;

    int len = h;  // a[i, i+(n>>len), i+2*(n>>len), ..] is transformed
    while (len) {
        if (len == 1) {
            int p = 1 << (h - len);
            mint irot = 1;
            for (int s = 0; s < (1 << (len - 1)); s++) {
                int offset = s << (h - len + 1);
                for (int i = 0; i < p; i++) {
                    auto l = a[i + offset];
                    auto r = a[i + offset + p];
                    a[i + offset] = l + r;
                    a[i + offset + p] =
                        (unsigned long long)((unsigned int)(l.val() - r.val()) + mint::mod()) *
                        irot.val();
                    ;
                }
                if (s + 1 != (1 << (len - 1)))
                    irot *= info.irate2[countr_zero(~(unsigned int)(s))];
            }
            len--;
        } else {
            // 4-base
            int p = 1 << (h - len);
            mint irot = 1, iimag = info.iroot[2];
            for (int s = 0; s < (1 << (len - 2)); s++) {
                mint irot2 = irot * irot;
                mint irot3 = irot2 * irot;
                int offset = s << (h - len + 2);
                for (int i = 0; i < p; i++) {
                    auto a0 = 1ULL * a[i + offset + 0 * p].val();
                    auto a1 = 1ULL * a[i + offset + 1 * p].val();
                    auto a2 = 1ULL * a[i + offset + 2 * p].val();
                    auto a3 = 1ULL * a[i + offset + 3 * p].val();

                    auto a2na3iimag =
                        1ULL *
                        mint((mint::mod() + a2 - a3) * iimag.val()).val();

                    a[i + offset] = a0 + a1 + a2 + a3;
                    a[i + offset + 1 * p] =
                        (a0 + (mint::mod() - a1) + a2na3iimag) * irot.val();
                    a[i + offset + 2 * p] =
                        (a0 + a1 + (mint::mod() - a2) + (mint::mod() - a3)) *
                        irot2.val();
                    a[i + offset + 3 * p] =
                        (a0 + (mint::mod() - a1) + (mint::mod() - a2na3iimag)) *
                        irot3.val();
                }
                if (s + 1 != (1 << (len - 2)))
                    irot *= info.irate3[countr_zero(~(unsigned int)(s))];
            }
            len -= 2;
        }
    }
}

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution_naive(const std::vector<mint>& a,
                                    const std::vector<mint>& b) {
    int n = int(a.size()), m = int(b.size());
    std::vector<mint> ans(n + m - 1);
    if (n < m) {
        for (int j = 0; j < m; j++) {
            for (int i = 0; i < n; i++) {
                ans[i + j] += a[i] * b[j];
            }
        }
    } else {
        for (int i = 0; i < n; i++) {
            for (int j = 0; j < m; j++) {
                ans[i + j] += a[i] * b[j];
            }
        }
    }
    return ans;
}

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution_fft(std::vector<mint> a, std::vector<mint> b) {
    int n = int(a.size()), m = int(b.size());
    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    a.resize(z);
    internal::butterfly(a);
    b.resize(z);
    internal::butterfly(b);
    for (int i = 0; i < z; i++) {
        a[i] *= b[i];
    }
    internal::butterfly_inv(a);
    a.resize(n + m - 1);
    mint iz = mint(z).inv();
    for (int i = 0; i < n + m - 1; i++) a[i] *= iz;
    return a;
}

}  // namespace internal

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution(std::vector<mint>&& a, std::vector<mint>&& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    assert((mint::mod() - 1) % z == 0);

    if (std::min(n, m) <= 60) return convolution_naive(std::move(a), std::move(b));
    return internal::convolution_fft(std::move(a), std::move(b));
}
template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution(const std::vector<mint>& a,
                              const std::vector<mint>& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    assert((mint::mod() - 1) % z == 0);

    if (std::min(n, m) <= 60) return convolution_naive(a, b);
    return internal::convolution_fft(a, b);
}

template <unsigned int mod = 998244353,
          class T,
          std::enable_if_t<internal::is_integral<T>::value>* = nullptr>
std::vector<T> convolution(const std::vector<T>& a, const std::vector<T>& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    using mint = static_modint<mod>;

    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    assert((mint::mod() - 1) % z == 0);

    std::vector<mint> a2(n), b2(m);
    for (int i = 0; i < n; i++) {
        a2[i] = mint(a[i]);
    }
    for (int i = 0; i < m; i++) {
        b2[i] = mint(b[i]);
    }
    auto c2 = convolution(std::move(a2), std::move(b2));
    std::vector<T> c(n + m - 1);
    for (int i = 0; i < n + m - 1; i++) {
        c[i] = c2[i].val();
    }
    return c;
}

std::vector<long long> convolution_ll(const std::vector<long long>& a,
                                      const std::vector<long long>& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    static constexpr unsigned long long MOD1 = 754974721;  // 2^24
    static constexpr unsigned long long MOD2 = 167772161;  // 2^25
    static constexpr unsigned long long MOD3 = 469762049;  // 2^26
    static constexpr unsigned long long M2M3 = MOD2 * MOD3;
    static constexpr unsigned long long M1M3 = MOD1 * MOD3;
    static constexpr unsigned long long M1M2 = MOD1 * MOD2;
    static constexpr unsigned long long M1M2M3 = MOD1 * MOD2 * MOD3;

    static constexpr unsigned long long i1 =
        internal::inv_gcd(MOD2 * MOD3, MOD1).second;
    static constexpr unsigned long long i2 =
        internal::inv_gcd(MOD1 * MOD3, MOD2).second;
    static constexpr unsigned long long i3 =
        internal::inv_gcd(MOD1 * MOD2, MOD3).second;

    static constexpr int MAX_AB_BIT = 24;
    static_assert(MOD1 % (1ull << MAX_AB_BIT) == 1, "MOD1 isn't enough to support an array length of 2^24.");
    static_assert(MOD2 % (1ull << MAX_AB_BIT) == 1, "MOD2 isn't enough to support an array length of 2^24.");
    static_assert(MOD3 % (1ull << MAX_AB_BIT) == 1, "MOD3 isn't enough to support an array length of 2^24.");
    assert(n + m - 1 <= (1 << MAX_AB_BIT));

    auto c1 = convolution<MOD1>(a, b);
    auto c2 = convolution<MOD2>(a, b);
    auto c3 = convolution<MOD3>(a, b);

    std::vector<long long> c(n + m - 1);
    for (int i = 0; i < n + m - 1; i++) {
        unsigned long long x = 0;
        x += (c1[i] * i1) % MOD1 * M2M3;
        x += (c2[i] * i2) % MOD2 * M1M3;
        x += (c3[i] * i3) % MOD3 * M1M2;
        // B = 2^63, -B <= x, r(real value) < B
        // (x, x - M, x - 2M, or x - 3M) = r (mod 2B)
        // r = c1[i] (mod MOD1)
        // focus on MOD1
        // r = x, x - M', x - 2M', x - 3M' (M' = M % 2^64) (mod 2B)
        // r = x,
        //     x - M' + (0 or 2B),
        //     x - 2M' + (0, 2B or 4B),
        //     x - 3M' + (0, 2B, 4B or 6B) (without mod!)
        // (r - x) = 0, (0)
        //           - M' + (0 or 2B), (1)
        //           -2M' + (0 or 2B or 4B), (2)
        //           -3M' + (0 or 2B or 4B or 6B) (3) (mod MOD1)
        // we checked that
        //   ((1) mod MOD1) mod 5 = 2
        //   ((2) mod MOD1) mod 5 = 3
        //   ((3) mod MOD1) mod 5 = 4
        long long diff =
            c1[i] - internal::safe_mod((long long)(x), (long long)(MOD1));
        if (diff < 0) diff += MOD1;
        static constexpr unsigned long long offset[5] = {
            0, 0, M1M2M3, 2 * M1M2M3, 3 * M1M2M3};
        x -= offset[diff % 5];
        c[i] = x;
    }

    return c;
}

}  // namespace atcoder

namespace noya {

/// @brief Remove trailing zero coefficients from a polynomial.
template <class T> void polynomial_trim(std::vector<T> &polynomial) {
  while (!polynomial.empty() && polynomial.back() == T{}) {
    polynomial.pop_back();
  }
}

/// @brief Return the formal derivative of a polynomial.
template <class T>
std::vector<T> polynomial_derivative(const std::vector<T> &polynomial) {
  if (polynomial.size() <= 1) {
    return {};
  }
  std::vector<T> result(polynomial.size() - 1);
  for (int i = 1; i < int(polynomial.size()); i++) {
    result[i - 1] = polynomial[i] * T(i);
  }
  return result;
}

/// @brief Return the formal integral with constant coefficient zero.
template <class T>
std::vector<T> polynomial_integral(const std::vector<T> &polynomial) {
  std::vector<T> result(polynomial.size() + 1);
  for (int i = 0; i < int(polynomial.size()); i++) {
    result[i + 1] = polynomial[i] / T(i + 1);
  }
  return result;
}

/// @brief Return the first n coefficients of 1/f using Newton iteration.
template <class Mint>
std::vector<Mint> polynomial_inverse_series(const std::vector<Mint> &f, int n) {
  assert(n >= 0);
  if (n == 0) {
    return {};
  }
  assert(!f.empty() && f[0] != Mint{});
  std::vector<Mint> inverse = {Mint(1) / f[0]};
  while (int(inverse.size()) < n) {
    int target = std::min(n, int(inverse.size()) * 2);
    std::vector<Mint> prefix(target);
    for (int i = 0; i < std::min(target, int(f.size())); i++) {
      prefix[i] = f[i];
    }
    std::vector<Mint> correction = atcoder::convolution(prefix, inverse);
    correction.resize(target);
    for (Mint &value : correction) {
      value = -value;
    }
    correction[0] += Mint(2);
    inverse = atcoder::convolution(inverse, correction);
    inverse.resize(target);
  }
  return inverse;
}

/// @brief Divide f by nonzero g and return (quotient, remainder).
template <class Mint>
std::pair<std::vector<Mint>, std::vector<Mint>>
polynomial_divmod(std::vector<Mint> f, std::vector<Mint> g) {
  polynomial_trim(f);
  polynomial_trim(g);
  assert(!g.empty());
  if (f.size() < g.size()) {
    return {{}, f};
  }
  int quotient_size = int(f.size() - g.size() + 1);
  std::vector<Mint> reversed_f(f.rbegin(), f.rend());
  std::vector<Mint> reversed_g(g.rbegin(), g.rend());
  reversed_f.resize(quotient_size);
  reversed_g.resize(quotient_size);
  std::vector<Mint> inverse =
      polynomial_inverse_series(reversed_g, quotient_size);
  std::vector<Mint> quotient = atcoder::convolution(reversed_f, inverse);
  quotient.resize(quotient_size);
  std::reverse(quotient.begin(), quotient.end());

  std::vector<Mint> product = atcoder::convolution(quotient, g);
  for (int i = 0; i < int(product.size()); i++) {
    f[i] -= product[i];
  }
  polynomial_trim(f);
  return {quotient, f};
}

/// @brief Return f(x + shift) in O(M(n)) time.
template <class Mint>
std::vector<Mint> polynomial_taylor_shift(const std::vector<Mint> &f,
                                          Mint shift) {
  int size = int(f.size());
  if (size == 0) {
    return {};
  }
  std::vector<Mint> factorial(size, Mint(1));
  std::vector<Mint> inverse_factorial(size, Mint(1));
  for (int i = 1; i < size; i++) {
    factorial[i] = factorial[i - 1] * Mint(i);
  }
  inverse_factorial.back() = Mint(1) / factorial.back();
  for (int i = size - 1; i > 0; i--) {
    inverse_factorial[i - 1] = inverse_factorial[i] * Mint(i);
  }

  std::vector<Mint> reversed(size), powers(size);
  Mint power = Mint(1);
  for (int i = 0; i < size; i++) {
    reversed[size - 1 - i] = f[i] * factorial[i];
    powers[i] = power * inverse_factorial[i];
    power *= shift;
  }
  std::vector<Mint> product = atcoder::convolution(reversed, powers);
  std::vector<Mint> result(size);
  for (int i = 0; i < size; i++) {
    result[i] = product[size - 1 - i] * inverse_factorial[i];
  }
  return result;
}

} // namespace noya

namespace noya {

/// @brief Return the first n coefficients of 1/f; requires f[0] != 0.
template <class Mint>
std::vector<Mint> fps_inverse(const std::vector<Mint> &f, int n) {
  return polynomial_inverse_series(f, n);
}

/// @brief Return log(f) modulo x^n; requires f[0] = 1.
template <class Mint>
std::vector<Mint> fps_logarithm(const std::vector<Mint> &f, int n) {
  assert(n >= 0);
  if (n == 0) {
    return {};
  }
  assert(!f.empty() && f[0] == Mint(1));
  std::vector<Mint> derivative = polynomial_derivative(f);
  std::vector<Mint> inverse = fps_inverse(f, n);
  std::vector<Mint> product = atcoder::convolution(derivative, inverse);
  product.resize(n - 1);
  std::vector<Mint> result = polynomial_integral(product);
  result.resize(n);
  return result;
}

/// @brief Return exp(f) modulo x^n; requires f[0] = 0.
template <class Mint>
std::vector<Mint> fps_exponential(const std::vector<Mint> &f, int n) {
  assert(n >= 0);
  if (n == 0) {
    return {};
  }
  assert(f.empty() || f[0] == Mint{});
  std::vector<Mint> result = {Mint(1)};
  while (int(result.size()) < n) {
    int target = std::min(n, int(result.size()) * 2);
    std::vector<Mint> logarithm = fps_logarithm(result, target);
    std::vector<Mint> correction(target);
    for (int i = 0; i < target; i++) {
      if (i < int(f.size())) {
        correction[i] += f[i];
      }
      correction[i] -= logarithm[i];
    }
    correction[0] += Mint(1);
    result = atcoder::convolution(result, correction);
    result.resize(target);
  }
  return result;
}

/// @brief Compute value^exponent by binary exponentiation.
template <class Mint>
Mint fps_scalar_power(Mint value, std::uint64_t exponent) {
  Mint result = Mint(1);
  while (exponent > 0) {
    if (exponent & 1) {
      result *= value;
    }
    value *= value;
    exponent >>= 1;
  }
  return result;
}

/// @brief Return f^exponent modulo x^n for a nonnegative exponent.
template <class Mint>
std::vector<Mint> fps_power(const std::vector<Mint> &f, std::uint64_t exponent,
                            int n) {
  assert(n >= 0);
  if (n == 0) {
    return {};
  }
  std::vector<Mint> zero(n);
  if (exponent == 0) {
    zero[0] = Mint(1);
    return zero;
  }
  int first = 0;
  while (first < int(f.size()) && f[first] == Mint{}) {
    first++;
  }
  if (first == int(f.size()) ||
      (first > 0 && exponent > std::uint64_t((n - 1) / first))) {
    return zero;
  }
  int shift = int(std::uint64_t(first) * exponent);
  int target = n - shift;
  Mint leading = f[first];
  std::vector<Mint> normalized(target);
  for (int i = 0; i < target && first + i < int(f.size()); i++) {
    normalized[i] = f[first + i] / leading;
  }
  std::vector<Mint> logarithm = fps_logarithm(normalized, target);
  Mint scalar_exponent = Mint(exponent);
  for (Mint &value : logarithm) {
    value *= scalar_exponent;
  }
  std::vector<Mint> powered = fps_exponential(logarithm, target);
  Mint leading_power = fps_scalar_power(leading, exponent);
  for (Mint &value : powered) {
    value *= leading_power;
  }
  std::vector<Mint> result(n);
  for (int i = 0; i < target; i++) {
    result[shift + i] = powered[i];
  }
  return result;
}

/// @brief Return a formal square root of f modulo x^n, if one exists.
template <class Mint>
std::optional<std::vector<Mint>> fps_square_root(const std::vector<Mint> &f,
                                                 int n) {
  assert(n >= 0);
  if (n == 0) {
    return std::vector<Mint>{};
  }
  int first = 0;
  while (first < std::min(n, int(f.size())) && f[first] == Mint{}) {
    first++;
  }
  if (first == std::min(n, int(f.size()))) {
    return std::vector<Mint>(n);
  }
  if (first & 1) {
    return std::nullopt;
  }
  if (first > 0) {
    int shift = first / 2;
    int target = n - first;
    std::vector<Mint> reduced(target);
    for (int i = 0; i < target && first + i < int(f.size()); i++) {
      reduced[i] = f[first + i];
    }
    auto root = fps_square_root(reduced, target);
    if (!root) {
      return std::nullopt;
    }
    std::vector<Mint> result(n);
    for (int i = 0; i < int(root->size()) && shift + i < n; i++) {
      result[shift + i] = (*root)[i];
    }
    return result;
  }

  auto constant_root =
      mod_sqrt(std::uint64_t(f[0].val()), std::uint64_t(Mint::mod()));
  if (!constant_root) {
    return std::nullopt;
  }
  std::vector<Mint> result = {Mint(*constant_root)};
  Mint inverse_two = Mint(1) / Mint(2);
  while (int(result.size()) < n) {
    int target = std::min(n, int(result.size()) * 2);
    std::vector<Mint> prefix(target);
    for (int i = 0; i < target && i < int(f.size()); i++) {
      prefix[i] = f[i];
    }
    std::vector<Mint> quotient = atcoder::convolution(
        prefix, polynomial_inverse_series(result, target));
    quotient.resize(target);
    result.resize(target);
    for (int i = 0; i < target; i++) {
      result[i] = (result[i] + quotient[i]) * inverse_two;
    }
  }
  return result;
}

} // namespace noya

namespace noya {

template <class Mint>
std::pair<std::vector<Mint>, std::vector<Mint>>
factorials_and_inverses(int n) {
  std::vector<Mint> factorial(n + 1, Mint(1));
  std::vector<Mint> inverse_factorial(n + 1, Mint(1));
  for (int i = 1; i <= n; i++) {
    factorial[i] = factorial[i - 1] * Mint(i);
  }
  inverse_factorial[n] = Mint(1) / factorial[n];
  for (int i = n; i > 0; i--) {
    inverse_factorial[i - 1] = inverse_factorial[i] * Mint(i);
  }
  return {factorial, inverse_factorial};
}

/// @brief Return B_0 through B_n. Their exponential generating function is
/// exp(exp(x)-1), so one FPS exponential followed by factorial scaling yields
/// all Bell numbers simultaneously.
template <class Mint> std::vector<Mint> bell_numbers(int n) {
  auto [factorial, inverse_factorial] = factorials_and_inverses<Mint>(n);
  std::vector<Mint> exponent(n + 1);
  for (int i = 1; i <= n; i++) {
    exponent[i] = inverse_factorial[i];
  }
  auto result = fps_exponential(exponent, n + 1);
  for (int i = 0; i <= n; i++) {
    result[i] *= factorial[i];
  }
  return result;
}

/// @brief Return B_0 through B_n with B_1=-1/2. Since
/// x/(exp(x)-1)=1/(sum_{i>=0} x^i/(i+1)!), a series inverse and factorial
/// scaling produce all Bernoulli numbers.
template <class Mint> std::vector<Mint> bernoulli_numbers(int n) {
  auto [factorial, inverse_factorial] =
      factorials_and_inverses<Mint>(n + 1);
  std::vector<Mint> denominator(n + 1);
  for (int i = 0; i <= n; i++) {
    denominator[i] = inverse_factorial[i + 1];
  }
  auto result = fps_inverse(denominator, n + 1);
  for (int i = 0; i <= n; i++) {
    result[i] *= factorial[i];
  }
  return result;
}

/// @brief Return p(0) through p(n). Taking the logarithm of Euler's product
/// gives log P(x)=sum_{m>=1}(sum_{d|m}1/d)x^m; exponentiating this divisor-sum
/// series recovers the partition generating function.
template <class Mint> std::vector<Mint> partition_numbers(int n) {
  std::vector<Mint> inverse(n + 1);
  if (n >= 1) {
    inverse[1] = Mint(1);
  }
  for (int i = 2; i <= n; i++) {
    inverse[i] = Mint(1) / Mint(i);
  }
  std::vector<Mint> logarithm(n + 1);
  for (int part = 1; part <= n; part++) {
    for (int count = 1; part * count <= n; count++) {
      logarithm[part * count] += inverse[count];
    }
  }
  return fps_exponential(logarithm, n + 1);
}

namespace combinatorial_sequences_detail {

template <class Mint>
std::vector<Mint> consecutive_linear_product(int left, int right) {
  if (right - left == 0) {
    return {Mint(1)};
  }
  if (right - left == 1) {
    return {-Mint(left), Mint(1)};
  }
  int middle = (left + right) / 2;
  auto first = consecutive_linear_product<Mint>(left, middle);
  auto second = consecutive_linear_product<Mint>(middle, right);
  return atcoder::convolution(first, second);
}

} // namespace combinatorial_sequences_detail

/// @brief Return the signed first-kind Stirling row s(n,0..n) by building the
/// product x(x-1)...(x-n+1) with a balanced convolution tree.
template <class Mint>
std::vector<Mint> stirling_first_kind_row(int n) {
  return combinatorial_sequences_detail::consecutive_linear_product<Mint>(0,
                                                                            n);
}

/// @brief Return the second-kind Stirling row S(n,0..n). Expanding
/// S(n,k)=1/k! sum_i (-1)^(k-i) binom(k,i)i^n turns the whole row into one
/// convolution of the sequences (-1)^i/i! and i^n/i!.
template <class Mint>
std::vector<Mint> stirling_second_kind_row(int n) {
  auto [factorial, inverse_factorial] = factorials_and_inverses<Mint>(n);
  std::vector<Mint> signs(n + 1), powers(n + 1);
  for (int i = 0; i <= n; i++) {
    signs[i] = (i & 1) ? -inverse_factorial[i] : inverse_factorial[i];
    powers[i] = Mint(i).pow(n) * inverse_factorial[i];
  }
  auto result = atcoder::convolution(signs, powers);
  result.resize(n + 1);
  return result;
}

/// @brief Return s(k,k) through s(n,k). The exponential generating function
/// for a fixed column is log(1+x)^k/k!; coefficient extraction only requires
/// one FPS power and factorial scaling.
template <class Mint>
std::vector<Mint> stirling_first_kind_fixed_column(int n, int k) {
  auto [factorial, inverse_factorial] = factorials_and_inverses<Mint>(n);
  std::vector<Mint> logarithm(n + 1);
  for (int i = 1; i <= n; i++) {
    logarithm[i] = Mint(1) / Mint(i);
    if (i % 2 == 0) {
      logarithm[i] = -logarithm[i];
    }
  }
  auto series = fps_power(logarithm, std::uint64_t(k), n + 1);
  std::vector<Mint> result(n - k + 1);
  for (int i = k; i <= n; i++) {
    result[i - k] = series[i] * factorial[i] * inverse_factorial[k];
  }
  return result;
}

/// @brief Return S(k,k) through S(n,k). The fixed-column exponential
/// generating function is (exp(x)-1)^k/k!, so FPS exponentiation and power
/// followed by factorial scaling produce the column.
template <class Mint>
std::vector<Mint> stirling_second_kind_fixed_column(int n, int k) {
  auto [factorial, inverse_factorial] = factorials_and_inverses<Mint>(n);
  std::vector<Mint> exponential(n + 1);
  for (int i = 1; i <= n; i++) {
    exponential[i] = inverse_factorial[i];
  }
  auto series = fps_power(exponential, std::uint64_t(k), n + 1);
  std::vector<Mint> result(n - k + 1);
  for (int i = k; i <= n; i++) {
    result[i - k] = series[i] * factorial[i] * inverse_factorial[k];
  }
  return result;
}

} // namespace noya

/// @complexity Time: O(M(n) log n) evaluation/interpolation.
/// Space: O(n log n) product tree.

namespace noya {

namespace polynomial_multipoint_internal {

template <class Mint> struct product_tree {
  int point_count = 0;
  int size = 1;
  std::vector<std::vector<Mint>> product;

  explicit product_tree(const std::vector<Mint> &points)
      : point_count(int(points.size())) {
    while (size < point_count) {
      size *= 2;
    }
    product.assign(size * 2, std::vector<Mint>{Mint(1)});
    for (int i = 0; i < point_count; i++) {
      product[size + i] = {-points[i], Mint(1)};
    }
    for (int id = size - 1; id > 0; id--) {
      product[id] =
          atcoder::convolution(product[id * 2], product[id * 2 + 1]);
    }
  }

  std::vector<Mint> evaluate(const std::vector<Mint> &polynomial) const {
    if (point_count == 0) {
      return {};
    }
    std::vector<std::vector<Mint>> remainder(size * 2);
    remainder[1] = polynomial_divmod(polynomial, product[1]).second;
    for (int id = 1; id < size; id++) {
      remainder[id * 2] =
          polynomial_divmod(remainder[id], product[id * 2]).second;
      remainder[id * 2 + 1] =
          polynomial_divmod(remainder[id], product[id * 2 + 1]).second;
    }
    std::vector<Mint> result(point_count);
    for (int i = 0; i < point_count; i++) {
      if (!remainder[size + i].empty()) {
        result[i] = remainder[size + i][0];
      }
    }
    return result;
  }
};

template <class Mint>
std::vector<Mint> add_polynomials(std::vector<Mint> left,
                                  const std::vector<Mint> &right) {
  left.resize(std::max(left.size(), right.size()));
  for (int i = 0; i < int(right.size()); i++) {
    left[i] += right[i];
  }
  polynomial_trim(left);
  return left;
}

} // namespace polynomial_multipoint_internal

/// @brief Evaluate a polynomial at all points in O((n + degree) log^2 n)
/// field operations using a product tree.
template <class Mint>
std::vector<Mint>
polynomial_multipoint_evaluation(const std::vector<Mint> &polynomial,
                                const std::vector<Mint> &points) {
  return polynomial_multipoint_internal::product_tree<Mint>(points).evaluate(
      polynomial);
}

/// @brief Interpolate the unique degree < n polynomial through n distinct
/// points in O(n log^2 n) field operations.
template <class Mint>
std::vector<Mint> polynomial_interpolation(const std::vector<Mint> &points,
                                           const std::vector<Mint> &values) {
  assert(points.size() == values.size());
  int n = int(points.size());
  if (n == 0) {
    return {};
  }
  polynomial_multipoint_internal::product_tree<Mint> tree(points);
  std::vector<Mint> derivative = polynomial_derivative(tree.product[1]);
  std::vector<Mint> denominators = tree.evaluate(derivative);
  std::vector<std::vector<Mint>> interpolation(tree.size * 2);
  for (int i = 0; i < n; i++) {
    assert(denominators[i] != Mint{});
    interpolation[tree.size + i] = {values[i] / denominators[i]};
  }
  for (int i = n; i < tree.size; i++) {
    interpolation[tree.size + i] = {};
  }
  using polynomial_multipoint_internal::add_polynomials;
  for (int id = tree.size - 1; id > 0; id--) {
    std::vector<Mint> left = atcoder::convolution(
        interpolation[id * 2], tree.product[id * 2 + 1]);
    std::vector<Mint> right = atcoder::convolution(
        interpolation[id * 2 + 1], tree.product[id * 2]);
    interpolation[id] = add_polynomials(std::move(left), right);
  }
  interpolation[1].resize(n);
  polynomial_trim(interpolation[1]);
  return interpolation[1];
}

} // namespace noya

/// @complexity Time: O(M(n + m)) for consecutive or geometric evaluation and
/// O(M(n)) for geometric interpolation, where M(n) is convolution time.
/// Space: O(n + m).

/// @complexity Time: O(n) field operations.
/// Space: O(n).

namespace noya {

/// @brief Invert a list of nonzero field elements with one division and O(n)
/// multiplications.
template <class T> std::vector<T> batch_inverse(const std::vector<T> &values) {
  std::vector<T> prefix(values.size() + 1, T(1));
  for (int index = 0; index < int(values.size()); index++) {
    assert(values[index] != T{});
    prefix[index + 1] = prefix[index] * values[index];
  }
  T suffix_inverse = T(1) / prefix.back();
  std::vector<T> result(values.size());
  for (int index = int(values.size()) - 1; index >= 0; index--) {
    result[index] = prefix[index] * suffix_inverse;
    suffix_inverse *= values[index];
  }
  return result;
}

} // namespace noya

namespace noya {

namespace polynomial_special_points_detail {

template <class Mint>
std::vector<Mint> factorial_inverses(int size) {
  std::vector<Mint> factorial(size, Mint(1));
  for (int i = 1; i < size; i++) {
    factorial[i] = factorial[i - 1] * Mint(i);
  }
  std::vector<Mint> inverse_factorial(size, Mint(1));
  if (size > 0) {
    inverse_factorial.back() = Mint(1) / factorial.back();
    for (int i = size - 1; i > 0; i--) {
      inverse_factorial[i - 1] = inverse_factorial[i] * Mint(i);
    }
  }
  return inverse_factorial;
}

template <class Mint>
std::vector<Mint> inverses_allowing_zero(const std::vector<Mint> &values) {
  std::vector<Mint> nonzero;
  nonzero.reserve(values.size());
  for (Mint value : values) {
    if (value != Mint{}) {
      nonzero.push_back(value);
    }
  }
  std::vector<Mint> inverted = batch_inverse(nonzero);
  std::vector<Mint> result(values.size());
  int at = 0;
  for (int i = 0; i < int(values.size()); i++) {
    if (values[i] != Mint{}) {
      result[i] = inverted[at++];
    }
  }
  return result;
}

} // namespace polynomial_special_points_detail

/// @brief Recover f(c),...,f(c+m-1) from f(0),...,f(n-1). Lagrange weights
/// turn every non-sampled value into one convolution with 1/(c+k-i); a
/// sliding product supplies prod_j(c+k-j). Positions that coincide with an
/// original sample are copied directly.
template <class Mint>
std::vector<Mint> polynomial_shift_samples(const std::vector<Mint> &samples,
                                           Mint c, int m) {
  assert(m >= 0);
  int n = int(samples.size());
  if (m == 0) {
    return {};
  }
  assert(n > 0);
  auto inverse_factorial =
      polynomial_special_points_detail::factorial_inverses<Mint>(n);

  std::vector<Mint> weights(n);
  for (int i = 0; i < n; i++) {
    weights[i] = samples[i] * inverse_factorial[i] *
                 inverse_factorial[n - 1 - i];
    if ((n - 1 - i) & 1) {
      weights[i] = -weights[i];
    }
  }
  std::vector<Mint> differences(n + m - 1);
  for (int offset = 1 - n; offset < m; offset++) {
    differences[offset + n - 1] = c + Mint(offset);
  }
  auto inverse_difference =
      polynomial_special_points_detail::inverses_allowing_zero(differences);
  std::vector<Mint> convolution =
      atcoder::convolution(weights, inverse_difference);

  std::vector<int> coincident_sample(m, -1);
  for (int position = 0; position < int(differences.size()); position++) {
    if (differences[position] != Mint{}) {
      continue;
    }
    int first_output = std::max(0, position - n + 1);
    int last_output = std::min(m - 1, position);
    for (int k = first_output; k <= last_output; k++) {
      coincident_sample[k] = k + n - 1 - position;
    }
  }

  int zero_count = 0;
  Mint nonzero_product = 1;
  for (int i = 0; i < n; i++) {
    Mint value = differences[n - 1 - i];
    if (value == Mint{}) {
      zero_count++;
    } else {
      nonzero_product *= value;
    }
  }

  std::vector<Mint> result(m);
  for (int k = 0; k < m; k++) {
    if (zero_count > 0) {
      int coincident = coincident_sample[k];
      assert(coincident >= 0);
      result[k] = samples[coincident];
    } else {
      result[k] = nonzero_product * convolution[k + n - 1];
    }
    if (k + 1 == m) {
      break;
    }
    Mint removed = differences[k];
    if (removed == Mint{}) {
      zero_count--;
    } else {
      nonzero_product *= inverse_difference[k];
    }
    Mint added = differences[k + n];
    if (added == Mint{}) {
      zero_count++;
    } else {
      nonzero_product *= added;
    }
  }
  return result;
}

/// @brief Evaluate f(a r^k) for k=0..m-1. Writing
/// r^(ik)=q(i+k)/(q(i)q(k)), q(t)=r^(t(t-1)/2), changes the Hankel product
/// into one ordinary convolution.
template <class Mint>
std::vector<Mint>
polynomial_evaluate_geometric(const std::vector<Mint> &polynomial, int m,
                              Mint a, Mint r) {
  assert(m >= 0);
  int n = int(polynomial.size());
  if (m == 0) {
    return {};
  }
  if (n == 0) {
    return std::vector<Mint>(m);
  }
  if (r == Mint{}) {
    std::vector<Mint> result(m, polynomial[0]);
    Mint value = 0;
    for (int i = n - 1; i >= 0; i--) {
      value = value * a + polynomial[i];
    }
    result[0] = value;
    return result;
  }

  std::vector<Mint> q(n + m);
  q[0] = 1;
  Mint power = 1;
  for (int i = 1; i < int(q.size()); i++) {
    q[i] = q[i - 1] * power;
    power *= r;
  }
  std::vector<Mint> inverse_q(q.begin(), q.begin() + std::max(n, m));
  inverse_q = batch_inverse(inverse_q);
  std::vector<Mint> left(n);
  Mint a_power = 1;
  for (int i = 0; i < n; i++) {
    left[n - 1 - i] = polynomial[i] * a_power * inverse_q[i];
    a_power *= a;
  }
  std::vector<Mint> product = atcoder::convolution(left, q);
  std::vector<Mint> result(m);
  for (int k = 0; k < m; k++) {
    result[k] = product[n - 1 + k] * inverse_q[k];
  }
  return result;
}

/// @brief Interpolate from the distinct points a,ar,...,ar^(n-1). Closed
/// forms for the product polynomial and its derivatives give barycentric
/// weights in linear time; a geometric evaluation computes all needed
/// moments, and one final convolution recovers the monomial coefficients.
template <class Mint>
std::vector<Mint>
polynomial_interpolate_geometric(const std::vector<Mint> &values, Mint a,
                                 Mint r) {
  int n = int(values.size());
  if (n == 0) {
    return {};
  }
  if (n == 1) {
    return values;
  }
  assert(a != Mint{});
  assert(r != Mint{});

  std::vector<Mint> one_minus_power(n);
  Mint r_power = r;
  for (int i = 1; i <= n; i++) {
    one_minus_power[i - 1] = Mint(1) - r_power;
    if (i < n) {
      assert(one_minus_power[i - 1] != Mint{});
    }
    r_power *= r;
  }
  std::vector<Mint> invertible_one_minus(one_minus_power.begin(),
                                         one_minus_power.end() - 1);
  auto inverse_one_minus = batch_inverse(invertible_one_minus);
  std::vector<Mint> prefix(n, Mint(1));
  for (int i = 1; i < n; i++) {
    prefix[i] = prefix[i - 1] * one_minus_power[i - 1];
  }

  std::vector<Mint> a_powers(n, Mint(1));
  std::vector<Mint> r_powers(n, Mint(1));
  for (int i = 1; i < n; i++) {
    a_powers[i] = a_powers[i - 1] * a;
    r_powers[i] = r_powers[i - 1] * r;
  }
  auto inverse_a_powers = batch_inverse(a_powers);
  std::vector<Mint> weights(n);
  for (int i = 0; i < n; i++) {
    long long exponent = 1LL * i * (i - 1) / 2 + 1LL * i * (n - 1 - i);
    Mint inverse_r_exponent = Mint(1);
    if (exponent > 0) {
      Mint base = Mint(1) / r;
      while (exponent > 0) {
        if (exponent & 1) {
          inverse_r_exponent *= base;
        }
        base *= base;
        exponent >>= 1;
      }
    }
    Mint derivative_inverse = inverse_a_powers[n - 1] *
                              inverse_r_exponent /
                              (prefix[i] * prefix[n - 1 - i]);
    if (i & 1) {
      derivative_inverse = -derivative_inverse;
    }
    weights[i] = values[i] * derivative_inverse;
  }

  std::vector<Mint> moments =
      polynomial_evaluate_geometric(weights, n, Mint(1), r);
  for (int i = 0; i < n; i++) {
    moments[i] *= a_powers[i];
  }

  std::vector<Mint> q_binomial(n + 1, Mint(1));
  for (int t = 1; t < n; t++) {
    q_binomial[t] = q_binomial[t - 1] * one_minus_power[n - t] *
                    inverse_one_minus[t - 1];
  }
  std::vector<Mint> product_polynomial(n + 1);
  Mint minus_a_power = 1;
  Mint q_factor = 1;
  Mint q_step = 1;
  for (int t = 0; t <= n; t++) {
    int coefficient = n - t;
    product_polynomial[coefficient] = q_binomial[t] * minus_a_power * q_factor;
    minus_a_power *= -a;
    q_factor *= q_step;
    q_step *= r;
  }

  std::vector<Mint> reversed_product(n);
  for (int i = 0; i < n; i++) {
    reversed_product[i] = product_polynomial[n - i];
  }
  std::vector<Mint> convolution =
      atcoder::convolution(reversed_product, moments);
  std::vector<Mint> result(n);
  for (int k = 0; k < n; k++) {
    result[k] = convolution[n - 1 - k];
  }
  return result;
}

/// @brief Multiply a sequence of polynomials by always combining the two
/// currently shortest factors. The Huffman-style merge order keeps the total
/// convolution work O(M(D) log n), where D is the final degree.
template <class Mint>
std::vector<Mint>
polynomial_product_sequence(std::vector<std::vector<Mint>> factors) {
  using item = std::pair<int, int>;
  std::priority_queue<item, std::vector<item>, std::greater<item>> queue;
  for (int i = 0; i < int(factors.size()); i++) {
    queue.emplace(int(factors[i].size()), i);
  }
  if (queue.empty()) {
    return {Mint(1)};
  }
  while (queue.size() > 1) {
    int first = queue.top().second;
    queue.pop();
    int second = queue.top().second;
    queue.pop();
    factors.push_back(atcoder::convolution(factors[first], factors[second]));
    queue.emplace(int(factors.back().size()), int(factors.size()) - 1);
  }
  return factors[queue.top().second];
}

} // namespace noya

namespace noya {

namespace large_factorial_detail {

template <class Mint>
std::vector<Mint> factorial_block_prefix(std::uint64_t maximum, int block) {
  int full_blocks = int(maximum / block);
  std::vector<Mint> prefix(full_blocks + 1, Mint(1));
  if (full_blocks == 0) {
    return prefix;
  }

  std::vector<std::vector<Mint>> factors;
  factors.reserve(block);
  for (int i = 1; i <= block; i++) {
    factors.push_back({Mint(i), Mint(1)});
  }
  std::vector<Mint> block_polynomial =
      polynomial_product_sequence(std::move(factors));
  std::vector<Mint> points(full_blocks);
  for (int i = 0; i < full_blocks; i++) {
    points[i] = Mint(std::uint64_t(i) * block);
  }
  std::vector<Mint> products =
      polynomial_multipoint_evaluation(block_polynomial, points);
  for (int i = 0; i < full_blocks; i++) {
    prefix[i + 1] = prefix[i] * products[i];
  }
  return prefix;
}

} // namespace large_factorial_detail

/// @brief Compute several factorials modulo a fixed prime without a linear
/// table. Split 1..N into blocks of length B about sqrt(N). The product inside
/// one block is the degree-B polynomial P(x)=prod_{i=1}^B(x+i); a product tree
/// constructs P and multipoint evaluation obtains P(0),P(B),P(2B),.... Prefix
/// products answer every full block, followed by at most B direct factors.
template <class Mint>
std::vector<Mint>
large_factorials(const std::vector<std::uint64_t> &queries) {
  if (queries.empty()) {
    return {};
  }
  std::uint64_t maximum =
      *std::max_element(queries.begin(), queries.end());
  assert(maximum < std::uint64_t(Mint::mod()));
  int block = int(std::sqrt(static_cast<long double>(maximum + 1)));
  block = std::max(block, 1);
  while (std::uint64_t(block) * block < maximum + 1) {
    block++;
  }
  std::vector<Mint> prefix =
      large_factorial_detail::factorial_block_prefix<Mint>(maximum, block);

  std::vector<Mint> result;
  result.reserve(queries.size());
  for (std::uint64_t n : queries) {
    std::uint64_t completed = n / block;
    Mint value = prefix[completed];
    for (std::uint64_t i = completed * block + 1; i <= n; i++) {
      value *= Mint(i);
    }
    result.push_back(value);
  }
  return result;
}

/// @brief Compute a large batch of factorials modulo a fixed prime. Boundary
/// values (kB)! are obtained by evaluating the block-product polynomial
/// prod_{i=1}^B(x+i). For a query n=qB+r, split the remaining product
/// (qB+1)...n into power-of-two suffixes. A suffix of length 2^b is the falling
/// factorial polynomial x(x-1)...(x-2^b+1) evaluated at its current right
/// endpoint. Queries sharing b are evaluated in batches of at most 2^b points,
/// so no query performs a linear tail scan.
template <class Mint>
std::vector<Mint>
many_factorials(const std::vector<std::uint32_t> &queries) {
  if (queries.empty()) {
    return {};
  }
  constexpr int log_block = 15;
  constexpr int block = 1 << log_block;
  std::uint32_t maximum =
      *std::max_element(queries.begin(), queries.end());
  assert(maximum < std::uint32_t(Mint::mod()));

  std::vector<Mint> prefix =
      large_factorial_detail::factorial_block_prefix<Mint>(maximum, block);
  std::vector<std::vector<std::pair<Mint, int>>> evaluation_points(log_block);
  std::vector<Mint> result(queries.size());
  for (int query = 0; query < int(queries.size()); query++) {
    std::uint32_t n = queries[query];
    int quotient = int(n / block);
    int remainder = int(n % block);
    result[query] = prefix[quotient];
    std::uint32_t endpoint = n;
    for (int bit = 0; bit < log_block; bit++) {
      if ((remainder >> bit) & 1) {
        evaluation_points[bit].emplace_back(Mint(endpoint), query);
        endpoint -= std::uint32_t(1) << bit;
      }
    }
    assert(endpoint == std::uint32_t(quotient * block));
  }

  for (int bit = 0; bit < log_block; bit++) {
    auto &items = evaluation_points[bit];
    if (items.empty()) {
      continue;
    }
    int length = 1 << bit;
    std::vector<Mint> falling_factorial =
        stirling_first_kind_row<Mint>(length);
    for (int left = 0; left < int(items.size()); left += length) {
      int right = std::min(left + length, int(items.size()));
      std::vector<Mint> points;
      points.reserve(right - left);
      for (int index = left; index < right; index++) {
        points.push_back(items[index].first);
      }
      std::vector<Mint> values = polynomial_multipoint_evaluation(
          falling_factorial, points);
      for (int index = left; index < right; index++) {
        result[items[index].second] *= values[index - left];
      }
    }
  }
  return result;
}

} // namespace noya