Skip to content

big_integer_multiplication.hpp

SECTIONMath INCLUDEnoya/big_integer_multiplication.hpp

用卷积乘任意长有符号整数,支持 2 到 36 进制输入输出。

\[ \displaystyle C = A B,\qquad A=\sum_{i=0}^{n-1}a_i B^i,\qquad B=\sum_{j=0}^{m-1}b_j B^j,\qquad C=\sum_{k=0}^{n+m-2}\left(\sum_{i+j=k}a_i b_j\right)B^k \]

Complexity: Time: O(n log n), where n is the larger digit count. Space: O(n).

AC 记录:multiplication_of_big_integers, multiplication_of_hex_big_integers

跳到代码 · GitHub ↗

Implementation

当前头文件,省略 include guard;依赖见 #include

/// @complexity Time: O(n log n), where n is the larger digit count.
/// Space: O(n).

#include "noya/big_integer_addition.hpp"
#include "noya/convolution_mod.hpp"

#include <algorithm>
#include <cassert>
#include <cstdint>
#include <string>
#include <string_view>
#include <vector>

namespace noya {

namespace big_integer_multiplication_internal {

inline std::vector<std::uint64_t> read_limbs(std::string_view mag, int bas,
                                             int dig) {
  std::vector<std::uint64_t> lim;
  for (std::size_t r = mag.size(); r > 0;) {
    std::size_t l = r > std::size_t(dig) ? r - dig : 0;
    std::uint64_t val = 0;
    for (std::size_t pos = l; pos < r; pos++) {
      val = val * bas + big_integer_addition_internal::digit_value(mag[pos]);
    }
    lim.push_back(val);
    r = l;
  }
  return lim;
}

inline std::vector<std::uint64_t>
multiply_limbs(const std::vector<std::uint64_t> &a,
               const std::vector<std::uint64_t> &b, std::uint64_t bs) {
  if (std::min(a.size(), b.size()) <= 32) {
    std::vector<std::uint64_t> res(a.size() + b.size());
    for (int l = 0; l < int(a.size()); l++) {
      unsigned __int128 car = 0;
      for (int r = 0; r < int(b.size()); r++) {
        unsigned __int128 val =
            res[l + r] + car + static_cast<unsigned __int128>(a[l]) * b[r];
        res[l + r] = std::uint64_t(val % bs);
        car = val / bs;
      }
      int pos = l + int(b.size());
      while (car > 0) {
        unsigned __int128 val = res[pos] + car;
        res[pos] = std::uint64_t(val % bs);
        car = val / bs;
        pos++;
      }
    }
    return res;
  }

  std::uint64_t ma = *std::max_element(a.begin(), a.end());
  std::uint64_t mb = *std::max_element(b.begin(), b.end());
  unsigned __int128 cb =
      static_cast<unsigned __int128>(std::min(a.size(), b.size())) * ma * mb;
  assert(cb < static_cast<unsigned __int128>(UINT64_MAX));
  // Every raw coefficient is below this modulus, so reduction after CRT does
  // not discard information.
  std::uint64_t mod = std::uint64_t(cb) + 1;
  std::vector<std::uint64_t> prd = convolution_mod(a, b, mod);
  std::uint64_t car = 0;
  for (std::uint64_t &val : prd) {
    unsigned __int128 cur = static_cast<unsigned __int128>(val) + car;
    val = std::uint64_t(cur % bs);
    car = std::uint64_t(cur / bs);
  }
  while (car > 0) {
    prd.push_back(car % bs);
    car /= bs;
  }
  return prd;
}

inline std::string write_limbs(std::vector<std::uint64_t> lim, int bas, int dig,
                               bool cap) {
  while (lim.size() > 1 && lim.back() == 0) {
    lim.pop_back();
  }
  std::string res;
  res.reserve(lim.size() * dig);
  for (int idx = int(lim.size()) - 1; idx >= 0; idx--) {
    std::uint64_t val = lim[idx];
    std::string blk;
    do {
      blk.push_back(
          big_integer_addition_internal::digit_character(int(val % bas), cap));
      val /= bas;
    } while (val > 0);
    if (idx + 1 != int(lim.size())) {
      blk.resize(dig, '0');
    }
    std::reverse(blk.begin(), blk.end());
    res += blk;
  }
  return res;
}

} // namespace big_integer_multiplication_internal

/// @brief Multiply two arbitrarily long signed integers in base 2..36. Digits
/// are packed into limbs whose square fits exact three-prime CRT convolution;
/// the product coefficients are then normalized by a linear carry pass.
inline std::string multiply_big_integers(std::string_view a, std::string_view b,
                                         int bas = 10, bool cap = false) {
  using namespace big_integer_addition_internal;
  using namespace big_integer_multiplication_internal;
  parsed_integer l = parse(a, bas);
  parsed_integer r = parse(b, bas);
  if (l.mag == "0" || r.mag == "0") {
    return "0";
  }

  constexpr std::uint64_t mbs = 10'000;
  int dig = 1;
  std::uint64_t bs = bas;
  while (bs <= mbs / std::uint64_t(bas)) {
    bs *= bas;
    dig++;
  }
  auto ll = read_limbs(l.mag, bas, dig);
  auto rl = read_limbs(r.mag, bas, dig);
  std::string res = write_limbs(multiply_limbs(ll, rl, bs), bas, dig, cap);
  if (l.neg != r.neg) {
    res.insert(res.begin(), '-');
  }
  return res;
}

} // namespace noya
#ifndef NOYA_BIG_INTEGER_MULTIPLICATION_HPP
#define NOYA_BIG_INTEGER_MULTIPLICATION_HPP 1

/// @complexity Time: O(n log n), where n is the larger digit count.
/// Space: O(n).

#include "noya/big_integer_addition.hpp"
#include "noya/convolution_mod.hpp"

#include <algorithm>
#include <cassert>
#include <cstdint>
#include <string>
#include <string_view>
#include <vector>

namespace noya {

namespace big_integer_multiplication_internal {

inline std::vector<std::uint64_t> read_limbs(std::string_view mag, int bas,
                                             int dig) {
  std::vector<std::uint64_t> lim;
  for (std::size_t r = mag.size(); r > 0;) {
    std::size_t l = r > std::size_t(dig) ? r - dig : 0;
    std::uint64_t val = 0;
    for (std::size_t pos = l; pos < r; pos++) {
      val = val * bas + big_integer_addition_internal::digit_value(mag[pos]);
    }
    lim.push_back(val);
    r = l;
  }
  return lim;
}

inline std::vector<std::uint64_t>
multiply_limbs(const std::vector<std::uint64_t> &a,
               const std::vector<std::uint64_t> &b, std::uint64_t bs) {
  if (std::min(a.size(), b.size()) <= 32) {
    std::vector<std::uint64_t> res(a.size() + b.size());
    for (int l = 0; l < int(a.size()); l++) {
      unsigned __int128 car = 0;
      for (int r = 0; r < int(b.size()); r++) {
        unsigned __int128 val =
            res[l + r] + car + static_cast<unsigned __int128>(a[l]) * b[r];
        res[l + r] = std::uint64_t(val % bs);
        car = val / bs;
      }
      int pos = l + int(b.size());
      while (car > 0) {
        unsigned __int128 val = res[pos] + car;
        res[pos] = std::uint64_t(val % bs);
        car = val / bs;
        pos++;
      }
    }
    return res;
  }

  std::uint64_t ma = *std::max_element(a.begin(), a.end());
  std::uint64_t mb = *std::max_element(b.begin(), b.end());
  unsigned __int128 cb =
      static_cast<unsigned __int128>(std::min(a.size(), b.size())) * ma * mb;
  assert(cb < static_cast<unsigned __int128>(UINT64_MAX));
  // Every raw coefficient is below this modulus, so reduction after CRT does
  // not discard information.
  std::uint64_t mod = std::uint64_t(cb) + 1;
  std::vector<std::uint64_t> prd = convolution_mod(a, b, mod);
  std::uint64_t car = 0;
  for (std::uint64_t &val : prd) {
    unsigned __int128 cur = static_cast<unsigned __int128>(val) + car;
    val = std::uint64_t(cur % bs);
    car = std::uint64_t(cur / bs);
  }
  while (car > 0) {
    prd.push_back(car % bs);
    car /= bs;
  }
  return prd;
}

inline std::string write_limbs(std::vector<std::uint64_t> lim, int bas, int dig,
                               bool cap) {
  while (lim.size() > 1 && lim.back() == 0) {
    lim.pop_back();
  }
  std::string res;
  res.reserve(lim.size() * dig);
  for (int idx = int(lim.size()) - 1; idx >= 0; idx--) {
    std::uint64_t val = lim[idx];
    std::string blk;
    do {
      blk.push_back(
          big_integer_addition_internal::digit_character(int(val % bas), cap));
      val /= bas;
    } while (val > 0);
    if (idx + 1 != int(lim.size())) {
      blk.resize(dig, '0');
    }
    std::reverse(blk.begin(), blk.end());
    res += blk;
  }
  return res;
}

} // namespace big_integer_multiplication_internal

/// @brief Multiply two arbitrarily long signed integers in base 2..36. Digits
/// are packed into limbs whose square fits exact three-prime CRT convolution;
/// the product coefficients are then normalized by a linear carry pass.
inline std::string multiply_big_integers(std::string_view a, std::string_view b,
                                         int bas = 10, bool cap = false) {
  using namespace big_integer_addition_internal;
  using namespace big_integer_multiplication_internal;
  parsed_integer l = parse(a, bas);
  parsed_integer r = parse(b, bas);
  if (l.mag == "0" || r.mag == "0") {
    return "0";
  }

  constexpr std::uint64_t mbs = 10'000;
  int dig = 1;
  std::uint64_t bs = bas;
  while (bs <= mbs / std::uint64_t(bas)) {
    bs *= bas;
    dig++;
  }
  auto ll = read_limbs(l.mag, bas, dig);
  auto rl = read_limbs(r.mag, bas, dig);
  std::string res = write_limbs(multiply_limbs(ll, rl, bs), bas, dig, cap);
  if (l.neg != r.neg) {
    res.insert(res.begin(), '-');
  }
  return res;
}

} // namespace noya

#endif // NOYA_BIG_INTEGER_MULTIPLICATION_HPP
#include <algorithm>
#include <array>
#include <cassert>
#include <cstdint>
#include <numeric>
#include <string>
#include <string_view>
#include <type_traits>
#include <utility>
#include <vector>

/// @complexity Time: O(n log n), where n is the larger digit count.
/// Space: O(n).

/// @complexity Time: O(|a| + |b|).
/// Space: O(max(|a|, |b|)) for the returned representation.

namespace noya {

namespace big_integer_addition_internal {

struct parsed_integer {
  bool neg = false;
  std::string_view mag;
};

inline int digit_value(char dig) {
  if ('0' <= dig && dig <= '9') {
    return dig - '0';
  }
  if ('a' <= dig && dig <= 'z') {
    return dig - 'a' + 10;
  }
  if ('A' <= dig && dig <= 'Z') {
    return dig - 'A' + 10;
  }
  return -1;
}

inline parsed_integer parse(std::string_view val, int bas) {
  assert(2 <= bas && bas <= 36);
  assert(!val.empty());
  bool neg = val.front() == '-';
  if (neg || val.front() == '+') {
    val.remove_prefix(1);
  }
  assert(!val.empty());
  for (char dig : val) {
    assert(0 <= digit_value(dig) && digit_value(dig) < bas);
  }
  while (val.size() > 1 && val.front() == '0') {
    val.remove_prefix(1);
  }
  if (val == "0") {
    neg = false;
  }
  return {neg, val};
}

inline int compare_magnitude(std::string_view a, std::string_view b) {
  if (a.size() != b.size()) {
    return a.size() < b.size() ? -1 : 1;
  }
  for (std::size_t idx = 0; idx < a.size(); idx++) {
    int l = digit_value(a[idx]);
    int r = digit_value(b[idx]);
    if (l != r) {
      return l < r ? -1 : 1;
    }
  }
  return 0;
}

inline char digit_character(int val, bool cap) {
  assert(0 <= val && val < 36);
  if (val < 10) {
    return char('0' + val);
  }
  return char((cap ? 'A' : 'a') + val - 10);
}

inline std::string add_magnitudes(std::string_view a, std::string_view b,
                                  int bas, bool cap) {
  std::string res;
  res.reserve(std::max(a.size(), b.size()) + 1);
  int car = 0;
  std::size_t l = a.size();
  std::size_t r = b.size();
  while (l > 0 || r > 0 || car != 0) {
    int val = car;
    if (l > 0) {
      val += digit_value(a[--l]);
    }
    if (r > 0) {
      val += digit_value(b[--r]);
    }
    res.push_back(digit_character(val % bas, cap));
    car = val / bas;
  }
  std::reverse(res.begin(), res.end());
  return res;
}

// Precondition: a >= b as unsigned magnitudes.
inline std::string subtract_magnitudes(std::string_view a, std::string_view b,
                                       int bas, bool cap) {
  std::string res;
  res.reserve(a.size());
  int bor = 0;
  std::size_t l = a.size();
  std::size_t r = b.size();
  while (l > 0) {
    int val = digit_value(a[--l]) - bor;
    if (r > 0) {
      val -= digit_value(b[--r]);
    }
    if (val < 0) {
      val += bas;
      bor = 1;
    } else {
      bor = 0;
    }
    res.push_back(digit_character(val, cap));
  }
  assert(bor == 0);
  while (res.size() > 1 && res.back() == '0') {
    res.pop_back();
  }
  std::reverse(res.begin(), res.end());
  return res;
}

} // namespace big_integer_addition_internal

/// @brief Add two arbitrarily long signed integers represented in base 2..36.
/// The result is canonical (no leading zeroes and no negative zero).
inline std::string add_big_integers(std::string_view a, std::string_view b,
                                    int bas = 10, bool cap = false) {
  using namespace big_integer_addition_internal;
  parsed_integer l = parse(a, bas);
  parsed_integer r = parse(b, bas);
  if (l.neg == r.neg) {
    std::string res = add_magnitudes(l.mag, r.mag, bas, cap);
    if (l.neg) {
      res.insert(res.begin(), '-');
    }
    return res;
  }
  int ord = compare_magnitude(l.mag, r.mag);
  if (ord == 0) {
    return "0";
  }
  bool neg = ord > 0 ? l.neg : r.neg;
  std::string res = ord > 0 ? subtract_magnitudes(l.mag, r.mag, bas, cap)
                            : subtract_magnitudes(r.mag, l.mag, bas, cap);
  if (neg) {
    res.insert(res.begin(), '-');
  }
  return res;
}

} // namespace noya

/// @complexity Time: O(n log n) for result length n.
/// Space: O(n).

#ifdef _MSC_VER
#include <intrin.h>
#endif

#if __cplusplus >= 202002L
#include <bit>
#endif

namespace atcoder {

namespace internal {

#if __cplusplus >= 202002L

using std::bit_ceil;

#else

// @return same with std::bit::bit_ceil
unsigned int bit_ceil(unsigned int n) {
    unsigned int x = 1;
    while (x < (unsigned int)(n)) x *= 2;
    return x;
}

#endif

// @param n `1 <= n`
// @return same with std::bit::countr_zero
int countr_zero(unsigned int n) {
#ifdef _MSC_VER
    unsigned long index;
    _BitScanForward(&index, n);
    return index;
#else
    return __builtin_ctz(n);
#endif
}

// @param n `1 <= n`
// @return same with std::bit::countr_zero
constexpr int countr_zero_constexpr(unsigned int n) {
    int x = 0;
    while (!(n & (1 << x))) x++;
    return x;
}

}  // namespace internal

}  // namespace atcoder

#ifdef _MSC_VER
#include <intrin.h>
#endif

#ifdef _MSC_VER
#include <intrin.h>
#endif

namespace atcoder {

namespace internal {

// @param m `1 <= m`
// @return x mod m
constexpr long long safe_mod(long long x, long long m) {
    x %= m;
    if (x < 0) x += m;
    return x;
}

// Fast modular multiplication by barrett reduction
// Reference: https://en.wikipedia.org/wiki/Barrett_reduction
// NOTE: reconsider after Ice Lake
struct barrett {
    unsigned int _m;
    unsigned long long im;

    // @param m `1 <= m`
    explicit barrett(unsigned int m) : _m(m), im((unsigned long long)(-1) / m + 1) {}

    // @return m
    unsigned int umod() const { return _m; }

    // @param a `0 <= a < m`
    // @param b `0 <= b < m`
    // @return `a * b % m`
    unsigned int mul(unsigned int a, unsigned int b) const {
        // [1] m = 1
        // a = b = im = 0, so okay

        // [2] m >= 2
        // im = ceil(2^64 / m)
        // -> im * m = 2^64 + r (0 <= r < m)
        // let z = a*b = c*m + d (0 <= c, d < m)
        // a*b * im = (c*m + d) * im = c*(im*m) + d*im = c*2^64 + c*r + d*im
        // c*r + d*im < m * m + m * im < m * m + 2^64 + m <= 2^64 + m * (m + 1) < 2^64 * 2
        // ((ab * im) >> 64) == c or c + 1
        unsigned long long z = a;
        z *= b;
#ifdef _MSC_VER
        unsigned long long x;
        _umul128(z, im, &x);
#else
        unsigned long long x =
            (unsigned long long)(((unsigned __int128)(z)*im) >> 64);
#endif
        unsigned long long y = x * _m;
        return (unsigned int)(z - y + (z < y ? _m : 0));
    }
};

// @param n `0 <= n`
// @param m `1 <= m`
// @return `(x ** n) % m`
constexpr long long pow_mod_constexpr(long long x, long long n, int m) {
    if (m == 1) return 0;
    unsigned int _m = (unsigned int)(m);
    unsigned long long r = 1;
    unsigned long long y = safe_mod(x, m);
    while (n) {
        if (n & 1) r = (r * y) % _m;
        y = (y * y) % _m;
        n >>= 1;
    }
    return r;
}

// Reference:
// M. Forisek and J. Jancina,
// Fast Primality Testing for Integers That Fit into a Machine Word
// @param n `0 <= n`
constexpr bool is_prime_constexpr(int n) {
    if (n <= 1) return false;
    if (n == 2 || n == 7 || n == 61) return true;
    if (n % 2 == 0) return false;
    long long d = n - 1;
    while (d % 2 == 0) d /= 2;
    constexpr long long bases[3] = {2, 7, 61};
    for (long long a : bases) {
        long long t = d;
        long long y = pow_mod_constexpr(a, t, n);
        while (t != n - 1 && y != 1 && y != n - 1) {
            y = y * y % n;
            t <<= 1;
        }
        if (y != n - 1 && t % 2 == 0) {
            return false;
        }
    }
    return true;
}
template <int n> constexpr bool is_prime = is_prime_constexpr(n);

// @param b `1 <= b`
// @return pair(g, x) s.t. g = gcd(a, b), xa = g (mod b), 0 <= x < b/g
constexpr std::pair<long long, long long> inv_gcd(long long a, long long b) {
    a = safe_mod(a, b);
    if (a == 0) return {b, 0};

    // Contracts:
    // [1] s - m0 * a = 0 (mod b)
    // [2] t - m1 * a = 0 (mod b)
    // [3] s * |m1| + t * |m0| <= b
    long long s = b, t = a;
    long long m0 = 0, m1 = 1;

    while (t) {
        long long u = s / t;
        s -= t * u;
        m0 -= m1 * u;  // |m1 * u| <= |m1| * s <= b

        // [3]:
        // (s - t * u) * |m1| + t * |m0 - m1 * u|
        // <= s * |m1| - t * u * |m1| + t * (|m0| + |m1| * u)
        // = s * |m1| + t * |m0| <= b

        auto tmp = s;
        s = t;
        t = tmp;
        tmp = m0;
        m0 = m1;
        m1 = tmp;
    }
    // by [3]: |m0| <= b/g
    // by g != b: |m0| < b/g
    if (m0 < 0) m0 += b / s;
    return {s, m0};
}

// Compile time primitive root
// @param m must be prime
// @return primitive root (and minimum in now)
constexpr int primitive_root_constexpr(int m) {
    if (m == 2) return 1;
    if (m == 167772161) return 3;
    if (m == 469762049) return 3;
    if (m == 754974721) return 11;
    if (m == 998244353) return 3;
    int divs[20] = {};
    divs[0] = 2;
    int cnt = 1;
    int x = (m - 1) / 2;
    while (x % 2 == 0) x /= 2;
    for (int i = 3; (long long)(i)*i <= x; i += 2) {
        if (x % i == 0) {
            divs[cnt++] = i;
            while (x % i == 0) {
                x /= i;
            }
        }
    }
    if (x > 1) {
        divs[cnt++] = x;
    }
    for (int g = 2;; g++) {
        bool ok = true;
        for (int i = 0; i < cnt; i++) {
            if (pow_mod_constexpr(g, (m - 1) / divs[i], m) == 1) {
                ok = false;
                break;
            }
        }
        if (ok) return g;
    }
}
template <int m> constexpr int primitive_root = primitive_root_constexpr(m);

// @param n `n < 2^32`
// @param m `1 <= m < 2^32`
// @return sum_{i=0}^{n-1} floor((ai + b) / m) (mod 2^64)
unsigned long long floor_sum_unsigned(unsigned long long n,
                                      unsigned long long m,
                                      unsigned long long a,
                                      unsigned long long b) {
    unsigned long long ans = 0;
    while (true) {
        if (a >= m) {
            ans += n * (n - 1) / 2 * (a / m);
            a %= m;
        }
        if (b >= m) {
            ans += n * (b / m);
            b %= m;
        }

        unsigned long long y_max = a * n + b;
        if (y_max < m) break;
        // y_max < m * (n + 1)
        // floor(y_max / m) <= n
        n = (unsigned long long)(y_max / m);
        b = (unsigned long long)(y_max % m);
        std::swap(m, a);
    }
    return ans;
}

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

#ifndef _MSC_VER
template <class T>
using is_signed_int128 =
    typename std::conditional<std::is_same<T, __int128_t>::value ||
                                  std::is_same<T, __int128>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using is_unsigned_int128 =
    typename std::conditional<std::is_same<T, __uint128_t>::value ||
                                  std::is_same<T, unsigned __int128>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using make_unsigned_int128 =
    typename std::conditional<std::is_same<T, __int128_t>::value,
                              __uint128_t,
                              unsigned __int128>;

template <class T>
using is_integral = typename std::conditional<std::is_integral<T>::value ||
                                                  is_signed_int128<T>::value ||
                                                  is_unsigned_int128<T>::value,
                                              std::true_type,
                                              std::false_type>::type;

template <class T>
using is_signed_int = typename std::conditional<(is_integral<T>::value &&
                                                 std::is_signed<T>::value) ||
                                                    is_signed_int128<T>::value,
                                                std::true_type,
                                                std::false_type>::type;

template <class T>
using is_unsigned_int =
    typename std::conditional<(is_integral<T>::value &&
                               std::is_unsigned<T>::value) ||
                                  is_unsigned_int128<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using to_unsigned = typename std::conditional<
    is_signed_int128<T>::value,
    make_unsigned_int128<T>,
    typename std::conditional<std::is_signed<T>::value,
                              std::make_unsigned<T>,
                              std::common_type<T>>::type>::type;

#else

template <class T> using is_integral = typename std::is_integral<T>;

template <class T>
using is_signed_int =
    typename std::conditional<is_integral<T>::value && std::is_signed<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using is_unsigned_int =
    typename std::conditional<is_integral<T>::value &&
                                  std::is_unsigned<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using to_unsigned = typename std::conditional<is_signed_int<T>::value,
                                              std::make_unsigned<T>,
                                              std::common_type<T>>::type;

#endif

template <class T>
using is_signed_int_t = std::enable_if_t<is_signed_int<T>::value>;

template <class T>
using is_unsigned_int_t = std::enable_if_t<is_unsigned_int<T>::value>;

template <class T> using to_unsigned_t = typename to_unsigned<T>::type;

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

struct modint_base {};
struct static_modint_base : modint_base {};

template <class T> using is_modint = std::is_base_of<modint_base, T>;
template <class T> using is_modint_t = std::enable_if_t<is_modint<T>::value>;

}  // namespace internal

template <int m, std::enable_if_t<(1 <= m)>* = nullptr>
struct static_modint : internal::static_modint_base {
    using mint = static_modint;

  public:
    static constexpr int mod() { return m; }
    static mint raw(int v) {
        mint x;
        x._v = v;
        return x;
    }

    static_modint() : _v(0) {}
    template <class T, internal::is_signed_int_t<T>* = nullptr>
    static_modint(T v) {
        long long x = (long long)(v % (long long)(umod()));
        if (x < 0) x += umod();
        _v = (unsigned int)(x);
    }
    template <class T, internal::is_unsigned_int_t<T>* = nullptr>
    static_modint(T v) {
        _v = (unsigned int)(v % umod());
    }

    int val() const { return _v; }

    mint& operator++() {
        _v++;
        if (_v == umod()) _v = 0;
        return *this;
    }
    mint& operator--() {
        if (_v == 0) _v = umod();
        _v--;
        return *this;
    }
    mint operator++(int) {
        mint result = *this;
        ++*this;
        return result;
    }
    mint operator--(int) {
        mint result = *this;
        --*this;
        return result;
    }

    mint& operator+=(const mint& rhs) {
        _v += rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator-=(const mint& rhs) {
        _v -= rhs._v;
        if (_v >= umod()) _v += umod();
        return *this;
    }
    mint& operator*=(const mint& rhs) {
        unsigned long long z = _v;
        z *= rhs._v;
        _v = (unsigned int)(z % umod());
        return *this;
    }
    mint& operator/=(const mint& rhs) { return *this = *this * rhs.inv(); }

    mint operator+() const { return *this; }
    mint operator-() const { return mint() - *this; }

    mint pow(long long n) const {
        assert(0 <= n);
        mint x = *this, r = 1;
        while (n) {
            if (n & 1) r *= x;
            x *= x;
            n >>= 1;
        }
        return r;
    }
    mint inv() const {
        if (prime) {
            assert(_v);
            return pow(umod() - 2);
        } else {
            auto eg = internal::inv_gcd(_v, m);
            assert(eg.first == 1);
            return eg.second;
        }
    }

    friend mint operator+(const mint& lhs, const mint& rhs) {
        return mint(lhs) += rhs;
    }
    friend mint operator-(const mint& lhs, const mint& rhs) {
        return mint(lhs) -= rhs;
    }
    friend mint operator*(const mint& lhs, const mint& rhs) {
        return mint(lhs) *= rhs;
    }
    friend mint operator/(const mint& lhs, const mint& rhs) {
        return mint(lhs) /= rhs;
    }
    friend bool operator==(const mint& lhs, const mint& rhs) {
        return lhs._v == rhs._v;
    }
    friend bool operator!=(const mint& lhs, const mint& rhs) {
        return lhs._v != rhs._v;
    }

  private:
    unsigned int _v;
    static constexpr unsigned int umod() { return m; }
    static constexpr bool prime = internal::is_prime<m>;
};

template <int id> struct dynamic_modint : internal::modint_base {
    using mint = dynamic_modint;

  public:
    static int mod() { return (int)(bt.umod()); }
    static void set_mod(int m) {
        assert(1 <= m);
        bt = internal::barrett(m);
    }
    static mint raw(int v) {
        mint x;
        x._v = v;
        return x;
    }

    dynamic_modint() : _v(0) {}
    template <class T, internal::is_signed_int_t<T>* = nullptr>
    dynamic_modint(T v) {
        long long x = (long long)(v % (long long)(mod()));
        if (x < 0) x += mod();
        _v = (unsigned int)(x);
    }
    template <class T, internal::is_unsigned_int_t<T>* = nullptr>
    dynamic_modint(T v) {
        _v = (unsigned int)(v % mod());
    }

    int val() const { return _v; }

    mint& operator++() {
        _v++;
        if (_v == umod()) _v = 0;
        return *this;
    }
    mint& operator--() {
        if (_v == 0) _v = umod();
        _v--;
        return *this;
    }
    mint operator++(int) {
        mint result = *this;
        ++*this;
        return result;
    }
    mint operator--(int) {
        mint result = *this;
        --*this;
        return result;
    }

    mint& operator+=(const mint& rhs) {
        _v += rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator-=(const mint& rhs) {
        _v += mod() - rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator*=(const mint& rhs) {
        _v = bt.mul(_v, rhs._v);
        return *this;
    }
    mint& operator/=(const mint& rhs) { return *this = *this * rhs.inv(); }

    mint operator+() const { return *this; }
    mint operator-() const { return mint() - *this; }

    mint pow(long long n) const {
        assert(0 <= n);
        mint x = *this, r = 1;
        while (n) {
            if (n & 1) r *= x;
            x *= x;
            n >>= 1;
        }
        return r;
    }
    mint inv() const {
        auto eg = internal::inv_gcd(_v, mod());
        assert(eg.first == 1);
        return eg.second;
    }

    friend mint operator+(const mint& lhs, const mint& rhs) {
        return mint(lhs) += rhs;
    }
    friend mint operator-(const mint& lhs, const mint& rhs) {
        return mint(lhs) -= rhs;
    }
    friend mint operator*(const mint& lhs, const mint& rhs) {
        return mint(lhs) *= rhs;
    }
    friend mint operator/(const mint& lhs, const mint& rhs) {
        return mint(lhs) /= rhs;
    }
    friend bool operator==(const mint& lhs, const mint& rhs) {
        return lhs._v == rhs._v;
    }
    friend bool operator!=(const mint& lhs, const mint& rhs) {
        return lhs._v != rhs._v;
    }

  private:
    unsigned int _v;
    static internal::barrett bt;
    static unsigned int umod() { return bt.umod(); }
};
template <int id> internal::barrett dynamic_modint<id>::bt(998244353);

using modint998244353 = static_modint<998244353>;
using modint1000000007 = static_modint<1000000007>;
using modint = dynamic_modint<-1>;

namespace internal {

template <class T>
using is_static_modint = std::is_base_of<internal::static_modint_base, T>;

template <class T>
using is_static_modint_t = std::enable_if_t<is_static_modint<T>::value>;

template <class> struct is_dynamic_modint : public std::false_type {};
template <int id>
struct is_dynamic_modint<dynamic_modint<id>> : public std::true_type {};

template <class T>
using is_dynamic_modint_t = std::enable_if_t<is_dynamic_modint<T>::value>;

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

template <class mint,
          int g = internal::primitive_root<mint::mod()>,
          internal::is_static_modint_t<mint>* = nullptr>
struct fft_info {
    static constexpr int rank2 = countr_zero_constexpr(mint::mod() - 1);
    std::array<mint, rank2 + 1> root;   // root[i]^(2^i) == 1
    std::array<mint, rank2 + 1> iroot;  // root[i] * iroot[i] == 1

    std::array<mint, std::max(0, rank2 - 2 + 1)> rate2;
    std::array<mint, std::max(0, rank2 - 2 + 1)> irate2;

    std::array<mint, std::max(0, rank2 - 3 + 1)> rate3;
    std::array<mint, std::max(0, rank2 - 3 + 1)> irate3;

    fft_info() {
        root[rank2] = mint(g).pow((mint::mod() - 1) >> rank2);
        iroot[rank2] = root[rank2].inv();
        for (int i = rank2 - 1; i >= 0; i--) {
            root[i] = root[i + 1] * root[i + 1];
            iroot[i] = iroot[i + 1] * iroot[i + 1];
        }

        {
            mint prod = 1, iprod = 1;
            for (int i = 0; i <= rank2 - 2; i++) {
                rate2[i] = root[i + 2] * prod;
                irate2[i] = iroot[i + 2] * iprod;
                prod *= iroot[i + 2];
                iprod *= root[i + 2];
            }
        }
        {
            mint prod = 1, iprod = 1;
            for (int i = 0; i <= rank2 - 3; i++) {
                rate3[i] = root[i + 3] * prod;
                irate3[i] = iroot[i + 3] * iprod;
                prod *= iroot[i + 3];
                iprod *= root[i + 3];
            }
        }
    }
};

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
void butterfly(std::vector<mint>& a) {
    int n = int(a.size());
    int h = internal::countr_zero((unsigned int)n);

    static const fft_info<mint> info;

    int len = 0;  // a[i, i+(n>>len), i+2*(n>>len), ..] is transformed
    while (len < h) {
        if (h - len == 1) {
            int p = 1 << (h - len - 1);
            mint rot = 1;
            for (int s = 0; s < (1 << len); s++) {
                int offset = s << (h - len);
                for (int i = 0; i < p; i++) {
                    auto l = a[i + offset];
                    auto r = a[i + offset + p] * rot;
                    a[i + offset] = l + r;
                    a[i + offset + p] = l - r;
                }
                if (s + 1 != (1 << len))
                    rot *= info.rate2[countr_zero(~(unsigned int)(s))];
            }
            len++;
        } else {
            // 4-base
            int p = 1 << (h - len - 2);
            mint rot = 1, imag = info.root[2];
            for (int s = 0; s < (1 << len); s++) {
                mint rot2 = rot * rot;
                mint rot3 = rot2 * rot;
                int offset = s << (h - len);
                for (int i = 0; i < p; i++) {
                    auto mod2 = 1ULL * mint::mod() * mint::mod();
                    auto a0 = 1ULL * a[i + offset].val();
                    auto a1 = 1ULL * a[i + offset + p].val() * rot.val();
                    auto a2 = 1ULL * a[i + offset + 2 * p].val() * rot2.val();
                    auto a3 = 1ULL * a[i + offset + 3 * p].val() * rot3.val();
                    auto a1na3imag =
                        1ULL * mint(a1 + mod2 - a3).val() * imag.val();
                    auto na2 = mod2 - a2;
                    a[i + offset] = a0 + a2 + a1 + a3;
                    a[i + offset + 1 * p] = a0 + a2 + (2 * mod2 - (a1 + a3));
                    a[i + offset + 2 * p] = a0 + na2 + a1na3imag;
                    a[i + offset + 3 * p] = a0 + na2 + (mod2 - a1na3imag);
                }
                if (s + 1 != (1 << len))
                    rot *= info.rate3[countr_zero(~(unsigned int)(s))];
            }
            len += 2;
        }
    }
}

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
void butterfly_inv(std::vector<mint>& a) {
    int n = int(a.size());
    int h = internal::countr_zero((unsigned int)n);

    static const fft_info<mint> info;

    int len = h;  // a[i, i+(n>>len), i+2*(n>>len), ..] is transformed
    while (len) {
        if (len == 1) {
            int p = 1 << (h - len);
            mint irot = 1;
            for (int s = 0; s < (1 << (len - 1)); s++) {
                int offset = s << (h - len + 1);
                for (int i = 0; i < p; i++) {
                    auto l = a[i + offset];
                    auto r = a[i + offset + p];
                    a[i + offset] = l + r;
                    a[i + offset + p] =
                        (unsigned long long)((unsigned int)(l.val() - r.val()) + mint::mod()) *
                        irot.val();
                    ;
                }
                if (s + 1 != (1 << (len - 1)))
                    irot *= info.irate2[countr_zero(~(unsigned int)(s))];
            }
            len--;
        } else {
            // 4-base
            int p = 1 << (h - len);
            mint irot = 1, iimag = info.iroot[2];
            for (int s = 0; s < (1 << (len - 2)); s++) {
                mint irot2 = irot * irot;
                mint irot3 = irot2 * irot;
                int offset = s << (h - len + 2);
                for (int i = 0; i < p; i++) {
                    auto a0 = 1ULL * a[i + offset + 0 * p].val();
                    auto a1 = 1ULL * a[i + offset + 1 * p].val();
                    auto a2 = 1ULL * a[i + offset + 2 * p].val();
                    auto a3 = 1ULL * a[i + offset + 3 * p].val();

                    auto a2na3iimag =
                        1ULL *
                        mint((mint::mod() + a2 - a3) * iimag.val()).val();

                    a[i + offset] = a0 + a1 + a2 + a3;
                    a[i + offset + 1 * p] =
                        (a0 + (mint::mod() - a1) + a2na3iimag) * irot.val();
                    a[i + offset + 2 * p] =
                        (a0 + a1 + (mint::mod() - a2) + (mint::mod() - a3)) *
                        irot2.val();
                    a[i + offset + 3 * p] =
                        (a0 + (mint::mod() - a1) + (mint::mod() - a2na3iimag)) *
                        irot3.val();
                }
                if (s + 1 != (1 << (len - 2)))
                    irot *= info.irate3[countr_zero(~(unsigned int)(s))];
            }
            len -= 2;
        }
    }
}

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution_naive(const std::vector<mint>& a,
                                    const std::vector<mint>& b) {
    int n = int(a.size()), m = int(b.size());
    std::vector<mint> ans(n + m - 1);
    if (n < m) {
        for (int j = 0; j < m; j++) {
            for (int i = 0; i < n; i++) {
                ans[i + j] += a[i] * b[j];
            }
        }
    } else {
        for (int i = 0; i < n; i++) {
            for (int j = 0; j < m; j++) {
                ans[i + j] += a[i] * b[j];
            }
        }
    }
    return ans;
}

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution_fft(std::vector<mint> a, std::vector<mint> b) {
    int n = int(a.size()), m = int(b.size());
    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    a.resize(z);
    internal::butterfly(a);
    b.resize(z);
    internal::butterfly(b);
    for (int i = 0; i < z; i++) {
        a[i] *= b[i];
    }
    internal::butterfly_inv(a);
    a.resize(n + m - 1);
    mint iz = mint(z).inv();
    for (int i = 0; i < n + m - 1; i++) a[i] *= iz;
    return a;
}

}  // namespace internal

template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution(std::vector<mint>&& a, std::vector<mint>&& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    assert((mint::mod() - 1) % z == 0);

    if (std::min(n, m) <= 60) return convolution_naive(std::move(a), std::move(b));
    return internal::convolution_fft(std::move(a), std::move(b));
}
template <class mint, internal::is_static_modint_t<mint>* = nullptr>
std::vector<mint> convolution(const std::vector<mint>& a,
                              const std::vector<mint>& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    assert((mint::mod() - 1) % z == 0);

    if (std::min(n, m) <= 60) return convolution_naive(a, b);
    return internal::convolution_fft(a, b);
}

template <unsigned int mod = 998244353,
          class T,
          std::enable_if_t<internal::is_integral<T>::value>* = nullptr>
std::vector<T> convolution(const std::vector<T>& a, const std::vector<T>& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    using mint = static_modint<mod>;

    int z = (int)internal::bit_ceil((unsigned int)(n + m - 1));
    assert((mint::mod() - 1) % z == 0);

    std::vector<mint> a2(n), b2(m);
    for (int i = 0; i < n; i++) {
        a2[i] = mint(a[i]);
    }
    for (int i = 0; i < m; i++) {
        b2[i] = mint(b[i]);
    }
    auto c2 = convolution(std::move(a2), std::move(b2));
    std::vector<T> c(n + m - 1);
    for (int i = 0; i < n + m - 1; i++) {
        c[i] = c2[i].val();
    }
    return c;
}

std::vector<long long> convolution_ll(const std::vector<long long>& a,
                                      const std::vector<long long>& b) {
    int n = int(a.size()), m = int(b.size());
    if (!n || !m) return {};

    static constexpr unsigned long long MOD1 = 754974721;  // 2^24
    static constexpr unsigned long long MOD2 = 167772161;  // 2^25
    static constexpr unsigned long long MOD3 = 469762049;  // 2^26
    static constexpr unsigned long long M2M3 = MOD2 * MOD3;
    static constexpr unsigned long long M1M3 = MOD1 * MOD3;
    static constexpr unsigned long long M1M2 = MOD1 * MOD2;
    static constexpr unsigned long long M1M2M3 = MOD1 * MOD2 * MOD3;

    static constexpr unsigned long long i1 =
        internal::inv_gcd(MOD2 * MOD3, MOD1).second;
    static constexpr unsigned long long i2 =
        internal::inv_gcd(MOD1 * MOD3, MOD2).second;
    static constexpr unsigned long long i3 =
        internal::inv_gcd(MOD1 * MOD2, MOD3).second;

    static constexpr int MAX_AB_BIT = 24;
    static_assert(MOD1 % (1ull << MAX_AB_BIT) == 1, "MOD1 isn't enough to support an array length of 2^24.");
    static_assert(MOD2 % (1ull << MAX_AB_BIT) == 1, "MOD2 isn't enough to support an array length of 2^24.");
    static_assert(MOD3 % (1ull << MAX_AB_BIT) == 1, "MOD3 isn't enough to support an array length of 2^24.");
    assert(n + m - 1 <= (1 << MAX_AB_BIT));

    auto c1 = convolution<MOD1>(a, b);
    auto c2 = convolution<MOD2>(a, b);
    auto c3 = convolution<MOD3>(a, b);

    std::vector<long long> c(n + m - 1);
    for (int i = 0; i < n + m - 1; i++) {
        unsigned long long x = 0;
        x += (c1[i] * i1) % MOD1 * M2M3;
        x += (c2[i] * i2) % MOD2 * M1M3;
        x += (c3[i] * i3) % MOD3 * M1M2;
        // B = 2^63, -B <= x, r(real value) < B
        // (x, x - M, x - 2M, or x - 3M) = r (mod 2B)
        // r = c1[i] (mod MOD1)
        // focus on MOD1
        // r = x, x - M', x - 2M', x - 3M' (M' = M % 2^64) (mod 2B)
        // r = x,
        //     x - M' + (0 or 2B),
        //     x - 2M' + (0, 2B or 4B),
        //     x - 3M' + (0, 2B, 4B or 6B) (without mod!)
        // (r - x) = 0, (0)
        //           - M' + (0 or 2B), (1)
        //           -2M' + (0 or 2B or 4B), (2)
        //           -3M' + (0 or 2B or 4B or 6B) (3) (mod MOD1)
        // we checked that
        //   ((1) mod MOD1) mod 5 = 2
        //   ((2) mod MOD1) mod 5 = 3
        //   ((3) mod MOD1) mod 5 = 4
        long long diff =
            c1[i] - internal::safe_mod((long long)(x), (long long)(MOD1));
        if (diff < 0) diff += MOD1;
        static constexpr unsigned long long offset[5] = {
            0, 0, M1M2M3, 2 * M1M2M3, 3 * M1M2M3};
        x -= offset[diff % 5];
        c[i] = x;
    }

    return c;
}

}  // namespace atcoder

namespace noya {
namespace convolution_mod_internal {

using u64 = std::uint64_t;
using u128 = unsigned __int128;

inline u64 power(u64 vl, u64 exp, u64 mod) {
  u64 res = 1;
  while (exp > 0) {
    if (exp & 1) {
      res = u64(u128(res) * vl % mod);
    }
    vl = u64(u128(vl) * vl % mod);
    exp >>= 1;
  }
  return res;
}

inline u64 inverse_prime(u64 vl, u64 p) { return power(vl, p - 2, p); }

template <int Mod>
std::vector<int> run_ntt(const std::vector<std::uint64_t> &lhs,
                         const std::vector<std::uint64_t> &rhs) {
  using mint = atcoder::static_modint<Mod>;
  std::vector<mint> a(lhs.size());
  std::vector<mint> b(rhs.size());
  for (int idx = 0; idx < int(lhs.size()); idx++) {
    a[idx] = lhs[idx] % Mod;
  }
  for (int idx = 0; idx < int(rhs.size()); idx++) {
    b[idx] = rhs[idx] % Mod;
  }
  auto prd = atcoder::convolution(a, b);
  std::vector<int> res(prd.size());
  for (int idx = 0; idx < int(prd.size()); idx++) {
    res[idx] = prd[idx].val();
  }
  return res;
}

} // namespace convolution_mod_internal

/// @brief Convolve nonnegative residues modulo an arbitrary modulus using
/// three NTT primes. Requires result length <= 2^24 and each integer
/// coefficient before reduction to be smaller than the product of the primes.
inline std::vector<std::uint64_t>
convolution_mod(const std::vector<std::uint64_t> &lhs,
                const std::vector<std::uint64_t> &rhs, std::uint64_t mod) {
  using namespace convolution_mod_internal;
  assert(mod >= 1);
  if (lhs.empty() || rhs.empty()) {
    return {};
  }
  constexpr u64 p1 = 167772161;
  constexpr u64 p2 = 469762049;
  constexpr u64 p3 = 1224736769;
  constexpr u128 pp = u128(p1) * p2 * p3;
  std::size_t nr = lhs.size() + rhs.size() - 1;
  assert(nr <= (std::size_t(1) << 24));
  u64 ma = 0;
  u64 mb = 0;
  for (u64 vl : lhs) {
    assert(vl < mod);
    ma = std::max(ma, vl);
  }
  for (u64 vl : rhs) {
    assert(vl < mod);
    mb = std::max(mb, vl);
  }
  assert(u128(std::min(lhs.size(), rhs.size())) * ma * mb < pp);

  auto rsd = run_ntt<p1>(lhs, rhs);
  auto re0 = run_ntt<p2>(lhs, rhs);
  auto re1 = run_ntt<p3>(lhs, rhs);
  const u64 i12 = inverse_prime(p1 % p2, p2);
  const u64 p12 = u64(u128(p1) * p2 % p3);
  const u64 i13 = inverse_prime(p12, p3);
  std::vector<u64> res(nr);
  for (int idx = 0; idx < int(nr); idx++) {
    u64 x1 = rsd[idx];
    u64 x2 = u64(u128((re0[idx] + p2 - x1 % p2) % p2) * i12 % p2);
    u64 x12 = u64((u128(x1) + u128(p1) * x2) % p3);
    u64 x3 = u64(u128((re1[idx] + p3 - x12) % p3) * i13 % p3);
    res[idx] = u64((u128(x1 % mod) + u128(p1 % mod) * x2 +
                    u128(u64(u128(p1) * p2 % mod)) * x3) %
                   mod);
  }
  return res;
}

} // namespace noya

namespace noya {

namespace big_integer_multiplication_internal {

inline std::vector<std::uint64_t> read_limbs(std::string_view mag, int bas,
                                             int dig) {
  std::vector<std::uint64_t> lim;
  for (std::size_t r = mag.size(); r > 0;) {
    std::size_t l = r > std::size_t(dig) ? r - dig : 0;
    std::uint64_t val = 0;
    for (std::size_t pos = l; pos < r; pos++) {
      val = val * bas + big_integer_addition_internal::digit_value(mag[pos]);
    }
    lim.push_back(val);
    r = l;
  }
  return lim;
}

inline std::vector<std::uint64_t>
multiply_limbs(const std::vector<std::uint64_t> &a,
               const std::vector<std::uint64_t> &b, std::uint64_t bs) {
  if (std::min(a.size(), b.size()) <= 32) {
    std::vector<std::uint64_t> res(a.size() + b.size());
    for (int l = 0; l < int(a.size()); l++) {
      unsigned __int128 car = 0;
      for (int r = 0; r < int(b.size()); r++) {
        unsigned __int128 val =
            res[l + r] + car + static_cast<unsigned __int128>(a[l]) * b[r];
        res[l + r] = std::uint64_t(val % bs);
        car = val / bs;
      }
      int pos = l + int(b.size());
      while (car > 0) {
        unsigned __int128 val = res[pos] + car;
        res[pos] = std::uint64_t(val % bs);
        car = val / bs;
        pos++;
      }
    }
    return res;
  }

  std::uint64_t ma = *std::max_element(a.begin(), a.end());
  std::uint64_t mb = *std::max_element(b.begin(), b.end());
  unsigned __int128 cb =
      static_cast<unsigned __int128>(std::min(a.size(), b.size())) * ma * mb;
  assert(cb < static_cast<unsigned __int128>(UINT64_MAX));
  // Every raw coefficient is below this modulus, so reduction after CRT does
  // not discard information.
  std::uint64_t mod = std::uint64_t(cb) + 1;
  std::vector<std::uint64_t> prd = convolution_mod(a, b, mod);
  std::uint64_t car = 0;
  for (std::uint64_t &val : prd) {
    unsigned __int128 cur = static_cast<unsigned __int128>(val) + car;
    val = std::uint64_t(cur % bs);
    car = std::uint64_t(cur / bs);
  }
  while (car > 0) {
    prd.push_back(car % bs);
    car /= bs;
  }
  return prd;
}

inline std::string write_limbs(std::vector<std::uint64_t> lim, int bas, int dig,
                               bool cap) {
  while (lim.size() > 1 && lim.back() == 0) {
    lim.pop_back();
  }
  std::string res;
  res.reserve(lim.size() * dig);
  for (int idx = int(lim.size()) - 1; idx >= 0; idx--) {
    std::uint64_t val = lim[idx];
    std::string blk;
    do {
      blk.push_back(
          big_integer_addition_internal::digit_character(int(val % bas), cap));
      val /= bas;
    } while (val > 0);
    if (idx + 1 != int(lim.size())) {
      blk.resize(dig, '0');
    }
    std::reverse(blk.begin(), blk.end());
    res += blk;
  }
  return res;
}

} // namespace big_integer_multiplication_internal

/// @brief Multiply two arbitrarily long signed integers in base 2..36. Digits
/// are packed into limbs whose square fits exact three-prime CRT convolution;
/// the product coefficients are then normalized by a linear carry pass.
inline std::string multiply_big_integers(std::string_view a, std::string_view b,
                                         int bas = 10, bool cap = false) {
  using namespace big_integer_addition_internal;
  using namespace big_integer_multiplication_internal;
  parsed_integer l = parse(a, bas);
  parsed_integer r = parse(b, bas);
  if (l.mag == "0" || r.mag == "0") {
    return "0";
  }

  constexpr std::uint64_t mbs = 10'000;
  int dig = 1;
  std::uint64_t bs = bas;
  while (bs <= mbs / std::uint64_t(bas)) {
    bs *= bas;
    dig++;
  }
  auto ll = read_limbs(l.mag, bas, dig);
  auto rl = read_limbs(r.mag, bas, dig);
  std::string res = write_limbs(multiply_limbs(ll, rl, bs), bas, dig, cap);
  if (l.neg != r.neg) {
    res.insert(res.begin(), '-');
  }
  return res;
}

} // namespace noya