Skip to content

multiplicative_function_sum.hpp

SECTIONMath INCLUDEnoya/multiplicative_function_sum.hpp

在较大 \(n\) 下计算积性函数的前缀和;适合能给出素数幂取值、但不能线性筛到 \(n\) 的数论求和。

\[ \displaystyle F(n)=\sum_{d|n}f(d)g(n/d) \]

Complexity: Time: O(n^(3/4) / log n) arithmetic operations. Space: O(sqrt(n)).

AC 记录:sum_of_multiplicative_function, sum_of_multiplicative_function_large

跳到代码 · GitHub ↗

Implementation

当前头文件,省略 include guard;依赖见 #include

/// @complexity Time: O(n^(3/4) / log n) arithmetic operations.
/// Space: O(sqrt(n)).

#include "atcoder/modint.hpp"

#include <algorithm>
#include <cassert>
#include <cmath>
#include <cstdint>
#include <vector>

namespace noya {

namespace multiplicative_function_sum_internal {

template <int Mod> class linear_prime_power_sum {
public:
  using mint = atcoder::static_modint<Mod>;
  using u64 = std::uint64_t;
  using u128 = unsigned __int128;

  linear_prime_power_sum(u64 lim, mint ce, mint cp)
      : lm_(lim), ce_(ce), cp_(cp) {
    if (lm_ == 0) {
      return;
    }
    sq_ = integer_square_root(lm_);
    hl_ = lm_ / sq_;
    if (hl_ != 1 && lm_ / (hl_ - 1) == sq_) {
      hl_--;
    }
    nb_ = int(hl_ + sq_);
    ps = enumerate_primes(int(sq_));
    build_prime_tables();
  }

  mint run() {
    if (lm_ == 0) {
      return 0;
    }
    an_ = mint(1) + prime_prefix(lm_);
    for (int idx = 0; idx < int(ps.size()); idx++) {
      enumerate_composites(idx, 1, ps[idx], mint(1));
    }
    return an_;
  }

private:
  struct prime_aggregate {
    mint cnt = 0;
    mint sum = 0;
  };

  u64 lm_ = 0;
  u64 sq_ = 0;
  u64 hl_ = 0;
  int nb_ = 0;
  mint ce_ = 0;
  mint cp_ = 0;
  std::vector<int> ps;
  std::vector<prime_aggregate> pt_;
  mint an_ = 0;

  static u64 integer_square_root(u64 val) {
    u64 rt = u64(std::sqrt(static_cast<long double>(val)));
    while (u128(rt + 1) * (rt + 1) <= val) {
      rt++;
    }
    while (u128(rt) * rt > val) {
      rt--;
    }
    return rt;
  }

  static std::vector<int> enumerate_primes(int lim) {
    if (lim < 2) {
      return {};
    }
    std::vector<bool> ip(lim + 1, true);
    ip[0] = ip[1] = false;
    for (int p = 2; 1LL * p * p <= lim; p++) {
      if (!ip[p]) {
        continue;
      }
      for (int mul = p * p; mul <= lim; mul += p) {
        ip[mul] = false;
      }
    }
    std::vector<int> res;
    for (int val = 2; val <= lim; val++) {
      if (ip[val]) {
        res.push_back(val);
      }
    }
    return res;
  }

  int index_of(u64 val) const {
    assert(1 <= val && val <= lm_);
    int idx = val <= sq_ ? nb_ - int(val) : int(lm_ / val);
    assert(0 <= idx && idx < nb_);
    return idx;
  }

  u64 value_at(int idx) const {
    assert(0 <= idx && idx < nb_);
    if (u64(idx) < hl_) {
      return idx == 0 ? 0 : lm_ / u64(idx);
    }
    return u64(nb_ - idx);
  }

  void build_prime_tables() {
    pt_.resize(nb_);
    mint iv2 = mint(2).inv();
    for (int idx = 1; idx < nb_; idx++) {
      u64 val = value_at(idx);
      pt_[idx].cnt = mint(val - 1);
      mint cnv = mint(val);
      pt_[idx].sum = cnv * (cnv + 1) * iv2 - 1;
    }

    for (int p : ps) {
      u64 squ = u64(p) * p;
      prime_aggregate pre = pt_[index_of(p - 1)];
      u64 mh = std::min(hl_, lm_ / squ + 1);
      for (u64 idx = 1; idx < mh; idx++) {
        u64 red = lm_ / (idx * u64(p));
        prime_aggregate dif{pt_[index_of(red)].cnt - pre.cnt,
                            pt_[index_of(red)].sum - pre.sum};
        pt_[int(idx)].cnt -= dif.cnt;
        pt_[int(idx)].sum -= mint(p) * dif.sum;
      }
      for (u64 val = sq_; val >= squ; val--) {
        int idx = index_of(val);
        prime_aggregate dif{pt_[index_of(val / p)].cnt - pre.cnt,
                            pt_[index_of(val / p)].sum - pre.sum};
        pt_[idx].cnt -= dif.cnt;
        pt_[idx].sum -= mint(p) * dif.sum;
      }
    }
  }

  mint prime_prefix(u64 val) const {
    const prime_aggregate &agg = pt_[index_of(val)];
    return ce_ * agg.cnt + cp_ * agg.sum;
  }

  mint prime_power_value(int p, int exp) const { return ce_ * exp + cp_ * p; }

  void enumerate_composites(int pi, int exp, u64 prd, mint pv) {
    int p = ps[pi];
    an_ += pv * prime_power_value(p, exp + 1);

    u64 lim = lm_ / prd;
    if (u128(p) * p <= lim) {
      enumerate_composites(pi, exp + 1, prd * u64(p), pv);
    }

    pv *= prime_power_value(p, exp);
    an_ += pv * (prime_prefix(lim) - prime_prefix(u64(p)));

    int nxt = pi + 1;
    while (nxt < int(ps.size()) && u128(ps[nxt]) * ps[nxt] * ps[nxt] <= lim) {
      enumerate_composites(nxt, 1, prd * u64(ps[nxt]), pv);
      nxt++;
    }
    while (nxt < int(ps.size()) && u128(ps[nxt]) * ps[nxt] <= lim) {
      int np = ps[nxt];
      mint con = prime_power_value(np, 2);
      con += prime_power_value(np, 1) *
             (prime_prefix(lim / u64(np)) - prime_prefix(u64(np)));
      an_ += pv * con;
      nxt++;
    }
  }
};

} // namespace multiplicative_function_sum_internal

/// @brief Sum a multiplicative function satisfying f(p^e)=a*e+b*p for every
/// prime power. Quotient blocks first store both pi(x) and the sum of primes
/// up to x. A recursion then fixes the least prime factor and its exponent;
/// the remaining one-prime tail is aggregated from those two tables instead
/// of visited individually. Each integer is represented by its unique prime
/// factorization, so the accumulated contributions are disjoint and complete.
template <int Mod>
atcoder::static_modint<Mod>
sum_linear_prime_power_multiplicative(std::uint64_t lim,
                                      atcoder::static_modint<Mod> ce,
                                      atcoder::static_modint<Mod> cp) {
  return multiplicative_function_sum_internal::linear_prime_power_sum<Mod>(
             lim, ce, cp)
      .run();
}

} // namespace noya
#ifndef NOYA_MULTIPLICATIVE_FUNCTION_SUM_HPP
#define NOYA_MULTIPLICATIVE_FUNCTION_SUM_HPP 1

/// @complexity Time: O(n^(3/4) / log n) arithmetic operations.
/// Space: O(sqrt(n)).

#include "atcoder/modint.hpp"

#include <algorithm>
#include <cassert>
#include <cmath>
#include <cstdint>
#include <vector>

namespace noya {

namespace multiplicative_function_sum_internal {

template <int Mod> class linear_prime_power_sum {
public:
  using mint = atcoder::static_modint<Mod>;
  using u64 = std::uint64_t;
  using u128 = unsigned __int128;

  linear_prime_power_sum(u64 lim, mint ce, mint cp)
      : lm_(lim), ce_(ce), cp_(cp) {
    if (lm_ == 0) {
      return;
    }
    sq_ = integer_square_root(lm_);
    hl_ = lm_ / sq_;
    if (hl_ != 1 && lm_ / (hl_ - 1) == sq_) {
      hl_--;
    }
    nb_ = int(hl_ + sq_);
    ps = enumerate_primes(int(sq_));
    build_prime_tables();
  }

  mint run() {
    if (lm_ == 0) {
      return 0;
    }
    an_ = mint(1) + prime_prefix(lm_);
    for (int idx = 0; idx < int(ps.size()); idx++) {
      enumerate_composites(idx, 1, ps[idx], mint(1));
    }
    return an_;
  }

private:
  struct prime_aggregate {
    mint cnt = 0;
    mint sum = 0;
  };

  u64 lm_ = 0;
  u64 sq_ = 0;
  u64 hl_ = 0;
  int nb_ = 0;
  mint ce_ = 0;
  mint cp_ = 0;
  std::vector<int> ps;
  std::vector<prime_aggregate> pt_;
  mint an_ = 0;

  static u64 integer_square_root(u64 val) {
    u64 rt = u64(std::sqrt(static_cast<long double>(val)));
    while (u128(rt + 1) * (rt + 1) <= val) {
      rt++;
    }
    while (u128(rt) * rt > val) {
      rt--;
    }
    return rt;
  }

  static std::vector<int> enumerate_primes(int lim) {
    if (lim < 2) {
      return {};
    }
    std::vector<bool> ip(lim + 1, true);
    ip[0] = ip[1] = false;
    for (int p = 2; 1LL * p * p <= lim; p++) {
      if (!ip[p]) {
        continue;
      }
      for (int mul = p * p; mul <= lim; mul += p) {
        ip[mul] = false;
      }
    }
    std::vector<int> res;
    for (int val = 2; val <= lim; val++) {
      if (ip[val]) {
        res.push_back(val);
      }
    }
    return res;
  }

  int index_of(u64 val) const {
    assert(1 <= val && val <= lm_);
    int idx = val <= sq_ ? nb_ - int(val) : int(lm_ / val);
    assert(0 <= idx && idx < nb_);
    return idx;
  }

  u64 value_at(int idx) const {
    assert(0 <= idx && idx < nb_);
    if (u64(idx) < hl_) {
      return idx == 0 ? 0 : lm_ / u64(idx);
    }
    return u64(nb_ - idx);
  }

  void build_prime_tables() {
    pt_.resize(nb_);
    mint iv2 = mint(2).inv();
    for (int idx = 1; idx < nb_; idx++) {
      u64 val = value_at(idx);
      pt_[idx].cnt = mint(val - 1);
      mint cnv = mint(val);
      pt_[idx].sum = cnv * (cnv + 1) * iv2 - 1;
    }

    for (int p : ps) {
      u64 squ = u64(p) * p;
      prime_aggregate pre = pt_[index_of(p - 1)];
      u64 mh = std::min(hl_, lm_ / squ + 1);
      for (u64 idx = 1; idx < mh; idx++) {
        u64 red = lm_ / (idx * u64(p));
        prime_aggregate dif{pt_[index_of(red)].cnt - pre.cnt,
                            pt_[index_of(red)].sum - pre.sum};
        pt_[int(idx)].cnt -= dif.cnt;
        pt_[int(idx)].sum -= mint(p) * dif.sum;
      }
      for (u64 val = sq_; val >= squ; val--) {
        int idx = index_of(val);
        prime_aggregate dif{pt_[index_of(val / p)].cnt - pre.cnt,
                            pt_[index_of(val / p)].sum - pre.sum};
        pt_[idx].cnt -= dif.cnt;
        pt_[idx].sum -= mint(p) * dif.sum;
      }
    }
  }

  mint prime_prefix(u64 val) const {
    const prime_aggregate &agg = pt_[index_of(val)];
    return ce_ * agg.cnt + cp_ * agg.sum;
  }

  mint prime_power_value(int p, int exp) const { return ce_ * exp + cp_ * p; }

  void enumerate_composites(int pi, int exp, u64 prd, mint pv) {
    int p = ps[pi];
    an_ += pv * prime_power_value(p, exp + 1);

    u64 lim = lm_ / prd;
    if (u128(p) * p <= lim) {
      enumerate_composites(pi, exp + 1, prd * u64(p), pv);
    }

    pv *= prime_power_value(p, exp);
    an_ += pv * (prime_prefix(lim) - prime_prefix(u64(p)));

    int nxt = pi + 1;
    while (nxt < int(ps.size()) && u128(ps[nxt]) * ps[nxt] * ps[nxt] <= lim) {
      enumerate_composites(nxt, 1, prd * u64(ps[nxt]), pv);
      nxt++;
    }
    while (nxt < int(ps.size()) && u128(ps[nxt]) * ps[nxt] <= lim) {
      int np = ps[nxt];
      mint con = prime_power_value(np, 2);
      con += prime_power_value(np, 1) *
             (prime_prefix(lim / u64(np)) - prime_prefix(u64(np)));
      an_ += pv * con;
      nxt++;
    }
  }
};

} // namespace multiplicative_function_sum_internal

/// @brief Sum a multiplicative function satisfying f(p^e)=a*e+b*p for every
/// prime power. Quotient blocks first store both pi(x) and the sum of primes
/// up to x. A recursion then fixes the least prime factor and its exponent;
/// the remaining one-prime tail is aggregated from those two tables instead
/// of visited individually. Each integer is represented by its unique prime
/// factorization, so the accumulated contributions are disjoint and complete.
template <int Mod>
atcoder::static_modint<Mod>
sum_linear_prime_power_multiplicative(std::uint64_t lim,
                                      atcoder::static_modint<Mod> ce,
                                      atcoder::static_modint<Mod> cp) {
  return multiplicative_function_sum_internal::linear_prime_power_sum<Mod>(
             lim, ce, cp)
      .run();
}

} // namespace noya

#endif // NOYA_MULTIPLICATIVE_FUNCTION_SUM_HPP
#include <algorithm>
#include <cassert>
#include <cmath>
#include <cstdint>
#include <numeric>
#include <type_traits>
#include <utility>
#include <vector>

/// @complexity Time: O(n^(3/4) / log n) arithmetic operations.
/// Space: O(sqrt(n)).

#ifdef _MSC_VER
#include <intrin.h>
#endif

#ifdef _MSC_VER
#include <intrin.h>
#endif

namespace atcoder {

namespace internal {

// @param m `1 <= m`
// @return x mod m
constexpr long long safe_mod(long long x, long long m) {
    x %= m;
    if (x < 0) x += m;
    return x;
}

// Fast modular multiplication by barrett reduction
// Reference: https://en.wikipedia.org/wiki/Barrett_reduction
// NOTE: reconsider after Ice Lake
struct barrett {
    unsigned int _m;
    unsigned long long im;

    // @param m `1 <= m`
    explicit barrett(unsigned int m) : _m(m), im((unsigned long long)(-1) / m + 1) {}

    // @return m
    unsigned int umod() const { return _m; }

    // @param a `0 <= a < m`
    // @param b `0 <= b < m`
    // @return `a * b % m`
    unsigned int mul(unsigned int a, unsigned int b) const {
        // [1] m = 1
        // a = b = im = 0, so okay

        // [2] m >= 2
        // im = ceil(2^64 / m)
        // -> im * m = 2^64 + r (0 <= r < m)
        // let z = a*b = c*m + d (0 <= c, d < m)
        // a*b * im = (c*m + d) * im = c*(im*m) + d*im = c*2^64 + c*r + d*im
        // c*r + d*im < m * m + m * im < m * m + 2^64 + m <= 2^64 + m * (m + 1) < 2^64 * 2
        // ((ab * im) >> 64) == c or c + 1
        unsigned long long z = a;
        z *= b;
#ifdef _MSC_VER
        unsigned long long x;
        _umul128(z, im, &x);
#else
        unsigned long long x =
            (unsigned long long)(((unsigned __int128)(z)*im) >> 64);
#endif
        unsigned long long y = x * _m;
        return (unsigned int)(z - y + (z < y ? _m : 0));
    }
};

// @param n `0 <= n`
// @param m `1 <= m`
// @return `(x ** n) % m`
constexpr long long pow_mod_constexpr(long long x, long long n, int m) {
    if (m == 1) return 0;
    unsigned int _m = (unsigned int)(m);
    unsigned long long r = 1;
    unsigned long long y = safe_mod(x, m);
    while (n) {
        if (n & 1) r = (r * y) % _m;
        y = (y * y) % _m;
        n >>= 1;
    }
    return r;
}

// Reference:
// M. Forisek and J. Jancina,
// Fast Primality Testing for Integers That Fit into a Machine Word
// @param n `0 <= n`
constexpr bool is_prime_constexpr(int n) {
    if (n <= 1) return false;
    if (n == 2 || n == 7 || n == 61) return true;
    if (n % 2 == 0) return false;
    long long d = n - 1;
    while (d % 2 == 0) d /= 2;
    constexpr long long bases[3] = {2, 7, 61};
    for (long long a : bases) {
        long long t = d;
        long long y = pow_mod_constexpr(a, t, n);
        while (t != n - 1 && y != 1 && y != n - 1) {
            y = y * y % n;
            t <<= 1;
        }
        if (y != n - 1 && t % 2 == 0) {
            return false;
        }
    }
    return true;
}
template <int n> constexpr bool is_prime = is_prime_constexpr(n);

// @param b `1 <= b`
// @return pair(g, x) s.t. g = gcd(a, b), xa = g (mod b), 0 <= x < b/g
constexpr std::pair<long long, long long> inv_gcd(long long a, long long b) {
    a = safe_mod(a, b);
    if (a == 0) return {b, 0};

    // Contracts:
    // [1] s - m0 * a = 0 (mod b)
    // [2] t - m1 * a = 0 (mod b)
    // [3] s * |m1| + t * |m0| <= b
    long long s = b, t = a;
    long long m0 = 0, m1 = 1;

    while (t) {
        long long u = s / t;
        s -= t * u;
        m0 -= m1 * u;  // |m1 * u| <= |m1| * s <= b

        // [3]:
        // (s - t * u) * |m1| + t * |m0 - m1 * u|
        // <= s * |m1| - t * u * |m1| + t * (|m0| + |m1| * u)
        // = s * |m1| + t * |m0| <= b

        auto tmp = s;
        s = t;
        t = tmp;
        tmp = m0;
        m0 = m1;
        m1 = tmp;
    }
    // by [3]: |m0| <= b/g
    // by g != b: |m0| < b/g
    if (m0 < 0) m0 += b / s;
    return {s, m0};
}

// Compile time primitive root
// @param m must be prime
// @return primitive root (and minimum in now)
constexpr int primitive_root_constexpr(int m) {
    if (m == 2) return 1;
    if (m == 167772161) return 3;
    if (m == 469762049) return 3;
    if (m == 754974721) return 11;
    if (m == 998244353) return 3;
    int divs[20] = {};
    divs[0] = 2;
    int cnt = 1;
    int x = (m - 1) / 2;
    while (x % 2 == 0) x /= 2;
    for (int i = 3; (long long)(i)*i <= x; i += 2) {
        if (x % i == 0) {
            divs[cnt++] = i;
            while (x % i == 0) {
                x /= i;
            }
        }
    }
    if (x > 1) {
        divs[cnt++] = x;
    }
    for (int g = 2;; g++) {
        bool ok = true;
        for (int i = 0; i < cnt; i++) {
            if (pow_mod_constexpr(g, (m - 1) / divs[i], m) == 1) {
                ok = false;
                break;
            }
        }
        if (ok) return g;
    }
}
template <int m> constexpr int primitive_root = primitive_root_constexpr(m);

// @param n `n < 2^32`
// @param m `1 <= m < 2^32`
// @return sum_{i=0}^{n-1} floor((ai + b) / m) (mod 2^64)
unsigned long long floor_sum_unsigned(unsigned long long n,
                                      unsigned long long m,
                                      unsigned long long a,
                                      unsigned long long b) {
    unsigned long long ans = 0;
    while (true) {
        if (a >= m) {
            ans += n * (n - 1) / 2 * (a / m);
            a %= m;
        }
        if (b >= m) {
            ans += n * (b / m);
            b %= m;
        }

        unsigned long long y_max = a * n + b;
        if (y_max < m) break;
        // y_max < m * (n + 1)
        // floor(y_max / m) <= n
        n = (unsigned long long)(y_max / m);
        b = (unsigned long long)(y_max % m);
        std::swap(m, a);
    }
    return ans;
}

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

#ifndef _MSC_VER
template <class T>
using is_signed_int128 =
    typename std::conditional<std::is_same<T, __int128_t>::value ||
                                  std::is_same<T, __int128>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using is_unsigned_int128 =
    typename std::conditional<std::is_same<T, __uint128_t>::value ||
                                  std::is_same<T, unsigned __int128>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using make_unsigned_int128 =
    typename std::conditional<std::is_same<T, __int128_t>::value,
                              __uint128_t,
                              unsigned __int128>;

template <class T>
using is_integral = typename std::conditional<std::is_integral<T>::value ||
                                                  is_signed_int128<T>::value ||
                                                  is_unsigned_int128<T>::value,
                                              std::true_type,
                                              std::false_type>::type;

template <class T>
using is_signed_int = typename std::conditional<(is_integral<T>::value &&
                                                 std::is_signed<T>::value) ||
                                                    is_signed_int128<T>::value,
                                                std::true_type,
                                                std::false_type>::type;

template <class T>
using is_unsigned_int =
    typename std::conditional<(is_integral<T>::value &&
                               std::is_unsigned<T>::value) ||
                                  is_unsigned_int128<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using to_unsigned = typename std::conditional<
    is_signed_int128<T>::value,
    make_unsigned_int128<T>,
    typename std::conditional<std::is_signed<T>::value,
                              std::make_unsigned<T>,
                              std::common_type<T>>::type>::type;

#else

template <class T> using is_integral = typename std::is_integral<T>;

template <class T>
using is_signed_int =
    typename std::conditional<is_integral<T>::value && std::is_signed<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using is_unsigned_int =
    typename std::conditional<is_integral<T>::value &&
                                  std::is_unsigned<T>::value,
                              std::true_type,
                              std::false_type>::type;

template <class T>
using to_unsigned = typename std::conditional<is_signed_int<T>::value,
                                              std::make_unsigned<T>,
                                              std::common_type<T>>::type;

#endif

template <class T>
using is_signed_int_t = std::enable_if_t<is_signed_int<T>::value>;

template <class T>
using is_unsigned_int_t = std::enable_if_t<is_unsigned_int<T>::value>;

template <class T> using to_unsigned_t = typename to_unsigned<T>::type;

}  // namespace internal

}  // namespace atcoder

namespace atcoder {

namespace internal {

struct modint_base {};
struct static_modint_base : modint_base {};

template <class T> using is_modint = std::is_base_of<modint_base, T>;
template <class T> using is_modint_t = std::enable_if_t<is_modint<T>::value>;

}  // namespace internal

template <int m, std::enable_if_t<(1 <= m)>* = nullptr>
struct static_modint : internal::static_modint_base {
    using mint = static_modint;

  public:
    static constexpr int mod() { return m; }
    static mint raw(int v) {
        mint x;
        x._v = v;
        return x;
    }

    static_modint() : _v(0) {}
    template <class T, internal::is_signed_int_t<T>* = nullptr>
    static_modint(T v) {
        long long x = (long long)(v % (long long)(umod()));
        if (x < 0) x += umod();
        _v = (unsigned int)(x);
    }
    template <class T, internal::is_unsigned_int_t<T>* = nullptr>
    static_modint(T v) {
        _v = (unsigned int)(v % umod());
    }

    int val() const { return _v; }

    mint& operator++() {
        _v++;
        if (_v == umod()) _v = 0;
        return *this;
    }
    mint& operator--() {
        if (_v == 0) _v = umod();
        _v--;
        return *this;
    }
    mint operator++(int) {
        mint result = *this;
        ++*this;
        return result;
    }
    mint operator--(int) {
        mint result = *this;
        --*this;
        return result;
    }

    mint& operator+=(const mint& rhs) {
        _v += rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator-=(const mint& rhs) {
        _v -= rhs._v;
        if (_v >= umod()) _v += umod();
        return *this;
    }
    mint& operator*=(const mint& rhs) {
        unsigned long long z = _v;
        z *= rhs._v;
        _v = (unsigned int)(z % umod());
        return *this;
    }
    mint& operator/=(const mint& rhs) { return *this = *this * rhs.inv(); }

    mint operator+() const { return *this; }
    mint operator-() const { return mint() - *this; }

    mint pow(long long n) const {
        assert(0 <= n);
        mint x = *this, r = 1;
        while (n) {
            if (n & 1) r *= x;
            x *= x;
            n >>= 1;
        }
        return r;
    }
    mint inv() const {
        if (prime) {
            assert(_v);
            return pow(umod() - 2);
        } else {
            auto eg = internal::inv_gcd(_v, m);
            assert(eg.first == 1);
            return eg.second;
        }
    }

    friend mint operator+(const mint& lhs, const mint& rhs) {
        return mint(lhs) += rhs;
    }
    friend mint operator-(const mint& lhs, const mint& rhs) {
        return mint(lhs) -= rhs;
    }
    friend mint operator*(const mint& lhs, const mint& rhs) {
        return mint(lhs) *= rhs;
    }
    friend mint operator/(const mint& lhs, const mint& rhs) {
        return mint(lhs) /= rhs;
    }
    friend bool operator==(const mint& lhs, const mint& rhs) {
        return lhs._v == rhs._v;
    }
    friend bool operator!=(const mint& lhs, const mint& rhs) {
        return lhs._v != rhs._v;
    }

  private:
    unsigned int _v;
    static constexpr unsigned int umod() { return m; }
    static constexpr bool prime = internal::is_prime<m>;
};

template <int id> struct dynamic_modint : internal::modint_base {
    using mint = dynamic_modint;

  public:
    static int mod() { return (int)(bt.umod()); }
    static void set_mod(int m) {
        assert(1 <= m);
        bt = internal::barrett(m);
    }
    static mint raw(int v) {
        mint x;
        x._v = v;
        return x;
    }

    dynamic_modint() : _v(0) {}
    template <class T, internal::is_signed_int_t<T>* = nullptr>
    dynamic_modint(T v) {
        long long x = (long long)(v % (long long)(mod()));
        if (x < 0) x += mod();
        _v = (unsigned int)(x);
    }
    template <class T, internal::is_unsigned_int_t<T>* = nullptr>
    dynamic_modint(T v) {
        _v = (unsigned int)(v % mod());
    }

    int val() const { return _v; }

    mint& operator++() {
        _v++;
        if (_v == umod()) _v = 0;
        return *this;
    }
    mint& operator--() {
        if (_v == 0) _v = umod();
        _v--;
        return *this;
    }
    mint operator++(int) {
        mint result = *this;
        ++*this;
        return result;
    }
    mint operator--(int) {
        mint result = *this;
        --*this;
        return result;
    }

    mint& operator+=(const mint& rhs) {
        _v += rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator-=(const mint& rhs) {
        _v += mod() - rhs._v;
        if (_v >= umod()) _v -= umod();
        return *this;
    }
    mint& operator*=(const mint& rhs) {
        _v = bt.mul(_v, rhs._v);
        return *this;
    }
    mint& operator/=(const mint& rhs) { return *this = *this * rhs.inv(); }

    mint operator+() const { return *this; }
    mint operator-() const { return mint() - *this; }

    mint pow(long long n) const {
        assert(0 <= n);
        mint x = *this, r = 1;
        while (n) {
            if (n & 1) r *= x;
            x *= x;
            n >>= 1;
        }
        return r;
    }
    mint inv() const {
        auto eg = internal::inv_gcd(_v, mod());
        assert(eg.first == 1);
        return eg.second;
    }

    friend mint operator+(const mint& lhs, const mint& rhs) {
        return mint(lhs) += rhs;
    }
    friend mint operator-(const mint& lhs, const mint& rhs) {
        return mint(lhs) -= rhs;
    }
    friend mint operator*(const mint& lhs, const mint& rhs) {
        return mint(lhs) *= rhs;
    }
    friend mint operator/(const mint& lhs, const mint& rhs) {
        return mint(lhs) /= rhs;
    }
    friend bool operator==(const mint& lhs, const mint& rhs) {
        return lhs._v == rhs._v;
    }
    friend bool operator!=(const mint& lhs, const mint& rhs) {
        return lhs._v != rhs._v;
    }

  private:
    unsigned int _v;
    static internal::barrett bt;
    static unsigned int umod() { return bt.umod(); }
};
template <int id> internal::barrett dynamic_modint<id>::bt(998244353);

using modint998244353 = static_modint<998244353>;
using modint1000000007 = static_modint<1000000007>;
using modint = dynamic_modint<-1>;

namespace internal {

template <class T>
using is_static_modint = std::is_base_of<internal::static_modint_base, T>;

template <class T>
using is_static_modint_t = std::enable_if_t<is_static_modint<T>::value>;

template <class> struct is_dynamic_modint : public std::false_type {};
template <int id>
struct is_dynamic_modint<dynamic_modint<id>> : public std::true_type {};

template <class T>
using is_dynamic_modint_t = std::enable_if_t<is_dynamic_modint<T>::value>;

}  // namespace internal

}  // namespace atcoder

namespace noya {

namespace multiplicative_function_sum_internal {

template <int Mod> class linear_prime_power_sum {
public:
  using mint = atcoder::static_modint<Mod>;
  using u64 = std::uint64_t;
  using u128 = unsigned __int128;

  linear_prime_power_sum(u64 lim, mint ce, mint cp)
      : lm_(lim), ce_(ce), cp_(cp) {
    if (lm_ == 0) {
      return;
    }
    sq_ = integer_square_root(lm_);
    hl_ = lm_ / sq_;
    if (hl_ != 1 && lm_ / (hl_ - 1) == sq_) {
      hl_--;
    }
    nb_ = int(hl_ + sq_);
    ps = enumerate_primes(int(sq_));
    build_prime_tables();
  }

  mint run() {
    if (lm_ == 0) {
      return 0;
    }
    an_ = mint(1) + prime_prefix(lm_);
    for (int idx = 0; idx < int(ps.size()); idx++) {
      enumerate_composites(idx, 1, ps[idx], mint(1));
    }
    return an_;
  }

private:
  struct prime_aggregate {
    mint cnt = 0;
    mint sum = 0;
  };

  u64 lm_ = 0;
  u64 sq_ = 0;
  u64 hl_ = 0;
  int nb_ = 0;
  mint ce_ = 0;
  mint cp_ = 0;
  std::vector<int> ps;
  std::vector<prime_aggregate> pt_;
  mint an_ = 0;

  static u64 integer_square_root(u64 val) {
    u64 rt = u64(std::sqrt(static_cast<long double>(val)));
    while (u128(rt + 1) * (rt + 1) <= val) {
      rt++;
    }
    while (u128(rt) * rt > val) {
      rt--;
    }
    return rt;
  }

  static std::vector<int> enumerate_primes(int lim) {
    if (lim < 2) {
      return {};
    }
    std::vector<bool> ip(lim + 1, true);
    ip[0] = ip[1] = false;
    for (int p = 2; 1LL * p * p <= lim; p++) {
      if (!ip[p]) {
        continue;
      }
      for (int mul = p * p; mul <= lim; mul += p) {
        ip[mul] = false;
      }
    }
    std::vector<int> res;
    for (int val = 2; val <= lim; val++) {
      if (ip[val]) {
        res.push_back(val);
      }
    }
    return res;
  }

  int index_of(u64 val) const {
    assert(1 <= val && val <= lm_);
    int idx = val <= sq_ ? nb_ - int(val) : int(lm_ / val);
    assert(0 <= idx && idx < nb_);
    return idx;
  }

  u64 value_at(int idx) const {
    assert(0 <= idx && idx < nb_);
    if (u64(idx) < hl_) {
      return idx == 0 ? 0 : lm_ / u64(idx);
    }
    return u64(nb_ - idx);
  }

  void build_prime_tables() {
    pt_.resize(nb_);
    mint iv2 = mint(2).inv();
    for (int idx = 1; idx < nb_; idx++) {
      u64 val = value_at(idx);
      pt_[idx].cnt = mint(val - 1);
      mint cnv = mint(val);
      pt_[idx].sum = cnv * (cnv + 1) * iv2 - 1;
    }

    for (int p : ps) {
      u64 squ = u64(p) * p;
      prime_aggregate pre = pt_[index_of(p - 1)];
      u64 mh = std::min(hl_, lm_ / squ + 1);
      for (u64 idx = 1; idx < mh; idx++) {
        u64 red = lm_ / (idx * u64(p));
        prime_aggregate dif{pt_[index_of(red)].cnt - pre.cnt,
                            pt_[index_of(red)].sum - pre.sum};
        pt_[int(idx)].cnt -= dif.cnt;
        pt_[int(idx)].sum -= mint(p) * dif.sum;
      }
      for (u64 val = sq_; val >= squ; val--) {
        int idx = index_of(val);
        prime_aggregate dif{pt_[index_of(val / p)].cnt - pre.cnt,
                            pt_[index_of(val / p)].sum - pre.sum};
        pt_[idx].cnt -= dif.cnt;
        pt_[idx].sum -= mint(p) * dif.sum;
      }
    }
  }

  mint prime_prefix(u64 val) const {
    const prime_aggregate &agg = pt_[index_of(val)];
    return ce_ * agg.cnt + cp_ * agg.sum;
  }

  mint prime_power_value(int p, int exp) const { return ce_ * exp + cp_ * p; }

  void enumerate_composites(int pi, int exp, u64 prd, mint pv) {
    int p = ps[pi];
    an_ += pv * prime_power_value(p, exp + 1);

    u64 lim = lm_ / prd;
    if (u128(p) * p <= lim) {
      enumerate_composites(pi, exp + 1, prd * u64(p), pv);
    }

    pv *= prime_power_value(p, exp);
    an_ += pv * (prime_prefix(lim) - prime_prefix(u64(p)));

    int nxt = pi + 1;
    while (nxt < int(ps.size()) && u128(ps[nxt]) * ps[nxt] * ps[nxt] <= lim) {
      enumerate_composites(nxt, 1, prd * u64(ps[nxt]), pv);
      nxt++;
    }
    while (nxt < int(ps.size()) && u128(ps[nxt]) * ps[nxt] <= lim) {
      int np = ps[nxt];
      mint con = prime_power_value(np, 2);
      con += prime_power_value(np, 1) *
             (prime_prefix(lim / u64(np)) - prime_prefix(u64(np)));
      an_ += pv * con;
      nxt++;
    }
  }
};

} // namespace multiplicative_function_sum_internal

/// @brief Sum a multiplicative function satisfying f(p^e)=a*e+b*p for every
/// prime power. Quotient blocks first store both pi(x) and the sum of primes
/// up to x. A recursion then fixes the least prime factor and its exponent;
/// the remaining one-prime tail is aggregated from those two tables instead
/// of visited individually. Each integer is represented by its unique prime
/// factorization, so the accumulated contributions are disjoint and complete.
template <int Mod>
atcoder::static_modint<Mod>
sum_linear_prime_power_multiplicative(std::uint64_t lim,
                                      atcoder::static_modint<Mod> ce,
                                      atcoder::static_modint<Mod> cp) {
  return multiplicative_function_sum_internal::linear_prime_power_sum<Mod>(
             lim, ce, cp)
      .run();
}

} // namespace noya