結果
| 問題 | No.3653 Space-Time Courier |
| コンテスト | |
| ユーザー |
apricity
|
| 提出日時 | 2026-08-28 23:35:23 |
| 言語 | C++23 (gcc 15.2.0 + boost 1.90.0) |
| 結果 |
AC
|
| 実行時間 | 2,105 ms / 4,000 ms |
| + 811µs | |
| コード長 | 36,286 bytes |
| 記録 | |
| コンパイル時間 | 3,000 ms |
| コンパイル使用メモリ | 389,644 KB |
| 実行使用メモリ | 103,552 KB |
| 最終ジャッジ日時 | 2026-08-28 23:37:06 |
| 合計ジャッジ時間 | 36,948 ms |
|
ジャッジサーバーID (参考情報) |
judge3_1 / judge2_0 |
(要ログイン)
| ファイルパターン | 結果 |
|---|---|
| sample | AC * 2 |
| other | AC * 28 |
ソースコード
#pragma GCC target("avx2")
#ifdef LOCAL
#include "template.hpp"
#else
#include<iostream>
#include<string>
#include<vector>
#include<algorithm>
#include<numeric>
#include<cmath>
#include<utility>
#include<tuple>
#include<array>
#include<cstdint>
#include<cstdio>
#include<iomanip>
#include<map>
#include<set>
#include<unordered_map>
#include<unordered_set>
#include<queue>
#include<stack>
#include<deque>
#include<bitset>
#include<cctype>
#include<chrono>
#include<random>
#include<cassert>
#include<cstddef>
#include<iterator>
#include<string_view>
#include<type_traits>
#include<functional>
using namespace std;
namespace io {
template <typename T, typename U>
istream &operator>>(istream &is, pair<T, U> &p) {
is >> p.first >> p.second;
return is;
}
template <size_t N = 0, typename T>
istream& cin_tuple_impl(istream &is, T &t) {
if constexpr (N < std::tuple_size<T>::value) {
auto &x = std::get<N>(t);
is >> x;
cin_tuple_impl<N + 1>(is, t);
}
return is;
}
template <class... T>
istream &operator>>(istream &is, tuple<T...> &t) {
return cin_tuple_impl(is, t);
}
template <typename T, size_t N = 0>
istream &operator>>(istream &is, array<T, N> &v) {
for (auto &x : v) is >> x;
return is;
}
template <typename T>
istream &operator>>(istream &is, vector<T> &v) {
for (auto &x : v) is >> x;
return is;
}
template<typename T, typename U>
ostream &operator<<(ostream &os, const pair<T, U> &p) {
os << p.first << " " << p.second;
return os;
}
template <size_t N = 0, typename T>
ostream& cout_tuple_impl(ostream &os, const T &t) {
if constexpr (N < std::tuple_size<T>::value) {
if constexpr (N > 0) os << " ";
const auto &x = std::get<N>(t);
os << x;
cout_tuple_impl<N + 1>(os, t);
}
return os;
}
template <class... T>
ostream &operator<<(ostream &os, const tuple<T...> &t) {
return cout_tuple_impl(os, t);
}
template<typename T, size_t N>
ostream &operator<<(ostream &os, const array<T, N> &v) {
size_t n = v.size();
for (size_t i = 0; i < n; i++) {
if (i) os << " ";
os << v[i];
}
return os;
}
template<typename T>
ostream &operator<<(ostream &os, const vector<T> &v) {
int s = (int)v.size();
for (int i = 0; i < s; i++) os << (i ? " " : "") << v[i];
return os;
}
void in() {}
template<typename T, class... U>
void in(T &t, U &...u) {
cin >> t;
in(u...);
}
void out() { cout << "\n"; }
template<typename T, class... U, char sep = ' '>
void out(const T &t, const U &...u) {
cout << t;
if (sizeof...(u)) cout << sep;
out(u...);
}
void outr() {}
template<typename T, class... U, char sep = ' '>
void outr(const T &t, const U &...u) {
cout << t;
outr(u...);
}
void __attribute__((constructor)) _c() {
ios_base::sync_with_stdio(false);
cin.tie(nullptr);
cout << fixed << setprecision(15);
}
} // namespace io
using io::in;
using io::out;
using io::outr;
#define SHOW(x) static_cast<void>(0)
using ll = long long;
using D = double;
using LD = long double;
using P = pair<ll, ll>;
using u8 = uint8_t;
using u16 = uint16_t;
using u32 = uint32_t;
using u64 = uint64_t;
using i128 = __int128;
using u128 = unsigned __int128;
using vi = vector<ll>;
template <class T> using vc = vector<T>;
template <class T> using vvc = vector<vc<T>>;
template <class T> using vvvc = vector<vvc<T>>;
template <class T> using vvvvc = vector<vvvc<T>>;
template <class T> using vvvvvc = vector<vvvvc<T>>;
#define vv(type, name, h, ...) \
vector<vector<type>> name(h, vector<type>(__VA_ARGS__))
#define vvv(type, name, h, w, ...) \
vector<vector<vector<type>>> name( \
h, vector<vector<type>>(w, vector<type>(__VA_ARGS__)))
#define vvvv(type, name, a, b, c, ...) \
vector<vector<vector<vector<type>>>> name( \
a, vector<vector<vector<type>>>( \
b, vector<vector<type>>(c, vector<type>(__VA_ARGS__))))
template<typename T> using PQ = priority_queue<T,vector<T>>;
template<typename T> using minPQ = priority_queue<T, vector<T>, greater<T>>;
#define rep1(a) for(ll i = 0; i < a; i++)
#define rep2(i, a) for(ll i = 0; i < a; i++)
#define rep3(i, a, b) for(ll i = a; i < b; i++)
#define rep4(i, a, b, c) for(ll i = a; i < b; i += c)
#define overload4(a, b, c, d, e, ...) e
#define rep(...) overload4(__VA_ARGS__, rep4, rep3, rep2, rep1)(__VA_ARGS__)
#define rrep1(a) for(ll i = (a)-1; i >= 0; i--)
#define rrep2(i, a) for(ll i = (a)-1; i >= 0; i--)
#define rrep3(i, a, b) for(ll i = (b)-1; i >= a; i--)
#define rrep4(i, a, b, c) for(ll i = (b)-1; i >= a; i -= c)
#define rrep(...) overload4(__VA_ARGS__, rrep4, rrep3, rrep2, rrep1)(__VA_ARGS__)
#define for_subset(t, s) for (ll t = (s); t >= 0; t = (t == 0 ? -1 : (t - 1) & (s)))
#define ALL(v) v.begin(), v.end()
#define RALL(v) v.rbegin(), v.rend()
#define UNIQUE(v) v.erase( unique(v.begin(), v.end()), v.end() )
#define SZ(v) ll(v.size())
#define MIN(v) *min_element(ALL(v))
#define MAX(v) *max_element(ALL(v))
#define LB(c, x) distance((c).begin(), lower_bound(ALL(c), (x)))
#define UB(c, x) distance((c).begin(), upper_bound(ALL(c), (x)))
template <typename T, typename U>
T SUM(const vector<U> &v) {
T res = 0;
for(auto &&a : v) res += a;
return res;
}
template <typename T>
vector<pair<T,int>> RLE(const vector<T> &v) {
if (v.empty()) return {};
T cur = v.front();
int cnt = 1;
vector<pair<T,int>> res;
for (int i = 1; i < (int)v.size(); i++) {
if (cur == v[i]) cnt++;
else {
res.emplace_back(cur, cnt);
cnt = 1; cur = v[i];
}
}
res.emplace_back(cur, cnt);
return res;
}
template<class T, class S>
inline bool chmax(T &a, const S &b) { return (a < b ? a = b, true : false); }
template<class T, class S>
inline bool chmin(T &a, const S &b) { return (a > b ? a = b, true : false); }
void YESNO(bool flag) { out(flag ? "YES" : "NO"); }
void yesno(bool flag) { out(flag ? "Yes" : "No"); }
int popcnt(int x) { return __builtin_popcount(x); }
int popcnt(u32 x) { return __builtin_popcount(x); }
int popcnt(ll x) { return __builtin_popcountll(x); }
int popcnt(u64 x) { return __builtin_popcountll(x); }
int popcnt_sgn(int x) { return (__builtin_parity(x) & 1 ? -1 : 1); }
int popcnt_sgn(u32 x) { return (__builtin_parity(x) & 1 ? -1 : 1); }
int popcnt_sgn(ll x) { return (__builtin_parityl(x) & 1 ? -1 : 1); }
int popcnt_sgn(u64 x) { return (__builtin_parityl(x) & 1 ? -1 : 1); }
int highbit(int x) { return (x == 0 ? -1 : 31 - __builtin_clz(x)); }
int highbit(u32 x) { return (x == 0 ? -1 : 31 - __builtin_clz(x)); }
int highbit(ll x) { return (x == 0 ? -1 : 63 - __builtin_clzll(x)); }
int highbit(u64 x) { return (x == 0 ? -1 : 63 - __builtin_clzll(x)); }
int lowbit(int x) { return (x == 0 ? -1 : __builtin_ctz(x)); }
int lowbit(u32 x) { return (x == 0 ? -1 : __builtin_ctz(x)); }
int lowbit(ll x) { return (x == 0 ? -1 : __builtin_ctzll(x)); }
int lowbit(u64 x) { return (x == 0 ? -1 : __builtin_ctzll(x)); }
template <typename T>
T get_bit(T x, int k) { return x >> k & 1; }
template <typename T>
T set_bit(T x, int k) { return x | T(1) << k; }
template <typename T>
T reset_bit(T x, int k) { return x & ~(T(1) << k); }
template <typename T>
T flip_bit(T x, int k) { return x ^ T(1) << k; }
template <typename T>
T popf(deque<T> &que) { T a = que.front(); que.pop_front(); return a; }
template <typename T>
T popb(deque<T> &que) { T a = que.back(); que.pop_back(); return a; }
template <typename T>
T pop(queue<T> &que) { T a = que.front(); que.pop(); return a; }
template <typename T>
T pop(stack<T> &que) { T a = que.top(); que.pop(); return a; }
template <typename T>
T pop(PQ<T> &que) { T a = que.top(); que.pop(); return a; }
template <typename T>
T pop(minPQ<T> &que) { T a = que.top(); que.pop(); return a; }
template <typename F>
ll binary_search(F check, ll ok, ll ng, bool check_ok = true) {
if (check_ok) assert(check(ok));
while (abs(ok - ng) > 1) {
ll mid = (ok + ng) / 2;
(check(mid) ? ok : ng) = mid;
}
return ok;
}
template <typename F>
double binary_search_real(F check, double ok, double ng, int iter = 60) {
for (int _ = 0; _ < iter; _++) {
double mid = (ok + ng) / 2;
(check(mid) ? ok : ng) = mid;
}
return (ok + ng) / 2;
}
// max x s.t. b*x <= a
ll div_floor(ll a, ll b) {
assert(b != 0);
if (b < 0) a = -a, b = -b;
return a / b - (a % b < 0);
}
// max x s.t. b*x < a
ll div_under(ll a, ll b) {
assert(b != 0);
if (b < 0) a = -a, b = -b;
return a / b - (a % b <= 0);
}
// min x s.t. b*x >= a
ll div_ceil(ll a, ll b) {
assert(b != 0);
if (b < 0) a = -a, b = -b;
return a / b + (a % b > 0);
}
// min x s.t. b*x > a
ll div_over(ll a, ll b) {
assert(b != 0);
if (b < 0) a = -a, b = -b;
return a / b + (a % b >= 0);
}
// x = a mod b (b > 0), 0 <= x < b
ll modulo(ll a, ll b) {
assert(b > 0);
ll c = a % b;
return c < 0 ? c + b : c;
}
// (q,r) s.t. a = b*q + r, 0 <= r < b (b > 0)
// div_floor(a,b), modulo(a,b)
pair<ll,ll> divmod(ll a, ll b) {
ll q = div_floor(a,b);
return {q, a - b*q};
}
#endif
// #pragma once
#include <cstdint>
#include <type_traits>
#include <string>
#include <algorithm>
#include <immintrin.h>
namespace quick_floyd_warshall {
namespace vectorize {
// instruction set for vectorization
enum class InstSet {
DEFAULT,
SSE4_2,
AVX2,
AVX512
};
std::string inst_set_to_str(InstSet inst_set) {
if (inst_set == InstSet::DEFAULT) return "DEFAULT";
if (inst_set == InstSet::SSE4_2 ) return "SSE4_2";
if (inst_set == InstSet::AVX2 ) return "AVX2";
if (inst_set == InstSet::AVX512 ) return "AVX512";
return "";
}
// wrapper of sse/avx intrinsics
template<InstSet inst_set> class vector_base_t;
template<> class vector_base_t<InstSet::SSE4_2> {
public:
static constexpr int SIZE = 16;
using internal_vector_t = __m128i;
internal_vector_t vec;
vector_base_t () = default;
vector_base_t (internal_vector_t vec_) : vec(vec_) {}
vector_base_t (void *ptr) : vec(_mm_load_si128((internal_vector_t *) ptr)) {}
vector_base_t &store(void *ptr) { _mm_store_si128((internal_vector_t *) ptr, vec); return *this; }
};
template<InstSet inst_set> class vector_base_t;
template<> class vector_base_t<InstSet::AVX2> {
public:
static constexpr int SIZE = 32;
using internal_vector_t = __m256i;
internal_vector_t vec;
vector_base_t () = default;
vector_base_t (internal_vector_t vec_) : vec(vec_) {}
vector_base_t (void *ptr) : vec(_mm256_load_si256((internal_vector_t *) ptr)) {}
vector_base_t &store(void *ptr) { _mm256_store_si256((internal_vector_t *) ptr, vec); return *this; }
};
template<> class vector_base_t<InstSet::AVX512> {
public:
static constexpr int SIZE = 64;
using internal_vector_t = __m512i;
internal_vector_t vec;
vector_base_t () = default;
vector_base_t (internal_vector_t vec_) : vec(vec_) {}
vector_base_t (void *ptr) : vec(_mm512_load_si512((internal_vector_t *) ptr)) {}
vector_base_t &store(void *ptr) { _mm512_store_si512((internal_vector_t *) ptr, vec); return *this; }
};
template<InstSet inst_set, typename T> class vector_t;
/*
vec.chmin_store(mem): mem[i] = min(mem[i], vec[i])
vec.chmax_store(mem): mem[i] = max(mem[i], vec[i])
*/
// DEFAULT / *
template<typename T> class vector_t<InstSet::DEFAULT, T> {
static_assert(std::is_same<T, int16_t>::value || std::is_same<T, int32_t>::value || std::is_same<T, int64_t>::value, "");
public:
static constexpr int SIZE = sizeof(T);
T val;
vector_t &store(void *ptr) { *((T *) ptr) = val; return *this; }
vector_t (void *val) : val(*((T *)val)) {}
vector_t (T val) : val(val) {}
vector_t operator + (const vector_t &rhs) const { return { T(val + rhs.val) }; }
vector_t operator - (const vector_t &rhs) const { return { T(val - rhs.val) }; }
vector_t operator - () const { return { -val }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { std::min(lhs.val, rhs.val) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { std::max(lhs.val, rhs.val) }; }
vector_t &chmin_store(void *ptr) { if (*((T *) ptr) > val) store(ptr); return *this; }
vector_t &chmax_store(void *ptr) { if (*((T *) ptr) < val) store(ptr); return *this; }
};
// SSE4.2 / int16_t
template<> class vector_t<InstSet::SSE4_2, int16_t> : public vector_base_t<InstSet::SSE4_2> {
public:
using vector_base_t<InstSet::SSE4_2>::vector_base_t;
vector_t (int16_t val) : vector_base_t(_mm_set1_epi16(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm_add_epi16(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm_sub_epi16(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm_sub_epi16(_mm_setzero_si128(), vec) }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { _mm_min_epi16(lhs.vec, rhs.vec) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { _mm_max_epi16(lhs.vec, rhs.vec) }; }
vector_t &chmin_store(void *ptr) { min(*this, vector_t(ptr)).store(ptr); return *this; }
vector_t &chmax_store(void *ptr) { max(*this, vector_t(ptr)).store(ptr); return *this; }
};
// SSE4.2 / int32_t
template<> class vector_t<InstSet::SSE4_2, int32_t> : public vector_base_t<InstSet::SSE4_2> {
public:
using vector_base_t<InstSet::SSE4_2>::vector_base_t;
vector_t (int32_t val) : vector_base_t(_mm_set1_epi32(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm_add_epi32(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm_sub_epi32(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm_sub_epi32(_mm_setzero_si128(), vec) }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { _mm_min_epi32(lhs.vec, rhs.vec) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { _mm_max_epi32(lhs.vec, rhs.vec) }; }
vector_t &chmin_store(void *ptr) { min(*this, vector_t(ptr)).store(ptr); return *this; }
vector_t &chmax_store(void *ptr) { max(*this, vector_t(ptr)).store(ptr); return *this; }
};
// SSE4.2 / int64_t
template<> class vector_t<InstSet::SSE4_2, int64_t> : public vector_base_t<InstSet::SSE4_2> {
public:
using vector_base_t<InstSet::SSE4_2>::vector_base_t;
vector_t (int64_t val) : vector_base_t(_mm_set1_epi64((__m64) val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm_add_epi64(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm_sub_epi64(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm_sub_epi64(_mm_setzero_si128(), vec) }; }
// SSE4 doesn't have _mm_min_epi64 / _mm_max_epi64
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return {
_mm_blendv_epi8(lhs.vec, rhs.vec, _mm_cmpgt_epi64(lhs.vec, rhs.vec))
}; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return {
_mm_blendv_epi8(lhs.vec, rhs.vec, _mm_cmpgt_epi64(rhs.vec, lhs.vec))
}; }
vector_t &chmin_store(void *ptr) { min(*this, vector_t(ptr)).store(ptr); return *this; }
vector_t &chmax_store(void *ptr) { max(*this, vector_t(ptr)).store(ptr); return *this; }
};
// AVX2 / int16_t
template<> class vector_t<InstSet::AVX2, int16_t> : public vector_base_t<InstSet::AVX2> {
public:
using vector_base_t<InstSet::AVX2>::vector_base_t;
vector_t (int16_t val) : vector_base_t(_mm256_set1_epi16(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm256_add_epi16(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm256_sub_epi16(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm256_sub_epi16(_mm256_setzero_si256(), vec) }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { _mm256_min_epi16(lhs.vec, rhs.vec) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { _mm256_max_epi16(lhs.vec, rhs.vec) }; }
vector_t &chmin_store(void *ptr) { min(*this, vector_t(ptr)).store(ptr); return *this; }
vector_t &chmax_store(void *ptr) { max(*this, vector_t(ptr)).store(ptr); return *this; }
};
// AVX2 / int32_t
template<> class vector_t<InstSet::AVX2, int32_t> : public vector_base_t<InstSet::AVX2> {
public:
using vector_base_t<InstSet::AVX2>::vector_base_t;
vector_t (int32_t val) : vector_base_t(_mm256_set1_epi32(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm256_add_epi32(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm256_sub_epi32(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm256_sub_epi32(_mm256_setzero_si256(), vec) }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { _mm256_min_epi32(lhs.vec, rhs.vec) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { _mm256_max_epi32(lhs.vec, rhs.vec) }; }
vector_t &chmin_store(void *ptr) { min(*this, vector_t(ptr)).store(ptr); return *this; }
vector_t &chmax_store(void *ptr) { max(*this, vector_t(ptr)).store(ptr); return *this; }
};
// AVX2 / int64_t
template<> class vector_t<InstSet::AVX2, int64_t> : public vector_base_t<InstSet::AVX2> {
public:
using vector_base_t<InstSet::AVX2>::vector_base_t;
vector_t (int64_t val) : vector_base_t(_mm256_set1_epi64x(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm256_add_epi64(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm256_sub_epi64(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm256_sub_epi64(_mm256_setzero_si256(), vec) }; }
// avx2 doesn't have _mm256_min_epi64 / _mm256_max_epi64
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return {
_mm256_blendv_epi8(lhs.vec, rhs.vec, _mm256_cmpgt_epi64(lhs.vec, rhs.vec))
}; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return {
_mm256_blendv_epi8(lhs.vec, rhs.vec, _mm256_cmpgt_epi64(rhs.vec, lhs.vec))
}; }
vector_t &chmin_store(void *ptr) { // slower because of separate load instruction in the 1st operand of cmpgt
_mm256_maskstore_epi64((long long *) ptr, _mm256_cmpgt_epi64(vector_t(ptr).vec, vec), vec);
return *this;
}
vector_t &chmax_store(void *ptr) { // faster because cmpgt allows memory address as the 2nd operand
_mm256_maskstore_epi64((long long *) ptr, _mm256_cmpgt_epi64(vec, vector_t(ptr).vec), vec);
return *this;
}
};
// AVX512 / int16_t
template<> class vector_t<InstSet::AVX512, int16_t> : public vector_base_t<InstSet::AVX512> {
public:
using vector_base_t<InstSet::AVX512>::vector_base_t;
vector_t (int16_t val) : vector_base_t(_mm512_set1_epi16(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm512_add_epi16(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm512_sub_epi16(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm512_sub_epi16(_mm512_setzero_si512(), vec) }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { _mm512_min_epi16(lhs.vec, rhs.vec) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { _mm512_max_epi16(lhs.vec, rhs.vec) }; }
vector_t &chmin_store(void *ptr) {
_mm512_mask_storeu_epi16((internal_vector_t *) (ptr),
_mm512_cmp_epi16_mask(vec, _mm512_load_si512((internal_vector_t *) ptr), _MM_CMPINT_LT), vec);
return *this;
}
vector_t &chmax_store(void *ptr) {
_mm512_mask_storeu_epi16((internal_vector_t *) (ptr),
_mm512_cmp_epi16_mask(vec, _mm512_load_si512((internal_vector_t *) ptr), _MM_CMPINT_GT), vec);
return *this;
}
};
// AVX512 / int32_t
template<> class vector_t<InstSet::AVX512, int32_t> : public vector_base_t<InstSet::AVX512> {
public:
using vector_base_t<InstSet::AVX512>::vector_base_t;
vector_t (int32_t val) : vector_base_t(_mm512_set1_epi32(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm512_add_epi32(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm512_sub_epi32(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm512_sub_epi32(_mm512_setzero_si512(), vec) }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { _mm512_min_epi32(lhs.vec, rhs.vec) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { _mm512_max_epi32(lhs.vec, rhs.vec) }; }
vector_t &chmin_store(void *ptr) {
_mm512_mask_store_epi32((internal_vector_t *) (ptr),
_mm512_cmp_epi32_mask(vec, _mm512_load_si512((internal_vector_t *) ptr), _MM_CMPINT_LT), vec);
return *this;
}
vector_t &chmax_store(void *ptr) {
_mm512_mask_store_epi32((internal_vector_t *) (ptr),
_mm512_cmp_epi32_mask(vec, _mm512_load_si512((internal_vector_t *) ptr), _MM_CMPINT_GT), vec);
return *this;
}
};
// AVX512 / int64_t
template<> class vector_t<InstSet::AVX512, int64_t> : public vector_base_t<InstSet::AVX512> {
public:
using vector_base_t<InstSet::AVX512>::vector_base_t;
vector_t (int64_t val) : vector_base_t(_mm512_set1_epi64(val)) {}
vector_t operator + (const vector_t &rhs) const { return { _mm512_add_epi64(vec, rhs.vec) }; }
vector_t operator - (const vector_t &rhs) const { return { _mm512_sub_epi64(vec, rhs.vec) }; }
vector_t operator - () const { return { _mm512_sub_epi64(_mm512_setzero_si512(), vec) }; }
friend vector_t min(const vector_t &lhs, const vector_t &rhs) { return { _mm512_min_epi64(lhs.vec, rhs.vec) }; }
friend vector_t max(const vector_t &lhs, const vector_t &rhs) { return { _mm512_max_epi64(lhs.vec, rhs.vec) }; }
vector_t &chmin_store(void *ptr) {
_mm512_mask_store_epi64((internal_vector_t *) (ptr),
_mm512_cmp_epi64_mask(vec, _mm512_load_si512((internal_vector_t *) ptr), _MM_CMPINT_LT), vec);
return *this;
}
vector_t &chmax_store(void *ptr) {
_mm512_mask_store_epi64((internal_vector_t *) (ptr),
_mm512_cmp_epi64_mask(vec, _mm512_load_si512((internal_vector_t *) ptr), _MM_CMPINT_GT), vec);
return *this;
}
};
template<InstSet inst_set, typename T> vector_t<inst_set, T> &operator += (
vector_t<inst_set, T> &lhs, const vector_t<inst_set, T> &rhs) {
lhs = lhs + rhs;
return lhs;
}
template<InstSet inst_set, typename T> vector_t<inst_set, T> &operator -= (
vector_t<inst_set, T> &lhs, const vector_t<inst_set, T> &rhs) {
lhs = lhs - rhs;
return lhs;
}
} // namespace vectorize
} // namespace quick_floyd_warshall
// #pragma once
#include <cassert>
#include <cstdlib>
#include <cstring>
#include <string>
#include <memory>
#include <type_traits>
#include <limits>
// #include "internal/vectorize.h"
namespace quick_floyd_warshall {
template <class T, class = void> struct is_complete : std::false_type {};
template <class T> struct is_complete<T, decltype(void(sizeof(T)))> : std::true_type {};
template<typename T> struct floyd_warshall_naive {
using value_t = T;
static constexpr T INF = std::numeric_limits<T>::max() / 2;
static std::string get_description() { return "naive<int" + std::to_string(sizeof(value_t) * 8) + "_t>"; }
static void run(int n, const T *input_matrix, T *output_matrix, bool symmetric = false) {
(void) symmetric;
T *buf = (T *) malloc(n * n * sizeof(T));
memcpy(buf, input_matrix, n * n * sizeof(T));
for (int k = 0; k < n; k++) for (int i = 0; i < n; i++) for (int j = 0; j < n; j++)
buf[i * n + j] = std::min<T>(buf[i * n + j], buf[i * n + k] + buf[k * n + j]);
memcpy(output_matrix, buf, n * n * sizeof(T));
free(buf);
}
};
using InstSet = vectorize::InstSet;
template<InstSet inst_set, typename T, int unroll_type> struct floyd_warshall {
public:
static constexpr T INF = std::numeric_limits<T>::max() / 2;
using value_t = T;
static std::string get_description() {
return "opt<" + vectorize::inst_set_to_str(inst_set) + ", " "int" + std::to_string(sizeof(value_t) * 8) + "_t, "
+ std::to_string(unroll_type) + ">";
}
private:
static constexpr int B = 64; // block size
using vector_t = vectorize::vector_t<inst_set, T>;
static_assert(is_complete<vector_t>::value, "Invalid inst_set or T");
static_assert(B % (vector_t::SIZE / sizeof(T)) == 0, "Invalid B value");
static_assert(unroll_type >= 0 && unroll_type <= 3, "Invalid unroll_type value");
/*
MaxPlusMul?(a, b, c) :
- [a, a + B * B), [b, b + B * B), [c, c + B * B) must not overlap
- equivalent to:
for all i, j in [0, B):
a[i * B + j] = max(a[i * B + j], max{b[i * B + k] + c[k * B + j] | k in [0, B)})
*/
static void MaxPlusMul0(T *a, T *b, T *c) {
constexpr int n = B;
for (int k = 0; k < n; k += 2) for (int i = 0; i < n; i += 2) {
vector_t coef00(b[(i + 0) * n + (k + 0)]);
vector_t coef01(b[(i + 0) * n + (k + 1)]);
vector_t coef10(b[(i + 1) * n + (k + 0)]);
vector_t coef11(b[(i + 1) * n + (k + 1)]);
T *aa = a + i * n;
T *bb = c + k * n;
for (int j = 0; j < n; j += vector_t::SIZE / sizeof(T)) {
vector_t t0(bb + j);
vector_t t1(bb + n + j);
max(t0 + coef00, t1 + coef01).chmax_store(aa + j);
max(t0 + coef10, t1 + coef11).chmax_store(aa + n + j);
}
}
}
static void MaxPlusMul1(T *a, T *b, T *c) {
constexpr int n = B;
for (int k = 0; k < n; k += 4) for (int i = 0; i < n; i += 2) {
vector_t coef00(b[(i + 0) * n + (k + 0)]);
vector_t coef01(b[(i + 0) * n + (k + 1)]);
vector_t coef02(b[(i + 0) * n + (k + 2)]);
vector_t coef03(b[(i + 0) * n + (k + 3)]);
vector_t coef10(b[(i + 1) * n + (k + 0)]);
vector_t coef11(b[(i + 1) * n + (k + 1)]);
vector_t coef12(b[(i + 1) * n + (k + 2)]);
vector_t coef13(b[(i + 1) * n + (k + 3)]);
T *aa = a + i * n;
T *bb = c + k * n;
for (int j = 0; j < n; j += vector_t::SIZE / sizeof(T)) {
vector_t t0(bb + j);
vector_t t1(bb + n + j);
vector_t t2(bb + n + n + j);
vector_t t3(bb + n + n + n + j);
max(max(t0 + coef00, t1 + coef01), max(t2 + coef02, t3 + coef03)).chmax_store(aa + j);
max(max(t0 + coef10, t1 + coef11), max(t2 + coef12, t3 + coef13)).chmax_store(aa + n + j);
}
}
}
static void MaxPlusMul2(T *a, T *b, T *c) {
constexpr int n = B;
for (int k = 0; k < n; k += 2) for (int i = 0; i < n; i += 4) {
vector_t coef00(b[(i + 0) * n + (k + 0)]);
vector_t coef01(b[(i + 0) * n + (k + 1)]);
vector_t coef10(b[(i + 1) * n + (k + 0)]);
vector_t coef11(b[(i + 1) * n + (k + 1)]);
vector_t coef20(b[(i + 2) * n + (k + 0)]);
vector_t coef21(b[(i + 2) * n + (k + 1)]);
vector_t coef30(b[(i + 3) * n + (k + 0)]);
vector_t coef31(b[(i + 3) * n + (k + 1)]);
T *aa = a + i * n;
T *bb = c + k * n;
for (int j = 0; j < n; j += vector_t::SIZE / sizeof(T)) {
vector_t t0(bb + j);
vector_t t1(bb + n + j);
max(t0 + coef00, t1 + coef01).chmax_store(aa + j);
max(t0 + coef10, t1 + coef11).chmax_store(aa + n + j);
max(t0 + coef20, t1 + coef21).chmax_store(aa + n + n + j);
max(t0 + coef30, t1 + coef31).chmax_store(aa + n + n + n + j);
}
}
}
static void MaxPlusMul3(T *a, T *b, T *c) {
constexpr int n = B;
for (int k = 0; k < n; k += 4) for (int i = 0; i < n; i += 4) {
vector_t coef00(b[(i + 0) * n + (k + 0)]);
vector_t coef01(b[(i + 0) * n + (k + 1)]);
vector_t coef02(b[(i + 0) * n + (k + 2)]);
vector_t coef03(b[(i + 0) * n + (k + 3)]);
vector_t coef10(b[(i + 1) * n + (k + 0)]);
vector_t coef11(b[(i + 1) * n + (k + 1)]);
vector_t coef12(b[(i + 1) * n + (k + 2)]);
vector_t coef13(b[(i + 1) * n + (k + 3)]);
vector_t coef20(b[(i + 2) * n + (k + 0)]);
vector_t coef21(b[(i + 2) * n + (k + 1)]);
vector_t coef22(b[(i + 2) * n + (k + 2)]);
vector_t coef23(b[(i + 2) * n + (k + 3)]);
vector_t coef30(b[(i + 3) * n + (k + 0)]);
vector_t coef31(b[(i + 3) * n + (k + 1)]);
vector_t coef32(b[(i + 3) * n + (k + 2)]);
vector_t coef33(b[(i + 3) * n + (k + 3)]);
T *aa = a + i * n;
T *bb = c + k * n;
for (int j = 0; j < n; j += vector_t::SIZE / sizeof(T)) {
vector_t t0(bb + j);
vector_t t1(bb + n + j);
vector_t t2(bb + n + n + j);
vector_t t3(bb + n + n + n + j);
max(max(t0 + coef00, t1 + coef01), max(t2 + coef02, t3 + coef03)).chmax_store(aa + j);
max(max(t0 + coef10, t1 + coef11), max(t2 + coef12, t3 + coef13)).chmax_store(aa + n + j);
max(max(t0 + coef20, t1 + coef21), max(t2 + coef22, t3 + coef23)).chmax_store(aa + n + n + j);
max(max(t0 + coef30, t1 + coef31), max(t2 + coef32, t3 + coef33)).chmax_store(aa + n + n + n + j);
}
}
}
static void FWI(T *a, T *b, T *c) {
if (a != b && a != c && b != c) {
if (unroll_type == 0) MaxPlusMul0(a, b, c);
if (unroll_type == 1) MaxPlusMul1(a, b, c);
if (unroll_type == 2) MaxPlusMul2(a, b, c);
if (unroll_type == 3) MaxPlusMul3(a, b, c);
return;
}
constexpr int n = B;
for (int k = 0; k < n; k++) for (int i = 0; i < n; i++) {
vector_t coef(b[i * n + k]);
T *aa = a + i * n;
T *bb = c + k * n;
for (int j = 0; j < n; j += vector_t::SIZE / sizeof(T))
(vector_t(bb + j) + coef).chmax_store(aa + j);
}
}
static void FWR(int n_blocks_power2, int n_blocks, int block_index0, int block_index1, int block_index2,
T **block_start, bool symmetric) {
if (block_index0 >= n_blocks || block_index1 >= n_blocks || block_index2 >= n_blocks) return;
if (n_blocks_power2 == 1) {
FWI(
block_start[block_index0 * n_blocks + block_index2],
block_start[block_index0 * n_blocks + block_index1],
block_start[block_index1 * n_blocks + block_index2]
);
} else {
int half = n_blocks_power2 >> 1;
if (!symmetric) {
FWR(half, n_blocks, block_index0 , block_index1 , block_index2 , block_start, false);
FWR(half, n_blocks, block_index0 , block_index1 , block_index2 + half, block_start, false);
FWR(half, n_blocks, block_index0 + half, block_index1 , block_index2 , block_start, false);
FWR(half, n_blocks, block_index0 + half, block_index1 , block_index2 + half, block_start, false);
FWR(half, n_blocks, block_index0 + half, block_index1 + half, block_index2 + half, block_start, false);
FWR(half, n_blocks, block_index0 + half, block_index1 + half, block_index2 , block_start, false);
FWR(half, n_blocks, block_index0 , block_index1 + half, block_index2 + half, block_start, false);
FWR(half, n_blocks, block_index0 , block_index1 + half, block_index2 , block_start, false);
} else {
// if symmetric, block_index0 = block_index1 = block_index2
FWR(half, n_blocks, block_index0 , block_index1 , block_index2 , block_start, true);
FWR(half, n_blocks, block_index0 , block_index1 , block_index2 + half, block_start, false);
transpose_copy(half, n_blocks, block_index0, block_index0 + half, block_start);
FWR(half, n_blocks, block_index0 + half, block_index1 , block_index2 + half, block_start, false);
FWR(half, n_blocks, block_index0 + half, block_index1 + half, block_index2 + half, block_start, true);
FWR(half, n_blocks, block_index0 + half, block_index1 + half, block_index2 , block_start, false);
transpose_copy(half, n_blocks, block_index0 + half, block_index0, block_start);
FWR(half, n_blocks, block_index0 , block_index1 + half, block_index2 , block_start, false);
}
}
}
// copy [block_row_offset:block_row_offset+n)[block_column_offset:block_column_offset+n) to its transposed posititon
// anything outside n_blocks * n_blocks blocks is ignored
static void transpose_copy(int n, int n_blocks, int block_row_offset, int block_column_offset, T **block_start) {
for (int i = block_row_offset; i < block_row_offset + n && i < n_blocks; i++)
for (int j = block_column_offset; j < block_column_offset + n && j < n_blocks; j++) {
T *src = block_start[i * n_blocks + j];
T *dst = block_start[j * n_blocks + i];
for (int y = 0; y < B; y++) for (int x = 0; x < B; x++) dst[x * B + y] = src[y * B + x];
}
}
/*
rev == false :
Copy the n * n elements in src to dst in the order like this(each src[i][j] is a BxB block):
dst: src[0][0], src[0][1], src[1][0], src[1][1],
src[0][2], src[0][3], src[1][2], src[1][3],
src[2][0], src[2][1], src[3][0], src[3][1],
src[2][2], src[2][3], src[3][2], src[3][3],
src[0][4], src[0][5], src[1][4], src[1][5], ...
and returns the pointer to the next element of the last element written in dst.
Elements outside the n_blocks * n_blocks blocks will be ignored
Elements in dst corresponding to elements outside the src_n * src_n but in n_blocks * n_blocks
will be filled
block_start[i * n_blocks + j] will point to the starting element in dst corresponding to (i, j) block
rev == true :
same as rev == false except that the copy direction is reversed and
elements in dst where INF would be contained if !rev are untouched
This function negates all the element and FWR handles everything with max instead of min.
This is because chmax(mem, reg) can be implemented faster than chmin(mem, reg) with avx2+int64_t
The cost of negation should be negligible for other combinations, where this trick is irrelevant
*/
static T *reorder(int src_n, int n_blocks_power2, T *dst_head, T *src, T **block_start,
int block_row, int block_column, bool rev) {
int n_blocks = (src_n + B - 1) / B;
if (block_row >= n_blocks || block_column >= n_blocks) return dst_head;
if (n_blocks_power2 == 1) {
T *src_base = src + (block_row * B * src_n + block_column * B);
for (int i = 0; i < B; i++) {
if (block_row * B + i < src_n) {
int length = std::min(B, src_n - block_column * B);
if (!rev) {
for (int j = 0; j < length; j++) dst_head[i * B + j] = -src_base[i * src_n + j];
for (int j = length; j < B; j++) dst_head[i * B + j] = -INF;
} else {
for (int j = 0; j < length; j++) src_base[i * src_n + j] = -dst_head[i * B + j];
}
} else {
if (!rev) std::fill(dst_head + i * B, dst_head + (i + 1) * B, -INF);
}
}
block_start[block_row * n_blocks + block_column] = dst_head;
return dst_head + B * B;
} else {
int n_blocks_p2_half = n_blocks_power2 >> 1;
// split into 2x2 recursively
for (int i = 0; i < 2; i++) for (int j = 0; j < 2; j++)
dst_head = reorder(src_n, n_blocks_p2_half, dst_head, src, block_start,
block_row + i * n_blocks_p2_half, block_column + j * n_blocks_p2_half, rev);
return dst_head;
}
}
public:
static void run(int src_n, const T *input_matrix, T *output_matrix, bool symmetric = false) {
assert(0 <= src_n && src_n < 65536);
if (src_n == 0) return;
int n_blocks = (src_n + B - 1) / B; // number of BxB blocks in a row
int n_blocks_power2 = 1; // smallest power of 2 >= src_n / B
while (n_blocks_power2 * B < src_n) n_blocks_power2 *= 2;
// allocate and align the needed buffers
const size_t reordered_needed_size = (B * n_blocks) * (B * n_blocks) * sizeof(T);
size_t reordered_buffer_size = reordered_needed_size + 64;
void *reordered_org = malloc(reordered_buffer_size);
assert(reordered_org);
void *reordered = reordered_org;
assert(std::align(64, reordered_needed_size, reordered, reordered_buffer_size));
// block_start[i][j] : pointer to the starting element of the (i, j) block in t
T **block_start = (T **) malloc(n_blocks * n_blocks * sizeof(T *));
std::fill(block_start, block_start + n_blocks * n_blocks, nullptr);
reorder(src_n, n_blocks_power2, (T *) reordered, const_cast<T *>(input_matrix), block_start, 0, 0, false);
FWR(n_blocks_power2, n_blocks, 0, 0, 0, block_start, symmetric);
reorder(src_n, n_blocks_power2, (T *) reordered, output_matrix, block_start, 0, 0, true);
free(reordered_org);
free(block_start);
}
};
} // namespace quick_floyd_warshall
//https://github.com/windows-server-2003/quick_floyd_warshall
using namespace quick_floyd_warshall;
int64_t mat[2500*2500];
void solve() {
int n,m; in(n,m);
vc<ll> p(n); in(p);
rep(i,n*n) mat[i] = floyd_warshall<InstSet::AVX2,int64_t, 3>::INF;
rep(i,m){
int u,v,t; in(u,v,t); u--; v--;
mat[u*n+v] = t;
}
floyd_warshall<InstSet::AVX2,int64_t,3>::run(n,mat,mat);
int ans = 0;
ll mn = 1e18;
rep(i,n)rep(j,n) if(i!=j) chmin(mn, mat[i*n+j] + p[i] + p[j]);
rep(i,n)rep(j,n) if(i!=j) ans += mat[i*n+j]+p[i]+p[j] == mn;
out(mn,ans);
}
int main() {
int tc = 1;
// in(tc);
while(tc--){
solve();
}
}
apricity