import macros;macro ImportExpand(s:untyped):untyped = parseStmt($s[2]) # source: src/cplib/tmpl/sheep.nim ImportExpand "cplib/tmpl/sheep" <=== "when not declared CPLIB_TMPL_SHEEP:\n const CPLIB_TMPL_SHEEP* = 1\n {.warning[UnusedImport]: off.}\n {.hint[XDeclaredButNotUsed]: off.}\n import algorithm\n import sequtils\n import tables\n import macros\n import math\n import sets\n import strutils\n import strformat\n import sugar\n import heapqueue\n import streams\n import deques\n import bitops\n import std/lenientops\n import options\n #入力系\n {.emit: \"\"\"\n #include \n #include \n #include \n #include \n #include \n\n namespace cplib_sheep_input {\n constexpr std::size_t buffer_size = 1U << 20;\n char buffer[buffer_size];\n std::size_t cursor = 0;\n std::size_t length = 0;\n const char* mapped = nullptr;\n bool initialized = false;\n\n inline void initialize() {\n if (initialized) return;\n initialized = true;\n\n struct stat st;\n const int fd = fileno(stdin);\n if (fstat(fd, &st) == 0 && S_ISREG(st.st_mode) && st.st_size > 0) {\n void* p = mmap(nullptr, static_cast(st.st_size),\n PROT_READ, MAP_PRIVATE, fd, 0);\n if (p != MAP_FAILED) {\n mapped = static_cast(p);\n length = static_cast(st.st_size);\n madvise(const_cast(mapped), length, MADV_SEQUENTIAL);\n }\n }\n }\n\n inline int get_char() {\n initialize();\n if (mapped != nullptr) {\n if (cursor == length) return -1;\n return static_cast(mapped[cursor++]);\n }\n\n if (cursor == length) {\n length = fread_unlocked(buffer, 1, buffer_size, stdin);\n cursor = 0;\n if (length == 0) return -1;\n }\n return static_cast(buffer[cursor++]);\n }\n\n inline bool refill() {\n length = fread_unlocked(buffer, 1, buffer_size, stdin);\n cursor = 0;\n return length != 0;\n }\n\n inline bool has_eight_digits(const char* source) {\n std::uint64_t bytes;\n std::memcpy(&bytes, source, sizeof(bytes));\n constexpr std::uint64_t high_nibbles = 0xf0f0f0f0f0f0f0f0ULL;\n return (bytes & high_nibbles) == 0x3030303030303030ULL &&\n ((bytes + 0x0606060606060606ULL) & high_nibbles) ==\n 0x3030303030303030ULL;\n }\n\n inline unsigned parse_eight_digits(const char* source) {\n#if defined(__BYTE_ORDER__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__\n std::uint64_t digits;\n std::memcpy(&digits, source, sizeof(digits));\n digits -= 0x3030303030303030ULL;\n digits = (digits * 10 + (digits >> 8)) & 0x00ff00ff00ff00ffULL;\n digits = (digits * 100 + (digits >> 16)) & 0x0000ffff0000ffffULL;\n return static_cast(\n (digits * 10000 + (digits >> 32)) & 0xffffffffULL);\n#else\n unsigned result = 0;\n for (int i = 0; i < 8; ++i) {\n result = result * 10U + static_cast(source[i] - '0');\n }\n return result;\n#endif\n }\n\n inline long long read_int() {\n initialize();\n\n if (mapped != nullptr) {\n while (cursor < length && mapped[cursor] <= ' ') ++cursor;\n if (cursor == length) return 0;\n\n const bool negative = mapped[cursor] == '-';\n if (negative) {\n ++cursor;\n if (cursor == length) return 0;\n }\n\n if (!negative && length - cursor >= 9 &&\n has_eight_digits(mapped + cursor)) {\n const unsigned value = parse_eight_digits(mapped + cursor);\n const unsigned ninth = static_cast(mapped[cursor + 8] - '0');\n if (ninth >= 10U) {\n cursor += 8;\n return static_cast(value);\n }\n if (length - cursor >= 10 &&\n static_cast(mapped[cursor + 9] - '0') >= 10U) {\n cursor += 9;\n return static_cast(value * 10U + ninth);\n }\n }\n\n long long value = 0;\n if (negative) {\n while (length - cursor >= 2) {\n const unsigned first = static_cast(mapped[cursor] - '0');\n const unsigned second = static_cast(mapped[cursor + 1] - '0');\n if (first >= 10U || second >= 10U) break;\n value = value * 100 - static_cast(first * 10U + second);\n cursor += 2;\n }\n if (cursor < length) {\n const unsigned digit = static_cast(mapped[cursor] - '0');\n if (digit < 10U) {\n value = value * 10 - static_cast(digit);\n ++cursor;\n }\n }\n } else {\n while (length - cursor >= 2) {\n const unsigned first = static_cast(mapped[cursor] - '0');\n const unsigned second = static_cast(mapped[cursor + 1] - '0');\n if (first >= 10U || second >= 10U) break;\n value = value * 100 + static_cast(first * 10U + second);\n cursor += 2;\n }\n if (cursor < length) {\n const unsigned digit = static_cast(mapped[cursor] - '0');\n if (digit < 10U) {\n value = value * 10 + static_cast(digit);\n ++cursor;\n }\n }\n }\n return value;\n }\n\n for (;;) {\n if (cursor == length && !refill()) return 0;\n while (cursor < length && buffer[cursor] <= ' ') ++cursor;\n if (cursor < length) break;\n }\n\n const bool negative = buffer[cursor] == '-';\n if (negative) ++cursor;\n long long value = 0;\n\n for (;;) {\n if (!negative && length - cursor >= 9 &&\n has_eight_digits(buffer + cursor)) {\n const unsigned first_eight = parse_eight_digits(buffer + cursor);\n const unsigned ninth = static_cast(buffer[cursor + 8] - '0');\n if (ninth >= 10U) {\n cursor += 8;\n return static_cast(first_eight);\n }\n if (length - cursor >= 10 &&\n static_cast(buffer[cursor + 9] - '0') >= 10U) {\n cursor += 9;\n return static_cast(first_eight * 10U + ninth);\n }\n }\n\n while (length - cursor >= 2) {\n const unsigned first = static_cast(buffer[cursor] - '0');\n const unsigned second = static_cast(buffer[cursor + 1] - '0');\n if (first >= 10U) return value;\n if (second >= 10U) {\n value = negative\n ? value * 10 - static_cast(first)\n : value * 10 + static_cast(first);\n ++cursor;\n return value;\n }\n value = negative\n ? value * 100 - static_cast(first * 10U + second)\n : value * 100 + static_cast(first * 10U + second);\n cursor += 2;\n }\n\n if (cursor < length) {\n const unsigned digit = static_cast(buffer[cursor] - '0');\n if (digit >= 10U) return value;\n value = negative\n ? value * 10 - static_cast(digit)\n : value * 10 + static_cast(digit);\n ++cursor;\n }\n if (!refill()) return value;\n }\n }\n\n template \n inline void read_int_array(T* output, std::size_t count) {\n for (std::size_t i = 0; i < count; ++i) {\n output[i] = static_cast(read_int());\n }\n }\n } // namespace cplib_sheep_input\n \"\"\".}\n\n proc sheepGetChar(): cint {.importcpp: \"cplib_sheep_input::get_char()\", nodecl, inline.}\n proc sheepReadInt(): clonglong {.importcpp: \"cplib_sheep_input::read_int()\", nodecl, inline.}\n proc sheepReadIntArray(values: ptr int, count: csize_t) {.importcpp: \"cplib_sheep_input::read_int_array(@)\", nodecl, inline.}\n\n proc ii(): int {.inline.} = sheepReadInt().int\n proc lii(N: int): seq[int] {.inline.} =\n result = newSeq[int](N)\n if N > 0:\n sheepReadIntArray(addr result[0], N.csize_t)\n\n proc si(): string {.inline.} =\n var c = sheepGetChar()\n while c >= 0 and c <= ord(' '):\n c = sheepGetChar()\n while c > ord(' '):\n result.add(char(c))\n c = sheepGetChar()\n \n # 出力系\n # 1. 実際の処理を行う proc (openArray を受け取る)\n proc print_internal(prop: tuple[f: File, sepc: string, endc: string, flush: bool], args: openArray[string]) =\n for i in 0 ..< args.len:\n prop.f.write(args[i])\n if i != args.len - 1:\n prop.f.write(prop.sepc)\n else:\n prop.f.write(prop.endc)\n if prop.flush:\n prop.f.flushFile()\n\n # 2. ユーザーが呼び出すためのインターフェース (varargs を受け取る)\n proc print*(prop: tuple[f: File, sepc: string, endc: string, flush: bool], args: varargs[string, `$`]) =\n # varargs は内部では openArray として扱えるので、そのまま渡せる\n print_internal(prop, args)\n\n proc print*(args: varargs[string, `$`]) =\n # こちらも内部用の proc を呼ぶ\n print_internal((f: stdout, sepc: \" \", endc: \"\\n\", flush: false), args)\n macro getSymbolName(x: typed): string = x.toStrLit\n macro debug*(args: varargs[untyped]): untyped =\n when defined(debug):\n result = newNimNode(nnkStmtList, args)\n template prop(e: string = \"\"): untyped = (f: stderr, sepc: \"\", endc: e, flush: true)\n for i, arg in args:\n if arg.kind == nnkStrLit:\n result.add(quote do: print(prop(), \"\\\"\", `arg`, \"\\\"\"))\n else:\n result.add(quote do: print(prop(\": \"), getSymbolName(`arg`)))\n result.add(quote do: print(prop(), `arg`))\n if i != args.len - 1: result.add(quote do: print(prop(), \", \"))\n else: result.add(quote do: print(prop(), \"\\n\"))\n else:\n return (quote do: discard)\n #chmin,chmax\n template `max=`(x, y) =\n let yVal = y # yが計算式の場合に評価を1回にするため\n if x < yVal:\n x = yVal\n\n template `min=`(x, y) =\n let yVal = y\n if x > yVal:\n x = yVal\n proc chmin[T](x: var T, y: T):bool=\n if x > y:\n x = y\n return true\n return false\n proc chmax[T](x: var T, y: T):bool=\n if x < y:\n x = y\n return true\n return false\n #bit演算\n proc `%`*(x: int, y: int): int =\n result = x mod y\n if y > 0 and result < 0: result += y\n if y < 0 and result > 0: result += y\n proc `//`*(x: int, y: int): int{.inline.} =\n result = x div y\n if y > 0 and result * y > x: result -= 1\n if y < 0 and result * y < x: result -= 1\n proc `%=`(x: var int, y: int): void = x = x%y\n proc `//=`(x: var int, y: int): void = x = x//y\n proc `**`(x: int, y: int): int = x^y\n proc `**=`(x: var int, y: int): void = x = x^y\n proc `^`(x: int, y: int): int = x xor y\n proc `|`(x: int, y: int): int = x or y\n proc `&`(x: int, y: int): int = x and y\n proc `>>`(x: int, y: int): int = x shr y\n proc `<<`(x: int, y: int): int = x shl y\n proc `~`(x: int): int = not x\n proc `^=`(x: var int, y: int): void = x = x ^ y\n proc `&=`(x: var int, y: int): void = x = x & y\n proc `|=`(x: var int, y: int): void = x = x | y\n proc `>>=`(x: var int, y: int): void = x = x >> y\n proc `<<=`(x: var int, y: int): void = x = x << y\n proc `[]`(x: int, n: int): bool = (x and (1 shl n)) != 0\n #便利な変換\n proc `!`(x: char, a = '0'): int = int(x)-int(a)\n #定数\n when not declared CPLIB_UTILS_CONSTANTS:\n const CPLIB_UTILS_CONSTANTS* = 1\n const INF32*: int32 = 1001000027.int32\n const INF64*: int = int(3300300300300300491)\n \n const INF = INF64\n #converter\n\n #range\n iterator range(start: int, ends: int, step: int): int =\n var i = start\n if step < 0:\n while i > ends:\n yield i\n i += step\n elif step > 0:\n while i < ends:\n yield i\n i += step\n iterator range(ends: int): int = (for i in 0.. r[i]:\n return false\n elif l[i] < r[i]:\n return true\n return len(l) < len(r)\n \n # Yes/No\n proc yes*(b: bool = true): void = print(if b: \"Yes\" else: \"No\")\n proc no*(b: bool = true): void = yes(not b)\n\n proc takahashi(b:bool = true) : void = print(if b: \"Takahashi\" else: \"Aoki\")\n proc aoki(b:bool = true) : void = takahashi(not b)\n\n template dblock(body: untyped) =\n when defined(debug):\n block:\n body\n" # source: src/cplib/graph/graph.nim ImportExpand "cplib/graph/graph" <=== "when not declared CPLIB_GRAPH_GRAPH:\n const CPLIB_GRAPH_GRAPH* = 1\n\n import sequtils\n import math\n type DynamicGraph*[T] = ref object of RootObj\n edges*: seq[seq[(int32, T)]]\n len*: int\n type StaticGraph*[T] = ref object of RootObj\n src*, dst*: seq[int32]\n cost*: seq[T]\n elist*: seq[(int32, T)]\n start*: seq[int32]\n len*: int\n\n type WeightedDirectedGraph*[T] = ref object of DynamicGraph[T]\n type WeightedUnDirectedGraph*[T] = ref object of DynamicGraph[T]\n type UnWeightedDirectedGraph* = ref object of DynamicGraph[int]\n type UnWeightedUnDirectedGraph* = ref object of DynamicGraph[int]\n type WeightedDirectedStaticGraph*[T] = ref object of StaticGraph[T]\n type WeightedUnDirectedStaticGraph*[T] = ref object of StaticGraph[T]\n type UnWeightedDirectedStaticGraph* = ref object of StaticGraph[int]\n type UnWeightedUnDirectedStaticGraph* = ref object of StaticGraph[int]\n\n type GraphTypes*[T] = DynamicGraph[T] or StaticGraph[T]\n type DirectedGraph* = WeightedDirectedGraph or UnWeightedDirectedGraph or WeightedDirectedStaticGraph or UnWeightedDirectedStaticGraph\n type UnDirectedGraph* = WeightedUnDirectedGraph or UnWeightedUnDirectedGraph or WeightedUnDirectedStaticGraph or UnWeightedUnDirectedStaticGraph\n type WeightedGraph*[T] = WeightedDirectedGraph[T] or WeightedUnDirectedGraph[T] or WeightedDirectedStaticGraph[T] or WeightedUnDirectedStaticGraph[T]\n type UnWeightedGraph* = UnWeightedDirectedGraph or UnWeightedUnDirectedGraph or UnWeightedDirectedStaticGraph or UnWeightedUnDirectedStaticGraph\n type DynamicGraphTypes* = WeightedDirectedGraph or UnWeightedDirectedGraph or WeightedUnDirectedGraph or UnWeightedUnDirectedGraph\n type StaticGraphTypes* = WeightedDirectedStaticGraph or UnWeightedDirectedStaticGraph or WeightedUnDirectedStaticGraph or UnWeightedUnDirectedStaticGraph\n\n proc add_edge_dynamic_impl*[T](g: DynamicGraph[T], u, v: int, cost: T, directed: bool) =\n g.edges[u].add((v.int32, cost))\n if not directed: g.edges[v].add((u.int32, cost))\n\n proc initWeightedDirectedGraph*(N: int, edgetype: typedesc = int): WeightedDirectedGraph[edgetype] =\n result = WeightedDirectedGraph[edgetype](edges: newSeq[seq[(int32, edgetype)]](N), len: N)\n proc add_edge*[T](g: var WeightedDirectedGraph[T], u, v: int, cost: T) =\n g.add_edge_dynamic_impl(u, v, cost, true)\n\n proc initWeightedUnDirectedGraph*(N: int, edgetype: typedesc = int): WeightedUnDirectedGraph[edgetype] =\n result = WeightedUnDirectedGraph[edgetype](edges: newSeq[seq[(int32, edgetype)]](N), len: N)\n proc add_edge*[T](g: var WeightedUnDirectedGraph[T], u, v: int, cost: T) =\n g.add_edge_dynamic_impl(u, v, cost, false)\n\n proc initUnWeightedDirectedGraph*(N: int): UnWeightedDirectedGraph =\n result = UnWeightedDirectedGraph(edges: newSeq[seq[(int32, int)]](N), len: N)\n proc add_edge*(g: var UnWeightedDirectedGraph, u, v: int) =\n g.add_edge_dynamic_impl(u, v, 1, true)\n\n proc initUnWeightedUnDirectedGraph*(N: int): UnWeightedUnDirectedGraph =\n result = UnWeightedUnDirectedGraph(edges: newSeq[seq[(int32, int)]](N), len: N)\n proc add_edge*(g: var UnWeightedUnDirectedGraph, u, v: int) =\n g.add_edge_dynamic_impl(u, v, 1, false)\n\n proc len*[T](G: WeightedGraph[T]): int = G.len\n proc len*(G: UnWeightedGraph): int = G.len\n\n iterator `[]`*[T](g: WeightedDirectedGraph[T] or WeightedUnDirectedGraph[T], x: int): (int, T) =\n for e in g.edges[x]: yield (e[0].int, e[1])\n iterator `[]`*(g: UnWeightedDirectedGraph or UnWeightedUnDirectedGraph, x: int): int =\n for e in g.edges[x]: yield e[0].int\n\n proc add_edge_static_impl*[T](g: StaticGraph[T], u, v: int, cost: T, directed: bool) =\n g.src.add(u.int32)\n g.dst.add(v.int32)\n g.cost.add(cost)\n if not directed:\n g.src.add(v.int32)\n g.dst.add(u.int32)\n g.cost.add(cost)\n\n proc build_impl*[T](g: StaticGraph[T]) =\n g.start = newSeqWith(g.len + 1, 0.int32)\n for i in 0.. 0, \"Static Graph must be initialized before use.\"\n\n iterator `[]`*[T](g: WeightedDirectedStaticGraph[T] or WeightedUnDirectedStaticGraph[T], x: int): (int, T) =\n g.static_graph_initialized_check()\n for i in g.start[x]..\n\n #include \n #include \n #include \n\n #ifndef CPLIB_WARSHALL_FLOYD_BLOCK_SIZE\n #define CPLIB_WARSHALL_FLOYD_BLOCK_SIZE 216\n #endif\n\n #ifndef CPLIB_WARSHALL_FLOYD_DENSE_BLOCK_SIZE\n #define CPLIB_WARSHALL_FLOYD_DENSE_BLOCK_SIZE 256\n #endif\n\n #ifndef CPLIB_WARSHALL_FLOYD_INT32_BLOCK_SIZE\n #define CPLIB_WARSHALL_FLOYD_INT32_BLOCK_SIZE 256\n #endif\n\n #pragma GCC push_options\n #pragma GCC target(\"avx2\")\n #pragma GCC optimize(\"O3\")\n\n static inline bool cplib_warshall_floyd_all_reachable_avx2(\n const std::int64_t* row,\n std::size_t begin,\n std::size_t end,\n __m256i inf4,\n std::int64_t inf) {\n std::size_t j = begin;\n for (; j + 4 <= end; j += 4) {\n const __m256i values = _mm256_loadu_si256(\n reinterpret_cast(row + j));\n const __m256i unreachable = _mm256_cmpeq_epi64(values, inf4);\n if (!_mm256_testz_si256(unreachable, unreachable)) return false;\n }\n for (; j < end; ++j) {\n if (row[j] == inf) return false;\n }\n return true;\n }\n\n static inline void cplib_warshall_floyd_relax_avx2(\n std::int64_t* row_i,\n const std::int64_t* row_k,\n std::int64_t dik,\n std::size_t begin,\n std::size_t end,\n __m256i inf4,\n std::int64_t inf,\n bool all_reachable) {\n const __m256i dik4 = _mm256_set1_epi64x(dik);\n std::size_t j = begin;\n if (all_reachable) {\n #pragma GCC unroll 8\n for (; j + 4 <= end; j += 4) {\n const __m256i dkj = _mm256_loadu_si256(\n reinterpret_cast(row_k + j));\n const __m256i dij = _mm256_loadu_si256(\n reinterpret_cast(row_i + j));\n const __m256i candidate = _mm256_add_epi64(dik4, dkj);\n const __m256i take = _mm256_cmpgt_epi64(dij, candidate);\n _mm256_maskstore_epi64(\n reinterpret_cast(row_i + j), take, candidate);\n }\n for (; j < end; ++j) {\n const std::int64_t candidate = dik + row_k[j];\n if (candidate < row_i[j]) row_i[j] = candidate;\n }\n } else {\n #pragma GCC unroll 4\n for (; j + 4 <= end; j += 4) {\n const __m256i dkj = _mm256_loadu_si256(\n reinterpret_cast(row_k + j));\n const __m256i dij = _mm256_loadu_si256(\n reinterpret_cast(row_i + j));\n const __m256i candidate = _mm256_add_epi64(dik4, dkj);\n const __m256i unreachable = _mm256_cmpeq_epi64(dkj, inf4);\n const __m256i improves = _mm256_cmpgt_epi64(dij, candidate);\n const __m256i take = _mm256_andnot_si256(\n unreachable, improves);\n _mm256_maskstore_epi64(\n reinterpret_cast(row_i + j), take, candidate);\n }\n for (; j < end; ++j) {\n if (row_k[j] != inf) {\n const std::int64_t candidate = dik + row_k[j];\n if (candidate < row_i[j]) row_i[j] = candidate;\n }\n }\n }\n }\n\n static inline void cplib_warshall_floyd_relax4_dense_avx2(\n std::int64_t* __restrict__ row0,\n std::int64_t* __restrict__ row1,\n std::int64_t* __restrict__ row2,\n std::int64_t* __restrict__ row3,\n const std::int64_t* __restrict__ row_k,\n std::int64_t dik0,\n std::int64_t dik1,\n std::int64_t dik2,\n std::int64_t dik3,\n std::size_t end) {\n const __m256i dik4_0 = _mm256_set1_epi64x(dik0);\n const __m256i dik4_1 = _mm256_set1_epi64x(dik1);\n const __m256i dik4_2 = _mm256_set1_epi64x(dik2);\n const __m256i dik4_3 = _mm256_set1_epi64x(dik3);\n std::size_t j = 0;\n for (; j + 4 <= end; j += 4) {\n const __m256i dkj = _mm256_load_si256(\n reinterpret_cast(row_k + j));\n const __m256i candidate0 = _mm256_add_epi64(dik4_0, dkj);\n const __m256i take0 = _mm256_cmpgt_epi64(\n _mm256_load_si256(\n reinterpret_cast(row0 + j)),\n candidate0);\n _mm256_maskstore_epi64(\n reinterpret_cast(row0 + j), take0, candidate0);\n\n const __m256i candidate1 = _mm256_add_epi64(dik4_1, dkj);\n const __m256i take1 = _mm256_cmpgt_epi64(\n _mm256_load_si256(\n reinterpret_cast(row1 + j)),\n candidate1);\n _mm256_maskstore_epi64(\n reinterpret_cast(row1 + j), take1, candidate1);\n\n const __m256i candidate2 = _mm256_add_epi64(dik4_2, dkj);\n const __m256i take2 = _mm256_cmpgt_epi64(\n _mm256_load_si256(\n reinterpret_cast(row2 + j)),\n candidate2);\n _mm256_maskstore_epi64(\n reinterpret_cast(row2 + j), take2, candidate2);\n\n const __m256i candidate3 = _mm256_add_epi64(dik4_3, dkj);\n const __m256i take3 = _mm256_cmpgt_epi64(\n _mm256_load_si256(\n reinterpret_cast(row3 + j)),\n candidate3);\n _mm256_maskstore_epi64(\n reinterpret_cast(row3 + j), take3, candidate3);\n }\n for (; j < end; ++j) {\n const std::int64_t dkj = row_k[j];\n const std::int64_t candidate0 = dik0 + dkj;\n const std::int64_t candidate1 = dik1 + dkj;\n const std::int64_t candidate2 = dik2 + dkj;\n const std::int64_t candidate3 = dik3 + dkj;\n if (candidate0 < row0[j]) row0[j] = candidate0;\n if (candidate1 < row1[j]) row1[j] = candidate1;\n if (candidate2 < row2[j]) row2[j] = candidate2;\n if (candidate3 < row3[j]) row3[j] = candidate3;\n }\n }\n\n static bool cplib_warshall_floyd_int64_sparse_avx2(\n void* raw_rows,\n std::size_t n,\n std::int64_t zero,\n std::int64_t inf) {\n std::int64_t** d = static_cast(raw_rows);\n const __m256i inf4 = _mm256_set1_epi64x(inf);\n constexpr std::size_t block_size =\n CPLIB_WARSHALL_FLOYD_BLOCK_SIZE;\n\n for (std::size_t i = 0; i < n; ++i) {\n if (d[i][i] < zero) return true;\n }\n\n for (std::size_t kk = 0; kk < n; kk += block_size) {\n const std::size_t kend =\n kk + block_size < n ? kk + block_size : n;\n\n // Phase 1: close the diagonal block.\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int64_t* const row_k = d[k];\n const bool all_reachable =\n cplib_warshall_floyd_all_reachable_avx2(\n row_k, kk, kend, inf4, inf);\n for (std::size_t i = kk; i < kend; ++i) {\n const std::int64_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_avx2(\n d[i], row_k, dik, kk, kend, inf4, inf,\n all_reachable);\n }\n }\n\n for (std::size_t i = 0; i < n; ++i) {\n if (d[i][i] < zero) return true;\n }\n\n // Phase 2a: update the blocks in the diagonal block row.\n for (std::size_t jj = 0; jj < n; jj += block_size) {\n if (jj == kk) continue;\n const std::size_t jend =\n jj + block_size < n ? jj + block_size : n;\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int64_t* const row_k = d[k];\n const bool all_reachable =\n cplib_warshall_floyd_all_reachable_avx2(\n row_k, jj, jend, inf4, inf);\n for (std::size_t i = kk; i < kend; ++i) {\n const std::int64_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_avx2(\n d[i], row_k, dik, jj, jend, inf4, inf,\n all_reachable);\n }\n }\n }\n\n // Phase 2b: update the blocks in the diagonal block column.\n for (std::size_t ii = 0; ii < n; ii += block_size) {\n if (ii == kk) continue;\n const std::size_t iend =\n ii + block_size < n ? ii + block_size : n;\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int64_t* const row_k = d[k];\n const bool all_reachable =\n cplib_warshall_floyd_all_reachable_avx2(\n row_k, kk, kend, inf4, inf);\n for (std::size_t i = ii; i < iend; ++i) {\n const std::int64_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_avx2(\n d[i], row_k, dik, kk, kend, inf4, inf,\n all_reachable);\n }\n }\n }\n\n // Phase 3: update all remaining blocks while the three tiles\n // stay cache-resident, reusing them across the inner loops.\n for (std::size_t ii = 0; ii < n; ii += block_size) {\n if (ii == kk) continue;\n const std::size_t iend =\n ii + block_size < n ? ii + block_size : n;\n for (std::size_t jj = 0; jj < n; jj += block_size) {\n if (jj == kk) continue;\n const std::size_t jend =\n jj + block_size < n ? jj + block_size : n;\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int64_t* const row_k = d[k];\n const bool all_reachable =\n cplib_warshall_floyd_all_reachable_avx2(\n row_k, jj, jend, inf4, inf);\n for (std::size_t i = ii; i < iend; ++i) {\n const std::int64_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_avx2(\n d[i], row_k, dik, jj, jend, inf4, inf,\n all_reachable);\n }\n }\n }\n }\n\n for (std::size_t i = 0; i < n; ++i) {\n if (d[i][i] < zero) return true;\n }\n }\n return false;\n }\n\n static bool cplib_warshall_floyd_int64_dense_tiled_avx2(\n std::int64_t** d,\n std::size_t n,\n std::int64_t zero,\n std::int64_t inf) {\n constexpr std::size_t block_size =\n CPLIB_WARSHALL_FLOYD_DENSE_BLOCK_SIZE;\n const std::size_t block_count =\n (n + block_size - 1) / block_size;\n const std::size_t padded_size = block_count * block_size;\n std::int64_t* const matrix = static_cast(\n _mm_malloc(padded_size * padded_size * sizeof(std::int64_t), 32));\n if (matrix == nullptr) {\n return cplib_warshall_floyd_int64_sparse_avx2(\n static_cast(d), n, zero, inf);\n }\n\n const auto tile = [&](std::size_t bi, std::size_t bj) {\n return matrix + (bi * block_count + bj) *\n block_size * block_size;\n };\n\n for (std::size_t i = 0; i < n; ++i) {\n const std::size_t bi = i / block_size;\n const std::size_t local_i = i % block_size;\n for (std::size_t bj = 0; bj < block_count; ++bj) {\n const std::size_t j_begin = bj * block_size;\n const std::size_t j_size =\n j_begin + block_size < n ? block_size : n - j_begin;\n std::memcpy(\n tile(bi, bj) + local_i * block_size,\n d[i] + j_begin,\n j_size * sizeof(std::int64_t));\n }\n }\n\n const __m256i inf4 = _mm256_set1_epi64x(inf);\n bool negative_cycle = false;\n for (std::size_t kb = 0; kb < block_count; ++kb) {\n const std::size_t k_begin = kb * block_size;\n const std::size_t k_size =\n k_begin + block_size < n ? block_size : n - k_begin;\n std::int64_t* const diagonal = tile(kb, kb);\n\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k =\n diagonal + k * block_size;\n for (std::size_t i = 0; i < k_size; ++i) {\n cplib_warshall_floyd_relax_avx2(\n diagonal + i * block_size, row_k,\n diagonal[i * block_size + k], 0, k_size,\n inf4, inf, true);\n }\n }\n\n for (std::size_t i = 0; i < k_size; ++i) {\n if (diagonal[i * block_size + i] < zero) {\n negative_cycle = true;\n break;\n }\n }\n if (negative_cycle) break;\n\n for (std::size_t jb = 0; jb < block_count; ++jb) {\n if (jb == kb) continue;\n const std::size_t j_begin = jb * block_size;\n const std::size_t j_size =\n j_begin + block_size < n ? block_size : n - j_begin;\n std::int64_t* const top = tile(kb, jb);\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k = top + k * block_size;\n for (std::size_t i = 0; i < k_size; ++i) {\n cplib_warshall_floyd_relax_avx2(\n top + i * block_size, row_k,\n diagonal[i * block_size + k], 0, j_size,\n inf4, inf, true);\n }\n }\n }\n\n for (std::size_t ib = 0; ib < block_count; ++ib) {\n if (ib == kb) continue;\n const std::size_t i_begin = ib * block_size;\n const std::size_t i_size =\n i_begin + block_size < n ? block_size : n - i_begin;\n std::int64_t* const left = tile(ib, kb);\n std::size_t i = 0;\n for (; i + 16 <= i_size; i += 16) {\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k =\n diagonal + k * block_size;\n cplib_warshall_floyd_relax4_dense_avx2(\n left + i * block_size,\n left + (i + 1) * block_size,\n left + (i + 2) * block_size,\n left + (i + 3) * block_size,\n row_k,\n left[i * block_size + k],\n left[(i + 1) * block_size + k],\n left[(i + 2) * block_size + k],\n left[(i + 3) * block_size + k],\n k_size);\n cplib_warshall_floyd_relax4_dense_avx2(\n left + (i + 4) * block_size,\n left + (i + 5) * block_size,\n left + (i + 6) * block_size,\n left + (i + 7) * block_size,\n row_k,\n left[(i + 4) * block_size + k],\n left[(i + 5) * block_size + k],\n left[(i + 6) * block_size + k],\n left[(i + 7) * block_size + k],\n k_size);\n cplib_warshall_floyd_relax4_dense_avx2(\n left + (i + 8) * block_size,\n left + (i + 9) * block_size,\n left + (i + 10) * block_size,\n left + (i + 11) * block_size,\n row_k,\n left[(i + 8) * block_size + k],\n left[(i + 9) * block_size + k],\n left[(i + 10) * block_size + k],\n left[(i + 11) * block_size + k],\n k_size);\n cplib_warshall_floyd_relax4_dense_avx2(\n left + (i + 12) * block_size,\n left + (i + 13) * block_size,\n left + (i + 14) * block_size,\n left + (i + 15) * block_size,\n row_k,\n left[(i + 12) * block_size + k],\n left[(i + 13) * block_size + k],\n left[(i + 14) * block_size + k],\n left[(i + 15) * block_size + k],\n k_size);\n }\n }\n for (; i + 4 <= i_size; i += 4) {\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k =\n diagonal + k * block_size;\n cplib_warshall_floyd_relax4_dense_avx2(\n left + i * block_size,\n left + (i + 1) * block_size,\n left + (i + 2) * block_size,\n left + (i + 3) * block_size,\n row_k,\n left[i * block_size + k],\n left[(i + 1) * block_size + k],\n left[(i + 2) * block_size + k],\n left[(i + 3) * block_size + k],\n k_size);\n }\n }\n for (; i < i_size; ++i) {\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k =\n diagonal + k * block_size;\n cplib_warshall_floyd_relax_avx2(\n left + i * block_size, row_k,\n left[i * block_size + k], 0, k_size,\n inf4, inf, true);\n }\n }\n }\n\n for (std::size_t ib = 0; ib < block_count; ++ib) {\n if (ib == kb) continue;\n const std::size_t i_begin = ib * block_size;\n const std::size_t i_size =\n i_begin + block_size < n ? block_size : n - i_begin;\n const std::int64_t* const left = tile(ib, kb);\n for (std::size_t jb = 0; jb < block_count; ++jb) {\n if (jb == kb) continue;\n const std::size_t j_begin = jb * block_size;\n const std::size_t j_size =\n j_begin + block_size < n ? block_size : n - j_begin;\n const std::int64_t* const top = tile(kb, jb);\n std::int64_t* const output = tile(ib, jb);\n std::size_t i = 0;\n for (; i + 16 <= i_size; i += 16) {\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k =\n top + k * block_size;\n cplib_warshall_floyd_relax4_dense_avx2(\n output + i * block_size,\n output + (i + 1) * block_size,\n output + (i + 2) * block_size,\n output + (i + 3) * block_size,\n row_k,\n left[i * block_size + k],\n left[(i + 1) * block_size + k],\n left[(i + 2) * block_size + k],\n left[(i + 3) * block_size + k],\n j_size);\n cplib_warshall_floyd_relax4_dense_avx2(\n output + (i + 4) * block_size,\n output + (i + 5) * block_size,\n output + (i + 6) * block_size,\n output + (i + 7) * block_size,\n row_k,\n left[(i + 4) * block_size + k],\n left[(i + 5) * block_size + k],\n left[(i + 6) * block_size + k],\n left[(i + 7) * block_size + k],\n j_size);\n cplib_warshall_floyd_relax4_dense_avx2(\n output + (i + 8) * block_size,\n output + (i + 9) * block_size,\n output + (i + 10) * block_size,\n output + (i + 11) * block_size,\n row_k,\n left[(i + 8) * block_size + k],\n left[(i + 9) * block_size + k],\n left[(i + 10) * block_size + k],\n left[(i + 11) * block_size + k],\n j_size);\n cplib_warshall_floyd_relax4_dense_avx2(\n output + (i + 12) * block_size,\n output + (i + 13) * block_size,\n output + (i + 14) * block_size,\n output + (i + 15) * block_size,\n row_k,\n left[(i + 12) * block_size + k],\n left[(i + 13) * block_size + k],\n left[(i + 14) * block_size + k],\n left[(i + 15) * block_size + k],\n j_size);\n }\n }\n for (; i + 4 <= i_size; i += 4) {\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k =\n top + k * block_size;\n cplib_warshall_floyd_relax4_dense_avx2(\n output + i * block_size,\n output + (i + 1) * block_size,\n output + (i + 2) * block_size,\n output + (i + 3) * block_size,\n row_k,\n left[i * block_size + k],\n left[(i + 1) * block_size + k],\n left[(i + 2) * block_size + k],\n left[(i + 3) * block_size + k],\n j_size);\n }\n }\n for (; i < i_size; ++i) {\n for (std::size_t k = 0; k < k_size; ++k) {\n const std::int64_t* const row_k =\n top + k * block_size;\n cplib_warshall_floyd_relax_avx2(\n output + i * block_size, row_k,\n left[i * block_size + k], 0, j_size,\n inf4, inf, true);\n }\n }\n }\n }\n }\n\n for (std::size_t i = 0; i < n; ++i) {\n const std::size_t bi = i / block_size;\n const std::size_t local_i = i % block_size;\n for (std::size_t bj = 0; bj < block_count; ++bj) {\n const std::size_t j_begin = bj * block_size;\n const std::size_t j_size =\n j_begin + block_size < n ? block_size : n - j_begin;\n std::memcpy(\n d[i] + j_begin,\n tile(bi, bj) + local_i * block_size,\n j_size * sizeof(std::int64_t));\n }\n }\n _mm_free(matrix);\n return negative_cycle;\n }\n\n extern \"C\" bool cplib_warshall_floyd_int64_avx2(\n void* raw_rows,\n std::size_t n,\n std::int64_t zero,\n std::int64_t inf) {\n std::int64_t** d = static_cast(raw_rows);\n const __m256i inf4 = _mm256_set1_epi64x(inf);\n bool dense = true;\n for (std::size_t i = 0; i < n && dense; ++i) {\n dense = cplib_warshall_floyd_all_reachable_avx2(\n d[i], 0, n, inf4, inf);\n }\n if (dense) {\n return cplib_warshall_floyd_int64_dense_tiled_avx2(\n d, n, zero, inf);\n }\n return cplib_warshall_floyd_int64_sparse_avx2(\n raw_rows, n, zero, inf);\n }\n\n static inline void cplib_warshall_floyd_relax_int32_avx2(\n std::int32_t* row_i,\n const std::int32_t* row_k,\n std::int32_t dik,\n std::size_t begin,\n std::size_t end,\n __m256i inf8,\n std::int32_t inf) {\n const __m256i dik8 = _mm256_set1_epi32(dik);\n std::size_t j = begin;\n for (; j + 8 <= end; j += 8) {\n const __m256i dkj = _mm256_loadu_si256(\n reinterpret_cast(row_k + j));\n const __m256i dij = _mm256_loadu_si256(\n reinterpret_cast(row_i + j));\n const __m256i candidate = _mm256_add_epi32(dik8, dkj);\n const __m256i unreachable = _mm256_cmpeq_epi32(dkj, inf8);\n const __m256i minimum = _mm256_min_epi32(dij, candidate);\n const __m256i updated = _mm256_blendv_epi8(\n minimum, dij, unreachable);\n _mm256_storeu_si256(\n reinterpret_cast<__m256i*>(row_i + j), updated);\n }\n for (; j < end; ++j) {\n if (row_k[j] != inf) {\n const std::int32_t candidate = dik + row_k[j];\n if (candidate < row_i[j]) row_i[j] = candidate;\n }\n }\n }\n\n extern \"C\" bool cplib_warshall_floyd_int32_avx2(\n void* raw_rows,\n std::size_t n,\n std::int32_t zero,\n std::int32_t inf) {\n std::int32_t** d = static_cast(raw_rows);\n const __m256i inf8 = _mm256_set1_epi32(inf);\n constexpr std::size_t block_size =\n CPLIB_WARSHALL_FLOYD_INT32_BLOCK_SIZE;\n\n for (std::size_t i = 0; i < n; ++i) {\n if (d[i][i] < zero) return true;\n }\n\n for (std::size_t kk = 0; kk < n; kk += block_size) {\n const std::size_t kend =\n kk + block_size < n ? kk + block_size : n;\n\n // Phase 1: close the diagonal block.\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int32_t* const row_k = d[k];\n for (std::size_t i = kk; i < kend; ++i) {\n const std::int32_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_int32_avx2(\n d[i], row_k, dik, kk, kend, inf8, inf);\n }\n }\n\n for (std::size_t i = 0; i < n; ++i) {\n if (d[i][i] < zero) return true;\n }\n\n // Phase 2a: update the blocks in the diagonal block row.\n for (std::size_t jj = 0; jj < n; jj += block_size) {\n if (jj == kk) continue;\n const std::size_t jend =\n jj + block_size < n ? jj + block_size : n;\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int32_t* const row_k = d[k];\n for (std::size_t i = kk; i < kend; ++i) {\n const std::int32_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_int32_avx2(\n d[i], row_k, dik, jj, jend, inf8, inf);\n }\n }\n }\n\n // Phase 2b: update the blocks in the diagonal block column.\n for (std::size_t ii = 0; ii < n; ii += block_size) {\n if (ii == kk) continue;\n const std::size_t iend =\n ii + block_size < n ? ii + block_size : n;\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int32_t* const row_k = d[k];\n for (std::size_t i = ii; i < iend; ++i) {\n const std::int32_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_int32_avx2(\n d[i], row_k, dik, kk, kend, inf8, inf);\n }\n }\n }\n\n // Phase 3: update all remaining cache-resident tiles.\n for (std::size_t ii = 0; ii < n; ii += block_size) {\n if (ii == kk) continue;\n const std::size_t iend =\n ii + block_size < n ? ii + block_size : n;\n for (std::size_t jj = 0; jj < n; jj += block_size) {\n if (jj == kk) continue;\n const std::size_t jend =\n jj + block_size < n ? jj + block_size : n;\n for (std::size_t k = kk; k < kend; ++k) {\n const std::int32_t* const row_k = d[k];\n for (std::size_t i = ii; i < iend; ++i) {\n const std::int32_t dik = d[i][k];\n if (dik != inf) cplib_warshall_floyd_relax_int32_avx2(\n d[i], row_k, dik, jj, jend, inf8, inf);\n }\n }\n }\n }\n\n for (std::size_t i = 0; i < n; ++i) {\n if (d[i][i] < zero) return true;\n }\n }\n return false;\n }\n\n #pragma GCC pop_options\n\n #endif\n \"\"\".}\n\n proc warshallFloydInt64Avx2(\n rows: pointer,\n n: csize_t,\n zero, inf: int\n ): bool {.importc: \"cplib_warshall_floyd_int64_avx2\".}\n\n proc warshallFloydInt32Avx2(\n rows: pointer,\n n: csize_t,\n zero, inf: int32\n ): bool {.importc: \"cplib_warshall_floyd_int32_avx2\".}\n\n proc warshall_floyd_impl[T](g: DynamicGraph[T] or StaticGraph[T], zero, inf: T): tuple[negative_cycle: bool, d: seq[seq[T]]] =\n var d = newSeqWith(g.len, newSeqWith(g.len, inf))\n for i in 0..