From dad6624e53476e105968c9ba2ff68a66e2a5c5a4 Mon Sep 17 00:00:00 2001 From: Thomas Lively Date: Thu, 1 Oct 2026 21:04:52 -0700 Subject: [PATCH 1/7] Add a dominator-tree WTO utility for reducible CFGs Add `src/cfg/wto.h` with `WeakTopologicalOrdering` (`WTO`) and `WTOWorklist` built on top of `DomTree`. In a reducible CFG ordered in reverse postorder, every cycle is a natural loop headed by a block that dominates all blocks in the cycle, allowing a Bourdoncle Weak Topological Ordering to be constructed directly from the dominator tree and natural loops of the CFG. Include unit tests in `test/gtest/wto.cpp` and `TODO` comments noting follow-on optimizations. --- src/cfg/wto.h | 292 +++++++++++++++++++++++ test/gtest/CMakeLists.txt | 11 +- test/gtest/wto.cpp | 476 ++++++++++++++++++++++++++++++++++++++ 3 files changed, 774 insertions(+), 5 deletions(-) create mode 100644 src/cfg/wto.h create mode 100644 test/gtest/wto.cpp diff --git a/src/cfg/wto.h b/src/cfg/wto.h new file mode 100644 index 00000000000..50936c60203 --- /dev/null +++ b/src/cfg/wto.h @@ -0,0 +1,292 @@ +/* + * Copyright 2026 WebAssembly Community Group participants + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// +// Weak Topological Ordering (WTO) and worklist runner for forward data flow +// analysis over reducible CFGs. +// +// A Weak Topological Ordering (Bourdoncle, "Efficient chaotic iteration +// strategies with widenings", 1993) is a hierarchical ordering of the reachable +// blocks of a directed graph in which strongly connected components (loops) are +// parenthesized into nested cycles. The first element of each cycle is its +// "head" (loop header), and the ordering satisfies two properties: +// +// 1. Every non-cycle edge u -> v goes forward in the flattened ordering +// (u appears before v). +// 2. Every backedge u -> v targets the head v of a cycle that encloses both +// u and v. +// +// Examples (writing `(h ...)` for a cycle with head `h`): +// +// - Diamond (0 -> 1, 0 -> 2, 1 -> 3, 2 -> 3): +// 0 1 2 3 +// +// - Simple loop (0 -> 1 -> 2 -> 1, 2 -> 3): +// 0 (1 2) 3 +// Here 1 is the cycle head, 1 and 2 form the cycle body, and the exit +// block 3 is outside the cycle. +// +// - Nested loops (0 -> 1 -> 2 -> 3 -> 2, 3 -> 4 -> 1, 4 -> 5): +// 0 (1 (2 3) 4) 5 +// Here outer cycle (1 (2 3) 4) with head 1 encloses inner cycle (2 3) with +// head 2. +// +// During forward dataflow analysis, elements of the WTO are evaluated +// left-to-right. When a cycle `(h ...)` is reached, its elements are evaluated +// repeatedly in order until the head `h` is no longer re-queued by a backedge. +// Inner cycles therefore stabilize completely on each iteration of an enclosing +// outer cycle before flow values propagate past the cycle. +// +// Algorithm sketch: +// +// In a reducible CFG whose blocks are ordered in reverse postorder (RPO, as +// produced by cfg-traversal.h), every cycle is a natural loop headed by a +// single entry block that dominates all blocks in the cycle, and every +// backedge `p -> h` satisfies `h` dominates `p` (with `h <= p` in RPO). We +// construct the WTO directly from the dominator tree in three steps: +// +// 1. Compute the dominator tree (`DomTree`) over the RPO-indexed blocks. +// 2. Discover natural loops from innermost to outermost by scanning candidate +// headers `h` in reverse RPO order (N - 1 down to 0). For each `h` that +// has at least one backedge `p -> h` (where `h` dominates `p`), run a +// backward DFS over predecessors starting from `p` and stopping at `h` to +// visit every block in `h`'s natural loop. Because inner loop headers have +// larger RPO indices than outer loop headers and are processed first, the +// first loop that visits a block `b != h` is its immediately enclosing +// loop (`loopParent[b] = h`). +// 3. Link each reachable block into the child list of its `loopParent` in +// increasing RPO order, then walk the resulting loop nesting forest to +// emit each loop header `h` and its children as a nested `Cycle`. +// + +#ifndef cfg_wto_h +#define cfg_wto_h + +#include +#include +#include +#include + +#include "cfg/domtree.h" +#include "wasm.h" + +namespace wasm { + +// The BasicBlock type is assumed to have an `in` vector of predecessor block +// pointers and a `contents.index` field of type `Index`. +template struct WeakTopologicalOrdering { + struct Cycle; + using Element = std::variant; + using List = std::vector; + + struct Cycle { + List elems; + + BasicBlock* head() const { return std::get(elems.front()); } + bool operator==(const Cycle&) const = default; + }; + + List elems; + + WeakTopologicalOrdering(std::vector>& blocks); +}; + +template +WeakTopologicalOrdering::WeakTopologicalOrdering( + std::vector>& blocks) { + Index numBlocks = blocks.size(); + if (numBlocks == 0) { + return; + } + + for (Index i = 0; i < numBlocks; ++i) { + blocks[i]->contents.index = i; + } + + // TODO: Avoid building an unordered_map of block indices in DomTree when + // BasicBlock already stores its RPO index on `contents`. + DomTree domTree(blocks); + + auto isReachable = [&](Index i) { + return i == 0 || domTree.iDoms[i] != domTree.nonsense; + }; + + auto dominates = [&](Index dom, Index node) { + assert(isReachable(dom)); + if (!isReachable(node)) { + return false; + } + Index curr = node; + while (curr > dom) { + curr = domTree.iDoms[curr]; + } + return curr == dom; + }; + + static constexpr Index NoIndex = Index(-1); + struct Node { + Index loopParent = NoIndex; + Index firstChild = NoIndex; + Index nextSibling = NoIndex; + Index lastVisitedBy = NoIndex; + bool isLoopHeader = false; + }; + std::vector nodes(numBlocks); + + // Discover natural loops from innermost to outermost (reverse RPO order). + // Because inner loops are processed before outer loops, the first loop whose + // natural loop body contains a block is its immediately enclosing loop. + // + // TODO: Collapse inner loops with union-find during natural loop discovery so + // outer loops do not re-traverse inner loop bodies. + std::vector worklist; + for (Index i = numBlocks; i > 0; --i) { + Index h = i - 1; + if (!isReachable(h)) { + continue; + } + nodes[h].lastVisitedBy = h; + for (auto* pred : blocks[h]->in) { + Index p = pred->contents.index; + if (dominates(h, p)) { + nodes[h].isLoopHeader = true; + if (nodes[p].lastVisitedBy != h) { + nodes[p].lastVisitedBy = h; + worklist.push_back(p); + } + } + } + while (!worklist.empty()) { + Index curr = worklist.back(); + worklist.pop_back(); + if (nodes[curr].loopParent == NoIndex) { + nodes[curr].loopParent = h; + } + for (auto* pred : blocks[curr]->in) { + Index p = pred->contents.index; + if (isReachable(p) && nodes[p].lastVisitedBy != h) { + assert(dominates(h, p) && "Expected reducible CFG"); + nodes[p].lastVisitedBy = h; + worklist.push_back(p); + } + } + } + } + + // Link each reachable block into its parent loop's intrusive child list. + // Prepending in reverse RPO order yields increasing RPO order. + Index topFirstChild = NoIndex; + for (Index i = numBlocks; i > 0; --i) { + Index idx = i - 1; + if (!isReachable(idx)) { + continue; + } + Index parent = nodes[idx].loopParent; + if (parent == NoIndex) { + nodes[idx].nextSibling = topFirstChild; + topFirstChild = idx; + } else { + nodes[idx].nextSibling = nodes[parent].firstChild; + nodes[parent].firstChild = idx; + } + } + + // TODO: Flatten the WTO into a single contiguous vector of entries with cycle + // jump targets to avoid per-cycle vector allocations and recursion. + auto buildList = [&](auto& self, Index firstChild, List& out) -> void { + for (Index curr = firstChild; curr != NoIndex; + curr = nodes[curr].nextSibling) { + auto* block = blocks[curr].get(); + if (nodes[curr].isLoopHeader) { + Cycle cycle; + cycle.elems.emplace_back(block); + self(self, nodes[curr].firstChild, cycle.elems); + out.emplace_back(std::move(cycle)); + } else { + out.emplace_back(block); + } + } + }; + + buildList(buildList, topFirstChild, elems); +} + +// Given a CFG in reverse postorder (e.g. from cfg-traversal), run a forward +// fixed-point analysis over its basic blocks using a Weak Topological Ordering. +// +// Usage: +// 1. Construct `WTOWorklist work(cfg);` (which initializes `inQueue` and +// `index` on each block's `contents`). +// 2. Seed the initial block(s) to evaluate via `work.push(cfg.entry);`. +// 3. Call `work.run([&](BasicBlock* block) { ... });`. Inside the visitor +// callback, evaluate the transfer function for `block` and call +// `work.push(next)` for any successor whose input state changed and needs +// to be (re-)evaluated. +// +// The BasicBlock `contents` of the CFG must contain two fields: +// +// bool inQueue; // whether scheduled for visitation +// Index index; // basic block index in RPO +// +template struct WTOWorklist { + using BasicBlock = typename CFG::BasicBlock; + + CFG& cfg; + + WTOWorklist(CFG& cfg) : cfg(cfg) { + auto& basicBlocks = cfg.basicBlocks; + for (Index i = 0; i < basicBlocks.size(); ++i) { + auto& contents = basicBlocks[i]->contents; + contents.inQueue = false; + contents.index = i; + } + } + + void push(BasicBlock* block) { block->contents.inQueue = true; } + + template void run(VisitFn&& visit) { + // TODO: Track the number of queued blocks to stop early once the worklist + // is empty. + // TODO: Fast-path initial entry singletons and CFGs without backedges + // without building DomTree or WTO, using CFGWalker::loopTops. + WeakTopologicalOrdering wto(cfg.basicBlocks); + auto evalList = + [&](auto& self, + const typename WeakTopologicalOrdering::List& list) + -> void { + for (const auto& elem : list) { + if (auto* block = std::get_if(&elem)) { + if ((*block)->contents.inQueue) { + (*block)->contents.inQueue = false; + visit(*block); + } + } else { + const auto& cycle = + std::get::Cycle>(elem); + BasicBlock* head = cycle.head(); + do { + self(self, cycle.elems); + } while (head->contents.inQueue); + } + } + }; + evalList(evalList, wto.elems); + } +}; + +} // namespace wasm + +#endif // cfg_wto_h diff --git a/test/gtest/CMakeLists.txt b/test/gtest/CMakeLists.txt index a87dfd965b2..19052a46147 100644 --- a/test/gtest/CMakeLists.txt +++ b/test/gtest/CMakeLists.txt @@ -17,17 +17,17 @@ set(unittest_SOURCES disjoint_sets.cpp dwarf-ranges.cpp effects.cpp - graph.cpp - int128.cpp - leaves.cpp glbs.cpp + graph.cpp inplace_vector.cpp + int128.cpp interpreter.cpp intervals.cpp istring.cpp js-embedded-module.cpp json.cpp lattices.cpp + leaves.cpp local-graph.cpp possible-contents.cpp principal-type.cpp @@ -35,6 +35,7 @@ set(unittest_SOURCES public-type-validator.cpp scc.cpp sizes.cpp + source-map.cpp span.cpp stringify.cpp subtype-exprs.cpp @@ -42,9 +43,9 @@ set(unittest_SOURCES topological-sort.cpp type-builder.cpp type-updating.cpp - wat-lexer.cpp validator.cpp - source-map.cpp + wat-lexer.cpp + wto.cpp ) if(BUILD_FUZZTEST) diff --git a/test/gtest/wto.cpp b/test/gtest/wto.cpp new file mode 100644 index 00000000000..026951248f0 --- /dev/null +++ b/test/gtest/wto.cpp @@ -0,0 +1,476 @@ +/* + * Copyright 2026 WebAssembly Community Group participants + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include +#include +#include +#include +#include +#include + +#include "cfg/wto.h" +#include "gtest/gtest.h" + +using namespace wasm; + +namespace { + +struct TestCFG { + struct Contents { + bool inQueue = false; + Index index = 0; + }; + + struct BasicBlock { + Contents contents; + std::vector out; + std::vector in; + }; + + std::vector> basicBlocks; + BasicBlock* entry = nullptr; + + explicit TestCFG(Index numBlocks) { + basicBlocks.reserve(numBlocks); + for (Index i = 0; i < numBlocks; ++i) { + auto block = std::make_unique(); + block->contents.index = i; + basicBlocks.push_back(std::move(block)); + } + if (numBlocks > 0) { + entry = basicBlocks[0].get(); + } + } + + void addEdge(Index u, Index v) { + assert(u < basicBlocks.size()); + assert(v < basicBlocks.size()); + basicBlocks[u]->out.push_back(basicBlocks[v].get()); + basicBlocks[v]->in.push_back(basicBlocks[u].get()); + } +}; + +// Index-based mirror of a Weak Topological Ordering used in tests so that: +// 1. Expected orderings can be written concisely with block indices (e.g. +// `WTOList{0, C({1, 2}), 3}`) and pretty-printed on failure. +// 2. Test assertions remain independent of the internal representation of +// `WeakTopologicalOrdering` (which will be flattened into a contiguous +// entry array in a follow-on commit). +struct WTOCycle; +struct WTOElem; +using WTOList = std::vector; + +struct WTOCycle { + WTOList elems; + + WTOCycle(std::initializer_list list); + explicit WTOCycle(WTOList elems); + + Index head() const; + bool operator==(const WTOCycle& other) const; +}; + +struct WTOElem : std::variant { + using Base = std::variant; + using Base::Base; + WTOElem(Index v) : Base(v) {} + WTOElem(int v) : Base(Index(v)) {} + WTOElem(WTOCycle c) : Base(std::move(c)) {} +}; + +WTOCycle::WTOCycle(std::initializer_list list) : elems(list) {} +WTOCycle::WTOCycle(WTOList elems) : elems(std::move(elems)) {} +Index WTOCycle::head() const { return std::get(elems.front()); } +bool WTOCycle::operator==(const WTOCycle& other) const = default; + +WTOCycle C(std::initializer_list list) { return WTOCycle(list); } + +std::ostream& operator<<(std::ostream& os, const WTOElem& elem); +std::ostream& operator<<(std::ostream& os, const WTOList& list); + +std::ostream& operator<<(std::ostream& os, const WTOCycle& cycle) { + return os << "(" << cycle.elems << ")"; +} + +std::ostream& operator<<(std::ostream& os, const WTOElem& elem) { + if (auto* v = std::get_if(&elem)) { + return os << *v; + } + return os << std::get(elem); +} + +std::ostream& operator<<(std::ostream& os, const WTOList& list) { + for (size_t i = 0; i < list.size(); ++i) { + if (i > 0) { + os << " "; + } + os << list[i]; + } + return os; +} + +using BasicBlock = TestCFG::BasicBlock; + +WTOList toIndexWTO(const WeakTopologicalOrdering::List& src) { + WTOList dst; + for (const auto& elem : src) { + if (auto* b = std::get_if(&elem)) { + dst.emplace_back((*b)->contents.index); + } else { + const auto& cycle = + std::get::Cycle>(elem); + EXPECT_EQ(cycle.head(), std::get(cycle.elems.front())); + dst.emplace_back(WTOCycle(toIndexWTO(cycle.elems))); + } + } + return dst; +} + +// Check the formal properties of a Weak Topological Ordering (Bourdoncle 1993, +// Definition 1) over the reachable subgraph of `cfg`: +// 1. Every vertex appears at most once in the flattened ordering. +// 2. Every cycle is non-empty and its head (first element) is a single vertex, +// not a nested cycle. +// 3. For every edge u -> v between reachable vertices: +// - Either u appears strictly before v in the flattened order, OR +// - v appears at or before u AND v is the head of a cycle containing both v +// and u. +void verifyWTOInvariants(const TestCFG& cfg, const WTOList& wto) { + std::vector flatOrder; + std::unordered_map pos; + std::unordered_map> cycleMembers; + + auto walk = [&](auto& self, + const WTOList& list, + std::vector& activeHeads) -> void { + for (const auto& elem : list) { + if (auto* v = std::get_if(&elem)) { + EXPECT_FALSE(pos.contains(*v)) << "Duplicate vertex " << *v; + pos[*v] = flatOrder.size(); + flatOrder.push_back(*v); + for (Index head : activeHeads) { + cycleMembers[head].insert(*v); + } + } else { + const auto& cycle = std::get(elem); + ASSERT_FALSE(cycle.elems.empty()) << "Empty cycle in WTO"; + ASSERT_TRUE(std::holds_alternative(cycle.elems.front())) + << "Cycle head must be a single vertex, not a nested cycle"; + Index head = cycle.head(); + activeHeads.push_back(head); + self(self, cycle.elems, activeHeads); + activeHeads.pop_back(); + } + } + }; + + std::vector activeHeads; + walk(walk, wto, activeHeads); + + for (Index u : flatOrder) { + for (auto* succ : cfg.basicBlocks[u]->out) { + Index v = succ->contents.index; + ASSERT_TRUE(pos.contains(v)) + << "Reachable vertex " << v << " missing from WTO"; + if (pos[u] >= pos[v]) { + ASSERT_TRUE(cycleMembers.contains(v)) + << "Back-edge " << u << " -> " << v << " targets non-head vertex " + << v; + EXPECT_TRUE(cycleMembers[v].contains(u)) + << "Back-edge " << u << " -> " << v + << " is not enclosed in the cycle headed by " << v; + } + } + } +} + +WTOList getWTO(TestCFG& cfg) { + WeakTopologicalOrdering wto(cfg.basicBlocks); + EXPECT_EQ(wto.elems, wto.elems); + auto list = toIndexWTO(wto.elems); + verifyWTOInvariants(cfg, list); + return list; +} + +} // namespace + +TEST(WTOTest, Empty) { + TestCFG cfg(0); + EXPECT_EQ(getWTO(cfg), WTOList{}); +} + +TEST(WTOTest, Singleton) { + TestCFG cfg(1); + EXPECT_EQ(getWTO(cfg), WTOList{0}); +} + +TEST(WTOTest, LinearChain) { + TestCFG cfg(3); + cfg.addEdge(0, 1); + cfg.addEdge(1, 2); + EXPECT_EQ(getWTO(cfg), (WTOList{0, 1, 2})); +} + +TEST(WTOTest, Diamond) { + { + TestCFG cfg(4); + cfg.addEdge(0, 1); + cfg.addEdge(0, 2); + cfg.addEdge(1, 3); + cfg.addEdge(2, 3); + EXPECT_EQ(getWTO(cfg), (WTOList{0, 1, 2, 3})); + } + { + // Reversed edge insertion order at the split and join. + TestCFG cfg(4); + cfg.addEdge(0, 2); + cfg.addEdge(0, 1); + cfg.addEdge(2, 3); + cfg.addEdge(1, 3); + EXPECT_EQ(getWTO(cfg), (WTOList{0, 1, 2, 3})); + } + { + // Asymmetric diamond (one arm has two blocks, the other has one) under both + // valid RPO block orderings. + TestCFG leftFirst(5); + leftFirst.addEdge(0, 1); + leftFirst.addEdge(1, 2); + leftFirst.addEdge(0, 3); + leftFirst.addEdge(2, 4); + leftFirst.addEdge(3, 4); + EXPECT_EQ(getWTO(leftFirst), (WTOList{0, 1, 2, 3, 4})); + + TestCFG rightFirst(5); + rightFirst.addEdge(0, 2); + rightFirst.addEdge(2, 3); + rightFirst.addEdge(0, 1); + rightFirst.addEdge(3, 4); + rightFirst.addEdge(1, 4); + EXPECT_EQ(getWTO(rightFirst), (WTOList{0, 1, 2, 3, 4})); + } +} + +TEST(WTOTest, SelfLoop) { + TestCFG cfg(3); + cfg.addEdge(0, 0); + cfg.addEdge(0, 1); + cfg.addEdge(1, 1); + cfg.addEdge(1, 2); + EXPECT_EQ(getWTO(cfg), (WTOList{C({0}), C({1}), 2})); +} + +TEST(WTOTest, SimpleCycle) { + TestCFG cfg(2); + cfg.addEdge(0, 1); + cfg.addEdge(1, 0); + EXPECT_EQ(getWTO(cfg), (WTOList{C({0, 1})})); +} + +TEST(WTOTest, ThreeNodeCycle) { + TestCFG cfg(3); + cfg.addEdge(0, 1); + cfg.addEdge(1, 2); + cfg.addEdge(2, 0); + EXPECT_EQ(getWTO(cfg), (WTOList{C({0, 1, 2})})); +} + +TEST(WTOTest, SharedLoopHeader) { + // Two loops sharing header 0: 0 -> 1 -> 0 and 0 -> 2 -> 0. + TestCFG cfg(3); + cfg.addEdge(0, 1); + cfg.addEdge(1, 0); + cfg.addEdge(0, 2); + cfg.addEdge(2, 0); + EXPECT_EQ(getWTO(cfg), (WTOList{C({0, 1, 2})})); +} + +TEST(WTOTest, BourdonclePaperExample) { + // The example control-flow graph from Bourdoncle's 1993 paper "Efficient + // chaotic iteration strategies with widenings", Figure 1 (0-indexed: vertices + // 0..7 correspond to 1..8 in the paper): + // 0 -> 1 + // 1 -> 2, 1 -> 7 + // 2 -> 3 + // 3 -> 4, 3 -> 6 + // 4 -> 5 + // 5 -> 4, 5 -> 6 + // 6 -> 2, 6 -> 7 + // Expected WTO from the paper: 0 1 (2 3 (4 5) 6) 7 + TestCFG cfg(8); + cfg.addEdge(0, 1); + cfg.addEdge(1, 2); + cfg.addEdge(1, 7); + cfg.addEdge(2, 3); + cfg.addEdge(3, 4); + cfg.addEdge(3, 6); + cfg.addEdge(4, 5); + cfg.addEdge(5, 4); + cfg.addEdge(5, 6); + cfg.addEdge(6, 2); + cfg.addEdge(6, 7); + auto wto = getWTO(cfg); + EXPECT_EQ(wto, (WTOList{0, 1, C({2, 3, C({4, 5}), 6}), 7})); + std::ostringstream ss; + ss << wto; + EXPECT_EQ(ss.str(), "0 1 (2 3 (4 5) 6) 7"); +} + +TEST(WTOTest, LoopHeaderDominatesExit) { + // Same as Bourdoncle's paper graph, except block 7 is only reachable from + // block 6 (no direct edge 1 -> 7). Block 2 dominates block 7 even though + // block 7 is outside the natural loop of 2. Block 7 must remain outside the + // cycle of 2. + TestCFG cfg(8); + cfg.addEdge(0, 1); + cfg.addEdge(1, 2); + cfg.addEdge(2, 3); + cfg.addEdge(3, 4); + cfg.addEdge(3, 6); + cfg.addEdge(4, 5); + cfg.addEdge(5, 4); + cfg.addEdge(5, 6); + cfg.addEdge(6, 2); + cfg.addEdge(6, 7); + EXPECT_EQ(getWTO(cfg), (WTOList{0, 1, C({2, 3, C({4, 5}), 6}), 7})); +} + +TEST(WTOTest, DiamondOfLoops) { + // 01 -> 23 -> 67 and 01 -> 45 -> 67, where each pair is a 2-block loop. + TestCFG cfg(8); + cfg.addEdge(0, 1); + cfg.addEdge(1, 0); + cfg.addEdge(1, 2); + cfg.addEdge(1, 4); + + cfg.addEdge(2, 3); + cfg.addEdge(3, 2); + cfg.addEdge(3, 6); + + cfg.addEdge(4, 5); + cfg.addEdge(5, 4); + cfg.addEdge(5, 6); + + cfg.addEdge(6, 7); + cfg.addEdge(7, 6); + + EXPECT_EQ(getWTO(cfg), (WTOList{C({0, 1}), C({2, 3}), C({4, 5}), C({6, 7})})); +} + +TEST(WTOTest, UnreachableBlocks) { + // Blocks 0, 1, 2 form a reachable loop 0 -> 1 -> 2 -> 1. + // Blocks 3, 4 form an unreachable cycle 3 -> 4 -> 3 with edges into 1 and 2. + TestCFG cfg(5); + cfg.addEdge(0, 1); + cfg.addEdge(1, 2); + cfg.addEdge(2, 1); + cfg.addEdge(3, 4); + cfg.addEdge(4, 3); + cfg.addEdge(3, 1); + cfg.addEdge(4, 2); + EXPECT_EQ(getWTO(cfg), (WTOList{0, C({1, 2})})); +} + +TEST(WTOTest, WorklistEvaluation) { + // Evaluate a chaotic iteration sequence on Bourdoncle's paper graph where the + // inner cycle (4 5) stabilizes in 2 iterations and the outer cycle + // (2 3 (4 5) 6) stabilizes in 2 iterations. + TestCFG cfg(8); + cfg.addEdge(0, 1); + cfg.addEdge(1, 2); + cfg.addEdge(1, 7); + cfg.addEdge(2, 3); + cfg.addEdge(3, 4); + cfg.addEdge(3, 6); + cfg.addEdge(4, 5); + cfg.addEdge(5, 4); + cfg.addEdge(5, 6); + cfg.addEdge(6, 2); + cfg.addEdge(6, 7); + + WTOWorklist work(cfg); + work.push(cfg.entry); + + std::vector visits; + unsigned count4 = 0; + unsigned count2 = 0; + work.run([&](BasicBlock* block) { + Index id = block->contents.index; + visits.push_back(id); + for (auto* out : block->out) { + Index outId = out->contents.index; + if (id == 5 && outId == 4) { + if (++count4 < 2) { + work.push(out); + } + } else if (id == 6 && outId == 2) { + if (++count2 < 2) { + count4 = 0; + work.push(out); + } + } else { + work.push(out); + } + } + }); + + // Expected recursive evaluation order: + // 0, 1, + // first iteration of (2 3 (4 5) 6): 2, 3, 4, 5, 4, 5, 6, + // second iteration of (2 3 (4 5) 6): 2, 3, 4, 5, 4, 5, 6, + // 7 + EXPECT_EQ( + visits, + (std::vector{0, 1, 2, 3, 4, 5, 4, 5, 6, 2, 3, 4, 5, 4, 5, 6, 7})); +} + +TEST(WTOTest, WorklistSelectivePropagation) { + // In Bourdoncle's graph, test when block 3 only queues block 6 (skipping the + // inner cycle (4 5) completely) and block 6 does not re-queue block 2. + TestCFG cfg(8); + cfg.addEdge(0, 1); + cfg.addEdge(1, 2); + cfg.addEdge(1, 7); + cfg.addEdge(2, 3); + cfg.addEdge(3, 4); + cfg.addEdge(3, 6); + cfg.addEdge(4, 5); + cfg.addEdge(5, 4); + cfg.addEdge(5, 6); + cfg.addEdge(6, 2); + cfg.addEdge(6, 7); + + WTOWorklist work(cfg); + work.push(cfg.entry); + + std::vector visits; + work.run([&](BasicBlock* block) { + Index id = block->contents.index; + visits.push_back(id); + if (id == 0) { + work.push(cfg.basicBlocks[1].get()); + } else if (id == 1) { + work.push(cfg.basicBlocks[2].get()); + } else if (id == 2) { + work.push(cfg.basicBlocks[3].get()); + } else if (id == 3) { + work.push(cfg.basicBlocks[6].get()); + } else if (id == 6) { + work.push(cfg.basicBlocks[7].get()); + } + }); + + EXPECT_EQ(visits, (std::vector{0, 1, 2, 3, 6, 7})); +} From 50f706451fdde318668cfa04cc0316c0b22b7558 Mon Sep 17 00:00:00 2001 From: Thomas Lively Date: Tue, 6 Oct 2026 11:29:12 -0700 Subject: [PATCH 2/7] Work around clang++-18 crash on defaulted WTOCycle::operator== --- test/gtest/wto.cpp | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/test/gtest/wto.cpp b/test/gtest/wto.cpp index 026951248f0..4a7ed21b7ac 100644 --- a/test/gtest/wto.cpp +++ b/test/gtest/wto.cpp @@ -94,7 +94,9 @@ struct WTOElem : std::variant { WTOCycle::WTOCycle(std::initializer_list list) : elems(list) {} WTOCycle::WTOCycle(WTOList elems) : elems(std::move(elems)) {} Index WTOCycle::head() const { return std::get(elems.front()); } -bool WTOCycle::operator==(const WTOCycle& other) const = default; +bool WTOCycle::operator==(const WTOCycle& other) const { + return elems == other.elems; +} WTOCycle C(std::initializer_list list) { return WTOCycle(list); } From 08750e7cb37fdf518a7a8d0bc56e7a9134bf5f77 Mon Sep 17 00:00:00 2001 From: Thomas Lively Date: Tue, 6 Oct 2026 14:35:34 -0700 Subject: [PATCH 3/7] tighten up definition --- src/cfg/wto.h | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/src/cfg/wto.h b/src/cfg/wto.h index 50936c60203..9475317af86 100644 --- a/src/cfg/wto.h +++ b/src/cfg/wto.h @@ -22,12 +22,13 @@ // strategies with widenings", 1993) is a hierarchical ordering of the reachable // blocks of a directed graph in which strongly connected components (loops) are // parenthesized into nested cycles. The first element of each cycle is its -// "head" (loop header), and the ordering satisfies two properties: +// "head" (loop header). Formally, the WTO of a directed graph is a hierarchical +// ordering of its vertices such that for every edge u -> v, either: // -// 1. Every non-cycle edge u -> v goes forward in the flattened ordering -// (u appears before v). -// 2. Every backedge u -> v targets the head v of a cycle that encloses both -// u and v. +// 1. u < v (i.e. this is a forward edge) and v is not the head of a cycle +// containing u. +// 2. u >= v (i.e. this is a backedge) and v is the head of a cycle containing +// u. // // Examples (writing `(h ...)` for a cycle with head `h`): // From 4aea7cde91922f5a22251a288e8465d46e80b827 Mon Sep 17 00:00:00 2001 From: Thomas Lively Date: Tue, 6 Oct 2026 16:35:56 -0700 Subject: [PATCH 4/7] Moar comments (I wrote them myself!) --- src/cfg/wto.h | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/src/cfg/wto.h b/src/cfg/wto.h index 9475317af86..7e4e8afa4d5 100644 --- a/src/cfg/wto.h +++ b/src/cfg/wto.h @@ -139,9 +139,15 @@ WeakTopologicalOrdering::WeakTopologicalOrdering( static constexpr Index NoIndex = Index(-1); struct Node { + // The innermost loop header for the cycle containing this block. Index loopParent = NoIndex; + // For loop headers, the index of their first child (i.e. the head of a + // linked list of children). Index firstChild = NoIndex; + // A linked list edge to the next child with the same loop header. Index nextSibling = NoIndex; + // The index of the loop header we last traversed this node for, used + // instead of a `visited` set during the DFS. Index lastVisitedBy = NoIndex; bool isLoopHeader = false; }; @@ -159,17 +165,27 @@ WeakTopologicalOrdering::WeakTopologicalOrdering( if (!isReachable(h)) { continue; } + // Check if h is the head of a loop. It is a loop header if and only if it + // dominates one of its predecessors. (We assume the CFG is reducible, so + // loop headers dominate all blocks in the loop bodies, including those that + // branch back to the header.) nodes[h].lastVisitedBy = h; for (auto* pred : blocks[h]->in) { Index p = pred->contents.index; if (dominates(h, p)) { nodes[h].isLoopHeader = true; + // Avoid repeat traversals by setting lastVisitedBy = h on visited + // blocks. if (nodes[p].lastVisitedBy != h) { nodes[p].lastVisitedBy = h; worklist.push_back(p); } } } + // We've initialized the worklist with all the loop tails that branch + // directly back to the loop header. DFS from those loop tails back to the + // loop header (but no further). All the blocks we find during the DFS are + // part of the loop body. while (!worklist.empty()) { Index curr = worklist.back(); worklist.pop_back(); @@ -178,6 +194,8 @@ WeakTopologicalOrdering::WeakTopologicalOrdering( } for (auto* pred : blocks[curr]->in) { Index p = pred->contents.index; + // The loop header has lastVisitedBy == h, so the search will stop + // there. if (isReachable(p) && nodes[p].lastVisitedBy != h) { assert(dominates(h, p) && "Expected reducible CFG"); nodes[p].lastVisitedBy = h; @@ -197,14 +215,17 @@ WeakTopologicalOrdering::WeakTopologicalOrdering( } Index parent = nodes[idx].loopParent; if (parent == NoIndex) { + // Prepend to top-level list. nodes[idx].nextSibling = topFirstChild; topFirstChild = idx; } else { + // Prepend to loop header's list. nodes[idx].nextSibling = nodes[parent].firstChild; nodes[parent].firstChild = idx; } } + // Traverse the linked lists of children, materializing them as WTO elements. // TODO: Flatten the WTO into a single contiguous vector of entries with cycle // jump targets to avoid per-cycle vector allocations and recursion. auto buildList = [&](auto& self, Index firstChild, List& out) -> void { From a469c96ae9b4cf0eaa74071b2e659675a189dd96 Mon Sep 17 00:00:00 2001 From: Thomas Lively Date: Tue, 6 Oct 2026 16:53:31 -0700 Subject: [PATCH 5/7] mini CFG comment --- test/gtest/wto.cpp | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/test/gtest/wto.cpp b/test/gtest/wto.cpp index 4a7ed21b7ac..5e8135d7669 100644 --- a/test/gtest/wto.cpp +++ b/test/gtest/wto.cpp @@ -28,6 +28,11 @@ using namespace wasm; namespace { +// The WTO utility is parameterized on the BasicBlock type associated with a +// CFG. Since BasicBlock itself contains a user-provided Contents type, there is +// no canonical BasicBlock type ready to use. Since we don't need a full +// CFGWalker, just create a mini version of CFGWalker for testing that has all +// the expected associated types and fields. struct TestCFG { struct Contents { bool inQueue = false; From f4d0e7631c6bad3414ce89e8c0f7acc1d7a45333 Mon Sep 17 00:00:00 2001 From: Thomas Lively Date: Tue, 6 Oct 2026 17:06:02 -0700 Subject: [PATCH 6/7] recursion comments --- src/cfg/wto.h | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/src/cfg/wto.h b/src/cfg/wto.h index 7e4e8afa4d5..55761e9a3c1 100644 --- a/src/cfg/wto.h +++ b/src/cfg/wto.h @@ -226,6 +226,8 @@ WeakTopologicalOrdering::WeakTopologicalOrdering( } // Traverse the linked lists of children, materializing them as WTO elements. + // Loop depth should be limited, so doing this recursively should be fine. If + // it ever causes an issue, we can un-recurse this. // TODO: Flatten the WTO into a single contiguous vector of entries with cycle // jump targets to avoid per-cycle vector allocations and recursion. auto buildList = [&](auto& self, Index firstChild, List& out) -> void { @@ -280,8 +282,12 @@ template struct WTOWorklist { void push(BasicBlock* block) { block->contents.inQueue = true; } template void run(VisitFn&& visit) { - // TODO: Track the number of queued blocks to stop early once the worklist - // is empty. + // Iterate through each element in the current cycle's list (or the + // top-level list), which will be in reverse postorder. Visit those that are + // in the queue, which may push later elements to the queue. When there is a + // nested cycle, repeatedly visit it recursively until it stabilizes before + // continuing on. We could un-recurse this, but the loop depth is expected + // to be acceptably small. // TODO: Fast-path initial entry singletons and CFGs without backedges // without building DomTree or WTO, using CFGWalker::loopTops. WeakTopologicalOrdering wto(cfg.basicBlocks); From f1a4061fa6351401b196f414908a59dd9d34656b Mon Sep 17 00:00:00 2001 From: Thomas Lively Date: Tue, 6 Oct 2026 17:42:14 -0700 Subject: [PATCH 7/7] Work around GCC 11 ICE on local static constexpr in WTO --- src/cfg/wto.h | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/src/cfg/wto.h b/src/cfg/wto.h index 55761e9a3c1..b12a57ff74e 100644 --- a/src/cfg/wto.h +++ b/src/cfg/wto.h @@ -89,6 +89,8 @@ namespace wasm { // The BasicBlock type is assumed to have an `in` vector of predecessor block // pointers and a `contents.index` field of type `Index`. template struct WeakTopologicalOrdering { + static constexpr Index NoIndex = Index(-1); + struct Cycle; using Element = std::variant; using List = std::vector; @@ -97,7 +99,7 @@ template struct WeakTopologicalOrdering { List elems; BasicBlock* head() const { return std::get(elems.front()); } - bool operator==(const Cycle&) const = default; + bool operator==(const Cycle& other) const { return elems == other.elems; } }; List elems; @@ -137,7 +139,6 @@ WeakTopologicalOrdering::WeakTopologicalOrdering( return curr == dom; }; - static constexpr Index NoIndex = Index(-1); struct Node { // The innermost loop header for the cycle containing this block. Index loopParent = NoIndex;