Skip to content

Commit 8962a2a

Browse files
committed
compiler: Keep the outer Dimension parallel on a short par-tile
When a multi `par-tile` entry is shorter than the blocked nest, the innermost Dimensions consume the available block sizes and the outermost one runs into a StopIteration. It was then dropped into `compact`, which promotes it back to its root Dimension, discarding its BlockDimension. On a device that is a serialization: the Dimension ends up outside the blocked nest, `filter_iterations` rejects it as non-parallel, and the kernel is launched over a 2D grid from a host loop iterating the outer Dimension one slice at a time. Pad such an entry with its innermost size instead, so the nest stays fully blocked. Whether to do so is decided when the par-tile is built, where the target is already known, rather than by inspecting a cluster back in the pass. It applies to multi par-tiles on a device only: a single user-supplied par-tile defines the block rank on purpose, and on a host a short entry is the documented way to ask for 2.5D blocking.
1 parent 33c3b12 commit 8962a2a

4 files changed

Lines changed: 72 additions & 5 deletions

File tree

‎devito/core/gpu.py‎

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -90,7 +90,8 @@ def _normalize_kwargs(cls, **kwargs):
9090
# GPU parallelism
9191
o['par-tile'] = ParTile(oo.pop('par-tile', False), default=(32, 4, 4),
9292
sparse=oo.pop('par-tile-sparse', None),
93-
reduce=oo.pop('par-tile-reduce', None))
93+
reduce=oo.pop('par-tile-reduce', None),
94+
stretch=True)
9495
o['par-collapse-ncores'] = 1 # Always collapse (meaningful if `par-tile=False`)
9596
o['par-collapse-work'] = 1 # Always collapse (meaningful if `par-tile=False`)
9697
o['par-chunk-nonaffine'] = oo.pop('par-chunk-nonaffine', cls.PAR_CHUNK_NONAFFINE)

‎devito/core/operator.py‎

Lines changed: 10 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -478,9 +478,16 @@ def __new__(cls, items, rule=None, tag=None):
478478

479479
class ParTile(UnboundedMultiTuple, OptOption):
480480

481-
def __new__(cls, items, default=None, sparse=None, reduce=None):
481+
# A short entry is padded to the nest it lands on rather than losing its
482+
# outermost BlockDimensions. Only meaningful on a device
483+
stretch = False
484+
485+
def __new__(cls, items, default=None, sparse=None, reduce=None,
486+
stretch=False):
482487
if not items:
483-
return UnboundedMultiTuple()
488+
obj = UnboundedMultiTuple()
489+
obj.stretch = False
490+
return obj
484491
elif isinstance(items, bool):
485492
if not default:
486493
raise ValueError("Expected `default` value, got None")
@@ -536,6 +543,7 @@ def __new__(cls, items, default=None, sparse=None, reduce=None):
536543
obj.default = as_tuple(default)
537544
obj.sparse = as_tuple(sparse)
538545
obj.reduce = as_tuple(reduce)
546+
obj.stretch = stretch
539547

540548
return obj
541549

‎devito/passes/clusters/blocking.py‎

Lines changed: 27 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -495,6 +495,7 @@ class BlockSizeGenerator:
495495

496496
def __init__(self, par_tile):
497497
self.umt = par_tile
498+
self.stretch = par_tile.stretch
498499

499500
if par_tile.is_multi:
500501
# The user has supplied one specific par-tile per blocked nest
@@ -522,7 +523,25 @@ def __init__(self, par_tile):
522523
else:
523524
self.umt_reduce = UnboundTuple(*par_tile.default, 1)
524525

526+
def stretched(self, umt, dims):
527+
"""
528+
Pad `umt` with its innermost size until it covers `dims`.
529+
530+
A user-supplied par-tile entry may be shorter than the nest it lands on.
531+
Consuming it as is would leave the outermost Dimensions without a block
532+
size, and dropping their BlockDimensions would cost them their
533+
parallelism -- on a device, that means iterating them on the host.
534+
"""
535+
if not umt or len(umt) >= len(dims):
536+
return umt
537+
538+
sizes = tuple(umt) + (umt[-1],) * (len(dims) - len(umt))
539+
540+
return UnboundTuple(*sizes)
541+
525542
def schedule(self, dims, clusters):
543+
stretch = False
544+
526545
if any(c.properties.is_parallel_atomic(dims) for c in clusters):
527546
# Correctness -- enforce blocking where necessary.
528547
# See also issue #276:PRO
@@ -537,9 +556,14 @@ def schedule(self, dims, clusters):
537556

538557
else:
539558
umt = self.umt
559+
stretch = self.stretch and umt.is_multi
540560

541561
umt.iter()
542562

563+
if stretch:
564+
# `umt` walks this nest's own entry, which may be too short
565+
return self.stretched(umt.curitem(), dims).reset()
566+
543567
return umt
544568

545569

@@ -622,10 +646,11 @@ def apply_par_tiles(clusters, options, **kwargs):
622646
Use the par-tile parameter to replace the symbolic BlockDimension sizes
623647
with actual integer numbers representing the block shape.
624648
"""
625-
if not options['par-tile']:
649+
par_tile = options['par-tile']
650+
if not par_tile:
626651
return clusters
627652

628-
blk_size_gen = BlockSizeGenerator(options['par-tile'])
653+
blk_size_gen = BlockSizeGenerator(par_tile)
629654

630655
key = lambda c: c.ispace.project(lambda d: d.is_Block)
631656

‎tests/test_gpu_openacc.py‎

Lines changed: 33 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -144,6 +144,39 @@ def test_multiple_tile_sizes(self, par_tile):
144144
assert trees[3][1].pragmas[0].ccode.value ==\
145145
f'acc parallel loop {sclause} present(src,src_gp,src_wx,src_wy,src_wz,u)'
146146

147+
def test_short_multi_tile_keeps_outer_dim_blocked(self):
148+
"""
149+
A multi `par-tile` entry shorter than the nest it lands on must not cost
150+
the outermost Dimension its BlockDimension: on a device, dropping it
151+
would leave `x` iterated outside the offloaded nest.
152+
"""
153+
grid = Grid(shape=(8, 8, 8))
154+
155+
u = TimeFunction(name="u", grid=grid, space_order=4)
156+
v = TimeFunction(name="v", grid=grid, space_order=4)
157+
158+
eqns = [Eq(u.forward, u.dx),
159+
Eq(v.forward, u.forward.dx)]
160+
161+
# The second entry is 2D, while the nest it lands on is 3D
162+
par_tile = ((32, 4, 4), (16, 4))
163+
164+
op = Operator(eqns, platform='nvidiaX', language='openacc',
165+
opt=(
166+
'advanced',
167+
{'par-tile': par_tile, 'blocklevels': 1, 'blockinner': True}))
168+
169+
bns, _ = assert_blocking(op, {'x0_blk0', 'x1_blk0'})
170+
171+
# The short entry is stretched by repeating its innermost size, so the
172+
# nest stays fully blocked rather than losing its outermost Dimension
173+
expected = ((4, 4, 32), (4, 4, 16))
174+
for root, v in zip(bns.values(), expected, strict=True):
175+
iters = FindNodes(Iteration).visit(root)
176+
iters = [i for i in iters if i.dim.is_Block and i.dim._depth == 1]
177+
assert len(iters) == len(v)
178+
assert all(i.step == j for i, j in zip(iters, v, strict=True))
179+
147180
def test_multi_tile_blocking_structure(self):
148181
grid = Grid(shape=(8, 8, 8))
149182

0 commit comments

Comments
 (0)