// Structured Program Candidate handoff modules for loom-lower-scf-to-dfg. // Module scope: optional llvm.func / func.func callables plus dataflow.thread // definitions; every spatial candidate is an explicit loom.spatial_region in // exactly one thread, holding fixed-domain graph-owned parallel SCF with // arbitrary nesting of scf.if / scf.for / scf.while. start: {new TCOUNT = random.randint(1, 2); new TI = 0; new PRELUDE = random.choice(['none', 'llvm', 'func'])} prelude threads; prelude: (PRELUDE == 'none') '' | (PRELUDE == 'llvm') 'llvm.func @imported_kernel(i64)\n\n' | (PRELUDE == 'func') 'func.func @native_helper(%arg0: index) -> index {\n return %arg0 : index\n}\n\n'; threads: (TI < TCOUNT) thread_def {TI += 1} threads | (TI == TCOUNT) ''; thread_def: {new RESIDENT = random.choice([0, 1]); new W = random.choice([1, 2, 4]); new PAR = random.choice([0, 1, 2]); new OUTER = random.choice([0, 1, 2]); new INNER = random.choice([0, 1, 2, 3])} 'dataflow.thread private @t' tname ' domain(#dataflow.thread_domain)(\n' ' %scratch: memref<8xindex>, %memory: memref<8xindex>, %grid: memref<4x4xindex>,\n' ' %n: index) ctrl (%ctrl: none) {\n' resident spatial_region ' dataflow.thread.yield\n}\n\n'; tname: [str(TI)]; // InstructionCore-resident thread body code outside the spatial boundary. resident: (RESIDENT == 0) '' | (RESIDENT == 1) ' %rzero = arith.constant 0 : index\n %rval = arith.constant 3 : index\n memref.store %rval, %scratch[%rzero] : memref<8xindex>\n'; spatial_region: ' "loom.spatial_region"(%n, %memory, %grid)\n' ' <{operandSegmentSizes = array,\n' ' resultSegmentSizes = array}> ({\n' ' ^bb0(%limit: index, %target: memref<8xindex>, %tile: memref<4x4xindex>):\n' consts outer ' "loom.spatial_yield"()\n' ' <{operandSegmentSizes = array}> : () -> ()\n' ' }) {graph_name = "g_t' tname '_0", source_maps = []} :\n' ' (index, memref<8xindex>, memref<4x4xindex>) -> ()\n'; consts: ' %c0 = arith.constant 0 : index\n' ' %c1 = arith.constant 1 : index\n' ' %cw = arith.constant ' width ' : index\n' ' %kv = arith.constant 7 : index\n'; width: [str(W)]; // Optional sequential nesting around the graph-owned parallel op. outer: (OUTER == 0) par | (OUTER == 1) ' %ocond = arith.cmpi slt, %c0, %limit : index\n scf.if %ocond {\n' par ' }\n' | (OUTER == 2) ' scf.for %oi = %c0 to %limit step %c1 {\n' par ' }\n'; // Fixed-width graph-owned parallel composition: effect-form scf.forall or // fixed-domain scf.parallel with compile-time constant bounds. par: (PAR == 0) ' scf.forall (%lane) in (' width ') {\n' inner ' }\n' | (PAR == 1) ' scf.parallel (%lane) = (%c0) to (%cw) step (%c1) {\n' inner ' scf.reduce\n }\n' | (PAR == 2) ' scf.parallel (%pi) = (%c0) to (%cw) step (%c1) {\n scf.parallel (%pj) = (%c0) to (%cw) step (%c1) {\n memref.store %kv, %tile[%pi, %pj] : memref<4x4xindex>\n scf.reduce\n }\n scf.reduce\n }\n'; // Lane-disjoint body: arbitrary nesting of scf.if / scf.for / scf.while. inner: (INNER == 0) ' memref.store %kv, %target[%lane] : memref<8xindex>\n' | (INNER == 1) ' %bcond = arith.cmpi slt, %lane, %cw : index\n scf.if %bcond {\n memref.store %kv, %target[%lane] : memref<8xindex>\n }\n' | (INNER == 2) ' %bsum = scf.for %bi = %c0 to %limit step %c1 iter_args(%bacc = %lane) -> (index) {\n %bnext = arith.addi %bacc, %c1 : index\n scf.yield %bnext : index\n }\n memref.store %bsum, %target[%lane] : memref<8xindex>\n' | (INNER == 3) ' %wres = scf.while (%wi = %c0) : (index) -> index {\n %wc = arith.cmpi slt, %wi, %limit : index\n scf.condition(%wc) %wi : index\n } do {\n ^bb0(%wb: index):\n %wn = arith.addi %wb, %c1 : index\n scf.yield %wn : index\n }\n memref.store %wres, %target[%lane] : memref<8xindex>\n';