EnzymeAD / EnzymeAD/Enzyme-JAX
Remaining all-reduce
- Dominant language
- MLIR
- Stars
- 131
- Forks
- 53
- Avg merge
- 1d 10h
- Merged PRs (30d)
- 193
Description
```
%701:26 = stablehlo.while(%iterArg = %377, %iterArg_9 = %arg12, %iterArg_10 = %476, %iterArg_11 = %477, %iterArg_12 = %473, %iterArg_13 = %93, %iterArg_14 = %93, %iterArg_15 = %93, %iterArg_16 = %407, %iterArg_17 = %408, %iterArg_18 = %92, **** %iterArg_19 ***** = %459, %iterArg_20 = %460, %iterArg_21 = %515, %iterArg_22 = %516, %iterArg_23 = %517, %iterArg_24 = %478, %iterArg_25 = %440, %iterArg_26 = %518, %iterArg_27 = %519, %iterArg_28 = %520, %iterArg_29 = %521, %iterArg_30 = %522, %iterArg_31 = %523, %iterArg_32 = %524, %iterArg_33 = %525) : tensor, tensor, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<20x1536x3056xf64>, tensor<20x1536x3056xf64>, tensor<4x1534x3070xf64>, tensor<6x1522x3056xf64>, tensor<6x1522x3056xf64>, tensor<4x1522x3058xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64> attributes {enzyme.disable_mincut, sdy.sharding = #sdy.sharding_per_value<[<@mesh, []>, <@mesh, []>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>, <@mesh, [{}, {"y"}, {"x"}]>]>}
cond {
%1962 = stablehlo.slice %iterArg_19 [0:1, 0:1, 0:3056] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<6x1522x3056xf64>) -> tensor<1x1x3056xf64> loc(#loc4137)
%1951 = stablehlo.slice %875 [0:4, 0:1, 0:3056] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<4x1520x3056xf64>) -> tensor<4x1x3056xf64> loc(#loc4135)
%1933 = stablehlo.slice %iterArg_19 [5:6, 0:1, 0:3056] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<6x1522x3056xf64>) -> tensor<1x1x3056xf64> loc(#loc369)
%1963 = stablehlo.concatenate %1962, %1951, %1933, dim = 0 {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<1x1x3056xf64>, tensor<4x1x3056xf64>, tensor<1x1x3056xf64>) -> tensor<6x1x3056xf64> loc(#loc4137)
%1954 = stablehlo.slice %875 [0:4, 1519:1520, 0:3056] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<4x1520x3056xf64>) -> tensor<4x1x3056xf64> loc(#loc4136)
%1970 = stablehlo.slice %iterArg_19 [0:6, 1521:1522, 0:3056] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<6x1522x3056xf64>) -> tensor<6x1x3056xf64> loc(#loc4139)
%1971 = stablehlo.dynamic_update_slice %1970, %1954, %c_7, %c_8, %c_8 {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<6x1x3056xf64>, tensor<4x1x3056xf64>, tensor, tensor, tensor) -> tensor<6x1x3056xf64> loc(#loc4139)
%1928 = stablehlo.slice %875 [0:1, 0:1520, 0:3056] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<4x1520x3056xf64>) -> tensor<1x1520x3056xf64> loc(#loc4128)
%1929 = stablehlo.slice %875 [3:4, 0:1520, 0:3056] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<4x1520x3056xf64>) -> tensor<1x1520x3056xf64> loc(#loc4128)
%1930 = stablehlo.concatenate %1928, %875, %1929, dim = 0 {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<1x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<1x1520x3056xf64>) -> tensor<6x1520x3056xf64> loc(#loc4128)
extend-like?
%1972 = stablehlo.pad %1963, %cst, low = [0, 0, 0], high = [0, 1521, 0], interior = [0, 0, 0] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<6x1x3056xf64>, tensor) -> tensor<6x1522x3056xf64> loc(#loc369)
%1973 = stablehlo.pad %1930, %cst, low = [0, 1, 0], high = [0, 1, 0], interior = [0, 0, 0] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<6x1520x3056xf64>, tensor) -> tensor<6x1522x3056xf64> loc(#loc369)
%1974 = stablehlo.pad %1971, %cst, low = [0, 1521, 0], high = [0, 0, 0], interior = [0, 0, 0] {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : (tensor<6x1x3056xf64>, tensor) -> tensor<6x1522x3056xf64> loc(#loc369)
%1975 = stablehlo.add %1972, %1973 {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : tensor<6x1522x3056xf64> loc(#loc369)
%1976 = stablehlo.add %1975, %1974 {sdy.sharding = #sdy.sharding_per_value<[<@mesh, [{}, {"y"}, {"x"}]>]>} : tensor<6x1522x3056xf64> loc(#loc369)
stablehlo.return %823, %1847, %1833, %1834, %1846, %1833, %1859, %1857, %1966, %1926, %2069, **** %1976 **** , %2000, %2595, %3892, %5154, %860, %859, %5782, %6404, %iterArg_22, %iterArg_23, %860, %859, %iterArg_26, %iterArg_27 : tensor, tensor, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<20x1536x3056xf64>, tensor<20x1536x3056xf64>, tensor<4x1534x3070xf64>, tensor<6x1522x3056xf64>, tensor<6x1522x3056xf64>, tensor<4x1522x3058xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<1x1520x3056xf64>, tensor<4x1520x3056xf64>, tensor<4x1520x3056xf64> loc(#loc)
```
Contributor guide
No contributing guide indexed for this repository
Research direction
The issue names no source files, tests, or entry points; begin by identifying the Enzyme-JAX pass or StableHLO transformation responsible for the remaining all-reduce in the shown while loop. Determine the intended elimination or lowering behavior and add a focused regression test demonstrating that the remaining all-reduce is handled.
Written by the indexing model from the issue text.
Assessment
- Domain
- compilers
- Issue type
- Feature
- Difficulty
- 5/5
- Estimated time
- Over a week
- Activity status
- Stale
- Clarity
- Needs clarification
- Newbie friendliness
- 15/100