Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 28 additions & 1 deletion include/dataflow-scheduler/Dialect/KTDF/KTDF.td
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ include "mlir/Interfaces/LoopLikeInterface.td"
include "mlir/Interfaces/SideEffectInterfaces.td"
include "mlir/IR/OpAsmInterface.td"

include "dataflow-scheduler/Dialect/KTDF/KTDFAttributes.td"
include "dataflow-scheduler/Dialect/KTDF/KTDFTypes.td"

// Base class for KTDF dialect operations
Expand Down Expand Up @@ -168,6 +169,18 @@ def KTDF_ReadFromFifoOp : KTDF_Op<"read_from_fifo", [
limited to the op's parent region scope, cannot leave that scope (eg via
yield) and does not alias any other memrefs.

## Splat mode

The optional `splat` attribute records how to form a full SIMD-width vector
from the subset of elements delivered over the FIFO. The FIFO carries fewer
than a full SIMD width of elements; the attribute names the shuffle that
fills in the remaining lanes to produce the final vector.

The scheduler sets the attribute on every `ktdf.read_from_fifo` that reads
such a transfer. The lowering to DFIR emits the named shuffle immediately
after the receive. Any device pattern that re-reads the slot as a buffer
must preserve the attribute, because the shuffle is still owed.

Example:
```mlir
// Read from a FIFO slot into a tensor
Expand All @@ -182,11 +195,17 @@ def KTDF_ReadFromFifoOp : KTDF_Op<"read_from_fifo", [
%size = arith.constant 8 : index
%dyn_slot = ktdf.fifo.allocate(%size) -> !ktdf.fifo.slot<"SFU" -> "L1SU", ?xf16>
%dyn_data = ktdf.read_from_fifo %dyn_slot : !ktdf.fifo.slot<"SFU" -> "L1SU", ?xf16> -> tensor<?xf16>

// Read a splat the load unit left one element per sub-SIMD group wide
%splat = ktdf.read_from_fifo %slot
{splat = #ktdf.splat<first_subsimd_lane_to_each_subsimd>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> memref<64xf16>
```
}];

let arguments = (ins
KTDF_FifoSlotType:$fifo_slot
KTDF_FifoSlotType:$fifo_slot,
OptionalAttr<KTDF_SplatModeAttr>:$splat
);

let results = (outs
Expand All @@ -199,6 +218,14 @@ def KTDF_ReadFromFifoOp : KTDF_Op<"read_from_fifo", [

let hasVerifier = 1;

let builders = [
// A read of a slot that owes no shuffle, i.e. every read without a splat
// attribute.
OpBuilder<(ins "::mlir::Type":$result, "::mlir::Value":$fifo_slot), [{
build($_builder, $_state, result, fifo_slot, /*splat=*/nullptr);
}]>
];

let extraClassDeclaration = [{
/// Returns the element type of the FIFO slot
auto getFifoElementType() -> Type {
Expand Down
27 changes: 27 additions & 0 deletions include/dataflow-scheduler/Dialect/KTDF/KTDFAttributes.td
Original file line number Diff line number Diff line change
Expand Up @@ -66,4 +66,31 @@ def KTDF_LoopTypeAttr : EnumAttr<KTDF_Dialect, KTDF_LoopTypeEnum, "loop_type"> {
let assemblyFormat = "`<` $value `>`";
}

//===----------------------------------------------------------------------===//
// SplatModeAttr
//===----------------------------------------------------------------------===//

def KTDF_SplatModeAttr : EnumAttr<KTDF_Dialect, KTDF_SplatModeEnum, "splat"> {
let summary = "KTDF splat mode attribute indicating how to form a full SIMD vector from a subset of elements";
let description = [{
Records how to form a full SIMD-width vector from the subset of elements
delivered over the FIFO. The FIFO carries fewer than a full SIMD width of
elements; the attribute names the shuffle that fills in the remaining lanes.

`first_subsimd_lane_to_each_subsimd` fills every lane of each sub-SIMD
group with the value held by its first lane. The group width is not encoded
here; it is read from the compute unit's
`ktdf_arch.feature.simd = { sub_simd_lanes }`.

Example:

```mlir
%d = ktdf.read_from_fifo %slot
{splat = #ktdf.splat<first_subsimd_lane_to_each_subsimd>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> memref<64xf16>
```
}];
let assemblyFormat = "`<` $value `>`";
}

#endif // DATAFLOW_SCHEDULER_DIALECT_KTDF_KTDFATTRIBUTES_TD
14 changes: 14 additions & 0 deletions include/dataflow-scheduler/Dialect/KTDF/KTDFEnums.td
Original file line number Diff line number Diff line change
Expand Up @@ -39,4 +39,18 @@ def KTDF_LoopTypeEnum : I32EnumAttr<"LoopType",
let genSpecializedAttr = 0;
}

//===----------------------------------------------------------------------===//
// Splat Mode Enum
//===----------------------------------------------------------------------===//

def KTDF_SplatModeEnum : I32EnumAttr<"SplatMode",
"KTDF splat mode attribute indicating how to form a full SIMD vector from a subset of elements",
[
I32EnumAttrCase<"FirstSubSimdLaneToEachSubSimd", 0,
"first_subsimd_lane_to_each_subsimd">
]> {
let cppNamespace = "::mlir::ktdf";
let genSpecializedAttr = 0;
}

#endif // DATAFLOW_SCHEDULER_DIALECT_KTDF_KTDFENUMS_TD
Original file line number Diff line number Diff line change
Expand Up @@ -363,6 +363,7 @@ struct SIMD : FeatureAttr<&KTDFArchDialect::getFeatureSIMDAttrName> {
static constexpr StringLiteral kSplatAttrName = "splat";
static constexpr StringLiteral kZeroPadAttrName = "zero_pad";
static constexpr StringLiteral kLanesAttrName = "lanes";
static constexpr StringLiteral kSubSimdLanesAttrName = "sub_simd_lanes";

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Keeping with the style of the existing attribute, this should be called sub_lanes I think.

There was also the open design question on whether SIMD lanes could be multidimensional, i.e., f16 = 64x64 to indicate matrix-type accelerators. I don't think this is quite the way to go, but it would be good to know whether sub SIMD lanes would be any different in that regard.


using LanesAttr = TypedMapAttr<TypeAttr, I64Attr>;

Expand Down Expand Up @@ -395,6 +396,21 @@ struct SIMD : FeatureAttr<&KTDFArchDialect::getFeatureSIMDAttrName> {
}
return 0;
}

/// Gets the width of a sub-SIMD group, in lanes, per scalar type. The lanes
/// of a group are what a shuffle mode naming sub-SIMD groups operates over.
[[nodiscard]] auto getSubSimdLanes() const -> LanesAttr {
return getAttr<LanesAttr>(kSubSimdLanesAttrName);
}
/// Gets the width of a sub-SIMD group, in lanes, for @p scalar_type .
///
/// @returns Group width for @p scalar_type , or 0 if not declared.
[[nodiscard]] auto getSubSimdLanes(Type scalar_type) const -> int64_t {
if (const auto lanes = getSubSimdLanes(); lanes) {
return lanes.getValue(TypeAttr::get(scalar_type)).value_or(0);
}
return 0;
}
};

// Indicates that the execution unit can perform store operations.
Expand Down
22 changes: 22 additions & 0 deletions lib/Dialect/KTDFArch/KTDFArchIntrinsics.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -515,6 +515,12 @@ auto feature::SIMD::verify(EmitErrorFn emit_error) const -> LogicalResult {
<< "' requires map from type to 64-bit integer";
}

const auto sub_simd_lanes = getAttr(kSubSimdLanesAttrName);
if (sub_simd_lanes && !isa<LanesAttr>(sub_simd_lanes)) {
return emit_error() << "attribute '" << kSubSimdLanesAttrName
<< "' requires map from type to 64-bit integer";
}

return success();
}

Expand Down Expand Up @@ -551,6 +557,22 @@ auto feature::SIMD::test(feature::SIMD requirements) const -> bool {
}
}

if (const auto required = requirements.getSubSimdLanes(); required) {
const auto required_lanes = required.getEntries();
const auto provided_lanes = getSubSimdLanes();

if (!provided_lanes && !required_lanes.empty()) {
return false;
}

if (llvm::any_of(required_lanes, [&](const auto& require) -> bool {
return provided_lanes.getValue(require.first) <
require.second.getValue();
})) {
return false;
}
}

return true;
}

Expand Down
29 changes: 29 additions & 0 deletions test/Dialect/KTDF/read-from-fifo-op.mlir
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,35 @@ module {
// Use the buffer
"test.op"(%buf0) : (memref<64xf16>) -> ()

return
}
}

// The splat mode rides along on either result form: it says the load unit left
// one live element per sub-SIMD group, and names the shuffle that finishes it.

// CHECK-LABEL: func.func @read_from_fifo_splat() {
// CHECK-NEXT: %[[FIFO_0:.*]] = ktdf.fifo.allocate() -> !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16>
// CHECK-NEXT: %[[READ_FROM_FIFO_0:.*]] = ktdf.read_from_fifo %[[FIFO_0]] {splat = #ktdf.splat<first_subsimd_lane_to_each_subsimd>} : <"L1LU" -> "SFU", 64xf16> -> tensor<64xf16>
// CHECK-NEXT: %[[READ_FROM_FIFO_1:.*]] = ktdf.read_from_fifo %[[FIFO_0]] {splat = #ktdf.splat<first_subsimd_lane_to_each_subsimd>} : <"L1LU" -> "SFU", 64xf16> -> memref<64xf16>
// CHECK-NEXT: "test.op"(%[[READ_FROM_FIFO_0]], %[[READ_FROM_FIFO_1]]) : (tensor<64xf16>, memref<64xf16>) -> ()
// CHECK-NEXT: return
// CHECK-NEXT: }

module {
func.func @read_from_fifo_splat() {
%slot0 = ktdf.fifo.allocate() -> !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16>

%data0 = ktdf.read_from_fifo %slot0
{splat = #ktdf.splat<first_subsimd_lane_to_each_subsimd>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> tensor<64xf16>

%buf0 = ktdf.read_from_fifo %slot0
{splat = #ktdf.splat<first_subsimd_lane_to_each_subsimd>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> memref<64xf16>

"test.op"(%data0, %buf0) : (tensor<64xf16>, memref<64xf16>) -> ()

return
}
}
7 changes: 7 additions & 0 deletions test/Dialect/KTDFArch/features-invalid.mlir
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,13 @@ ktdf_arch.device @simd_invalid_lanes {

// -----

ktdf_arch.device @simd_invalid_sub_simd_lanes {
// expected-error@+1 {{'sub_simd_lanes' requires map from type to 64-bit integer}}
exec_unit { ktdf_arch.features = { ktdf_arch.feature.simd = { sub_simd_lanes = 1 } } }
}

// -----

ktdf_arch.device @queue_not_on_link {
// expected-error@+1 {{only valid on links}}
exec_unit { ktdf_arch.features = { ktdf_arch.feature.queue = { size = "a" } } }
Expand Down
5 changes: 4 additions & 1 deletion test/Dialect/KTDFArch/intrinsics.mlir
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,10 @@ ktdf_arch.device @my_device attributes {version = 1} {
kind = "CPU",
ktdf_arch.features = {
ktdf_arch.feature.compute,
ktdf_arch.feature.simd = { lanes = #ktdf_arch.map<f32 = 4, f16 = 8> }
ktdf_arch.feature.simd = {
lanes = #ktdf_arch.map<f32 = 4, f16 = 8>,
sub_simd_lanes = #ktdf_arch.map<f32 = 2, f16 = 4>
}
}
}
%ls = exec_unit {
Expand Down
24 changes: 24 additions & 0 deletions unittest/Dialect/KTDFArch/Features.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -154,6 +154,30 @@ TEST_CASE("mlir::ktdf_arch::feature::SIMD") {
"{ lanes = #ktdf_arch.map<f16 = 4> }",
"{ lanes = #ktdf_arch.map<f16 = 2> }"));
}

SUBCASE("sub_simd_lanes") {
CHECK(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<> }", "{ }"));
CHECK(testFeature<feature::SIMD>(
&context, "{ }",
"{ sub_simd_lanes = #ktdf_arch.map<> }"));
CHECK(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<> }",
"{ sub_simd_lanes = #ktdf_arch.map<> }"));

CHECK_FALSE(testFeature<feature::SIMD>(
&context, "{ }", "{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));
CHECK_FALSE(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<f16 = 1> }",
"{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));
CHECK_FALSE(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<f32 = 1> }",
"{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));

CHECK(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<f16 = 4> }",
"{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));
}
}

TEST_CASE("mlir::ktdf_arch::feature::Queue") {
Expand Down
Loading