Skip to content
Open
Show file tree
Hide file tree
Changes from 3 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 29 additions & 1 deletion include/dataflow-scheduler/Dialect/KTDF/KTDF.td
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ include "mlir/Interfaces/LoopLikeInterface.td"
include "mlir/Interfaces/SideEffectInterfaces.td"
include "mlir/IR/OpAsmInterface.td"

include "dataflow-scheduler/Dialect/KTDF/KTDFAttributes.td"
include "dataflow-scheduler/Dialect/KTDF/KTDFTypes.td"

// Base class for KTDF dialect operations
Expand Down Expand Up @@ -168,6 +169,19 @@ def KTDF_ReadFromFifoOp : KTDF_Op<"read_from_fifo", [
limited to the op's parent region scope, cannot leave that scope (eg via
yield) and does not alias any other memrefs.

## Splat mode

The optional `splat` attribute records that the load unit left a splat half
Comment thread
msdataei marked this conversation as resolved.
Outdated
done and names the shuffle that finishes it. When a load unit's smallest
access granularity is wider than the source element, the hardware loads the
whole granule and splats that; one element per sub-SIMD group therefore
arrives over the FIFO rather than one for the whole vector.

The scheduler sets the attribute on every `ktdf.read_from_fifo` that reads
such a transfer. The lowering to DFIR emits the named shuffle immediately
after the receive. Any device pattern that re-reads the slot as a buffer
must preserve the attribute, because the shuffle is still owed.

Example:
```mlir
// Read from a FIFO slot into a tensor
Expand All @@ -182,11 +196,17 @@ def KTDF_ReadFromFifoOp : KTDF_Op<"read_from_fifo", [
%size = arith.constant 8 : index
%dyn_slot = ktdf.fifo.allocate(%size) -> !ktdf.fifo.slot<"SFU" -> "L1SU", ?xf16>
%dyn_data = ktdf.read_from_fifo %dyn_slot : !ktdf.fifo.slot<"SFU" -> "L1SU", ?xf16> -> tensor<?xf16>

// Read a splat the load unit left one element per sub-SIMD group wide
%splat = ktdf.read_from_fifo %slot
{splat = #ktdf.splat<first_subsimd_lane_to_all_subsimd_lanes>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> memref<64xf16>
```
}];

let arguments = (ins
KTDF_FifoSlotType:$fifo_slot
KTDF_FifoSlotType:$fifo_slot,
OptionalAttr<KTDF_SplatModeAttr>:$splat
);

let results = (outs
Expand All @@ -199,6 +219,14 @@ def KTDF_ReadFromFifoOp : KTDF_Op<"read_from_fifo", [

let hasVerifier = 1;

let builders = [
// A read of a slot that owes no shuffle, which is every read but the one a
// half-done splat feeds.
Comment thread
msdataei marked this conversation as resolved.
Outdated
OpBuilder<(ins "::mlir::Type":$result, "::mlir::Value":$fifo_slot), [{
build($_builder, $_state, result, fifo_slot, /*splat=*/nullptr);
}]>
];

let extraClassDeclaration = [{
/// Returns the element type of the FIFO slot
auto getFifoElementType() -> Type {
Expand Down
29 changes: 29 additions & 0 deletions include/dataflow-scheduler/Dialect/KTDF/KTDFAttributes.td
Original file line number Diff line number Diff line change
Expand Up @@ -66,4 +66,33 @@ def KTDF_LoopTypeAttr : EnumAttr<KTDF_Dialect, KTDF_LoopTypeEnum, "loop_type"> {
let assemblyFormat = "`<` $value `>`";
}

//===----------------------------------------------------------------------===//
// SplatModeAttr
//===----------------------------------------------------------------------===//

def KTDF_SplatModeAttr : EnumAttr<KTDF_Dialect, KTDF_SplatModeEnum, "splat"> {
let summary = "KTDF splat mode attribute naming the shuffle a splat still needs";
Comment thread
msdataei marked this conversation as resolved.
Outdated
let description = [{
Records that a load unit left a splat half done and names the shuffle that
finishes it. When a load unit's smallest access granularity is wider than
the source element, the hardware loads the whole granule and splats that;
one element per sub-SIMD group therefore arrives over the FIFO rather than
one for the whole vector.
Comment thread
msdataei marked this conversation as resolved.
Outdated

`first_subsimd_lane_to_all_subsimd_lanes` fills every lane of each sub-SIMD
group with the value held by its first lane. The group width is not encoded
here; it is read from the compute unit's
`ktdf_arch.feature.simd = { sub_simd_lanes }`.

Example:

```mlir
%d = ktdf.read_from_fifo %slot
{splat = #ktdf.splat<first_subsimd_lane_to_all_subsimd_lanes>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> memref<64xf16>
```
}];
let assemblyFormat = "`<` $value `>`";
}

#endif // DATAFLOW_SCHEDULER_DIALECT_KTDF_KTDFATTRIBUTES_TD
14 changes: 14 additions & 0 deletions include/dataflow-scheduler/Dialect/KTDF/KTDFEnums.td
Original file line number Diff line number Diff line change
Expand Up @@ -39,4 +39,18 @@ def KTDF_LoopTypeEnum : I32EnumAttr<"LoopType",
let genSpecializedAttr = 0;
}

//===----------------------------------------------------------------------===//
// Splat Mode Enum
//===----------------------------------------------------------------------===//

def KTDF_SplatModeEnum : I32EnumAttr<"SplatMode",
"Splat a load unit left half done, named by the shuffle that finishes it",
Comment thread
msdataei marked this conversation as resolved.
Outdated
[
I32EnumAttrCase<"FirstSubSimdLaneToAllSubSimdLanes", 0,
"first_subsimd_lane_to_all_subsimd_lanes">
]> {
let cppNamespace = "::mlir::ktdf";
let genSpecializedAttr = 0;
}

#endif // DATAFLOW_SCHEDULER_DIALECT_KTDF_KTDFENUMS_TD
Original file line number Diff line number Diff line change
Expand Up @@ -363,6 +363,7 @@ struct SIMD : FeatureAttr<&KTDFArchDialect::getFeatureSIMDAttrName> {
static constexpr StringLiteral kSplatAttrName = "splat";
static constexpr StringLiteral kZeroPadAttrName = "zero_pad";
static constexpr StringLiteral kLanesAttrName = "lanes";
static constexpr StringLiteral kSubSimdLanesAttrName = "sub_simd_lanes";

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Keeping with the style of the existing attribute, this should be called sub_lanes I think.

There was also the open design question on whether SIMD lanes could be multidimensional, i.e., f16 = 64x64 to indicate matrix-type accelerators. I don't think this is quite the way to go, but it would be good to know whether sub SIMD lanes would be any different in that regard.


using LanesAttr = TypedMapAttr<TypeAttr, I64Attr>;

Expand Down Expand Up @@ -395,6 +396,21 @@ struct SIMD : FeatureAttr<&KTDFArchDialect::getFeatureSIMDAttrName> {
}
return 0;
}

/// Gets the width of a sub-SIMD group, in lanes, per scalar type. The lanes
/// of a group are what a shuffle mode naming sub-SIMD groups operates over.
[[nodiscard]] auto getSubSimdLanes() const -> LanesAttr {
return getAttr<LanesAttr>(kSubSimdLanesAttrName);
}
/// Gets the width of a sub-SIMD group, in lanes, for @p scalar_type .
///
/// @returns Group width for @p scalar_type , or 0 if not declared.
[[nodiscard]] auto getSubSimdLanes(Type scalar_type) const -> int64_t {
if (const auto lanes = getSubSimdLanes(); lanes) {
return lanes.getValue(TypeAttr::get(scalar_type)).value_or(0);
}
return 0;
}
};

// Indicates that the execution unit can perform store operations.
Expand Down
22 changes: 22 additions & 0 deletions lib/Dialect/KTDFArch/KTDFArchIntrinsics.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -515,6 +515,12 @@ auto feature::SIMD::verify(EmitErrorFn emit_error) const -> LogicalResult {
<< "' requires map from type to 64-bit integer";
}

const auto sub_simd_lanes = getAttr(kSubSimdLanesAttrName);
if (sub_simd_lanes && !isa<LanesAttr>(sub_simd_lanes)) {
return emit_error() << "attribute '" << kSubSimdLanesAttrName
<< "' requires map from type to 64-bit integer";
}

return success();
}

Expand Down Expand Up @@ -551,6 +557,22 @@ auto feature::SIMD::test(feature::SIMD requirements) const -> bool {
}
}

if (const auto required = requirements.getSubSimdLanes(); required) {
const auto required_lanes = required.getEntries();
const auto provided_lanes = getSubSimdLanes();

if (!provided_lanes && !required_lanes.empty()) {
return false;
}

if (llvm::any_of(required_lanes, [&](const auto& require) -> bool {
return provided_lanes.getValue(require.first) <
require.second.getValue();
})) {
return false;
}
}

return true;
}

Expand Down
29 changes: 29 additions & 0 deletions test/Dialect/KTDF/read-from-fifo-op.mlir
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,35 @@ module {
// Use the buffer
"test.op"(%buf0) : (memref<64xf16>) -> ()

return
}
}

// The splat mode rides along on either result form: it says the load unit left
// one live element per sub-SIMD group, and names the shuffle that finishes it.

// CHECK-LABEL: func.func @read_from_fifo_splat() {
// CHECK-NEXT: %[[FIFO_0:.*]] = ktdf.fifo.allocate() -> !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16>
// CHECK-NEXT: %[[READ_FROM_FIFO_0:.*]] = ktdf.read_from_fifo %[[FIFO_0]] {splat = #ktdf.splat<first_subsimd_lane_to_all_subsimd_lanes>} : <"L1LU" -> "SFU", 64xf16> -> tensor<64xf16>
// CHECK-NEXT: %[[READ_FROM_FIFO_1:.*]] = ktdf.read_from_fifo %[[FIFO_0]] {splat = #ktdf.splat<first_subsimd_lane_to_all_subsimd_lanes>} : <"L1LU" -> "SFU", 64xf16> -> memref<64xf16>
// CHECK-NEXT: "test.op"(%[[READ_FROM_FIFO_0]], %[[READ_FROM_FIFO_1]]) : (tensor<64xf16>, memref<64xf16>) -> ()
// CHECK-NEXT: return
// CHECK-NEXT: }

module {
func.func @read_from_fifo_splat() {
%slot0 = ktdf.fifo.allocate() -> !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16>

%data0 = ktdf.read_from_fifo %slot0
{splat = #ktdf.splat<first_subsimd_lane_to_all_subsimd_lanes>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> tensor<64xf16>

%buf0 = ktdf.read_from_fifo %slot0
{splat = #ktdf.splat<first_subsimd_lane_to_all_subsimd_lanes>}
: !ktdf.fifo.slot<"L1LU" -> "SFU", 64xf16> -> memref<64xf16>

"test.op"(%data0, %buf0) : (tensor<64xf16>, memref<64xf16>) -> ()

return
}
}
7 changes: 7 additions & 0 deletions test/Dialect/KTDFArch/features-invalid.mlir
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,13 @@ ktdf_arch.device @simd_invalid_lanes {

// -----

ktdf_arch.device @simd_invalid_sub_simd_lanes {
// expected-error@+1 {{'sub_simd_lanes' requires map from type to 64-bit integer}}
exec_unit { ktdf_arch.features = { ktdf_arch.feature.simd = { sub_simd_lanes = 1 } } }
}

// -----

ktdf_arch.device @queue_not_on_link {
// expected-error@+1 {{only valid on links}}
exec_unit { ktdf_arch.features = { ktdf_arch.feature.queue = { size = "a" } } }
Expand Down
5 changes: 4 additions & 1 deletion test/Dialect/KTDFArch/intrinsics.mlir
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,10 @@ ktdf_arch.device @my_device attributes {version = 1} {
kind = "CPU",
ktdf_arch.features = {
ktdf_arch.feature.compute,
ktdf_arch.feature.simd = { lanes = #ktdf_arch.map<f32 = 4, f16 = 8> }
ktdf_arch.feature.simd = {
lanes = #ktdf_arch.map<f32 = 4, f16 = 8>,
sub_simd_lanes = #ktdf_arch.map<f32 = 2, f16 = 4>
}
}
}
%ls = exec_unit {
Expand Down
24 changes: 24 additions & 0 deletions unittest/Dialect/KTDFArch/Features.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -154,6 +154,30 @@ TEST_CASE("mlir::ktdf_arch::feature::SIMD") {
"{ lanes = #ktdf_arch.map<f16 = 4> }",
"{ lanes = #ktdf_arch.map<f16 = 2> }"));
}

SUBCASE("sub_simd_lanes") {
CHECK(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<> }", "{ }"));
CHECK(testFeature<feature::SIMD>(
&context, "{ }",
"{ sub_simd_lanes = #ktdf_arch.map<> }"));
CHECK(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<> }",
"{ sub_simd_lanes = #ktdf_arch.map<> }"));

CHECK_FALSE(testFeature<feature::SIMD>(
&context, "{ }", "{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));
CHECK_FALSE(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<f16 = 1> }",
"{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));
CHECK_FALSE(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<f32 = 1> }",
"{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));

CHECK(testFeature<feature::SIMD>(
&context, "{ sub_simd_lanes = #ktdf_arch.map<f16 = 4> }",
"{ sub_simd_lanes = #ktdf_arch.map<f16 = 2> }"));
}
}

TEST_CASE("mlir::ktdf_arch::feature::Queue") {
Expand Down
Loading