Skip to content
Merged
Show file tree
Hide file tree
Changes from 5 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
187 changes: 187 additions & 0 deletions lib/Conversion/LlvmToNeura/LlvmToNeuraPass.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -290,6 +290,45 @@ struct LlvmFMulAddToNeuraFMulFAdd : public OpRewritePattern<mlir::LLVM::FMulAddO
}
};

// Handles LLVM intrinsic memset operations
struct LlvmMemsetToNeuraOps : public OpRewritePattern<LLVM::MemsetOp> {
using OpRewritePattern::OpRewritePattern;

LogicalResult matchAndRewrite(LLVM::MemsetOp op,
PatternRewriter &rewriter) const override {
// Get the operands: dest, value, len, isvolatile
auto dest = op.getDst();
auto value = op.getVal();
// Note: len and isvolatile are not used in this simplified implementation
// but are kept for future enhancement
(void)op.getLen(); // len - not used in current implementation
(void)op.getIsVolatile(); // isvolatile - not used in current implementation

// For CGRA, we can implement memset as a simple store operation
// Create a constant for the value to set (use the actual value if available)
Attribute valueAttr;
Comment thread
tancheng marked this conversation as resolved.
Outdated
Comment thread
tancheng marked this conversation as resolved.
Outdated
if (auto constOp = value.getDefiningOp<neura::ConstantOp>()) {
valueAttr = constOp->getAttr("value");
} else {
// Default to 0 if we can't determine the value
valueAttr = rewriter.getI8IntegerAttr(0);
}

auto valueConst = rewriter.create<neura::ConstantOp>(
op.getLoc(), rewriter.getI8Type(), valueAttr);

// Create a store operation to set the memory
// Note: This is a simplified implementation
// In a real implementation, we might need to handle the length parameter
Comment thread
tancheng marked this conversation as resolved.
Outdated
auto storeOp = rewriter.create<neura::StoreOp>(
op.getLoc(), valueConst, dest);

// Replace the memset with the store operation
rewriter.replaceOp(op, storeOp->getResults());
return success();
}
};

struct LlvmVFMulToNeuraVFMul : public OpRewritePattern<mlir::LLVM::FMulOp> {
using OpRewritePattern::OpRewritePattern;

Expand Down Expand Up @@ -529,6 +568,98 @@ struct LlvmReturnToNeuraReturn : public OpRewritePattern<LLVM::ReturnOp> {
}
};

struct LlvmFNegToNeuraFSub : public OpRewritePattern<LLVM::FNegOp> {
using OpRewritePattern::OpRewritePattern;

LogicalResult matchAndRewrite(LLVM::FNegOp op,
PatternRewriter &rewriter) const override {
// Gets operand.
Value operand = op.getOperand();
Type result_type = op.getType();
Location loc = op.getLoc();

// Creates constant 0.0.
auto zero_attr = rewriter.getFloatAttr(result_type, 0.0);
auto zero_const = rewriter.create<neura::ConstantOp>(loc, result_type, zero_attr);

// Replaces with 0.0 - operand.
rewriter.replaceOpWithNewOp<neura::FSubOp>(op, result_type, zero_const, operand);
Comment thread
tancheng marked this conversation as resolved.
Outdated
return success();
}
};
Comment thread
tancheng marked this conversation as resolved.
Outdated

struct LlvmSubToNeuraSub : public OpRewritePattern<LLVM::SubOp> {
using OpRewritePattern::OpRewritePattern;

LogicalResult matchAndRewrite(LLVM::SubOp op,
PatternRewriter &rewriter) const override {
// Gets operands.
Value lhs = op.getLhs();
Value rhs = op.getRhs();
Type result_type = op.getType();

// Replaces with neura.sub.
rewriter.replaceOpWithNewOp<neura::SubOp>(op, result_type, lhs, rhs);
return success();
}
};

// TODO: Implement LlvmXOrToNeuraOr - used in ADPCM coder and FFT kernels
// llvm.xor operations appear in:
// - adpcm_coder-kernel.mlir (line 104: %87 = llvm.xor %29, %19 : i1)
// - fft_kernel.mlir (line 19: %11 = llvm.xor %10, %3 : i32)
// Implementation: xor(a, b) = or(a, b) for boolean values

// TODO: Implement LlvmAndToNeuraMul - used in ADPCM coder and MVT kernels
// llvm.and operations appear in:
// - adpcm_coder-kernel.mlir (lines 55, 94: bitwise AND operations)
// - mvt-kernel.mlir (lines 44, 47, 50, 53: vector and scalar AND operations)
// Implementation: and(a, b) = mul(a, b) for boolean values

// TODO: Implement LlvmAllocaToNeuraOps - used in DTW kernel
// llvm.alloca operations appear in:
// - dtw-kernel-O0.mlir (lines 19-23: multiple stack allocations)
// Implementation: For CGRA, erase alloca or convert to register allocation

// TODO: Implement LlvmLShrToNeuraShl - used in ADPCM coder/decoder and FFT kernels
// llvm.lshr operations appear in:
// - adpcm_coder-kernel.mlir (line 54: %42 = llvm.lshr %40, %7 : i32)
// - adpcm_decoder-kernel.ll (line 35: %30 = lshr i32 %29, 4)
// - fft_kernel.mlir (line 67: %49 = llvm.lshr %7, %1 : i32)
// Implementation: Need proper logical right shift (lshr(x,n) != shl(x,-n))

// TODO: Implement LlvmAShrToNeuraAShr - used in ADPCM coder/decoder kernels
// llvm.ashr operations appear in:
// - adpcm_coder-kernel.mlir (lines 57, 63, 70: multiple ashr operations)
// - adpcm_decoder-kernel.ll (lines 49, 56, 61: ashr i32 %20, 3/1/2)
// Implementation: Need proper arithmetic right shift (preserves sign bit)


struct LlvmSMaxToNeuraSMax : public OpRewritePattern<LLVM::SMaxOp> {
using OpRewritePattern::OpRewritePattern;

LogicalResult matchAndRewrite(LLVM::SMaxOp op,
PatternRewriter &rewriter) const override {
// Gets operands.
Value lhs = op.getA();
Value rhs = op.getB();
Type result_type = op.getType();
Location loc = op.getLoc();

// Implements smax(a, b) = a >= b ? a : b
auto cmp = rewriter.create<neura::ICmpOp>(loc, rewriter.getI1Type(),
lhs, rhs,
rewriter.getStringAttr("sge"));

// Select: a >= b ? a : b
rewriter.replaceOpWithNewOp<neura::SelOp>(op, result_type, cmp, lhs, rhs);
return success();
}
};

// TODO: Implement LlvmAbsToNeuraAbs - used in ADPCM coder kernel
// llvm.intr.abs operations appear in adpcm_coder-kernel.mlir
// Implementation: abs(x) = x >= 0 ? x : -x (using ICmpOp + SelOp)

struct FuncReturnToNeuraReturn : public OpRewritePattern<func::ReturnOp> {
using OpRewritePattern::OpRewritePattern;
Expand Down Expand Up @@ -602,6 +733,46 @@ struct LlvmZExtToNeuraZExt : public OpRewritePattern<LLVM::ZExtOp> {
}
};

struct LlvmTruncToNeuraCast : public OpRewritePattern<LLVM::TruncOp> {
using OpRewritePattern<LLVM::TruncOp>::OpRewritePattern;

LogicalResult matchAndRewrite(LLVM::TruncOp op,
PatternRewriter &rewriter) const override {
// Trunc is a simple cast operation
auto result = rewriter.create<neura::CastOp>(
op.getLoc(), op.getType(), op.getArg(),
rewriter.getStringAttr("trunc"));
rewriter.replaceOp(op, result.getResult());
return success();
}
};

struct LlvmUDivToNeuraDiv : public OpRewritePattern<LLVM::UDivOp> {
using OpRewritePattern<LLVM::UDivOp>::OpRewritePattern;

LogicalResult matchAndRewrite(LLVM::UDivOp op,
PatternRewriter &rewriter) const override {
// UDiv is unsigned division
auto result = rewriter.create<neura::DivOp>(
op.getLoc(), op.getType(), op.getLhs(), op.getRhs());
rewriter.replaceOp(op, result.getResult());
return success();
}
};

struct LlvmURemToNeuraRem : public OpRewritePattern<LLVM::URemOp> {
using OpRewritePattern<LLVM::URemOp>::OpRewritePattern;

LogicalResult matchAndRewrite(LLVM::URemOp op,
PatternRewriter &rewriter) const override {
// URem is unsigned remainder
auto result = rewriter.create<neura::RemOp>(
op.getLoc(), op.getType(), op.getLhs(), op.getRhs());
rewriter.replaceOp(op, result.getResult());
return success();
}
};

struct LlvmMulToNeuraMul : public OpRewritePattern<LLVM::MulOp> {
using OpRewritePattern::OpRewritePattern;

Expand Down Expand Up @@ -783,6 +954,22 @@ struct LowerLlvmToNeuraPass
patterns.add<LlvmFPToSIToNeuraCast>(&getContext());
patterns.add<LlvmFMulAddToNeuraFMulFAdd>(&getContext());
patterns.add<LlvmSelectToNeuraSel>(&getContext());
patterns.add<LlvmMemsetToNeuraOps>(&getContext());
patterns.add<LlvmFNegToNeuraFSub>(&getContext());
patterns.add<LlvmSubToNeuraSub>(&getContext());
patterns.add<LlvmTruncToNeuraCast>(&getContext());
patterns.add<LlvmUDivToNeuraDiv>(&getContext());
patterns.add<LlvmURemToNeuraRem>(&getContext());
patterns.add<LlvmSMaxToNeuraSMax>(&getContext());
// TODO: Add more LLVM to Neura conversion patterns as needed
// patterns.add<LlvmXOrToNeuraOr>(&getContext()); // TODO: Used in ADPCM coder + FFT kernels
// patterns.add<LlvmAndToNeuraMul>(&getContext()); // TODO: Used in ADPCM coder + MVT kernels
// patterns.add<LlvmAllocaToNeuraOps>(&getContext()); // TODO: Used in DTW kernel
// TODO: Fix right shift implementations - current implementations are incorrect
// patterns.add<LlvmLShrToNeuraShl>(&getContext()); // TODO: Used in ADPCM coder/decoder + FFT kernels
// patterns.add<LlvmAShrToNeuraAShr>(&getContext()); // TODO: Used in ADPCM coder/decoder kernels
// patterns.add<LlvmAbsToNeuraAbs>(&getContext()); // TODO: Used in ADPCM coder kernel


FrozenRewritePatternSet frozen(std::move(patterns));

Expand Down
128 changes: 128 additions & 0 deletions test/e2e/bicg/bicg_kernel.mlir
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
// Compile the C kernel to LLVM IR (let clang handle headers and macros).
// Use -I %S so local headers (bicg.h, polybench.h) are visible.
// RUN: clang -S -emit-llvm -O3 -fno-vectorize -fno-unroll-loops -std=c11 \
// RUN: -I %S/../../benchmark/CGRA-Bench/kernels/bicg -DSMALL_DATASET \
// RUN: -o %t-kernel-full.ll %S/../../benchmark/CGRA-Bench/kernels/bicg/bicg.c

// Extract only the kernel function(s). PolyBench typically uses kernel_bicg,
// so a regex keeps this robust across name variants.
// RUN: llvm-extract --rfunc=".*kernel.*" %t-kernel-full.ll -o %t-kernel-only.ll

// Import the LLVM IR into MLIR (LLVM dialect).
// RUN: mlir-translate --import-llvm %t-kernel-only.ll -o %t-kernel.mlir

// Lower and map to the Neura accelerator, then generate code.
// Exact mapping (tiles, II, etc.) depends on the architecture/heuristics,
// so checks below focus on structural properties for stability.
// RUN: mlir-neura-opt %t-kernel.mlir \
// RUN: --assign-accelerator \
// RUN: --lower-llvm-to-neura \
// RUN: --promote-func-arg-to-const \
// RUN: --fold-constant \
// RUN: -o %t-before-canonicalize.mlir

// Check the IR before canonicalize-live-in to verify constant folding behavior
// RUN: FileCheck %s --input-file=%t-before-canonicalize.mlir -check-prefix=BEFORE_CANONICALIZE
// BEFORE_CANONICALIZE: module attributes
// BEFORE_CANONICALIZE: func.func @kernel
// BEFORE_CANONICALIZE: %0 = "neura.constant"() <{value = "%arg0"}> : () -> i32
// BEFORE_CANONICALIZE: %1 = "neura.constant"() <{value = "%arg1"}> : () -> i32
// BEFORE_CANONICALIZE: %2 = "neura.constant"() <{value = 0 : i64}> : () -> i64
// BEFORE_CANONICALIZE: %3 = "neura.icmp"(%0) <{cmpType = "sgt"}> {rhs_value = 0 : i32} : (i32) -> i1
// BEFORE_CANONICALIZE: neura.cond_br %3 : i1 then to ^bb1 else to ^bb2
// BEFORE_CANONICALIZE: ^bb1: // pred: ^bb0
// BEFORE_CANONICALIZE: %4 = "neura.constant"() <{value = 0 : i8}> : () -> i8
// BEFORE_CANONICALIZE: "neura.store"(%4) {rhs_value = "%arg3"} : (i8) -> ()
// BEFORE_CANONICALIZE: %5 = "neura.icmp"(%1) <{cmpType = "sgt"}> {rhs_value = 0 : i32} : (i32) -> i1
// BEFORE_CANONICALIZE: neura.cond_br %5 : i1 then to ^bb4 else to ^bb8
// BEFORE_CANONICALIZE: ^bb2: // pred: ^bb0
// BEFORE_CANONICALIZE: %6 = "neura.icmp"(%1) <{cmpType = "sgt"}> {rhs_value = 0 : i32} : (i32) -> i1
// BEFORE_CANONICALIZE: neura.cond_br %6 : i1 then to ^bb3 else to ^bb8
// BEFORE_CANONICALIZE: ^bb3: // pred: ^bb2
// BEFORE_CANONICALIZE: %7 = "neura.constant"() <{value = 0 : i8}> : () -> i8
// BEFORE_CANONICALIZE: "neura.store"(%7) {rhs_value = "%arg4"} : (i8) -> ()
// BEFORE_CANONICALIZE: neura.br to ^bb8
// BEFORE_CANONICALIZE: ^bb4: // pred: ^bb1
// BEFORE_CANONICALIZE: %8 = neura.zext %1 : i32 -> i64
// BEFORE_CANONICALIZE: %9 = neura.zext %0 : i32 -> i64
// BEFORE_CANONICALIZE: neura.br %2 : i64 to ^bb5
// BEFORE_CANONICALIZE: ^bb5(%10: i64): // 2 preds: ^bb4, ^bb7
// BEFORE_CANONICALIZE: %11 = "neura.gep"(%10) <{operandSegmentSizes = array<i32: 0, 1>}> {lhs_value = "%arg4"} : (i64) -> !llvm.ptr
// BEFORE_CANONICALIZE: "neura.store"(%11) {lhs_value = 0.000000e+00 : f64} : (!llvm.ptr) -> ()
// BEFORE_CANONICALIZE: %12 = "neura.gep"(%10) <{operandSegmentSizes = array<i32: 0, 1>}> {lhs_value = "%arg6"} : (i64) -> !llvm.ptr
// BEFORE_CANONICALIZE: neura.br %2 : i64 to ^bb6
// BEFORE_CANONICALIZE: ^bb6(%13: i64): // 2 preds: ^bb5, ^bb6
// BEFORE_CANONICALIZE: %14 = "neura.gep"(%13) <{operandSegmentSizes = array<i32: 0, 1>}> {lhs_value = "%arg3"} : (i64) -> !llvm.ptr
// BEFORE_CANONICALIZE: %15 = "neura.load"(%14) : (!llvm.ptr) -> f64
// BEFORE_CANONICALIZE: %16 = "neura.load"(%12) : (!llvm.ptr) -> f64
// BEFORE_CANONICALIZE: %17 = "neura.gep"(%10, %13) <{operandSegmentSizes = array<i32: 0, 2>}> {lhs_value = "%arg2"} : (i64, i64) -> !llvm.ptr
// BEFORE_CANONICALIZE: %18 = "neura.load"(%17) : (!llvm.ptr) -> f64
// BEFORE_CANONICALIZE: %19 = "neura.fmul_fadd"(%16, %18, %15) : (f64, f64, f64) -> f64
// BEFORE_CANONICALIZE: "neura.store"(%19, %14) : (f64, !llvm.ptr) -> ()
// BEFORE_CANONICALIZE: %20 = "neura.load"(%11) : (!llvm.ptr) -> f64
// BEFORE_CANONICALIZE: %21 = "neura.load"(%17) : (!llvm.ptr) -> f64
// BEFORE_CANONICALIZE: %22 = "neura.gep"(%13) <{operandSegmentSizes = array<i32: 0, 1>}> {lhs_value = "%arg5"} : (i64) -> !llvm.ptr
// BEFORE_CANONICALIZE: %23 = "neura.load"(%22) : (!llvm.ptr) -> f64
// BEFORE_CANONICALIZE: %24 = "neura.fmul_fadd"(%21, %23, %20) : (f64, f64, f64) -> f64
// BEFORE_CANONICALIZE: "neura.store"(%24, %11) : (f64, !llvm.ptr) -> ()
// BEFORE_CANONICALIZE: %25 = "neura.add"(%13) {rhs_value = 1 : i64} : (i64) -> i64
// BEFORE_CANONICALIZE: %26 = "neura.icmp"(%25, %9) <{cmpType = "eq"}> : (i64, i64) -> i1
// BEFORE_CANONICALIZE: neura.cond_br %26 : i1 then to ^bb7 else %25 : i64 to ^bb6
// BEFORE_CANONICALIZE: ^bb7: // pred: ^bb6
// BEFORE_CANONICALIZE: %27 = "neura.add"(%10) {rhs_value = 1 : i64} : (i64) -> i64
// BEFORE_CANONICALIZE: %28 = "neura.icmp"(%27, %8) <{cmpType = "eq"}> : (i64, i64) -> i1
// BEFORE_CANONICALIZE: neura.cond_br %28 : i1 then to ^bb8 else %27 : i64 to ^bb5
// BEFORE_CANONICALIZE: ^bb8: // 4 preds: ^bb1, ^bb2, ^bb3, ^bb7
// BEFORE_CANONICALIZE: "neura.return"() : () -> ()
// BEFORE_CANONICALIZE: }
// BEFORE_CANONICALIZE: }

// RUN: mlir-neura-opt %t-kernel.mlir \
// RUN: --assign-accelerator \
// RUN: --lower-llvm-to-neura \
// RUN: --promote-func-arg-to-const \
// RUN: --fold-constant \
// RUN: --canonicalize-live-in \
// RUN: --leverage-predicated-value \
// RUN: --transform-ctrl-to-data-flow \
// RUN: --fold-constant \
// RUN: --insert-data-mov \
// RUN: --map-to-accelerator="mapping-strategy=heuristic" \
// RUN: --architecture-spec=%S/../../arch_spec/architecture.yaml \
// RUN: --generate-code -o %t-mapping.mlir
Comment thread
tancheng marked this conversation as resolved.

Comment thread
tancheng marked this conversation as resolved.
// Sanity-check the mapped MLIR contains a module/func and neura ops.
// RUN: FileCheck %s --input-file=%t-mapping.mlir -check-prefix=MAPPING
// MAPPING: module
// MAPPING: func.func
Comment thread
tancheng marked this conversation as resolved.
Outdated
// MAPPING: neura.
// MAPPING: neura.return

// Verify the generated YAML/ASM artifacts look well-formed.
// RUN: FileCheck %s --input-file=tmp-generated-instructions.yaml --check-prefix=YAML
// YAML: array_config:
// YAML-NEXT: columns: 4
// YAML-NEXT: rows: 4
// YAML-NEXT: compiled_ii: 12
// YAML-NEXT: cores:
// YAML-NEXT: - column: 0
// YAML-NEXT: row: 0
// YAML-NEXT: core_id: "0"
// YAML-NEXT: entries:
// YAML-NEXT: - entry_id: "entry0"
// YAML-NEXT: instructions:
// YAML-NEXT: - timestep: 0
// YAML-NEXT: operations:
// YAML-NEXT: - opcode: "CONSTANT"

// RUN: FileCheck %s --input-file=tmp-generated-instructions.asm --check-prefix=ASM
// ASM: PE(0,0):
// ASM-NEXT: {
// ASM-NEXT: CONSTANT, [#0] -> [EAST, RED]
// ASM-NEXT: } (t=0)
// ASM-NEXT: {
// ASM-NEXT: GRANT_ONCE, [] -> [EAST, RED]
// ASM-NEXT: } (t=2)
// ASM-NEXT: {
// ASM-NEXT: GRANT_ONCE, [#0] -> [NORTH, RED]
// ASM-NEXT: } (t=3)
Loading