================
@@ -811,6 +811,53 @@ static mlir::Value emitX86MaskedLoad(CIRGenBuilderTy
&builder,
return builder.createMaskedLoad(loc, ty, ptr, alignment, maskVec, ops[1]);
}
+static mlir::Value emitX86VPerm2f128(CIRGenBuilderTy &builder,
+ mlir::Location loc, mlir::Type resType,
+ llvm::SmallVector<mlir::Value> ops) {
+ auto inputType = cast<cir::VectorType>(ops[0].getType());
+ const unsigned imm = CIRGenFunction::getZExtIntValueFromConstOp(ops[2]);
+ const uint8_t zeroMask = 0x08, controlMask = 0x0F;
+
+ // Mirror hardware and OGCG behaviour returning a zero vector
+ if ((imm & zeroMask) && (imm & zeroMask << 4))
+ return builder.getZero(loc, resType);
+
+ mlir::Value lanes[2];
+ llvm::SmallVector<int64_t, 64> mask;
+ const unsigned numElts = inputType.getSize();
+
+ // We must evaluated each lane(128 bits) separetely
+ for (auto lane : llvm::seq(0, 2)) {
+ uint8_t controlBits = (imm >> (lane * 4)) & controlMask;
----------------
lucaslive974 wrote:
I modified the bit checking logic to mirror original codegen. I didn't thought
about the futures changes on original codegen that would be harder to synch.
https://github.com/llvm/llvm-project/pull/208851
_______________________________________________
cfe-commits mailing list
[email protected]
https://lists.llvm.org/cgi-bin/mailman/listinfo/cfe-commits