1. Jun 15, 2022
    • Nico Weber's avatar
      [gn build] (semi-automatically) port fb34d531 · 794d080e
      Nico Weber authored
      794d080e
    • Nico Weber's avatar
      [gn build] (semi-automatically) port 8bc0bb95 · 650c0b6e
      Nico Weber authored
      650c0b6e
    • Thomas Joerg's avatar
      Revert "Reland "[X86][RFC] Enable `_Float16` type support on X86 following the psABI"" · 37455b1f
      Thomas Joerg authored
      This reverts commit 6e02e275.
      
      This introduces a crash in the backend. Reproducer in MLIR's LLVM
      dialect follows. Let me know if you have trouble reproducing this.
      
      module {
        llvm.func @malloc(i64) -> !llvm.ptr<i8>
        llvm.func @_mlir_ciface_tf_report_error(!llvm.ptr<i8>, i32, !llvm.ptr<i8>)
        llvm.mlir.global internal constant @error_message_2208944672953921889("failed to allocate memory at loc(\22-\22:3:8)\00")
        llvm.func @_mlir_ciface_tf_alloc(!llvm.ptr<i8>, i64, i64, i32, i32, !llvm.ptr<i32>) -> !llvm.ptr<i8>
        llvm.func @Rsqrt_CPU_DT_HALF_DT_HALF(%arg0: !llvm.ptr<i8>, %arg1: i64, %arg2: !llvm.ptr<i8>) -> !llvm.struct<(i64, ptr<i8>)> attributes {llvm.emit_c_interface, tf_entry} {
          %0 = llvm.mlir.constant(8 : i32) : i32
          %1 = llvm.mlir.constant(8 : index) : i64
          %2 = llvm.mlir.constant(2 : index) : i64
          %3 = llvm.mlir.constant(dense<0.000000e+00> : vector<4xf16>) : vector<4xf16>
          %4 = llvm.mlir.constant(dense<[0, 1, 2, 3]> : vector<4xi32>) : vector<4xi32>
          %5 = llvm.mlir.constant(dense<1.000000e+00> : vector<4xf16>) : vector<4xf16>
          %6 = llvm.mlir.constant(false) : i1
          %7 = llvm.mlir.constant(1 : i32) : i32
          %8 = llvm.mlir.constant(0 : i32) : i32
          %9 = llvm.mlir.constant(4 : index) : i64
          %10 = llvm.mlir.constant(0 : index) : i64
          %11 = llvm.mlir.constant(1 : index) : i64
          %12 = llvm.mlir.constant(-1 : index) : i64
          %13 = llvm.mlir.null : !llvm.ptr<f16>
          %14 = llvm.getelementptr %13[%9] : (!llvm.ptr<f16>, i64) -> !llvm.ptr<f16>
          %15 = llvm.ptrtoint %14 : !llvm.ptr<f16> to i64
          %16 = llvm.alloca %15 x f16 {alignment = 32 : i64} : (i64) -> !llvm.ptr<f16>
          %17 = llvm.alloca %15 x f16 {alignment = 32 : i64} : (i64) -> !llvm.ptr<f16>
          %18 = llvm.mlir.null : !llvm.ptr<i64>
          %19 = llvm.getelementptr %18[%arg1] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          %20 = llvm.ptrtoint %19 : !llvm.ptr<i64> to i64
          %21 = llvm.alloca %20 x i64 : (i64) -> !llvm.ptr<i64>
          llvm.br ^bb1(%10 : i64)
        ^bb1(%22: i64):  // 2 preds: ^bb0, ^bb2
          %23 = llvm.icmp "slt" %22, %arg1 : i64
          llvm.cond_br %23, ^bb2, ^bb3
        ^bb2:  // pred: ^bb1
          %24 = llvm.bitcast %arg2 : !llvm.ptr<i8> to !llvm.ptr<struct<(ptr<f16>, ptr<f16>, i64)>>
          %25 = llvm.getelementptr %24[%10, 2] : (!llvm.ptr<struct<(ptr<f16>, ptr<f16>, i64)>>, i64) -> !llvm.ptr<i64>
          %26 = llvm.add %22, %11  : i64
          %27 = llvm.getelementptr %25[%26] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          %28 = llvm.load %27 : !llvm.ptr<i64>
          %29 = llvm.getelementptr %21[%22] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          llvm.store %28, %29 : !llvm.ptr<i64>
          llvm.br ^bb1(%26 : i64)
        ^bb3:  // pred: ^bb1
          llvm.br ^bb4(%10, %11 : i64, i64)
        ^bb4(%30: i64, %31: i64):  // 2 preds: ^bb3, ^bb5
          %32 = llvm.icmp "slt" %30, %arg1 : i64
          llvm.cond_br %32, ^bb5, ^bb6
        ^bb5:  // pred: ^bb4
          %33 = llvm.bitcast %arg2 : !llvm.ptr<i8> to !llvm.ptr<struct<(ptr<f16>, ptr<f16>, i64)>>
          %34 = llvm.getelementptr %33[%10, 2] : (!llvm.ptr<struct<(ptr<f16>, ptr<f16>, i64)>>, i64) -> !llvm.ptr<i64>
          %35 = llvm.add %30, %11  : i64
          %36 = llvm.getelementptr %34[%35] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          %37 = llvm.load %36 : !llvm.ptr<i64>
          %38 = llvm.mul %37, %31  : i64
          llvm.br ^bb4(%35, %38 : i64, i64)
        ^bb6:  // pred: ^bb4
          %39 = llvm.bitcast %arg2 : !llvm.ptr<i8> to !llvm.ptr<ptr<f16>>
          %40 = llvm.getelementptr %39[%11] : (!llvm.ptr<ptr<f16>>, i64) -> !llvm.ptr<ptr<f16>>
          %41 = llvm.load %40 : !llvm.ptr<ptr<f16>>
          %42 = llvm.getelementptr %13[%11] : (!llvm.ptr<f16>, i64) -> !llvm.ptr<f16>
          %43 = llvm.ptrtoint %42 : !llvm.ptr<f16> to i64
          %44 = llvm.alloca %7 x i32 : (i32) -> !llvm.ptr<i32>
          llvm.store %8, %44 : !llvm.ptr<i32>
          %45 = llvm.call @_mlir_ciface_tf_alloc(%arg0, %31, %43, %8, %7, %44) : (!llvm.ptr<i8>, i64, i64, i32, i32, !llvm.ptr<i32>) -> !llvm.ptr<i8>
          %46 = llvm.bitcast %45 : !llvm.ptr<i8> to !llvm.ptr<f16>
          %47 = llvm.icmp "eq" %31, %10 : i64
          %48 = llvm.or %6, %47  : i1
          %49 = llvm.mlir.null : !llvm.ptr<i8>
          %50 = llvm.icmp "ne" %45, %49 : !llvm.ptr<i8>
          %51 = llvm.or %50, %48  : i1
          llvm.cond_br %51, ^bb7, ^bb13
        ^bb7:  // pred: ^bb6
          %52 = llvm.urem %31, %9  : i64
          %53 = llvm.sub %31, %52  : i64
          llvm.br ^bb8(%10 : i64)
        ^bb8(%54: i64):  // 2 preds: ^bb7, ^bb9
          %55 = llvm.icmp "slt" %54, %53 : i64
          llvm.cond_br %55, ^bb9, ^bb10
        ^bb9:  // pred: ^bb8
          %56 = llvm.mul %54, %11  : i64
          %57 = llvm.add %56, %10  : i64
          %58 = llvm.add %57, %10  : i64
          %59 = llvm.getelementptr %41[%58] : (!llvm.ptr<f16>, i64) -> !llvm.ptr<f16>
          %60 = llvm.bitcast %59 : !llvm.ptr<f16> to !llvm.ptr<vector<4xf16>>
          %61 = llvm.load %60 {alignment = 2 : i64} : !llvm.ptr<vector<4xf16>>
          %62 = "llvm.intr.sqrt"(%61) : (vector<4xf16>) -> vector<4xf16>
          %63 = llvm.fdiv %5, %62  : vector<4xf16>
          %64 = llvm.getelementptr %46[%58] : (!llvm.ptr<f16>, i64) -> !llvm.ptr<f16>
          %65 = llvm.bitcast %64 : !llvm.ptr<f16> to !llvm.ptr<vector<4xf16>>
          llvm.store %63, %65 {alignment = 2 : i64} : !llvm.ptr<vector<4xf16>>
          %66 = llvm.add %54, %9  : i64
          llvm.br ^bb8(%66 : i64)
        ^bb10:  // pred: ^bb8
          %67 = llvm.icmp "ult" %53, %31 : i64
          llvm.cond_br %67, ^bb11, ^bb12
        ^bb11:  // pred: ^bb10
          %68 = llvm.mul %53, %12  : i64
          %69 = llvm.add %31, %68  : i64
          %70 = llvm.mul %53, %11  : i64
          %71 = llvm.add %70, %10  : i64
          %72 = llvm.trunc %69 : i64 to i32
          %73 = llvm.mlir.undef : vector<4xi32>
          %74 = llvm.insertelement %72, %73[%8 : i32] : vector<4xi32>
          %75 = llvm.shufflevector %74, %73 [0 : i32, 0 : i32, 0 : i32, 0 : i32] : vector<4xi32>, vector<4xi32>
          %76 = llvm.icmp "slt" %4, %75 : vector<4xi32>
          %77 = llvm.add %71, %10  : i64
          %78 = llvm.getelementptr %41[%77] : (!llvm.ptr<f16>, i64) -> !llvm.ptr<f16>
          %79 = llvm.bitcast %78 : !llvm.ptr<f16> to !llvm.ptr<vector<4xf16>>
          %80 = llvm.intr.masked.load %79, %76, %3 {alignment = 2 : i32} : (!llvm.ptr<vector<4xf16>>, vector<4xi1>, vector<4xf16>) -> vector<4xf16>
          %81 = llvm.bitcast %16 : !llvm.ptr<f16> to !llvm.ptr<vector<4xf16>>
          llvm.store %80, %81 : !llvm.ptr<vector<4xf16>>
          %82 = llvm.load %81 {alignment = 2 : i64} : !llvm.ptr<vector<4xf16>>
          %83 = "llvm.intr.sqrt"(%82) : (vector<4xf16>) -> vector<4xf16>
          %84 = llvm.fdiv %5, %83  : vector<4xf16>
          %85 = llvm.bitcast %17 : !llvm.ptr<f16> to !llvm.ptr<vector<4xf16>>
          llvm.store %84, %85 {alignment = 2 : i64} : !llvm.ptr<vector<4xf16>>
          %86 = llvm.load %85 : !llvm.ptr<vector<4xf16>>
          %87 = llvm.getelementptr %46[%77] : (!llvm.ptr<f16>, i64) -> !llvm.ptr<f16>
          %88 = llvm.bitcast %87 : !llvm.ptr<f16> to !llvm.ptr<vector<4xf16>>
          llvm.intr.masked.store %86, %88, %76 {alignment = 2 : i32} : vector<4xf16>, vector<4xi1> into !llvm.ptr<vector<4xf16>>
          llvm.br ^bb12
        ^bb12:  // 2 preds: ^bb10, ^bb11
          %89 = llvm.mul %2, %1  : i64
          %90 = llvm.mul %arg1, %2  : i64
          %91 = llvm.add %90, %11  : i64
          %92 = llvm.mul %91, %1  : i64
          %93 = llvm.add %89, %92  : i64
          %94 = llvm.alloca %93 x i8 : (i64) -> !llvm.ptr<i8>
          %95 = llvm.bitcast %94 : !llvm.ptr<i8> to !llvm.ptr<ptr<f16>>
          llvm.store %46, %95 : !llvm.ptr<ptr<f16>>
          %96 = llvm.getelementptr %95[%11] : (!llvm.ptr<ptr<f16>>, i64) -> !llvm.ptr<ptr<f16>>
          llvm.store %46, %96 : !llvm.ptr<ptr<f16>>
          %97 = llvm.getelementptr %95[%2] : (!llvm.ptr<ptr<f16>>, i64) -> !llvm.ptr<ptr<f16>>
          %98 = llvm.bitcast %97 : !llvm.ptr<ptr<f16>> to !llvm.ptr<i64>
          llvm.store %10, %98 : !llvm.ptr<i64>
          %99 = llvm.bitcast %94 : !llvm.ptr<i8> to !llvm.ptr<struct<(ptr<f16>, ptr<f16>, i64, i64)>>
          %100 = llvm.getelementptr %99[%10, 3] : (!llvm.ptr<struct<(ptr<f16>, ptr<f16>, i64, i64)>>, i64) -> !llvm.ptr<i64>
          %101 = llvm.getelementptr %100[%arg1] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          %102 = llvm.sub %arg1, %11  : i64
          llvm.br ^bb14(%102, %11 : i64, i64)
        ^bb13:  // pred: ^bb6
          %103 = llvm.mlir.addressof @error_message_2208944672953921889 : !llvm.ptr<array<42 x i8>>
          %104 = llvm.getelementptr %103[%10, %10] : (!llvm.ptr<array<42 x i8>>, i64, i64) -> !llvm.ptr<i8>
          llvm.call @_mlir_ciface_tf_report_error(%arg0, %0, %104) : (!llvm.ptr<i8>, i32, !llvm.ptr<i8>) -> ()
          %105 = llvm.mul %2, %1  : i64
          %106 = llvm.mul %2, %10  : i64
          %107 = llvm.add %106, %11  : i64
          %108 = llvm.mul %107, %1  : i64
          %109 = llvm.add %105, %108  : i64
          %110 = llvm.alloca %109 x i8 : (i64) -> !llvm.ptr<i8>
          %111 = llvm.bitcast %110 : !llvm.ptr<i8> to !llvm.ptr<ptr<f16>>
          llvm.store %13, %111 : !llvm.ptr<ptr<f16>>
          %112 = llvm.getelementptr %111[%11] : (!llvm.ptr<ptr<f16>>, i64) -> !llvm.ptr<ptr<f16>>
          llvm.store %13, %112 : !llvm.ptr<ptr<f16>>
          %113 = llvm.getelementptr %111[%2] : (!llvm.ptr<ptr<f16>>, i64) -> !llvm.ptr<ptr<f16>>
          %114 = llvm.bitcast %113 : !llvm.ptr<ptr<f16>> to !llvm.ptr<i64>
          llvm.store %10, %114 : !llvm.ptr<i64>
          %115 = llvm.call @malloc(%109) : (i64) -> !llvm.ptr<i8>
          "llvm.intr.memcpy"(%115, %110, %109, %6) : (!llvm.ptr<i8>, !llvm.ptr<i8>, i64, i1) -> ()
          %116 = llvm.mlir.undef : !llvm.struct<(i64, ptr<i8>)>
          %117 = llvm.insertvalue %10, %116[0] : !llvm.struct<(i64, ptr<i8>)>
          %118 = llvm.insertvalue %115, %117[1] : !llvm.struct<(i64, ptr<i8>)>
          llvm.return %118 : !llvm.struct<(i64, ptr<i8>)>
        ^bb14(%119: i64, %120: i64):  // 2 preds: ^bb12, ^bb15
          %121 = llvm.icmp "sge" %119, %10 : i64
          llvm.cond_br %121, ^bb15, ^bb16
        ^bb15:  // pred: ^bb14
          %122 = llvm.getelementptr %21[%119] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          %123 = llvm.load %122 : !llvm.ptr<i64>
          %124 = llvm.getelementptr %100[%119] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          llvm.store %123, %124 : !llvm.ptr<i64>
          %125 = llvm.getelementptr %101[%119] : (!llvm.ptr<i64>, i64) -> !llvm.ptr<i64>
          llvm.store %120, %125 : !llvm.ptr<i64>
          %126 = llvm.mul %120, %123  : i64
          %127 = llvm.sub %119, %11  : i64
          llvm.br ^bb14(%127, %126 : i64, i64)
        ^bb16:  // pred: ^bb14
          %128 = llvm.call @malloc(%93) : (i64) -> !llvm.ptr<i8>
          "llvm.intr.memcpy"(%128, %94, %93, %6) : (!llvm.ptr<i8>, !llvm.ptr<i8>, i64, i1) -> ()
          %129 = llvm.mlir.undef : !llvm.struct<(i64, ptr<i8>)>
          %130 = llvm.insertvalue %arg1, %129[0] : !llvm.struct<(i64, ptr<i8>)>
          %131 = llvm.insertvalue %128, %130[1] : !llvm.struct<(i64, ptr<i8>)>
          llvm.return %131 : !llvm.struct<(i64, ptr<i8>)>
        }
        llvm.func @_mlir_ciface_Rsqrt_CPU_DT_HALF_DT_HALF(%arg0: !llvm.ptr<struct<(i64, ptr<i8>)>>, %arg1: !llvm.ptr<i8>, %arg2: !llvm.ptr<struct<(i64, ptr<i8>)>>) attributes {llvm.emit_c_interface, tf_entry} {
          %0 = llvm.load %arg2 : !llvm.ptr<struct<(i64, ptr<i8>)>>
          %1 = llvm.extractvalue %0[0] : !llvm.struct<(i64, ptr<i8>)>
          %2 = llvm.extractvalue %0[1] : !llvm.struct<(i64, ptr<i8>)>
          %3 = llvm.call @Rsqrt_CPU_DT_HALF_DT_HALF(%arg1, %1, %2) : (!llvm.ptr<i8>, i64, !llvm.ptr<i8>) -> !llvm.struct<(i64, ptr<i8>)>
          llvm.store %3, %arg0 : !llvm.ptr<struct<(i64, ptr<i8>)>>
          llvm.return
        }
      }
      37455b1f
    • Nikita Popov's avatar
      [BitcodeReader] Remove unnecessary argument defaults (NFC) · 7cc4d449
      Nikita Popov authored
      This is an internal method that is always called with all arguments.
      7cc4d449
    • Simon Pilgrim's avatar
    • Simon Pilgrim's avatar
      [AArch64] Add test case from D127354 · adfcdb0d
      Simon Pilgrim authored
      adfcdb0d
    • Benjamin Kramer's avatar
      Add a conversion from double to bf16 · 8bc0bb95
      Benjamin Kramer authored
      This introduces a new compiler-rt function `__truncdfbf2`.
      8bc0bb95
    • Benjamin Kramer's avatar
      Promote bf16 to f32 when the target doesn't support it · fb34d531
      Benjamin Kramer authored
      This is modeled after the half-precision fp support. Two new nodes are
      introduced for casting from and to bf16. Since casting from bf16 is a
      simple operation I opted to always directly lower it to integer
      arithmetic. The other way round is more complicated if you want to
      preserve IEEE semantics, so it's handled by a new __truncsfbf2
      compiler-rt builtin.
      
      This is of course very bare bones, but sufficient to get a semi-softened
      fadd on x86.
      
      Possible future improvements:
       - Targets with bf16 conversion instructions can now make fp_to_bf16 legal
       - The software conversion to bf16 can be replaced by a trivial
         implementation under fast math.
      
      Differential Revision: https://reviews.llvm.org/D126953
      fb34d531
    • Simon Pilgrim's avatar
      Fix signed/unsigned comparison warning · 43e7ba64
      Simon Pilgrim authored
      43e7ba64
    • Keith Walker's avatar
      [DebugInfo][ARM] Not readonly check for RWPI globals · 94fac097
      Keith Walker authored
      When compiling for the RWPI relocation model [1], the debug information
      is wrong for readonly global variables.
      
      Writable global variables are accessed by the static base register (R9
      on ARM) in the RWPI relocation model.  This is being correctly generated
      
      Readonly global variables are not accessed by the static base register
      in the RWPI relocation model. This case is incorrectly generating the
      same debugging information as for writable global variables.
      
      References:
      [1] ARM Read-Write Position Independence: https://github.com/ARM-software/abi-aa/blob/main/aapcs32/aapcs32.rst#read-write-position-independence-rwpi
      
      Differential Revision: https://reviews.llvm.org/D126361
      94fac097
    • Benjamin Kramer's avatar
      170ca11a
    • Nabeel Omer's avatar
      [X86][SLP] Basic test coverage for llvm.powi · 245604a9
      Nabeel Omer authored
      This patch introduces basic test coverage for llvm.powi.* intrinsics.
      
      Differential Revision: https://reviews.llvm.org/D127492
      245604a9
    • David Sherwood's avatar
    • Simon Pilgrim's avatar
      [DAG] Fix SDLoc mismatch in (shl (srl x, c1), c2) -> and(shift(x,c3)) fold · f096d592
      Simon Pilgrim authored
      Noticed by @craig.topper on D125836 which uses a tweaked copy of the same code.
      
      Differential Revision: https://reviews.llvm.org/D127772
      f096d592
    • Stanislav Gatev's avatar
      [clang][dataflow] Add support for correlated branches to optional model · 8fcdd625
      Stanislav Gatev authored
      Add support for correlated branches to the std::optional dataflow model.
      
      Differential Revision: https://reviews.llvm.org/D125931
      
      Reviewed-by: ymandel, xazax.hun
      8fcdd625
    • Martin Boehme's avatar
      [clang] Reject non-declaration C++11 attributes on declarations · 8c7b64b5
      Martin Boehme authored
      For backwards compatiblity, we emit only a warning instead of an error if the
      attribute is one of the existing type attributes that we have historically
      allowed to "slide" to the `DeclSpec` just as if it had been specified in GNU
      syntax. (We will call these "legacy type attributes" below.)
      
      The high-level changes that achieve this are:
      
      - We introduce a new field `Declarator::DeclarationAttrs` (with appropriate
        accessors) to store C++11 attributes occurring in the attribute-specifier-seq
        at the beginning of a simple-declaration (and other similar declarations).
        Previously, these attributes were placed on the `DeclSpec`, which made it
        impossible to reconstruct later on whether the attributes had in fact been
        placed on the decl-specifier-seq or ahead of the declaration.
      
      - In the parser, we propgate declaration attributes and decl-specifier-seq
        attributes separately until we can place them in
        `Declarator::DeclarationAttrs` or `Dec...
      8c7b64b5
    • Sven van Haastregt's avatar
      [OpenCL] Reword unknown extension pragma diagnostic · 7acc88be
      Sven van Haastregt authored
      For newer OpenCL extensions that do not require a pragma, such as
      `cl_khr_subgroup_shuffle`, a user could still accidentally attempt to
      use a pragma.  This would result in a warning
        "unknown OpenCL extension 'cl_khr_subgroup_shuffle' - ignoring"
      which could be mistakenly interpreted as "clang does not support this
      extension at all" instead of "clang does not require any pragma for
      this extension".
      
      Differential Revision: https://reviews.llvm.org/D126660
      7acc88be
    • Simon Pilgrim's avatar
      [X86] needCarryOrOverflowFlag/onlyZeroFlagUsed - merge identical switch cases. NFCI. · 4fd56141
      Simon Pilgrim authored
      Makes it easier to grok and fixes various bugprone-branch-clone warnings.
      4fd56141
    • David Sherwood's avatar
      [AArch64][SME] Add SME read/write intrinsics that map to the mova instruction · 5fa2416e
      David Sherwood authored
      This patch adds implementations for the read/write SME ACLE intrinsics:
      
        @llvm.aarch64.sme.read.horiz
        @llvm.aarch64.sme.read.vert
        @llvm.aarch64.sme.write.horiz
        @llvm.aarch64.sme.write.vert
      
      These all map to the SME mova instruction.
      
      Differential Revision: https://reviews.llvm.org/D127414
      5fa2416e
    • Martin Boehme's avatar
    • Ilya Biryukov's avatar
      [libcxx] Fix allocator<void>::pointer in C++20 with removed members · 374f938f
      Ilya Biryukov authored
      When compiled with `-D_LIBCPP_ENABLE_CXX20_REMOVED_ALLOCATOR_MEMBERS`
      uses of `allocator<void>::pointer` resulted in compiler errors after D104323.
      If we instantiate the primary template, `allocator<void>::reference` produces
      an error 'cannot form references to void'.
      
      To workaround this, allow to bring back the `allocator<void>` specialization by defining the new `_LIBCPP_ENABLE_CXX20_REMOVED_ALLOCATOR_VOID_SPECIALIZATION` macro.
      
      To make sure the code that uses `allocator<void>` and the removed members does not break,
      both `_LIBCPP_ENABLE_CXX20_REMOVED_ALLOCATOR_MEMBERS` and `_LIBCPP_ENABLE_CXX20_REMOVED_ALLOCATOR_MEMBERS` have to be defined.
      
      Reviewed By: ldionne, #libc, philnik
      
      Differential Revision: https://reviews.llvm.org/D126210
      374f938f
    • Martin Boehme's avatar
    • David Sherwood's avatar
      [NFC][AArch64] Minor refactor of AArch64InstPrinter::printMatrixTileList · 8f9d73fb
      David Sherwood authored
      We can remove the MatrixZADRegisterTable table of tile registers and
      just calculate the register index directly.
      
      Differential Revision: https://reviews.llvm.org/D127757
      8f9d73fb
    • Kadir Cetinkaya's avatar
      [clangd] Enable AKA type printing by default · a67beef3
      Kadir Cetinkaya authored
      This has been tested on a large set of c++ developers for a long while,
      without any crashes or complaints.
      
      Differential Revision: https://reviews.llvm.org/D127833
      a67beef3
    • owenca's avatar
      462b49f1
    • Benjamin Kramer's avatar
      [mlir][Arith] Fix a use-after-free after rewriting ops to unsigned · 0886ea90
      Benjamin Kramer authored
      Just short-circuit when a change was made, the erased value is invalid
      after that. Found by asan.
      
      This pass looks like it could use rewrite patterns instead which don't
      have this issue, but let's fix the asan build first.
      0886ea90
    • Kito Cheng's avatar
      [RISCV] Fixing undefined physical register issue when subreg liveness tracking enabled. · 687e5661
      Kito Cheng authored
      RISC-V expand register tuple spilling into series of register spilling after
      register allocation phase by the pseudo instruction expansion, however part of
      register tuple might be still undefined during spilling, machine verifier will
      complain the spill instruction is using an undefined physical register.
      
      Optimal solution should be doing liveness analysis and do not emit spill
      and reload for those undefined parts, but accurate liveness info at that point
      is not so easy to get.
      
      So the suboptimal solution is still spill and reload those undefined parts, but
      adding implicit-use of super register to spill function, then machine
      verifier will only report report using undefined physical register if
      the when whole super register is undefined, and this behavior are also
      documented in MachineVerifier::checkLiveness[1].
      
      Example for demo what happend:
      
      ```
        v10m2 = xxx
        # v12m2 not define yet
        PseudoVSPILL2_M2 v10m2_v12m2
        ...
      ```
      
      After expansion:
      ```
        v10m2 = xxx
        # v12m2 not define yet
        # Expand PseudoVSPILL2_M2 v10m2_v12m2 to 2 vs2r
        VS2R_V v10m2
        VS2R_V v12m2 # Use undef reg!
      ```
      
      What this patch did:
      ```
        v10m2 = xxx
        # v12m2 not define yet
        # Expand PseudoVSPILL2_M2 v10m2_v12m2 to 2 vs2r
        VS2R_V v10m2 implicit v10m2_v12m2
        # Use undef reg (v12m2), but v10m2_v12m2 ins't totally undef, so
        # that's OK.
        VS2R_V v12m2 implicit v10m2_v12m2
      ```
      
      [1] https://github.com/llvm-mirror/llvm/blob/master/lib/CodeGen/MachineVerifier.cpp#L2016-L2019
      
      Reviewed By: craig.topper
      
      Differential Revision: https://reviews.llvm.org/D127642
      687e5661
    • Matthias Springer's avatar
      [mlir][bufferize] Better implementation of AnalysisState::isTensorYielded · a36c801d
      Matthias Springer authored
      If `create-deallocs=0`, mark all bufferization.alloc_tensor ops as escaping. (Unless they already have an `escape` attribute.) In the absence of analysis information, check SSA use-def chains to see if the value may be yielded.
      
      Differential Revision: https://reviews.llvm.org/D127302
      a36c801d
    • Siva Chandra Reddy's avatar
      0f72a0d2
    • Matthias Springer's avatar
      [mlir][bufferize][NFC] Merge AlwaysCopyAnalysisState into AnalysisState · a3bca118
      Matthias Springer authored
      `AnalysisState` now has default implementations of all virtual functions.
      
      Differential Revision: https://reviews.llvm.org/D127301
      a3bca118
    • Heejin Ahn's avatar
      [InstCombine] Improve check for catchswitch BBs (NFC) · b2f4112f
      Heejin Ahn authored
      Reviewed By: nikic
      
      Differential Revision: https://reviews.llvm.org/D127810
      b2f4112f
    • Matthias Springer's avatar
      [mlir][bufferize][NFC] Make func BufferizableOpInterface impl compatible with One-Shot Bufferize · cd80617a
      Matthias Springer authored
      Bufferization of the func dialect must go through `OneShotModuleBufferize`. With this change, the analysis interface methods of the BufferizableOpInterface of func dialect ops can be used together with the normal `OneShotBufferize`. (In the absence of analysis information, they will return conservative results.)
      
      Differential Revision: https://reviews.llvm.org/D127299
      cd80617a
    • Peixin-Qiao's avatar
      [flang][OpenMP] Add one semantic check for data-sharing clauses · 9441003b
      Peixin-Qiao authored
      As OpenMP 5.0, for firstprivate, lastprivate, copyin, and copyprivate
      clauses, if the list item is a polymorphic variable with the allocatable
      attribute, the behavior is unspecified.
      
      Reviewed By: kiranchandramohan
      
      Differential Revision: https://reviews.llvm.org/D127601
      9441003b
    • Matthias Springer's avatar
      [mlir][linalg][bufferize] Remove always-aliasing-with-dest option · ad2e635f
      Matthias Springer authored
      This flag was introduced for a use case in IREE, but it is no longer needed.
      
      Differential Revision: https://reviews.llvm.org/D126965
      ad2e635f
    • Martin Boehme's avatar
      [Clang] Add the `annotate_type` attribute · 665da187
      Martin Boehme authored
      This is an analog to the `annotate` attribute but for types. The intent is to allow adding arbitrary annotations to types for use in static analysis tools.
      
      For details, see this RFC:
      
      https://discourse.llvm.org/t/rfc-new-attribute-annotate-type-iteration-2/61378
      
      Reviewed By: aaron.ballman
      
      Differential Revision: https://reviews.llvm.org/D111548
      665da187
    • Peixin-Qiao's avatar
      [flang] Change C889 from error into warning · 3151fb5e
      Peixin-Qiao authored
      This constraint is used in OMP2012 benchmark, and other compilers do not
      enforce it. Change it into one warning. This addresses the issue
      https://github.com/llvm/llvm-project/issues/56003.
      
      Reviewed By: klausler, kiranchandramohan
      
      Differential Revision: https://reviews.llvm.org/D127740
      3151fb5e
    • Nikita Popov's avatar
      [SimplifyLibCalls] Drop duplicate check (NFC) · 2dac2c4f
      Nikita Popov authored
      The same condition already exists inside optimizeMemCmpConstantSize().
      2dac2c4f
    • Austin Kerbow's avatar
      [AMDGPU] Fix buildbot failures after 48ebc1af · 4bba8211
      Austin Kerbow authored
      Some buildbots (lto, windows) were failing due to some function reference
      variables being improperly initialized.
      4bba8211
    • Siva Chandra Reddy's avatar
    • Petr Hosek's avatar
      [libFuzzer] Use the compiler to link the relocatable object · 7524fe96
      Petr Hosek authored
      Rather than invoking the linker directly, let the compiler driver
      handle it. This ensures that we use the correct linker in the case
      of cross-compiling.
      
      Differential Revision: https://reviews.llvm.org/D127828
      7524fe96