CPUExecutionProvider
    -64 Add
    -64 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +64 SkipLayerNormalization epsilon=9.999999747378752e-06
    432 nodes, weights 814130f59635
    FuseSkipLayerNorm x63 f04b9281
    FuseSkipLayerNorm, _FuseClassTokenPrepend x1 ed3bb91a
    RemoveOptionalBiasFromConv x1 99a652fc
    _ConstantifyReshapeTarget x1 f786528a
    _FuseClassTokenPrepend x1 408ab1ab
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x363 3b561fe8

CUDAExecutionProvider
    -64 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +64 SkipLayerNormalization epsilon=9.999999747378752e-06
    -63 Add
    +2 Reshape
    -1 Conv auto_pad=NOTSET dilations=(1, 1) group=1 pads=(0, 0, 0, 0) strides=(14, 14)
    +1 MatMul
    -1 Reshape allowzero=1
    -1 Transpose perm=(0, 2, 1)
    +1 Transpose perm=(0, 2, 1, 3)
    -1 Transpose perm=(0, 3, 1, 2)
    433 nodes, weights fb96e86083fd
    FuseSkipLayerNorm x63 f04b9281
    FuseSkipLayerNorm, _FuseClassTokenPrepend x1 ed3bb91a
    PatchEmbedToMatMul, _ConstantifyReshapeTarget, RemoveOptionalBiasFromConv x5 132b0235
    _FuseClassTokenPrepend x1 1c1de38b
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x361 0e4c6d7f

CoreMLExecutionProvider
    +192 MatMul
    +160 Slice
    +128 Add
    +128 Reshape
    +128 Transpose perm=(0, 2, 1, 3)
    +33 Mul
    -32 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +32 Softmax axis=-1
    +32 Transpose perm=(0, 1, 3, 2)
    -1 ReduceL2 keepdims=1 noop_with_empty_axes=0
    +1 ReduceSum keepdims=1
    +1 Sqrt
    -1 Transpose perm=(0, 3, 1, 2)
    1297 nodes, weights 4bfab0ef7a3a
    DecomposeAttention x416 c8637215
    DecomposeReduceL2 x3 a60b2157
    RemoveOptionalBiasFromConv x1 8217276f
    SplitLargeReduction x448 db6e4678
    _ConstantifyReshapeTarget x1 f786528a
    _FuseClassTokenPrepend x2 64b83b02
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x424 7017bb0a

MIGraphXExecutionProvider
    +128 Reshape
    +128 Transpose perm=(0, 2, 1, 3)
    +64 MatMul
    -32 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +32 Mul
    +32 Softmax axis=-1
    +32 Transpose perm=(0, 1, 3, 2)
    880 nodes, weights 0807b20e6c7b
    DecomposeAttention x416 c8637215
    RemoveOptionalBiasFromConv x1 99a652fc
    _ConstantifyReshapeTarget x1 f786528a
    _FuseClassTokenPrepend x2 64b83b02
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x458 bd3a24f4

NvTensorRTRTXExecutionProvider
    +130 Reshape
    +129 Transpose perm=(0, 2, 1, 3)
    +65 MatMul
    -32 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +32 Mul
    +32 Softmax axis=-1
    +32 Transpose perm=(0, 1, 3, 2)
    +1 Add
    -1 Conv auto_pad=NOTSET dilations=(1, 1) group=1 pads=(0, 0, 0, 0) strides=(14, 14)
    -1 Reshape allowzero=1
    -1 Transpose perm=(0, 2, 1)
    -1 Transpose perm=(0, 3, 1, 2)
    881 nodes, weights 9530acb7fa18
    DecomposeAttention x416 c8637215
    PatchEmbedToMatMul, _ConstantifyReshapeTarget, RemoveOptionalBiasFromConv x5 132b0235
    _FuseClassTokenPrepend x2 db84f216
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x456 fc84bd55

OpenVINOExecutionProvider
    +130 Reshape
    +129 Transpose perm=(0, 2, 1, 3)
    +65 MatMul
    -32 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +32 Mul
    +32 Softmax axis=-1
    +32 Transpose perm=(0, 1, 3, 2)
    +1 Add
    -1 Conv auto_pad=NOTSET dilations=(1, 1) group=1 pads=(0, 0, 0, 0) strides=(14, 14)
    -1 Reshape allowzero=1
    -1 Transpose perm=(0, 2, 1)
    -1 Transpose perm=(0, 3, 1, 2)
    881 nodes, weights 9530acb7fa18
    DecomposeAttention x416 c8637215
    PatchEmbedToMatMul, _ConstantifyReshapeTarget, RemoveOptionalBiasFromConv x5 132b0235
    _FuseClassTokenPrepend x2 db84f216
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x456 fc84bd55

RKNPU static
    +611 Slice
    +547 MatMul
    +515 Add
    +130 Reshape
    +129 Transpose perm=(0, 2, 1, 3)
    +96 Mul
    -32 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +32 Div
    +32 Erf
    -32 Gelu approximate=none
    +32 Softmax axis=-1
    +32 Transpose perm=(0, 1, 3, 2)
    -1 Cast to=1
    -1 Conv auto_pad=NOTSET dilations=(1, 1) group=1 pads=(0, 0, 0, 0) strides=(14, 14)
    -1 Reshape allowzero=1
    -1 Transpose perm=(0, 2, 1)
    -1 Transpose perm=(0, 3, 1, 2)
    2583 nodes, weights 110cfc5824ea
    contract {"dims":[{}]}
    opset 19
    DecomposeAttention x416 c8637215
    DecomposeGelu x160 c1b925e9
    PatchEmbedToMatMul, _ConstantifyReshapeTarget, RemoveOptionalBiasFromConv x5 37ecaa30
    SplitLargeReduction x1704 e3da81ed
    _FuseClassTokenPrepend x2 db84f216
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x294 227f5d24

TensorrtExecutionProvider
    +130 Reshape
    +129 Transpose perm=(0, 2, 1, 3)
    +65 MatMul
    -32 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +32 Mul
    +32 Softmax axis=-1
    +32 Transpose perm=(0, 1, 3, 2)
    +1 Add
    -1 Conv auto_pad=NOTSET dilations=(1, 1) group=1 pads=(0, 0, 0, 0) strides=(14, 14)
    -1 Reshape allowzero=1
    -1 Transpose perm=(0, 2, 1)
    -1 Transpose perm=(0, 3, 1, 2)
    881 nodes, weights 9530acb7fa18
    DecomposeAttention x416 c8637215
    PatchEmbedToMatMul, _ConstantifyReshapeTarget, RemoveOptionalBiasFromConv x5 132b0235
    _FuseClassTokenPrepend x2 db84f216
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 cde77a8c
    _SelectBeforeLayerNorm x1 ab62bee7
    unstamped x456 fc84bd55
