CPUExecutionProvider
    -48 Add
    -48 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +48 SkipLayerNormalization epsilon=9.999999747378752e-06
    377 nodes, weights c9f0babe98d4
    FuseSkipLayerNorm x48 4d0888df
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x324 3356fae1

CUDAExecutionProvider
    -48 Add
    -48 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +48 SkipLayerNormalization epsilon=9.999999747378752e-06
    377 nodes, weights c9f0babe98d4
    FuseSkipLayerNorm x48 4d0888df
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x324 3356fae1

CoreMLExecutionProvider
    +216 MatMul
    +192 Add
    +192 Slice
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +25 Mul
    -24 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    -1 ReduceL2 keepdims=1 noop_with_empty_axes=0
    +1 ReduceSum keepdims=1
    +1 Sqrt
    1267 nodes, weights 12d081ec1aac
    DecomposeAttention x336 c838ca50
    DecomposeReduceL2 x3 80ef778a
    SplitLargeReduction x552 5b0984f7
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x371 1c964288

MIGraphXExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    -24 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    +23 Add
    736 nodes, weights 3f010c89bcb3
    DecomposeAttention x312 d09fab13
    ElideMaskQueryAxis, DecomposeAttention, DecomposeAttention x24 494fe2fd
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x395 8bf2a8aa

NvTensorRTRTXExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    737 nodes, weights 653863cf8f02
    DecomposeAttention x336 c838ca50
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x396 e34b33dd

OpenVINOExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    737 nodes, weights 653863cf8f02
    DecomposeAttention x336 c838ca50
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x396 e34b33dd

RKNPU static
    +624 Slice
    +528 MatMul
    +504 Add
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    -24 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    -2 Gather axis=0
    +1 Abs
    +1 Cast to=1
    +1 Clip
    +1 Sub
    2323 nodes, weights 199d9ebcc5ee
    contract {"dims":[{}],"embedding":"text.transformer.embed_tokens.weight_fp16"}
    opset 19
    DecomposeAttention x336 bf925f8e
    FloatifyPadKeep x4 51ada845
    SplitLargeReduction x1728 82e5d128
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x1 ee8d996d
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x251 7735bac0

TensorrtExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    737 nodes, weights 653863cf8f02
    DecomposeAttention x336 c838ca50
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x396 e34b33dd
