CPUExecutionProvider
    -24 Add
    -24 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +24 SkipLayerNormalization epsilon=9.999999747378752e-06
    197 nodes, weights a4ff6f547354
    FuseSkipLayerNorm x24 d36fffc0
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x168 ca66a873

CUDAExecutionProvider
    -24 Add
    -24 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +24 SkipLayerNormalization epsilon=9.999999747378752e-06
    197 nodes, weights a4ff6f547354
    FuseSkipLayerNorm x24 d36fffc0
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x168 ca66a873

CoreMLExecutionProvider
    +60 MatMul
    +48 Add
    +48 Reshape
    +48 Slice
    +48 Transpose perm=(0, 2, 1, 3)
    +13 Mul
    -12 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    -1 ReduceL2 keepdims=1 noop_with_empty_axes=0
    +1 ReduceSum keepdims=1
    +1 Sqrt
    499 nodes, weights c4543946b442
    DecomposeAttention x168 eb0af97d
    DecomposeReduceL2 x3 8bc63742
    SplitLargeReduction x132 e083d438
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x191 7440c948

MIGraphXExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    -12 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    +11 Add
    376 nodes, weights 021a02d8d672
    DecomposeAttention x156 60fe5ac6
    ElideMaskQueryAxis, DecomposeAttention, DecomposeAttention x12 95ec0977
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x203 689273a6

NvTensorRTRTXExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +12 Add
    -12 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    377 nodes, weights 6d4bd14a83eb
    DecomposeAttention x168 eb0af97d
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x204 8ca92728

OpenVINOExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +12 Add
    -12 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    377 nodes, weights 6d4bd14a83eb
    DecomposeAttention x168 eb0af97d
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x204 8ca92728

RKNPU static
    +216 Slice
    +168 MatMul
    +156 Add
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    -12 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    -2 Gather axis=0
    +1 Abs
    +1 Cast to=1
    +1 Clip
    +1 Sub
    883 nodes, weights c31587ee0131
    contract {"dims":[{}],"embedding":"text.transformer.embed_tokens.weight_fp16"}
    opset 19
    DecomposeAttention x168 e5e34075
    FloatifyPadKeep x4 49f1d7fc
    SplitLargeReduction x576 de19bfa3
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x1 ee8d996d
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x131 870d5eef

TensorrtExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +12 Add
    -12 Attention is_causal=0 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 scale=0.125 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    377 nodes, weights 6d4bd14a83eb
    DecomposeAttention x168 eb0af97d
    _ConstantifyReshapeTarget x1 018d14a3
    _FoldEmbeddingScale, _Fp16TokenEmbedding, _Fp16TokenEmbedding x2 52890805
    _ScalarGatherToSlice, _SelectBeforeLayerNorm x1 03ddb995
    _SelectBeforeLayerNorm x1 06a4559b
    unstamped x204 8ca92728
