CPUExecutionProvider
    -48 Add
    -48 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +48 SkipLayerNormalization epsilon=9.999999747378752e-06
    326 nodes, weights 803a064e3d96
    FuseSkipLayerNorm x47 0953f037
    FuseSkipLayerNorm, _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 cec5f0c2
    _EotOneHotSelect x3 e4ec520d
    _FlipCausalAttention x24 704ac85f
    _Fp16TokenEmbedding x2 777a3503
    unstamped x249 25fbd5cb

CUDAExecutionProvider
    -48 Add
    -48 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +48 SkipLayerNormalization epsilon=9.999999747378752e-06
    326 nodes, weights 803a064e3d96
    FuseSkipLayerNorm x47 0953f037
    FuseSkipLayerNorm, _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 cec5f0c2
    _EotOneHotSelect x3 e4ec520d
    _FlipCausalAttention x24 704ac85f
    _Fp16TokenEmbedding x2 777a3503
    unstamped x249 25fbd5cb

CoreMLExecutionProvider
    +120 MatMul
    +96 Add
    +96 Reshape
    +96 Slice
    +96 Transpose perm=(0, 2, 1, 3)
    +25 Mul
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    -1 ReduceL2 keepdims=1 noop_with_empty_axes=0
    +1 ReduceSum keepdims=1
    +1 Sqrt
    928 nodes, weights a8e127819f21
    DecomposeAttention, _FlipCausalAttention x336 2cd44200
    DecomposeReduceL2 x3 cb1a64d8
    SplitLargeReduction x264 e93f19bb
    _EotOneHotSelect x3 fc34fbbf
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x319 1c685130

MIGraphXExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    686 nodes, weights faf21f9664d3
    DecomposeAttention, _FlipCausalAttention x336 2cd44200
    _EotOneHotSelect x3 fc34fbbf
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x344 d9f6795d

NvTensorRTRTXExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    686 nodes, weights faf21f9664d3
    DecomposeAttention, _FlipCausalAttention x336 2cd44200
    _EotOneHotSelect x3 fc34fbbf
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x344 d9f6795d

OpenVINOExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    686 nodes, weights faf21f9664d3
    DecomposeAttention, _FlipCausalAttention x336 2cd44200
    _EotOneHotSelect x3 fc34fbbf
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x344 d9f6795d

RKNPU static
    +338 Slice
    +289 Add
    +289 MatMul
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +72 Mul
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Div
    +24 Erf
    -24 Gelu approximate=none
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    -1 Gather axis=0
    1601 nodes, weights 9998e2fc5e77
    contract {"dims":[{}],"embedding":"token_embedding.weight_fp16"}
    opset 19
    DecomposeAttention, _FlipCausalAttention x336 2cd44200
    DecomposeGelu x120 faaef6cd
    SplitLargeReduction x917 14f60a84
    _EotOneHotSelect x3 fc34fbbf
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x1 ee8d996d
    unstamped x223 95ee846e

TensorrtExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    686 nodes, weights faf21f9664d3
    DecomposeAttention, _FlipCausalAttention x336 2cd44200
    _EotOneHotSelect x3 fc34fbbf
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x344 d9f6795d
