CPUExecutionProvider
    -48 Add
    -48 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +48 SkipLayerNormalization epsilon=9.999999747378752e-06
    374 nodes, weights f14929ebb2ab
    FuseSkipLayerNorm x47 0953f037
    FuseSkipLayerNorm, _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 cec5f0c2
    _EotOneHotSelect x3 156bebc6
    _FlipCausalAttention x24 704ac85f
    _Fp16TokenEmbedding x2 777a3503
    unstamped x297 5a35c262

CUDAExecutionProvider
    -48 Add
    -48 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +48 SkipLayerNormalization epsilon=9.999999747378752e-06
    374 nodes, weights f14929ebb2ab
    FuseSkipLayerNorm x47 0953f037
    FuseSkipLayerNorm, _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 cec5f0c2
    _EotOneHotSelect x3 156bebc6
    _FlipCausalAttention x24 704ac85f
    _Fp16TokenEmbedding x2 777a3503
    unstamped x297 5a35c262

CoreMLExecutionProvider
    +120 MatMul
    +96 Add
    +96 Reshape
    +96 Slice
    +96 Transpose perm=(0, 2, 1, 3)
    +25 Mul
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    -1 ReduceL2 keepdims=1 noop_with_empty_axes=0
    +1 ReduceSum keepdims=1
    +1 Sqrt
    976 nodes, weights cd7801024584
    DecomposeAttention, _FlipCausalAttention x336 ecf28c36
    DecomposeReduceL2 x3 50783ecd
    SplitLargeReduction x264 748c75b1
    _EotOneHotSelect x3 2f44effc
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x367 5402f86b

MIGraphXExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    734 nodes, weights 8eac61876b41
    DecomposeAttention, _FlipCausalAttention x336 ecf28c36
    _EotOneHotSelect x3 2f44effc
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x392 3a7204d5

NvTensorRTRTXExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    734 nodes, weights 8eac61876b41
    DecomposeAttention, _FlipCausalAttention x336 ecf28c36
    _EotOneHotSelect x3 2f44effc
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x392 3a7204d5

OpenVINOExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    734 nodes, weights 8eac61876b41
    DecomposeAttention, _FlipCausalAttention x336 ecf28c36
    _EotOneHotSelect x3 2f44effc
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x392 3a7204d5

RKNPU static
    +338 Slice
    +289 MatMul
    +265 Add
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    -1 Gather axis=0
    1553 nodes, weights b25beeb6f683
    contract {"dims":[{}],"embedding":"token_embedding.weight_fp16"}
    opset 19
    DecomposeAttention, _FlipCausalAttention x336 ecf28c36
    SplitLargeReduction x917 a55b0027
    _EotOneHotSelect x3 2f44effc
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x1 ee8d996d
    unstamped x295 8f73c43c

TensorrtExecutionProvider
    +96 Reshape
    +96 Transpose perm=(0, 2, 1, 3)
    +48 MatMul
    +24 Add
    -24 Attention is_causal=1 kv_num_heads=16 q_num_heads=16 qk_matmul_output_mode=0 softcap=0.0
    +24 Mul
    +24 Softmax axis=-1
    +24 Transpose perm=(0, 1, 3, 2)
    734 nodes, weights 8eac61876b41
    DecomposeAttention, _FlipCausalAttention x336 ecf28c36
    _EotOneHotSelect x3 2f44effc
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x392 3a7204d5
