CPUExecutionProvider
    -24 Add
    -24 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +24 SkipLayerNormalization epsilon=9.999999747378752e-06
    170 nodes, weights f2036e33fe86
    FuseSkipLayerNorm x23 43fd47b1
    FuseSkipLayerNorm, _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 cec5f0c2
    _EotOneHotSelect x3 5dd15e84
    _FlipCausalAttention x12 0cd68cd8
    _Fp16TokenEmbedding x2 777a3503
    unstamped x129 6aa868cb

CUDAExecutionProvider
    -24 Add
    -24 LayerNormalization axis=-1 epsilon=9.999999747378752e-06 stash_type=1
    +24 SkipLayerNormalization epsilon=9.999999747378752e-06
    170 nodes, weights f2036e33fe86
    FuseSkipLayerNorm x23 43fd47b1
    FuseSkipLayerNorm, _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 cec5f0c2
    _EotOneHotSelect x3 5dd15e84
    _FlipCausalAttention x12 0cd68cd8
    _Fp16TokenEmbedding x2 777a3503
    unstamped x129 6aa868cb

CoreMLExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +13 Mul
    +12 Add
    -12 Attention is_causal=1 kv_num_heads=10 q_num_heads=10 qk_matmul_output_mode=0 softcap=0.0
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    -1 ReduceL2 keepdims=1 noop_with_empty_axes=0
    +1 ReduceSum keepdims=1
    +1 Sqrt
    352 nodes, weights e6866eef7332
    DecomposeAttention, _FlipCausalAttention x168 06d4787f
    DecomposeReduceL2 x3 f3571c6d
    _EotOneHotSelect x3 b8d02656
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x175 7f731d81

MIGraphXExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +12 Add
    -12 Attention is_causal=1 kv_num_heads=10 q_num_heads=10 qk_matmul_output_mode=0 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    350 nodes, weights e6866eef7332
    DecomposeAttention, _FlipCausalAttention x168 06d4787f
    _EotOneHotSelect x3 b8d02656
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x176 97f916dc

NvTensorRTRTXExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +12 Add
    -12 Attention is_causal=1 kv_num_heads=10 q_num_heads=10 qk_matmul_output_mode=0 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    350 nodes, weights e6866eef7332
    DecomposeAttention, _FlipCausalAttention x168 06d4787f
    _EotOneHotSelect x3 b8d02656
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x176 97f916dc

OpenVINOExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +12 Add
    -12 Attention is_causal=1 kv_num_heads=10 q_num_heads=10 qk_matmul_output_mode=0 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    350 nodes, weights e6866eef7332
    DecomposeAttention, _FlipCausalAttention x168 06d4787f
    _EotOneHotSelect x3 b8d02656
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x176 97f916dc

RKNPU static
    +72 Add
    +72 MatMul
    +60 Slice
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +36 Mul
    -12 Attention is_causal=1 kv_num_heads=10 q_num_heads=10 qk_matmul_output_mode=0 softcap=0.0
    +12 Div
    +12 Erf
    -12 Gelu approximate=none
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    -1 Gather axis=0
    553 nodes, weights b069384d7a30
    contract {"dims":[{}],"embedding":"token_embedding.weight_fp16"}
    opset 19
    DecomposeAttention, _FlipCausalAttention x168 06d4787f
    DecomposeGelu x60 55e0b8fd
    SplitLargeReduction x168 91a1c6ab
    _EotOneHotSelect x3 b8d02656
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x1 ee8d996d
    unstamped x152 f7a05fb9

TensorrtExecutionProvider
    +48 Reshape
    +48 Transpose perm=(0, 2, 1, 3)
    +24 MatMul
    +12 Add
    -12 Attention is_causal=1 kv_num_heads=10 q_num_heads=10 qk_matmul_output_mode=0 softcap=0.0
    +12 Mul
    +12 Softmax axis=-1
    +12 Transpose perm=(0, 1, 3, 2)
    350 nodes, weights e6866eef7332
    DecomposeAttention, _FlipCausalAttention x168 06d4787f
    _EotOneHotSelect x3 b8d02656
    _EotSelectBeforeLayerNorm, _EotOneHotSelect x1 9780cd73
    _Fp16TokenEmbedding x2 777a3503
    unstamped x176 97f916dc
