mirror of
https://github.com/immich-app/ml-models.git
synced 2026-09-30 13:22:55 +08:00
* feat: torch-free export core, shared graph/IR helpers, and CLI scaffolding * feat: fused InsightFace face exporter * feat: RKNN export path with per-SoC compilation * feat: single-partition CoreML CLIP export and runtime rewrite API * feat: refactor to passes, ep-specific fixes and optimizations * feat: check command diffing committed graph renderings, wired into CI * chore: graph and rewrite-plan renderings for the catalog * use pokedex large
471 lines
32 KiB
Plaintext
471 lines
32 KiB
Plaintext
<
|
|
ir_version: 10,
|
|
opset_import: ["" : 23],
|
|
producer_name: "pytorch"
|
|
>
|
|
main_graph (uint8[batch,224,224,3] image) => (float[batch,1024] image_embedding)
|
|
<
|
|
float[batch,1024,14,14] add_1016
|
|
float[batch,1024,14,14] add_1092
|
|
float[batch,2048,7,7] add_1188
|
|
float[batch,2048,7,7] add_1264
|
|
float[batch,2048,7,7] add_1340
|
|
float[50,batch,2048] add_1370
|
|
float[batch,256,56,56] add_140
|
|
float[batch,256,56,56] add_216
|
|
float[batch,256,56,56] add_292
|
|
float[batch,512,28,28] add_388
|
|
float[batch,512,28,28] add_464
|
|
float[batch,512,28,28] add_540
|
|
float[batch,512,28,28] add_616
|
|
float[batch,1024,14,14] add_712
|
|
float[batch,1024,14,14] add_788
|
|
float[batch,1024,14,14] add_864
|
|
float[batch,1024,14,14] add_940
|
|
float[batch,64,56,56] avg_pool2d
|
|
float[batch,128,28,28] avg_pool2d_2
|
|
float[batch,256,28,28] avg_pool2d_3
|
|
float[batch,256,14,14] avg_pool2d_4
|
|
float[batch,512,14,14] avg_pool2d_5
|
|
float[batch,512,7,7] avg_pool2d_6
|
|
float[batch,1024,7,7] avg_pool2d_7
|
|
float[50,batch,2048] cat
|
|
float[batch,1] clamp_min
|
|
float[batch,32,112,112] getitem
|
|
float[batch,256,14,14] getitem_102
|
|
float[batch,1024,14,14] getitem_105
|
|
float[batch,256,14,14] getitem_108
|
|
float[batch,256,14,14] getitem_111
|
|
float[batch,1024,14,14] getitem_114
|
|
float[batch,256,14,14] getitem_117
|
|
float[batch,64,56,56] getitem_12
|
|
float[batch,256,14,14] getitem_120
|
|
float[batch,1024,14,14] getitem_123
|
|
float[batch,256,14,14] getitem_126
|
|
float[batch,256,14,14] getitem_129
|
|
float[batch,1024,14,14] getitem_132
|
|
float[batch,512,14,14] getitem_135
|
|
float[batch,512,14,14] getitem_138
|
|
float[batch,2048,7,7] getitem_141
|
|
float[batch,2048,7,7] getitem_144
|
|
float[batch,512,7,7] getitem_147
|
|
float[batch,256,56,56] getitem_15
|
|
float[batch,512,7,7] getitem_150
|
|
float[batch,2048,7,7] getitem_153
|
|
float[batch,512,7,7] getitem_156
|
|
float[batch,512,7,7] getitem_159
|
|
float[batch,2048,7,7] getitem_162
|
|
float[batch,256,56,56] getitem_18
|
|
float[batch,64,56,56] getitem_21
|
|
float[batch,64,56,56] getitem_24
|
|
float[batch,256,56,56] getitem_27
|
|
float[batch,32,112,112] getitem_3
|
|
float[batch,64,56,56] getitem_30
|
|
float[batch,64,56,56] getitem_33
|
|
float[batch,256,56,56] getitem_36
|
|
float[batch,128,56,56] getitem_39
|
|
float[batch,128,56,56] getitem_42
|
|
float[batch,512,28,28] getitem_45
|
|
float[batch,512,28,28] getitem_48
|
|
float[batch,128,28,28] getitem_51
|
|
float[batch,128,28,28] getitem_54
|
|
float[batch,512,28,28] getitem_57
|
|
float[batch,64,112,112] getitem_6
|
|
float[batch,128,28,28] getitem_60
|
|
float[batch,128,28,28] getitem_63
|
|
float[batch,512,28,28] getitem_66
|
|
float[batch,128,28,28] getitem_69
|
|
float[batch,128,28,28] getitem_72
|
|
float[batch,512,28,28] getitem_75
|
|
float[batch,256,28,28] getitem_78
|
|
float[batch,256,28,28] getitem_81
|
|
float[batch,1024,14,14] getitem_84
|
|
float[batch,1024,14,14] getitem_87
|
|
float[batch,64,56,56] getitem_9
|
|
float[batch,256,14,14] getitem_90
|
|
float[batch,256,14,14] getitem_93
|
|
float[batch,1024,14,14] getitem_96
|
|
float[batch,256,14,14] getitem_99
|
|
float[batch,3,224,224] image_chw
|
|
float[batch,224,224,3] image_f32
|
|
float[batch,224,224,3] image_shifted
|
|
float[batch,1] linalg_vector_norm
|
|
float[1,batch,2048] linear
|
|
float[50,batch,2048] linear_1
|
|
float[50,batch,2048] linear_2
|
|
float[batch,1024] linear_3
|
|
float[1,batch,2048] mean
|
|
float[1,batch,2048] node_scaled_dot_product_attention_q_row
|
|
float[49,batch,2048] permute_1
|
|
float[1,batch,32,64] permute_2
|
|
float[batch,32,112,112] relu
|
|
float[batch,32,112,112] relu_1
|
|
float[batch,64,56,56] relu_10
|
|
float[batch,256,56,56] relu_11
|
|
float[batch,128,56,56] relu_12
|
|
float[batch,128,56,56] relu_13
|
|
float[batch,512,28,28] relu_14
|
|
float[batch,128,28,28] relu_15
|
|
float[batch,128,28,28] relu_16
|
|
float[batch,512,28,28] relu_17
|
|
float[batch,128,28,28] relu_18
|
|
float[batch,128,28,28] relu_19
|
|
float[batch,64,112,112] relu_2
|
|
float[batch,512,28,28] relu_20
|
|
float[batch,128,28,28] relu_21
|
|
float[batch,128,28,28] relu_22
|
|
float[batch,512,28,28] relu_23
|
|
float[batch,256,28,28] relu_24
|
|
float[batch,256,28,28] relu_25
|
|
float[batch,1024,14,14] relu_26
|
|
float[batch,256,14,14] relu_27
|
|
float[batch,256,14,14] relu_28
|
|
float[batch,1024,14,14] relu_29
|
|
float[batch,64,56,56] relu_3
|
|
float[batch,256,14,14] relu_30
|
|
float[batch,256,14,14] relu_31
|
|
float[batch,1024,14,14] relu_32
|
|
float[batch,256,14,14] relu_33
|
|
float[batch,256,14,14] relu_34
|
|
float[batch,1024,14,14] relu_35
|
|
float[batch,256,14,14] relu_36
|
|
float[batch,256,14,14] relu_37
|
|
float[batch,1024,14,14] relu_38
|
|
float[batch,256,14,14] relu_39
|
|
float[batch,64,56,56] relu_4
|
|
float[batch,256,14,14] relu_40
|
|
float[batch,1024,14,14] relu_41
|
|
float[batch,512,14,14] relu_42
|
|
float[batch,512,14,14] relu_43
|
|
float[batch,2048,7,7] relu_44
|
|
float[batch,512,7,7] relu_45
|
|
float[batch,512,7,7] relu_46
|
|
float[batch,2048,7,7] relu_47
|
|
float[batch,512,7,7] relu_48
|
|
float[batch,512,7,7] relu_49
|
|
float[batch,256,56,56] relu_5
|
|
float[batch,2048,7,7] relu_50
|
|
float[batch,64,56,56] relu_6
|
|
float[batch,64,56,56] relu_7
|
|
float[batch,256,56,56] relu_8
|
|
float[batch,64,56,56] relu_9
|
|
float[batch,32,1,64] scaled_dot_product_attention
|
|
float[batch,1024] select
|
|
float[2048] split_split_0
|
|
float[2048] split_split_1
|
|
float[2048] split_split_2
|
|
float[unk__1,1,64] transpose
|
|
float[unk__1,50,64] transpose_1
|
|
float[unk__1,50,64] transpose_2
|
|
float[50,1,2048] unsqueeze
|
|
float[1,batch,2048] val_7
|
|
float[50,batch,2048] val_8
|
|
float[50,batch,2048] val_9
|
|
float[1,batch,1024] view_10
|
|
float[batch,2048,49] view_2
|
|
float[1,unk__1,64] view_3
|
|
float[50,unk__1,64] view_4
|
|
float[50,unk__1,64] view_5
|
|
float[batch,32,1,64] view_6
|
|
float[batch,32,50,64] view_7
|
|
float[batch,32,50,64] view_8
|
|
float[batch,2048] view_9
|
|
>
|
|
{
|
|
[pre_cast] image_f32 = Cast <to: int = 1> (image)
|
|
[pre_shift] image_shifted = Sub (image_f32, image_shift)
|
|
[pre_nhwc_to_nchw] image_chw = Transpose <perm: ints = [0, 3, 1, 2]> (image_shifted)
|
|
getitem = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [2, 2]> (image_chw, "visual.conv1.weight", "visual.conv1.weight_bias")
|
|
[node_relu] relu = Relu (getitem)
|
|
getitem_3 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu, "visual.conv2.weight", "visual.conv2.weight_bias")
|
|
relu_1 = Relu (getitem_3)
|
|
getitem_6 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_1, "visual.conv3.weight", "visual.conv3.weight_bias")
|
|
relu_2 = Relu (getitem_6)
|
|
[node_avg_pool2d] avg_pool2d = AveragePool <auto_pad: string = "NOTSET", ceil_mode: int = 0, count_include_pad: int = 1, kernel_shape: ints = [2, 2], pads: ints = [0, 0, 0, 0], strides: ints = [2, 2]> (relu_2)
|
|
getitem_9 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d, "visual.layer1.0.conv1.weight", "visual.layer1.0.conv1.weight_bias")
|
|
relu_3 = Relu (getitem_9)
|
|
getitem_12 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_3, "visual.layer1.0.conv2.weight", "visual.layer1.0.conv2.weight_bias")
|
|
relu_4 = Relu (getitem_12)
|
|
getitem_15 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_4, "visual.layer1.0.conv3.weight", "visual.layer1.0.conv3.weight_bias")
|
|
getitem_18 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d, "visual.layer1.0.downsample.0.weight", "visual.layer1.0.downsample.0.weight_bias")
|
|
add_140 = Add (getitem_15, getitem_18)
|
|
relu_5 = Relu (add_140)
|
|
getitem_21 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_5, "visual.layer1.1.conv1.weight", "visual.layer1.1.conv1.weight_bias")
|
|
relu_6 = Relu (getitem_21)
|
|
getitem_24 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_6, "visual.layer1.1.conv2.weight", "visual.layer1.1.conv2.weight_bias")
|
|
relu_7 = Relu (getitem_24)
|
|
getitem_27 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_7, "visual.layer1.1.conv3.weight", "visual.layer1.1.conv3.weight_bias")
|
|
add_216 = Add (getitem_27, relu_5)
|
|
relu_8 = Relu (add_216)
|
|
getitem_30 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_8, "visual.layer1.2.conv1.weight", "visual.layer1.2.conv1.weight_bias")
|
|
relu_9 = Relu (getitem_30)
|
|
getitem_33 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_9, "visual.layer1.2.conv2.weight", "visual.layer1.2.conv2.weight_bias")
|
|
relu_10 = Relu (getitem_33)
|
|
getitem_36 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_10, "visual.layer1.2.conv3.weight", "visual.layer1.2.conv3.weight_bias")
|
|
add_292 = Add (getitem_36, relu_8)
|
|
relu_11 = Relu (add_292)
|
|
getitem_39 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_11, "visual.layer2.0.conv1.weight", "visual.layer2.0.conv1.weight_bias")
|
|
relu_12 = Relu (getitem_39)
|
|
getitem_42 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_12, "visual.layer2.0.conv2.weight", "visual.layer2.0.conv2.weight_bias")
|
|
relu_13 = Relu (getitem_42)
|
|
avg_pool2d_2 = AveragePool <auto_pad: string = "NOTSET", ceil_mode: int = 0, count_include_pad: int = 1, kernel_shape: ints = [2, 2], pads: ints = [0, 0, 0, 0], strides: ints = [2, 2]> (relu_13)
|
|
getitem_45 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d_2, "visual.layer2.0.conv3.weight", "visual.layer2.0.conv3.weight_bias")
|
|
avg_pool2d_3 = AveragePool <auto_pad: string = "NOTSET", ceil_mode: int = 0, count_include_pad: int = 1, kernel_shape: ints = [2, 2], pads: ints = [0, 0, 0, 0], strides: ints = [2, 2]> (relu_11)
|
|
getitem_48 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d_3, "visual.layer2.0.downsample.0.weight", "visual.layer2.0.downsample.0.weight_bias")
|
|
add_388 = Add (getitem_45, getitem_48)
|
|
relu_14 = Relu (add_388)
|
|
getitem_51 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_14, "visual.layer2.1.conv1.weight", "visual.layer2.1.conv1.weight_bias")
|
|
relu_15 = Relu (getitem_51)
|
|
getitem_54 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_15, "visual.layer2.1.conv2.weight", "visual.layer2.1.conv2.weight_bias")
|
|
relu_16 = Relu (getitem_54)
|
|
getitem_57 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_16, "visual.layer2.1.conv3.weight", "visual.layer2.1.conv3.weight_bias")
|
|
add_464 = Add (getitem_57, relu_14)
|
|
relu_17 = Relu (add_464)
|
|
getitem_60 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_17, "visual.layer2.2.conv1.weight", "visual.layer2.2.conv1.weight_bias")
|
|
relu_18 = Relu (getitem_60)
|
|
getitem_63 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_18, "visual.layer2.2.conv2.weight", "visual.layer2.2.conv2.weight_bias")
|
|
relu_19 = Relu (getitem_63)
|
|
getitem_66 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_19, "visual.layer2.2.conv3.weight", "visual.layer2.2.conv3.weight_bias")
|
|
add_540 = Add (getitem_66, relu_17)
|
|
relu_20 = Relu (add_540)
|
|
getitem_69 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_20, "visual.layer2.3.conv1.weight", "visual.layer2.3.conv1.weight_bias")
|
|
relu_21 = Relu (getitem_69)
|
|
getitem_72 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_21, "visual.layer2.3.conv2.weight", "visual.layer2.3.conv2.weight_bias")
|
|
relu_22 = Relu (getitem_72)
|
|
getitem_75 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_22, "visual.layer2.3.conv3.weight", "visual.layer2.3.conv3.weight_bias")
|
|
add_616 = Add (getitem_75, relu_20)
|
|
relu_23 = Relu (add_616)
|
|
getitem_78 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_23, "visual.layer3.0.conv1.weight", "visual.layer3.0.conv1.weight_bias")
|
|
relu_24 = Relu (getitem_78)
|
|
getitem_81 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_24, "visual.layer3.0.conv2.weight", "visual.layer3.0.conv2.weight_bias")
|
|
relu_25 = Relu (getitem_81)
|
|
avg_pool2d_4 = AveragePool <auto_pad: string = "NOTSET", ceil_mode: int = 0, count_include_pad: int = 1, kernel_shape: ints = [2, 2], pads: ints = [0, 0, 0, 0], strides: ints = [2, 2]> (relu_25)
|
|
getitem_84 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d_4, "visual.layer3.0.conv3.weight", "visual.layer3.0.conv3.weight_bias")
|
|
avg_pool2d_5 = AveragePool <auto_pad: string = "NOTSET", ceil_mode: int = 0, count_include_pad: int = 1, kernel_shape: ints = [2, 2], pads: ints = [0, 0, 0, 0], strides: ints = [2, 2]> (relu_23)
|
|
getitem_87 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d_5, "visual.layer3.0.downsample.0.weight", "visual.layer3.0.downsample.0.weight_bias")
|
|
add_712 = Add (getitem_84, getitem_87)
|
|
relu_26 = Relu (add_712)
|
|
getitem_90 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_26, "visual.layer3.1.conv1.weight", "visual.layer3.1.conv1.weight_bias")
|
|
relu_27 = Relu (getitem_90)
|
|
getitem_93 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_27, "visual.layer3.1.conv2.weight", "visual.layer3.1.conv2.weight_bias")
|
|
relu_28 = Relu (getitem_93)
|
|
getitem_96 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_28, "visual.layer3.1.conv3.weight", "visual.layer3.1.conv3.weight_bias")
|
|
add_788 = Add (getitem_96, relu_26)
|
|
relu_29 = Relu (add_788)
|
|
getitem_99 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_29, "visual.layer3.2.conv1.weight", "visual.layer3.2.conv1.weight_bias")
|
|
relu_30 = Relu (getitem_99)
|
|
getitem_102 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_30, "visual.layer3.2.conv2.weight", "visual.layer3.2.conv2.weight_bias")
|
|
relu_31 = Relu (getitem_102)
|
|
getitem_105 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_31, "visual.layer3.2.conv3.weight", "visual.layer3.2.conv3.weight_bias")
|
|
add_864 = Add (getitem_105, relu_29)
|
|
relu_32 = Relu (add_864)
|
|
getitem_108 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_32, "visual.layer3.3.conv1.weight", "visual.layer3.3.conv1.weight_bias")
|
|
relu_33 = Relu (getitem_108)
|
|
getitem_111 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_33, "visual.layer3.3.conv2.weight", "visual.layer3.3.conv2.weight_bias")
|
|
relu_34 = Relu (getitem_111)
|
|
getitem_114 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_34, "visual.layer3.3.conv3.weight", "visual.layer3.3.conv3.weight_bias")
|
|
add_940 = Add (getitem_114, relu_32)
|
|
relu_35 = Relu (add_940)
|
|
getitem_117 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_35, "visual.layer3.4.conv1.weight", "visual.layer3.4.conv1.weight_bias")
|
|
relu_36 = Relu (getitem_117)
|
|
getitem_120 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_36, "visual.layer3.4.conv2.weight", "visual.layer3.4.conv2.weight_bias")
|
|
relu_37 = Relu (getitem_120)
|
|
getitem_123 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_37, "visual.layer3.4.conv3.weight", "visual.layer3.4.conv3.weight_bias")
|
|
add_1016 = Add (getitem_123, relu_35)
|
|
relu_38 = Relu (add_1016)
|
|
getitem_126 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_38, "visual.layer3.5.conv1.weight", "visual.layer3.5.conv1.weight_bias")
|
|
relu_39 = Relu (getitem_126)
|
|
getitem_129 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_39, "visual.layer3.5.conv2.weight", "visual.layer3.5.conv2.weight_bias")
|
|
relu_40 = Relu (getitem_129)
|
|
getitem_132 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_40, "visual.layer3.5.conv3.weight", "visual.layer3.5.conv3.weight_bias")
|
|
add_1092 = Add (getitem_132, relu_38)
|
|
relu_41 = Relu (add_1092)
|
|
getitem_135 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_41, "visual.layer4.0.conv1.weight", "visual.layer4.0.conv1.weight_bias")
|
|
relu_42 = Relu (getitem_135)
|
|
getitem_138 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_42, "visual.layer4.0.conv2.weight", "visual.layer4.0.conv2.weight_bias")
|
|
relu_43 = Relu (getitem_138)
|
|
avg_pool2d_6 = AveragePool <auto_pad: string = "NOTSET", ceil_mode: int = 0, count_include_pad: int = 1, kernel_shape: ints = [2, 2], pads: ints = [0, 0, 0, 0], strides: ints = [2, 2]> (relu_43)
|
|
getitem_141 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d_6, "visual.layer4.0.conv3.weight", "visual.layer4.0.conv3.weight_bias")
|
|
avg_pool2d_7 = AveragePool <auto_pad: string = "NOTSET", ceil_mode: int = 0, count_include_pad: int = 1, kernel_shape: ints = [2, 2], pads: ints = [0, 0, 0, 0], strides: ints = [2, 2]> (relu_41)
|
|
getitem_144 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (avg_pool2d_7, "visual.layer4.0.downsample.0.weight", "visual.layer4.0.downsample.0.weight_bias")
|
|
add_1188 = Add (getitem_141, getitem_144)
|
|
relu_44 = Relu (add_1188)
|
|
getitem_147 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_44, "visual.layer4.1.conv1.weight", "visual.layer4.1.conv1.weight_bias")
|
|
relu_45 = Relu (getitem_147)
|
|
getitem_150 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_45, "visual.layer4.1.conv2.weight", "visual.layer4.1.conv2.weight_bias")
|
|
relu_46 = Relu (getitem_150)
|
|
getitem_153 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_46, "visual.layer4.1.conv3.weight", "visual.layer4.1.conv3.weight_bias")
|
|
add_1264 = Add (getitem_153, relu_44)
|
|
relu_47 = Relu (add_1264)
|
|
getitem_156 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_47, "visual.layer4.2.conv1.weight", "visual.layer4.2.conv1.weight_bias")
|
|
relu_48 = Relu (getitem_156)
|
|
getitem_159 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [1, 1, 1, 1], strides: ints = [1, 1]> (relu_48, "visual.layer4.2.conv2.weight", "visual.layer4.2.conv2.weight_bias")
|
|
relu_49 = Relu (getitem_159)
|
|
getitem_162 = Conv <auto_pad: string = "NOTSET", dilations: ints = [1, 1], group: int = 1, pads: ints = [0, 0, 0, 0], strides: ints = [1, 1]> (relu_49, "visual.layer4.2.conv3.weight", "visual.layer4.2.conv3.weight_bias")
|
|
add_1340 = Add (getitem_162, relu_47)
|
|
relu_50 = Relu (add_1340)
|
|
view_2 = Reshape <allowzero: int = 1> (relu_50, view_2_target)
|
|
permute_1 = Transpose <perm: ints = [2, 0, 1]> (view_2)
|
|
[node_mean] mean = ReduceMean <keepdims: int = 1, noop_with_empty_axes: int = 0> (permute_1, val_0)
|
|
[node_cat] cat = Concat <axis: int = 0> (mean, permute_1)
|
|
[node_unsqueeze] unsqueeze = Unsqueeze ("visual.attnpool.positional_embedding", val_3)
|
|
add_1370 = Add (cat, unsqueeze)
|
|
split_split_0, split_split_1, split_split_2 = Split <axis: int = 0, num_outputs: int = 3> (cat_1)
|
|
node_scaled_dot_product_attention_q_row = Slice (add_1370, val_0, val_3, val_0)
|
|
val_7 = MatMul (node_scaled_dot_product_attention_q_row, val_4)
|
|
[node_linear] linear = Add (val_7, split_split_0)
|
|
val_8 = MatMul (add_1370, val_5)
|
|
linear_1 = Add (val_8, split_split_1)
|
|
val_9 = MatMul (add_1370, val_6)
|
|
linear_2 = Add (val_9, split_split_2)
|
|
view_3 = Reshape <allowzero: int = 1> (linear, node_scaled_dot_product_attention_q_unpack_1)
|
|
[node_transpose] transpose = Transpose <perm: ints = [1, 0, 2]> (view_3)
|
|
view_4 = Reshape <allowzero: int = 1> (linear_1, view_4_target)
|
|
transpose_1 = Transpose <perm: ints = [1, 0, 2]> (view_4)
|
|
view_5 = Reshape <allowzero: int = 1> (linear_2, view_4_target)
|
|
transpose_2 = Transpose <perm: ints = [1, 0, 2]> (view_5)
|
|
view_6 = Reshape <allowzero: int = 1> (transpose, node_scaled_dot_product_attention_q_pack_1)
|
|
view_7 = Reshape <allowzero: int = 1> (transpose_1, view_7_target)
|
|
view_8 = Reshape <allowzero: int = 1> (transpose_2, view_7_target)
|
|
[node_scaled_dot_product_attention] scaled_dot_product_attention = Attention <is_causal: int = 0, qk_matmul_output_mode: int = 0, softcap: float = 0> (view_6, view_7, view_8)
|
|
permute_2 = Transpose <perm: ints = [2, 0, 1, 3]> (scaled_dot_product_attention)
|
|
view_9 = Reshape <allowzero: int = 1> (permute_2, view_9_target)
|
|
linear_3 = Gemm <alpha: float = 1, beta: float = 1, transA: int = 0, transB: int = 1> (view_9, "visual.attnpool.c_proj.weight", "visual.attnpool.c_proj.bias")
|
|
view_10 = Reshape <allowzero: int = 1> (linear_3, node_scaled_dot_product_attention_out_1)
|
|
select = Squeeze (view_10, val_0)
|
|
[node_linalg_vector_norm] linalg_vector_norm = ReduceL2 <keepdims: int = 1, noop_with_empty_axes: int = 0> (select, val_1)
|
|
[node_clamp_min] clamp_min = Clip (linalg_vector_norm, val_2)
|
|
[node_div] image_embedding = Div (select, clamp_min)
|
|
}
|
|
|
|
weights:
|
|
cat_1 FLOAT[6144] a211dfbeaa31
|
|
image_shift FLOAT[3] 2f7a50e604ad
|
|
node_scaled_dot_product_attention_out_1 INT64[3] 7162728d1394
|
|
node_scaled_dot_product_attention_q_pack_1 INT64[4] f7b9b5620534
|
|
node_scaled_dot_product_attention_q_unpack_1 INT64[3] cfe34a386daf
|
|
val_0 INT64[1] af5570f5a181
|
|
val_1 INT64[1] 12a3ae445661
|
|
val_2 FLOAT[] 6708d9be4956
|
|
val_3 INT64[1] 7c9fa136d441
|
|
val_4 FLOAT[2048,2048] 2b70cefcc885
|
|
val_5 FLOAT[2048,2048] ee5a737b87af
|
|
val_6 FLOAT[2048,2048] 34f890ff6a2d
|
|
view_2_target INT64[3] 68ffb9ecb5ec
|
|
view_4_target INT64[3] c7a3d94c4eb2
|
|
view_7_target INT64[4] e0ea1841387b
|
|
view_9_target INT64[2] 76aecb4697fd
|
|
visual.attnpool.c_proj.bias FLOAT[1024] c92df736f314
|
|
visual.attnpool.c_proj.weight FLOAT[1024,2048] b1fb51d9ce4b
|
|
visual.attnpool.positional_embedding FLOAT[50,2048] 8b84485dd347
|
|
visual.conv1.weight FLOAT[32,3,3,3] d2cc7115d426
|
|
visual.conv1.weight_bias FLOAT[32] 695b387fcc62
|
|
visual.conv2.weight FLOAT[32,32,3,3] 044e0a4e083d
|
|
visual.conv2.weight_bias FLOAT[32] 371becd9c858
|
|
visual.conv3.weight FLOAT[64,32,3,3] 340789ffde55
|
|
visual.conv3.weight_bias FLOAT[64] 61cc233f23a8
|
|
visual.layer1.0.conv1.weight FLOAT[64,64,1,1] fa178cbe4e5f
|
|
visual.layer1.0.conv1.weight_bias FLOAT[64] 4eeac23d37ce
|
|
visual.layer1.0.conv2.weight FLOAT[64,64,3,3] e095bc4eda5b
|
|
visual.layer1.0.conv2.weight_bias FLOAT[64] 606d19b8261c
|
|
visual.layer1.0.conv3.weight FLOAT[256,64,1,1] 4760a24797db
|
|
visual.layer1.0.conv3.weight_bias FLOAT[256] 59c77b9ae9dd
|
|
visual.layer1.0.downsample.0.weight FLOAT[256,64,1,1] 7d93282aaada
|
|
visual.layer1.0.downsample.0.weight_bias FLOAT[256] e94f0c9e3deb
|
|
visual.layer1.1.conv1.weight FLOAT[64,256,1,1] bd243ee08c48
|
|
visual.layer1.1.conv1.weight_bias FLOAT[64] 826820ddc106
|
|
visual.layer1.1.conv2.weight FLOAT[64,64,3,3] e7b6b45eb035
|
|
visual.layer1.1.conv2.weight_bias FLOAT[64] 452ba6c62dc2
|
|
visual.layer1.1.conv3.weight FLOAT[256,64,1,1] a02760e51a5b
|
|
visual.layer1.1.conv3.weight_bias FLOAT[256] a57e8a8f92f0
|
|
visual.layer1.2.conv1.weight FLOAT[64,256,1,1] 22d1c70a1696
|
|
visual.layer1.2.conv1.weight_bias FLOAT[64] 73017849a289
|
|
visual.layer1.2.conv2.weight FLOAT[64,64,3,3] f3e2b9166e84
|
|
visual.layer1.2.conv2.weight_bias FLOAT[64] 76e51ee6eff2
|
|
visual.layer1.2.conv3.weight FLOAT[256,64,1,1] eaaab987d273
|
|
visual.layer1.2.conv3.weight_bias FLOAT[256] 54d00b32994d
|
|
visual.layer2.0.conv1.weight FLOAT[128,256,1,1] c50c62f216b5
|
|
visual.layer2.0.conv1.weight_bias FLOAT[128] 6a244223cbe6
|
|
visual.layer2.0.conv2.weight FLOAT[128,128,3,3] dff98875cd90
|
|
visual.layer2.0.conv2.weight_bias FLOAT[128] e0e32b8526d2
|
|
visual.layer2.0.conv3.weight FLOAT[512,128,1,1] 7e152e384ea6
|
|
visual.layer2.0.conv3.weight_bias FLOAT[512] 4e69afa0af73
|
|
visual.layer2.0.downsample.0.weight FLOAT[512,256,1,1] 7ff5e4f182ce
|
|
visual.layer2.0.downsample.0.weight_bias FLOAT[512] eb385c390151
|
|
visual.layer2.1.conv1.weight FLOAT[128,512,1,1] 44ec8207910d
|
|
visual.layer2.1.conv1.weight_bias FLOAT[128] 702866a78e30
|
|
visual.layer2.1.conv2.weight FLOAT[128,128,3,3] 8de0646c8931
|
|
visual.layer2.1.conv2.weight_bias FLOAT[128] 6ec23bb9e1da
|
|
visual.layer2.1.conv3.weight FLOAT[512,128,1,1] 04197d826c8c
|
|
visual.layer2.1.conv3.weight_bias FLOAT[512] f241b8d03420
|
|
visual.layer2.2.conv1.weight FLOAT[128,512,1,1] 95dc5c6635b1
|
|
visual.layer2.2.conv1.weight_bias FLOAT[128] f0fac6025567
|
|
visual.layer2.2.conv2.weight FLOAT[128,128,3,3] 6375013414f8
|
|
visual.layer2.2.conv2.weight_bias FLOAT[128] e9c229adc89e
|
|
visual.layer2.2.conv3.weight FLOAT[512,128,1,1] 0c31f12f77c0
|
|
visual.layer2.2.conv3.weight_bias FLOAT[512] bf63039053a9
|
|
visual.layer2.3.conv1.weight FLOAT[128,512,1,1] c618a95eb085
|
|
visual.layer2.3.conv1.weight_bias FLOAT[128] 2e5648e16153
|
|
visual.layer2.3.conv2.weight FLOAT[128,128,3,3] ffd039798312
|
|
visual.layer2.3.conv2.weight_bias FLOAT[128] 4ed886f87269
|
|
visual.layer2.3.conv3.weight FLOAT[512,128,1,1] daf1d1e1acb5
|
|
visual.layer2.3.conv3.weight_bias FLOAT[512] 423d8771195e
|
|
visual.layer3.0.conv1.weight FLOAT[256,512,1,1] 03d1b441b2fe
|
|
visual.layer3.0.conv1.weight_bias FLOAT[256] 320490904373
|
|
visual.layer3.0.conv2.weight FLOAT[256,256,3,3] bf0ba36b31a5
|
|
visual.layer3.0.conv2.weight_bias FLOAT[256] 5936953c0909
|
|
visual.layer3.0.conv3.weight FLOAT[1024,256,1,1] d4fc223568d9
|
|
visual.layer3.0.conv3.weight_bias FLOAT[1024] f9b4e95210ea
|
|
visual.layer3.0.downsample.0.weight FLOAT[1024,512,1,1] f58d6ae305a7
|
|
visual.layer3.0.downsample.0.weight_bias FLOAT[1024] 6acd05f34e1a
|
|
visual.layer3.1.conv1.weight FLOAT[256,1024,1,1] a9286b47def4
|
|
visual.layer3.1.conv1.weight_bias FLOAT[256] 4a79119db266
|
|
visual.layer3.1.conv2.weight FLOAT[256,256,3,3] 39e2be9ea8ea
|
|
visual.layer3.1.conv2.weight_bias FLOAT[256] d77e1ddceac9
|
|
visual.layer3.1.conv3.weight FLOAT[1024,256,1,1] fc24820d73ef
|
|
visual.layer3.1.conv3.weight_bias FLOAT[1024] 2852a976e9de
|
|
visual.layer3.2.conv1.weight FLOAT[256,1024,1,1] 10da72685abb
|
|
visual.layer3.2.conv1.weight_bias FLOAT[256] be9cd27d2ad9
|
|
visual.layer3.2.conv2.weight FLOAT[256,256,3,3] 91ab75999e84
|
|
visual.layer3.2.conv2.weight_bias FLOAT[256] 217603ce252d
|
|
visual.layer3.2.conv3.weight FLOAT[1024,256,1,1] 77dd5fe1b228
|
|
visual.layer3.2.conv3.weight_bias FLOAT[1024] 5cceab1488ee
|
|
visual.layer3.3.conv1.weight FLOAT[256,1024,1,1] 594f943d7a4f
|
|
visual.layer3.3.conv1.weight_bias FLOAT[256] 40393b6854ae
|
|
visual.layer3.3.conv2.weight FLOAT[256,256,3,3] da5be985f1a6
|
|
visual.layer3.3.conv2.weight_bias FLOAT[256] 162b683e7fbd
|
|
visual.layer3.3.conv3.weight FLOAT[1024,256,1,1] 2668069c0532
|
|
visual.layer3.3.conv3.weight_bias FLOAT[1024] 267c2dae45b4
|
|
visual.layer3.4.conv1.weight FLOAT[256,1024,1,1] 41d365b4a0ed
|
|
visual.layer3.4.conv1.weight_bias FLOAT[256] 50d08e8e6775
|
|
visual.layer3.4.conv2.weight FLOAT[256,256,3,3] b6ff66f0a791
|
|
visual.layer3.4.conv2.weight_bias FLOAT[256] 45c2067b1c4c
|
|
visual.layer3.4.conv3.weight FLOAT[1024,256,1,1] 7cce15c37b0b
|
|
visual.layer3.4.conv3.weight_bias FLOAT[1024] 341230fbe255
|
|
visual.layer3.5.conv1.weight FLOAT[256,1024,1,1] acb072d3b4f3
|
|
visual.layer3.5.conv1.weight_bias FLOAT[256] 45282e68b6e9
|
|
visual.layer3.5.conv2.weight FLOAT[256,256,3,3] 85417aebcf05
|
|
visual.layer3.5.conv2.weight_bias FLOAT[256] 08534e693484
|
|
visual.layer3.5.conv3.weight FLOAT[1024,256,1,1] 5382d9eafe46
|
|
visual.layer3.5.conv3.weight_bias FLOAT[1024] cc8879f02347
|
|
visual.layer4.0.conv1.weight FLOAT[512,1024,1,1] 1bffc1ead8d8
|
|
visual.layer4.0.conv1.weight_bias FLOAT[512] ec9e7dc0dc3a
|
|
visual.layer4.0.conv2.weight FLOAT[512,512,3,3] d218ccb5c5cc
|
|
visual.layer4.0.conv2.weight_bias FLOAT[512] 122fb87e4d29
|
|
visual.layer4.0.conv3.weight FLOAT[2048,512,1,1] 6365ca4c0d83
|
|
visual.layer4.0.conv3.weight_bias FLOAT[2048] 25f97f110683
|
|
visual.layer4.0.downsample.0.weight FLOAT[2048,1024,1,1] 0765d2036025
|
|
visual.layer4.0.downsample.0.weight_bias FLOAT[2048] 20051736989b
|
|
visual.layer4.1.conv1.weight FLOAT[512,2048,1,1] ce0d72308d32
|
|
visual.layer4.1.conv1.weight_bias FLOAT[512] 2e921063c057
|
|
visual.layer4.1.conv2.weight FLOAT[512,512,3,3] ff5407e36069
|
|
visual.layer4.1.conv2.weight_bias FLOAT[512] 6eb91b41308f
|
|
visual.layer4.1.conv3.weight FLOAT[2048,512,1,1] 77b9eed3aa1b
|
|
visual.layer4.1.conv3.weight_bias FLOAT[2048] 092fd6e88004
|
|
visual.layer4.2.conv1.weight FLOAT[512,2048,1,1] d98cefef114c
|
|
visual.layer4.2.conv1.weight_bias FLOAT[512] d9a3a246e2e5
|
|
visual.layer4.2.conv2.weight FLOAT[512,512,3,3] 95bb518e6393
|
|
visual.layer4.2.conv2.weight_bias FLOAT[512] 20a7a637b746
|
|
visual.layer4.2.conv3.weight FLOAT[2048,512,1,1] 0e179ab51133
|
|
visual.layer4.2.conv3.weight_bias FLOAT[2048] 9da3b4ee3e83
|