feat: full models list in models.yaml (#15)

* feat: full models list in models.yaml

* feat: special case runner for large models (#16)

* chore: put more models on mich (#17)

* fix: pr feedback, formatting

* chore: use implicit HF_TOKEN env var

* fix python version

* update dependencies

* fixes

* add onnxscript

* override numpy

* remove sdpa workaround

* dependency hell

* polish cli

* include setuptools

* refactor

* fix gha

* fix: run on pokedex-large instead of mich

* fix: concurrency block for push

* fix: better runner sizing

* fix: enable caching for setup-uv

* fix: use pytorch-cpu

* fix m-clip

* fix gelu, fuse attention

* model versioning

---------

Co-authored-by: mertalev <101130780+mertalev@users.noreply.github.com>
This commit is contained in:
bo0tzz
2026-07-21 21:01:51 -04:00
committed by GitHub
co-authored by mertalev
parent 8897fc0c23
commit 6ce4836cb4
28 changed files with 2409 additions and 1878 deletions
+7 -1
View File
@@ -28,7 +28,8 @@ module.exports = ({core}) => {
if (!m || !n) return false;
return m["name"] === n["name"] &&
m["source"] === n["source"] &&
m["hf-name"] === n["hf-name"];
m["hf-name"] === n["hf-name"] &&
m["runner"] === n["runner"];
}
for (const key of keys) {
@@ -45,6 +46,11 @@ module.exports = ({core}) => {
// !n: deleted, which we ignore
}
// Potential shape change:
// Instead of this binary choice, output all the models
// And annotate each model with which tasks should run for it
// (like export, benchmark, upload, etc)
// That means every model will make it to the index job through the same path
core.setOutput('to_export', JSON.stringify(to_export));
core.setOutput('unchanged', JSON.stringify(unchanged));
}
+41 -8
View File
@@ -1,7 +1,7 @@
on:
workflow_call:
secrets:
HF_AUTH_TOKEN:
HF_TOKEN:
required: true
inputs:
model-name:
@@ -10,6 +10,9 @@ on:
model-source:
required: true
type: string
runner:
required: true
type: string
hf-name:
required: false
type: string
@@ -19,15 +22,38 @@ on:
jobs:
export:
runs-on: ubuntu-latest
runs-on: ${{ inputs.runner }}
steps:
- name: Checkout
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
- run: uv sync # TODO: cache uv env (does setup-uv do that already automatically?)
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv run --group onnx immich-model export "$MODEL_NAME" "$MODEL_SOURCE" --hf-model-name "$HF_MODEL_NAME"
env:
MODEL_NAME: ${{ inputs.model-name }}
MODEL_SOURCE: ${{ inputs.model-source }}
HF_MODEL_NAME: ${{ inputs.hf-name || inputs.model-name }}
HF_TOKEN: ${{ secrets.HF_TOKEN }}
# CLIP models publish at an opset RKNN can't ingest, so build the RKNN binaries from a lower-opset copy
- run: |
if [ "$MODEL_SOURCE" = insightface ]; then
uv run --group rknn immich-model compile "$MODEL_NAME"
else
uv run --group onnx immich-model export "$MODEL_NAME" "$MODEL_SOURCE" \
--hf-model-name "$HF_MODEL_NAME" --opset 20 --output-dir rknn-build
uv run --group rknn immich-model compile "$MODEL_NAME" --input-dir rknn-build
fi
env:
MODEL_NAME: ${{ inputs.model-name }}
MODEL_SOURCE: ${{ inputs.model-source }}
HF_MODEL_NAME: ${{ inputs.hf-name || inputs.model-name }}
HF_TOKEN: ${{ secrets.HF_TOKEN }}
- run: uv run immich_model_exporter export "${{ inputs.model-name }}" "${{ inputs.model-source }}" --hf-model-name "${{ inputs.hf-name }}"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: ${{ inputs.model-name }}
@@ -38,14 +64,17 @@ jobs:
upload:
if: ${{ inputs.upload }}
runs-on: ubuntu-latest
runs-on: ${{ inputs.runner }}
needs: export
steps:
- name: Checkout
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
- run: uv sync # TODO: cache uv env (does setup-uv do that already automatically?)
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
@@ -53,6 +82,10 @@ jobs:
merge-multiple: 'true' # We don't actually have multiple, but this also sets the subfolder naming behaviour
path: models/
- run: uv run immich_model_exporter upload "${{ inputs.model-name }}" --hf-model-name "${{ inputs.hf-name }}" --hf-organization immich-testing
- run: |
uv run immich-model upload "$MODEL_NAME" \
--hf-model-name "$HF_MODEL_NAME" --hf-organization immich-testing --revision v2
env:
HF_AUTH_TOKEN: ${{ secrets.HF_AUTH_TOKEN }}
MODEL_NAME: ${{ inputs.model-name }}
HF_MODEL_NAME: ${{ inputs.hf-name || inputs.model-name }}
HF_TOKEN: ${{ secrets.HF_TOKEN }}
+7 -3
View File
@@ -12,7 +12,9 @@ on:
required: false
description: 'Force export all models'
# TODO: concurrency block?
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
configure:
@@ -59,13 +61,14 @@ jobs:
force: ${{ inputs.force || 'false' }}
oldModels: ${{ steps.old-models.outputs.result }}
newModels: ${{ steps.new-models.outputs.result }}
oldHash: ${{ hashFiles('./before/immich_model_exporter/exporters/**') }}
newHash: ${{ hashFiles('./immich_model_exporter/exporters/**') }}
oldHash: ${{ hashFiles('./before/immich_model/exporters/**') }}
newHash: ${{ hashFiles('./immich_model/exporters/**') }}
script: |
const script = require('./.github/scripts/scope.js')
script({core})
# TODO: Explicitly skip this if nothing to do
export:
uses: ./.github/workflows/export.yaml
needs: configure
@@ -80,4 +83,5 @@ jobs:
model-name: ${{ matrix.name }}
model-source: ${{ matrix.source }}
hf-name: ${{ matrix.hf-name }}
runner: ${{ matrix.runner || 'ubuntu-latest'}}
upload: ${{ inputs.force || github.event_name == 'release'}}
+1 -1
View File
@@ -11,7 +11,7 @@ onnx__*
*.latent
*.pos_embed
vocab.txt
immich_model_exporter/models/**/README.md
immich_model/models/**/README.md
tokenizer.json
tokenizer_config.json
special_tokens_map.json
+1 -1
View File
@@ -1 +1 @@
3.14
3.12
+164
View File
@@ -0,0 +1,164 @@
import json
from pathlib import Path
from typing import Annotated
from typer import Argument, Exit, Option, Typer, echo
from .constants import DELETE_PATTERNS, SOURCE_TO_METADATA, ModelFormat, ModelSource
app = Typer(
no_args_is_help=True,
pretty_exceptions_show_locals=False,
help="Export models used by Immich to ONNX and compile them for on-device runtimes.",
)
ModelName = Annotated[str, Argument(help="Model name; also the per-model output subdirectory name.")]
OutputDir = Annotated[Path, Option(help="Base directory holding per-model output directories.")]
Cache = Annotated[bool, Option(help="Reuse existing outputs instead of regenerating them.")]
def generate_readme(model_name: str, model_source: ModelSource) -> str:
name, link, type = SOURCE_TO_METADATA[model_source]
match model_source:
case ModelSource.MCLIP:
tags = ["immich", "clip", "multilingual"]
case ModelSource.OPENCLIP:
tags = ["immich", "clip"]
lowered = model_name.lower()
if "xlm" in lowered or "nllb" in lowered:
tags.append("multilingual")
case ModelSource.INSIGHTFACE:
tags = ["immich", "facial-recognition"]
case _:
raise ValueError(f"Unsupported model source {model_source}")
return f"""---
tags:
{" - " + "\n - ".join(tags)}
---
# Model Description
This repo contains ONNX exports for the associated {type} model by {name}. See the [{name}]({link}) repo for more info.
This repo is specifically intended for use with [Immich](https://immich.app/), a self-hosted photo library.
"""
@app.command()
def export(
model_name: ModelName,
model_source: Annotated[ModelSource, Argument(help="Upstream source the model comes from.")],
hf_model_name: Annotated[str | None, Option(help="Hugging Face repo to fetch; defaults to model_name.")] = None,
output_dir: OutputDir = Path("models"),
opset: Annotated[int, Option(help="ONNX opset for the exported model.")] = 23,
cache: Cache = True,
) -> None:
"""Export a model to ONNX (plus tokenizer/config) under <output-dir>/<model-name>."""
from . import onnx
if not hf_model_name:
hf_model_name = model_name
output_dir = output_dir / model_name
match model_source:
case ModelSource.MCLIP | ModelSource.OPENCLIP:
output_dir.mkdir(parents=True, exist_ok=True)
onnx.export(hf_model_name, model_source, output_dir, opset=opset, cache=cache)
case ModelSource.INSIGHTFACE:
from huggingface_hub import snapshot_download
# TODO: start from insightface dump instead of downloading from HF
snapshot_download(f"immich-app/{hf_model_name}", local_dir=output_dir)
case _:
raise ValueError(f"Unsupported model source {model_source}")
readme_path = output_dir / "README.md"
if not (cache or readme_path.exists()):
with open(readme_path, "w") as f:
f.write(generate_readme(model_name, model_source))
@app.command()
def compile(
model_name: ModelName,
input_dir: Annotated[
Path, Option(help="Base directory holding the ONNX to compile; defaults to --output-dir.")
] = Path("models"),
output_dir: OutputDir = Path("models"),
cache: Cache = True,
) -> None:
"""Compile an exported ONNX model into a device binary, writing <output-dir>/<model-name>/**/rknpu."""
from . import rknn
try:
rknn.compile(input_dir / model_name, output_dir / model_name, cache=cache)
except Exception as e:
echo(f"Failed to compile {model_name} to RKNN: {e}", err=True)
raise Exit(code=1)
@app.command()
def profile(
model_name: ModelName,
model_format: Annotated[ModelFormat, Option("--format", help="Artifact format to profile.")] = ModelFormat.ONNX,
base_dir: OutputDir = Path("models"),
soc: Annotated[str, Option(help="RKNN target SoC (only for --format rknn; needs an attached NPU).")] = "rk3588",
output_path: Annotated[
Path | None, Option(help="Profile JSON path; defaults to profiling/<model-name>.<format>.json.")
] = None,
) -> None:
"""Benchmark an exported model per-node/per-layer, writing a JSON report."""
model_dir = base_dir / model_name
match model_format:
case ModelFormat.ONNX:
from . import onnx
result = onnx.profile(model_dir)
case ModelFormat.RKNN:
from . import rknn
result = rknn.profile(model_dir, soc)
case _:
raise ValueError(f"Profiling not supported for format {model_format}")
if output_path is None:
output_path = Path("profiling") / f"{model_name}.{model_format}.json"
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(json.dumps(result, indent=2))
for sub, data in result["submodels"].items():
echo(f" {sub}: {data['summary']}")
echo(f"wrote {output_path}")
@app.command()
def upload(
model_name: ModelName,
hf_branch: Annotated[str, Option(help="Repo branch to upload to.")],
hf_model_name: Annotated[str | None, Option(help="Target Hugging Face repo name; defaults to model_name.")] = None,
hf_organization: Annotated[str, Option(help="Hugging Face organization to upload under.")] = "immich-app",
input_dir: Annotated[Path, Option(help="Base directory holding the exported model directory.")] = Path("models"),
) -> None:
"""Upload an exported model directory (<input-dir>/<model-name>) to a Hugging Face repo."""
from huggingface_hub import create_branch, create_repo, upload_folder
from tenacity import retry, stop_after_attempt, wait_fixed
if not hf_model_name:
hf_model_name = model_name
model_dir = input_dir / model_name
repo_id = f"{hf_organization}/{hf_model_name}"
@retry(stop=stop_after_attempt(5), wait=wait_fixed(5))
def upload_model() -> None:
create_repo(repo_id, exist_ok=True)
if hf_branch != "main":
create_branch(repo_id, branch=hf_branch, exist_ok=True)
upload_folder(
repo_id=repo_id,
folder_path=model_dir,
revision=hf_branch,
# remote repo files to be deleted before uploading
# deletion is in the same commit as the upload, so it's atomic
delete_patterns=DELETE_PATTERNS,
)
upload_model()
+3
View File
@@ -0,0 +1,3 @@
from immich_model import app
app()
@@ -13,6 +13,11 @@ class ModelTask(StrEnum):
SEARCH = "clip"
class ModelFormat(StrEnum):
ONNX = "onnx"
RKNN = "rknn"
class SourceMetadata(NamedTuple):
name: str
link: str
+4
View File
@@ -0,0 +1,4 @@
from .export import export
from .profile import profile
__all__ = ["export", "profile"]
@@ -4,17 +4,15 @@ from ..constants import ModelSource
from .models import mclip, openclip
def export(
model_name: str, model_source: ModelSource, output_dir: Path, opset_version: int = 19, cache: bool = True
) -> None:
def export(model_name: str, model_source: ModelSource, output_dir: Path, opset: int, cache: bool = True) -> None:
visual_dir = output_dir / "visual"
textual_dir = output_dir / "textual"
match model_source:
case ModelSource.MCLIP:
mclip.to_onnx(model_name, opset_version, visual_dir, textual_dir, cache=cache)
mclip.to_onnx(model_name, opset, visual_dir, textual_dir, cache=cache)
case ModelSource.OPENCLIP:
name, _, pretrained = model_name.partition("__")
config = openclip.OpenCLIPModelConfig(name, pretrained)
openclip.to_onnx(config, opset_version, visual_dir, textual_dir, cache=cache)
openclip.to_onnx(config, opset, visual_dir, textual_dir, cache=cache)
case _:
raise ValueError(f"Unsupported model source {model_source}")
@@ -23,13 +23,9 @@ def to_onnx(
) -> tuple[Path, Path]:
textual_path = get_model_path(output_dir_textual)
if not cache or not textual_path.exists():
import torch
from multilingual_clip.pt_multilingual_clip import MultilingualCLIP
from transformers import AutoTokenizer
torch.backends.mha.set_fastpath_enabled(False)
model = MultilingualCLIP.from_pretrained(model_name)
model = _load_model(model_name)
AutoTokenizer.from_pretrained(model_name).save_pretrained(output_dir_textual)
model.eval()
@@ -39,11 +35,31 @@ def to_onnx(
_export_text_encoder(model, textual_path, opset_version)
else:
print(f"Model {textual_path} already exists, skipping")
visual_path, _ = openclip_to_onnx(_MCLIP_TO_OPENCLIP[model_name], opset_version, output_dir_visual, cache=cache)
# Keep the original activation since M-CLIP's text encoder was aligned against it
visual_path, _ = openclip_to_onnx(
_MCLIP_TO_OPENCLIP[model_name], opset_version, output_dir_visual, cache=cache, force_quick_gelu=False
)
assert visual_path is not None, "Visual model export failed"
return visual_path, textual_path
def _load_model(model_name: str) -> Any:
# transformers 5 breaks multilingual_clip, so we instantiate the model manually.
import torch
from huggingface_hub import hf_hub_download
from multilingual_clip import Config_MCLIP
from multilingual_clip.pt_multilingual_clip import MultilingualCLIP
config = Config_MCLIP.MCLIPConfig.from_pretrained(model_name)
model = MultilingualCLIP(config)
weights_path = hf_hub_download(model_name, "pytorch_model.bin")
state_dict = torch.load(weights_path, map_location="cpu", weights_only=True)
missing, _ = model.load_state_dict(state_dict, strict=False)
assert not missing, f"Missing weights when loading {model_name}: {missing}"
return model
def _export_text_encoder(model: Any, output_path: Path | str, opset_version: int) -> None:
import torch
from multilingual_clip.pt_multilingual_clip import MultilingualCLIP
@@ -68,7 +84,7 @@ def _export_text_encoder(model: Any, output_path: Path | str, opset_version: int
args,
output_path.as_posix(),
input_names=["input_ids", "attention_mask"],
output_names=["embedding"],
output_names=["text_embedding"],
opset_version=opset_version,
# dynamic_axes={
# "input_ids": {0: "batch_size", 1: "sequence_length"},
@@ -38,6 +38,7 @@ def to_onnx(
output_dir_visual: Path | str | None = None,
output_dir_textual: Path | str | None = None,
cache: bool = True,
force_quick_gelu: bool | None = None,
) -> tuple[Path | None, Path | None]:
visual_path = None
textual_path = None
@@ -54,14 +55,12 @@ def to_onnx(
return visual_path, textual_path
import open_clip
import torch
from transformers import AutoTokenizer
torch.backends.mha.set_fastpath_enabled(False)
model = open_clip.create_model(
model_cfg.name,
pretrained=model_cfg.pretrained,
force_quick_gelu=force_quick_gelu or model_cfg.pretrained == "openai",
jit=False,
require_pretrained=True,
)
@@ -116,7 +115,7 @@ def _export_image_encoder(
args,
output_path.as_posix(),
input_names=["image"],
output_names=["embedding"],
output_names=["image_embedding"],
opset_version=opset_version,
# dynamic_axes={"image": {0: "batch_size"}},
)
@@ -145,7 +144,7 @@ def _export_text_encoder(
args,
output_path.as_posix(),
input_names=["text"],
output_names=["embedding"],
output_names=["text_embedding"],
opset_version=opset_version,
# dynamic_axes={"text": {0: "batch_size"}},
)
+54
View File
@@ -0,0 +1,54 @@
import json
from collections import defaultdict
from pathlib import Path
from typing import Any
# per-model subdirectories that hold a model.onnx
SUBMODELS = ["textual", "visual", "detection", "recognition"]
def profile(model_dir: Path, runs: int = 20, provider: str = "CPUExecutionProvider") -> dict[str, Any]:
"""Profile every ONNX submodel via onnxruntime, returning per-op-node costs.
Uses onnxruntime's built-in profiler (enable_profiling), which records a
kernel-time trace per node; costs are aggregated by op type.
"""
import numpy as np
import onnxruntime as ort
subs = [s for s in SUBMODELS if (model_dir / s / "model.onnx").is_file()]
if not subs:
raise RuntimeError(f"No ONNX model found under {model_dir}")
result: dict[str, Any] = {"model": model_dir.name, "format": "onnx", "provider": provider, "submodels": {}}
for sub in subs:
so = ort.SessionOptions()
so.enable_profiling = True
sess = ort.InferenceSession((model_dir / sub / "model.onnx").as_posix(), sess_options=so, providers=[provider])
feeds = {i.name: _rand_input(i, np) for i in sess.get_inputs()}
for _ in range(runs): # profiling records every run; the first (cold) run is averaged in
sess.run(None, feeds)
events = json.load(open(sess.end_profiling()))
per_op: dict[str, list[float]] = defaultdict(lambda: [0, 0.0])
for e in events:
if e.get("cat") == "Node" and e.get("name", "").endswith("_kernel_time"):
op = e["args"].get("op_name", "?")
per_op[op][0] += 1
per_op[op][1] += e["dur"]
ranked = sorted(per_op.items(), key=lambda kv: -kv[1][1])
ops = [{"op": op, "count": int(n // runs), "us_per_run": round(us / runs, 1)} for op, (n, us) in ranked]
mean_ms = round(sum(us for _, us in per_op.values()) / runs / 1000, 3)
hottest = f", hottest {ops[0]['op']} ({ops[0]['us_per_run']}us)" if ops else ""
result["submodels"][sub] = {"summary": f"{mean_ms} ms/run{hottest}", "mean_ms": mean_ms, "ops": ops}
return result
def _rand_input(node: Any, np: Any) -> Any:
shape = [d if isinstance(d, int) else 1 for d in node.shape]
if "int" in node.type: # token ids etc. — values don't matter for latency
return np.zeros(shape, dtype=np.int64 if "int64" in node.type else np.int32)
return np.random.rand(*shape).astype(np.float32)
+4
View File
@@ -0,0 +1,4 @@
from .compile import compile
from .profile import profile
__all__ = ["compile", "profile"]
+84
View File
@@ -0,0 +1,84 @@
import math
from pathlib import Path
from typing import Any
# tanh-GELU coefficients
_GELU_C0 = math.sqrt(2 / math.pi)
_GELU_C1 = 0.044715
def prepare_for_rknn(onnx_path: Path, work_dir: Path) -> Path:
"""Return an ONNX path that ``rknn.build`` can ingest.
rknn-toolkit2 rejects opset > 19 and has no NPU `Erf` kernel, so each native `Gelu` node
is rewritten into its tanh approximation and the opset is pinned to 19.
"""
import onnx
model = onnx.load(onnx_path.as_posix())
opset = max((o.version for o in model.opset_import if o.domain in ("", "ai.onnx")), default=0)
has_gelu = any(node.op_type == "Gelu" for node in model.graph.node)
if opset <= 19 and not has_gelu:
return onnx_path
if has_gelu:
_decompose_gelu_to_tanh(model)
_pin_opset(model, 19)
out_path = work_dir / "model.onnx"
onnx.save(
model,
out_path.as_posix(),
save_as_external_data=True,
all_tensors_to_one_file=True,
location="model.onnx.data",
)
return out_path
def _decompose_gelu_to_tanh(model: Any) -> None:
"""Replace every native `Gelu` node with its tanh-approximation subgraph, in place."""
import numpy as np
from onnx import helper, numpy_helper
graph = model.graph
consts: dict[str, Any] = {}
def const(name: str, value: float) -> str:
if name not in consts:
consts[name] = numpy_helper.from_array(np.array(value, dtype=np.float32), name)
return name
new_nodes = []
for node in graph.node:
if node.op_type != "Gelu":
new_nodes.append(node)
continue
x, y = node.input[0], node.output[0]
p = node.name or y
c0, c1 = const("gelu_c0", _GELU_C0), const("gelu_c1", _GELU_C1)
half, one = const("gelu_half", 0.5), const("gelu_one", 1.0)
new_nodes += [
helper.make_node("Mul", [x, x], [f"{p}_x2"]),
helper.make_node("Mul", [f"{p}_x2", x], [f"{p}_x3"]),
helper.make_node("Mul", [f"{p}_x3", c1], [f"{p}_c1x3"]),
helper.make_node("Add", [x, f"{p}_c1x3"], [f"{p}_inner"]),
helper.make_node("Mul", [f"{p}_inner", c0], [f"{p}_scaled"]),
helper.make_node("Tanh", [f"{p}_scaled"], [f"{p}_tanh"]),
helper.make_node("Add", [f"{p}_tanh", one], [f"{p}_1ptanh"]),
helper.make_node("Mul", [x, half], [f"{p}_halfx"]),
helper.make_node("Mul", [f"{p}_halfx", f"{p}_1ptanh"], [y]),
]
del graph.node[:]
graph.node.extend(new_nodes)
graph.initializer.extend(consts.values())
def _pin_opset(model: Any, version: int) -> None:
"""Force the ai.onnx opset to `version`."""
from onnx import helper
kept = [opset for opset in model.opset_import if opset.domain not in ("", "ai.onnx")]
kept.append(helper.make_operatorsetid("", version))
del model.opset_import[:]
model.opset_import.extend(kept)
+113
View File
@@ -0,0 +1,113 @@
import tempfile
from pathlib import Path
from ..constants import RKNN_SOCS
from ._onnx import prepare_for_rknn
def _export_platform(
onnx_path: Path,
output_dir: Path,
target_platform: str,
inputs: list[str] | None = None,
input_size_list: list[list[int]] | None = None,
fuse_matmul_softmax_matmul_to_sdpa: bool = True,
) -> None:
from rknn.api import RKNN
output_path = output_dir / "rknpu" / target_platform / "model.rknn"
print(f"Exporting {onnx_path} to {output_path}")
def check(ret: int, step: str) -> None:
if ret != 0:
raise RuntimeError(f"RKNN {step} failed for {target_platform} (code {ret})")
rknn = RKNN(verbose=False)
rknn.config(
target_platform=target_platform,
disable_rules=[] if fuse_matmul_softmax_matmul_to_sdpa else ["fuse_matmul_softmax_matmul_to_sdpa"],
enable_flash_attention=False,
model_pruning=True,
)
check(rknn.load_onnx(model=onnx_path.as_posix(), inputs=inputs, input_size_list=input_size_list), "load")
check(rknn.build(do_quantization=False), "build")
output_path.parent.mkdir(parents=True, exist_ok=True)
check(rknn.export_rknn(output_path.as_posix()), "export")
def _export_platforms(
input_dir: Path,
output_dir: Path,
inputs: list[str] | None = None,
input_size_list: list[list[int]] | None = None,
cache: bool = True,
) -> None:
socs = []
for soc in RKNN_SOCS:
if cache and (model_path := output_dir / "rknpu" / soc / "model.rknn").exists():
print(f"{model_path} already exists, skipping")
else:
socs.append(soc)
if not socs:
return
with tempfile.TemporaryDirectory() as work_dir:
# rknn.build can't take opset > 19 or exact GELU, so normalise the ONNX once for all SoCs.
onnx_path = prepare_for_rknn(input_dir / "model.onnx", Path(work_dir))
def attempt(soc: str, fuse: bool) -> None:
_export_platform(
onnx_path,
output_dir,
soc,
inputs=inputs,
input_size_list=input_size_list,
fuse_matmul_softmax_matmul_to_sdpa=fuse,
)
fuse = True
failed: list[str] = []
for soc in socs:
try:
attempt(soc, fuse)
except Exception as e:
# This fusion isn't valid for every model; drop it (for this and later SoCs) and retry.
if fuse and "inputs or 'outputs' must be set" in str(e):
print(f"Retrying {soc} without fuse_matmul_softmax_matmul_to_sdpa")
fuse = False
try:
attempt(soc, fuse)
continue
except Exception as retry_error:
e = retry_error
print(f"Failed to export {input_dir.name} for {soc}: {e}")
failed.append(soc)
if failed:
raise RuntimeError(f"RKNN export failed for {input_dir.name} on: {', '.join(failed)}")
def compile(input_dir: Path, output_dir: Path, cache: bool = True) -> None:
"""Compile each ONNX submodel under input_dir into RKNN binaries under output_dir/<sub>/rknpu."""
# (subdirectory, inputs, input_size_list) — inputs/sizes are only needed for the face models
sub_models: list[tuple[str, list[str] | None, list[list[int]] | None]] = [
("textual", None, None),
("visual", None, None),
("detection", ["input.1"], [[1, 3, 640, 640]]),
("recognition", ["input.1"], [[1, 3, 112, 112]]),
]
present = [(sub, inputs, sizes) for sub, inputs, sizes in sub_models if (input_dir / sub).is_dir()]
if not present:
raise RuntimeError(f"No exportable model found under {input_dir}")
errors: list[str] = []
for sub, inputs, input_size_list in present:
try:
_export_platforms(
input_dir / sub, output_dir / sub, inputs=inputs, input_size_list=input_size_list, cache=cache
)
except Exception as e:
errors.append(str(e))
if errors:
raise RuntimeError("; ".join(errors))
+67
View File
@@ -0,0 +1,67 @@
from pathlib import Path
from typing import Any
# per-model subdirectories that hold rknpu/<soc>/model.rknn
SUBMODELS = ["textual", "visual", "detection", "recognition"]
def profile(model_dir: Path, soc: str = "rk3588") -> dict[str, Any]:
"""Profile every RKNN submodel on an attached NPU via eval_perf (per-op NPU/CPU costs)."""
from rknn.api import RKNN
subs = [s for s in SUBMODELS if (model_dir / s / "rknpu" / soc / "model.rknn").is_file()]
if not subs:
raise RuntimeError(f"No RKNN model for {soc} found under {model_dir}")
result: dict[str, Any] = {"model": model_dir.name, "format": "rknn", "soc": soc, "submodels": {}}
for sub in subs:
rknn = RKNN(verbose=False)
try:
if rknn.load_rknn((model_dir / sub / "rknpu" / soc / "model.rknn").as_posix()) != 0:
raise RuntimeError(f"load_rknn failed for {sub}")
if rknn.init_runtime(target=soc, perf_debug=True) != 0:
raise RuntimeError(f"init_runtime failed for {sub} (is a {soc} NPU attached?)")
result["submodels"][sub] = _parse_perf(rknn.eval_perf(is_print=False))
finally:
rknn.release()
return result
def _parse_perf(report: str) -> dict[str, Any]:
"""Parse eval_perf's per-op-type summary table + CPU/NPU totals out of its report string."""
ops: list[dict[str, Any]] = []
totals = {"cpu_us": 0, "npu_us": 0, "total_us": 0}
in_table = False
for line in report.splitlines():
s = line.strip()
if s.startswith("OpType") and "CallNumber" in s:
in_table = True
continue
if not in_table or not s or s.startswith("-"):
continue
parts = s.split()
if parts[0] == "Total": # Total <cpu> <gpu> <npu> <total>
nums = [int(p) for p in parts[1:] if p.lstrip("-").isdigit()]
if len(nums) >= 4:
totals = {"cpu_us": nums[0], "npu_us": nums[2], "total_us": nums[3]}
break
if len(parts) >= 6 and parts[1].isdigit(): # <op> <calls> <cpu> <gpu> <npu> <total> <ratio%>
ops.append(
{
"op": parts[0],
"calls": int(parts[1]),
"cpu_us": int(parts[2]),
"npu_us": int(parts[4]),
"total_us": int(parts[5]),
}
)
total = totals["total_us"] or 1
cpu_pct, npu_pct = round(100 * totals["cpu_us"] / total), round(100 * totals["npu_us"] / total)
return {
"summary": f"{totals['total_us'] / 1000:.1f} ms/frame (CPU {cpu_pct}%, NPU {npu_pct}%)",
"total_ms": round(totals["total_us"] / 1000, 2),
"cpu_pct": cpu_pct,
"npu_pct": npu_pct,
"ops": ops,
}
-171
View File
@@ -1,171 +0,0 @@
import json
import resource
from pathlib import Path
import typer
from tenacity import retry, stop_after_attempt, wait_fixed
from typing_extensions import Annotated
from .exporters.constants import DELETE_PATTERNS, SOURCE_TO_METADATA, ModelSource, ModelTask
from .exporters.onnx import export as onnx_export
from .exporters.rknn import export as rknn_export
app = typer.Typer(pretty_exceptions_show_locals=False)
def generate_readme(model_name: str, model_source: ModelSource) -> str:
(name, link, type) = SOURCE_TO_METADATA[model_source]
match model_source:
case ModelSource.MCLIP:
tags = ["immich", "clip", "multilingual"]
case ModelSource.OPENCLIP:
tags = ["immich", "clip"]
lowered = model_name.lower()
if "xlm" in lowered or "nllb" in lowered:
tags.append("multilingual")
case ModelSource.INSIGHTFACE:
tags = ["immich", "facial-recognition"]
case _:
raise ValueError(f"Unsupported model source {model_source}")
return f"""---
tags:
{" - " + "\n - ".join(tags)}
---
# Model Description
This repo contains ONNX exports for the associated {type} model by {name}. See the [{name}]({link}) repo for more info.
This repo is specifically intended for use with [Immich](https://immich.app/), a self-hosted photo library.
"""
@app.command()
def export(
model_name: str,
model_source: ModelSource,
hf_model_name: str | None = None,
output_dir: Path = Path("models"),
cache: bool = True,
) -> None:
if not hf_model_name:
hf_model_name = model_name
output_dir = output_dir / model_name
match model_source:
case ModelSource.MCLIP | ModelSource.OPENCLIP:
output_dir.mkdir(parents=True, exist_ok=True)
onnx_export(hf_model_name, model_source, output_dir, cache=cache)
case ModelSource.INSIGHTFACE:
from huggingface_hub import snapshot_download
# TODO: start from insightface dump instead of downloading from HF
snapshot_download(f"immich-app/{hf_model_name}", local_dir=output_dir)
case _:
raise ValueError(f"Unsupported model source {model_source}")
try:
rknn_export(output_dir, cache=cache)
except Exception as e:
print(f"Failed to export model {model_name} to rknn: {e}")
(output_dir / "rknpu").unlink(missing_ok=True)
readme_path = output_dir / "README.md"
if not (cache or readme_path.exists()):
with open(readme_path, "w") as f:
f.write(generate_readme(model_name, model_source))
# TODO: Args shape parity with the other commands? (eg taking model_name, default base dir)
@app.command()
def profile(model_dir: Path, model_task: ModelTask, output_path: Path) -> None:
from timeit import timeit
import numpy as np
import onnxruntime as ort
np.random.seed(0)
sess_options = ort.SessionOptions()
sess_options.enable_cpu_mem_arena = False
providers = ["CPUExecutionProvider"]
provider_options = [{"arena_extend_strategy": "kSameAsRequested"}]
match model_task:
case ModelTask.SEARCH:
textual = ort.InferenceSession(
model_dir / "textual" / "model.onnx",
sess_options=sess_options,
providers=providers,
provider_options=provider_options,
)
tokens = {node.name: np.random.rand(*node.shape).astype(np.int32) for node in textual.get_inputs()}
visual = ort.InferenceSession(
model_dir / "visual" / "model.onnx",
sess_options=sess_options,
providers=providers,
provider_options=provider_options,
)
image = {node.name: np.random.rand(*node.shape).astype(np.float32) for node in visual.get_inputs()}
def predict() -> None:
textual.run(None, tokens)
visual.run(None, image)
case ModelTask.FACIAL_RECOGNITION:
detection = ort.InferenceSession(
model_dir / "detection" / "model.onnx",
sess_options=sess_options,
providers=providers,
provider_options=provider_options,
)
image = {node.name: np.random.rand(1, 3, 640, 640).astype(np.float32) for node in detection.get_inputs()}
recognition = ort.InferenceSession(
model_dir / "recognition" / "model.onnx",
sess_options=sess_options,
providers=providers,
provider_options=provider_options,
)
face = {node.name: np.random.rand(1, 3, 112, 112).astype(np.float32) for node in recognition.get_inputs()}
def predict() -> None:
detection.run(None, image)
recognition.run(None, face)
case _:
raise ValueError(f"Unsupported model task {model_task}")
predict()
ms = timeit(predict, number=100)
rss = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
json.dump({"pretrained_model": model_dir.name, "peak_rss": rss, "exec_time_ms": ms}, output_path.open("w"))
print(f"Model {model_dir.name} took {ms:.2f}ms per iteration using {rss / 1024:.2f}MiB of memory")
@app.command()
def upload(
model_name: str,
input_dir: Path = Path("models"),
hf_model_name: str | None = None,
hf_organization: str = "immich-app",
hf_auth_token: Annotated[str | None, typer.Option(envvar="HF_AUTH_TOKEN")] = None,
) -> None:
from huggingface_hub import create_repo, upload_folder
if not hf_model_name:
hf_model_name = model_name
model_dir = input_dir / hf_model_name
repo_id = f"{hf_organization}/{hf_model_name}"
@retry(stop=stop_after_attempt(5), wait=wait_fixed(5))
def upload_model() -> None:
create_repo(repo_id, exist_ok=True, token=hf_auth_token)
upload_folder(
repo_id=repo_id,
folder_path=model_dir,
# remote repo files to be deleted before uploading
# deletion is in the same commit as the upload, so it's atomic
delete_patterns=DELETE_PATTERNS,
token=hf_auth_token,
)
upload_model()
-3
View File
@@ -1,3 +0,0 @@
from immich_model_exporter import app
app()
-96
View File
@@ -1,96 +0,0 @@
from pathlib import Path
from .constants import RKNN_SOCS
def _export_platform(
model_dir: Path,
target_platform: str,
inputs: list[str] | None = None,
input_size_list: list[list[int]] | None = None,
fuse_matmul_softmax_matmul_to_sdpa: bool = True,
cache: bool = True,
) -> None:
from rknn.api import RKNN
input_path = model_dir / "model.onnx"
output_path = model_dir / "rknpu" / target_platform / "model.rknn"
if cache and output_path.exists():
print(f"Model {input_path} already exists at {output_path}, skipping")
return
print(f"Exporting model {input_path} to {output_path}")
rknn = RKNN(verbose=False)
rknn.config(
target_platform=target_platform,
disable_rules=["fuse_matmul_softmax_matmul_to_sdpa"] if not fuse_matmul_softmax_matmul_to_sdpa else [],
enable_flash_attention=False,
model_pruning=True,
)
ret = rknn.load_onnx(model=input_path.as_posix(), inputs=inputs, input_size_list=input_size_list)
if ret != 0:
raise RuntimeError("Load failed!")
ret = rknn.build(do_quantization=False)
if ret != 0:
raise RuntimeError("Build failed!")
output_path.parent.mkdir(parents=True, exist_ok=True)
ret = rknn.export_rknn(output_path.as_posix())
if ret != 0:
raise RuntimeError("Export rknn model failed!")
def _export_platforms(
model_dir: Path,
inputs: list[str] | None = None,
input_size_list: list[list[int]] | None = None,
cache: bool = True,
) -> None:
fuse_matmul_softmax_matmul_to_sdpa = True
for soc in RKNN_SOCS:
try:
_export_platform(
model_dir,
soc,
inputs=inputs,
input_size_list=input_size_list,
fuse_matmul_softmax_matmul_to_sdpa=fuse_matmul_softmax_matmul_to_sdpa,
cache=cache,
)
except Exception as e:
print(f"Failed to export model for {soc}: {e}")
if "inputs or 'outputs' must be set" in str(e):
print("Retrying without fuse_matmul_softmax_matmul_to_sdpa")
fuse_matmul_softmax_matmul_to_sdpa = False
_export_platform(
model_dir,
soc,
inputs=inputs,
input_size_list=input_size_list,
fuse_matmul_softmax_matmul_to_sdpa=fuse_matmul_softmax_matmul_to_sdpa,
cache=cache,
)
def export(model_dir: Path, cache: bool = True) -> None:
textual = model_dir / "textual"
visual = model_dir / "visual"
detection = model_dir / "detection"
recognition = model_dir / "recognition"
if textual.is_dir():
_export_platforms(textual, cache=cache)
if visual.is_dir():
_export_platforms(visual, cache=cache)
if detection.is_dir():
_export_platforms(detection, inputs=["input.1"], input_size_list=[[1, 3, 640, 640]], cache=cache)
if recognition.is_dir():
_export_platforms(recognition, inputs=["input.1"], input_size_list=[[1, 3, 112, 112]], cache=cache)
-242
View File
@@ -1,242 +0,0 @@
import subprocess
from pathlib import Path
from exporters.constants import ModelSource
from immich_model_exporter import clean_name
from immich_model_exporter.exporters.constants import SOURCE_TO_TASK
mclip = [
"M-CLIP/LABSE-Vit-L-14",
"M-CLIP/XLM-Roberta-Large-Vit-B-16Plus",
"M-CLIP/XLM-Roberta-Large-Vit-B-32",
"M-CLIP/XLM-Roberta-Large-Vit-L-14",
]
openclip = [
"RN101__openai",
"RN101__yfcc15m",
"RN50__cc12m",
"RN50__openai",
"RN50__yfcc15m",
"RN50x16__openai",
"RN50x4__openai",
"RN50x64__openai",
"ViT-B-16-SigLIP-256__webli",
"ViT-B-16-SigLIP-384__webli",
"ViT-B-16-SigLIP-512__webli",
"ViT-B-16-SigLIP-i18n-256__webli",
"ViT-B-16-SigLIP2__webli",
"ViT-B-16-SigLIP__webli",
"ViT-B-16-plus-240__laion400m_e31",
"ViT-B-16-plus-240__laion400m_e32",
"ViT-B-16__laion400m_e31",
"ViT-B-16__laion400m_e32",
"ViT-B-16__openai",
"ViT-B-32-SigLIP2-256__webli",
"ViT-B-32__laion2b-s34b-b79k",
"ViT-B-32__laion2b_e16",
"ViT-B-32__laion400m_e31",
"ViT-B-32__laion400m_e32",
"ViT-B-32__openai",
"ViT-H-14-378-quickgelu__dfn5b",
"ViT-H-14-quickgelu__dfn5b",
"ViT-H-14__laion2b-s32b-b79k",
"ViT-L-14-336__openai",
"ViT-L-14-quickgelu__dfn2b",
"ViT-L-14__laion2b-s32b-b82k",
"ViT-L-14__laion400m_e31",
"ViT-L-14__laion400m_e32",
"ViT-L-14__openai",
"ViT-L-16-SigLIP-256__webli",
"ViT-L-16-SigLIP-384__webli",
"ViT-L-16-SigLIP2-256__webli",
"ViT-L-16-SigLIP2-384__webli",
"ViT-L-16-SigLIP2-512__webli",
"ViT-SO400M-14-SigLIP-384__webli",
"ViT-SO400M-14-SigLIP2-378__webli",
"ViT-SO400M-14-SigLIP2__webli",
"ViT-SO400M-16-SigLIP2-256__webli",
"ViT-SO400M-16-SigLIP2-384__webli",
"ViT-SO400M-16-SigLIP2-512__webli",
"ViT-gopt-16-SigLIP2-256__webli",
"ViT-gopt-16-SigLIP2-384__webli",
"nllb-clip-base-siglip__mrl",
"nllb-clip-base-siglip__v1",
"nllb-clip-large-siglip__mrl",
"nllb-clip-large-siglip__v1",
"xlm-roberta-base-ViT-B-32__laion5b_s13b_b90k",
"xlm-roberta-large-ViT-H-14__frozen_laion5b_s13b_b90k",
]
insightface = [
"antelopev2",
"buffalo_l",
"buffalo_m",
"buffalo_s",
]
def export_models(models: list[str], source: ModelSource) -> None:
profiling_dir = Path("profiling")
profiling_dir.mkdir(exist_ok=True)
for model in models:
try:
model_dir = f"models/{clean_name(model)}"
task = SOURCE_TO_TASK[source]
print(f"Processing model {model}")
subprocess.check_call(["python", "-m", "immich_model_exporter", "export", model, source])
subprocess.check_call(
[
"python",
"-m",
"immich_model_exporter",
"profile",
model_dir,
task,
"--output_path",
profiling_dir / f"{model}.json",
]
)
subprocess.check_call(["python", "-m", "immich_model_exporter", "upload", model_dir])
except Exception as e:
print(f"Failed to export model {model}: {e}")
if __name__ == "__main__":
export_models(mclip, ModelSource.MCLIP)
export_models(openclip, ModelSource.OPENCLIP)
export_models(insightface, ModelSource.INSIGHTFACE)
Path("results").mkdir(exist_ok=True)
dataset_root = Path("datasets")
dataset_root.mkdir(exist_ok=True)
crossmodal3600_root = dataset_root / "crossmodal3600"
subprocess.check_call(
[
"clip_benchmark",
"eval",
"--pretrained_model",
*[name.replace("__", ",") for name in openclip],
"--task",
"zeroshot_retrieval",
"--dataset",
"crossmodal3600",
"--dataset_root",
crossmodal3600_root.as_posix(),
"--batch_size",
"64",
"--language",
"ar",
"bn",
"cs",
"da",
"de",
"el",
"en",
"es",
"fa",
"fi",
"fil",
"fr",
"he",
"hi",
"hr",
"hu",
"id",
"it",
"ja",
"ko",
"mi",
"nl",
"no",
"pl",
"pt",
"quz",
"ro",
"ru",
"sv",
"sw",
"te",
"th",
"tr",
"uk",
"vi",
"zh",
"--recall_k",
"1",
"5",
"10",
"--no_amp",
"--output",
"results/{dataset}_{language}_{model}_{pretrained}.json",
]
)
xtd10_root = dataset_root / "xtd10"
subprocess.check_call(
[
"clip_benchmark",
"eval",
"--pretrained_model",
*[name.replace("__", ",") for name in openclip],
"--task",
"zeroshot_retrieval",
"--dataset",
"xtd10",
"--dataset_root",
xtd10_root.as_posix(),
"--batch_size",
"64",
"--language",
"de",
"en",
"es",
"fr",
"it",
"jp",
"ko",
"pl",
"ru",
"tr",
"zh",
"--recall_k",
"1",
"5",
"10",
"--no_amp",
"--output",
"results/{dataset}_{language}_{model}_{pretrained}.json",
]
)
flickr30k_root = dataset_root / "flickr30k"
# note: need ~/.kaggle/kaggle.json to download the dataset automatically
subprocess.check_call(
[
"clip_benchmark",
"eval",
"--pretrained_model",
*[name.replace("__", ",") for name in openclip],
"--task",
"zeroshot_retrieval",
"--dataset",
"flickr30k",
"--dataset_root",
flickr30k_root.as_posix(),
"--batch_size",
"64",
"--language",
"en",
"zh",
"--recall_k",
"1",
"5",
"10",
"--no_amp",
"--output",
"results/{dataset}_{language}_{model}_{pretrained}.json",
]
)
+140 -6
View File
@@ -1,11 +1,145 @@
models:
- name: 'LABSE-Vit-L-14'
hf-name: 'M-CLIP/LABSE-Vit-L-14' # Do this for all mclip models
hf-name: 'M-CLIP/LABSE-Vit-L-14'
source: 'mclip'
- name: 'XLM-Roberta-Large-Vit-B-16Plus'
hf-name: 'M-CLIP/XLM-Roberta-Large-Vit-B-16Plus'
source: 'mclip'
- name: 'XLM-Roberta-Large-Vit-B-32'
hf-name: 'M-CLIP/XLM-Roberta-Large-Vit-B-32'
source: 'mclip'
- name: 'XLM-Roberta-Large-Vit-L-14'
hf-name: 'M-CLIP/XLM-Roberta-Large-Vit-L-14'
source: 'mclip'
- name: 'XLM-Roberta-Base-ViT-B-32__laion5b_s13b_b90k'
hf-name: 'xlm-roberta-base-ViT-B-32__laion5b_s13b_b90k' # Do this for both xlm models
source: 'openclip'
- name: 'RN101__openai'
source: 'openclip'
- name: 'antelopev2'
source: 'insightface'
- name: 'buffalo_l'
source: 'insightface'
- name: 'buffalo_m'
source: 'insightface'
- name: 'buffalo_s'
source: 'insightface'
- name: 'RN101__openai'
source: 'openclip'
- name: 'RN101__yfcc15m'
source: 'openclip'
- name: 'RN50__cc12m'
source: 'openclip'
- name: 'RN50__openai'
source: 'openclip'
- name: 'RN50__yfcc15m'
source: 'openclip'
- name: 'RN50x16__openai'
source: 'openclip'
- name: 'RN50x4__openai'
source: 'openclip'
- name: 'RN50x64__openai'
source: 'openclip'
- name: 'ViT-B-16-SigLIP-256__webli'
source: 'openclip'
- name: 'ViT-B-16-SigLIP-384__webli'
source: 'openclip'
- name: 'ViT-B-16-SigLIP-512__webli'
source: 'openclip'
- name: 'ViT-B-16-SigLIP-i18n-256__webli'
source: 'openclip'
- name: 'ViT-B-16-SigLIP2__webli'
source: 'openclip'
- name: 'ViT-B-16-SigLIP__webli'
source: 'openclip'
- name: 'ViT-B-16-plus-240__laion400m_e31'
source: 'openclip'
- name: 'ViT-B-16-plus-240__laion400m_e32'
source: 'openclip'
- name: 'ViT-B-16__laion400m_e31'
source: 'openclip'
- name: 'ViT-B-16__laion400m_e32'
source: 'openclip'
- name: 'ViT-B-16__openai'
source: 'openclip'
- name: 'ViT-B-32-SigLIP2-256__webli'
source: 'openclip'
- name: 'ViT-B-32__laion2b-s34b-b79k'
source: 'openclip'
- name: 'ViT-B-32__laion2b_e16'
source: 'openclip'
- name: 'ViT-B-32__laion400m_e31'
source: 'openclip'
- name: 'ViT-B-32__laion400m_e32'
source: 'openclip'
- name: 'ViT-B-32__openai'
source: 'openclip'
- name: 'ViT-H-14-378-quickgelu__dfn5b'
source: 'openclip'
runner: 'pokedex-large'
- name: 'ViT-H-14-quickgelu__dfn5b'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'ViT-H-14__laion2b-s32b-b79k'
source: 'openclip'
- name: 'ViT-L-14-336__openai'
source: 'openclip'
- name: 'ViT-L-14-quickgelu__dfn2b'
source: 'openclip'
- name: 'ViT-L-14__laion2b-s32b-b82k'
source: 'openclip'
- name: 'ViT-L-14__laion400m_e31'
source: 'openclip'
- name: 'ViT-L-14__laion400m_e32'
source: 'openclip'
- name: 'ViT-L-14__openai'
source: 'openclip'
- name: 'ViT-L-16-SigLIP-256__webli'
source: 'openclip'
- name: 'ViT-L-16-SigLIP-384__webli'
source: 'openclip'
- name: 'ViT-L-16-SigLIP2-256__webli'
source: 'openclip'
- name: 'ViT-L-16-SigLIP2-384__webli'
source: 'openclip'
- name: 'ViT-L-16-SigLIP2-512__webli'
source: 'openclip'
- name: 'ViT-SO400M-14-SigLIP-384__webli'
source: 'openclip'
runner: 'pokedex-large'
- name: 'ViT-SO400M-14-SigLIP2-378__webli'
source: 'openclip'
runner: 'pokedex-large'
- name: 'ViT-SO400M-14-SigLIP2__webli'
source: 'openclip'
runner: 'pokedex-large'
- name: 'ViT-SO400M-16-SigLIP2-256__webli'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'ViT-SO400M-16-SigLIP2-384__webli'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'ViT-SO400M-16-SigLIP2-512__webli'
source: 'openclip'
runner: 'pokedex-large'
- name: 'ViT-gopt-16-SigLIP2-256__webli'
source: 'openclip'
runner: 'pokedex-large'
- name: 'ViT-gopt-16-SigLIP2-384__webli'
source: 'openclip'
runner: 'pokedex-huge'
- name: 'nllb-clip-base-siglip__mrl'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'nllb-clip-base-siglip__v1'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'nllb-clip-large-siglip__mrl'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'nllb-clip-large-siglip__v1'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'XLM-Roberta-Base-ViT-B-32__laion5b_s13b_b90k'
hf-name: 'xlm-roberta-base-ViT-B-32__laion5b_s13b_b90k'
source: 'openclip'
runner: 'pokedex-medium'
- name: 'XLM-Roberta-Large-ViT-H-14__frozen_laion5b_s13b_b90k'
hf-name: 'xlm-roberta-large-ViT-H-14__frozen_laion5b_s13b_b90k'
source: 'openclip'
runner: 'pokedex-medium'
+34 -18
View File
@@ -1,43 +1,59 @@
[project]
name = "immich_model_exporter"
name = "immich_model"
version = "0.1.0"
description = "Add your description here"
description = "Export Immich's CLIP and face models to ONNX and compile them for on-device runtimes."
readme = "README.md"
requires-python = ">=3.10, <4.0"
requires-python = ">=3.10,<3.13"
dependencies = [
"huggingface-hub>=0.29.3",
"multilingual-clip>=1.0.10",
"onnx>=1.14.1",
"onnxruntime>=1.16.0",
"open-clip-torch>=2.31.0",
"typer>=0.15.2",
"rknn-toolkit2>=2.3.0",
"transformers>=4.49.0",
"tenacity>=9.0.0",
"polars>=1.25.2",
"kaggle>=1.7.4.2",
"clip-benchmark",
]
[project.scripts]
immich-model = "immich_model:app"
[dependency-groups]
dev = ["black>=23.3.0", "mypy>=1.3.0", "ruff>=0.0.272"]
[tool.uv]
override-dependencies = [
"onnx>=1.16.0,<2",
"onnxruntime>=1.18.2,<2",
"torch>=2.4",
"torchvision>=0.21",
onnx = [
"onnx>=1.18.0",
"onnxruntime>=1.18.2",
"onnxscript>=0.7.0",
"ml-dtypes>=0.5.0",
"open-clip-torch>=2.31.0",
"multilingual-clip>=1.0.10",
"transformers>=4.49.0",
"clip-benchmark",
"numpy>=2.1.0",
"torch",
"torchvision",
]
rknn = ["rknn-toolkit2>=2.3.0", "setuptools<81", "torch"]
[tool.uv]
# `onnx` and `rknn` have mutually incompatible pins (numpy/protobuf/onnx/torch)
conflicts = [[{ group = "onnx" }, { group = "rknn" }]]
override-dependencies = ["onnxoptimizer>=0.4.2"]
[tool.uv.sources]
clip-benchmark = { git = "https://github.com/mertalev/CLIP_benchmark.git", rev = "77e733ee241399611296d3c4aca583ec89cf5190" }
torch = { index = "pytorch-cpu" }
torchvision = { index = "pytorch-cpu" }
[[tool.uv.index]]
name = "pytorch-cpu"
url = "https://download.pytorch.org/whl/cpu"
explicit = true
[tool.hatch.build.targets.sdist]
include = ["immich_model_exporter"]
include = ["immich_model"]
[tool.hatch.build.targets.wheel]
include = ["immich_model_exporter"]
include = ["immich_model"]
[build-system]
requires = ["hatchling"]
Generated
+1650 -1311
View File
File diff suppressed because it is too large Load Diff