mirror of
https://github.com/immich-app/ml-models.git
synced 2026-09-30 13:22:55 +08:00
feat: full models list in models.yaml (#15)
* feat: full models list in models.yaml * feat: special case runner for large models (#16) * chore: put more models on mich (#17) * fix: pr feedback, formatting * chore: use implicit HF_TOKEN env var * fix python version * update dependencies * fixes * add onnxscript * override numpy * remove sdpa workaround * dependency hell * polish cli * include setuptools * refactor * fix gha * fix: run on pokedex-large instead of mich * fix: concurrency block for push * fix: better runner sizing * fix: enable caching for setup-uv * fix: use pytorch-cpu * fix m-clip * fix gelu, fuse attention * model versioning --------- Co-authored-by: mertalev <101130780+mertalev@users.noreply.github.com>
This commit is contained in:
@@ -28,7 +28,8 @@ module.exports = ({core}) => {
|
||||
if (!m || !n) return false;
|
||||
return m["name"] === n["name"] &&
|
||||
m["source"] === n["source"] &&
|
||||
m["hf-name"] === n["hf-name"];
|
||||
m["hf-name"] === n["hf-name"] &&
|
||||
m["runner"] === n["runner"];
|
||||
}
|
||||
|
||||
for (const key of keys) {
|
||||
@@ -45,6 +46,11 @@ module.exports = ({core}) => {
|
||||
// !n: deleted, which we ignore
|
||||
}
|
||||
|
||||
// Potential shape change:
|
||||
// Instead of this binary choice, output all the models
|
||||
// And annotate each model with which tasks should run for it
|
||||
// (like export, benchmark, upload, etc)
|
||||
// That means every model will make it to the index job through the same path
|
||||
core.setOutput('to_export', JSON.stringify(to_export));
|
||||
core.setOutput('unchanged', JSON.stringify(unchanged));
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
on:
|
||||
workflow_call:
|
||||
secrets:
|
||||
HF_AUTH_TOKEN:
|
||||
HF_TOKEN:
|
||||
required: true
|
||||
inputs:
|
||||
model-name:
|
||||
@@ -10,6 +10,9 @@ on:
|
||||
model-source:
|
||||
required: true
|
||||
type: string
|
||||
runner:
|
||||
required: true
|
||||
type: string
|
||||
hf-name:
|
||||
required: false
|
||||
type: string
|
||||
@@ -19,15 +22,38 @@ on:
|
||||
|
||||
jobs:
|
||||
export:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ${{ inputs.runner }}
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
||||
- run: uv sync # TODO: cache uv env (does setup-uv do that already automatically?)
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: uv.lock
|
||||
|
||||
- run: uv run --group onnx immich-model export "$MODEL_NAME" "$MODEL_SOURCE" --hf-model-name "$HF_MODEL_NAME"
|
||||
env:
|
||||
MODEL_NAME: ${{ inputs.model-name }}
|
||||
MODEL_SOURCE: ${{ inputs.model-source }}
|
||||
HF_MODEL_NAME: ${{ inputs.hf-name || inputs.model-name }}
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN }}
|
||||
|
||||
# CLIP models publish at an opset RKNN can't ingest, so build the RKNN binaries from a lower-opset copy
|
||||
- run: |
|
||||
if [ "$MODEL_SOURCE" = insightface ]; then
|
||||
uv run --group rknn immich-model compile "$MODEL_NAME"
|
||||
else
|
||||
uv run --group onnx immich-model export "$MODEL_NAME" "$MODEL_SOURCE" \
|
||||
--hf-model-name "$HF_MODEL_NAME" --opset 20 --output-dir rknn-build
|
||||
uv run --group rknn immich-model compile "$MODEL_NAME" --input-dir rknn-build
|
||||
fi
|
||||
env:
|
||||
MODEL_NAME: ${{ inputs.model-name }}
|
||||
MODEL_SOURCE: ${{ inputs.model-source }}
|
||||
HF_MODEL_NAME: ${{ inputs.hf-name || inputs.model-name }}
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN }}
|
||||
|
||||
- run: uv run immich_model_exporter export "${{ inputs.model-name }}" "${{ inputs.model-source }}" --hf-model-name "${{ inputs.hf-name }}"
|
||||
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: ${{ inputs.model-name }}
|
||||
@@ -38,14 +64,17 @@ jobs:
|
||||
|
||||
upload:
|
||||
if: ${{ inputs.upload }}
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ${{ inputs.runner }}
|
||||
needs: export
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b # v8.1.0
|
||||
- run: uv sync # TODO: cache uv env (does setup-uv do that already automatically?)
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: uv.lock
|
||||
- run: uv sync
|
||||
|
||||
- uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
@@ -53,6 +82,10 @@ jobs:
|
||||
merge-multiple: 'true' # We don't actually have multiple, but this also sets the subfolder naming behaviour
|
||||
path: models/
|
||||
|
||||
- run: uv run immich_model_exporter upload "${{ inputs.model-name }}" --hf-model-name "${{ inputs.hf-name }}" --hf-organization immich-testing
|
||||
- run: |
|
||||
uv run immich-model upload "$MODEL_NAME" \
|
||||
--hf-model-name "$HF_MODEL_NAME" --hf-organization immich-testing --revision v2
|
||||
env:
|
||||
HF_AUTH_TOKEN: ${{ secrets.HF_AUTH_TOKEN }}
|
||||
MODEL_NAME: ${{ inputs.model-name }}
|
||||
HF_MODEL_NAME: ${{ inputs.hf-name || inputs.model-name }}
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN }}
|
||||
|
||||
@@ -12,7 +12,9 @@ on:
|
||||
required: false
|
||||
description: 'Force export all models'
|
||||
|
||||
# TODO: concurrency block?
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
jobs:
|
||||
configure:
|
||||
@@ -59,13 +61,14 @@ jobs:
|
||||
force: ${{ inputs.force || 'false' }}
|
||||
oldModels: ${{ steps.old-models.outputs.result }}
|
||||
newModels: ${{ steps.new-models.outputs.result }}
|
||||
oldHash: ${{ hashFiles('./before/immich_model_exporter/exporters/**') }}
|
||||
newHash: ${{ hashFiles('./immich_model_exporter/exporters/**') }}
|
||||
oldHash: ${{ hashFiles('./before/immich_model/exporters/**') }}
|
||||
newHash: ${{ hashFiles('./immich_model/exporters/**') }}
|
||||
script: |
|
||||
const script = require('./.github/scripts/scope.js')
|
||||
script({core})
|
||||
|
||||
|
||||
# TODO: Explicitly skip this if nothing to do
|
||||
export:
|
||||
uses: ./.github/workflows/export.yaml
|
||||
needs: configure
|
||||
@@ -80,4 +83,5 @@ jobs:
|
||||
model-name: ${{ matrix.name }}
|
||||
model-source: ${{ matrix.source }}
|
||||
hf-name: ${{ matrix.hf-name }}
|
||||
runner: ${{ matrix.runner || 'ubuntu-latest'}}
|
||||
upload: ${{ inputs.force || github.event_name == 'release'}}
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@ onnx__*
|
||||
*.latent
|
||||
*.pos_embed
|
||||
vocab.txt
|
||||
immich_model_exporter/models/**/README.md
|
||||
immich_model/models/**/README.md
|
||||
tokenizer.json
|
||||
tokenizer_config.json
|
||||
special_tokens_map.json
|
||||
|
||||
+1
-1
@@ -1 +1 @@
|
||||
3.14
|
||||
3.12
|
||||
|
||||
@@ -0,0 +1,164 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Annotated
|
||||
|
||||
from typer import Argument, Exit, Option, Typer, echo
|
||||
|
||||
from .constants import DELETE_PATTERNS, SOURCE_TO_METADATA, ModelFormat, ModelSource
|
||||
|
||||
app = Typer(
|
||||
no_args_is_help=True,
|
||||
pretty_exceptions_show_locals=False,
|
||||
help="Export models used by Immich to ONNX and compile them for on-device runtimes.",
|
||||
)
|
||||
|
||||
ModelName = Annotated[str, Argument(help="Model name; also the per-model output subdirectory name.")]
|
||||
OutputDir = Annotated[Path, Option(help="Base directory holding per-model output directories.")]
|
||||
Cache = Annotated[bool, Option(help="Reuse existing outputs instead of regenerating them.")]
|
||||
|
||||
|
||||
def generate_readme(model_name: str, model_source: ModelSource) -> str:
|
||||
name, link, type = SOURCE_TO_METADATA[model_source]
|
||||
match model_source:
|
||||
case ModelSource.MCLIP:
|
||||
tags = ["immich", "clip", "multilingual"]
|
||||
case ModelSource.OPENCLIP:
|
||||
tags = ["immich", "clip"]
|
||||
lowered = model_name.lower()
|
||||
if "xlm" in lowered or "nllb" in lowered:
|
||||
tags.append("multilingual")
|
||||
case ModelSource.INSIGHTFACE:
|
||||
tags = ["immich", "facial-recognition"]
|
||||
case _:
|
||||
raise ValueError(f"Unsupported model source {model_source}")
|
||||
|
||||
return f"""---
|
||||
tags:
|
||||
{" - " + "\n - ".join(tags)}
|
||||
---
|
||||
# Model Description
|
||||
|
||||
This repo contains ONNX exports for the associated {type} model by {name}. See the [{name}]({link}) repo for more info.
|
||||
|
||||
This repo is specifically intended for use with [Immich](https://immich.app/), a self-hosted photo library.
|
||||
"""
|
||||
|
||||
|
||||
@app.command()
|
||||
def export(
|
||||
model_name: ModelName,
|
||||
model_source: Annotated[ModelSource, Argument(help="Upstream source the model comes from.")],
|
||||
hf_model_name: Annotated[str | None, Option(help="Hugging Face repo to fetch; defaults to model_name.")] = None,
|
||||
output_dir: OutputDir = Path("models"),
|
||||
opset: Annotated[int, Option(help="ONNX opset for the exported model.")] = 23,
|
||||
cache: Cache = True,
|
||||
) -> None:
|
||||
"""Export a model to ONNX (plus tokenizer/config) under <output-dir>/<model-name>."""
|
||||
from . import onnx
|
||||
|
||||
if not hf_model_name:
|
||||
hf_model_name = model_name
|
||||
output_dir = output_dir / model_name
|
||||
match model_source:
|
||||
case ModelSource.MCLIP | ModelSource.OPENCLIP:
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
onnx.export(hf_model_name, model_source, output_dir, opset=opset, cache=cache)
|
||||
case ModelSource.INSIGHTFACE:
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
# TODO: start from insightface dump instead of downloading from HF
|
||||
snapshot_download(f"immich-app/{hf_model_name}", local_dir=output_dir)
|
||||
case _:
|
||||
raise ValueError(f"Unsupported model source {model_source}")
|
||||
|
||||
readme_path = output_dir / "README.md"
|
||||
if not (cache or readme_path.exists()):
|
||||
with open(readme_path, "w") as f:
|
||||
f.write(generate_readme(model_name, model_source))
|
||||
|
||||
|
||||
@app.command()
|
||||
def compile(
|
||||
model_name: ModelName,
|
||||
input_dir: Annotated[
|
||||
Path, Option(help="Base directory holding the ONNX to compile; defaults to --output-dir.")
|
||||
] = Path("models"),
|
||||
output_dir: OutputDir = Path("models"),
|
||||
cache: Cache = True,
|
||||
) -> None:
|
||||
"""Compile an exported ONNX model into a device binary, writing <output-dir>/<model-name>/**/rknpu."""
|
||||
from . import rknn
|
||||
|
||||
try:
|
||||
rknn.compile(input_dir / model_name, output_dir / model_name, cache=cache)
|
||||
except Exception as e:
|
||||
echo(f"Failed to compile {model_name} to RKNN: {e}", err=True)
|
||||
raise Exit(code=1)
|
||||
|
||||
|
||||
@app.command()
|
||||
def profile(
|
||||
model_name: ModelName,
|
||||
model_format: Annotated[ModelFormat, Option("--format", help="Artifact format to profile.")] = ModelFormat.ONNX,
|
||||
base_dir: OutputDir = Path("models"),
|
||||
soc: Annotated[str, Option(help="RKNN target SoC (only for --format rknn; needs an attached NPU).")] = "rk3588",
|
||||
output_path: Annotated[
|
||||
Path | None, Option(help="Profile JSON path; defaults to profiling/<model-name>.<format>.json.")
|
||||
] = None,
|
||||
) -> None:
|
||||
"""Benchmark an exported model per-node/per-layer, writing a JSON report."""
|
||||
model_dir = base_dir / model_name
|
||||
match model_format:
|
||||
case ModelFormat.ONNX:
|
||||
from . import onnx
|
||||
|
||||
result = onnx.profile(model_dir)
|
||||
case ModelFormat.RKNN:
|
||||
from . import rknn
|
||||
|
||||
result = rknn.profile(model_dir, soc)
|
||||
case _:
|
||||
raise ValueError(f"Profiling not supported for format {model_format}")
|
||||
|
||||
if output_path is None:
|
||||
output_path = Path("profiling") / f"{model_name}.{model_format}.json"
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
output_path.write_text(json.dumps(result, indent=2))
|
||||
|
||||
for sub, data in result["submodels"].items():
|
||||
echo(f" {sub}: {data['summary']}")
|
||||
echo(f"wrote {output_path}")
|
||||
|
||||
|
||||
@app.command()
|
||||
def upload(
|
||||
model_name: ModelName,
|
||||
hf_branch: Annotated[str, Option(help="Repo branch to upload to.")],
|
||||
hf_model_name: Annotated[str | None, Option(help="Target Hugging Face repo name; defaults to model_name.")] = None,
|
||||
hf_organization: Annotated[str, Option(help="Hugging Face organization to upload under.")] = "immich-app",
|
||||
input_dir: Annotated[Path, Option(help="Base directory holding the exported model directory.")] = Path("models"),
|
||||
) -> None:
|
||||
"""Upload an exported model directory (<input-dir>/<model-name>) to a Hugging Face repo."""
|
||||
from huggingface_hub import create_branch, create_repo, upload_folder
|
||||
from tenacity import retry, stop_after_attempt, wait_fixed
|
||||
|
||||
if not hf_model_name:
|
||||
hf_model_name = model_name
|
||||
model_dir = input_dir / model_name
|
||||
repo_id = f"{hf_organization}/{hf_model_name}"
|
||||
|
||||
@retry(stop=stop_after_attempt(5), wait=wait_fixed(5))
|
||||
def upload_model() -> None:
|
||||
create_repo(repo_id, exist_ok=True)
|
||||
if hf_branch != "main":
|
||||
create_branch(repo_id, branch=hf_branch, exist_ok=True)
|
||||
upload_folder(
|
||||
repo_id=repo_id,
|
||||
folder_path=model_dir,
|
||||
revision=hf_branch,
|
||||
# remote repo files to be deleted before uploading
|
||||
# deletion is in the same commit as the upload, so it's atomic
|
||||
delete_patterns=DELETE_PATTERNS,
|
||||
)
|
||||
|
||||
upload_model()
|
||||
@@ -0,0 +1,3 @@
|
||||
from immich_model import app
|
||||
|
||||
app()
|
||||
@@ -13,6 +13,11 @@ class ModelTask(StrEnum):
|
||||
SEARCH = "clip"
|
||||
|
||||
|
||||
class ModelFormat(StrEnum):
|
||||
ONNX = "onnx"
|
||||
RKNN = "rknn"
|
||||
|
||||
|
||||
class SourceMetadata(NamedTuple):
|
||||
name: str
|
||||
link: str
|
||||
@@ -0,0 +1,4 @@
|
||||
from .export import export
|
||||
from .profile import profile
|
||||
|
||||
__all__ = ["export", "profile"]
|
||||
@@ -4,17 +4,15 @@ from ..constants import ModelSource
|
||||
from .models import mclip, openclip
|
||||
|
||||
|
||||
def export(
|
||||
model_name: str, model_source: ModelSource, output_dir: Path, opset_version: int = 19, cache: bool = True
|
||||
) -> None:
|
||||
def export(model_name: str, model_source: ModelSource, output_dir: Path, opset: int, cache: bool = True) -> None:
|
||||
visual_dir = output_dir / "visual"
|
||||
textual_dir = output_dir / "textual"
|
||||
match model_source:
|
||||
case ModelSource.MCLIP:
|
||||
mclip.to_onnx(model_name, opset_version, visual_dir, textual_dir, cache=cache)
|
||||
mclip.to_onnx(model_name, opset, visual_dir, textual_dir, cache=cache)
|
||||
case ModelSource.OPENCLIP:
|
||||
name, _, pretrained = model_name.partition("__")
|
||||
config = openclip.OpenCLIPModelConfig(name, pretrained)
|
||||
openclip.to_onnx(config, opset_version, visual_dir, textual_dir, cache=cache)
|
||||
openclip.to_onnx(config, opset, visual_dir, textual_dir, cache=cache)
|
||||
case _:
|
||||
raise ValueError(f"Unsupported model source {model_source}")
|
||||
+23
-7
@@ -23,13 +23,9 @@ def to_onnx(
|
||||
) -> tuple[Path, Path]:
|
||||
textual_path = get_model_path(output_dir_textual)
|
||||
if not cache or not textual_path.exists():
|
||||
import torch
|
||||
from multilingual_clip.pt_multilingual_clip import MultilingualCLIP
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
torch.backends.mha.set_fastpath_enabled(False)
|
||||
|
||||
model = MultilingualCLIP.from_pretrained(model_name)
|
||||
model = _load_model(model_name)
|
||||
AutoTokenizer.from_pretrained(model_name).save_pretrained(output_dir_textual)
|
||||
|
||||
model.eval()
|
||||
@@ -39,11 +35,31 @@ def to_onnx(
|
||||
_export_text_encoder(model, textual_path, opset_version)
|
||||
else:
|
||||
print(f"Model {textual_path} already exists, skipping")
|
||||
visual_path, _ = openclip_to_onnx(_MCLIP_TO_OPENCLIP[model_name], opset_version, output_dir_visual, cache=cache)
|
||||
# Keep the original activation since M-CLIP's text encoder was aligned against it
|
||||
visual_path, _ = openclip_to_onnx(
|
||||
_MCLIP_TO_OPENCLIP[model_name], opset_version, output_dir_visual, cache=cache, force_quick_gelu=False
|
||||
)
|
||||
assert visual_path is not None, "Visual model export failed"
|
||||
return visual_path, textual_path
|
||||
|
||||
|
||||
def _load_model(model_name: str) -> Any:
|
||||
# transformers 5 breaks multilingual_clip, so we instantiate the model manually.
|
||||
import torch
|
||||
from huggingface_hub import hf_hub_download
|
||||
from multilingual_clip import Config_MCLIP
|
||||
from multilingual_clip.pt_multilingual_clip import MultilingualCLIP
|
||||
|
||||
config = Config_MCLIP.MCLIPConfig.from_pretrained(model_name)
|
||||
model = MultilingualCLIP(config)
|
||||
|
||||
weights_path = hf_hub_download(model_name, "pytorch_model.bin")
|
||||
state_dict = torch.load(weights_path, map_location="cpu", weights_only=True)
|
||||
missing, _ = model.load_state_dict(state_dict, strict=False)
|
||||
assert not missing, f"Missing weights when loading {model_name}: {missing}"
|
||||
return model
|
||||
|
||||
|
||||
def _export_text_encoder(model: Any, output_path: Path | str, opset_version: int) -> None:
|
||||
import torch
|
||||
from multilingual_clip.pt_multilingual_clip import MultilingualCLIP
|
||||
@@ -68,7 +84,7 @@ def _export_text_encoder(model: Any, output_path: Path | str, opset_version: int
|
||||
args,
|
||||
output_path.as_posix(),
|
||||
input_names=["input_ids", "attention_mask"],
|
||||
output_names=["embedding"],
|
||||
output_names=["text_embedding"],
|
||||
opset_version=opset_version,
|
||||
# dynamic_axes={
|
||||
# "input_ids": {0: "batch_size", 1: "sequence_length"},
|
||||
+4
-5
@@ -38,6 +38,7 @@ def to_onnx(
|
||||
output_dir_visual: Path | str | None = None,
|
||||
output_dir_textual: Path | str | None = None,
|
||||
cache: bool = True,
|
||||
force_quick_gelu: bool | None = None,
|
||||
) -> tuple[Path | None, Path | None]:
|
||||
visual_path = None
|
||||
textual_path = None
|
||||
@@ -54,14 +55,12 @@ def to_onnx(
|
||||
return visual_path, textual_path
|
||||
|
||||
import open_clip
|
||||
import torch
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
torch.backends.mha.set_fastpath_enabled(False)
|
||||
|
||||
model = open_clip.create_model(
|
||||
model_cfg.name,
|
||||
pretrained=model_cfg.pretrained,
|
||||
force_quick_gelu=force_quick_gelu or model_cfg.pretrained == "openai",
|
||||
jit=False,
|
||||
require_pretrained=True,
|
||||
)
|
||||
@@ -116,7 +115,7 @@ def _export_image_encoder(
|
||||
args,
|
||||
output_path.as_posix(),
|
||||
input_names=["image"],
|
||||
output_names=["embedding"],
|
||||
output_names=["image_embedding"],
|
||||
opset_version=opset_version,
|
||||
# dynamic_axes={"image": {0: "batch_size"}},
|
||||
)
|
||||
@@ -145,7 +144,7 @@ def _export_text_encoder(
|
||||
args,
|
||||
output_path.as_posix(),
|
||||
input_names=["text"],
|
||||
output_names=["embedding"],
|
||||
output_names=["text_embedding"],
|
||||
opset_version=opset_version,
|
||||
# dynamic_axes={"text": {0: "batch_size"}},
|
||||
)
|
||||
@@ -0,0 +1,54 @@
|
||||
import json
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
# per-model subdirectories that hold a model.onnx
|
||||
SUBMODELS = ["textual", "visual", "detection", "recognition"]
|
||||
|
||||
|
||||
def profile(model_dir: Path, runs: int = 20, provider: str = "CPUExecutionProvider") -> dict[str, Any]:
|
||||
"""Profile every ONNX submodel via onnxruntime, returning per-op-node costs.
|
||||
|
||||
Uses onnxruntime's built-in profiler (enable_profiling), which records a
|
||||
kernel-time trace per node; costs are aggregated by op type.
|
||||
"""
|
||||
import numpy as np
|
||||
import onnxruntime as ort
|
||||
|
||||
subs = [s for s in SUBMODELS if (model_dir / s / "model.onnx").is_file()]
|
||||
if not subs:
|
||||
raise RuntimeError(f"No ONNX model found under {model_dir}")
|
||||
|
||||
result: dict[str, Any] = {"model": model_dir.name, "format": "onnx", "provider": provider, "submodels": {}}
|
||||
for sub in subs:
|
||||
so = ort.SessionOptions()
|
||||
so.enable_profiling = True
|
||||
sess = ort.InferenceSession((model_dir / sub / "model.onnx").as_posix(), sess_options=so, providers=[provider])
|
||||
|
||||
feeds = {i.name: _rand_input(i, np) for i in sess.get_inputs()}
|
||||
for _ in range(runs): # profiling records every run; the first (cold) run is averaged in
|
||||
sess.run(None, feeds)
|
||||
events = json.load(open(sess.end_profiling()))
|
||||
|
||||
per_op: dict[str, list[float]] = defaultdict(lambda: [0, 0.0])
|
||||
for e in events:
|
||||
if e.get("cat") == "Node" and e.get("name", "").endswith("_kernel_time"):
|
||||
op = e["args"].get("op_name", "?")
|
||||
per_op[op][0] += 1
|
||||
per_op[op][1] += e["dur"]
|
||||
|
||||
ranked = sorted(per_op.items(), key=lambda kv: -kv[1][1])
|
||||
ops = [{"op": op, "count": int(n // runs), "us_per_run": round(us / runs, 1)} for op, (n, us) in ranked]
|
||||
mean_ms = round(sum(us for _, us in per_op.values()) / runs / 1000, 3)
|
||||
hottest = f", hottest {ops[0]['op']} ({ops[0]['us_per_run']}us)" if ops else ""
|
||||
result["submodels"][sub] = {"summary": f"{mean_ms} ms/run{hottest}", "mean_ms": mean_ms, "ops": ops}
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def _rand_input(node: Any, np: Any) -> Any:
|
||||
shape = [d if isinstance(d, int) else 1 for d in node.shape]
|
||||
if "int" in node.type: # token ids etc. — values don't matter for latency
|
||||
return np.zeros(shape, dtype=np.int64 if "int64" in node.type else np.int32)
|
||||
return np.random.rand(*shape).astype(np.float32)
|
||||
@@ -0,0 +1,4 @@
|
||||
from .compile import compile
|
||||
from .profile import profile
|
||||
|
||||
__all__ = ["compile", "profile"]
|
||||
@@ -0,0 +1,84 @@
|
||||
import math
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
# tanh-GELU coefficients
|
||||
_GELU_C0 = math.sqrt(2 / math.pi)
|
||||
_GELU_C1 = 0.044715
|
||||
|
||||
|
||||
def prepare_for_rknn(onnx_path: Path, work_dir: Path) -> Path:
|
||||
"""Return an ONNX path that ``rknn.build`` can ingest.
|
||||
|
||||
rknn-toolkit2 rejects opset > 19 and has no NPU `Erf` kernel, so each native `Gelu` node
|
||||
is rewritten into its tanh approximation and the opset is pinned to 19.
|
||||
"""
|
||||
import onnx
|
||||
|
||||
model = onnx.load(onnx_path.as_posix())
|
||||
opset = max((o.version for o in model.opset_import if o.domain in ("", "ai.onnx")), default=0)
|
||||
has_gelu = any(node.op_type == "Gelu" for node in model.graph.node)
|
||||
if opset <= 19 and not has_gelu:
|
||||
return onnx_path
|
||||
|
||||
if has_gelu:
|
||||
_decompose_gelu_to_tanh(model)
|
||||
_pin_opset(model, 19)
|
||||
|
||||
out_path = work_dir / "model.onnx"
|
||||
onnx.save(
|
||||
model,
|
||||
out_path.as_posix(),
|
||||
save_as_external_data=True,
|
||||
all_tensors_to_one_file=True,
|
||||
location="model.onnx.data",
|
||||
)
|
||||
return out_path
|
||||
|
||||
|
||||
def _decompose_gelu_to_tanh(model: Any) -> None:
|
||||
"""Replace every native `Gelu` node with its tanh-approximation subgraph, in place."""
|
||||
import numpy as np
|
||||
from onnx import helper, numpy_helper
|
||||
|
||||
graph = model.graph
|
||||
consts: dict[str, Any] = {}
|
||||
|
||||
def const(name: str, value: float) -> str:
|
||||
if name not in consts:
|
||||
consts[name] = numpy_helper.from_array(np.array(value, dtype=np.float32), name)
|
||||
return name
|
||||
|
||||
new_nodes = []
|
||||
for node in graph.node:
|
||||
if node.op_type != "Gelu":
|
||||
new_nodes.append(node)
|
||||
continue
|
||||
x, y = node.input[0], node.output[0]
|
||||
p = node.name or y
|
||||
c0, c1 = const("gelu_c0", _GELU_C0), const("gelu_c1", _GELU_C1)
|
||||
half, one = const("gelu_half", 0.5), const("gelu_one", 1.0)
|
||||
new_nodes += [
|
||||
helper.make_node("Mul", [x, x], [f"{p}_x2"]),
|
||||
helper.make_node("Mul", [f"{p}_x2", x], [f"{p}_x3"]),
|
||||
helper.make_node("Mul", [f"{p}_x3", c1], [f"{p}_c1x3"]),
|
||||
helper.make_node("Add", [x, f"{p}_c1x3"], [f"{p}_inner"]),
|
||||
helper.make_node("Mul", [f"{p}_inner", c0], [f"{p}_scaled"]),
|
||||
helper.make_node("Tanh", [f"{p}_scaled"], [f"{p}_tanh"]),
|
||||
helper.make_node("Add", [f"{p}_tanh", one], [f"{p}_1ptanh"]),
|
||||
helper.make_node("Mul", [x, half], [f"{p}_halfx"]),
|
||||
helper.make_node("Mul", [f"{p}_halfx", f"{p}_1ptanh"], [y]),
|
||||
]
|
||||
del graph.node[:]
|
||||
graph.node.extend(new_nodes)
|
||||
graph.initializer.extend(consts.values())
|
||||
|
||||
|
||||
def _pin_opset(model: Any, version: int) -> None:
|
||||
"""Force the ai.onnx opset to `version`."""
|
||||
from onnx import helper
|
||||
|
||||
kept = [opset for opset in model.opset_import if opset.domain not in ("", "ai.onnx")]
|
||||
kept.append(helper.make_operatorsetid("", version))
|
||||
del model.opset_import[:]
|
||||
model.opset_import.extend(kept)
|
||||
@@ -0,0 +1,113 @@
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
from ..constants import RKNN_SOCS
|
||||
from ._onnx import prepare_for_rknn
|
||||
|
||||
|
||||
def _export_platform(
|
||||
onnx_path: Path,
|
||||
output_dir: Path,
|
||||
target_platform: str,
|
||||
inputs: list[str] | None = None,
|
||||
input_size_list: list[list[int]] | None = None,
|
||||
fuse_matmul_softmax_matmul_to_sdpa: bool = True,
|
||||
) -> None:
|
||||
from rknn.api import RKNN
|
||||
|
||||
output_path = output_dir / "rknpu" / target_platform / "model.rknn"
|
||||
print(f"Exporting {onnx_path} to {output_path}")
|
||||
|
||||
def check(ret: int, step: str) -> None:
|
||||
if ret != 0:
|
||||
raise RuntimeError(f"RKNN {step} failed for {target_platform} (code {ret})")
|
||||
|
||||
rknn = RKNN(verbose=False)
|
||||
rknn.config(
|
||||
target_platform=target_platform,
|
||||
disable_rules=[] if fuse_matmul_softmax_matmul_to_sdpa else ["fuse_matmul_softmax_matmul_to_sdpa"],
|
||||
enable_flash_attention=False,
|
||||
model_pruning=True,
|
||||
)
|
||||
check(rknn.load_onnx(model=onnx_path.as_posix(), inputs=inputs, input_size_list=input_size_list), "load")
|
||||
check(rknn.build(do_quantization=False), "build")
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
check(rknn.export_rknn(output_path.as_posix()), "export")
|
||||
|
||||
|
||||
def _export_platforms(
|
||||
input_dir: Path,
|
||||
output_dir: Path,
|
||||
inputs: list[str] | None = None,
|
||||
input_size_list: list[list[int]] | None = None,
|
||||
cache: bool = True,
|
||||
) -> None:
|
||||
socs = []
|
||||
for soc in RKNN_SOCS:
|
||||
if cache and (model_path := output_dir / "rknpu" / soc / "model.rknn").exists():
|
||||
print(f"{model_path} already exists, skipping")
|
||||
else:
|
||||
socs.append(soc)
|
||||
if not socs:
|
||||
return
|
||||
|
||||
with tempfile.TemporaryDirectory() as work_dir:
|
||||
# rknn.build can't take opset > 19 or exact GELU, so normalise the ONNX once for all SoCs.
|
||||
onnx_path = prepare_for_rknn(input_dir / "model.onnx", Path(work_dir))
|
||||
|
||||
def attempt(soc: str, fuse: bool) -> None:
|
||||
_export_platform(
|
||||
onnx_path,
|
||||
output_dir,
|
||||
soc,
|
||||
inputs=inputs,
|
||||
input_size_list=input_size_list,
|
||||
fuse_matmul_softmax_matmul_to_sdpa=fuse,
|
||||
)
|
||||
|
||||
fuse = True
|
||||
failed: list[str] = []
|
||||
for soc in socs:
|
||||
try:
|
||||
attempt(soc, fuse)
|
||||
except Exception as e:
|
||||
# This fusion isn't valid for every model; drop it (for this and later SoCs) and retry.
|
||||
if fuse and "inputs or 'outputs' must be set" in str(e):
|
||||
print(f"Retrying {soc} without fuse_matmul_softmax_matmul_to_sdpa")
|
||||
fuse = False
|
||||
try:
|
||||
attempt(soc, fuse)
|
||||
continue
|
||||
except Exception as retry_error:
|
||||
e = retry_error
|
||||
print(f"Failed to export {input_dir.name} for {soc}: {e}")
|
||||
failed.append(soc)
|
||||
|
||||
if failed:
|
||||
raise RuntimeError(f"RKNN export failed for {input_dir.name} on: {', '.join(failed)}")
|
||||
|
||||
|
||||
def compile(input_dir: Path, output_dir: Path, cache: bool = True) -> None:
|
||||
"""Compile each ONNX submodel under input_dir into RKNN binaries under output_dir/<sub>/rknpu."""
|
||||
# (subdirectory, inputs, input_size_list) — inputs/sizes are only needed for the face models
|
||||
sub_models: list[tuple[str, list[str] | None, list[list[int]] | None]] = [
|
||||
("textual", None, None),
|
||||
("visual", None, None),
|
||||
("detection", ["input.1"], [[1, 3, 640, 640]]),
|
||||
("recognition", ["input.1"], [[1, 3, 112, 112]]),
|
||||
]
|
||||
present = [(sub, inputs, sizes) for sub, inputs, sizes in sub_models if (input_dir / sub).is_dir()]
|
||||
if not present:
|
||||
raise RuntimeError(f"No exportable model found under {input_dir}")
|
||||
|
||||
errors: list[str] = []
|
||||
for sub, inputs, input_size_list in present:
|
||||
try:
|
||||
_export_platforms(
|
||||
input_dir / sub, output_dir / sub, inputs=inputs, input_size_list=input_size_list, cache=cache
|
||||
)
|
||||
except Exception as e:
|
||||
errors.append(str(e))
|
||||
|
||||
if errors:
|
||||
raise RuntimeError("; ".join(errors))
|
||||
@@ -0,0 +1,67 @@
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
# per-model subdirectories that hold rknpu/<soc>/model.rknn
|
||||
SUBMODELS = ["textual", "visual", "detection", "recognition"]
|
||||
|
||||
|
||||
def profile(model_dir: Path, soc: str = "rk3588") -> dict[str, Any]:
|
||||
"""Profile every RKNN submodel on an attached NPU via eval_perf (per-op NPU/CPU costs)."""
|
||||
from rknn.api import RKNN
|
||||
|
||||
subs = [s for s in SUBMODELS if (model_dir / s / "rknpu" / soc / "model.rknn").is_file()]
|
||||
if not subs:
|
||||
raise RuntimeError(f"No RKNN model for {soc} found under {model_dir}")
|
||||
|
||||
result: dict[str, Any] = {"model": model_dir.name, "format": "rknn", "soc": soc, "submodels": {}}
|
||||
for sub in subs:
|
||||
rknn = RKNN(verbose=False)
|
||||
try:
|
||||
if rknn.load_rknn((model_dir / sub / "rknpu" / soc / "model.rknn").as_posix()) != 0:
|
||||
raise RuntimeError(f"load_rknn failed for {sub}")
|
||||
if rknn.init_runtime(target=soc, perf_debug=True) != 0:
|
||||
raise RuntimeError(f"init_runtime failed for {sub} (is a {soc} NPU attached?)")
|
||||
result["submodels"][sub] = _parse_perf(rknn.eval_perf(is_print=False))
|
||||
finally:
|
||||
rknn.release()
|
||||
return result
|
||||
|
||||
|
||||
def _parse_perf(report: str) -> dict[str, Any]:
|
||||
"""Parse eval_perf's per-op-type summary table + CPU/NPU totals out of its report string."""
|
||||
ops: list[dict[str, Any]] = []
|
||||
totals = {"cpu_us": 0, "npu_us": 0, "total_us": 0}
|
||||
in_table = False
|
||||
for line in report.splitlines():
|
||||
s = line.strip()
|
||||
if s.startswith("OpType") and "CallNumber" in s:
|
||||
in_table = True
|
||||
continue
|
||||
if not in_table or not s or s.startswith("-"):
|
||||
continue
|
||||
parts = s.split()
|
||||
if parts[0] == "Total": # Total <cpu> <gpu> <npu> <total>
|
||||
nums = [int(p) for p in parts[1:] if p.lstrip("-").isdigit()]
|
||||
if len(nums) >= 4:
|
||||
totals = {"cpu_us": nums[0], "npu_us": nums[2], "total_us": nums[3]}
|
||||
break
|
||||
if len(parts) >= 6 and parts[1].isdigit(): # <op> <calls> <cpu> <gpu> <npu> <total> <ratio%>
|
||||
ops.append(
|
||||
{
|
||||
"op": parts[0],
|
||||
"calls": int(parts[1]),
|
||||
"cpu_us": int(parts[2]),
|
||||
"npu_us": int(parts[4]),
|
||||
"total_us": int(parts[5]),
|
||||
}
|
||||
)
|
||||
|
||||
total = totals["total_us"] or 1
|
||||
cpu_pct, npu_pct = round(100 * totals["cpu_us"] / total), round(100 * totals["npu_us"] / total)
|
||||
return {
|
||||
"summary": f"{totals['total_us'] / 1000:.1f} ms/frame (CPU {cpu_pct}%, NPU {npu_pct}%)",
|
||||
"total_ms": round(totals["total_us"] / 1000, 2),
|
||||
"cpu_pct": cpu_pct,
|
||||
"npu_pct": npu_pct,
|
||||
"ops": ops,
|
||||
}
|
||||
@@ -1,171 +0,0 @@
|
||||
import json
|
||||
import resource
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
from tenacity import retry, stop_after_attempt, wait_fixed
|
||||
from typing_extensions import Annotated
|
||||
|
||||
from .exporters.constants import DELETE_PATTERNS, SOURCE_TO_METADATA, ModelSource, ModelTask
|
||||
from .exporters.onnx import export as onnx_export
|
||||
from .exporters.rknn import export as rknn_export
|
||||
|
||||
app = typer.Typer(pretty_exceptions_show_locals=False)
|
||||
|
||||
|
||||
def generate_readme(model_name: str, model_source: ModelSource) -> str:
|
||||
(name, link, type) = SOURCE_TO_METADATA[model_source]
|
||||
match model_source:
|
||||
case ModelSource.MCLIP:
|
||||
tags = ["immich", "clip", "multilingual"]
|
||||
case ModelSource.OPENCLIP:
|
||||
tags = ["immich", "clip"]
|
||||
lowered = model_name.lower()
|
||||
if "xlm" in lowered or "nllb" in lowered:
|
||||
tags.append("multilingual")
|
||||
case ModelSource.INSIGHTFACE:
|
||||
tags = ["immich", "facial-recognition"]
|
||||
case _:
|
||||
raise ValueError(f"Unsupported model source {model_source}")
|
||||
|
||||
return f"""---
|
||||
tags:
|
||||
{" - " + "\n - ".join(tags)}
|
||||
---
|
||||
# Model Description
|
||||
|
||||
This repo contains ONNX exports for the associated {type} model by {name}. See the [{name}]({link}) repo for more info.
|
||||
|
||||
This repo is specifically intended for use with [Immich](https://immich.app/), a self-hosted photo library.
|
||||
"""
|
||||
|
||||
|
||||
@app.command()
|
||||
def export(
|
||||
model_name: str,
|
||||
model_source: ModelSource,
|
||||
hf_model_name: str | None = None,
|
||||
output_dir: Path = Path("models"),
|
||||
cache: bool = True,
|
||||
) -> None:
|
||||
if not hf_model_name:
|
||||
hf_model_name = model_name
|
||||
output_dir = output_dir / model_name
|
||||
match model_source:
|
||||
case ModelSource.MCLIP | ModelSource.OPENCLIP:
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
onnx_export(hf_model_name, model_source, output_dir, cache=cache)
|
||||
case ModelSource.INSIGHTFACE:
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
# TODO: start from insightface dump instead of downloading from HF
|
||||
snapshot_download(f"immich-app/{hf_model_name}", local_dir=output_dir)
|
||||
case _:
|
||||
raise ValueError(f"Unsupported model source {model_source}")
|
||||
|
||||
try:
|
||||
rknn_export(output_dir, cache=cache)
|
||||
except Exception as e:
|
||||
print(f"Failed to export model {model_name} to rknn: {e}")
|
||||
(output_dir / "rknpu").unlink(missing_ok=True)
|
||||
|
||||
readme_path = output_dir / "README.md"
|
||||
if not (cache or readme_path.exists()):
|
||||
with open(readme_path, "w") as f:
|
||||
f.write(generate_readme(model_name, model_source))
|
||||
|
||||
|
||||
# TODO: Args shape parity with the other commands? (eg taking model_name, default base dir)
|
||||
@app.command()
|
||||
def profile(model_dir: Path, model_task: ModelTask, output_path: Path) -> None:
|
||||
from timeit import timeit
|
||||
|
||||
import numpy as np
|
||||
import onnxruntime as ort
|
||||
|
||||
np.random.seed(0)
|
||||
|
||||
sess_options = ort.SessionOptions()
|
||||
sess_options.enable_cpu_mem_arena = False
|
||||
providers = ["CPUExecutionProvider"]
|
||||
provider_options = [{"arena_extend_strategy": "kSameAsRequested"}]
|
||||
match model_task:
|
||||
case ModelTask.SEARCH:
|
||||
textual = ort.InferenceSession(
|
||||
model_dir / "textual" / "model.onnx",
|
||||
sess_options=sess_options,
|
||||
providers=providers,
|
||||
provider_options=provider_options,
|
||||
)
|
||||
tokens = {node.name: np.random.rand(*node.shape).astype(np.int32) for node in textual.get_inputs()}
|
||||
|
||||
visual = ort.InferenceSession(
|
||||
model_dir / "visual" / "model.onnx",
|
||||
sess_options=sess_options,
|
||||
providers=providers,
|
||||
provider_options=provider_options,
|
||||
)
|
||||
image = {node.name: np.random.rand(*node.shape).astype(np.float32) for node in visual.get_inputs()}
|
||||
|
||||
def predict() -> None:
|
||||
textual.run(None, tokens)
|
||||
visual.run(None, image)
|
||||
|
||||
case ModelTask.FACIAL_RECOGNITION:
|
||||
detection = ort.InferenceSession(
|
||||
model_dir / "detection" / "model.onnx",
|
||||
sess_options=sess_options,
|
||||
providers=providers,
|
||||
provider_options=provider_options,
|
||||
)
|
||||
image = {node.name: np.random.rand(1, 3, 640, 640).astype(np.float32) for node in detection.get_inputs()}
|
||||
|
||||
recognition = ort.InferenceSession(
|
||||
model_dir / "recognition" / "model.onnx",
|
||||
sess_options=sess_options,
|
||||
providers=providers,
|
||||
provider_options=provider_options,
|
||||
)
|
||||
face = {node.name: np.random.rand(1, 3, 112, 112).astype(np.float32) for node in recognition.get_inputs()}
|
||||
|
||||
def predict() -> None:
|
||||
detection.run(None, image)
|
||||
recognition.run(None, face)
|
||||
|
||||
case _:
|
||||
raise ValueError(f"Unsupported model task {model_task}")
|
||||
predict()
|
||||
ms = timeit(predict, number=100)
|
||||
rss = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
|
||||
json.dump({"pretrained_model": model_dir.name, "peak_rss": rss, "exec_time_ms": ms}, output_path.open("w"))
|
||||
print(f"Model {model_dir.name} took {ms:.2f}ms per iteration using {rss / 1024:.2f}MiB of memory")
|
||||
|
||||
|
||||
@app.command()
|
||||
def upload(
|
||||
model_name: str,
|
||||
input_dir: Path = Path("models"),
|
||||
hf_model_name: str | None = None,
|
||||
hf_organization: str = "immich-app",
|
||||
hf_auth_token: Annotated[str | None, typer.Option(envvar="HF_AUTH_TOKEN")] = None,
|
||||
) -> None:
|
||||
from huggingface_hub import create_repo, upload_folder
|
||||
|
||||
if not hf_model_name:
|
||||
hf_model_name = model_name
|
||||
model_dir = input_dir / hf_model_name
|
||||
repo_id = f"{hf_organization}/{hf_model_name}"
|
||||
|
||||
@retry(stop=stop_after_attempt(5), wait=wait_fixed(5))
|
||||
def upload_model() -> None:
|
||||
create_repo(repo_id, exist_ok=True, token=hf_auth_token)
|
||||
upload_folder(
|
||||
repo_id=repo_id,
|
||||
folder_path=model_dir,
|
||||
# remote repo files to be deleted before uploading
|
||||
# deletion is in the same commit as the upload, so it's atomic
|
||||
delete_patterns=DELETE_PATTERNS,
|
||||
token=hf_auth_token,
|
||||
)
|
||||
|
||||
upload_model()
|
||||
@@ -1,3 +0,0 @@
|
||||
from immich_model_exporter import app
|
||||
|
||||
app()
|
||||
@@ -1,96 +0,0 @@
|
||||
from pathlib import Path
|
||||
|
||||
from .constants import RKNN_SOCS
|
||||
|
||||
|
||||
def _export_platform(
|
||||
model_dir: Path,
|
||||
target_platform: str,
|
||||
inputs: list[str] | None = None,
|
||||
input_size_list: list[list[int]] | None = None,
|
||||
fuse_matmul_softmax_matmul_to_sdpa: bool = True,
|
||||
cache: bool = True,
|
||||
) -> None:
|
||||
from rknn.api import RKNN
|
||||
|
||||
input_path = model_dir / "model.onnx"
|
||||
output_path = model_dir / "rknpu" / target_platform / "model.rknn"
|
||||
if cache and output_path.exists():
|
||||
print(f"Model {input_path} already exists at {output_path}, skipping")
|
||||
return
|
||||
|
||||
print(f"Exporting model {input_path} to {output_path}")
|
||||
|
||||
rknn = RKNN(verbose=False)
|
||||
|
||||
rknn.config(
|
||||
target_platform=target_platform,
|
||||
disable_rules=["fuse_matmul_softmax_matmul_to_sdpa"] if not fuse_matmul_softmax_matmul_to_sdpa else [],
|
||||
enable_flash_attention=False,
|
||||
model_pruning=True,
|
||||
)
|
||||
ret = rknn.load_onnx(model=input_path.as_posix(), inputs=inputs, input_size_list=input_size_list)
|
||||
|
||||
if ret != 0:
|
||||
raise RuntimeError("Load failed!")
|
||||
|
||||
ret = rknn.build(do_quantization=False)
|
||||
|
||||
if ret != 0:
|
||||
raise RuntimeError("Build failed!")
|
||||
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
ret = rknn.export_rknn(output_path.as_posix())
|
||||
if ret != 0:
|
||||
raise RuntimeError("Export rknn model failed!")
|
||||
|
||||
|
||||
def _export_platforms(
|
||||
model_dir: Path,
|
||||
inputs: list[str] | None = None,
|
||||
input_size_list: list[list[int]] | None = None,
|
||||
cache: bool = True,
|
||||
) -> None:
|
||||
fuse_matmul_softmax_matmul_to_sdpa = True
|
||||
for soc in RKNN_SOCS:
|
||||
try:
|
||||
_export_platform(
|
||||
model_dir,
|
||||
soc,
|
||||
inputs=inputs,
|
||||
input_size_list=input_size_list,
|
||||
fuse_matmul_softmax_matmul_to_sdpa=fuse_matmul_softmax_matmul_to_sdpa,
|
||||
cache=cache,
|
||||
)
|
||||
except Exception as e:
|
||||
print(f"Failed to export model for {soc}: {e}")
|
||||
if "inputs or 'outputs' must be set" in str(e):
|
||||
print("Retrying without fuse_matmul_softmax_matmul_to_sdpa")
|
||||
fuse_matmul_softmax_matmul_to_sdpa = False
|
||||
_export_platform(
|
||||
model_dir,
|
||||
soc,
|
||||
inputs=inputs,
|
||||
input_size_list=input_size_list,
|
||||
fuse_matmul_softmax_matmul_to_sdpa=fuse_matmul_softmax_matmul_to_sdpa,
|
||||
cache=cache,
|
||||
)
|
||||
|
||||
|
||||
def export(model_dir: Path, cache: bool = True) -> None:
|
||||
textual = model_dir / "textual"
|
||||
visual = model_dir / "visual"
|
||||
detection = model_dir / "detection"
|
||||
recognition = model_dir / "recognition"
|
||||
|
||||
if textual.is_dir():
|
||||
_export_platforms(textual, cache=cache)
|
||||
|
||||
if visual.is_dir():
|
||||
_export_platforms(visual, cache=cache)
|
||||
|
||||
if detection.is_dir():
|
||||
_export_platforms(detection, inputs=["input.1"], input_size_list=[[1, 3, 640, 640]], cache=cache)
|
||||
|
||||
if recognition.is_dir():
|
||||
_export_platforms(recognition, inputs=["input.1"], input_size_list=[[1, 3, 112, 112]], cache=cache)
|
||||
@@ -1,242 +0,0 @@
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
from exporters.constants import ModelSource
|
||||
|
||||
from immich_model_exporter import clean_name
|
||||
from immich_model_exporter.exporters.constants import SOURCE_TO_TASK
|
||||
|
||||
mclip = [
|
||||
"M-CLIP/LABSE-Vit-L-14",
|
||||
"M-CLIP/XLM-Roberta-Large-Vit-B-16Plus",
|
||||
"M-CLIP/XLM-Roberta-Large-Vit-B-32",
|
||||
"M-CLIP/XLM-Roberta-Large-Vit-L-14",
|
||||
]
|
||||
|
||||
openclip = [
|
||||
"RN101__openai",
|
||||
"RN101__yfcc15m",
|
||||
"RN50__cc12m",
|
||||
"RN50__openai",
|
||||
"RN50__yfcc15m",
|
||||
"RN50x16__openai",
|
||||
"RN50x4__openai",
|
||||
"RN50x64__openai",
|
||||
"ViT-B-16-SigLIP-256__webli",
|
||||
"ViT-B-16-SigLIP-384__webli",
|
||||
"ViT-B-16-SigLIP-512__webli",
|
||||
"ViT-B-16-SigLIP-i18n-256__webli",
|
||||
"ViT-B-16-SigLIP2__webli",
|
||||
"ViT-B-16-SigLIP__webli",
|
||||
"ViT-B-16-plus-240__laion400m_e31",
|
||||
"ViT-B-16-plus-240__laion400m_e32",
|
||||
"ViT-B-16__laion400m_e31",
|
||||
"ViT-B-16__laion400m_e32",
|
||||
"ViT-B-16__openai",
|
||||
"ViT-B-32-SigLIP2-256__webli",
|
||||
"ViT-B-32__laion2b-s34b-b79k",
|
||||
"ViT-B-32__laion2b_e16",
|
||||
"ViT-B-32__laion400m_e31",
|
||||
"ViT-B-32__laion400m_e32",
|
||||
"ViT-B-32__openai",
|
||||
"ViT-H-14-378-quickgelu__dfn5b",
|
||||
"ViT-H-14-quickgelu__dfn5b",
|
||||
"ViT-H-14__laion2b-s32b-b79k",
|
||||
"ViT-L-14-336__openai",
|
||||
"ViT-L-14-quickgelu__dfn2b",
|
||||
"ViT-L-14__laion2b-s32b-b82k",
|
||||
"ViT-L-14__laion400m_e31",
|
||||
"ViT-L-14__laion400m_e32",
|
||||
"ViT-L-14__openai",
|
||||
"ViT-L-16-SigLIP-256__webli",
|
||||
"ViT-L-16-SigLIP-384__webli",
|
||||
"ViT-L-16-SigLIP2-256__webli",
|
||||
"ViT-L-16-SigLIP2-384__webli",
|
||||
"ViT-L-16-SigLIP2-512__webli",
|
||||
"ViT-SO400M-14-SigLIP-384__webli",
|
||||
"ViT-SO400M-14-SigLIP2-378__webli",
|
||||
"ViT-SO400M-14-SigLIP2__webli",
|
||||
"ViT-SO400M-16-SigLIP2-256__webli",
|
||||
"ViT-SO400M-16-SigLIP2-384__webli",
|
||||
"ViT-SO400M-16-SigLIP2-512__webli",
|
||||
"ViT-gopt-16-SigLIP2-256__webli",
|
||||
"ViT-gopt-16-SigLIP2-384__webli",
|
||||
"nllb-clip-base-siglip__mrl",
|
||||
"nllb-clip-base-siglip__v1",
|
||||
"nllb-clip-large-siglip__mrl",
|
||||
"nllb-clip-large-siglip__v1",
|
||||
"xlm-roberta-base-ViT-B-32__laion5b_s13b_b90k",
|
||||
"xlm-roberta-large-ViT-H-14__frozen_laion5b_s13b_b90k",
|
||||
]
|
||||
|
||||
insightface = [
|
||||
"antelopev2",
|
||||
"buffalo_l",
|
||||
"buffalo_m",
|
||||
"buffalo_s",
|
||||
]
|
||||
|
||||
|
||||
def export_models(models: list[str], source: ModelSource) -> None:
|
||||
profiling_dir = Path("profiling")
|
||||
profiling_dir.mkdir(exist_ok=True)
|
||||
for model in models:
|
||||
try:
|
||||
model_dir = f"models/{clean_name(model)}"
|
||||
task = SOURCE_TO_TASK[source]
|
||||
|
||||
print(f"Processing model {model}")
|
||||
subprocess.check_call(["python", "-m", "immich_model_exporter", "export", model, source])
|
||||
subprocess.check_call(
|
||||
[
|
||||
"python",
|
||||
"-m",
|
||||
"immich_model_exporter",
|
||||
"profile",
|
||||
model_dir,
|
||||
task,
|
||||
"--output_path",
|
||||
profiling_dir / f"{model}.json",
|
||||
]
|
||||
)
|
||||
subprocess.check_call(["python", "-m", "immich_model_exporter", "upload", model_dir])
|
||||
except Exception as e:
|
||||
print(f"Failed to export model {model}: {e}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
export_models(mclip, ModelSource.MCLIP)
|
||||
export_models(openclip, ModelSource.OPENCLIP)
|
||||
export_models(insightface, ModelSource.INSIGHTFACE)
|
||||
|
||||
Path("results").mkdir(exist_ok=True)
|
||||
dataset_root = Path("datasets")
|
||||
dataset_root.mkdir(exist_ok=True)
|
||||
|
||||
crossmodal3600_root = dataset_root / "crossmodal3600"
|
||||
subprocess.check_call(
|
||||
[
|
||||
"clip_benchmark",
|
||||
"eval",
|
||||
"--pretrained_model",
|
||||
*[name.replace("__", ",") for name in openclip],
|
||||
"--task",
|
||||
"zeroshot_retrieval",
|
||||
"--dataset",
|
||||
"crossmodal3600",
|
||||
"--dataset_root",
|
||||
crossmodal3600_root.as_posix(),
|
||||
"--batch_size",
|
||||
"64",
|
||||
"--language",
|
||||
"ar",
|
||||
"bn",
|
||||
"cs",
|
||||
"da",
|
||||
"de",
|
||||
"el",
|
||||
"en",
|
||||
"es",
|
||||
"fa",
|
||||
"fi",
|
||||
"fil",
|
||||
"fr",
|
||||
"he",
|
||||
"hi",
|
||||
"hr",
|
||||
"hu",
|
||||
"id",
|
||||
"it",
|
||||
"ja",
|
||||
"ko",
|
||||
"mi",
|
||||
"nl",
|
||||
"no",
|
||||
"pl",
|
||||
"pt",
|
||||
"quz",
|
||||
"ro",
|
||||
"ru",
|
||||
"sv",
|
||||
"sw",
|
||||
"te",
|
||||
"th",
|
||||
"tr",
|
||||
"uk",
|
||||
"vi",
|
||||
"zh",
|
||||
"--recall_k",
|
||||
"1",
|
||||
"5",
|
||||
"10",
|
||||
"--no_amp",
|
||||
"--output",
|
||||
"results/{dataset}_{language}_{model}_{pretrained}.json",
|
||||
]
|
||||
)
|
||||
|
||||
xtd10_root = dataset_root / "xtd10"
|
||||
subprocess.check_call(
|
||||
[
|
||||
"clip_benchmark",
|
||||
"eval",
|
||||
"--pretrained_model",
|
||||
*[name.replace("__", ",") for name in openclip],
|
||||
"--task",
|
||||
"zeroshot_retrieval",
|
||||
"--dataset",
|
||||
"xtd10",
|
||||
"--dataset_root",
|
||||
xtd10_root.as_posix(),
|
||||
"--batch_size",
|
||||
"64",
|
||||
"--language",
|
||||
"de",
|
||||
"en",
|
||||
"es",
|
||||
"fr",
|
||||
"it",
|
||||
"jp",
|
||||
"ko",
|
||||
"pl",
|
||||
"ru",
|
||||
"tr",
|
||||
"zh",
|
||||
"--recall_k",
|
||||
"1",
|
||||
"5",
|
||||
"10",
|
||||
"--no_amp",
|
||||
"--output",
|
||||
"results/{dataset}_{language}_{model}_{pretrained}.json",
|
||||
]
|
||||
)
|
||||
|
||||
flickr30k_root = dataset_root / "flickr30k"
|
||||
# note: need ~/.kaggle/kaggle.json to download the dataset automatically
|
||||
subprocess.check_call(
|
||||
[
|
||||
"clip_benchmark",
|
||||
"eval",
|
||||
"--pretrained_model",
|
||||
*[name.replace("__", ",") for name in openclip],
|
||||
"--task",
|
||||
"zeroshot_retrieval",
|
||||
"--dataset",
|
||||
"flickr30k",
|
||||
"--dataset_root",
|
||||
flickr30k_root.as_posix(),
|
||||
"--batch_size",
|
||||
"64",
|
||||
"--language",
|
||||
"en",
|
||||
"zh",
|
||||
"--recall_k",
|
||||
"1",
|
||||
"5",
|
||||
"10",
|
||||
"--no_amp",
|
||||
"--output",
|
||||
"results/{dataset}_{language}_{model}_{pretrained}.json",
|
||||
]
|
||||
)
|
||||
+140
-6
@@ -1,11 +1,145 @@
|
||||
models:
|
||||
- name: 'LABSE-Vit-L-14'
|
||||
hf-name: 'M-CLIP/LABSE-Vit-L-14' # Do this for all mclip models
|
||||
hf-name: 'M-CLIP/LABSE-Vit-L-14'
|
||||
source: 'mclip'
|
||||
- name: 'XLM-Roberta-Large-Vit-B-16Plus'
|
||||
hf-name: 'M-CLIP/XLM-Roberta-Large-Vit-B-16Plus'
|
||||
source: 'mclip'
|
||||
- name: 'XLM-Roberta-Large-Vit-B-32'
|
||||
hf-name: 'M-CLIP/XLM-Roberta-Large-Vit-B-32'
|
||||
source: 'mclip'
|
||||
- name: 'XLM-Roberta-Large-Vit-L-14'
|
||||
hf-name: 'M-CLIP/XLM-Roberta-Large-Vit-L-14'
|
||||
source: 'mclip'
|
||||
- name: 'XLM-Roberta-Base-ViT-B-32__laion5b_s13b_b90k'
|
||||
hf-name: 'xlm-roberta-base-ViT-B-32__laion5b_s13b_b90k' # Do this for both xlm models
|
||||
source: 'openclip'
|
||||
- name: 'RN101__openai'
|
||||
source: 'openclip'
|
||||
- name: 'antelopev2'
|
||||
source: 'insightface'
|
||||
- name: 'buffalo_l'
|
||||
source: 'insightface'
|
||||
- name: 'buffalo_m'
|
||||
source: 'insightface'
|
||||
- name: 'buffalo_s'
|
||||
source: 'insightface'
|
||||
- name: 'RN101__openai'
|
||||
source: 'openclip'
|
||||
- name: 'RN101__yfcc15m'
|
||||
source: 'openclip'
|
||||
- name: 'RN50__cc12m'
|
||||
source: 'openclip'
|
||||
- name: 'RN50__openai'
|
||||
source: 'openclip'
|
||||
- name: 'RN50__yfcc15m'
|
||||
source: 'openclip'
|
||||
- name: 'RN50x16__openai'
|
||||
source: 'openclip'
|
||||
- name: 'RN50x4__openai'
|
||||
source: 'openclip'
|
||||
- name: 'RN50x64__openai'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-SigLIP-256__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-SigLIP-384__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-SigLIP-512__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-SigLIP-i18n-256__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-SigLIP2__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-SigLIP__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-plus-240__laion400m_e31'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16-plus-240__laion400m_e32'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16__laion400m_e31'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16__laion400m_e32'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-16__openai'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-32-SigLIP2-256__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-32__laion2b-s34b-b79k'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-32__laion2b_e16'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-32__laion400m_e31'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-32__laion400m_e32'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-B-32__openai'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-H-14-378-quickgelu__dfn5b'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-large'
|
||||
- name: 'ViT-H-14-quickgelu__dfn5b'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'ViT-H-14__laion2b-s32b-b79k'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-14-336__openai'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-14-quickgelu__dfn2b'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-14__laion2b-s32b-b82k'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-14__laion400m_e31'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-14__laion400m_e32'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-14__openai'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-16-SigLIP-256__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-16-SigLIP-384__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-16-SigLIP2-256__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-16-SigLIP2-384__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-L-16-SigLIP2-512__webli'
|
||||
source: 'openclip'
|
||||
- name: 'ViT-SO400M-14-SigLIP-384__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-large'
|
||||
- name: 'ViT-SO400M-14-SigLIP2-378__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-large'
|
||||
- name: 'ViT-SO400M-14-SigLIP2__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-large'
|
||||
- name: 'ViT-SO400M-16-SigLIP2-256__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'ViT-SO400M-16-SigLIP2-384__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'ViT-SO400M-16-SigLIP2-512__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-large'
|
||||
- name: 'ViT-gopt-16-SigLIP2-256__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-large'
|
||||
- name: 'ViT-gopt-16-SigLIP2-384__webli'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-huge'
|
||||
- name: 'nllb-clip-base-siglip__mrl'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'nllb-clip-base-siglip__v1'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'nllb-clip-large-siglip__mrl'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'nllb-clip-large-siglip__v1'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'XLM-Roberta-Base-ViT-B-32__laion5b_s13b_b90k'
|
||||
hf-name: 'xlm-roberta-base-ViT-B-32__laion5b_s13b_b90k'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
- name: 'XLM-Roberta-Large-ViT-H-14__frozen_laion5b_s13b_b90k'
|
||||
hf-name: 'xlm-roberta-large-ViT-H-14__frozen_laion5b_s13b_b90k'
|
||||
source: 'openclip'
|
||||
runner: 'pokedex-medium'
|
||||
|
||||
+34
-18
@@ -1,43 +1,59 @@
|
||||
[project]
|
||||
name = "immich_model_exporter"
|
||||
name = "immich_model"
|
||||
version = "0.1.0"
|
||||
description = "Add your description here"
|
||||
description = "Export Immich's CLIP and face models to ONNX and compile them for on-device runtimes."
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.10, <4.0"
|
||||
requires-python = ">=3.10,<3.13"
|
||||
dependencies = [
|
||||
"huggingface-hub>=0.29.3",
|
||||
"multilingual-clip>=1.0.10",
|
||||
"onnx>=1.14.1",
|
||||
"onnxruntime>=1.16.0",
|
||||
"open-clip-torch>=2.31.0",
|
||||
"typer>=0.15.2",
|
||||
"rknn-toolkit2>=2.3.0",
|
||||
"transformers>=4.49.0",
|
||||
"tenacity>=9.0.0",
|
||||
"polars>=1.25.2",
|
||||
"kaggle>=1.7.4.2",
|
||||
"clip-benchmark",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
immich-model = "immich_model:app"
|
||||
|
||||
[dependency-groups]
|
||||
dev = ["black>=23.3.0", "mypy>=1.3.0", "ruff>=0.0.272"]
|
||||
|
||||
[tool.uv]
|
||||
override-dependencies = [
|
||||
"onnx>=1.16.0,<2",
|
||||
"onnxruntime>=1.18.2,<2",
|
||||
"torch>=2.4",
|
||||
"torchvision>=0.21",
|
||||
onnx = [
|
||||
"onnx>=1.18.0",
|
||||
"onnxruntime>=1.18.2",
|
||||
"onnxscript>=0.7.0",
|
||||
"ml-dtypes>=0.5.0",
|
||||
"open-clip-torch>=2.31.0",
|
||||
"multilingual-clip>=1.0.10",
|
||||
"transformers>=4.49.0",
|
||||
"clip-benchmark",
|
||||
"numpy>=2.1.0",
|
||||
"torch",
|
||||
"torchvision",
|
||||
]
|
||||
|
||||
rknn = ["rknn-toolkit2>=2.3.0", "setuptools<81", "torch"]
|
||||
|
||||
[tool.uv]
|
||||
# `onnx` and `rknn` have mutually incompatible pins (numpy/protobuf/onnx/torch)
|
||||
conflicts = [[{ group = "onnx" }, { group = "rknn" }]]
|
||||
override-dependencies = ["onnxoptimizer>=0.4.2"]
|
||||
|
||||
[tool.uv.sources]
|
||||
clip-benchmark = { git = "https://github.com/mertalev/CLIP_benchmark.git", rev = "77e733ee241399611296d3c4aca583ec89cf5190" }
|
||||
torch = { index = "pytorch-cpu" }
|
||||
torchvision = { index = "pytorch-cpu" }
|
||||
|
||||
[[tool.uv.index]]
|
||||
name = "pytorch-cpu"
|
||||
url = "https://download.pytorch.org/whl/cpu"
|
||||
explicit = true
|
||||
|
||||
[tool.hatch.build.targets.sdist]
|
||||
include = ["immich_model_exporter"]
|
||||
include = ["immich_model"]
|
||||
|
||||
[tool.hatch.build.targets.wheel]
|
||||
include = ["immich_model_exporter"]
|
||||
include = ["immich_model"]
|
||||
|
||||
[build-system]
|
||||
requires = ["hatchling"]
|
||||
|
||||
Reference in New Issue
Block a user