polyvoice 0.12.0

Speaker diarization for Rust — who spoke when. ONNX path optional: default features are empty (ort-free BYO-embedder core); enable onnx for Silero VAD, WeSpeaker embeddings, and Pyannote segmentation.
Documentation
schema = "polyvoice-models-v2"

# Stage-scoped version aliases (Deepgram diarize_model=latest|v1 pattern).
# Resolution is logged so DER reports record the pin that was actually used.
[aliases.segmenter]
latest = "powerset_fp32"
v1 = "powerset_fp32"

[aliases.embedder]
latest = "wespeaker_resnet34"
v1 = "wespeaker_resnet34"

[aliases.vad]
latest = "silero_vad"
v1 = "silero_vad"

# v0.5 legacy profiles — point at proven FP32 models until v1.0 components
# are validated. Mobile uses CAM++ (512-dim), Balanced uses WeSpeaker ResNet34
# (256-dim). Both share Silero VAD for speech segmentation.
[profiles.mobile]
segmenter = "powerset_fp32"
embedder  = "wespeaker_resnet34"

[profiles.balanced]
segmenter = "powerset_fp32"
embedder  = "wespeaker_resnet34"

# Legacy v0.5 entries — kept for back-compat callers that pass the model id
# directly to ModelRegistry::ensure(). Profiles do not reference them anymore.
[models.silero_vad]
url      = "https://github.com/snakers4/silero-vad/raw/master/src/silero_vad/data/silero_vad.onnx"
sha256   = "1a153a22f4509e292a94e67d6f9b85e8deb25b4988682b7e174c65279d8788e3"
size     = 2327524
filename = "silero_vad.onnx"
license  = "MIT"
license_url = "https://github.com/snakers4/silero-vad/blob/master/LICENSE"
provenance = "snakers4/silero-vad upstream ONNX"
adapter_type = "silero"
version  = "1.0"
sample_rate = 16000
signature = '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhZU2jujLiO9QWKC7WPSGEE1crTXcP/4t+FGlbFrJ9+JLTVuj7Om/zXdAo1Aak/nvQo3/7xXev41Qn10+VSda/wE=
trusted comment: polyvoice v0.6.0-alpha.3 | silero_vad.onnx
ECe8pg8lcsO5MxlmAjaLRUIFC2t5TRt8gGrl3boQO7PiVJFZkzgFlgI74YH9T1Dp0bKgaMBJ0kSBIJpmyZVmDw==
'''

[models.wespeaker_resnet34]
url      = "https://huggingface.co/Wespeaker/wespeaker-voxceleb-resnet34/resolve/main/voxceleb_resnet34.onnx?download=true"
sha256   = "9fea6516d7ad6bf0a76c7689f5a49b65d330fad6dde96c91bb4435ffbfe056a1"
size     = 26534127
filename = "wespeaker_resnet34.onnx"
license  = "Apache-2.0"
license_url = "https://huggingface.co/Wespeaker/wespeaker-voxceleb-resnet34"
provenance = "Wespeaker/wespeaker-voxceleb-resnet34 Hugging Face"
adapter_type = "wespeaker-resnet34"
version  = "1.0"
sample_rate = 16000
embedding_dim = 256
signature = '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhf99QhdU+yew+XOYA3tx+dixo9cqxxz0Y7xlIOiiLhFTFKbsKeiH5OkiGn9GPbzP8TwGGLjKGLVRsOdFHSueOAs=
trusted comment: polyvoice v0.6.0-alpha.3 | wespeaker_resnet34.onnx
BzPwRAu4i4ABLmMMqmgv++OWy+3tbmdf9FrCIgXtB/zfdXXwckWIQE7vcCIfpuLPS0BUWDbkbbV6n3mugnlCBw==
'''

[models.powerset_fp32]
url      = "https://huggingface.co/csukuangfj/sherpa-onnx-pyannote-segmentation-3-0/resolve/main/model.onnx"
sha256   = "220ad67ca923bef2fa91f2390c786097bf305bceb5e261d4af67b38e938e1079"
size     = 5992913
filename = "powerset_fp32.onnx"
license  = "MIT"
license_url = "https://github.com/k2-fsa/sherpa-onnx"
provenance = "sherpa-onnx-pyannote-segmentation-3-0 (pyannote/segmentation-3.0)"
adapter_type = "powerset-v1"
version  = "3.0"
sample_rate = 16000
window_secs = 10.0
hop_secs = 1.0
num_speakers = 3
signature = '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhQ10AL95cKcDAXudXyE1DdH7VfQpci6E/PZHNlI6W19DEjsqPi8tZ7GC8PZkaHeRJ4ZnjAKTQCvkRWYoByTjuAk=
trusted comment: polyvoice v0.6.0-alpha.3 | powerset_fp32.onnx
e33pe01miZQKvp1AoCQcv6Oa3vVmxOBcxNiOkasmsCxhRq5ix1uqMWjah8IB6YieUjvHYj4hd9j1OH6wPAm/AA==
'''

[models.cam_pp_fp32]
url      = "https://huggingface.co/Wespeaker/wespeaker-voxceleb-campplus/resolve/main/voxceleb_CAM%2B%2B.onnx?download=true"
sha256   = "b50810498b5bcf5773d086f6993d344476bd0c88b566a41e8d801aaf8461efad"
size     = 29292449
filename = "cam_pp_fp32.onnx"
license  = "Apache-2.0"
license_url = "https://huggingface.co/Wespeaker/wespeaker-voxceleb-campplus"
provenance = "Wespeaker/wespeaker-voxceleb-campplus Hugging Face"
adapter_type = "cam++"
version  = "1.0"
sample_rate = 16000
embedding_dim = 512
signature = '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhZxjhT1+WXcKvuaszc4e4LHkeeySB9tzOhQLjB23OeDyBKfzQWQdfp1JfLylKJ3fvlH92eix78kBHpFruwFCFAc=
trusted comment: polyvoice v0.6.0-alpha.3 | cam_pp_fp32.onnx
pnLedo8TIV+K14fAyL54r1d9mWm1rN18C50BA/QQclyfP01tOaJVbxcJB+GZ4+AA/fluvTDJewlPn1gMo33wBA==
'''

# v1.0 INT8 artifacts (M5). Hashes/sizes are real, taken from
# `bash scripts/publish-models.sh` output. Calibration set: VoxConverse-dev
# random 500-sample (seed 42). See docs/calibration/<date>-int8-validation.md.
#
# NOTE (2026-05-07): hashes below are PROVISIONAL — produced from the M5
# preview calibration that used voxconverse-test as a stand-in calibration set
# (because the dev split download was still in progress). They will be
# overwritten by `publish-models.sh` after the full VoxConverse-dev calibration
# completes. Sizes are stable across calibration sets (compression depends on
# weight statistics, not calibration distribution).
[models.powerset_int8]
url      = "https://github.com/ekhodzitsky/polyvoice/releases/download/v0.6.0-alpha.2/powerset_int8.onnx"
sha256   = "ef549ac4b068fdb8df273d2df43cd9c150a3edc26f859b0c9b5c07f2db7914aa"
size     = 5737909
filename = "powerset_int8.onnx"
calibration = "voxconverse_dev_500_samples_seed_42"
license  = "MIT"
license_url = "https://github.com/k2-fsa/sherpa-onnx"
provenance = "INT8 quant of sherpa-onnx-pyannote-segmentation-3-0"
adapter_type = "powerset-v1"
version  = "3.0-int8"
sample_rate = 16000
window_secs = 10.0
hop_secs = 1.0
num_speakers = 3
signature = '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhU/BRu4ryq+ErSzXtl11bOOsOU0N43fBlBw5PlG0CcAVg1tcdHocfuTBfslpnb1igiBfkQBU9ZGFaf/Ec2Yu3go=
trusted comment: polyvoice v0.6.0-alpha.3 | powerset_int8.onnx
WCgtdOCkchC7LjeKxisHOzcAS3+84kMP9F2so0dd2vs1WLP0r2sv0rCzKPYovV+l/Nk2EG+RYbh1eisAj1knAw==
'''

[models.cam_pp_int8]
url      = "https://github.com/ekhodzitsky/polyvoice/releases/download/v0.6.0-alpha.2/cam_pp_int8.onnx"
sha256   = "cca48a4b36c1b46e48432b1eb1461dd69f9cf113cf506f3f660de808c93b9a85"
size     = 8803007
filename = "cam_pp_int8.onnx"
calibration = "voxconverse_dev_500_samples_seed_42"
license  = "Apache-2.0"
license_url = "https://huggingface.co/Wespeaker/wespeaker-voxceleb-campplus"
provenance = "INT8 quant of Wespeaker CAM++"
adapter_type = "cam++"
version  = "1.0-int8"
sample_rate = 16000
embedding_dim = 512
signature = '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhRQLqEqyoP/TNkcz2seLIK19JzqgIbPWHKFDVHMTA4+2hdmwZA5t0M6msDTE8LPEQXVpmqlb+jC4IzfAm518JgE=
trusted comment: polyvoice v0.6.0-alpha.3 | cam_pp_int8.onnx
XHIN9IgSVAhMFVb0TazR7NjXwO/ba2iCHQzrXsJEJ6hM4KpnrpjUgFaxTHkH4nVxlq8/5Pp4ODUc4qku3cKDAA==
'''

[models.resnet34_int8]
url      = "https://github.com/ekhodzitsky/polyvoice/releases/download/v0.6.0-alpha.2/resnet34_int8.onnx"
sha256   = "d4528ed19bac510e9f8dfe08515bc3c2860f8f4d135aa5ce875b346ed0f3bbae"
size     = 6766646
filename = "resnet34_int8.onnx"
calibration = "voxconverse_dev_500_samples_seed_42"
license  = "Apache-2.0"
license_url = "https://huggingface.co/Wespeaker/wespeaker-voxceleb-resnet34"
provenance = "INT8 quant of Wespeaker ResNet34"
adapter_type = "wespeaker-resnet34"
version  = "1.0-int8"
sample_rate = 16000
embedding_dim = 256
signature = '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhfT3/3pshTKu6WUH1VBohHK2UjcgjH77Gd6GHqQJJXp74rJtfoiEUx6e3jfPsfAIt6N7NmpjyL4xmPugQ+d9uAs=
trusted comment: polyvoice v0.6.0-alpha.3 | resnet34_int8.onnx
OdzDgAbIm1wmfjPhvZFqPYFl5dvBTlquEGjx1ZCJ10xpvY2IVP1xFvOtzGPT4hrzuf9h94KFUuya8/yvcSN/DQ==
'''

# Optional ERes2NetV2 (Apache-2.0, ungated). Short-utterance-oriented 192-d
# embedder. NEVER profile-default / never bundled. SHA-256 from upstream LFS oid.
# Signature omitted until a release-signed asset exists.
[models.eres2netv2]
url      = "https://huggingface.co/csukuangfj/speaker-embedding-models/resolve/main/3dspeaker_speech_eres2netv2_sv_zh-cn_16k-common.onnx?download=true"
sha256   = "bf1a75b9930474cf3389ef415e6e5d38ca96fea4a3a00f7e301d080a58ee2239"
size     = 71441526
filename = "3dspeaker_speech_eres2netv2_sv_zh-cn_16k-common.onnx"
license  = "Apache-2.0"
license_url = "https://www.apache.org/licenses/LICENSE-2.0"
provenance = "3D-Speaker / ModelScope iic speech_eres2netv2_sv_zh-cn_16k-common; ONNX mirror csukuangfj/speaker-embedding-models"
adapter_type = "eres2netv2"
version  = "1.0"
sample_rate = 16000
embedding_dim = 192

# Optional CAM++ trained on 200k Chinese speakers (domain variant for CJK audio).
# Same fbank pipeline as default CAM++; select via adapter "cam++-zh".
[models.cam_pp_zh]
url      = "https://huggingface.co/csukuangfj/speaker-embedding-models/resolve/main/3dspeaker_speech_campplus_sv_zh-cn_16k-common.onnx?download=true"
sha256   = "f682b514c05d947ee3fa91cd6ec6c5c7543479a128373fa29b1faedccd21fd11"
size     = 28281138
filename = "3dspeaker_speech_campplus_sv_zh-cn_16k-common.onnx"
license  = "Apache-2.0"
license_url = "https://www.apache.org/licenses/LICENSE-2.0"
provenance = "3D-Speaker / ModelScope campplus zh-cn common; ONNX mirror csukuangfj/speaker-embedding-models"
adapter_type = "cam++-zh"
version  = "1.0"
sample_rate = 16000
embedding_dim = 192

# Optional E2E Streaming Sortformer v2 (≤4 speakers). NEVER default, NEVER
# bundled, NEVER pulled by profile resolution or default CI. Download only via
# explicit ModelRegistry::ensure("sortformer_v2") under the `sortformer` feature.
# Signature omitted until a release artifact is signed with the project key —
# SHA-256 alone gates integrity for this optional download path.
[models.sortformer_v2]
url      = "https://huggingface.co/cgus/diar_streaming_sortformer_4spk-v2-onnx/resolve/main/diar_streaming_sortformer_4spk-v2.onnx?download=true"
sha256   = "7dbfc7cba4615e07b679f7d65b5e0edd22a4b7b1ab69505a594f2f90421bd9c1"
size     = 492242946
filename = "diar_streaming_sortformer_4spk-v2.onnx"
license  = "CC-BY-4.0"
license_url = "https://creativecommons.org/licenses/by/4.0/"
provenance = "NVIDIA diar_streaming_sortformer_4spk-v2 (CC-BY-4.0); community ONNX export by Altunenes for parakeet-rs, mirrored at cgus/diar_streaming_sortformer_4spk-v2-onnx"
adapter_type = "sortformer-v2"
version  = "2.0"
sample_rate = 16000
num_speakers = 4

# ---------------------------------------------------------------------------
# VBx PLDA weights (six precomputed .npy files, ~265 KB total). NOT profile
# defaults: resolved only when the `vbx` clusterer is selected and no local
# `--vbx-plda-dir` / POLYVOICE_VBX_PLDA_DIR is set. Signature omitted until a
# release engineer signs with the project minisign key — SHA-256 alone gates
# integrity for this optional download path (same pattern as sortformer_v2 /
# eres2netv2). Hosted from the commit that introduced the fixtures; pin keeps
# the URL content-stable even if master moves.
# ---------------------------------------------------------------------------
[models.vbx_plda_transform]
url      = "https://raw.githubusercontent.com/ekhodzitsky/polyvoice/0623ae0b6773db7731f0fb3e75d8950347be94db/fixtures/vbx-plda/plda_transform.npy"
sha256   = "90261469714415743f4b8a86ee6b89466db858bde3c5944367cccfb7abd34f14"
size     = 131200
filename = "plda_transform.npy"
license  = "CC-BY-4.0"
license_url = "https://creativecommons.org/licenses/by/4.0/"
provenance = "pyannote speaker-diarization-community-1 PLDA params, diagonalized offline via scripts/build-vbx-plda.py; fixtures/vbx-plda"
adapter_type = "vbx-plda"
version  = "1.0"

[models.vbx_plda_phi_computed]
url      = "https://raw.githubusercontent.com/ekhodzitsky/polyvoice/0623ae0b6773db7731f0fb3e75d8950347be94db/fixtures/vbx-plda/plda_phi_computed.npy"
sha256   = "6ef7cf2f5a23a45b66f440f9a996a4cf5c047b369829af695d50ef18aa0a35e3"
size     = 1152
filename = "plda_phi_computed.npy"
license  = "CC-BY-4.0"
license_url = "https://creativecommons.org/licenses/by/4.0/"
provenance = "pyannote speaker-diarization-community-1 PLDA params, diagonalized offline via scripts/build-vbx-plda.py; fixtures/vbx-plda"
adapter_type = "vbx-plda"
version  = "1.0"

[models.vbx_plda_mean1]
url      = "https://raw.githubusercontent.com/ekhodzitsky/polyvoice/0623ae0b6773db7731f0fb3e75d8950347be94db/fixtures/vbx-plda/plda_mean1.npy"
sha256   = "e424c0c352182aa8e0f555dec1f3b30e29a20b9ed6b25d339f112af92e51e36f"
size     = 2176
filename = "plda_mean1.npy"
license  = "CC-BY-4.0"
license_url = "https://creativecommons.org/licenses/by/4.0/"
provenance = "pyannote speaker-diarization-community-1 PLDA params, diagonalized offline via scripts/build-vbx-plda.py; fixtures/vbx-plda"
adapter_type = "vbx-plda"
version  = "1.0"

[models.vbx_plda_mean2]
url      = "https://raw.githubusercontent.com/ekhodzitsky/polyvoice/0623ae0b6773db7731f0fb3e75d8950347be94db/fixtures/vbx-plda/plda_mean2.npy"
sha256   = "6f6fb708a2037197b5b84ffeaa8f140cb878088fbecd6ab042ad26a7691bd2cf"
size     = 640
filename = "plda_mean2.npy"
license  = "CC-BY-4.0"
license_url = "https://creativecommons.org/licenses/by/4.0/"
provenance = "pyannote speaker-diarization-community-1 PLDA params, diagonalized offline via scripts/build-vbx-plda.py; fixtures/vbx-plda"
adapter_type = "vbx-plda"
version  = "1.0"

[models.vbx_plda_lda]
url      = "https://raw.githubusercontent.com/ekhodzitsky/polyvoice/0623ae0b6773db7731f0fb3e75d8950347be94db/fixtures/vbx-plda/plda_lda.npy"
sha256   = "e20c9b012bebd1aabda5a38a127e63a43cf35debdc502715fc143e2fb6bc3c4b"
size     = 131200
filename = "plda_lda.npy"
license  = "CC-BY-4.0"
license_url = "https://creativecommons.org/licenses/by/4.0/"
provenance = "pyannote speaker-diarization-community-1 PLDA params, diagonalized offline via scripts/build-vbx-plda.py; fixtures/vbx-plda"
adapter_type = "vbx-plda"
version  = "1.0"

[models.vbx_plda_mu]
url      = "https://raw.githubusercontent.com/ekhodzitsky/polyvoice/0623ae0b6773db7731f0fb3e75d8950347be94db/fixtures/vbx-plda/plda_mu.npy"
sha256   = "d286d48acf99bbc1ed1502fed0a3e361ae5626ce1870c8be9f7397c5e47886c6"
size     = 1152
filename = "plda_mu.npy"
license  = "CC-BY-4.0"
license_url = "https://creativecommons.org/licenses/by/4.0/"
provenance = "pyannote speaker-diarization-community-1 PLDA params, diagonalized offline via scripts/build-vbx-plda.py; fixtures/vbx-plda"
adapter_type = "vbx-plda"
version  = "1.0"