1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
= "polyvoice-models-v2"
# Stage-scoped version aliases (Deepgram diarize_model=latest|v1 pattern).
# Resolution is logged so DER reports record the pin that was actually used.
[]
= "powerset_fp32"
= "powerset_fp32"
[]
= "wespeaker_resnet34"
= "wespeaker_resnet34"
[]
= "silero_vad"
= "silero_vad"
# v0.5 legacy profiles — point at proven FP32 models until v1.0 components
# are validated. Mobile uses CAM++ (512-dim), Balanced uses WeSpeaker ResNet34
# (256-dim). Both share Silero VAD for speech segmentation.
[]
= "powerset_fp32"
= "wespeaker_resnet34"
[]
= "powerset_fp32"
= "wespeaker_resnet34"
# Legacy v0.5 entries — kept for back-compat callers that pass the model id
# directly to ModelRegistry::ensure(). Profiles do not reference them anymore.
[]
= "https://github.com/snakers4/silero-vad/raw/master/src/silero_vad/data/silero_vad.onnx"
= "1a153a22f4509e292a94e67d6f9b85e8deb25b4988682b7e174c65279d8788e3"
= 2327524
= "silero_vad.onnx"
= "MIT"
= "https://github.com/snakers4/silero-vad/blob/master/LICENSE"
= "snakers4/silero-vad upstream ONNX"
= "silero"
= "1.0"
= 16000
= '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhZU2jujLiO9QWKC7WPSGEE1crTXcP/4t+FGlbFrJ9+JLTVuj7Om/zXdAo1Aak/nvQo3/7xXev41Qn10+VSda/wE=
trusted comment: polyvoice v0.6.0-alpha.3 | silero_vad.onnx
ECe8pg8lcsO5MxlmAjaLRUIFC2t5TRt8gGrl3boQO7PiVJFZkzgFlgI74YH9T1Dp0bKgaMBJ0kSBIJpmyZVmDw==
'''
[]
= "https://huggingface.co/Wespeaker/wespeaker-voxceleb-resnet34/resolve/main/voxceleb_resnet34.onnx?download=true"
= "9fea6516d7ad6bf0a76c7689f5a49b65d330fad6dde96c91bb4435ffbfe056a1"
= 26534127
= "wespeaker_resnet34.onnx"
= "Apache-2.0"
= "https://huggingface.co/Wespeaker/wespeaker-voxceleb-resnet34"
= "Wespeaker/wespeaker-voxceleb-resnet34 Hugging Face"
= "wespeaker-resnet34"
= "1.0"
= 16000
= 256
= '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhf99QhdU+yew+XOYA3tx+dixo9cqxxz0Y7xlIOiiLhFTFKbsKeiH5OkiGn9GPbzP8TwGGLjKGLVRsOdFHSueOAs=
trusted comment: polyvoice v0.6.0-alpha.3 | wespeaker_resnet34.onnx
BzPwRAu4i4ABLmMMqmgv++OWy+3tbmdf9FrCIgXtB/zfdXXwckWIQE7vcCIfpuLPS0BUWDbkbbV6n3mugnlCBw==
'''
[]
= "https://huggingface.co/csukuangfj/sherpa-onnx-pyannote-segmentation-3-0/resolve/main/model.onnx"
= "220ad67ca923bef2fa91f2390c786097bf305bceb5e261d4af67b38e938e1079"
= 5992913
= "powerset_fp32.onnx"
= "MIT"
= "https://github.com/k2-fsa/sherpa-onnx"
= "sherpa-onnx-pyannote-segmentation-3-0 (pyannote/segmentation-3.0)"
= "powerset-v1"
= "3.0"
= 16000
= 10.0
= 1.0
= 3
= '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhQ10AL95cKcDAXudXyE1DdH7VfQpci6E/PZHNlI6W19DEjsqPi8tZ7GC8PZkaHeRJ4ZnjAKTQCvkRWYoByTjuAk=
trusted comment: polyvoice v0.6.0-alpha.3 | powerset_fp32.onnx
e33pe01miZQKvp1AoCQcv6Oa3vVmxOBcxNiOkasmsCxhRq5ix1uqMWjah8IB6YieUjvHYj4hd9j1OH6wPAm/AA==
'''
[]
= "https://huggingface.co/Wespeaker/wespeaker-voxceleb-campplus/resolve/main/voxceleb_CAM%2B%2B.onnx?download=true"
= "b50810498b5bcf5773d086f6993d344476bd0c88b566a41e8d801aaf8461efad"
= 29292449
= "cam_pp_fp32.onnx"
= "Apache-2.0"
= "https://huggingface.co/Wespeaker/wespeaker-voxceleb-campplus"
= "Wespeaker/wespeaker-voxceleb-campplus Hugging Face"
= "cam++"
= "1.0"
= 16000
= 512
= '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhZxjhT1+WXcKvuaszc4e4LHkeeySB9tzOhQLjB23OeDyBKfzQWQdfp1JfLylKJ3fvlH92eix78kBHpFruwFCFAc=
trusted comment: polyvoice v0.6.0-alpha.3 | cam_pp_fp32.onnx
pnLedo8TIV+K14fAyL54r1d9mWm1rN18C50BA/QQclyfP01tOaJVbxcJB+GZ4+AA/fluvTDJewlPn1gMo33wBA==
'''
# v1.0 INT8 artifacts (M5). Hashes/sizes are real, taken from
# `bash scripts/publish-models.sh` output. Calibration set: VoxConverse-dev
# random 500-sample (seed 42). See docs/calibration/<date>-int8-validation.md.
#
# NOTE (2026-05-07): hashes below are PROVISIONAL — produced from the M5
# preview calibration that used voxconverse-test as a stand-in calibration set
# (because the dev split download was still in progress). They will be
# overwritten by `publish-models.sh` after the full VoxConverse-dev calibration
# completes. Sizes are stable across calibration sets (compression depends on
# weight statistics, not calibration distribution).
[]
= "https://github.com/ekhodzitsky/polyvoice/releases/download/v0.6.0-alpha.2/powerset_int8.onnx"
= "ef549ac4b068fdb8df273d2df43cd9c150a3edc26f859b0c9b5c07f2db7914aa"
= 5737909
= "powerset_int8.onnx"
= "voxconverse_dev_500_samples_seed_42"
= "MIT"
= "https://github.com/k2-fsa/sherpa-onnx"
= "INT8 quant of sherpa-onnx-pyannote-segmentation-3-0"
= "powerset-v1"
= "3.0-int8"
= 16000
= 10.0
= 1.0
= 3
= '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhU/BRu4ryq+ErSzXtl11bOOsOU0N43fBlBw5PlG0CcAVg1tcdHocfuTBfslpnb1igiBfkQBU9ZGFaf/Ec2Yu3go=
trusted comment: polyvoice v0.6.0-alpha.3 | powerset_int8.onnx
WCgtdOCkchC7LjeKxisHOzcAS3+84kMP9F2so0dd2vs1WLP0r2sv0rCzKPYovV+l/Nk2EG+RYbh1eisAj1knAw==
'''
[]
= "https://github.com/ekhodzitsky/polyvoice/releases/download/v0.6.0-alpha.2/cam_pp_int8.onnx"
= "cca48a4b36c1b46e48432b1eb1461dd69f9cf113cf506f3f660de808c93b9a85"
= 8803007
= "cam_pp_int8.onnx"
= "voxconverse_dev_500_samples_seed_42"
= "Apache-2.0"
= "https://huggingface.co/Wespeaker/wespeaker-voxceleb-campplus"
= "INT8 quant of Wespeaker CAM++"
= "cam++"
= "1.0-int8"
= 16000
= 512
= '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhRQLqEqyoP/TNkcz2seLIK19JzqgIbPWHKFDVHMTA4+2hdmwZA5t0M6msDTE8LPEQXVpmqlb+jC4IzfAm518JgE=
trusted comment: polyvoice v0.6.0-alpha.3 | cam_pp_int8.onnx
XHIN9IgSVAhMFVb0TazR7NjXwO/ba2iCHQzrXsJEJ6hM4KpnrpjUgFaxTHkH4nVxlq8/5Pp4ODUc4qku3cKDAA==
'''
[]
= "https://github.com/ekhodzitsky/polyvoice/releases/download/v0.6.0-alpha.2/resnet34_int8.onnx"
= "d4528ed19bac510e9f8dfe08515bc3c2860f8f4d135aa5ce875b346ed0f3bbae"
= 6766646
= "resnet34_int8.onnx"
= "voxconverse_dev_500_samples_seed_42"
= "Apache-2.0"
= "https://huggingface.co/Wespeaker/wespeaker-voxceleb-resnet34"
= "INT8 quant of Wespeaker ResNet34"
= "wespeaker-resnet34"
= "1.0-int8"
= 16000
= 256
= '''
untrusted comment: signature from minisign secret key
RUQGu9FvZMmIhfT3/3pshTKu6WUH1VBohHK2UjcgjH77Gd6GHqQJJXp74rJtfoiEUx6e3jfPsfAIt6N7NmpjyL4xmPugQ+d9uAs=
trusted comment: polyvoice v0.6.0-alpha.3 | resnet34_int8.onnx
OdzDgAbIm1wmfjPhvZFqPYFl5dvBTlquEGjx1ZCJ10xpvY2IVP1xFvOtzGPT4hrzuf9h94KFUuya8/yvcSN/DQ==
'''
# Optional ERes2NetV2 (Apache-2.0, ungated). Short-utterance-oriented 192-d
# embedder. NEVER profile-default / never bundled. SHA-256 from upstream LFS oid.
# Signature omitted until a release-signed asset exists.
[]
= "https://huggingface.co/csukuangfj/speaker-embedding-models/resolve/main/3dspeaker_speech_eres2netv2_sv_zh-cn_16k-common.onnx?download=true"
= "bf1a75b9930474cf3389ef415e6e5d38ca96fea4a3a00f7e301d080a58ee2239"
= 71441526
= "3dspeaker_speech_eres2netv2_sv_zh-cn_16k-common.onnx"
= "Apache-2.0"
= "https://www.apache.org/licenses/LICENSE-2.0"
= "3D-Speaker / ModelScope iic speech_eres2netv2_sv_zh-cn_16k-common; ONNX mirror csukuangfj/speaker-embedding-models"
= "eres2netv2"
= "1.0"
= 16000
= 192
# Optional CAM++ trained on 200k Chinese speakers (domain variant for CJK audio).
# Same fbank pipeline as default CAM++; select via adapter "cam++-zh".
[]
= "https://huggingface.co/csukuangfj/speaker-embedding-models/resolve/main/3dspeaker_speech_campplus_sv_zh-cn_16k-common.onnx?download=true"
= "f682b514c05d947ee3fa91cd6ec6c5c7543479a128373fa29b1faedccd21fd11"
= 28281138
= "3dspeaker_speech_campplus_sv_zh-cn_16k-common.onnx"
= "Apache-2.0"
= "https://www.apache.org/licenses/LICENSE-2.0"
= "3D-Speaker / ModelScope campplus zh-cn common; ONNX mirror csukuangfj/speaker-embedding-models"
= "cam++-zh"
= "1.0"
= 16000
= 192
# Optional E2E Streaming Sortformer v2 (≤4 speakers). NEVER default, NEVER
# bundled, NEVER pulled by profile resolution or default CI. Download only via
# explicit ModelRegistry::ensure("sortformer_v2") under the `sortformer` feature.
# Signature omitted until a release artifact is signed with the project key —
# SHA-256 alone gates integrity for this optional download path.
[]
= "https://huggingface.co/cgus/diar_streaming_sortformer_4spk-v2-onnx/resolve/main/diar_streaming_sortformer_4spk-v2.onnx?download=true"
= "7dbfc7cba4615e07b679f7d65b5e0edd22a4b7b1ab69505a594f2f90421bd9c1"
= 492242946
= "diar_streaming_sortformer_4spk-v2.onnx"
= "CC-BY-4.0"
= "https://creativecommons.org/licenses/by/4.0/"
= "NVIDIA diar_streaming_sortformer_4spk-v2 (CC-BY-4.0); community ONNX export by Altunenes for parakeet-rs, mirrored at cgus/diar_streaming_sortformer_4spk-v2-onnx"
= "sortformer-v2"
= "2.0"
= 16000
= 4