-
Notifications
You must be signed in to change notification settings - Fork 3
Expand file tree
/
Copy pathDiffSingerDeclarations.cs
More file actions
445 lines (399 loc) · 27.4 KB
/
Copy pathDiffSingerDeclarations.cs
File metadata and controls
445 lines (399 loc) · 27.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
using System;
using System.Collections.Generic;
using System.IO;
using System.Linq;
using TuneLab.Foundation;
using TuneLab.SDK;
namespace DiffSingerForTuneLab;
// 声明面(解析上下文 PartContext 的纯函数):自动化轨 / 回显轨 / part 面板 / note 面板。
// 据 use_*_embed 暴露可编辑曲线、据 predict_* 暴露只读回显轨;据 manifest 暴露 voice→model→version 与白名单。
// 无 manifest(legacy)→ 退化为今天行为:整模型 1 voice、speaker 下拉 + 混音容器。
// 有 manifest → voice 即说话人、暴露 model/version 下拉、混音候选 = 同包其它暴露 voice、seed 轨按 retake 声明 gating。
// 规格集中在此、单一真相源;会话运行时复用本类的轨 key 与 variance/gender/speed/speaker 集合。设计见 docs/tunelab-voicebank-schema.md。
public static class DiffSingerDeclarations
{
// 插件独立用户数据根:Voices(默认声库扫描根)/ Vocoders(声码器)/ Cache(张量缓存)均由此派生。
public static string UserDataRoot => Path.Combine(
Environment.GetFolderPath(Environment.SpecialFolder.ApplicationData), "DiffSingerForTuneLab");
// —— 暴露给用户的参数键(避开宿主保留名 Volume / VibratoEnvelope)——
public const string KeyGender = "gender";
public const string KeySpeed = "speed";
public const string KeyMouthOpening = "shift_mouth_opening"; // SHMC alpha:键取偏移语义,给将来绝对口型轨留 "mouth_opening"
public const string KeyExpressiveness = "expressiveness"; // pitch 表现力(PEXP):仅 dspitch use_expr 时暴露
public const string KeyToneShift = "tone_shift"; // 音区偏移(SHFC):仅 pitch_controllable 声码器时暴露
public const string KeySpeaker = "speaker"; // legacy:part 默认说话人下拉
public const string KeyLanguage = "language";
public const string KeyModel = "model"; // manifest:模型下拉(part 属性)
public const string KeyVersion = "version"; // manifest:版本下拉(part 属性,"" = 最新跟随)
// 随机种子轨(逐帧 = 时间维×值维 → 区域独立 take):按 manifest retake 三位 gating。
public const string KeySeedPitch = "seed_pitch";
public const string KeySeedVariance = "seed_variance";
public const string KeySeedAcoustic = "seed_acoustic";
// seed 轨归一化为 [0,1](最通用刻度,别的 retake 引擎天然兼容、撞键也不越界);
// 合成期 round(v·uint.MaxValue) 放大到 uint32 喂位置寻址哈希(哈希白化 → 刻度不影响噪声质量,见 DiffSingerNoise)。
public const double SeedCurveMax = 1;
public const string KeyMix = "speaker_mix"; // part 属性:说话人混合变长键控容器,条目键 = suffix
public const string KeyMixPrefix = "mix:"; // 说话人混合自动化轨 key 前缀:mix:<suffix>
// —— 音素混合(帧级包络,N 槽;与说话人混合正交)。需重导出模型带 tokens_b/blend/encoder_out_b 输入,
// 无该输入的老库合成期 HasInput gating 静默忽略。键按槽号 1..N(SlotKey),槽号即张量行 = 号-1——
public const string KeyMixSlots = "phoneme_mix_slots"; // part 属性:音素混合槽数 N(0=关;N=槽 1..N)
public const string KeyMixPhoneme = "mix_phoneme"; // per-phoneme 前缀:目标音素 → SlotKey mix_phoneme:k
public const string KeyMixLanguage = "mix_phoneme_lang"; // per-phoneme 前缀:目标语言 → mix_phoneme_lang:k("" = 跟随本音素)
public const string KeyMixCurve = "phoneme_mix"; // part 级前缀:比例包络曲线 [0,1] → phoneme_mix:k
public const int MaxMixSlots = 8; // 槽数上限(UI 理智值;模型动态支持任意 N)
// 槽键:base:k(k 从 1)。号 = 张量槽行 k-1。
public static string SlotKey(string baseKey, int slot) => $"{baseKey}:{slot}";
// 槽显示名:仅一槽时省略序号(key 不受影响,槽数变化时同一 key 显示名会跟着变)。
public static string SlotDisplay(string label, int slot, int slots) => slots == 1 ? label : $"{label} {slot}";
// part 属性读槽数,clamp [0, MaxMixSlots]。
public static int MixSlots(PropertyObject partProperties)
=> Math.Clamp((int)Math.Round(partProperties.GetDouble(KeyMixSlots, 0)), 0, MaxMixSlots);
// 归一化刻度:gender 中性 0、量程 [-1,1];speed 中性 1(=原速)、量程 [0,2](百分比小数化)。
public const double GenderBaseline = 0, GenderMin = -1, GenderMax = 1;
public const double SpeedBaseline = 1, SpeedMin = 0, SpeedMax = 2;
// SHMC 口型偏移 alpha:模型原生量程即 [-1,1]、0 = 不干预(相对模型隐式基线),无 convert 直接透传。
public const double MouthOpeningBaseline = 0, MouthOpeningMin = -1, MouthOpeningMax = 1;
// 表现力:混合比(OpenUtau PEXP 0~100 的小数化),1 = 满表现力(模型自由轮廓)、0 = 贴谱面音高。
public const double ExpressivenessBaseline = 1, ExpressivenessMin = 0, ExpressivenessMax = 1;
// 音区偏移:半音(物理单位,不归一——对齐 §14.2「归一化只对可任意定标的量做」;OpenUtau SHFC ±1200 音分 = ±12)。
// 听感音高不变(声码器吃原始 f0)、声学按偏移后的 f0 取音色——「用另一个音区的嗓子唱这个音」。
public const double ToneShiftBaseline = 0, ToneShiftMin = -12, ToneShiftMax = 12;
public readonly record struct VarianceSpec(
string Key, string Display, string Color,
Func<VoicebankConfig, bool> Use, Func<VoicebankConfig, bool> Predict,
double EditMin, double EditMax, double Neutral,
double AcousticMin, double AcousticMax,
Func<float, float, float> Delta);
// 编辑轨(delta 语义)归一化到小数:energy/breath/tension 中性 0、量程 [-1,1];voicing 中性 1、量程 [0,1.25]。
// Delta(x=预测声学值, y=用户归一化值) 系数随之 ×100:y=1 等价旧 y=100(energy/breath ±12dB、tension ±5)。
// AcousticMin/Max 为回显轨的真实声学单位(dB)值域;EditMin/Max 兼作合成期输入 clamp(宿主数据层无量程硬契约)。
// voicing 有意偏离 OpenUtau(其满偏 −12dB 够不着静音底、无法做纯气声段),分两支:
// 下行(y≤1)饱和有理项 + 十二次幂跳水项:起步斜率 48(浅笔即可闻)、消声点实测 ≈ y 0.2、y=0 恒精确触底 −96。
// 导数刻意非单调(先陡后缓再跳水):起步斜率 48 > 可听区平均斜率 ~32,中段必须放缓找平,
// 末端 ~70dB 挤在最后两成行程——三项需求联立的必然形状,勿当 bug 修。
// 上行(y>1)线性 +48(y−1):满偏 1.25 = +12dB(与 energy 正向满偏对齐、避开 0dBFS 过载区);
// 中性线两侧斜率相等(48)、过线无顿挫(仅二阶导跳变)。
// 预测 < −72dB 的帧(本就近静音)下行中段轻微下越 −96,由合成期 clamp 兜住。详见 schema 文档 §14.2。
public static readonly VarianceSpec[] Variances =
{
new("energy", "Energy", "#E573A5", c => c.UseEnergyEmbed, c => c.PredictEnergy, -1, 1, 0, -96, 0, (x, y) => x + y * 12),
new("breathiness", "Breathiness", "#73E5C2", c => c.UseBreathinessEmbed, c => c.PredictBreathiness, -1, 1, 0, -96, 0, (x, y) => x + y * 12),
new("voicing", "Voicing", "#C2E573", c => c.UseVoicingEmbed, c => c.PredictVoicing, 0, 1.25, 1, -96, 0,
(x, y) => y > 1 ? x + 48 * (y - 1)
: x - 48 * (1 - y) / (2 - y) - (x + 72) * MathF.Pow(1 - y, 12)),
new("tension", "Tension", "#A573E5", c => c.UseTensionEmbed, c => c.PredictTension, -1, 1, 0, -10, 10, (x, y) => x + y * 5),
};
// manifest retake 三位(legacy → 全 false ⇒ 不暴露任何 seed 轨)。
public static (bool Acoustic, bool Pitch, bool Variance) RetakeOf(ResolvedVoice resolved)
=> resolved.Manifest is { } m ? (m.RetakeAcoustic, m.RetakePitch, m.RetakeVariance) : (false, false, false);
// 音素混合能力(manifest phoneme_mix;legacy/未重导出 → false ⇒ 不暴露任何音素混合 UI)。
public static bool HasPhonemeMix(PartContext pc) => pc.Resolved.Manifest?.PhonemeMix ?? false;
// —— 自动化轨(可编辑曲线)= 固定轨 + 已启用的说话人混合轨 ——
public static OrderedMap<PropertyKey, AutomationConfig> BuildAutomationConfigs(PartContext pc, PropertyObject partProperties)
{
var map = BuildFixedAutomationConfigs(pc.Config, RetakeOf(pc.Resolved));
var set = SpeakerSet.Compute(pc.Resolved);
foreach (var (suffix, display, color) in SelectedMixTracks(set, partProperties))
map.Add((KeyMixPrefix + suffix, display), Continuous(color, 0, 0, 1)); // 混合权重归一化 [0,1](0=不混入)
// 音素混合比例包络:按槽数 N 生成 phoneme_mix:1..N(逐帧 [0,1]、基线 0)。仅能力声库暴露;目标音素在 per-phoneme 面板设。
int mixSlots = HasPhonemeMix(pc) ? MixSlots(partProperties) : 0;
for (int k = 1; k <= mixSlots; k++)
map.Add((SlotKey(KeyMixCurve, k), SlotDisplay(L.Tr("Phoneme mix"), k, mixSlots)),
Continuous(MixColors[(k - 1) % MixColors.Length], 0, 0, 1));
return map;
}
// 固定轨(与 part 属性无关):variance(按声库能力)+ Gender/Speed + seed(按 retake gating)。
public static OrderedMap<PropertyKey, AutomationConfig> BuildFixedAutomationConfigs(
VoicebankConfig config, (bool Acoustic, bool Pitch, bool Variance) retake)
{
var map = new OrderedMap<PropertyKey, AutomationConfig>();
foreach (var v in Variances)
if (v.Use(config))
map.Add((v.Key, L.Tr(v.Display)), Continuous(v.Color, v.Neutral, v.EditMin, v.EditMax));
if (config.UseKeyShiftEmbed)
map.Add((KeyGender, L.Tr("Gender")), Continuous("#E5A573", GenderBaseline, GenderMin, GenderMax));
if (config.UseSpeedEmbed)
map.Add((KeySpeed, L.Tr("Speed")), Continuous("#73B5E5", SpeedBaseline, SpeedMin, SpeedMax));
if (config.UseShiftMouthOpeningEmbed)
map.Add((KeyMouthOpening, L.Tr("Mouth opening")),
Continuous("#C273E5", MouthOpeningBaseline, MouthOpeningMin, MouthOpeningMax));
// 表现力(pitch 模型 expr 口,dspitch use_expr 声明);音区偏移(宿主 f0 技巧,仅 pitch_controllable 声码器可行)。
if (config.UseExpr)
map.Add((KeyExpressiveness, L.Tr("Expressiveness")), Continuous("#E573E5", ExpressivenessBaseline, ExpressivenessMin, ExpressivenessMax));
if (config.VocoderPitchControllable)
map.Add((KeyToneShift, L.Tr("Tone shift")), Continuous("#73E573", ToneShiftBaseline, ToneShiftMin, ToneShiftMax));
// seed 轨:仅当 manifest 声明对应 retake 能力时暴露(连续、基线 0、量程 [0,SeedCurveMax])。
if (retake.Pitch)
map.Add((KeySeedPitch, L.Tr("Pitch seed")), Continuous("#9E9E9E", 0, 0, SeedCurveMax, randomizable: true));
if (retake.Variance)
map.Add((KeySeedVariance, L.Tr("Variance seed")), Continuous("#9E9E9E", 0, 0, SeedCurveMax, randomizable: true));
if (retake.Acoustic)
map.Add((KeySeedAcoustic, L.Tr("Timbre seed")), Continuous("#9E9E9E", 0, 0, SeedCurveMax, randomizable: true));
return map;
}
// 已启用混合的说话人 (suffix, 显示名, 稳定配色):遍历说话人集合,排除默认 speaker,
// 过滤出 part 属性 speaker_mix 容器里已存在的键。配色按固定索引轮转(manifest voice 自带 color 则优先)。
public static IEnumerable<(string Suffix, string Display, string Color)> SelectedMixTracks(SpeakerSet set, PropertyObject partProperties)
{
var selected = partProperties.GetObject(KeyMix);
int i = 0;
foreach (var opt in set.Options)
{
string color = opt.Color ?? MixColors[i % MixColors.Length];
i++;
if (opt.Suffix == set.DefaultSuffix)
continue;
if (selected.Map.ContainsKey(opt.Suffix))
yield return (opt.Suffix, opt.Display, color);
}
}
// 只读回显轨:仅当声学接受该量为输入且方差器能产基线时——显示实参(预测 + 用户 delta 合成后)。
public static OrderedMap<PropertyKey, AutomationConfig> BuildReadbackConfigs(VoicebankConfig config)
{
var map = new OrderedMap<PropertyKey, AutomationConfig>();
foreach (var v in Variances)
if (v.Use(config) && v.Predict(config))
map.Add((v.Key, L.Tr(v.Display)), Piecewise(v.Color, v.AcousticMin, v.AcousticMax));
return map;
}
// part 级面板:manifest 暴露 model/version 下拉;legacy 暴露 speaker 下拉;两者皆可有混音容器 + 默认语言。
public static ObjectConfig BuildPartConfig(PartContext pc, PropertyObject mergedProps)
{
var properties = new OrderedMap<PropertyKey, IControllerConfig>();
var registry = pc.Registry;
// 模型下拉(含此 voice 的模型 >1 时):默认 = released 最新(浮动)。
if (registry.ModelOptions(pc.VoiceId) is { } modelOptions)
{
properties.Add((KeyModel, L.Tr("Model")), ComboBoxConfig
.Create(modelOptions.Select(o => new ComboBoxItem(PropertyValue.Create(o.Value), o.Display)).ToList())
.WithDefault(PropertyValue.Create(string.Empty))); // "" = 最新 sentinel(浮动跟随最新模型)
}
// 版本下拉(当前选中模型版本 >1 时):首项 "" = 最新(跟随)。
string? selectedModel = mergedProps.GetString(KeyModel, string.Empty) is { Length: > 0 } sm ? sm : null;
if (registry.VersionOptions(pc.VoiceId, selectedModel) is { } versionOptions)
{
properties.Add((KeyVersion, L.Tr("Version")), ComboBoxConfig
.Create(versionOptions.Select(o => new ComboBoxItem(PropertyValue.Create(o.Value), o.Display)).ToList())
.WithDefault(PropertyValue.Create(string.Empty)));
}
// 默认语言:多语言切换属高频操作,排在进阶的混音容器之上。
if (HasLanguageChoice(pc))
properties.Add((KeyLanguage, L.Tr("Language")), LanguageCombo(EffectiveLanguages(pc), DefaultLanguageId(pc.Config, pc.Resolved)));
// 说话人混合容器:候选 = 同包暴露 voice 中除当前 voice 外的(legacy 即旧 speaker 下拉里的其他人)。
var set = SpeakerSet.Compute(pc.Resolved);
if (set.Options.Any(o => o.Suffix != set.DefaultSuffix))
properties.Add((KeyMix, L.Tr("Speaker mix")), BuildSpeakerMixConfig(set, mergedProps));
// 音素混合槽数(仅能力声库暴露):0=关;调到 N → 自动化面板出 N 条 phoneme_mix:k 曲线、每个音素出 N 组目标音素/语言。
if (HasPhonemeMix(pc))
properties.Add((KeyMixSlots, L.Tr("Phoneme mix slots")),
DraggableNumberBoxConfig.Integer(0).WithRange(0, MaxMixSlots));
return ObjectConfig.Create(properties);
}
// 说话人混合容器:Properties = 当前已选(present 键);AddableElements = 候选(除默认)。条目皆纯 presence。
static ExtensibleObjectConfig BuildSpeakerMixConfig(SpeakerSet set, PropertyObject partProperties)
{
var selected = partProperties.GetObject(KeyMix);
var props = new OrderedMap<PropertyKey, IControllerConfig>();
var addable = new List<AddableKey>();
foreach (var opt in set.Options)
{
if (opt.Suffix == set.DefaultSuffix)
continue;
if (selected.Map.ContainsKey(opt.Suffix))
props.Add((opt.Suffix, opt.Display), EmptyEntry());
addable.Add(new AddableKey((opt.Suffix, opt.Display), EmptyEntry()));
}
return ExtensibleObjectConfig.Create(props, addable);
}
static ObjectConfig EmptyEntry() => ObjectConfig.Create(new OrderedMap<PropertyKey, IControllerConfig>());
// note 级面板:多语言暴露 per-note 语言覆盖;默认值取 part 当前默认语言。
public static ObjectConfig BuildNoteConfig(PartContext pc, IVoiceSynthesisNotePropertyContext context)
{
var properties = new OrderedMap<PropertyKey, IControllerConfig>();
if (HasLanguageChoice(pc))
{
var partDefault = context.Part.PartProperties.GetString(KeyLanguage, DefaultLanguageId(pc.Config, pc.Resolved));
properties.Add((KeyLanguage, L.Tr("Language")), LanguageCombo(EffectiveLanguages(pc), partDefault));
}
return ObjectConfig.Create(properties);
}
// per-phoneme 面板:多语言时暴露 per-phoneme 语言覆盖(默认空 = 跟随 note);按 part 槽数 N 暴露 N 组音素混合(目标 + 语言)。
// slots 来自 part 属性 phoneme_mix_slots;每槽 k 一组 mix_phoneme:k + mix_phoneme_lang:k。比例由 part 级 phoneme_mix:k 曲线给。
public static ObjectConfig BuildPhonemeConfig(PartContext pc, int slots)
{
var props = new OrderedMap<PropertyKey, IControllerConfig>();
if (HasLanguageChoice(pc))
{
var options = new List<ComboBoxItem> { new(PropertyValue.Create(string.Empty), L.Tr("(follow note)")) };
foreach (var (id, display) in EffectiveLanguages(pc))
options.Add(new ComboBoxItem(PropertyValue.Create(id), display));
props.Add((KeyLanguage, L.Tr("Language")), ComboBoxConfig.Create(options).WithDefault(PropertyValue.Create(string.Empty)));
}
// 音素混合(帧级包络):每槽 k = 目标音素输入框(裸符号,空=该槽不混)+ 语言下拉(多语言库才显示,默认跟随本音素)。
// 比例由 part 级「Phoneme mix k」包络曲线逐帧给。调整任一项 → 宿主自动钉死该 note;目标符号合成期按语言解析、查不到即不混。
for (int k = 1; k <= slots; k++)
{
props.Add((SlotKey(KeyMixPhoneme, k), SlotDisplay(L.Tr("Mix phoneme"), k, slots)), TextBoxConfig.Create());
if (HasLanguageChoice(pc))
{
var mixLangs = new List<ComboBoxItem> { new(PropertyValue.Create(string.Empty), L.Tr("(follow phoneme)")) };
foreach (var (id, display) in EffectiveLanguages(pc))
mixLangs.Add(new ComboBoxItem(PropertyValue.Create(id), display));
props.Add((SlotKey(KeyMixLanguage, k), SlotDisplay(L.Tr("Mix language"), k, slots)),
ComboBoxConfig.Create(mixLangs).WithDefault(PropertyValue.Create(string.Empty)));
}
}
return ObjectConfig.Create(props);
}
// 多语言判据只看语言表:languages 表管「有哪些语言」;use_lang_id(→ ONNX languages 输入)只管「要不要 lang embed」,
// 前缀音素库(多词典多语言)use_lang_id=false 照样多语言(对齐 OpenUtau:语言=phonemizer 选择,与 use_lang_id 解耦)。
public static bool HasLanguageChoice(PartContext pc)
=> EffectiveLanguages(pc).Count > 1;
// 有效语言 (id, 显示名):id 来自 dsconfig 语言表;manifest 的 languages.expose 非空则按其白名单 + 顺序 + 显示名/i18n 叠加。
public static List<(string Id, string Display)> EffectiveLanguages(PartContext pc)
{
var configLangs = pc.Config.Languages;
var manifest = pc.Resolved.Manifest;
if (manifest is not null && manifest.Languages.Count > 0)
{
var set = new HashSet<string>(configLangs, StringComparer.Ordinal);
var result = manifest.Languages
.Where(l => set.Contains(l.Id))
.Select(l => (l.Id, I18n.Resolve(string.IsNullOrWhiteSpace(l.Name) ? l.Id : l.Name!, l.NameI18n, HostLang)))
.ToList();
if (result.Count > 0)
return result;
}
return configLangs.Select(id => (id, id)).ToList();
}
// 默认语言回退链:manifest 模型级 default → 当前 voice 的 default → character.yaml default_phonemizer
// (OpenUtau 生态既有的作者声明,类名关键词映射语言)→ 宿主界面语言命中 → 语言表首键。
// 每层都要求命中 dsconfig 语言表才采用;语言表键序来自训练配置、与“主语言”无必然关系,故首键仅作末位兜底。
public static string DefaultLanguageId(VoicebankConfig config, ResolvedVoice resolved)
{
var (id, source) = Resolve();
LogDefaultLanguageOnce(resolved.RootPath, id, source);
return id;
(string Id, string Source) Resolve()
{
var ids = new HashSet<string>(config.Languages, StringComparer.Ordinal);
if (resolved.Manifest?.DefaultLanguage is { } md && ids.Contains(md))
return (md, "manifest");
if (resolved.CurrentVoice?.DefaultLanguage is { } vd && ids.Contains(vd))
return (vd, "voice");
if (PhonemizerLanguage(CharacterMetadata.Read(resolved.RootPath).DefaultPhonemizer) is { } pl && ids.Contains(pl))
return (pl, "default_phonemizer");
if (HostLang is { } host)
{
var hostLang = LangPart(host);
if (config.Languages.FirstOrDefault(x => string.Equals(x, hostLang, StringComparison.OrdinalIgnoreCase)) is { } hl)
return (hl, "host");
}
return config.Languages.Count > 0 ? (config.Languages[0], "first-key") : (string.Empty, "none");
}
}
// 诊断:每 (包, 宿主语言, 结果, 判据) 只记一次——“默认语言不符预期”类问题的现场证据,量极小不刷屏。
static readonly HashSet<string> mLoggedLangDefaults = [];
static void LogDefaultLanguageOnce(string rootPath, string id, string source)
{
var host = HostLang ?? "(null)";
lock (mLoggedLangDefaults)
{
if (!mLoggedLangDefaults.Add($"{rootPath}|{host}|{id}|{source}"))
return;
}
TuneLabContext.Global.GetLogger().Info(
$"DiffSinger:默认语言 \"{id}\"(判据 {source},宿主语言 \"{host}\")—— {Path.GetFileName(rootPath)}");
}
// OpenUtau phonemizer 类名 → 语言 id(按关键词包含匹配;如 DiffSingerARPAPlusEnglishPhonemizer → en)。
// 查无返回 null,由上层回退链继续。Jyutping 须先于 Chinese 无所谓——两词无包含关系,表序仅习惯。
static readonly (string Keyword, string Lang)[] PhonemizerLangKeywords =
{
("Jyutping", "yue"), ("Cantonese", "yue"), ("Chinese", "zh"), ("English", "en"),
("Japanese", "ja"), ("Korean", "ko"), ("Russian", "ru"), ("Spanish", "es"),
("German", "de"), ("French", "fr"), ("Italian", "it"), ("Portuguese", "pt"),
};
static string? PhonemizerLanguage(string? phonemizer)
{
if (string.IsNullOrWhiteSpace(phonemizer))
return null;
foreach (var (keyword, lang) in PhonemizerLangKeywords)
if (phonemizer.Contains(keyword, StringComparison.OrdinalIgnoreCase))
return lang;
return null;
}
// "zh-CN" → "zh"(宿主 culture 的语言部;语言表键即语言码)。
static string LangPart(string locale)
{
int dash = locale.IndexOf('-');
return dash > 0 ? locale[..dash] : locale;
}
static string? HostLang => TuneLabContext.Global.Language;
static ComboBoxConfig LanguageCombo(List<(string Id, string Display)> langs, string defaultValue)
=> ComboBoxConfig
.Create(langs.Select(l => new ComboBoxItem(PropertyValue.Create(l.Id), l.Display)).ToList())
.WithDefault(PropertyValue.Create(defaultValue));
// 说话人混合轨 (key=mix:<suffix>, suffix):每候选一条;同 suffix 去重。会话据此订阅 + 渲染期解析。
public static IEnumerable<(string Key, string Suffix)> SpeakerMixTracks(SpeakerSet set)
{
foreach (var opt in set.Options)
yield return (KeyMixPrefix + opt.Suffix, opt.Suffix);
}
// 全部候选 mix 轨 key(从解析包的 ExposedVoices)——供会话订阅期使用(彼时只有实时属性、无 PropertyObject 快照)。
public static IEnumerable<(string Key, string Suffix)> MixTrackKeys(ResolvedVoice resolved)
{
var seen = new HashSet<string>(StringComparer.Ordinal);
foreach (var v in resolved.ExposedVoices)
{
var suffix = Suffix(v.Speaker);
if (!string.IsNullOrEmpty(suffix) && seen.Add(suffix))
yield return (KeyMixPrefix + suffix, suffix);
}
}
// dsconfig 说话人条目 → 显示/匹配用 suffix("260509a.Miku" → "Miku";无点则原样)。
public static string Suffix(string speaker)
{
int dot = speaker.LastIndexOf('.');
return dot >= 0 && dot < speaker.Length - 1 ? speaker[(dot + 1)..] : speaker;
}
static readonly string[] MixColors =
{
"#E57373", "#F06292", "#BA68C8", "#9575CD", "#7986CB", "#64B5F6",
"#4DD0E1", "#4DB6AC", "#81C784", "#DCE775", "#FFD54F", "#FFB74D",
};
// 分段轨:不调 WithDefault → DefaultValue 保持 NaN(= 分段、无基线)。
static AutomationConfig Piecewise(string color, double min, double max)
=> AutomationConfig.Create(min, max).WithColor(color);
// 连续轨:WithDefault 给实数基线 → 连续;按需 WithRandomizable。
static AutomationConfig Continuous(string color, double baseline, double min, double max, bool randomizable = false)
=> AutomationConfig.Create(min, max).WithDefault(baseline).WithColor(color).WithRandomizable(randomizable);
}
// 一个 part 当前的“可用说话人集合”:驱动默认说话人、混音候选/轨、嵌入解析。
// 统一从解析包的 ExposedVoices 取(manifest 显式声明;legacy 由 dsconfig speakers 自动派生)——无 legacy 特例分支。
// Options = 同包暴露 voices(显示=voice 名,可 i18n / 自带 color);默认 = 当前 voice 的 speaker。
public sealed class SpeakerSet
{
public IReadOnlyList<SpeakerOption> Options { get; }
public string DefaultSuffix { get; }
SpeakerSet(IReadOnlyList<SpeakerOption> options, string defaultSuffix)
{
Options = options;
DefaultSuffix = defaultSuffix;
}
public static SpeakerSet Compute(ResolvedVoice resolved)
{
string? host = TuneLabContext.Global.Language;
var seen = new HashSet<string>(StringComparer.Ordinal);
var options = new List<SpeakerOption>();
foreach (var v in resolved.ExposedVoices)
{
var suffix = DiffSingerDeclarations.Suffix(v.Speaker);
if (string.IsNullOrEmpty(suffix) || !seen.Add(suffix))
continue;
var display = I18n.Resolve(string.IsNullOrWhiteSpace(v.Name) ? suffix : v.Name!, v.NameI18n, host);
options.Add(new SpeakerOption(suffix, display, v.Color));
}
return new SpeakerSet(options, DiffSingerDeclarations.Suffix(resolved.VoiceSpeaker ?? string.Empty));
}
}
public sealed record SpeakerOption(string Suffix, string Display, string? Color);