fix(lyrics): preserve Apple vocal sides for group parts

This commit is contained in:
zarzet committed 2026-09-27 04:06:15 +07:00
1 parent fedf5f137a
commit 34cda0a802
4 files changed
+221 -10

No files matched your search

+172 -7
View File
@@ -1,7 +1,9 @@
use super::{LyricsResponse, LyricsWord, json, lrc};
use regex::Regex;
use serde::Serialize;
use serde_json::value::RawValue;
use std::collections::{BTreeMap, BTreeSet};
use std::sync::LazyLock;
#[derive(Clone, Debug, Default, Serialize)]
pub struct PaxDetail {
@@ -23,12 +25,28 @@ pub struct PaxLine {
pub background: bool,
pub background_text: Option<Vec<PaxDetail>>,
pub endtime: isize,
#[serde(skip_serializing_if = "String::is_empty")]
pub agent: String,
}
json::go_deserialize!(PaxLine {
"text" => text, "timestamp" => timestamp, "oppositeturn" => opposite_turn,
"background" => background, "backgroundtext" => background_text, "endtime" => endtime,
"agent" => agent,
});
#[derive(Default)]
struct AppleAgent {
id: String,
kind: String,
}
json::go_deserialize!(AppleAgent { "id" => id, "type" => kind, });
#[derive(Default)]
struct AppleMetadata {
agents: Option<Vec<AppleAgent>>,
}
json::go_deserialize!(AppleMetadata { "agents" => agents, });
#[derive(Default)]
struct ApplePayload {
kind: String,
@@ -37,12 +55,86 @@ struct ApplePayload {
elrc_multi_person: String,
plain: String,
ttml_content: String,
metadata: Option<AppleMetadata>,
}
json::go_deserialize!(ApplePayload {
"type" => kind, "content" => content, "elrc" => elrc,
"elrcmultiperson" => elrc_multi_person, "plain" => plain, "ttmlcontent" => ttml_content,
"metadata" => metadata,
});
static APPLE_VOCAL_LINE: LazyLock<Regex> = LazyLock::new(|| {
Regex::new(r"(?i)^\[([0-9]{1,3}):([0-9]{1,2})(?:[.:]([0-9]{1,3}))?\]\s*(v[1-9][0-9]*):")
.unwrap()
});
/// The proxy's oppositeTurn flag can turn a group agent into the second
/// singer. Recover the original roles while retaining every eLRC word time,
/// space and backing part. Group vocals use the primary side, as in TTML.
fn apple_vocal_sides(text: &str, payload: &ApplePayload) -> String {
let agents = payload
.metadata
.as_ref()
.and_then(|metadata| metadata.agents.as_deref())
.unwrap_or_default();
let mut voices = BTreeMap::new();
let mut person = 0;
for agent in agents {
let voice = match agent.kind.as_str() {
"person" => {
person += 1;
person
}
"group" => 1,
_ => continue,
};
voices.insert(agent.id.as_str(), format!("v{voice}"));
}
let lines = payload.content.as_deref().unwrap_or_default();
if voices.is_empty() || lines.is_empty() {
return text.into();
}
text.lines()
.map(|line| {
let Some(captures) = APPLE_VOCAL_LINE.captures(line) else {
return line.to_owned();
};
let fraction = captures.get(3).map_or(0, |value| {
value.as_str().parse::<i64>().unwrap() * 10_i64.pow(3 - value.as_str().len() as u32)
});
let start = captures[1].parse::<i64>().unwrap() * 60_000
+ captures[2].parse::<i64>().unwrap() * 1000
+ fraction;
// Proxy eLRC may round to centiseconds; prefer an exact match.
let source = lines
.iter()
.filter(|source| !source.agent.is_empty())
.min_by_key(|source| (source.timestamp as i64 - start).abs())
.filter(|source| (source.timestamp as i64 - start).abs() <= 10);
let Some(source) = source else {
return line.to_owned();
};
let Some(voice) = voices.get(source.agent.as_str()) else {
return line.to_owned();
};
let distance = (source.timestamp as i64 - start).abs();
if lines.iter().any(|other| {
(other.timestamp as i64 - start).abs() == distance
&& voices.get(other.agent.as_str()) != Some(voice)
}) {
// Simultaneous independent lines cannot be identified by time
// alone. Retain their supplied labels rather than swapping them.
return line.to_owned();
}
let prefix = captures.get(4).unwrap();
let mut corrected = line.to_owned();
corrected.replace_range(prefix.range(), voice);
corrected
})
.collect::<Vec<_>>()
.join("\n")
}
#[derive(Default)]
struct ProxyPayload {
kind: String,
@@ -134,7 +226,7 @@ pub fn format_apple(raw: &str, multi_person: bool, word_timing: bool) -> Result<
.any(|value| !value.trim().is_empty()))
{
if word_timing && multi_person && !value.elrc_multi_person.trim().is_empty() {
return Ok(value.elrc_multi_person.trim().into());
return Ok(apple_vocal_sides(value.elrc_multi_person.trim(), &value));
}
if word_timing && !value.elrc.trim().is_empty() {
return Ok(value.elrc.trim().into());
@@ -146,12 +238,12 @@ pub fn format_apple(raw: &str, multi_person: bool, word_timing: bool) -> Result<
if content.is_empty() {
return Err("unsupported apple music lyrics payload".into());
}
return Ok(format_pax_content(
&value.kind,
content,
multi_person,
word_timing,
));
let text = format_pax_content(&value.kind, content, multi_person, word_timing);
return Ok(if multi_person {
apple_vocal_sides(&text, &value)
} else {
text
});
}
if let Ok(Some(lines)) = json::decode::<Option<Vec<PaxLine>>>(raw)
&& !lines.is_empty()
@@ -462,6 +554,79 @@ pub fn format_kpoe(response: &KpoeResponse, multi_person: bool, word_timing: boo
mod supplement_tests {
use super::*;
#[test]
fn apple_agents_restore_vocal_sides_without_reformatting_word_times() {
let raw = serde_json::json!({
"type": "Syllable",
"elrcMultiPerson": "[00:01.01]v1: <00:01.009>Lead<00:02.00>\n[bg:<00:01.50>Echo<00:02.50>]\n[00:03.00]v2: <00:03.00>Guest<00:04.00>\n[00:05.00]v2: <00:05.00>Together<00:06.00>\n[00:07.00]v2: Third",
"content": [
{"timestamp": 1009, "agent": "lead"},
{"timestamp": 3000, "agent": "guest"},
{"timestamp": 5000, "agent": "all"},
{"timestamp": 7000, "agent": "third"}
],
"metadata": {"agents": [
{"id": "lead", "type": "person"},
{"id": "guest", "type": "person"},
{"id": "all", "type": "group"},
{"id": "third", "type": "person"}
]}
});
let text = format_apple(&raw.to_string(), true, true).unwrap();
assert_eq!(
text,
"[00:01.01]v1: <00:01.009>Lead<00:02.00>\n[bg:<00:01.50>Echo<00:02.50>]\n[00:03.00]v2: <00:03.00>Guest<00:04.00>\n[00:05.00]v1: <00:05.00>Together<00:06.00>\n[00:07.00]v3: Third"
);
let lyrics = LyricsResponse::from_text(&text, "Apple Music", "Apple Music");
let stored = lrc::with_metadata(&lyrics, "Track", "Artist");
assert!(stored.contains("[00:05.00]v1: <00:05.00>Together<00:06.00>"));
assert!(stored.contains("[bg:<00:01.50>Echo<00:02.50>]"));
}
#[test]
fn apple_content_fallback_honors_agents_only_when_multi_person_is_enabled() {
let raw = serde_json::json!({
"type": "Syllable",
"content": [{"timestamp": 1000, "oppositeTurn": true, "agent": "group", "text": [
{"text": "Together", "timestamp": 1000, "endtime": 2000}
]}],
"metadata": {"agents": [{"id": "group", "type": "group"}]}
})
.to_string();
for timing in [false, true] {
assert!(
format_apple(&raw, true, timing)
.unwrap()
.starts_with("[00:01.00]v1:")
);
assert!(!format_apple(&raw, false, timing).unwrap().contains("v1:"));
assert!(!format_apple(&raw, false, timing).unwrap().contains("v2:"));
}
}
#[test]
fn apple_voice_correction_preserves_unknown_or_ambiguous_lines() {
let text = "[00:01.00]v2: Unknown\n[00:02.00]v2: Too far\n[00:03.00]v2: Ambiguous\n[00:04.00]v2: Exact";
let mut raw = serde_json::json!({"elrcMultiPerson": text});
assert_eq!(format_apple(&raw.to_string(), true, true).unwrap(), text);
raw["metadata"] = serde_json::json!({"agents": [
{"id": "v1", "type": "person"}, {"id": "v2", "type": "person"},
{"id": "v3", "type": "group"}
]});
raw["content"] = serde_json::json!([
{"timestamp": 1000, "agent": "unknown"},
{"timestamp": 2011, "agent": "v3"},
{"timestamp": 3000, "agent": "v3"},
{"timestamp": 3000, "agent": "v2"},
{"timestamp": 3999, "agent": "v2"},
{"timestamp": 4000, "agent": "v3"}
]);
assert_eq!(
format_apple(&raw.to_string(), true, true).unwrap(),
text.replace("[00:04.00]v2:", "[00:04.00]v1:")
);
}
fn payload(lang: &str) -> serde_json::Value {
serde_json::json!({
"type": "Syllable",