fix: sync lyric romanization and clear stale selection menus

Preserve romanization and translation metadata through Rust exports and cache, match rounded line times, and retain word endings. Highlight romanization with the main lyric clock in Material and Mornye.

Clear selection overlay ownership safely when collection screens are disposed. Add parser, rendering, cache and overlay lifecycle regression coverage.
This commit is contained in:
zarzet
2026-09-23 12:42:08 +07:00
parent ebcd1fd877
commit ba44bafa92
14 changed files with 1286 additions and 67 deletions
+23
View File
@@ -41,18 +41,41 @@ pub fn text_from_bytes(bytes: &[u8]) -> String {
crate::text::utf8(bytes)
}
#[derive(Clone, Debug, Default, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct LyricsWord {
pub text: String,
pub start_time_ms: i64,
pub end_time_ms: i64,
}
json::go_deserialize!(LyricsWord {
"text" => text,
"starttimems" => start_time_ms,
"endtimems" => end_time_ms,
});
#[derive(Clone, Debug, Default, PartialEq, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct LyricsLine {
pub start_time_ms: i64,
pub words: String,
pub end_time_ms: i64,
#[serde(skip_serializing_if = "Option::is_none")]
pub romanization: Option<String>,
#[serde(skip_serializing_if = "Option::is_none")]
pub romanization_words: Option<Vec<LyricsWord>>,
#[serde(skip_serializing_if = "Option::is_none")]
pub translation: Option<String>,
}
json::go_deserialize!(LyricsLine {
"starttimems" => start_time_ms,
"words" => words,
"endtimems" => end_time_ms,
"romanization" => romanization,
"romanizationwords" => romanization_words,
"translation" => translation,
});
#[derive(Clone, Debug, Default, PartialEq, Serialize)]
+65 -7
View File
@@ -1,12 +1,13 @@
use super::{LyricsLine, LyricsResponse};
use crate::matching::lowercase;
use base64::{Engine, engine::general_purpose::STANDARD};
use regex::Regex;
use std::sync::LazyLock;
static TIMED_LINE: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"\[([0-9]{2}):([0-9]{2})\.([0-9]{2,3})\](.*)").unwrap());
static METADATA: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"(?i)^\[[a-z][a-z0-9_]*:.*\]$").unwrap());
LazyLock::new(|| Regex::new(r"(?i)^\[[a-z][a-z0-9_-]*:.*\]$").unwrap());
static BACKGROUND: LazyLock<Regex> = LazyLock::new(|| Regex::new(r"(?i)^\[bg:(.*)\]$").unwrap());
static LEADING_TIME: LazyLock<Regex> =
LazyLock::new(|| Regex::new(r"^\[[0-9]{1,3}:[0-9]{1,2}(?:[.:][0-9]{1,3})?\]").unwrap());
@@ -78,6 +79,7 @@ pub fn parse_synced(raw: &str) -> Option<Vec<LyricsLine>> {
start_time_ms: minutes * 60_000 + seconds * 1000 + fraction,
words: words.into(),
end_time_ms: 0,
..LyricsLine::default()
});
}
}
@@ -113,13 +115,14 @@ pub fn plain_from_timed_lines(lines: &[LyricsLine]) -> String {
}
pub fn timestamp_inline(ms: i64) -> String {
let ms = ms.max(0);
let seconds = ms / 1000;
format!(
"{:02}:{:02}.{:02}",
seconds / 60,
seconds % 60,
ms % 1000 / 10
)
let fraction = if ms % 10 == 0 {
format!("{:02}", ms % 1000 / 10)
} else {
format!("{:03}", ms % 1000)
};
format!("{:02}:{:02}.{fraction}", seconds / 60, seconds % 60)
}
pub fn timestamp(ms: i64) -> String {
@@ -175,6 +178,32 @@ pub fn with_metadata(lyrics: &LyricsResponse, track: &str, artist: &str) -> Stri
output.push_str(&format!(" (source: {source})"));
}
output.push_str("]\n\n");
if lyrics.sync_type == "LINE_SYNCED" {
for line in lyrics.lines() {
if let Some(words) = &line.romanization_words
&& !words.is_empty()
&& let Ok(json) = serde_json::to_vec(words)
{
output.push_str(&format!(
"[x-romaji-words:{}:{}]\n",
line.start_time_ms,
STANDARD.encode(json)
));
}
for (tag, text) in [
("x-romaji", &line.romanization),
("x-translation", &line.translation),
] {
if let Some(text) = text.as_deref().filter(|text| !text.trim().is_empty()) {
output.push_str(&format!(
"[{tag}:{}:{}]\n",
line.start_time_ms,
STANDARD.encode(text.trim())
));
}
}
}
}
for line in lyrics.lines() {
if line.words.is_empty() {
continue;
@@ -187,3 +216,32 @@ pub fn with_metadata(lyrics: &LyricsResponse, track: &str, artist: &str) -> Stri
}
output
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn writing_timestamps_never_discards_milliseconds() {
for time in [0, 1000, 1009, 12345, 59999, 60000, 3599999] {
let raw = format!("{}Original", timestamp(time));
assert_eq!(parse_synced(&raw).unwrap()[0].start_time_ms, time);
}
assert_eq!(timestamp_inline(1000), "00:01.00");
assert_eq!(timestamp_inline(1009), "00:01.009");
}
#[test]
fn optional_text_is_safe_metadata_not_lyric_content() {
let raw = "[x-romaji:1000:YQ==]\n[x-translation:1000:Yg==]";
assert!(!has_usable_content(raw));
let mut lyrics =
LyricsResponse::from_text("[00:01.009]Original", "Apple Music", "Apple Music");
lyrics.lines.as_mut().unwrap()[0].translation = Some("Text ]\nnext line 日本語".into());
let lrc = with_metadata(&lyrics, "Track", "Artist");
let encoded = STANDARD.encode("Text ]\nnext line 日本語");
assert!(lrc.contains(&format!("[x-translation:1009:{encoded}]")));
assert!(lrc.contains("[00:01.009]Original"));
assert!(has_usable_content(&lrc));
}
}
+346 -2
View File
@@ -1,7 +1,7 @@
use super::{LyricsResponse, json, lrc};
use super::{LyricsResponse, LyricsWord, json, lrc};
use serde::Serialize;
use serde_json::value::RawValue;
use std::collections::BTreeMap;
use std::collections::{BTreeMap, BTreeSet};
#[derive(Clone, Debug, Default, Serialize)]
pub struct PaxDetail {
@@ -166,6 +166,153 @@ pub fn format_apple(raw: &str, multi_person: bool, word_timing: bool) -> Result<
Err("failed to parse pax lyrics response".into())
}
/// Attach optional Apple text to the actual output lines once. The proxy's
/// eLRC can round in either direction, so metadata timestamps need not be
/// identical to the selected line timestamps. Exact matches win collisions.
pub fn apple_supplements(raw: &str, lyrics: &mut LyricsResponse) {
if lyrics.sync_type != "LINE_SYNCED" {
return;
}
#[derive(Default)]
struct Payload {
metadata: Option<Box<RawValue>>,
}
json::go_deserialize!(Payload { "metadata" => metadata, });
let metadata = json::decode::<Payload>(raw)
.ok()
.and_then(|payload| payload.metadata)
.and_then(|raw| json::decode::<serde_json::Value>(raw.get()).ok());
let Some(metadata) = metadata else { return };
let starts: BTreeSet<_> = lyrics
.lines()
.iter()
.map(|line| line.start_time_ms)
.collect();
let language = metadata["language"].as_str().unwrap_or_default();
let romanization =
apple_supplement_lines(&metadata["transliterations"], &starts, language, |lang| {
lang.split('-')
.any(|part| part.eq_ignore_ascii_case("Latn"))
});
let translation = apple_supplement_lines(&metadata["translations"], &starts, "en", |lang| {
lang.split('-')
.next()
.is_some_and(|part| part.eq_ignore_ascii_case("en"))
});
for line in lyrics.lines.iter_mut().flatten() {
if let Some(supplement) = romanization.get(&line.start_time_ms) {
line.romanization = Some(supplement.text.clone());
line.romanization_words = supplement.words.clone();
}
line.translation = translation
.get(&line.start_time_ms)
.map(|line| line.text.clone());
}
}
struct AppleSupplement {
text: String,
words: Option<Vec<LyricsWord>>,
}
// Spans are syllables, not necessarily words: "utsu" + "kushii" must remain
// joined. Recover spaces from the full text instead of adding one per span.
fn apple_supplement_words(text: &str, spans: &serde_json::Value) -> Option<Vec<LyricsWord>> {
let mut words: Vec<LyricsWord> = Vec::new();
let mut cursor = 0;
for span in spans.as_array()? {
let part = span["text"].as_str()?.trim();
if part.is_empty() {
continue;
}
let start = span["begin"].as_i64()?;
let end = span["end"].as_i64()?;
if start < 0 || end < start || words.last().is_some_and(|word| word.start_time_ms > start) {
return None;
}
let offset = text[cursor..].find(part)?;
let gap = &text[cursor..cursor + offset];
if !gap.chars().all(char::is_whitespace) {
return None;
}
if let Some(previous) = words.last_mut() {
previous.text.push_str(gap);
}
words.push(LyricsWord {
text: part.into(),
start_time_ms: start,
end_time_ms: end,
});
cursor += offset + part.len();
}
if words.is_empty() || !text[cursor..].trim().is_empty() {
return None;
}
Some(words)
}
fn apple_supplement_lines(
groups: &serde_json::Value,
starts: &BTreeSet<i64>,
language: &str,
accepts_language: impl Fn(&str) -> bool,
) -> BTreeMap<i64, AppleSupplement> {
let mut best = BTreeMap::new();
let mut best_rank = (false, 0);
for group in groups.as_array().into_iter().flatten() {
let Some(lang) = group["lang"].as_str().filter(|lang| accepts_language(lang)) else {
continue;
};
let mut matched: BTreeMap<i64, (u64, AppleSupplement)> = BTreeMap::new();
for line in group["lines"].as_array().into_iter().flatten() {
let (Some(time), Some(text)) = (line["timestamp"].as_i64(), line["text"].as_str())
else {
continue;
};
if time < 0 || text.trim().is_empty() {
continue;
}
let Some(&start) = starts
.range(time.saturating_sub(10)..=time.saturating_add(10))
.min_by_key(|&&start| (start.abs_diff(time), start))
else {
continue;
};
let distance = start.abs_diff(time);
if matched.get(&start).is_none_or(|(old, _)| distance < *old) {
let text = text.trim();
matched.insert(
start,
(
distance,
AppleSupplement {
text: text.into(),
words: apple_supplement_words(text, &line["spans"]),
},
),
);
}
}
// Prefer the song's language, then coverage. Empty or malformed
// optional groups never suppress a usable later alternative.
let same_language = lang
.split('-')
.next()
.unwrap_or_default()
.eq_ignore_ascii_case(language.split('-').next().unwrap_or_default());
let rank = (same_language, matched.len());
if !matched.is_empty() && (best.is_empty() || rank > best_rank) {
best_rank = rank;
best = matched
.into_iter()
.map(|(start, (_, text))| (start, text))
.collect();
}
}
best
}
/// Failure strings describe unavailable proxy payloads, not an absent track.
pub fn parse_proxy(
raw: &str,
@@ -310,3 +457,200 @@ pub fn format_kpoe(response: &KpoeResponse, multi_person: bool, word_timing: boo
}
lines.join("\n").trim().into()
}
#[cfg(test)]
mod supplement_tests {
use super::*;
fn payload(lang: &str) -> serde_json::Value {
serde_json::json!({
"type": "Syllable",
"elrc": "[00:01.010]<00:01.009>Original <00:01.307>\n[00:02.000]Next",
"content": [{"timestamp": 1009, "text": [
{"text": "Original", "timestamp": 1009, "endtime": 1307}
]}, {"timestamp": 2001, "text": [{"text": "Next"}]}],
"metadata": {
"language": lang,
"transliterations": [{"lang": format!("{lang}-Latn"), "lines": [
{"timestamp": 1009, "text": "Romanized"},
{"timestamp": 2001, "text": "Next romanized"}
]}],
"translations": [{"lang": "en-US", "lines": [
{"timestamp": 1009, "text": "English"},
{"timestamp": 2001, "text": "Next English"}
]}]
}
})
}
#[test]
fn supplements_follow_output_timestamps_in_both_apple_modes() {
for lang in ["ja", "ko", "zh-Hans"] {
for word_timing in [false, true] {
let raw = payload(lang).to_string();
let text = format_apple(&raw, false, word_timing).unwrap();
let mut lyrics = LyricsResponse::from_text(&text, "Apple Music", "Apple Music");
apple_supplements(&raw, &mut lyrics);
assert_eq!(lyrics.lines()[0].romanization.as_deref(), Some("Romanized"));
assert_eq!(
lyrics.lines()[1].translation.as_deref(),
Some("Next English")
);
let start = if word_timing { 1010 } else { 1009 };
assert_eq!(lyrics.lines()[0].start_time_ms, start);
let exported = lrc::with_metadata(&lyrics, "Track", "Artist");
assert!(exported.contains(&format!("[x-romaji:{start}:Um9tYW5pemVk]")));
assert!(exported.contains(&lrc::timestamp(start)));
if word_timing {
assert!(exported.contains("<00:01.307>"));
}
// Persistence uses this same JSON response contract.
let restored: LyricsResponse =
serde_json::from_str(&serde_json::to_string(&lyrics).unwrap()).unwrap();
assert_eq!(restored, lyrics);
}
}
}
#[test]
fn nearest_matching_is_bounded_and_exact_matches_win() {
let mut raw = payload("ja");
raw["metadata"]["transliterations"][0]["lines"] = serde_json::json!([
{"timestamp": 1009, "text": "Rounded"},
{"timestamp": 1010, "text": "Exact"},
{"timestamp": 1001, "text": "Less accurate"},
{"timestamp": 2011, "text": "Too far"},
{"timestamp": 2990, "text": "Boundary"},
{"timestamp": 4005, "text": "Tie"}
]);
let mut lyrics = LyricsResponse::from_text(
"[00:01.010]A\n[00:02.00]B\n[00:03.00]C\n[00:04.00]D\n[00:04.010]E",
"",
"",
);
apple_supplements(&raw.to_string(), &mut lyrics);
let actual: Vec<_> = lyrics
.lines()
.iter()
.map(|line| line.romanization.as_deref())
.collect();
assert_eq!(
actual,
[Some("Exact"), None, Some("Boundary"), Some("Tie"), None]
);
}
#[test]
fn malformed_optional_lines_do_not_hide_other_supplements() {
let mut raw = payload("ja");
raw["metadata"]["transliterations"][0]["lines"] = serde_json::json!([
null, {"text": "Missing time"}, {"timestamp": "1009", "text": "Bad time"},
{"timestamp": 1009, "text": []}, {"timestamp": -1, "text": "Negative"},
{"timestamp": 2001, "text": " Next romanized "}
]);
let mut lyrics = LyricsResponse::from_text("[00:01.01]A\n[00:02.00]B", "", "");
apple_supplements(&raw.to_string(), &mut lyrics);
assert!(lyrics.lines()[0].romanization.is_none());
assert_eq!(lyrics.lines()[0].translation.as_deref(), Some("English"));
assert_eq!(
lyrics.lines()[1].romanization.as_deref(),
Some("Next romanized")
);
}
#[test]
fn language_selection_prefers_song_language_and_usable_groups() {
let mut raw = payload("ja");
raw["metadata"]["transliterations"] = serde_json::json!([
{"lang": "ja-Latn", "lines": []},
{"lang": "ko-Latn", "lines": [{"timestamp": 1009, "text": "Other language"}]},
{"lang": "ja-Kana", "lines": [{"timestamp": 1009, "text": "Not Latin"}]},
{"lang": "ja-Latn", "lines": [{"timestamp": 1009, "text": "Romanized"}]}
]);
raw["metadata"]["translations"] = serde_json::json!([
{"lang": "en", "lines": null},
{"lang": "fr", "lines": [{"timestamp": 1009, "text": "French"}]},
{"lang": "en-GB", "lines": [{"timestamp": 1009, "text": "English"}]}
]);
let mut lyrics = LyricsResponse::from_text("[00:01.01]Original", "", "");
apple_supplements(&raw.to_string(), &mut lyrics);
assert_eq!(lyrics.lines()[0].romanization.as_deref(), Some("Romanized"));
assert_eq!(lyrics.lines()[0].translation.as_deref(), Some("English"));
}
#[test]
fn absent_or_invalid_metadata_preserves_original_lyrics() {
for raw in [
"{}",
"not json",
r#"{"metadata":null}"#,
r#"{"metadata":{"translations":5}}"#,
] {
let mut lyrics = LyricsResponse::from_text("[00:01.00]Original", "", "");
let original = lyrics.clone();
apple_supplements(raw, &mut lyrics);
assert_eq!(lyrics, original);
}
let mut plain = LyricsResponse::from_text("Original", "", "");
apple_supplements(&payload("ja").to_string(), &mut plain);
assert!(plain.lines()[0].romanization.is_none());
}
#[test]
fn romanization_spans_keep_syllables_spaces_and_original_word_times() {
let mut raw = payload("ja");
raw["metadata"]["transliterations"][0]["lines"][0] = serde_json::json!({
"timestamp": 1009, "text": "Roma nized", "spans": [
{"text": "Ro", "begin": 1009, "end": 1107},
{"text": "ma", "begin": 1107, "end": 1307},
{"text": "nized", "begin": 2003, "end": 2497}
]
});
let mut lyrics = LyricsResponse::from_text("[00:01.010]Original", "", "");
apple_supplements(&raw.to_string(), &mut lyrics);
let words = lyrics.lines()[0].romanization_words.as_ref().unwrap();
assert_eq!(
words
.iter()
.map(|word| word.text.as_str())
.collect::<Vec<_>>(),
["Ro", "ma ", "nized"]
);
assert_eq!(words[0].start_time_ms, 1009);
assert_eq!(words[1].end_time_ms, 1307);
assert_eq!(words[2].start_time_ms, 2003);
assert_eq!(words[2].end_time_ms, 2497);
let output = lrc::with_metadata(&lyrics, "Track", "Artist");
let encoded = output
.lines()
.find_map(|line| {
line.strip_prefix("[x-romaji-words:1010:")
.and_then(|line| line.strip_suffix(']'))
})
.unwrap();
use base64::Engine;
let decoded = base64::engine::general_purpose::STANDARD
.decode(encoded)
.unwrap();
let restored: Vec<LyricsWord> = serde_json::from_slice(&decoded).unwrap();
assert_eq!(&restored, words);
}
#[test]
fn invalid_romanization_timing_falls_back_to_readable_text() {
for spans in [
serde_json::json!([{"text": "Other", "begin": 1009, "end": 1307}]),
serde_json::json!([{"text": "Romanized", "begin": 1009, "end": 900}]),
serde_json::json!([{"text": "Romanized", "begin": 1009}]),
serde_json::json!([{"text": "Ro", "begin": 1009, "end": 1307}]),
] {
let mut raw = payload("ja");
raw["metadata"]["transliterations"][0]["lines"][0]["spans"] = spans;
let mut lyrics = LyricsResponse::from_text("[00:01.010]Original", "", "");
apple_supplements(&raw.to_string(), &mut lyrics);
assert_eq!(lyrics.lines()[0].romanization.as_deref(), Some("Romanized"));
assert!(lyrics.lines()[0].romanization_words.is_none());
}
}
}
@@ -166,7 +166,9 @@ impl BuiltinLyricsClient {
Err(error) if raw.starts_with(['{', '[']) => return Err(LyricsError::Other(error)),
Err(_) => raw.into(),
};
from_text(&text, "Apple Music", "Apple Music")
let mut lyrics = from_text(&text, "Apple Music", "Apple Music")?;
payloads::apple_supplements(raw, &mut lyrics);
Ok(lyrics)
}
}
@@ -13,6 +13,7 @@ use std::time::{Duration, SystemTime, UNIX_EPOCH};
pub const MAX_ENTRIES: usize = 500;
pub const TTL: Duration = Duration::from_secs(24 * 60 * 60);
const MAX_PERSISTED_BYTES: u64 = 64 << 20;
const SNAPSHOT_VERSION: u32 = 3;
#[derive(Clone)]
struct Entry {
@@ -143,7 +144,7 @@ impl LyricsCache {
let loaded = read_snapshot(path).unwrap_or_default();
let mut state = self.inner.state.lock().expect("lyrics cache lock");
state.path = Some(path.to_owned());
if loaded.version == 1 {
if (1..=SNAPSHOT_VERSION).contains(&loaded.version) {
for (key, entry) in loaded.entries {
if state.entries.len() >= MAX_ENTRIES {
break;
@@ -151,6 +152,11 @@ impl LyricsCache {
let Some(response) = entry.response else {
continue;
};
// Older versions discarded Apple text or romanization timing.
// Refetch once, preserving other providers' caches.
if loaded.version < SNAPSHOT_VERSION && response.provider == "Apple Music" {
continue;
}
let Some(expires_at) = u64::try_from(entry.expires_at)
.ok()
.and_then(|seconds| UNIX_EPOCH.checked_add(Duration::from_secs(seconds)))
@@ -267,7 +273,7 @@ fn write_snapshot(path: &Path, entries: &BTreeMap<String, Entry>) -> std::io::Re
entries: BTreeMap<&'a str, BorrowedEntry<'a>>,
}
let snapshot = BorrowedSnapshot {
version: 1,
version: SNAPSHOT_VERSION,
entries: entries
.iter()
.map(|(key, entry)| {
@@ -305,3 +311,54 @@ fn write_snapshot(path: &Path, entries: &BTreeMap<String, Entry>) -> std::io::Re
temporary.persist(path).map_err(|error| error.error)?;
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn legacy_snapshot_refetches_apple_but_preserves_other_providers() {
let file = tempfile::NamedTempFile::new().unwrap();
let now = UNIX_EPOCH + Duration::from_secs(100);
let apple = LyricsResponse::from_text("[00:01.00]Original", "Apple Music", "Apple Music");
let other = LyricsResponse::from_text("[00:01.00]Original", "LRCLIB", "LRCLIB");
for version in 1..SNAPSHOT_VERSION {
let snapshot = serde_json::json!({"version": version, "entries": {
"apple": {"response": apple, "expires_at": 200},
"other": {"response": other, "expires_at": 200}
}});
fs::write(file.path(), serde_json::to_vec(&snapshot).unwrap()).unwrap();
let cache = LyricsCache::default();
cache.set_persistence_path(file.path(), now);
assert!(cache.get("apple", now).is_none());
assert_eq!(cache.get("other", now).unwrap().provider, "LRCLIB");
}
}
#[test]
fn current_snapshot_restores_supplementary_text() {
let file = tempfile::NamedTempFile::new().unwrap();
let now = UNIX_EPOCH + Duration::from_secs(100);
let mut lyrics =
LyricsResponse::from_text("[00:01.009]Original", "Apple Music", "Apple Music");
let line = &mut lyrics.lines.as_mut().unwrap()[0];
line.romanization = Some("Romanized".into());
line.romanization_words = Some(vec![spotiflac_core::lyrics::LyricsWord {
text: "Romanized".into(),
start_time_ms: 1009,
end_time_ms: 1307,
}]);
line.translation = Some("English".into());
let entries = BTreeMap::from([(
"apple".into(),
Entry {
response: Arc::new(lyrics.clone()),
expires_at: now + Duration::from_secs(100),
},
)]);
write_snapshot(file.path(), &entries).unwrap();
let cache = LyricsCache::default();
cache.set_persistence_path(file.path(), now);
assert_eq!(cache.get("apple", now), Some(lyrics));
}
}