mirror of
https://github.com/wiltodelta/remove-ai-watermarks.git
synced 2026-08-19 20:17:12 +02:00
Evaluate selective text restoration
This commit is contained in:
@@ -0,0 +1,34 @@
|
||||
"""Pure text-normalization helpers shared by evaluation scripts."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unicodedata
|
||||
|
||||
|
||||
def normalize_text(text: str) -> str:
|
||||
"""Normalize text for layout-independent evaluation comparisons."""
|
||||
return "".join(unicodedata.normalize("NFC", text).casefold().split())
|
||||
|
||||
|
||||
def levenshtein_normalized(left: str, right: str) -> float:
|
||||
"""Return normalized Levenshtein distance without changing either input."""
|
||||
if not left and not right:
|
||||
return 0.0
|
||||
previous = list(range(len(right) + 1))
|
||||
for left_index, left_character in enumerate(left, start=1):
|
||||
current = [left_index]
|
||||
for right_index, right_character in enumerate(right, start=1):
|
||||
current.append(
|
||||
min(
|
||||
current[-1] + 1,
|
||||
previous[right_index] + 1,
|
||||
previous[right_index - 1] + (left_character != right_character),
|
||||
)
|
||||
)
|
||||
previous = current
|
||||
return previous[-1] / max(len(left), len(right))
|
||||
|
||||
|
||||
def normalized_edit_distance(left: str, right: str) -> float:
|
||||
"""Normalize text, then return Levenshtein distance over the result."""
|
||||
return levenshtein_normalized(normalize_text(left), normalize_text(right))
|
||||
Reference in New Issue
Block a user