Preserve Arabic spelling choices
This commit is contained in:
+101
-1
@@ -14,7 +14,11 @@ public struct ArabicPronunciationProcessor: Sendable {
|
||||
)
|
||||
}
|
||||
|
||||
var processedText = lyrics.precomposedStringWithCanonicalMapping
|
||||
let protection = protectPreservedSpellings(
|
||||
in: lyrics,
|
||||
preservedSpellings: settings.preservedSpellings
|
||||
)
|
||||
var processedText = protection.text.precomposedStringWithCanonicalMapping
|
||||
var notes = processingNotes(for: processedText, settings: settings)
|
||||
|
||||
if settings.tanweenPolicy == .removeWhenUnwanted {
|
||||
@@ -27,6 +31,15 @@ public struct ArabicPronunciationProcessor: Sendable {
|
||||
notes.append(.tanweenAdditionNeedsReview)
|
||||
}
|
||||
|
||||
processedText = restorePreservedSpellings(
|
||||
in: processedText,
|
||||
replacements: protection.replacements
|
||||
)
|
||||
|
||||
if protection.matchCount > 0 {
|
||||
notes.append(.preservedSpellingsProtected(count: protection.matchCount))
|
||||
}
|
||||
|
||||
return ArabicPronunciationProcessingResult(
|
||||
text: processedText,
|
||||
notes: notes
|
||||
@@ -53,6 +66,81 @@ public struct ArabicPronunciationProcessor: Sendable {
|
||||
}
|
||||
}
|
||||
|
||||
private func protectPreservedSpellings(
|
||||
in text: String,
|
||||
preservedSpellings: [String]
|
||||
) -> PreservedSpellingProtection {
|
||||
let spellings = Set(
|
||||
preservedSpellings.map {
|
||||
$0.trimmingCharacters(in: .whitespacesAndNewlines)
|
||||
}
|
||||
)
|
||||
.filter { !$0.isEmpty }
|
||||
.sorted {
|
||||
if $0.count == $1.count {
|
||||
return $0 < $1
|
||||
}
|
||||
return $0.count > $1.count
|
||||
}
|
||||
|
||||
var protectedText = text
|
||||
var replacements: [PreservedSpellingReplacement] = []
|
||||
var matchCount = 0
|
||||
|
||||
for (index, spelling) in spellings.enumerated() {
|
||||
let occurrenceCount = protectedText.components(separatedBy: spelling).count - 1
|
||||
guard occurrenceCount > 0 else {
|
||||
continue
|
||||
}
|
||||
|
||||
let placeholder = uniquePlaceholder(
|
||||
for: index,
|
||||
in: protectedText
|
||||
)
|
||||
protectedText = protectedText.replacingOccurrences(
|
||||
of: spelling,
|
||||
with: placeholder
|
||||
)
|
||||
replacements.append(
|
||||
PreservedSpellingReplacement(
|
||||
placeholder: placeholder,
|
||||
spelling: spelling
|
||||
)
|
||||
)
|
||||
matchCount += occurrenceCount
|
||||
}
|
||||
|
||||
return PreservedSpellingProtection(
|
||||
text: protectedText,
|
||||
replacements: replacements,
|
||||
matchCount: matchCount
|
||||
)
|
||||
}
|
||||
|
||||
private func uniquePlaceholder(for index: Int, in text: String) -> String {
|
||||
var collisionIndex = 0
|
||||
var placeholder = "[[music-assistant-preserved-\(index)]]"
|
||||
|
||||
while text.contains(placeholder) {
|
||||
collisionIndex += 1
|
||||
placeholder = "[[music-assistant-preserved-\(index)-\(collisionIndex)]]"
|
||||
}
|
||||
|
||||
return placeholder
|
||||
}
|
||||
|
||||
private func restorePreservedSpellings(
|
||||
in text: String,
|
||||
replacements: [PreservedSpellingReplacement]
|
||||
) -> String {
|
||||
replacements.reduce(text) { result, replacement in
|
||||
result.replacingOccurrences(
|
||||
of: replacement.placeholder,
|
||||
with: replacement.spelling
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
private func removeTanween(from text: String) -> String {
|
||||
String(text.unicodeScalars.filter { scalar in
|
||||
!isTanween(scalar)
|
||||
@@ -85,6 +173,17 @@ public struct ArabicPronunciationProcessor: Sendable {
|
||||
}
|
||||
}
|
||||
|
||||
private struct PreservedSpellingProtection {
|
||||
let text: String
|
||||
let replacements: [PreservedSpellingReplacement]
|
||||
let matchCount: Int
|
||||
}
|
||||
|
||||
private struct PreservedSpellingReplacement {
|
||||
let placeholder: String
|
||||
let spelling: String
|
||||
}
|
||||
|
||||
public struct ArabicPronunciationProcessingResult: Equatable, Sendable {
|
||||
public var text: String
|
||||
public var notes: [ArabicPronunciationProcessingNote]
|
||||
@@ -101,4 +200,5 @@ public enum ArabicPronunciationProcessingNote: Equatable, Sendable {
|
||||
case pronunciationTargetedDiacriticsNeedReview
|
||||
case fullTashkeelNeedsReview
|
||||
case tanweenAdditionNeedsReview
|
||||
case preservedSpellingsProtected(count: Int)
|
||||
}
|
||||
|
||||
@@ -64,4 +64,43 @@ final class ArabicPronunciationProcessorTests: XCTestCase {
|
||||
[.fullTashkeelNeedsReview, .tanweenAdditionNeedsReview]
|
||||
)
|
||||
}
|
||||
|
||||
func testPreservedSpellingsKeepTheirTanweenWhileOtherTextIsProcessed() {
|
||||
let protectedSpelling = "\u{0647}\u{064F}\u{062F}\u{064B}\u{0649}"
|
||||
let otherText = "\u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{064B}\u{0627}"
|
||||
let lyrics = "\(protectedSpelling) \(otherText) \(protectedSpelling)"
|
||||
|
||||
let result = processor.process(
|
||||
lyrics: lyrics,
|
||||
settings: ArabicPronunciationSettings(
|
||||
isEnabled: true,
|
||||
diacritizationPolicy: .pronunciationTargeted,
|
||||
tanweenPolicy: .removeWhenUnwanted,
|
||||
preservedSpellings: [protectedSpelling]
|
||||
)
|
||||
)
|
||||
|
||||
XCTAssertEqual(
|
||||
result.text,
|
||||
"\(protectedSpelling) \u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{0627} \(protectedSpelling)"
|
||||
)
|
||||
XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 2)])
|
||||
}
|
||||
|
||||
func testPreservedSpellingsKeepTheirOriginalDiacriticOrder() {
|
||||
let preservedSpelling = "\u{0645}\u{0651}\u{064E}"
|
||||
let otherText = "\u{0628}\u{0651}\u{064E}"
|
||||
|
||||
let result = processor.process(
|
||||
lyrics: "\(preservedSpelling) \(otherText)",
|
||||
settings: ArabicPronunciationSettings(
|
||||
isEnabled: true,
|
||||
diacritizationPolicy: .pronunciationTargeted,
|
||||
preservedSpellings: [preservedSpelling]
|
||||
)
|
||||
)
|
||||
|
||||
XCTAssertEqual(result.text, "\(preservedSpelling) \u{0628}\u{064E}\u{0651}")
|
||||
XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 1)])
|
||||
}
|
||||
}
|
||||
|
||||
@@ -66,7 +66,9 @@ tanween policy without choosing a diacritization policy for the user.
|
||||
It must not invent missing vowel marks or tanween when no reliable
|
||||
linguistic source is available; instead, it returns a review note so a
|
||||
later user-review flow can present the unresolved text. This processor
|
||||
does not use external services or APIs.
|
||||
does not use external services or APIs. Before processing, it protects
|
||||
the user's exact preserved spellings and dialect phrases so normalization
|
||||
or tanween removal cannot alter them.
|
||||
|
||||
## Security
|
||||
|
||||
|
||||
+1
-1
@@ -69,7 +69,7 @@ requirement is missing and blocks implementation, record it in
|
||||
|
||||
- [x] Add Arabic-specific settings UI.
|
||||
- [x] Support diacritics/harakat/tanween processing.
|
||||
- [ ] Preserve intentional spelling/dialect choices where possible.
|
||||
- [x] Preserve intentional spelling/dialect choices where possible.
|
||||
- [ ] Allow user to compare/edit processed Arabic before Suno handoff.
|
||||
- [ ] Add Arabic test fixtures covering multiple dialects.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user