Preserve Arabic spelling choices

This commit is contained in:
diyaa
2026-09-13 22:33:57 +02:00
parent f5ab97e6a0
commit 3865f77dc0
4 changed files with 144 additions and 3 deletions
@@ -14,7 +14,11 @@ public struct ArabicPronunciationProcessor: Sendable {
) )
} }
var processedText = lyrics.precomposedStringWithCanonicalMapping let protection = protectPreservedSpellings(
in: lyrics,
preservedSpellings: settings.preservedSpellings
)
var processedText = protection.text.precomposedStringWithCanonicalMapping
var notes = processingNotes(for: processedText, settings: settings) var notes = processingNotes(for: processedText, settings: settings)
if settings.tanweenPolicy == .removeWhenUnwanted { if settings.tanweenPolicy == .removeWhenUnwanted {
@@ -27,6 +31,15 @@ public struct ArabicPronunciationProcessor: Sendable {
notes.append(.tanweenAdditionNeedsReview) notes.append(.tanweenAdditionNeedsReview)
} }
processedText = restorePreservedSpellings(
in: processedText,
replacements: protection.replacements
)
if protection.matchCount > 0 {
notes.append(.preservedSpellingsProtected(count: protection.matchCount))
}
return ArabicPronunciationProcessingResult( return ArabicPronunciationProcessingResult(
text: processedText, text: processedText,
notes: notes notes: notes
@@ -53,6 +66,81 @@ public struct ArabicPronunciationProcessor: Sendable {
} }
} }
private func protectPreservedSpellings(
in text: String,
preservedSpellings: [String]
) -> PreservedSpellingProtection {
let spellings = Set(
preservedSpellings.map {
$0.trimmingCharacters(in: .whitespacesAndNewlines)
}
)
.filter { !$0.isEmpty }
.sorted {
if $0.count == $1.count {
return $0 < $1
}
return $0.count > $1.count
}
var protectedText = text
var replacements: [PreservedSpellingReplacement] = []
var matchCount = 0
for (index, spelling) in spellings.enumerated() {
let occurrenceCount = protectedText.components(separatedBy: spelling).count - 1
guard occurrenceCount > 0 else {
continue
}
let placeholder = uniquePlaceholder(
for: index,
in: protectedText
)
protectedText = protectedText.replacingOccurrences(
of: spelling,
with: placeholder
)
replacements.append(
PreservedSpellingReplacement(
placeholder: placeholder,
spelling: spelling
)
)
matchCount += occurrenceCount
}
return PreservedSpellingProtection(
text: protectedText,
replacements: replacements,
matchCount: matchCount
)
}
private func uniquePlaceholder(for index: Int, in text: String) -> String {
var collisionIndex = 0
var placeholder = "[[music-assistant-preserved-\(index)]]"
while text.contains(placeholder) {
collisionIndex += 1
placeholder = "[[music-assistant-preserved-\(index)-\(collisionIndex)]]"
}
return placeholder
}
private func restorePreservedSpellings(
in text: String,
replacements: [PreservedSpellingReplacement]
) -> String {
replacements.reduce(text) { result, replacement in
result.replacingOccurrences(
of: replacement.placeholder,
with: replacement.spelling
)
}
}
private func removeTanween(from text: String) -> String { private func removeTanween(from text: String) -> String {
String(text.unicodeScalars.filter { scalar in String(text.unicodeScalars.filter { scalar in
!isTanween(scalar) !isTanween(scalar)
@@ -85,6 +173,17 @@ public struct ArabicPronunciationProcessor: Sendable {
} }
} }
private struct PreservedSpellingProtection {
let text: String
let replacements: [PreservedSpellingReplacement]
let matchCount: Int
}
private struct PreservedSpellingReplacement {
let placeholder: String
let spelling: String
}
public struct ArabicPronunciationProcessingResult: Equatable, Sendable { public struct ArabicPronunciationProcessingResult: Equatable, Sendable {
public var text: String public var text: String
public var notes: [ArabicPronunciationProcessingNote] public var notes: [ArabicPronunciationProcessingNote]
@@ -101,4 +200,5 @@ public enum ArabicPronunciationProcessingNote: Equatable, Sendable {
case pronunciationTargetedDiacriticsNeedReview case pronunciationTargetedDiacriticsNeedReview
case fullTashkeelNeedsReview case fullTashkeelNeedsReview
case tanweenAdditionNeedsReview case tanweenAdditionNeedsReview
case preservedSpellingsProtected(count: Int)
} }
@@ -64,4 +64,43 @@ final class ArabicPronunciationProcessorTests: XCTestCase {
[.fullTashkeelNeedsReview, .tanweenAdditionNeedsReview] [.fullTashkeelNeedsReview, .tanweenAdditionNeedsReview]
) )
} }
func testPreservedSpellingsKeepTheirTanweenWhileOtherTextIsProcessed() {
let protectedSpelling = "\u{0647}\u{064F}\u{062F}\u{064B}\u{0649}"
let otherText = "\u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{064B}\u{0627}"
let lyrics = "\(protectedSpelling) \(otherText) \(protectedSpelling)"
let result = processor.process(
lyrics: lyrics,
settings: ArabicPronunciationSettings(
isEnabled: true,
diacritizationPolicy: .pronunciationTargeted,
tanweenPolicy: .removeWhenUnwanted,
preservedSpellings: [protectedSpelling]
)
)
XCTAssertEqual(
result.text,
"\(protectedSpelling) \u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{0627} \(protectedSpelling)"
)
XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 2)])
}
func testPreservedSpellingsKeepTheirOriginalDiacriticOrder() {
let preservedSpelling = "\u{0645}\u{0651}\u{064E}"
let otherText = "\u{0628}\u{0651}\u{064E}"
let result = processor.process(
lyrics: "\(preservedSpelling) \(otherText)",
settings: ArabicPronunciationSettings(
isEnabled: true,
diacritizationPolicy: .pronunciationTargeted,
preservedSpellings: [preservedSpelling]
)
)
XCTAssertEqual(result.text, "\(preservedSpelling) \u{0628}\u{064E}\u{0651}")
XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 1)])
}
} }
+3 -1
View File
@@ -66,7 +66,9 @@ tanween policy without choosing a diacritization policy for the user.
It must not invent missing vowel marks or tanween when no reliable It must not invent missing vowel marks or tanween when no reliable
linguistic source is available; instead, it returns a review note so a linguistic source is available; instead, it returns a review note so a
later user-review flow can present the unresolved text. This processor later user-review flow can present the unresolved text. This processor
does not use external services or APIs. does not use external services or APIs. Before processing, it protects
the user's exact preserved spellings and dialect phrases so normalization
or tanween removal cannot alter them.
## Security ## Security
+1 -1
View File
@@ -69,7 +69,7 @@ requirement is missing and blocks implementation, record it in
- [x] Add Arabic-specific settings UI. - [x] Add Arabic-specific settings UI.
- [x] Support diacritics/harakat/tanween processing. - [x] Support diacritics/harakat/tanween processing.
- [ ] Preserve intentional spelling/dialect choices where possible. - [x] Preserve intentional spelling/dialect choices where possible.
- [ ] Allow user to compare/edit processed Arabic before Suno handoff. - [ ] Allow user to compare/edit processed Arabic before Suno handoff.
- [ ] Add Arabic test fixtures covering multiple dialects. - [ ] Add Arabic test fixtures covering multiple dialects.