Preserve Arabic spelling choices
This commit is contained in:
+101
-1
@@ -14,7 +14,11 @@ public struct ArabicPronunciationProcessor: Sendable {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
var processedText = lyrics.precomposedStringWithCanonicalMapping
|
let protection = protectPreservedSpellings(
|
||||||
|
in: lyrics,
|
||||||
|
preservedSpellings: settings.preservedSpellings
|
||||||
|
)
|
||||||
|
var processedText = protection.text.precomposedStringWithCanonicalMapping
|
||||||
var notes = processingNotes(for: processedText, settings: settings)
|
var notes = processingNotes(for: processedText, settings: settings)
|
||||||
|
|
||||||
if settings.tanweenPolicy == .removeWhenUnwanted {
|
if settings.tanweenPolicy == .removeWhenUnwanted {
|
||||||
@@ -27,6 +31,15 @@ public struct ArabicPronunciationProcessor: Sendable {
|
|||||||
notes.append(.tanweenAdditionNeedsReview)
|
notes.append(.tanweenAdditionNeedsReview)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
processedText = restorePreservedSpellings(
|
||||||
|
in: processedText,
|
||||||
|
replacements: protection.replacements
|
||||||
|
)
|
||||||
|
|
||||||
|
if protection.matchCount > 0 {
|
||||||
|
notes.append(.preservedSpellingsProtected(count: protection.matchCount))
|
||||||
|
}
|
||||||
|
|
||||||
return ArabicPronunciationProcessingResult(
|
return ArabicPronunciationProcessingResult(
|
||||||
text: processedText,
|
text: processedText,
|
||||||
notes: notes
|
notes: notes
|
||||||
@@ -53,6 +66,81 @@ public struct ArabicPronunciationProcessor: Sendable {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private func protectPreservedSpellings(
|
||||||
|
in text: String,
|
||||||
|
preservedSpellings: [String]
|
||||||
|
) -> PreservedSpellingProtection {
|
||||||
|
let spellings = Set(
|
||||||
|
preservedSpellings.map {
|
||||||
|
$0.trimmingCharacters(in: .whitespacesAndNewlines)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
.filter { !$0.isEmpty }
|
||||||
|
.sorted {
|
||||||
|
if $0.count == $1.count {
|
||||||
|
return $0 < $1
|
||||||
|
}
|
||||||
|
return $0.count > $1.count
|
||||||
|
}
|
||||||
|
|
||||||
|
var protectedText = text
|
||||||
|
var replacements: [PreservedSpellingReplacement] = []
|
||||||
|
var matchCount = 0
|
||||||
|
|
||||||
|
for (index, spelling) in spellings.enumerated() {
|
||||||
|
let occurrenceCount = protectedText.components(separatedBy: spelling).count - 1
|
||||||
|
guard occurrenceCount > 0 else {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
let placeholder = uniquePlaceholder(
|
||||||
|
for: index,
|
||||||
|
in: protectedText
|
||||||
|
)
|
||||||
|
protectedText = protectedText.replacingOccurrences(
|
||||||
|
of: spelling,
|
||||||
|
with: placeholder
|
||||||
|
)
|
||||||
|
replacements.append(
|
||||||
|
PreservedSpellingReplacement(
|
||||||
|
placeholder: placeholder,
|
||||||
|
spelling: spelling
|
||||||
|
)
|
||||||
|
)
|
||||||
|
matchCount += occurrenceCount
|
||||||
|
}
|
||||||
|
|
||||||
|
return PreservedSpellingProtection(
|
||||||
|
text: protectedText,
|
||||||
|
replacements: replacements,
|
||||||
|
matchCount: matchCount
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
private func uniquePlaceholder(for index: Int, in text: String) -> String {
|
||||||
|
var collisionIndex = 0
|
||||||
|
var placeholder = "[[music-assistant-preserved-\(index)]]"
|
||||||
|
|
||||||
|
while text.contains(placeholder) {
|
||||||
|
collisionIndex += 1
|
||||||
|
placeholder = "[[music-assistant-preserved-\(index)-\(collisionIndex)]]"
|
||||||
|
}
|
||||||
|
|
||||||
|
return placeholder
|
||||||
|
}
|
||||||
|
|
||||||
|
private func restorePreservedSpellings(
|
||||||
|
in text: String,
|
||||||
|
replacements: [PreservedSpellingReplacement]
|
||||||
|
) -> String {
|
||||||
|
replacements.reduce(text) { result, replacement in
|
||||||
|
result.replacingOccurrences(
|
||||||
|
of: replacement.placeholder,
|
||||||
|
with: replacement.spelling
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
private func removeTanween(from text: String) -> String {
|
private func removeTanween(from text: String) -> String {
|
||||||
String(text.unicodeScalars.filter { scalar in
|
String(text.unicodeScalars.filter { scalar in
|
||||||
!isTanween(scalar)
|
!isTanween(scalar)
|
||||||
@@ -85,6 +173,17 @@ public struct ArabicPronunciationProcessor: Sendable {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
private struct PreservedSpellingProtection {
|
||||||
|
let text: String
|
||||||
|
let replacements: [PreservedSpellingReplacement]
|
||||||
|
let matchCount: Int
|
||||||
|
}
|
||||||
|
|
||||||
|
private struct PreservedSpellingReplacement {
|
||||||
|
let placeholder: String
|
||||||
|
let spelling: String
|
||||||
|
}
|
||||||
|
|
||||||
public struct ArabicPronunciationProcessingResult: Equatable, Sendable {
|
public struct ArabicPronunciationProcessingResult: Equatable, Sendable {
|
||||||
public var text: String
|
public var text: String
|
||||||
public var notes: [ArabicPronunciationProcessingNote]
|
public var notes: [ArabicPronunciationProcessingNote]
|
||||||
@@ -101,4 +200,5 @@ public enum ArabicPronunciationProcessingNote: Equatable, Sendable {
|
|||||||
case pronunciationTargetedDiacriticsNeedReview
|
case pronunciationTargetedDiacriticsNeedReview
|
||||||
case fullTashkeelNeedsReview
|
case fullTashkeelNeedsReview
|
||||||
case tanweenAdditionNeedsReview
|
case tanweenAdditionNeedsReview
|
||||||
|
case preservedSpellingsProtected(count: Int)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -64,4 +64,43 @@ final class ArabicPronunciationProcessorTests: XCTestCase {
|
|||||||
[.fullTashkeelNeedsReview, .tanweenAdditionNeedsReview]
|
[.fullTashkeelNeedsReview, .tanweenAdditionNeedsReview]
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func testPreservedSpellingsKeepTheirTanweenWhileOtherTextIsProcessed() {
|
||||||
|
let protectedSpelling = "\u{0647}\u{064F}\u{062F}\u{064B}\u{0649}"
|
||||||
|
let otherText = "\u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{064B}\u{0627}"
|
||||||
|
let lyrics = "\(protectedSpelling) \(otherText) \(protectedSpelling)"
|
||||||
|
|
||||||
|
let result = processor.process(
|
||||||
|
lyrics: lyrics,
|
||||||
|
settings: ArabicPronunciationSettings(
|
||||||
|
isEnabled: true,
|
||||||
|
diacritizationPolicy: .pronunciationTargeted,
|
||||||
|
tanweenPolicy: .removeWhenUnwanted,
|
||||||
|
preservedSpellings: [protectedSpelling]
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
XCTAssertEqual(
|
||||||
|
result.text,
|
||||||
|
"\(protectedSpelling) \u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{0627} \(protectedSpelling)"
|
||||||
|
)
|
||||||
|
XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 2)])
|
||||||
|
}
|
||||||
|
|
||||||
|
func testPreservedSpellingsKeepTheirOriginalDiacriticOrder() {
|
||||||
|
let preservedSpelling = "\u{0645}\u{0651}\u{064E}"
|
||||||
|
let otherText = "\u{0628}\u{0651}\u{064E}"
|
||||||
|
|
||||||
|
let result = processor.process(
|
||||||
|
lyrics: "\(preservedSpelling) \(otherText)",
|
||||||
|
settings: ArabicPronunciationSettings(
|
||||||
|
isEnabled: true,
|
||||||
|
diacritizationPolicy: .pronunciationTargeted,
|
||||||
|
preservedSpellings: [preservedSpelling]
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
XCTAssertEqual(result.text, "\(preservedSpelling) \u{0628}\u{064E}\u{0651}")
|
||||||
|
XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 1)])
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -66,7 +66,9 @@ tanween policy without choosing a diacritization policy for the user.
|
|||||||
It must not invent missing vowel marks or tanween when no reliable
|
It must not invent missing vowel marks or tanween when no reliable
|
||||||
linguistic source is available; instead, it returns a review note so a
|
linguistic source is available; instead, it returns a review note so a
|
||||||
later user-review flow can present the unresolved text. This processor
|
later user-review flow can present the unresolved text. This processor
|
||||||
does not use external services or APIs.
|
does not use external services or APIs. Before processing, it protects
|
||||||
|
the user's exact preserved spellings and dialect phrases so normalization
|
||||||
|
or tanween removal cannot alter them.
|
||||||
|
|
||||||
## Security
|
## Security
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -69,7 +69,7 @@ requirement is missing and blocks implementation, record it in
|
|||||||
|
|
||||||
- [x] Add Arabic-specific settings UI.
|
- [x] Add Arabic-specific settings UI.
|
||||||
- [x] Support diacritics/harakat/tanween processing.
|
- [x] Support diacritics/harakat/tanween processing.
|
||||||
- [ ] Preserve intentional spelling/dialect choices where possible.
|
- [x] Preserve intentional spelling/dialect choices where possible.
|
||||||
- [ ] Allow user to compare/edit processed Arabic before Suno handoff.
|
- [ ] Allow user to compare/edit processed Arabic before Suno handoff.
|
||||||
- [ ] Add Arabic test fixtures covering multiple dialects.
|
- [ ] Add Arabic test fixtures covering multiple dialects.
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user