From 3865f77dc0c16cf1f1a3385bc801edb214a30831 Mon Sep 17 00:00:00 2001 From: diyaa Date: Sun, 13 Sep 2026 22:33:57 +0200 Subject: [PATCH] Preserve Arabic spelling choices --- .../ArabicPronunciationProcessor.swift | 102 +++++++++++++++++- .../ArabicPronunciationProcessorTests.swift | 39 +++++++ docs/ARCHITECTURE.md | 4 +- docs/TASKS.md | 2 +- 4 files changed, 144 insertions(+), 3 deletions(-) diff --git a/Sources/MusicAssistantCore/Services/ArabicPronunciation/ArabicPronunciationProcessor.swift b/Sources/MusicAssistantCore/Services/ArabicPronunciation/ArabicPronunciationProcessor.swift index 984a4f8..887107b 100644 --- a/Sources/MusicAssistantCore/Services/ArabicPronunciation/ArabicPronunciationProcessor.swift +++ b/Sources/MusicAssistantCore/Services/ArabicPronunciation/ArabicPronunciationProcessor.swift @@ -14,7 +14,11 @@ public struct ArabicPronunciationProcessor: Sendable { ) } - var processedText = lyrics.precomposedStringWithCanonicalMapping + let protection = protectPreservedSpellings( + in: lyrics, + preservedSpellings: settings.preservedSpellings + ) + var processedText = protection.text.precomposedStringWithCanonicalMapping var notes = processingNotes(for: processedText, settings: settings) if settings.tanweenPolicy == .removeWhenUnwanted { @@ -27,6 +31,15 @@ public struct ArabicPronunciationProcessor: Sendable { notes.append(.tanweenAdditionNeedsReview) } + processedText = restorePreservedSpellings( + in: processedText, + replacements: protection.replacements + ) + + if protection.matchCount > 0 { + notes.append(.preservedSpellingsProtected(count: protection.matchCount)) + } + return ArabicPronunciationProcessingResult( text: processedText, notes: notes @@ -53,6 +66,81 @@ public struct ArabicPronunciationProcessor: Sendable { } } + private func protectPreservedSpellings( + in text: String, + preservedSpellings: [String] + ) -> PreservedSpellingProtection { + let spellings = Set( + preservedSpellings.map { + $0.trimmingCharacters(in: .whitespacesAndNewlines) + } + ) + .filter { !$0.isEmpty } + .sorted { + if $0.count == $1.count { + return $0 < $1 + } + return $0.count > $1.count + } + + var protectedText = text + var replacements: [PreservedSpellingReplacement] = [] + var matchCount = 0 + + for (index, spelling) in spellings.enumerated() { + let occurrenceCount = protectedText.components(separatedBy: spelling).count - 1 + guard occurrenceCount > 0 else { + continue + } + + let placeholder = uniquePlaceholder( + for: index, + in: protectedText + ) + protectedText = protectedText.replacingOccurrences( + of: spelling, + with: placeholder + ) + replacements.append( + PreservedSpellingReplacement( + placeholder: placeholder, + spelling: spelling + ) + ) + matchCount += occurrenceCount + } + + return PreservedSpellingProtection( + text: protectedText, + replacements: replacements, + matchCount: matchCount + ) + } + + private func uniquePlaceholder(for index: Int, in text: String) -> String { + var collisionIndex = 0 + var placeholder = "[[music-assistant-preserved-\(index)]]" + + while text.contains(placeholder) { + collisionIndex += 1 + placeholder = "[[music-assistant-preserved-\(index)-\(collisionIndex)]]" + } + + return placeholder + } + + private func restorePreservedSpellings( + in text: String, + replacements: [PreservedSpellingReplacement] + ) -> String { + replacements.reduce(text) { result, replacement in + result.replacingOccurrences( + of: replacement.placeholder, + with: replacement.spelling + ) + } + } + private func removeTanween(from text: String) -> String { String(text.unicodeScalars.filter { scalar in !isTanween(scalar) @@ -85,6 +173,17 @@ public struct ArabicPronunciationProcessor: Sendable { } } +private struct PreservedSpellingProtection { + let text: String + let replacements: [PreservedSpellingReplacement] + let matchCount: Int +} + +private struct PreservedSpellingReplacement { + let placeholder: String + let spelling: String +} + public struct ArabicPronunciationProcessingResult: Equatable, Sendable { public var text: String public var notes: [ArabicPronunciationProcessingNote] @@ -101,4 +200,5 @@ public enum ArabicPronunciationProcessingNote: Equatable, Sendable { case pronunciationTargetedDiacriticsNeedReview case fullTashkeelNeedsReview case tanweenAdditionNeedsReview + case preservedSpellingsProtected(count: Int) } diff --git a/Tests/MusicAssistantCoreTests/ArabicPronunciationProcessorTests.swift b/Tests/MusicAssistantCoreTests/ArabicPronunciationProcessorTests.swift index 813db78..258ba6b 100644 --- a/Tests/MusicAssistantCoreTests/ArabicPronunciationProcessorTests.swift +++ b/Tests/MusicAssistantCoreTests/ArabicPronunciationProcessorTests.swift @@ -64,4 +64,43 @@ final class ArabicPronunciationProcessorTests: XCTestCase { [.fullTashkeelNeedsReview, .tanweenAdditionNeedsReview] ) } + + func testPreservedSpellingsKeepTheirTanweenWhileOtherTextIsProcessed() { + let protectedSpelling = "\u{0647}\u{064F}\u{062F}\u{064B}\u{0649}" + let otherText = "\u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{064B}\u{0627}" + let lyrics = "\(protectedSpelling) \(otherText) \(protectedSpelling)" + + let result = processor.process( + lyrics: lyrics, + settings: ArabicPronunciationSettings( + isEnabled: true, + diacritizationPolicy: .pronunciationTargeted, + tanweenPolicy: .removeWhenUnwanted, + preservedSpellings: [protectedSpelling] + ) + ) + + XCTAssertEqual( + result.text, + "\(protectedSpelling) \u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{0627} \(protectedSpelling)" + ) + XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 2)]) + } + + func testPreservedSpellingsKeepTheirOriginalDiacriticOrder() { + let preservedSpelling = "\u{0645}\u{0651}\u{064E}" + let otherText = "\u{0628}\u{0651}\u{064E}" + + let result = processor.process( + lyrics: "\(preservedSpelling) \(otherText)", + settings: ArabicPronunciationSettings( + isEnabled: true, + diacritizationPolicy: .pronunciationTargeted, + preservedSpellings: [preservedSpelling] + ) + ) + + XCTAssertEqual(result.text, "\(preservedSpelling) \u{0628}\u{064E}\u{0651}") + XCTAssertEqual(result.notes, [.preservedSpellingsProtected(count: 1)]) + } } diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index e1293de..c620bc1 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -66,7 +66,9 @@ tanween policy without choosing a diacritization policy for the user. It must not invent missing vowel marks or tanween when no reliable linguistic source is available; instead, it returns a review note so a later user-review flow can present the unresolved text. This processor -does not use external services or APIs. +does not use external services or APIs. Before processing, it protects +the user's exact preserved spellings and dialect phrases so normalization +or tanween removal cannot alter them. ## Security diff --git a/docs/TASKS.md b/docs/TASKS.md index 6eda652..ce4f234 100644 --- a/docs/TASKS.md +++ b/docs/TASKS.md @@ -69,7 +69,7 @@ requirement is missing and blocks implementation, record it in - [x] Add Arabic-specific settings UI. - [x] Support diacritics/harakat/tanween processing. -- [ ] Preserve intentional spelling/dialect choices where possible. +- [x] Preserve intentional spelling/dialect choices where possible. - [ ] Allow user to compare/edit processed Arabic before Suno handoff. - [ ] Add Arabic test fixtures covering multiple dialects.