Add Arabic pronunciation processing
This commit is contained in:
+104
@@ -0,0 +1,104 @@
|
||||
import Foundation
|
||||
|
||||
public struct ArabicPronunciationProcessor: Sendable {
|
||||
public init() {}
|
||||
|
||||
public func process(
|
||||
lyrics: String,
|
||||
settings: ArabicPronunciationSettings
|
||||
) -> ArabicPronunciationProcessingResult {
|
||||
guard settings.isEnabled else {
|
||||
return ArabicPronunciationProcessingResult(
|
||||
text: lyrics,
|
||||
notes: [.processingDisabled]
|
||||
)
|
||||
}
|
||||
|
||||
var processedText = lyrics.precomposedStringWithCanonicalMapping
|
||||
var notes = processingNotes(for: processedText, settings: settings)
|
||||
|
||||
if settings.tanweenPolicy == .removeWhenUnwanted {
|
||||
processedText = removeTanween(from: processedText)
|
||||
}
|
||||
|
||||
if settings.tanweenPolicy == .addWhenPronunciationRequires,
|
||||
containsArabicLetter(in: processedText),
|
||||
!containsTanween(in: processedText) {
|
||||
notes.append(.tanweenAdditionNeedsReview)
|
||||
}
|
||||
|
||||
return ArabicPronunciationProcessingResult(
|
||||
text: processedText,
|
||||
notes: notes
|
||||
)
|
||||
}
|
||||
|
||||
private func processingNotes(
|
||||
for text: String,
|
||||
settings: ArabicPronunciationSettings
|
||||
) -> [ArabicPronunciationProcessingNote] {
|
||||
guard containsArabicLetter(in: text) else {
|
||||
return []
|
||||
}
|
||||
|
||||
switch settings.diacritizationPolicy {
|
||||
case .unspecified:
|
||||
return [.diacritizationPolicyUnspecified]
|
||||
case .pronunciationTargeted where !containsDiacritics(in: text):
|
||||
return [.pronunciationTargetedDiacriticsNeedReview]
|
||||
case .fullTashkeel where !containsDiacritics(in: text):
|
||||
return [.fullTashkeelNeedsReview]
|
||||
default:
|
||||
return []
|
||||
}
|
||||
}
|
||||
|
||||
private func removeTanween(from text: String) -> String {
|
||||
String(text.unicodeScalars.filter { scalar in
|
||||
!isTanween(scalar)
|
||||
})
|
||||
}
|
||||
|
||||
private func containsArabicLetter(in text: String) -> Bool {
|
||||
text.unicodeScalars.contains { scalar in
|
||||
let isArabicBlock = (0x0600...0x06FF).contains(scalar.value)
|
||||
|| (0x0750...0x077F).contains(scalar.value)
|
||||
|| (0x08A0...0x08FF).contains(scalar.value)
|
||||
return isArabicBlock && CharacterSet.letters.contains(scalar)
|
||||
}
|
||||
}
|
||||
|
||||
private func containsDiacritics(in text: String) -> Bool {
|
||||
text.unicodeScalars.contains(where: isArabicDiacritic)
|
||||
}
|
||||
|
||||
private func containsTanween(in text: String) -> Bool {
|
||||
text.unicodeScalars.contains(where: isTanween)
|
||||
}
|
||||
|
||||
private func isArabicDiacritic(_ scalar: Unicode.Scalar) -> Bool {
|
||||
(0x064B...0x065F).contains(scalar.value) || scalar.value == 0x0670
|
||||
}
|
||||
|
||||
private func isTanween(_ scalar: Unicode.Scalar) -> Bool {
|
||||
(0x064B...0x064D).contains(scalar.value)
|
||||
}
|
||||
}
|
||||
|
||||
public struct ArabicPronunciationProcessingResult: Equatable, Sendable {
|
||||
public var text: String
|
||||
public var notes: [ArabicPronunciationProcessingNote]
|
||||
|
||||
public init(text: String, notes: [ArabicPronunciationProcessingNote] = []) {
|
||||
self.text = text
|
||||
self.notes = notes
|
||||
}
|
||||
}
|
||||
|
||||
public enum ArabicPronunciationProcessingNote: Equatable, Sendable {
|
||||
case processingDisabled
|
||||
case diacritizationPolicyUnspecified
|
||||
case pronunciationTargetedDiacriticsNeedReview
|
||||
case fullTashkeelNeedsReview
|
||||
case tanweenAdditionNeedsReview
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
import MusicAssistantCore
|
||||
import XCTest
|
||||
|
||||
final class ArabicPronunciationProcessorTests: XCTestCase {
|
||||
private let processor = ArabicPronunciationProcessor()
|
||||
|
||||
func testDisabledProcessingLeavesLyricsUntouched() {
|
||||
let lyrics = "\u{0645}\u{0651}\u{064E}\u{0631}\u{062D}\u{064E}\u{0628}\u{064B}\u{0627}"
|
||||
|
||||
let result = processor.process(
|
||||
lyrics: lyrics,
|
||||
settings: ArabicPronunciationSettings(isEnabled: false)
|
||||
)
|
||||
|
||||
XCTAssertEqual(result.text, lyrics)
|
||||
XCTAssertEqual(result.notes, [.processingDisabled])
|
||||
}
|
||||
|
||||
func testProcessingNormalizesSuppliedDiacritics() {
|
||||
let lyrics = "\u{0645}\u{0651}\u{064E}\u{0631}\u{062D}\u{064E}\u{0628}\u{064B}\u{0627}"
|
||||
|
||||
let result = processor.process(
|
||||
lyrics: lyrics,
|
||||
settings: ArabicPronunciationSettings(
|
||||
isEnabled: true,
|
||||
diacritizationPolicy: .pronunciationTargeted
|
||||
)
|
||||
)
|
||||
|
||||
XCTAssertEqual(
|
||||
result.text,
|
||||
"\u{0645}\u{064E}\u{0651}\u{0631}\u{062D}\u{064E}\u{0628}\u{064B}\u{0627}"
|
||||
)
|
||||
XCTAssertTrue(result.notes.isEmpty)
|
||||
}
|
||||
|
||||
func testRemovingTanweenKeepsOtherDiacritics() {
|
||||
let result = processor.process(
|
||||
lyrics: "\u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{064B}\u{0627}",
|
||||
settings: ArabicPronunciationSettings(
|
||||
isEnabled: true,
|
||||
tanweenPolicy: .removeWhenUnwanted
|
||||
)
|
||||
)
|
||||
|
||||
XCTAssertEqual(result.text, "\u{0634}\u{064F}\u{0643}\u{0652}\u{0631}\u{0627}")
|
||||
}
|
||||
|
||||
func testFullTashkeelDoesNotInventMissingMarks() {
|
||||
let lyrics = "\u{0645}\u{0631}\u{062D}\u{0628}\u{0627}"
|
||||
|
||||
let result = processor.process(
|
||||
lyrics: lyrics,
|
||||
settings: ArabicPronunciationSettings(
|
||||
isEnabled: true,
|
||||
diacritizationPolicy: .fullTashkeel,
|
||||
tanweenPolicy: .addWhenPronunciationRequires
|
||||
)
|
||||
)
|
||||
|
||||
XCTAssertEqual(result.text, lyrics)
|
||||
XCTAssertEqual(
|
||||
result.notes,
|
||||
[.fullTashkeelNeedsReview, .tanweenAdditionNeedsReview]
|
||||
)
|
||||
}
|
||||
}
|
||||
@@ -58,6 +58,16 @@ retry delay when available, and propagates cancellation without retrying.
|
||||
A deterministic layer converts the approved SongProject into the final
|
||||
Suno-facing lyrics/style content. User choices override AI suggestions.
|
||||
|
||||
## Arabic Pronunciation Processor
|
||||
|
||||
The Arabic Pronunciation Processor is a local Application Service. It
|
||||
normalizes user-supplied Arabic diacritics and applies the selected
|
||||
tanween policy without choosing a diacritization policy for the user.
|
||||
It must not invent missing vowel marks or tanween when no reliable
|
||||
linguistic source is available; instead, it returns a review note so a
|
||||
later user-review flow can present the unresolved text. This processor
|
||||
does not use external services or APIs.
|
||||
|
||||
## Security
|
||||
|
||||
- Never commit API keys to source control.
|
||||
|
||||
+1
-1
@@ -68,7 +68,7 @@ requirement is missing and blocks implementation, record it in
|
||||
## Phase 5 --- Arabic Lyrics Processing
|
||||
|
||||
- [x] Add Arabic-specific settings UI.
|
||||
- [ ] Support diacritics/harakat/tanween processing.
|
||||
- [x] Support diacritics/harakat/tanween processing.
|
||||
- [ ] Preserve intentional spelling/dialect choices where possible.
|
||||
- [ ] Allow user to compare/edit processed Arabic before Suno handoff.
|
||||
- [ ] Add Arabic test fixtures covering multiple dialects.
|
||||
|
||||
Reference in New Issue
Block a user