205 lines
6.3 KiB
Swift
205 lines
6.3 KiB
Swift
import Foundation
|
|
|
|
public struct ArabicPronunciationProcessor: Sendable {
|
|
public init() {}
|
|
|
|
public func process(
|
|
lyrics: String,
|
|
settings: ArabicPronunciationSettings
|
|
) -> ArabicPronunciationProcessingResult {
|
|
guard settings.isEnabled else {
|
|
return ArabicPronunciationProcessingResult(
|
|
text: lyrics,
|
|
notes: [.processingDisabled]
|
|
)
|
|
}
|
|
|
|
let protection = protectPreservedSpellings(
|
|
in: lyrics,
|
|
preservedSpellings: settings.preservedSpellings
|
|
)
|
|
var processedText = protection.text.precomposedStringWithCanonicalMapping
|
|
var notes = processingNotes(for: processedText, settings: settings)
|
|
|
|
if settings.tanweenPolicy == .removeWhenUnwanted {
|
|
processedText = removeTanween(from: processedText)
|
|
}
|
|
|
|
if settings.tanweenPolicy == .addWhenPronunciationRequires,
|
|
containsArabicLetter(in: processedText),
|
|
!containsTanween(in: processedText) {
|
|
notes.append(.tanweenAdditionNeedsReview)
|
|
}
|
|
|
|
processedText = restorePreservedSpellings(
|
|
in: processedText,
|
|
replacements: protection.replacements
|
|
)
|
|
|
|
if protection.matchCount > 0 {
|
|
notes.append(.preservedSpellingsProtected(count: protection.matchCount))
|
|
}
|
|
|
|
return ArabicPronunciationProcessingResult(
|
|
text: processedText,
|
|
notes: notes
|
|
)
|
|
}
|
|
|
|
private func processingNotes(
|
|
for text: String,
|
|
settings: ArabicPronunciationSettings
|
|
) -> [ArabicPronunciationProcessingNote] {
|
|
guard containsArabicLetter(in: text) else {
|
|
return []
|
|
}
|
|
|
|
switch settings.diacritizationPolicy {
|
|
case .unspecified:
|
|
return [.diacritizationPolicyUnspecified]
|
|
case .pronunciationTargeted where !containsDiacritics(in: text):
|
|
return [.pronunciationTargetedDiacriticsNeedReview]
|
|
case .fullTashkeel where !containsDiacritics(in: text):
|
|
return [.fullTashkeelNeedsReview]
|
|
default:
|
|
return []
|
|
}
|
|
}
|
|
|
|
private func protectPreservedSpellings(
|
|
in text: String,
|
|
preservedSpellings: [String]
|
|
) -> PreservedSpellingProtection {
|
|
let spellings = Set(
|
|
preservedSpellings.map {
|
|
$0.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
}
|
|
)
|
|
.filter { !$0.isEmpty }
|
|
.sorted {
|
|
if $0.count == $1.count {
|
|
return $0 < $1
|
|
}
|
|
return $0.count > $1.count
|
|
}
|
|
|
|
var protectedText = text
|
|
var replacements: [PreservedSpellingReplacement] = []
|
|
var matchCount = 0
|
|
|
|
for (index, spelling) in spellings.enumerated() {
|
|
let occurrenceCount = protectedText.components(separatedBy: spelling).count - 1
|
|
guard occurrenceCount > 0 else {
|
|
continue
|
|
}
|
|
|
|
let placeholder = uniquePlaceholder(
|
|
for: index,
|
|
in: protectedText
|
|
)
|
|
protectedText = protectedText.replacingOccurrences(
|
|
of: spelling,
|
|
with: placeholder
|
|
)
|
|
replacements.append(
|
|
PreservedSpellingReplacement(
|
|
placeholder: placeholder,
|
|
spelling: spelling
|
|
)
|
|
)
|
|
matchCount += occurrenceCount
|
|
}
|
|
|
|
return PreservedSpellingProtection(
|
|
text: protectedText,
|
|
replacements: replacements,
|
|
matchCount: matchCount
|
|
)
|
|
}
|
|
|
|
private func uniquePlaceholder(for index: Int, in text: String) -> String {
|
|
var collisionIndex = 0
|
|
var placeholder = "[[music-assistant-preserved-\(index)]]"
|
|
|
|
while text.contains(placeholder) {
|
|
collisionIndex += 1
|
|
placeholder = "[[music-assistant-preserved-\(index)-\(collisionIndex)]]"
|
|
}
|
|
|
|
return placeholder
|
|
}
|
|
|
|
private func restorePreservedSpellings(
|
|
in text: String,
|
|
replacements: [PreservedSpellingReplacement]
|
|
) -> String {
|
|
replacements.reduce(text) { result, replacement in
|
|
result.replacingOccurrences(
|
|
of: replacement.placeholder,
|
|
with: replacement.spelling
|
|
)
|
|
}
|
|
}
|
|
|
|
private func removeTanween(from text: String) -> String {
|
|
String(text.unicodeScalars.filter { scalar in
|
|
!isTanween(scalar)
|
|
})
|
|
}
|
|
|
|
private func containsArabicLetter(in text: String) -> Bool {
|
|
text.unicodeScalars.contains { scalar in
|
|
let isArabicBlock = (0x0600...0x06FF).contains(scalar.value)
|
|
|| (0x0750...0x077F).contains(scalar.value)
|
|
|| (0x08A0...0x08FF).contains(scalar.value)
|
|
return isArabicBlock && CharacterSet.letters.contains(scalar)
|
|
}
|
|
}
|
|
|
|
private func containsDiacritics(in text: String) -> Bool {
|
|
text.unicodeScalars.contains(where: isArabicDiacritic)
|
|
}
|
|
|
|
private func containsTanween(in text: String) -> Bool {
|
|
text.unicodeScalars.contains(where: isTanween)
|
|
}
|
|
|
|
private func isArabicDiacritic(_ scalar: Unicode.Scalar) -> Bool {
|
|
(0x064B...0x065F).contains(scalar.value) || scalar.value == 0x0670
|
|
}
|
|
|
|
private func isTanween(_ scalar: Unicode.Scalar) -> Bool {
|
|
(0x064B...0x064D).contains(scalar.value)
|
|
}
|
|
}
|
|
|
|
private struct PreservedSpellingProtection {
|
|
let text: String
|
|
let replacements: [PreservedSpellingReplacement]
|
|
let matchCount: Int
|
|
}
|
|
|
|
private struct PreservedSpellingReplacement {
|
|
let placeholder: String
|
|
let spelling: String
|
|
}
|
|
|
|
public struct ArabicPronunciationProcessingResult: Equatable, Sendable {
|
|
public var text: String
|
|
public var notes: [ArabicPronunciationProcessingNote]
|
|
|
|
public init(text: String, notes: [ArabicPronunciationProcessingNote] = []) {
|
|
self.text = text
|
|
self.notes = notes
|
|
}
|
|
}
|
|
|
|
public enum ArabicPronunciationProcessingNote: Equatable, Sendable {
|
|
case processingDisabled
|
|
case diacritizationPolicyUnspecified
|
|
case pronunciationTargetedDiacriticsNeedReview
|
|
case fullTashkeelNeedsReview
|
|
case tanweenAdditionNeedsReview
|
|
case preservedSpellingsProtected(count: Int)
|
|
}
|