You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
276 lines
13 KiB
276 lines
13 KiB
import Foundation
|
|
|
|
public final class TextProcessor {
|
|
public static let shared = TextProcessor()
|
|
|
|
private init() {}
|
|
|
|
public func process(text: String) -> String {
|
|
var result = text.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
if result.isEmpty { return "" }
|
|
|
|
// 0. Check for Hermes / Bot Messenger Commands (/steer, /queue, /cancel)
|
|
if let botCmd = BotCommandProcessor.parse(text: result) {
|
|
let processedBody: String
|
|
if !botCmd.body.isEmpty {
|
|
var bodyText = botCmd.body
|
|
if SonioxSettings.shared.voicePunctuation {
|
|
bodyText = applyVoicePunctuation(bodyText)
|
|
}
|
|
if SonioxSettings.shared.domainVocabulary {
|
|
bodyText = applyDomainVocabulary(bodyText)
|
|
}
|
|
bodyText = applyNumberFormatting(bodyText, format: SonioxSettings.shared.numberFormat)
|
|
if SonioxSettings.shared.normalizeText {
|
|
bodyText = normalizeText(bodyText)
|
|
}
|
|
processedBody = bodyText.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
} else {
|
|
processedBody = ""
|
|
}
|
|
|
|
if processedBody.isEmpty {
|
|
return botCmd.command
|
|
} else {
|
|
return "\(botCmd.command)\n\(processedBody)"
|
|
}
|
|
}
|
|
|
|
// 1. Voice Punctuation Commands
|
|
if SonioxSettings.shared.voicePunctuation {
|
|
result = applyVoicePunctuation(result)
|
|
}
|
|
|
|
// 2. Domain Vocabulary Enhancement
|
|
if SonioxSettings.shared.domainVocabulary {
|
|
result = applyDomainVocabulary(result)
|
|
}
|
|
|
|
// 3. Number Formatting
|
|
result = applyNumberFormatting(result, format: SonioxSettings.shared.numberFormat)
|
|
|
|
// 4. Text Normalization
|
|
if SonioxSettings.shared.normalizeText {
|
|
result = normalizeText(result)
|
|
}
|
|
|
|
// 5. Wake Word & End Trigger sanitization
|
|
result = stripTrailingEndTrigger(result)
|
|
|
|
return result.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
}
|
|
|
|
private func applyVoicePunctuation(_ text: String) -> String {
|
|
var t = " " + text + " "
|
|
|
|
// Persian Voice Commands
|
|
let persianMappings: [(String, String)] = [
|
|
(" علامت سوال", "؟"),
|
|
(" علامت سؤال", "؟"),
|
|
(" علامت پرسش", "؟"),
|
|
(" علامت تعجب", "!"),
|
|
(" نقطه ویرگول", "؛"),
|
|
(" خط بعد", "\n"),
|
|
(" سطر بعد", "\n"),
|
|
(" سر خط", "\n"),
|
|
(" اینتر", "\n"),
|
|
(" برو خط بعد", "\n"),
|
|
(" نقطه", "."),
|
|
(" ویرگول", "،"),
|
|
(" کاما", "،"),
|
|
(" دو نقطه", ":"),
|
|
(" سه نقطه", "..."),
|
|
(" گیومه باز", " «"),
|
|
(" گیومه بسته", "» "),
|
|
(" پرانتز باز", " ("),
|
|
(" پرانتز بسته", ") ")
|
|
]
|
|
|
|
for (cmd, rep) in persianMappings {
|
|
t = t.replacingOccurrences(of: cmd, with: rep, options: .caseInsensitive)
|
|
}
|
|
|
|
// English Voice Commands
|
|
let englishMappings: [(String, String)] = [
|
|
(" question mark", "?"),
|
|
(" exclamation mark", "!"),
|
|
(" exclamation point", "!"),
|
|
(" semi colon", ";"),
|
|
(" semicolon", ";"),
|
|
(" new line", "\n"),
|
|
(" enter key", "\n"),
|
|
(" full stop", "."),
|
|
(" period", "."),
|
|
(" comma", ","),
|
|
(" colon", ":"),
|
|
(" ellipsis", "..."),
|
|
(" open quote", " \""),
|
|
(" close quote", "\" "),
|
|
(" open parenthesis", " ("),
|
|
(" close parenthesis", ") ")
|
|
]
|
|
|
|
for (cmd, rep) in englishMappings {
|
|
t = t.replacingOccurrences(of: cmd, with: rep, options: .caseInsensitive)
|
|
}
|
|
|
|
return t
|
|
}
|
|
|
|
private func applyDomainVocabulary(_ text: String) -> String {
|
|
var t = text
|
|
let terms: [(String, String)] = [
|
|
("نیوهورایزن", "NewHorizon"),
|
|
("نیو هورایزن", "NewHorizon"),
|
|
("نیوهوریزن", "NewHorizon"),
|
|
("زر محک", "ZarMahak"),
|
|
("زرمحک", "ZarMahak"),
|
|
("سانی اوکس", "Soniox"),
|
|
("سانیوکس", "Soniox"),
|
|
("سانی اکس", "Soniox"),
|
|
("تیم بای", "TeamBy"),
|
|
("تیمبی", "TeamBy"),
|
|
("دوودی", "Dovodi"),
|
|
("داودی", "Dovodi"),
|
|
("داکر", "Docker"),
|
|
("کوبرنتیز", "Kubernetes"),
|
|
("کوبرنتس", "Kubernetes"),
|
|
("ای اس ایکس آی", "ESXi"),
|
|
("ای اس ایکس ای", "ESXi"),
|
|
("فست ای پی آی", "FastAPI"),
|
|
("فست ای پی ای", "FastAPI"),
|
|
("پستگرس کیو ال", "PostgreSQL"),
|
|
("پستگرس", "PostgreSQL"),
|
|
("ردیس", "Redis"),
|
|
("فلاتر", "Flutter"),
|
|
("وایارگارد", "WireGuard"),
|
|
("وایرگارد", "WireGuard"),
|
|
("میکروتیک", "MikroTik"),
|
|
("هرمس", "Hermes"),
|
|
("نکست جی اس", "Next.js"),
|
|
("ری اکت", "React"),
|
|
("پایتون", "Python"),
|
|
("سویفت", "Swift")
|
|
]
|
|
|
|
for (target, canonical) in terms {
|
|
t = t.replacingOccurrences(of: target, with: canonical, options: .caseInsensitive)
|
|
}
|
|
return t
|
|
}
|
|
|
|
private func applyNumberFormatting(_ text: String, format: NumberFormatOption) -> String {
|
|
switch format {
|
|
case .raw:
|
|
return text
|
|
|
|
case .digits, .persianDigits:
|
|
// Convert spoken Persian numbers to digits
|
|
var t = " " + text + " "
|
|
|
|
// Compound tens and units
|
|
let compounds: [(String, String)] = [
|
|
(" بیست و نه ", " 29 "), (" بیست و هشت ", " 28 "), (" بیست و هفت ", " 27 "), (" بیست و شش ", " 26 "),
|
|
(" بیست و پنج ", " 25 "), (" بیست و چهار ", " 24 "), (" بیست و سه ", " 23 "), (" بیست و دو ", " 22 "), (" بیست و یک ", " 21 "),
|
|
(" سی و نه ", " 39 "), (" سی و هشت ", " 38 "), (" سی و هفت ", " 37 "), (" سی و شش ", " 36 "),
|
|
(" سی و پنج ", " 35 "), (" سی و چهار ", " 34 "), (" سی و سه ", " 33 "), (" سی و دو ", " 32 "), (" سی و یک ", " 31 "),
|
|
(" چهل و نه ", " 49 "), (" چهل و هشت ", " 48 "), (" چهل و هفت ", " 47 "), (" چهل و شش ", " 46 "),
|
|
(" چهل و پنج ", " 45 "), (" چهل و چهار ", " 44 "), (" چهل و سه ", " 43 "), (" چهل و دو ", " 42 "), (" چهل و یک ", " 41 "),
|
|
(" پنجاه و نه ", " 59 "), (" پنجاه و هشت ", " 58 "), (" پنجاه و هفت ", " 57 "), (" پنجاه و شش ", " 56 "),
|
|
(" پنجاه و پنج ", " 55 "), (" پنجاه و چهار ", " 54 "), (" پنجاه و سه ", " 53 "), (" پنجاه و دو ", " 52 "), (" پنجاه و یک ", " 51 "),
|
|
(" شصت و نه ", " 69 "), (" شصت و هشت ", " 68 "), (" شصت و هفت ", " 67 "), (" شصت و شش ", " 66 "),
|
|
(" شصت و پنج ", " 65 "), (" شصت و چهار ", " 64 "), (" شصت و سه ", " 63 "), (" شصت و دو ", " 62 "), (" شصت و یک ", " 61 "),
|
|
(" هفتاد و نه ", " 79 "), (" هفتاد و هشت ", " 78 "), (" هفتاد و هفت ", " 77 "), (" هفتاد و شش ", " 76 "),
|
|
(" هفتاد و پنج ", " 75 "), (" هفتاد و چهار ", " 74 "), (" هفتاد و سه ", " 73 "), (" هفتاد و دو ", " 72 "), (" هفتاد و یک ", " 71 "),
|
|
(" هشتاد و نه ", " 89 "), (" هشتاد و هشت ", " 88 "), (" هشتاد و هفت ", " 87 "), (" هشتاد و شش ", " 86 "),
|
|
(" هشتاد و پنج ", " 85 "), (" هشتاد و چهار ", " 84 "), (" هشتاد و سه ", " 83 "), (" هشتاد و دو ", " 82 "), (" هشتاد و یک ", " 81 "),
|
|
(" نود و نه ", " 99 "), (" نود و هشت ", " 98 "), (" نود و هفت ", " 97 "), (" نود و شش ", " 96 "),
|
|
(" نود و پنج ", " 95 "), (" نود و چهار ", " 94 "), (" نود و سه ", " 93 "), (" نود و دو ", " 92 "), (" نود و یک ", " 91 ")
|
|
]
|
|
|
|
for (word, num) in compounds {
|
|
t = t.replacingOccurrences(of: word, with: num)
|
|
}
|
|
|
|
// Standard single numbers & round numbers
|
|
let simples: [(String, String)] = [
|
|
(" ده ", " 10 "), (" یازده ", " 11 "), (" دوازده ", " 12 "), (" سیزده ", " 13 "), (" چهارده ", " 14 "),
|
|
(" پانزده ", " 15 "), (" شانزده ", " 16 "), (" هفده ", " 17 "), (" هجده ", " 18 "), (" نوزده ", " 19 "),
|
|
(" بیست ", " 20 "), (" سی ", " 30 "), (" چهل ", " 40 "), (" پنجاه ", " 50 "),
|
|
(" شصت ", " 60 "), (" هفتاد ", " 70 "), (" هشتاد ", " 80 "), (" نود ", " 90 "),
|
|
(" صد ", " 100 "), (" دویست ", " 200 "), (" سیصد ", " 300 "), (" چهارصد ", " 400 "),
|
|
(" پانصد ", " 500 "), (" ششصد ", " 600 "), (" هفتصد ", " 700 "), (" هشتصد ", " 800 "), (" نهصد ", " 900 "),
|
|
(" هزار ", " 1000 "), (" میلیون ", " 1000000 "),
|
|
(" صفر ", " 0 "), (" یک ", " 1 "), (" دو ", " 2 "), (" سه ", " 3 "), (" چهار ", " 4 "),
|
|
(" پنج ", " 5 "), (" شش ", " 6 "), (" هفت ", " 7 "), (" هشت ", " 8 "), (" نه ", " 9 ")
|
|
]
|
|
|
|
for (word, num) in simples {
|
|
t = t.replacingOccurrences(of: word, with: num)
|
|
}
|
|
|
|
if format == .persianDigits {
|
|
let pDigits = ["0":"۰", "1":"۱", "2":"۲", "3":"۳", "4":"۴", "5":"۵", "6":"۶", "7":"۷", "8":"۸", "9":"۹"]
|
|
for (en, fa) in pDigits {
|
|
t = t.replacingOccurrences(of: en, with: fa)
|
|
}
|
|
}
|
|
return t.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
|
|
case .words:
|
|
var t = text
|
|
let toWords = [
|
|
"0": "صفر", "1": "یک", "2": "دو", "3": "سه", "4": "چهار",
|
|
"5": "پنج", "6": "شش", "7": "هفت", "8": "هشت", "9": "نه",
|
|
"۰": "صفر", "۱": "یک", "۲": "دو", "۳": "سه", "۴": "چهار",
|
|
"۵": "پنج", "۶": "شش", "۷": "هفت", "۸": "هشت", "۹": "نه"
|
|
]
|
|
for (digit, word) in toWords {
|
|
t = t.replacingOccurrences(of: digit, with: word)
|
|
}
|
|
return t
|
|
}
|
|
}
|
|
|
|
private func normalizeText(_ text: String) -> String {
|
|
var t = text
|
|
// Clean double spaces
|
|
while t.contains(" ") {
|
|
t = t.replacingOccurrences(of: " ", with: " ")
|
|
}
|
|
// Remove space before punctuation
|
|
let punctuationList = [".", "،", "؛", "؟", "!", ":", ")", "»"]
|
|
for p in punctuationList {
|
|
t = t.replacingOccurrences(of: " " + p, with: p)
|
|
}
|
|
// Ensure space after punctuation (unless end of line)
|
|
for p in [".", "،", "؛", "؟", "!"] {
|
|
t = t.replacingOccurrences(of: p, with: p + " ")
|
|
}
|
|
// Fix spaces inside parentheses / quotes
|
|
t = t.replacingOccurrences(of: "( ", with: "(")
|
|
t = t.replacingOccurrences(of: "« ", with: "«")
|
|
|
|
while t.contains(" ") {
|
|
t = t.replacingOccurrences(of: " ", with: " ")
|
|
}
|
|
return t.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
}
|
|
|
|
|
|
public func stripTrailingEndTrigger(_ text: String) -> String {
|
|
var t = text.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
let pattern = #"[،,:\s]+(and|تمام|end)[.!?]?$"#
|
|
if let regex = try? NSRegularExpression(pattern: pattern, options: .caseInsensitive) {
|
|
let range = NSRange(location: 0, length: t.utf16.count)
|
|
t = regex.stringByReplacingMatches(in: t, options: [], range: range, withTemplate: "")
|
|
}
|
|
let singlePattern = #"^(and|تمام|end)[.!?]?$"#
|
|
if let regex = try? NSRegularExpression(pattern: singlePattern, options: .caseInsensitive) {
|
|
let range = NSRange(location: 0, length: t.utf16.count)
|
|
t = regex.stringByReplacingMatches(in: t, options: [], range: range, withTemplate: "")
|
|
}
|
|
return t.trimmingCharacters(in: .whitespacesAndNewlines)
|
|
}
|
|
|
|
}
|