import scala.collection.mutable
object RemoveDuplicateWordsFreeText {
/*
splitWords
Splits free text into words using a Unicode-aware regex.
We use String.split instead of Regex.split because:
- It supports Unicode properties like \p{L}
- It avoids Scala's Regex.split deprecation warnings
- It is efficient and idiomatic
Regex:
[^\p{L}]+ → any sequence of NON-letter characters
*/
def splitWords(text: String): Seq[String] = {
val trimmed: String = text.trim
// Split on any sequence of non-letter characters
val parts: Array[String] = trimmed.split("[^\\p{L}]+")
parts.filter(_.nonEmpty)
}
/*
removeDuplicateWords
Removes duplicate words while preserving:
- original order
- original casing of first occurrence
- case-insensitive comparison
Uses mutable.HashSet for O(1) lookup.
*/
def removeDuplicateWords(text: String): String = {
val words: Seq[String] = splitWords(text)
val seen: mutable.HashSet[String] = mutable.HashSet.empty[String]
val unique: mutable.ListBuffer[String] = mutable.ListBuffer.empty[String]
for (word <- words) {
val key: String = word.toLowerCase // Unicode-aware lowercase
if (!seen.contains(key)) {
seen.add(key)
unique += word // preserve original casing
}
}
// Reassemble into a space-separated string
unique.mkString(" ")
}
def main(args: Array[String]): Unit = {
val input: String =
"Hello, hello! This is a test. A TEST, hello universe... " +
"UNIVERSE! Hello; *** Is Anybody There?"
val output: String = removeDuplicateWords(input)
println(output)
}
}
/*
run:
Hello This is a test universe Anybody There
*/