Files
SkillCompiler/data/skills-bench/tasks/python-scala-translation/src/main/scala/tokenizer/Tokenizer.scala
T
2026-09-04 14:58:42 +08:00

737 lines
21 KiBLFS
Scala

package tokenizer
/**
* Tokenizer module for converting various input types to standardized string tokens.
*
* This module demonstrates Scala's type system with:
* - Generic types with covariant/contravariant relationships
* - Union types via Either and sealed traits
* - Proper immutability patterns
* - Structural typing via type classes
* - Compile-time type safety
*/
import java.time.{LocalDate, LocalDateTime}
import java.time.format.DateTimeFormatter
import scala.collection.mutable
import scala.util.{Try, Success, Failure}
import io.circe._
import io.circe.syntax._
import io.circe.parser._
// ============================================================================
// Protocol definitions (type classes - Scala's structural typing)
// ============================================================================
/**
* Type class for any object that can be converted to a token string.
* This is the Scala equivalent of Python's Protocol/duck typing.
*/
trait Tokenizable[A] {
def toToken(a: A): String
}
object Tokenizable {
def apply[A](implicit ev: Tokenizable[A]): Tokenizable[A] = ev
// Syntax extension for cleaner usage
implicit class TokenizableOps[A](val a: A) extends AnyVal {
def toToken(implicit ev: Tokenizable[A]): String = ev.toToken(a)
}
// Default instances
implicit val stringTokenizable: Tokenizable[String] = (a: String) => a
implicit val intTokenizable: Tokenizable[Int] = (a: Int) => a.toString
implicit val doubleTokenizable: Tokenizable[Double] = (a: Double) => a.toString
implicit val boolTokenizable: Tokenizable[Boolean] = (a: Boolean) => a.toString
}
/**
* Type class for objects with length.
*/
trait HasLength[A] {
def length(a: A): Int
}
object HasLength {
implicit val stringHasLength: HasLength[String] = (a: String) => a.length
implicit def seqHasLength[T]: HasLength[Seq[T]] = (a: Seq[T]) => a.length
}
/**
* Contravariant processor that consumes tokens.
*/
trait TokenProcessor[-A] {
def process(item: A): Unit
}
// ============================================================================
// Enums and Constants
// ============================================================================
/**
* Token type enumeration.
*/
sealed trait TokenType {
def value: String
}
object TokenType {
case object STRING extends TokenType { val value = "string" }
case object NUMERIC extends TokenType { val value = "numeric" }
case object TEMPORAL extends TokenType { val value = "temporal" }
case object STRUCTURED extends TokenType { val value = "structured" }
case object BINARY extends TokenType { val value = "binary" }
case object NULL extends TokenType { val value = "null" }
val values: Seq[TokenType] = Seq(STRING, NUMERIC, TEMPORAL, STRUCTURED, BINARY, NULL)
def fromString(s: String): Option[TokenType] = values.find(_.value == s)
}
// ============================================================================
// Core Token Classes
// ============================================================================
/**
* Immutable token representation.
* Scala case classes are naturally immutable, unlike Python dataclasses.
*/
final case class Token(
value: String,
tokenType: TokenType,
metadata: Map[String, Any] = Map.empty
) {
/**
* Return new token with additional metadata.
*/
def withMetadata(newMeta: (String, Any)*): Token =
copy(metadata = metadata ++ newMeta.toMap)
}
/**
* Mutable batch of tokens - contrast with immutable Token.
* Uses Scala's mutable collections explicitly.
*/
final class MutableTokenBatch {
private val _tokens: mutable.ListBuffer[Token] = mutable.ListBuffer.empty
private var _processed: Boolean = false
def tokens: List[Token] = _tokens.toList
def add(token: Token): Unit = {
if (_processed) {
throw new RuntimeException("Batch already processed")
}
_tokens += token
}
def markProcessed(): Unit = {
_processed = true
}
def isProcessed: Boolean = _processed
}
// ============================================================================
// Generic Container Classes
// ============================================================================
/**
* Covariant container - can return subtypes.
* The +A indicates covariance in Scala.
*/
class TokenContainer[+A](items: Seq[A]) {
private val _items: Vector[A] = items.toVector
def getAll: Vector[A] = _items
def mapTokens[B](func: A => B): Vector[B] = _items.map(func)
def size: Int = _items.size
}
/**
* Contravariant sink - can accept supertypes.
* The -A indicates contravariance in Scala.
*/
class TokenSink[-A] {
private val _received: mutable.ListBuffer[Any] = mutable.ListBuffer.empty
def receive(item: A): Unit = {
_received += item
}
def drain(): List[Any] = {
val result = _received.toList
_received.clear()
result
}
}
/**
* Invariant handler - exact type matching required.
* No variance annotation means invariant.
*/
class BivariantHandler[A](private var _value: A) {
def get: A = _value
def set(value: A): Unit = {
_value = value
}
def transform(func: A => A): A = {
_value = func(_value)
_value
}
}
// ============================================================================
// Tokenizer Implementations
// ============================================================================
/**
* Abstract base tokenizer with generic input type.
*/
abstract class BaseTokenizer[A] {
def tokenize(value: A): Token
/**
* Lazy tokenization of multiple values using Scala's Iterator.
*/
def tokenizeBatch(values: Iterable[A]): Iterator[Token] =
values.iterator.map(tokenize)
}
/**
* Union type for String or Array[Byte].
* Scala doesn't have Python's Union, so we use a sealed trait.
*/
sealed trait StrOrBytes {
def asString(encoding: String): String
}
object StrOrBytes {
final case class Str(value: String) extends StrOrBytes {
def asString(encoding: String): String = value
}
final case class Bytes(value: Array[Byte]) extends StrOrBytes {
def asString(encoding: String): String = new String(value, encoding)
}
// Implicit conversions for convenience
implicit def fromString(s: String): StrOrBytes = Str(s)
implicit def fromBytes(b: Array[Byte]): StrOrBytes = Bytes(b)
}
/**
* Tokenizer for string and bytes types.
*/
class StringTokenizer(
encoding: String = "UTF-8",
normalizer: String => String = identity
) extends BaseTokenizer[StrOrBytes] {
override def tokenize(value: StrOrBytes): Token = {
val strValue = value.asString(encoding)
val normalized = normalizer(strValue)
Token(normalized, TokenType.STRING)
}
// Convenience method for direct string tokenization
def tokenizeString(value: String): Token =
tokenize(StrOrBytes.Str(value))
}
/**
* Numeric type wrapper for tokenization.
* Scala uses BigDecimal instead of Python's Decimal.
*/
sealed trait NumericValue {
def typeName: String
}
object NumericValue {
final case class IntValue(value: Int) extends NumericValue {
val typeName = "Int"
}
final case class LongValue(value: Long) extends NumericValue {
val typeName = "Long"
}
final case class FloatValue(value: Float) extends NumericValue {
val typeName = "Float"
}
final case class DoubleValue(value: Double) extends NumericValue {
val typeName = "Double"
}
final case class BigDecimalValue(value: BigDecimal) extends NumericValue {
val typeName = "BigDecimal"
}
// Implicit conversions
implicit def fromInt(i: Int): NumericValue = IntValue(i)
implicit def fromLong(l: Long): NumericValue = LongValue(l)
implicit def fromFloat(f: Float): NumericValue = FloatValue(f)
implicit def fromDouble(d: Double): NumericValue = DoubleValue(d)
implicit def fromBigDecimal(bd: BigDecimal): NumericValue = BigDecimalValue(bd)
}
/**
* Tokenizer for numeric types with precision handling.
*
* Note: Unlike Python, Scala doesn't allow mutable default arguments,
* so we use immutable Map by default.
*/
class NumericTokenizer(
precision: Int = 6,
formatOptions: Map[String, Any] = Map.empty
) extends BaseTokenizer[NumericValue] {
private val formatString = s"%.${precision}f"
override def tokenize(value: NumericValue): Token = {
val strValue = value match {
case NumericValue.BigDecimalValue(bd) =>
bd.setScale(precision, BigDecimal.RoundingMode.HALF_UP).toString()
case NumericValue.DoubleValue(d) =>
formatString.format(d)
case NumericValue.FloatValue(f) =>
formatString.format(f)
case NumericValue.IntValue(i) =>
i.toString
case NumericValue.LongValue(l) =>
l.toString
}
Token(strValue, TokenType.NUMERIC, Map("original_type" -> value.typeName))
}
// Convenience methods for direct numeric tokenization
def tokenizeInt(value: Int): Token = tokenize(NumericValue.IntValue(value))
def tokenizeDouble(value: Double): Token = tokenize(NumericValue.DoubleValue(value))
def tokenizeBigDecimal(value: BigDecimal): Token = tokenize(NumericValue.BigDecimalValue(value))
}
/**
* Temporal value wrapper.
*/
sealed trait TemporalValue
object TemporalValue {
final case class DateTime(value: LocalDateTime) extends TemporalValue
final case class Date(value: LocalDate) extends TemporalValue
implicit def fromLocalDateTime(dt: LocalDateTime): TemporalValue = DateTime(dt)
implicit def fromLocalDate(d: LocalDate): TemporalValue = Date(d)
}
/**
* Tokenizer for date/time types.
*/
class TemporalTokenizer(
formatStr: Option[String] = None
) extends BaseTokenizer[TemporalValue] {
private val IsoFormat = DateTimeFormatter.ofPattern("yyyy-MM-dd'T'HH:mm:ss")
private val DateFormat = DateTimeFormatter.ofPattern("yyyy-MM-dd")
override def tokenize(value: TemporalValue): Token = {
val formatter = formatStr match {
case Some(fmt) => DateTimeFormatter.ofPattern(fmt)
case None => value match {
case _: TemporalValue.DateTime => IsoFormat
case _: TemporalValue.Date => DateFormat
}
}
val strValue = value match {
case TemporalValue.DateTime(dt) => dt.format(formatter)
case TemporalValue.Date(d) => d.format(formatter)
}
Token(strValue, TokenType.TEMPORAL)
}
// Convenience methods
def tokenizeDateTime(value: LocalDateTime): Token =
tokenize(TemporalValue.DateTime(value))
def tokenizeDate(value: LocalDate): Token =
tokenize(TemporalValue.Date(value))
}
// ============================================================================
// Advanced: Union Types via Sealed Traits
// ============================================================================
/**
* Universal tokenizable value - Scala's approach to Python's Union type.
*/
sealed trait TokenizableValue
object TokenizableValue {
final case class StringVal(value: String) extends TokenizableValue
final case class BytesVal(value: Array[Byte]) extends TokenizableValue
final case class IntVal(value: Int) extends TokenizableValue
final case class LongVal(value: Long) extends TokenizableValue
final case class DoubleVal(value: Double) extends TokenizableValue
final case class BigDecimalVal(value: BigDecimal) extends TokenizableValue
final case class DateTimeVal(value: LocalDateTime) extends TokenizableValue
final case class DateVal(value: LocalDate) extends TokenizableValue
final case class CustomVal[A](value: A)(implicit ev: Tokenizable[A]) extends TokenizableValue {
def toToken: String = ev.toToken(value)
}
case object NullVal extends TokenizableValue
// Implicit conversions
implicit def fromString(s: String): TokenizableValue = StringVal(s)
implicit def fromInt(i: Int): TokenizableValue = IntVal(i)
implicit def fromDouble(d: Double): TokenizableValue = DoubleVal(d)
implicit def fromDateTime(dt: LocalDateTime): TokenizableValue = DateTimeVal(dt)
}
/**
* Tokenizer that handles multiple types with pattern matching.
* This is Scala's idiomatic approach to Python's overloaded methods.
*/
class UniversalTokenizer {
private val stringTokenizer = new StringTokenizer()
private val numericTokenizer = new NumericTokenizer()
private val temporalTokenizer = new TemporalTokenizer()
/**
* Dispatch to appropriate tokenizer based on the value type.
* Scala's pattern matching provides compile-time exhaustiveness checking,
* unlike Python's runtime isinstance checks.
*/
def tokenize(value: TokenizableValue): Token = value match {
case TokenizableValue.NullVal =>
Token("NULL", TokenType.NULL)
case TokenizableValue.StringVal(s) =>
stringTokenizer.tokenizeString(s)
case TokenizableValue.BytesVal(b) =>
stringTokenizer.tokenize(StrOrBytes.Bytes(b))
case TokenizableValue.IntVal(i) =>
numericTokenizer.tokenizeInt(i)
case TokenizableValue.LongVal(l) =>
numericTokenizer.tokenize(NumericValue.LongValue(l))
case TokenizableValue.DoubleVal(d) =>
numericTokenizer.tokenizeDouble(d)
case TokenizableValue.BigDecimalVal(bd) =>
numericTokenizer.tokenizeBigDecimal(bd)
case TokenizableValue.DateTimeVal(dt) =>
temporalTokenizer.tokenizeDateTime(dt)
case TokenizableValue.DateVal(d) =>
temporalTokenizer.tokenizeDate(d)
case c: TokenizableValue.CustomVal[_] =>
Token(c.toToken, TokenType.STRUCTURED)
}
// Convenience overloads for common types
def tokenize(value: String): Token = tokenize(TokenizableValue.StringVal(value))
def tokenize(value: Int): Token = tokenize(TokenizableValue.IntVal(value))
def tokenize(value: Double): Token = tokenize(TokenizableValue.DoubleVal(value))
def tokenize(value: LocalDateTime): Token = tokenize(TokenizableValue.DateTimeVal(value))
def tokenize(value: LocalDate): Token = tokenize(TokenizableValue.DateVal(value))
def tokenizeNull: Token = tokenize(TokenizableValue.NullVal)
}
// ============================================================================
// Complex Nested Generics
// ============================================================================
/**
* Registry with complex nested generic types.
*
* In Scala, we maintain explicit type safety throughout,
* unlike Python's runtime type flexibility.
*/
class TokenRegistry[A] {
private val _registry: mutable.Map[String, TokenContainer[A]] = mutable.Map.empty
private val _handlers: mutable.ListBuffer[A => Option[Token]] = mutable.ListBuffer.empty
def register(key: String, container: TokenContainer[A]): Unit = {
_registry(key) = container
}
def addHandler(handler: A => Option[Token]): Unit = {
_handlers += handler
}
/**
* Process all items in a container through all handlers.
*/
def process(key: String): List[Option[Token]] = {
_registry.get(key) match {
case None => Nil
case Some(container) =>
container.getAll.map { item =>
_handlers.iterator.map(_(item)).find(_.isDefined).flatten
}.toList
}
}
}
// ============================================================================
// Higher-Kinded Type Simulation / Functor & Monad
// ============================================================================
/**
* Functor for tokens.
*
* Scala can express true higher-kinded types, making this a proper functor.
*/
class TokenFunctor[A](protected val _value: A) {
def map[B](func: A => B): TokenFunctor[B] =
new TokenFunctor(func(_value))
def flatMap[B](func: A => TokenFunctor[B]): TokenFunctor[B] =
func(_value)
def getOrElse(default: => A): A =
if (_value != null) _value else default
def get: A = _value
}
/**
* Monad for tokens with extended operations.
*/
class TokenMonad[A](value: A) extends TokenFunctor[A](value) {
override def map[B](func: A => B): TokenMonad[B] =
new TokenMonad(func(_value))
override def flatMap[B](func: A => TokenFunctor[B]): TokenMonad[B] =
func(_value) match {
case tm: TokenMonad[B] => tm
case tf => new TokenMonad(tf.get)
}
/**
* Applicative apply.
*/
def ap[B](funcWrapped: TokenMonad[A => B]): TokenMonad[B] =
new TokenMonad(funcWrapped._value(_value))
}
object TokenMonad {
def pure[A](value: A): TokenMonad[A] = new TokenMonad(value)
}
// ============================================================================
// JSON Structure Tokenization
// ============================================================================
/**
* Tokenizer for JSON structures using Circe.
*
* Circe provides type-safe JSON handling in Scala,
* unlike Python's dynamic json module.
*/
class JsonTokenizer(pretty: Boolean = false) {
def tokenize(value: Json): Token = {
val jsonStr = if (pretty) {
value.spaces2
} else {
value.noSpaces
}
Token(jsonStr, TokenType.STRUCTURED, Map("json" -> true))
}
/**
* Parse and tokenize a JSON string.
*/
def tokenizeString(jsonString: String): Either[ParsingFailure, Token] = {
parse(jsonString).map(tokenize)
}
/**
* Extract and tokenize a value at a JSON path.
*/
def tokenizePath(value: Json, path: String): Option[Token] = {
val parts = path.split('.')
def navigate(current: Json, remainingParts: List[String]): Option[Json] = {
remainingParts match {
case Nil => Some(current)
case part :: rest =>
if (part.forall(_.isDigit)) {
// Array index
val idx = part.toInt
current.asArray.flatMap(_.lift(idx)).flatMap(navigate(_, rest))
} else {
// Object key
current.asObject.flatMap(_.apply(part)).flatMap(navigate(_, rest))
}
}
}
navigate(value, parts.toList).map(tokenize)
}
}
// ============================================================================
// Whitespace Tokenizer - Basic Text Tokenization
// ============================================================================
/**
* Basic tokenizer that splits text by whitespace.
*
* This is a fundamental tokenization operation used in NLP and text processing.
*/
class WhitespaceTokenizer(
lowercase: Boolean = false,
minLength: Int = 0,
maxLength: Option[Int] = None,
stripPunctuation: Boolean = false
) {
private val punctuation: Set[Char] = Set('.', ',', '!', '?', ';', ':', '\'', '"', '(', ')', '[', ']', '{', '}')
private def processToken(word: String): Option[String] = {
var processed = word
if (stripPunctuation) {
processed = processed.dropWhile(punctuation.contains).reverse.dropWhile(punctuation.contains).reverse
}
if (lowercase) {
processed = processed.toLowerCase
}
if (processed.length < minLength) {
return None
}
maxLength.foreach { max =>
if (processed.length > max) {
processed = processed.take(max)
}
}
if (processed.isEmpty) None else Some(processed)
}
/**
* Split text by whitespace and return list of tokens.
*/
def tokenize(text: String): List[Token] = {
val words = text.split("\\s+").toList.filter(_.nonEmpty)
words.zipWithIndex.flatMap { case (word, i) =>
processToken(word).map { processed =>
Token(
value = processed,
tokenType = TokenType.STRING,
metadata = Map("position" -> i, "original" -> word)
)
}
}
}
/**
* Convenience method that returns just the string values.
*/
def tokenizeToStrings(text: String): List[String] =
tokenize(text).map(_.value)
/**
* Tokenize text and return tokens with character positions.
*/
def tokenizeWithPositions(text: String): List[(String, Int, Int)] = {
val words = text.split("\\s+").toList.filter(_.nonEmpty)
var currentPos = 0
words.flatMap { word =>
val start = text.indexOf(word, currentPos)
val end = start + word.length
currentPos = end
processToken(word).map(processed => (processed, start, end))
}
}
/**
* Return the number of tokens in the text.
*/
def countTokens(text: String): Int = tokenize(text).size
}
// ============================================================================
// Builder Pattern with Fluent Interface
// ============================================================================
/**
* Fluent builder for creating tokenizers.
*
* Scala's type system allows for type-safe method chaining.
*/
class TokenizerBuilder[A] {
private val _normalizers: mutable.ListBuffer[String => String] = mutable.ListBuffer.empty
private val _validators: mutable.ListBuffer[A => Boolean] = mutable.ListBuffer.empty
private var _metadata: Map[String, Any] = Map.empty
def withNormalizer(normalizer: String => String): TokenizerBuilder[A] = {
_normalizers += normalizer
this
}
def withValidator(validator: A => Boolean): TokenizerBuilder[A] = {
_validators += validator
this
}
def withMetadata(meta: (String, Any)*): TokenizerBuilder[A] = {
_metadata = _metadata ++ meta.toMap
this
}
/**
* Build the final tokenizer function.
*/
def build(): A => Token = {
val normalizers = _normalizers.toList
val validators = _validators.toList
val metadata = _metadata
(value: A) => {
// Validate
validators.foreach { validator =>
if (!validator(value)) {
throw new IllegalArgumentException(s"Validation failed for $value")
}
}
// Convert to string
var strValue = value.toString
// Normalize
normalizers.foreach { normalizer =>
strValue = normalizer(strValue)
}
Token(strValue, TokenType.STRING, metadata)
}
}
}
object TokenizerBuilder {
def apply[A](): TokenizerBuilder[A] = new TokenizerBuilder[A]()
}