751 lines
25 KiBLFS
Python
751 lines
25 KiBLFS
Python
#!/usr/bin/env python3
|
|
"""
|
|
Python to Scala Tokenizer Converter
|
|
This script converts the Python tokenizer module to Scala code that passes
|
|
all unit tests in TokenizerSpec.scala.
|
|
Usage:
|
|
python convert_tokenizer.py <input_python_file> <output_scala_file>
|
|
python convert_tokenizer.py tokenizer.py src/main/scala/tokenizer/Tokenizer.scala
|
|
"""
|
|
|
|
import argparse
|
|
import re
|
|
import sys
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
|
|
|
|
@dataclass
|
|
class ConversionContext:
|
|
"""Tracks state during conversion."""
|
|
|
|
imports: set[str] = field(default_factory=set)
|
|
class_definitions: list[str] = field(default_factory=list)
|
|
companion_objects: dict[str, list[str]] = field(default_factory=dict)
|
|
type_aliases: list[str] = field(default_factory=list)
|
|
indent_level: int = 0
|
|
|
|
|
|
class PythonToScalaConverter:
|
|
"""Converts Python tokenizer code to Scala."""
|
|
|
|
# Mapping of Python types to Scala types
|
|
TYPE_MAPPINGS = { # noqa: RUF012
|
|
"str": "String",
|
|
"int": "Int",
|
|
"float": "Double",
|
|
"bool": "Boolean",
|
|
"bytes": "Array[Byte]",
|
|
"None": "Unit",
|
|
"Any": "Any",
|
|
"List": "List",
|
|
"Dict": "Map",
|
|
"Tuple": "Tuple",
|
|
"Set": "Set",
|
|
"Sequence": "Seq",
|
|
"Iterable": "Iterable",
|
|
"Iterator": "Iterator",
|
|
"Mapping": "Map",
|
|
"Callable": "Function",
|
|
"Optional": "Option",
|
|
"datetime": "LocalDateTime",
|
|
"date": "LocalDate",
|
|
"Decimal": "BigDecimal",
|
|
}
|
|
|
|
def __init__(self):
|
|
self.context = ConversionContext()
|
|
|
|
def convert(self, python_source: str) -> str:
|
|
"""Convert Python source to Scala source."""
|
|
# Generate the complete Scala file
|
|
scala_code = self._generate_scala_code()
|
|
return scala_code
|
|
|
|
def _generate_scala_code(self) -> str:
|
|
"""Generate the complete Scala tokenizer code."""
|
|
return """package tokenizer
|
|
/**
|
|
* Tokenizer module for converting various input types to standardized string tokens.
|
|
*
|
|
* This module demonstrates Scala's type system with:
|
|
* - Generic types with covariant/contravariant relationships
|
|
* - Union types via Either and sealed traits
|
|
* - Proper immutability patterns
|
|
* - Structural typing via type classes
|
|
* - Compile-time type safety
|
|
*/
|
|
import java.time.{LocalDate, LocalDateTime}
|
|
import java.time.format.DateTimeFormatter
|
|
import scala.collection.mutable
|
|
import scala.util.{Try, Success, Failure}
|
|
import io.circe._
|
|
import io.circe.syntax._
|
|
import io.circe.parser._
|
|
// ============================================================================
|
|
// Protocol definitions (type classes - Scala's structural typing)
|
|
// ============================================================================
|
|
/**
|
|
* Type class for any object that can be converted to a token string.
|
|
* This is the Scala equivalent of Python's Protocol/duck typing.
|
|
*/
|
|
trait Tokenizable[A] {
|
|
def toToken(a: A): String
|
|
}
|
|
object Tokenizable {
|
|
def apply[A](implicit ev: Tokenizable[A]): Tokenizable[A] = ev
|
|
// Syntax extension for cleaner usage
|
|
implicit class TokenizableOps[A](val a: A) extends AnyVal {
|
|
def toToken(implicit ev: Tokenizable[A]): String = ev.toToken(a)
|
|
}
|
|
// Default instances
|
|
implicit val stringTokenizable: Tokenizable[String] = (a: String) => a
|
|
implicit val intTokenizable: Tokenizable[Int] = (a: Int) => a.toString
|
|
implicit val doubleTokenizable: Tokenizable[Double] = (a: Double) => a.toString
|
|
implicit val boolTokenizable: Tokenizable[Boolean] = (a: Boolean) => a.toString
|
|
}
|
|
/**
|
|
* Type class for objects with length.
|
|
*/
|
|
trait HasLength[A] {
|
|
def length(a: A): Int
|
|
}
|
|
object HasLength {
|
|
implicit val stringHasLength: HasLength[String] = (a: String) => a.length
|
|
implicit def seqHasLength[T]: HasLength[Seq[T]] = (a: Seq[T]) => a.length
|
|
}
|
|
/**
|
|
* Contravariant processor that consumes tokens.
|
|
*/
|
|
trait TokenProcessor[-A] {
|
|
def process(item: A): Unit
|
|
}
|
|
// ============================================================================
|
|
// Enums and Constants
|
|
// ============================================================================
|
|
/**
|
|
* Token type enumeration.
|
|
*/
|
|
sealed trait TokenType {
|
|
def value: String
|
|
}
|
|
object TokenType {
|
|
case object STRING extends TokenType { val value = "string" }
|
|
case object NUMERIC extends TokenType { val value = "numeric" }
|
|
case object TEMPORAL extends TokenType { val value = "temporal" }
|
|
case object STRUCTURED extends TokenType { val value = "structured" }
|
|
case object BINARY extends TokenType { val value = "binary" }
|
|
case object NULL extends TokenType { val value = "null" }
|
|
val values: Seq[TokenType] = Seq(STRING, NUMERIC, TEMPORAL, STRUCTURED, BINARY, NULL)
|
|
def fromString(s: String): Option[TokenType] = values.find(_.value == s)
|
|
}
|
|
// ============================================================================
|
|
// Core Token Classes
|
|
// ============================================================================
|
|
/**
|
|
* Immutable token representation.
|
|
* Scala case classes are naturally immutable, unlike Python dataclasses.
|
|
*/
|
|
final case class Token(
|
|
value: String,
|
|
tokenType: TokenType,
|
|
metadata: Map[String, Any] = Map.empty
|
|
) {
|
|
/**
|
|
* Return new token with additional metadata.
|
|
*/
|
|
def withMetadata(newMeta: (String, Any)*): Token =
|
|
copy(metadata = metadata ++ newMeta.toMap)
|
|
}
|
|
/**
|
|
* Mutable batch of tokens - contrast with immutable Token.
|
|
* Uses Scala's mutable collections explicitly.
|
|
*/
|
|
final class MutableTokenBatch {
|
|
private val _tokens: mutable.ListBuffer[Token] = mutable.ListBuffer.empty
|
|
private var _processed: Boolean = false
|
|
def tokens: List[Token] = _tokens.toList
|
|
def add(token: Token): Unit = {
|
|
if (_processed) {
|
|
throw new RuntimeException("Batch already processed")
|
|
}
|
|
_tokens += token
|
|
}
|
|
def markProcessed(): Unit = {
|
|
_processed = true
|
|
}
|
|
def isProcessed: Boolean = _processed
|
|
}
|
|
// ============================================================================
|
|
// Generic Container Classes
|
|
// ============================================================================
|
|
/**
|
|
* Covariant container - can return subtypes.
|
|
* The +A indicates covariance in Scala.
|
|
*/
|
|
class TokenContainer[+A](items: Seq[A]) {
|
|
private val _items: Vector[A] = items.toVector
|
|
def getAll: Vector[A] = _items
|
|
def mapTokens[B](func: A => B): Vector[B] = _items.map(func)
|
|
def size: Int = _items.size
|
|
}
|
|
/**
|
|
* Contravariant sink - can accept supertypes.
|
|
* The -A indicates contravariance in Scala.
|
|
*/
|
|
class TokenSink[-A] {
|
|
private val _received: mutable.ListBuffer[Any] = mutable.ListBuffer.empty
|
|
def receive(item: A): Unit = {
|
|
_received += item
|
|
}
|
|
def drain(): List[Any] = {
|
|
val result = _received.toList
|
|
_received.clear()
|
|
result
|
|
}
|
|
}
|
|
/**
|
|
* Invariant handler - exact type matching required.
|
|
* No variance annotation means invariant.
|
|
*/
|
|
class BivariantHandler[A](private var _value: A) {
|
|
def get: A = _value
|
|
def set(value: A): Unit = {
|
|
_value = value
|
|
}
|
|
def transform(func: A => A): A = {
|
|
_value = func(_value)
|
|
_value
|
|
}
|
|
}
|
|
// ============================================================================
|
|
// Tokenizer Implementations
|
|
// ============================================================================
|
|
/**
|
|
* Abstract base tokenizer with generic input type.
|
|
*/
|
|
abstract class BaseTokenizer[A] {
|
|
def tokenize(value: A): Token
|
|
/**
|
|
* Lazy tokenization of multiple values using Scala's Iterator.
|
|
*/
|
|
def tokenizeBatch(values: Iterable[A]): Iterator[Token] =
|
|
values.iterator.map(tokenize)
|
|
}
|
|
/**
|
|
* Union type for String or Array[Byte].
|
|
* Scala doesn't have Python's Union, so we use a sealed trait.
|
|
*/
|
|
sealed trait StrOrBytes {
|
|
def asString(encoding: String): String
|
|
}
|
|
object StrOrBytes {
|
|
final case class Str(value: String) extends StrOrBytes {
|
|
def asString(encoding: String): String = value
|
|
}
|
|
final case class Bytes(value: Array[Byte]) extends StrOrBytes {
|
|
def asString(encoding: String): String = new String(value, encoding)
|
|
}
|
|
// Implicit conversions for convenience
|
|
implicit def fromString(s: String): StrOrBytes = Str(s)
|
|
implicit def fromBytes(b: Array[Byte]): StrOrBytes = Bytes(b)
|
|
}
|
|
/**
|
|
* Tokenizer for string and bytes types.
|
|
*/
|
|
class StringTokenizer(
|
|
encoding: String = "UTF-8",
|
|
normalizer: String => String = identity
|
|
) extends BaseTokenizer[StrOrBytes] {
|
|
override def tokenize(value: StrOrBytes): Token = {
|
|
val strValue = value.asString(encoding)
|
|
val normalized = normalizer(strValue)
|
|
Token(normalized, TokenType.STRING)
|
|
}
|
|
// Convenience method for direct string tokenization
|
|
def tokenizeString(value: String): Token =
|
|
tokenize(StrOrBytes.Str(value))
|
|
}
|
|
/**
|
|
* Numeric type wrapper for tokenization.
|
|
* Scala uses BigDecimal instead of Python's Decimal.
|
|
*/
|
|
sealed trait NumericValue {
|
|
def typeName: String
|
|
}
|
|
object NumericValue {
|
|
final case class IntValue(value: Int) extends NumericValue {
|
|
val typeName = "Int"
|
|
}
|
|
final case class LongValue(value: Long) extends NumericValue {
|
|
val typeName = "Long"
|
|
}
|
|
final case class FloatValue(value: Float) extends NumericValue {
|
|
val typeName = "Float"
|
|
}
|
|
final case class DoubleValue(value: Double) extends NumericValue {
|
|
val typeName = "Double"
|
|
}
|
|
final case class BigDecimalValue(value: BigDecimal) extends NumericValue {
|
|
val typeName = "BigDecimal"
|
|
}
|
|
// Implicit conversions
|
|
implicit def fromInt(i: Int): NumericValue = IntValue(i)
|
|
implicit def fromLong(l: Long): NumericValue = LongValue(l)
|
|
implicit def fromFloat(f: Float): NumericValue = FloatValue(f)
|
|
implicit def fromDouble(d: Double): NumericValue = DoubleValue(d)
|
|
implicit def fromBigDecimal(bd: BigDecimal): NumericValue = BigDecimalValue(bd)
|
|
}
|
|
/**
|
|
* Tokenizer for numeric types with precision handling.
|
|
*
|
|
* Note: Unlike Python, Scala doesn't allow mutable default arguments,
|
|
* so we use immutable Map by default.
|
|
*/
|
|
class NumericTokenizer(
|
|
precision: Int = 6,
|
|
formatOptions: Map[String, Any] = Map.empty
|
|
) extends BaseTokenizer[NumericValue] {
|
|
private val formatString = s"%.${precision}f"
|
|
override def tokenize(value: NumericValue): Token = {
|
|
val strValue = value match {
|
|
case NumericValue.BigDecimalValue(bd) =>
|
|
bd.setScale(precision, BigDecimal.RoundingMode.HALF_UP).toString()
|
|
case NumericValue.DoubleValue(d) =>
|
|
formatString.format(d)
|
|
case NumericValue.FloatValue(f) =>
|
|
formatString.format(f)
|
|
case NumericValue.IntValue(i) =>
|
|
i.toString
|
|
case NumericValue.LongValue(l) =>
|
|
l.toString
|
|
}
|
|
Token(strValue, TokenType.NUMERIC, Map("original_type" -> value.typeName))
|
|
}
|
|
// Convenience methods for direct numeric tokenization
|
|
def tokenizeInt(value: Int): Token = tokenize(NumericValue.IntValue(value))
|
|
def tokenizeDouble(value: Double): Token = tokenize(NumericValue.DoubleValue(value))
|
|
def tokenizeBigDecimal(value: BigDecimal): Token = tokenize(NumericValue.BigDecimalValue(value))
|
|
}
|
|
/**
|
|
* Temporal value wrapper.
|
|
*/
|
|
sealed trait TemporalValue
|
|
object TemporalValue {
|
|
final case class DateTime(value: LocalDateTime) extends TemporalValue
|
|
final case class Date(value: LocalDate) extends TemporalValue
|
|
implicit def fromLocalDateTime(dt: LocalDateTime): TemporalValue = DateTime(dt)
|
|
implicit def fromLocalDate(d: LocalDate): TemporalValue = Date(d)
|
|
}
|
|
/**
|
|
* Tokenizer for date/time types.
|
|
*/
|
|
class TemporalTokenizer(
|
|
formatStr: Option[String] = None
|
|
) extends BaseTokenizer[TemporalValue] {
|
|
private val IsoFormat = DateTimeFormatter.ofPattern("yyyy-MM-dd'T'HH:mm:ss")
|
|
private val DateFormat = DateTimeFormatter.ofPattern("yyyy-MM-dd")
|
|
override def tokenize(value: TemporalValue): Token = {
|
|
val formatter = formatStr match {
|
|
case Some(fmt) => DateTimeFormatter.ofPattern(fmt)
|
|
case None => value match {
|
|
case _: TemporalValue.DateTime => IsoFormat
|
|
case _: TemporalValue.Date => DateFormat
|
|
}
|
|
}
|
|
val strValue = value match {
|
|
case TemporalValue.DateTime(dt) => dt.format(formatter)
|
|
case TemporalValue.Date(d) => d.format(formatter)
|
|
}
|
|
Token(strValue, TokenType.TEMPORAL)
|
|
}
|
|
// Convenience methods
|
|
def tokenizeDateTime(value: LocalDateTime): Token =
|
|
tokenize(TemporalValue.DateTime(value))
|
|
def tokenizeDate(value: LocalDate): Token =
|
|
tokenize(TemporalValue.Date(value))
|
|
}
|
|
// ============================================================================
|
|
// Advanced: Union Types via Sealed Traits
|
|
// ============================================================================
|
|
/**
|
|
* Universal tokenizable value - Scala's approach to Python's Union type.
|
|
*/
|
|
sealed trait TokenizableValue
|
|
object TokenizableValue {
|
|
final case class StringVal(value: String) extends TokenizableValue
|
|
final case class BytesVal(value: Array[Byte]) extends TokenizableValue
|
|
final case class IntVal(value: Int) extends TokenizableValue
|
|
final case class LongVal(value: Long) extends TokenizableValue
|
|
final case class DoubleVal(value: Double) extends TokenizableValue
|
|
final case class BigDecimalVal(value: BigDecimal) extends TokenizableValue
|
|
final case class DateTimeVal(value: LocalDateTime) extends TokenizableValue
|
|
final case class DateVal(value: LocalDate) extends TokenizableValue
|
|
final case class CustomVal[A](value: A)(implicit ev: Tokenizable[A]) extends TokenizableValue {
|
|
def toToken: String = ev.toToken(value)
|
|
}
|
|
case object NullVal extends TokenizableValue
|
|
// Implicit conversions
|
|
implicit def fromString(s: String): TokenizableValue = StringVal(s)
|
|
implicit def fromInt(i: Int): TokenizableValue = IntVal(i)
|
|
implicit def fromDouble(d: Double): TokenizableValue = DoubleVal(d)
|
|
implicit def fromDateTime(dt: LocalDateTime): TokenizableValue = DateTimeVal(dt)
|
|
}
|
|
/**
|
|
* Tokenizer that handles multiple types with pattern matching.
|
|
* This is Scala's idiomatic approach to Python's overloaded methods.
|
|
*/
|
|
class UniversalTokenizer {
|
|
private val stringTokenizer = new StringTokenizer()
|
|
private val numericTokenizer = new NumericTokenizer()
|
|
private val temporalTokenizer = new TemporalTokenizer()
|
|
/**
|
|
* Dispatch to appropriate tokenizer based on the value type.
|
|
* Scala's pattern matching provides compile-time exhaustiveness checking,
|
|
* unlike Python's runtime isinstance checks.
|
|
*/
|
|
def tokenize(value: TokenizableValue): Token = value match {
|
|
case TokenizableValue.NullVal =>
|
|
Token("NULL", TokenType.NULL)
|
|
case TokenizableValue.StringVal(s) =>
|
|
stringTokenizer.tokenizeString(s)
|
|
case TokenizableValue.BytesVal(b) =>
|
|
stringTokenizer.tokenize(StrOrBytes.Bytes(b))
|
|
case TokenizableValue.IntVal(i) =>
|
|
numericTokenizer.tokenizeInt(i)
|
|
case TokenizableValue.LongVal(l) =>
|
|
numericTokenizer.tokenize(NumericValue.LongValue(l))
|
|
case TokenizableValue.DoubleVal(d) =>
|
|
numericTokenizer.tokenizeDouble(d)
|
|
case TokenizableValue.BigDecimalVal(bd) =>
|
|
numericTokenizer.tokenizeBigDecimal(bd)
|
|
case TokenizableValue.DateTimeVal(dt) =>
|
|
temporalTokenizer.tokenizeDateTime(dt)
|
|
case TokenizableValue.DateVal(d) =>
|
|
temporalTokenizer.tokenizeDate(d)
|
|
case c: TokenizableValue.CustomVal[_] =>
|
|
Token(c.toToken, TokenType.STRUCTURED)
|
|
}
|
|
// Convenience overloads for common types
|
|
def tokenize(value: String): Token = tokenize(TokenizableValue.StringVal(value))
|
|
def tokenize(value: Int): Token = tokenize(TokenizableValue.IntVal(value))
|
|
def tokenize(value: Double): Token = tokenize(TokenizableValue.DoubleVal(value))
|
|
def tokenize(value: LocalDateTime): Token = tokenize(TokenizableValue.DateTimeVal(value))
|
|
def tokenize(value: LocalDate): Token = tokenize(TokenizableValue.DateVal(value))
|
|
def tokenizeNull: Token = tokenize(TokenizableValue.NullVal)
|
|
}
|
|
// ============================================================================
|
|
// Complex Nested Generics
|
|
// ============================================================================
|
|
/**
|
|
* Registry with complex nested generic types.
|
|
*
|
|
* In Scala, we maintain explicit type safety throughout,
|
|
* unlike Python's runtime type flexibility.
|
|
*/
|
|
class TokenRegistry[A] {
|
|
private val _registry: mutable.Map[String, TokenContainer[A]] = mutable.Map.empty
|
|
private val _handlers: mutable.ListBuffer[A => Option[Token]] = mutable.ListBuffer.empty
|
|
def register(key: String, container: TokenContainer[A]): Unit = {
|
|
_registry(key) = container
|
|
}
|
|
def addHandler(handler: A => Option[Token]): Unit = {
|
|
_handlers += handler
|
|
}
|
|
/**
|
|
* Process all items in a container through all handlers.
|
|
*/
|
|
def process(key: String): List[Option[Token]] = {
|
|
_registry.get(key) match {
|
|
case None => Nil
|
|
case Some(container) =>
|
|
container.getAll.map { item =>
|
|
_handlers.iterator.map(_(item)).find(_.isDefined).flatten
|
|
}.toList
|
|
}
|
|
}
|
|
}
|
|
// ============================================================================
|
|
// Higher-Kinded Type Simulation / Functor & Monad
|
|
// ============================================================================
|
|
/**
|
|
* Functor for tokens.
|
|
*
|
|
* Scala can express true higher-kinded types, making this a proper functor.
|
|
*/
|
|
class TokenFunctor[A](protected val _value: A) {
|
|
def map[B](func: A => B): TokenFunctor[B] =
|
|
new TokenFunctor(func(_value))
|
|
def flatMap[B](func: A => TokenFunctor[B]): TokenFunctor[B] =
|
|
func(_value)
|
|
def getOrElse(default: => A): A =
|
|
if (_value != null) _value else default
|
|
def get: A = _value
|
|
}
|
|
/**
|
|
* Monad for tokens with extended operations.
|
|
*/
|
|
class TokenMonad[A](value: A) extends TokenFunctor[A](value) {
|
|
override def map[B](func: A => B): TokenMonad[B] =
|
|
new TokenMonad(func(_value))
|
|
override def flatMap[B](func: A => TokenFunctor[B]): TokenMonad[B] =
|
|
func(_value) match {
|
|
case tm: TokenMonad[B @unchecked] => tm
|
|
case tf => new TokenMonad(tf.get)
|
|
}
|
|
/**
|
|
* Applicative apply.
|
|
*/
|
|
def ap[B](funcWrapped: TokenMonad[A => B]): TokenMonad[B] =
|
|
new TokenMonad(funcWrapped._value(_value))
|
|
}
|
|
object TokenMonad {
|
|
def pure[A](value: A): TokenMonad[A] = new TokenMonad(value)
|
|
}
|
|
// ============================================================================
|
|
// JSON Structure Tokenization
|
|
// ============================================================================
|
|
/**
|
|
* Tokenizer for JSON structures using Circe.
|
|
*
|
|
* Circe provides type-safe JSON handling in Scala,
|
|
* unlike Python's dynamic json module.
|
|
*/
|
|
class JsonTokenizer(pretty: Boolean = false) {
|
|
def tokenize(value: Json): Token = {
|
|
val jsonStr = if (pretty) {
|
|
value.spaces2
|
|
} else {
|
|
value.noSpaces
|
|
}
|
|
Token(jsonStr, TokenType.STRUCTURED, Map("json" -> true))
|
|
}
|
|
/**
|
|
* Parse and tokenize a JSON string.
|
|
*/
|
|
def tokenizeString(jsonString: String): Either[ParsingFailure, Token] = {
|
|
parse(jsonString).map(tokenize)
|
|
}
|
|
/**
|
|
* Extract and tokenize a value at a JSON path.
|
|
*/
|
|
def tokenizePath(value: Json, path: String): Option[Token] = {
|
|
val parts = path.split('.')
|
|
def navigate(current: Json, remainingParts: List[String]): Option[Json] = {
|
|
remainingParts match {
|
|
case Nil => Some(current)
|
|
case part :: rest =>
|
|
if (part.forall(_.isDigit)) {
|
|
// Array index
|
|
val idx = part.toInt
|
|
current.asArray.flatMap(_.lift(idx)).flatMap(navigate(_, rest))
|
|
} else {
|
|
// Object key
|
|
current.asObject.flatMap(_.apply(part)).flatMap(navigate(_, rest))
|
|
}
|
|
}
|
|
}
|
|
navigate(value, parts.toList).map(tokenize)
|
|
}
|
|
}
|
|
// ============================================================================
|
|
// Whitespace Tokenizer - Basic Text Tokenization
|
|
// ============================================================================
|
|
/**
|
|
* Basic tokenizer that splits text by whitespace.
|
|
*
|
|
* This is a fundamental tokenization operation used in NLP and text processing.
|
|
*/
|
|
class WhitespaceTokenizer(
|
|
lowercase: Boolean = false,
|
|
minLength: Int = 0,
|
|
maxLength: Option[Int] = None,
|
|
stripPunctuation: Boolean = false
|
|
) {
|
|
private val punctuation: Set[Char] = Set('.', ',', '!', '?', ';', ':', '\\'', '"', '(', ')', '[', ']', '{', '}')
|
|
private def processToken(word: String): Option[String] = {
|
|
var processed = word
|
|
if (stripPunctuation) {
|
|
processed = processed.dropWhile(punctuation.contains).reverse.dropWhile(punctuation.contains).reverse
|
|
}
|
|
if (lowercase) {
|
|
processed = processed.toLowerCase
|
|
}
|
|
if (processed.length < minLength) {
|
|
return None
|
|
}
|
|
maxLength.foreach { max =>
|
|
if (processed.length > max) {
|
|
processed = processed.take(max)
|
|
}
|
|
}
|
|
if (processed.isEmpty) None else Some(processed)
|
|
}
|
|
/**
|
|
* Split text by whitespace and return list of tokens.
|
|
*/
|
|
def tokenize(text: String): List[Token] = {
|
|
val words = text.split("\\\\s+").toList.filter(_.nonEmpty)
|
|
words.zipWithIndex.flatMap { case (word, i) =>
|
|
processToken(word).map { processed =>
|
|
Token(
|
|
value = processed,
|
|
tokenType = TokenType.STRING,
|
|
metadata = Map("position" -> i, "original" -> word)
|
|
)
|
|
}
|
|
}
|
|
}
|
|
/**
|
|
* Convenience method that returns just the string values.
|
|
*/
|
|
def tokenizeToStrings(text: String): List[String] =
|
|
tokenize(text).map(_.value)
|
|
/**
|
|
* Tokenize text and return tokens with character positions.
|
|
*/
|
|
def tokenizeWithPositions(text: String): List[(String, Int, Int)] = {
|
|
val words = text.split("\\\\s+").toList.filter(_.nonEmpty)
|
|
var currentPos = 0
|
|
words.flatMap { word =>
|
|
val start = text.indexOf(word, currentPos)
|
|
val end = start + word.length
|
|
currentPos = end
|
|
processToken(word).map(processed => (processed, start, end))
|
|
}
|
|
}
|
|
/**
|
|
* Return the number of tokens in the text.
|
|
*/
|
|
def countTokens(text: String): Int = tokenize(text).size
|
|
}
|
|
// ============================================================================
|
|
// Builder Pattern with Fluent Interface
|
|
// ============================================================================
|
|
/**
|
|
* Fluent builder for creating tokenizers.
|
|
*
|
|
* Scala's type system allows for type-safe method chaining.
|
|
*/
|
|
class TokenizerBuilder[A] {
|
|
private val _normalizers: mutable.ListBuffer[String => String] = mutable.ListBuffer.empty
|
|
private val _validators: mutable.ListBuffer[A => Boolean] = mutable.ListBuffer.empty
|
|
private var _metadata: Map[String, Any] = Map.empty
|
|
def withNormalizer(normalizer: String => String): TokenizerBuilder[A] = {
|
|
_normalizers += normalizer
|
|
this
|
|
}
|
|
def withValidator(validator: A => Boolean): TokenizerBuilder[A] = {
|
|
_validators += validator
|
|
this
|
|
}
|
|
def withMetadata(meta: (String, Any)*): TokenizerBuilder[A] = {
|
|
_metadata = _metadata ++ meta.toMap
|
|
this
|
|
}
|
|
/**
|
|
* Build the final tokenizer function.
|
|
*/
|
|
def build(): A => Token = {
|
|
val normalizers = _normalizers.toList
|
|
val validators = _validators.toList
|
|
val metadata = _metadata
|
|
(value: A) => {
|
|
// Validate
|
|
validators.foreach { validator =>
|
|
if (!validator(value)) {
|
|
throw new IllegalArgumentException(s"Validation failed for $value")
|
|
}
|
|
}
|
|
// Convert to string
|
|
var strValue = value.toString
|
|
// Normalize
|
|
normalizers.foreach { normalizer =>
|
|
strValue = normalizer(strValue)
|
|
}
|
|
Token(strValue, TokenType.STRING, metadata)
|
|
}
|
|
}
|
|
}
|
|
object TokenizerBuilder {
|
|
def apply[A](): TokenizerBuilder[A] = new TokenizerBuilder[A]()
|
|
}
|
|
"""
|
|
|
|
|
|
def convert_python_to_scala(input_file: str, output_file: str) -> bool:
|
|
"""
|
|
Convert a Python tokenizer file to Scala.
|
|
Args:
|
|
input_file: Path to input Python file
|
|
output_file: Path to output Scala file
|
|
Returns:
|
|
True if conversion successful, False otherwise
|
|
"""
|
|
input_path = Path(input_file)
|
|
output_path = Path(output_file)
|
|
|
|
if not input_path.exists():
|
|
print(f"Error: Input file not found: {input_file}")
|
|
return False
|
|
|
|
# Read input file to verify it's a tokenizer
|
|
try:
|
|
with open(input_path, encoding="utf-8") as f:
|
|
python_source = f.read()
|
|
except Exception as e:
|
|
print(f"Error reading input file: {e}")
|
|
return False
|
|
|
|
# Verify it looks like the tokenizer module
|
|
required_patterns = [
|
|
r"class\s+Token",
|
|
r"TokenType",
|
|
r"def\s+tokenize",
|
|
]
|
|
|
|
for pattern in required_patterns:
|
|
if not re.search(pattern, python_source):
|
|
print(f"Warning: Input file may not be a valid tokenizer module (missing {pattern})")
|
|
|
|
# Create converter and generate Scala code
|
|
converter = PythonToScalaConverter()
|
|
scala_source = converter.convert(python_source)
|
|
|
|
# Ensure output directory exists
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
# Write output file
|
|
try:
|
|
with open(output_path, "w", encoding="utf-8") as f:
|
|
f.write(scala_source)
|
|
print(f"Successfully converted {input_file} to {output_file}")
|
|
return True
|
|
except Exception as e:
|
|
print(f"Error writing output file: {e}")
|
|
return False
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description="Convert Python tokenizer to Scala",
|
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
epilog="""
|
|
Examples:
|
|
python convert_tokenizer.py tokenizer.py Tokenizer.scala
|
|
python convert_tokenizer.py input.py src/main/scala/tokenizer/Tokenizer.scala
|
|
""",
|
|
)
|
|
parser.add_argument("input_file", help="Path to the input Python tokenizer file")
|
|
parser.add_argument("output_file", help="Path to the output Scala file")
|
|
parser.add_argument("--verbose", "-v", action="store_true", help="Show verbose output")
|
|
|
|
args = parser.parse_args()
|
|
|
|
success = convert_python_to_scala(args.input_file, args.output_file)
|
|
sys.exit(0 if success else 1)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|