""" Tokenizer module for converting various input types to standardized string tokens. This module demonstrates Python's flexible type system with: - Generic types with covariant/contravariant relationships - Union types and Optional handling - Mutable default arguments (Python anti-pattern) - Duck typing with Protocol classes - Runtime type flexibility that static typing can't capture """ import json from abc import ABC, abstractmethod from collections.abc import Callable, Iterable, Iterator, Sequence from dataclasses import dataclass, field from datetime import date, datetime from decimal import Decimal from enum import Enum from typing import Any, Generic, Protocol, TypeVar, Union, overload, runtime_checkable # ============================================================================ # Type Variables with various constraints # ============================================================================ T = TypeVar("T") T_co = TypeVar("T_co", covariant=True) T_contra = TypeVar("T_contra", contravariant=True) NumericT = TypeVar("NumericT", int, float, Decimal) StrOrBytes = TypeVar("StrOrBytes", str, bytes) # ============================================================================ # Protocol definitions (structural typing) # ============================================================================ @runtime_checkable class Tokenizable(Protocol): """Any object that can be converted to a token string.""" def to_token(self) -> str: ... @runtime_checkable class HasLength(Protocol): """Any object with a length.""" def __len__(self) -> int: ... class TokenProcessor(Protocol[T_contra]): """Contravariant processor that consumes tokens.""" def process(self, item: T_contra) -> None: ... # ============================================================================ # Enums and Constants # ============================================================================ class TokenType(Enum): STRING = "string" NUMERIC = "numeric" TEMPORAL = "temporal" STRUCTURED = "structured" BINARY = "binary" NULL = "null" # ============================================================================ # Core Token Classes # ============================================================================ @dataclass(frozen=True) class Token: """Immutable token representation.""" value: str token_type: TokenType metadata: dict[str, Any] = field(default_factory=dict) def with_metadata(self, **kwargs: Any) -> "Token": """Return new token with additional metadata.""" new_meta = {**self.metadata, **kwargs} return Token(self.value, self.token_type, new_meta) @dataclass class MutableTokenBatch: """Mutable batch of tokens - contrast with immutable Token.""" tokens: list[Token] = field(default_factory=list) _processed: bool = False def add(self, token: Token) -> None: if self._processed: raise RuntimeError("Batch already processed") self.tokens.append(token) def mark_processed(self) -> None: self._processed = True # ============================================================================ # Generic Container Classes # ============================================================================ class TokenContainer(Generic[T_co]): """Covariant container - can return subtypes.""" def __init__(self, items: Sequence[T_co]) -> None: self._items: tuple[T_co, ...] = tuple(items) def get_all(self) -> tuple[T_co, ...]: return self._items def map_tokens(self, func: Callable[[T_co], str]) -> list[str]: return [func(item) for item in self._items] class TokenSink(Generic[T_contra]): """Contravariant sink - can accept supertypes.""" def __init__(self) -> None: self._received: list[Any] = [] def receive(self, item: T_contra) -> None: self._received.append(item) def drain(self) -> list[Any]: result = self._received.copy() self._received.clear() return result class BivariantHandler(Generic[T]): """Invariant handler - exact type matching required.""" def __init__(self, default: T) -> None: self._value: T = default def get(self) -> T: return self._value def set(self, value: T) -> None: self._value = value def transform(self, func: Callable[[T], T]) -> T: self._value = func(self._value) return self._value # ============================================================================ # Tokenizer Implementations # ============================================================================ class BaseTokenizer(ABC, Generic[T]): """Abstract base tokenizer with generic input type.""" @abstractmethod def tokenize(self, value: T) -> Token: """Convert value to token.""" pass def tokenize_batch(self, values: Iterable[T]) -> Iterator[Token]: """Lazy tokenization of multiple values.""" for v in values: yield self.tokenize(v) class StringTokenizer(BaseTokenizer[StrOrBytes]): """Tokenizer for string and bytes types.""" def __init__(self, encoding: str = "utf-8", normalizer: Callable[[str], str] | None = None) -> None: self.encoding = encoding self.normalizer = normalizer or (lambda x: x) def tokenize(self, value: StrOrBytes) -> Token: if isinstance(value, bytes): str_value = value.decode(self.encoding) else: str_value = value normalized = self.normalizer(str_value) return Token(normalized, TokenType.STRING) class NumericTokenizer(BaseTokenizer[NumericT]): """Tokenizer for numeric types with precision handling.""" def __init__( self, precision: int = 6, # DANGER: Mutable default argument - Python allows, Scala doesn't format_options: dict[str, Any] = {}, # noqa: B006 ) -> None: self.precision = precision self.format_options = format_options def tokenize(self, value: NumericT) -> Token: if isinstance(value, Decimal): str_value = f"{value:.{self.precision}f}" elif isinstance(value, float): str_value = f"{value:.{self.precision}f}" else: str_value = str(value) return Token(str_value, TokenType.NUMERIC, {"original_type": type(value).__name__}) class TemporalTokenizer(BaseTokenizer[Union[datetime, date]]): # noqa: UP007 """Tokenizer for date/time types.""" ISO_FORMAT = "%Y-%m-%dT%H:%M:%S" DATE_FORMAT = "%Y-%m-%d" def __init__(self, format_str: str | None = None) -> None: self.format_str = format_str def tokenize(self, value: datetime | date) -> Token: if self.format_str: fmt = self.format_str elif isinstance(value, datetime): fmt = self.ISO_FORMAT else: fmt = self.DATE_FORMAT return Token(value.strftime(fmt), TokenType.TEMPORAL) # ============================================================================ # Advanced: Union Types and Overloads # ============================================================================ class UniversalTokenizer: """Tokenizer that handles multiple types with overloaded methods.""" def __init__(self) -> None: self._string_tokenizer = StringTokenizer() self._numeric_tokenizer = NumericTokenizer() self._temporal_tokenizer = TemporalTokenizer() # Overloaded signatures for type-specific behavior @overload def tokenize(self, value: str) -> Token: ... @overload def tokenize(self, value: bytes) -> Token: ... @overload def tokenize(self, value: int) -> Token: ... @overload def tokenize(self, value: float) -> Token: ... @overload def tokenize(self, value: datetime) -> Token: ... @overload def tokenize(self, value: None) -> Token: ... @overload def tokenize(self, value: Tokenizable) -> Token: ... def tokenize(self, value: Any) -> Token: """ Dispatch to appropriate tokenizer based on runtime type. Note: Python allows this duck-typing dispatch that Scala's static type system would reject without explicit type classes. """ if value is None: return Token("NULL", TokenType.NULL) if isinstance(value, Tokenizable): return Token(value.to_token(), TokenType.STRUCTURED) if isinstance(value, (str, bytes)): return self._string_tokenizer.tokenize(value) if isinstance(value, (int, float, Decimal)): return self._numeric_tokenizer.tokenize(value) if isinstance(value, (datetime, date)): return self._temporal_tokenizer.tokenize(value) # Fallback: try str() conversion return Token(str(value), TokenType.STRING, {"fallback": True}) # ============================================================================ # Complex Nested Generics # ============================================================================ class TokenRegistry(Generic[T]): """ Registry with complex nested generic types. This pattern is particularly challenging to translate to Scala due to the mixing of mutable and immutable collections with generics. """ def __init__(self) -> None: # Nested generics with mixed mutability self._registry: dict[str, TokenContainer[T]] = {} self._handlers: list[Callable[[T], Token | None]] = [] def register(self, key: str, container: TokenContainer[T]) -> None: self._registry[key] = container def add_handler(self, handler: Callable[[T], Token | None]) -> None: self._handlers.append(handler) def process(self, key: str) -> list[Token | None]: """Process all items in a container through all handlers.""" container = self._registry.get(key) if container is None: return [] results: list[Token | None] = [] for item in container.get_all(): for handler in self._handlers: result = handler(item) if result is not None: results.append(result) break else: results.append(None) return results # ============================================================================ # Higher-Kinded Type Simulation # ============================================================================ F = TypeVar("F") # Type constructor placeholder class TokenFunctor(Generic[T]): """ Simulated functor for tokens. Python can't express true higher-kinded types (HKT), but Scala can. This simulation needs to become a proper type class in Scala. """ def __init__(self, value: T) -> None: self._value = value def map(self, func: Callable[[T], Any]) -> "TokenFunctor[Any]": return TokenFunctor(func(self._value)) def flat_map(self, func: Callable[[T], "TokenFunctor[Any]"]) -> "TokenFunctor[Any]": return func(self._value) def get_or_else(self, default: T) -> T: return self._value if self._value is not None else default class TokenMonad(TokenFunctor[T]): """Extended monad operations.""" @classmethod def pure(cls, value: T) -> "TokenMonad[T]": return cls(value) def ap(self, func_wrapped: "TokenMonad[Callable[[T], Any]]") -> "TokenMonad[Any]": """Applicative apply.""" return TokenMonad(func_wrapped._value(self._value)) # ============================================================================ # JSON Structure Tokenization # ============================================================================ JsonValue = Union[str, int, float, bool, None, list["JsonValue"], dict[str, "JsonValue"]] # noqa UP007 class JsonTokenizer: """ Tokenizer for JSON structures with recursive types. The recursive JsonValue type alias is tricky in Scala due to the need for explicit recursive type definitions. """ def __init__(self, pretty: bool = False) -> None: self.pretty = pretty def tokenize(self, value: JsonValue) -> Token: if self.pretty: json_str = json.dumps(value, indent=2) else: json_str = json.dumps(value) return Token(json_str, TokenType.STRUCTURED, {"json": True}) def tokenize_path(self, value: JsonValue, path: str) -> Token | None: """Extract and tokenize a value at a JSON path.""" parts = path.split(".") current: Any = value for part in parts: if isinstance(current, dict) and part in current: current = current[part] elif isinstance(current, list) and part.isdigit(): idx = int(part) if 0 <= idx < len(current): current = current[idx] else: return None else: return None return self.tokenize(current) # ============================================================================ # Whitespace Tokenizer - Basic Text Tokenization # ============================================================================ class WhitespaceTokenizer: """ Basic tokenizer that splits text by whitespace. This is a fundamental tokenization operation used in NLP and text processing. Demonstrates a simple, practical use case for the tokenizer module. """ def __init__(self, lowercase: bool = False, min_length: int = 0, max_length: int | None = None, strip_punctuation: bool = False) -> None: """ Initialize whitespace tokenizer with options. Args: lowercase: Convert all tokens to lowercase min_length: Minimum token length (shorter tokens are filtered out) max_length: Maximum token length (longer tokens are truncated) strip_punctuation: Remove leading/trailing punctuation from tokens """ self.lowercase = lowercase self.min_length = min_length self.max_length = max_length self.strip_punctuation = strip_punctuation self._punctuation = set(".,!?;:'\"()[]{}") def _process_token(self, word: str) -> str | None: """Process a single word into a token string.""" if self.strip_punctuation: word = word.strip("".join(self._punctuation)) if self.lowercase: word = word.lower() if len(word) < self.min_length: return None if self.max_length is not None and len(word) > self.max_length: word = word[: self.max_length] return word if word else None def tokenize(self, text: str) -> list[Token]: """ Split text by whitespace and return list of tokens. Args: text: Input text to tokenize Returns: List of Token objects, one per word """ words = text.split() tokens: list[Token] = [] for i, word in enumerate(words): processed = self._process_token(word) if processed is not None: token = Token(value=processed, token_type=TokenType.STRING, metadata={"position": i, "original": word}) tokens.append(token) return tokens def tokenize_to_strings(self, text: str) -> list[str]: """ Split text by whitespace and return list of token strings. Convenience method that returns just the string values. Args: text: Input text to tokenize Returns: List of token strings """ return [t.value for t in self.tokenize(text)] def tokenize_with_positions(self, text: str) -> list[tuple[str, int, int]]: """ Tokenize text and return tokens with character positions. Args: text: Input text to tokenize Returns: List of tuples (token_string, start_pos, end_pos) """ result: list[tuple[str, int, int]] = [] current_pos = 0 for word in text.split(): # Find the actual position in the original text start = text.find(word, current_pos) end = start + len(word) processed = self._process_token(word) if processed is not None: result.append((processed, start, end)) current_pos = end return result def count_tokens(self, text: str) -> int: """Return the number of tokens in the text.""" return len(self.tokenize(text)) # ============================================================================ # Builder Pattern with Fluent Interface # ============================================================================ class TokenizerBuilder(Generic[T]): """ Fluent builder for creating tokenizers. The method chaining with generic return types requires careful handling of type bounds in Scala. """ def __init__(self) -> None: self._normalizers: list[Callable[[str], str]] = [] self._validators: list[Callable[[T], bool]] = [] self._metadata: dict[str, Any] = {} def with_normalizer(self, normalizer: Callable[[str], str]) -> "TokenizerBuilder[T]": self._normalizers.append(normalizer) return self def with_validator(self, validator: Callable[[T], bool]) -> "TokenizerBuilder[T]": self._validators.append(validator) return self def with_metadata(self, **kwargs: Any) -> "TokenizerBuilder[T]": self._metadata.update(kwargs) return self def build(self) -> Callable[[T], Token]: """Build the final tokenizer function.""" normalizers = self._normalizers.copy() validators = self._validators.copy() metadata = self._metadata.copy() def tokenize(value: T) -> Token: # Validate for validator in validators: if not validator(value): raise ValueError(f"Validation failed for {value}") # Convert to string str_value = str(value) # Normalize for normalizer in normalizers: str_value = normalizer(str_value) return Token(str_value, TokenType.STRING, metadata) return tokenize