Instructions to use ArthaLabs/panini-tokenizer with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use ArthaLabs/panini-tokenizer with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("ArthaLabs/panini-tokenizer", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Upload folder using huggingface_hub
Browse files- app.py +1 -1
- src/__init__.py +5 -3
- src/sandhi_engine.py +107 -0
- src/splitter.py +155 -5
app.py
CHANGED
|
@@ -98,7 +98,7 @@ def tokenize_with_panini(text: str) -> list:
|
|
| 98 |
|
| 99 |
for i, word in enumerate(words):
|
| 100 |
prefix = "▁" if i == 0 else ""
|
| 101 |
-
split_result = PANINI_SPLITTER.
|
| 102 |
|
| 103 |
if split_result.is_compound and len(split_result.components) > 1:
|
| 104 |
for j, comp in enumerate(split_result.components):
|
|
|
|
| 98 |
|
| 99 |
for i, word in enumerate(words):
|
| 100 |
prefix = "▁" if i == 0 else ""
|
| 101 |
+
split_result = PANINI_SPLITTER.split_v4(word) # V1.5: Uses sandhi expansion
|
| 102 |
|
| 103 |
if split_result.is_compound and len(split_result.components) > 1:
|
| 104 |
for j, comp in enumerate(split_result.components):
|
src/__init__.py
CHANGED
|
@@ -1,10 +1,11 @@
|
|
| 1 |
"""
|
| 2 |
-
Panini Tokenizer
|
| 3 |
-
Morphology-aware Sanskrit tokenizer
|
| 4 |
"""
|
| 5 |
|
| 6 |
from .analyzer import VidyutAnalyzer, MorphParse
|
| 7 |
from .splitter import SamasaSplitter, CompoundSplit
|
|
|
|
| 8 |
from .tokenizer import PaniniTokenizerV3, create_tokenizer
|
| 9 |
|
| 10 |
__all__ = [
|
|
@@ -12,8 +13,9 @@ __all__ = [
|
|
| 12 |
"MorphParse",
|
| 13 |
"SamasaSplitter",
|
| 14 |
"CompoundSplit",
|
|
|
|
| 15 |
"PaniniTokenizerV3",
|
| 16 |
"create_tokenizer",
|
| 17 |
]
|
| 18 |
|
| 19 |
-
__version__ = "
|
|
|
|
| 1 |
"""
|
| 2 |
+
Panini Tokenizer
|
| 3 |
+
Morphology-aware Sanskrit tokenizer with Sandhi Expansion.
|
| 4 |
"""
|
| 5 |
|
| 6 |
from .analyzer import VidyutAnalyzer, MorphParse
|
| 7 |
from .splitter import SamasaSplitter, CompoundSplit
|
| 8 |
+
from .sandhi_engine import SandhiEngine
|
| 9 |
from .tokenizer import PaniniTokenizerV3, create_tokenizer
|
| 10 |
|
| 11 |
__all__ = [
|
|
|
|
| 13 |
"MorphParse",
|
| 14 |
"SamasaSplitter",
|
| 15 |
"CompoundSplit",
|
| 16 |
+
"SandhiEngine",
|
| 17 |
"PaniniTokenizerV3",
|
| 18 |
"create_tokenizer",
|
| 19 |
]
|
| 20 |
|
| 21 |
+
__version__ = "1.5.0"
|
src/sandhi_engine.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Sandhi Engine for Panini Tokenizer V4
|
| 3 |
+
Generates pre-sandhi hypotheses for Sanskrit compound splitting.
|
| 4 |
+
Handles vowel coalescence (ac-sandhi) and visarga/consonant assimilation.
|
| 5 |
+
|
| 6 |
+
Uses table-driven design for maintainability.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from typing import List, Tuple, Generator
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
class SandhiEngine:
|
| 13 |
+
"""
|
| 14 |
+
Generates pre-sandhi hypotheses for Sanskrit compound splitting.
|
| 15 |
+
Handles vowel coalescence (ac-sandhi) and visarga/consonant assimilation.
|
| 16 |
+
"""
|
| 17 |
+
|
| 18 |
+
def __init__(self):
|
| 19 |
+
# ac-sandhi (vowel merger) tables
|
| 20 |
+
# Key = surface char, Value = list of (left_end, right_start) pairs
|
| 21 |
+
self.VOWEL_SPLITS = {
|
| 22 |
+
# Guṇa
|
| 23 |
+
'e': [('a', 'i'), ('A', 'i'), ('a', 'I'), ('A', 'I')],
|
| 24 |
+
'o': [('a', 'u'), ('A', 'u'), ('a', 'U'), ('A', 'U')],
|
| 25 |
+
'ar': [('a', 'f'), ('A', 'f'), ('a', 'F'), ('A', 'F')], # maharzi -> mahA + fzi
|
| 26 |
+
|
| 27 |
+
# Vṛddhi
|
| 28 |
+
'E': [('a', 'e'), ('A', 'e'), ('a', 'E'), ('A', 'E')], # ai
|
| 29 |
+
'O': [('a', 'o'), ('A', 'o'), ('a', 'O'), ('A', 'O')], # au
|
| 30 |
+
|
| 31 |
+
# Dīrgha (savarṇa dīrgha) - critical for long vowel restoration
|
| 32 |
+
'A': [('a', 'a'), ('a', 'A'), ('A', 'a'), ('A', 'A')],
|
| 33 |
+
'I': [('i', 'i'), ('i', 'I'), ('I', 'i'), ('I', 'I')],
|
| 34 |
+
'U': [('u', 'u'), ('u', 'U'), ('U', 'u'), ('U', 'U')],
|
| 35 |
+
}
|
| 36 |
+
|
| 37 |
+
# Consonant categories
|
| 38 |
+
self.VOICED = set(['g', 'G', 'j', 'J', 'd', 'D', 'b', 'B', 'n', 'N', 'm', 'y', 'r', 'l', 'v', 'h'])
|
| 39 |
+
self.HARD = set(['k', 'K', 'c', 'C', 't', 'T', 'w', 'W', 'p', 'P', 'S', 's'])
|
| 40 |
+
|
| 41 |
+
def generate_splits(self, word: str, i: int) -> Generator[Tuple[str, str], None, None]:
|
| 42 |
+
"""
|
| 43 |
+
Yields (left, right) tuples for a split AT index i.
|
| 44 |
+
i is the index of the character being considered as the 'pivot'.
|
| 45 |
+
"""
|
| 46 |
+
if i < 1 or i >= len(word):
|
| 47 |
+
return
|
| 48 |
+
|
| 49 |
+
char = word[i]
|
| 50 |
+
|
| 51 |
+
# 1. Default: hard cut (no sandhi)
|
| 52 |
+
# Split BEFORE char: word[:i] | word[i:]
|
| 53 |
+
yield (word[:i], word[i:])
|
| 54 |
+
|
| 55 |
+
# 2. Vowel coalescence (the char IS the result of merger)
|
| 56 |
+
# e.g. gaṇ[e]śa -> left ends with 'a', right starts with 'i'
|
| 57 |
+
if char in self.VOWEL_SPLITS:
|
| 58 |
+
for left_end, right_start in self.VOWEL_SPLITS[char]:
|
| 59 |
+
# Replace char at i with the split pair
|
| 60 |
+
yield (word[:i] + left_end, right_start + word[i+1:])
|
| 61 |
+
|
| 62 |
+
# 3. Yān sandhi (y -> i/I, v -> u/U)
|
| 63 |
+
# e.g. praty[e]kam -> prati + ekam
|
| 64 |
+
# CAUTION: Yān happens BEFORE a vowel, check word[i+1]
|
| 65 |
+
if i + 1 < len(word):
|
| 66 |
+
next_char = word[i+1]
|
| 67 |
+
if char == 'y': # y -> i/I
|
| 68 |
+
for v in ['i', 'I']:
|
| 69 |
+
yield (word[:i] + v, word[i+1:])
|
| 70 |
+
elif char == 'v': # v -> u/U
|
| 71 |
+
for v in ['u', 'U']:
|
| 72 |
+
yield (word[:i] + v, word[i+1:])
|
| 73 |
+
|
| 74 |
+
# 4. Visarga sandhi restoration
|
| 75 |
+
# 'o' before voiced consonant -> 'aH'
|
| 76 |
+
if char == 'o' and i + 1 < len(word):
|
| 77 |
+
if word[i+1] in self.VOICED:
|
| 78 |
+
yield (word[:i] + "aH", word[i+1:])
|
| 79 |
+
|
| 80 |
+
# 'r' before voiced -> 'H' (punarjanma -> punaH + janma)
|
| 81 |
+
if char == 'r' and i + 1 < len(word):
|
| 82 |
+
if word[i+1] in self.VOICED:
|
| 83 |
+
yield (word[:i] + "H", word[i+1:])
|
| 84 |
+
|
| 85 |
+
# 's'/'S' before hard consonant -> 'H'
|
| 86 |
+
if char in ['s', 'S'] and i + 1 < len(word):
|
| 87 |
+
if word[i+1] in self.HARD:
|
| 88 |
+
yield (word[:i] + "H", word[i+1:])
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
# --- TEST ---
|
| 92 |
+
if __name__ == "__main__":
|
| 93 |
+
engine = SandhiEngine()
|
| 94 |
+
|
| 95 |
+
print("Testing SandhiEngine...")
|
| 96 |
+
|
| 97 |
+
test_cases = [
|
| 98 |
+
("gaReSa", 3), # e: should yield gaRa + iSa
|
| 99 |
+
("devendra", 3), # e: should yield deva + indra
|
| 100 |
+
("rAmo", 3), # o: should yield rAmaH before voiced
|
| 101 |
+
("punarjanma", 4), # r: should yield punaH + janma
|
| 102 |
+
]
|
| 103 |
+
|
| 104 |
+
for word, pos in test_cases:
|
| 105 |
+
print(f"\n {word} at pos {pos}:")
|
| 106 |
+
for left, right in engine.generate_splits(word, pos):
|
| 107 |
+
print(f" {left} | {right}")
|
src/splitter.py
CHANGED
|
@@ -6,11 +6,9 @@ Detects and splits Sanskrit compound words at their boundaries.
|
|
| 6 |
from typing import List, Tuple, Optional
|
| 7 |
from dataclasses import dataclass
|
| 8 |
|
| 9 |
-
# Import analyzer for Kosha access
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
except ImportError:
|
| 13 |
-
from analyzer import VidyutAnalyzer, MorphParse
|
| 14 |
|
| 15 |
|
| 16 |
@dataclass
|
|
@@ -57,6 +55,7 @@ class SamasaSplitter:
|
|
| 57 |
def __init__(self, analyzer: Optional[VidyutAnalyzer] = None):
|
| 58 |
"""Initialize with optional shared analyzer."""
|
| 59 |
self.analyzer = analyzer or VidyutAnalyzer(preload_cache=False)
|
|
|
|
| 60 |
|
| 61 |
# Sandhi reversal rules: (surface_ending, possible_original_endings)
|
| 62 |
# These are common consonant/vowel Sandhi transformations to reverse
|
|
@@ -154,6 +153,27 @@ class SamasaSplitter:
|
|
| 154 |
if candidate.endswith('U') and self.analyzer._in_kosha(candidate[:-1] + 'u'):
|
| 155 |
return True
|
| 156 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 157 |
# Try PRATYAYA STRIPPING (grammatical suffix removal)
|
| 158 |
# This is Panini's kRt/taddhita system - generalizes to ALL Sanskrit
|
| 159 |
PRATYAYAS = [
|
|
@@ -167,6 +187,12 @@ class SamasaSplitter:
|
|
| 167 |
('in', 2), # ṇini: possessor
|
| 168 |
('ika', 3), # ṭhak: related to
|
| 169 |
('Iya', 3), # cha: related to
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 170 |
]
|
| 171 |
|
| 172 |
for suffix, min_root in PRATYAYAS:
|
|
@@ -175,6 +201,13 @@ class SamasaSplitter:
|
|
| 175 |
# Try the root in Kosha
|
| 176 |
if self.analyzer._in_kosha(root):
|
| 177 |
return True
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 178 |
# Try Sandhi reversal on root
|
| 179 |
for r in self._try_sandhi_reversal(root):
|
| 180 |
if self.analyzer._in_kosha(r):
|
|
@@ -698,6 +731,123 @@ class SamasaSplitter:
|
|
| 698 |
compound_type=None # We don't classify samāsa types
|
| 699 |
)
|
| 700 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 701 |
def split_multiple(self, words: List[str]) -> List[CompoundSplit]:
|
| 702 |
"""Split multiple words."""
|
| 703 |
return [self.split(w) for w in words]
|
|
|
|
| 6 |
from typing import List, Tuple, Optional
|
| 7 |
from dataclasses import dataclass
|
| 8 |
|
| 9 |
+
# Import analyzer for Kosha access
|
| 10 |
+
from .analyzer import VidyutAnalyzer, MorphParse
|
| 11 |
+
from .sandhi_engine import SandhiEngine
|
|
|
|
|
|
|
| 12 |
|
| 13 |
|
| 14 |
@dataclass
|
|
|
|
| 55 |
def __init__(self, analyzer: Optional[VidyutAnalyzer] = None):
|
| 56 |
"""Initialize with optional shared analyzer."""
|
| 57 |
self.analyzer = analyzer or VidyutAnalyzer(preload_cache=False)
|
| 58 |
+
self.sandhi_engine = SandhiEngine() # V4: Generative sandhi expansion
|
| 59 |
|
| 60 |
# Sandhi reversal rules: (surface_ending, possible_original_endings)
|
| 61 |
# These are common consonant/vowel Sandhi transformations to reverse
|
|
|
|
| 153 |
if candidate.endswith('U') and self.analyzer._in_kosha(candidate[:-1] + 'u'):
|
| 154 |
return True
|
| 155 |
|
| 156 |
+
# Try VISARGA STRIPPING (vAlmIkiH → vAlmIki)
|
| 157 |
+
if surface.endswith('H') and len(surface) > 2:
|
| 158 |
+
base = surface[:-1]
|
| 159 |
+
if self.analyzer._in_kosha(base):
|
| 160 |
+
return True
|
| 161 |
+
|
| 162 |
+
# Try VIBHAKTI STRIPPING (nominal case endings)
|
| 163 |
+
VIBHAKTI_ENDINGS = [
|
| 164 |
+
'am', 'aH', 'ena', 'Aya', 'At', 'asya', 'e', 'AH', # Masculine a-stem
|
| 165 |
+
'An', 'EH', 'eBya', 'AnAm', 'ezu', # Masculine a-stem plural
|
| 166 |
+
'au', 'OH', 'AvyAm', # Dual
|
| 167 |
+
]
|
| 168 |
+
for ending in sorted(VIBHAKTI_ENDINGS, key=len, reverse=True):
|
| 169 |
+
if surface.endswith(ending) and len(surface) > len(ending) + 2:
|
| 170 |
+
stem = surface[:-len(ending)]
|
| 171 |
+
if self.analyzer._in_kosha(stem):
|
| 172 |
+
return True
|
| 173 |
+
# Try with 'a' restoration (munipuMgavam → munipuMgava)
|
| 174 |
+
if self.analyzer._in_kosha(stem + 'a'):
|
| 175 |
+
return True
|
| 176 |
+
|
| 177 |
# Try PRATYAYA STRIPPING (grammatical suffix removal)
|
| 178 |
# This is Panini's kRt/taddhita system - generalizes to ALL Sanskrit
|
| 179 |
PRATYAYAS = [
|
|
|
|
| 187 |
('in', 2), # ṇini: possessor
|
| 188 |
('ika', 3), # ṭhak: related to
|
| 189 |
('Iya', 3), # cha: related to
|
| 190 |
+
# Feminine/agent kṛdanta suffixes (Fix 2)
|
| 191 |
+
('iRi', 3), # iṇī: feminine agent (ākarṣiṇī)
|
| 192 |
+
('iRI', 3), # iṇī: alt spelling
|
| 193 |
+
('inI', 3), # inī: feminine possessor (yoginī)
|
| 194 |
+
('ikA', 3), # ikā: feminine derivative (nāyikā)
|
| 195 |
+
('trI', 3), # trī: feminine agent (kartrī)
|
| 196 |
]
|
| 197 |
|
| 198 |
for suffix, min_root in PRATYAYAS:
|
|
|
|
| 201 |
# Try the root in Kosha
|
| 202 |
if self.analyzer._in_kosha(root):
|
| 203 |
return True
|
| 204 |
+
# Try with guṇa 'a' restoration
|
| 205 |
+
if self.analyzer._in_kosha(root + 'a'):
|
| 206 |
+
return True
|
| 207 |
+
# Try R→f transliteration (MW uses f for ṛ: kartRI → kartf)
|
| 208 |
+
root_f = root.replace('R', 'f')
|
| 209 |
+
if root_f != root and self.analyzer._in_kosha(root_f):
|
| 210 |
+
return True
|
| 211 |
# Try Sandhi reversal on root
|
| 212 |
for r in self._try_sandhi_reversal(root):
|
| 213 |
if self.analyzer._in_kosha(r):
|
|
|
|
| 731 |
compound_type=None # We don't classify samāsa types
|
| 732 |
)
|
| 733 |
|
| 734 |
+
def _split_dp(self, word: str, memo: dict = None) -> List[List[str]]:
|
| 735 |
+
"""
|
| 736 |
+
V4 Algorithm: Memoized Dynamic Programming with Sandhi Expansion.
|
| 737 |
+
|
| 738 |
+
Returns all valid splits, cached by suffix.
|
| 739 |
+
Handles coalescent sandhi (e=a+i, o=a+u, etc.) that V3 misses.
|
| 740 |
+
"""
|
| 741 |
+
if memo is None:
|
| 742 |
+
memo = {}
|
| 743 |
+
|
| 744 |
+
if word in memo:
|
| 745 |
+
return memo[word]
|
| 746 |
+
|
| 747 |
+
# Base: too short to split
|
| 748 |
+
if len(word) <= 2:
|
| 749 |
+
if self._is_valid_stem(word):
|
| 750 |
+
return [[word]]
|
| 751 |
+
return []
|
| 752 |
+
|
| 753 |
+
valid_splits = []
|
| 754 |
+
|
| 755 |
+
# 1. OPTION A: The whole word is a stem (Lexicalized)
|
| 756 |
+
if self._is_valid_stem(word):
|
| 757 |
+
valid_splits.append([word])
|
| 758 |
+
# DO NOT RETURN EARLY. Keep looking for splits!
|
| 759 |
+
|
| 760 |
+
# 2. OPTION B: Split it (Generative Sandhi)
|
| 761 |
+
# Try each split position with sandhi expansion
|
| 762 |
+
for i in range(2, len(word) - 1):
|
| 763 |
+
for left, right in self.sandhi_engine.generate_splits(word, i):
|
| 764 |
+
if len(left) < 2 or len(right) < 2:
|
| 765 |
+
continue
|
| 766 |
+
|
| 767 |
+
if self._is_valid_stem(left):
|
| 768 |
+
# Recurse on right (memoized!)
|
| 769 |
+
right_splits = self._split_dp(right, memo)
|
| 770 |
+
for rs in right_splits:
|
| 771 |
+
valid_splits.append([left] + rs)
|
| 772 |
+
|
| 773 |
+
memo[word] = valid_splits
|
| 774 |
+
return valid_splits
|
| 775 |
+
|
| 776 |
+
def split_v4(self, word: str) -> CompoundSplit:
|
| 777 |
+
"""
|
| 778 |
+
V4 Split: Uses generative sandhi expansion for coalescent sandhi.
|
| 779 |
+
|
| 780 |
+
Handles:
|
| 781 |
+
- Vowel coalescence: gaṇeśa → gaṇa + īśa (e = a+i)
|
| 782 |
+
- Visarga sandhi: punarjanma → punaH + janma
|
| 783 |
+
- Vṛddhi: tavaiva → tava + eva
|
| 784 |
+
"""
|
| 785 |
+
if len(word) < 4:
|
| 786 |
+
return CompoundSplit(
|
| 787 |
+
surface=word, components=[word],
|
| 788 |
+
split_points=[], is_compound=False, compound_type=None
|
| 789 |
+
)
|
| 790 |
+
|
| 791 |
+
# Use V4 DP algorithm
|
| 792 |
+
all_splits = self._split_dp(word)
|
| 793 |
+
|
| 794 |
+
if not all_splits:
|
| 795 |
+
return CompoundSplit(
|
| 796 |
+
surface=word, components=[word],
|
| 797 |
+
split_points=[], is_compound=False, compound_type=None
|
| 798 |
+
)
|
| 799 |
+
|
| 800 |
+
# SCORING STRATEGY:
|
| 801 |
+
# Balance: prefer splits, but penalize over-fragmentation.
|
| 802 |
+
# 1. Penalize short components (< 3 chars) heavily
|
| 803 |
+
# 2. Prefer 2-component splits over 3+ components
|
| 804 |
+
# 3. Single long tokens get moderate penalty
|
| 805 |
+
# 4. Bonus for components directly in kosha (prefer mahA+rAja over maha+arAja)
|
| 806 |
+
def score_split(components):
|
| 807 |
+
base_score = sum(len(c)**2 for c in components)
|
| 808 |
+
|
| 809 |
+
# Penalize short components (garbage like 'ma', 'at')
|
| 810 |
+
short_penalty = sum(10 for c in components if len(c) < 3)
|
| 811 |
+
base_score -= short_penalty * 5
|
| 812 |
+
|
| 813 |
+
# Bonus for 2-component splits (optimal granularity)
|
| 814 |
+
if len(components) == 2:
|
| 815 |
+
base_score += 20
|
| 816 |
+
|
| 817 |
+
# Penalty for single long tokens (prefer analysis)
|
| 818 |
+
if len(components) == 1 and len(components[0]) > 6:
|
| 819 |
+
base_score -= 15
|
| 820 |
+
|
| 821 |
+
# Bonus for components directly in kosha (prefer clean stems)
|
| 822 |
+
kosha_bonus = sum(25 for c in components if self.analyzer._in_kosha(c))
|
| 823 |
+
base_score += kosha_bonus
|
| 824 |
+
|
| 825 |
+
# Prefer balanced splits (similar length components)
|
| 826 |
+
if len(components) == 2:
|
| 827 |
+
len_diff = abs(len(components[0]) - len(components[1]))
|
| 828 |
+
if len_diff <= 1:
|
| 829 |
+
base_score += 10 # Bonus for balanced split
|
| 830 |
+
|
| 831 |
+
# Penalize splits with expanded length (sandhi artifacts add characters)
|
| 832 |
+
total_len = sum(len(c) for c in components)
|
| 833 |
+
if total_len > len(word):
|
| 834 |
+
base_score -= (total_len - len(word)) * 10 # Penalty per extra char
|
| 835 |
+
|
| 836 |
+
return base_score
|
| 837 |
+
|
| 838 |
+
best_split = max(all_splits, key=score_split)
|
| 839 |
+
|
| 840 |
+
if len(best_split) <= 1:
|
| 841 |
+
return CompoundSplit(
|
| 842 |
+
surface=word, components=[word],
|
| 843 |
+
split_points=[], is_compound=False, compound_type=None
|
| 844 |
+
)
|
| 845 |
+
|
| 846 |
+
return CompoundSplit(
|
| 847 |
+
surface=word, components=best_split,
|
| 848 |
+
split_points=[], is_compound=True, compound_type=None
|
| 849 |
+
)
|
| 850 |
+
|
| 851 |
def split_multiple(self, words: List[str]) -> List[CompoundSplit]:
|
| 852 |
"""Split multiple words."""
|
| 853 |
return [self.split(w) for w in words]
|