ArthaLabs commited on
Commit
77111fb
·
verified ·
1 Parent(s): 9166de1

Upload folder using huggingface_hub

Browse files
Files changed (4) hide show
  1. app.py +1 -1
  2. src/__init__.py +5 -3
  3. src/sandhi_engine.py +107 -0
  4. src/splitter.py +155 -5
app.py CHANGED
@@ -98,7 +98,7 @@ def tokenize_with_panini(text: str) -> list:
98
 
99
  for i, word in enumerate(words):
100
  prefix = "▁" if i == 0 else ""
101
- split_result = PANINI_SPLITTER.split(word)
102
 
103
  if split_result.is_compound and len(split_result.components) > 1:
104
  for j, comp in enumerate(split_result.components):
 
98
 
99
  for i, word in enumerate(words):
100
  prefix = "▁" if i == 0 else ""
101
+ split_result = PANINI_SPLITTER.split_v4(word) # V1.5: Uses sandhi expansion
102
 
103
  if split_result.is_compound and len(split_result.components) > 1:
104
  for j, comp in enumerate(split_result.components):
src/__init__.py CHANGED
@@ -1,10 +1,11 @@
1
  """
2
- Panini Tokenizer V3
3
- Morphology-aware Sanskrit tokenizer using Vidyut.
4
  """
5
 
6
  from .analyzer import VidyutAnalyzer, MorphParse
7
  from .splitter import SamasaSplitter, CompoundSplit
 
8
  from .tokenizer import PaniniTokenizerV3, create_tokenizer
9
 
10
  __all__ = [
@@ -12,8 +13,9 @@ __all__ = [
12
  "MorphParse",
13
  "SamasaSplitter",
14
  "CompoundSplit",
 
15
  "PaniniTokenizerV3",
16
  "create_tokenizer",
17
  ]
18
 
19
- __version__ = "3.0.0"
 
1
  """
2
+ Panini Tokenizer
3
+ Morphology-aware Sanskrit tokenizer with Sandhi Expansion.
4
  """
5
 
6
  from .analyzer import VidyutAnalyzer, MorphParse
7
  from .splitter import SamasaSplitter, CompoundSplit
8
+ from .sandhi_engine import SandhiEngine
9
  from .tokenizer import PaniniTokenizerV3, create_tokenizer
10
 
11
  __all__ = [
 
13
  "MorphParse",
14
  "SamasaSplitter",
15
  "CompoundSplit",
16
+ "SandhiEngine",
17
  "PaniniTokenizerV3",
18
  "create_tokenizer",
19
  ]
20
 
21
+ __version__ = "1.5.0"
src/sandhi_engine.py ADDED
@@ -0,0 +1,107 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ Sandhi Engine for Panini Tokenizer V4
3
+ Generates pre-sandhi hypotheses for Sanskrit compound splitting.
4
+ Handles vowel coalescence (ac-sandhi) and visarga/consonant assimilation.
5
+
6
+ Uses table-driven design for maintainability.
7
+ """
8
+
9
+ from typing import List, Tuple, Generator
10
+
11
+
12
+ class SandhiEngine:
13
+ """
14
+ Generates pre-sandhi hypotheses for Sanskrit compound splitting.
15
+ Handles vowel coalescence (ac-sandhi) and visarga/consonant assimilation.
16
+ """
17
+
18
+ def __init__(self):
19
+ # ac-sandhi (vowel merger) tables
20
+ # Key = surface char, Value = list of (left_end, right_start) pairs
21
+ self.VOWEL_SPLITS = {
22
+ # Guṇa
23
+ 'e': [('a', 'i'), ('A', 'i'), ('a', 'I'), ('A', 'I')],
24
+ 'o': [('a', 'u'), ('A', 'u'), ('a', 'U'), ('A', 'U')],
25
+ 'ar': [('a', 'f'), ('A', 'f'), ('a', 'F'), ('A', 'F')], # maharzi -> mahA + fzi
26
+
27
+ # Vṛddhi
28
+ 'E': [('a', 'e'), ('A', 'e'), ('a', 'E'), ('A', 'E')], # ai
29
+ 'O': [('a', 'o'), ('A', 'o'), ('a', 'O'), ('A', 'O')], # au
30
+
31
+ # Dīrgha (savarṇa dīrgha) - critical for long vowel restoration
32
+ 'A': [('a', 'a'), ('a', 'A'), ('A', 'a'), ('A', 'A')],
33
+ 'I': [('i', 'i'), ('i', 'I'), ('I', 'i'), ('I', 'I')],
34
+ 'U': [('u', 'u'), ('u', 'U'), ('U', 'u'), ('U', 'U')],
35
+ }
36
+
37
+ # Consonant categories
38
+ self.VOICED = set(['g', 'G', 'j', 'J', 'd', 'D', 'b', 'B', 'n', 'N', 'm', 'y', 'r', 'l', 'v', 'h'])
39
+ self.HARD = set(['k', 'K', 'c', 'C', 't', 'T', 'w', 'W', 'p', 'P', 'S', 's'])
40
+
41
+ def generate_splits(self, word: str, i: int) -> Generator[Tuple[str, str], None, None]:
42
+ """
43
+ Yields (left, right) tuples for a split AT index i.
44
+ i is the index of the character being considered as the 'pivot'.
45
+ """
46
+ if i < 1 or i >= len(word):
47
+ return
48
+
49
+ char = word[i]
50
+
51
+ # 1. Default: hard cut (no sandhi)
52
+ # Split BEFORE char: word[:i] | word[i:]
53
+ yield (word[:i], word[i:])
54
+
55
+ # 2. Vowel coalescence (the char IS the result of merger)
56
+ # e.g. gaṇ[e]śa -> left ends with 'a', right starts with 'i'
57
+ if char in self.VOWEL_SPLITS:
58
+ for left_end, right_start in self.VOWEL_SPLITS[char]:
59
+ # Replace char at i with the split pair
60
+ yield (word[:i] + left_end, right_start + word[i+1:])
61
+
62
+ # 3. Yān sandhi (y -> i/I, v -> u/U)
63
+ # e.g. praty[e]kam -> prati + ekam
64
+ # CAUTION: Yān happens BEFORE a vowel, check word[i+1]
65
+ if i + 1 < len(word):
66
+ next_char = word[i+1]
67
+ if char == 'y': # y -> i/I
68
+ for v in ['i', 'I']:
69
+ yield (word[:i] + v, word[i+1:])
70
+ elif char == 'v': # v -> u/U
71
+ for v in ['u', 'U']:
72
+ yield (word[:i] + v, word[i+1:])
73
+
74
+ # 4. Visarga sandhi restoration
75
+ # 'o' before voiced consonant -> 'aH'
76
+ if char == 'o' and i + 1 < len(word):
77
+ if word[i+1] in self.VOICED:
78
+ yield (word[:i] + "aH", word[i+1:])
79
+
80
+ # 'r' before voiced -> 'H' (punarjanma -> punaH + janma)
81
+ if char == 'r' and i + 1 < len(word):
82
+ if word[i+1] in self.VOICED:
83
+ yield (word[:i] + "H", word[i+1:])
84
+
85
+ # 's'/'S' before hard consonant -> 'H'
86
+ if char in ['s', 'S'] and i + 1 < len(word):
87
+ if word[i+1] in self.HARD:
88
+ yield (word[:i] + "H", word[i+1:])
89
+
90
+
91
+ # --- TEST ---
92
+ if __name__ == "__main__":
93
+ engine = SandhiEngine()
94
+
95
+ print("Testing SandhiEngine...")
96
+
97
+ test_cases = [
98
+ ("gaReSa", 3), # e: should yield gaRa + iSa
99
+ ("devendra", 3), # e: should yield deva + indra
100
+ ("rAmo", 3), # o: should yield rAmaH before voiced
101
+ ("punarjanma", 4), # r: should yield punaH + janma
102
+ ]
103
+
104
+ for word, pos in test_cases:
105
+ print(f"\n {word} at pos {pos}:")
106
+ for left, right in engine.generate_splits(word, pos):
107
+ print(f" {left} | {right}")
src/splitter.py CHANGED
@@ -6,11 +6,9 @@ Detects and splits Sanskrit compound words at their boundaries.
6
  from typing import List, Tuple, Optional
7
  from dataclasses import dataclass
8
 
9
- # Import analyzer for Kosha access (use absolute import for standalone execution)
10
- try:
11
- from .analyzer import VidyutAnalyzer, MorphParse
12
- except ImportError:
13
- from analyzer import VidyutAnalyzer, MorphParse
14
 
15
 
16
  @dataclass
@@ -57,6 +55,7 @@ class SamasaSplitter:
57
  def __init__(self, analyzer: Optional[VidyutAnalyzer] = None):
58
  """Initialize with optional shared analyzer."""
59
  self.analyzer = analyzer or VidyutAnalyzer(preload_cache=False)
 
60
 
61
  # Sandhi reversal rules: (surface_ending, possible_original_endings)
62
  # These are common consonant/vowel Sandhi transformations to reverse
@@ -154,6 +153,27 @@ class SamasaSplitter:
154
  if candidate.endswith('U') and self.analyzer._in_kosha(candidate[:-1] + 'u'):
155
  return True
156
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
157
  # Try PRATYAYA STRIPPING (grammatical suffix removal)
158
  # This is Panini's kRt/taddhita system - generalizes to ALL Sanskrit
159
  PRATYAYAS = [
@@ -167,6 +187,12 @@ class SamasaSplitter:
167
  ('in', 2), # ṇini: possessor
168
  ('ika', 3), # ṭhak: related to
169
  ('Iya', 3), # cha: related to
 
 
 
 
 
 
170
  ]
171
 
172
  for suffix, min_root in PRATYAYAS:
@@ -175,6 +201,13 @@ class SamasaSplitter:
175
  # Try the root in Kosha
176
  if self.analyzer._in_kosha(root):
177
  return True
 
 
 
 
 
 
 
178
  # Try Sandhi reversal on root
179
  for r in self._try_sandhi_reversal(root):
180
  if self.analyzer._in_kosha(r):
@@ -698,6 +731,123 @@ class SamasaSplitter:
698
  compound_type=None # We don't classify samāsa types
699
  )
700
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
701
  def split_multiple(self, words: List[str]) -> List[CompoundSplit]:
702
  """Split multiple words."""
703
  return [self.split(w) for w in words]
 
6
  from typing import List, Tuple, Optional
7
  from dataclasses import dataclass
8
 
9
+ # Import analyzer for Kosha access
10
+ from .analyzer import VidyutAnalyzer, MorphParse
11
+ from .sandhi_engine import SandhiEngine
 
 
12
 
13
 
14
  @dataclass
 
55
  def __init__(self, analyzer: Optional[VidyutAnalyzer] = None):
56
  """Initialize with optional shared analyzer."""
57
  self.analyzer = analyzer or VidyutAnalyzer(preload_cache=False)
58
+ self.sandhi_engine = SandhiEngine() # V4: Generative sandhi expansion
59
 
60
  # Sandhi reversal rules: (surface_ending, possible_original_endings)
61
  # These are common consonant/vowel Sandhi transformations to reverse
 
153
  if candidate.endswith('U') and self.analyzer._in_kosha(candidate[:-1] + 'u'):
154
  return True
155
 
156
+ # Try VISARGA STRIPPING (vAlmIkiH → vAlmIki)
157
+ if surface.endswith('H') and len(surface) > 2:
158
+ base = surface[:-1]
159
+ if self.analyzer._in_kosha(base):
160
+ return True
161
+
162
+ # Try VIBHAKTI STRIPPING (nominal case endings)
163
+ VIBHAKTI_ENDINGS = [
164
+ 'am', 'aH', 'ena', 'Aya', 'At', 'asya', 'e', 'AH', # Masculine a-stem
165
+ 'An', 'EH', 'eBya', 'AnAm', 'ezu', # Masculine a-stem plural
166
+ 'au', 'OH', 'AvyAm', # Dual
167
+ ]
168
+ for ending in sorted(VIBHAKTI_ENDINGS, key=len, reverse=True):
169
+ if surface.endswith(ending) and len(surface) > len(ending) + 2:
170
+ stem = surface[:-len(ending)]
171
+ if self.analyzer._in_kosha(stem):
172
+ return True
173
+ # Try with 'a' restoration (munipuMgavam → munipuMgava)
174
+ if self.analyzer._in_kosha(stem + 'a'):
175
+ return True
176
+
177
  # Try PRATYAYA STRIPPING (grammatical suffix removal)
178
  # This is Panini's kRt/taddhita system - generalizes to ALL Sanskrit
179
  PRATYAYAS = [
 
187
  ('in', 2), # ṇini: possessor
188
  ('ika', 3), # ṭhak: related to
189
  ('Iya', 3), # cha: related to
190
+ # Feminine/agent kṛdanta suffixes (Fix 2)
191
+ ('iRi', 3), # iṇī: feminine agent (ākarṣiṇī)
192
+ ('iRI', 3), # iṇī: alt spelling
193
+ ('inI', 3), # inī: feminine possessor (yoginī)
194
+ ('ikA', 3), # ikā: feminine derivative (nāyikā)
195
+ ('trI', 3), # trī: feminine agent (kartrī)
196
  ]
197
 
198
  for suffix, min_root in PRATYAYAS:
 
201
  # Try the root in Kosha
202
  if self.analyzer._in_kosha(root):
203
  return True
204
+ # Try with guṇa 'a' restoration
205
+ if self.analyzer._in_kosha(root + 'a'):
206
+ return True
207
+ # Try R→f transliteration (MW uses f for ṛ: kartRI → kartf)
208
+ root_f = root.replace('R', 'f')
209
+ if root_f != root and self.analyzer._in_kosha(root_f):
210
+ return True
211
  # Try Sandhi reversal on root
212
  for r in self._try_sandhi_reversal(root):
213
  if self.analyzer._in_kosha(r):
 
731
  compound_type=None # We don't classify samāsa types
732
  )
733
 
734
+ def _split_dp(self, word: str, memo: dict = None) -> List[List[str]]:
735
+ """
736
+ V4 Algorithm: Memoized Dynamic Programming with Sandhi Expansion.
737
+
738
+ Returns all valid splits, cached by suffix.
739
+ Handles coalescent sandhi (e=a+i, o=a+u, etc.) that V3 misses.
740
+ """
741
+ if memo is None:
742
+ memo = {}
743
+
744
+ if word in memo:
745
+ return memo[word]
746
+
747
+ # Base: too short to split
748
+ if len(word) <= 2:
749
+ if self._is_valid_stem(word):
750
+ return [[word]]
751
+ return []
752
+
753
+ valid_splits = []
754
+
755
+ # 1. OPTION A: The whole word is a stem (Lexicalized)
756
+ if self._is_valid_stem(word):
757
+ valid_splits.append([word])
758
+ # DO NOT RETURN EARLY. Keep looking for splits!
759
+
760
+ # 2. OPTION B: Split it (Generative Sandhi)
761
+ # Try each split position with sandhi expansion
762
+ for i in range(2, len(word) - 1):
763
+ for left, right in self.sandhi_engine.generate_splits(word, i):
764
+ if len(left) < 2 or len(right) < 2:
765
+ continue
766
+
767
+ if self._is_valid_stem(left):
768
+ # Recurse on right (memoized!)
769
+ right_splits = self._split_dp(right, memo)
770
+ for rs in right_splits:
771
+ valid_splits.append([left] + rs)
772
+
773
+ memo[word] = valid_splits
774
+ return valid_splits
775
+
776
+ def split_v4(self, word: str) -> CompoundSplit:
777
+ """
778
+ V4 Split: Uses generative sandhi expansion for coalescent sandhi.
779
+
780
+ Handles:
781
+ - Vowel coalescence: gaṇeśa → gaṇa + īśa (e = a+i)
782
+ - Visarga sandhi: punarjanma → punaH + janma
783
+ - Vṛddhi: tavaiva → tava + eva
784
+ """
785
+ if len(word) < 4:
786
+ return CompoundSplit(
787
+ surface=word, components=[word],
788
+ split_points=[], is_compound=False, compound_type=None
789
+ )
790
+
791
+ # Use V4 DP algorithm
792
+ all_splits = self._split_dp(word)
793
+
794
+ if not all_splits:
795
+ return CompoundSplit(
796
+ surface=word, components=[word],
797
+ split_points=[], is_compound=False, compound_type=None
798
+ )
799
+
800
+ # SCORING STRATEGY:
801
+ # Balance: prefer splits, but penalize over-fragmentation.
802
+ # 1. Penalize short components (< 3 chars) heavily
803
+ # 2. Prefer 2-component splits over 3+ components
804
+ # 3. Single long tokens get moderate penalty
805
+ # 4. Bonus for components directly in kosha (prefer mahA+rAja over maha+arAja)
806
+ def score_split(components):
807
+ base_score = sum(len(c)**2 for c in components)
808
+
809
+ # Penalize short components (garbage like 'ma', 'at')
810
+ short_penalty = sum(10 for c in components if len(c) < 3)
811
+ base_score -= short_penalty * 5
812
+
813
+ # Bonus for 2-component splits (optimal granularity)
814
+ if len(components) == 2:
815
+ base_score += 20
816
+
817
+ # Penalty for single long tokens (prefer analysis)
818
+ if len(components) == 1 and len(components[0]) > 6:
819
+ base_score -= 15
820
+
821
+ # Bonus for components directly in kosha (prefer clean stems)
822
+ kosha_bonus = sum(25 for c in components if self.analyzer._in_kosha(c))
823
+ base_score += kosha_bonus
824
+
825
+ # Prefer balanced splits (similar length components)
826
+ if len(components) == 2:
827
+ len_diff = abs(len(components[0]) - len(components[1]))
828
+ if len_diff <= 1:
829
+ base_score += 10 # Bonus for balanced split
830
+
831
+ # Penalize splits with expanded length (sandhi artifacts add characters)
832
+ total_len = sum(len(c) for c in components)
833
+ if total_len > len(word):
834
+ base_score -= (total_len - len(word)) * 10 # Penalty per extra char
835
+
836
+ return base_score
837
+
838
+ best_split = max(all_splits, key=score_split)
839
+
840
+ if len(best_split) <= 1:
841
+ return CompoundSplit(
842
+ surface=word, components=[word],
843
+ split_points=[], is_compound=False, compound_type=None
844
+ )
845
+
846
+ return CompoundSplit(
847
+ surface=word, components=best_split,
848
+ split_points=[], is_compound=True, compound_type=None
849
+ )
850
+
851
  def split_multiple(self, words: List[str]) -> List[CompoundSplit]:
852
  """Split multiple words."""
853
  return [self.split(w) for w in words]