--- cdli/cdliSplitter.py	2007/10/24 20:36:06	1.7.2.4
+++ cdli/cdliSplitter.py	2008/01/14 18:43:20	1.7.2.13
@@ -24,12 +24,16 @@ def getSupportedEncoding(encodings):
 ignoreLines=['$','@','#','&','>']
 separators=['']
 # kommas relevant for graphemes will not be deleted
-komma_exception="([^sStThH])," 
+komma_exception="([^sStThH]),"
+komma_exceptionex=re.compile(komma_exception)
 # grapheme boundaries
-graphemeBounds="\{|\}|<|>|\(|\)|-|_|\#|,|\||\]|\[|\!|\?"
+#graphemeBounds="\{|\}|<|>|\(|\)|-|_|\#|,|\||\]|\[|\!|\?"
+graphemeBounds="\{|\}|<|>|-|_|\#|,|\]|\[|\!|\?|\""
+graphemeIgnore="<|>|\#|\||\]|\[|\!|\?\*|;"
 # for words 
-wordBounds="<|>|\(|\)|_|\#|,|\||\]|\[|\!|\?"
-
+#wordBounds="<|>|\(|\)|_|\#|,|\||\]|\[|\!|\?"
+wordBounds="_|,|\""
+wordIgnore="<|>|\#|\||\]|\[|\!|\?\*|;"
            
 class cdliSplitter:
     """base class for splitter. 
@@ -38,6 +42,9 @@ class cdliSplitter:
     
     default_encoding = "utf-8"
     bounds=graphemeBounds
+    boundsex=re.compile(graphemeBounds)
+    ignore=graphemeIgnore
+    ignorex=re.compile(graphemeIgnore)
     indexName="cdliSplitter"
     
     
@@ -66,7 +73,7 @@ class cdliSplitter:
                         
                     elif not (s[0] in ignoreLines):
                         # regular line
-                        lineparts=s.split(".")
+                        lineparts=s.split(". ",1)
                         if len(lineparts)==1: 
                             # no line number
                             txt=s
@@ -76,9 +83,11 @@ class cdliSplitter:
                             lineNum=lineparts[0] 
                             
                         # delete kommata except kommata relevant for graphemes
-                        txt = re.sub(komma_exception,r"\1",txt)
+                        txt = komma_exceptionex.sub(r"\1",txt)
                         # replace word boundaries by spaces
-                        txt = re.sub(self.bounds,' ',txt)
+                        txt = self.boundsex.sub(' ',txt)
+                        # replace letters to be ignored
+                        txt = self.ignorex.sub('',txt)
                         # split words
                         words = txt.split(" ")
                         for w in words:
@@ -86,16 +95,22 @@ class cdliSplitter:
                             if not (w==''):
                                 result.append(w)
 
-        logging.debug("split '%s' into %s"%(lst,repr(result)))
+        #logging.debug("split '%s' into %s"%(lst,repr(result)))
         return result
 
 
 class graphemeSplitter(cdliSplitter):
     bounds=graphemeBounds
+    boundsex=re.compile(graphemeBounds)
+    ignore=graphemeIgnore
+    ignorex=re.compile(graphemeIgnore)
     indexName="graphemeSplitter"
     
 class wordSplitter(cdliSplitter):
     bounds=wordBounds
+    boundsex=re.compile(wordBounds)
+    ignore=wordIgnore
+    ignorex=re.compile(wordIgnore)
     indexName="wordSplitter"
       
 try: