--- cdli/cdliSplitter.py 2008/01/02 15:52:01 1.7.2.10 +++ cdli/cdliSplitter.py 2008/01/21 17:19:01 1.8 @@ -29,11 +29,11 @@ komma_exceptionex=re.compile(komma_excep # grapheme boundaries #graphemeBounds="\{|\}|<|>|\(|\)|-|_|\#|,|\||\]|\[|\!|\?" graphemeBounds="\{|\}|<|>|-|_|\#|,|\]|\[|\!|\?|\"" -graphemeIgnore="<|>|\#|\||\]|\[|\!|\?\*" +graphemeIgnore="<|>|\#|\||\]|\[|\!|\?\*|;" # for words #wordBounds="<|>|\(|\)|_|\#|,|\||\]|\[|\!|\?" wordBounds="_|,|\"" -wordIgnore="<|>|\#|\||\]|\[|\!|\?\*" +wordIgnore="<|>|\#|\||\]|\[|\!|\?\*|;" class cdliSplitter: """base class for splitter. @@ -73,7 +73,7 @@ class cdliSplitter: elif not (s[0] in ignoreLines): # regular line - lineparts=s.split(".") + lineparts=s.split(". ",1) if len(lineparts)==1: # no line number txt=s