utils(string): add support for floating point numbers
This commit is contained in:
@@ -138,6 +138,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|
||||||
|
- Fixed a `match_endofsentence` issue that would result in floating point
|
||||||
|
numbers to be considered an end of sentence.
|
||||||
|
|
||||||
- Fixed a `match_endofsentence` issue that would result in emails to be
|
- Fixed a `match_endofsentence` issue that would result in emails to be
|
||||||
considered an end of sentence.
|
considered an end of sentence.
|
||||||
|
|
||||||
|
|||||||
@@ -8,7 +8,8 @@ import re
|
|||||||
|
|
||||||
ENDOFSENTENCE_PATTERN_STR = r"""
|
ENDOFSENTENCE_PATTERN_STR = r"""
|
||||||
(?<![A-Z]) # Negative lookbehind: not preceded by an uppercase letter (e.g., "U.S.A.")
|
(?<![A-Z]) # Negative lookbehind: not preceded by an uppercase letter (e.g., "U.S.A.")
|
||||||
(?<!\d) # Negative lookbehind: not preceded by a digit (e.g., "1. Let's start")
|
(?<!\d\.\d) # Not preceded by a decimal number (e.g., "3.14159")
|
||||||
|
(?<!^\d\.) # Not preceded by a numbered list item (e.g., "1. Let's start")
|
||||||
(?<!\d\s[ap]) # Negative lookbehind: not preceded by time (e.g., "3:00 a.m.")
|
(?<!\d\s[ap]) # Negative lookbehind: not preceded by time (e.g., "3:00 a.m.")
|
||||||
(?<!Mr|Ms|Dr) # Negative lookbehind: not preceded by Mr, Ms, Dr (combined bc. length is the same)
|
(?<!Mr|Ms|Dr) # Negative lookbehind: not preceded by Mr, Ms, Dr (combined bc. length is the same)
|
||||||
(?<!Mrs) # Negative lookbehind: not preceded by "Mrs"
|
(?<!Mrs) # Negative lookbehind: not preceded by "Mrs"
|
||||||
@@ -22,19 +23,30 @@ ENDOFSENTENCE_PATTERN = re.compile(ENDOFSENTENCE_PATTERN_STR, re.VERBOSE)
|
|||||||
|
|
||||||
EMAIL_PATTERN = re.compile(r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}")
|
EMAIL_PATTERN = re.compile(r"[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}")
|
||||||
|
|
||||||
|
NUMBER_PATTERN = re.compile(r"[+-]?(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?")
|
||||||
|
|
||||||
|
|
||||||
|
def replace_match(text: str, match: re.Match, old: str, new: str) -> str:
|
||||||
|
start = match.start()
|
||||||
|
end = match.end()
|
||||||
|
replacement = text[start:end].replace(old, new)
|
||||||
|
text = text[:start] + replacement + text[end:]
|
||||||
|
return text
|
||||||
|
|
||||||
|
|
||||||
def match_endofsentence(text: str) -> int:
|
def match_endofsentence(text: str) -> int:
|
||||||
text = text.rstrip()
|
text = text.rstrip()
|
||||||
|
|
||||||
# Find all emails.
|
# Replace email dots by ampersands so we can find the end of sentence. For
|
||||||
|
# example, first.last@email.com becomes first&last@email&com.
|
||||||
emails = list(EMAIL_PATTERN.finditer(text))
|
emails = list(EMAIL_PATTERN.finditer(text))
|
||||||
|
|
||||||
# Replace email dots by ampersands so we can find the end of sentence.
|
|
||||||
for email_match in emails:
|
for email_match in emails:
|
||||||
start = email_match.start()
|
text = replace_match(text, email_match, ".", "&")
|
||||||
end = email_match.end()
|
|
||||||
new_email = text[start:end].replace(".", "&")
|
# Replace number dots by ampersands so we can find the end of sentence.
|
||||||
text = text[:start] + new_email + text[end:]
|
numbers = list(NUMBER_PATTERN.finditer(text))
|
||||||
|
for number_match in numbers:
|
||||||
|
text = replace_match(text, number_match, ".", "&")
|
||||||
|
|
||||||
# Match against the new text.
|
# Match against the new text.
|
||||||
match = ENDOFSENTENCE_PATTERN.search(text)
|
match = ENDOFSENTENCE_PATTERN.search(text)
|
||||||
|
|||||||
@@ -21,6 +21,11 @@ class TestUtilsString(unittest.IsolatedAsyncioTestCase):
|
|||||||
assert match_endofsentence("This is for Mr. and Mrs. Jones.") == 31
|
assert match_endofsentence("This is for Mr. and Mrs. Jones.") == 31
|
||||||
assert match_endofsentence("U.S.A and U.S.A..") == 17
|
assert match_endofsentence("U.S.A and U.S.A..") == 17
|
||||||
assert match_endofsentence("My emails are foo@pipecat.ai and bar@pipecat.ai.") == 48
|
assert match_endofsentence("My emails are foo@pipecat.ai and bar@pipecat.ai.") == 48
|
||||||
|
assert match_endofsentence("My email is foo.bar@pipecat.ai.") == 31
|
||||||
|
assert match_endofsentence("My email is spell(foo.bar@pipecat.ai).") == 38
|
||||||
|
assert match_endofsentence("The number pi is 3.14159.") == 25
|
||||||
|
assert match_endofsentence("Valid scientific notation 1.23e4.") == 33
|
||||||
|
assert match_endofsentence("Valid scientific notation 0.e4.") == 31
|
||||||
assert not match_endofsentence("This is not a sentence")
|
assert not match_endofsentence("This is not a sentence")
|
||||||
assert not match_endofsentence("This is not a sentence,")
|
assert not match_endofsentence("This is not a sentence,")
|
||||||
assert not match_endofsentence("This is not a sentence, ")
|
assert not match_endofsentence("This is not a sentence, ")
|
||||||
@@ -32,6 +37,7 @@ class TestUtilsString(unittest.IsolatedAsyncioTestCase):
|
|||||||
assert not match_endofsentence("America, or the U.") # U.S.A.
|
assert not match_endofsentence("America, or the U.") # U.S.A.
|
||||||
assert not match_endofsentence("It still early, it's 3:00 a.") # 3:00 a.m.
|
assert not match_endofsentence("It still early, it's 3:00 a.") # 3:00 a.m.
|
||||||
assert not match_endofsentence("My emails are foo@pipecat.ai and bar@pipecat.ai")
|
assert not match_endofsentence("My emails are foo@pipecat.ai and bar@pipecat.ai")
|
||||||
|
assert not match_endofsentence("The number pi is 3.14159")
|
||||||
|
|
||||||
async def test_endofsentence_zh(self):
|
async def test_endofsentence_zh(self):
|
||||||
chinese_sentences = [
|
chinese_sentences = [
|
||||||
|
|||||||
Reference in New Issue
Block a user