Skip to content

Commit df185df

Browse files
authored
Merge pull request #197 from MAIF/feature/update_segmentation
⚡️Enhance segmentation regex patterns
2 parents 71bb10e + 8a2b6b8 commit df185df

4 files changed

Lines changed: 127 additions & 28 deletions

File tree

‎melusine/__init__.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -7,7 +7,7 @@
77

88
__all__ = ["config", "MelusinePipeline"]
99

10-
VERSION = (3, 3, 1)
10+
VERSION = (3, 3, 2)
1111
__version__ = ".".join(map(str, VERSION))
1212

1313
# ------------------------------- #

‎melusine/processors.py‎

Lines changed: 57 additions & 25 deletions
Original file line numberDiff line numberDiff line change
@@ -440,10 +440,16 @@ def create_segmentation_regex_list() -> Iterable[str]:
440440
List of segmentation regexs
441441
442442
"""
443+
# Match everything until the end of the line.
444+
tolerant_line_start = r"^.{,5}"
445+
semicolon_pattern = r" ?\n? ?: *\n?"
446+
dash_pattern = r"(?:[\n ]*--+)"
447+
# Match anything until first line break and following line breaks
448+
end_pattern = r"[^\n]*\n[\n ]*"
449+
443450
# Meta patterns of the form "META_KEYWORD : META_CONTENT"
444451
# Ex: "De : jean@gmail.com"
445-
meta_keywords_list_with_semicolon = [
446-
r"Date",
452+
mandatory_meta_keywords_list_with_semicolon = [
447453
r"De",
448454
r"Exp[ée]diteur",
449455
r"[ÀA]",
@@ -455,26 +461,47 @@ def create_segmentation_regex_list() -> Iterable[str]:
455461
r"To",
456462
r"Sent",
457463
r"Cc",
464+
]
465+
optional_meta_keywords_list_with_semicolon = [
466+
r"Date",
458467
r"Copie",
459468
r"Attachments",
469+
r"Objet",
470+
r"Object",
471+
r"Subject",
472+
r"Sujet",
460473
]
461-
piped_keywords_with_semicolon = "(?:" + "|".join(meta_keywords_list_with_semicolon) + ")"
462-
# Ajout d'un pattern pour gérer les cas sans ':' mais avec retour à la ligne
463-
starter_pattern_with_semicolon = rf"^.{{,5}}(?:{piped_keywords_with_semicolon} ?\n? ?: *\n?)"
464-
starter_pattern_with_newline = rf"^.{{,5}}(?:{piped_keywords_with_semicolon})\n"
474+
mandatory_pattern = (
475+
"(?:"
476+
+ tolerant_line_start
477+
+ "(?:"
478+
+ "|".join(mandatory_meta_keywords_list_with_semicolon)
479+
+ ")"
480+
+ semicolon_pattern
481+
+ ")"
482+
)
483+
optional_pattern = (
484+
"(?:"
485+
+ tolerant_line_start
486+
+ "(?:"
487+
+ "|".join(optional_meta_keywords_list_with_semicolon)
488+
+ ")"
489+
+ semicolon_pattern
490+
+ ")"
491+
)
465492

466493
# Meta patterns of the form "META_KEYWORD"
467494
# Ex: "Transféré par jean@gmail.com"
468495
# (pas de ":")
469496
# Ex: "------ Message transmis ------"
470497

471498
regex_weekdays = (
472-
r"(?:[Ll]undi|[Ll]un\.|[Mm]ardi|[Mm]ar\.|[Mm]ercredi|[Mm]er\.|[Jj]eudi|[Jj]eu\.|" # noqa
473-
r"[Vv]endredi|[Vv]en\.|[Ss]amedi|[Ss]am\.|[Dd]imanche|[Dd]im\.)" # noqa
499+
r"(?:[Ll]undi|[Ll]un\.|[Mm]ardi|[Mm]ar\.|[Mm]ercredi|[Mm]er\.|[Jj]eudi|[Jj]eu\.|"
500+
r"[Vv]endredi|[Vv]en\.|[Ss]amedi|[Ss]am\.|[Dd]imanche|[Dd]im\.)"
474501
)
475502
regex_months = (
476-
r"(?:[Jj]anvier|[Ff][ée]vrier|[Mm]ars|[Aa]vril|[Mm]ai|[Jj]uin|[Jj]uillet|" # noqa
477-
r"[Aa]o[ûu]t|[Ss]eptembre|[Oo]ctobre|[Nn]ovembre|[Dd][eé]cembre|" # noqa
503+
r"(?:[Jj]anvier|[Ff][ée]vrier|[Mm]ars|[Aa]vril|[Mm]ai|[Jj]uin|[Jj]uillet|"
504+
r"[Aa]o[ûu]t|[Ss]eptembre|[Oo]ctobre|[Nn]ovembre|[Dd][eé]cembre|"
478505
r"(?:janv?|f[ée]vr?|mar|avr|juil?|sept?|oct|nov|d[ée]c)\.)"
479506
)
480507

@@ -483,10 +510,10 @@ def create_segmentation_regex_list() -> Iterable[str]:
483510
# Le 02 juillet 1991 à 11:20 jane@gmail.fr a écrit :
484511
# Le mardi 31 août 2021 à 11:09, <ville@maif.fr> a écrit :
485512
(
486-
rf"\bLe (?:"
487-
rf"\d{{2}}/\d{{2}}/\d{{4}}|\d{{4}}-\d{{2}}-\d{{2}}|{regex_weekdays}|" # noqa
513+
r"\bLe (?:"
514+
rf"\d{{2}}/\d{{2}}/\d{{4}}|\d{{4}}-\d{{2}}-\d{{2}}|{regex_weekdays}|"
488515
rf"\d{{1,2}} {regex_months})(?:.|\n)"
489-
rf"{{,30}}\d{{2}}:\d{{2}}(?:.|\n){{,50}}(?:\<.{{,30}}\>.{{,5}})?\ba [éecrit]" # noqa
516+
r"{,30}\d{2}:\d{2}(?:.|\n){,50}(?:\<.{,30}\>.{,5})?\ba [ée]crit"
490517
),
491518
r"Transf[ée]r[ée] par",
492519
r"D[ée]but du message transf[ée]r[ée] :",
@@ -500,22 +527,22 @@ def create_segmentation_regex_list() -> Iterable[str]:
500527
r"Forwarded by",
501528
]
502529
piped_keywords_without_semicolon = "(?:" + "|".join(meta_keywords_list_without_semicolon) + ")" # noqa
503-
starter_pattern_without_semicolon = f"{piped_keywords_without_semicolon}(?:[\n ]*--+)?"
504530

505-
# Combine pattern with and without semicolon, et avec retour à la ligne
506-
starter_pattern = (
507-
rf"(?:{starter_pattern_with_semicolon}|{starter_pattern_with_newline}|{starter_pattern_without_semicolon})"
531+
mandatory_pattern_without_semicolon = (
532+
"(?:" + "(?:" + "|".join(meta_keywords_list_without_semicolon) + ")" + rf"{dash_pattern}?" + ")"
508533
)
509534

510-
# Match everything until the end of the line.
511-
# Match End of line "\n" and "space" characters
512-
end_pattern = r".*[\n ]*"
535+
# Must match at least one mandatory pattern (ex: "De :) and any optional patterns. Exemple :
536+
# Copy : blabla (optional)
537+
# De : test_from@maif.fr (mandatory)
538+
# A : test_to@maif.fr (mandatory)
539+
# Attachments : test.pdf, blop.png (optional)
540+
# Cc : test_cc@gmail.co (mandatory)
541+
any_line = rf"(?:(?:{optional_pattern}|{mandatory_pattern}|{mandatory_pattern_without_semicolon}){end_pattern})"
542+
mandatory_line = rf"(?:(?:{mandatory_pattern_without_semicolon}|{mandatory_pattern}){end_pattern})"
543+
544+
full_generic_meta_pattern = rf"{any_line}*{mandatory_line}{any_line}*"
513545

514-
# Object / Subject pattern (These patterns are not sufficient to trigger segmentation)
515-
object_line_pattern = "(?:^.{,5}(?:Objet|Subject|Sujet) ?\n? ?: *\n?)" + end_pattern
516-
full_generic_meta_pattern = (
517-
rf"(?:(?:{object_line_pattern})?{starter_pattern}{end_pattern}(?:{object_line_pattern})*)+"
518-
)
519546
pattern_list = (full_generic_meta_pattern,)
520547
return pattern_list
521548

@@ -1233,10 +1260,15 @@ def GREETINGS(self) -> str | list[str] | re.Pattern:
12331260
r"^.{0,3}Bonne r[ée]ception.{0,3}$",
12341261
r"^.{0,3}votre bien d[ée]vou[ée]e?.{0,3}$",
12351262
r"^.{0,3}amicalement votre.{0,3}$",
1263+
(
1264+
r"^.{0,3}(?:Je|Nous) vous pri(?:e|ons) d'(?:agr[ée]er|accepter)(.{0,4}madame)?(.{0,4}monsieur)?"
1265+
r"(?:.{0,50}(consideration|salutations|sentiments).{0,20})?.{0,10}$"
1266+
),
12361267
(
12371268
r"^.{,3}je vous prie de croire.{,50}"
12381269
r"(expression|assurance)?.{,50}(consideration|salutations|sentiments).{,30}$"
12391270
),
1271+
r"^.{0,3}(expression|assurance)?.{,50}(consideration|salutations|sentiments).{,30}$",
12401272
# English
12411273
r"^.{0,3}regards.{0,3}$",
12421274
r"^.{0,3}(best|warm|kind|my) *(regards|wishes)?.{0,3}$",

‎tests/processors/test_content_refined_tagger.py‎

Lines changed: 37 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -264,12 +264,17 @@ def test_content_tagger_split_text(text, expected_parts):
264264
{"base_text": "John Smith", "base_tag": "BODY", "refined_tag": "SIGNATURE_NAME"},
265265
],
266266
),
267-
(
267+
pytest.param(
268268
(
269269
"chère madame,\n"
270270
"URGENT URGENT\n"
271271
"Merci de me faire suivre les docs à ma nouvelle adresse qui est 0 rue du parc, 75000 Paris. "
272-
"Merci d'avance. \nRecevez nos salutations,\nVous en souhaitant bonne réception"
272+
"Merci d'avance. \nRecevez nos salutations,\nVous en souhaitant bonne réception\n"
273+
"Nous vous prions d'agréer.\n"
274+
"Nous vous prions d'agréer do not match\n"
275+
"Nous vous prions d'agréer, Madame, Monsieur, l'expression de nos salutations distinguées.\n"
276+
"Nous vous prions d'accepter, Madame, Monsieur, l'assurance de nos sentiments sincères.\n"
277+
"L'expression de mes salutations distinguées.\n"
273278
),
274279
[
275280
{"base_text": "chère madame,", "base_tag": "HELLO", "refined_tag": "HELLO"},
@@ -286,7 +291,37 @@ def test_content_tagger_split_text(text, expected_parts):
286291
"base_tag": "GREETINGS",
287292
"refined_tag": "GREETINGS",
288293
},
294+
{
295+
"base_text": ("Nous vous prions d'agréer."),
296+
"base_tag": "GREETINGS",
297+
"refined_tag": "GREETINGS",
298+
},
299+
{
300+
"base_text": ("Nous vous prions d'agréer do not match"),
301+
"base_tag": "BODY",
302+
"refined_tag": "BODY",
303+
},
304+
{
305+
"base_text": (
306+
"Nous vous prions d'agréer, Madame, Monsieur, l'expression de nos salutations distinguées."
307+
),
308+
"base_tag": "GREETINGS",
309+
"refined_tag": "GREETINGS",
310+
},
311+
{
312+
"base_text": (
313+
"Nous vous prions d'accepter, Madame, Monsieur, l'assurance de nos sentiments sincères."
314+
),
315+
"base_tag": "GREETINGS",
316+
"refined_tag": "GREETINGS",
317+
},
318+
{
319+
"base_text": ("L'expression de mes salutations distinguées."),
320+
"base_tag": "GREETINGS",
321+
"refined_tag": "GREETINGS",
322+
},
289323
],
324+
id="Test_different_greetings",
290325
),
291326
pytest.param(
292327
"Un témoignage sous X\nEnvoyé depuis mon téléphone Orange",

‎tests/processors/test_processors.py‎

Lines changed: 32 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -163,6 +163,38 @@ def test_segmenter(input_text, expected_messages):
163163
assert message.text == expected_messages[i].text
164164

165165

166+
@pytest.mark.parametrize(
167+
"input_text, expected_messages",
168+
[
169+
pytest.param(
170+
"I want to break free.\nAfin de\ntest test test test\n:\n● I want to break free",
171+
[
172+
Message(meta="", text="I want to break free.\nAfin de\ntest test test test\n:\n● I want to break free"),
173+
],
174+
id="Afin de",
175+
),
176+
pytest.param(
177+
"Hello.\nCopie: test test\nCopie\n: test test\n Copie :\ntest test\nDe: test@gmail.com\nsome text",
178+
[
179+
Message(meta="", text="Hello."),
180+
Message(
181+
meta="Copie: test test\nCopie\n: test test\n Copie :\ntest test\nDe: test@gmail.com",
182+
text="some text",
183+
),
184+
],
185+
id="Copie keyword",
186+
),
187+
],
188+
)
189+
def test_segmenter_2(input_text, expected_messages):
190+
"""Test"""
191+
segmenter = Segmenter()
192+
result = segmenter.segment_text(input_text)
193+
for i, message in enumerate(result):
194+
assert message.meta == expected_messages[i].meta
195+
assert message.text == expected_messages[i].text
196+
197+
166198
@pytest.mark.parametrize(
167199
"input_message_list, expected_text",
168200
[

0 commit comments

Comments
 (0)