diff options
| author | Bobby <[email protected]> | 2026-07-17 14:43:56 +0530 |
|---|---|---|
| committer | GitHub <[email protected]> | 2026-07-17 14:43:56 +0530 |
| commit | d27e5f4145972990544975a033d03abdd3a5f596 (patch) | |
| tree | b0d069809b22ee040e3bd294b6b7a868d926fde8 /tests | |
| parent | 8bff9a0ec3c7b7207ecd994ace4c696b71fd84f5 (diff) | |
| parent | 5d71c2f3b7218895a10bf485ddd3a04df2c4f66e (diff) | |
| download | edify-main.tar.xz edify-main.zip | |
Multi-locale phone validator, expanded corpora, and all-validator round-trip + operator-algebra coverage (#288)HEADmain
Completes the remaining non-docs library items on the v1.0.0 milestone:
a genuinely multi-locale `phone` validator, real per-locale corpus
hardening, and an end-to-end coverage layer that exercises every shipped
validator and the operator algebra.
## Multi-locale phone (#154)
`phone` is rebuilt as a table-driven multi-locale model: an optional `+`
or `00` prefix, two to eight digit groups of one to four digits each,
single-character space/dot/dash separators, and any group optionally
parenthesised. It now accepts the display forms it previously missed —
leading and inline parenthesised area codes such as `(555) 123-4567` and
`+44 (0)20 7946 0958`, and variable national grouping from three-group
North American through five-group French — plus two-to-six-digit service
and short codes. Doubled separators, bare prefixes, and letters are
still rejected. The corpus grows from a seven-string smoke test to
twenty-one real numbers across eleven locales with ten adversarial
rejects.
## Postal locale coverage (#155)
Multi-locale postal codes are delivered by the existing `postal`
validator; this pins it down with a twenty-code corpus spanning North
America, Europe, Asia, and Oceania (including alphanumeric UK, Irish,
Canadian, and Dutch shapes) with ten adversarial rejects. `zip_code`
stays deliberately US-scoped, with `postal` as its locale-complete
counterpart.
## Validator snapshots (#191)
Every per-validator migration to a callable `Pattern` has landed, so
regenerating the built-in snapshot corpus is a no-op apart from
`phone`'s new emitted shape, which is refreshed here.
## End-to-end coverage of every construct (#258)
- Every registered library validator now round-trips through both dict
and JSON serialization with its emitted regex preserved — a property
test over the full validator set, alongside the existing compiled-source
invariant.
- The operator algebra (`+`, `|`, and `.use()`) is asserted identical to
its fluent-chain equivalent across every character-class constant, at
both the emitted-string and compiled-matcher level, so the two surfaces
can never drift.
- Hardening corpora expand beyond the smoke-test handful to `phone`,
`postal`, `semver`, `slug`, `hostname`, `port`, `isbn`, and `vin`, each
with real-world accepts and adversarial rejects.
Closes #154, Closes #155, Closes #191, Closes #258
Diffstat (limited to 'tests')
| -rw-r--r-- | tests/library/algebra.test.py | 77 | ||||
| -rw-r--r-- | tests/library/corpora/hostname.toml | 15 | ||||
| -rw-r--r-- | tests/library/corpora/isbn.toml | 13 | ||||
| -rw-r--r-- | tests/library/corpora/phone.toml | 25 | ||||
| -rw-r--r-- | tests/library/corpora/port.toml | 17 | ||||
| -rw-r--r-- | tests/library/corpora/postal.toml | 35 | ||||
| -rw-r--r-- | tests/library/corpora/semver.toml | 18 | ||||
| -rw-r--r-- | tests/library/corpora/slug.toml | 17 | ||||
| -rw-r--r-- | tests/library/corpora/vin.toml | 11 | ||||
| -rw-r--r-- | tests/library/invariants.test.py | 23 | ||||
| -rw-r--r-- | tests/snapshots/library/phone.regex | 2 |
11 files changed, 244 insertions, 9 deletions
diff --git a/tests/library/algebra.test.py b/tests/library/algebra.test.py new file mode 100644 index 0000000..ee4005d --- /dev/null +++ b/tests/library/algebra.test.py @@ -0,0 +1,77 @@ +"""Chain <-> operator-algebra equivalence: ``+``, ``|`` and ``.use()`` emit the fluent-chain regex. + +Every character-class module constant is driven through both surfaces and the two +emitted regex strings are asserted identical, so the operator algebra can never +drift from the fluent builder it stands in for. +""" + +import pytest + +from edify import ( + ALPHANUMERIC, + ANY_CHAR, + CARRIAGE_RETURN, + DIGIT, + LETTER, + LOWERCASE, + NEW_LINE, + NON_DIGIT, + NON_WHITESPACE, + NON_WORD, + NULL_BYTE, + TAB, + UPPERCASE, + WHITESPACE, + WORD, + Pattern, +) + +_CONSTANTS: dict[str, Pattern] = { + "ANY_CHAR": ANY_CHAR, + "WHITESPACE": WHITESPACE, + "NON_WHITESPACE": NON_WHITESPACE, + "DIGIT": DIGIT, + "NON_DIGIT": NON_DIGIT, + "WORD": WORD, + "NON_WORD": NON_WORD, + "NEW_LINE": NEW_LINE, + "CARRIAGE_RETURN": CARRIAGE_RETURN, + "TAB": TAB, + "NULL_BYTE": NULL_BYTE, + "LETTER": LETTER, + "UPPERCASE": UPPERCASE, + "LOWERCASE": LOWERCASE, + "ALPHANUMERIC": ALPHANUMERIC, +} +_IDS = list(_CONSTANTS) +_VALUES = list(_CONSTANTS.values()) + + [email protected]("constant", _VALUES, ids=_IDS) +def test_plus_matches_chain_subexpression(constant: Pattern): + operator_form = (constant + DIGIT).to_regex_string() + chain_form = Pattern().subexpression(constant).subexpression(DIGIT).to_regex_string() + assert operator_form == chain_form + + [email protected]("constant", _VALUES, ids=_IDS) +def test_or_matches_chain_any_of(constant: Pattern): + operator_form = (constant | DIGIT).to_regex_string() + chain_form = ( + Pattern().any_of().subexpression(constant).subexpression(DIGIT).end().to_regex_string() + ) + assert operator_form == chain_form + + [email protected]("constant", _VALUES, ids=_IDS) +def test_use_matches_chain_subexpression(constant: Pattern): + use_form = Pattern().use(constant).to_regex_string() + chain_form = Pattern().subexpression(constant).to_regex_string() + assert use_form == chain_form + + [email protected]("constant", _VALUES, ids=_IDS) +def test_operator_and_chain_compile_to_the_same_matcher(constant: Pattern): + operator_compiled = (constant + DIGIT).to_regex().source + chain_compiled = Pattern().subexpression(constant).subexpression(DIGIT).to_regex().source + assert operator_compiled == chain_compiled diff --git a/tests/library/corpora/hostname.toml b/tests/library/corpora/hostname.toml new file mode 100644 index 0000000..23da560 --- /dev/null +++ b/tests/library/corpora/hostname.toml @@ -0,0 +1,15 @@ +accepts = [ + "example.com", + "sub.example.co.uk", + "localhost", + "a.b.c", + "xn--d1acufc.example", +] + +rejects = [ + "", + "-bad.com", + ".com", + "a..b", + "exa mple.com", +] diff --git a/tests/library/corpora/isbn.toml b/tests/library/corpora/isbn.toml new file mode 100644 index 0000000..9028bc5 --- /dev/null +++ b/tests/library/corpora/isbn.toml @@ -0,0 +1,13 @@ +accepts = [ + "978-3-16-148410-0", + "0-306-40615-2", + "9783161484100", + "0306406152", +] + +rejects = [ + "", + "123", + "abc", + "978-3-16-148410-X", +] diff --git a/tests/library/corpora/phone.toml b/tests/library/corpora/phone.toml index 7abe2f7..f7f2b35 100644 --- a/tests/library/corpora/phone.toml +++ b/tests/library/corpora/phone.toml @@ -1,17 +1,36 @@ accepts = [ "+1 555 123 4567", - "+44 20 7946 0958", + "+1-555-123-4567", "555-123-4567", "555.123.4567", - "+1-555-123-4567", + "(555) 123-4567", "5551234567", + "1-800-555-0199", + "+44 20 7946 0958", + "+44 (0)20 7946 0958", "+81 3-1234-5678", + "+91 98765 43210", + "+49 30 12345678", + "+33 1 42 68 53 00", + "+61 2 9374 4000", + "+55 11 91234-5678", + "00 1 555 123 4567", + "112", + "911", + "999", + "311", + "08000", ] rejects = [ "", + " ", "not a phone", "abc-def-ghij", - " ", "555--123--4567", + "++15551234567", + "+", + "1", + "555-", + "(555 123-4567", ] diff --git a/tests/library/corpora/port.toml b/tests/library/corpora/port.toml new file mode 100644 index 0000000..4c715cc --- /dev/null +++ b/tests/library/corpora/port.toml @@ -0,0 +1,17 @@ +accepts = [ + "0", + "1", + "80", + "443", + "8080", + "65535", +] + +rejects = [ + "", + "65536", + "99999", + "-1", + "abc", + "8080a", +] diff --git a/tests/library/corpora/postal.toml b/tests/library/corpora/postal.toml new file mode 100644 index 0000000..5bf05fd --- /dev/null +++ b/tests/library/corpora/postal.toml @@ -0,0 +1,35 @@ +accepts = [ + "90210", + "90210-1234", + "10115", + "75008", + "110001", + "1011 AB", + "NL-1011 AB", + "SW1A 1AA", + "EC1A 1BB", + "GIR 0AA", + "K1A 0B1", + "D02 AF30", + "100-0001", + "01310-100", + "2000", + "00184", + "28001", + "114 55", + "00-001", + "101000", +] + +rejects = [ + "", + " ", + "notazip", + "abcde", + "!!!!!", + "90210-12", + "90210-", + "ZZ99 9ZZ", + "123456789", + "----", +] diff --git a/tests/library/corpora/semver.toml b/tests/library/corpora/semver.toml new file mode 100644 index 0000000..287c6c2 --- /dev/null +++ b/tests/library/corpora/semver.toml @@ -0,0 +1,18 @@ +accepts = [ + "1.0.0", + "2.1.3", + "10.20.30", + "1.0.0-alpha", + "1.0.0-alpha.1", + "1.0.0-rc.1+build.5", + "1.0.0+build.1", +] + +rejects = [ + "", + "1", + "1.0", + "v1.0.0", + "1.0.0.0", + "01.0.0", +] diff --git a/tests/library/corpora/slug.toml b/tests/library/corpora/slug.toml new file mode 100644 index 0000000..87d5407 --- /dev/null +++ b/tests/library/corpora/slug.toml @@ -0,0 +1,17 @@ +accepts = [ + "hello-world", + "my-post-123", + "a", + "abc", + "x-y-z", +] + +rejects = [ + "", + " ", + "Hello-World", + "hello_world", + "-abc", + "abc-", + "a--b", +] diff --git a/tests/library/corpora/vin.toml b/tests/library/corpora/vin.toml new file mode 100644 index 0000000..bc5909e --- /dev/null +++ b/tests/library/corpora/vin.toml @@ -0,0 +1,11 @@ +accepts = [ + "1HGBH41JXMN109186", + "5YJSA1E26HF000001", +] + +rejects = [ + "", + "1HGBH41JXMN10918", + "ABC", + "1HGBH41JXMN109186Q", +] diff --git a/tests/library/invariants.test.py b/tests/library/invariants.test.py index 662e81c..340f6a4 100644 --- a/tests/library/invariants.test.py +++ b/tests/library/invariants.test.py @@ -18,13 +18,10 @@ def _registered_patterns() -> Iterator[tuple[str, Pattern]]: REGISTERED_PATTERNS: list[tuple[str, Pattern]] = list(_registered_patterns()) +_IDS = [registered_name for registered_name, _ in REGISTERED_PATTERNS] - ("name", "pattern"), - REGISTERED_PATTERNS, - ids=[registered_name for registered_name, _ in REGISTERED_PATTERNS], -) [email protected](("name", "pattern"), REGISTERED_PATTERNS, ids=_IDS) def test_to_regex_string_matches_the_compiled_pattern_source(name: str, pattern: Pattern): emitted_source = pattern.to_regex_string() compiled_source = pattern.to_regex().source @@ -32,3 +29,19 @@ def test_to_regex_string_matches_the_compiled_pattern_source(name: str, pattern: f"library pattern {name!r} emits {emitted_source!r} but its compiled Regex reports " f"{compiled_source!r} — the terminals have drifted apart." ) + + [email protected](("name", "pattern"), REGISTERED_PATTERNS, ids=_IDS) +def test_dict_round_trip_preserves_the_emitted_regex(name: str, pattern: Pattern): + restored = Pattern.from_dict(pattern.to_dict()) + assert restored.to_regex_string() == pattern.to_regex_string(), ( + f"library pattern {name!r} does not survive a to_dict/from_dict round trip." + ) + + [email protected](("name", "pattern"), REGISTERED_PATTERNS, ids=_IDS) +def test_json_round_trip_preserves_the_emitted_regex(name: str, pattern: Pattern): + restored = Pattern.from_json(pattern.to_json()) + assert restored.to_regex_string() == pattern.to_regex_string(), ( + f"library pattern {name!r} does not survive a to_json/from_json round trip." + ) diff --git a/tests/snapshots/library/phone.regex b/tests/snapshots/library/phone.regex index 609f322..49c4e2d 100644 --- a/tests/snapshots/library/phone.regex +++ b/tests/snapshots/library/phone.regex @@ -1 +1 @@ -^(?:\+?\d{1,4}?(?:\s|[\-\.])?\(?\d{1,3}?\)?(?:\s|[\-\.])?\d{1,4}(?:\s|[\-\.])?\d{1,4}(?:\s|[\-\.])?\d{1,9}|\d{2,4})$
\ No newline at end of file +^(?:(?:(?:00|[\+])[ .-]?)?(?:\d{1,4}|\(\d{1,4}\))(?:[ .-]?(?:\d{1,4}|\(\d{1,4}\))){1,7}|\d{2,6})$
\ No newline at end of file |
