aboutsummaryrefslogtreecommitdiff
path: root/tests
diff options
context:
space:
mode:
authorBobby <[email protected]>2026-07-17 14:43:56 +0530
committerGitHub <[email protected]>2026-07-17 14:43:56 +0530
commitd27e5f4145972990544975a033d03abdd3a5f596 (patch)
treeb0d069809b22ee040e3bd294b6b7a868d926fde8 /tests
parent8bff9a0ec3c7b7207ecd994ace4c696b71fd84f5 (diff)
parent5d71c2f3b7218895a10bf485ddd3a04df2c4f66e (diff)
downloadedify-main.tar.xz
edify-main.zip
Multi-locale phone validator, expanded corpora, and all-validator round-trip + operator-algebra coverage (#288)HEADmain
Completes the remaining non-docs library items on the v1.0.0 milestone: a genuinely multi-locale `phone` validator, real per-locale corpus hardening, and an end-to-end coverage layer that exercises every shipped validator and the operator algebra. ## Multi-locale phone (#154) `phone` is rebuilt as a table-driven multi-locale model: an optional `+` or `00` prefix, two to eight digit groups of one to four digits each, single-character space/dot/dash separators, and any group optionally parenthesised. It now accepts the display forms it previously missed — leading and inline parenthesised area codes such as `(555) 123-4567` and `+44 (0)20 7946 0958`, and variable national grouping from three-group North American through five-group French — plus two-to-six-digit service and short codes. Doubled separators, bare prefixes, and letters are still rejected. The corpus grows from a seven-string smoke test to twenty-one real numbers across eleven locales with ten adversarial rejects. ## Postal locale coverage (#155) Multi-locale postal codes are delivered by the existing `postal` validator; this pins it down with a twenty-code corpus spanning North America, Europe, Asia, and Oceania (including alphanumeric UK, Irish, Canadian, and Dutch shapes) with ten adversarial rejects. `zip_code` stays deliberately US-scoped, with `postal` as its locale-complete counterpart. ## Validator snapshots (#191) Every per-validator migration to a callable `Pattern` has landed, so regenerating the built-in snapshot corpus is a no-op apart from `phone`'s new emitted shape, which is refreshed here. ## End-to-end coverage of every construct (#258) - Every registered library validator now round-trips through both dict and JSON serialization with its emitted regex preserved — a property test over the full validator set, alongside the existing compiled-source invariant. - The operator algebra (`+`, `|`, and `.use()`) is asserted identical to its fluent-chain equivalent across every character-class constant, at both the emitted-string and compiled-matcher level, so the two surfaces can never drift. - Hardening corpora expand beyond the smoke-test handful to `phone`, `postal`, `semver`, `slug`, `hostname`, `port`, `isbn`, and `vin`, each with real-world accepts and adversarial rejects. Closes #154, Closes #155, Closes #191, Closes #258
Diffstat (limited to 'tests')
-rw-r--r--tests/library/algebra.test.py77
-rw-r--r--tests/library/corpora/hostname.toml15
-rw-r--r--tests/library/corpora/isbn.toml13
-rw-r--r--tests/library/corpora/phone.toml25
-rw-r--r--tests/library/corpora/port.toml17
-rw-r--r--tests/library/corpora/postal.toml35
-rw-r--r--tests/library/corpora/semver.toml18
-rw-r--r--tests/library/corpora/slug.toml17
-rw-r--r--tests/library/corpora/vin.toml11
-rw-r--r--tests/library/invariants.test.py23
-rw-r--r--tests/snapshots/library/phone.regex2
11 files changed, 244 insertions, 9 deletions
diff --git a/tests/library/algebra.test.py b/tests/library/algebra.test.py
new file mode 100644
index 0000000..ee4005d
--- /dev/null
+++ b/tests/library/algebra.test.py
@@ -0,0 +1,77 @@
+"""Chain <-> operator-algebra equivalence: ``+``, ``|`` and ``.use()`` emit the fluent-chain regex.
+
+Every character-class module constant is driven through both surfaces and the two
+emitted regex strings are asserted identical, so the operator algebra can never
+drift from the fluent builder it stands in for.
+"""
+
+import pytest
+
+from edify import (
+ ALPHANUMERIC,
+ ANY_CHAR,
+ CARRIAGE_RETURN,
+ DIGIT,
+ LETTER,
+ LOWERCASE,
+ NEW_LINE,
+ NON_DIGIT,
+ NON_WHITESPACE,
+ NON_WORD,
+ NULL_BYTE,
+ TAB,
+ UPPERCASE,
+ WHITESPACE,
+ WORD,
+ Pattern,
+)
+
+_CONSTANTS: dict[str, Pattern] = {
+ "ANY_CHAR": ANY_CHAR,
+ "WHITESPACE": WHITESPACE,
+ "NON_WHITESPACE": NON_WHITESPACE,
+ "DIGIT": DIGIT,
+ "NON_DIGIT": NON_DIGIT,
+ "WORD": WORD,
+ "NON_WORD": NON_WORD,
+ "NEW_LINE": NEW_LINE,
+ "CARRIAGE_RETURN": CARRIAGE_RETURN,
+ "TAB": TAB,
+ "NULL_BYTE": NULL_BYTE,
+ "LETTER": LETTER,
+ "UPPERCASE": UPPERCASE,
+ "LOWERCASE": LOWERCASE,
+ "ALPHANUMERIC": ALPHANUMERIC,
+}
+_IDS = list(_CONSTANTS)
+_VALUES = list(_CONSTANTS.values())
+
+
[email protected]("constant", _VALUES, ids=_IDS)
+def test_plus_matches_chain_subexpression(constant: Pattern):
+ operator_form = (constant + DIGIT).to_regex_string()
+ chain_form = Pattern().subexpression(constant).subexpression(DIGIT).to_regex_string()
+ assert operator_form == chain_form
+
+
[email protected]("constant", _VALUES, ids=_IDS)
+def test_or_matches_chain_any_of(constant: Pattern):
+ operator_form = (constant | DIGIT).to_regex_string()
+ chain_form = (
+ Pattern().any_of().subexpression(constant).subexpression(DIGIT).end().to_regex_string()
+ )
+ assert operator_form == chain_form
+
+
[email protected]("constant", _VALUES, ids=_IDS)
+def test_use_matches_chain_subexpression(constant: Pattern):
+ use_form = Pattern().use(constant).to_regex_string()
+ chain_form = Pattern().subexpression(constant).to_regex_string()
+ assert use_form == chain_form
+
+
[email protected]("constant", _VALUES, ids=_IDS)
+def test_operator_and_chain_compile_to_the_same_matcher(constant: Pattern):
+ operator_compiled = (constant + DIGIT).to_regex().source
+ chain_compiled = Pattern().subexpression(constant).subexpression(DIGIT).to_regex().source
+ assert operator_compiled == chain_compiled
diff --git a/tests/library/corpora/hostname.toml b/tests/library/corpora/hostname.toml
new file mode 100644
index 0000000..23da560
--- /dev/null
+++ b/tests/library/corpora/hostname.toml
@@ -0,0 +1,15 @@
+accepts = [
+ "example.com",
+ "sub.example.co.uk",
+ "localhost",
+ "a.b.c",
+ "xn--d1acufc.example",
+]
+
+rejects = [
+ "",
+ "-bad.com",
+ ".com",
+ "a..b",
+ "exa mple.com",
+]
diff --git a/tests/library/corpora/isbn.toml b/tests/library/corpora/isbn.toml
new file mode 100644
index 0000000..9028bc5
--- /dev/null
+++ b/tests/library/corpora/isbn.toml
@@ -0,0 +1,13 @@
+accepts = [
+ "978-3-16-148410-0",
+ "0-306-40615-2",
+ "9783161484100",
+ "0306406152",
+]
+
+rejects = [
+ "",
+ "123",
+ "abc",
+ "978-3-16-148410-X",
+]
diff --git a/tests/library/corpora/phone.toml b/tests/library/corpora/phone.toml
index 7abe2f7..f7f2b35 100644
--- a/tests/library/corpora/phone.toml
+++ b/tests/library/corpora/phone.toml
@@ -1,17 +1,36 @@
accepts = [
"+1 555 123 4567",
- "+44 20 7946 0958",
+ "+1-555-123-4567",
"555-123-4567",
"555.123.4567",
- "+1-555-123-4567",
+ "(555) 123-4567",
"5551234567",
+ "1-800-555-0199",
+ "+44 20 7946 0958",
+ "+44 (0)20 7946 0958",
"+81 3-1234-5678",
+ "+91 98765 43210",
+ "+49 30 12345678",
+ "+33 1 42 68 53 00",
+ "+61 2 9374 4000",
+ "+55 11 91234-5678",
+ "00 1 555 123 4567",
+ "112",
+ "911",
+ "999",
+ "311",
+ "08000",
]
rejects = [
"",
+ " ",
"not a phone",
"abc-def-ghij",
- " ",
"555--123--4567",
+ "++15551234567",
+ "+",
+ "1",
+ "555-",
+ "(555 123-4567",
]
diff --git a/tests/library/corpora/port.toml b/tests/library/corpora/port.toml
new file mode 100644
index 0000000..4c715cc
--- /dev/null
+++ b/tests/library/corpora/port.toml
@@ -0,0 +1,17 @@
+accepts = [
+ "0",
+ "1",
+ "80",
+ "443",
+ "8080",
+ "65535",
+]
+
+rejects = [
+ "",
+ "65536",
+ "99999",
+ "-1",
+ "abc",
+ "8080a",
+]
diff --git a/tests/library/corpora/postal.toml b/tests/library/corpora/postal.toml
new file mode 100644
index 0000000..5bf05fd
--- /dev/null
+++ b/tests/library/corpora/postal.toml
@@ -0,0 +1,35 @@
+accepts = [
+ "90210",
+ "90210-1234",
+ "10115",
+ "75008",
+ "110001",
+ "1011 AB",
+ "NL-1011 AB",
+ "SW1A 1AA",
+ "EC1A 1BB",
+ "GIR 0AA",
+ "K1A 0B1",
+ "D02 AF30",
+ "100-0001",
+ "01310-100",
+ "2000",
+ "00184",
+ "28001",
+ "114 55",
+ "00-001",
+ "101000",
+]
+
+rejects = [
+ "",
+ " ",
+ "notazip",
+ "abcde",
+ "!!!!!",
+ "90210-12",
+ "90210-",
+ "ZZ99 9ZZ",
+ "123456789",
+ "----",
+]
diff --git a/tests/library/corpora/semver.toml b/tests/library/corpora/semver.toml
new file mode 100644
index 0000000..287c6c2
--- /dev/null
+++ b/tests/library/corpora/semver.toml
@@ -0,0 +1,18 @@
+accepts = [
+ "1.0.0",
+ "2.1.3",
+ "10.20.30",
+ "1.0.0-alpha",
+ "1.0.0-alpha.1",
+ "1.0.0-rc.1+build.5",
+ "1.0.0+build.1",
+]
+
+rejects = [
+ "",
+ "1",
+ "1.0",
+ "v1.0.0",
+ "1.0.0.0",
+ "01.0.0",
+]
diff --git a/tests/library/corpora/slug.toml b/tests/library/corpora/slug.toml
new file mode 100644
index 0000000..87d5407
--- /dev/null
+++ b/tests/library/corpora/slug.toml
@@ -0,0 +1,17 @@
+accepts = [
+ "hello-world",
+ "my-post-123",
+ "a",
+ "abc",
+ "x-y-z",
+]
+
+rejects = [
+ "",
+ " ",
+ "Hello-World",
+ "hello_world",
+ "-abc",
+ "abc-",
+ "a--b",
+]
diff --git a/tests/library/corpora/vin.toml b/tests/library/corpora/vin.toml
new file mode 100644
index 0000000..bc5909e
--- /dev/null
+++ b/tests/library/corpora/vin.toml
@@ -0,0 +1,11 @@
+accepts = [
+ "1HGBH41JXMN109186",
+ "5YJSA1E26HF000001",
+]
+
+rejects = [
+ "",
+ "1HGBH41JXMN10918",
+ "ABC",
+ "1HGBH41JXMN109186Q",
+]
diff --git a/tests/library/invariants.test.py b/tests/library/invariants.test.py
index 662e81c..340f6a4 100644
--- a/tests/library/invariants.test.py
+++ b/tests/library/invariants.test.py
@@ -18,13 +18,10 @@ def _registered_patterns() -> Iterator[tuple[str, Pattern]]:
REGISTERED_PATTERNS: list[tuple[str, Pattern]] = list(_registered_patterns())
+_IDS = [registered_name for registered_name, _ in REGISTERED_PATTERNS]
- ("name", "pattern"),
- REGISTERED_PATTERNS,
- ids=[registered_name for registered_name, _ in REGISTERED_PATTERNS],
-)
[email protected](("name", "pattern"), REGISTERED_PATTERNS, ids=_IDS)
def test_to_regex_string_matches_the_compiled_pattern_source(name: str, pattern: Pattern):
emitted_source = pattern.to_regex_string()
compiled_source = pattern.to_regex().source
@@ -32,3 +29,19 @@ def test_to_regex_string_matches_the_compiled_pattern_source(name: str, pattern:
f"library pattern {name!r} emits {emitted_source!r} but its compiled Regex reports "
f"{compiled_source!r} — the terminals have drifted apart."
)
+
+
[email protected](("name", "pattern"), REGISTERED_PATTERNS, ids=_IDS)
+def test_dict_round_trip_preserves_the_emitted_regex(name: str, pattern: Pattern):
+ restored = Pattern.from_dict(pattern.to_dict())
+ assert restored.to_regex_string() == pattern.to_regex_string(), (
+ f"library pattern {name!r} does not survive a to_dict/from_dict round trip."
+ )
+
+
[email protected](("name", "pattern"), REGISTERED_PATTERNS, ids=_IDS)
+def test_json_round_trip_preserves_the_emitted_regex(name: str, pattern: Pattern):
+ restored = Pattern.from_json(pattern.to_json())
+ assert restored.to_regex_string() == pattern.to_regex_string(), (
+ f"library pattern {name!r} does not survive a to_json/from_json round trip."
+ )
diff --git a/tests/snapshots/library/phone.regex b/tests/snapshots/library/phone.regex
index 609f322..49c4e2d 100644
--- a/tests/snapshots/library/phone.regex
+++ b/tests/snapshots/library/phone.regex
@@ -1 +1 @@
-^(?:\+?\d{1,4}?(?:\s|[\-\.])?\(?\d{1,3}?\)?(?:\s|[\-\.])?\d{1,4}(?:\s|[\-\.])?\d{1,4}(?:\s|[\-\.])?\d{1,9}|\d{2,4})$ \ No newline at end of file
+^(?:(?:(?:00|[\+])[ .-]?)?(?:\d{1,4}|\(\d{1,4}\))(?:[ .-]?(?:\d{1,4}|\(\d{1,4}\))){1,7}|\d{2,6})$ \ No newline at end of file