Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
51 changes: 49 additions & 2 deletions src/freshdata/fieldcheck.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,10 @@ class to an action. The default policy is non-destructive: nothing is deleted,
"PolicyResult",
"validate_fields",
"apply_field_policy",
"looks_like_postal_code",
"looks_like_uk_postal_code",
"looks_like_ca_postal_code",
"looks_like_de_postal_code",
"CLASSIFICATIONS",
"ACTIONS",
]
Expand Down Expand Up @@ -84,13 +88,50 @@ class to an action. The default policy is non-destructive: nothing is deleted,
_PHONE_MAX_DIGITS = 15
_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._\-/]*$")

#: Postal codes for the ``postal_code`` semantic type: UK, Canada and the
#: German PLZ. Case and whitespace are normalised before matching, so
#: "sw1a 1aa" and "SW1A 1AA" read as the same code.
#:
#: UK: 1-2 letters plus a district that may carry a second digit/letter
#: (M1, W1A, B33, CR2, SN10); the inward is a digit and two letters.
#: Canada is letter/digit alternating; Germany is a 5-digit number.
_UK_POSTAL_RE = re.compile(r"^[A-Z]{1,2}[0-9][A-Z]{0,1}[0-9][A-Z]{2}$")
_CA_POSTAL_RE = re.compile(r"^[A-Z][0-9][A-Z][0-9][A-Z][0-9]$")
_DE_POSTAL_RE = re.compile(r"^[0-9]{5}$")
#: Space-free union of the three, used by the vectorised pre-screen; it
#: matches only the canonical uppercase form, so spaced/lowercase values
#: fall through to the per-cell check, which normalises first.
_POSTAL_COMPACT_RE = re.compile(
r"^([A-Z]{1,2}[0-9][A-Z]{0,1}[0-9][A-Z]{2}|[A-Z][0-9][A-Z][0-9][A-Z][0-9]|[0-9]{5})$"
)


def _is_phone(s: str) -> bool:
"""Scalar phone check; mirrors the vectorised check in the suspect scan."""
if not _PHONE_RE.match(s):
return False
return _PHONE_MIN_DIGITS <= sum(c.isdigit() for c in s) <= _PHONE_MAX_DIGITS

def _normalize_postal_code(value: str) -> str:
"""Upper-cased, whitespace-free form of a postal code value."""
return re.sub(r"\s+", "", value).upper()

def looks_like_postal_code(s: str) -> bool:
"""True when ``s`` is a valid UK, Canadian or German postal code."""
return bool(_POSTAL_COMPACT_RE.match(_normalize_postal_code(s)))

def looks_like_uk_postal_code(s: str) -> bool:
"""UK code, e.g. "SW1A 1AA" (case- and whitespace-insensitive)."""
return bool(_UK_POSTAL_RE.match(_normalize_postal_code(s)))

def looks_like_ca_postal_code(s: str) -> bool:
"""Canadian code, e.g. "K1A 0B1" (case- and whitespace-insensitive)."""
return bool(_CA_POSTAL_RE.match(_normalize_postal_code(s)))

def looks_like_de_postal_code(s: str) -> bool:
"""German PLZ, a 5-digit numeric code, e.g. "10115"."""
return bool(_DE_POSTAL_RE.match(_normalize_postal_code(s)))


def _safe_fullmatch(pattern: str, value: str) -> bool:
"""``re.fullmatch`` that never raises.
Expand Down Expand Up @@ -183,7 +224,7 @@ def _as_text(series: pd.Series, dtype: Any = "string") -> pd.Series:
_KNOWN_SEMANTIC_TYPES = _NUMERIC_TYPES | _DATE_TYPES | frozenset({
"company_name", "entity_name", "person_name", "city", "country",
"free_text", "text", "identifier", "account_number", "ticker",
"stock_ticker", "email", "url", "phone",
"stock_ticker", "email", "url", "phone", "postal_code",
})


Expand Down Expand Up @@ -612,6 +653,12 @@ def issue(classification: str, reason: str, rule: str, *,
return issue("semantic_mismatch", f"{s!r} is not a valid URL", "url_format")
if spec.semantic_type == "phone" and not _is_phone(s):
return issue("semantic_mismatch", f"{s!r} is not a plausible phone number", "phone_format")
if spec.semantic_type == "postal_code" and not looks_like_postal_code(s):
return issue(
"domain_mismatch",
f"{s!r} is not a valid UK, Canadian or German postal code",
"postal_code_format",
)

if spec.max_length is not None and len(s) > spec.max_length:
return issue(
Expand Down Expand Up @@ -711,7 +758,7 @@ def _suspect_rows(series: pd.Series, spec: FieldSpec) -> pd.Index:
type_res = {
"identifier": _ID_RE, "account_number": _ID_RE,
"ticker": _TICKER_RE, "stock_ticker": _TICKER_RE,
"email": _EMAIL_RE, "url": _URL_RE,
"email": _EMAIL_RE, "url": _URL_RE, "postal_code": _POSTAL_COMPACT_RE,
}
if spec.semantic_type in type_res:
fine &= strs.str.fullmatch(type_res[spec.semantic_type].pattern).fillna(False)
Expand Down
55 changes: 55 additions & 0 deletions tests/test_fieldcheck.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,10 @@
RemediationPolicy,
apply_field_policy,
detect_value_type,
looks_like_ca_postal_code,
looks_like_de_postal_code,
looks_like_postal_code,
looks_like_uk_postal_code,
validate_fields,
)

Expand Down Expand Up @@ -805,3 +809,54 @@ def test_numeric_outliers_still_flagged_after_the_bool_guard():
df = pd.DataFrame({"amount": [1.0, 2.0, 1.5, 2.5, 1.2, 2.2, 1.8, 2.8, 10_000.0]})
report = validate_fields(df, {"amount": FieldSpec(semantic_type="currency_amount")})
assert [i.row for i in report.issues if i.severity == "warning"] == [8]

# ---------------------------------------------------------------------------
# postal_code semantic type (UK / Canada / German PLZ)
# ---------------------------------------------------------------------------

POSTAL_SCHEMA = {"postal_code": FieldSpec(semantic_type="postal_code")}


def test_valid_postal_codes_pass_without_a_schema_override():
df = pd.DataFrame({"postal_code": ["SW1A 1AA", "K1A 0B1", "10115"]})
assert issues_for(df, POSTAL_SCHEMA, "postal_code") == []


def test_invalid_postal_code_is_domain_mismatch():
df = pd.DataFrame({"postal_code": ["SW1A 1AA", "ABC", "K1A 0B1"]})
[issue] = issues_for(df, POSTAL_SCHEMA, "postal_code")
assert issue.classification == "domain_mismatch"
assert issue.rule == "postal_code_format"
assert issue.row == 1


def test_lowercase_and_spaced_postal_codes_stay_valid():
df = pd.DataFrame({"postal_code": ["sw1a 1aa", "K1A 0B1"]})
assert issues_for(df, POSTAL_SCHEMA, "postal_code") == []


def test_postal_code_helpers_accept_and_reject_shapes():
assert looks_like_uk_postal_code("SW1A 1AA")
assert looks_like_uk_postal_code("M1 1AA")
assert not looks_like_uk_postal_code("10115")
assert looks_like_ca_postal_code("K1A 0B1")
assert not looks_like_ca_postal_code("SW1A 1AA")
assert looks_like_de_postal_code("10115")
assert not looks_like_de_postal_code("1011")
for value in ("SW1A 1AA", "K1A 0B1", "10115"):
assert looks_like_postal_code(value)
assert not looks_like_postal_code("SW1A")
assert not looks_like_postal_code("")


def test_postal_code_vector_prescreen_and_per_cell_fallback():
# the compact regex in _suspect_rows only knows the space-free
# uppercase form; spaced and lowercase values land in the slow path,
# which normalises before deciding (same rhythm as email/url).
canonical = pd.DataFrame({"postal_code": ["SW1A1AA", "K1A0B1"]})
assert issues_for(canonical, POSTAL_SCHEMA, "postal_code") == []
spaced = pd.DataFrame({"postal_code": ["SW1A 1AA", "sw1a 1aa", "10 115"]})
assert issues_for(spaced, POSTAL_SCHEMA, "postal_code") == []
junk = pd.DataFrame({"postal_code": ["SW1A 1AA", "10 115", "!!"]})
[issue] = issues_for(junk, POSTAL_SCHEMA, "postal_code")
assert issue.row == 2 and issue.rule == "postal_code_format"