Repository navigation
Expand file tree
/
Copy path03_duplicate_detection.py
More file actions
82 lines (65 loc) · 2.97 KB
/
Copy path03_duplicate_detection.py
File metadata and controls
82 lines (65 loc) · 2.97 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
"""03_duplicate_detection.py — Duplicate Row Detection and Safe Resolution.
Problem:
--------
Blindly running `df.drop_duplicates()` in automated data pipelines is dangerous.
In many real-world systems (e.g. retail orders, audit logs, sensor telemetry),
identical rows can represent valid, distinct events rather than technical
glitches. Furthermore, if a faulty database join creates an unexpected 40%
duplicate ratio, silently dropping rows masks an upstream outage.
Dataset:
--------
A realistic transaction dataset containing both valid repeated customer orders
and accidentally duplicated webhook payloads.
Installation:
-------------
pip install freshdata-cleaner
Expected Result:
----------------
FreshData detects duplicate rows and evaluates the duplicate ratio:
- By default, duplicates are reported and warned upon, but NOT silently removed.
- When opt-in `drop_duplicates=True` is supplied, FreshData safely removes
redundant rows according to `duplicate_keep` semantics and records the
exact count in the audit trail.
- If the duplicate ratio exceeds safe thresholds, an alert is raised.
"""
from __future__ import annotations
import pandas as pd
import freshdata as fd
def create_order_data() -> pd.DataFrame:
return pd.DataFrame({
"order_id": ["ORD-101", "ORD-102", "ORD-102", "ORD-103", "ORD-104", "ORD-104"],
"customer_id": ["C-1", "C-2", "C-2", "C-3", "C-4", "C-4"],
"amount": [49.99, 120.00, 120.00, 15.50, 89.00, 89.00],
"status": ["completed", "completed", "completed", "shipped", "pending", "pending"],
})
def main() -> None:
print("=== FreshData Flagship Example 03: Duplicate Detection & Resolution ===")
df = create_order_data()
print(f"\n[1] Raw DataFrame ({len(df)} rows):")
print(df)
# 1. Safe default behavior: FreshData warns and reports, but does NOT silently drop
print("\n--- [A] Default Mode: Detection & Reporting Only (Safe Default) ---")
cleaned_default, report_default = fd.clean(df, return_report=True)
print(f"Rows after default clean: {len(cleaned_default)} (preserves all rows)")
if report_default.warnings:
print("Maintainer Warnings:")
for w in report_default.warnings:
print(f" ! {w}")
# 2. Opt-in resolution: Explicitly instruct FreshData to resolve duplicates
print("\n--- [B] Opt-in Mode: Controlled Duplicate Resolution ---")
cleaned_resolved, report_resolved = fd.clean(
df,
drop_duplicates=True,
duplicate_keep="first",
return_report=True,
)
dups = report_resolved.duplicates_removed
print(f"Rows after resolution: {len(cleaned_resolved)} (removed {dups} duplicates)")
print("\nResolved DataFrame:")
print(cleaned_resolved)
print("\nAudit Report Actions:")
for a in report_resolved.actions:
if "duplicate" in a.step or "duplicate" in a.description.lower():
print(f" - [{a.step}] {a.description} (risk: {a.risk}, count: {a.count})")
if __name__ == "__main__":
main()