← Back to the playground

Lab · Master data management

One engine,
two languages.

The playground runs in TypeScript, in your browser. Data teams work in Python. So the same engine exists in both — reading the same data, and proven to produce identical results.

TypeScript

For the live experience

Runs instantly in the visitor's browser — every slider recalculates on the spot, with no server and no data leaving the page.

Python

For data teams

The language of analysis and data engineering — with pandas, tests and a notebook walkthrough, ready to run against real volumes.

Identical results — verified on every build

Default settings · shared sample data

ResultTypeScriptPythonMatch
Customers77✓
Households44✓
Pairs for review22✓
Pair scores compared4545✓

Every pair score, golden record and household is compared — not just these totals. If the two engines ever disagree, the website build fails.

Side by side

The same logic, step by step. These excerpts are read from the real source files when the site is built, so they are always the code that runs.

  1. Step 01

    How alike are two names?

    Jaro–Winkler similarity scores two strings from 0 to 1, forgiving small typos like Smith and Smyth and rewarding a shared start.

    TypeScriptsrc/lib/mdm.ts ↗
    /** Jaro–Winkler similarity, 0–1. Good at typos and short name variants. */
    export function jaroWinkler(a: string, b: string) {
      if (!a || !b) return 0;
      if (a === b) return 1;
      const range = Math.max(0, Math.floor(Math.max(a.length, b.length) / 2) - 1);
      const aMatch = new Array(a.length).fill(false);
      const bMatch = new Array(b.length).fill(false);
      let matches = 0;
      for (let i = 0; i < a.length; i++) {
        for (let j = Math.max(0, i - range); j < Math.min(b.length, i + range + 1); j++) {
          if (bMatch[j] || a[i] !== b[j]) continue;
          aMatch[i] = bMatch[j] = true;
          matches++;
          break;
        }
      }
      if (matches === 0) return 0;
      let transpositions = 0;
      for (let i = 0, k = 0; i < a.length; i++) {
        if (!aMatch[i]) continue;
        while (!bMatch[k]) k++;
        if (a[i] !== b[k++]) transpositions++;
      }
      const m = matches;
      const jaro = (m / a.length + m / b.length + (m - transpositions / 2) / m) / 3;
      let prefix = 0;
      while (prefix < 4 && a[prefix] === b[prefix]) prefix++;
      return jaro + prefix * 0.1 * (1 - jaro);
    }
    Pythonpython/mdm.py ↗
    def jaro_winkler(a: str, b: str) -> float:
        """Jaro–Winkler similarity, 0–1. Good at typos and short name variants."""
        if not a or not b:
            return 0.0
        if a == b:
            return 1.0
        match_range = max(0, max(len(a), len(b)) // 2 - 1)
        a_match = [False] * len(a)
        b_match = [False] * len(b)
        matches = 0
        for i, ch in enumerate(a):
            for j in range(max(0, i - match_range), min(len(b), i + match_range + 1)):
                if b_match[j] or ch != b[j]:
                    continue
                a_match[i] = b_match[j] = True
                matches += 1
                break
        if matches == 0:
            return 0.0
        transpositions = 0
        k = 0
        for i, ch in enumerate(a):
            if not a_match[i]:
                continue
            while not b_match[k]:
                k += 1
            if ch != b[k]:
                transpositions += 1
            k += 1
        m = matches
        jaro = (m / len(a) + m / len(b) + (m - transpositions / 2) / m) / 3
        prefix = 0
        while prefix < 4 and prefix < len(a) and prefix < len(b) and a[prefix] == b[prefix]:
            prefix += 1
        return jaro + prefix * 0.1 * (1 - jaro)
  2. Step 02

    Scoring a pair of records

    Field similarities are weighted and combined into a score out of 100. Missing fields don't count, and the thresholds turn the score into a decision.

    TypeScriptsrc/lib/mdm.ts ↗
    /** Weighted score over the fields both records have, scaled to 0–100. */
    export function scorePair(a: SourceRecord, b: SourceRecord, weights: Weights, t: Thresholds): Pair {
      const scores = fieldScores(a, b);
      let total = 0;
      let weight = 0;
      for (const f of Object.keys(scores) as Field[]) {
        const s = scores[f];
        if (s === null) continue;
        total += weights[f] * s;
        weight += weights[f];
      }
      const score = weight ? Math.round((total / weight) * 100) : 0;
      const decision: Decision = score >= t.auto ? "match" : score >= t.review ? "review" : "no-match";
      return { a, b, scores, score, decision };
    }
    Pythonpython/mdm.py ↗
    def score_pair(a: dict, b: dict, weights: dict, thresholds: dict) -> dict:
        """Weighted score over the fields both records have, scaled to 0–100."""
        scores = field_scores(a, b)
        total = weight = 0.0
        for field in FIELDS:
            if scores[field] is None:
                continue
            total += weights[field] * scores[field]
            weight += weights[field]
        # Round half up, exactly like JavaScript's Math.round for positive numbers.
        score = math.floor((total / weight) * 100 + 0.5) if weight else 0
        if score >= thresholds["auto"]:
            decision = "match"
        elif score >= thresholds["review"]:
            decision = "review"
        else:
            decision = "no-match"
        return {"a": a, "b": b, "scores": scores, "score": score, "decision": decision}
  3. Step 03

    Grouping matches into customers

    Union–find follows the automatic matches, so if A matches B and B matches C, all three become one customer.

    TypeScriptsrc/lib/mdm.ts ↗
    /** Group records into customers by following automatic matches (union–find). */
    export function cluster(records: SourceRecord[], pairs: Pair[]) {
      const parent = new Map(records.map((r) => [r.id, r.id]));
      const find = (id: string): string => (parent.get(id) === id ? id : find(parent.get(id)!));
      for (const p of pairs) if (p.decision === "match") parent.set(find(p.a.id), find(p.b.id));
      const groups = new Map<string, SourceRecord[]>();
      for (const r of records) groups.set(find(r.id), [...(groups.get(find(r.id)) ?? []), r]);
      return [...groups.values()];
    }
    Pythonpython/mdm.py ↗
    def cluster(records: list, pairs: list) -> list:
        """Group records into customers by following automatic matches (union–find)."""
        parent = {r["id"]: r["id"] for r in records}
    
        def find(record_id: str) -> str:
            return record_id if parent[record_id] == record_id else find(parent[record_id])
    
        for p in pairs:
            if p["decision"] == "match":
                parent[find(p["a"]["id"])] = find(p["b"]["id"])
        groups: dict[str, list] = {}
        for r in records:
            groups.setdefault(find(r["id"]), []).append(r)
        return list(groups.values())
  4. Step 04

    Building the golden record

    Survivorship rules decide which source wins each field. By default the policy system — the highest-quality source — wins every field it holds.

    TypeScriptsrc/lib/mdm.ts ↗
    export function survive(group: SourceRecord[], rules: Survivorship): GoldenRecord {
      const pick = (field: Field, order: (a: SourceRecord, b: SourceRecord) => number) => {
        const winner = [...group].filter((r) => r[field]).sort(order)[0];
        return { value: winner?.[field] ?? "", from: winner?.source ?? null };
      };
      return {
        id: group.map((r) => r.id).join("+"),
        members: group,
        values: {
          name:
            rules.name === "most-complete"
              ? pick("name", (a, b) => completeness(b) - completeness(a) || byTrust(a, b))
              : pick("name", byTrust),
          dob: pick("dob", byTrust), // date of birth: always the most trusted source
          address: rules.address === "most-recent" ? pick("address", byRecency) : pick("address", byTrust),
          email: pick("email", byTrust), // email: from the most trusted source
        },
      };
    }
    Pythonpython/mdm.py ↗
    def survive(group: list, rules: dict) -> dict:
        """Build one golden record; rules decide which source wins each field."""
    
        def pick(field: str, order) -> dict:
            candidates = sorted((r for r in group if r[field]), key=cmp_to_key(order))
            winner = candidates[0] if candidates else None
            return {"value": winner[field] if winner else "", "from": winner["source"] if winner else None}
    
        def most_complete(a: dict, b: dict) -> int:
            return (_completeness(b) - _completeness(a)) or _by_trust(a, b)
    
        return {
            "id": "+".join(r["id"] for r in group),
            "members": group,
            "values": {
                "name": pick("name", most_complete if rules["name"] == "most-complete" else _by_trust),
                "dob": pick("dob", _by_trust),  # date of birth: always the most trusted source
                "address": pick("address", _by_recency if rules["address"] == "most-recent" else _by_trust),
                "email": pick("email", _by_trust),  # email: from the most trusted source
            },
        }
  5. Step 05

    Grouping households

    Golden records with the same house number and postcode form a household.

    TypeScriptsrc/lib/mdm.ts ↗
    /** Group golden records that share an address into households. */
    export function households(golden: GoldenRecord[]) {
      const map = new Map<string, GoldenRecord[]>();
      for (const g of golden) {
        const key = householdKey(g.values.address.value);
        map.set(key, [...(map.get(key) ?? []), g]);
      }
      return [...map.entries()].map(([key, members]) => ({ key, members }));
    }
    Pythonpython/mdm.py ↗
    def households(golden: list) -> list:
        """Group golden records that share an address into households."""
        found: dict[str, list] = {}
        for g in golden:
            found.setdefault(household_key(g["values"]["address"]["value"]), []).append(g)
        return [{"key": key, "members": members} for key, members in found.items()]

Built with Claude Code; reviewed and understood by me. A simplified illustration with fictional data.

Contact

Let's talk.

I'm based in London, UK. Whether it's a leadership role, an advisory engagement or a transformation that needs shaping — my inbox is open.

srini.vankee@gmail.com