Files

483 lines
22 KiB
Python

#!/usr/bin/env python3
"""
SODH - BR-013 alignment, last step of the v1.1 migration
=========================================================
Regenerates the Metric and Data Element sections of instances/sodh.ttl from the
v0.7 back-doc, which has been ahead of the TTL since the divergence found at
audit time. This is the migration that puts the flow back the right way round:
after it, the TTL is the source and the workbook becomes a generated artefact.
USAGE
python3 scripts/apply_br013_sodh.py # dry run
python3 scripts/apply_br013_sodh.py --apply # rewrite, .bak kept
WHAT IT FIXES
37 BLOCKING DataElementMeaningShape -- no element had a route to meaning
15 BLOCKING physicalName on a Metric -- layer leak
15 WARNING hasGranularity on a Metric
6 WARNING monitoredBy on a Data Object
WHAT IT DOES
1. 15 metrics -> 10 mother metrics carrying the harmonized calculation
rule, each with a formula, a unit and a measured Concept. No physical
name, no granularity: both belong elsewhere now.
2. 95 measure Data Elements generated, each computedBy its mother metric,
each carrying the physical name the metric used to hold.
3. 37 dimensional Data Elements given a represents towards their Concept.
With 2, this closes the XOR: every element reaches meaning by exactly
one route.
4. Three concepts created because the XOR forces them to be named --
Country, Marketing Entity, Fiscal Period had elements but no notion.
TO_ARBITRATE like the others.
5. Steward removed from the 6 Data Objects: it is inherited now.
6. hasGrainElement on the 4 DIMENSION Data Objects only.
WHY FACT TABLES GET NO GRAIN ELEMENTS
A fact table's grain is carried by its foreign keys, and OW-006 says a
foreign key is not a Data Element -- so a fact object has no element of its
own to point at. Its grain is already expressed, by references towards the
dimension objects it joins. Forcing hasGrainElement onto facts would mean
either breaking OW-006 or pointing at another object's elements, which
GrainConsistencyShape rejects. Nothing is lost: the information is there.
"""
import os
import re
import shutil
import sys
REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
TTL = os.path.join(REPO, "instances", "sodh.ttl")
BACKDOC = os.path.join(REPO, "instances", "SODH_data.xlsx")
W = 78
# ---- mother metric -> the Concept it measures ------------------------------
METRIC_CONCEPT = {
"M-06.01-001": "ex:BC_06_01_001", # Sell Out Volume -> Sell Out
"M-06.01-002": "ex:BC_06_01_001", # Sell Out Value -> Sell Out
"M-06.01-003": "ex:BC_06_01_004", # Numeric Distribution -> Retail Distribution
"M-06.01-004": "ex:BC_06_01_004", # Weighted Distribution -> Retail Distribution
"M-06.01-005": "ex:BC_06_01_004", # Total Distribution Pts -> Retail Distribution
"M-06.01-006": "ex:BC_06_01_002", # Forward Stock Volume -> Retailer Stock
"M-06.01-007": "ex:BC_06_01_002", # Days of Coverage -> Retailer Stock
"M-06.01-008": "ex:BC_06_01_002", # Stock Share -> Retailer Stock
"M-06.01-009": "ex:BC_06_01_005", # Baseline Volume -> Sell Out Baseline
"M-06.01-010": "ex:BC_06_01_005", # Incremental Volume -> Sell Out Baseline
}
# ---- dimensional element -> the Concept it represents ----------------------
# Exact ids first, then id prefixes. Every dimensional element must land
# somewhere: that is what the XOR is for.
DE_CONCEPT_EXACT = {
"DE-04.01-0006": "ex:BC_04_01_002", # Outlet Type -> Outlet
"DE-04.01-0007": "ex:BC_04_01_002", # Channel -> Outlet
"DE-04.01-0010": "ex:BC_04_01_002", # Point of Sale Count -> Outlet
"DE-04.01-0008": "ex:BC_04_03_001", # Country Code -> Country
"DE-04.01-0009": "ex:BC_04_03_001", # Country Name -> Country
"DE-16.01-0001": "ex:BC_16_01_001", # Currency Code -> Currency
"DE-16.01-0002": "ex:BC_16_01_002", # FX Rate to-Euro -> Exchange Rate
"DE-16.01-0003": "ex:BC_16_01_002", # FX Rate to-USD -> Exchange Rate
"DE-16.02-0001": "ex:BC_16_03_001", # Marketing Entity -> Marketing Entity
"DE-16.02-0002": "ex:BC_16_02_001", # Fiscal Period Code -> Fiscal Period
"DE-16.02-0003": "ex:BC_16_02_001", # Fiscal Quarter Code -> Fiscal Period
}
DE_CONCEPT_PREFIX = [
("DE-10.01", "ex:BC_10_01_001"), # product hierarchy -> Product
("DE-04.01", "ex:BC_04_01_001"), # trade hierarchy -> Customer
("DE-21.01", "ex:BC_21_01_001"), # calendar -> Calendar Date
("DE-05.01", "ex:BC_05_01_001"), # promotion flags -> Promotion
]
# ---- concepts the XOR forces us to name ------------------------------------
NEW_CONCEPTS = [
# Final identifiers from the outset. Creating them under one id and
# renumbering them later broke idempotence: the second run no longer found
# the original id and recreated the concept alongside the renamed one.
("ex:BC_04_03_001", "BC-04.03-001", "Country", "ex:DD_04", "ex:ST_DD_04",
"A sovereign territory used as the geographic frame for retail measurement, "
"identified by its ISO 3166 code. Distinct from Customer and Outlet: it is where "
"they operate, not what they are."),
("ex:BC_16_03_001", "BC-16.03-001", "Marketing Entity", "ex:DD_16", "ex:ST_DD_16",
"An organisational unit of the group holding commercial responsibility for a market, "
"and the level at which financial results are consolidated."),
("ex:BC_16_02_001", "BC-16.02-001", "Fiscal Period", "ex:DD_16", "ex:ST_DD_16",
"A reporting interval of the Pernod Ricard fiscal year, which runs July to June. "
"Distinct from Calendar Date: the fiscal frame does not align with the Gregorian one."),
]
# ---- grain of the DIMENSION objects (fact grain lives in references) -------
GRAIN = {
"ex:DO_10_01_001": ["ex:DE_10_01_0005"], # Product Dimension -> SKU Code
"ex:DO_04_01_001": ["ex:DE_04_01_0005"], # Customer Dimension -> Customer Tier-2 Code
"ex:DO_16_01_001": ["ex:DE_16_01_0001"], # Currency Dimension -> Currency Code
"ex:DO_21_01_001": ["ex:DE_21_01_0001"], # Calendar Dimension -> Calendar Date
}
UNIT_FIX = {"9L / L": "9L", "Currency EUR/USD/LC": "EUR", "%": "%",
"Index": "Index", "Days": "Days"}
def iri(ident):
"""BO-06.01-001 -> ex:BO_06_01_001"""
return "ex:" + ident.replace("-", "_", 1).replace(".", "_").replace("-", "_")
def esc(text):
return text.replace("\\", "\\\\").replace('"', '\\"')
def _xlsx_rows(path):
"""
Minimal .xlsx reader: zipfile + ElementTree, no third-party dependency.
openpyxl would do this in three lines, but the validation venv is already
pinned tightly (pyshacl 0.26.0 for Python 3.9) and adding a dependency to a
migration script that runs once is a poor trade. Handles what a back-doc
needs: shared strings, inline strings, numbers, and sheet names.
"""
import zipfile
import xml.etree.ElementTree as ET
NS = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
REL = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}"
PKG = "{http://schemas.openxmlformats.org/package/2006/relationships}"
with zipfile.ZipFile(path) as z:
shared = []
if "xl/sharedStrings.xml" in z.namelist():
for si in ET.fromstring(z.read("xl/sharedStrings.xml")):
shared.append("".join(t.text or "" for t in si.iter(NS + "t")))
rels = {}
for rel in ET.fromstring(z.read("xl/_rels/workbook.xml.rels")):
rels[rel.get("Id")] = rel.get("Target").lstrip("/")
sheets = []
for sh in ET.fromstring(z.read("xl/workbook.xml")).iter(NS + "sheet"):
target = rels.get(sh.get(REL + "id"), "")
if not target.startswith("xl/"):
target = "xl/" + target
sheets.append((sh.get("name"), target))
for name, target in sheets:
if target not in z.namelist():
continue
rows = []
for row in ET.fromstring(z.read(target)).iter(NS + "row"):
cells = []
for c in row.iter(NS + "c"):
v = c.find(NS + "v")
if c.get("t") == "s" and v is not None:
cells.append(shared[int(v.text)])
elif c.get("t") == "inlineStr":
cells.append("".join(t.text or "" for t in c.iter(NS + "t")))
else:
cells.append(v.text if v is not None else "")
rows.append([(x or "").strip() for x in cells])
yield name, rows
def _text_rows(path):
"""Tab-separated fallback, for back-docs exported as plain text."""
sheet, rows = None, []
for raw in open(path, encoding="utf-8", errors="replace"):
if raw.startswith("## Sheet:"):
if sheet:
yield sheet, rows
sheet, rows = raw.split(":", 1)[1].strip(), []
elif sheet is not None:
rows.append([c.strip() for c in raw.rstrip("\n").split("\t")])
if sheet:
yield sheet, rows
def parse_backdoc(path):
"""
Read the back-doc, whichever form it takes.
A real .xlsx is a ZIP (magic PK\x03\x04); some pipelines hand over a
tab-separated text export of the same content. Sniff rather than assume:
guessing from the extension is what made this script fail the first time.
"""
with open(path, "rb") as fh:
is_zip = fh.read(4) == b"PK\x03\x04"
reader = _xlsx_rows if is_zip else _text_rows
out = {"metrics": [], "elements": []}
for name, rows in reader(path):
key = None
if name.strip().startswith("6"):
key = "metrics"
elif name.strip().startswith("7"):
key = "elements"
if not key:
continue
for cells in rows:
if len(cells) < 5 or not cells[0]:
continue
if key == "metrics" and cells[0].startswith("M-"):
out[key].append(cells)
elif key == "elements" and cells[0].startswith("DE-"):
out[key].append(cells)
if not out["metrics"] or not out["elements"]:
raise SystemExit(
"Back-doc read but empty: %d metrics, %d elements.\n"
"Expected sheets starting with '6.' (Mother Metrics) and '7.' "
"(Data Elements) in %s" % (len(out["metrics"]), len(out["elements"]), path))
return out
def concept_for(de_id):
if de_id in DE_CONCEPT_EXACT:
return DE_CONCEPT_EXACT[de_id]
for prefix, bc in DE_CONCEPT_PREFIX:
if de_id.startswith(prefix):
return bc
return None
def build_metrics(metrics):
out = ["# --- mother metrics, BR-013 -----------------------------------------",
"# One harmonized calculation rule each. The granular variants are Data",
"# Elements linked by computedBy. No physical name (layer leak) and no",
"# granularity (a calculation rule has no rows).", ""]
for r in metrics:
mid, name, bo, formula, unit = r[0], r[1], r[2], r[3], r[4]
unit = UNIT_FIX.get(unit.split(" (")[0], unit.split(" (")[0])
out.append("%s a pr:Metric ;" % iri(mid))
out.append(' pr:hasIdentifier "%s" ; pr:hasName "%s" ;' % (mid, esc(name)))
out.append(" pr:owningDomain ex:DD_06 ; pr:ownedBy ex:ST_DD_06 ;")
out.append(' pr:hasFormula "%s" ;' % esc(formula))
out.append(' pr:hasUnit "%s" ;' % esc(unit))
out.append(" pr:measures %s ;" % METRIC_CONCEPT[mid])
out.append(' pr:hasStatus "DRAFT" ; pr:hasVersion "1.1" ; '
'pr:hasSource "SODH back-doc v0.7" .')
return "\n".join(out)
def build_elements(elements):
measure, dimensional = [], []
for r in elements:
(deid, name, domain, computed, fmt, unit, source, phys, do) = r[:9]
dom = "ex:DD_" + deid.split("-")[1].split(".")[0]
block = ["%s a pr:DataElement ;" % iri(deid),
' pr:hasIdentifier "%s" ; pr:hasName "%s" ;' % (deid, esc(name)),
" pr:owningDomain %s ;" % dom,
' pr:hasFormat "%s" ;' % esc(fmt)]
if unit and unit != "-":
block.append(' pr:hasUnit "%s" ;' % esc(unit))
if phys and phys != "-":
block.append(' pr:physicalName "%s" ;' % esc(phys))
if computed == "(dimensional)":
bc = concept_for(deid)
if not bc:
raise SystemExit("No concept mapped for %s -- the XOR would fail." % deid)
block.append(" pr:represents %s ;" % bc)
block.append(' pr:hasStatus "DRAFT" ; pr:hasVersion "1.1" ; '
'pr:hasSource "%s" .' % esc(source))
dimensional.append("\n".join(block))
else:
block.append(' pr:hasStatus "DRAFT" ; pr:hasVersion "1.1" ; '
'pr:hasSource "%s" .' % esc(source))
measure.append("\n".join(block))
return measure, dimensional
def build_computed_by(elements):
"""Metric -> its granular elements. Asserted from the metric side."""
by_metric = {}
name_to_id = {r[1]: r[0] for r in parse_backdoc(BACKDOC)["metrics"]}
for r in elements:
if r[3] == "(dimensional)":
continue
mid = name_to_id.get(r[3])
if mid:
by_metric.setdefault(mid, []).append(iri(r[0]))
out = ["# --- BR-013 wiring: each mother metric and the variants that implement it",
""]
for mid, des in by_metric.items():
chunks = [des[i:i + 4] for i in range(0, len(des), 4)]
lines = [" , ".join(c) for c in chunks]
out.append("%s pr:computedBy %s ." % (iri(mid), " ,\n ".join(lines)))
return "\n".join(out), sum(len(v) for v in by_metric.values())
def rebuild_has_element(text, elements):
"""
Re-attach every Data Element to its Data Object, from the back-doc column.
Without this the 95 generated measure elements would be orphans: no path to
the physical layer, and no steward, since stewardship is inherited through
the object. Not caught by any shape -- which is why it is worth doing here
rather than waiting for the validator to complain.
"""
by_do = {}
for r in elements:
do = r[8].strip()
if do and do != "-":
by_do.setdefault(iri(do), []).append(iri(r[0]))
n = 0
for do, des in by_do.items():
pat = re.compile(r'(^%s a pr:DataObject ;.*?)\n\s*pr:hasElement[^;]*;' % re.escape(do),
re.S | re.M)
chunks = [des[i:i + 4] for i in range(0, len(des), 4)]
clause = "\n pr:hasElement " + " ,\n ".join(" , ".join(c) for c in chunks) + " ;"
if pat.search(text):
text = pat.sub(lambda m: m.group(1) + clause, text, count=1)
n += 1
return text, n, {k: len(v) for k, v in by_do.items()}
def rebuild_has_metric(text, metrics):
"""
Re-point hasMetric on the Business Objects at the mother metrics.
Removing the old metric definitions is not enough: the Business Objects
still name them, and because hasMetric has rdfs:range pr:Metric, RDFS
entailment types those dangling IRIs as metrics. They then become focus
nodes carrying no identifier, no name, no formula -- ten ghosts, six
violations each. Deleting a subject means deleting what points at it.
"""
by_bo = {}
for r in metrics:
by_bo.setdefault(iri(r[2]), []).append(iri(r[0]))
text = re.sub(r'\n\s*pr:hasMetric[^;]*;', '', text)
n = 0
for bo, ms in by_bo.items():
pat = re.compile(r'(^%s a pr:BusinessObject ;.*?)(\n\s*pr:hasStatus)' % re.escape(bo),
re.S | re.M)
if pat.search(text):
text = pat.sub(r'\1\n pr:hasMetric %s ;\2' % " , ".join(ms), text, count=1)
n += 1
return text, n, by_bo
def concept_block(c):
i, ident, name, dom, st, definition = c
return ("%s a pr:BusinessConcept ;\n"
' pr:hasIdentifier "%s" ; pr:hasName "%s" ;\n'
" pr:owningDomain %s ; pr:ownedBy %s ;\n"
' pr:hasBusinessDefinition "%s" ;\n'
' pr:arbitrationStatus "TO_ARBITRATE" ;\n'
' pr:hasStatus "DRAFT" ; pr:hasVersion "1.1" ; '
'pr:hasSource "DGO proposal, pending ratification by the owning domain" .'
% (i, ident, esc(name), dom, st, esc(definition)))
def main():
apply_changes = "--apply" in sys.argv
for f in (TTL, BACKDOC):
if not os.path.exists(f):
print("Not found: %s" % f)
sys.exit(2)
doc = parse_backdoc(BACKDOC)
text = original = open(TTL, encoding="utf-8").read()
report = []
# 1. drop the old metric and data element blocks
n_old_m = len(re.findall(r'^ex:M_\w+ a pr:Metric ;', text, re.M))
n_old_de = len(re.findall(r'^ex:DE_\w+ a pr:DataElement ;', text, re.M))
text = re.sub(r'^ex:M_\w+ a pr:Metric ;.*?\.\s*\n(?=^ex:|\Z)', '', text, flags=re.S | re.M)
text = re.sub(r'^ex:DE_\w+ a pr:DataElement ;.*?\.\s*\n(?=^ex:|\Z)', '', text, flags=re.S | re.M)
report.append("removed %d old metrics and %d old data elements" % (n_old_m, n_old_de))
# 2. steward is inherited, not declared on a Data Object
n_st = 0
for m in re.finditer(r'^ex:DO_\w+ a pr:DataObject ;.*?\.\s*\n', text, re.S | re.M):
block = m.group(0)
new = re.sub(r'\s*pr:monitoredBy\s+ex:\w+\s*;', ' ;', block)
new = re.sub(r';\s*;', ' ;', new)
if new != block:
n_st += 1
text = text.replace(block, new, 1)
report.append("removed monitoredBy from %d Data Objects (inherited now)" % n_st)
# 3. grain, dimension objects only
n_g = 0
for do, grain in GRAIN.items():
pat = re.compile(r'(^%s a pr:DataObject ;.*?)(\n\s*pr:hasStatus)' % re.escape(do),
re.S | re.M)
if pat.search(text):
text = pat.sub(r'\1\n pr:hasGrainElement %s ;\2' % " , ".join(grain), text, count=1)
n_g += 1
report.append("grain declared on %d dimension Data Objects "
"(fact grain stays in references)" % n_g)
# 4. the three concepts the XOR forces us to name
# idempotent: a concept already in the file is left alone. Duplicated
# blocks would be INVISIBLE to SHACL -- identical triples merge under RDF
# set semantics, so the graph validates while the file carries redundant
# text. A script that can be re-run must check before it inserts.
todo = [c for c in NEW_CONCEPTS
if not re.search(r'^%s a ' % re.escape(c[0]), text, re.M)]
if len(todo) < len(NEW_CONCEPTS):
report.append("skipped %d concept(s) already present"
% (len(NEW_CONCEPTS) - len(todo)))
anchor = re.search(r'\n(?=ex:BO_\w+ a pr:BusinessObject ;)', text)
blocks = "\n".join(concept_block(c) for c in todo)
if todo:
text = (text[:anchor.start()] + "\n\n"
+ "# --- concepts required by the meaning XOR ---------------------------\n"
+ "# Country, Marketing Entity and Fiscal Period had Data Elements but no\n"
+ "# notion behind them. The rule forced them to be named.\n"
+ blocks + "\n" + text[anchor.start():])
report.append("added %d concepts required by the XOR (TO_ARBITRATE)" % len(todo))
# 5. metrics, elements, wiring
measure, dimensional = build_elements(doc["elements"])
wiring, n_wired = build_computed_by(doc["elements"])
text = text.rstrip() + "\n\n\n" + build_metrics(doc["metrics"]) + "\n\n"
text += ("# --- dimensional data elements --------------------------------------\n"
"# Each represents the Concept it carries: the first of the two routes\n"
"# to business meaning.\n\n" + "\n".join(dimensional) + "\n\n")
text += ("# --- granular measure data elements ---------------------------------\n"
"# Same calculation rule as their mother metric, different analysis\n"
"# context. They reach meaning through computedBy, never directly.\n\n"
+ "\n".join(measure) + "\n\n" + wiring + "\n")
report.append("wrote %d mother metrics, %d dimensional and %d measure elements"
% (len(doc["metrics"]), len(dimensional), len(measure)))
report.append("wired %d computedBy links" % n_wired)
text, n_bo, per_bo = rebuild_has_metric(text, doc["metrics"])
report.append("re-pointed hasMetric on %d Business Objects (%s)"
% (n_bo, ", ".join("%s=%d" % (k.replace("ex:BO_", "BO-"), len(v))
for k, v in sorted(per_bo.items()))))
text, n_do, counts = rebuild_has_element(text, doc["elements"])
report.append("re-attached elements to %d Data Objects (%s)"
% (n_do, ", ".join("%s=%d" % (k.replace("ex:DO_", "DO-"), v)
for k, v in sorted(counts.items()))))
print()
print("SODH BR-013 ALIGNMENT %s" % ("APPLY" if apply_changes else "DRY RUN"))
print("=" * W)
for line in report:
print(" " + line)
print("=" * W)
defined = set(re.findall(r'^(ex:M_\w+) a pr:Metric ;', text, re.M))
referenced = set(re.findall(r'ex:M_\w+', text))
dangling = referenced - defined
print(" dangling metric references: %s"
% (", ".join(sorted(dangling)) if dangling else "none"))
print(" Metrics %d | Data Elements %d | Concepts %d | computedBy %d | "
"physicalName on Metric %d"
% (len(re.findall(r'a pr:Metric', text)),
len(re.findall(r'a pr:DataElement', text)),
len(re.findall(r'a pr:BusinessConcept', text)),
n_wired,
len(re.findall(r'a pr:Metric ;[^.]*?physicalName', text, re.S))))
print("=" * W)
if text != original and apply_changes:
shutil.copy2(TTL, TTL + ".bak")
open(TTL, "w", encoding="utf-8").write(text)
print(" written, backup at %s.bak" % os.path.basename(TTL))
elif text != original:
print(" dry run -- re-run with --apply to write")
print()
if __name__ == "__main__":
main()