diff --git a/governance/generate_bpmn.py b/governance/generate_bpmn.py new file mode 100644 index 0000000..943e1d9 --- /dev/null +++ b/governance/generate_bpmn.py @@ -0,0 +1,206 @@ +#!/usr/bin/env python3 +""" +Generate strict, queryable BPMN 2.0 files from processes.yaml (EV-008). + +USAGE + python3 generate_bpmn.py [output_dir] + +One file per procedure, in bpmn/. Rule identifiers are carried in +extensionElements under the project namespace, so a rule can be traced to every +procedure that enforces it with a single XPath expression: + + //bpmn:process[.//pr:rule='TN-002']/@id + +Diagram interchange is emitted with a simple left-to-right layout so the files +open directly in any BPMN editor. Layout is cosmetic and never authoritative. +""" +import os +import sys +from xml.sax.saxutils import escape + +import yaml + +HERE = os.path.dirname(os.path.abspath(__file__)) +SRC = os.path.join(HERE, "processes.yaml") +OUTDIR = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "bpmn") + +BPMN = "http://www.omg.org/spec/BPMN/20100524/MODEL" +BPMNDI = "http://www.omg.org/spec/BPMN/20100524/DI" +DI = "http://www.omg.org/spec/DD/20100524/DI" +DC = "http://www.omg.org/spec/DD/20100524/DC" + +TAG = { + "start": "startEvent", + "end": "endEvent", + "userTask": "userTask", + "scriptTask": "scriptTask", + "callActivity": "callActivity", + "gateway": "exclusiveGateway", +} +W = {"start": 36, "end": 36, "gateway": 50} +H = {"start": 36, "end": 36, "gateway": 50} +DEFAULT_W, DEFAULT_H = 150, 70 +COL, ROW = 200, 130 + + +def selfcheck(doc, rule_ids): + """A procedure must be a connected graph, and every rule it cites must exist.""" + problems = [] + ids = {p["id"] for p in doc["processes"]} + for p in doc["processes"]: + nodes = {n["id"] for n in p["flow"]} + for f in p["flows"]: + for side in ("from", "to"): + if f[side] not in nodes: + problems.append("%s: flow references unknown node %s" % (p["id"], f[side])) + targets = {f["to"] for f in p["flows"]} + sources = {f["from"] for f in p["flows"]} + for n in p["flow"]: + if n["type"] == "start" and n["id"] in targets: + problems.append("%s: start event %s has an incoming flow" % (p["id"], n["id"])) + if n["type"] == "end" and n["id"] in sources: + problems.append("%s: end event %s has an outgoing flow" % (p["id"], n["id"])) + if n["type"] not in ("start", "end") and n["id"] not in targets: + problems.append("%s: node %s is unreachable" % (p["id"], n["id"])) + if n["type"] != "end" and n["id"] not in sources: + problems.append("%s: node %s is a dead end" % (p["id"], n["id"])) + if n["type"] == "callActivity" and n.get("calls") not in ids: + problems.append("%s: %s calls unknown procedure %s" + % (p["id"], n["id"], n.get("calls"))) + for r in n.get("rules") or []: + if rule_ids and r not in rule_ids: + problems.append("%s/%s: unknown rule %s" % (p["id"], n["id"], r)) + return problems + + +def layout(proc): + """Assign a column by longest path from start, a row to disambiguate siblings.""" + succ = {} + for f in proc["flows"]: + succ.setdefault(f["from"], []).append(f["to"]) + starts = [n["id"] for n in proc["flow"] if n["type"] == "start"] + depth = {} + for s in starts: + stack = [(s, 0)] + seen = set() + while stack: + node, d = stack.pop() + if (node, d) in seen: + continue + seen.add((node, d)) + if depth.get(node, -1) < d: + depth[node] = d + for nxt in succ.get(node, []): + if d < 60: + stack.append((nxt, d + 1)) + used = {} + pos = {} + for n in proc["flow"]: + c = depth.get(n["id"], 0) + r = used.get(c, 0) + used[c] = r + 1 + pos[n["id"]] = (60 + c * COL, 80 + r * ROW) + return pos + + +def build(proc, ns): + pid = proc["id"] + o = [] + a = o.append + a('') + a('' % (BPMN, BPMNDI, DI, DC, ns, pid, ns)) + a(' ' + % (pid, escape(proc["name"]))) + doc = "%s | Scope: %s | Trigger: %s" % ( + proc["name"], proc.get("scope", ""), proc.get("trigger", "")) + a(' %s' % escape(doc)) + + incoming, outgoing = {}, {} + for i, f in enumerate(proc["flows"], start=1): + fid = "flow_%d" % i + f["_id"] = fid + outgoing.setdefault(f["from"], []).append(fid) + incoming.setdefault(f["to"], []).append(fid) + + for n in proc["flow"]: + tag = TAG[n["type"]] + attrs = 'id="%s" name="%s"' % (n["id"], escape(n["name"])) + if n["type"] == "callActivity": + attrs += ' calledElement="%s"' % n["calls"] + a(' ' % (tag, attrs)) + rules = n.get("rules") or [] + if rules: + a(' ') + a(' ') + for r in rules: + a(' %s' % r) + a(' ') + a(' ') + for fid in incoming.get(n["id"], []): + a(' %s' % fid) + for fid in outgoing.get(n["id"], []): + a(' %s' % fid) + a(' ' % tag) + + for f in proc["flows"]: + nm = ' name="%s"' % escape(f["condition"]) if f.get("condition") else "" + a(' ' + % (f["_id"], nm, f["from"], f["to"])) + a(' ') + + pos = layout(proc) + a(' ' % pid) + a(' ' % (pid, pid)) + for n in proc["flow"]: + x, y = pos[n["id"]] + w, h = W.get(n["type"], DEFAULT_W), H.get(n["type"], DEFAULT_H) + a(' ' % (n["id"], n["id"])) + a(' ' % (x, y, w, h)) + a(' ') + for f in proc["flows"]: + sx, sy = pos[f["from"]] + tx, ty = pos[f["to"]] + sw = W.get(next(n["type"] for n in proc["flow"] if n["id"] == f["from"]), DEFAULT_W) + sh = H.get(next(n["type"] for n in proc["flow"] if n["id"] == f["from"]), DEFAULT_H) + th = H.get(next(n["type"] for n in proc["flow"] if n["id"] == f["to"]), DEFAULT_H) + a(' ' % (f["_id"], f["_id"])) + a(' ' % (sx + sw, sy + sh // 2)) + a(' ' % (tx, ty + th // 2)) + a(' ') + a(' ') + a(' ') + a('') + return "\n".join(o) + "\n" + + +def main(): + doc = yaml.safe_load(open(SRC, encoding="utf-8")) + rule_ids = set() + rb = os.path.join(HERE, "rules.yaml") + if os.path.exists(rb): + rule_ids = {r["id"] for r in yaml.safe_load(open(rb, encoding="utf-8"))["rules"]} + + problems = selfcheck(doc, rule_ids) + if problems: + print("SELF-CHECK FAILED") + for p in problems: + print(" " + p) + sys.exit(1) + + os.makedirs(OUTDIR, exist_ok=True) + ns = doc["meta"]["namespace"] + for proc in doc["processes"]: + path = os.path.join(OUTDIR, "%s.bpmn" % proc["id"]) + open(path, "w", encoding="utf-8").write(build(proc, ns)) + n_rules = len({r for n in proc["flow"] for r in (n.get("rules") or [])}) + print(" %-4s %-46s %2d nodes, %2d flows, %2d rules" + % (proc["id"], proc["name"], len(proc["flow"]), + len(proc["flows"]), n_rules)) + print("self-check passed: %d procedures, graphs connected, every rule known" + % len(doc["processes"])) + + +if __name__ == "__main__": + main() diff --git a/governance/generate_processbook_md.py b/governance/generate_processbook_md.py new file mode 100644 index 0000000..dfe5abc --- /dev/null +++ b/governance/generate_processbook_md.py @@ -0,0 +1,157 @@ +#!/usr/bin/env python3 +""" +Generate the Markdown processbook, with Mermaid views, from processes.yaml +(EV-008). The BPMN files remain the authoritative form; the views here are +derived from the same source and cannot drift from it. + +USAGE + python3 generate_processbook_md.py [output.md] +""" +import os +import sys + +import yaml + +HERE = os.path.dirname(os.path.abspath(__file__)) +SRC = os.path.join(HERE, "processes.yaml") +OUT = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "PR_TBox_Processbook.md") + +SHAPE = { + "start": ("([", "])"), + "end": ("([", "])"), + "userTask": ("[", "]"), + "scriptTask": ("[", "]"), + "callActivity": ("[[", "]]"), + "gateway": ("{", "}"), +} +KIND = { + "userTask": "human", + "scriptTask": "script", + "callActivity": "calls", + "gateway": "decision", + "start": "start", + "end": "end", +} + + +def flow(t): + return " ".join((t or "").split()) + + +def mermaid(proc): + out = ["```mermaid", "flowchart TD"] + for n in proc["flow"]: + o, c = SHAPE[n["type"]] + label = n["name"].replace('"', "'") + if n["type"] == "callActivity": + label = "%s: %s" % (n["calls"], label) + out.append(' %s%s"%s"%s' % (n["id"], o, label, c)) + for f in proc["flows"]: + if f.get("condition"): + out.append(' %s -->|%s| %s' % (f["from"], f["condition"], f["to"])) + else: + out.append(" %s --> %s" % (f["from"], f["to"])) + for n in proc["flow"]: + if n["type"] == "userTask": + out.append(" class %s human;" % n["id"]) + elif n["type"] == "gateway": + out.append(" class %s decision;" % n["id"]) + out.append(" classDef human fill:#FFF4CE,stroke:#B08900;") + out.append(" classDef decision fill:#E8F0F7,stroke:#3E6E96;") + out.append("```") + return out + + +def main(): + doc = yaml.safe_load(open(SRC, encoding="utf-8")) + m, out = doc["meta"], [] + w = out.append + + w("# %s" % m["title"]) + w("") + w("**Version %s** — %s — generated %s" % (m["version"], m["status"], m["date"])) + w("") + w("> Generated from `processes.yaml`. The BPMN files in `bpmn/` are the " + "authoritative form; the views below are derived from the same source. " + "Do not edit this document (EV-007, EV-008).") + w("") + w(flow(m["scope"])) + w("") + w("## Procedures") + w("") + w("| Id | Family | Procedure | Scope |") + w("|---|---|---|---|") + for p in doc["processes"]: + w("| **%s** | %s | %s | %s |" % (p["id"], p["family"], p["name"], p.get("scope", ""))) + w("") + w("## Reading a diagram") + w("") + w("- A **rounded** node is a start or end state.") + w("- A **diamond** is a decision; every outgoing path is labelled.") + w("- A **shaded** box is a human task: it cannot be automated, and the work " + "stops there until someone acts.") + w("- A **double-bordered** box calls another procedure by its identifier.") + w("") + + for p in doc["processes"]: + w("---") + w("") + w("# %s — %s" % (p["id"], p["name"])) + w("") + w("| | |") + w("|---|---|") + w("| Family | %s |" % p["family"]) + w("| Scope | %s |" % p.get("scope", "")) + w("| Trigger | %s |" % p.get("trigger", "")) + w("| Inputs | %s |" % ", ".join(p.get("inputs") or [])) + w("| Outputs | %s |" % ", ".join(p.get("outputs") or [])) + rules = sorted({r for n in p["flow"] for r in (n.get("rules") or [])}) + w("| Rules enforced | %s |" % (", ".join(rules) or "-")) + calls = sorted({n["calls"] for n in p["flow"] if n["type"] == "callActivity"}) + w("| Calls | %s |" % (", ".join(calls) or "-")) + w("") + if p.get("parameter"): + w("**Parameter.** %s" % flow(p["parameter"])) + w("") + if p.get("note"): + w("**Note.** %s" % flow(p["note"])) + w("") + out.extend(mermaid(p)) + w("") + w("| Step | Type | Rules |") + w("|---|---|---|") + for n in p["flow"]: + if n["type"] in ("start", "end"): + continue + name = n["name"] + if n["type"] == "callActivity": + name = "%s (%s)" % (name, n["calls"]) + w("| %s | %s | %s |" + % (name, KIND[n["type"]], ", ".join(n.get("rules") or []) or "-")) + w("") + + w("---") + w("") + w("# Rule coverage") + w("") + w("Which procedures enforce each rule. Derived, not maintained: a rule cited " + "in a diagram appears here automatically.") + w("") + cover = {} + for p in doc["processes"]: + for n in p["flow"]: + for r in n.get("rules") or []: + cover.setdefault(r, set()).add(p["id"]) + w("| Rule | Procedures |") + w("|---|---|") + for r in sorted(cover): + w("| %s | %s |" % (r, ", ".join(sorted(cover[r])))) + w("") + + open(OUT, "w", encoding="utf-8").write("\n".join(out)) + print("written: %s (%d lines, %d procedures, %d rules covered)" + % (OUT, len(out), len(doc["processes"]), len(cover))) + + +if __name__ == "__main__": + main() diff --git a/governance/generate_rulebook_md.py b/governance/generate_rulebook_md.py new file mode 100644 index 0000000..25c4326 --- /dev/null +++ b/governance/generate_rulebook_md.py @@ -0,0 +1,154 @@ +#!/usr/bin/env python3 +""" +Generate the Markdown rulebook from rules.yaml (EV-008). + +USAGE + python3 generate_rulebook_md.py [output.md] + +rules.yaml is the single source of truth. This script never edits it. +The output must never be edited by hand (EV-007). +""" +import os +import sys + +import yaml + +HERE = os.path.dirname(os.path.abspath(__file__)) +SRC = os.path.join(HERE, "rules.yaml") +OUT = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "PR_TBox_Rulebook.md") + + +def selfcheck(doc): + """EV-011 and EV-012 applied to the rulebook itself.""" + problems = [] + for r in doc["rules"]: + ctl = r.get("control") or {} + if not ctl.get("tier"): + problems.append("%s: no control tier (EV-012)" % r["id"]) + if r.get("severity") == "BLOCKING" and not ctl.get("executor"): + problems.append("%s: BLOCKING without an executor (EV-011)" % r["id"]) + if r.get("category") not in doc["categories"]: + problems.append("%s: unknown category" % r["id"]) + return problems + + +def flow(text): + """Collapse a YAML folded scalar into a single paragraph.""" + return " ".join((text or "").split()) + + +def joined(v): + return ", ".join(v) if isinstance(v, list) else (v or "-") + + +def main(): + doc = yaml.safe_load(open(SRC, encoding="utf-8")) + problems = selfcheck(doc) + if problems: + print("SELF-CHECK FAILED") + for p in problems: + print(" " + p) + sys.exit(1) + + m, out = doc["meta"], [] + w = out.append + + w("# %s" % m["title"]) + w("") + w("**Version %s** — %s — generated %s" % (m["version"], m["status"], m["date"])) + w("") + w("> Generated from `rules.yaml`. Do not edit this document: edit the source " + "and regenerate (EV-007, EV-008).") + w("") + w("## Scope") + w("") + w(flow(m["scope"])) + w("") + w(flow(m["audience"])) + w("") + w("## Categories") + w("") + w("| Category | Subject | Rules |") + w("|---|---|---|") + for k, v in doc["categories"].items(): + n = sum(1 for r in doc["rules"] if r["category"] == k) + w("| **%s** | %s | %d |" % (k, v["title"], n)) + w("") + w("## Severity") + w("") + for k, v in doc["severity_model"].items(): + w("- **%s** — %s" % (k, flow(v))) + w("") + + for cat, info in doc["categories"].items(): + w("---") + w("") + w("# %s — %s" % (cat, info["title"])) + w("") + w(flow(info["intent"])) + w("") + for r in [x for x in doc["rules"] if x["category"] == cat]: + ctl = r.get("control") or {} + w("## %s — %s" % (r["id"], r["title"])) + w("") + w("**Statement.** %s" % flow(r["statement"])) + w("") + w("| | |") + w("|---|---|") + w("| Severity | **%s** |" % r["severity"]) + w("| Applies to | %s |" % joined(r.get("scope"))) + w("| Enforced at | %s |" % joined(ctl.get("tier"))) + w("| Executor | `%s` |" % (ctl.get("executor") or "-")) + w("| Procedure | %s |" % (ctl.get("procedure") or "_pending_")) + if r.get("filiation"): + w("| Related | %s |" % r["filiation"]) + w("") + if r.get("rationale"): + w("**Why.** %s" % flow(r["rationale"])) + w("") + if r.get("examples"): + w("| From | To | Note |") + w("|---|---|---|") + for e in r["examples"]: + w("| %s | %s | %s |" % (e["from"], e["to"], e.get("note") or "")) + w("") + + ab = doc["abstractness"] + w("---") + w("") + w("# Annex A — Abstractness") + w("") + w("Applies TN-006 and TN-007 to the model. %d classes, of which **%d abstract** " + "and **%d concrete**. Maintained with the model, not after it." + % (len(ab), sum(1 for a in ab if a["is_abstract"]), + sum(1 for a in ab if not a["is_abstract"]))) + w("") + w("| Class | isAbstract | Note |") + w("|---|---|---|") + for a in ab: + w("| `%s` | %s | %s |" + % (a["term"], "**true**" if a["is_abstract"] else "false", a.get("note") or "")) + w("") + w("---") + w("") + w("# Annex B — Display") + w("") + w("Applies TN-011 and TN-022. The short label is what screens display and what " + "relation names reuse; the acronym is a search key, never an identity.") + w("") + w("| IRI | rdfs:label | shortLabel | acronym | Note |") + w("|---|---|---|---|---|") + for d in doc["display"]: + w("| `%s` | %s | %s | %s | %s |" + % (d["iri"], d["label"], d.get("short_label") or "-", + d.get("acronym") or "-", d.get("note") or "")) + w("") + + open(OUT, "w", encoding="utf-8").write("\n".join(out)) + print("self-check passed: %d rules, every tier assigned, no blocking rule " + "without an executor" % len(doc["rules"])) + print("written: %s (%d lines)" % (OUT, len(out))) + + +if __name__ == "__main__": + main() diff --git a/governance/generate_rulebook_xlsx.py b/governance/generate_rulebook_xlsx.py new file mode 100644 index 0000000..e6c8597 --- /dev/null +++ b/governance/generate_rulebook_xlsx.py @@ -0,0 +1,218 @@ +#!/usr/bin/env python3 +""" +Generate the Excel rulebook from rules.yaml (EV-008). + +USAGE + python3 generate_rulebook_xlsx.py [output.xlsx] + +rules.yaml is the single source of truth. This script never edits it. +The output must never be edited by hand (EV-007). +""" +import os +import sys + +import yaml +from openpyxl import Workbook +from openpyxl.styles import Alignment, Border, Font, PatternFill, Side +from openpyxl.utils import get_column_letter + +HERE = os.path.dirname(os.path.abspath(__file__)) +SRC = os.path.join(HERE, "rules.yaml") +OUT = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "PR_TBox_Rulebook.xlsx") + +FONT = "Arial" +INK = "1F2933" +HEAD_FILL = PatternFill("solid", fgColor="1F2933") +BAND = { + "Identifier": "E8F0F7", + "Label": "EAF3EC", + "Declaration": "FBF0E4", + "Evolution": "F2EAF5", +} +MAJOR_FILL = PatternFill("solid", fgColor="FFF4CE") +THIN = Side(style="thin", color="C9CFD6") +BORDER = Border(left=THIN, right=THIN, top=THIN, bottom=THIN) + + +def selfcheck(doc): + """EV-011 and EV-012 applied to the rulebook itself.""" + problems = [] + for r in doc["rules"]: + ctl = r.get("control") or {} + if not ctl.get("tier"): + problems.append("%s: no control tier (EV-012)" % r["id"]) + if r.get("severity") == "BLOCKING" and not ctl.get("executor"): + problems.append("%s: BLOCKING without an executor (EV-011)" % r["id"]) + if r.get("category") not in doc["categories"]: + problems.append("%s: unknown category" % r["id"]) + return problems + + +def flow(t): + return " ".join((t or "").split()) + + +def joined(v): + return ", ".join(v) if isinstance(v, list) else (v if v else "") + + +def header(ws, cols): + for i, (title, width) in enumerate(cols, start=1): + c = ws.cell(row=1, column=i, value=title) + c.font = Font(name=FONT, size=10, bold=True, color="FFFFFF") + c.fill = HEAD_FILL + c.alignment = Alignment(vertical="center", horizontal="left", wrap_text=True) + c.border = BORDER + ws.column_dimensions[get_column_letter(i)].width = width + ws.row_dimensions[1].height = 28 + ws.freeze_panes = ws.cell(row=2, column=1) + ws.auto_filter.ref = "A1:%s1" % get_column_letter(len(cols)) + + +def put(ws, r, c, value, wrap=False, bold=False, fill=None, size=10): + cell = ws.cell(row=r, column=c, value=value) + cell.font = Font(name=FONT, size=size, bold=bold, color=INK) + cell.alignment = Alignment(vertical="top", wrap_text=wrap) + cell.border = BORDER + if fill: + cell.fill = fill + return cell + + +def sheet_cover(wb, doc): + m = doc["meta"] + ws = wb.create_sheet("Cover") + ws.column_dimensions["A"].width = 24 + ws.column_dimensions["B"].width = 100 + put(ws, 1, 1, m["title"], bold=True, size=14) + + rows = [ + ("Version", "%s — %s" % (m["version"], m["status"])), + ("Generated", "%s from rules.yaml" % m["date"]), + ("Scope", flow(m["scope"])), + ("Audience", flow(m["audience"])), + ("Generation", "Generated from rules.yaml by generate_rulebook_xlsx.py. " + "Do not edit this workbook: edit the source and regenerate " + "(EV-007, EV-008)."), + ] + r = 3 + for k, v in rows: + put(ws, r, 1, k, bold=True) + put(ws, r, 2, v, wrap=True) + ws.row_dimensions[r].height = 15 * (len(v) // 95 + 1) + r += 1 + + r += 1 + put(ws, r, 1, "Categories", bold=True, size=12) + r += 1 + for k, v in doc["categories"].items(): + put(ws, r, 1, k, bold=True, fill=PatternFill("solid", fgColor=BAND[k])) + put(ws, r, 2, "%s — %s" % (v["title"], flow(v["intent"])), wrap=True) + ws.row_dimensions[r].height = 30 + r += 1 + + r += 1 + put(ws, r, 1, "Counts", bold=True, size=12) + r += 1 + for label, formula in [ + ("Rules", "=COUNTA(Rules!A2:A200)"), + ("of which blocking", '=COUNTIF(Rules!D2:D200,"BLOCKING")'), + ("Examples", "=COUNTA(Examples!A2:A400)"), + ("Classes", "=COUNTA(Abstractness!A2:A200)"), + ("of which abstract", "=COUNTIF(Abstractness!B2:B200,TRUE)"), + ("Display entries", "=COUNTA(Display!A2:A200)"), + ]: + put(ws, r, 1, label) + put(ws, r, 2, formula) + r += 1 + + +def sheet_rules(wb, doc): + ws = wb.create_sheet("Rules") + header(ws, [("Rule", 10), ("Category", 14), ("Title", 40), ("Severity", 11), + ("Statement", 70), ("Applies to", 24), ("Enforced at", 14), + ("Executor", 26), ("Procedure", 16), ("Related", 16), + ("Why", 90)]) + r = 2 + for x in doc["rules"]: + ctl = x.get("control") or {} + band = PatternFill("solid", fgColor=BAND[x["category"]]) + put(ws, r, 1, x["id"], bold=True, fill=band) + put(ws, r, 2, x["category"], fill=band) + put(ws, r, 3, x["title"], wrap=True, bold=True) + put(ws, r, 4, x["severity"], + fill=MAJOR_FILL if x["severity"] != "BLOCKING" else None) + put(ws, r, 5, flow(x["statement"]), wrap=True) + put(ws, r, 6, joined(x.get("scope")), wrap=True) + put(ws, r, 7, joined(ctl.get("tier"))) + put(ws, r, 8, ctl.get("executor") or "") + put(ws, r, 9, ctl.get("procedure") or "pending") + put(ws, r, 10, x.get("filiation") or "") + put(ws, r, 11, flow(x.get("rationale")), wrap=True) + ws.row_dimensions[r].height = 78 + r += 1 + + +def sheet_examples(wb, doc): + ws = wb.create_sheet("Examples") + header(ws, [("Rule", 10), ("Category", 14), ("From", 48), ("To", 48), ("Note", 62)]) + r = 2 + for x in doc["rules"]: + for e in x.get("examples") or []: + band = PatternFill("solid", fgColor=BAND[x["category"]]) + put(ws, r, 1, x["id"], bold=True, fill=band) + put(ws, r, 2, x["category"], fill=band) + put(ws, r, 3, str(e["from"]), wrap=True) + put(ws, r, 4, str(e["to"]), wrap=True) + put(ws, r, 5, e.get("note") or "", wrap=True) + r += 1 + + +def sheet_abstractness(wb, doc): + ws = wb.create_sheet("Abstractness") + header(ws, [("Class", 36), ("isAbstract", 12), ("Note", 96)]) + r = 2 + for a in doc["abstractness"]: + put(ws, r, 1, a["term"], bold=a["is_abstract"]) + put(ws, r, 2, a["is_abstract"]) + put(ws, r, 3, a.get("note") or "", wrap=True) + r += 1 + + +def sheet_display(wb, doc): + ws = wb.create_sheet("Display") + header(ws, [("IRI", 36), ("rdfs:label", 36), ("shortLabel", 22), + ("acronym", 10), ("Note", 80)]) + r = 2 + for d in doc["display"]: + put(ws, r, 1, d["iri"], bold=True) + put(ws, r, 2, d["label"]) + put(ws, r, 3, d.get("short_label") or "") + put(ws, r, 4, d.get("acronym") or "") + put(ws, r, 5, d.get("note") or "", wrap=True) + r += 1 + + +def main(): + doc = yaml.safe_load(open(SRC, encoding="utf-8")) + problems = selfcheck(doc) + if problems: + print("SELF-CHECK FAILED") + for p in problems: + print(" " + p) + sys.exit(1) + + wb = Workbook() + wb.remove(wb.active) + sheet_cover(wb, doc) + sheet_rules(wb, doc) + sheet_examples(wb, doc) + sheet_abstractness(wb, doc) + sheet_display(wb, doc) + wb.save(OUT) + print("self-check passed: %d rules" % len(doc["rules"])) + print("written: %s (%d sheets)" % (OUT, len(wb.sheetnames))) + + +if __name__ == "__main__": + main() diff --git a/governance/migration/migrate_tbox_v2_0.py b/governance/migration/migrate_tbox_v2_0.py new file mode 100644 index 0000000..9501a1d --- /dev/null +++ b/governance/migration/migrate_tbox_v2_0.py @@ -0,0 +1,357 @@ +#!/usr/bin/env python3 +""" +Migrate the Pernod Ricard Data MetaModel from v1.6 to v2.0. + +USAGE + python3 migrate_tbox_v2_0.py # dry run, writes nothing + python3 migrate_tbox_v2_0.py --apply # writes in place + python3 migrate_tbox_v2_0.py --apply --log-dir logs/ + + --ontology path to the ontology TTL (default ../ontology/pr_metamodel.ttl) + --instances path to an instance TTL (repeatable) + --shapes path to a shapes TTL (repeatable) + +DESIGN + EV-004 dry run is the default; nothing is written without --apply + EV-005 every edit goes through rdflib, never a regex on the text + EV-006 guards test the target state, so a replay reports zero change + EV-015 an execution log is written for every attempt + +Run on GrosseBertha, inside the pinned venv: + . venv/bin/activate && python3 migrate_tbox_v2_0.py +""" +import argparse +import datetime +import hashlib +import json +import os +import re +import sys +from collections import OrderedDict + +import yaml +from rdflib import Graph, Literal, Namespace, RDF, RDFS, OWL, URIRef, XSD + +HERE = os.path.dirname(os.path.abspath(__file__)) +SPEC = os.path.join(HERE, "renames.yaml") +RULES = os.path.normpath(os.path.join(HERE, "..", "rules.yaml")) + +DCTERMS = Namespace("http://purl.org/dc/terms/") + + +# ----------------------------------------------------------------- utilities + +class Report(object): + """Counts every step, so that idempotence is provable and not merely hoped.""" + + def __init__(self): + self.steps = OrderedDict() + self.notes = [] + + def add(self, step, n, detail=None): + self.steps[step] = self.steps.get(step, 0) + n + if detail: + self.notes.append("%s: %s" % (step, detail)) + + @property + def total(self): + return sum(self.steps.values()) + + +def md5(path): + h = hashlib.md5() + with open(path, "rb") as f: + for chunk in iter(lambda: f.read(65536), b""): + h.update(chunk) + return h.hexdigest() + + +def local(uri, ns): + s = str(uri) + return s[len(ns):] if s.startswith(ns) else None + + +def decamelise(name, is_class): + """TN-018. Consecutive capitals are kept together: they carry an acronym.""" + spaced = re.sub(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])", " ", name) + return spaced if is_class else spaced[0].lower() + spaced[1:] + + +# ------------------------------------------------------------------ the work + +def rename(graph, pr, old, new, report, step): + """EV-001. Rewrite every triple naming the identifier, in all three positions.""" + src, dst = pr[old], pr[new] + if (dst, None, None) in graph and (src, None, None) not in graph: + return 0 # EV-006: already done + n = 0 + for s, p, o in list(graph): + ns, np_, no = s, p, o + if s == src: + ns = dst + if p == src: + np_ = dst + if o == src: + no = dst + if (ns, np_, no) != (s, p, o): + graph.remove((s, p, o)) + graph.add((ns, np_, no)) + n += 1 + report.add(step, n) + return n + + +def merge(graph, pr, src_name, dst_name, report): + """The source disappears into an existing target, which keeps its declaration.""" + src, dst = pr[src_name], pr[dst_name] + n = 0 + for s, p, o in list(graph.triples((None, src, None))): + graph.remove((s, p, o)) + graph.add((s, dst, o)) + n += 1 + for s, p, o in list(graph.triples((src, None, None))): + graph.remove((s, p, o)) # drop the source declaration + n += 1 + report.add("merge", n) + return n + + +def drop_subject(graph, subject, report, step): + """EV-003. Remove the block AND every reference naming it, or inference rebuilds it.""" + n = 0 + for t in list(graph.triples((subject, None, None))): + graph.remove(t) + n += 1 + for t in list(graph.triples((None, None, subject))): + graph.remove(t) + n += 1 + for t in list(graph.triples((None, subject, None))): + graph.remove(t) + n += 1 + report.add(step, n) + return n + + +def count_instances(graphs, term): + """EV-002. The proof required before any permanent withdrawal.""" + n = 0 + for g in graphs: + n += len(list(g.triples((None, RDF.type, term)))) + n += len(list(g.triples((None, term, None)))) + n += len(list(g.triples((None, None, term)))) + return n + + +def migrate(onto, others, spec, rules, report): + ns = spec["meta"]["namespace"] + pr = Namespace(ns) + all_graphs = [onto] + others + + # 1 — deprecated terms are withdrawn outright (phase clause), proof first + if spec["structural"].get("drop_deprecated"): + for subj in list(onto.subjects(OWL.deprecated, Literal(True))): + name = local(subj, ns) or str(subj) + used = sum(count_instances([g], subj) for g in others) + if used: + report.add("deprecated_kept", 1, "%s still used %d times" % (name, used)) + continue + for g in all_graphs: + drop_subject(g, subj, report, "deprecated_dropped") + + # 2 — renames, in dependency order + for order in (1, 2, 3): + for r in [x for x in spec["renames"] if x["order"] == order]: + for g in all_graphs: + rename(g, pr, r["from"], r["to"], report, "rename_order_%d" % order) + + # 3 — merges + for m in spec.get("merges") or []: + for g in all_graphs: + merge(g, pr, m["from"], m["into"], report) + + # 4 — reclassify display properties (TN-003) + for rc in spec["structural"]["reclassify"]: + term = pr[rc["term"]] + if (term, RDF.type, OWL.AnnotationProperty) not in onto: + onto.remove((term, RDF.type, getattr(OWL, rc["from"]))) + onto.add((term, RDF.type, OWL.AnnotationProperty)) + report.add("reclassified", 1, rc["term"]) + + # 5 — create the governance layer root (TN-027) + for c in spec["structural"].get("create_classes") or []: + term = pr[c["term"]] + if (term, RDF.type, OWL.Class) not in onto: + onto.add((term, RDF.type, OWL.Class)) + onto.add((term, RDFS.subClassOf, pr[c["parent"]])) + onto.add((term, RDFS.label, Literal(c["label"]))) + onto.add((term, RDFS.comment, Literal(" ".join(c["comment"].split())))) + onto.add((term, pr.isAbstract, Literal(True))) + report.add("classes_created", 1, c["term"]) + + # 6 — reparent (TN-027) + for rp in spec["structural"].get("reparent") or []: + term, parent = pr[rp["term"]], pr[rp["parent"]] + if (term, RDFS.subClassOf, parent) not in onto: + onto.add((term, RDFS.subClassOf, parent)) + report.add("reparented", 1, "%s -> %s" % (rp["term"], rp["parent"])) + + # 7 — sub-properties (TN-028) + for sp in spec["structural"].get("subproperties") or []: + term, parent = pr[sp["term"]], pr[sp["parent"]] + if (term, RDFS.subPropertyOf, parent) not in onto: + onto.add((term, RDFS.subPropertyOf, parent)) + report.add("subproperties", 1, "%s -> %s" % (sp["term"], sp["parent"])) + + # 8 — controlled values become literals (TN-016, TN-017) + for conv in spec["structural"].get("to_literal") or []: + prop = pr[conv["property"]] + if (prop, RDF.type, OWL.DatatypeProperty) not in onto: + onto.remove((prop, RDF.type, OWL.ObjectProperty)) + onto.add((prop, RDF.type, OWL.DatatypeProperty)) + onto.remove((prop, RDFS.range, None)) + onto.add((prop, RDFS.range, XSD.string)) + onto.add((prop, RDFS.domain, pr[conv["domain"]])) + report.add("to_literal_property", 1, conv["property"]) + for g in all_graphs: # rewrite the asserted values + for s, p, o in list(g.triples((None, prop, None))): + name = local(o, ns) + if name and name in conv["value_map"]: + g.remove((s, p, o)) + g.add((s, p, Literal(conv["value_map"][name]))) + report.add("to_literal_values", 1) + for ind in conv["drop_individuals"] + [conv["drop_class"]]: + subj = pr[ind] + if (subj, None, None) in onto: + for g in all_graphs: + drop_subject(g, subj, report, "to_literal_dropped") + + # 9 — abstractness, from the rulebook annex (TN-006, TN-007) + for a in rules["abstractness"]: + term = pr[a["term"]] + if (term, RDF.type, OWL.Class) not in onto: + report.add("abstract_missing_class", 1, a["term"]) + continue + want = Literal(bool(a["is_abstract"])) + if (term, pr.isAbstract, want) not in onto: + onto.remove((term, pr.isAbstract, None)) + onto.add((term, pr.isAbstract, want)) + report.add("isAbstract_declared", 1) + + # 10 — display annotations, from the rulebook annex (TN-022) + for d in rules["display"]: + term = pr[d["iri"]] + for prop, value in ((pr.shortLabel, d.get("short_label")), + (pr.acronym, d.get("acronym"))): + if not value: + continue + if (term, prop, Literal(value)) not in onto: + onto.remove((term, prop, None)) + onto.add((term, prop, Literal(value))) + report.add("display_set", 1) + + # 11 — regenerate every label by derivation (TN-018, TN-023) + kinds = (OWL.Class, OWL.ObjectProperty, OWL.DatatypeProperty, OWL.AnnotationProperty) + for kind in kinds: + for term in set(onto.subjects(RDF.type, kind)): + name = local(term, ns) + if not name: + continue + want = Literal(decamelise(name, kind == OWL.Class)) + if (term, RDFS.label, want) not in onto: + onto.remove((term, RDFS.label, None)) + onto.add((term, RDFS.label, want)) + report.add("labels_regenerated", 1) + + # 12 — bump the ontology version (EV-014: the namespace itself never moves) + target = URIRef(ns.rstrip("/") + "/" + spec["meta"]["to_version"]) + for onto_iri in set(onto.subjects(RDF.type, OWL.Ontology)): + if (onto_iri, OWL.versionIRI, target) not in onto: + onto.remove((onto_iri, OWL.versionIRI, None)) + onto.add((onto_iri, OWL.versionIRI, target)) + report.add("version_bumped", 1, str(target)) + + +# ----------------------------------------------------------------------- main + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--apply", action="store_true", help="write the files (default: dry run)") + ap.add_argument("--ontology", default=os.path.join(HERE, "..", "ontology", "pr_metamodel.ttl")) + ap.add_argument("--instances", action="append", default=[]) + ap.add_argument("--shapes", action="append", default=[]) + ap.add_argument("--log-dir", default=os.path.join(HERE, "logs")) + args = ap.parse_args() + + spec = yaml.safe_load(open(SPEC, encoding="utf-8")) + rules = yaml.safe_load(open(RULES, encoding="utf-8")) + + paths = [args.ontology] + args.instances + args.shapes + missing = [p for p in paths if not os.path.exists(p)] + if missing: + print("MISSING INPUT") + for p in missing: + print(" " + p) + sys.exit(1) + + inputs = OrderedDict((p, md5(p)) for p in paths) + + onto = Graph() + onto.parse(args.ontology, format="turtle") + others, other_paths = [], args.instances + args.shapes + for p in other_paths: + g = Graph() + g.parse(p, format="turtle") + others.append(g) + + before = [len(onto)] + [len(g) for g in others] + report = Report() + migrate(onto, others, spec, rules, report) + after = [len(onto)] + [len(g) for g in others] + + print("MIGRATION %s -> %s %s" + % (spec["meta"]["from_version"], spec["meta"]["to_version"], + "APPLY" if args.apply else "DRY RUN")) + print() + for step, n in report.steps.items(): + print(" %-28s %6d" % (step, n)) + print(" %-28s %6d" % ("total changes", report.total)) + print() + for p, b, a in zip(paths, before, after): + print(" %-46s %6d -> %6d triples" % (os.path.basename(p), b, a)) + if report.notes: + print() + for n in report.notes: + print(" note: " + n) + + log = { + "attempt_timestamp": datetime.datetime.now().isoformat(timespec="seconds"), + "mode": "apply" if args.apply else "dry-run", + "from_version": spec["meta"]["from_version"], + "to_version": spec["meta"]["to_version"], + "input_checksums": inputs, + "steps": report.steps, + "total_changes": report.total, + "triples_before": dict(zip(paths, before)), + "triples_after": dict(zip(paths, after)), + "notes": report.notes, + } + os.makedirs(args.log_dir, exist_ok=True) + stamp = datetime.datetime.now().strftime("%Y%m%dT%H%M%S") + log_path = os.path.join(args.log_dir, "migration_%s.json" % stamp) + json.dump(log, open(log_path, "w", encoding="utf-8"), indent=2) + print("\n log: %s" % log_path) + + if not args.apply: + print("\n DRY RUN — nothing written. Re-run with --apply when the counts " + "above are what you expect.") + return + + onto.serialize(destination=args.ontology, format="turtle") + for p, g in zip(other_paths, others): + g.serialize(destination=p, format="turtle") + print("\n written. Replay this script now: a second run must report zero " + "changes (EV-006).") + + +if __name__ == "__main__": + main() diff --git a/governance/migration/renames.yaml b/governance/migration/renames.yaml new file mode 100644 index 0000000..2cf5be3 --- /dev/null +++ b/governance/migration/renames.yaml @@ -0,0 +1,154 @@ +# T-BOX MIGRATION TO v2.0 — RENAME MAP AND STRUCTURAL OPERATIONS +# ============================================================================= +# Data consumed by migrate_tbox_v2_0.py. Kept separate from the script so that +# the transformation is reviewable without reading code (EV-004). +# +# Not listed here and derived from other sources at run time: +# - isAbstract declarations -> rules.yaml, annex `abstractness` +# - shortLabel and acronym -> rules.yaml, annex `display` +# - deprecated terms to drop -> found in the graph by owl:deprecated true +# Deriving them keeps a single source of truth for each fact (EV-008). +# ============================================================================= + +meta: + from_version: "1.6" + to_version: "2.0" + namespace: "https://ontology.pernod-ricard.com/metamodel/" + +# ----------------------------------------------------------------------------- +# RENAMES — 48. Applied everywhere the identifier appears: subject, predicate, +# object. Nothing cascades in RDF (EV-001). +# +# `order` is a migration constraint, not a preference: +# 1 ordinary terms +# 2 terms named in SHACL shapes or SPARQL constraints +# 3 the four attributes borne by MetaModelObject, therefore by every subject +# ----------------------------------------------------------------------------- + +renames: + +# --- classes +- {from: SubDomain, to: DataSubDomain, kind: class, order: 1} +- {from: SubDomainOwner, to: DataSubDomainOwner, kind: class, order: 1} +- {from: KPI, to: KeyPerformanceIndicator, kind: class, order: 1} +- {from: BIWorkspace, to: BusinessIntelligenceWorkspace, kind: class, order: 1} +- {from: BIDataSource, to: BusinessIntelligenceDataSource, kind: class, order: 1} +- {from: BIField, to: BusinessIntelligenceField, kind: class, order: 1} +- {from: BIReport, to: BusinessIntelligenceReport, kind: class, order: 1} + +# --- individual +- {from: BITool, to: BusinessIntelligenceTool, kind: individual, order: 1} + +# --- relations +- {from: hasDGL, to: hasGovernanceLead, kind: relation, order: 1} +- {from: aboutConcept, to: isAbout, kind: relation, order: 1} +- {from: primaryLocation, to: primarilyStoredIn, kind: relation, order: 1} +- {from: inDatabase, to: isInDatabase, kind: relation, order: 1} +- {from: inSchema, to: isInSchema, kind: relation, order: 1} +- {from: inWorkspace, to: isInBIWorkspace, kind: relation, order: 1} +- {from: inBIDataSource, to: isInBIDataSource, kind: relation, order: 1} +- {from: hasBIField, to: containsBIField, kind: relation, order: 1} +- {from: hasBIOwner, to: hasPublisher, kind: relation, order: 1} +- {from: fromField, to: referencesSourceField, kind: relation, order: 1} +- {from: toField, to: referencesTargetField, kind: relation, order: 1} +- {from: biDerivedFrom, to: derivedFromSourceField, kind: relation, order: 1} +- {from: owningDomain, to: ownedByDomain, kind: relation, order: 2} + +# --- attributes +- {from: hasBusinessDefinition, to: businessDefinition, kind: attribute, order: 1} +- {from: hasTechnicalDefinition, to: technicalDefinition, kind: attribute, order: 1} +- {from: hasSynonym, to: synonym, kind: attribute, order: 1} +- {from: hasFormula, to: formula, kind: attribute, order: 1} +- {from: hasUnit, to: unit, kind: attribute, order: 1} +- {from: hasTimeAggregation, to: timeAggregation, kind: attribute, order: 1} +- {from: hasFormat, to: logicalFormat, kind: attribute, order: 1} +- {from: hasPhysicalDataType, to: physicalDataType, kind: attribute, order: 1} +- {from: hasViewDefinition, to: viewDefinition, kind: attribute, order: 1} +- {from: hasExpression, to: expression, kind: attribute, order: 1} +- {from: hasBusinessRule, to: businessRule, kind: attribute, order: 1} +- {from: hasExampleValue, to: exampleValue, kind: attribute, order: 1} +- {from: createdOn, to: creationDate, kind: attribute, order: 1} +- {from: lastReviewedOn, to: lastReviewDate, kind: attribute, order: 1} +- {from: harvestedOn, to: harvestDate, kind: attribute, order: 1} +- {from: lastQueriedOn, to: lastQueryDate, kind: attribute, order: 1} +- {from: definedBy, to: definitionAddress, kind: attribute, order: 1} +- {from: queryCount30d, to: queryCount, kind: attribute, order: 1} +- {from: hasShortLabel, to: shortLabel, kind: annotation, order: 2} +- {from: hasAcronym, to: acronym, kind: annotation, order: 2} +- {from: hasIdentifier, to: identifier, kind: attribute, order: 3} +- {from: hasName, to: canonicalName, kind: attribute, order: 3} +- {from: hasStatus, to: status, kind: attribute, order: 3} +- {from: hasVersion, to: version, kind: attribute, order: 3} + +# --- relations that change nature (TN-016): object property -> datatype property +- {from: hasActivationStatus, to: activationStatus, kind: relation_to_attribute, order: 1} +- {from: deployedIn, to: environment, kind: relation_to_attribute, order: 1} + +# ----------------------------------------------------------------------------- +# MERGES — the source identifier disappears into an existing target. +# Unlike a rename, the target already exists and keeps its own declaration. +# ----------------------------------------------------------------------------- + +merges: +- from: biConsumes + into: consumes + widen_domain: true + reason: > + Same verb, same target, identical meaning. The domain widens and scope is + controlled per class by SHACL. Ranges are compatible, so no RDFS retyping + is introduced (TN-028). + +# ----------------------------------------------------------------------------- +# STRUCTURAL OPERATIONS +# ----------------------------------------------------------------------------- + +structural: + + reclassify: + - {term: shortLabel, from: DatatypeProperty, to: AnnotationProperty, rule: TN-003} + - {term: acronym, from: DatatypeProperty, to: AnnotationProperty, rule: TN-003} + + create_classes: + - term: GovernanceLayerObject + parent: MetaModelObject + is_abstract: true + label: Governance Layer Object + comment: > + Root of the governance layer. Holds the actors and governance objects that + are not data objects and therefore sit outside the data layers. Created by + insertion above Actor rather than by renaming it, so that every range + pointing at Actor keeps stating the nature of its target rather than its + position in the model. + rule: TN-027 + + reparent: + - {term: Actor, parent: GovernanceLayerObject, rule: TN-027} + - {term: PhysicalLayerObject, parent: MetaModelObject, rule: TN-027} + + subproperties: + - {term: primarilyStoredIn, parent: storedIn, rule: TN-028} + + # TN-016: closed governance states become literals. The classes and their + # individuals are withdrawn, and the properties become datatype properties. + to_literal: + - property: activationStatus + domain: DataDomain + drop_class: ActivationStatus + drop_individuals: [NotActivated, LightActivation, FullActivation] + value_map: + NotActivated: NOT_ACTIVATED + LightActivation: LIGHT_ACTIVATION + FullActivation: FULL_ACTIVATION + - property: environment + domain: CapturedObject + drop_class: Environment + drop_individuals: [Development, UserAcceptance, Production] + value_map: + Development: DEVELOPMENT + UserAcceptance: USER_ACCEPTANCE + Production: PRODUCTION + + # Phase clause: before v2.0 is published the project is in design, so + # deprecated terms are removed outright rather than kept as stubs. EV-002 + # still applies — the count must be zero. + drop_deprecated: true diff --git a/governance/processes.yaml b/governance/processes.yaml new file mode 100644 index 0000000..71a8e16 --- /dev/null +++ b/governance/processes.yaml @@ -0,0 +1,375 @@ +# PERNOD RICARD DATA METAMODEL — T-BOX PROCESSBOOK +# ============================================================================= +# SINGLE SOURCE OF TRUTH for the procedures. The BPMN 2.0 files, the Mermaid +# views and the Markdown document are GENERATED from this file (EV-008). +# +# python3 generate_bpmn.py -> bpmn/.bpmn (strict, queryable) +# python3 generate_processbook_md.py -> PR_TBox_Processbook.md (Mermaid views) +# +# NODE TYPES +# start | end events +# userTask a human decides or writes; cannot be automated +# scriptTask fully automated +# callActivity invokes another procedure by its id +# gateway exclusive decision; every outgoing flow is guarded +# +# Every node may carry `rules`, the identifiers of the rulebook rules it +# enforces. That list is what makes the BPMN queryable and what feeds the +# control.procedure field back into the rulebook. +# ============================================================================= + +meta: + title: Pernod Ricard Data MetaModel — T-Box Processbook + version: "0.1" + status: Draft for review + date: "2026-08-03" + scope: > + Procedures for changing the vocabulary of the model. Each procedure names + the rules it enforces, the steps a person must perform, and the points at + which the work stops rather than continues. + namespace: "https://ontology.pernod-ricard.com/process/" + +families: + Create: Bringing a new element into the vocabulary. + Verify: Establishing that what exists conforms. + Update: Changing an element that already exists. + Delete: Removing an element from the vocabulary. + Release: Propagating a change and publishing a version. + +processes: + +# ---------------------------------------------------------------------- X1 +- id: X1 + family: Release + name: Propagate a change and publish a version + scope: Any validated modification of the vocabulary. + trigger: A change to the vocabulary has been agreed and is ready to apply. + inputs: [the agreed change, the target version, the list of affected artifacts] + outputs: [a merged branch, a tagged version, an execution log] + note: > + The most frequently invoked procedure of the processbook: every other + procedure ends by calling it. Two of its steps are gateways rather than + checks, because an idempotence failure and a validation failure must stop + the work rather than be noted in passing. + flow: + - {id: start, type: start, name: Change agreed} + - {id: branch, type: scriptTask, name: Create the migration branch, rules: [EV-009]} + - {id: apply, type: scriptTask, name: Apply the change through an RDF parser, rules: [EV-005, EV-006]} + - {id: labels, type: scriptTask, name: Regenerate labels by derivation, rules: [TN-023]} + - {id: abox, type: scriptTask, name: Propagate to the instances, rules: [EV-009]} + - {id: shapes, type: scriptTask, name: Update the shapes and align conformsTo, rules: [EV-010]} + - {id: artifacts, type: scriptTask, name: Regenerate every derived artifact, rules: [EV-008]} + - {id: replay, type: scriptTask, name: Replay the chain a second time, rules: [EV-006, EV-007]} + - {id: idem, type: gateway, name: "Second run reports zero change?"} + - {id: verify_voc, type: callActivity, calls: R1, name: Verify the vocabulary} + - {id: verify_abox, type: callActivity, calls: R2, name: Validate the instances} + - {id: log, type: scriptTask, name: Write the execution log for this attempt, rules: [EV-015]} + - {id: clean, type: gateway, name: "Any violation?"} + - {id: decide, type: userTask, name: Correct or abandon, rules: [EV-004, EV-007]} + - {id: drop, type: scriptTask, name: Destroy the branch and restart from the published version, rules: [EV-007]} + - {id: review, type: userTask, name: Review the merge, rules: [EV-004, EV-009]} + - {id: merge, type: scriptTask, name: Merge as a block, tag, push the tag separately, rules: [EV-009]} + - {id: bump, type: scriptTask, name: Increment the ontology version, rules: [EV-014]} + - {id: end, type: end, name: Version published} + - {id: aborted, type: end, name: Change abandoned} + flows: + - {from: start, to: branch} + - {from: branch, to: apply} + - {from: apply, to: labels} + - {from: labels, to: abox} + - {from: abox, to: shapes} + - {from: shapes, to: artifacts} + - {from: artifacts, to: replay} + - {from: replay, to: idem} + - {from: idem, to: verify_voc, condition: "zero change"} + - {from: idem, to: decide, condition: "the chain is not replayable"} + - {from: verify_voc, to: verify_abox} + - {from: verify_abox, to: log} + - {from: log, to: clean} + - {from: clean, to: review, condition: "none"} + - {from: clean, to: decide, condition: "at least one"} + - {from: decide, to: apply, condition: correct} + - {from: decide, to: drop, condition: abandon} + - {from: drop, to: aborted} + - {from: review, to: merge} + - {from: merge, to: bump} + - {from: bump, to: end} + +# ---------------------------------------------------------------------- C1 +- id: C1 + family: Create + name: Create a term + scope: class | relation | attribute | annotation + parameter: > + The nature of the term selects the naming rules applied at the naming step: + a class takes TN-001, TN-002, TN-004, TN-005, TN-006, TN-007 and TN-027; a + relation takes TN-001, TN-002, TN-009, TN-010, TN-011 and TN-028; an + attribute takes TN-001, TN-002, TN-004, TN-012, TN-013, TN-014 and TN-015; + an annotation takes TN-001, TN-002 and TN-003. + trigger: A missing concept, edge or field has been identified. + inputs: [the intended meaning, the nature, the target layer, the provenance, the parent or the domain and range] + outputs: [a declared term, a published version] + note: > + The first step has no tool today. Checking that no existing term already + covers the need is the semantic uniqueness control that SHACL cannot + perform, and it stays a human task until R3 exists. + flow: + - {id: start, type: start, name: Need identified} + - {id: unique, type: userTask, name: Check that no existing term covers the need} + - {id: exists, type: gateway, name: "A term already covers it?"} + - {id: reuse, type: end, name: Reuse the existing term} + - {id: name, type: scriptTask, name: Name the term according to its nature, rules: [TN-001, TN-002, TN-003, TN-004, TN-005, TN-006, TN-009, TN-010, TN-011, TN-012, TN-013, TN-014, TN-015, TN-028]} + - {id: label, type: scriptTask, name: Derive the label, rules: [TN-018, TN-019, TN-020, TN-021, TN-023]} + - {id: short, type: userTask, name: Decide the short label and the acronym, rules: [TN-022]} + - {id: comment, type: userTask, name: Write the comment, rules: [TN-024]} + - {id: declare, type: scriptTask, name: Declare typing, provenance and attachment, rules: [TN-007, TN-025, TN-026, TN-027]} + - {id: validate, type: userTask, name: Validate before applying, rules: [EV-004]} + - {id: check, type: callActivity, calls: R1, name: Verify against the rulebook} + - {id: release, type: callActivity, calls: X1, name: Propagate and publish} + - {id: end, type: end, name: Term available} + flows: + - {from: start, to: unique} + - {from: unique, to: exists} + - {from: exists, to: reuse, condition: "yes"} + - {from: exists, to: name, condition: "no"} + - {from: name, to: label} + - {from: label, to: short} + - {from: short, to: comment} + - {from: comment, to: declare} + - {from: declare, to: validate} + - {from: validate, to: check} + - {from: check, to: release} + - {from: release, to: end} + +# ---------------------------------------------------------------------- C2 +- id: C2 + family: Create + name: Create a controlled value + scope: typed individual | literal + parameter: > + The form is not a style choice but the outcome of the first decision. The + two branches have different consequences: the literal branch edits the + shapes and therefore the vocabulary, so it falls under the version freeze; + the individual branch touches nothing else. + trigger: A new value is needed in a controlled set. + inputs: [the value, the set it belongs to, whether a domain may add others] + outputs: [a declared value, a published version] + flow: + - {id: start, type: start, name: New value needed} + - {id: decide, type: userTask, name: "Must it be defined, owned or extended by a domain?", rules: [TN-016]} + - {id: form, type: gateway, name: "Which form?"} + - {id: indiv, type: scriptTask, name: Create the typed individual and derive its label, rules: [TN-001, TN-018, TN-022]} + - {id: literal, type: scriptTask, name: Write the literal in upper snake case, rules: [TN-017]} + - {id: shape, type: callActivity, calls: C3, name: Extend the closed list in the shapes} + - {id: validate, type: userTask, name: Validate before applying, rules: [EV-004]} + - {id: check, type: callActivity, calls: R1, name: Verify against the rulebook} + - {id: release, type: callActivity, calls: X1, name: Propagate and publish} + - {id: end, type: end, name: Value available} + flows: + - {from: start, to: decide} + - {from: decide, to: form} + - {from: form, to: indiv, condition: "extensible by a domain"} + - {from: form, to: literal, condition: "closed governance state"} + - {from: indiv, to: validate} + - {from: literal, to: shape} + - {from: shape, to: validate} + - {from: validate, to: check} + - {from: check, to: release} + - {from: release, to: end} + +# ---------------------------------------------------------------------- C3 +- id: C3 + family: Create + name: Create a shape + scope: Any SHACL constraint added to the validation set. + trigger: A rule needs an executor, or a new class enters the validation perimeter. + inputs: [the rule to enforce, the target class or property] + outputs: [a shape, a rulebook entry naming it as executor] + note: > + The step that is forgotten is the last one before release: recording the + shape as the executor of its rule. Without it a rule stays declared + blocking with nothing enforcing it, which EV-011 forbids. + flow: + - {id: start, type: start, name: Constraint needed} + - {id: write, type: userTask, name: Write the constraint} + - {id: pattern, type: scriptTask, name: Check that every pattern is expressed positively, rules: [EV-013]} + - {id: portable, type: gateway, name: "Expressible within the specification?"} + - {id: demote, type: userTask, name: Move the rule to the script tier, rules: [EV-012]} + - {id: conforms, type: scriptTask, name: Align the declared ontology version, rules: [EV-010]} + - {id: test, type: callActivity, calls: R2, name: Test against real instances} + - {id: record, type: userTask, name: Record the shape as the executor of its rule, rules: [EV-011, EV-012]} + - {id: release, type: callActivity, calls: X1, name: Propagate and publish} + - {id: end, type: end, name: Shape in force} + flows: + - {from: start, to: write} + - {from: write, to: pattern} + - {from: pattern, to: portable} + - {from: portable, to: conforms, condition: "yes"} + - {from: portable, to: demote, condition: "no"} + - {from: demote, to: record} + - {from: conforms, to: test} + - {from: test, to: record} + - {from: record, to: release} + - {from: release, to: end} + +# ---------------------------------------------------------------------- D1 +- id: D1 + family: Delete + name: Deprecate a term + scope: class | relation | attribute | annotation | individual + trigger: A term is superseded or no longer needed. + inputs: [the term, its replacement if any] + outputs: [a deprecated stub, a published version] + note: > + The count at step two is informative, not a condition: a term may be + deprecated whether or not it is instantiated. It becomes a condition only + in D2. + flow: + - {id: start, type: start, name: Term superseded} + - {id: replacement, type: userTask, name: Identify the replacement, or record that there is none} + - {id: count, type: scriptTask, name: Count the instances for information, rules: [EV-015]} + - {id: mark, type: scriptTask, name: Mark deprecated and declare the replacement, rules: [EV-001]} + - {id: strip, type: scriptTask, name: Strip the stub of every edge, rules: [EV-001]} + - {id: label, type: scriptTask, name: Remove any mention of state from the label, rules: [TN-020]} + - {id: release, type: callActivity, calls: X1, name: Propagate and publish} + - {id: end, type: end, name: Term deprecated} + flows: + - {from: start, to: replacement} + - {from: replacement, to: count} + - {from: count, to: mark} + - {from: mark, to: strip} + - {from: strip, to: label} + - {from: label, to: release} + - {from: release, to: end} + +# ---------------------------------------------------------------------- D2 +- id: D2 + family: Delete + name: Withdraw a term permanently + scope: class | relation | attribute | annotation | individual + trigger: A deprecated term is to be removed from the vocabulary. + inputs: [the deprecated term] + outputs: [a vocabulary without the term, a proof of non-instantiation, a published version] + note: > + The reference cleaning step is what prevents phantom nodes: a subject + removed while its identifier is still cited elsewhere is reconstructed by + inference, present in traversals and absent from every control. It is a step + of the procedure, not a check at the end of a script. + flow: + - {id: start, type: start, name: Withdrawal requested} + - {id: count, type: scriptTask, name: Run the counting query, rules: [EV-002, EV-015]} + - {id: instantiated, type: gateway, name: "Any instance found?"} + - {id: keep, type: end, name: Kept deprecated} + - {id: proof, type: scriptTask, name: Attach the proof to the commit, rules: [EV-002, EV-015]} + - {id: remove, type: scriptTask, name: Remove the subject block} + - {id: refs, type: scriptTask, name: Remove every reference naming the subject, rules: [EV-003]} + - {id: check_voc, type: callActivity, calls: R1, name: Verify the vocabulary} + - {id: check_abox, type: callActivity, calls: R2, name: Validate the instances} + - {id: release, type: callActivity, calls: X1, name: Propagate and publish} + - {id: end, type: end, name: Term withdrawn} + flows: + - {from: start, to: count} + - {from: count, to: instantiated} + - {from: instantiated, to: keep, condition: "yes"} + - {from: instantiated, to: proof, condition: "no"} + - {from: proof, to: remove} + - {from: remove, to: refs} + - {from: refs, to: check_voc} + - {from: check_voc, to: check_abox} + - {from: check_abox, to: release} + - {from: release, to: end} + +# ---------------------------------------------------------------------- R1 +- id: R1 + family: Verify + name: Verify the vocabulary against the rulebook + scope: The whole vocabulary, or the subset touched by a change. + trigger: A term has been created, changed or withdrawn, or a release is prepared. + inputs: [the vocabulary, the rulebook source] + outputs: [a conformance report] + note: > + Two tiers run in sequence, not in parallel: the script settles everything + mechanisable, and a person answers only for the rules no script can judge. + Reversing the order wastes review time on findings the script would have + caught. + flow: + - {id: start, type: start, name: Verification requested} + - {id: script, type: scriptTask, name: "Run the naming and declaration checks", rules: [TN-001, TN-002, TN-003, TN-004, TN-005, TN-006, TN-007, TN-009, TN-010, TN-011, TN-012, TN-013, TN-014, TN-015, TN-017, TN-018, TN-019, TN-020, TN-021, TN-023, TN-025, TN-026, TN-027, TN-028, EV-014]} + - {id: mechanised, type: gateway, name: "Any mechanised violation?"} + - {id: report_fail, type: end, name: Report returned with violations} + - {id: human, type: userTask, name: "Review the rules no script can judge", rules: [TN-008, TN-016, TN-022, TN-024]} + - {id: judged, type: gateway, name: "Reviewer raises an issue?"} + - {id: report_ok, type: end, name: Vocabulary conforms} + flows: + - {from: start, to: script} + - {from: script, to: mechanised} + - {from: mechanised, to: report_fail, condition: "at least one"} + - {from: mechanised, to: human, condition: none} + - {from: human, to: judged} + - {from: judged, to: report_fail, condition: "yes"} + - {from: judged, to: report_ok, condition: "no"} + +# ---------------------------------------------------------------------- R2 +- id: R2 + family: Verify + name: Validate instances against the shapes + scope: Any instance graph in the validation perimeter. + trigger: Instances have changed, shapes have changed, or a release is prepared. + inputs: [the instance graph, the shapes, the ontology] + outputs: [a validation report] + note: > + The version check comes first and aborts rather than warns. Validating + against shapes that target another version of the ontology does not fail + loudly: it returns a long list of violations that reads exactly like a + regression of the model. + flow: + - {id: start, type: start, name: Validation requested} + - {id: conforms, type: scriptTask, name: "Compare the declared target version with the ontology version", rules: [EV-010]} + - {id: match, type: gateway, name: "Versions match?"} + - {id: abort, type: end, name: Aborted on version mismatch} + - {id: run, type: scriptTask, name: Run the shape validation} + - {id: violations, type: gateway, name: "Any violation?"} + - {id: fail, type: end, name: Report returned with violations} + - {id: ok, type: end, name: Instances conform} + flows: + - {from: start, to: conforms} + - {from: conforms, to: match} + - {from: match, to: abort, condition: "no"} + - {from: match, to: run, condition: "yes"} + - {from: run, to: violations} + - {from: violations, to: fail, condition: "at least one"} + - {from: violations, to: ok, condition: none} + +# ---------------------------------------------------------------------- R3 +- id: R3 + family: Verify + name: Audit what the shapes cannot see + scope: The whole vocabulary and its source file. + trigger: Periodic audit, or before a version is published. + inputs: [the vocabulary, its source file] + outputs: [a shortlist of suspected duplicates, a list of duplicated blocks] + note: > + Shape validation reads a graph, not a file and not meaning. Two identifiers + standing for the same notion produce two individually valid graphs; a + duplicated block of text produces identical triples and no complaint. This + procedure is the tier those rules fall to, and its output is a shortlist for + a person rather than a verdict. + flow: + - {id: start, type: start, name: Audit requested} + - {id: index, type: scriptTask, name: "Build the normalised label index", rules: [EV-012]} + - {id: hash, type: scriptTask, name: "Hash every normalised subject block", rules: [EV-012]} + - {id: shortlist, type: gateway, name: "Any candidate found?"} + - {id: clean, type: end, name: Nothing to arbitrate} + - {id: review, type: userTask, name: Arbitrate each candidate} + - {id: act, type: gateway, name: "Duplication confirmed?"} + - {id: merge, type: end, name: Referred to the merge procedure} + - {id: dismissed, type: end, name: Candidates dismissed} + flows: + - {from: start, to: index} + - {from: index, to: hash} + - {from: hash, to: shortlist} + - {from: shortlist, to: clean, condition: none} + - {from: shortlist, to: review, condition: "at least one"} + - {from: review, to: act} + - {from: act, to: merge, condition: "yes"} + - {from: act, to: dismissed, condition: "no"} diff --git a/governance/requirements.txt b/governance/requirements.txt new file mode 100644 index 0000000..e76bbcf --- /dev/null +++ b/governance/requirements.txt @@ -0,0 +1,3 @@ + +PyYAML>=6.0,<7 +openpyxl>=3.1,<3.2 diff --git a/governance/rules.yaml b/governance/rules.yaml new file mode 100644 index 0000000..3fcfd71 --- /dev/null +++ b/governance/rules.yaml @@ -0,0 +1,1045 @@ +# PERNOD RICARD DATA METAMODEL — T-BOX RULEBOOK +# ============================================================================= +# SINGLE SOURCE OF TRUTH. The .md and .xlsx deliverables are GENERATED from this +# file and must never be edited by hand (EV-007, EV-008). +# +# python3 generate_rulebook_md.py +# python3 generate_rulebook_xlsx.py +# +# severity : BLOCKING | MAJOR | GUIDELINE — one per rule. The rule is the +# rule; transitional tolerance belongs to a migration process, +# not to a normative document. +# control.tier : shacl | script | human (EV-012 — where the rule is enforced) +# control.executor : the script or shape that enforces it (EV-011 — no blocking +# rule without an executor) +# control.procedure: filled in by the processbook. null until then. +# ============================================================================= + +meta: + title: Pernod Ricard Data MetaModel — T-Box Rulebook + version: "1.1" + status: Draft for review + date: "2026-08-03" + scope: > + Governs the vocabulary of the model itself: the IRIs, labels and declarative + axioms of every class, property and enumeration individual. Does not govern + instances, which remain under the A-Box rulebook. + audience: > + Written to be read and applied directly, by a person or by a language model, + without further context. Every rule states what must hold, why, and what + enforces it. + +categories: + Identifier: + title: Form of the IRI + intent: > + The IRI is identity. It is immutable in the RDF sense — changing one is a + migration, never an edit — so it must be unambiguous, searchable and free + of local convention. + Label: + title: Derivation and display + intent: > + The label is derived, not authored. Anything a reader needs that the + derivation cannot produce lives in a dedicated channel: the short label + for the spoken form, the comment for meaning. + Declaration: + title: Declarative completeness + intent: > + Nothing is left to be guessed. What a term is, what it applies to, where + it sits and where it came from are all stated, never inferred from + structure or from silence. + Evolution: + title: Change and propagation + intent: > + Every rule here is anchored on a real failure. They govern how the model + changes without breaking what depends on it. + +severity_model: + BLOCKING: > + The commit does not pass. Enforced pre-commit and in continuous integration + by the named executor. + MAJOR: > + Reviewed and answered for, but not machine-enforceable. Reserved for rules + that require human judgement, per EV-011. + GUIDELINE: > + Recommended practice with no gate. + +rules: + +# ---------------------------------------------------------------- IDENTIFIER + +- id: TN-001 + category: Identifier + title: CamelCase IRI with no separator + statement: > + The local name of an IRI is strict CamelCase with no separator of any kind: + no hyphen, no underscore, no dot, no space. Initial capital for a class or + an enumeration individual; initial lower case for a property, whether + object, datatype or annotation. Consecutive capitals are admitted only where + they carry an acronym reproduced from a target class short label under + TN-011, and nowhere else. + scope: [class, object_property, datatype_property, annotation_property, individual] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A hyphen forces escaping in certain Turtle and SPARQL contexts and makes + decamelisation ambiguous: Sub-Domain has no single correct expansion. + Typographic convention is a display concern and belongs to the short label, + never to identity. + examples: + - {from: "pr:data_domain", to: "pr:DataDomain", note: no underscore} + - {from: "pr:Sub-Domain", to: "pr:DataSubDomain", note: no hyphen} + - {from: "pr:BelongsTo", to: "pr:belongsTo", note: a property starts lower case} + +- id: TN-002 + category: Identifier + title: No acronym in the IRI + statement: > + The local name of an IRI contains no acronym, initialism or abbreviation. + Every short form is expanded in full. One exception, and one only: a + relation reproducing the short label of the class it targets carries that + short label as it stands, acronym included, under TN-011. No other term may + carry an acronym, and no exception list is maintained. + scope: [class, object_property, datatype_property, annotation_property, individual] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: A-Box NR-004 + rationale: > + An acronym in an identifier is a local convention pretending to be a name. + It is unsearchable by anyone who does not already know it, and it collides + across domains: PO is a Product Owner in one team and a Purchase Order in + another. Length is not a counter-argument — an IRI is written by tooling and + read in context, and the short form remains available through the short + label and the acronym annotation. The single exception exists because a + relation name is read on every edge of the graph and in every query, where + the spoken form is what makes it legible; reproducing the target's short + label verbatim also keeps the relation name derivable from the class rather + than invented, which a truncated form would not. + examples: + - {from: "pr:KPI", to: "pr:KeyPerformanceIndicator", note: "short form moves to pr:acronym"} + - {from: "pr:BIDataSource", to: "pr:BusinessIntelligenceDataSource", note: null} + - {from: "pr:definitionUri", to: "pr:definitionAddress", note: "URI is itself an acronym"} + - {from: "a relation targeting a class whose short label is BI Field", to: "pr:containsBIField", note: "the one exception — see TN-011"} + +- id: TN-003 + category: Identifier + title: Display properties are annotations + statement: > + pr:acronym and pr:shortLabel are declared owl:AnnotationProperty, never + owl:DatatypeProperty. + scope: [annotation_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A datatype property applies to an individual. Annotating a term of the + vocabulary with one takes the ontology out of OWL DL into OWL Full, which + reasoners reject and strict validators flag. An annotation property applies + to anything — class, property or individual. Nothing is lost on the control + side: SHACL validates triples and does not read OWL declarations, so every + shape targeting these properties keeps working unchanged. + examples: + - {from: "pr:shortLabel a owl:DatatypeProperty", to: "pr:shortLabel a owl:AnnotationProperty", note: null} + - {from: "pr:acronym a owl:DatatypeProperty", to: "pr:acronym a owl:AnnotationProperty", note: null} + +- id: TN-004 + category: Identifier + title: No digit in the IRI + statement: > + The local name of a vocabulary term contains no digit. A numeric parameter — + a window, a threshold, a version — is a value, never an identity. + scope: [class, object_property, datatype_property, annotation_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: A-Box NR-014 (deliberate opposite) + rationale: > + An instance identifier does encode numbers, and that is its purpose: it + carries a position in the domain tree. A vocabulary term encodes nothing but + its own meaning. The opposition is deliberate and is stated in the ontology + header so that no reader takes one for a violation of the other. A term + named after a thirty-day window starts lying the day the window becomes + ninety, and nothing detects it. + examples: + - {from: "pr:queryCount30d", to: "pr:queryCount", note: the window belongs to the harvesting specification} + +- id: TN-005 + category: Identifier + title: A class is a singular common noun + statement: > + The local name of a class is a singular noun phrase. No plural, no verb. No + type word such as Object, Entity or Item unless it names a genuine + abstraction of the model. No suffix rule is imposed on abstract classes. + scope: [class] + severity: BLOCKING + control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null} + filiation: A-Box NR-008, NR-013 + rationale: > + Abstractness is not inferable from a name and must not be encoded in one. + The correspondence fails in both directions: classes can be abstract without + any suffix, and classes ending in Object can be among the most heavily + instantiated in the model. Renaming a central business term to satisfy a + naming rule would trade meaning for symmetry. Abstractness is declared + instead, by TN-007. + examples: + - {from: "pr:DataDomains", to: "pr:DataDomain", note: singular} + - {from: "pr:ActorObject", to: "pr:Actor", note: no type word added for its own sake} + +- id: TN-006 + category: Identifier + title: Every layer root is abstract + statement: > + A class that serves as the root of a layer carries pr:isAbstract true. + scope: [class] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A non-abstract layer root allows an instance to be typed as an object of + that layer without saying which kind of object it is. That statement carries + no information and no downstream control can repair it. + examples: + - {from: "pr:PhysicalLayerObject with no declaration", to: "pr:isAbstract true", note: null} + +- id: TN-007 + category: Identifier + title: Every class declares whether it is abstract + statement: > + Every owl:Class carries pr:isAbstract with an explicit boolean value. + Absence of the property is a violation, not a default. + scope: [class] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + Where declaration is optional, nothing distinguishes a class deliberately + left concrete from one where the question was never asked. Two situations + make the compulsion necessary. A class with subclasses may be perfectly + instantiable, and has been wrongly marked abstract by inference. And a class + used as a classification axis may have no declared subclass at all, its + members being typed by multiple typing on instances — its abstractness is + then structurally unverifiable and only the explicit declaration carries it. + examples: + - {from: "a class with no declaration", to: "pr:isAbstract true or false", note: silence is a violation} + - {from: "pr:Metric inferred abstract because it has a subclass", to: "pr:isAbstract false", note: concrete despite having a subclass} + +- id: TN-008 + category: Identifier + title: Abstractness is declared, never inferred + statement: > + Abstractness is read from pr:isAbstract and from nothing else. No consumer + of the model — viewer, export, script or documentation generator — may + derive it from the presence of subclasses, from the absence of instances, or + from any other structural signal. + scope: [class, consumer] + severity: BLOCKING + control: {tier: human, executor: consumer contract review, procedure: null} + filiation: null + rationale: > + A declaration rule holds in the source only if it also binds every consumer. + Otherwise each one re-invents its own inference and the model acquires as + many answers as it has readers. This is the same failure mode that justified + a governed short label rather than a display dictionary per viewer. + examples: + - {from: "isAbstract = term.subclasses.length > 0", to: "isAbstract = term.isAbstract === true", note: consumer contract} + +- id: TN-009 + category: Identifier + title: A relation begins with a verb + statement: > + The local name of an object property begins with a verb, in lower case. Both + the active form and the passive or participial form are admitted: a past + participle is a verb in first position. No relation begins with a + preposition or with a noun. + scope: [object_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A relation reads as a verb and an attribute reads as a noun. The distinction + is what lets a reader tell an edge from a field without opening the + declaration. The explicit clause on participial forms is required: read + literally, a verb-first rule would condemn a whole family of sound relations + such as computedBy, storedIn and derivedFrom. + examples: + - {from: "pr:inDatabase", to: "pr:isInDatabase", note: a preposition is not a verb} + - {from: "pr:primaryLocation", to: "pr:primarilyStoredIn", note: a noun is not a verb} + - {from: "pr:computedBy", to: "pr:computedBy", note: participial form is conforming} + +- id: TN-010 + category: Identifier + title: The is prefix is either a copula or a predicate + statement: > + On an owl:ObjectProperty, is acts as a COPULA — a state verb before a + preposition or adjective, whose target is a node of the graph. On a boolean + datatype or annotation property, is acts as a PREDICATE — it introduces an + adjective or participle and the value is true or false. No other use is + admitted. + scope: [object_property, datatype_property, annotation_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + Conformance of a name beginning with is is never checkable on the name + alone. A control asserting that every such term is a boolean produces false + positives on every copula; the reverse control lets real problems through. + The correct check reads the declared type of the property first, then + applies the matching clause. + examples: + - {from: "pr:isInSchema", to: copula, note: object property, target is a node} + - {from: "pr:isNullable", to: predicate, note: datatype property of range xsd:boolean} + +- id: TN-011 + category: Identifier + title: A relation names its target by its short label + statement: > + A relation need not mention its target class at all. When it does, it uses + the target class's pr:shortLabel AS IT STANDS — not the full label, not a + truncation of the short label, and not a form invented for the occasion. + Where the short label carries an acronym, the acronym is reproduced with it: + this is the single exception admitted by TN-002, and TN-001 admits the + consecutive capitals it produces. + scope: [object_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + Relation names are what is read on the edges of the graph, constantly, in + every viewer and every query. Full labels make them unreadable on an arrow. + The short label is the form the organisation actually speaks, and reusing it + makes a relation name predictable from the class it points at instead of + arbitrary. Reproducing it whole is what preserves that property: a + truncation would be a form the model invented, unpredictable from the class + and unverifiable against it. A corollary: relations do not normally carry a + short label of their own, since this rule already keeps their IRIs short. + examples: + - {from: "pr:hasDomainOwner", to: "pr:hasDomainOwner", note: "target DataDomainOwner, short label Domain Owner — conforming"} + - {from: "pr:hasDGL", to: "pr:hasGovernanceLead", note: "target short label is Governance Lead"} + - {from: "pr:inBIDataSource", to: "pr:isInBIDataSource", note: "short label BI Data Source reproduced whole, acronym included"} + - {from: "pr:containsField", to: "pr:containsBIField", note: "a truncated short label is not admitted"} + +- id: TN-012 + category: Identifier + title: An attribute is a noun + statement: > + The local name of a datatype property is a noun phrase. It begins with no + verb, and in particular with no has. + scope: [datatype_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A has prefix carries no information — rdfs:domain already says who has the + attribute — and it costs the reader the distinction TN-009 exists to make + visible. Under TN-018 the derived label would read as a verb phrase on a + field, so a relation and an attribute would become indistinguishable in + every display. + examples: + - {from: "pr:hasFormula", to: "pr:formula", note: null} + - {from: "pr:hasName", to: "pr:canonicalName", note: name alone is too generic to stand as an identity} + +- id: TN-013 + category: Identifier + title: A date attribute is a noun, not a participle + statement: > + An attribute holding a date or timestamp is named as a noun. + scope: [datatype_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + The participial form is idiomatic English but creates a standing family of + exceptions to TN-012 that must be defended at every review. The nominal form + is consistent with the other attributes of the model and needs no defence. + examples: + - {from: "pr:createdOn", to: "pr:creationDate", note: null} + - {from: "pr:lastQueriedOn", to: "pr:lastQueryDate", note: null} + +- id: TN-014 + category: Identifier + title: An attribute never takes a relational suffix + statement: > + A datatype property never takes a form ending in By, In, From or On. Those + forms are reserved for object properties. The only exception is the + predicative is of booleans, covered by TN-015. + scope: [datatype_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + An attribute named like a relation will be read as one. A reader + encountering a term in the shape of computedBy or ownedBy expects a node at + the other end, not a string. The tell is usually visible in the label: a + parenthetical gloss added to explain what the value actually holds is + evidence that the name itself is wrong. + examples: + - {from: "pr:definedBy holding a string", to: "pr:definitionAddress", note: "address, not location — location is taken by the storage side"} + +- id: TN-015 + category: Identifier + title: A boolean attribute begins with is + statement: > + A property of range xsd:boolean begins with is, followed by an adjective or + participle. This is the single explicit exception to TN-012. + scope: [datatype_property, annotation_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A bare adjective does not read as a statement, and some of them collide with + reserved words in the languages that consume the model. + examples: + - {from: "pr:nullable", to: "pr:isNullable", note: null} + - {from: "pr:abstract", to: "pr:isAbstract", note: "abstract is a reserved word in several target languages"} + +- id: TN-016 + category: Identifier + title: A controlled value is an individual or a literal + statement: > + A value becomes a TYPED INDIVIDUAL if it needs to be defined, owned, or + extended by a domain without modifying the model. It stays a LITERAL if it + is a closed technical state whose list is the property of governance and + must precisely not be extended. The model chooses one form per kind of value + and states the choice. + scope: [class, individual, datatype_property] + severity: BLOCKING + control: {tier: [shacl, human], executor: pr_metamodel_shapes.ttl, procedure: null} + filiation: A-Box NR-016 + rationale: > + A literal cannot carry a definition, an owner or a link, and extending its + list means editing the shapes and therefore the frozen vocabulary. A typed + individual can be added by a domain without touching the model, appears as a + node in every viewer, and can be reached by following an edge. The test is + not how the value looks but who is allowed to add one. + examples: + - {from: "an activation level defined in a playbook, not in the graph", to: "literal constrained by sh:in", note: not domain-extensible} + - {from: "a kind of system a domain may need to add", to: typed individual, note: extensible and worth defining} + +- id: TN-017 + category: Identifier + title: Controlled literals are UPPER_SNAKE_CASE + statement: > + Controlled literal values are written UPPER_SNAKE_CASE. This is the only + place in the model where that casing is used. + scope: [datatype_property] + severity: BLOCKING + control: {tier: shacl, executor: pr_metamodel_shapes.ttl, procedure: null} + filiation: null + rationale: > + The casing exists so that a value drawn from a closed list is + distinguishable from free text at a glance, in the source and in any export. + examples: + - {from: "\"Full activation\"", to: "\"FULL_ACTIVATION\"", note: null} + - {from: "\"Published\"", to: "\"PUBLISHED\"", note: null} + +# --------------------------------------------------------------------- LABEL + +- id: TN-018 + category: Label + title: The label is derived from the IRI + statement: > + rdfs:label is obtained from the local name of the IRI by inserting a space + at every case boundary. No word added, removed, reordered or substituted. + The words of a class label keep their initial capital; the words of a + property label are entirely lower case. + scope: [class, object_property, datatype_property, annotation_property, individual] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: A-Box NR-005 (deliberate opposite) + rationale: > + A label carrying meaning the IRI does not is the symptom of a badly named + IRI or of an incomplete comment — never of a legitimate exception to the + label. Where a label has been quietly enriched to explain a term, the + explanation belongs in rdfs:comment and the short form, if the organisation + speaks one, in pr:shortLabel. + examples: + - {from: "pr:BusinessIntelligenceDataSource", to: "Business Intelligence Data Source", note: a class label keeps its capitals} + - {from: "pr:primarilyStoredIn", to: "primarily stored in", note: a property label is lower case} + - {from: "\"Ownership and Categorization Layer Object\"", to: "\"Ownership Layer Object\"", note: the added meaning moves to rdfs:comment} + +- id: TN-019 + category: Label + title: No acronym in the label + statement: > + The label introduces no acronym or abbreviation that the IRI does not + already carry, including in a recapitalised form that the derivation would + not produce. Where the IRI legitimately carries an acronym under TN-011, the + derived label carries it too and nothing further is added. + scope: [class, object_property, datatype_property, annotation_property, individual] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A direct corollary of TN-018, kept as a separate rule because it catches + labels whose IRI is innocent: a term correctly named in full can still carry + a hand-written label that reintroduces the short form. Stated as an + introduction rather than a prohibition, it stays a pure corollary — whatever + the IRI legitimately holds, the derivation reproduces, and nothing else. + examples: + - {from: "\"BI expression\"", to: "\"expression\"", note: the IRI was already conforming} + +- id: TN-020 + category: Label + title: No state in the label + statement: > + The label mentions no status, version or deprecation. State lives in an + axiom — owl:deprecated, pr:status — never in text. Deprecated and active + terms are never mixed in a single view: a consumer filters on the axiom at + query time and presents deprecated terms separately. + scope: [class, object_property, datatype_property, annotation_property, individual, consumer] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A term whose label repeats its state carries that state twice. The day one + is changed without the other, the two contradict each other and nothing + detects it. With the rule in force, the question of whether a term is + deprecated has exactly one possible answer. + examples: + - {from: "\"belongs to domain (deprecated)\"", to: "\"belongs to domain\"", note: owl:deprecated carries the state} + +- id: TN-021 + category: Label + title: No gloss or parenthesis in the label + statement: > + No parenthesis, qualification or example appears in the label. They belong + in rdfs:comment. + scope: [class, object_property, datatype_property, annotation_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A parenthetical gloss is almost always the symptom of an IRI that does not + say what it holds. The right response is usually to rename the term, not to + annotate the label. + examples: + - {from: "\"defined by (code URI)\"", to: "\"definition address\"", note: rename the term rather than gloss the label} + - {from: "\"query count (30 days)\"", to: "\"query count\"", note: null} + +- id: TN-022 + category: Label + title: The short label carries the spoken form + statement: > + A term carries pr:shortLabel whenever the form derived under TN-018 is not + what the organisation writes or says. The short label is free: acronyms, + abbreviations, hyphens, short forms. + scope: [class, individual] + severity: MAJOR + control: {tier: human, executor: review checklist, procedure: null} + filiation: null + rationale: > + The short label is what makes the derivation rule bearable. Without it, a + mechanical label impoverishes every screen; with it, display becomes + explicit and centrally governed instead of being improvised by each + consumer. It is MAJOR rather than BLOCKING because judging whether a derived + form matches what people actually say is human work, and EV-011 forbids + declaring a rule blocking with no executor. + examples: + - {from: "\"Data Sub Domain Owner\"", to: "short label Sub Domain Owner", note: the word Data is dropped in the spoken form} + - {from: "\"Key Performance Indicator\"", to: "short label KPI, acronym KPI", note: the acronym channel is what makes TN-002 bearable} + +- id: TN-023 + category: Label + title: The label is generated, never typed + statement: > + rdfs:label is never written by hand. It is produced by the derivation + function from the IRI and stored in the source. A hand-typed label is a + violation even when its value happens to be correct. + scope: [class, object_property, datatype_property, annotation_property, individual] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + Generate and store, rather than store only or compute at read time. The + label stays in the source so SPARQL and third-party tooling behave normally, + while a script regenerates it at every commit and the gate fails if a stored + label differs from the derived one. Redundancy becomes harmless because it + is checked: rather than hoping two facts stay consistent, the model makes + inconsistency detectable. + examples: + - {from: label absent, to: violation, note: null} + - {from: label differing from the derivation, to: "violation — drift", note: null} + +# --------------------------------------------------------------- DECLARATION + +- id: TN-024 + category: Declaration + title: Every active term carries a comment + statement: > + rdfs:comment is mandatory on every active term. On a property the comment is + structured: the definition, plus the scope constraint or the confusion to be + avoided. On a class the form is free. + scope: [class, object_property, datatype_property, annotation_property] + severity: BLOCKING + control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A comment that paraphrases the name prevents nothing. A good one prevents a + specific error: it says where the property may and may not be declared, or + which neighbouring term it must not be confused with. The structure is + imposed on properties because they are where ambiguity costs most — they are + asserted thousands of times by people who will not read the model. The + script checks presence; only a reviewer checks value, which is why the rule + names two tiers. + examples: + - {from: "\"the owner\"", to: "definition plus where it may be declared and what it excludes", note: null} + +- id: TN-025 + category: Declaration + title: Every property declares its domain and range + statement: > + rdfs:domain and rdfs:range are mandatory on every property, EXCEPT where the + property is deliberately polymorphic. A polymorphic property states so in + its comment and has its scope declared in SHACL. Silent absence is a + violation; documented absence is not. + scope: [object_property, datatype_property] + severity: BLOCKING + control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A property with no domain cannot be targeted by any control. But some + properties are polymorphic by design, their scope controlled class by class + in SHACL rather than by twin properties; giving those an rdfs:domain would + trigger the RDFS retyping described in TN-028. The rule therefore separates + the two cases rather than demanding a domain everywhere. + examples: + - {from: a polymorphic property with no domain and no comment, to: violation, note: silence is indistinguishable from omission} + - {from: a polymorphic property with no domain, documented, to: conforming, note: scope declared in SHACL} + +- id: TN-026 + category: Declaration + title: Every term declares how it was authored + statement: > + pr:authoringMode is mandatory on every term. pr:harvestSource is mandatory + if and only if the mode is HARVESTED. + scope: [class, object_property, datatype_property, annotation_property] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + Provenance decides who may edit a term and what a divergence means. A term + stating where its data comes from without stating that it is harvested, or + declaring itself harvested without naming a source, is half-declared in a + way no control can catch. Declared symmetrically, provenance also makes a + harvester specifiable from the model itself rather than from a side + document. + examples: + - {from: harvestSource present, authoringMode absent, to: "authoringMode HARVESTED", note: null} + - {from: "authoringMode HARVESTED, harvestSource absent", to: harvestSource declared, note: null} + +- id: TN-027 + category: Declaration + title: A concrete class has one layer and one provenance + statement: > + Every concrete class descends from exactly one layer root and from exactly + one provenance axis. Controlled-vocabulary classes are outside the layers by + nature — a named and closed exception. + scope: [class] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A class with no layer is invisible to every layer-scoped control and to + every view organised by layer; a class with two is ambiguous in both. Where + a family of classes genuinely sits outside the data layers — actors, + governance objects — the answer is an explicit root of its own, not silence. + Adding such a root by INSERTION above an existing abstraction, rather than + by renaming it, keeps the ranges that point at that abstraction readable: a + range must state the nature of its target, not its position in the model. + examples: + - {from: a class with no parent, to: attached to its layer root, note: null} + - {from: renaming an abstraction into a layer root, to: inserting the layer root above it, note: preserves the meaning of every range that points at it} + +- id: TN-028 + category: Declaration + title: A sub-property inherits the domain of its parent + statement: > + Declaring rdfs:subPropertyOf does not shield a property from the + rdfs:domain of its parent. Every assertion of a sub-property is also an + assertion of the parent, so the parent's domain applies and its RDFS + retyping takes effect. + scope: [object_property, datatype_property] + severity: BLOCKING + control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + In RDFS a domain is not a rejecting constraint but a retyping inference. + Attaching a sub-property to a parent whose domain does not fit will silently + retype its subjects — an object of one layer quietly becomes an object of + another, entering counts, controls and traversals it was never part of. + Where two relations share a verb but not a domain, they must not be merged + and must not be linked by subPropertyOf; the unification, if one is needed, + belongs on a shared upper property with no domain of its own. + examples: + - {from: "a dashboard field asserted through a property whose domain is a warehouse column", to: "the dashboard field is inferred to BE a warehouse column", note: silent retyping} + - {from: "linking the two by rdfs:subPropertyOf", to: "no protection — the parent's domain still applies", note: null} + - {from: "a shared upper property with no domain", to: "end-to-end traversal without merging IRIs", note: the correct unification} + +# ----------------------------------------------------------------- EVOLUTION + +- id: EV-001 + category: Evolution + title: A rename is never an edit in place + statement: > + A rename creates a new term and keeps the old one with owl:deprecated true + and dcterms:isReplacedBy. No IRI is modified in place. The deprecated stub + carries no edge: no subClassOf, no label, no comment — only its type and the + two axioms. + scope: [procedure] + severity: BLOCKING + control: {tier: human, executor: commit review, procedure: null} + filiation: A-Box LC-002 + rationale: > + Nothing propagates in RDF. An IRI is a string copied into every triple, so + renaming means rewriting every triple where it appears as subject, predicate + or object — declaration, subclass axioms, domain, range, instance typing, + shape targets and paths, queries, viewers, documentation. Edges are moved, + not duplicated. The stub exists so that a consumer holding an older export + can still be told what replaced the term. + examples: + - {from: editing an IRI in place, to: new term plus deprecated stub, note: null} + +- id: EV-002 + category: Evolution + title: A withdrawal requires proof of non-instantiation + statement: > + A term is removed only after a count proving it was never instantiated, the + count being attached to the commit. Without proof, deprecation only. + scope: [procedure] + severity: BLOCKING + control: {tier: script, executor: count query attached to the commit, procedure: null} + filiation: null + rationale: > + Removal is the one irreversible operation on a vocabulary. The proof is + cheap — a single counting query — and it is what separates a safe withdrawal + from a silent data loss. + examples: + - {from: deleting a term, to: "SELECT (COUNT(?s) AS ?n) WHERE { ?s a pr:Term }", note: must return zero} + +- id: EV-003 + category: Evolution + title: A deletion removes every reference to the subject + statement: > + A subject removed while its IRI is still cited elsewhere is recreated by + RDFS inference. Every reference is removed in the same operation. + scope: [procedure] + severity: BLOCKING + control: {tier: script, executor: dangling-reference check in the migration script, procedure: null} + filiation: null + rationale: > + A graph has no foreign keys. Deleting the block that declares a subject + leaves every triple that names it intact, and a reasoner will reconstruct a + hollow node from them — present in traversals, absent from every control. + examples: + - {from: deleting a subject block, to: deleting the block and every triple naming it, note: null} + +- id: EV-004 + category: Evolution + title: A change is validated before it is applied + statement: > + Every modification of the model is agreed before any artifact is written. + Dry run first, apply second. + scope: [procedure] + severity: BLOCKING + control: {tier: human, executor: commit review, procedure: null} + filiation: null + rationale: > + The damage from a bad change is rarely in the change itself but in + everything that silently depended on the old state. A dry run that reports + what it would touch is what makes that dependency visible while it can still + be discussed. + examples: + - {from: "migrate.py --apply as the first run", to: "migrate.py, then --apply", note: dry run is the default} + +- id: EV-005 + category: Evolution + title: Graph edits go through an RDF parser + statement: > + Any transformation of the graph is performed with an RDF parser. No regular + expression applied line by line to Turtle. + scope: [procedure, script] + severity: BLOCKING + control: {tier: human, executor: script review, procedure: null} + filiation: null + rationale: > + Turtle is a structured syntax and a line is not a unit of meaning. A literal + may contain the very characters a regex splits on, so a text-level edit that + looks correct on every example can truncate a value on the one that matters. + A parser knows the difference between a delimiter and a character inside a + string; a regex does not, and every workaround for that is a partial + reimplementation of the parser. + examples: + - {from: "re.sub on the file text", to: remove and add on parsed triples, note: null} + - {from: hand-written clause splitting, to: parser, note: a workaround is evidence the wrong tool is in use} + +- id: EV-006 + category: Evolution + title: Every script is idempotent + statement: > + A script may be replayed without changing the result. Its guard tests the + TARGET state, not the source state, and it counts its own result. Acceptance + test: two consecutive runs, the second reporting zero modifications. + scope: [script] + severity: BLOCKING + control: {tier: script, executor: double run in continuous integration, procedure: null} + filiation: null + rationale: > + A chain of transformations is only safe if any step can be replayed. A guard + that tests whether work remains to be done breaks as soon as another script + has already changed the source; a guard that tests whether the target state + exists survives it. The self-count is what turns the property into a + verifiable one. + examples: + - {from: appending unconditionally, to: appending only if the target is absent, note: guard on the target state} + +- id: EV-007 + category: Evolution + title: The replayable source is the start of the chain + statement: > + The replayable source is the starting file plus the ordered sequence of + scripts, never an intermediate committed state. Hand-editing a file produced + by a script is forbidden, including for a typo: the script is corrected and + the chain replayed. + scope: [procedure] + severity: BLOCKING + control: {tier: human, executor: commit review, procedure: null} + filiation: null + rationale: > + A committed intermediate file is a convenience, not a source. As soon as one + hand edit exists in it that no script performs, replaying the chain no + longer reproduces the committed state and nobody can say which of the two is + right. Keeping the chain authoritative is also what makes a failed migration + recoverable: restart from the last true source with a corrected script, + rather than repair a half-migrated file by hand. + examples: + - {from: fixing a typo in a generated file, to: fixing the script and replaying, note: null} + +- id: EV-008 + category: Evolution + title: Every artifact has a producer script + statement: > + No artifact contains hard-coded data. Every deliverable — viewer, + documentation, workbook, diagram — is regenerated from a versioned source by + a script. + scope: [script, artifact] + severity: BLOCKING + control: {tier: human, executor: artifact review, procedure: null} + filiation: null + rationale: > + An artifact that cannot be regenerated becomes stale the moment the model + moves, and there is no way to tell whether it is stale or merely different. + This rulebook applies the rule to itself: the source is one file, the + document and the workbook are generated from it. + examples: + - {from: a hand-maintained workbook, to: a source file plus a generator, note: null} + +- id: EV-009 + category: Evolution + title: A change propagates in a single merge + statement: > + A change is propagated to the vocabulary, the instances, the shapes, the + consumers and the documentation in one branch, merged as a block. The main + line never sees a state where one has moved and another has not. + scope: [procedure] + severity: BLOCKING + control: {tier: human, executor: merge review, procedure: null} + filiation: null + rationale: > + Partial propagation is not a smaller version of a change, it is a different + and invalid state. Its worst form is silent: a gate that validates the + stale half and reports success. A single merge makes the intermediate state + unreachable rather than merely discouraged. + examples: + - {from: committing the vocabulary and updating the shapes later, to: one branch, one merge, one tag, note: null} + +- id: EV-010 + category: Evolution + title: Shapes declare the ontology version they target + statement: > + The shapes graph declares dcterms:conformsTo equal to the owl:versionIRI of + the ontology. Any divergence aborts validation immediately. + scope: [shape, script] + severity: BLOCKING + control: {tier: script, executor: run_shacl_validation.py, procedure: null} + filiation: null + rationale: > + Validating against the wrong shapes does not fail loudly. It produces a long + list of violations that reads exactly like a regression of the model, and + the time goes into interpreting them rather than noticing the mismatch. One + declared triple and one check turn that into a single line naming the real + problem. + examples: + - {from: hundreds of violations to interpret, to: "ABORT — shapes target a different ontology version", note: null} + +- id: EV-011 + category: Evolution + title: No blocking rule without an executor + statement: > + A rule may be declared BLOCKING only if a named script or shape enforces it. + A blocking rule with no control is a statement of intent. + scope: [rule] + severity: BLOCKING + control: {tier: script, executor: rulebook source self-check, procedure: null} + filiation: null + rationale: > + An unenforced blocking rule is worse than an absent one: it is cited as + though it held, and the gap only surfaces when something it should have + caught reaches production. Making the executor a mandatory field turns the + omission into something a script can detect on the rulebook itself. + examples: + - {from: "severity BLOCKING with no executor", to: violation, note: detected by the generators before they write anything} + +- id: EV-012 + category: Evolution + title: Every rule names the tier that enforces it + statement: > + Controls exist at three tiers — SHACL for the shape of the graph, scripts + for the file, the lexicon and naming, human review for semantics. Every rule + names its tier. A rule with no assigned tier is a violation. + scope: [rule] + severity: BLOCKING + control: {tier: script, executor: rulebook source self-check, procedure: null} + filiation: null + rationale: > + SHACL validates a graph, not a file, and not meaning. Two IRIs standing for + the same notion produce two individually valid graphs; a duplicated block of + text produces identical triples and no complaint. Naming the tier forces the + question of what actually catches each rule, and pushes the ones SHACL + cannot see to a control that can: a normalised-label index reviewed by a + person, a file-level hash, a naming script, a review checklist. + examples: + - {from: two IRIs for one notion, to: normalised label index plus human review, note: SHACL cannot see it} + - {from: a duplicated text block, to: file-level block hashing, note: identical triples, no violation} + +- id: EV-013 + category: Evolution + title: No sh:pattern outside the SHACL specification + statement: > + sh:pattern uses XPath 2.0 regular expressions, which support neither + lookbehind nor lookahead. A constraint is expressed positively — what the + string must be — rather than negatively. Where that is impossible the rule + leaves SHACL and moves to the script tier under EV-012. + scope: [shape] + severity: BLOCKING + control: {tier: script, executor: run_shacl_validation.py, procedure: null} + filiation: A-Box NR-001 + rationale: > + A pattern written in a richer regex dialect does not degrade gracefully: the + validator fails to compile it, and the rule that depended on it silently + stops being enforced while still being declared blocking. + examples: + - {from: "a pattern using (? + The namespace of terms is not versioned. The version lives in + owl:versionIRI alone. + scope: [ontology] + severity: BLOCKING + control: {tier: script, executor: check_tbox_naming.py, procedure: null} + filiation: null + rationale: > + A versioned term namespace changes every IRI at every release, breaking + every consumer for no benefit. It also fails silently in the other + direction: a file bound to an outdated namespace targets IRIs that no longer + exist, so every shape matches nothing and validation passes while testing + nothing at all. + examples: + - {from: "a prefix bound to a versioned namespace", to: "a prefix bound to the stable namespace", note: the version lives in owl:versionIRI} + +- id: EV-015 + category: Evolution + title: Every procedure that writes produces an execution log + statement: > + Any procedure that modifies the graph writes a versioned execution log, + committed alongside the change. The log records the attempt number and + timestamp, a checksum of every input consumed, the ontology version and the + version the shapes declare they target, a count per step, the full + validation report, and a difference against the previous attempt. + scope: [procedure, script] + severity: BLOCKING + control: {tier: script, executor: the procedure's own runner, procedure: null} + filiation: null + rationale: > + A counter printed to the terminal disappears with the terminal. When a + validation run fails, the question is never only what failed but whether it + is the same failure as the previous attempt: a violation that persists + unchanged after a correction says the cause lies elsewhere, while one that + moves every time says the change is being patched rather than fixed. Those + are opposite diagnoses and neither is available without a record of the + previous run. The input checksums serve the same purpose one level down, + separating a wrong model from a stale file — the ambiguity that made a past + incident expensive to explain. The log is also the natural carrier of the + proof of non-instantiation required by EV-002, which otherwise has no + defined format. + examples: + - {from: a counter printed to standard output, to: a versioned log committed with the change, note: null} + - {from: "a failed run with no record of the previous one", to: "a difference against the previous attempt", note: what makes correct-or-abandon decidable} + +# ====== ANNEX: ABSTRACTNESS (TN-006, TN-007) ====== +abstractness: +- {term: MetaModelObject, is_abstract: true, note: "root of the model"} +- {term: DefinedObject, is_abstract: true, note: "provenance axis"} +- {term: CapturedObject, is_abstract: true, note: "provenance axis"} +- {term: OwnershipLayerObject, is_abstract: true, note: "layer root"} +- {term: BusinessLayerObject, is_abstract: true, note: "layer root"} +- {term: LogicalLayerObject, is_abstract: true, note: "layer root"} +- {term: PhysicalLayerObject, is_abstract: true, note: "NEW — omission in v1.6"} +- {term: DeliveryLayerObject, is_abstract: true, note: "layer root"} +- {term: ConsumptionLayerObject, is_abstract: true, note: "layer root"} +- {term: GovernanceLayerObject, is_abstract: true, note: "NEW — seventh layer root"} +- {term: DataStructure, is_abstract: true, note: "harvesting types the concrete sort"} +- {term: KeyConstraint, is_abstract: true, note: "only PrimaryKey and ForeignKey instantiate"} +- {term: Actor, is_abstract: true, note: "renaming would degrade the range of ownedBy and hasPublisher"} +- {term: DataDomain, is_abstract: false, note: null} +- {term: DataSubDomain, is_abstract: false, note: null} +- {term: BusinessObject, is_abstract: false, note: "ends in Object and is heavily instantiated - refutes any suffix rule"} +- {term: BusinessConcept, is_abstract: false, note: null} +- {term: Metric, is_abstract: false, note: "concrete despite having KeyPerformanceIndicator beneath it"} +- {term: KeyPerformanceIndicator, is_abstract: false, note: null} +- {term: DataObject, is_abstract: false, note: null} +- {term: DataElement, is_abstract: false, note: null} +- {term: Database, is_abstract: false, note: null} +- {term: Schema, is_abstract: false, note: null} +- {term: Field, is_abstract: false, note: null} +- {term: System, is_abstract: false, note: null} +- {term: Transformation, is_abstract: false, note: null} +- {term: BaseTable, is_abstract: false, note: null} +- {term: ExternalTable, is_abstract: false, note: null} +- {term: View, is_abstract: false, note: null} +- {term: PrimaryKey, is_abstract: false, note: null} +- {term: ForeignKey, is_abstract: false, note: null} +- {term: DataProduct, is_abstract: false, note: null} +- {term: DataContract, is_abstract: false, note: null} +- {term: DataInterface, is_abstract: false, note: null} +- {term: BusinessIntelligenceWorkspace, is_abstract: false, note: null} +- {term: BusinessIntelligenceDataSource, is_abstract: false, note: null} +- {term: BusinessIntelligenceField, is_abstract: false, note: null} +- {term: BusinessIntelligenceReport, is_abstract: false, note: null} +- {term: DataDomainOwner, is_abstract: false, note: null} +- {term: DataSubDomainOwner, is_abstract: false, note: null} +- {term: DataProductOwner, is_abstract: false, note: null} +- {term: DataSteward, is_abstract: false, note: null} +- {term: DataGovernanceLead, is_abstract: false, note: null} +- {term: SystemType, is_abstract: false, note: "must stay concrete — it types five individuals"} + +# ====== ANNEX: DISPLAY (TN-011, TN-022) ====== +display: +- {iri: DataDomain, label: "Data Domain", short_label: "Domain", acronym: "DD", note: "DD is already the short form used inside A-Box identifiers under NR-014"} +- {iri: DataSubDomain, label: "Data Sub Domain", short_label: "Sub Domain", acronym: "SD", note: "no hyphen"} +- {iri: DataDomainOwner, label: "Data Domain Owner", short_label: "Domain Owner", acronym: "DDO", note: null} +- {iri: DataSubDomainOwner, label: "Data Sub Domain Owner", short_label: "Sub Domain Owner", acronym: "SDO", note: null} +- {iri: DataProductOwner, label: "Data Product Owner", short_label: "Product Owner", acronym: "PO", note: "NOT DPO — the initialism is taken by Data Protection Officer"} +- {iri: DataSteward, label: "Data Steward", short_label: "Steward", acronym: null, note: null} +- {iri: DataGovernanceLead, label: "Data Governance Lead", short_label: "Governance Lead", acronym: "DGL", note: null} +- {iri: DataElement, label: "Data Element", short_label: "Element", acronym: null, note: "drives the name of hasElement and hasGrainElement"} +- {iri: BusinessConcept, label: "Business Concept", short_label: "Concept", acronym: null, note: "drives the name of usesConcept"} +- {iri: BusinessObject, label: "Business Object", short_label: null, acronym: "BO", note: null} +- {iri: KeyPerformanceIndicator, label: "Key Performance Indicator", short_label: "KPI", acronym: "KPI", note: "the acronym channel is what makes TN-002 bearable"} +- {iri: BusinessIntelligenceWorkspace, label: "Business Intelligence Workspace", short_label: "BI Workspace", acronym: null, note: null} +- {iri: BusinessIntelligenceDataSource, label: "Business Intelligence Data Source", short_label: "BI Data Source", acronym: null, note: null} +- {iri: BusinessIntelligenceField, label: "Business Intelligence Field", short_label: "BI Field", acronym: null, note: null} +- {iri: BusinessIntelligenceReport, label: "Business Intelligence Report", short_label: "BI Report", acronym: null, note: null} +- {iri: OwnershipLayerObject, label: "Ownership Layer Object", short_label: null, acronym: null, note: "the categorisation role moves into rdfs:comment"}