diff --git a/governance/generate_bpmn.py b/governance/generate_bpmn.py
new file mode 100644
index 0000000..943e1d9
--- /dev/null
+++ b/governance/generate_bpmn.py
@@ -0,0 +1,206 @@
+#!/usr/bin/env python3
+"""
+Generate strict, queryable BPMN 2.0 files from processes.yaml (EV-008).
+
+USAGE
+ python3 generate_bpmn.py [output_dir]
+
+One file per procedure, in bpmn/. Rule identifiers are carried in
+extensionElements under the project namespace, so a rule can be traced to every
+procedure that enforces it with a single XPath expression:
+
+ //bpmn:process[.//pr:rule='TN-002']/@id
+
+Diagram interchange is emitted with a simple left-to-right layout so the files
+open directly in any BPMN editor. Layout is cosmetic and never authoritative.
+"""
+import os
+import sys
+from xml.sax.saxutils import escape
+
+import yaml
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+SRC = os.path.join(HERE, "processes.yaml")
+OUTDIR = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "bpmn")
+
+BPMN = "http://www.omg.org/spec/BPMN/20100524/MODEL"
+BPMNDI = "http://www.omg.org/spec/BPMN/20100524/DI"
+DI = "http://www.omg.org/spec/DD/20100524/DI"
+DC = "http://www.omg.org/spec/DD/20100524/DC"
+
+TAG = {
+ "start": "startEvent",
+ "end": "endEvent",
+ "userTask": "userTask",
+ "scriptTask": "scriptTask",
+ "callActivity": "callActivity",
+ "gateway": "exclusiveGateway",
+}
+W = {"start": 36, "end": 36, "gateway": 50}
+H = {"start": 36, "end": 36, "gateway": 50}
+DEFAULT_W, DEFAULT_H = 150, 70
+COL, ROW = 200, 130
+
+
+def selfcheck(doc, rule_ids):
+ """A procedure must be a connected graph, and every rule it cites must exist."""
+ problems = []
+ ids = {p["id"] for p in doc["processes"]}
+ for p in doc["processes"]:
+ nodes = {n["id"] for n in p["flow"]}
+ for f in p["flows"]:
+ for side in ("from", "to"):
+ if f[side] not in nodes:
+ problems.append("%s: flow references unknown node %s" % (p["id"], f[side]))
+ targets = {f["to"] for f in p["flows"]}
+ sources = {f["from"] for f in p["flows"]}
+ for n in p["flow"]:
+ if n["type"] == "start" and n["id"] in targets:
+ problems.append("%s: start event %s has an incoming flow" % (p["id"], n["id"]))
+ if n["type"] == "end" and n["id"] in sources:
+ problems.append("%s: end event %s has an outgoing flow" % (p["id"], n["id"]))
+ if n["type"] not in ("start", "end") and n["id"] not in targets:
+ problems.append("%s: node %s is unreachable" % (p["id"], n["id"]))
+ if n["type"] != "end" and n["id"] not in sources:
+ problems.append("%s: node %s is a dead end" % (p["id"], n["id"]))
+ if n["type"] == "callActivity" and n.get("calls") not in ids:
+ problems.append("%s: %s calls unknown procedure %s"
+ % (p["id"], n["id"], n.get("calls")))
+ for r in n.get("rules") or []:
+ if rule_ids and r not in rule_ids:
+ problems.append("%s/%s: unknown rule %s" % (p["id"], n["id"], r))
+ return problems
+
+
+def layout(proc):
+ """Assign a column by longest path from start, a row to disambiguate siblings."""
+ succ = {}
+ for f in proc["flows"]:
+ succ.setdefault(f["from"], []).append(f["to"])
+ starts = [n["id"] for n in proc["flow"] if n["type"] == "start"]
+ depth = {}
+ for s in starts:
+ stack = [(s, 0)]
+ seen = set()
+ while stack:
+ node, d = stack.pop()
+ if (node, d) in seen:
+ continue
+ seen.add((node, d))
+ if depth.get(node, -1) < d:
+ depth[node] = d
+ for nxt in succ.get(node, []):
+ if d < 60:
+ stack.append((nxt, d + 1))
+ used = {}
+ pos = {}
+ for n in proc["flow"]:
+ c = depth.get(n["id"], 0)
+ r = used.get(c, 0)
+ used[c] = r + 1
+ pos[n["id"]] = (60 + c * COL, 80 + r * ROW)
+ return pos
+
+
+def build(proc, ns):
+ pid = proc["id"]
+ o = []
+ a = o.append
+ a('')
+ a('' % (BPMN, BPMNDI, DI, DC, ns, pid, ns))
+ a(' '
+ % (pid, escape(proc["name"])))
+ doc = "%s | Scope: %s | Trigger: %s" % (
+ proc["name"], proc.get("scope", ""), proc.get("trigger", ""))
+ a(' %s' % escape(doc))
+
+ incoming, outgoing = {}, {}
+ for i, f in enumerate(proc["flows"], start=1):
+ fid = "flow_%d" % i
+ f["_id"] = fid
+ outgoing.setdefault(f["from"], []).append(fid)
+ incoming.setdefault(f["to"], []).append(fid)
+
+ for n in proc["flow"]:
+ tag = TAG[n["type"]]
+ attrs = 'id="%s" name="%s"' % (n["id"], escape(n["name"]))
+ if n["type"] == "callActivity":
+ attrs += ' calledElement="%s"' % n["calls"]
+ a(' ' % (tag, attrs))
+ rules = n.get("rules") or []
+ if rules:
+ a(' ')
+ a(' ')
+ for r in rules:
+ a(' %s' % r)
+ a(' ')
+ a(' ')
+ for fid in incoming.get(n["id"], []):
+ a(' %s' % fid)
+ for fid in outgoing.get(n["id"], []):
+ a(' %s' % fid)
+ a(' ' % tag)
+
+ for f in proc["flows"]:
+ nm = ' name="%s"' % escape(f["condition"]) if f.get("condition") else ""
+ a(' '
+ % (f["_id"], nm, f["from"], f["to"]))
+ a(' ')
+
+ pos = layout(proc)
+ a(' ' % pid)
+ a(' ' % (pid, pid))
+ for n in proc["flow"]:
+ x, y = pos[n["id"]]
+ w, h = W.get(n["type"], DEFAULT_W), H.get(n["type"], DEFAULT_H)
+ a(' ' % (n["id"], n["id"]))
+ a(' ' % (x, y, w, h))
+ a(' ')
+ for f in proc["flows"]:
+ sx, sy = pos[f["from"]]
+ tx, ty = pos[f["to"]]
+ sw = W.get(next(n["type"] for n in proc["flow"] if n["id"] == f["from"]), DEFAULT_W)
+ sh = H.get(next(n["type"] for n in proc["flow"] if n["id"] == f["from"]), DEFAULT_H)
+ th = H.get(next(n["type"] for n in proc["flow"] if n["id"] == f["to"]), DEFAULT_H)
+ a(' ' % (f["_id"], f["_id"]))
+ a(' ' % (sx + sw, sy + sh // 2))
+ a(' ' % (tx, ty + th // 2))
+ a(' ')
+ a(' ')
+ a(' ')
+ a('')
+ return "\n".join(o) + "\n"
+
+
+def main():
+ doc = yaml.safe_load(open(SRC, encoding="utf-8"))
+ rule_ids = set()
+ rb = os.path.join(HERE, "rules.yaml")
+ if os.path.exists(rb):
+ rule_ids = {r["id"] for r in yaml.safe_load(open(rb, encoding="utf-8"))["rules"]}
+
+ problems = selfcheck(doc, rule_ids)
+ if problems:
+ print("SELF-CHECK FAILED")
+ for p in problems:
+ print(" " + p)
+ sys.exit(1)
+
+ os.makedirs(OUTDIR, exist_ok=True)
+ ns = doc["meta"]["namespace"]
+ for proc in doc["processes"]:
+ path = os.path.join(OUTDIR, "%s.bpmn" % proc["id"])
+ open(path, "w", encoding="utf-8").write(build(proc, ns))
+ n_rules = len({r for n in proc["flow"] for r in (n.get("rules") or [])})
+ print(" %-4s %-46s %2d nodes, %2d flows, %2d rules"
+ % (proc["id"], proc["name"], len(proc["flow"]),
+ len(proc["flows"]), n_rules))
+ print("self-check passed: %d procedures, graphs connected, every rule known"
+ % len(doc["processes"]))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/governance/generate_processbook_md.py b/governance/generate_processbook_md.py
new file mode 100644
index 0000000..dfe5abc
--- /dev/null
+++ b/governance/generate_processbook_md.py
@@ -0,0 +1,157 @@
+#!/usr/bin/env python3
+"""
+Generate the Markdown processbook, with Mermaid views, from processes.yaml
+(EV-008). The BPMN files remain the authoritative form; the views here are
+derived from the same source and cannot drift from it.
+
+USAGE
+ python3 generate_processbook_md.py [output.md]
+"""
+import os
+import sys
+
+import yaml
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+SRC = os.path.join(HERE, "processes.yaml")
+OUT = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "PR_TBox_Processbook.md")
+
+SHAPE = {
+ "start": ("([", "])"),
+ "end": ("([", "])"),
+ "userTask": ("[", "]"),
+ "scriptTask": ("[", "]"),
+ "callActivity": ("[[", "]]"),
+ "gateway": ("{", "}"),
+}
+KIND = {
+ "userTask": "human",
+ "scriptTask": "script",
+ "callActivity": "calls",
+ "gateway": "decision",
+ "start": "start",
+ "end": "end",
+}
+
+
+def flow(t):
+ return " ".join((t or "").split())
+
+
+def mermaid(proc):
+ out = ["```mermaid", "flowchart TD"]
+ for n in proc["flow"]:
+ o, c = SHAPE[n["type"]]
+ label = n["name"].replace('"', "'")
+ if n["type"] == "callActivity":
+ label = "%s: %s" % (n["calls"], label)
+ out.append(' %s%s"%s"%s' % (n["id"], o, label, c))
+ for f in proc["flows"]:
+ if f.get("condition"):
+ out.append(' %s -->|%s| %s' % (f["from"], f["condition"], f["to"]))
+ else:
+ out.append(" %s --> %s" % (f["from"], f["to"]))
+ for n in proc["flow"]:
+ if n["type"] == "userTask":
+ out.append(" class %s human;" % n["id"])
+ elif n["type"] == "gateway":
+ out.append(" class %s decision;" % n["id"])
+ out.append(" classDef human fill:#FFF4CE,stroke:#B08900;")
+ out.append(" classDef decision fill:#E8F0F7,stroke:#3E6E96;")
+ out.append("```")
+ return out
+
+
+def main():
+ doc = yaml.safe_load(open(SRC, encoding="utf-8"))
+ m, out = doc["meta"], []
+ w = out.append
+
+ w("# %s" % m["title"])
+ w("")
+ w("**Version %s** — %s — generated %s" % (m["version"], m["status"], m["date"]))
+ w("")
+ w("> Generated from `processes.yaml`. The BPMN files in `bpmn/` are the "
+ "authoritative form; the views below are derived from the same source. "
+ "Do not edit this document (EV-007, EV-008).")
+ w("")
+ w(flow(m["scope"]))
+ w("")
+ w("## Procedures")
+ w("")
+ w("| Id | Family | Procedure | Scope |")
+ w("|---|---|---|---|")
+ for p in doc["processes"]:
+ w("| **%s** | %s | %s | %s |" % (p["id"], p["family"], p["name"], p.get("scope", "")))
+ w("")
+ w("## Reading a diagram")
+ w("")
+ w("- A **rounded** node is a start or end state.")
+ w("- A **diamond** is a decision; every outgoing path is labelled.")
+ w("- A **shaded** box is a human task: it cannot be automated, and the work "
+ "stops there until someone acts.")
+ w("- A **double-bordered** box calls another procedure by its identifier.")
+ w("")
+
+ for p in doc["processes"]:
+ w("---")
+ w("")
+ w("# %s — %s" % (p["id"], p["name"]))
+ w("")
+ w("| | |")
+ w("|---|---|")
+ w("| Family | %s |" % p["family"])
+ w("| Scope | %s |" % p.get("scope", ""))
+ w("| Trigger | %s |" % p.get("trigger", ""))
+ w("| Inputs | %s |" % ", ".join(p.get("inputs") or []))
+ w("| Outputs | %s |" % ", ".join(p.get("outputs") or []))
+ rules = sorted({r for n in p["flow"] for r in (n.get("rules") or [])})
+ w("| Rules enforced | %s |" % (", ".join(rules) or "-"))
+ calls = sorted({n["calls"] for n in p["flow"] if n["type"] == "callActivity"})
+ w("| Calls | %s |" % (", ".join(calls) or "-"))
+ w("")
+ if p.get("parameter"):
+ w("**Parameter.** %s" % flow(p["parameter"]))
+ w("")
+ if p.get("note"):
+ w("**Note.** %s" % flow(p["note"]))
+ w("")
+ out.extend(mermaid(p))
+ w("")
+ w("| Step | Type | Rules |")
+ w("|---|---|---|")
+ for n in p["flow"]:
+ if n["type"] in ("start", "end"):
+ continue
+ name = n["name"]
+ if n["type"] == "callActivity":
+ name = "%s (%s)" % (name, n["calls"])
+ w("| %s | %s | %s |"
+ % (name, KIND[n["type"]], ", ".join(n.get("rules") or []) or "-"))
+ w("")
+
+ w("---")
+ w("")
+ w("# Rule coverage")
+ w("")
+ w("Which procedures enforce each rule. Derived, not maintained: a rule cited "
+ "in a diagram appears here automatically.")
+ w("")
+ cover = {}
+ for p in doc["processes"]:
+ for n in p["flow"]:
+ for r in n.get("rules") or []:
+ cover.setdefault(r, set()).add(p["id"])
+ w("| Rule | Procedures |")
+ w("|---|---|")
+ for r in sorted(cover):
+ w("| %s | %s |" % (r, ", ".join(sorted(cover[r]))))
+ w("")
+
+ open(OUT, "w", encoding="utf-8").write("\n".join(out))
+ print("written: %s (%d lines, %d procedures, %d rules covered)"
+ % (OUT, len(out), len(doc["processes"]), len(cover)))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/governance/generate_rulebook_md.py b/governance/generate_rulebook_md.py
new file mode 100644
index 0000000..25c4326
--- /dev/null
+++ b/governance/generate_rulebook_md.py
@@ -0,0 +1,154 @@
+#!/usr/bin/env python3
+"""
+Generate the Markdown rulebook from rules.yaml (EV-008).
+
+USAGE
+ python3 generate_rulebook_md.py [output.md]
+
+rules.yaml is the single source of truth. This script never edits it.
+The output must never be edited by hand (EV-007).
+"""
+import os
+import sys
+
+import yaml
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+SRC = os.path.join(HERE, "rules.yaml")
+OUT = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "PR_TBox_Rulebook.md")
+
+
+def selfcheck(doc):
+ """EV-011 and EV-012 applied to the rulebook itself."""
+ problems = []
+ for r in doc["rules"]:
+ ctl = r.get("control") or {}
+ if not ctl.get("tier"):
+ problems.append("%s: no control tier (EV-012)" % r["id"])
+ if r.get("severity") == "BLOCKING" and not ctl.get("executor"):
+ problems.append("%s: BLOCKING without an executor (EV-011)" % r["id"])
+ if r.get("category") not in doc["categories"]:
+ problems.append("%s: unknown category" % r["id"])
+ return problems
+
+
+def flow(text):
+ """Collapse a YAML folded scalar into a single paragraph."""
+ return " ".join((text or "").split())
+
+
+def joined(v):
+ return ", ".join(v) if isinstance(v, list) else (v or "-")
+
+
+def main():
+ doc = yaml.safe_load(open(SRC, encoding="utf-8"))
+ problems = selfcheck(doc)
+ if problems:
+ print("SELF-CHECK FAILED")
+ for p in problems:
+ print(" " + p)
+ sys.exit(1)
+
+ m, out = doc["meta"], []
+ w = out.append
+
+ w("# %s" % m["title"])
+ w("")
+ w("**Version %s** — %s — generated %s" % (m["version"], m["status"], m["date"]))
+ w("")
+ w("> Generated from `rules.yaml`. Do not edit this document: edit the source "
+ "and regenerate (EV-007, EV-008).")
+ w("")
+ w("## Scope")
+ w("")
+ w(flow(m["scope"]))
+ w("")
+ w(flow(m["audience"]))
+ w("")
+ w("## Categories")
+ w("")
+ w("| Category | Subject | Rules |")
+ w("|---|---|---|")
+ for k, v in doc["categories"].items():
+ n = sum(1 for r in doc["rules"] if r["category"] == k)
+ w("| **%s** | %s | %d |" % (k, v["title"], n))
+ w("")
+ w("## Severity")
+ w("")
+ for k, v in doc["severity_model"].items():
+ w("- **%s** — %s" % (k, flow(v)))
+ w("")
+
+ for cat, info in doc["categories"].items():
+ w("---")
+ w("")
+ w("# %s — %s" % (cat, info["title"]))
+ w("")
+ w(flow(info["intent"]))
+ w("")
+ for r in [x for x in doc["rules"] if x["category"] == cat]:
+ ctl = r.get("control") or {}
+ w("## %s — %s" % (r["id"], r["title"]))
+ w("")
+ w("**Statement.** %s" % flow(r["statement"]))
+ w("")
+ w("| | |")
+ w("|---|---|")
+ w("| Severity | **%s** |" % r["severity"])
+ w("| Applies to | %s |" % joined(r.get("scope")))
+ w("| Enforced at | %s |" % joined(ctl.get("tier")))
+ w("| Executor | `%s` |" % (ctl.get("executor") or "-"))
+ w("| Procedure | %s |" % (ctl.get("procedure") or "_pending_"))
+ if r.get("filiation"):
+ w("| Related | %s |" % r["filiation"])
+ w("")
+ if r.get("rationale"):
+ w("**Why.** %s" % flow(r["rationale"]))
+ w("")
+ if r.get("examples"):
+ w("| From | To | Note |")
+ w("|---|---|---|")
+ for e in r["examples"]:
+ w("| %s | %s | %s |" % (e["from"], e["to"], e.get("note") or ""))
+ w("")
+
+ ab = doc["abstractness"]
+ w("---")
+ w("")
+ w("# Annex A — Abstractness")
+ w("")
+ w("Applies TN-006 and TN-007 to the model. %d classes, of which **%d abstract** "
+ "and **%d concrete**. Maintained with the model, not after it."
+ % (len(ab), sum(1 for a in ab if a["is_abstract"]),
+ sum(1 for a in ab if not a["is_abstract"])))
+ w("")
+ w("| Class | isAbstract | Note |")
+ w("|---|---|---|")
+ for a in ab:
+ w("| `%s` | %s | %s |"
+ % (a["term"], "**true**" if a["is_abstract"] else "false", a.get("note") or ""))
+ w("")
+ w("---")
+ w("")
+ w("# Annex B — Display")
+ w("")
+ w("Applies TN-011 and TN-022. The short label is what screens display and what "
+ "relation names reuse; the acronym is a search key, never an identity.")
+ w("")
+ w("| IRI | rdfs:label | shortLabel | acronym | Note |")
+ w("|---|---|---|---|---|")
+ for d in doc["display"]:
+ w("| `%s` | %s | %s | %s | %s |"
+ % (d["iri"], d["label"], d.get("short_label") or "-",
+ d.get("acronym") or "-", d.get("note") or ""))
+ w("")
+
+ open(OUT, "w", encoding="utf-8").write("\n".join(out))
+ print("self-check passed: %d rules, every tier assigned, no blocking rule "
+ "without an executor" % len(doc["rules"]))
+ print("written: %s (%d lines)" % (OUT, len(out)))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/governance/generate_rulebook_xlsx.py b/governance/generate_rulebook_xlsx.py
new file mode 100644
index 0000000..e6c8597
--- /dev/null
+++ b/governance/generate_rulebook_xlsx.py
@@ -0,0 +1,218 @@
+#!/usr/bin/env python3
+"""
+Generate the Excel rulebook from rules.yaml (EV-008).
+
+USAGE
+ python3 generate_rulebook_xlsx.py [output.xlsx]
+
+rules.yaml is the single source of truth. This script never edits it.
+The output must never be edited by hand (EV-007).
+"""
+import os
+import sys
+
+import yaml
+from openpyxl import Workbook
+from openpyxl.styles import Alignment, Border, Font, PatternFill, Side
+from openpyxl.utils import get_column_letter
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+SRC = os.path.join(HERE, "rules.yaml")
+OUT = sys.argv[1] if len(sys.argv) > 1 else os.path.join(HERE, "PR_TBox_Rulebook.xlsx")
+
+FONT = "Arial"
+INK = "1F2933"
+HEAD_FILL = PatternFill("solid", fgColor="1F2933")
+BAND = {
+ "Identifier": "E8F0F7",
+ "Label": "EAF3EC",
+ "Declaration": "FBF0E4",
+ "Evolution": "F2EAF5",
+}
+MAJOR_FILL = PatternFill("solid", fgColor="FFF4CE")
+THIN = Side(style="thin", color="C9CFD6")
+BORDER = Border(left=THIN, right=THIN, top=THIN, bottom=THIN)
+
+
+def selfcheck(doc):
+ """EV-011 and EV-012 applied to the rulebook itself."""
+ problems = []
+ for r in doc["rules"]:
+ ctl = r.get("control") or {}
+ if not ctl.get("tier"):
+ problems.append("%s: no control tier (EV-012)" % r["id"])
+ if r.get("severity") == "BLOCKING" and not ctl.get("executor"):
+ problems.append("%s: BLOCKING without an executor (EV-011)" % r["id"])
+ if r.get("category") not in doc["categories"]:
+ problems.append("%s: unknown category" % r["id"])
+ return problems
+
+
+def flow(t):
+ return " ".join((t or "").split())
+
+
+def joined(v):
+ return ", ".join(v) if isinstance(v, list) else (v if v else "")
+
+
+def header(ws, cols):
+ for i, (title, width) in enumerate(cols, start=1):
+ c = ws.cell(row=1, column=i, value=title)
+ c.font = Font(name=FONT, size=10, bold=True, color="FFFFFF")
+ c.fill = HEAD_FILL
+ c.alignment = Alignment(vertical="center", horizontal="left", wrap_text=True)
+ c.border = BORDER
+ ws.column_dimensions[get_column_letter(i)].width = width
+ ws.row_dimensions[1].height = 28
+ ws.freeze_panes = ws.cell(row=2, column=1)
+ ws.auto_filter.ref = "A1:%s1" % get_column_letter(len(cols))
+
+
+def put(ws, r, c, value, wrap=False, bold=False, fill=None, size=10):
+ cell = ws.cell(row=r, column=c, value=value)
+ cell.font = Font(name=FONT, size=size, bold=bold, color=INK)
+ cell.alignment = Alignment(vertical="top", wrap_text=wrap)
+ cell.border = BORDER
+ if fill:
+ cell.fill = fill
+ return cell
+
+
+def sheet_cover(wb, doc):
+ m = doc["meta"]
+ ws = wb.create_sheet("Cover")
+ ws.column_dimensions["A"].width = 24
+ ws.column_dimensions["B"].width = 100
+ put(ws, 1, 1, m["title"], bold=True, size=14)
+
+ rows = [
+ ("Version", "%s — %s" % (m["version"], m["status"])),
+ ("Generated", "%s from rules.yaml" % m["date"]),
+ ("Scope", flow(m["scope"])),
+ ("Audience", flow(m["audience"])),
+ ("Generation", "Generated from rules.yaml by generate_rulebook_xlsx.py. "
+ "Do not edit this workbook: edit the source and regenerate "
+ "(EV-007, EV-008)."),
+ ]
+ r = 3
+ for k, v in rows:
+ put(ws, r, 1, k, bold=True)
+ put(ws, r, 2, v, wrap=True)
+ ws.row_dimensions[r].height = 15 * (len(v) // 95 + 1)
+ r += 1
+
+ r += 1
+ put(ws, r, 1, "Categories", bold=True, size=12)
+ r += 1
+ for k, v in doc["categories"].items():
+ put(ws, r, 1, k, bold=True, fill=PatternFill("solid", fgColor=BAND[k]))
+ put(ws, r, 2, "%s — %s" % (v["title"], flow(v["intent"])), wrap=True)
+ ws.row_dimensions[r].height = 30
+ r += 1
+
+ r += 1
+ put(ws, r, 1, "Counts", bold=True, size=12)
+ r += 1
+ for label, formula in [
+ ("Rules", "=COUNTA(Rules!A2:A200)"),
+ ("of which blocking", '=COUNTIF(Rules!D2:D200,"BLOCKING")'),
+ ("Examples", "=COUNTA(Examples!A2:A400)"),
+ ("Classes", "=COUNTA(Abstractness!A2:A200)"),
+ ("of which abstract", "=COUNTIF(Abstractness!B2:B200,TRUE)"),
+ ("Display entries", "=COUNTA(Display!A2:A200)"),
+ ]:
+ put(ws, r, 1, label)
+ put(ws, r, 2, formula)
+ r += 1
+
+
+def sheet_rules(wb, doc):
+ ws = wb.create_sheet("Rules")
+ header(ws, [("Rule", 10), ("Category", 14), ("Title", 40), ("Severity", 11),
+ ("Statement", 70), ("Applies to", 24), ("Enforced at", 14),
+ ("Executor", 26), ("Procedure", 16), ("Related", 16),
+ ("Why", 90)])
+ r = 2
+ for x in doc["rules"]:
+ ctl = x.get("control") or {}
+ band = PatternFill("solid", fgColor=BAND[x["category"]])
+ put(ws, r, 1, x["id"], bold=True, fill=band)
+ put(ws, r, 2, x["category"], fill=band)
+ put(ws, r, 3, x["title"], wrap=True, bold=True)
+ put(ws, r, 4, x["severity"],
+ fill=MAJOR_FILL if x["severity"] != "BLOCKING" else None)
+ put(ws, r, 5, flow(x["statement"]), wrap=True)
+ put(ws, r, 6, joined(x.get("scope")), wrap=True)
+ put(ws, r, 7, joined(ctl.get("tier")))
+ put(ws, r, 8, ctl.get("executor") or "")
+ put(ws, r, 9, ctl.get("procedure") or "pending")
+ put(ws, r, 10, x.get("filiation") or "")
+ put(ws, r, 11, flow(x.get("rationale")), wrap=True)
+ ws.row_dimensions[r].height = 78
+ r += 1
+
+
+def sheet_examples(wb, doc):
+ ws = wb.create_sheet("Examples")
+ header(ws, [("Rule", 10), ("Category", 14), ("From", 48), ("To", 48), ("Note", 62)])
+ r = 2
+ for x in doc["rules"]:
+ for e in x.get("examples") or []:
+ band = PatternFill("solid", fgColor=BAND[x["category"]])
+ put(ws, r, 1, x["id"], bold=True, fill=band)
+ put(ws, r, 2, x["category"], fill=band)
+ put(ws, r, 3, str(e["from"]), wrap=True)
+ put(ws, r, 4, str(e["to"]), wrap=True)
+ put(ws, r, 5, e.get("note") or "", wrap=True)
+ r += 1
+
+
+def sheet_abstractness(wb, doc):
+ ws = wb.create_sheet("Abstractness")
+ header(ws, [("Class", 36), ("isAbstract", 12), ("Note", 96)])
+ r = 2
+ for a in doc["abstractness"]:
+ put(ws, r, 1, a["term"], bold=a["is_abstract"])
+ put(ws, r, 2, a["is_abstract"])
+ put(ws, r, 3, a.get("note") or "", wrap=True)
+ r += 1
+
+
+def sheet_display(wb, doc):
+ ws = wb.create_sheet("Display")
+ header(ws, [("IRI", 36), ("rdfs:label", 36), ("shortLabel", 22),
+ ("acronym", 10), ("Note", 80)])
+ r = 2
+ for d in doc["display"]:
+ put(ws, r, 1, d["iri"], bold=True)
+ put(ws, r, 2, d["label"])
+ put(ws, r, 3, d.get("short_label") or "")
+ put(ws, r, 4, d.get("acronym") or "")
+ put(ws, r, 5, d.get("note") or "", wrap=True)
+ r += 1
+
+
+def main():
+ doc = yaml.safe_load(open(SRC, encoding="utf-8"))
+ problems = selfcheck(doc)
+ if problems:
+ print("SELF-CHECK FAILED")
+ for p in problems:
+ print(" " + p)
+ sys.exit(1)
+
+ wb = Workbook()
+ wb.remove(wb.active)
+ sheet_cover(wb, doc)
+ sheet_rules(wb, doc)
+ sheet_examples(wb, doc)
+ sheet_abstractness(wb, doc)
+ sheet_display(wb, doc)
+ wb.save(OUT)
+ print("self-check passed: %d rules" % len(doc["rules"]))
+ print("written: %s (%d sheets)" % (OUT, len(wb.sheetnames)))
+
+
+if __name__ == "__main__":
+ main()
diff --git a/governance/migration/migrate_tbox_v2_0.py b/governance/migration/migrate_tbox_v2_0.py
new file mode 100644
index 0000000..9501a1d
--- /dev/null
+++ b/governance/migration/migrate_tbox_v2_0.py
@@ -0,0 +1,357 @@
+#!/usr/bin/env python3
+"""
+Migrate the Pernod Ricard Data MetaModel from v1.6 to v2.0.
+
+USAGE
+ python3 migrate_tbox_v2_0.py # dry run, writes nothing
+ python3 migrate_tbox_v2_0.py --apply # writes in place
+ python3 migrate_tbox_v2_0.py --apply --log-dir logs/
+
+ --ontology path to the ontology TTL (default ../ontology/pr_metamodel.ttl)
+ --instances path to an instance TTL (repeatable)
+ --shapes path to a shapes TTL (repeatable)
+
+DESIGN
+ EV-004 dry run is the default; nothing is written without --apply
+ EV-005 every edit goes through rdflib, never a regex on the text
+ EV-006 guards test the target state, so a replay reports zero change
+ EV-015 an execution log is written for every attempt
+
+Run on GrosseBertha, inside the pinned venv:
+ . venv/bin/activate && python3 migrate_tbox_v2_0.py
+"""
+import argparse
+import datetime
+import hashlib
+import json
+import os
+import re
+import sys
+from collections import OrderedDict
+
+import yaml
+from rdflib import Graph, Literal, Namespace, RDF, RDFS, OWL, URIRef, XSD
+
+HERE = os.path.dirname(os.path.abspath(__file__))
+SPEC = os.path.join(HERE, "renames.yaml")
+RULES = os.path.normpath(os.path.join(HERE, "..", "rules.yaml"))
+
+DCTERMS = Namespace("http://purl.org/dc/terms/")
+
+
+# ----------------------------------------------------------------- utilities
+
+class Report(object):
+ """Counts every step, so that idempotence is provable and not merely hoped."""
+
+ def __init__(self):
+ self.steps = OrderedDict()
+ self.notes = []
+
+ def add(self, step, n, detail=None):
+ self.steps[step] = self.steps.get(step, 0) + n
+ if detail:
+ self.notes.append("%s: %s" % (step, detail))
+
+ @property
+ def total(self):
+ return sum(self.steps.values())
+
+
+def md5(path):
+ h = hashlib.md5()
+ with open(path, "rb") as f:
+ for chunk in iter(lambda: f.read(65536), b""):
+ h.update(chunk)
+ return h.hexdigest()
+
+
+def local(uri, ns):
+ s = str(uri)
+ return s[len(ns):] if s.startswith(ns) else None
+
+
+def decamelise(name, is_class):
+ """TN-018. Consecutive capitals are kept together: they carry an acronym."""
+ spaced = re.sub(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])", " ", name)
+ return spaced if is_class else spaced[0].lower() + spaced[1:]
+
+
+# ------------------------------------------------------------------ the work
+
+def rename(graph, pr, old, new, report, step):
+ """EV-001. Rewrite every triple naming the identifier, in all three positions."""
+ src, dst = pr[old], pr[new]
+ if (dst, None, None) in graph and (src, None, None) not in graph:
+ return 0 # EV-006: already done
+ n = 0
+ for s, p, o in list(graph):
+ ns, np_, no = s, p, o
+ if s == src:
+ ns = dst
+ if p == src:
+ np_ = dst
+ if o == src:
+ no = dst
+ if (ns, np_, no) != (s, p, o):
+ graph.remove((s, p, o))
+ graph.add((ns, np_, no))
+ n += 1
+ report.add(step, n)
+ return n
+
+
+def merge(graph, pr, src_name, dst_name, report):
+ """The source disappears into an existing target, which keeps its declaration."""
+ src, dst = pr[src_name], pr[dst_name]
+ n = 0
+ for s, p, o in list(graph.triples((None, src, None))):
+ graph.remove((s, p, o))
+ graph.add((s, dst, o))
+ n += 1
+ for s, p, o in list(graph.triples((src, None, None))):
+ graph.remove((s, p, o)) # drop the source declaration
+ n += 1
+ report.add("merge", n)
+ return n
+
+
+def drop_subject(graph, subject, report, step):
+ """EV-003. Remove the block AND every reference naming it, or inference rebuilds it."""
+ n = 0
+ for t in list(graph.triples((subject, None, None))):
+ graph.remove(t)
+ n += 1
+ for t in list(graph.triples((None, None, subject))):
+ graph.remove(t)
+ n += 1
+ for t in list(graph.triples((None, subject, None))):
+ graph.remove(t)
+ n += 1
+ report.add(step, n)
+ return n
+
+
+def count_instances(graphs, term):
+ """EV-002. The proof required before any permanent withdrawal."""
+ n = 0
+ for g in graphs:
+ n += len(list(g.triples((None, RDF.type, term))))
+ n += len(list(g.triples((None, term, None))))
+ n += len(list(g.triples((None, None, term))))
+ return n
+
+
+def migrate(onto, others, spec, rules, report):
+ ns = spec["meta"]["namespace"]
+ pr = Namespace(ns)
+ all_graphs = [onto] + others
+
+ # 1 — deprecated terms are withdrawn outright (phase clause), proof first
+ if spec["structural"].get("drop_deprecated"):
+ for subj in list(onto.subjects(OWL.deprecated, Literal(True))):
+ name = local(subj, ns) or str(subj)
+ used = sum(count_instances([g], subj) for g in others)
+ if used:
+ report.add("deprecated_kept", 1, "%s still used %d times" % (name, used))
+ continue
+ for g in all_graphs:
+ drop_subject(g, subj, report, "deprecated_dropped")
+
+ # 2 — renames, in dependency order
+ for order in (1, 2, 3):
+ for r in [x for x in spec["renames"] if x["order"] == order]:
+ for g in all_graphs:
+ rename(g, pr, r["from"], r["to"], report, "rename_order_%d" % order)
+
+ # 3 — merges
+ for m in spec.get("merges") or []:
+ for g in all_graphs:
+ merge(g, pr, m["from"], m["into"], report)
+
+ # 4 — reclassify display properties (TN-003)
+ for rc in spec["structural"]["reclassify"]:
+ term = pr[rc["term"]]
+ if (term, RDF.type, OWL.AnnotationProperty) not in onto:
+ onto.remove((term, RDF.type, getattr(OWL, rc["from"])))
+ onto.add((term, RDF.type, OWL.AnnotationProperty))
+ report.add("reclassified", 1, rc["term"])
+
+ # 5 — create the governance layer root (TN-027)
+ for c in spec["structural"].get("create_classes") or []:
+ term = pr[c["term"]]
+ if (term, RDF.type, OWL.Class) not in onto:
+ onto.add((term, RDF.type, OWL.Class))
+ onto.add((term, RDFS.subClassOf, pr[c["parent"]]))
+ onto.add((term, RDFS.label, Literal(c["label"])))
+ onto.add((term, RDFS.comment, Literal(" ".join(c["comment"].split()))))
+ onto.add((term, pr.isAbstract, Literal(True)))
+ report.add("classes_created", 1, c["term"])
+
+ # 6 — reparent (TN-027)
+ for rp in spec["structural"].get("reparent") or []:
+ term, parent = pr[rp["term"]], pr[rp["parent"]]
+ if (term, RDFS.subClassOf, parent) not in onto:
+ onto.add((term, RDFS.subClassOf, parent))
+ report.add("reparented", 1, "%s -> %s" % (rp["term"], rp["parent"]))
+
+ # 7 — sub-properties (TN-028)
+ for sp in spec["structural"].get("subproperties") or []:
+ term, parent = pr[sp["term"]], pr[sp["parent"]]
+ if (term, RDFS.subPropertyOf, parent) not in onto:
+ onto.add((term, RDFS.subPropertyOf, parent))
+ report.add("subproperties", 1, "%s -> %s" % (sp["term"], sp["parent"]))
+
+ # 8 — controlled values become literals (TN-016, TN-017)
+ for conv in spec["structural"].get("to_literal") or []:
+ prop = pr[conv["property"]]
+ if (prop, RDF.type, OWL.DatatypeProperty) not in onto:
+ onto.remove((prop, RDF.type, OWL.ObjectProperty))
+ onto.add((prop, RDF.type, OWL.DatatypeProperty))
+ onto.remove((prop, RDFS.range, None))
+ onto.add((prop, RDFS.range, XSD.string))
+ onto.add((prop, RDFS.domain, pr[conv["domain"]]))
+ report.add("to_literal_property", 1, conv["property"])
+ for g in all_graphs: # rewrite the asserted values
+ for s, p, o in list(g.triples((None, prop, None))):
+ name = local(o, ns)
+ if name and name in conv["value_map"]:
+ g.remove((s, p, o))
+ g.add((s, p, Literal(conv["value_map"][name])))
+ report.add("to_literal_values", 1)
+ for ind in conv["drop_individuals"] + [conv["drop_class"]]:
+ subj = pr[ind]
+ if (subj, None, None) in onto:
+ for g in all_graphs:
+ drop_subject(g, subj, report, "to_literal_dropped")
+
+ # 9 — abstractness, from the rulebook annex (TN-006, TN-007)
+ for a in rules["abstractness"]:
+ term = pr[a["term"]]
+ if (term, RDF.type, OWL.Class) not in onto:
+ report.add("abstract_missing_class", 1, a["term"])
+ continue
+ want = Literal(bool(a["is_abstract"]))
+ if (term, pr.isAbstract, want) not in onto:
+ onto.remove((term, pr.isAbstract, None))
+ onto.add((term, pr.isAbstract, want))
+ report.add("isAbstract_declared", 1)
+
+ # 10 — display annotations, from the rulebook annex (TN-022)
+ for d in rules["display"]:
+ term = pr[d["iri"]]
+ for prop, value in ((pr.shortLabel, d.get("short_label")),
+ (pr.acronym, d.get("acronym"))):
+ if not value:
+ continue
+ if (term, prop, Literal(value)) not in onto:
+ onto.remove((term, prop, None))
+ onto.add((term, prop, Literal(value)))
+ report.add("display_set", 1)
+
+ # 11 — regenerate every label by derivation (TN-018, TN-023)
+ kinds = (OWL.Class, OWL.ObjectProperty, OWL.DatatypeProperty, OWL.AnnotationProperty)
+ for kind in kinds:
+ for term in set(onto.subjects(RDF.type, kind)):
+ name = local(term, ns)
+ if not name:
+ continue
+ want = Literal(decamelise(name, kind == OWL.Class))
+ if (term, RDFS.label, want) not in onto:
+ onto.remove((term, RDFS.label, None))
+ onto.add((term, RDFS.label, want))
+ report.add("labels_regenerated", 1)
+
+ # 12 — bump the ontology version (EV-014: the namespace itself never moves)
+ target = URIRef(ns.rstrip("/") + "/" + spec["meta"]["to_version"])
+ for onto_iri in set(onto.subjects(RDF.type, OWL.Ontology)):
+ if (onto_iri, OWL.versionIRI, target) not in onto:
+ onto.remove((onto_iri, OWL.versionIRI, None))
+ onto.add((onto_iri, OWL.versionIRI, target))
+ report.add("version_bumped", 1, str(target))
+
+
+# ----------------------------------------------------------------------- main
+
+def main():
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--apply", action="store_true", help="write the files (default: dry run)")
+ ap.add_argument("--ontology", default=os.path.join(HERE, "..", "ontology", "pr_metamodel.ttl"))
+ ap.add_argument("--instances", action="append", default=[])
+ ap.add_argument("--shapes", action="append", default=[])
+ ap.add_argument("--log-dir", default=os.path.join(HERE, "logs"))
+ args = ap.parse_args()
+
+ spec = yaml.safe_load(open(SPEC, encoding="utf-8"))
+ rules = yaml.safe_load(open(RULES, encoding="utf-8"))
+
+ paths = [args.ontology] + args.instances + args.shapes
+ missing = [p for p in paths if not os.path.exists(p)]
+ if missing:
+ print("MISSING INPUT")
+ for p in missing:
+ print(" " + p)
+ sys.exit(1)
+
+ inputs = OrderedDict((p, md5(p)) for p in paths)
+
+ onto = Graph()
+ onto.parse(args.ontology, format="turtle")
+ others, other_paths = [], args.instances + args.shapes
+ for p in other_paths:
+ g = Graph()
+ g.parse(p, format="turtle")
+ others.append(g)
+
+ before = [len(onto)] + [len(g) for g in others]
+ report = Report()
+ migrate(onto, others, spec, rules, report)
+ after = [len(onto)] + [len(g) for g in others]
+
+ print("MIGRATION %s -> %s %s"
+ % (spec["meta"]["from_version"], spec["meta"]["to_version"],
+ "APPLY" if args.apply else "DRY RUN"))
+ print()
+ for step, n in report.steps.items():
+ print(" %-28s %6d" % (step, n))
+ print(" %-28s %6d" % ("total changes", report.total))
+ print()
+ for p, b, a in zip(paths, before, after):
+ print(" %-46s %6d -> %6d triples" % (os.path.basename(p), b, a))
+ if report.notes:
+ print()
+ for n in report.notes:
+ print(" note: " + n)
+
+ log = {
+ "attempt_timestamp": datetime.datetime.now().isoformat(timespec="seconds"),
+ "mode": "apply" if args.apply else "dry-run",
+ "from_version": spec["meta"]["from_version"],
+ "to_version": spec["meta"]["to_version"],
+ "input_checksums": inputs,
+ "steps": report.steps,
+ "total_changes": report.total,
+ "triples_before": dict(zip(paths, before)),
+ "triples_after": dict(zip(paths, after)),
+ "notes": report.notes,
+ }
+ os.makedirs(args.log_dir, exist_ok=True)
+ stamp = datetime.datetime.now().strftime("%Y%m%dT%H%M%S")
+ log_path = os.path.join(args.log_dir, "migration_%s.json" % stamp)
+ json.dump(log, open(log_path, "w", encoding="utf-8"), indent=2)
+ print("\n log: %s" % log_path)
+
+ if not args.apply:
+ print("\n DRY RUN — nothing written. Re-run with --apply when the counts "
+ "above are what you expect.")
+ return
+
+ onto.serialize(destination=args.ontology, format="turtle")
+ for p, g in zip(other_paths, others):
+ g.serialize(destination=p, format="turtle")
+ print("\n written. Replay this script now: a second run must report zero "
+ "changes (EV-006).")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/governance/migration/renames.yaml b/governance/migration/renames.yaml
new file mode 100644
index 0000000..2cf5be3
--- /dev/null
+++ b/governance/migration/renames.yaml
@@ -0,0 +1,154 @@
+# T-BOX MIGRATION TO v2.0 — RENAME MAP AND STRUCTURAL OPERATIONS
+# =============================================================================
+# Data consumed by migrate_tbox_v2_0.py. Kept separate from the script so that
+# the transformation is reviewable without reading code (EV-004).
+#
+# Not listed here and derived from other sources at run time:
+# - isAbstract declarations -> rules.yaml, annex `abstractness`
+# - shortLabel and acronym -> rules.yaml, annex `display`
+# - deprecated terms to drop -> found in the graph by owl:deprecated true
+# Deriving them keeps a single source of truth for each fact (EV-008).
+# =============================================================================
+
+meta:
+ from_version: "1.6"
+ to_version: "2.0"
+ namespace: "https://ontology.pernod-ricard.com/metamodel/"
+
+# -----------------------------------------------------------------------------
+# RENAMES — 48. Applied everywhere the identifier appears: subject, predicate,
+# object. Nothing cascades in RDF (EV-001).
+#
+# `order` is a migration constraint, not a preference:
+# 1 ordinary terms
+# 2 terms named in SHACL shapes or SPARQL constraints
+# 3 the four attributes borne by MetaModelObject, therefore by every subject
+# -----------------------------------------------------------------------------
+
+renames:
+
+# --- classes
+- {from: SubDomain, to: DataSubDomain, kind: class, order: 1}
+- {from: SubDomainOwner, to: DataSubDomainOwner, kind: class, order: 1}
+- {from: KPI, to: KeyPerformanceIndicator, kind: class, order: 1}
+- {from: BIWorkspace, to: BusinessIntelligenceWorkspace, kind: class, order: 1}
+- {from: BIDataSource, to: BusinessIntelligenceDataSource, kind: class, order: 1}
+- {from: BIField, to: BusinessIntelligenceField, kind: class, order: 1}
+- {from: BIReport, to: BusinessIntelligenceReport, kind: class, order: 1}
+
+# --- individual
+- {from: BITool, to: BusinessIntelligenceTool, kind: individual, order: 1}
+
+# --- relations
+- {from: hasDGL, to: hasGovernanceLead, kind: relation, order: 1}
+- {from: aboutConcept, to: isAbout, kind: relation, order: 1}
+- {from: primaryLocation, to: primarilyStoredIn, kind: relation, order: 1}
+- {from: inDatabase, to: isInDatabase, kind: relation, order: 1}
+- {from: inSchema, to: isInSchema, kind: relation, order: 1}
+- {from: inWorkspace, to: isInBIWorkspace, kind: relation, order: 1}
+- {from: inBIDataSource, to: isInBIDataSource, kind: relation, order: 1}
+- {from: hasBIField, to: containsBIField, kind: relation, order: 1}
+- {from: hasBIOwner, to: hasPublisher, kind: relation, order: 1}
+- {from: fromField, to: referencesSourceField, kind: relation, order: 1}
+- {from: toField, to: referencesTargetField, kind: relation, order: 1}
+- {from: biDerivedFrom, to: derivedFromSourceField, kind: relation, order: 1}
+- {from: owningDomain, to: ownedByDomain, kind: relation, order: 2}
+
+# --- attributes
+- {from: hasBusinessDefinition, to: businessDefinition, kind: attribute, order: 1}
+- {from: hasTechnicalDefinition, to: technicalDefinition, kind: attribute, order: 1}
+- {from: hasSynonym, to: synonym, kind: attribute, order: 1}
+- {from: hasFormula, to: formula, kind: attribute, order: 1}
+- {from: hasUnit, to: unit, kind: attribute, order: 1}
+- {from: hasTimeAggregation, to: timeAggregation, kind: attribute, order: 1}
+- {from: hasFormat, to: logicalFormat, kind: attribute, order: 1}
+- {from: hasPhysicalDataType, to: physicalDataType, kind: attribute, order: 1}
+- {from: hasViewDefinition, to: viewDefinition, kind: attribute, order: 1}
+- {from: hasExpression, to: expression, kind: attribute, order: 1}
+- {from: hasBusinessRule, to: businessRule, kind: attribute, order: 1}
+- {from: hasExampleValue, to: exampleValue, kind: attribute, order: 1}
+- {from: createdOn, to: creationDate, kind: attribute, order: 1}
+- {from: lastReviewedOn, to: lastReviewDate, kind: attribute, order: 1}
+- {from: harvestedOn, to: harvestDate, kind: attribute, order: 1}
+- {from: lastQueriedOn, to: lastQueryDate, kind: attribute, order: 1}
+- {from: definedBy, to: definitionAddress, kind: attribute, order: 1}
+- {from: queryCount30d, to: queryCount, kind: attribute, order: 1}
+- {from: hasShortLabel, to: shortLabel, kind: annotation, order: 2}
+- {from: hasAcronym, to: acronym, kind: annotation, order: 2}
+- {from: hasIdentifier, to: identifier, kind: attribute, order: 3}
+- {from: hasName, to: canonicalName, kind: attribute, order: 3}
+- {from: hasStatus, to: status, kind: attribute, order: 3}
+- {from: hasVersion, to: version, kind: attribute, order: 3}
+
+# --- relations that change nature (TN-016): object property -> datatype property
+- {from: hasActivationStatus, to: activationStatus, kind: relation_to_attribute, order: 1}
+- {from: deployedIn, to: environment, kind: relation_to_attribute, order: 1}
+
+# -----------------------------------------------------------------------------
+# MERGES — the source identifier disappears into an existing target.
+# Unlike a rename, the target already exists and keeps its own declaration.
+# -----------------------------------------------------------------------------
+
+merges:
+- from: biConsumes
+ into: consumes
+ widen_domain: true
+ reason: >
+ Same verb, same target, identical meaning. The domain widens and scope is
+ controlled per class by SHACL. Ranges are compatible, so no RDFS retyping
+ is introduced (TN-028).
+
+# -----------------------------------------------------------------------------
+# STRUCTURAL OPERATIONS
+# -----------------------------------------------------------------------------
+
+structural:
+
+ reclassify:
+ - {term: shortLabel, from: DatatypeProperty, to: AnnotationProperty, rule: TN-003}
+ - {term: acronym, from: DatatypeProperty, to: AnnotationProperty, rule: TN-003}
+
+ create_classes:
+ - term: GovernanceLayerObject
+ parent: MetaModelObject
+ is_abstract: true
+ label: Governance Layer Object
+ comment: >
+ Root of the governance layer. Holds the actors and governance objects that
+ are not data objects and therefore sit outside the data layers. Created by
+ insertion above Actor rather than by renaming it, so that every range
+ pointing at Actor keeps stating the nature of its target rather than its
+ position in the model.
+ rule: TN-027
+
+ reparent:
+ - {term: Actor, parent: GovernanceLayerObject, rule: TN-027}
+ - {term: PhysicalLayerObject, parent: MetaModelObject, rule: TN-027}
+
+ subproperties:
+ - {term: primarilyStoredIn, parent: storedIn, rule: TN-028}
+
+ # TN-016: closed governance states become literals. The classes and their
+ # individuals are withdrawn, and the properties become datatype properties.
+ to_literal:
+ - property: activationStatus
+ domain: DataDomain
+ drop_class: ActivationStatus
+ drop_individuals: [NotActivated, LightActivation, FullActivation]
+ value_map:
+ NotActivated: NOT_ACTIVATED
+ LightActivation: LIGHT_ACTIVATION
+ FullActivation: FULL_ACTIVATION
+ - property: environment
+ domain: CapturedObject
+ drop_class: Environment
+ drop_individuals: [Development, UserAcceptance, Production]
+ value_map:
+ Development: DEVELOPMENT
+ UserAcceptance: USER_ACCEPTANCE
+ Production: PRODUCTION
+
+ # Phase clause: before v2.0 is published the project is in design, so
+ # deprecated terms are removed outright rather than kept as stubs. EV-002
+ # still applies — the count must be zero.
+ drop_deprecated: true
diff --git a/governance/processes.yaml b/governance/processes.yaml
new file mode 100644
index 0000000..71a8e16
--- /dev/null
+++ b/governance/processes.yaml
@@ -0,0 +1,375 @@
+# PERNOD RICARD DATA METAMODEL — T-BOX PROCESSBOOK
+# =============================================================================
+# SINGLE SOURCE OF TRUTH for the procedures. The BPMN 2.0 files, the Mermaid
+# views and the Markdown document are GENERATED from this file (EV-008).
+#
+# python3 generate_bpmn.py -> bpmn/.bpmn (strict, queryable)
+# python3 generate_processbook_md.py -> PR_TBox_Processbook.md (Mermaid views)
+#
+# NODE TYPES
+# start | end events
+# userTask a human decides or writes; cannot be automated
+# scriptTask fully automated
+# callActivity invokes another procedure by its id
+# gateway exclusive decision; every outgoing flow is guarded
+#
+# Every node may carry `rules`, the identifiers of the rulebook rules it
+# enforces. That list is what makes the BPMN queryable and what feeds the
+# control.procedure field back into the rulebook.
+# =============================================================================
+
+meta:
+ title: Pernod Ricard Data MetaModel — T-Box Processbook
+ version: "0.1"
+ status: Draft for review
+ date: "2026-08-03"
+ scope: >
+ Procedures for changing the vocabulary of the model. Each procedure names
+ the rules it enforces, the steps a person must perform, and the points at
+ which the work stops rather than continues.
+ namespace: "https://ontology.pernod-ricard.com/process/"
+
+families:
+ Create: Bringing a new element into the vocabulary.
+ Verify: Establishing that what exists conforms.
+ Update: Changing an element that already exists.
+ Delete: Removing an element from the vocabulary.
+ Release: Propagating a change and publishing a version.
+
+processes:
+
+# ---------------------------------------------------------------------- X1
+- id: X1
+ family: Release
+ name: Propagate a change and publish a version
+ scope: Any validated modification of the vocabulary.
+ trigger: A change to the vocabulary has been agreed and is ready to apply.
+ inputs: [the agreed change, the target version, the list of affected artifacts]
+ outputs: [a merged branch, a tagged version, an execution log]
+ note: >
+ The most frequently invoked procedure of the processbook: every other
+ procedure ends by calling it. Two of its steps are gateways rather than
+ checks, because an idempotence failure and a validation failure must stop
+ the work rather than be noted in passing.
+ flow:
+ - {id: start, type: start, name: Change agreed}
+ - {id: branch, type: scriptTask, name: Create the migration branch, rules: [EV-009]}
+ - {id: apply, type: scriptTask, name: Apply the change through an RDF parser, rules: [EV-005, EV-006]}
+ - {id: labels, type: scriptTask, name: Regenerate labels by derivation, rules: [TN-023]}
+ - {id: abox, type: scriptTask, name: Propagate to the instances, rules: [EV-009]}
+ - {id: shapes, type: scriptTask, name: Update the shapes and align conformsTo, rules: [EV-010]}
+ - {id: artifacts, type: scriptTask, name: Regenerate every derived artifact, rules: [EV-008]}
+ - {id: replay, type: scriptTask, name: Replay the chain a second time, rules: [EV-006, EV-007]}
+ - {id: idem, type: gateway, name: "Second run reports zero change?"}
+ - {id: verify_voc, type: callActivity, calls: R1, name: Verify the vocabulary}
+ - {id: verify_abox, type: callActivity, calls: R2, name: Validate the instances}
+ - {id: log, type: scriptTask, name: Write the execution log for this attempt, rules: [EV-015]}
+ - {id: clean, type: gateway, name: "Any violation?"}
+ - {id: decide, type: userTask, name: Correct or abandon, rules: [EV-004, EV-007]}
+ - {id: drop, type: scriptTask, name: Destroy the branch and restart from the published version, rules: [EV-007]}
+ - {id: review, type: userTask, name: Review the merge, rules: [EV-004, EV-009]}
+ - {id: merge, type: scriptTask, name: Merge as a block, tag, push the tag separately, rules: [EV-009]}
+ - {id: bump, type: scriptTask, name: Increment the ontology version, rules: [EV-014]}
+ - {id: end, type: end, name: Version published}
+ - {id: aborted, type: end, name: Change abandoned}
+ flows:
+ - {from: start, to: branch}
+ - {from: branch, to: apply}
+ - {from: apply, to: labels}
+ - {from: labels, to: abox}
+ - {from: abox, to: shapes}
+ - {from: shapes, to: artifacts}
+ - {from: artifacts, to: replay}
+ - {from: replay, to: idem}
+ - {from: idem, to: verify_voc, condition: "zero change"}
+ - {from: idem, to: decide, condition: "the chain is not replayable"}
+ - {from: verify_voc, to: verify_abox}
+ - {from: verify_abox, to: log}
+ - {from: log, to: clean}
+ - {from: clean, to: review, condition: "none"}
+ - {from: clean, to: decide, condition: "at least one"}
+ - {from: decide, to: apply, condition: correct}
+ - {from: decide, to: drop, condition: abandon}
+ - {from: drop, to: aborted}
+ - {from: review, to: merge}
+ - {from: merge, to: bump}
+ - {from: bump, to: end}
+
+# ---------------------------------------------------------------------- C1
+- id: C1
+ family: Create
+ name: Create a term
+ scope: class | relation | attribute | annotation
+ parameter: >
+ The nature of the term selects the naming rules applied at the naming step:
+ a class takes TN-001, TN-002, TN-004, TN-005, TN-006, TN-007 and TN-027; a
+ relation takes TN-001, TN-002, TN-009, TN-010, TN-011 and TN-028; an
+ attribute takes TN-001, TN-002, TN-004, TN-012, TN-013, TN-014 and TN-015;
+ an annotation takes TN-001, TN-002 and TN-003.
+ trigger: A missing concept, edge or field has been identified.
+ inputs: [the intended meaning, the nature, the target layer, the provenance, the parent or the domain and range]
+ outputs: [a declared term, a published version]
+ note: >
+ The first step has no tool today. Checking that no existing term already
+ covers the need is the semantic uniqueness control that SHACL cannot
+ perform, and it stays a human task until R3 exists.
+ flow:
+ - {id: start, type: start, name: Need identified}
+ - {id: unique, type: userTask, name: Check that no existing term covers the need}
+ - {id: exists, type: gateway, name: "A term already covers it?"}
+ - {id: reuse, type: end, name: Reuse the existing term}
+ - {id: name, type: scriptTask, name: Name the term according to its nature, rules: [TN-001, TN-002, TN-003, TN-004, TN-005, TN-006, TN-009, TN-010, TN-011, TN-012, TN-013, TN-014, TN-015, TN-028]}
+ - {id: label, type: scriptTask, name: Derive the label, rules: [TN-018, TN-019, TN-020, TN-021, TN-023]}
+ - {id: short, type: userTask, name: Decide the short label and the acronym, rules: [TN-022]}
+ - {id: comment, type: userTask, name: Write the comment, rules: [TN-024]}
+ - {id: declare, type: scriptTask, name: Declare typing, provenance and attachment, rules: [TN-007, TN-025, TN-026, TN-027]}
+ - {id: validate, type: userTask, name: Validate before applying, rules: [EV-004]}
+ - {id: check, type: callActivity, calls: R1, name: Verify against the rulebook}
+ - {id: release, type: callActivity, calls: X1, name: Propagate and publish}
+ - {id: end, type: end, name: Term available}
+ flows:
+ - {from: start, to: unique}
+ - {from: unique, to: exists}
+ - {from: exists, to: reuse, condition: "yes"}
+ - {from: exists, to: name, condition: "no"}
+ - {from: name, to: label}
+ - {from: label, to: short}
+ - {from: short, to: comment}
+ - {from: comment, to: declare}
+ - {from: declare, to: validate}
+ - {from: validate, to: check}
+ - {from: check, to: release}
+ - {from: release, to: end}
+
+# ---------------------------------------------------------------------- C2
+- id: C2
+ family: Create
+ name: Create a controlled value
+ scope: typed individual | literal
+ parameter: >
+ The form is not a style choice but the outcome of the first decision. The
+ two branches have different consequences: the literal branch edits the
+ shapes and therefore the vocabulary, so it falls under the version freeze;
+ the individual branch touches nothing else.
+ trigger: A new value is needed in a controlled set.
+ inputs: [the value, the set it belongs to, whether a domain may add others]
+ outputs: [a declared value, a published version]
+ flow:
+ - {id: start, type: start, name: New value needed}
+ - {id: decide, type: userTask, name: "Must it be defined, owned or extended by a domain?", rules: [TN-016]}
+ - {id: form, type: gateway, name: "Which form?"}
+ - {id: indiv, type: scriptTask, name: Create the typed individual and derive its label, rules: [TN-001, TN-018, TN-022]}
+ - {id: literal, type: scriptTask, name: Write the literal in upper snake case, rules: [TN-017]}
+ - {id: shape, type: callActivity, calls: C3, name: Extend the closed list in the shapes}
+ - {id: validate, type: userTask, name: Validate before applying, rules: [EV-004]}
+ - {id: check, type: callActivity, calls: R1, name: Verify against the rulebook}
+ - {id: release, type: callActivity, calls: X1, name: Propagate and publish}
+ - {id: end, type: end, name: Value available}
+ flows:
+ - {from: start, to: decide}
+ - {from: decide, to: form}
+ - {from: form, to: indiv, condition: "extensible by a domain"}
+ - {from: form, to: literal, condition: "closed governance state"}
+ - {from: indiv, to: validate}
+ - {from: literal, to: shape}
+ - {from: shape, to: validate}
+ - {from: validate, to: check}
+ - {from: check, to: release}
+ - {from: release, to: end}
+
+# ---------------------------------------------------------------------- C3
+- id: C3
+ family: Create
+ name: Create a shape
+ scope: Any SHACL constraint added to the validation set.
+ trigger: A rule needs an executor, or a new class enters the validation perimeter.
+ inputs: [the rule to enforce, the target class or property]
+ outputs: [a shape, a rulebook entry naming it as executor]
+ note: >
+ The step that is forgotten is the last one before release: recording the
+ shape as the executor of its rule. Without it a rule stays declared
+ blocking with nothing enforcing it, which EV-011 forbids.
+ flow:
+ - {id: start, type: start, name: Constraint needed}
+ - {id: write, type: userTask, name: Write the constraint}
+ - {id: pattern, type: scriptTask, name: Check that every pattern is expressed positively, rules: [EV-013]}
+ - {id: portable, type: gateway, name: "Expressible within the specification?"}
+ - {id: demote, type: userTask, name: Move the rule to the script tier, rules: [EV-012]}
+ - {id: conforms, type: scriptTask, name: Align the declared ontology version, rules: [EV-010]}
+ - {id: test, type: callActivity, calls: R2, name: Test against real instances}
+ - {id: record, type: userTask, name: Record the shape as the executor of its rule, rules: [EV-011, EV-012]}
+ - {id: release, type: callActivity, calls: X1, name: Propagate and publish}
+ - {id: end, type: end, name: Shape in force}
+ flows:
+ - {from: start, to: write}
+ - {from: write, to: pattern}
+ - {from: pattern, to: portable}
+ - {from: portable, to: conforms, condition: "yes"}
+ - {from: portable, to: demote, condition: "no"}
+ - {from: demote, to: record}
+ - {from: conforms, to: test}
+ - {from: test, to: record}
+ - {from: record, to: release}
+ - {from: release, to: end}
+
+# ---------------------------------------------------------------------- D1
+- id: D1
+ family: Delete
+ name: Deprecate a term
+ scope: class | relation | attribute | annotation | individual
+ trigger: A term is superseded or no longer needed.
+ inputs: [the term, its replacement if any]
+ outputs: [a deprecated stub, a published version]
+ note: >
+ The count at step two is informative, not a condition: a term may be
+ deprecated whether or not it is instantiated. It becomes a condition only
+ in D2.
+ flow:
+ - {id: start, type: start, name: Term superseded}
+ - {id: replacement, type: userTask, name: Identify the replacement, or record that there is none}
+ - {id: count, type: scriptTask, name: Count the instances for information, rules: [EV-015]}
+ - {id: mark, type: scriptTask, name: Mark deprecated and declare the replacement, rules: [EV-001]}
+ - {id: strip, type: scriptTask, name: Strip the stub of every edge, rules: [EV-001]}
+ - {id: label, type: scriptTask, name: Remove any mention of state from the label, rules: [TN-020]}
+ - {id: release, type: callActivity, calls: X1, name: Propagate and publish}
+ - {id: end, type: end, name: Term deprecated}
+ flows:
+ - {from: start, to: replacement}
+ - {from: replacement, to: count}
+ - {from: count, to: mark}
+ - {from: mark, to: strip}
+ - {from: strip, to: label}
+ - {from: label, to: release}
+ - {from: release, to: end}
+
+# ---------------------------------------------------------------------- D2
+- id: D2
+ family: Delete
+ name: Withdraw a term permanently
+ scope: class | relation | attribute | annotation | individual
+ trigger: A deprecated term is to be removed from the vocabulary.
+ inputs: [the deprecated term]
+ outputs: [a vocabulary without the term, a proof of non-instantiation, a published version]
+ note: >
+ The reference cleaning step is what prevents phantom nodes: a subject
+ removed while its identifier is still cited elsewhere is reconstructed by
+ inference, present in traversals and absent from every control. It is a step
+ of the procedure, not a check at the end of a script.
+ flow:
+ - {id: start, type: start, name: Withdrawal requested}
+ - {id: count, type: scriptTask, name: Run the counting query, rules: [EV-002, EV-015]}
+ - {id: instantiated, type: gateway, name: "Any instance found?"}
+ - {id: keep, type: end, name: Kept deprecated}
+ - {id: proof, type: scriptTask, name: Attach the proof to the commit, rules: [EV-002, EV-015]}
+ - {id: remove, type: scriptTask, name: Remove the subject block}
+ - {id: refs, type: scriptTask, name: Remove every reference naming the subject, rules: [EV-003]}
+ - {id: check_voc, type: callActivity, calls: R1, name: Verify the vocabulary}
+ - {id: check_abox, type: callActivity, calls: R2, name: Validate the instances}
+ - {id: release, type: callActivity, calls: X1, name: Propagate and publish}
+ - {id: end, type: end, name: Term withdrawn}
+ flows:
+ - {from: start, to: count}
+ - {from: count, to: instantiated}
+ - {from: instantiated, to: keep, condition: "yes"}
+ - {from: instantiated, to: proof, condition: "no"}
+ - {from: proof, to: remove}
+ - {from: remove, to: refs}
+ - {from: refs, to: check_voc}
+ - {from: check_voc, to: check_abox}
+ - {from: check_abox, to: release}
+ - {from: release, to: end}
+
+# ---------------------------------------------------------------------- R1
+- id: R1
+ family: Verify
+ name: Verify the vocabulary against the rulebook
+ scope: The whole vocabulary, or the subset touched by a change.
+ trigger: A term has been created, changed or withdrawn, or a release is prepared.
+ inputs: [the vocabulary, the rulebook source]
+ outputs: [a conformance report]
+ note: >
+ Two tiers run in sequence, not in parallel: the script settles everything
+ mechanisable, and a person answers only for the rules no script can judge.
+ Reversing the order wastes review time on findings the script would have
+ caught.
+ flow:
+ - {id: start, type: start, name: Verification requested}
+ - {id: script, type: scriptTask, name: "Run the naming and declaration checks", rules: [TN-001, TN-002, TN-003, TN-004, TN-005, TN-006, TN-007, TN-009, TN-010, TN-011, TN-012, TN-013, TN-014, TN-015, TN-017, TN-018, TN-019, TN-020, TN-021, TN-023, TN-025, TN-026, TN-027, TN-028, EV-014]}
+ - {id: mechanised, type: gateway, name: "Any mechanised violation?"}
+ - {id: report_fail, type: end, name: Report returned with violations}
+ - {id: human, type: userTask, name: "Review the rules no script can judge", rules: [TN-008, TN-016, TN-022, TN-024]}
+ - {id: judged, type: gateway, name: "Reviewer raises an issue?"}
+ - {id: report_ok, type: end, name: Vocabulary conforms}
+ flows:
+ - {from: start, to: script}
+ - {from: script, to: mechanised}
+ - {from: mechanised, to: report_fail, condition: "at least one"}
+ - {from: mechanised, to: human, condition: none}
+ - {from: human, to: judged}
+ - {from: judged, to: report_fail, condition: "yes"}
+ - {from: judged, to: report_ok, condition: "no"}
+
+# ---------------------------------------------------------------------- R2
+- id: R2
+ family: Verify
+ name: Validate instances against the shapes
+ scope: Any instance graph in the validation perimeter.
+ trigger: Instances have changed, shapes have changed, or a release is prepared.
+ inputs: [the instance graph, the shapes, the ontology]
+ outputs: [a validation report]
+ note: >
+ The version check comes first and aborts rather than warns. Validating
+ against shapes that target another version of the ontology does not fail
+ loudly: it returns a long list of violations that reads exactly like a
+ regression of the model.
+ flow:
+ - {id: start, type: start, name: Validation requested}
+ - {id: conforms, type: scriptTask, name: "Compare the declared target version with the ontology version", rules: [EV-010]}
+ - {id: match, type: gateway, name: "Versions match?"}
+ - {id: abort, type: end, name: Aborted on version mismatch}
+ - {id: run, type: scriptTask, name: Run the shape validation}
+ - {id: violations, type: gateway, name: "Any violation?"}
+ - {id: fail, type: end, name: Report returned with violations}
+ - {id: ok, type: end, name: Instances conform}
+ flows:
+ - {from: start, to: conforms}
+ - {from: conforms, to: match}
+ - {from: match, to: abort, condition: "no"}
+ - {from: match, to: run, condition: "yes"}
+ - {from: run, to: violations}
+ - {from: violations, to: fail, condition: "at least one"}
+ - {from: violations, to: ok, condition: none}
+
+# ---------------------------------------------------------------------- R3
+- id: R3
+ family: Verify
+ name: Audit what the shapes cannot see
+ scope: The whole vocabulary and its source file.
+ trigger: Periodic audit, or before a version is published.
+ inputs: [the vocabulary, its source file]
+ outputs: [a shortlist of suspected duplicates, a list of duplicated blocks]
+ note: >
+ Shape validation reads a graph, not a file and not meaning. Two identifiers
+ standing for the same notion produce two individually valid graphs; a
+ duplicated block of text produces identical triples and no complaint. This
+ procedure is the tier those rules fall to, and its output is a shortlist for
+ a person rather than a verdict.
+ flow:
+ - {id: start, type: start, name: Audit requested}
+ - {id: index, type: scriptTask, name: "Build the normalised label index", rules: [EV-012]}
+ - {id: hash, type: scriptTask, name: "Hash every normalised subject block", rules: [EV-012]}
+ - {id: shortlist, type: gateway, name: "Any candidate found?"}
+ - {id: clean, type: end, name: Nothing to arbitrate}
+ - {id: review, type: userTask, name: Arbitrate each candidate}
+ - {id: act, type: gateway, name: "Duplication confirmed?"}
+ - {id: merge, type: end, name: Referred to the merge procedure}
+ - {id: dismissed, type: end, name: Candidates dismissed}
+ flows:
+ - {from: start, to: index}
+ - {from: index, to: hash}
+ - {from: hash, to: shortlist}
+ - {from: shortlist, to: clean, condition: none}
+ - {from: shortlist, to: review, condition: "at least one"}
+ - {from: review, to: act}
+ - {from: act, to: merge, condition: "yes"}
+ - {from: act, to: dismissed, condition: "no"}
diff --git a/governance/requirements.txt b/governance/requirements.txt
new file mode 100644
index 0000000..e76bbcf
--- /dev/null
+++ b/governance/requirements.txt
@@ -0,0 +1,3 @@
+
+PyYAML>=6.0,<7
+openpyxl>=3.1,<3.2
diff --git a/governance/rules.yaml b/governance/rules.yaml
new file mode 100644
index 0000000..3fcfd71
--- /dev/null
+++ b/governance/rules.yaml
@@ -0,0 +1,1045 @@
+# PERNOD RICARD DATA METAMODEL — T-BOX RULEBOOK
+# =============================================================================
+# SINGLE SOURCE OF TRUTH. The .md and .xlsx deliverables are GENERATED from this
+# file and must never be edited by hand (EV-007, EV-008).
+#
+# python3 generate_rulebook_md.py
+# python3 generate_rulebook_xlsx.py
+#
+# severity : BLOCKING | MAJOR | GUIDELINE — one per rule. The rule is the
+# rule; transitional tolerance belongs to a migration process,
+# not to a normative document.
+# control.tier : shacl | script | human (EV-012 — where the rule is enforced)
+# control.executor : the script or shape that enforces it (EV-011 — no blocking
+# rule without an executor)
+# control.procedure: filled in by the processbook. null until then.
+# =============================================================================
+
+meta:
+ title: Pernod Ricard Data MetaModel — T-Box Rulebook
+ version: "1.1"
+ status: Draft for review
+ date: "2026-08-03"
+ scope: >
+ Governs the vocabulary of the model itself: the IRIs, labels and declarative
+ axioms of every class, property and enumeration individual. Does not govern
+ instances, which remain under the A-Box rulebook.
+ audience: >
+ Written to be read and applied directly, by a person or by a language model,
+ without further context. Every rule states what must hold, why, and what
+ enforces it.
+
+categories:
+ Identifier:
+ title: Form of the IRI
+ intent: >
+ The IRI is identity. It is immutable in the RDF sense — changing one is a
+ migration, never an edit — so it must be unambiguous, searchable and free
+ of local convention.
+ Label:
+ title: Derivation and display
+ intent: >
+ The label is derived, not authored. Anything a reader needs that the
+ derivation cannot produce lives in a dedicated channel: the short label
+ for the spoken form, the comment for meaning.
+ Declaration:
+ title: Declarative completeness
+ intent: >
+ Nothing is left to be guessed. What a term is, what it applies to, where
+ it sits and where it came from are all stated, never inferred from
+ structure or from silence.
+ Evolution:
+ title: Change and propagation
+ intent: >
+ Every rule here is anchored on a real failure. They govern how the model
+ changes without breaking what depends on it.
+
+severity_model:
+ BLOCKING: >
+ The commit does not pass. Enforced pre-commit and in continuous integration
+ by the named executor.
+ MAJOR: >
+ Reviewed and answered for, but not machine-enforceable. Reserved for rules
+ that require human judgement, per EV-011.
+ GUIDELINE: >
+ Recommended practice with no gate.
+
+rules:
+
+# ---------------------------------------------------------------- IDENTIFIER
+
+- id: TN-001
+ category: Identifier
+ title: CamelCase IRI with no separator
+ statement: >
+ The local name of an IRI is strict CamelCase with no separator of any kind:
+ no hyphen, no underscore, no dot, no space. Initial capital for a class or
+ an enumeration individual; initial lower case for a property, whether
+ object, datatype or annotation. Consecutive capitals are admitted only where
+ they carry an acronym reproduced from a target class short label under
+ TN-011, and nowhere else.
+ scope: [class, object_property, datatype_property, annotation_property, individual]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A hyphen forces escaping in certain Turtle and SPARQL contexts and makes
+ decamelisation ambiguous: Sub-Domain has no single correct expansion.
+ Typographic convention is a display concern and belongs to the short label,
+ never to identity.
+ examples:
+ - {from: "pr:data_domain", to: "pr:DataDomain", note: no underscore}
+ - {from: "pr:Sub-Domain", to: "pr:DataSubDomain", note: no hyphen}
+ - {from: "pr:BelongsTo", to: "pr:belongsTo", note: a property starts lower case}
+
+- id: TN-002
+ category: Identifier
+ title: No acronym in the IRI
+ statement: >
+ The local name of an IRI contains no acronym, initialism or abbreviation.
+ Every short form is expanded in full. One exception, and one only: a
+ relation reproducing the short label of the class it targets carries that
+ short label as it stands, acronym included, under TN-011. No other term may
+ carry an acronym, and no exception list is maintained.
+ scope: [class, object_property, datatype_property, annotation_property, individual]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: A-Box NR-004
+ rationale: >
+ An acronym in an identifier is a local convention pretending to be a name.
+ It is unsearchable by anyone who does not already know it, and it collides
+ across domains: PO is a Product Owner in one team and a Purchase Order in
+ another. Length is not a counter-argument — an IRI is written by tooling and
+ read in context, and the short form remains available through the short
+ label and the acronym annotation. The single exception exists because a
+ relation name is read on every edge of the graph and in every query, where
+ the spoken form is what makes it legible; reproducing the target's short
+ label verbatim also keeps the relation name derivable from the class rather
+ than invented, which a truncated form would not.
+ examples:
+ - {from: "pr:KPI", to: "pr:KeyPerformanceIndicator", note: "short form moves to pr:acronym"}
+ - {from: "pr:BIDataSource", to: "pr:BusinessIntelligenceDataSource", note: null}
+ - {from: "pr:definitionUri", to: "pr:definitionAddress", note: "URI is itself an acronym"}
+ - {from: "a relation targeting a class whose short label is BI Field", to: "pr:containsBIField", note: "the one exception — see TN-011"}
+
+- id: TN-003
+ category: Identifier
+ title: Display properties are annotations
+ statement: >
+ pr:acronym and pr:shortLabel are declared owl:AnnotationProperty, never
+ owl:DatatypeProperty.
+ scope: [annotation_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A datatype property applies to an individual. Annotating a term of the
+ vocabulary with one takes the ontology out of OWL DL into OWL Full, which
+ reasoners reject and strict validators flag. An annotation property applies
+ to anything — class, property or individual. Nothing is lost on the control
+ side: SHACL validates triples and does not read OWL declarations, so every
+ shape targeting these properties keeps working unchanged.
+ examples:
+ - {from: "pr:shortLabel a owl:DatatypeProperty", to: "pr:shortLabel a owl:AnnotationProperty", note: null}
+ - {from: "pr:acronym a owl:DatatypeProperty", to: "pr:acronym a owl:AnnotationProperty", note: null}
+
+- id: TN-004
+ category: Identifier
+ title: No digit in the IRI
+ statement: >
+ The local name of a vocabulary term contains no digit. A numeric parameter —
+ a window, a threshold, a version — is a value, never an identity.
+ scope: [class, object_property, datatype_property, annotation_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: A-Box NR-014 (deliberate opposite)
+ rationale: >
+ An instance identifier does encode numbers, and that is its purpose: it
+ carries a position in the domain tree. A vocabulary term encodes nothing but
+ its own meaning. The opposition is deliberate and is stated in the ontology
+ header so that no reader takes one for a violation of the other. A term
+ named after a thirty-day window starts lying the day the window becomes
+ ninety, and nothing detects it.
+ examples:
+ - {from: "pr:queryCount30d", to: "pr:queryCount", note: the window belongs to the harvesting specification}
+
+- id: TN-005
+ category: Identifier
+ title: A class is a singular common noun
+ statement: >
+ The local name of a class is a singular noun phrase. No plural, no verb. No
+ type word such as Object, Entity or Item unless it names a genuine
+ abstraction of the model. No suffix rule is imposed on abstract classes.
+ scope: [class]
+ severity: BLOCKING
+ control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null}
+ filiation: A-Box NR-008, NR-013
+ rationale: >
+ Abstractness is not inferable from a name and must not be encoded in one.
+ The correspondence fails in both directions: classes can be abstract without
+ any suffix, and classes ending in Object can be among the most heavily
+ instantiated in the model. Renaming a central business term to satisfy a
+ naming rule would trade meaning for symmetry. Abstractness is declared
+ instead, by TN-007.
+ examples:
+ - {from: "pr:DataDomains", to: "pr:DataDomain", note: singular}
+ - {from: "pr:ActorObject", to: "pr:Actor", note: no type word added for its own sake}
+
+- id: TN-006
+ category: Identifier
+ title: Every layer root is abstract
+ statement: >
+ A class that serves as the root of a layer carries pr:isAbstract true.
+ scope: [class]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A non-abstract layer root allows an instance to be typed as an object of
+ that layer without saying which kind of object it is. That statement carries
+ no information and no downstream control can repair it.
+ examples:
+ - {from: "pr:PhysicalLayerObject with no declaration", to: "pr:isAbstract true", note: null}
+
+- id: TN-007
+ category: Identifier
+ title: Every class declares whether it is abstract
+ statement: >
+ Every owl:Class carries pr:isAbstract with an explicit boolean value.
+ Absence of the property is a violation, not a default.
+ scope: [class]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ Where declaration is optional, nothing distinguishes a class deliberately
+ left concrete from one where the question was never asked. Two situations
+ make the compulsion necessary. A class with subclasses may be perfectly
+ instantiable, and has been wrongly marked abstract by inference. And a class
+ used as a classification axis may have no declared subclass at all, its
+ members being typed by multiple typing on instances — its abstractness is
+ then structurally unverifiable and only the explicit declaration carries it.
+ examples:
+ - {from: "a class with no declaration", to: "pr:isAbstract true or false", note: silence is a violation}
+ - {from: "pr:Metric inferred abstract because it has a subclass", to: "pr:isAbstract false", note: concrete despite having a subclass}
+
+- id: TN-008
+ category: Identifier
+ title: Abstractness is declared, never inferred
+ statement: >
+ Abstractness is read from pr:isAbstract and from nothing else. No consumer
+ of the model — viewer, export, script or documentation generator — may
+ derive it from the presence of subclasses, from the absence of instances, or
+ from any other structural signal.
+ scope: [class, consumer]
+ severity: BLOCKING
+ control: {tier: human, executor: consumer contract review, procedure: null}
+ filiation: null
+ rationale: >
+ A declaration rule holds in the source only if it also binds every consumer.
+ Otherwise each one re-invents its own inference and the model acquires as
+ many answers as it has readers. This is the same failure mode that justified
+ a governed short label rather than a display dictionary per viewer.
+ examples:
+ - {from: "isAbstract = term.subclasses.length > 0", to: "isAbstract = term.isAbstract === true", note: consumer contract}
+
+- id: TN-009
+ category: Identifier
+ title: A relation begins with a verb
+ statement: >
+ The local name of an object property begins with a verb, in lower case. Both
+ the active form and the passive or participial form are admitted: a past
+ participle is a verb in first position. No relation begins with a
+ preposition or with a noun.
+ scope: [object_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A relation reads as a verb and an attribute reads as a noun. The distinction
+ is what lets a reader tell an edge from a field without opening the
+ declaration. The explicit clause on participial forms is required: read
+ literally, a verb-first rule would condemn a whole family of sound relations
+ such as computedBy, storedIn and derivedFrom.
+ examples:
+ - {from: "pr:inDatabase", to: "pr:isInDatabase", note: a preposition is not a verb}
+ - {from: "pr:primaryLocation", to: "pr:primarilyStoredIn", note: a noun is not a verb}
+ - {from: "pr:computedBy", to: "pr:computedBy", note: participial form is conforming}
+
+- id: TN-010
+ category: Identifier
+ title: The is prefix is either a copula or a predicate
+ statement: >
+ On an owl:ObjectProperty, is acts as a COPULA — a state verb before a
+ preposition or adjective, whose target is a node of the graph. On a boolean
+ datatype or annotation property, is acts as a PREDICATE — it introduces an
+ adjective or participle and the value is true or false. No other use is
+ admitted.
+ scope: [object_property, datatype_property, annotation_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ Conformance of a name beginning with is is never checkable on the name
+ alone. A control asserting that every such term is a boolean produces false
+ positives on every copula; the reverse control lets real problems through.
+ The correct check reads the declared type of the property first, then
+ applies the matching clause.
+ examples:
+ - {from: "pr:isInSchema", to: copula, note: object property, target is a node}
+ - {from: "pr:isNullable", to: predicate, note: datatype property of range xsd:boolean}
+
+- id: TN-011
+ category: Identifier
+ title: A relation names its target by its short label
+ statement: >
+ A relation need not mention its target class at all. When it does, it uses
+ the target class's pr:shortLabel AS IT STANDS — not the full label, not a
+ truncation of the short label, and not a form invented for the occasion.
+ Where the short label carries an acronym, the acronym is reproduced with it:
+ this is the single exception admitted by TN-002, and TN-001 admits the
+ consecutive capitals it produces.
+ scope: [object_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ Relation names are what is read on the edges of the graph, constantly, in
+ every viewer and every query. Full labels make them unreadable on an arrow.
+ The short label is the form the organisation actually speaks, and reusing it
+ makes a relation name predictable from the class it points at instead of
+ arbitrary. Reproducing it whole is what preserves that property: a
+ truncation would be a form the model invented, unpredictable from the class
+ and unverifiable against it. A corollary: relations do not normally carry a
+ short label of their own, since this rule already keeps their IRIs short.
+ examples:
+ - {from: "pr:hasDomainOwner", to: "pr:hasDomainOwner", note: "target DataDomainOwner, short label Domain Owner — conforming"}
+ - {from: "pr:hasDGL", to: "pr:hasGovernanceLead", note: "target short label is Governance Lead"}
+ - {from: "pr:inBIDataSource", to: "pr:isInBIDataSource", note: "short label BI Data Source reproduced whole, acronym included"}
+ - {from: "pr:containsField", to: "pr:containsBIField", note: "a truncated short label is not admitted"}
+
+- id: TN-012
+ category: Identifier
+ title: An attribute is a noun
+ statement: >
+ The local name of a datatype property is a noun phrase. It begins with no
+ verb, and in particular with no has.
+ scope: [datatype_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A has prefix carries no information — rdfs:domain already says who has the
+ attribute — and it costs the reader the distinction TN-009 exists to make
+ visible. Under TN-018 the derived label would read as a verb phrase on a
+ field, so a relation and an attribute would become indistinguishable in
+ every display.
+ examples:
+ - {from: "pr:hasFormula", to: "pr:formula", note: null}
+ - {from: "pr:hasName", to: "pr:canonicalName", note: name alone is too generic to stand as an identity}
+
+- id: TN-013
+ category: Identifier
+ title: A date attribute is a noun, not a participle
+ statement: >
+ An attribute holding a date or timestamp is named as a noun.
+ scope: [datatype_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ The participial form is idiomatic English but creates a standing family of
+ exceptions to TN-012 that must be defended at every review. The nominal form
+ is consistent with the other attributes of the model and needs no defence.
+ examples:
+ - {from: "pr:createdOn", to: "pr:creationDate", note: null}
+ - {from: "pr:lastQueriedOn", to: "pr:lastQueryDate", note: null}
+
+- id: TN-014
+ category: Identifier
+ title: An attribute never takes a relational suffix
+ statement: >
+ A datatype property never takes a form ending in By, In, From or On. Those
+ forms are reserved for object properties. The only exception is the
+ predicative is of booleans, covered by TN-015.
+ scope: [datatype_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ An attribute named like a relation will be read as one. A reader
+ encountering a term in the shape of computedBy or ownedBy expects a node at
+ the other end, not a string. The tell is usually visible in the label: a
+ parenthetical gloss added to explain what the value actually holds is
+ evidence that the name itself is wrong.
+ examples:
+ - {from: "pr:definedBy holding a string", to: "pr:definitionAddress", note: "address, not location — location is taken by the storage side"}
+
+- id: TN-015
+ category: Identifier
+ title: A boolean attribute begins with is
+ statement: >
+ A property of range xsd:boolean begins with is, followed by an adjective or
+ participle. This is the single explicit exception to TN-012.
+ scope: [datatype_property, annotation_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A bare adjective does not read as a statement, and some of them collide with
+ reserved words in the languages that consume the model.
+ examples:
+ - {from: "pr:nullable", to: "pr:isNullable", note: null}
+ - {from: "pr:abstract", to: "pr:isAbstract", note: "abstract is a reserved word in several target languages"}
+
+- id: TN-016
+ category: Identifier
+ title: A controlled value is an individual or a literal
+ statement: >
+ A value becomes a TYPED INDIVIDUAL if it needs to be defined, owned, or
+ extended by a domain without modifying the model. It stays a LITERAL if it
+ is a closed technical state whose list is the property of governance and
+ must precisely not be extended. The model chooses one form per kind of value
+ and states the choice.
+ scope: [class, individual, datatype_property]
+ severity: BLOCKING
+ control: {tier: [shacl, human], executor: pr_metamodel_shapes.ttl, procedure: null}
+ filiation: A-Box NR-016
+ rationale: >
+ A literal cannot carry a definition, an owner or a link, and extending its
+ list means editing the shapes and therefore the frozen vocabulary. A typed
+ individual can be added by a domain without touching the model, appears as a
+ node in every viewer, and can be reached by following an edge. The test is
+ not how the value looks but who is allowed to add one.
+ examples:
+ - {from: "an activation level defined in a playbook, not in the graph", to: "literal constrained by sh:in", note: not domain-extensible}
+ - {from: "a kind of system a domain may need to add", to: typed individual, note: extensible and worth defining}
+
+- id: TN-017
+ category: Identifier
+ title: Controlled literals are UPPER_SNAKE_CASE
+ statement: >
+ Controlled literal values are written UPPER_SNAKE_CASE. This is the only
+ place in the model where that casing is used.
+ scope: [datatype_property]
+ severity: BLOCKING
+ control: {tier: shacl, executor: pr_metamodel_shapes.ttl, procedure: null}
+ filiation: null
+ rationale: >
+ The casing exists so that a value drawn from a closed list is
+ distinguishable from free text at a glance, in the source and in any export.
+ examples:
+ - {from: "\"Full activation\"", to: "\"FULL_ACTIVATION\"", note: null}
+ - {from: "\"Published\"", to: "\"PUBLISHED\"", note: null}
+
+# --------------------------------------------------------------------- LABEL
+
+- id: TN-018
+ category: Label
+ title: The label is derived from the IRI
+ statement: >
+ rdfs:label is obtained from the local name of the IRI by inserting a space
+ at every case boundary. No word added, removed, reordered or substituted.
+ The words of a class label keep their initial capital; the words of a
+ property label are entirely lower case.
+ scope: [class, object_property, datatype_property, annotation_property, individual]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: A-Box NR-005 (deliberate opposite)
+ rationale: >
+ A label carrying meaning the IRI does not is the symptom of a badly named
+ IRI or of an incomplete comment — never of a legitimate exception to the
+ label. Where a label has been quietly enriched to explain a term, the
+ explanation belongs in rdfs:comment and the short form, if the organisation
+ speaks one, in pr:shortLabel.
+ examples:
+ - {from: "pr:BusinessIntelligenceDataSource", to: "Business Intelligence Data Source", note: a class label keeps its capitals}
+ - {from: "pr:primarilyStoredIn", to: "primarily stored in", note: a property label is lower case}
+ - {from: "\"Ownership and Categorization Layer Object\"", to: "\"Ownership Layer Object\"", note: the added meaning moves to rdfs:comment}
+
+- id: TN-019
+ category: Label
+ title: No acronym in the label
+ statement: >
+ The label introduces no acronym or abbreviation that the IRI does not
+ already carry, including in a recapitalised form that the derivation would
+ not produce. Where the IRI legitimately carries an acronym under TN-011, the
+ derived label carries it too and nothing further is added.
+ scope: [class, object_property, datatype_property, annotation_property, individual]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A direct corollary of TN-018, kept as a separate rule because it catches
+ labels whose IRI is innocent: a term correctly named in full can still carry
+ a hand-written label that reintroduces the short form. Stated as an
+ introduction rather than a prohibition, it stays a pure corollary — whatever
+ the IRI legitimately holds, the derivation reproduces, and nothing else.
+ examples:
+ - {from: "\"BI expression\"", to: "\"expression\"", note: the IRI was already conforming}
+
+- id: TN-020
+ category: Label
+ title: No state in the label
+ statement: >
+ The label mentions no status, version or deprecation. State lives in an
+ axiom — owl:deprecated, pr:status — never in text. Deprecated and active
+ terms are never mixed in a single view: a consumer filters on the axiom at
+ query time and presents deprecated terms separately.
+ scope: [class, object_property, datatype_property, annotation_property, individual, consumer]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A term whose label repeats its state carries that state twice. The day one
+ is changed without the other, the two contradict each other and nothing
+ detects it. With the rule in force, the question of whether a term is
+ deprecated has exactly one possible answer.
+ examples:
+ - {from: "\"belongs to domain (deprecated)\"", to: "\"belongs to domain\"", note: owl:deprecated carries the state}
+
+- id: TN-021
+ category: Label
+ title: No gloss or parenthesis in the label
+ statement: >
+ No parenthesis, qualification or example appears in the label. They belong
+ in rdfs:comment.
+ scope: [class, object_property, datatype_property, annotation_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A parenthetical gloss is almost always the symptom of an IRI that does not
+ say what it holds. The right response is usually to rename the term, not to
+ annotate the label.
+ examples:
+ - {from: "\"defined by (code URI)\"", to: "\"definition address\"", note: rename the term rather than gloss the label}
+ - {from: "\"query count (30 days)\"", to: "\"query count\"", note: null}
+
+- id: TN-022
+ category: Label
+ title: The short label carries the spoken form
+ statement: >
+ A term carries pr:shortLabel whenever the form derived under TN-018 is not
+ what the organisation writes or says. The short label is free: acronyms,
+ abbreviations, hyphens, short forms.
+ scope: [class, individual]
+ severity: MAJOR
+ control: {tier: human, executor: review checklist, procedure: null}
+ filiation: null
+ rationale: >
+ The short label is what makes the derivation rule bearable. Without it, a
+ mechanical label impoverishes every screen; with it, display becomes
+ explicit and centrally governed instead of being improvised by each
+ consumer. It is MAJOR rather than BLOCKING because judging whether a derived
+ form matches what people actually say is human work, and EV-011 forbids
+ declaring a rule blocking with no executor.
+ examples:
+ - {from: "\"Data Sub Domain Owner\"", to: "short label Sub Domain Owner", note: the word Data is dropped in the spoken form}
+ - {from: "\"Key Performance Indicator\"", to: "short label KPI, acronym KPI", note: the acronym channel is what makes TN-002 bearable}
+
+- id: TN-023
+ category: Label
+ title: The label is generated, never typed
+ statement: >
+ rdfs:label is never written by hand. It is produced by the derivation
+ function from the IRI and stored in the source. A hand-typed label is a
+ violation even when its value happens to be correct.
+ scope: [class, object_property, datatype_property, annotation_property, individual]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ Generate and store, rather than store only or compute at read time. The
+ label stays in the source so SPARQL and third-party tooling behave normally,
+ while a script regenerates it at every commit and the gate fails if a stored
+ label differs from the derived one. Redundancy becomes harmless because it
+ is checked: rather than hoping two facts stay consistent, the model makes
+ inconsistency detectable.
+ examples:
+ - {from: label absent, to: violation, note: null}
+ - {from: label differing from the derivation, to: "violation — drift", note: null}
+
+# --------------------------------------------------------------- DECLARATION
+
+- id: TN-024
+ category: Declaration
+ title: Every active term carries a comment
+ statement: >
+ rdfs:comment is mandatory on every active term. On a property the comment is
+ structured: the definition, plus the scope constraint or the confusion to be
+ avoided. On a class the form is free.
+ scope: [class, object_property, datatype_property, annotation_property]
+ severity: BLOCKING
+ control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A comment that paraphrases the name prevents nothing. A good one prevents a
+ specific error: it says where the property may and may not be declared, or
+ which neighbouring term it must not be confused with. The structure is
+ imposed on properties because they are where ambiguity costs most — they are
+ asserted thousands of times by people who will not read the model. The
+ script checks presence; only a reviewer checks value, which is why the rule
+ names two tiers.
+ examples:
+ - {from: "\"the owner\"", to: "definition plus where it may be declared and what it excludes", note: null}
+
+- id: TN-025
+ category: Declaration
+ title: Every property declares its domain and range
+ statement: >
+ rdfs:domain and rdfs:range are mandatory on every property, EXCEPT where the
+ property is deliberately polymorphic. A polymorphic property states so in
+ its comment and has its scope declared in SHACL. Silent absence is a
+ violation; documented absence is not.
+ scope: [object_property, datatype_property]
+ severity: BLOCKING
+ control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A property with no domain cannot be targeted by any control. But some
+ properties are polymorphic by design, their scope controlled class by class
+ in SHACL rather than by twin properties; giving those an rdfs:domain would
+ trigger the RDFS retyping described in TN-028. The rule therefore separates
+ the two cases rather than demanding a domain everywhere.
+ examples:
+ - {from: a polymorphic property with no domain and no comment, to: violation, note: silence is indistinguishable from omission}
+ - {from: a polymorphic property with no domain, documented, to: conforming, note: scope declared in SHACL}
+
+- id: TN-026
+ category: Declaration
+ title: Every term declares how it was authored
+ statement: >
+ pr:authoringMode is mandatory on every term. pr:harvestSource is mandatory
+ if and only if the mode is HARVESTED.
+ scope: [class, object_property, datatype_property, annotation_property]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ Provenance decides who may edit a term and what a divergence means. A term
+ stating where its data comes from without stating that it is harvested, or
+ declaring itself harvested without naming a source, is half-declared in a
+ way no control can catch. Declared symmetrically, provenance also makes a
+ harvester specifiable from the model itself rather than from a side
+ document.
+ examples:
+ - {from: harvestSource present, authoringMode absent, to: "authoringMode HARVESTED", note: null}
+ - {from: "authoringMode HARVESTED, harvestSource absent", to: harvestSource declared, note: null}
+
+- id: TN-027
+ category: Declaration
+ title: A concrete class has one layer and one provenance
+ statement: >
+ Every concrete class descends from exactly one layer root and from exactly
+ one provenance axis. Controlled-vocabulary classes are outside the layers by
+ nature — a named and closed exception.
+ scope: [class]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A class with no layer is invisible to every layer-scoped control and to
+ every view organised by layer; a class with two is ambiguous in both. Where
+ a family of classes genuinely sits outside the data layers — actors,
+ governance objects — the answer is an explicit root of its own, not silence.
+ Adding such a root by INSERTION above an existing abstraction, rather than
+ by renaming it, keeps the ranges that point at that abstraction readable: a
+ range must state the nature of its target, not its position in the model.
+ examples:
+ - {from: a class with no parent, to: attached to its layer root, note: null}
+ - {from: renaming an abstraction into a layer root, to: inserting the layer root above it, note: preserves the meaning of every range that points at it}
+
+- id: TN-028
+ category: Declaration
+ title: A sub-property inherits the domain of its parent
+ statement: >
+ Declaring rdfs:subPropertyOf does not shield a property from the
+ rdfs:domain of its parent. Every assertion of a sub-property is also an
+ assertion of the parent, so the parent's domain applies and its RDFS
+ retyping takes effect.
+ scope: [object_property, datatype_property]
+ severity: BLOCKING
+ control: {tier: [script, human], executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ In RDFS a domain is not a rejecting constraint but a retyping inference.
+ Attaching a sub-property to a parent whose domain does not fit will silently
+ retype its subjects — an object of one layer quietly becomes an object of
+ another, entering counts, controls and traversals it was never part of.
+ Where two relations share a verb but not a domain, they must not be merged
+ and must not be linked by subPropertyOf; the unification, if one is needed,
+ belongs on a shared upper property with no domain of its own.
+ examples:
+ - {from: "a dashboard field asserted through a property whose domain is a warehouse column", to: "the dashboard field is inferred to BE a warehouse column", note: silent retyping}
+ - {from: "linking the two by rdfs:subPropertyOf", to: "no protection — the parent's domain still applies", note: null}
+ - {from: "a shared upper property with no domain", to: "end-to-end traversal without merging IRIs", note: the correct unification}
+
+# ----------------------------------------------------------------- EVOLUTION
+
+- id: EV-001
+ category: Evolution
+ title: A rename is never an edit in place
+ statement: >
+ A rename creates a new term and keeps the old one with owl:deprecated true
+ and dcterms:isReplacedBy. No IRI is modified in place. The deprecated stub
+ carries no edge: no subClassOf, no label, no comment — only its type and the
+ two axioms.
+ scope: [procedure]
+ severity: BLOCKING
+ control: {tier: human, executor: commit review, procedure: null}
+ filiation: A-Box LC-002
+ rationale: >
+ Nothing propagates in RDF. An IRI is a string copied into every triple, so
+ renaming means rewriting every triple where it appears as subject, predicate
+ or object — declaration, subclass axioms, domain, range, instance typing,
+ shape targets and paths, queries, viewers, documentation. Edges are moved,
+ not duplicated. The stub exists so that a consumer holding an older export
+ can still be told what replaced the term.
+ examples:
+ - {from: editing an IRI in place, to: new term plus deprecated stub, note: null}
+
+- id: EV-002
+ category: Evolution
+ title: A withdrawal requires proof of non-instantiation
+ statement: >
+ A term is removed only after a count proving it was never instantiated, the
+ count being attached to the commit. Without proof, deprecation only.
+ scope: [procedure]
+ severity: BLOCKING
+ control: {tier: script, executor: count query attached to the commit, procedure: null}
+ filiation: null
+ rationale: >
+ Removal is the one irreversible operation on a vocabulary. The proof is
+ cheap — a single counting query — and it is what separates a safe withdrawal
+ from a silent data loss.
+ examples:
+ - {from: deleting a term, to: "SELECT (COUNT(?s) AS ?n) WHERE { ?s a pr:Term }", note: must return zero}
+
+- id: EV-003
+ category: Evolution
+ title: A deletion removes every reference to the subject
+ statement: >
+ A subject removed while its IRI is still cited elsewhere is recreated by
+ RDFS inference. Every reference is removed in the same operation.
+ scope: [procedure]
+ severity: BLOCKING
+ control: {tier: script, executor: dangling-reference check in the migration script, procedure: null}
+ filiation: null
+ rationale: >
+ A graph has no foreign keys. Deleting the block that declares a subject
+ leaves every triple that names it intact, and a reasoner will reconstruct a
+ hollow node from them — present in traversals, absent from every control.
+ examples:
+ - {from: deleting a subject block, to: deleting the block and every triple naming it, note: null}
+
+- id: EV-004
+ category: Evolution
+ title: A change is validated before it is applied
+ statement: >
+ Every modification of the model is agreed before any artifact is written.
+ Dry run first, apply second.
+ scope: [procedure]
+ severity: BLOCKING
+ control: {tier: human, executor: commit review, procedure: null}
+ filiation: null
+ rationale: >
+ The damage from a bad change is rarely in the change itself but in
+ everything that silently depended on the old state. A dry run that reports
+ what it would touch is what makes that dependency visible while it can still
+ be discussed.
+ examples:
+ - {from: "migrate.py --apply as the first run", to: "migrate.py, then --apply", note: dry run is the default}
+
+- id: EV-005
+ category: Evolution
+ title: Graph edits go through an RDF parser
+ statement: >
+ Any transformation of the graph is performed with an RDF parser. No regular
+ expression applied line by line to Turtle.
+ scope: [procedure, script]
+ severity: BLOCKING
+ control: {tier: human, executor: script review, procedure: null}
+ filiation: null
+ rationale: >
+ Turtle is a structured syntax and a line is not a unit of meaning. A literal
+ may contain the very characters a regex splits on, so a text-level edit that
+ looks correct on every example can truncate a value on the one that matters.
+ A parser knows the difference between a delimiter and a character inside a
+ string; a regex does not, and every workaround for that is a partial
+ reimplementation of the parser.
+ examples:
+ - {from: "re.sub on the file text", to: remove and add on parsed triples, note: null}
+ - {from: hand-written clause splitting, to: parser, note: a workaround is evidence the wrong tool is in use}
+
+- id: EV-006
+ category: Evolution
+ title: Every script is idempotent
+ statement: >
+ A script may be replayed without changing the result. Its guard tests the
+ TARGET state, not the source state, and it counts its own result. Acceptance
+ test: two consecutive runs, the second reporting zero modifications.
+ scope: [script]
+ severity: BLOCKING
+ control: {tier: script, executor: double run in continuous integration, procedure: null}
+ filiation: null
+ rationale: >
+ A chain of transformations is only safe if any step can be replayed. A guard
+ that tests whether work remains to be done breaks as soon as another script
+ has already changed the source; a guard that tests whether the target state
+ exists survives it. The self-count is what turns the property into a
+ verifiable one.
+ examples:
+ - {from: appending unconditionally, to: appending only if the target is absent, note: guard on the target state}
+
+- id: EV-007
+ category: Evolution
+ title: The replayable source is the start of the chain
+ statement: >
+ The replayable source is the starting file plus the ordered sequence of
+ scripts, never an intermediate committed state. Hand-editing a file produced
+ by a script is forbidden, including for a typo: the script is corrected and
+ the chain replayed.
+ scope: [procedure]
+ severity: BLOCKING
+ control: {tier: human, executor: commit review, procedure: null}
+ filiation: null
+ rationale: >
+ A committed intermediate file is a convenience, not a source. As soon as one
+ hand edit exists in it that no script performs, replaying the chain no
+ longer reproduces the committed state and nobody can say which of the two is
+ right. Keeping the chain authoritative is also what makes a failed migration
+ recoverable: restart from the last true source with a corrected script,
+ rather than repair a half-migrated file by hand.
+ examples:
+ - {from: fixing a typo in a generated file, to: fixing the script and replaying, note: null}
+
+- id: EV-008
+ category: Evolution
+ title: Every artifact has a producer script
+ statement: >
+ No artifact contains hard-coded data. Every deliverable — viewer,
+ documentation, workbook, diagram — is regenerated from a versioned source by
+ a script.
+ scope: [script, artifact]
+ severity: BLOCKING
+ control: {tier: human, executor: artifact review, procedure: null}
+ filiation: null
+ rationale: >
+ An artifact that cannot be regenerated becomes stale the moment the model
+ moves, and there is no way to tell whether it is stale or merely different.
+ This rulebook applies the rule to itself: the source is one file, the
+ document and the workbook are generated from it.
+ examples:
+ - {from: a hand-maintained workbook, to: a source file plus a generator, note: null}
+
+- id: EV-009
+ category: Evolution
+ title: A change propagates in a single merge
+ statement: >
+ A change is propagated to the vocabulary, the instances, the shapes, the
+ consumers and the documentation in one branch, merged as a block. The main
+ line never sees a state where one has moved and another has not.
+ scope: [procedure]
+ severity: BLOCKING
+ control: {tier: human, executor: merge review, procedure: null}
+ filiation: null
+ rationale: >
+ Partial propagation is not a smaller version of a change, it is a different
+ and invalid state. Its worst form is silent: a gate that validates the
+ stale half and reports success. A single merge makes the intermediate state
+ unreachable rather than merely discouraged.
+ examples:
+ - {from: committing the vocabulary and updating the shapes later, to: one branch, one merge, one tag, note: null}
+
+- id: EV-010
+ category: Evolution
+ title: Shapes declare the ontology version they target
+ statement: >
+ The shapes graph declares dcterms:conformsTo equal to the owl:versionIRI of
+ the ontology. Any divergence aborts validation immediately.
+ scope: [shape, script]
+ severity: BLOCKING
+ control: {tier: script, executor: run_shacl_validation.py, procedure: null}
+ filiation: null
+ rationale: >
+ Validating against the wrong shapes does not fail loudly. It produces a long
+ list of violations that reads exactly like a regression of the model, and
+ the time goes into interpreting them rather than noticing the mismatch. One
+ declared triple and one check turn that into a single line naming the real
+ problem.
+ examples:
+ - {from: hundreds of violations to interpret, to: "ABORT — shapes target a different ontology version", note: null}
+
+- id: EV-011
+ category: Evolution
+ title: No blocking rule without an executor
+ statement: >
+ A rule may be declared BLOCKING only if a named script or shape enforces it.
+ A blocking rule with no control is a statement of intent.
+ scope: [rule]
+ severity: BLOCKING
+ control: {tier: script, executor: rulebook source self-check, procedure: null}
+ filiation: null
+ rationale: >
+ An unenforced blocking rule is worse than an absent one: it is cited as
+ though it held, and the gap only surfaces when something it should have
+ caught reaches production. Making the executor a mandatory field turns the
+ omission into something a script can detect on the rulebook itself.
+ examples:
+ - {from: "severity BLOCKING with no executor", to: violation, note: detected by the generators before they write anything}
+
+- id: EV-012
+ category: Evolution
+ title: Every rule names the tier that enforces it
+ statement: >
+ Controls exist at three tiers — SHACL for the shape of the graph, scripts
+ for the file, the lexicon and naming, human review for semantics. Every rule
+ names its tier. A rule with no assigned tier is a violation.
+ scope: [rule]
+ severity: BLOCKING
+ control: {tier: script, executor: rulebook source self-check, procedure: null}
+ filiation: null
+ rationale: >
+ SHACL validates a graph, not a file, and not meaning. Two IRIs standing for
+ the same notion produce two individually valid graphs; a duplicated block of
+ text produces identical triples and no complaint. Naming the tier forces the
+ question of what actually catches each rule, and pushes the ones SHACL
+ cannot see to a control that can: a normalised-label index reviewed by a
+ person, a file-level hash, a naming script, a review checklist.
+ examples:
+ - {from: two IRIs for one notion, to: normalised label index plus human review, note: SHACL cannot see it}
+ - {from: a duplicated text block, to: file-level block hashing, note: identical triples, no violation}
+
+- id: EV-013
+ category: Evolution
+ title: No sh:pattern outside the SHACL specification
+ statement: >
+ sh:pattern uses XPath 2.0 regular expressions, which support neither
+ lookbehind nor lookahead. A constraint is expressed positively — what the
+ string must be — rather than negatively. Where that is impossible the rule
+ leaves SHACL and moves to the script tier under EV-012.
+ scope: [shape]
+ severity: BLOCKING
+ control: {tier: script, executor: run_shacl_validation.py, procedure: null}
+ filiation: A-Box NR-001
+ rationale: >
+ A pattern written in a richer regex dialect does not degrade gracefully: the
+ validator fails to compile it, and the rule that depended on it silently
+ stops being enforced while still being declared blocking.
+ examples:
+ - {from: "a pattern using (?
+ The namespace of terms is not versioned. The version lives in
+ owl:versionIRI alone.
+ scope: [ontology]
+ severity: BLOCKING
+ control: {tier: script, executor: check_tbox_naming.py, procedure: null}
+ filiation: null
+ rationale: >
+ A versioned term namespace changes every IRI at every release, breaking
+ every consumer for no benefit. It also fails silently in the other
+ direction: a file bound to an outdated namespace targets IRIs that no longer
+ exist, so every shape matches nothing and validation passes while testing
+ nothing at all.
+ examples:
+ - {from: "a prefix bound to a versioned namespace", to: "a prefix bound to the stable namespace", note: the version lives in owl:versionIRI}
+
+- id: EV-015
+ category: Evolution
+ title: Every procedure that writes produces an execution log
+ statement: >
+ Any procedure that modifies the graph writes a versioned execution log,
+ committed alongside the change. The log records the attempt number and
+ timestamp, a checksum of every input consumed, the ontology version and the
+ version the shapes declare they target, a count per step, the full
+ validation report, and a difference against the previous attempt.
+ scope: [procedure, script]
+ severity: BLOCKING
+ control: {tier: script, executor: the procedure's own runner, procedure: null}
+ filiation: null
+ rationale: >
+ A counter printed to the terminal disappears with the terminal. When a
+ validation run fails, the question is never only what failed but whether it
+ is the same failure as the previous attempt: a violation that persists
+ unchanged after a correction says the cause lies elsewhere, while one that
+ moves every time says the change is being patched rather than fixed. Those
+ are opposite diagnoses and neither is available without a record of the
+ previous run. The input checksums serve the same purpose one level down,
+ separating a wrong model from a stale file — the ambiguity that made a past
+ incident expensive to explain. The log is also the natural carrier of the
+ proof of non-instantiation required by EV-002, which otherwise has no
+ defined format.
+ examples:
+ - {from: a counter printed to standard output, to: a versioned log committed with the change, note: null}
+ - {from: "a failed run with no record of the previous one", to: "a difference against the previous attempt", note: what makes correct-or-abandon decidable}
+
+# ====== ANNEX: ABSTRACTNESS (TN-006, TN-007) ======
+abstractness:
+- {term: MetaModelObject, is_abstract: true, note: "root of the model"}
+- {term: DefinedObject, is_abstract: true, note: "provenance axis"}
+- {term: CapturedObject, is_abstract: true, note: "provenance axis"}
+- {term: OwnershipLayerObject, is_abstract: true, note: "layer root"}
+- {term: BusinessLayerObject, is_abstract: true, note: "layer root"}
+- {term: LogicalLayerObject, is_abstract: true, note: "layer root"}
+- {term: PhysicalLayerObject, is_abstract: true, note: "NEW — omission in v1.6"}
+- {term: DeliveryLayerObject, is_abstract: true, note: "layer root"}
+- {term: ConsumptionLayerObject, is_abstract: true, note: "layer root"}
+- {term: GovernanceLayerObject, is_abstract: true, note: "NEW — seventh layer root"}
+- {term: DataStructure, is_abstract: true, note: "harvesting types the concrete sort"}
+- {term: KeyConstraint, is_abstract: true, note: "only PrimaryKey and ForeignKey instantiate"}
+- {term: Actor, is_abstract: true, note: "renaming would degrade the range of ownedBy and hasPublisher"}
+- {term: DataDomain, is_abstract: false, note: null}
+- {term: DataSubDomain, is_abstract: false, note: null}
+- {term: BusinessObject, is_abstract: false, note: "ends in Object and is heavily instantiated - refutes any suffix rule"}
+- {term: BusinessConcept, is_abstract: false, note: null}
+- {term: Metric, is_abstract: false, note: "concrete despite having KeyPerformanceIndicator beneath it"}
+- {term: KeyPerformanceIndicator, is_abstract: false, note: null}
+- {term: DataObject, is_abstract: false, note: null}
+- {term: DataElement, is_abstract: false, note: null}
+- {term: Database, is_abstract: false, note: null}
+- {term: Schema, is_abstract: false, note: null}
+- {term: Field, is_abstract: false, note: null}
+- {term: System, is_abstract: false, note: null}
+- {term: Transformation, is_abstract: false, note: null}
+- {term: BaseTable, is_abstract: false, note: null}
+- {term: ExternalTable, is_abstract: false, note: null}
+- {term: View, is_abstract: false, note: null}
+- {term: PrimaryKey, is_abstract: false, note: null}
+- {term: ForeignKey, is_abstract: false, note: null}
+- {term: DataProduct, is_abstract: false, note: null}
+- {term: DataContract, is_abstract: false, note: null}
+- {term: DataInterface, is_abstract: false, note: null}
+- {term: BusinessIntelligenceWorkspace, is_abstract: false, note: null}
+- {term: BusinessIntelligenceDataSource, is_abstract: false, note: null}
+- {term: BusinessIntelligenceField, is_abstract: false, note: null}
+- {term: BusinessIntelligenceReport, is_abstract: false, note: null}
+- {term: DataDomainOwner, is_abstract: false, note: null}
+- {term: DataSubDomainOwner, is_abstract: false, note: null}
+- {term: DataProductOwner, is_abstract: false, note: null}
+- {term: DataSteward, is_abstract: false, note: null}
+- {term: DataGovernanceLead, is_abstract: false, note: null}
+- {term: SystemType, is_abstract: false, note: "must stay concrete — it types five individuals"}
+
+# ====== ANNEX: DISPLAY (TN-011, TN-022) ======
+display:
+- {iri: DataDomain, label: "Data Domain", short_label: "Domain", acronym: "DD", note: "DD is already the short form used inside A-Box identifiers under NR-014"}
+- {iri: DataSubDomain, label: "Data Sub Domain", short_label: "Sub Domain", acronym: "SD", note: "no hyphen"}
+- {iri: DataDomainOwner, label: "Data Domain Owner", short_label: "Domain Owner", acronym: "DDO", note: null}
+- {iri: DataSubDomainOwner, label: "Data Sub Domain Owner", short_label: "Sub Domain Owner", acronym: "SDO", note: null}
+- {iri: DataProductOwner, label: "Data Product Owner", short_label: "Product Owner", acronym: "PO", note: "NOT DPO — the initialism is taken by Data Protection Officer"}
+- {iri: DataSteward, label: "Data Steward", short_label: "Steward", acronym: null, note: null}
+- {iri: DataGovernanceLead, label: "Data Governance Lead", short_label: "Governance Lead", acronym: "DGL", note: null}
+- {iri: DataElement, label: "Data Element", short_label: "Element", acronym: null, note: "drives the name of hasElement and hasGrainElement"}
+- {iri: BusinessConcept, label: "Business Concept", short_label: "Concept", acronym: null, note: "drives the name of usesConcept"}
+- {iri: BusinessObject, label: "Business Object", short_label: null, acronym: "BO", note: null}
+- {iri: KeyPerformanceIndicator, label: "Key Performance Indicator", short_label: "KPI", acronym: "KPI", note: "the acronym channel is what makes TN-002 bearable"}
+- {iri: BusinessIntelligenceWorkspace, label: "Business Intelligence Workspace", short_label: "BI Workspace", acronym: null, note: null}
+- {iri: BusinessIntelligenceDataSource, label: "Business Intelligence Data Source", short_label: "BI Data Source", acronym: null, note: null}
+- {iri: BusinessIntelligenceField, label: "Business Intelligence Field", short_label: "BI Field", acronym: null, note: null}
+- {iri: BusinessIntelligenceReport, label: "Business Intelligence Report", short_label: "BI Report", acronym: null, note: null}
+- {iri: OwnershipLayerObject, label: "Ownership Layer Object", short_label: null, acronym: null, note: "the categorisation role moves into rdfs:comment"}