tbox: v2.2 - polymorphisme declare sur 19 proprietes, 17 domaines poses, provenance portee par les classes concretes, GovernanceLayerObject rattachee, affectations de gouvernance completes (12 completees, 10 SDO, 1 PO, 16 stewards par objet metier), Panel Coverage gouverne. Validation SHACL sans violation bloquante
This commit is contained in:
@@ -0,0 +1,488 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Enforce the naming and declaration rules of the T-Box rulebook.
|
||||
|
||||
USAGE
|
||||
python3 check_tbox_naming.py [--ontology PATH] [--json PATH] [--quiet]
|
||||
|
||||
Named as the executor of 25 rules. EV-011 forbids declaring a rule blocking
|
||||
without one, so this script is what makes those declarations true.
|
||||
|
||||
THREE VERDICTS
|
||||
VIOLATION the rule is settled mechanically and is broken -> exit 1
|
||||
REVIEW the checker cannot conclude; a person must -> exit 0
|
||||
PASS conforming
|
||||
|
||||
The middle verdict is the point. Without it the choice would be between
|
||||
blocking on false positives and staying silent on real problems. What the
|
||||
checker cannot decide alone comes from naming_lexicon.yaml; a word absent from
|
||||
the lexicon produces a REVIEW, never a VIOLATION.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections import OrderedDict
|
||||
|
||||
import yaml
|
||||
from rdflib import Graph, Literal, Namespace, RDF, RDFS, OWL, URIRef, XSD
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
RULES = os.path.join(HERE, "rules.yaml")
|
||||
LEXICON = os.path.join(HERE, "naming_lexicon.yaml")
|
||||
DEFAULT_ONTOLOGY = os.path.normpath(os.path.join(HERE, "..", "ontology", "pr_metamodel.ttl"))
|
||||
|
||||
NS = "https://ontology.pernod-ricard.com/metamodel/"
|
||||
PR = Namespace(NS)
|
||||
|
||||
VIOLATION, REVIEW = "VIOLATION", "REVIEW"
|
||||
|
||||
KINDS = OrderedDict([
|
||||
(OWL.Class, "class"),
|
||||
(OWL.ObjectProperty, "relation"),
|
||||
(OWL.DatatypeProperty, "attribute"),
|
||||
(OWL.AnnotationProperty, "annotation"),
|
||||
])
|
||||
|
||||
|
||||
# ------------------------------------------------------------------- findings
|
||||
|
||||
class Findings(object):
|
||||
def __init__(self):
|
||||
self.items = []
|
||||
|
||||
def add(self, verdict, rule, term, message):
|
||||
self.items.append({"verdict": verdict, "rule": rule,
|
||||
"term": term, "message": message})
|
||||
|
||||
def violation(self, rule, term, message):
|
||||
self.add(VIOLATION, rule, term, message)
|
||||
|
||||
def review(self, rule, term, message):
|
||||
self.add(REVIEW, rule, term, message)
|
||||
|
||||
@property
|
||||
def violations(self):
|
||||
return [i for i in self.items if i["verdict"] == VIOLATION]
|
||||
|
||||
@property
|
||||
def reviews(self):
|
||||
return [i for i in self.items if i["verdict"] == REVIEW]
|
||||
|
||||
|
||||
# -------------------------------------------------------------------- helpers
|
||||
|
||||
def local(uri):
|
||||
s = str(uri)
|
||||
return s[len(NS):] if s.startswith(NS) else None
|
||||
|
||||
|
||||
def tokens(name):
|
||||
"""Split a CamelCase local name. A run of capitals is one token: an acronym."""
|
||||
return re.findall(r"[A-Z]+(?![a-z])|[A-Z][a-z0-9]*|^[a-z0-9]+|[a-z0-9]+", name)
|
||||
|
||||
|
||||
def decamelise(name, is_class):
|
||||
"""TN-018. A run of capitals is kept together: it carries an acronym."""
|
||||
spaced = re.sub(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])", " ", name)
|
||||
return spaced if is_class else spaced[0].lower() + spaced[1:]
|
||||
|
||||
|
||||
def acronyms_in(name, lex):
|
||||
"""Runs of two or more capitals, plus any known short form appearing whole.
|
||||
|
||||
The negative lookahead matters: without it, BIField yields BIF rather than
|
||||
BI, because the run swallows the initial capital of the word that follows.
|
||||
"""
|
||||
found = set(re.findall(r"[A-Z]{2,}(?![a-z])", name))
|
||||
for short in lex["known_acronyms"]:
|
||||
if re.search(r"(?:^|(?<=[a-z]))%s(?![a-z])" % re.escape(short), name):
|
||||
found.add(short)
|
||||
return sorted(found)
|
||||
|
||||
|
||||
def strip_case(text):
|
||||
return re.sub(r"[^a-z0-9]", "", (text or "").lower())
|
||||
|
||||
|
||||
# --------------------------------------------------------------------- checks
|
||||
|
||||
def provenance_axis(g, term):
|
||||
"""The provenance axis a class descends from, or None."""
|
||||
seen, stack = set(), [term]
|
||||
while stack:
|
||||
node = stack.pop()
|
||||
for parent in g.objects(node, RDFS.subClassOf):
|
||||
if isinstance(parent, URIRef) and parent not in seen:
|
||||
seen.add(parent)
|
||||
stack.append(parent)
|
||||
if PR.CapturedObject in seen:
|
||||
return "CapturedObject"
|
||||
if PR.DefinedObject in seen:
|
||||
return "DefinedObject"
|
||||
return None
|
||||
|
||||
|
||||
def check_identifier(g, terms, lex, f):
|
||||
"""TN-001 to TN-015 — the form of the IRI."""
|
||||
short_labels = {t: str(v) for t in terms
|
||||
for v in g.objects(t, PR.shortLabel)}
|
||||
allowed_acronyms = set()
|
||||
for value in short_labels.values():
|
||||
allowed_acronyms |= set(re.findall(r"[A-Z]{2,}", value))
|
||||
|
||||
for term, kind in terms.items():
|
||||
name = local(term)
|
||||
if name is None:
|
||||
continue
|
||||
|
||||
# TN-001 — CamelCase, no separator, correct initial
|
||||
if re.search(r"[^A-Za-z0-9]", name):
|
||||
f.violation("TN-001", name, "the local name carries a separator")
|
||||
expect_upper = kind in ("class", "individual")
|
||||
if name and name[0].isupper() != expect_upper:
|
||||
f.violation("TN-001", name,
|
||||
"a %s starts with %s case" %
|
||||
(kind, "upper" if expect_upper else "lower"))
|
||||
|
||||
# TN-004 — no digit
|
||||
if any(c.isdigit() for c in name):
|
||||
f.violation("TN-004", name, "a digit belongs to a value, not an identity")
|
||||
|
||||
# TN-002 — no acronym, except one reproduced from a target short label
|
||||
for a in acronyms_in(name, lex):
|
||||
if kind == "relation" and a in allowed_acronyms:
|
||||
continue # TN-011 exception
|
||||
expansion = lex["known_acronyms"].get(a)
|
||||
f.violation("TN-002", name, "carries the acronym %s%s"
|
||||
% (a, " (%s)" % expansion if expansion else ""))
|
||||
|
||||
if kind == "class":
|
||||
check_class_name(name, lex, f)
|
||||
elif kind == "relation":
|
||||
check_relation_name(g, term, name, lex, f)
|
||||
elif kind == "attribute":
|
||||
check_attribute_name(g, term, name, lex, f)
|
||||
|
||||
|
||||
def check_class_name(name, lex, f):
|
||||
"""TN-005 — a singular common noun, no type word."""
|
||||
if name in lex["type_word_exceptions"]:
|
||||
return
|
||||
for w in lex["type_words"]:
|
||||
if re.search(r"(?:^|(?<=[a-z]))%s(?![a-z])" % w, name):
|
||||
f.violation("TN-005", name,
|
||||
"carries the type word %s and is not a recorded abstraction" % w)
|
||||
last = tokens(name)[-1] if tokens(name) else name
|
||||
if last.endswith("s") and last not in lex["not_plural"]:
|
||||
f.review("TN-005", name, "%s looks plural; a class name is singular" % last)
|
||||
|
||||
|
||||
def check_relation_name(g, term, name, lex, f):
|
||||
"""TN-009, TN-010, TN-011 — verb first, is disambiguated, target named whole."""
|
||||
toks = tokens(name)
|
||||
head = toks[0] if toks else name
|
||||
|
||||
# TN-009 — an adverb is transparent, the verb follows
|
||||
if head in lex["adverbs"]:
|
||||
toks = toks[1:]
|
||||
head = toks[0].lower() if toks else head
|
||||
head = head.lower()
|
||||
|
||||
if head in lex["copula"]:
|
||||
pass # TN-010, copula use
|
||||
elif head in lex["active_verbs"] or head in lex["participles"]:
|
||||
pass
|
||||
else:
|
||||
f.review("TN-009", name,
|
||||
"%s is not a verb known to the lexicon; add it there or rename" % head)
|
||||
|
||||
# TN-011 — if the target class is named, it is named by its short label whole
|
||||
for rng in g.objects(term, RDFS.range):
|
||||
target = local(rng)
|
||||
if not target:
|
||||
continue
|
||||
short = next((str(v) for v in g.objects(rng, PR.shortLabel)), None)
|
||||
reference = strip_case(short) if short else strip_case(target)
|
||||
tail = strip_case("".join(toks[1:]))
|
||||
if not tail or tail == reference:
|
||||
continue
|
||||
if reference.endswith(tail) or tail in reference:
|
||||
f.violation("TN-011", name,
|
||||
"names its target by a truncation; %s expects %s"
|
||||
% (target, short or target))
|
||||
|
||||
|
||||
def check_attribute_name(g, term, name, lex, f):
|
||||
"""TN-012 to TN-015 — a noun, no relational suffix, booleans predicative."""
|
||||
ranges = [str(r) for r in g.objects(term, RDFS.range)]
|
||||
is_boolean = str(XSD.boolean) in ranges
|
||||
is_date = any(r in lex["date_ranges"] for r in ranges)
|
||||
toks = tokens(name)
|
||||
head = (toks[0] if toks else name).lower()
|
||||
|
||||
if is_boolean:
|
||||
if head not in lex["copula"]:
|
||||
f.violation("TN-015", name, "a boolean attribute is predicative and begins with is")
|
||||
return
|
||||
|
||||
# TN-012 — a noun, and in particular no has
|
||||
if head == "has":
|
||||
f.violation("TN-012", name, "an attribute is a noun; the domain already says who has it")
|
||||
elif head in lex["active_verbs"] or head in lex["copula"]:
|
||||
f.violation("TN-012", name, "%s is a verb; an attribute is a noun" % head)
|
||||
elif head in lex["participles"]:
|
||||
f.review("TN-012", name, "%s is a participle; confirm this reads as a noun" % head)
|
||||
|
||||
# TN-014 — relational suffixes are reserved for relations
|
||||
last = toks[-1] if toks else name
|
||||
if last in lex["relational_suffixes"]:
|
||||
f.violation("TN-014", name,
|
||||
"ends in %s, a form reserved for object properties" % last)
|
||||
|
||||
# TN-013 — a date attribute is nominal
|
||||
if is_date and (last in lex["relational_suffixes"] or head in lex["participles"]):
|
||||
f.violation("TN-013", name, "a date attribute is a noun, not a participle")
|
||||
|
||||
|
||||
def check_declaration(g, terms, lex, f, abstractness):
|
||||
"""TN-003, TN-006, TN-007, TN-024, TN-025, TN-026, TN-027, TN-028."""
|
||||
layer_roots = {t for t in terms
|
||||
if terms[t] == "class"
|
||||
and (local(t) or "").endswith("LayerObject")}
|
||||
|
||||
for term, kind in terms.items():
|
||||
name = local(term)
|
||||
if name is None:
|
||||
continue
|
||||
|
||||
# TN-024 — every active term carries a comment
|
||||
if not list(g.objects(term, RDFS.comment)):
|
||||
f.violation("TN-024", name, "carries no rdfs:comment")
|
||||
|
||||
# TN-026 — only a concrete class declares how its instances are produced
|
||||
modes = [str(m) for m in g.objects(term, PR.authoringMode)]
|
||||
concrete = (kind == "class"
|
||||
and Literal(False) in set(g.objects(term, PR.isAbstract)))
|
||||
if modes and not concrete:
|
||||
f.violation("TN-026", name,
|
||||
"declares pr:authoringMode; only a concrete class does, "
|
||||
"since a vocabulary term is declared by definition")
|
||||
elif concrete:
|
||||
if not modes:
|
||||
f.violation("TN-026", name, "does not declare pr:authoringMode")
|
||||
else:
|
||||
if "HARVESTED" in modes and not list(g.objects(term, PR.harvestSource)):
|
||||
f.violation("TN-026", name, "is HARVESTED and names no harvestSource")
|
||||
axis = provenance_axis(g, term)
|
||||
if axis == "CapturedObject" and "HARVESTED" not in modes:
|
||||
f.violation("TN-026", name,
|
||||
"descends from CapturedObject and must declare HARVESTED")
|
||||
if axis == "DefinedObject" and "ASSERTED" not in modes:
|
||||
f.violation("TN-026", name,
|
||||
"descends from DefinedObject and must declare ASSERTED")
|
||||
|
||||
if kind == "class":
|
||||
values = list(g.objects(term, PR.isAbstract))
|
||||
# TN-007 — declared exactly once, explicitly
|
||||
if len(values) != 1:
|
||||
f.violation("TN-007", name,
|
||||
"declares pr:isAbstract %d times; exactly one is required"
|
||||
% len(values))
|
||||
# TN-006 — a layer root is abstract
|
||||
elif term in layer_roots and values[0].toPython() is not True:
|
||||
f.violation("TN-006", name, "is a layer root and must be abstract")
|
||||
# TN-027 — a concrete class has one layer and one provenance
|
||||
elif values[0].toPython() is False:
|
||||
check_attachment(g, term, name, layer_roots, f)
|
||||
|
||||
if kind in ("relation", "attribute"):
|
||||
check_property_typing(g, term, name, lex, f)
|
||||
|
||||
# TN-003 — the display properties are annotations
|
||||
for display in (PR.shortLabel, PR.acronym):
|
||||
name = local(display)
|
||||
types = set(g.objects(display, RDF.type))
|
||||
if not types:
|
||||
continue
|
||||
if OWL.AnnotationProperty not in types:
|
||||
f.violation("TN-003", name, "is declared %s and must be an annotation property"
|
||||
% ", ".join(sorted(local(t) or str(t) for t in types)))
|
||||
|
||||
|
||||
def check_attachment(g, term, name, layer_roots, f):
|
||||
"""TN-027 — exactly one layer root and one provenance axis above."""
|
||||
seen, stack = set(), [term]
|
||||
while stack:
|
||||
node = stack.pop()
|
||||
for parent in g.objects(node, RDFS.subClassOf):
|
||||
if isinstance(parent, URIRef) and parent not in seen:
|
||||
seen.add(parent)
|
||||
stack.append(parent)
|
||||
layers = seen & layer_roots
|
||||
axes = seen & {PR.DefinedObject, PR.CapturedObject}
|
||||
if len(layers) != 1:
|
||||
f.violation("TN-027", name, "descends from %d layer roots; exactly one is required"
|
||||
% len(layers))
|
||||
if len(axes) != 1:
|
||||
f.violation("TN-027", name, "descends from %d provenance axes; exactly one is required"
|
||||
% len(axes))
|
||||
|
||||
|
||||
def check_property_typing(g, term, name, lex, f):
|
||||
"""TN-025 and TN-028 — domain, range, and inherited domain."""
|
||||
domains = list(g.objects(term, RDFS.domain))
|
||||
ranges = list(g.objects(term, RDFS.range))
|
||||
|
||||
if not ranges and Literal(True) not in set(g.objects(term, PR.polymorphic)):
|
||||
f.violation("TN-025", name, "declares no rdfs:range")
|
||||
polymorphic = Literal(True) in set(g.objects(term, PR.polymorphic))
|
||||
if not domains:
|
||||
if polymorphic:
|
||||
f.review("TN-025", name,
|
||||
"declares itself polymorphic; confirm the scope is declared in SHACL")
|
||||
else:
|
||||
f.violation("TN-025", name,
|
||||
"omits its domain in silence; declare pr:polymorphic true "
|
||||
"or state the domain")
|
||||
|
||||
# TN-028 — a sub-property inherits the domain of its parent
|
||||
for parent in g.objects(term, RDFS.subPropertyOf):
|
||||
if not isinstance(parent, URIRef) or local(parent) is None:
|
||||
continue
|
||||
parent_domains = set(g.objects(parent, RDFS.domain))
|
||||
if not parent_domains or not domains:
|
||||
continue
|
||||
if set(domains) & parent_domains:
|
||||
continue
|
||||
compatible = False
|
||||
for d in domains:
|
||||
ancestors, stack = set(), [d]
|
||||
while stack:
|
||||
node = stack.pop()
|
||||
for up in g.objects(node, RDFS.subClassOf):
|
||||
if isinstance(up, URIRef) and up not in ancestors:
|
||||
ancestors.add(up)
|
||||
stack.append(up)
|
||||
if ancestors & parent_domains:
|
||||
compatible = True
|
||||
if not compatible:
|
||||
f.review("TN-028", name,
|
||||
"is a sub-property of %s, whose domain does not cover its own; "
|
||||
"asserting it will retype the subject"
|
||||
% local(parent))
|
||||
|
||||
|
||||
def check_label(g, terms, f):
|
||||
"""TN-018 to TN-023 — the label is derived, and carries nothing else."""
|
||||
for term, kind in terms.items():
|
||||
name = local(term)
|
||||
if name is None:
|
||||
continue
|
||||
labels = list(g.objects(term, RDFS.label))
|
||||
if not labels:
|
||||
f.violation("TN-023", name, "carries no rdfs:label")
|
||||
continue
|
||||
if len(labels) > 1:
|
||||
f.violation("TN-023", name, "carries %d labels; the derivation yields one"
|
||||
% len(labels))
|
||||
text = str(labels[0])
|
||||
expected = decamelise(name, kind in ("class", "individual"))
|
||||
if text != expected:
|
||||
f.violation("TN-018", name, "label is %r; the derivation yields %r"
|
||||
% (text, expected))
|
||||
if "(" in text or ")" in text:
|
||||
f.violation("TN-021", name, "the label carries a gloss; it belongs in the comment")
|
||||
if re.search(r"\b(deprecated|obsolete|draft|published|v\d)\b", text, re.I):
|
||||
f.violation("TN-020", name, "the label carries state; the axiom carries it")
|
||||
|
||||
|
||||
def check_namespace(g, f):
|
||||
"""EV-014 — the term namespace is never versioned."""
|
||||
for prefix, uri in g.namespaces():
|
||||
if str(uri).startswith("https://ontology.pernod-ricard.com/metamodel/") \
|
||||
and str(uri) != NS:
|
||||
f.violation("EV-014", str(uri),
|
||||
"the prefix %s binds a versioned namespace; the version lives "
|
||||
"in owl:versionIRI alone" % prefix)
|
||||
|
||||
|
||||
# ------------------------------------------------------------------------ main
|
||||
|
||||
def collect(g):
|
||||
terms = OrderedDict()
|
||||
for rdf_type, kind in KINDS.items():
|
||||
for term in sorted(g.subjects(RDF.type, rdf_type), key=str):
|
||||
if local(term) is None:
|
||||
continue
|
||||
if (term, OWL.deprecated, Literal(True)) in g:
|
||||
continue
|
||||
terms[term] = kind
|
||||
return terms
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--ontology", default=DEFAULT_ONTOLOGY)
|
||||
ap.add_argument("--json", default=None, help="write the findings as JSON")
|
||||
ap.add_argument("--quiet", action="store_true", help="print the summary only")
|
||||
args = ap.parse_args()
|
||||
|
||||
lex = yaml.safe_load(open(LEXICON, encoding="utf-8"))
|
||||
rules = yaml.safe_load(open(RULES, encoding="utf-8"))
|
||||
abstractness = {a["term"]: a["is_abstract"] for a in rules["abstractness"]}
|
||||
|
||||
g = Graph()
|
||||
g.parse(args.ontology, format="turtle")
|
||||
terms = collect(g)
|
||||
|
||||
f = Findings()
|
||||
check_identifier(g, terms, lex, f)
|
||||
check_declaration(g, terms, lex, f, abstractness)
|
||||
check_label(g, terms, f)
|
||||
check_namespace(g, f)
|
||||
|
||||
titles = {r["id"]: r["title"] for r in rules["rules"]}
|
||||
print("=" * 78)
|
||||
print("T-BOX NAMING CHECK")
|
||||
print(" ontology : %s" % args.ontology)
|
||||
print(" terms : %d active" % len(terms))
|
||||
print("=" * 78)
|
||||
|
||||
if not args.quiet:
|
||||
for verdict in (VIOLATION, REVIEW):
|
||||
items = [i for i in f.items if i["verdict"] == verdict]
|
||||
if not items:
|
||||
continue
|
||||
print("\n%s — %d" % (verdict, len(items)))
|
||||
by_rule = OrderedDict()
|
||||
for i in items:
|
||||
by_rule.setdefault(i["rule"], []).append(i)
|
||||
for rule, group in sorted(by_rule.items()):
|
||||
print("\n [%s] %s (%d)" % (rule, titles.get(rule, ""), len(group)))
|
||||
for i in group[:12]:
|
||||
print(" %-34s %s" % (i["term"], i["message"]))
|
||||
if len(group) > 12:
|
||||
print(" ... and %d more" % (len(group) - 12))
|
||||
|
||||
print("\n" + "=" * 78)
|
||||
print("RESULT : %d violation(s) | %d for review | %d term(s) checked"
|
||||
% (len(f.violations), len(f.reviews), len(terms)))
|
||||
print("=" * 78)
|
||||
|
||||
if args.json:
|
||||
directory = os.path.dirname(os.path.abspath(args.json))
|
||||
if directory:
|
||||
os.makedirs(directory, exist_ok=True)
|
||||
json.dump({"ontology": args.ontology, "terms": len(terms),
|
||||
"violations": len(f.violations), "reviews": len(f.reviews),
|
||||
"findings": f.items},
|
||||
open(args.json, "w", encoding="utf-8"), indent=2)
|
||||
print("json: %s" % args.json)
|
||||
|
||||
sys.exit(1 if f.violations else 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user