#!/usr/bin/env python3 """ Enforce the naming and declaration rules of the T-Box rulebook. USAGE python3 check_tbox_naming.py [--ontology PATH] [--json PATH] [--quiet] Named as the executor of 25 rules. EV-011 forbids declaring a rule blocking without one, so this script is what makes those declarations true. THREE VERDICTS VIOLATION the rule is settled mechanically and is broken -> exit 1 REVIEW the checker cannot conclude; a person must -> exit 0 PASS conforming The middle verdict is the point. Without it the choice would be between blocking on false positives and staying silent on real problems. What the checker cannot decide alone comes from naming_lexicon.yaml; a word absent from the lexicon produces a REVIEW, never a VIOLATION. """ import argparse import json import os import re import sys from collections import OrderedDict import yaml from rdflib import Graph, Literal, Namespace, RDF, RDFS, OWL, URIRef, XSD HERE = os.path.dirname(os.path.abspath(__file__)) RULES = os.path.join(HERE, "rules.yaml") LEXICON = os.path.join(HERE, "naming_lexicon.yaml") DEFAULT_ONTOLOGY = os.path.normpath(os.path.join(HERE, "..", "ontology", "pr_metamodel.ttl")) NS = "https://ontology.pernod-ricard.com/metamodel/" PR = Namespace(NS) VIOLATION, REVIEW = "VIOLATION", "REVIEW" KINDS = OrderedDict([ (OWL.Class, "class"), (OWL.ObjectProperty, "relation"), (OWL.DatatypeProperty, "attribute"), (OWL.AnnotationProperty, "annotation"), ]) # ------------------------------------------------------------------- findings class Findings(object): def __init__(self): self.items = [] def add(self, verdict, rule, term, message): self.items.append({"verdict": verdict, "rule": rule, "term": term, "message": message}) def violation(self, rule, term, message): self.add(VIOLATION, rule, term, message) def review(self, rule, term, message): self.add(REVIEW, rule, term, message) @property def violations(self): return [i for i in self.items if i["verdict"] == VIOLATION] @property def reviews(self): return [i for i in self.items if i["verdict"] == REVIEW] # -------------------------------------------------------------------- helpers def local(uri): s = str(uri) return s[len(NS):] if s.startswith(NS) else None def tokens(name): """Split a CamelCase local name. A run of capitals is one token: an acronym.""" return re.findall(r"[A-Z]+(?![a-z])|[A-Z][a-z0-9]*|^[a-z0-9]+|[a-z0-9]+", name) def decamelise(name, is_class): """TN-018. A run of capitals is kept together: it carries an acronym.""" spaced = re.sub(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])", " ", name) return spaced if is_class else spaced[0].lower() + spaced[1:] def acronyms_in(name, lex): """Runs of two or more capitals, plus any known short form appearing whole. The negative lookahead matters: without it, BIField yields BIF rather than BI, because the run swallows the initial capital of the word that follows. """ found = set(re.findall(r"[A-Z]{2,}(?![a-z])", name)) for short in lex["known_acronyms"]: if re.search(r"(?:^|(?<=[a-z]))%s(?![a-z])" % re.escape(short), name): found.add(short) return sorted(found) def strip_case(text): return re.sub(r"[^a-z0-9]", "", (text or "").lower()) # --------------------------------------------------------------------- checks def provenance_axis(g, term): """The provenance axis a class descends from, or None.""" seen, stack = set(), [term] while stack: node = stack.pop() for parent in g.objects(node, RDFS.subClassOf): if isinstance(parent, URIRef) and parent not in seen: seen.add(parent) stack.append(parent) if PR.CapturedObject in seen: return "CapturedObject" if PR.DefinedObject in seen: return "DefinedObject" return None def check_identifier(g, terms, lex, f): """TN-001 to TN-015 — the form of the IRI.""" short_labels = {t: str(v) for t in terms for v in g.objects(t, PR.shortLabel)} allowed_acronyms = set() for value in short_labels.values(): allowed_acronyms |= set(re.findall(r"[A-Z]{2,}", value)) for term, kind in terms.items(): name = local(term) if name is None: continue # TN-001 — CamelCase, no separator, correct initial if re.search(r"[^A-Za-z0-9]", name): f.violation("TN-001", name, "the local name carries a separator") expect_upper = kind in ("class", "individual") if name and name[0].isupper() != expect_upper: f.violation("TN-001", name, "a %s starts with %s case" % (kind, "upper" if expect_upper else "lower")) # TN-004 — no digit if any(c.isdigit() for c in name): f.violation("TN-004", name, "a digit belongs to a value, not an identity") # TN-002 — no acronym, except one reproduced from a target short label for a in acronyms_in(name, lex): if kind == "relation" and a in allowed_acronyms: continue # TN-011 exception expansion = lex["known_acronyms"].get(a) f.violation("TN-002", name, "carries the acronym %s%s" % (a, " (%s)" % expansion if expansion else "")) if kind == "class": check_class_name(name, lex, f) elif kind == "relation": check_relation_name(g, term, name, lex, f) elif kind == "attribute": check_attribute_name(g, term, name, lex, f) def check_class_name(name, lex, f): """TN-005 — a singular common noun, no type word.""" if name in lex["type_word_exceptions"]: return for w in lex["type_words"]: if re.search(r"(?:^|(?<=[a-z]))%s(?![a-z])" % w, name): f.violation("TN-005", name, "carries the type word %s and is not a recorded abstraction" % w) last = tokens(name)[-1] if tokens(name) else name if last.endswith("s") and last not in lex["not_plural"]: f.review("TN-005", name, "%s looks plural; a class name is singular" % last) def check_relation_name(g, term, name, lex, f): """TN-009, TN-010, TN-011 — verb first, is disambiguated, target named whole.""" toks = tokens(name) head = toks[0] if toks else name # TN-009 — an adverb is transparent, the verb follows if head in lex["adverbs"]: toks = toks[1:] head = toks[0].lower() if toks else head head = head.lower() if head in lex["copula"]: pass # TN-010, copula use elif head in lex["active_verbs"] or head in lex["participles"]: pass else: f.review("TN-009", name, "%s is not a verb known to the lexicon; add it there or rename" % head) # TN-011 — if the target class is named, it is named by its short label whole for rng in g.objects(term, RDFS.range): target = local(rng) if not target: continue short = next((str(v) for v in g.objects(rng, PR.shortLabel)), None) reference = strip_case(short) if short else strip_case(target) tail = strip_case("".join(toks[1:])) if not tail or tail == reference: continue if reference.endswith(tail) or tail in reference: f.violation("TN-011", name, "names its target by a truncation; %s expects %s" % (target, short or target)) def check_attribute_name(g, term, name, lex, f): """TN-012 to TN-015 — a noun, no relational suffix, booleans predicative.""" ranges = [str(r) for r in g.objects(term, RDFS.range)] is_boolean = str(XSD.boolean) in ranges is_date = any(r in lex["date_ranges"] for r in ranges) toks = tokens(name) head = (toks[0] if toks else name).lower() if is_boolean: if head not in lex["copula"]: f.violation("TN-015", name, "a boolean attribute is predicative and begins with is") return # TN-012 — a noun, and in particular no has if head == "has": f.violation("TN-012", name, "an attribute is a noun; the domain already says who has it") elif head in lex["active_verbs"] or head in lex["copula"]: f.violation("TN-012", name, "%s is a verb; an attribute is a noun" % head) elif head in lex["participles"]: f.review("TN-012", name, "%s is a participle; confirm this reads as a noun" % head) # TN-014 — relational suffixes are reserved for relations last = toks[-1] if toks else name if last in lex["relational_suffixes"]: f.violation("TN-014", name, "ends in %s, a form reserved for object properties" % last) # TN-013 — a date attribute is nominal if is_date and (last in lex["relational_suffixes"] or head in lex["participles"]): f.violation("TN-013", name, "a date attribute is a noun, not a participle") def check_declaration(g, terms, lex, f, abstractness): """TN-003, TN-006, TN-007, TN-024, TN-025, TN-026, TN-027, TN-028.""" layer_roots = {t for t in terms if terms[t] == "class" and (local(t) or "").endswith("LayerObject")} for term, kind in terms.items(): name = local(term) if name is None: continue # TN-024 — every active term carries a comment if not list(g.objects(term, RDFS.comment)): f.violation("TN-024", name, "carries no rdfs:comment") # TN-026 — only a concrete class declares how its instances are produced modes = [str(m) for m in g.objects(term, PR.authoringMode)] concrete = (kind == "class" and Literal(False) in set(g.objects(term, PR.isAbstract))) if modes and not concrete: f.violation("TN-026", name, "declares pr:authoringMode; only a concrete class does, " "since a vocabulary term is declared by definition") elif concrete: if not modes: f.violation("TN-026", name, "does not declare pr:authoringMode") else: if "HARVESTED" in modes and not list(g.objects(term, PR.harvestSource)): f.violation("TN-026", name, "is HARVESTED and names no harvestSource") axis = provenance_axis(g, term) if axis == "CapturedObject" and "HARVESTED" not in modes: f.violation("TN-026", name, "descends from CapturedObject and must declare HARVESTED") if axis == "DefinedObject" and "ASSERTED" not in modes: f.violation("TN-026", name, "descends from DefinedObject and must declare ASSERTED") if kind == "class": values = list(g.objects(term, PR.isAbstract)) # TN-007 — declared exactly once, explicitly if len(values) != 1: f.violation("TN-007", name, "declares pr:isAbstract %d times; exactly one is required" % len(values)) # TN-006 — a layer root is abstract elif term in layer_roots and values[0].toPython() is not True: f.violation("TN-006", name, "is a layer root and must be abstract") # TN-027 — a concrete class has one layer and one provenance elif values[0].toPython() is False: check_attachment(g, term, name, layer_roots, f) if kind in ("relation", "attribute"): check_property_typing(g, term, name, lex, f) # TN-003 — the display properties are annotations for display in (PR.shortLabel, PR.acronym): name = local(display) types = set(g.objects(display, RDF.type)) if not types: continue if OWL.AnnotationProperty not in types: f.violation("TN-003", name, "is declared %s and must be an annotation property" % ", ".join(sorted(local(t) or str(t) for t in types))) def check_attachment(g, term, name, layer_roots, f): """TN-027 — exactly one layer root and one provenance axis above.""" seen, stack = set(), [term] while stack: node = stack.pop() for parent in g.objects(node, RDFS.subClassOf): if isinstance(parent, URIRef) and parent not in seen: seen.add(parent) stack.append(parent) layers = seen & layer_roots axes = seen & {PR.DefinedObject, PR.CapturedObject} if len(layers) != 1: f.violation("TN-027", name, "descends from %d layer roots; exactly one is required" % len(layers)) if len(axes) != 1: f.violation("TN-027", name, "descends from %d provenance axes; exactly one is required" % len(axes)) def check_property_typing(g, term, name, lex, f): """TN-025 and TN-028 — domain, range, and inherited domain.""" domains = list(g.objects(term, RDFS.domain)) ranges = list(g.objects(term, RDFS.range)) if not ranges and Literal(True) not in set(g.objects(term, PR.polymorphic)): f.violation("TN-025", name, "declares no rdfs:range") polymorphic = Literal(True) in set(g.objects(term, PR.polymorphic)) if not domains: if polymorphic: f.review("TN-025", name, "declares itself polymorphic; confirm the scope is declared in SHACL") else: f.violation("TN-025", name, "omits its domain in silence; declare pr:polymorphic true " "or state the domain") # TN-028 — a sub-property inherits the domain of its parent for parent in g.objects(term, RDFS.subPropertyOf): if not isinstance(parent, URIRef) or local(parent) is None: continue parent_domains = set(g.objects(parent, RDFS.domain)) if not parent_domains or not domains: continue if set(domains) & parent_domains: continue compatible = False for d in domains: ancestors, stack = set(), [d] while stack: node = stack.pop() for up in g.objects(node, RDFS.subClassOf): if isinstance(up, URIRef) and up not in ancestors: ancestors.add(up) stack.append(up) if ancestors & parent_domains: compatible = True if not compatible: f.review("TN-028", name, "is a sub-property of %s, whose domain does not cover its own; " "asserting it will retype the subject" % local(parent)) def check_label(g, terms, f): """TN-018 to TN-023 — the label is derived, and carries nothing else.""" for term, kind in terms.items(): name = local(term) if name is None: continue labels = list(g.objects(term, RDFS.label)) if not labels: f.violation("TN-023", name, "carries no rdfs:label") continue if len(labels) > 1: f.violation("TN-023", name, "carries %d labels; the derivation yields one" % len(labels)) text = str(labels[0]) expected = decamelise(name, kind in ("class", "individual")) if text != expected: f.violation("TN-018", name, "label is %r; the derivation yields %r" % (text, expected)) if "(" in text or ")" in text: f.violation("TN-021", name, "the label carries a gloss; it belongs in the comment") if re.search(r"\b(deprecated|obsolete|draft|published|v\d)\b", text, re.I): f.violation("TN-020", name, "the label carries state; the axiom carries it") def check_namespace(g, f): """EV-014 — the term namespace is never versioned.""" for prefix, uri in g.namespaces(): if str(uri).startswith("https://ontology.pernod-ricard.com/metamodel/") \ and str(uri) != NS: f.violation("EV-014", str(uri), "the prefix %s binds a versioned namespace; the version lives " "in owl:versionIRI alone" % prefix) # ------------------------------------------------------------------------ main def collect(g): terms = OrderedDict() for rdf_type, kind in KINDS.items(): for term in sorted(g.subjects(RDF.type, rdf_type), key=str): if local(term) is None: continue if (term, OWL.deprecated, Literal(True)) in g: continue terms[term] = kind return terms def main(): ap = argparse.ArgumentParser() ap.add_argument("--ontology", default=DEFAULT_ONTOLOGY) ap.add_argument("--json", default=None, help="write the findings as JSON") ap.add_argument("--quiet", action="store_true", help="print the summary only") args = ap.parse_args() lex = yaml.safe_load(open(LEXICON, encoding="utf-8")) rules = yaml.safe_load(open(RULES, encoding="utf-8")) abstractness = {a["term"]: a["is_abstract"] for a in rules["abstractness"]} g = Graph() g.parse(args.ontology, format="turtle") terms = collect(g) f = Findings() check_identifier(g, terms, lex, f) check_declaration(g, terms, lex, f, abstractness) check_label(g, terms, f) check_namespace(g, f) titles = {r["id"]: r["title"] for r in rules["rules"]} print("=" * 78) print("T-BOX NAMING CHECK") print(" ontology : %s" % args.ontology) print(" terms : %d active" % len(terms)) print("=" * 78) if not args.quiet: for verdict in (VIOLATION, REVIEW): items = [i for i in f.items if i["verdict"] == verdict] if not items: continue print("\n%s — %d" % (verdict, len(items))) by_rule = OrderedDict() for i in items: by_rule.setdefault(i["rule"], []).append(i) for rule, group in sorted(by_rule.items()): print("\n [%s] %s (%d)" % (rule, titles.get(rule, ""), len(group))) for i in group[:12]: print(" %-34s %s" % (i["term"], i["message"])) if len(group) > 12: print(" ... and %d more" % (len(group) - 12)) print("\n" + "=" * 78) print("RESULT : %d violation(s) | %d for review | %d term(s) checked" % (len(f.violations), len(f.reviews), len(terms))) print("=" * 78) if args.json: directory = os.path.dirname(os.path.abspath(args.json)) if directory: os.makedirs(directory, exist_ok=True) json.dump({"ontology": args.ontology, "terms": len(terms), "violations": len(f.violations), "reviews": len(f.reviews), "findings": f.items}, open(args.json, "w", encoding="utf-8"), indent=2) print("json: %s" % args.json) sys.exit(1 if f.violations else 0) if __name__ == "__main__": main()