#!/usr/bin/env python3 """ Migrate the Pernod Ricard Data MetaModel from v1.6 to v2.0. USAGE python3 migrate_tbox_v2_0.py # dry run, writes nothing python3 migrate_tbox_v2_0.py --apply # writes in place python3 migrate_tbox_v2_0.py --apply --log-dir logs/ --ontology path to the ontology TTL (default ../ontology/pr_metamodel.ttl) --instances path to an instance TTL (repeatable) --shapes path to a shapes TTL (repeatable) DESIGN EV-004 dry run is the default; nothing is written without --apply EV-005 every edit goes through rdflib, never a regex on the text EV-006 guards test the target state, so a replay reports zero change EV-015 an execution log is written for every attempt Run on GrosseBertha, inside the pinned venv: . venv/bin/activate && python3 migrate_tbox_v2_0.py """ import argparse import datetime import hashlib import json import os import re import sys from collections import OrderedDict import yaml from rdflib import Graph, Literal, Namespace, RDF, RDFS, OWL, URIRef, XSD HERE = os.path.dirname(os.path.abspath(__file__)) SPEC = os.path.join(HERE, "renames.yaml") RULES = os.path.normpath(os.path.join(HERE, "..", "rules.yaml")) DCTERMS = Namespace("http://purl.org/dc/terms/") # ----------------------------------------------------------------- utilities class Report(object): """Counts every step, so that idempotence is provable and not merely hoped.""" def __init__(self): self.steps = OrderedDict() self.notes = [] def add(self, step, n, detail=None): self.steps[step] = self.steps.get(step, 0) + n if detail: self.notes.append("%s: %s" % (step, detail)) @property def total(self): return sum(self.steps.values()) def md5(path): h = hashlib.md5() with open(path, "rb") as f: for chunk in iter(lambda: f.read(65536), b""): h.update(chunk) return h.hexdigest() def local(uri, ns): s = str(uri) return s[len(ns):] if s.startswith(ns) else None def decamelise(name, is_class): """TN-018. Consecutive capitals are kept together: they carry an acronym.""" spaced = re.sub(r"(?<=[a-z0-9])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])", " ", name) return spaced if is_class else spaced[0].lower() + spaced[1:] # ------------------------------------------------------------------ the work def rename(graph, pr, old, new, report, step): """EV-001. Rewrite every triple naming the identifier, in all three positions.""" src, dst = pr[old], pr[new] if (dst, None, None) in graph and (src, None, None) not in graph: return 0 # EV-006: already done n = 0 for s, p, o in list(graph): ns, np_, no = s, p, o if s == src: ns = dst if p == src: np_ = dst if o == src: no = dst if (ns, np_, no) != (s, p, o): graph.remove((s, p, o)) graph.add((ns, np_, no)) n += 1 report.add(step, n) return n def merge(graph, pr, src_name, dst_name, report): """The source disappears into an existing target, which keeps its declaration.""" src, dst = pr[src_name], pr[dst_name] n = 0 for s, p, o in list(graph.triples((None, src, None))): graph.remove((s, p, o)) graph.add((s, dst, o)) n += 1 for s, p, o in list(graph.triples((src, None, None))): graph.remove((s, p, o)) # drop the source declaration n += 1 report.add("merge", n) return n def drop_subject(graph, subject, report, step): """EV-003. Remove the block AND every reference naming it, or inference rebuilds it.""" n = 0 for t in list(graph.triples((subject, None, None))): graph.remove(t) n += 1 for t in list(graph.triples((None, None, subject))): graph.remove(t) n += 1 for t in list(graph.triples((None, subject, None))): graph.remove(t) n += 1 report.add(step, n) return n def count_instances(graphs, term): """EV-002. The proof required before any permanent withdrawal.""" n = 0 for g in graphs: n += len(list(g.triples((None, RDF.type, term)))) n += len(list(g.triples((None, term, None)))) n += len(list(g.triples((None, None, term)))) return n def migrate(onto, others, spec, rules, report): ns = spec["meta"]["namespace"] pr = Namespace(ns) all_graphs = [onto] + others # 1 — deprecated terms are withdrawn outright (phase clause), proof first if spec["structural"].get("drop_deprecated"): for subj in list(onto.subjects(OWL.deprecated, Literal(True))): name = local(subj, ns) or str(subj) used = sum(count_instances([g], subj) for g in others) if used: report.add("deprecated_kept", 1, "%s still used %d times" % (name, used)) continue for g in all_graphs: drop_subject(g, subj, report, "deprecated_dropped") # 2 — renames, in dependency order for order in (1, 2, 3): for r in [x for x in spec["renames"] if x["order"] == order]: for g in all_graphs: rename(g, pr, r["from"], r["to"], report, "rename_order_%d" % order) # 3 — merges for m in spec.get("merges") or []: for g in all_graphs: merge(g, pr, m["from"], m["into"], report) # 4 — reclassify display properties (TN-003) for rc in spec["structural"]["reclassify"]: term = pr[rc["term"]] if (term, RDF.type, OWL.AnnotationProperty) not in onto: onto.remove((term, RDF.type, getattr(OWL, rc["from"]))) onto.add((term, RDF.type, OWL.AnnotationProperty)) report.add("reclassified", 1, rc["term"]) # 5 — create the governance layer root (TN-027) for c in spec["structural"].get("create_classes") or []: term = pr[c["term"]] if (term, RDF.type, OWL.Class) not in onto: onto.add((term, RDF.type, OWL.Class)) onto.add((term, RDFS.subClassOf, pr[c["parent"]])) onto.add((term, RDFS.label, Literal(c["label"]))) onto.add((term, RDFS.comment, Literal(" ".join(c["comment"].split())))) onto.add((term, pr.isAbstract, Literal(True))) report.add("classes_created", 1, c["term"]) # 6 — reparent (TN-027) for rp in spec["structural"].get("reparent") or []: term, parent = pr[rp["term"]], pr[rp["parent"]] if (term, RDFS.subClassOf, parent) not in onto: onto.add((term, RDFS.subClassOf, parent)) report.add("reparented", 1, "%s -> %s" % (rp["term"], rp["parent"])) # 7 — sub-properties (TN-028) for sp in spec["structural"].get("subproperties") or []: term, parent = pr[sp["term"]], pr[sp["parent"]] if (term, RDFS.subPropertyOf, parent) not in onto: onto.add((term, RDFS.subPropertyOf, parent)) report.add("subproperties", 1, "%s -> %s" % (sp["term"], sp["parent"])) # 8 — controlled values become literals (TN-016, TN-017) for conv in spec["structural"].get("to_literal") or []: prop = pr[conv["property"]] if (prop, RDF.type, OWL.DatatypeProperty) not in onto: onto.remove((prop, RDF.type, OWL.ObjectProperty)) onto.add((prop, RDF.type, OWL.DatatypeProperty)) onto.remove((prop, RDFS.range, None)) onto.add((prop, RDFS.range, XSD.string)) onto.add((prop, RDFS.domain, pr[conv["domain"]])) report.add("to_literal_property", 1, conv["property"]) for g in all_graphs: # rewrite the asserted values for s, p, o in list(g.triples((None, prop, None))): name = local(o, ns) if name and name in conv["value_map"]: g.remove((s, p, o)) g.add((s, p, Literal(conv["value_map"][name]))) report.add("to_literal_values", 1) for ind in conv["drop_individuals"] + [conv["drop_class"]]: subj = pr[ind] if (subj, None, None) in onto: for g in all_graphs: drop_subject(g, subj, report, "to_literal_dropped") # 9 — abstractness, from the rulebook annex (TN-006, TN-007) for a in rules["abstractness"]: term = pr[a["term"]] if (term, RDF.type, OWL.Class) not in onto: report.add("abstract_missing_class", 1, a["term"]) continue want = Literal(bool(a["is_abstract"])) if (term, pr.isAbstract, want) not in onto: onto.remove((term, pr.isAbstract, None)) onto.add((term, pr.isAbstract, want)) report.add("isAbstract_declared", 1) # 10 — display annotations, from the rulebook annex (TN-022) for d in rules["display"]: term = pr[d["iri"]] for prop, value in ((pr.shortLabel, d.get("short_label")), (pr.acronym, d.get("acronym"))): if not value: continue if (term, prop, Literal(value)) not in onto: onto.remove((term, prop, None)) onto.add((term, prop, Literal(value))) report.add("display_set", 1) # 11 — regenerate every label by derivation (TN-018, TN-023) kinds = (OWL.Class, OWL.ObjectProperty, OWL.DatatypeProperty, OWL.AnnotationProperty) for kind in kinds: for term in set(onto.subjects(RDF.type, kind)): name = local(term, ns) if not name: continue want = Literal(decamelise(name, kind == OWL.Class)) if (term, RDFS.label, want) not in onto: onto.remove((term, RDFS.label, None)) onto.add((term, RDFS.label, want)) report.add("labels_regenerated", 1) # 12 — bump the ontology version (EV-014: the namespace itself never moves) target = URIRef(ns.rstrip("/") + "/" + spec["meta"]["to_version"]) for onto_iri in set(onto.subjects(RDF.type, OWL.Ontology)): if (onto_iri, OWL.versionIRI, target) not in onto: onto.remove((onto_iri, OWL.versionIRI, None)) onto.add((onto_iri, OWL.versionIRI, target)) report.add("version_bumped", 1, str(target)) # ----------------------------------------------------------------------- main def main(): ap = argparse.ArgumentParser() ap.add_argument("--apply", action="store_true", help="write the files (default: dry run)") ap.add_argument("--ontology", default=os.path.join(HERE, "..", "ontology", "pr_metamodel.ttl")) ap.add_argument("--instances", action="append", default=[]) ap.add_argument("--shapes", action="append", default=[]) ap.add_argument("--log-dir", default=os.path.join(HERE, "logs")) args = ap.parse_args() spec = yaml.safe_load(open(SPEC, encoding="utf-8")) rules = yaml.safe_load(open(RULES, encoding="utf-8")) paths = [args.ontology] + args.instances + args.shapes missing = [p for p in paths if not os.path.exists(p)] if missing: print("MISSING INPUT") for p in missing: print(" " + p) sys.exit(1) inputs = OrderedDict((p, md5(p)) for p in paths) onto = Graph() onto.parse(args.ontology, format="turtle") others, other_paths = [], args.instances + args.shapes for p in other_paths: g = Graph() g.parse(p, format="turtle") others.append(g) before = [len(onto)] + [len(g) for g in others] report = Report() migrate(onto, others, spec, rules, report) after = [len(onto)] + [len(g) for g in others] print("MIGRATION %s -> %s %s" % (spec["meta"]["from_version"], spec["meta"]["to_version"], "APPLY" if args.apply else "DRY RUN")) print() for step, n in report.steps.items(): print(" %-28s %6d" % (step, n)) print(" %-28s %6d" % ("total changes", report.total)) print() for p, b, a in zip(paths, before, after): print(" %-46s %6d -> %6d triples" % (os.path.basename(p), b, a)) if report.notes: print() for n in report.notes: print(" note: " + n) log = { "attempt_timestamp": datetime.datetime.now().isoformat(timespec="seconds"), "mode": "apply" if args.apply else "dry-run", "from_version": spec["meta"]["from_version"], "to_version": spec["meta"]["to_version"], "input_checksums": inputs, "steps": report.steps, "total_changes": report.total, "triples_before": dict(zip(paths, before)), "triples_after": dict(zip(paths, after)), "notes": report.notes, } os.makedirs(args.log_dir, exist_ok=True) stamp = datetime.datetime.now().strftime("%Y%m%dT%H%M%S") log_path = os.path.join(args.log_dir, "migration_%s.json" % stamp) json.dump(log, open(log_path, "w", encoding="utf-8"), indent=2) print("\n log: %s" % log_path) if not args.apply: print("\n DRY RUN — nothing written. Re-run with --apply when the counts " "above are what you expect.") return onto.serialize(destination=args.ontology, format="turtle") for p, g in zip(other_paths, others): g.serialize(destination=p, format="turtle") print("\n written. Replay this script now: a second run must report zero " "changes (EV-006).") if __name__ == "__main__": main()