Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions packages/cli/src/mdl_cli/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -110,6 +110,12 @@ def _ontology():
try:
import mdl_ontology

# mdl_ontology exports lazily (PEP 562), so `import mdl_ontology` alone no longer
# pulls the RDF backend — that is deliberate, so pyoxigraph-free members like Lock
# stay reachable on a core install. Force-resolve one backend-dependent export here
# so a missing backend fails INSIDE this guard (→ the install hint), not later as a
# raw ImportError in the command body.
_ = mdl_ontology.serialize # rdf_export → _rdf → pyoxigraph
return mdl_ontology
except ImportError as e:
typer.secho(
Expand Down
25 changes: 25 additions & 0 deletions packages/cli/tests/test_core_without_ontology.py
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,31 @@ def test_ontology_commands_fail_with_install_hint(argv, tmp_path: Path):
assert "modelith-dbt[ontology]" in combined, combined


def test_init_scaffolds_without_rdflib(tmp_path: Path):
"""`mdl init` writes .mdl/lock.yaml with no RDF backend. Regression: scaffolding
imports mdl_ontology.lock (for Lock), which must NOT drag in the pyoxigraph-backed
providers — a lazy mdl_ontology.__init__ keeps Lock reachable on a core install."""
with _blocked_backend():
mod = importlib.reload(importlib.import_module("mdl_cli.main"))
r = runner.invoke(mod.app, ["init", str(tmp_path / "proj")])
assert r.exit_code == 0, r.output
assert (tmp_path / "proj" / ".mdl" / "lock.yaml").exists()


def test_import_erwin_without_rdflib(tmp_path: Path):
"""`mdl import erwin` into an empty folder scaffolds a model with no RDF backend."""
fixture = Path(__file__).resolve().parents[2] / "reverse" / "tests" / "test_erwin.py"
xml = tmp_path / "m.xml"
xml.write_text(fixture.read_text().split('_ERWIN = """')[1].split('"""')[0])
out = tmp_path / "proj"
with _blocked_backend():
mod = importlib.reload(importlib.import_module("mdl_cli.main"))
r = runner.invoke(mod.app, ["import", "erwin", str(xml), "-o", str(out)])
assert r.exit_code == 0, r.output
assert (out / ".mdl" / "lock.yaml").exists()
assert (out / "logical" / "entities" / "counterparty.yaml").exists()


def test_ontology_still_works_when_backend_present(tmp_path: Path):
"""Sanity: with the backend installed (the normal test env), an ontology command is
NOT gated — it runs. Guards against the helper over-blocking."""
Expand Down
172 changes: 106 additions & 66 deletions packages/ontology/src/mdl_ontology/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,72 +2,112 @@

Vocabulary-agnostic: FIBO is one reference bundle; ACORD/FHIR/ISO 20022/GS1/custom
vocabularies plug in by declaration. Depends only on `modelith-core` (§1.3).

Public names are exported LAZILY (PEP 562). Most of this package's surface pulls in the
RDF backend (pyoxigraph, the optional `[ontology]` extra), but a few members — notably
`Lock` (`.mdl/lock.yaml`, used by `mdl init` / import scaffolding) — do not. Eagerly
importing everything here meant that merely reaching `mdl_ontology.lock` dragged in
pyoxigraph, so `mdl init` / `mdl import erwin` crashed on an install without the extra —
breaking the promise that the core CLI runs without it. Resolving names on first access
instead keeps the pyoxigraph-free members reachable on a core-only install; a name that
does need the backend raises the ModuleNotFoundError only when it is actually used.
"""

from mdl_ontology.align import (
AlignmentProposal,
Candidate,
LexicalMatcher,
Matcher,
align_model,
confidence_band,
)
from mdl_ontology.fetch import (
FetchError,
FetchResult,
compute_lock,
fetch_all,
fetch_layer,
)
from mdl_ontology.ingest import save_ontology_upload
from mdl_ontology.layers import CoverageReport, check_layers, coverage_report
from mdl_ontology.lock import CACHE_REL, LOCK_MODES, Lock, OntologyLayerLock
from mdl_ontology.providers.cache import cache_from_registry, cache_resolved_term
from mdl_ontology.r2rml_export import (
R2RMLCoverage,
UnmappedError,
export_r2rml,
r2rml_coverage,
)
from mdl_ontology.rdf_export import export_rdf, export_shacl, serialize
from mdl_ontology.registry import (
OntologyRegistry,
ResolvedTerm,
VocabularySource,
build_registry,
)
# The TYPE_CHECKING block below re-imports every public name purely so type checkers and
# IDEs resolve them; at runtime __getattr__ does the real (lazy) import. Those imports read
# as unused to the linter, so F401 is suppressed for this file.
# ruff: noqa: F401

from __future__ import annotations

from typing import TYPE_CHECKING

# public name -> submodule it lives in. Resolved on first attribute access.
_EXPORTS = {
"align_model": "align",
"AlignmentProposal": "align",
"Candidate": "align",
"Matcher": "align",
"LexicalMatcher": "align",
"confidence_band": "align",
"FetchError": "fetch",
"FetchResult": "fetch",
"compute_lock": "fetch",
"fetch_all": "fetch",
"fetch_layer": "fetch",
"save_ontology_upload": "ingest",
"CoverageReport": "layers",
"check_layers": "layers",
"coverage_report": "layers",
"CACHE_REL": "lock",
"LOCK_MODES": "lock",
"Lock": "lock",
"OntologyLayerLock": "lock",
"cache_from_registry": "providers.cache",
"cache_resolved_term": "providers.cache",
"R2RMLCoverage": "r2rml_export",
"UnmappedError": "r2rml_export",
"export_r2rml": "r2rml_export",
"r2rml_coverage": "r2rml_export",
"export_rdf": "rdf_export",
"export_shacl": "rdf_export",
"serialize": "rdf_export",
"OntologyRegistry": "registry",
"ResolvedTerm": "registry",
"VocabularySource": "registry",
"build_registry": "registry",
}

__all__ = list(_EXPORTS)


def __getattr__(name: str):
"""PEP 562 lazy export: import the owning submodule only when the name is accessed."""
module = _EXPORTS.get(name)
if module is None:
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
import importlib

mod = importlib.import_module(f"{__name__}.{module}")
return getattr(mod, name)


def __dir__() -> list[str]:
return sorted(__all__)


__all__ = [
"OntologyRegistry",
"VocabularySource",
"ResolvedTerm",
"build_registry",
"check_layers",
"coverage_report",
"CoverageReport",
"export_rdf",
"export_shacl",
"export_r2rml",
"r2rml_coverage",
"R2RMLCoverage",
"UnmappedError",
"serialize",
"Lock",
"OntologyLayerLock",
"LOCK_MODES",
"CACHE_REL",
"fetch_all",
"fetch_layer",
"compute_lock",
"FetchError",
"FetchResult",
"cache_resolved_term",
"cache_from_registry",
"save_ontology_upload",
"align_model",
"AlignmentProposal",
"Candidate",
"Matcher",
"LexicalMatcher",
"confidence_band",
]
if TYPE_CHECKING: # keep static analysers / IDEs seeing the real symbols
# These are re-exports resolved lazily at runtime via __getattr__; the block exists
# only so type checkers and IDEs see the real symbols (hence the file-level noqa: F401).
from mdl_ontology.align import (
AlignmentProposal,
Candidate,
LexicalMatcher,
Matcher,
align_model,
confidence_band,
)
from mdl_ontology.fetch import (
FetchError,
FetchResult,
compute_lock,
fetch_all,
fetch_layer,
)
from mdl_ontology.ingest import save_ontology_upload
from mdl_ontology.layers import CoverageReport, check_layers, coverage_report
from mdl_ontology.lock import CACHE_REL, LOCK_MODES, Lock, OntologyLayerLock
from mdl_ontology.providers.cache import cache_from_registry, cache_resolved_term
from mdl_ontology.r2rml_export import (
R2RMLCoverage,
UnmappedError,
export_r2rml,
r2rml_coverage,
)
from mdl_ontology.rdf_export import export_rdf, export_shacl, serialize
from mdl_ontology.registry import (
OntologyRegistry,
ResolvedTerm,
VocabularySource,
build_registry,
)
43 changes: 37 additions & 6 deletions packages/reverse/src/mdl_reverse/writer.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@

from __future__ import annotations

import re
from pathlib import Path

from mdl_core.ir import Model
Expand Down Expand Up @@ -65,17 +66,34 @@ def _write_project_config(model: Model, root: Path) -> None:
def write_model(model: Model, root: Path) -> list[str]:
root = Path(root)
written: list[str] = []
used: set[str] = set() # rel paths already taken this write, for collision-avoidance

# Collection fields that default to []: `exclude_none` does not drop an empty
# list, so a reversed model would carry a noise `members: []` on every object.
_EMPTY_OK = ("members", "synonyms", "subtypes", "ontology_refs", "values")

def _unique(rel: str) -> str:
"""Disambiguate a filename collision. Slugging is lossy (two names can map to one
filename, e.g. 'Order Line' and 'Order/Line' -> 'order_line'), so a clash would
silently overwrite one object with another. Append -2, -3… on the stem instead."""
if rel not in used:
used.add(rel)
return rel
stem, dot, ext = rel.rpartition(".")
n = 2
while f"{stem}-{n}{dot}{ext}" in used:
n += 1
cand = f"{stem}-{n}{dot}{ext}"
used.add(cand)
return cand

def dump(rel: str, obj) -> None:
data = obj.model_dump(by_alias=True, exclude_none=True, mode="json")
for key in _EMPTY_OK:
if data.get(key) == []:
data.pop(key)
# `kind` is an enum -> its value; pydantic mode="json" already handles it.
rel = _unique(rel)
path = root / rel
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(dump_str(data), encoding="utf-8")
Expand All @@ -87,25 +105,25 @@ def dump(rel: str, obj) -> None:
written.append("mdl-project.yaml")

for sa in model.subject_areas.values():
dump(f"conceptual/subject-areas/{sa.name.lower().replace(' ', '_')}.yaml", sa)
dump(f"conceptual/subject-areas/{_slug(sa.name)}.yaml", sa)
for ce in model.conceptual_entities.values():
dump(f"conceptual/entities/{_slug(ce.name)}.yaml", ce)
for term in model.terms.values():
dump(f"conceptual/terms/{_slug(term.name)}.yaml", term)
for dom in model.domains.values():
dump(f"logical/domains/{dom.name}.yaml", dom)
dump(f"logical/domains/{_slug(dom.name)}.yaml", dom)
for cs in model.code_sets.values():
dump(f"logical/value-sets/{_slug(cs.name)}.yaml", cs)
for le in model.logical_entities.values():
dump(f"logical/entities/{le.name}.yaml", le)
dump(f"logical/entities/{_slug(le.name)}.yaml", le)
for rel in model.relationships.values():
dump(f"logical/relationships/{rel.name}.yaml", rel)
dump(f"logical/relationships/{_slug(rel.name)}.yaml", rel)
for kg in model.key_groups.values():
dump(f"logical/key-groups/{_slug(kg.name)}.yaml", kg)
for cat in model.categories.values():
dump(f"logical/categories/{_slug(cat.name)}.yaml", cat)
for pt in model.physical_tables.values():
dump(f"physical/{pt.target}/tables/{pt.name.lower()}.yaml", pt)
dump(f"physical/{_slug(pt.target)}/tables/{_slug(pt.name)}.yaml", pt)

# Prune stale object files from a PRIOR reverse into this dir: an entity that no
# longer exists (e.g. now excluded by an edited reverse.exclude) would otherwise
Expand Down Expand Up @@ -145,5 +163,18 @@ def _prune_stale(root: Path, written: set[str]) -> None:
pass


# Characters illegal in a Windows filename (< > : " / \ | ? *) plus control chars. A
# reversed/imported object name is used verbatim as a filename, and an erwin export can
# carry names like "<root>" or "Order:Line" that crash open() on Windows (OSError 22) and
# make a portable git repo impossible. Map every unsafe run to a single underscore.
_UNSAFE_FS = re.compile(r'[<>:"/\\|?*\x00-\x1f]+')


def _slug(name: str) -> str:
return name.lower().replace(" ", "_")
"""A filesystem-safe, portable slug for a filename: lowercased, spaces and any
Windows-illegal characters collapsed to underscores, and trailing dots/spaces (also
illegal on Windows) stripped. Never returns empty."""
s = _UNSAFE_FS.sub("_", name).lower().replace(" ", "_")
# Windows also forbids a name ending in a dot or space, and bare dots are confusing.
s = s.strip("._")
return s or "unnamed"
43 changes: 43 additions & 0 deletions packages/reverse/tests/test_writer_filenames.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,43 @@
"""The writer must produce portable, filesystem-safe filenames.

A reversed/imported object name is used to derive its YAML filename. An erwin export can
carry names with characters that are illegal in a Windows filename (< > : " / \\ | ? *),
e.g. a domain named "<root>", which crashes open() on Windows (OSError 22). The writer
slugs every name to a safe form, and disambiguates collisions the slugging can create.
"""

from __future__ import annotations

from mdl_core.ir import Domain, LogicalEntity, Model, ProjectConfig
from mdl_reverse.writer import _slug, write_model


def test_slug_strips_windows_illegal_chars():
assert _slug("<root>") == "root"
assert _slug("Order:Line") == "order_line"
assert _slug("Foo/Bar") == "foo_bar"
assert _slug('a"b|c?d*e') == "a_b_c_d_e"
assert _slug("trailing.") == "trailing" # Windows forbids a trailing dot
assert _slug(" ") == "unnamed" # never empty
assert _slug("Counterparty") == "counterparty" # a clean name is unchanged


def test_writer_handles_illegal_domain_name(tmp_path):
# a domain literally named "<root>" (erwin's structural node) must not crash the write
m = Model(ProjectConfig(name="m", dbt_target="duckdb_dev"))
m.add(Domain(id="01J000000000000000000DOM01", name="<root>", base_type="string"))
written = write_model(m, tmp_path)
# it landed at a safe filename, and no angle brackets reached the filesystem
assert (tmp_path / "logical" / "domains" / "root.yaml").exists()
assert any(w.endswith("root.yaml") for w in written)


def test_writer_disambiguates_filename_collisions(tmp_path):
# two entities whose names slug to the SAME filename must both survive, not overwrite
m = Model(ProjectConfig(name="m", dbt_target="duckdb_dev"))
m.add(LogicalEntity(id="01J000000000000000000ENT01", name="Order Line", realises=None))
m.add(LogicalEntity(id="01J000000000000000000ENT02", name="Order/Line", realises=None))
write_model(m, tmp_path)
files = sorted(p.name for p in (tmp_path / "logical" / "entities").glob("*.yaml"))
# both were written (one disambiguated with a -2 suffix), neither clobbered
assert files == ["order_line-2.yaml", "order_line.yaml"]
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
# project, so the install name is `modelith-dbt` (install: uv tool install
# modelith-dbt). The command stays `mdl`; the import modules stay mdl_*.
name = "modelith-dbt"
version = "0.6.8"
version = "0.6.10"
description = "Ontology-anchored, git-native data modeling tool for dbt-core teams"
readme = "README.md"
requires-python = ">=3.11"
Expand Down
24 changes: 24 additions & 0 deletions sandbox/clean-install/Dockerfile
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
# Clean-install verification: the CORE-ONLY install a user gets from
# pip install modelith-dbt (NOT 'modelith-dbt[ontology]')
#
# This is the environment our dev workspace hides: `uv run` always has every extra
# (pyoxigraph included), so a command that imports the optional RDF stack passes
# locally and crashes on a real core-only install. That is exactly the class of bug
# that shipped — `mdl import erwin` into an empty folder pulled mdl_ontology → the
# pyoxigraph backend, which a core install does not have, and died with
# ModuleNotFoundError: No module named 'pyoxigraph'.
#
# The image is a faithful clean machine: python:3.12-slim, a non-root user, pip only.
# It installs a LOCALLY BUILT wheel (mounted at /tmp/wheels) so an unreleased branch
# is tested, and it installs NO extras — the point is to prove the core CLI's logical
# flows run without them. run-clean.sh does the asserts and exits non-zero on failure,
# so it doubles as a CI gate.
FROM python:3.12-slim

RUN useradd -m -s /bin/bash user
USER user
WORKDIR /home/user

COPY --chown=user:user run-clean.sh /home/user/run-clean.sh

ENTRYPOINT ["/bin/bash", "/home/user/run-clean.sh"]
Loading
Loading