Skip to content

File operations

kurra.file

RDF file manipulation functions.

FailOnChangeError

Bases: Exception

Raised when running format, if check is set and the file would change as a result of the operation.

merge

merge(*files: Path, destination: Optional[Path] = None, output_format: str = 'longturtle') -> None

Merge RDF files into one RDF graph or dataset document.

Named graphs are preserved. Triple-only inputs are placed in DEFAULT_GRAPH_IRI when the output format supports datasets, and named graphs are flattened when it only supports triples.

Parameters:

Name Type Description Default
files Path

The RDF files to merge.

()
destination Optional[Path]

The output file path. If omitted, the merged RDF is printed.

None
output_format str

The RDFLib serialization format for the merged RDF.

'longturtle'

Raises:

Type Description
ValueError

If output_format is not a supported RDF format.

Source code in kurra/file.py
def merge(
    *files: Path,
    destination: Optional[Path] = None,
    output_format: str = "longturtle",
) -> None:
    """Merge RDF files into one RDF graph or dataset document.

    Named graphs are preserved. Triple-only inputs are placed in `DEFAULT_GRAPH_IRI` when the output format supports datasets, and named graphs are flattened when it only supports triples.

    Args:
        files: The RDF files to merge.
        destination: The output file path. If omitted, the merged RDF is printed.
        output_format: The RDFLib serialization format for the merged RDF.

    Raises:
        ValueError: If `output_format` is not a supported RDF format.
    """
    if output_format not in RDF_FILE_SUFFIXES:
        raise ValueError(
            "Unsupported output_format. It must be one of "
            f"{', '.join(RDF_FILE_SUFFIXES)}"
        )

    input_formats = [_format_for_suffix(Path(file).suffix) for file in files]
    # JSON-LD can represent either a graph or a dataset. Keep triple-only
    # JSON-LD graph-shaped, but use its dataset form when named graphs occur.
    output_is_dataset = output_format in RDF_GRAPH_AWARE_FORMATS and (
        output_format != "json-ld"
        or any(code in RDF_GRAPH_AWARE_FORMATS for code in input_formats)
    )
    merged = Dataset() if output_is_dataset else Graph()
    for file, input_format in zip(files, input_formats):
        path = Path(file)
        if input_format in RDF_GRAPH_AWARE_FORMATS:
            parsed = _parse_dataset(path, format=input_format)
            if output_is_dataset:
                for quad in parsed.quads():
                    merged.add(quad)
            else:
                for subject, predicate, obj, _ in parsed.quads():
                    merged.add((subject, predicate, obj))
        else:
            parsed = Graph().parse(source=path, format=input_format)
            if output_is_dataset:
                for triple in parsed:
                    merged.add((*triple, DEFAULT_GRAPH_IRI))
            else:
                merged += parsed

    serialized = merged.serialize(format=output_format)

    if destination is None:
        print(serialized, end="" if serialized.endswith("\n") else "\n")
    else:
        Path(destination).write_text(serialized, encoding="utf-8")

do_format

do_format(content: str, output_format: keys() = 'longturtle', input_format: str | None = None) -> Tuple[str, bool]

Reformat RDF content to a given serialization format.

Parameters:

Name Type Description Default
content str

The RDF content to reformat, as a string.

required
output_format keys()

The RDFLib serialization format to write.

'longturtle'
input_format str | None

The RDFLib format of content. Detected automatically if not given.

None

Returns:

Type Description
Tuple[str, bool]

A tuple of the reformatted content and whether it changed from the input.

Source code in kurra/file.py
def do_format(
    content: str,
    output_format: RDF_FILE_SUFFIXES.keys() = "longturtle",
    input_format: str | None = None,
) -> Tuple[str, bool]:
    """Reformat RDF content to a given serialization format.

    Args:
        content: The RDF content to reformat, as a string.
        output_format: The RDFLib serialization format to write.
        input_format: The RDFLib format of `content`. Detected automatically if not given.

    Returns:
        A tuple of the reformatted content and whether it changed from the input.
    """
    if output_format not in RDF_FILE_SUFFIXES:
        raise ValueError(
            "Unsupported output_format. It must be one of "
            f"{', '.join(RDF_FILE_SUFFIXES)}"
        )
    if output_format in ["turtle", "longturtle", "ttl", "nt", "n3"]:
        lines = content.split("\n")
        comments = []
        for i, line in enumerate(lines):
            if line.startswith("#") or line == "":
                comments.append(line)
            else:
                break

        content_no_comments = "\n".join(lines[len(comments) :])
        graph = (
            load_graph(content_no_comments)
            if input_format is None
            else Graph().parse(data=content_no_comments, format=input_format)
        )
        if comments != []:
            header = "\n".join(comments) + "\n"
        else:
            header = ""
        new_content = header + graph.serialize(format=output_format, canon=True)
    else:
        clean_content = ""
        for line in content.split("\n"):
            if not line.startswith("#") and line != "":
                clean_content += line + "\n"

        source_format = input_format or output_format
        if source_format in RDF_GRAPH_AWARE_FORMATS:
            graph = _parse_dataset(data=clean_content, format=source_format)
        else:
            graph = Graph().parse(data=clean_content, format=source_format)

        output_is_dataset = output_format in RDF_GRAPH_AWARE_FORMATS and (
            output_format != "json-ld" or source_format in RDF_GRAPH_AWARE_FORMATS
        )
        if output_is_dataset and not isinstance(graph, Dataset):
            dataset = Dataset()
            for triple in graph:
                dataset.add((*triple, DEFAULT_GRAPH_IRI))
            graph = dataset
        elif not output_is_dataset and isinstance(graph, Dataset):
            flattened = Graph()
            for subject, predicate, obj, _ in graph.quads():
                flattened.add((subject, predicate, obj))
            graph = flattened
        new_content = graph.serialize(format=output_format, canon=True)

    changed = content != new_content
    return new_content, changed

reformat

reformat(path: Path, check: bool, output_format: keys() = 'longturtle', output_filename: Path = None) -> None

Reformat one RDF file or every RDF file in a directory to a given format.

Parameters:

Name Type Description Default
path Path

The file or directory of RDF files to be formatted.

required
check bool

If True, check whether files will be changed by this command without applying the effect.

required
output_format keys()

The RDFLib serialization format to write.

'longturtle'
output_filename Path

The name of the file to write the reformatted content to. Only valid when path is a single file.

None

Raises:

Type Description
ValueError

If output_filename is given while reformatting a directory.

FailOnChangeError

If check is True and reformatting a directory would change any file.

Source code in kurra/file.py
def reformat(
    path: Path,
    check: bool,
    output_format: RDF_FILE_SUFFIXES.keys() = "longturtle",
    output_filename: Path = None,
) -> None:
    """Reformat one RDF file or every RDF file in a directory to a given format.

    Args:
        path: The file or directory of RDF files to be formatted.
        check: If True, check whether files will be changed by this command without applying the effect.
        output_format: The RDFLib serialization format to write.
        output_filename: The name of the file to write the reformatted content to. Only valid when `path` is a single file.

    Raises:
        ValueError: If `output_filename` is given while reformatting a directory.
        FailOnChangeError: If `check` is True and reformatting a directory would change any file.
    """
    path = Path(path).resolve()

    if path.is_dir():
        if output_filename is not None:
            raise ValueError(
                "You cannot specify an output filename if converting multiple files"
            )

        types = [f"**/*{ft}" for ft in RDF_SUFFIX_MAP]
        files = list(
            itertools.chain.from_iterable(path.glob(pattern) for pattern in types)
        )

        changed_files = []

        for file in files:
            try:
                changed = _format_file(
                    file,
                    check,
                    output_format=output_format,
                    output_filename=output_filename,
                )
                if changed:
                    changed_files.append(file)
            except FailOnChangeError as err:
                print(err)
                changed_files.append(file)

        if check and changed_files:
            if changed_files:
                raise FailOnChangeError(
                    f"{len(changed_files)} out of {len(files)} files will change."
                )
            else:
                print(
                    f"{len(changed_files)} out of {len(files)} files will change.",
                )
        else:
            print(
                f"{len(changed_files)} out of {len(files)} files changed.",
            )
    else:
        try:
            _format_file(
                path,
                check,
                output_format=output_format,
                output_filename=output_filename,
            )
        except FailOnChangeError as err:
            print(err)

make_dataset

make_dataset(path_str_or_graph: Union[Path, str, Graph], graph_iri: Union[str, URIRef]) -> Dataset

Wrap a graph, or an RDF file or string of triples, as a Dataset under a given graph IRI.

Parameters:

Name Type Description Default
path_str_or_graph Union[Path, str, Graph]

The RDF source to wrap as a Graph, file path, or string of triples.

required
graph_iri Union[str, URIRef]

The IRI to assign every triple's named graph.

required

Returns:

Type Description
Dataset

A Dataset containing the source's triples under graph_iri.

Source code in kurra/file.py
def make_dataset(
    path_str_or_graph: Union[Path, str, Graph], graph_iri: Union[str, URIRef]
) -> Dataset:
    """Wrap a graph, or an RDF file or string of triples, as a Dataset under a given graph IRI.

    Args:
        path_str_or_graph: The RDF source to wrap as a Graph, file path, or string of triples.
        graph_iri: The IRI to assign every triple's named graph.

    Returns:
        A Dataset containing the source's triples under `graph_iri`.
    """

    # TODO: make a Dataset from a Graph or Datatset
    # - override option to replace existing graph
    # - set default union graph
    # - set default graph
    if not isinstance(graph_iri, URIRef):
        graph_iri = URIRef(graph_iri)

    g = load_graph(path_str_or_graph)

    d = Dataset()
    for s, p, o in g:
        d.add((s, p, o, graph_iri))

    return d

hierarchy

hierarchy(path_str_or_graph: Union[Path, str, Graph], graph_iri: Optional[Union[str, URIRef]] = None, use_names: bool = False) -> None

Print the class, property, or concept hierarchy found in an RDF source.

Resources are displayed as namespace-qualified names where possible, or by their preferred label if use_names is set. Separate hierarchy roots, and separate hierarchy kinds, are divided by a blank line.

Parameters:

Name Type Description Default
path_str_or_graph Union[Path, str, Graph]

An RDF file, serialized RDF string, or RDFLib Graph.

required
graph_iri Optional[Union[str, URIRef]]

The named graph to use. Only valid for a remote URL or a .trig/.jsonld file; without it, the source is parsed as a context-less graph.

None
use_names bool

If True, select each resource's name from (in order): skos:prefLabel, dcterms:title, schema:name, or rdfs:label, falling back to the IRI if none is found.

False

Raises:

Type Description
ValueError

If graph_iri is given for a source other than a remote URL or .trig/.jsonld file, or if a cycle is detected in the hierarchy.

Source code in kurra/file.py
def hierarchy(
    path_str_or_graph: Union[Path, str, Graph],
    graph_iri: Optional[Union[str, URIRef]] = None,
    use_names: bool = False,
) -> None:
    """Print the class, property, or concept hierarchy found in an RDF source.

    Resources are displayed as namespace-qualified names where possible, or by their preferred label if `use_names` is set. Separate hierarchy roots, and separate hierarchy kinds, are divided by a blank line.

    Args:
        path_str_or_graph: An RDF file, serialized RDF string, or RDFLib Graph.
        graph_iri: The named graph to use. Only valid for a remote URL or a `.trig`/`.jsonld` file; without it, the source is parsed as a context-less graph.
        use_names: If True, select each resource's name from (in order): `skos:prefLabel`, `dcterms:title`, `schema:name`, or `rdfs:label`, falling back to the IRI if none is found.

    Raises:
        ValueError: If `graph_iri` is given for a source other than a remote URL or `.trig`/`.jsonld` file, or if a cycle is detected in the hierarchy.
    """
    is_remote = isinstance(path_str_or_graph, str) and path_str_or_graph.startswith(
        "http"
    )
    is_named_graph_file = isinstance(path_str_or_graph, Path) and (
        path_str_or_graph.suffix.lower() in {".trig", ".jsonld"}
    )

    if graph_iri is not None:
        if not (is_remote or is_named_graph_file):
            raise ValueError(
                "graph_iri is only allowed for a remote HTTP source or a "
                ".trig/.jsonld file"
            )
        if not isinstance(graph_iri, URIRef):
            graph_iri = URIRef(graph_iri)
        dataset = _parse_dataset(path_str_or_graph)
        graph = dataset.graph(graph_iri)
    elif isinstance(path_str_or_graph, Graph):
        graph = Graph()
        for prefix, namespace in path_str_or_graph.namespaces():
            graph.bind(prefix, namespace)
        for triple in path_str_or_graph:
            graph.add(triple)
    elif isinstance(path_str_or_graph, Path):
        graph = Graph().parse(path_str_or_graph)
    else:
        graph = load_graph(path_str_or_graph)

    class_types = {OWL.Class, RDFS.Class}
    property_types = {
        RDF.Property,
        URIRef(f"{RDFS}Property"),
        OWL.ObjectProperty,
        OWL.DatatypeProperty,
        OWL.AnnotationProperty,
        OWL.FunctionalProperty,
        OWL.InverseFunctionalProperty,
        OWL.SymmetricProperty,
        OWL.TransitiveProperty,
    }

    def typed_resources(types: set[URIRef]) -> set:
        return {
            subject
            for rdf_type in types
            for subject in graph.subjects(RDF.type, rdf_type)
        }

    def display_name(resource) -> str:
        if use_names:
            name_predicates = (
                SKOS.prefLabel,
                DCTERMS.title,
                URIRef("https://schema.org/name"),
                RDFS.label,
            )
            for predicate in name_predicates:
                values = sorted(graph.objects(resource, predicate), key=str)
                if values:
                    return str(values[0])
        if isinstance(resource, URIRef):
            try:
                return graph.namespace_manager.normalizeUri(resource)
            except Exception:  # RDFLib may reject an IRI it cannot compact.
                return f"<{resource}>"
        return resource.n3(graph.namespace_manager)

    def forests(
        nodes: set,
        predicates: tuple[URIRef, ...],
        inverse=(),
        root_links: tuple[URIRef, ...] = (),
        inverse_root_links: tuple[URIRef, ...] = (),
        membership_predicate: Optional[URIRef] = None,
        container_roots: set = frozenset(),
    ) -> list[str]:
        children = {node: set() for node in nodes}
        parents = {node: set() for node in nodes}

        def add_edge(parent, child) -> None:
            if parent in nodes and child in nodes:
                children[parent].add(child)
                parents[child].add(parent)

        for predicate in predicates:
            for child, parent in graph.subject_objects(predicate):
                add_edge(parent, child)
        for predicate in inverse:
            for parent, child in graph.subject_objects(predicate):
                add_edge(parent, child)
        for predicate in root_links:
            for child, parent in graph.subject_objects(predicate):
                add_edge(parent, child)
        for predicate in inverse_root_links:
            for parent, child in graph.subject_objects(predicate):
                add_edge(parent, child)

        if membership_predicate is not None:
            for child, parent in graph.subject_objects(membership_predicate):
                # inScheme expresses membership, not a direct hierarchy edge.
                # Attach only concepts that do not already have a broader
                # concept or an explicit top-concept relationship.
                if child in nodes and not parents[child]:
                    add_edge(parent, child)

        # Ontologies do not have a standard predicate linking them to every
        # declared class or property. Likewise, some SKOS sources omit
        # inScheme/top-concept links. When there is one unambiguous container,
        # place all otherwise top-level resources beneath it.
        if len(container_roots) == 1:
            container = next(iter(container_roots))
            for node in sorted(nodes - container_roots, key=display_name):
                if not parents[node]:
                    add_edge(container, node)

        visited = set()
        active = []
        active_set = set()

        def check_for_cycle(node) -> None:
            if node in active_set:
                cycle_start = active.index(node)
                cycle = active[cycle_start:] + [node]
                raise ValueError(
                    "Cycle detected in hierarchy: "
                    + " -> ".join(display_name(item) for item in cycle)
                )
            if node in visited:
                return
            active.append(node)
            active_set.add(node)
            for child in sorted(children[node], key=display_name):
                check_for_cycle(child)
            active.pop()
            active_set.remove(node)
            visited.add(node)

        for node in sorted(nodes, key=display_name):
            check_for_cycle(node)

        connected = {node for node in nodes if children[node] or parents[node]}
        roots = sorted(
            (node for node in connected if not parents[node]), key=display_name
        )
        covered = set()
        output = []

        def render(node, prefix="", connector=""):
            output.append(f"{prefix}{connector}{display_name(node)}")
            covered.add(node)
            descendants = sorted(children[node], key=display_name)
            for index, child in enumerate(descendants):
                last = index == len(descendants) - 1
                render(
                    child,
                    prefix
                    + ("    " if connector == "└── " else "│   " if connector else ""),
                    "└── " if last else "├── ",
                )

        for root in roots:
            if output:
                output.append("")
            render(root)

        # This also covers a component reached from an already-rendered node in
        # a hierarchy where a resource has more than one parent.
        for node in sorted(connected - covered, key=display_name):
            if node in covered:
                continue
            if output:
                output.append("")
            render(node)
        return output

    sections = []
    ontologies = typed_resources({OWL.Ontology})
    class_lines = forests(
        typed_resources(class_types) | ontologies,
        (RDFS.subClassOf,),
        container_roots=ontologies,
    )
    if class_lines:
        sections.append("\n".join(class_lines))
    property_lines = forests(
        typed_resources(property_types) | ontologies,
        (RDFS.subPropertyOf,),
        container_roots=ontologies,
    )
    if property_lines:
        sections.append("\n".join(property_lines))
    concepts = typed_resources({SKOS.Concept})
    concept_schemes = typed_resources({SKOS.ConceptScheme})
    concept_lines = forests(
        concepts | concept_schemes,
        (SKOS.broader,),
        (SKOS.narrower,),
        root_links=(SKOS.topConceptOf,),
        inverse_root_links=(SKOS.hasTopConcept,),
        membership_predicate=SKOS.inScheme,
        container_roots=concept_schemes,
    )
    if concept_lines:
        sections.append("\n".join(concept_lines))

    if sections:
        print("\n\n".join(sections))

export_quads

export_quads(path_str_or_dataset: Union[Path, str, Dataset], destination: Optional[Path] = None) -> bool | str

Export triples from a given Dataset, quads string in trig format, or a quads file specified by Path.

Parameters:

Name Type Description Default
path_str_or_dataset Union[Path, str, Dataset]

A Dataset, a trig file path, or a string of trig data.

required
destination Optional[Path]

The file to write the quads to. If given and the file already exists, its quads are merged in.

None

Returns:

Type Description
bool | str

True if written to destination, otherwise the serialized quads as a string.

Source code in kurra/file.py
def export_quads(
    path_str_or_dataset: Union[Path, str, Dataset], destination: Optional[Path] = None
) -> bool | str:
    """Export triples from a given Dataset, quads string in trig format, or a quads file specified by Path. 

    Args:
        path_str_or_dataset: A Dataset, a trig file path, or a string of trig data.
        destination: The file to write the quads to. If given and the file already exists, its quads are merged in.

    Returns:
        True if written to `destination`, otherwise the serialized quads as a string.
    """
    if isinstance(path_str_or_dataset, Path):
        d = _parse_dataset(path_str_or_dataset)
    elif isinstance(path_str_or_dataset, str):
        d = _parse_dataset(data=path_str_or_dataset, format="trig")
    else:  # Dataset
        d = path_str_or_dataset

    if destination is not None:
        if Path(destination).is_file():
            d2 = _parse_dataset(destination)
            d3 = d + d2
            _serialize_dataset(d3, destination=destination)
        else:
            _serialize_dataset(d, destination=destination)

        return True
    else:
        return _serialize_dataset(d)