Skip to content

Utilities

kurra.utils

Utilities used by the other modules.

GspType

Bases: str, Enum

Graph Store Protocol operation types: get, put, post, or delete.

RenderFormat

Bases: str, Enum

Output formats for render_sparql_result: original, json or markdown.

guess_format_from_data

guess_format_from_data(rdf: str) -> str | None

Guess an RDF media type from a string of RDF data.

Parameters:

Name Type Description Default
rdf str

The RDF data to inspect.

required

Returns:

Type Description
str | None

The guessed media type, or None if rdf is None.

Source code in kurra/utils.py
def guess_format_from_data(rdf: str) -> str | None:
    """Guess an RDF media type from a string of RDF data.

    Args:
        rdf: The RDF data to inspect.

    Returns:
        The guessed media type, or None if `rdf` is None.
    """
    if rdf is not None:
        rdf = rdf.strip()
        if rdf.startswith("PREFIX") or rdf.startswith("@prefix"):
            return "text/turtle"
        elif rdf.startswith("{") or rdf.startswith("["):
            return "application/ld+json"
        elif rdf.startswith("<?xml") or rdf.startswith("<rdf"):
            return "application/rdf+xml"
        elif rdf.startswith("<http"):
            return "application/n-triples"
        else:
            return "application/n-triples"
    else:
        return None

load_graph

load_graph(source: Union[GraphInput, list[GraphInput], tuple[GraphInput, ...]], *additional_graph_paths_or_str: GraphInput, recursive: bool = False) -> Graph

Return an RDFLib Graph from one or more existing Graphs, pickle-cached RDF files, RDF files or directories, remote RDF URLs, or RDF data strings.

Parameters:

Name Type Description Default
source Union[GraphInput, list[GraphInput], tuple[GraphInput, ...]]

A single input, or a list/tuple of inputs, to combine into one graph.

required
*additional_graph_paths_or_str GraphInput

Further inputs, combined with source into one graph.

()
recursive bool

If True, recurse into subdirectories when source is a directory.

False

Returns:

Type Description
Graph

The combined Graph.

Raises:

Type Description
FileNotFoundError

If a given filesystem path does not exist.

Source code in kurra/utils.py
def load_graph(
    source: Union[GraphInput, list[GraphInput], tuple[GraphInput, ...]],
    *additional_graph_paths_or_str: GraphInput,
    recursive: bool = False,
) -> Graph:
    """Return an RDFLib Graph from one or more existing Graphs, pickle-cached RDF files, RDF files or directories, remote RDF URLs, or RDF data strings.

    Args:
        source: A single input, or a list/tuple of inputs, to combine into one graph.
        *additional_graph_paths_or_str: Further inputs, combined with `source` into one graph.
        recursive: If True, recurse into subdirectories when `source` is a directory.

    Returns:
        The combined Graph.

    Raises:
        FileNotFoundError: If a given filesystem path does not exist.
    """
    # Preserve the former ``load_graph(path, recursive)`` positional call form.
    if len(additional_graph_paths_or_str) == 1 and isinstance(
        additional_graph_paths_or_str[0], bool
    ):
        recursive = additional_graph_paths_or_str[0]
        additional_graph_paths_or_str = ()

    if isinstance(source, (list, tuple)):
        graph_inputs = (*source, *additional_graph_paths_or_str)
    else:
        graph_inputs = (source, *additional_graph_paths_or_str)

    if not graph_inputs:
        return Graph()

    if len(graph_inputs) > 1:
        graph = Graph()
        for graph_input in graph_inputs:
            graph += load_graph(graph_input, recursive=recursive)
        return graph

    source = graph_inputs[0]

    # Pre-existing Graph
    if isinstance(source, Graph):
        return source

    # Serialized RDF file or dir of files, optionally using a sibling pickle cache
    if isinstance(source, Path):
        if source.is_file():
            pkl_path = source.with_suffix(".pkl")
            if pkl_path.is_file():
                with pkl_path.open("rb") as pickle_file:
                    return pickle.load(pickle_file)
            if source.suffix.lower() == ".trig":
                return _parse_dataset(source)
            return _parse_graph(source)
        elif source.is_dir():
            g = Graph()
            if recursive:
                gl = source.rglob("*.ttl")
            else:
                gl = source.glob("*.ttl")
            for f in gl:
                if f.is_file():
                    g.parse(f)
            return g
        raise FileNotFoundError(f"Graph path does not exist: {source}")

    # A remote file via HTTP
    elif isinstance(source, str) and source.startswith("http"):
        return _parse_graph(source)

    # RDF data in a string
    else:
        return _parse_graph(
            data=source,
            format=guess_format_from_data(source),
        )

render_sparql_result

render_sparql_result(r: dict | str | Graph, rf: RenderFormat = RenderFormat.markdown) -> str

Render a SPARQL result as plain JSON, Markdown, or its original form.

Parameters:

Name Type Description Default
r dict | str | Graph

The SPARQL result to render as a dict, a JSON string, or a Graph (for CONSTRUCT/DESCRIBE).

required
rf RenderFormat

The format to render as.

markdown

Returns:

Type Description
str

The rendered result.

Source code in kurra/utils.py
def render_sparql_result(
    r: dict | str | Graph, rf: RenderFormat = RenderFormat.markdown
) -> str:
    """Render a SPARQL result as plain JSON, Markdown, or its original form.

    Args:
        r: The SPARQL result to render as a dict, a JSON string, or a Graph (for CONSTRUCT/DESCRIBE).
        rf: The format to render as.

    Returns:
        The rendered result.
    """
    if rf == RenderFormat.original:
        return r

    elif rf == RenderFormat.json:
        if isinstance(r, dict):
            return json.dumps(r, indent=4)
        elif isinstance(r, str):
            return json.dumps(json.loads(r), indent=4)
        elif isinstance(r, Graph):
            return r.serialize(format="json-ld", indent=4)

    elif rf == RenderFormat.markdown:
        if isinstance(r, Graph):  # CONSTRUCT: RDF GRaph
            output = "```turtle\n" + r.serialize(format="longturtle") + "```\n"
        else:  # SELECT or ASK: Python dict or JSON

            def render_sparql_value(v: dict) -> str:
                # TODO: handle v["datatype"]
                if v is None:
                    return ""
                elif isinstance(v, URIRef) or isinstance(v, str):
                    return f"[{v.split('/')[-1].split('#')[-1]}]({v})"
                elif isinstance(v, Literal):
                    return v
                elif isinstance(v, BNode):
                    return f"BN: {v:>6}"
                elif v["type"] == "uri":
                    return f"[{v['value'].split('/')[-1].split('#')[-1]}]({v['value']})"
                elif v["type"] == "literal":
                    return v["value"]
                elif v["type"] == "bnode":
                    return f"BN: {v['value']:>6}"

            if isinstance(r, str):
                r = json.loads(r)

            output = ""
            header = ["", ""]
            body = []

            if r.get("head") is not None:
                # SELECT
                if r["head"].get("vars") is not None:
                    for col in r["head"]["vars"]:
                        header[0] += f"{col} | "
                        header[1] += f"--- | "
                    output = (
                        "| " + header[0].strip() + "\n| " + header[1].strip() + "\n"
                    )

            if r.get("results"):
                if r["results"].get("bindings"):
                    for row in r["results"]["bindings"]:
                        row_cols = []
                        for k in r["head"]["vars"]:
                            v = row.get(k)
                            if v is not None:
                                # ignore the k
                                row_cols.append(render_sparql_value(v))
                            else:
                                row_cols.append("")
                        body.append(" | ".join(row_cols))

                output += "\n| ".join(body) + " |\n"

            if r.get("boolean") is not None:
                output = str(bool(r.get("boolean")))

        return output

make_httpx_client

make_httpx_client(sparql_username: str | None = None, sparql_password: str | None = None, timeout: int = 60) -> httpx.Client

Create an HTTPX client, with Basic Auth if a username and password are given.

Parameters:

Name Type Description Default
sparql_username str | None

The username for Basic Auth.

None
sparql_password str | None

The password for Basic Auth.

None
timeout int

The client's request timeout, in seconds.

60

Returns:

Type Description
Client

A configured HTTPX Client.

Source code in kurra/utils.py
def make_httpx_client(
    sparql_username: str | None = None,
    sparql_password: str | None = None,
    timeout: int = 60,
) -> httpx.Client:
    """Create an HTTPX client, with Basic Auth if a username and password are given.

    Args:
        sparql_username: The username for Basic Auth.
        sparql_password: The password for Basic Auth.
        timeout: The client's request timeout, in seconds.

    Returns:
        A configured HTTPX Client.
    """
    auth = None
    if sparql_username:
        if sparql_password:
            auth = httpx.BasicAuth(sparql_username, sparql_password)
    return httpx.Client(auth=auth, timeout=timeout)

convert_sparql_json_to_python

convert_sparql_json_to_python(j: Union[str, bytes, Response], return_bindings_only: bool = False) -> dict

Convert a SPARQL JSON results response into native Python types.

Parameters:

Name Type Description Default
j Union[str, bytes, Response]

The SPARQL JSON results, as a string, bytes, or an HTTPX Response.

required
return_bindings_only bool

If True, return just the result bindings (for SELECT) or a bool (for ASK), rather than the full SPARQL results structure.

False

Returns:

Type Description
dict

The converted result.

Source code in kurra/utils.py
def convert_sparql_json_to_python(
    j: Union[str, bytes, httpx.Response], return_bindings_only: bool = False
) -> dict:
    """Convert a SPARQL JSON results response into native Python types.

    Args:
        j: The SPARQL JSON results, as a string, bytes, or an HTTPX Response.
        return_bindings_only: If True, return just the result bindings (for SELECT) or a bool (for ASK), rather than the full SPARQL results structure.

    Returns:
        The converted result.
    """
    if isinstance(j, str):
        r = json.loads(j)
    elif isinstance(j, bytes):
        r = json.loads(j.decode())
    elif isinstance(j, httpx.Response):
        r = j.json()

    if r.get("results") is not None:  # SELECT
        for row in r["results"]["bindings"]:
            for k, v in row.items():
                if v["type"] == "literal":
                    if v.get("datatype") is not None:
                        row[k] = Literal(v["value"], datatype=v["datatype"]).toPython()
                    else:
                        row[k] = Literal(v["value"]).toPython()
                elif v["type"] == "uri":
                    row[k] = v["value"]
        if return_bindings_only:
            r = r["results"]["bindings"]
        return r
    elif r.get("boolean") is not None:  # ASK
        if return_bindings_only:
            return bool(r["boolean"])
        else:
            return r
    else:
        return r

sparql_statement_return_type

sparql_statement_return_type(query: str, statement: SparqlStatementType | None = None) -> str

Get the media type a SPARQL endpoint should return for a query.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
str

"text/turtle" for CONSTRUCT/DESCRIBE, otherwise "application/sparql-results+json".

Source code in kurra/utils.py
def sparql_statement_return_type(
    query: str, statement: SparqlStatementType | None = None
) -> str:
    """Get the media type a SPARQL endpoint should return for a query.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        `"text/turtle"` for CONSTRUCT/DESCRIBE, otherwise `"application/sparql-results+json"`.
    """
    statement = _ensure_statement_type(query, statement)
    if is_construct_or_describe_query(query, statement):
        return "text/turtle"
    return "application/sparql-results+json"

statement_type_for_query

statement_type_for_query(query: str) -> SparqlStatementType

Get the statement type of a SPARQL query/update string.

Parameters:

Name Type Description Default
query str

The SPARQL query or update.

required

Returns:

Type Description
SparqlStatementType

The statement type.

Source code in kurra/utils.py
def statement_type_for_query(query: str) -> SparqlStatementType:
    """Get the statement type of a SPARQL query/update string.

    Args:
        query: The SPARQL query or update.

    Returns:
        The statement type.
    """
    return statement_type_from_string(query)

is_construct_query

is_construct_query(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL CONSTRUCT query.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is a CONSTRUCT query, False otherwise.

Source code in kurra/utils.py
def is_construct_query(
    query: str, statement: SparqlStatementType | None = None
) -> bool:
    """Check whether a given query is a SPARQL CONSTRUCT query.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is a CONSTRUCT query, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return (
        statement.type == SparqlType.QUERY
        and statement.subtype == QuerySubType.CONSTRUCT
    )

is_describe_query

is_describe_query(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL DESCRIBE query.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is a DESCRIBE query, False otherwise.

Source code in kurra/utils.py
def is_describe_query(query: str, statement: SparqlStatementType | None = None) -> bool:
    """Check whether a given query is a SPARQL DESCRIBE query.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is a DESCRIBE query, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return (
        statement.type == SparqlType.QUERY
        and statement.subtype == QuerySubType.DESCRIBE
    )

is_select_query

is_select_query(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL SELECT query.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is a SELECT query, False otherwise.

Source code in kurra/utils.py
def is_select_query(query: str, statement: SparqlStatementType | None = None) -> bool:
    """Check whether a given query is a SPARQL SELECT query.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is a SELECT query, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return (
        statement.type == SparqlType.QUERY and statement.subtype == QuerySubType.SELECT
    )

is_ask_query

is_ask_query(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL ASK query.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is an ASK query, False otherwise.

Source code in kurra/utils.py
def is_ask_query(query: str, statement: SparqlStatementType | None = None) -> bool:
    """Check whether a given query is a SPARQL ASK query.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is an ASK query, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return statement.type == SparqlType.QUERY and statement.subtype == QuerySubType.ASK

is_construct_or_describe_query

is_construct_or_describe_query(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL CONSTRUCT or DESCRIBE query.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is a CONSTRUCT or DESCRIBE query, False otherwise.

Source code in kurra/utils.py
def is_construct_or_describe_query(
    query: str, statement: SparqlStatementType | None = None
) -> bool:
    """Check whether a given query is a SPARQL CONSTRUCT or DESCRIBE query.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is a CONSTRUCT or DESCRIBE query, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return statement.type == SparqlType.QUERY and statement.subtype in {
        QuerySubType.CONSTRUCT,
        QuerySubType.DESCRIBE,
    }

is_select_or_ask_query

is_select_or_ask_query(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL SELECT or ASK query.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is a SELECT or ASK query, False otherwise.

Source code in kurra/utils.py
def is_select_or_ask_query(
    query: str, statement: SparqlStatementType | None = None
) -> bool:
    """Check whether a given query is a SPARQL SELECT or ASK query.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is a SELECT or ASK query, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return statement.type == SparqlType.QUERY and statement.subtype in {
        QuerySubType.SELECT,
        QuerySubType.ASK,
    }

is_update_query

is_update_query(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL update.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is an update, False otherwise.

Source code in kurra/utils.py
def is_update_query(query: str, statement: SparqlStatementType | None = None) -> bool:
    """Check whether a given query is a SPARQL update.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is an update, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return statement.type == SparqlType.UPDATE

is_drop_update

is_drop_update(query: str, statement: SparqlStatementType | None = None) -> bool

Check whether a given query is a SPARQL DROP update.

Parameters:

Name Type Description Default
query str

The SPARQL query.

required
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None

Returns:

Type Description
bool

True if query is a DROP update, False otherwise.

Source code in kurra/utils.py
def is_drop_update(query: str, statement: SparqlStatementType | None = None) -> bool:
    """Check whether a given query is a SPARQL DROP update.

    Args:
        query: The SPARQL query.
        statement: The query's parsed statement type, if already known.

    Returns:
        True if `query` is a DROP update, False otherwise.
    """
    statement = _ensure_statement_type(query, statement)
    return (
        statement.type == SparqlType.UPDATE and statement.subtype == UpdateSubType.DROP
    )

make_sparql_dataframe

make_sparql_dataframe(sparql_result: dict) -> DataFrame

Convert a parsed SPARQL SELECT or ASK result into a pandas DataFrame.

Parameters:

Name Type Description Default
sparql_result dict

The parsed SPARQL JSON result.

required

Returns:

Type Description
DataFrame

A DataFrame of the result's bindings (SELECT) or its boolean value (ASK).

Raises:

Type Description
ValueError

If pandas is not installed.

Source code in kurra/utils.py
def make_sparql_dataframe(sparql_result: dict) -> "DataFrame":
    """Convert a parsed SPARQL SELECT or ASK result into a pandas DataFrame.

    Args:
        sparql_result: The parsed SPARQL JSON result.

    Returns:
        A DataFrame of the result's bindings (SELECT) or its boolean value (ASK).

    Raises:
        ValueError: If pandas is not installed.
    """
    try:
        from pandas import DataFrame
    except ImportError:
        raise ValueError(
            'You selected the output format "dataframe" but the pandas Python package is not installed.'
        )

    if sparql_result.get("results") is not None:  # SELECT
        df = DataFrame(columns=sparql_result["head"]["vars"])
        for i, row in enumerate(sparql_result["results"]["bindings"]):
            new_row = {}
            for k, v in row.items():
                if v["type"] == "literal":
                    if v.get("datatype") is not None:
                        new_row[k] = Literal(
                            v["value"], datatype=v["datatype"]
                        ).toPython()
                    else:
                        new_row[k] = Literal(v["value"]).toPython()
                else:
                    new_row[k] = v["value"]
            df.loc[i] = new_row
        return df
    else:  # ASK
        df = DataFrame(columns=["boolean"])
        df.loc[0] = sparql_result["boolean"]

    return df

add_namespaces_to_query_or_data

add_namespaces_to_query_or_data(q: str, namespaces: dict) -> str

Prepend PREFIX declarations to a SPARQL query or RDF data string.

Parameters:

Name Type Description Default
q str

The SPARQL query or RDF data to prepend to.

required
namespaces dict

Namespace prefixes to declare, keyed by prefix.

required

Returns:

Type Description
str

q, with the PREFIX declarations prepended.

Source code in kurra/utils.py
def add_namespaces_to_query_or_data(q: str, namespaces: dict) -> str:
    """Prepend PREFIX declarations to a SPARQL query or RDF data string.

    Args:
        q: The SPARQL query or RDF data to prepend to.
        namespaces: Namespace prefixes to declare, keyed by prefix.

    Returns:
        `q`, with the PREFIX declarations prepended.
    """
    preamble = ""
    for k, v in namespaces.items():
        preamble += f"PREFIX {k}: <{v}>\n"
    preamble += "\n"
    return preamble + q

get_system_graph

get_system_graph(system_graph_source: str | Path | Dataset | Graph = None, http_client: Client | None = None) -> Graph | int

Load Olis's System Graph from a file, Graph, Dataset, or SPARQL endpoint.

Parameters:

Name Type Description Default
system_graph_source str | Path | Dataset | Graph

The source to load from as an RDF file path, Graph, Dataset, SPARQL endpoint URL, or None for an empty System Graph.

None
http_client Client | None

An optional HTTPX client to contain credentials if needed to access a SPARQL endpoint. A new one is created if not given.

None

Returns:

Type Description
Graph | int

The System Graph, or the HTTP status code if fetching from a remote endpoint failed.

Raises:

Type Description
ValueError

If system_graph_source is a Path that does not exist, or is of an unsupported type.

Source code in kurra/utils.py
def get_system_graph(
    system_graph_source: str | Path | Dataset | Graph = None,
    http_client: httpx.Client | None = None,
) -> Graph | int:
    """Load Olis's System Graph from a file, Graph, Dataset, or SPARQL endpoint.

    Args:
        system_graph_source: The source to load from as an RDF file path, Graph, Dataset, SPARQL endpoint URL, or None for an empty System Graph.
        http_client: An optional HTTPX client to contain credentials if needed to access a SPARQL endpoint. A new one is created if not given.

    Returns:
        The System Graph, or the HTTP status code if fetching from a remote endpoint failed.

    Raises:
        ValueError: If `system_graph_source` is a Path that does not exist, or is of an unsupported type.
    """
    system_graph = Graph(identifier=SYSTEM_GRAPH_IRI)
    system_graph.bind("olis", OLIS)
    if system_graph_source is None:
        # no incoming System Graph
        pass
    elif isinstance(system_graph_source, Path):
        # we have a Graph or Dataset file, so read it
        if not system_graph_source.is_file():
            raise ValueError(
                f"system_graph_source must be an existing RDF file. Value supplied was {system_graph_source}"
            )

        if system_graph_source.suffix == ".trig":
            system_graph += _parse_dataset(system_graph_source, format="trig").graph(
                SYSTEM_GRAPH_IRI
            )
        else:
            system_graph += load_graph(system_graph_source)
    elif isinstance(system_graph_source, Graph):
        # we have a Graph, so assume it's a System Graph and load it
        system_graph += system_graph_source
    elif isinstance(system_graph_source, Dataset):
        # we have a Dataset object, so load its system Graph
        system_graph += system_graph_source.graph(SYSTEM_GRAPH_IRI)
    elif system_graph_source and system_graph_source.startswith("http"):
        # we have a remote SPARQL Endpoint, so read the System Graph
        # this is simplified GSP get()
        close_http_client = False
        if http_client is None:
            http_client = httpx.Client()
            close_http_client = True

        r = http_client.get(
            str(system_graph_source),
            params={"graph": SYSTEM_GRAPH_IRI},
            headers={"Accept": "text/turtle"},
        )

        if close_http_client:
            http_client.close()

        if r.is_success:
            system_graph += Graph().parse(data=r.text, format="turtle")
        else:
            return r.status_code
    elif system_graph_source and not system_graph_source.startswith("http"):
        system_graph += load_graph(system_graph_source)
    else:
        raise ValueError(
            "The parameter system_graph_source must be either None, a Path to an RDF Graph or Dataset serialised "
            "in Turtle or Trig, an RDFLib Graph object assumed to be a System Graph, an RDFLib Dataset object containing"
            "a System Graph or a string URL for a SPARQL Endpoint."
        )

    return system_graph

put_system_graph

put_system_graph(system_graph: Graph, system_graph_source: str | Path | Dataset | Graph | None = None, http_client: Client | None = None) -> Graph | int | None

Write a System Graph back to its file, Dataset, or SPARQL endpoint source.

Parameters:

Name Type Description Default
system_graph Graph

The System Graph to write.

required
system_graph_source str | Path | Dataset | Graph | None

An RDF file path, a Dataset, a SPARQL endpoint URL, or None to return system_graph unwritten.

None
http_client Client | None

An optional HTTPX client to contain credentials if needed to access a SPARQL endpoint. A new one is created if not given.

None

Returns:

Type Description
Graph | int | None

The HTTP status code if writing to a remote endpoint failed, system_graph if system_graph_source was None or a local (non-HTTP) string, otherwise None.

Source code in kurra/utils.py
def put_system_graph(
    system_graph: Graph,
    system_graph_source: str | Path | Dataset | Graph | None = None,
    http_client: httpx.Client | None = None,
) -> Graph | int | None:
    """Write a System Graph back to its file, Dataset, or SPARQL endpoint source.

    Args:
        system_graph: The System Graph to write.
        system_graph_source: An RDF file path, a Dataset, a SPARQL endpoint URL, or None to return `system_graph` unwritten.
        http_client: An optional HTTPX client to contain credentials if needed to access a SPARQL endpoint. A new one is created if not given.

    Returns:
        The HTTP status code if writing to a remote endpoint failed, `system_graph` if `system_graph_source` was None or a local (non-HTTP) string, otherwise None.
    """
    if system_graph_source is None:
        return system_graph
    elif isinstance(system_graph_source, Path):
        if system_graph_source.suffix == ".trig":
            # TODO: deduplicate the Dataset parse in get_system_graph
            d = _parse_dataset(system_graph_source, format="trig")
            d.remove_graph(SYSTEM_GRAPH_IRI)
            d.add_graph(system_graph)
            _serialize_dataset(d, destination=system_graph_source)
        else:
            system_graph.serialize(destination=system_graph_source, format="longturtle")

        return None
    elif isinstance(system_graph_source, Graph):
        system_graph_source = system_graph
        return None
    elif isinstance(system_graph_source, Dataset):
        system_graph_source.remove_graph(SYSTEM_GRAPH_IRI)
        system_graph_source.add_graph(system_graph)
        return None
    elif system_graph_source and system_graph_source.startswith("http"):
        # this is simplified GSP put()
        close_http_client = False
        if http_client is None:
            http_client = httpx.Client()
            close_http_client = True

        r = http_client.put(
            system_graph_source,
            params={"graph": SYSTEM_GRAPH_IRI},
            headers={"Content-Type": "text/turtle"},
            content=system_graph.serialize(format="text/turtle"),
        )

        if close_http_client:
            http_client.close()

        if r.is_success:
            return None
        else:
            return r.status_code
    elif system_graph_source and not system_graph_source.startswith("http"):
        return system_graph
    else:
        return None

make_system_specific_sparql_endpoint

make_system_specific_sparql_endpoint(sparql_endpoint: str, q: str | None = None, statement: SparqlStatementType | None = None, gsp_query_type: GspType | None = None) -> str

Adjust a SPARQL endpoint URL to match a specific triplestore's conventions.

For example, GraphDB requires /statements appended to its base SPARQL endpoint for updates.

Parameters:

Name Type Description Default
sparql_endpoint str

The base SPARQL endpoint URL.

required
q str | None

The SPARQL query or update, used to detect GraphDB update endpoints.

None
statement SparqlStatementType | None

The query's parsed statement type, if already known.

None
gsp_query_type GspType | None

The Graph Store Protocol operation type, if this is a GSP request.

None

Returns:

Type Description
str

The adjusted endpoint URL, or sparql_endpoint unchanged if no adjustment applies.

Source code in kurra/utils.py
def make_system_specific_sparql_endpoint(
    sparql_endpoint: str,
    q: str | None = None,
    statement: SparqlStatementType | None = None,
    gsp_query_type: GspType | None = None,
) -> str:
    """Adjust a SPARQL endpoint URL to match a specific triplestore's conventions.

    For example, GraphDB requires `/statements` appended to its base SPARQL endpoint for updates.

    Args:
        sparql_endpoint: The base SPARQL endpoint URL.
        q: The SPARQL query or update, used to detect GraphDB update endpoints.
        statement: The query's parsed statement type, if already known.
        gsp_query_type: The Graph Store Protocol operation type, if this is a GSP request.

    Returns:
        The adjusted endpoint URL, or `sparql_endpoint` unchanged if no adjustment applies.
    """

    # GraphDB SPARQL
    if q is not None and statement is not None:
        # GraphDB: Update
        if (
            "/repositories/" in sparql_endpoint
            and is_update_query(q, statement)
            and not sparql_endpoint.endswith("/statements")
        ):
            return sparql_endpoint + "/statements"

    # GraphDB GSP
    if gsp_query_type is not None:
        if "/repositories/" in sparql_endpoint:
            return sparql_endpoint + "/rdf-graphs/service"

    return sparql_endpoint

iter_iris

iter_iris(graph: Graph) -> Iterable[URIRef]

Iterate over every IRI referenced in a graph's triples.

Parameters:

Name Type Description Default
graph Graph

The graph to scan.

required

Returns:

Type Description
Iterable[URIRef]

Each IRI found.

Source code in kurra/utils.py
def iter_iris(graph: Graph) -> Iterable[URIRef]:
    """Iterate over every IRI referenced in a graph's triples.

    Args:
        graph: The graph to scan.

    Returns:
        Each IRI found.
    """
    for triple in graph:
        for node in triple:
            if isinstance(node, URIRef):
                yield node

build_values_clause

build_values_clause(values: dict[str, Iterable[Node]]) -> str

Build a SPARQL VALUES clause for the given values.

Parameters:

Name Type Description Default
values dict[str, Iterable[Node]]

A dictionary where keys are variable names and values are iterables of RDF nodes.

required

Returns:

Type Description
str

A string containing the SPARQL VALUES clause.

Source code in kurra/utils.py
def build_values_clause(values: dict[str, Iterable[Node]]) -> str:
    """Build a SPARQL VALUES clause for the given values.

    Args:
        values: A dictionary where keys are variable names and values are iterables of RDF nodes.

    Returns:
        A string containing the SPARQL VALUES clause.
    """
    var_names = " ".join([f"?{k}" for k in values.keys()])
    lines = [
        f"VALUES ({var_names}) {{",
    ]
    for row in zip(*values.values()):
        row_values = " ".join(
            [f"<{v}>" if isinstance(v, URIRef) else f'"{v}"' for v in row]
        )
        lines.append(f"  ({row_values})")
    lines.append("}")
    return "\n".join(lines)

is_class

is_class(graph: Graph, iri: URIRef) -> bool

Check whether the given IRI is a class in the provided RDF graph.

Parameters:

Name Type Description Default
graph Graph

An RDFLib Graph object.

required
iri URIRef

The IRI to check.

required

Returns:

Type Description
bool

True if the IRI is a class, False otherwise.

Source code in kurra/utils.py
def is_class(graph: Graph, iri: URIRef) -> bool:
    """Check whether the given IRI is a class in the provided RDF graph.

    Args:
        graph: An RDFLib Graph object.
        iri: The IRI to check.

    Returns:
        True if the IRI is a class, False otherwise.
    """
    # Could later improve this using SPARQL to check for subclasses of rdfs:class
    return (iri, RDF.type, OWL.Class) in graph or (iri, RDF.type, RDFS.Class) in graph