Skip to content

SPARQL endpoints

kurra.db.sparql

SPARQL functions for remote SPARQL endpoints (not local files).

query

query(sparql_endpoint: str, q: str | Path, namespaces: dict[str, str] | None = None, http_client: Client | None = None, return_format: Literal['original'] = 'original', return_bindings_only: bool = False, user_agent: str = USER_AGENT_STRING) -> str
query(sparql_endpoint: str, q: str | Path, namespaces: dict[str, str] | None, http_client: Client | None, return_format: Literal['python'], return_bindings_only: bool = False, user_agent: str = USER_AGENT_STRING) -> Graph
query(sparql_endpoint: str, q: str | Path, namespaces: dict[str, str] | None, http_client: Client | None, return_format: Literal['dataframe'], return_bindings_only: bool = False, user_agent: str = USER_AGENT_STRING) -> DataFrame
query(sparql_endpoint: str, q: str | Path, namespaces: dict[str, str] | None = None, http_client: Client | None = None, return_format: Literal['original', 'python', 'dataframe'] = 'original', return_bindings_only: bool = False, user_agent: str = USER_AGENT_STRING) -> str | Graph | dict | DataFrame

Run a SPARQL query or update against a remote SPARQL endpoint.

Parameters:

Name Type Description Default
sparql_endpoint str

The SPARQL endpoint URL to query.

required
q str | Path

The SPARQL query or update, as a string or a path to a file containing one.

required
namespaces dict[str, str] | None

Namespace prefixes to add to q before running it.

None
http_client Client | None

An optional HTTPX client to contain credentials if needed. A new one is created if not given.

None
return_format Literal['original', 'python', 'dataframe']

"original" for the endpoint's raw response, "python" for parsed Python objects, or "dataframe" for a pandas DataFrame (SELECT/ASK only).

'original'
return_bindings_only bool

If True, return just the result bindings rather than the full SPARQL results structure.

False
user_agent str

The User-Agent header to send.

USER_AGENT_STRING

Returns:

Type Description
str | Graph | dict | DataFrame

The query result, in the requested return_format. CONSTRUCT/DESCRIBE queries always return the endpoint's raw text.

Raises:

Type Description
ValueError

If sparql_endpoint or q is not given, return_format is invalid, return_format is "dataframe" for a non-SELECT/ASK query, or pandas is not installed for "dataframe".

RuntimeError

If the endpoint responds with an error status.

Source code in kurra/db/sparql.py
def query(
    sparql_endpoint: str,
    q: str | Path,
    namespaces: dict[str, str] | None = None,
    http_client: httpx.Client | None = None,
    return_format: LiteralType["original", "python", "dataframe"] = "original",
    return_bindings_only: bool = False,
    user_agent: str = USER_AGENT_STRING,
) -> "str | Graph | dict | DataFrame":
    """Run a SPARQL query or update against a remote SPARQL endpoint.

    Args:
        sparql_endpoint: The SPARQL endpoint URL to query.
        q: The SPARQL query or update, as a string or a path to a file containing one.
        namespaces: Namespace prefixes to add to `q` before running it.
        http_client: An optional HTTPX client to contain credentials if needed. A new one is created if not given.
        return_format: `"original"` for the endpoint's raw response, `"python"` for parsed Python objects, or `"dataframe"` for a pandas DataFrame (SELECT/ASK only).
        return_bindings_only: If True, return just the result bindings rather than the full SPARQL results structure.
        user_agent: The User-Agent header to send.

    Returns:
        The query result, in the requested `return_format`. CONSTRUCT/DESCRIBE queries always return the endpoint's raw text.

    Raises:
        ValueError: If `sparql_endpoint` or `q` is not given, `return_format` is invalid, `return_format` is `"dataframe"` for a non-SELECT/ASK query, or pandas is not installed for `"dataframe"`.
        RuntimeError: If the endpoint responds with an error status.
    """
    if sparql_endpoint is None:
        raise ValueError("You must supply a sparql_endpoint")

    if q is None:
        raise ValueError("You must supply a query")

    if isinstance(q, str):
        if len(q) < 260:
            if Path(q).is_file():
                q = Path(q).read_text()

    if return_format not in ["original", "python", "dataframe"]:
        raise ValueError(
            f"return_format {return_format} must be either 'original', 'python' or 'dataframe'"
        )

    if namespaces is not None:
        q = add_namespaces_to_query_or_data(q, namespaces)

    if http_client is None:
        http_client = httpx.Client()

    headers = {}
    headers["Content-Type"] = "application/sparql-update"

    statement = statement_type_for_query(q)

    if return_format == "dataframe":
        if not is_select_or_ask_query(q, statement):
            raise ValueError(
                'Only SELECT and ASK queries can have return_format set to "dataframe"'
            )

        try:
            from pandas import DataFrame
        except ImportError:
            raise ValueError(
                'You selected the output format "dataframe" but the pandas Python package is not installed.'
            )

    if is_update_query(q, statement):
        headers["Content-Type"] = "application/sparql-update"
    else:
        headers = {"Content-Type": "application/sparql-query"}

    headers["Accept"] = sparql_statement_return_type(q, statement)
    headers["User-Agent"] = user_agent

    ssse = make_system_specific_sparql_endpoint(sparql_endpoint, q, statement)

    r = http_client.post(
        ssse,
        headers=headers,
        content=str(q),
        follow_redirects=True,
        timeout=25,
    )

    status_code = r.status_code

    # in case the endpoint doesn't allow POST
    if 400 <= status_code < 600:
        r = http_client.get(
            sparql_endpoint,
            headers=headers,
            params={"query": str(q)},
            follow_redirects=True,
            timeout=25,
        )

        status_code = r.status_code

    if status_code != 200 and status_code != 201 and status_code != 204:
        raise RuntimeError(f"ERROR {status_code}: {r.text}")

    if status_code == 204:
        return ""

    if is_construct_or_describe_query(q, statement):
        return r.text

    if return_format == "python":
        return convert_sparql_json_to_python(r, return_bindings_only)

    elif return_format == "dataframe":
        return make_sparql_dataframe(r.json())

    # original format - JSON
    return r.text