758 lines
27 KiB
Python
758 lines
27 KiB
Python
import json
|
|
import logging
|
|
import re
|
|
from collections import defaultdict
|
|
from collections.abc import Iterable, Mapping
|
|
from contextvars import ContextVar
|
|
from typing import Any, Union
|
|
|
|
from pydantic import GetCoreSchemaHandler
|
|
from pydantic_core import core_schema
|
|
from pyld import jsonld
|
|
from rdflib import Graph, Literal, Namespace, URIRef
|
|
from rdflib.namespace import NamespaceManager
|
|
|
|
from ontocast.onto.constants import COMMON_PREFIXES
|
|
from ontocast.util import render_text_hash
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
PREFIX_PATTERN = re.compile(r"@prefix\s+(\w+):\s+<[^>]+>\s+\.")
|
|
# Pattern to match prefix usage: prefix:something (not in @prefix declarations)
|
|
PREFIX_USAGE_PATTERN = re.compile(r"\b([a-zA-Z_][a-zA-Z0-9_]*):[^\s]")
|
|
|
|
# Context variable to store known prefixes during parsing
|
|
_known_prefixes_context: ContextVar[dict[str, str] | None] = ContextVar[
|
|
dict[str, str] | None
|
|
]("known_prefixes", default=None)
|
|
|
|
|
|
class RDFGraph(Graph):
|
|
"""Subclass of rdflib.Graph with Pydantic schema support.
|
|
|
|
This class extends rdflib.Graph to provide serialization and deserialization
|
|
capabilities for Pydantic models, with special handling for Turtle format.
|
|
"""
|
|
|
|
@classmethod
|
|
def __get_pydantic_core_schema__(cls, _source_type, handler: GetCoreSchemaHandler):
|
|
"""Get the Pydantic core schema for this class.
|
|
|
|
Args:
|
|
_source_type: The source type.
|
|
handler: The core schema handler.
|
|
|
|
Returns:
|
|
A union schema that handles both Graph instances and string conversion.
|
|
Supports both Turtle and JSON-LD string formats.
|
|
"""
|
|
return core_schema.union_schema(
|
|
[
|
|
core_schema.is_instance_schema(cls),
|
|
core_schema.chain_schema(
|
|
[
|
|
core_schema.str_schema(),
|
|
core_schema.no_info_plain_validator_function(cls._from_str),
|
|
]
|
|
),
|
|
],
|
|
serialization=core_schema.plain_serializer_function_ser_schema(
|
|
cls._to_turtle_str,
|
|
info_arg=False,
|
|
return_schema=core_schema.str_schema(),
|
|
),
|
|
)
|
|
|
|
def __add__(self, other: Union["RDFGraph", Graph, Iterable]) -> "RDFGraph":
|
|
"""Addition operator for RDFGraph instances.
|
|
|
|
Merges the RDF graphs while maintaining the RDFGraph type.
|
|
|
|
Args:
|
|
other: The graph to add to this one.
|
|
|
|
Returns:
|
|
RDFGraph: A new RDFGraph containing the merged triples.
|
|
"""
|
|
# Create a new RDFGraph instance
|
|
result = RDFGraph()
|
|
|
|
# Copy all triples from both graphs
|
|
for triple in self:
|
|
result.add(triple)
|
|
for triple in other:
|
|
result.add(triple)
|
|
|
|
# Copy namespace bindings from self
|
|
for prefix, uri in self.namespaces():
|
|
result.bind(prefix, uri)
|
|
|
|
# Copy namespace bindings from other if it's a Graph
|
|
if isinstance(other, Graph):
|
|
for prefix, uri in other.namespaces():
|
|
result.bind(prefix, uri)
|
|
|
|
return result
|
|
|
|
def __iadd__(self, other: Union["RDFGraph", Graph, Iterable]) -> "RDFGraph":
|
|
"""In-place addition operator for RDFGraph instances.
|
|
|
|
Merges the RDF graphs while maintaining the RDFGraph type and binding prefixes.
|
|
|
|
Args:
|
|
other: The graph to add to this one.
|
|
|
|
Returns:
|
|
RDFGraph: self after modification.
|
|
"""
|
|
# Use __add__ to get the merged result with proper prefix binding
|
|
result = self.__add__(other)
|
|
|
|
# Clear current graph and copy the result
|
|
self.remove((None, None, None)) # Remove all triples
|
|
|
|
# Copy all triples from result
|
|
for triple in result:
|
|
self.add(triple)
|
|
|
|
# Copy namespace bindings from result
|
|
for prefix, uri in result.namespaces():
|
|
self.bind(prefix, uri)
|
|
|
|
return self
|
|
|
|
def copy(self) -> "RDFGraph":
|
|
"""Create a copy of this RDFGraph.
|
|
|
|
Returns:
|
|
RDFGraph: A new RDFGraph instance with all triples and namespace bindings copied.
|
|
"""
|
|
result = RDFGraph()
|
|
|
|
# Copy all triples
|
|
for triple in self:
|
|
result.add(triple)
|
|
|
|
# Copy namespace bindings
|
|
for prefix, uri in self.namespaces():
|
|
result.bind(prefix, uri)
|
|
|
|
return result
|
|
|
|
@staticmethod
|
|
def _ensure_prefixes(turtle_str: str) -> str:
|
|
"""Ensure all common prefixes and used custom prefixes are declared in the Turtle string.
|
|
|
|
This method:
|
|
1. Adds missing common prefixes (rdf, rdfs, owl, etc.)
|
|
2. Detects prefixes that are used but not declared
|
|
3. Adds declarations for used prefixes if they're available in the context
|
|
|
|
Args:
|
|
turtle_str: The input Turtle string.
|
|
|
|
Returns:
|
|
str: The Turtle string with all necessary prefixes declared.
|
|
"""
|
|
declared_prefixes = set(
|
|
match.group(1) for match in PREFIX_PATTERN.finditer(turtle_str)
|
|
)
|
|
|
|
# Add missing common prefixes
|
|
missing_common = {
|
|
prefix: uri
|
|
for prefix, uri in COMMON_PREFIXES.items()
|
|
if prefix not in declared_prefixes
|
|
}
|
|
|
|
# Detect prefixes that are used but not declared
|
|
used_prefixes = set()
|
|
for match in PREFIX_USAGE_PATTERN.finditer(turtle_str):
|
|
prefix = match.group(1)
|
|
# Skip if already declared or is a common prefix we're about to add
|
|
if prefix not in declared_prefixes and prefix not in missing_common:
|
|
used_prefixes.add(prefix)
|
|
|
|
# Get known prefixes from context (set by caller)
|
|
known_prefixes = _known_prefixes_context.get()
|
|
missing_custom = {}
|
|
if known_prefixes and used_prefixes:
|
|
for prefix in used_prefixes:
|
|
if prefix in known_prefixes:
|
|
namespace_uri = known_prefixes[prefix]
|
|
# Format as Turtle prefix declaration
|
|
if not namespace_uri.startswith("<"):
|
|
namespace_uri = f"<{namespace_uri}>"
|
|
missing_custom[prefix] = namespace_uri
|
|
|
|
all_missing = {**missing_common, **missing_custom}
|
|
|
|
if not all_missing:
|
|
return turtle_str
|
|
|
|
prefix_block = (
|
|
"\n".join(
|
|
f"@prefix {prefix}: {uri} ." for prefix, uri in all_missing.items()
|
|
)
|
|
+ "\n\n"
|
|
)
|
|
|
|
return prefix_block + turtle_str
|
|
|
|
@staticmethod
|
|
def _is_jsonld_str(s: str) -> bool:
|
|
"""Check if a string appears to be JSON-LD format.
|
|
|
|
Args:
|
|
s: The string to check.
|
|
|
|
Returns:
|
|
bool: True if the string appears to be JSON-LD.
|
|
"""
|
|
s = s.strip()
|
|
if not (s.startswith("{") or s.startswith("[")):
|
|
return False
|
|
try:
|
|
# Try to parse as JSON
|
|
data = json.loads(s)
|
|
# Check if it's a dict/object with @context or @id, or an array containing such objects
|
|
if isinstance(data, dict):
|
|
return "@context" in data or "@id" in data
|
|
elif isinstance(data, list):
|
|
return any(
|
|
isinstance(item, dict) and ("@context" in item or "@id" in item)
|
|
for item in data
|
|
)
|
|
return False
|
|
except (json.JSONDecodeError, ValueError):
|
|
return False
|
|
|
|
@classmethod
|
|
def _from_str(cls, data_str: str) -> "RDFGraph":
|
|
"""Create an RDFGraph instance from a string (Turtle or JSON-LD).
|
|
|
|
Automatically detects the format and parses accordingly.
|
|
|
|
Args:
|
|
data_str: The input string in Turtle or JSON-LD format.
|
|
|
|
Returns:
|
|
RDFGraph: A new RDFGraph instance.
|
|
"""
|
|
if cls._is_jsonld_str(data_str):
|
|
return cls._from_jsonld_str(data_str)
|
|
else:
|
|
return cls._from_turtle_str(data_str)
|
|
|
|
@classmethod
|
|
def _from_turtle_str(cls, turtle_str: str) -> "RDFGraph":
|
|
"""Create an RDFGraph instance from a Turtle string.
|
|
|
|
This method uses context variables to access known prefixes that may be
|
|
needed to complete missing prefix declarations in the Turtle string.
|
|
|
|
Args:
|
|
turtle_str: The input Turtle string.
|
|
|
|
Returns:
|
|
RDFGraph: A new RDFGraph instance.
|
|
"""
|
|
turtle_str = bytes(turtle_str, "utf-8").decode("unicode_escape")
|
|
patched_turtle = cls._ensure_prefixes(turtle_str)
|
|
g = cls()
|
|
try:
|
|
g.parse(data=patched_turtle, format="turtle")
|
|
return g
|
|
except Exception as parse_error:
|
|
# Typical LLM truncation: dangling ';' or ',' at EOF in property list.
|
|
if "EOF found when expected verb in property list" not in str(parse_error):
|
|
raise
|
|
repaired_turtle = cls._repair_truncated_turtle(patched_turtle)
|
|
if repaired_turtle == patched_turtle:
|
|
raise
|
|
logger.warning(
|
|
"Recovering truncated Turtle by closing dangling property list punctuation."
|
|
)
|
|
repaired_graph = cls()
|
|
repaired_graph.parse(data=repaired_turtle, format="turtle")
|
|
return repaired_graph
|
|
|
|
@staticmethod
|
|
def _repair_truncated_turtle(turtle_str: str) -> str:
|
|
"""Repair common LLM Turtle truncation patterns.
|
|
|
|
This only applies a minimal fix when content ends with dangling property-list
|
|
punctuation (';' or ',') and no terminating '.'.
|
|
"""
|
|
stripped = turtle_str.rstrip()
|
|
if not stripped:
|
|
return turtle_str
|
|
if stripped.endswith(";") or stripped.endswith(","):
|
|
return f"{stripped[:-1].rstrip()} .\n"
|
|
return turtle_str
|
|
|
|
@classmethod
|
|
def set_known_prefixes(cls, prefixes: dict[str, str] | None) -> None:
|
|
"""Set known prefixes in the context for use during parsing.
|
|
|
|
This should be called before parsing TTL strings that may use prefixes
|
|
from an ontology or other source. The prefixes will be automatically
|
|
added if they're used but not declared in the TTL string.
|
|
|
|
Args:
|
|
prefixes: Dictionary mapping prefix names to namespace URIs.
|
|
Example: {"fcaont": "https://growgraph.dev/fcaont#"}
|
|
"""
|
|
_known_prefixes_context.set(prefixes)
|
|
|
|
@classmethod
|
|
def get_known_prefixes(cls) -> dict[str, str] | None:
|
|
"""Get currently known prefixes from context.
|
|
|
|
Returns:
|
|
Dictionary mapping prefix names to namespace URIs, or None.
|
|
"""
|
|
return _known_prefixes_context.get()
|
|
|
|
@classmethod
|
|
def _from_jsonld_str(cls, jsonld_str: str) -> "RDFGraph":
|
|
"""Create an RDFGraph instance from a JSON-LD string.
|
|
|
|
Args:
|
|
jsonld_str: The input JSON-LD string.
|
|
|
|
Returns:
|
|
RDFGraph: A new RDFGraph instance with namespace prefixes extracted from @context.
|
|
"""
|
|
# Use pyld to convert JSON-LD to n-quads, then parse to avoid rdflib's deprecated ConjunctiveGraph
|
|
# This adapts to the new convention by using pyld directly instead of rdflib's JSON-LD parser
|
|
jsonld_data = json.loads(jsonld_str)
|
|
normalized = jsonld.normalize(
|
|
jsonld_data,
|
|
{"algorithm": "URDNA2015", "format": "application/n-quads"},
|
|
)
|
|
|
|
# jsonld.normalize returns a string when format is "application/n-quads"
|
|
normalized_str = normalized if isinstance(normalized, str) else str(normalized)
|
|
|
|
# Parse the normalized n-quads into RDFGraph
|
|
g = cls()
|
|
g.parse(data=normalized_str, format="nquads")
|
|
|
|
# Extract prefixes from @context in JSON-LD and bind them
|
|
try:
|
|
context = None
|
|
|
|
# Handle single object or array
|
|
if isinstance(jsonld_data, dict):
|
|
context = jsonld_data.get("@context")
|
|
elif isinstance(jsonld_data, list) and jsonld_data:
|
|
# For arrays, check first item for @context
|
|
first_item = jsonld_data[0]
|
|
if isinstance(first_item, dict):
|
|
context = first_item.get("@context")
|
|
|
|
# Bind prefixes from @context
|
|
if context and isinstance(context, dict):
|
|
for prefix, uri in context.items():
|
|
if isinstance(uri, str) and not prefix.startswith("@"):
|
|
# Skip JSON-LD keywords (starting with @)
|
|
try:
|
|
g.bind(prefix, uri)
|
|
except Exception as e:
|
|
logger.debug(f"Failed to bind prefix '{prefix}': {e}")
|
|
|
|
except (json.JSONDecodeError, ValueError, AttributeError) as e:
|
|
logger.debug(f"Could not extract prefixes from JSON-LD @context: {e}")
|
|
|
|
return g
|
|
|
|
@staticmethod
|
|
def _to_turtle_str(g: Any) -> str:
|
|
"""Convert an RDFGraph to a Turtle string.
|
|
|
|
For graphs backed by the *oxigraph* store the serialisation is
|
|
delegated to ``pyoxigraph`` so that RDF 1.2 triple-term syntax
|
|
(``<<( s p o )>>``) is emitted correctly.
|
|
|
|
Args:
|
|
g: The RDFGraph instance.
|
|
|
|
Returns:
|
|
str: The Turtle (or Turtle-star) string representation.
|
|
"""
|
|
if hasattr(g, "store") and type(g.store).__name__ == "OxigraphStore":
|
|
return g.serialize_turtle_star()
|
|
return g.serialize(format="turtle")
|
|
|
|
def serialize_turtle_star(self) -> str:
|
|
"""Serialize an oxigraph-backed graph to Turtle-star via *pyoxigraph*.
|
|
|
|
This method extracts all quads belonging to this graph's context
|
|
from the underlying ``pyoxigraph.Store`` and serialises them into
|
|
the default graph using ``pyoxigraph.serialize`` with the Turtle
|
|
format, which natively supports RDF 1.2 ``<<( … )>>`` syntax.
|
|
|
|
Returns:
|
|
Turtle-star string.
|
|
|
|
Raises:
|
|
RuntimeError: If the graph is not backed by an oxigraph store.
|
|
"""
|
|
try:
|
|
import pyoxigraph as ox
|
|
from oxrdflib._converter import to_ox
|
|
except ImportError as exc:
|
|
raise RuntimeError(
|
|
"pyoxigraph / oxrdflib must be installed for Turtle-star serialisation"
|
|
) from exc
|
|
|
|
inner_store: ox.Store = self.store._inner # type: ignore[attr-defined]
|
|
graph_ctx_raw = to_ox(self.identifier)
|
|
assert isinstance(
|
|
graph_ctx_raw,
|
|
(ox.NamedNode, ox.BlankNode, ox.DefaultGraph),
|
|
)
|
|
graph_ctx: ox.NamedNode | ox.BlankNode | ox.DefaultGraph = graph_ctx_raw
|
|
|
|
# Copy quads into a temporary store under the default graph so
|
|
# that ``ox.serialize`` can emit plain Turtle (Turtle-star).
|
|
tmp = ox.Store()
|
|
used_iri_terms: set[str] = set()
|
|
|
|
def _collect_used_iris(term: Any) -> None:
|
|
if isinstance(term, ox.NamedNode):
|
|
used_iri_terms.add(term.value)
|
|
return
|
|
if isinstance(term, ox.Triple):
|
|
_collect_used_iris(term.subject)
|
|
_collect_used_iris(term.predicate)
|
|
_collect_used_iris(term.object)
|
|
|
|
for quad in inner_store.quads_for_pattern(
|
|
None,
|
|
None,
|
|
None,
|
|
graph_ctx,
|
|
):
|
|
_collect_used_iris(quad.subject)
|
|
_collect_used_iris(quad.predicate)
|
|
_collect_used_iris(quad.object)
|
|
tmp.add(
|
|
ox.Quad(quad.subject, quad.predicate, quad.object, ox.DefaultGraph())
|
|
)
|
|
|
|
namespace_to_prefix: dict[str, str] = {}
|
|
for prefix, namespace in self.namespaces():
|
|
if not prefix:
|
|
continue
|
|
prefix_str = str(prefix)
|
|
namespace_str = str(namespace)
|
|
current = namespace_to_prefix.get(namespace_str)
|
|
if current is None or (len(prefix_str), prefix_str) < (
|
|
len(current),
|
|
current,
|
|
):
|
|
namespace_to_prefix[namespace_str] = prefix_str
|
|
|
|
prefixes = {
|
|
prefix: namespace
|
|
for namespace, prefix in namespace_to_prefix.items()
|
|
if any(iri.startswith(namespace) for iri in used_iri_terms)
|
|
}
|
|
raw: bytes = tmp.dump(
|
|
format=ox.RdfFormat.TURTLE,
|
|
from_graph=ox.DefaultGraph(),
|
|
prefixes=prefixes or None,
|
|
) # type: ignore[assignment]
|
|
return raw.decode()
|
|
|
|
def __new__(cls, *args, **kwargs):
|
|
"""Create a new RDFGraph instance."""
|
|
instance = super().__new__(cls)
|
|
return instance
|
|
|
|
def serialize(
|
|
self,
|
|
destination: Any = None,
|
|
format: str = "turtle",
|
|
base: str | None = None,
|
|
encoding: str | None = None,
|
|
**args: Any,
|
|
) -> Any:
|
|
"""Serialize the graph, delegating to pyoxigraph for oxigraph stores.
|
|
|
|
When the graph is backed by an *oxigraph* store and the requested
|
|
format is ``"turtle"`` (or ``"ttl"``), serialisation is handled by
|
|
``pyoxigraph`` which natively supports RDF 1.2 triple terms.
|
|
For all other stores or formats the default rdflib serialiser is
|
|
used.
|
|
"""
|
|
is_ox = type(self.store).__name__ == "OxigraphStore"
|
|
if is_ox and format in ("turtle", "ttl"):
|
|
ttl = self.serialize_turtle_star()
|
|
if destination is not None:
|
|
enc = encoding or "utf-8"
|
|
with open(destination, "w", encoding=enc) as fh:
|
|
fh.write(ttl)
|
|
return None
|
|
return ttl
|
|
return super().serialize(
|
|
destination=destination,
|
|
format=format,
|
|
base=base,
|
|
encoding=encoding,
|
|
**args,
|
|
)
|
|
|
|
def update(
|
|
self,
|
|
update_object: Any,
|
|
processor: Any = "sparql",
|
|
initNs: Mapping[str, Any] | None = None,
|
|
initBindings: Mapping[str, Any] | None = None,
|
|
use_store_provided: bool = True,
|
|
**kwargs: Any,
|
|
) -> None:
|
|
"""Execute SPARQL update using a base Graph view.
|
|
|
|
rdflib's SPARQL update engine has internal checks that branch on exact
|
|
``Graph`` type, which can break for subclasses on ``INSERT/DELETE ... WHERE``.
|
|
Running updates through a base ``Graph`` view avoids that edge case while
|
|
still operating on the same underlying store/identifier.
|
|
"""
|
|
graph_view = Graph(store=self.store, identifier=self.identifier)
|
|
graph_view.namespace_manager = self.namespace_manager
|
|
graph_view.update(
|
|
update_object=update_object,
|
|
processor=processor,
|
|
initNs=initNs,
|
|
initBindings=initBindings,
|
|
use_store_provided=use_store_provided,
|
|
**kwargs,
|
|
)
|
|
return None
|
|
|
|
def sanitize_prefixes_namespaces(self):
|
|
"""
|
|
Rematches prefixes in an RDFLib graph to correct namespaces when a namespace
|
|
with the same URI exists. Handles cases where prefixes might not be bound
|
|
as namespaces.
|
|
|
|
Args:
|
|
self (RDFGraph): The RDFLib graph to process
|
|
|
|
Returns:
|
|
RDFGraph: The graph with corrected prefix-namespace mappings
|
|
"""
|
|
# Get the namespace manager
|
|
ns_manager = self.namespace_manager
|
|
|
|
# Collect all current prefix-URI mappings
|
|
current_prefixes = dict(ns_manager.namespaces())
|
|
|
|
# Group URIs by their string representation to find duplicates
|
|
uri_to_prefixes = defaultdict(list)
|
|
for prefix, uri in current_prefixes.items():
|
|
uri_to_prefixes[str(uri)].append((prefix, uri))
|
|
|
|
# Find the "canonical" namespace objects for each URI
|
|
# (the actual Namespace objects that might be registered)
|
|
canonical_namespaces = {}
|
|
|
|
# Check if any of the URIs correspond to well-known namespaces
|
|
# by trying to create Namespace objects and seeing if they're already registered
|
|
for uri_str, prefix_uri_pairs in uri_to_prefixes.items():
|
|
# Try to find if there's already a proper Namespace object for this URI
|
|
namespace_candidates = []
|
|
|
|
for prefix, uri_obj in prefix_uri_pairs:
|
|
# Check if this is already a proper Namespace object
|
|
if isinstance(uri_obj, Namespace):
|
|
namespace_candidates.append(uri_obj)
|
|
else:
|
|
# Try to create a Namespace and see if it matches existing ones
|
|
try:
|
|
ns = Namespace(uri_str)
|
|
namespace_candidates.append(ns)
|
|
except:
|
|
continue
|
|
|
|
# Use the first valid namespace candidate as canonical
|
|
if namespace_candidates:
|
|
canonical_namespaces[uri_str] = namespace_candidates[0]
|
|
|
|
# Now rebuild the namespace manager with corrected mappings
|
|
# Clear existing bindings first
|
|
new_ns_manager = NamespaceManager(self)
|
|
|
|
# Track which prefixes we want to keep/reassign
|
|
final_mappings = {}
|
|
|
|
for uri_str, prefix_uri_pairs in uri_to_prefixes.items():
|
|
if len(prefix_uri_pairs) == 1:
|
|
# No duplicates, keep as-is but ensure we use canonical namespace
|
|
prefix, _ = prefix_uri_pairs[0]
|
|
canonical_ns = canonical_namespaces.get(uri_str)
|
|
if canonical_ns:
|
|
final_mappings[prefix] = canonical_ns
|
|
else:
|
|
# Fallback to creating a new Namespace
|
|
final_mappings[prefix] = Namespace(uri_str)
|
|
else:
|
|
# Multiple prefixes for same URI - need to decide which to keep
|
|
# Priority: 1) Proper Namespace objects,
|
|
# 2) Shorter prefixes,
|
|
# 3) Alphabetical
|
|
prefix_uri_pairs.sort(
|
|
key=lambda x: (
|
|
not isinstance(x[1], Namespace), # Namespace objects first
|
|
len(x[0]), # Shorter prefixes next
|
|
x[0], # Alphabetical order
|
|
)
|
|
)
|
|
|
|
# Keep the best prefix, map others to it if needed
|
|
best_prefix, _ = prefix_uri_pairs[0]
|
|
canonical_ns = canonical_namespaces.get(uri_str, Namespace(uri_str))
|
|
final_mappings[best_prefix] = canonical_ns
|
|
|
|
other_prefixes = [p for p, _ in prefix_uri_pairs[1:]]
|
|
if other_prefixes:
|
|
logger.debug(
|
|
f"Consolidating prefixes {other_prefixes} "
|
|
f"-> '{best_prefix}' for URI: {uri_str}"
|
|
)
|
|
|
|
# Apply the final mappings
|
|
for prefix, namespace in final_mappings.items():
|
|
new_ns_manager.bind(prefix, namespace, override=True)
|
|
|
|
# Replace the graph's namespace manager
|
|
self.namespace_manager = new_ns_manager
|
|
|
|
def unbind_chunk_namespaces(self, chunk_pattern="/chunk/") -> "RDFGraph":
|
|
"""
|
|
Unbinds namespace prefixes that point to URIs containing a chunk pattern.
|
|
Returns a new graph with chunk namespaces dereferenced (expanded to full URIs).
|
|
|
|
Args:
|
|
chunk_pattern (str): The pattern to look for in URIs (default: "/chunk/")
|
|
|
|
Returns:
|
|
RDFGraph: New graph with chunk-related namespaces unbound
|
|
"""
|
|
current_prefixes = dict(self.namespace_manager.namespaces())
|
|
|
|
# Find prefixes that point to URIs containing the chunk pattern
|
|
chunk_prefixes = []
|
|
for prefix, uri in current_prefixes.items():
|
|
uri_str = str(uri)
|
|
if chunk_pattern in uri_str:
|
|
chunk_prefixes.append((prefix, uri_str))
|
|
|
|
# Create new graph
|
|
new_graph = RDFGraph()
|
|
|
|
# Copy all triples (URIs are already expanded internally)
|
|
for triple in self:
|
|
new_graph.add(triple)
|
|
|
|
# Bind only non-chunk namespace prefixes to the new graph
|
|
for prefix, uri in current_prefixes.items():
|
|
uri_str = str(uri)
|
|
if chunk_pattern not in uri_str:
|
|
new_graph.bind(prefix, uri)
|
|
|
|
# Log what was removed
|
|
if chunk_prefixes:
|
|
logger.debug(f"Unbound {len(chunk_prefixes)} chunk-related namespace(s):")
|
|
for prefix, uri in chunk_prefixes:
|
|
logger.debug(f" - '{prefix}': {uri}")
|
|
|
|
return new_graph
|
|
|
|
def remap_namespaces(self, old_namespace, new_namespace) -> None:
|
|
updates = {}
|
|
for s, p, o in self:
|
|
new_s, new_p, new_o = s, p, o
|
|
if isinstance(s, URIRef) and str(s).startswith(str(old_namespace)):
|
|
new_s = URIRef(
|
|
str(s).replace(str(old_namespace), str(new_namespace), 1)
|
|
)
|
|
if isinstance(p, URIRef) and str(p).startswith(str(old_namespace)):
|
|
new_p = URIRef(
|
|
str(p).replace(str(old_namespace), str(new_namespace), 1)
|
|
)
|
|
if isinstance(o, URIRef) and str(o).startswith(str(old_namespace)):
|
|
new_o = URIRef(
|
|
str(o).replace(str(old_namespace), str(new_namespace), 1)
|
|
)
|
|
|
|
if (new_s, new_p, new_o) != (s, p, o):
|
|
updates[(s, p, o)] = (new_s, new_p, new_o)
|
|
|
|
for (s, p, o), (new_s, new_p, new_o) in updates.items():
|
|
self.remove((s, p, o))
|
|
self.add((new_s, new_p, new_o))
|
|
|
|
def add_triple(self, subject: str, predicate: str, object_: str) -> None:
|
|
"""Add a triple to the graph.
|
|
|
|
Args:
|
|
subject: Subject URI as string
|
|
predicate: Predicate URI as string
|
|
object_: Object URI as string or literal value
|
|
"""
|
|
# Convert strings to appropriate RDFLib objects
|
|
subj = URIRef(subject)
|
|
pred = URIRef(predicate)
|
|
|
|
# Handle object - could be URI or literal
|
|
if object_.startswith("http://") or object_.startswith("https://"):
|
|
obj = URIRef(object_)
|
|
else:
|
|
# Treat as literal
|
|
obj = Literal(object_)
|
|
|
|
self.add((subj, pred, obj))
|
|
logger.debug(f"Added triple: {subj} {pred} {obj}")
|
|
|
|
def remove_triple(self, subject: str, predicate: str, object_: str) -> None:
|
|
"""Remove a triple from the graph.
|
|
|
|
Args:
|
|
subject: Subject URI as string
|
|
predicate: Predicate URI as string
|
|
object_: Object URI as string or literal value
|
|
"""
|
|
# Convert strings to appropriate RDFLib objects
|
|
subj = URIRef(subject)
|
|
pred = URIRef(predicate)
|
|
|
|
# Handle object - could be URI or literal
|
|
if object_.startswith("http://") or object_.startswith("https://"):
|
|
obj = URIRef(object_)
|
|
else:
|
|
# Treat as literal
|
|
obj = Literal(object_)
|
|
|
|
self.remove((subj, pred, obj))
|
|
logger.debug(f"Removed triple: {subj} {pred} {obj}")
|
|
|
|
def hash(self: Graph) -> str:
|
|
# Serialize to JSON-LD
|
|
data = self.serialize(format="json-ld")
|
|
|
|
# Parse the JSON string
|
|
doc = json.loads(data)
|
|
|
|
# Canonicalize using URDNA2015 normalization
|
|
normalized = jsonld.normalize(
|
|
doc,
|
|
{"algorithm": "URDNA2015", "format": "application/n-quads"},
|
|
)
|
|
# jsonld.normalize returns a string when format is "application/n-quads"
|
|
normalized_str = normalized if isinstance(normalized, str) else str(normalized)
|
|
return render_text_hash(normalized_str, digits=None)
|