From 03f92d460e8426da88888bb9226c60d782fcfe68 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martynas=20Jusevi=C4=8Dius?= Date: Sun, 2 Aug 2026 20:34:40 +0200 Subject: [PATCH 1/7] Serve raw ontology graphs without RDFS inference MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolve the application ontology's owl:imports closure natively via ontapi (OntModelFactory.createModel over a ScopedGraphRepository view) instead of manually flattening the closure and materializing an RDFS-inferred model. No inference is applied anymore — every consumer (constructor/constraint inheritance, client-side (rdfs:subClassOf)* queries) traverses hierarchies explicitly. rdfs:Class terms are promoted to owl:Class in a separate union member so no document graph is polluted; closure union graphs are cached in a bounded map on Application, keyed by ontology URI. The shared repository now only ever holds raw per-document graphs, which ProxyRequestFilter serves directly for closure documents — asserted triples only, identical to a direct document GET. The DESCRIBE fallback over the in-memory closure (terms minted in external namespaces) is inference-free as well. This fixes inferred rdf:type rdfs:Resource leaking into proxied namespace documents, where the extra type produced multi-token @typeof and silently degraded View blocks to a generic property list. Co-Authored-By: Claude Fable 5 --- CHANGELOG.md | 8 ++ http-tests/proxy/GET-proxied-ontology-ns.sh | 25 +++- .../atomgraph/linkeddatahub/Application.java | 20 ++- .../linkeddatahub/resource/Namespace.java | 6 +- .../resource/admin/ClearOntology.java | 6 +- .../server/filter/request/OntologyFilter.java | 85 ++++++------- .../filter/request/ProxyRequestFilter.java | 28 ++++- .../server/io/ValidatingModelProvider.java | 3 +- .../server/util/ScopedGraphRepository.java | 119 ++++++++++++++++++ .../OntologyImportsCharacterizationTest.java | 66 +++++++--- .../util/SPINConstraintValidationTest.java | 7 +- 11 files changed, 288 insertions(+), 85 deletions(-) create mode 100644 src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java diff --git a/CHANGELOG.md b/CHANGELOG.md index c5ed2dae2..ad83ac92f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,11 @@ +## [Unreleased] +### Changed +- Application ontologies resolved as a native ontapi `owl:imports` union graph (cached per ontology URI) instead of a manually flattened, RDFS-materialized model — no RDFS inference +- `Namespace` no-query GET serves the raw ontology graph from the shared repository instead of rebuilding a repository per request + +### Fixed +- Raw ontology graphs no longer leak inferred `rdf:type rdfs:Resource`, which produced multi-token `@typeof` that broke View block rendering + ## [5.7.1] - 2026-08-06 ### Changed - RDFa editor: annotation overlay rebuilt on demand (`rdfa-editor/overlay.xsl`) diff --git a/http-tests/proxy/GET-proxied-ontology-ns.sh b/http-tests/proxy/GET-proxied-ontology-ns.sh index c97f48505..152360441 100755 --- a/http-tests/proxy/GET-proxied-ontology-ns.sh +++ b/http-tests/proxy/GET-proxied-ontology-ns.sh @@ -49,9 +49,9 @@ clear-ontology.sh \ --ontology "$namespace" # request the namespace document URI (without fragment) via ?uri= proxy. -# the namespace document is not DataManager-mapped and not a registered app, -# so ProxyRequestFilter falls through to the OntModel DESCRIBE path, which -# returns descriptions of all #-fragment terms in that namespace. +# the namespace document is not DataManager-mapped, not a registered app and not a graph +# in the ontology closure, so ProxyRequestFilter falls through to the closure DESCRIBE +# fallback, which returns descriptions of all #-fragment terms in that namespace. response=$(curl -k -f -s \ -G \ @@ -60,7 +60,24 @@ response=$(curl -k -f -s \ --data-urlencode "uri=${namespace_uri}" \ "$END_USER_BASE_URL") -# verify both class descriptions are present in the response +# verify both class descriptions are present in the response and no inferred triples leak echo "$response" | grep -q "$class1" echo "$response" | grep -q "$class2" +! echo "$response" | grep -q "http://www.w3.org/2000/01/rdf-schema#Resource" + +# request the ontology document itself via ?uri= proxy: it is a graph in the ontology closure, +# so it must be served with its raw graph — asserted triples only, identical to a direct +# document GET. Inferred rdf:type rdfs:Resource used to leak from the RDFS-materialized +# in-memory model here, breaking client-side @typeof matching of View blocks. + +doc_response=$(curl -k -f -s \ + -G \ + -E "$AGENT_CERT_FILE":"$AGENT_CERT_PWD" \ + -H "Accept: application/n-triples" \ + --data-urlencode "uri=${ontology_doc}" \ + "$END_USER_BASE_URL") + +echo "$doc_response" | grep -q "$class1" +echo "$doc_response" | grep -q "$class2" +! echo "$doc_response" | grep -q "http://www.w3.org/2000/01/rdf-schema#Resource" diff --git a/src/main/java/com/atomgraph/linkeddatahub/Application.java b/src/main/java/com/atomgraph/linkeddatahub/Application.java index 9147b37fc..b83be580b 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/Application.java +++ b/src/main/java/com/atomgraph/linkeddatahub/Application.java @@ -161,6 +161,7 @@ import javax.net.ssl.TrustManagerFactory; import jakarta.servlet.ServletContext; import javax.xml.transform.Source; +import org.apache.jena.ontapi.UnionGraph; import org.apache.jena.ontapi.model.OntModel; import org.apache.jena.query.Dataset; import org.apache.jena.query.Query; @@ -199,6 +200,7 @@ import javax.xml.parsers.ParserConfigurationException; import javax.xml.transform.TransformerException; import javax.xml.transform.stream.StreamSource; +import net.jodah.expiringmap.ExpirationPolicy; import net.jodah.expiringmap.ExpiringMap; import net.sf.saxon.om.TreeInfo; import net.sf.saxon.s9api.Processor; @@ -288,6 +290,7 @@ public class Application extends ResourceConfig private final KeyStore keyStore, trustStore; private final URI secretaryWebIDURI; private final List supportedLanguages; + private final ExpiringMap ontologyGraphs = ExpiringMap.builder().maxSize(1000).expirationPolicy(ExpirationPolicy.ACCESSED).expiration(1, TimeUnit.HOURS).build(); // assembled ontology imports-closure union graphs, keyed by ontology URI; evicted entries are transparently rebuilt by OntologyFilter on the next cache miss private final ExpiringMap webIDmodelCache = ExpiringMap.builder().expiration(Long.parseLong(System.getProperty("com.atomgraph.linkeddatahub.webIDCacheExpiration", "86400")), TimeUnit.SECONDS).build(); // TTL (seconds) configurable via WEBID_CACHE_EXPIRATION; a lower value bounds how long a revoked WebID stays cached private final ExpiringMap oidcModelCache = ExpiringMap.builder().variableExpiration().build(); private final ExpiringMap jwksCache = ExpiringMap.builder().expiration(Long.parseLong(System.getProperty("com.atomgraph.linkeddatahub.jwksCacheExpiration", "86400")), TimeUnit.SECONDS).build(); // Cache JWKS responses; TTL (seconds) configurable via JWKS_CACHE_EXPIRATION @@ -1804,10 +1807,23 @@ public OntologyRepository createRepository(EndUserApplication app) return appRepository; } - + + /** + * Returns the cache of assembled ontology imports-closure union graphs, keyed by ontology URI + * (origin-scoped per dataspace, so a single map cannot collide across applications). + * The union graph is ontapi's view over the raw per-document graphs cached in the (per-app or + * system) repository; it is not a document graph itself and is never served on the wire. + * + * @return ontology URI to union graph map + */ + public Map getOntologyGraphs() + { + return ontologyGraphs; + } + /** * Returns a registry of readable and writeable media types. - * + * * @return registry object */ public MediaTypes getMediaTypes() diff --git a/src/main/java/com/atomgraph/linkeddatahub/resource/Namespace.java b/src/main/java/com/atomgraph/linkeddatahub/resource/Namespace.java index 5ce68929e..50cd157ca 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/resource/Namespace.java +++ b/src/main/java/com/atomgraph/linkeddatahub/resource/Namespace.java @@ -140,9 +140,9 @@ public Response get(@QueryParam(QUERY) Query query, // the application ontology MUST use a URI! This is the URI this ontology endpoint is deployed on by the Dispatcher class String ontologyURI = getApplication().getOntology().getURI(); if (log.isDebugEnabled()) log.debug("Returning raw namespace ontology: {}", ontologyURI); - // not returning the injected in-memory ontology because it has inferences applied to it; - // a fresh, mapping-seeded repository serves the raw SPARQL-loaded ontology - OntologyRepository repository = getSystem().createRepository(getApplication().as(EndUserApplication.class)); + // not returning the injected in-memory ontology because it is the full imports closure (a union view); + // the shared repository serves the standalone raw ontology graph + OntologyRepository repository = getSystem().getRepository(getApplication().as(EndUserApplication.class)); return getResponseBuilder(org.apache.jena.rdf.model.ModelFactory.createModelForGraph(repository.get(ontologyURI))).build(); } else throw new BadRequestException("SPARQL query string not provided"); diff --git a/src/main/java/com/atomgraph/linkeddatahub/resource/admin/ClearOntology.java b/src/main/java/com/atomgraph/linkeddatahub/resource/admin/ClearOntology.java index 9ce40f1a6..190fbd346 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/resource/admin/ClearOntology.java +++ b/src/main/java/com/atomgraph/linkeddatahub/resource/admin/ClearOntology.java @@ -78,12 +78,14 @@ public Response post(@FormParam("uri") String ontologyURI, @HeaderParam("Referer EndUserApplication endUserApp = getApplication().as(AdminApplication.class).getEndUserApplication(); // we're assuming the current app is admin OntologyRepository repository = getSystem().getRepository(endUserApp); - if (repository.isCached(ontologyURI)) + if (repository.isCached(ontologyURI) || getSystem().getOntologyGraphs().containsKey(ontologyURI)) { if (log.isDebugEnabled()) log.debug("Clearing ontology with URI '{}' from memory", ontologyURI); repository.remove(ontologyURI); + getSystem().getOntologyGraphs().remove(ontologyURI); URI ontologyDocURI = UriBuilder.fromUri(ontologyURI).fragment(null).build(); // skip fragment from the ontology URI to get its graph URI + repository.remove(ontologyDocURI.toString()); // the raw graph is also aliased under the fragment-stripped document URI // frontend proxy still uses URL-pattern BAN for direct document GETs (until Stage 3 brings xkey tagging to varnish-frontend). // xkey purge covers proxied SPARQL CONSTRUCT/SELECT responses tagged by their backend (varnish-admin / varnish-end-user). URI frontendProxy = getSystem().getFrontendProxy(); @@ -110,7 +112,7 @@ public Response post(@FormParam("uri") String ontologyURI, @HeaderParam("Referer } // !!! we need to reload the ontology model before returning a response, to make sure the next request already gets the new version !!! - OntologyFilter.loadOntology(repository, ontologyURI); + getSystem().getOntologyGraphs().put(ontologyURI, OntologyFilter.loadOntology(repository, ontologyURI)); } if (referer != null) return Response.seeOther(referer).build(); diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java index e9bd7af71..653012665 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java +++ b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java @@ -19,14 +19,13 @@ import com.atomgraph.linkeddatahub.apps.model.Application; import com.atomgraph.linkeddatahub.apps.model.EndUserApplication; import com.atomgraph.client.util.jena.PrefixGraphRepository; +import com.atomgraph.linkeddatahub.server.util.ScopedGraphRepository; import com.atomgraph.linkeddatahub.vocabulary.LAPP; import com.atomgraph.server.exception.OntologyException; import java.io.IOException; import java.net.URI; import java.net.URISyntaxException; -import java.util.HashSet; import java.util.Optional; -import java.util.Set; import jakarta.annotation.Priority; import jakarta.inject.Inject; import jakarta.ws.rs.container.ContainerRequestContext; @@ -34,12 +33,13 @@ import jakarta.ws.rs.container.PreMatching; import org.apache.jena.ontapi.OntModelFactory; import org.apache.jena.ontapi.OntSpecification; +import org.apache.jena.ontapi.UnionGraph; import org.apache.jena.ontapi.model.OntModel; import org.apache.jena.rdf.model.Model; import org.apache.jena.rdf.model.ModelFactory; -import org.apache.jena.vocabulary.OWL; import org.apache.jena.vocabulary.RDF; import org.apache.jena.vocabulary.RDFS; +import org.apache.jena.vocabulary.OWL; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -131,8 +131,9 @@ public OntModel getOntology(Application app) } /** - * Loads the ontology model for the specified ontology URI, building its owl:imports closure with - * RDFS inference and materializing the inferences into the repository cache. + * Returns the ontology model for the specified ontology URI, assembling its owl:imports closure + * on a cache miss. The returned model is a fresh per-request wrapper over the shared closure + * union graph. * * @param app application resource * @param uri ontology URI @@ -146,64 +147,54 @@ public OntModel getOntology(Application app, String uri) final PrefixGraphRepository repository = app.canAs(EndUserApplication.class) ? getSystem().getRepository(app.as(EndUserApplication.class)) : getSystem().getRepository(); - // only build the materialized model if the ontology is not already cached; the double check under the - // repository lock ensures a single thread materializes it (loadOntology is a compound load + inference + - // put, not atomic), so concurrent cold requests don't duplicate the work or race each other's writes - if (!repository.isCached(uri)) + // only assemble the closure if it is not already cached; the double check under the repository + // lock ensures a single thread assembles it (loadOntology is a compound load + union build, not + // atomic), so concurrent cold requests don't duplicate the work or race each other's writes + UnionGraph union = getSystem().getOntologyGraphs().get(uri); + if (union == null) { synchronized (repository) { - if (!repository.isCached(uri)) loadOntology(repository, uri); + union = getSystem().getOntologyGraphs().get(uri); + if (union == null) + { + union = loadOntology(repository, uri); + getSystem().getOntologyGraphs().put(uri, union); + } } } - return OntModelFactory.createModel(repository.get(uri), OntSpecification.OWL2_FULL_MEM); + return OntModelFactory.createModel(union, OntSpecification.OWL2_FULL_MEM); } /** - * Builds and caches the materialized ontology model. Assembles the owl:imports closure into a single - * graph (so ontapi never manages a union-graph hierarchy over the shared repository), applies RDFS - * inference over the flattened closure, and materializes the inferences into the repository cache so - * the rules engine is not invoked on every request. + * Assembles the ontology's owl:imports closure as a union graph. ontapi resolves the closure through + * a scoped repository view: raw per-document graphs are read through (and cached in) the shared + * repository, while ontapi's union-graph bookkeeping stays local to the view — the shared repository + * keeps serving raw document graphs, and duplicate ontology IDs across applications cannot collide. + * No inference is applied: all consumers traverse class/property hierarchies explicitly. * * @param repository graph repository * @param uri ontology URI + * @return closure union graph */ - public static void loadOntology(PrefixGraphRepository repository, String uri) + public static UnionGraph loadOntology(PrefixGraphRepository repository, String uri) { if (log.isDebugEnabled()) log.debug("Started loading ontology with URI '{}'", uri); - Model union = ModelFactory.createDefaultModel(); - Set closure = new HashSet<>(); - loadClosure(repository, uri, union, closure); - OntModel inferred = OntModelFactory.createModel(union.getGraph(), OntSpecification.OWL2_FULL_MEM_RDFS_INF); - OntModel materialized = OntModelFactory.createModel(OntSpecification.OWL2_FULL_MEM); - materialized.add(inferred); - // promote rdfs:Class to owl:Class so OWL2 profiles recognise third-party vocab terms (e.g. sp:Describe in sp.ttl) - inferred.listSubjectsWithProperty(RDF.type, RDFS.Class).forEach(r -> materialized.add(r, RDF.type, OWL.Class)); - repository.put(uri, materialized.getGraph()); - // cache imported graphs under their fragment-stripped document URIs too - closure.stream().filter(closureURI -> !closureURI.equals(uri)).forEach(importURI -> addDocumentModel(repository, importURI)); + ScopedGraphRepository scoped = new ScopedGraphRepository(repository); + OntModel ontology = OntModelFactory.createModel(repository.get(uri), OntSpecification.OWL2_FULL_MEM, scoped); + UnionGraph union = (UnionGraph)ontology.getGraph(); + // promote rdfs:Class to owl:Class so the OWL2 profile recognises third-party vocab terms (e.g. sp:Describe + // in sp.ttl) as named classes. The promotions live in their own union member so no document graph is + // polluted; carrying no owl:Ontology header, the member is ignored by ontapi's union-graph listener + Model promotions = ModelFactory.createDefaultModel(); + ontology.listSubjectsWithProperty(RDF.type, RDFS.Class).forEach(r -> promotions.add(r, RDF.type, OWL.Class)); + if (!promotions.isEmpty()) union.addSubGraph(promotions.getGraph()); + // cache closure graphs under their fragment-stripped document URIs too + scoped.ids().filter(closureURI -> closureURI.startsWith("http://") || closureURI.startsWith("https://")). + forEach(closureURI -> addDocumentModel(repository, closureURI)); if (log.isDebugEnabled()) log.debug("Finished loading ontology with URI '{}'", uri); - } - - /** - * Recursively loads the transitive owl:imports closure of an ontology into a single union model, - * fetching each graph via the repository (SPARQL-first / bundled mappings). - * - * @param repository graph repository - * @param uri ontology URI - * @param union accumulator model - * @param seen accumulator of visited URIs (prevents cycles) - */ - public static void loadClosure(PrefixGraphRepository repository, String uri, Model union, Set seen) - { - if (!seen.add(uri)) return; - Model model = ModelFactory.createModelForGraph(repository.get(uri)); - union.add(model); - model.listObjectsOfProperty(OWL.imports).toList().forEach(imp -> - { - if (imp.isURIResource()) loadClosure(repository, imp.asResource().getURI(), union, seen); - }); + return union; } /** diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java index 9166b28ea..4a87ab3a0 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java +++ b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java @@ -21,7 +21,10 @@ import com.atomgraph.client.vocabulary.AC; import com.atomgraph.core.exception.BadGatewayException; import com.atomgraph.core.util.ModelUtils; +import com.atomgraph.client.util.jena.PrefixGraphRepository; +import com.atomgraph.linkeddatahub.apps.model.Application; import com.atomgraph.linkeddatahub.apps.model.Dataset; +import com.atomgraph.linkeddatahub.apps.model.EndUserApplication; import org.apache.jena.ontapi.model.OntModel; import com.atomgraph.linkeddatahub.client.GraphStoreClient; import com.atomgraph.linkeddatahub.client.filter.auth.IDTokenDelegationFilter; @@ -186,12 +189,29 @@ public void filter(ContainerRequestContext requestContext) throws IOException return; } - // serve terms from the app's in-memory namespace ontology (full imports closure) via DESCRIBE. - // covers both slash-based term URIs (e.g. schema:category) and hash-based namespaces - // (e.g. sioc:UserAccount → ac:document-uri strips to sioc:ns, so we also describe all - // ?term where STR(?term) starts with "#") if (isSafeMethod && getOntology().isPresent()) { + // serve documents of the app's ontology imports closure with their raw graphs — asserted triples + // only, identical to a direct document GET. Hash-based term URIs (e.g. ldh:View) arrive here + // fragment-stripped as their document URI. The isCached guard restricts serving to graphs already + // loaded as part of the closure, so arbitrary external URIs cannot trigger repository loading + // outside the SSRF-validated external client path below. + Optional appOpt = (Optional)requestContext.getProperty(LAPP.Application.getURI()); + PrefixGraphRepository repository = appOpt != null && appOpt.isPresent() && appOpt.get().canAs(EndUserApplication.class) ? + getSystem().getRepository(appOpt.get().as(EndUserApplication.class)) : getSystem().getRepository(); + if (repository.isCached(targetURI.toString())) + { + if (log.isDebugEnabled()) log.debug("Serving URI from the ontology closure cache: {}", targetURI); + Model model = org.apache.jena.rdf.model.ModelFactory.createModelForGraph(repository.get(targetURI.toString())); + requestContext.abortWith(getResponse(model, Response.Status.OK)); + return; + } + + // fall back to DESCRIBE over the in-memory closure union (asserted triples only — no inference) + // for term URIs whose document is not a closure graph: terms minted in external namespaces but + // defined in the app ontology. Covers both slash-based term URIs (e.g. schema:category) and + // hash-based namespaces (e.g. sioc:UserAccount → ac:document-uri strips to sioc:ns, so we also + // describe all ?term where STR(?term) starts with "#") ParameterizedSparqlString pss = new ParameterizedSparqlString( "DESCRIBE ?doc ?term WHERE { ?term ?p ?o FILTER(STRSTARTS(STR(?term), CONCAT(STR(?doc), \"#\"))) }"); pss.setIri("doc", targetURI.toString()); diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/io/ValidatingModelProvider.java b/src/main/java/com/atomgraph/linkeddatahub/server/io/ValidatingModelProvider.java index 018bca6c2..757c86bc1 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/server/io/ValidatingModelProvider.java +++ b/src/main/java/com/atomgraph/linkeddatahub/server/io/ValidatingModelProvider.java @@ -239,8 +239,9 @@ public Resource processRead(Resource resource) // this logic really belongs in a if (getApplication().isPresent() && getApplication().get().canAs(AdminApplication.class) && resource.hasProperty(RDF.type, OWL.Ontology)) { - // clear cached OntModel if ontology is updated. TO-DO: send event instead + // clear cached raw graph and closure union graph if ontology is updated. TO-DO: send event instead getSystem().getRepository().remove(resource.getURI()); + getSystem().getOntologyGraphs().remove(resource.getURI()); } if (getApplication().isPresent() && resource.hasProperty(RDF.type, ACL.Authorization)) diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java b/src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java new file mode 100644 index 000000000..c41190223 --- /dev/null +++ b/src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java @@ -0,0 +1,119 @@ +/** + * Copyright 2026 Martynas Jusevičius + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + */ +package com.atomgraph.linkeddatahub.server.util; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.stream.Stream; +import org.apache.jena.graph.Graph; +import org.apache.jena.ontapi.GraphRepository; + +/** + * A graph repository view that reads through a shared backing repository but keeps writes local. + *

+ * Handed to {@code OntModelFactory.createModel(Graph, OntSpecification, GraphRepository)} so that + * ontapi's union-graph bookkeeping — a {@code UnionGraph} wrapper per ontology in the imports + * closure, with listeners — lands in this instance's private store instead of the shared repository. + * The shared repository must keep answering {@code get(uri)} with the raw per-document graph (that is + * what proxied and direct document GETs serve), and duplicate ontology IDs across applications must + * not collide in one store. Reads fall through to the backing repository, triggering its + * SPARQL-first/mapped/HTTP loading and raw caching as usual, so resolving an imports closure through + * this view populates the shared raw cache as a side effect. + *

+ * After model construction, {@link #ids()} equals the set of resolved closure ontology IDs. + * + * @author Martynas Jusevičius {@literal } + */ +public class ScopedGraphRepository implements GraphRepository +{ + + private final GraphRepository backing; + private final Map local = new HashMap<>(); + + /** + * Constructs the view over a shared backing repository. + * + * @param backing shared graph repository + */ + public ScopedGraphRepository(GraphRepository backing) + { + this.backing = backing; + } + + @Override + public Graph get(String id) + { + Graph graph = local.get(id); + if (graph != null) return graph; + + return getBacking().get(id); + } + + @Override + public Stream ids() + { + return List.copyOf(local.keySet()).stream(); + } + + @Override + public Graph put(String id, Graph graph) + { + return local.put(id, graph); + } + + @Override + public Graph remove(String id) + { + return local.remove(id); + } + + @Override + public void clear() + { + local.clear(); + } + + @Override + public boolean contains(String id) + { + return local.containsKey(id) || getBacking().contains(id); + } + + @Override + public long count() + { + return local.size(); + } + + @Override + public Stream graphs() + { + return List.copyOf(local.values()).stream(); + } + + /** + * Returns the shared backing repository. + * + * @return graph repository + */ + public GraphRepository getBacking() + { + return backing; + } + +} diff --git a/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyImportsCharacterizationTest.java b/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyImportsCharacterizationTest.java index 312a61415..bafae14a9 100644 --- a/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyImportsCharacterizationTest.java +++ b/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyImportsCharacterizationTest.java @@ -19,6 +19,7 @@ import com.atomgraph.client.util.jena.PrefixGraphRepository; import org.apache.jena.ontapi.OntModelFactory; import org.apache.jena.ontapi.OntSpecification; +import org.apache.jena.ontapi.UnionGraph; import org.apache.jena.ontapi.model.OntModel; import org.apache.jena.rdf.model.Model; import org.apache.jena.rdf.model.ModelFactory; @@ -31,9 +32,10 @@ import static org.junit.jupiter.api.Assertions.*; /** - * Pins {@link OntologyFilter#loadOntology}: it flattens the owl:imports closure into one graph, - * applies RDFS inference, and materializes the inferences into the repository cache — without ontapi - * managing a union-graph hierarchy over the shared repository (which collides on duplicate ontology IDs). + * Pins {@link OntologyFilter#loadOntology}: it assembles the owl:imports closure as a union graph + * (resolved natively by ontapi over a scoped repository view), applies no inference, and + * leaves the shared repository holding raw per-document graphs — which is what proxied and direct + * document GETs serve. * * @author Martynas Jusevičius {@literal } */ @@ -45,7 +47,7 @@ public class OntologyImportsCharacterizationTest private static final String NS = "http://example.org/ns#"; @Test - public void testLoadOntologyFlattensClosureWithMaterializedRDFSInference() + public void testLoadOntologyResolvesClosureWithoutInference() { PrefixGraphRepository repository = new PrefixGraphRepository(null); @@ -70,22 +72,52 @@ public void testLoadOntologyFlattensClosureWithMaterializedRDFSInference() base.add(baseOnt, OWL.imports, base.createResource(IMPORT_URI)); repository.put(BASE_URI, base.getGraph()); - OntologyFilter.loadOntology(repository, BASE_URI); + UnionGraph union = OntologyFilter.loadOntology(repository, BASE_URI); + Model closure = ModelFactory.createModelForGraph(union); - Model result = ModelFactory.createModelForGraph(repository.get(BASE_URI)); - // (a) imported terms flattened into the cached graph - assertTrue(result.contains(b, RDFS.subClassOf, a), "imported terms should be flattened in"); - // (b) RDFS inference materialized as a concrete triple: x a A - assertTrue(result.contains(x, RDF.type, a), "RDFS-inferred 'x a A' should be materialized in the cached graph"); - // (c) the import is also cached under its (fragment-stripped) document URI - assertTrue(repository.isCached(IMPORT_URI), "import should remain cached"); - // (d) REGRESSION GUARD: both owl:Class and rdfs:Class-only terms must be recognized as OntClasses by the returned - // model, so GET /ns?forClass= resolves the class and runs its SPIN constructor. - // OntologyFilter promotes all rdfs:Class subjects to owl:Class so OWL2 profiles (which do not recognize bare - // rdfs:Class) can find third-party vocab terms like sp:Describe. - OntModel ontology = OntModelFactory.createModel(repository.get(BASE_URI), OntSpecification.OWL2_FULL_MEM); + // (a) imported terms are visible through the closure union + assertTrue(closure.contains(b, RDFS.subClassOf, a), "imported terms should be visible through the closure union"); + // (b) no inference: neither type propagation nor vacuous rdfs:Resource typing appears + assertFalse(closure.contains(x, RDF.type, a), "no RDFS type propagation expected in the closure"); + assertFalse(closure.contains(x, RDF.type, RDFS.Resource), "no vacuous rdfs:Resource typing expected in the closure"); + // (c) the shared repository still holds the RAW document graphs — this is what document GETs serve + assertTrue(ModelFactory.createModelForGraph(repository.get(BASE_URI)).isIsomorphicWith(base), "repository must keep serving the raw base ontology graph"); + assertTrue(ModelFactory.createModelForGraph(repository.get(IMPORT_URI)).isIsomorphicWith(imported), "repository must keep serving the raw imported ontology graph"); + // (d) REGRESSION GUARD: both owl:Class and rdfs:Class-only terms must be recognized as OntClasses by the model + // wrapped over the union, so GET /ns?forClass= resolves the class and runs its SPIN constructor. + // OntologyFilter promotes rdfs:Class subjects to owl:Class in a separate union member so the OWL2 profile + // (which does not recognize bare rdfs:Class) can find third-party vocab terms like sp:Describe. + OntModel ontology = OntModelFactory.createModel(union, OntSpecification.OWL2_FULL_MEM); assertNotNull(ontology.getOntClass(NS + "A"), "owl:Class term must be recognized as an OntClass under OWL2_FULL_MEM"); assertNotNull(ontology.getOntClass(NS + "B"), "rdfs:Class-only term must be recognized as an OntClass after promotion"); + // the promotion must not leak into the raw document graphs + assertFalse(ModelFactory.createModelForGraph(repository.get(IMPORT_URI)).contains(b, RDF.type, OWL.Class), "owl:Class promotion must not be written into the raw document graph"); + } + + @Test + public void testLoadOntologyToleratesImportCycles() + { + PrefixGraphRepository repository = new PrefixGraphRepository(null); + + String firstURI = "http://example.org/first"; + String secondURI = "http://example.org/second"; + Resource term = ResourceFactory.createResource(NS + "Term"); + + Model first = ModelFactory.createDefaultModel(); + Resource firstOnt = first.createResource(firstURI); + first.add(firstOnt, RDF.type, OWL.Ontology); + first.add(firstOnt, OWL.imports, first.createResource(secondURI)); + repository.put(firstURI, first.getGraph()); + + Model second = ModelFactory.createDefaultModel(); + Resource secondOnt = second.createResource(secondURI); + second.add(secondOnt, RDF.type, OWL.Ontology); + second.add(secondOnt, OWL.imports, second.createResource(firstURI)); + second.add(term, RDF.type, OWL.Class); + repository.put(secondURI, second.getGraph()); + + UnionGraph union = OntologyFilter.loadOntology(repository, firstURI); + assertTrue(ModelFactory.createModelForGraph(union).contains(term, RDF.type, OWL.Class), "cyclic imports must resolve without recursing infinitely"); } } diff --git a/src/test/java/com/atomgraph/linkeddatahub/server/util/SPINConstraintValidationTest.java b/src/test/java/com/atomgraph/linkeddatahub/server/util/SPINConstraintValidationTest.java index 6d2062f9c..c2a69790a 100644 --- a/src/test/java/com/atomgraph/linkeddatahub/server/util/SPINConstraintValidationTest.java +++ b/src/test/java/com/atomgraph/linkeddatahub/server/util/SPINConstraintValidationTest.java @@ -63,11 +63,8 @@ private OntModel loadOntology() "com/atomgraph/linkeddatahub/ldh.ttl" }) RDFDataMgr.read(closure, classpath); - // mirror OntologyFilter.loadOntology: RDFS-infer then materialize into a plain OWL2_FULL_MEM graph - OntModel inferred = OntModelFactory.createModel(closure.getGraph(), OntSpecification.OWL2_FULL_MEM_RDFS_INF); - OntModel materialized = OntModelFactory.createModel(OntSpecification.OWL2_FULL_MEM); - materialized.add(inferred); - return materialized; + // mirror OntologyFilter.loadOntology: a plain OWL2_FULL_MEM model over the assembled closure, no inference + return OntModelFactory.createModel(closure.getGraph(), OntSpecification.OWL2_FULL_MEM); } /** A dh:Item with NO dct:title — violates the MissingTitle constraint. */ From caa7b624dc1c2da62ddcd8dc9ff6368fbc56adb5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martynas=20Jusevi=C4=8Dius?= Date: Sun, 2 Aug 2026 14:36:27 +0200 Subject: [PATCH 2/7] Fix ScopedGraphRepository.contains() dropping resolvable imports MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ontapi consults GraphRepository.contains() before get() when resolving an ontology's imports closure. PrefixGraphRepository.contains() reports cache state (loaded graphs), not resolvability, so every import that resolves through a bundled location mapping (dh, sp, spin, foaf, sioc, sd), SPARQL-first loading or HTTP was answered with false on first resolution — and ontapi silently substituted an empty ontology graph for it (its ignoreUnresolvedImports fallback). The closure kept its shape but lost the content of every such import: SPIN constraints vanished, so validation enforced nothing (422 tests wrote through, eventually applying invalid dataspace settings and cascading into NPEs), and vocabulary term lookups came up empty. This is what failed the HTTP test suite in CI. contains() now attempts resolution through the backing repository (which loads and caches the graph) after the cache checks, reporting absent only for genuinely unresolvable ids. Adds a production-shaped regression test: ns# ontology importing the SPARQL-seeded ldh# vocabulary with its transitive imports resolved through the real bundled location mappings. Co-Authored-By: Claude Fable 5 --- .../server/util/ScopedGraphRepository.java | 16 ++- .../request/OntologyClosureCIReproTest.java | 101 ++++++++++++++++++ 2 files changed, 116 insertions(+), 1 deletion(-) create mode 100644 src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java b/src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java index c41190223..4fec524ed 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java +++ b/src/main/java/com/atomgraph/linkeddatahub/server/util/ScopedGraphRepository.java @@ -91,7 +91,21 @@ public void clear() @Override public boolean contains(String id) { - return local.containsKey(id) || getBacking().contains(id); + if (local.containsKey(id) || getBacking().contains(id)) return true; + + // the backing repository's contains() only reports already-cached graphs, but ontapi consults + // contains() before get() when resolving imports — a false negative for a resolvable id (bundled + // mapping, SPARQL-first, HTTP) makes ontapi silently substitute an empty ontology graph for the + // import. Attempt resolution instead: the backing repository loads and caches the graph, and only + // a genuinely unresolvable id reports absent + try + { + return getBacking().get(id) != null; + } + catch (RuntimeException ex) + { + return false; + } } @Override diff --git a/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java b/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java new file mode 100644 index 000000000..3a3bc03ac --- /dev/null +++ b/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java @@ -0,0 +1,101 @@ +/** + * Copyright 2026 Martynas Jusevičius + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + * + */ +package com.atomgraph.linkeddatahub.server.filter.request; + +import com.atomgraph.client.util.jena.PrefixGraphRepository; +import org.apache.jena.ontapi.UnionGraph; +import org.apache.jena.rdf.model.Model; +import org.apache.jena.rdf.model.ModelFactory; +import org.apache.jena.rdf.model.Resource; +import org.apache.jena.rdf.model.ResourceFactory; +import org.apache.jena.riot.RDFParser; +import org.apache.jena.vocabulary.OWL; +import org.apache.jena.vocabulary.RDF; +import org.apache.jena.vocabulary.RDFS; +import org.junit.jupiter.api.Test; +import static org.junit.jupiter.api.Assertions.*; + +/** + * Production-shaped regression guard for the imports closure: the end-user ns# ontology as the SPARQL + * CONSTRUCT returns it (header + imports + unrelated document resources), importing the SPARQL-loaded + * ldh# vocabulary, whose transitive imports resolve through the real bundled location mappings + * (dh, spin, sp, foaf, sioc, sd) plus store-seeded stubs for the non-mapped ones (ac, nfo, owl). + *

+ * Pins the fix for ontapi consulting {@code contains()} before {@code get()} during import resolution: + * a cache-state (rather than resolvability) answer made ontapi silently substitute empty ontology + * graphs for every bundled-mapped import, stripping SPIN constraints and vocabularies from the closure. + * + * @author Martynas Jusevičius {@literal } + */ +public class OntologyClosureCIReproTest +{ + + private static final String NS = "https://localhost:4443/ns#"; + private static final String LDH = "https://w3id.org/atomgraph/linkeddatahub#"; + + @Test + public void productionShapedClosureContainsAllImports() + { + PrefixGraphRepository repository = new PrefixGraphRepository(null); + + // real bundled mappings + Model mappingModel = ModelFactory.createDefaultModel(); + RDFParser.create().source("location-mapping.ttl").streamManager(repository.getStreamManager()).build().parse(mappingModel); + repository.processConfig(mappingModel); + + // mimic SPARQL-first load result: the ldh# vocabulary as a store graph + Model ldh = ModelFactory.createDefaultModel(); + RDFParser.create().source("com/atomgraph/linkeddatahub/ldh.ttl").base(LDH).streamManager(repository.getStreamManager()).build().parse(ldh); + repository.put(LDH, ldh.getGraph()); + + // stub graphs for ldh#'s non-mapped imports (SPARQL/HTTP-loaded in production) + for (String stub : new String[] { + "https://w3id.org/atomgraph/client#", + "http://www.semanticdesktop.org/ontologies/2007/03/22/nfo#", + "http://www.w3.org/2002/07/owl#" }) + { + Model m = ModelFactory.createDefaultModel(); + m.add(m.createResource(stub), RDF.type, OWL.Ontology); + repository.put(stub, m.getGraph()); + } + + // ns# base graph as the ontology CONSTRUCT returns it: ontology header + imports + document resource + Model ns = ModelFactory.createDefaultModel(); + Resource nsOnt = ns.createResource(NS); + ns.add(nsOnt, RDF.type, OWL.Ontology); + ns.add(nsOnt, OWL.imports, ns.createResource(LDH)); + Resource doc = ns.createResource("https://admin.localhost:4443/ontologies/namespace/"); + ns.add(doc, RDF.type, ns.createResource("https://www.w3.org/ns/ldt/document-hierarchy#Item")); + ns.add(doc, ResourceFactory.createProperty("http://purl.org/dc/terms/title"), "Namespace"); + repository.put(NS, ns.getGraph()); + + UnionGraph union = OntologyFilter.loadOntology(repository, NS); + Model closure = ModelFactory.createModelForGraph(union); + + // direct import: ldh.ttl content + assertTrue(closure.contains(closure.createResource(LDH + "View"), RDF.type, RDFS.Class), "ldh# (direct import) must be in the closure"); + // transitive via ldh#: dh.ttl (bundled mapping) + assertTrue(closure.contains(closure.createResource("https://www.w3.org/ns/ldt/document-hierarchy#Item"), RDF.type, OWL.Class), "dh# (transitive, bundled) must be in the closure"); + // transitive via ldh#: spin.ttl imported as http://spinrdf.org/spin (no hash) + assertTrue(closure.contains(closure.createResource("http://spinrdf.org/spin#constraint"), RDF.type, RDF.Property), "spin (transitive, bundled, hashless import URI) must be in the closure"); + // transitive via dh#: sp.ttl imported as http://spinrdf.org/sp# + assertTrue(closure.contains(closure.createResource("http://spinrdf.org/sp#text"), RDF.type, RDF.Property), "sp# (transitive via dh#, bundled) must be in the closure"); + // transitive via dh#: foaf (bundled) + assertFalse(closure.listStatements(closure.createResource("http://xmlns.com/foaf/0.1/Agent"), null, (org.apache.jena.rdf.model.RDFNode)null).toList().isEmpty(), "foaf (transitive, bundled) must be in the closure"); + } + +} From feb004fd356e7b57bd8df9e932de294e540b958a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martynas=20Jusevi=C4=8Dius?= Date: Sun, 2 Aug 2026 14:48:56 +0200 Subject: [PATCH 3/7] Alias only repository-held closure ids under their document URIs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ontapi keys imports under their declared ontology IRIs, which need not be repository entries: a content-addressed upload is cached under its uploads/ URI while declaring a foreign ontology IRI. The doc-URI aliasing loop called repository.get() on such declared IRIs, which fell through to an HTTP dereference of the foreign IRI (e.g. https://example.org/test) during ontology load — failing the load and the ontology-import-upload-no-deadlock HTTP test. Guard the loop with isCached() so only graphs the shared repository actually holds get aliased. Adds a mismatched-IRI import case to the closure regression test. Co-Authored-By: Claude Fable 5 --- .../server/filter/request/OntologyFilter.java | 6 +++- .../request/OntologyClosureCIReproTest.java | 30 +++++++++++++++++++ 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java index 653012665..319cf68e3 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java +++ b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyFilter.java @@ -190,8 +190,12 @@ public static UnionGraph loadOntology(PrefixGraphRepository repository, String u Model promotions = ModelFactory.createDefaultModel(); ontology.listSubjectsWithProperty(RDF.type, RDFS.Class).forEach(r -> promotions.add(r, RDF.type, OWL.Class)); if (!promotions.isEmpty()) union.addSubGraph(promotions.getGraph()); - // cache closure graphs under their fragment-stripped document URIs too + // cache closure graphs under their fragment-stripped document URIs too. ontapi keys imports under + // their declared ontology IRIs, which need not be repository entries (a content-addressed upload is + // cached under its uploads/ URI while declaring a foreign ontology IRI) — only alias ids the shared + // repository actually holds, lest the lookup dereference a foreign IRI over HTTP scoped.ids().filter(closureURI -> closureURI.startsWith("http://") || closureURI.startsWith("https://")). + filter(repository::isCached). forEach(closureURI -> addDocumentModel(repository, closureURI)); if (log.isDebugEnabled()) log.debug("Finished loading ontology with URI '{}'", uri); return union; diff --git a/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java b/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java index 3a3bc03ac..6cf9bfd2a 100644 --- a/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java +++ b/src/test/java/com/atomgraph/linkeddatahub/server/filter/request/OntologyClosureCIReproTest.java @@ -98,4 +98,34 @@ public void productionShapedClosureContainsAllImports() assertFalse(closure.listStatements(closure.createResource("http://xmlns.com/foaf/0.1/Agent"), null, (org.apache.jena.rdf.model.RDFNode)null).toList().isEmpty(), "foaf (transitive, bundled) must be in the closure"); } + @Test + public void importedDocumentWithMismatchedOntologyIRIIsInTheClosure() + { + // content-addressed uploads: the document URI (uploads/) necessarily differs from the + // ontology IRI the uploaded file declares — both must still land in the closure + String uploadURI = "https://localhost:4443/uploads/da39a3ee5e6b4b0d3255bfef95601890afd80709"; + String declaredURI = "https://example.org/test#"; + + PrefixGraphRepository repository = new PrefixGraphRepository(null); + + Model uploaded = ModelFactory.createDefaultModel(); + Resource declaredOnt = uploaded.createResource(declaredURI); + uploaded.add(declaredOnt, RDF.type, OWL.Ontology); + Resource testClass = uploaded.createResource(declaredURI + "TestClass"); + uploaded.add(testClass, RDF.type, OWL.Class); + uploaded.add(testClass, RDFS.label, "Test Class"); + repository.put(uploadURI, uploaded.getGraph()); + + Model ns = ModelFactory.createDefaultModel(); + Resource nsOnt = ns.createResource(NS); + ns.add(nsOnt, RDF.type, OWL.Ontology); + ns.add(nsOnt, OWL.imports, ns.createResource(uploadURI)); + repository.put(NS, ns.getGraph()); + + UnionGraph union = OntologyFilter.loadOntology(repository, NS); + Model closure = ModelFactory.createModelForGraph(union); + + assertTrue(closure.contains(testClass, RDFS.label, closure.createLiteral("Test Class")), "content of an import whose declared ontology IRI differs from its document URI must be in the closure"); + } + } From 15a72533cd1dbdc8ec89e9167187bd623fa9c80a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martynas=20Jusevi=C4=8Dius?= Date: Sun, 2 Aug 2026 15:01:57 +0200 Subject: [PATCH 4/7] Fix GET-proxied-ontology-ns.sh assertions for closure DESCRIBE semantics MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The admin ontologies/namespace/ document stores the ontology but the closure keys it under the ontology URI, so a proxied GET of the document URI is answered by the closure DESCRIBE fallback — the document's own #-fragment term descriptions — not the raw graph branch. Assert on a class minted in the document's hash namespace (mirroring the original #related_View regression) instead of the made-up-namespace classes, which only appear under their own namespace document URI. Co-Authored-By: Claude Fable 5 --- http-tests/proxy/GET-proxied-ontology-ns.sh | 24 +++++++++++++++------ 1 file changed, 17 insertions(+), 7 deletions(-) diff --git a/http-tests/proxy/GET-proxied-ontology-ns.sh b/http-tests/proxy/GET-proxied-ontology-ns.sh index 152360441..8efdf65b8 100755 --- a/http-tests/proxy/GET-proxied-ontology-ns.sh +++ b/http-tests/proxy/GET-proxied-ontology-ns.sh @@ -20,9 +20,11 @@ namespace_uri="http://made-up-test-ns.example/ns" class1="${namespace_uri}#ClassOne" class2="${namespace_uri}#ClassTwo" ontology_doc="${ADMIN_BASE_URL}ontologies/namespace/" +class3="${ontology_doc}#ClassThree" namespace="${END_USER_BASE_URL}ns#" -# add two classes with URIs in the made-up namespace to the app's ontology +# add two classes with URIs in the made-up namespace to the app's ontology, +# plus one in the ontology document's own hash namespace add-class.sh \ -f "$OWNER_CERT_FILE" \ @@ -40,6 +42,14 @@ add-class.sh \ --label "Class Two" \ "$ontology_doc" +add-class.sh \ + -f "$OWNER_CERT_FILE" \ + -p "$OWNER_CERT_PWD" \ + -b "$ADMIN_BASE_URL" \ + --uri "$class3" \ + --label "Class Three" \ + "$ontology_doc" + # clear the in-memory ontology so the new classes are present on next request clear-ontology.sh \ @@ -66,10 +76,11 @@ echo "$response" | grep -q "$class1" echo "$response" | grep -q "$class2" ! echo "$response" | grep -q "http://www.w3.org/2000/01/rdf-schema#Resource" -# request the ontology document itself via ?uri= proxy: it is a graph in the ontology closure, -# so it must be served with its raw graph — asserted triples only, identical to a direct -# document GET. Inferred rdf:type rdfs:Resource used to leak from the RDFS-materialized -# in-memory model here, breaking client-side @typeof matching of View blocks. +# request the ontology document itself via ?uri= proxy. The ontology is stored in this admin +# document but keyed in the closure under the ontology URI, so this too is answered by the +# closure DESCRIBE fallback: descriptions of the document's own #-fragment terms, asserted triples +# only. Inferred rdf:type rdfs:Resource used to leak from the RDFS-materialized in-memory model +# here, producing multi-token @typeof that broke client-side matching of View blocks. doc_response=$(curl -k -f -s \ -G \ @@ -78,6 +89,5 @@ doc_response=$(curl -k -f -s \ --data-urlencode "uri=${ontology_doc}" \ "$END_USER_BASE_URL") -echo "$doc_response" | grep -q "$class1" -echo "$doc_response" | grep -q "$class2" +echo "$doc_response" | grep -q "$class3" ! echo "$doc_response" | grep -q "http://www.w3.org/2000/01/rdf-schema#Resource" From 30ca2191fcd057f1423b33b26a2a016f6ff6a24a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martynas=20Jusevi=C4=8Dius?= Date: Thu, 6 Aug 2026 12:55:42 +0200 Subject: [PATCH 5/7] Remove Linked Data proxy DESCRIBE-over-closure fallback The proxy is document-keyed transport; term lookups over the ontology closure belong on /ns (SPARQL), not on a DESCRIBE synthesized from the proxy target URI. Drop the fallback and its ParameterizedSparqlString/ QueryExecution imports; the isCached closure-cache branch (raw per-doc graphs, no inference) still serves closure documents. Realign tests: delete GET-proxied-ontology-ns.sh (it only exercised the removed DESCRIBE path and dereferenced ontology terms via the admin doc URI, which was never a supported path). Add GET-ns-no-query.sh (raw /ns ontology graph, no rdfs:Resource leak) and GET-proxied-mapped-vocab.sh (dct:title, foaf:Person, skos:Concept served from the static prefix mapping; skos also covers fragment-strip + xml:base resolution). Co-Authored-By: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 3 + http-tests/proxy/GET-proxied-mapped-vocab.sh | 58 ++++++++++++ http-tests/proxy/GET-proxied-ontology-ns.sh | 93 ------------------- .../sparql-protocol/query/GET-ns-no-query.sh | 42 +++++++++ .../filter/request/ProxyRequestFilter.java | 23 +---- 5 files changed, 104 insertions(+), 115 deletions(-) create mode 100755 http-tests/proxy/GET-proxied-mapped-vocab.sh delete mode 100755 http-tests/proxy/GET-proxied-ontology-ns.sh create mode 100755 http-tests/sparql-protocol/query/GET-ns-no-query.sh diff --git a/CHANGELOG.md b/CHANGELOG.md index ad83ac92f..109d3ac07 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,9 @@ ### Fixed - Raw ontology graphs no longer leak inferred `rdf:type rdfs:Resource`, which produced multi-token `@typeof` that broke View block rendering +### Removed +- The Linked Data proxy no longer serves ontology terms; it is now dumb transport (bundled-vocab file cache + SSRF-checked external fetch), with ontology terms served by `/ns` + ## [5.7.1] - 2026-08-06 ### Changed - RDFa editor: annotation overlay rebuilt on demand (`rdfa-editor/overlay.xsl`) diff --git a/http-tests/proxy/GET-proxied-mapped-vocab.sh b/http-tests/proxy/GET-proxied-mapped-vocab.sh new file mode 100755 index 000000000..d7577a33f --- /dev/null +++ b/http-tests/proxy/GET-proxied-mapped-vocab.sh @@ -0,0 +1,58 @@ +#!/usr/bin/env bash +set -euo pipefail + +initialize_dataset "$END_USER_BASE_URL" "$TMP_END_USER_DATASET" "$END_USER_ENDPOINT_URL" +initialize_dataset "$ADMIN_BASE_URL" "$TMP_ADMIN_DATASET" "$ADMIN_ENDPOINT_URL" +purge_cache "$END_USER_VARNISH_SERVICE" +purge_cache "$ADMIN_VARNISH_SERVICE" +purge_cache "$FRONTEND_VARNISH_SERVICE" + +# add agent to the readers group to be able to read documents + +add-agent-to-group.sh \ + -f "$OWNER_CERT_FILE" \ + -p "$OWNER_CERT_PWD" \ + --agent "$AGENT_URI" \ + "${ADMIN_BASE_URL}acl/groups/readers/" + +# well-known vocab terms that are statically prefix-mapped to bundled documents +# (src/main/resources/prefix-mapping.ttl), so the proxy serves them straight from +# that cache (isMapped branch) instead of dereferencing the network. + +# dct:title - slash-based namespace (http://purl.org/dc/terms/); the proxy request +# URI equals the term URI itself + +dct_response=$(curl -k -f -s \ + -G \ + -E "$AGENT_CERT_FILE":"$AGENT_CERT_PWD" \ + -H "Accept: application/n-triples" \ + --data-urlencode "uri=http://purl.org/dc/terms/title" \ + "$END_USER_BASE_URL") + +echo "$dct_response" | grep -q ' "Title"' + +# foaf:Person - also slash-based (http://xmlns.com/foaf/0.1/) + +foaf_response=$(curl -k -f -s \ + -G \ + -E "$AGENT_CERT_FILE":"$AGENT_CERT_PWD" \ + -H "Accept: application/n-triples" \ + --data-urlencode "uri=http://xmlns.com/foaf/0.1/Person" \ + "$END_USER_BASE_URL") + +echo "$foaf_response" | grep -q ' "Person"' + +# skos:Concept - hash-based namespace (http://www.w3.org/2004/02/skos/core#); the +# request carries a #fragment that ProxyRequestFilter strips before matching the +# mapped prefix, and the bundled document declares terms as relative (#Concept) +# under its own xml:base, so this also confirms that base resolves back to the +# full hash URI rather than leaking a bare fragment or the classpath location + +skos_response=$(curl -k -f -s \ + -G \ + -E "$AGENT_CERT_FILE":"$AGENT_CERT_PWD" \ + -H "Accept: application/n-triples" \ + --data-urlencode "uri=http://www.w3.org/2004/02/skos/core#Concept" \ + "$END_USER_BASE_URL") + +echo "$skos_response" | grep -q ' "Concept"' diff --git a/http-tests/proxy/GET-proxied-ontology-ns.sh b/http-tests/proxy/GET-proxied-ontology-ns.sh deleted file mode 100755 index 8efdf65b8..000000000 --- a/http-tests/proxy/GET-proxied-ontology-ns.sh +++ /dev/null @@ -1,93 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -initialize_dataset "$END_USER_BASE_URL" "$TMP_END_USER_DATASET" "$END_USER_ENDPOINT_URL" -initialize_dataset "$ADMIN_BASE_URL" "$TMP_ADMIN_DATASET" "$ADMIN_ENDPOINT_URL" -purge_cache "$END_USER_VARNISH_SERVICE" -purge_cache "$ADMIN_VARNISH_SERVICE" -purge_cache "$FRONTEND_VARNISH_SERVICE" - -# add agent to the readers group to be able to read documents - -add-agent-to-group.sh \ - -f "$OWNER_CERT_FILE" \ - -p "$OWNER_CERT_PWD" \ - --agent "$AGENT_URI" \ - "${ADMIN_BASE_URL}acl/groups/readers/" - -# use a made-up hash-based namespace: not mapped as a static file, not a registered app -namespace_uri="http://made-up-test-ns.example/ns" -class1="${namespace_uri}#ClassOne" -class2="${namespace_uri}#ClassTwo" -ontology_doc="${ADMIN_BASE_URL}ontologies/namespace/" -class3="${ontology_doc}#ClassThree" -namespace="${END_USER_BASE_URL}ns#" - -# add two classes with URIs in the made-up namespace to the app's ontology, -# plus one in the ontology document's own hash namespace - -add-class.sh \ - -f "$OWNER_CERT_FILE" \ - -p "$OWNER_CERT_PWD" \ - -b "$ADMIN_BASE_URL" \ - --uri "$class1" \ - --label "Class One" \ - "$ontology_doc" - -add-class.sh \ - -f "$OWNER_CERT_FILE" \ - -p "$OWNER_CERT_PWD" \ - -b "$ADMIN_BASE_URL" \ - --uri "$class2" \ - --label "Class Two" \ - "$ontology_doc" - -add-class.sh \ - -f "$OWNER_CERT_FILE" \ - -p "$OWNER_CERT_PWD" \ - -b "$ADMIN_BASE_URL" \ - --uri "$class3" \ - --label "Class Three" \ - "$ontology_doc" - -# clear the in-memory ontology so the new classes are present on next request - -clear-ontology.sh \ - -f "$OWNER_CERT_FILE" \ - -p "$OWNER_CERT_PWD" \ - -b "$ADMIN_BASE_URL" \ - --ontology "$namespace" - -# request the namespace document URI (without fragment) via ?uri= proxy. -# the namespace document is not DataManager-mapped, not a registered app and not a graph -# in the ontology closure, so ProxyRequestFilter falls through to the closure DESCRIBE -# fallback, which returns descriptions of all #-fragment terms in that namespace. - -response=$(curl -k -f -s \ - -G \ - -E "$AGENT_CERT_FILE":"$AGENT_CERT_PWD" \ - -H "Accept: application/n-triples" \ - --data-urlencode "uri=${namespace_uri}" \ - "$END_USER_BASE_URL") - -# verify both class descriptions are present in the response and no inferred triples leak - -echo "$response" | grep -q "$class1" -echo "$response" | grep -q "$class2" -! echo "$response" | grep -q "http://www.w3.org/2000/01/rdf-schema#Resource" - -# request the ontology document itself via ?uri= proxy. The ontology is stored in this admin -# document but keyed in the closure under the ontology URI, so this too is answered by the -# closure DESCRIBE fallback: descriptions of the document's own #-fragment terms, asserted triples -# only. Inferred rdf:type rdfs:Resource used to leak from the RDFS-materialized in-memory model -# here, producing multi-token @typeof that broke client-side matching of View blocks. - -doc_response=$(curl -k -f -s \ - -G \ - -E "$AGENT_CERT_FILE":"$AGENT_CERT_PWD" \ - -H "Accept: application/n-triples" \ - --data-urlencode "uri=${ontology_doc}" \ - "$END_USER_BASE_URL") - -echo "$doc_response" | grep -q "$class3" -! echo "$doc_response" | grep -q "http://www.w3.org/2000/01/rdf-schema#Resource" diff --git a/http-tests/sparql-protocol/query/GET-ns-no-query.sh b/http-tests/sparql-protocol/query/GET-ns-no-query.sh new file mode 100755 index 000000000..c2e99f759 --- /dev/null +++ b/http-tests/sparql-protocol/query/GET-ns-no-query.sh @@ -0,0 +1,42 @@ +#!/usr/bin/env bash +set -euo pipefail + +initialize_dataset "$END_USER_BASE_URL" "$TMP_END_USER_DATASET" "$END_USER_ENDPOINT_URL" +initialize_dataset "$ADMIN_BASE_URL" "$TMP_ADMIN_DATASET" "$ADMIN_ENDPOINT_URL" +purge_cache "$END_USER_VARNISH_SERVICE" +purge_cache "$ADMIN_VARNISH_SERVICE" +purge_cache "$FRONTEND_VARNISH_SERVICE" + +# add a class to the app's namespace ontology + +namespace_doc="${END_USER_BASE_URL}ns" +namespace="${namespace_doc}#" +ontology_doc="${ADMIN_BASE_URL}ontologies/namespace/" +class="${namespace}ClassThree" + +add-class.sh \ + -f "$OWNER_CERT_FILE" \ + -p "$OWNER_CERT_PWD" \ + -b "$ADMIN_BASE_URL" \ + --uri "$class" \ + --label "Class Three" \ + "$ontology_doc" + +# clear ontology from memory so the new class is loaded on next request + +clear-ontology.sh \ + -f "$OWNER_CERT_FILE" \ + -p "$OWNER_CERT_PWD" \ + -b "$ADMIN_BASE_URL" \ + --ontology "$namespace" + +# GET with no ?query= should return the raw namespace ontology graph (asserted +# triples only, no RDFS materialization) rather than run a SPARQL query + +response=$(curl -k -f -s \ + -E "$OWNER_CERT_FILE":"$OWNER_CERT_PWD" \ + -H "Accept: application/n-triples" \ + "$namespace_doc") + +echo "$response" | grep -q "$class" +! echo "$response" | grep -q "http://www.w3.org/2000/01/rdf-schema#Resource" diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java index 4a87ab3a0..71eddd0dc 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java +++ b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java @@ -41,8 +41,6 @@ import java.util.List; import java.util.Optional; import java.util.Set; -import org.apache.jena.query.ParameterizedSparqlString; -import org.apache.jena.query.QueryExecution; import jakarta.annotation.Priority; import jakarta.inject.Inject; import jakarta.ws.rs.NotAllowedException; @@ -206,25 +204,6 @@ public void filter(ContainerRequestContext requestContext) throws IOException requestContext.abortWith(getResponse(model, Response.Status.OK)); return; } - - // fall back to DESCRIBE over the in-memory closure union (asserted triples only — no inference) - // for term URIs whose document is not a closure graph: terms minted in external namespaces but - // defined in the app ontology. Covers both slash-based term URIs (e.g. schema:category) and - // hash-based namespaces (e.g. sioc:UserAccount → ac:document-uri strips to sioc:ns, so we also - // describe all ?term where STR(?term) starts with "#") - ParameterizedSparqlString pss = new ParameterizedSparqlString( - "DESCRIBE ?doc ?term WHERE { ?term ?p ?o FILTER(STRSTARTS(STR(?term), CONCAT(STR(?doc), \"#\"))) }"); - pss.setIri("doc", targetURI.toString()); - try (QueryExecution qe = QueryExecution.create(pss.asQuery(), getOntology().get())) - { - Model description = qe.execDescribe(); - if (!description.isEmpty()) - { - if (log.isDebugEnabled()) log.debug("Serving URI from namespace ontology: {}", targetURI); - requestContext.abortWith(getResponse(description, Response.Status.OK)); - return; - } - } } boolean isRegisteredApp = getSystem().matchApp(targetURI) != null; @@ -439,7 +418,7 @@ private Response overlayHeaders(Response response, Response clientResponse, bool /** * Builds a response for the given RDF model with type-appropriate content negotiation. - * Used for locally-served responses (DataManager cache, namespace ontology DESCRIBE) and for + * Used for locally-served responses (DataManager cache, ontology closure cache) and for * the proxy's Model branch. * * @param model RDF model From e87c78f0ff3a0f36aee57bf156ecf8639e09178d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martynas=20Jusevi=C4=8Dius?= Date: Fri, 7 Aug 2026 15:58:20 +0200 Subject: [PATCH 6/7] Make the Linked Data proxy dumb: drop ontology-closure serving Remove the getOntology()/isCached branch that answered ?uri= requests from the app ontology owl:imports closure cache, plus the now-unused ontology injection, getOntology(), and PrefixGraphRepository/Application/EndUserApplication/ OntModel imports. The proxy is now dumb transport: bundled-vocab file cache (isMapped) + SSRF-checked external fetch. Ontology terms are served by /ns. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../filter/request/ProxyRequestFilter.java | 36 +------------------ 1 file changed, 1 insertion(+), 35 deletions(-) diff --git a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java index 71eddd0dc..da1050384 100644 --- a/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java +++ b/src/main/java/com/atomgraph/linkeddatahub/server/filter/request/ProxyRequestFilter.java @@ -21,11 +21,7 @@ import com.atomgraph.client.vocabulary.AC; import com.atomgraph.core.exception.BadGatewayException; import com.atomgraph.core.util.ModelUtils; -import com.atomgraph.client.util.jena.PrefixGraphRepository; -import com.atomgraph.linkeddatahub.apps.model.Application; import com.atomgraph.linkeddatahub.apps.model.Dataset; -import com.atomgraph.linkeddatahub.apps.model.EndUserApplication; -import org.apache.jena.ontapi.model.OntModel; import com.atomgraph.linkeddatahub.client.GraphStoreClient; import com.atomgraph.linkeddatahub.client.filter.auth.IDTokenDelegationFilter; import com.atomgraph.linkeddatahub.client.filter.auth.WebIDDelegationFilter; @@ -134,7 +130,6 @@ public class ProxyRequestFilter implements ContainerRequestFilter "Age"); @Inject com.atomgraph.linkeddatahub.Application system; - @Inject jakarta.inject.Provider> ontology; @Inject MediaTypes mediaTypes; @Context Request request; @@ -187,25 +182,6 @@ public void filter(ContainerRequestContext requestContext) throws IOException return; } - if (isSafeMethod && getOntology().isPresent()) - { - // serve documents of the app's ontology imports closure with their raw graphs — asserted triples - // only, identical to a direct document GET. Hash-based term URIs (e.g. ldh:View) arrive here - // fragment-stripped as their document URI. The isCached guard restricts serving to graphs already - // loaded as part of the closure, so arbitrary external URIs cannot trigger repository loading - // outside the SSRF-validated external client path below. - Optional appOpt = (Optional)requestContext.getProperty(LAPP.Application.getURI()); - PrefixGraphRepository repository = appOpt != null && appOpt.isPresent() && appOpt.get().canAs(EndUserApplication.class) ? - getSystem().getRepository(appOpt.get().as(EndUserApplication.class)) : getSystem().getRepository(); - if (repository.isCached(targetURI.toString())) - { - if (log.isDebugEnabled()) log.debug("Serving URI from the ontology closure cache: {}", targetURI); - Model model = org.apache.jena.rdf.model.ModelFactory.createModelForGraph(repository.get(targetURI.toString())); - requestContext.abortWith(getResponse(model, Response.Status.OK)); - return; - } - } - boolean isRegisteredApp = getSystem().matchApp(targetURI) != null; if (!isRegisteredApp && !getSystem().isEnableLinkedDataProxy()) throw new NotAllowedException("Linked Data proxy not enabled"); @@ -418,7 +394,7 @@ private Response overlayHeaders(Response response, Response clientResponse, bool /** * Builds a response for the given RDF model with type-appropriate content negotiation. - * Used for locally-served responses (DataManager cache, ontology closure cache) and for + * Used for locally-served responses (DataManager cache) and for * the proxy's Model branch. * * @param model RDF model @@ -480,16 +456,6 @@ public com.atomgraph.linkeddatahub.Application getSystem() return system; } - /** - * Returns the current application's namespace ontology, if available. - * - * @return optional ontology - */ - public Optional getOntology() - { - return ontology.get(); - } - /** * Returns the media types registry used for content negotiation and outbound {@code Accept} headers. * From 538ec9d41c19b64e645c56d4493cc1987e013b31 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martynas=20Jusevi=C4=8Dius?= Date: Fri, 7 Aug 2026 18:16:33 +0200 Subject: [PATCH 7/7] Fix GET-proxied-mapped-vocab SIGPIPE on large vocab responses MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The assertions piped the whole vocabulary graph through `echo | grep -q`. `grep -q` exits on first match and closes the pipe while `echo` is still writing; with `set -o pipefail` the SIGPIPE'd `echo` (write error: broken pipe, exit 141) fails the pipeline whenever the response exceeds the ~64 KiB pipe buffer — so a *successful* match killed the test. Only this test trips it, being the only one that returns entire vocabulary documents (40-113 KiB). Read from a here-string instead (temp file, no pipe to break), and match the full language-tagged label literal the bundled documents actually carry. Co-Authored-By: Claude Opus 4.8 (1M context) --- http-tests/proxy/GET-proxied-mapped-vocab.sh | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/http-tests/proxy/GET-proxied-mapped-vocab.sh b/http-tests/proxy/GET-proxied-mapped-vocab.sh index d7577a33f..f2250e07a 100755 --- a/http-tests/proxy/GET-proxied-mapped-vocab.sh +++ b/http-tests/proxy/GET-proxied-mapped-vocab.sh @@ -18,6 +18,13 @@ add-agent-to-group.sh \ # well-known vocab terms that are statically prefix-mapped to bundled documents # (src/main/resources/prefix-mapping.ttl), so the proxy serves them straight from # that cache (isMapped branch) instead of dereferencing the network. +# +# The whole vocabulary graph is returned (tens of KiB), so assertions read from a +# here-string rather than `echo "$response" | grep -q`: `grep -q` closes the pipe on +# first match, and with `set -o pipefail` the SIGPIPE'd `echo` (write error: broken +# pipe) fails the whole pipeline whenever the response exceeds the ~64 KiB pipe buffer. +# Labels are language-tagged in the bundled documents, so the expected literal is +# matched in full including its tag. # dct:title - slash-based namespace (http://purl.org/dc/terms/); the proxy request # URI equals the term URI itself @@ -29,9 +36,9 @@ dct_response=$(curl -k -f -s \ --data-urlencode "uri=http://purl.org/dc/terms/title" \ "$END_USER_BASE_URL") -echo "$dct_response" | grep -q ' "Title"' +grep -qF ' "Title"@en-US' <<< "$dct_response" -# foaf:Person - also slash-based (http://xmlns.com/foaf/0.1/) +# foaf:Person - also slash-based (http://xmlns.com/foaf/0.1/); label is a plain literal foaf_response=$(curl -k -f -s \ -G \ @@ -40,7 +47,7 @@ foaf_response=$(curl -k -f -s \ --data-urlencode "uri=http://xmlns.com/foaf/0.1/Person" \ "$END_USER_BASE_URL") -echo "$foaf_response" | grep -q ' "Person"' +grep -qF ' "Person"' <<< "$foaf_response" # skos:Concept - hash-based namespace (http://www.w3.org/2004/02/skos/core#); the # request carries a #fragment that ProxyRequestFilter strips before matching the @@ -55,4 +62,4 @@ skos_response=$(curl -k -f -s \ --data-urlencode "uri=http://www.w3.org/2004/02/skos/core#Concept" \ "$END_USER_BASE_URL") -echo "$skos_response" | grep -q ' "Concept"' +grep -qF ' "Concept"@en' <<< "$skos_response"