Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions build.gradle.kts
Original file line number Diff line number Diff line change
Expand Up @@ -550,6 +550,11 @@ if (nativeBuild) {
"--initialize-at-build-time=org.junit.platform.launcher",
"--initialize-at-build-time=org.junit.platform.engine",
"--initialize-at-build-time=org.junit.jupiter.engine.descriptor",
// Spring AOT registers every *.json resource, which covers shows.json;
// the CSV, XML and Markdown flavours of the sample dataset need an
// explicit include or ShowsSampleDataTest cannot load them natively.
"-H:IncludeResources=shows\\.(csv|xml)$",
"-H:IncludeResources=shows-markdown/.*\\.md$",
)
}
}
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -313,11 +313,12 @@ public String indexCsvDocuments(@McpToolParam(description = "Solr collection to
* <strong>Supported XML Formats:</strong>
*
* <ul>
* <li><strong>Single Document</strong>: Root element treated as one document
* <li><strong>Multiple Documents</strong>: Child elements with 'doc', 'item',
* or 'record' names treated as separate documents
* <li><strong>Nested Elements</strong>: Automatically flattened with underscore
* notation
* <li><strong>Single Document</strong>: Root element treated as one document;
* its child elements are the fields
* <li><strong>Multiple Documents</strong>: Repeated child elements of the root
* (any name) are separate documents; each one's child elements are its fields
* <li><strong>Nested Elements</strong>: Flattened below the field with
* underscore notation ({@code author_name})
* <li><strong>Attributes</strong>: Converted to fields with "_attr" suffix
* <li><strong>Mixed Data Types</strong>: Automatic type detection by Solr
* </ul>
Expand Down Expand Up @@ -387,9 +388,12 @@ public String indexCsvDocuments(@McpToolParam(description = "Solr collection to
@McpTool(
name = "index-xml-documents",
annotations = @McpTool.McpAnnotations(idempotentHint = true),
description = "Index documents from XML string into Solr collection. Element names are"
+ " sanitized for Solr compatibility (lowercased, special characters replaced"
+ " with underscores); the response lists the field names as indexed")
description = "Index documents from XML string into Solr collection: repeated child elements of the"
+ " root are the documents and each one's child elements are its fields (<show><id>x</id>"
+ "<title>T</title></show> yields id and title; nested elements flatten with underscores,"
+ " attributes get an _attr suffix). Element names are sanitized for Solr compatibility"
+ " (lowercased, special characters replaced with underscores); the response lists the field"
+ " names as indexed")
public String indexXmlDocuments(@McpToolParam(description = "Solr collection to index into") String collection,
@McpToolParam(description = "XML string containing documents to index") String xml)
throws ParserConfigurationException, SAXException, IOException, SolrServerException {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -58,7 +58,10 @@ public XmlDocumentCreator() {
* This method parses the XML and creates documents based on the structure: - If
* the XML has multiple child elements with the same tag name (indicating
* repeated structures), each child element becomes a separate document -
* Otherwise, the entire XML structure is treated as a single document
* Otherwise, the entire XML structure is treated as a single document. In both
* cases the record element is a wrapper: its child elements are the fields,
* named after themselves, so {@code <show><id>x</id><title>T</title></show>}
* yields {@code id} and {@code title}, not {@code show_id}.
*
* <p>
* This approach is flexible and doesn't rely on hardcoded element names,
Expand Down Expand Up @@ -147,7 +150,7 @@ private List<SolrInputDocument> createDocumentsFromChildren(List<Element> childE

for (Element childElement : childElements) {
SolrInputDocument solrDoc = new SolrInputDocument();
addXmlElementFields(solrDoc, childElement, "");
addRecordFields(solrDoc, childElement);
if (!solrDoc.isEmpty()) {
documents.add(solrDoc);
}
Expand All @@ -156,11 +159,34 @@ private List<SolrInputDocument> createDocumentsFromChildren(List<Element> childE
return documents;
}

/**
* Adds the fields of one record element. The record element itself (the root
* for a single document, or each repeated child) is a wrapper, not a field: its
* child elements become fields named after themselves ({@code <title>} is
* {@code title}, a repeated {@code <genres>} is multi-valued), nested elements
* flatten below that with underscores, and the record's own attributes keep the
* {@code _attr} suffix.
*/
private void addRecordFields(SolrInputDocument doc, Element record) {
String recordName = FieldNameSanitizer.sanitizeFieldName(record.getTagName());
// The record's own attributes are unqualified (id_attr); a child element's
// attributes are qualified by that element (name_lang_attr).
processXmlAttributes(doc, record, "", "");
NodeList children = record.getChildNodes();
processXmlTextContent(doc, recordName, recordName, "", hasChildElements(children), children);
for (int i = 0; i < children.getLength(); i++) {
Node child = children.item(i);
if (child.getNodeType() == Node.ELEMENT_NODE) {
addXmlElementFields(doc, (Element) child, "");
}
}
}

/** Creates a single document from the root element. */
private List<SolrInputDocument> createSingleDocument(Element rootElement) {
List<SolrInputDocument> documents = new ArrayList<>();
SolrInputDocument solrDoc = new SolrInputDocument();
addXmlElementFields(solrDoc, rootElement, "");
addRecordFields(solrDoc, rootElement);

if (!solrDoc.isEmpty()) {
documents.add(solrDoc);
Expand Down Expand Up @@ -229,7 +255,7 @@ private void processXmlAttributes(SolrInputDocument doc, Element element, String
for (int i = 0; i < element.getAttributes().getLength(); i++) {
Node attr = element.getAttributes().item(i);
String attrName = FieldNameSanitizer.sanitizeFieldName(attr.getNodeName()) + "_attr";
String fieldName = prefix.isEmpty() ? attrName : currentPrefix + "_" + attrName;
String fieldName = currentPrefix.isEmpty() ? attrName : currentPrefix + "_" + attrName;
String attrValue = attr.getNodeValue();

if (attrValue != null && !attrValue.trim().isEmpty()) {
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,138 @@
/*
* Licensed to the Apache Software Foundation (ASF) under one or more
* contributor license agreements. See the NOTICE file distributed with
* this work for additional information regarding copyright ownership.
* The ASF licenses this file to You under the Apache License, Version 2.0
* (the "License"); you may not use this file except in compliance with
* the License. You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/
package org.apache.solr.mcp.server.indexing;

import static org.assertj.core.api.Assertions.assertThat;

import com.fasterxml.jackson.databind.ObjectMapper;
import java.io.IOException;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.TreeMap;
import org.apache.solr.common.SolrInputDocument;
import org.apache.solr.mcp.server.indexing.documentcreator.CsvDocumentCreator;
import org.apache.solr.mcp.server.indexing.documentcreator.JsonDocumentCreator;
import org.apache.solr.mcp.server.indexing.documentcreator.MarkdownDocumentCreator;
import org.apache.solr.mcp.server.indexing.documentcreator.XmlDocumentCreator;
import org.junit.jupiter.api.Test;

/**
* The {@code shows} sample dataset ships in every format the indexing tools
* accept: {@code shows.json}, {@code shows.csv}, {@code shows.xml} and one
* Markdown file per show under {@code shows-markdown/}. These tests pin that
* the four representations parse to the same 61 documents, so a tutorial step
* written against one format holds for the others.
*
* <p>
* Two representation choices are worth knowing:
* <ul>
* <li>CSV carries multi-valued fields as <em>repeated column headers</em>
* ({@code genres,genres,genres}); the parser adds one value per non-empty cell
* under the same field name.</li>
* <li>XML uses one {@code <show>} element per record; the record element is a
* wrapper, so its children become the fields ({@code id}, {@code title}, a
* repeated {@code <genres>} is multi-valued), the same names as JSON.</li>
* </ul>
*/
class ShowsSampleDataTest {

private static final int SHOWS = 61;

private final JsonDocumentCreator json = new JsonDocumentCreator(new ObjectMapper());

@Test
void jsonHas61ShowsWithUniqueIds() throws Exception {
Map<String, Map<String, List<String>>> shows = byId(json.create(resource("/shows.json")), "id");

assertThat(shows).hasSize(SHOWS);
assertThat(shows.get("netflix-001")).containsEntry("title", List.of("Stranger Things")).containsEntry("genres",
List.of("Sci-Fi", "Horror", "Drama"));
}

@Test
void csvParsesToTheSameDocumentsAsJson() throws Exception {
Map<String, Map<String, List<String>>> expected = byId(json.create(resource("/shows.json")), "id");
Map<String, Map<String, List<String>>> actual = byId(new CsvDocumentCreator().create(resource("/shows.csv")),
"id");

assertThat(actual).containsExactlyEntriesOf(expected);
}

@Test
void markdownFilesParseToTheSameDocumentsAsJson() throws Exception {
Map<String, Map<String, List<String>>> expected = byId(json.create(resource("/shows.json")), "id");
MarkdownDocumentCreator markdown = new MarkdownDocumentCreator();

for (Map.Entry<String, Map<String, List<String>>> show : expected.entrySet()) {
List<SolrInputDocument> docs = markdown.create(resource("/shows-markdown/" + show.getKey() + ".md"));
assertThat(docs).as(show.getKey()).hasSize(1);
Map<String, List<String>> fields = fields(docs.getFirst(), "");

// The description is the document body, so it comes back as content
// (with the title heading) rather than as a front matter field.
Map<String, List<String>> frontMatter = new TreeMap<>(show.getValue());
List<String> description = frontMatter.remove("description");
assertThat(fields.remove("content")).as("%s content", show.getKey()).hasSize(1).first().asString()
.contains(description.getFirst());
assertThat(fields.remove("headings")).as("%s headings", show.getKey())
.isEqualTo(show.getValue().get("title"));
assertThat(fields).as(show.getKey()).containsExactlyEntriesOf(frontMatter);
}
}

@Test
void xmlParsesToTheSameDocumentsAsJson() throws Exception {
Map<String, Map<String, List<String>>> expected = byId(json.create(resource("/shows.json")), "id");
Map<String, Map<String, List<String>>> actual = byId(new XmlDocumentCreator().create(resource("/shows.xml")),
"id");

assertThat(actual).containsExactlyEntriesOf(expected);
}

private static Map<String, Map<String, List<String>>> byId(List<SolrInputDocument> docs, String idField) {
Map<String, Map<String, List<String>>> byId = new TreeMap<>();
for (SolrInputDocument doc : docs) {
Map<String, List<String>> fields = fields(doc, "");
String id = Objects.requireNonNull(fields.get(idField), "missing " + idField).getFirst();
assertThat(byId.put(id, fields)).as("duplicate id %s", id).isNull();
}
return byId;
}

/**
* Field name to string values, with {@code prefix} stripped from every name.
*/
private static Map<String, List<String>> fields(SolrInputDocument doc, String prefix) {
Map<String, List<String>> fields = new TreeMap<>();
for (String name : doc.getFieldNames()) {
String key = name.startsWith(prefix) ? name.substring(prefix.length()) : name;
fields.put(key, doc.getFieldValues(name).stream().map(String::valueOf).toList());
}
return new LinkedHashMap<>(fields);
}

private static String resource(String path) throws IOException {
try (InputStream in = Objects.requireNonNull(ShowsSampleDataTest.class.getResourceAsStream(path),
"missing test resource " + path)) {
return new String(in.readAllBytes(), StandardCharsets.UTF_8);
}
}
}
Loading