refactor(aihr): extract shared document parser

This commit is contained in:
2026-07-12 02:33:38 +08:00
parent 655d784beb
commit 36ff3c79f8
6 changed files with 237 additions and 19 deletions
@@ -0,0 +1,9 @@
package org.dromara.aihr.knowledge.parse;
/**
* Stateless byte-document parser shared by knowledge ingestion flows.
*/
public interface KnowledgeDocumentParser {
ParsedDocument parse(String fileName, String contentType, byte[] bytes);
}
@@ -0,0 +1,37 @@
package org.dromara.aihr.knowledge.parse;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
public record ParsedDocument(String text, String mimeType, Map<String, String> metadata) {
public ParsedDocument {
text = text == null ? "" : text;
mimeType = mimeType == null ? "application/octet-stream" : mimeType;
metadata = metadata == null ? Map.of() : Map.copyOf(metadata);
}
public List<String> chunks(int blockSize, int overlap) {
if (blockSize <= 0 || overlap < 0 || overlap >= blockSize) {
throw new IllegalArgumentException("invalid chunk settings");
}
if (text.isBlank()) {
return List.of();
}
List<String> chunks = new ArrayList<>();
int step = blockSize - overlap;
for (int start = 0; start < text.length(); start += step) {
int end = Math.min(text.length(), start + blockSize);
String chunk = text.substring(start, end).trim();
if (!chunk.isEmpty()) {
chunks.add(chunk);
}
if (end == text.length()) {
break;
}
}
return List.copyOf(chunks);
}
}
@@ -0,0 +1,92 @@
package org.dromara.aihr.knowledge.parse;
import org.apache.tika.exception.WriteLimitReachedException;
import org.apache.tika.metadata.Metadata;
import org.apache.tika.metadata.TikaCoreProperties;
import org.apache.tika.parser.AutoDetectParser;
import org.apache.tika.parser.ParseContext;
import org.apache.tika.sax.BodyContentHandler;
import org.springframework.stereotype.Component;
import java.io.ByteArrayInputStream;
import java.util.LinkedHashMap;
import java.util.Locale;
import java.util.Map;
@Component
public class TikaKnowledgeDocumentParser implements KnowledgeDocumentParser {
static final int DEFAULT_MAX_EXPANDED_CHARS = 2_000_000;
private final int maxExpandedChars;
public TikaKnowledgeDocumentParser() {
this(DEFAULT_MAX_EXPANDED_CHARS);
}
TikaKnowledgeDocumentParser(int maxExpandedChars) {
if (maxExpandedChars <= 0) {
throw new IllegalArgumentException("max expanded characters must be positive");
}
this.maxExpandedChars = maxExpandedChars;
}
@Override
public ParsedDocument parse(String fileName, String contentType, byte[] bytes) {
if (bytes == null || bytes.length == 0) {
throw new IllegalArgumentException("document content is empty");
}
Metadata metadata = new Metadata();
if (fileName != null && !fileName.isBlank()) {
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName.trim());
}
if (contentType != null && !contentType.isBlank()) {
metadata.set(Metadata.CONTENT_TYPE, contentType.trim());
}
BodyContentHandler handler = new BodyContentHandler(maxExpandedChars + 1);
try (ByteArrayInputStream input = new ByteArrayInputStream(bytes)) {
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
} catch (Exception e) {
if (WriteLimitReachedException.isWriteLimitReached(e)) {
throw new IllegalArgumentException("document expanded text exceeds limit", e);
}
throw new IllegalArgumentException("document parsing failed", e);
}
String text = handler.toString().trim();
if (text.isEmpty()) {
throw new IllegalArgumentException("document contains no text");
}
if (text.length() > maxExpandedChars) {
throw new IllegalArgumentException("document expanded text exceeds limit");
}
return new ParsedDocument(text, resolveMimeType(contentType, metadata), metadataMap(metadata));
}
private static String resolveMimeType(String suppliedContentType, Metadata metadata) {
String candidate = suppliedContentType;
if (candidate == null || candidate.isBlank()) {
candidate = metadata.get(Metadata.CONTENT_TYPE);
}
if (candidate == null || candidate.isBlank()) {
return "application/octet-stream";
}
int parameterStart = candidate.indexOf(';');
String mimeType = (parameterStart >= 0 ? candidate.substring(0, parameterStart) : candidate).trim();
return mimeType.isEmpty() ? "application/octet-stream" : mimeType.toLowerCase(Locale.ROOT);
}
private static Map<String, String> metadataMap(Metadata metadata) {
Map<String, String> values = new LinkedHashMap<>();
for (String name : metadata.names()) {
String value = metadata.get(name);
if (value != null) {
values.put(name, value);
}
}
return values;
}
}
@@ -34,12 +34,8 @@ import org.dromara.aihr.domain.AihrSopDto.SummaryCardResponse;
import org.dromara.aihr.domain.AihrSopDto.UploadResponse;
import org.dromara.aihr.domain.AihrSopDto.VectorIndexStatusResponse;
import org.dromara.aihr.domain.AihrSopDto.VectorizeResponse;
import org.dromara.aihr.knowledge.parse.KnowledgeDocumentParser;
import org.dromara.common.core.exception.ServiceException;
import org.apache.tika.metadata.Metadata;
import org.apache.tika.metadata.TikaCoreProperties;
import org.apache.tika.parser.AutoDetectParser;
import org.apache.tika.parser.ParseContext;
import org.apache.tika.sax.BodyContentHandler;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.dao.DataAccessException;
import org.springframework.jdbc.core.PreparedStatementCreator;
@@ -120,19 +116,22 @@ public class AihrSopSeedService {
private final String importRootConfig;
private final ScheduledExecutorService scheduledExecutorService;
private final AihrVideoService videoService;
private final KnowledgeDocumentParser knowledgeDocumentParser;
private volatile boolean knowledgeGapTableReady;
private volatile boolean sopReviewTableReady;
private volatile boolean answerFeedbackTableReady;
public AihrSopSeedService(ObjectMapper objectMapper, JdbcTemplate jdbcTemplate, ISysOssService ossService,
@Value("${aihr.import.root:}") String importRootConfig,
ScheduledExecutorService scheduledExecutorService, AihrVideoService videoService) {
ScheduledExecutorService scheduledExecutorService, AihrVideoService videoService,
KnowledgeDocumentParser knowledgeDocumentParser) {
this.objectMapper = objectMapper;
this.jdbcTemplate = jdbcTemplate;
this.ossService = ossService;
this.importRootConfig = importRootConfig;
this.scheduledExecutorService = scheduledExecutorService;
this.videoService = videoService;
this.knowledgeDocumentParser = knowledgeDocumentParser;
}
public SearchResponse search(SearchRequest request) {
@@ -2947,12 +2946,8 @@ public class AihrSopSeedService {
}
return "";
}
try (InputStream input = file.getInputStream()) {
BodyContentHandler handler = new BodyContentHandler(-1);
Metadata metadata = new Metadata();
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
return normalizeExtractedText(handler.toString());
try {
return parseDocument(fileName, file.getContentType(), file.getBytes());
} catch (Exception e) {
throw new ServiceException("文件解析失败");
}
@@ -2990,17 +2985,17 @@ public class AihrSopSeedService {
}
return "";
}
try (InputStream input = Files.newInputStream(file)) {
BodyContentHandler handler = new BodyContentHandler(-1);
Metadata metadata = new Metadata();
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
return normalizeExtractedText(handler.toString());
try {
return parseDocument(fileName, Files.probeContentType(file), Files.readAllBytes(file));
} catch (Exception e) {
throw new ServiceException("文件解析失败");
}
}
private String parseDocument(String fileName, String contentType, byte[] bytes) {
return normalizeExtractedText(knowledgeDocumentParser.parse(fileName, contentType, bytes).text());
}
private static FileFingerprint fileFingerprint(MultipartFile file) {
try (InputStream input = file.getInputStream()) {
return fileFingerprint(input);
@@ -0,0 +1,85 @@
package org.dromara.aihr.knowledge.parse;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
import java.nio.charset.StandardCharsets;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
@Tag("dev")
class TikaKnowledgeDocumentParserTest {
@Test
void parsesUtf8TextAndCreatesOverlappingChunks() {
KnowledgeDocumentParser parser = new TikaKnowledgeDocumentParser();
ParsedDocument document = parser.parse(
"fee-guide.txt",
"text/plain; charset=UTF-8",
"收费沟通先说明费用构成".getBytes(StandardCharsets.UTF_8)
);
assertEquals("收费沟通先说明费用构成", document.text());
assertEquals("text/plain", document.mimeType());
assertFalse(document.metadata().isEmpty());
assertEquals(
java.util.List.of("收费沟通先", "通先说明费", "明费用构成"),
document.chunks(5, 2)
);
}
@Test
void rejectsInvalidChunkSettings() {
ParsedDocument document = new ParsedDocument("有效内容", "text/plain", java.util.Map.of());
IllegalArgumentException zeroBlock = assertThrows(
IllegalArgumentException.class,
() -> document.chunks(0, 0)
);
IllegalArgumentException negativeOverlap = assertThrows(
IllegalArgumentException.class,
() -> document.chunks(4, -1)
);
IllegalArgumentException fullOverlap = assertThrows(
IllegalArgumentException.class,
() -> document.chunks(4, 4)
);
assertEquals("invalid chunk settings", zeroBlock.getMessage());
assertEquals("invalid chunk settings", negativeOverlap.getMessage());
assertEquals("invalid chunk settings", fullOverlap.getMessage());
}
@Test
void rejectsEmptyOrWhitespaceOnlyContent() {
KnowledgeDocumentParser parser = new TikaKnowledgeDocumentParser();
assertThrows(IllegalArgumentException.class, () -> parser.parse("empty.txt", "text/plain", new byte[0]));
assertThrows(
IllegalArgumentException.class,
() -> parser.parse("blank.txt", "text/plain", " \n\t".getBytes(StandardCharsets.UTF_8))
);
}
@Test
void enforcesExpandedTextLimitAtBoundary() {
KnowledgeDocumentParser parser = new TikaKnowledgeDocumentParser(10);
ParsedDocument accepted = parser.parse(
"boundary.txt",
"text/plain",
"1234567890".getBytes(StandardCharsets.UTF_8)
);
IllegalArgumentException rejected = assertThrows(
IllegalArgumentException.class,
() -> parser.parse("too-long.txt", "text/plain", "12345678901".getBytes(StandardCharsets.UTF_8))
);
assertEquals("1234567890", accepted.text());
assertTrue(rejected.getMessage().contains("exceeds"));
}
}
@@ -62,7 +62,7 @@ public class AihrPracticeSeedServiceTest {
RecordingTransactionManager transactionManager = new RecordingTransactionManager();
TransactionTemplate transactionTemplate = new TransactionTemplate(transactionManager);
PilotExportJdbcTemplate jdbcTemplate = new PilotExportJdbcTemplate(transactionManager);
AihrSopSeedService sopSeedService = new AihrSopSeedService(new ObjectMapper(), jdbcTemplate, null, "", null, null);
AihrSopSeedService sopSeedService = new AihrSopSeedService(new ObjectMapper(), jdbcTemplate, null, "", null, null, null);
Constructor<AihrPracticeSeedService> constructor = (Constructor<AihrPracticeSeedService>) AihrPracticeSeedService.class
.getDeclaredConstructor(ObjectMapper.class, JdbcTemplate.class, AihrPracticeLlmService.class,
AihrSopSeedService.class, TransactionTemplate.class);