refactor(aihr): extract shared document parser
This commit is contained in:
+9
@@ -0,0 +1,9 @@
|
||||
package org.dromara.aihr.knowledge.parse;
|
||||
|
||||
/**
|
||||
* Stateless byte-document parser shared by knowledge ingestion flows.
|
||||
*/
|
||||
public interface KnowledgeDocumentParser {
|
||||
|
||||
ParsedDocument parse(String fileName, String contentType, byte[] bytes);
|
||||
}
|
||||
+37
@@ -0,0 +1,37 @@
|
||||
package org.dromara.aihr.knowledge.parse;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
public record ParsedDocument(String text, String mimeType, Map<String, String> metadata) {
|
||||
|
||||
public ParsedDocument {
|
||||
text = text == null ? "" : text;
|
||||
mimeType = mimeType == null ? "application/octet-stream" : mimeType;
|
||||
metadata = metadata == null ? Map.of() : Map.copyOf(metadata);
|
||||
}
|
||||
|
||||
public List<String> chunks(int blockSize, int overlap) {
|
||||
if (blockSize <= 0 || overlap < 0 || overlap >= blockSize) {
|
||||
throw new IllegalArgumentException("invalid chunk settings");
|
||||
}
|
||||
if (text.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
List<String> chunks = new ArrayList<>();
|
||||
int step = blockSize - overlap;
|
||||
for (int start = 0; start < text.length(); start += step) {
|
||||
int end = Math.min(text.length(), start + blockSize);
|
||||
String chunk = text.substring(start, end).trim();
|
||||
if (!chunk.isEmpty()) {
|
||||
chunks.add(chunk);
|
||||
}
|
||||
if (end == text.length()) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return List.copyOf(chunks);
|
||||
}
|
||||
}
|
||||
+92
@@ -0,0 +1,92 @@
|
||||
package org.dromara.aihr.knowledge.parse;
|
||||
|
||||
import org.apache.tika.exception.WriteLimitReachedException;
|
||||
import org.apache.tika.metadata.Metadata;
|
||||
import org.apache.tika.metadata.TikaCoreProperties;
|
||||
import org.apache.tika.parser.AutoDetectParser;
|
||||
import org.apache.tika.parser.ParseContext;
|
||||
import org.apache.tika.sax.BodyContentHandler;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
|
||||
@Component
|
||||
public class TikaKnowledgeDocumentParser implements KnowledgeDocumentParser {
|
||||
|
||||
static final int DEFAULT_MAX_EXPANDED_CHARS = 2_000_000;
|
||||
|
||||
private final int maxExpandedChars;
|
||||
|
||||
public TikaKnowledgeDocumentParser() {
|
||||
this(DEFAULT_MAX_EXPANDED_CHARS);
|
||||
}
|
||||
|
||||
TikaKnowledgeDocumentParser(int maxExpandedChars) {
|
||||
if (maxExpandedChars <= 0) {
|
||||
throw new IllegalArgumentException("max expanded characters must be positive");
|
||||
}
|
||||
this.maxExpandedChars = maxExpandedChars;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ParsedDocument parse(String fileName, String contentType, byte[] bytes) {
|
||||
if (bytes == null || bytes.length == 0) {
|
||||
throw new IllegalArgumentException("document content is empty");
|
||||
}
|
||||
|
||||
Metadata metadata = new Metadata();
|
||||
if (fileName != null && !fileName.isBlank()) {
|
||||
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName.trim());
|
||||
}
|
||||
if (contentType != null && !contentType.isBlank()) {
|
||||
metadata.set(Metadata.CONTENT_TYPE, contentType.trim());
|
||||
}
|
||||
|
||||
BodyContentHandler handler = new BodyContentHandler(maxExpandedChars + 1);
|
||||
try (ByteArrayInputStream input = new ByteArrayInputStream(bytes)) {
|
||||
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
|
||||
} catch (Exception e) {
|
||||
if (WriteLimitReachedException.isWriteLimitReached(e)) {
|
||||
throw new IllegalArgumentException("document expanded text exceeds limit", e);
|
||||
}
|
||||
throw new IllegalArgumentException("document parsing failed", e);
|
||||
}
|
||||
|
||||
String text = handler.toString().trim();
|
||||
if (text.isEmpty()) {
|
||||
throw new IllegalArgumentException("document contains no text");
|
||||
}
|
||||
if (text.length() > maxExpandedChars) {
|
||||
throw new IllegalArgumentException("document expanded text exceeds limit");
|
||||
}
|
||||
|
||||
return new ParsedDocument(text, resolveMimeType(contentType, metadata), metadataMap(metadata));
|
||||
}
|
||||
|
||||
private static String resolveMimeType(String suppliedContentType, Metadata metadata) {
|
||||
String candidate = suppliedContentType;
|
||||
if (candidate == null || candidate.isBlank()) {
|
||||
candidate = metadata.get(Metadata.CONTENT_TYPE);
|
||||
}
|
||||
if (candidate == null || candidate.isBlank()) {
|
||||
return "application/octet-stream";
|
||||
}
|
||||
int parameterStart = candidate.indexOf(';');
|
||||
String mimeType = (parameterStart >= 0 ? candidate.substring(0, parameterStart) : candidate).trim();
|
||||
return mimeType.isEmpty() ? "application/octet-stream" : mimeType.toLowerCase(Locale.ROOT);
|
||||
}
|
||||
|
||||
private static Map<String, String> metadataMap(Metadata metadata) {
|
||||
Map<String, String> values = new LinkedHashMap<>();
|
||||
for (String name : metadata.names()) {
|
||||
String value = metadata.get(name);
|
||||
if (value != null) {
|
||||
values.put(name, value);
|
||||
}
|
||||
}
|
||||
return values;
|
||||
}
|
||||
}
|
||||
+13
-18
@@ -34,12 +34,8 @@ import org.dromara.aihr.domain.AihrSopDto.SummaryCardResponse;
|
||||
import org.dromara.aihr.domain.AihrSopDto.UploadResponse;
|
||||
import org.dromara.aihr.domain.AihrSopDto.VectorIndexStatusResponse;
|
||||
import org.dromara.aihr.domain.AihrSopDto.VectorizeResponse;
|
||||
import org.dromara.aihr.knowledge.parse.KnowledgeDocumentParser;
|
||||
import org.dromara.common.core.exception.ServiceException;
|
||||
import org.apache.tika.metadata.Metadata;
|
||||
import org.apache.tika.metadata.TikaCoreProperties;
|
||||
import org.apache.tika.parser.AutoDetectParser;
|
||||
import org.apache.tika.parser.ParseContext;
|
||||
import org.apache.tika.sax.BodyContentHandler;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.dao.DataAccessException;
|
||||
import org.springframework.jdbc.core.PreparedStatementCreator;
|
||||
@@ -120,19 +116,22 @@ public class AihrSopSeedService {
|
||||
private final String importRootConfig;
|
||||
private final ScheduledExecutorService scheduledExecutorService;
|
||||
private final AihrVideoService videoService;
|
||||
private final KnowledgeDocumentParser knowledgeDocumentParser;
|
||||
private volatile boolean knowledgeGapTableReady;
|
||||
private volatile boolean sopReviewTableReady;
|
||||
private volatile boolean answerFeedbackTableReady;
|
||||
|
||||
public AihrSopSeedService(ObjectMapper objectMapper, JdbcTemplate jdbcTemplate, ISysOssService ossService,
|
||||
@Value("${aihr.import.root:}") String importRootConfig,
|
||||
ScheduledExecutorService scheduledExecutorService, AihrVideoService videoService) {
|
||||
ScheduledExecutorService scheduledExecutorService, AihrVideoService videoService,
|
||||
KnowledgeDocumentParser knowledgeDocumentParser) {
|
||||
this.objectMapper = objectMapper;
|
||||
this.jdbcTemplate = jdbcTemplate;
|
||||
this.ossService = ossService;
|
||||
this.importRootConfig = importRootConfig;
|
||||
this.scheduledExecutorService = scheduledExecutorService;
|
||||
this.videoService = videoService;
|
||||
this.knowledgeDocumentParser = knowledgeDocumentParser;
|
||||
}
|
||||
|
||||
public SearchResponse search(SearchRequest request) {
|
||||
@@ -2947,12 +2946,8 @@ public class AihrSopSeedService {
|
||||
}
|
||||
return "";
|
||||
}
|
||||
try (InputStream input = file.getInputStream()) {
|
||||
BodyContentHandler handler = new BodyContentHandler(-1);
|
||||
Metadata metadata = new Metadata();
|
||||
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
|
||||
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
|
||||
return normalizeExtractedText(handler.toString());
|
||||
try {
|
||||
return parseDocument(fileName, file.getContentType(), file.getBytes());
|
||||
} catch (Exception e) {
|
||||
throw new ServiceException("文件解析失败");
|
||||
}
|
||||
@@ -2990,17 +2985,17 @@ public class AihrSopSeedService {
|
||||
}
|
||||
return "";
|
||||
}
|
||||
try (InputStream input = Files.newInputStream(file)) {
|
||||
BodyContentHandler handler = new BodyContentHandler(-1);
|
||||
Metadata metadata = new Metadata();
|
||||
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
|
||||
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
|
||||
return normalizeExtractedText(handler.toString());
|
||||
try {
|
||||
return parseDocument(fileName, Files.probeContentType(file), Files.readAllBytes(file));
|
||||
} catch (Exception e) {
|
||||
throw new ServiceException("文件解析失败");
|
||||
}
|
||||
}
|
||||
|
||||
private String parseDocument(String fileName, String contentType, byte[] bytes) {
|
||||
return normalizeExtractedText(knowledgeDocumentParser.parse(fileName, contentType, bytes).text());
|
||||
}
|
||||
|
||||
private static FileFingerprint fileFingerprint(MultipartFile file) {
|
||||
try (InputStream input = file.getInputStream()) {
|
||||
return fileFingerprint(input);
|
||||
|
||||
+85
@@ -0,0 +1,85 @@
|
||||
package org.dromara.aihr.knowledge.parse;
|
||||
|
||||
import org.junit.jupiter.api.Tag;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
@Tag("dev")
|
||||
class TikaKnowledgeDocumentParserTest {
|
||||
|
||||
@Test
|
||||
void parsesUtf8TextAndCreatesOverlappingChunks() {
|
||||
KnowledgeDocumentParser parser = new TikaKnowledgeDocumentParser();
|
||||
|
||||
ParsedDocument document = parser.parse(
|
||||
"fee-guide.txt",
|
||||
"text/plain; charset=UTF-8",
|
||||
"收费沟通先说明费用构成".getBytes(StandardCharsets.UTF_8)
|
||||
);
|
||||
|
||||
assertEquals("收费沟通先说明费用构成", document.text());
|
||||
assertEquals("text/plain", document.mimeType());
|
||||
assertFalse(document.metadata().isEmpty());
|
||||
assertEquals(
|
||||
java.util.List.of("收费沟通先", "通先说明费", "明费用构成"),
|
||||
document.chunks(5, 2)
|
||||
);
|
||||
}
|
||||
|
||||
@Test
|
||||
void rejectsInvalidChunkSettings() {
|
||||
ParsedDocument document = new ParsedDocument("有效内容", "text/plain", java.util.Map.of());
|
||||
|
||||
IllegalArgumentException zeroBlock = assertThrows(
|
||||
IllegalArgumentException.class,
|
||||
() -> document.chunks(0, 0)
|
||||
);
|
||||
IllegalArgumentException negativeOverlap = assertThrows(
|
||||
IllegalArgumentException.class,
|
||||
() -> document.chunks(4, -1)
|
||||
);
|
||||
IllegalArgumentException fullOverlap = assertThrows(
|
||||
IllegalArgumentException.class,
|
||||
() -> document.chunks(4, 4)
|
||||
);
|
||||
|
||||
assertEquals("invalid chunk settings", zeroBlock.getMessage());
|
||||
assertEquals("invalid chunk settings", negativeOverlap.getMessage());
|
||||
assertEquals("invalid chunk settings", fullOverlap.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void rejectsEmptyOrWhitespaceOnlyContent() {
|
||||
KnowledgeDocumentParser parser = new TikaKnowledgeDocumentParser();
|
||||
|
||||
assertThrows(IllegalArgumentException.class, () -> parser.parse("empty.txt", "text/plain", new byte[0]));
|
||||
assertThrows(
|
||||
IllegalArgumentException.class,
|
||||
() -> parser.parse("blank.txt", "text/plain", " \n\t".getBytes(StandardCharsets.UTF_8))
|
||||
);
|
||||
}
|
||||
|
||||
@Test
|
||||
void enforcesExpandedTextLimitAtBoundary() {
|
||||
KnowledgeDocumentParser parser = new TikaKnowledgeDocumentParser(10);
|
||||
|
||||
ParsedDocument accepted = parser.parse(
|
||||
"boundary.txt",
|
||||
"text/plain",
|
||||
"1234567890".getBytes(StandardCharsets.UTF_8)
|
||||
);
|
||||
IllegalArgumentException rejected = assertThrows(
|
||||
IllegalArgumentException.class,
|
||||
() -> parser.parse("too-long.txt", "text/plain", "12345678901".getBytes(StandardCharsets.UTF_8))
|
||||
);
|
||||
|
||||
assertEquals("1234567890", accepted.text());
|
||||
assertTrue(rejected.getMessage().contains("exceeds"));
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -62,7 +62,7 @@ public class AihrPracticeSeedServiceTest {
|
||||
RecordingTransactionManager transactionManager = new RecordingTransactionManager();
|
||||
TransactionTemplate transactionTemplate = new TransactionTemplate(transactionManager);
|
||||
PilotExportJdbcTemplate jdbcTemplate = new PilotExportJdbcTemplate(transactionManager);
|
||||
AihrSopSeedService sopSeedService = new AihrSopSeedService(new ObjectMapper(), jdbcTemplate, null, "", null, null);
|
||||
AihrSopSeedService sopSeedService = new AihrSopSeedService(new ObjectMapper(), jdbcTemplate, null, "", null, null, null);
|
||||
Constructor<AihrPracticeSeedService> constructor = (Constructor<AihrPracticeSeedService>) AihrPracticeSeedService.class
|
||||
.getDeclaredConstructor(ObjectMapper.class, JdbcTemplate.class, AihrPracticeLlmService.class,
|
||||
AihrSopSeedService.class, TransactionTemplate.class);
|
||||
|
||||
Reference in New Issue
Block a user