fix(knowledge): parse PDFs through safe Tika adapter

This commit is contained in:
2026-07-22 21:44:40 +08:00
parent 455267cf6f
commit 4ab0fccc73
2 changed files with 50 additions and 15 deletions
@@ -38,15 +38,12 @@ import org.dromara.aihr.domain.AihrSopDto.UploadResponse;
import org.dromara.aihr.domain.AihrSopDto.VectorIndexStatusResponse;
import org.dromara.aihr.domain.AihrSopDto.VectorizeResponse;
import org.dromara.aihr.knowledge.domain.AihrKnowledgeSpaceDto.UnbindDocumentResponse;
import org.dromara.aihr.knowledge.parse.KnowledgeDocumentParser;
import org.dromara.aihr.knowledge.parse.TikaKnowledgeDocumentParser;
import org.dromara.aihr.personal.service.EnterpriseKnowledgeAccessPolicy.EnterpriseKnowledgeGrant;
import org.dromara.aihr.personal.support.PersonalOwner;
import org.dromara.common.core.exception.ServiceException;
import org.dromara.common.tenant.helper.TenantHelper;
import org.apache.tika.metadata.Metadata;
import org.apache.tika.metadata.TikaCoreProperties;
import org.apache.tika.parser.AutoDetectParser;
import org.apache.tika.parser.ParseContext;
import org.apache.tika.sax.BodyContentHandler;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.dao.DataAccessException;
@@ -113,6 +110,7 @@ public class AihrSopSeedService {
private static final String LOCAL_EMBEDDING_BASE_URL = "local";
private static final int LOCAL_EMBEDDING_DIMENSION = 1536;
private static final int MAX_EVIDENCE_SNIPPETS = 3;
private static final KnowledgeDocumentParser KNOWLEDGE_DOCUMENT_PARSER = new TikaKnowledgeDocumentParser();
private static final String NO_CLEAR_SOP_EVIDENCE = "未在已入库 SOP 中找到明确依据";
private static final Set<String> FEEDBACK_REASON_CODES = Set.of(
"no_answer", "not_specific", "wrong_reference", "outdated", "not_applicable"
@@ -3844,11 +3842,7 @@ public class AihrSopSeedService {
return "";
}
try (InputStream input = file.getInputStream()) {
BodyContentHandler handler = new BodyContentHandler(-1);
Metadata metadata = new Metadata();
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
return normalizeExtractedText(handler.toString());
return readKnowledgeDocument(input, fileName);
} catch (Exception e) {
throw new ServiceException("文件解析失败");
}
@@ -3887,16 +3881,16 @@ public class AihrSopSeedService {
return "";
}
try (InputStream input = Files.newInputStream(file)) {
BodyContentHandler handler = new BodyContentHandler(-1);
Metadata metadata = new Metadata();
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
return normalizeExtractedText(handler.toString());
return readKnowledgeDocument(input, fileName);
} catch (Exception e) {
throw new ServiceException("文件解析失败");
}
}
static String readKnowledgeDocument(InputStream input, String fileName) {
return normalizeExtractedText(KNOWLEDGE_DOCUMENT_PARSER.parse(fileName, null, input).text());
}
private static FileFingerprint fileFingerprint(MultipartFile file) {
try (InputStream input = file.getInputStream()) {
return fileFingerprint(input);
@@ -0,0 +1,41 @@
package org.dromara.aihr.service;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import static org.junit.jupiter.api.Assertions.assertTrue;
class AihrSopSeedServiceDocumentParserTest {
@Test
@Tag("dev")
void stagedPdfUsesTheNonArchiveSafeParser() throws IOException {
try (ByteArrayInputStream input = new ByteArrayInputStream(pdfBytes("Fee guide"))) {
assertTrue(AihrSopSeedService.readKnowledgeDocument(input, "fee-guide.pdf").contains("Fee guide"));
}
}
private static byte[] pdfBytes(String text) throws IOException {
try (PDDocument document = new PDDocument(); ByteArrayOutputStream output = new ByteArrayOutputStream()) {
document.addPage(new PDPage());
try (PDPageContentStream content = new PDPageContentStream(document, document.getPage(0))) {
content.beginText();
content.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
content.newLineAtOffset(72, 720);
content.showText(text);
content.endText();
}
document.save(output);
return output.toByteArray();
}
}
}