fix(knowledge): parse PDFs through safe Tika adapter
This commit is contained in:
+9
-15
@@ -38,15 +38,12 @@ import org.dromara.aihr.domain.AihrSopDto.UploadResponse;
|
||||
import org.dromara.aihr.domain.AihrSopDto.VectorIndexStatusResponse;
|
||||
import org.dromara.aihr.domain.AihrSopDto.VectorizeResponse;
|
||||
import org.dromara.aihr.knowledge.domain.AihrKnowledgeSpaceDto.UnbindDocumentResponse;
|
||||
import org.dromara.aihr.knowledge.parse.KnowledgeDocumentParser;
|
||||
import org.dromara.aihr.knowledge.parse.TikaKnowledgeDocumentParser;
|
||||
import org.dromara.aihr.personal.service.EnterpriseKnowledgeAccessPolicy.EnterpriseKnowledgeGrant;
|
||||
import org.dromara.aihr.personal.support.PersonalOwner;
|
||||
import org.dromara.common.core.exception.ServiceException;
|
||||
import org.dromara.common.tenant.helper.TenantHelper;
|
||||
import org.apache.tika.metadata.Metadata;
|
||||
import org.apache.tika.metadata.TikaCoreProperties;
|
||||
import org.apache.tika.parser.AutoDetectParser;
|
||||
import org.apache.tika.parser.ParseContext;
|
||||
import org.apache.tika.sax.BodyContentHandler;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.dao.DataAccessException;
|
||||
@@ -113,6 +110,7 @@ public class AihrSopSeedService {
|
||||
private static final String LOCAL_EMBEDDING_BASE_URL = "local";
|
||||
private static final int LOCAL_EMBEDDING_DIMENSION = 1536;
|
||||
private static final int MAX_EVIDENCE_SNIPPETS = 3;
|
||||
private static final KnowledgeDocumentParser KNOWLEDGE_DOCUMENT_PARSER = new TikaKnowledgeDocumentParser();
|
||||
private static final String NO_CLEAR_SOP_EVIDENCE = "未在已入库 SOP 中找到明确依据";
|
||||
private static final Set<String> FEEDBACK_REASON_CODES = Set.of(
|
||||
"no_answer", "not_specific", "wrong_reference", "outdated", "not_applicable"
|
||||
@@ -3844,11 +3842,7 @@ public class AihrSopSeedService {
|
||||
return "";
|
||||
}
|
||||
try (InputStream input = file.getInputStream()) {
|
||||
BodyContentHandler handler = new BodyContentHandler(-1);
|
||||
Metadata metadata = new Metadata();
|
||||
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
|
||||
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
|
||||
return normalizeExtractedText(handler.toString());
|
||||
return readKnowledgeDocument(input, fileName);
|
||||
} catch (Exception e) {
|
||||
throw new ServiceException("文件解析失败");
|
||||
}
|
||||
@@ -3887,16 +3881,16 @@ public class AihrSopSeedService {
|
||||
return "";
|
||||
}
|
||||
try (InputStream input = Files.newInputStream(file)) {
|
||||
BodyContentHandler handler = new BodyContentHandler(-1);
|
||||
Metadata metadata = new Metadata();
|
||||
metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, fileName);
|
||||
new AutoDetectParser().parse(input, handler, metadata, new ParseContext());
|
||||
return normalizeExtractedText(handler.toString());
|
||||
return readKnowledgeDocument(input, fileName);
|
||||
} catch (Exception e) {
|
||||
throw new ServiceException("文件解析失败");
|
||||
}
|
||||
}
|
||||
|
||||
static String readKnowledgeDocument(InputStream input, String fileName) {
|
||||
return normalizeExtractedText(KNOWLEDGE_DOCUMENT_PARSER.parse(fileName, null, input).text());
|
||||
}
|
||||
|
||||
private static FileFingerprint fileFingerprint(MultipartFile file) {
|
||||
try (InputStream input = file.getInputStream()) {
|
||||
return fileFingerprint(input);
|
||||
|
||||
+41
@@ -0,0 +1,41 @@
|
||||
package org.dromara.aihr.service;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDPageContentStream;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType1Font;
|
||||
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
|
||||
import org.junit.jupiter.api.Tag;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
class AihrSopSeedServiceDocumentParserTest {
|
||||
|
||||
@Test
|
||||
@Tag("dev")
|
||||
void stagedPdfUsesTheNonArchiveSafeParser() throws IOException {
|
||||
try (ByteArrayInputStream input = new ByteArrayInputStream(pdfBytes("Fee guide"))) {
|
||||
assertTrue(AihrSopSeedService.readKnowledgeDocument(input, "fee-guide.pdf").contains("Fee guide"));
|
||||
}
|
||||
}
|
||||
|
||||
private static byte[] pdfBytes(String text) throws IOException {
|
||||
try (PDDocument document = new PDDocument(); ByteArrayOutputStream output = new ByteArrayOutputStream()) {
|
||||
document.addPage(new PDPage());
|
||||
try (PDPageContentStream content = new PDPageContentStream(document, document.getPage(0))) {
|
||||
content.beginText();
|
||||
content.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
content.newLineAtOffset(72, 720);
|
||||
content.showText(text);
|
||||
content.endText();
|
||||
}
|
||||
document.save(output);
|
||||
return output.toByteArray();
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user