From 12491b372484f8387596d4038359fbc4cea31696 Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?=E9=99=88=E5=AD=90=E9=BB=98?= <925456043@qq.com>
Date: Wed, 29 Jul 2026 01:03:42 +0800
Subject: [PATCH] =?UTF-8?q?feat:=20=E5=A2=9E=E5=BC=BA=E5=B8=B8=E8=A7=84?=
=?UTF-8?q?=E6=96=87=E6=A1=A3=E8=BD=BB=E9=87=8F=E8=AF=BB=E5=8F=96=E8=83=BD?=
=?UTF-8?q?=E5=8A=9B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
- 支持 PDF、Office、表格、TXT 与 Markdown 的结构化轻量读取
- 增加结构上限、取消信号、稳定定位与读取错误分类
- 使用事件流读取表格并补充核心读取测试
---
.../core/file2text/DocumentReadErrorCode.java | 37 +++
.../core/file2text/DocumentReadException.java | 56 ++++
.../core/file2text/DocumentReadSupport.java | 107 ++++++++
.../core/file2text/DocumentTextSegment.java | 80 ++++++
.../core/file2text/File2TextService.java | 80 +++++-
.../LightweightDocumentReadRequest.java | 100 +++++++
.../LightweightDocumentReadResult.java | 91 +++++++
.../extractor/ExtractorRegistry.java | 3 +
.../file2text/extractor/FileExtractor.java | 37 ++-
.../extractor/impl/DocExtractor.java | 52 +++-
.../extractor/impl/DocxExtractor.java | 130 +++++++--
.../extractor/impl/PdfTextExtractor.java | 55 +++-
.../extractor/impl/PlainTextExtractor.java | 181 ++++++++++++-
.../extractor/impl/PptExtractor.java | 122 +++++++++
.../extractor/impl/PptxExtractor.java | 51 +++-
.../impl/SpreadsheetReadSupport.java | 76 ++++++
.../extractor/impl/XlsExtractor.java | 256 ++++++++++++++++++
.../extractor/impl/XlsxExtractor.java | 228 ++++++++++++++++
.../File2TextServiceLightweightReadTest.java | 162 +++++++++++
19 files changed, 1832 insertions(+), 72 deletions(-)
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadErrorCode.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadException.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadSupport.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentTextSegment.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadRequest.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadResult.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptExtractor.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/SpreadsheetReadSupport.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsExtractor.java
create mode 100644 easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsxExtractor.java
create mode 100644 easy-agents-core/src/test/java/com/easyagents/core/file2text/File2TextServiceLightweightReadTest.java
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadErrorCode.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadErrorCode.java
new file mode 100644
index 0000000..8632c51
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadErrorCode.java
@@ -0,0 +1,37 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0 (the "License");
+ * you may not use this file except in compliance with the License.
+ * You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package com.easyagents.core.file2text;
+
+/**
+ * 轻量文档读取错误码。
+ */
+public enum DocumentReadErrorCode {
+
+ /** 不支持的文档类型。 */
+ UNSUPPORTED_DOCUMENT_TYPE,
+ /** 文档结构超过安全上限。 */
+ DOCUMENT_STRUCTURE_LIMIT_EXCEEDED,
+ /** 文档已加密。 */
+ DOCUMENT_ENCRYPTED,
+ /** 文档损坏或容器不合法。 */
+ DOCUMENT_CORRUPTED,
+ /** 文档中没有可读取文字。 */
+ DOCUMENT_NO_READABLE_TEXT,
+ /** 文档读取已取消。 */
+ DOCUMENT_READ_CANCELLED,
+ /** 未分类的文档读取失败。 */
+ DOCUMENT_READ_FAILED
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadException.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadException.java
new file mode 100644
index 0000000..c63b78f
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadException.java
@@ -0,0 +1,56 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0 (the "License");
+ * you may not use this file except in compliance with the License.
+ * You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package com.easyagents.core.file2text;
+
+/**
+ * 轻量文档读取异常。
+ */
+public class DocumentReadException extends RuntimeException {
+
+ private final DocumentReadErrorCode errorCode;
+
+ /**
+ * 创建文档读取异常。
+ *
+ * @param errorCode 错误码
+ * @param message 错误消息
+ */
+ public DocumentReadException(DocumentReadErrorCode errorCode, String message) {
+ super(message);
+ this.errorCode = errorCode;
+ }
+
+ /**
+ * 创建带原因的文档读取异常。
+ *
+ * @param errorCode 错误码
+ * @param message 错误消息
+ * @param cause 原始异常
+ */
+ public DocumentReadException(DocumentReadErrorCode errorCode, String message, Throwable cause) {
+ super(message, cause);
+ this.errorCode = errorCode;
+ }
+
+ /**
+ * 获取错误码。
+ *
+ * @return 错误码
+ */
+ public DocumentReadErrorCode getErrorCode() {
+ return errorCode;
+ }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadSupport.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadSupport.java
new file mode 100644
index 0000000..dbedeba
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentReadSupport.java
@@ -0,0 +1,107 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text;
+
+import com.easyagents.core.file2text.source.DocumentSource;
+
+import java.util.List;
+
+/**
+ * 结构化文档读取结果构造工具。
+ */
+public final class DocumentReadSupport {
+
+ private DocumentReadSupport() {
+ }
+
+ /**
+ * 创建带定位的文本片段。
+ *
+ * @param segmentId 片段 ID
+ * @param text 文本
+ * @param locatorType 定位类型
+ * @param locatorLabel 定位标签
+ * @param startIndex 起始字符下标
+ * @param headingPath 标题路径
+ * @return 文本片段
+ */
+ public static DocumentTextSegment segment(String segmentId,
+ String text,
+ String locatorType,
+ String locatorLabel,
+ int startIndex,
+ List headingPath) {
+ String safeText = text == null ? "" : text.trim();
+ DocumentTextSegment segment = new DocumentTextSegment();
+ segment.setSegmentId(segmentId);
+ segment.setText(safeText);
+ segment.setLocatorType(locatorType);
+ segment.setLocatorLabel(locatorLabel);
+ segment.setStartIndex(startIndex);
+ segment.setEndIndex(startIndex + safeText.length());
+ segment.setHeadingPath(headingPath);
+ segment.setTokenEstimate(estimateTokens(safeText));
+ return segment;
+ }
+
+ /**
+ * 汇总结构化读取结果。
+ *
+ * @param source 文档来源
+ * @param segments 片段
+ * @param request 读取请求
+ * @return 读取结果
+ * @throws DocumentReadException 结果为空或超过边界
+ */
+ public static LightweightDocumentReadResult result(DocumentSource source,
+ List segments,
+ LightweightDocumentReadRequest request)
+ throws DocumentReadException {
+ List nonEmpty = segments.stream()
+ .filter(item -> item.getText() != null && !item.getText().isBlank())
+ .toList();
+ if (nonEmpty.isEmpty()) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_NO_READABLE_TEXT,
+ "No readable text detected");
+ }
+ long charCount = nonEmpty.stream().mapToLong(item -> item.getText().length()).sum()
+ + Math.max(0, nonEmpty.size() - 1L);
+ if (charCount > request.getMaxExpandedChars() || charCount > Integer.MAX_VALUE) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_STRUCTURE_LIMIT_EXCEEDED,
+ "Expanded document text exceeds the configured limit");
+ }
+ LightweightDocumentReadResult result = new LightweightDocumentReadResult();
+ result.setFileName(source.getFileName());
+ result.setMimeType(source.getMimeType());
+ result.setSegments(nonEmpty);
+ result.setCharCount((int) charCount);
+ result.setTokenEstimate(nonEmpty.stream().mapToInt(DocumentTextSegment::getTokenEstimate).sum());
+ return result;
+ }
+
+ /**
+ * 使用偏保守的字符规则估算 Token 数。
+ *
+ * @param text 文本
+ * @return Token 估算
+ */
+ public static int estimateTokens(String text) {
+ if (text == null || text.isEmpty()) {
+ return 0;
+ }
+ double tokens = 0;
+ for (int offset = 0; offset < text.length();) {
+ int codePoint = text.codePointAt(offset);
+ Character.UnicodeScript script = Character.UnicodeScript.of(codePoint);
+ tokens += switch (script) {
+ case HAN, HANGUL, HIRAGANA, KATAKANA -> 1.0d;
+ default -> Character.isWhitespace(codePoint) ? 0.1d : 0.25d;
+ };
+ offset += Character.charCount(codePoint);
+ }
+ return Math.max(1, (int) Math.ceil(tokens));
+ }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentTextSegment.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentTextSegment.java
new file mode 100644
index 0000000..774be37
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/DocumentTextSegment.java
@@ -0,0 +1,80 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0 (the "License");
+ * you may not use this file except in compliance with the License.
+ * You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text;
+
+import java.util.ArrayList;
+import java.util.List;
+
+/**
+ * 带稳定定位信息的文档文本片段。
+ */
+public class DocumentTextSegment {
+
+ private String segmentId;
+ private String text;
+ private String locatorType;
+ private String locatorLabel;
+ private int startIndex;
+ private int endIndex;
+ private List headingPath = new ArrayList<>();
+ private int tokenEstimate;
+
+ /** @return 片段 ID */
+ public String getSegmentId() { return segmentId; }
+
+ /** @param segmentId 片段 ID */
+ public void setSegmentId(String segmentId) { this.segmentId = segmentId; }
+
+ /** @return 片段文本 */
+ public String getText() { return text; }
+
+ /** @param text 片段文本 */
+ public void setText(String text) { this.text = text; }
+
+ /** @return 定位类型 */
+ public String getLocatorType() { return locatorType; }
+
+ /** @param locatorType 定位类型 */
+ public void setLocatorType(String locatorType) { this.locatorType = locatorType; }
+
+ /** @return 可读定位标签 */
+ public String getLocatorLabel() { return locatorLabel; }
+
+ /** @param locatorLabel 可读定位标签 */
+ public void setLocatorLabel(String locatorLabel) { this.locatorLabel = locatorLabel; }
+
+ /** @return 全文起始字符下标 */
+ public int getStartIndex() { return startIndex; }
+
+ /** @param startIndex 全文起始字符下标 */
+ public void setStartIndex(int startIndex) { this.startIndex = startIndex; }
+
+ /** @return 全文结束字符下标 */
+ public int getEndIndex() { return endIndex; }
+
+ /** @param endIndex 全文结束字符下标 */
+ public void setEndIndex(int endIndex) { this.endIndex = endIndex; }
+
+ /** @return 标题路径 */
+ public List getHeadingPath() { return headingPath; }
+
+ /** @param headingPath 标题路径 */
+ public void setHeadingPath(List headingPath) {
+ this.headingPath = headingPath == null ? new ArrayList<>() : new ArrayList<>(headingPath);
+ }
+
+ /** @return Token 估算 */
+ public int getTokenEstimate() { return tokenEstimate; }
+
+ /** @param tokenEstimate Token 估算 */
+ public void setTokenEstimate(int tokenEstimate) { this.tokenEstimate = tokenEstimate; }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/File2TextService.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/File2TextService.java
index 281a1ae..e65a2bb 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/File2TextService.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/File2TextService.java
@@ -22,9 +22,13 @@ import com.easyagents.core.file2text.source.*;
import java.io.File;
import java.io.InputStream;
+import java.io.IOException;
import java.util.List;
import java.util.stream.Collectors;
+/**
+ * 文档轻量读取服务。
+ */
public class File2TextService {
private static final org.slf4j.Logger log = org.slf4j.LoggerFactory.getLogger(File2TextService.class);
private final ExtractorRegistry registry;
@@ -75,6 +79,44 @@ public class File2TextService {
* @throws IllegalArgumentException 输入源为空
*/
public String extractTextFromSource(DocumentSource source) {
+ return readFromSource(new LightweightDocumentReadRequest(source)).getText();
+ }
+
+ /**
+ * 从文件读取结构化文档内容。
+ *
+ * @param file 文档文件
+ * @return 结构化结果
+ */
+ public LightweightDocumentReadResult readFromFile(File file) {
+ return readFromSource(new LightweightDocumentReadRequest(new FileDocumentSource(file)));
+ }
+
+ /**
+ * 从输入流读取结构化文档内容。
+ *
+ * @param inputStream 文档输入流
+ * @param fileName 文件名
+ * @param mimeType MIME 类型
+ * @return 结构化结果
+ */
+ public LightweightDocumentReadResult readFromStream(InputStream inputStream, String fileName, String mimeType) {
+ return readFromSource(new LightweightDocumentReadRequest(
+ new ByteStreamDocumentSource(inputStream, fileName, mimeType)));
+ }
+
+ /**
+ * 按请求读取结构化文档内容。
+ *
+ * @param request 读取请求
+ * @return 结构化结果
+ * @throws DocumentReadException 不支持、空文本或读取失败
+ */
+ public LightweightDocumentReadResult readFromSource(LightweightDocumentReadRequest request) {
+ if (request == null || request.getSource() == null) {
+ throw new IllegalArgumentException("Document read request cannot be null");
+ }
+ DocumentSource source = request.getSource();
if (source == null) {
throw new IllegalArgumentException("DocumentSource cannot be null");
}
@@ -83,8 +125,8 @@ public class File2TextService {
// 获取可用的 Extractor(按优先级排序)
List candidates = registry.findExtractors(source);
if (candidates.isEmpty()) {
- log.warn("No extractor supports this document: " + safeFileName(source));
- return null;
+ throw new DocumentReadException(DocumentReadErrorCode.UNSUPPORTED_DOCUMENT_TYPE,
+ "Unsupported document type: " + safeFileName(source));
}
// 日志:输出候选 Extractor
@@ -93,29 +135,41 @@ public class File2TextService {
.map(e -> e.getClass().getSimpleName())
.collect(Collectors.joining(", ")));
-
+ DocumentReadException lastFailure = null;
for (FileExtractor extractor : candidates) {
try {
log.debug("Trying {} on {}", extractor.getClass().getSimpleName(), safeFileName(source));
-
- String text = extractor.extractText(source);
- if (text != null && !text.trim().isEmpty()) {
+ LightweightDocumentReadResult result = extractor.read(request);
+ if (result != null && !result.getSegments().isEmpty()) {
log.debug("Success with {}: extracted {} chars",
- extractor.getClass().getSimpleName(), text.length());
- return text;
- } else {
- log.debug("Extractor {} returned null", extractor.getClass().getSimpleName());
+ extractor.getClass().getSimpleName(), result.getCharCount());
+ return result;
}
- } catch (Exception e) {
+ } catch (DocumentReadException e) {
+ lastFailure = e;
+ log.warn("Extractor {} rejected {} with {}: {}",
+ extractor.getClass().getSimpleName(), safeFileName(source),
+ e.getErrorCode(), e.getMessage());
+ } catch (IOException e) {
+ lastFailure = new DocumentReadException(DocumentReadErrorCode.DOCUMENT_READ_FAILED,
+ "Failed to read document: " + safeFileName(source), e);
log.warn("Extractor {} failed on {}: {}",
extractor.getClass().getSimpleName(),
safeFileName(source),
e.toString());
+ } catch (RuntimeException e) {
+ lastFailure = new DocumentReadException(DocumentReadErrorCode.DOCUMENT_READ_FAILED,
+ "Failed to read document: " + safeFileName(source), e);
+ log.warn("Extractor {} failed on {}: {}",
+ extractor.getClass().getSimpleName(), safeFileName(source), e.toString());
}
}
- log.warn(String.format("All %d extractors failed for: %s", candidates.size(), safeFileName(source)));
- return null;
+ if (lastFailure != null) {
+ throw lastFailure;
+ }
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_NO_READABLE_TEXT,
+ "No readable text detected: " + safeFileName(source));
} finally {
source.cleanup();
}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadRequest.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadRequest.java
new file mode 100644
index 0000000..1b8d835
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadRequest.java
@@ -0,0 +1,100 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text;
+
+import com.easyagents.core.file2text.source.DocumentSource;
+
+import java.util.Objects;
+import java.util.function.BooleanSupplier;
+
+/**
+ * 轻量文档读取请求及结构安全边界。
+ */
+public class LightweightDocumentReadRequest {
+
+ private final DocumentSource source;
+ private int maxPdfPages = 200;
+ private int maxSlides = 200;
+ private int maxSheets = 20;
+ private int maxNonEmptyCells = 50_000;
+ private long maxExpandedChars = 150L * 1024L * 1024L;
+ private BooleanSupplier cancelled = () -> false;
+
+ /**
+ * 创建读取请求。
+ *
+ * @param source 文档来源
+ */
+ public LightweightDocumentReadRequest(DocumentSource source) {
+ this.source = Objects.requireNonNull(source, "DocumentSource cannot be null");
+ }
+
+ /** @return 文档来源 */
+ public DocumentSource getSource() { return source; }
+
+ /** @return 最大 PDF 页数 */
+ public int getMaxPdfPages() { return maxPdfPages; }
+
+ /** @param maxPdfPages 最大 PDF 页数 */
+ public void setMaxPdfPages(int maxPdfPages) { this.maxPdfPages = positive(maxPdfPages, "maxPdfPages"); }
+
+ /** @return 最大幻灯片数 */
+ public int getMaxSlides() { return maxSlides; }
+
+ /** @param maxSlides 最大幻灯片数 */
+ public void setMaxSlides(int maxSlides) { this.maxSlides = positive(maxSlides, "maxSlides"); }
+
+ /** @return 最大工作表数 */
+ public int getMaxSheets() { return maxSheets; }
+
+ /** @param maxSheets 最大工作表数 */
+ public void setMaxSheets(int maxSheets) { this.maxSheets = positive(maxSheets, "maxSheets"); }
+
+ /** @return 最大非空单元格数 */
+ public int getMaxNonEmptyCells() { return maxNonEmptyCells; }
+
+ /** @param maxNonEmptyCells 最大非空单元格数 */
+ public void setMaxNonEmptyCells(int maxNonEmptyCells) {
+ this.maxNonEmptyCells = positive(maxNonEmptyCells, "maxNonEmptyCells");
+ }
+
+ /** @return 最大展开字符数 */
+ public long getMaxExpandedChars() { return maxExpandedChars; }
+
+ /** @param maxExpandedChars 最大展开字符数 */
+ public void setMaxExpandedChars(long maxExpandedChars) {
+ if (maxExpandedChars <= 0) {
+ throw new IllegalArgumentException("maxExpandedChars must be positive");
+ }
+ this.maxExpandedChars = maxExpandedChars;
+ }
+
+ /** @return 取消检查器 */
+ public BooleanSupplier getCancelled() { return cancelled; }
+
+ /** @param cancelled 取消检查器 */
+ public void setCancelled(BooleanSupplier cancelled) {
+ this.cancelled = cancelled == null ? () -> false : cancelled;
+ }
+
+ /**
+ * 检查当前读取是否已取消。
+ *
+ * @throws DocumentReadException 已取消时抛出
+ */
+ public void checkCancelled() throws DocumentReadException {
+ if (cancelled.getAsBoolean() || Thread.currentThread().isInterrupted()) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_READ_CANCELLED, "Document read cancelled");
+ }
+ }
+
+ private int positive(int value, String name) {
+ if (value <= 0) {
+ throw new IllegalArgumentException(name + " must be positive");
+ }
+ return value;
+ }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadResult.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadResult.java
new file mode 100644
index 0000000..b1494cd
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/LightweightDocumentReadResult.java
@@ -0,0 +1,91 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text;
+
+import java.util.ArrayList;
+import java.util.List;
+
+/**
+ * 轻量文档结构化读取结果。
+ */
+public class LightweightDocumentReadResult {
+
+ /** 当前读取器版本。 */
+ public static final String READER_VERSION = "v1";
+ /** 当前读取策略版本。 */
+ public static final String READ_POLICY_VERSION = "v1";
+
+ private String fileName;
+ private String mimeType;
+ private String readerVersion = READER_VERSION;
+ private String readPolicyVersion = READ_POLICY_VERSION;
+ private int charCount;
+ private int tokenEstimate;
+ private List segments = new ArrayList<>();
+
+ /** @return 文件名 */
+ public String getFileName() { return fileName; }
+
+ /** @param fileName 文件名 */
+ public void setFileName(String fileName) { this.fileName = fileName; }
+
+ /** @return MIME 类型 */
+ public String getMimeType() { return mimeType; }
+
+ /** @param mimeType MIME 类型 */
+ public void setMimeType(String mimeType) { this.mimeType = mimeType; }
+
+ /** @return 读取器版本 */
+ public String getReaderVersion() { return readerVersion; }
+
+ /** @param readerVersion 读取器版本 */
+ public void setReaderVersion(String readerVersion) { this.readerVersion = readerVersion; }
+
+ /** @return 读取策略版本 */
+ public String getReadPolicyVersion() { return readPolicyVersion; }
+
+ /** @param readPolicyVersion 读取策略版本 */
+ public void setReadPolicyVersion(String readPolicyVersion) { this.readPolicyVersion = readPolicyVersion; }
+
+ /** @return 字符数 */
+ public int getCharCount() { return charCount; }
+
+ /** @param charCount 字符数 */
+ public void setCharCount(int charCount) { this.charCount = charCount; }
+
+ /** @return Token 估算 */
+ public int getTokenEstimate() { return tokenEstimate; }
+
+ /** @param tokenEstimate Token 估算 */
+ public void setTokenEstimate(int tokenEstimate) { this.tokenEstimate = tokenEstimate; }
+
+ /** @return 文本片段 */
+ public List getSegments() { return segments; }
+
+ /** @param segments 文本片段 */
+ public void setSegments(List segments) {
+ this.segments = segments == null ? new ArrayList<>() : new ArrayList<>(segments);
+ }
+
+ /**
+ * 按片段顺序拼接兼容纯文本。
+ *
+ * @return 拼接后的文本
+ */
+ public String getText() {
+ StringBuilder text = new StringBuilder(Math.max(0, charCount));
+ for (DocumentTextSegment segment : segments) {
+ if (segment.getText() == null || segment.getText().isBlank()) {
+ continue;
+ }
+ if (!text.isEmpty()) {
+ text.append('\n');
+ }
+ text.append(segment.getText());
+ }
+ return text.toString();
+ }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/ExtractorRegistry.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/ExtractorRegistry.java
index a2a905a..96d5694 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/ExtractorRegistry.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/ExtractorRegistry.java
@@ -35,7 +35,10 @@ public class ExtractorRegistry {
register(new PdfTextExtractor());
register(new DocxExtractor());
register(new DocExtractor());
+ register(new PptExtractor());
register(new PptxExtractor());
+ register(new XlsExtractor());
+ register(new XlsxExtractor());
register(new HtmlExtractor());
register(new PlainTextExtractor());
}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/FileExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/FileExtractor.java
index 62d50ca..44893fd 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/FileExtractor.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/FileExtractor.java
@@ -15,24 +15,59 @@
*/
package com.easyagents.core.file2text.extractor;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
import com.easyagents.core.file2text.source.DocumentSource;
import java.io.IOException;
import java.util.Comparator;
+import java.util.List;
+/**
+ * 文档文本提取器。
+ */
public interface FileExtractor {
Comparator ORDER_COMPARATOR =
Comparator.comparingInt(FileExtractor::getOrder);
/**
- * 判断该 Extractor 是否支持处理此文档
+ * 判断该 Extractor 是否支持处理此文档。
+ *
+ * @param source 文档来源
+ * @return 是否支持
*/
boolean supports(DocumentSource source);
+ /**
+ * 提取兼容纯文本。
+ *
+ * @param source 文档来源
+ * @return 提取文本
+ * @throws IOException 文档读取失败
+ */
String extractText(DocumentSource source) throws IOException;
+ /**
+ * 提取结构化文档片段。
+ *
+ * @param request 读取请求
+ * @return 结构化结果
+ * @throws IOException 文档读取失败
+ */
+ default LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ String text = extractText(request.getSource());
+ return DocumentReadSupport.result(request.getSource(),
+ List.of(DocumentReadSupport.segment("segment-1", text, "DOCUMENT", "全文", 0, List.of())),
+ request);
+ }
+ /**
+ * 获取读取器优先级。
+ *
+ * @return 越小越优先
+ */
default int getOrder() {
return 100;
}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocExtractor.java
index 46f8d22..f1d0aa1 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocExtractor.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocExtractor.java
@@ -15,6 +15,12 @@
*/
package com.easyagents.core.file2text.extractor.impl;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.DocumentTextSegment;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
import com.easyagents.core.file2text.extractor.FileExtractor;
import com.easyagents.core.file2text.source.DocumentSource;
import org.apache.poi.hwpf.HWPFDocument;
@@ -24,7 +30,9 @@ import org.apache.poi.poifs.filesystem.POIFSFileSystem;
import java.io.IOException;
import java.io.InputStream;
import java.util.Collections;
+import java.util.ArrayList;
import java.util.HashSet;
+import java.util.List;
import java.util.Set;
/**
@@ -68,25 +76,47 @@ public class DocExtractor implements FileExtractor {
@Override
public String extractText(DocumentSource source) throws IOException {
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+
+ /**
+ * 按段落读取 Word 97-2003 文档。
+ *
+ * @param request 读取请求
+ * @return 段落结构化结果
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
try (InputStream is = source.openStream();
POIFSFileSystem fs = new POIFSFileSystem(is);
HWPFDocument doc = new HWPFDocument(fs)) {
- WordExtractor extractor = new WordExtractor(doc);
- String[] paragraphs = extractor.getParagraphText();
-
- StringBuilder text = new StringBuilder();
- for (String para : paragraphs) {
+ try (WordExtractor extractor = new WordExtractor(doc)) {
+ String[] paragraphs = extractor.getParagraphText();
+ List segments = new ArrayList<>();
+ int offset = 0;
+ for (int index = 0; index < paragraphs.length; index++) {
+ request.checkCancelled();
+ String para = paragraphs[index];
// 清理控制字符
- String clean = para.replaceAll("[\\r\\001]+", "").trim();
- if (!clean.isEmpty()) {
- text.append(clean).append("\n");
+ String clean = para.replaceAll("[\\r\\001]+", "").trim();
+ if (!clean.isEmpty()) {
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "paragraph-" + (index + 1), clean, "PARAGRAPH",
+ "第 " + (index + 1) + " 段", offset, List.of());
+ segments.add(segment);
+ offset = segment.getEndIndex() + 1;
+ }
}
+ return DocumentReadSupport.result(source, segments, request);
}
-
- return text.toString().trim();
+ } catch (DocumentReadException e) {
+ throw e;
} catch (Exception e) {
- throw new IOException("Failed to extract .doc file: " + e.getMessage(), e);
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Failed to extract .doc file", e);
}
}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocxExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocxExtractor.java
index 3f8dc4d..1e74973 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocxExtractor.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/DocxExtractor.java
@@ -16,12 +16,19 @@
package com.easyagents.core.file2text.extractor.impl;
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.DocumentTextSegment;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
import com.easyagents.core.file2text.extractor.FileExtractor;
import com.easyagents.core.file2text.source.DocumentSource;
import org.apache.poi.xwpf.usermodel.*;
import java.io.IOException;
import java.io.InputStream;
+import java.util.ArrayList;
import java.util.Collections;
import java.util.HashSet;
import java.util.List;
@@ -79,37 +86,66 @@ public class DocxExtractor implements FileExtractor {
@Override
public String extractText(DocumentSource source) throws IOException {
- StringBuilder text = new StringBuilder();
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+ /**
+ * 按正文顺序读取 DOCX 段落和表格。
+ *
+ * @param request 读取请求
+ * @return 结构化读取结果
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
try (InputStream is = source.openStream();
XWPFDocument document = new XWPFDocument(is)) {
-
- // 提取段落
- for (XWPFParagraph paragraph : document.getParagraphs()) {
- String paraText = getParagraphText(paragraph);
- if (paraText != null && !paraText.trim().isEmpty()) {
- text.append(paraText).append("\n");
+ List segments = new ArrayList<>();
+ List headingPath = new ArrayList<>();
+ int paragraphIndex = 0;
+ int tableIndex = 0;
+ int offset = 0;
+ for (IBodyElement element : document.getBodyElements()) {
+ request.checkCancelled();
+ if (element instanceof XWPFParagraph paragraph) {
+ paragraphIndex++;
+ String text = getParagraphText(paragraph);
+ if (text == null || text.isBlank()) {
+ continue;
+ }
+ int headingLevel = headingLevel(paragraph);
+ if (headingLevel > 0) {
+ while (headingPath.size() >= headingLevel) {
+ headingPath.remove(headingPath.size() - 1);
+ }
+ headingPath.add(text.trim());
+ }
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "paragraph-" + paragraphIndex, text, "PARAGRAPH",
+ "第 " + paragraphIndex + " 段", offset, headingPath);
+ segments.add(segment);
+ offset = segment.getEndIndex() + 1;
+ } else if (element instanceof XWPFTable table) {
+ tableIndex++;
+ String tableText = getTableText(table);
+ if (tableText.isBlank()) {
+ continue;
+ }
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "table-" + tableIndex, tableText, "TABLE",
+ "第 " + tableIndex + " 个表格", offset, headingPath);
+ segments.add(segment);
+ offset = segment.getEndIndex() + 1;
}
}
-
- // 提取表格
- for (XWPFTable table : document.getTables()) {
- text.append("\n[Table Start]\n");
- for (XWPFTableRow row : table.getRows()) {
- List cellTexts = row.getTableCells().stream()
- .map(this::getCellText)
- .map(String::trim)
- .collect(Collectors.toList());
- text.append(cellTexts).append("\n");
- }
- text.append("[Table End]\n\n");
- }
-
+ return DocumentReadSupport.result(source, segments, request);
+ } catch (DocumentReadException e) {
+ throw e;
} catch (Exception e) {
- throw new IOException("Failed to extract DOCX: " + e.getMessage(), e);
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Failed to extract DOCX", e);
}
-
- return text.toString().trim();
}
private String getParagraphText(XWPFParagraph paragraph) {
@@ -138,6 +174,52 @@ public class DocxExtractor implements FileExtractor {
return text.toString().trim();
}
+ /**
+ * 获取表格的显示文本。
+ *
+ * @param table 表格
+ * @return 表格文本
+ */
+ private String getTableText(XWPFTable table) {
+ StringBuilder text = new StringBuilder();
+ for (XWPFTableRow row : table.getRows()) {
+ List cellTexts = row.getTableCells().stream()
+ .map(this::getCellText)
+ .map(String::trim)
+ .collect(Collectors.toList());
+ if (cellTexts.stream().anyMatch(item -> !item.isEmpty())) {
+ text.append(String.join(" | ", cellTexts)).append('\n');
+ }
+ }
+ return text.toString().trim();
+ }
+
+ /**
+ * 解析常见 Word 标题样式层级。
+ *
+ * @param paragraph 段落
+ * @return 标题层级,非标题返回 0
+ */
+ private int headingLevel(XWPFParagraph paragraph) {
+ String style = paragraph.getStyle();
+ if (style == null) {
+ return 0;
+ }
+ String normalized = style.replaceAll("\\s+", "").toLowerCase();
+ if (!normalized.startsWith("heading") && !normalized.startsWith("标题")) {
+ return 0;
+ }
+ String digits = normalized.replaceAll("\\D+", "");
+ if (digits.isEmpty()) {
+ return 1;
+ }
+ try {
+ return Math.max(1, Math.min(9, Integer.parseInt(digits)));
+ } catch (NumberFormatException ignored) {
+ return 1;
+ }
+ }
+
@Override
public int getOrder() {
return 10;
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PdfTextExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PdfTextExtractor.java
index 8c28c08..61acbf4 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PdfTextExtractor.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PdfTextExtractor.java
@@ -15,15 +15,24 @@
*/
package com.easyagents.core.file2text.extractor.impl;
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.DocumentTextSegment;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
import com.easyagents.core.file2text.extractor.FileExtractor;
import com.easyagents.core.file2text.source.DocumentSource;
import org.apache.pdfbox.pdmodel.PDDocument;
+import org.apache.pdfbox.pdmodel.encryption.InvalidPasswordException;
import org.apache.pdfbox.text.PDFTextStripper;
import java.io.IOException;
import java.io.InputStream;
import java.util.Collections;
+import java.util.ArrayList;
import java.util.HashSet;
+import java.util.List;
import java.util.Set;
/**
@@ -64,12 +73,54 @@ public class PdfTextExtractor implements FileExtractor {
@Override
public String extractText(DocumentSource source) throws IOException {
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+
+ /**
+ * 按页读取 PDF 文本层。
+ *
+ * @param request 读取请求
+ * @return 按页结构化结果
+ * @throws IOException PDF I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
try (InputStream is = source.openStream();
PDDocument doc = PDDocument.load(is)) {
+ if (doc.isEncrypted()) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_ENCRYPTED,
+ "Encrypted PDF is not supported");
+ }
+ int pages = doc.getNumberOfPages();
+ if (pages > request.getMaxPdfPages()) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_STRUCTURE_LIMIT_EXCEEDED,
+ "PDF page count exceeds " + request.getMaxPdfPages());
+ }
PDFTextStripper stripper = new PDFTextStripper();
- return stripper.getText(doc).trim();
+ List segments = new ArrayList<>();
+ int offset = 0;
+ for (int page = 1; page <= pages; page++) {
+ request.checkCancelled();
+ stripper.setStartPage(page);
+ stripper.setEndPage(page);
+ String text = stripper.getText(doc).trim();
+ if (!text.isEmpty()) {
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "page-" + page, text, "PAGE", "第 " + page + " 页", offset, List.of());
+ segments.add(segment);
+ offset = segment.getEndIndex() + 1;
+ }
+ }
+ return DocumentReadSupport.result(source, segments, request);
+ } catch (InvalidPasswordException e) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_ENCRYPTED,
+ "Encrypted PDF is not supported", e);
+ } catch (DocumentReadException e) {
+ throw e;
} catch (Exception e) {
- throw new IOException("Failed to extract PDF text: " + e.getMessage(), e);
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Failed to extract PDF text", e);
}
}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PlainTextExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PlainTextExtractor.java
index 2029f8a..10add06 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PlainTextExtractor.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PlainTextExtractor.java
@@ -16,6 +16,12 @@
package com.easyagents.core.file2text.extractor.impl;
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.DocumentTextSegment;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
import com.easyagents.core.file2text.extractor.FileExtractor;
import com.easyagents.core.file2text.source.DocumentSource;
@@ -23,9 +29,17 @@ import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
+import java.nio.charset.CharacterCodingException;
+import java.nio.charset.Charset;
+import java.nio.charset.CodingErrorAction;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
import java.util.Collections;
import java.util.HashSet;
+import java.util.List;
import java.util.Set;
+import java.util.regex.Matcher;
+import java.util.regex.Pattern;
/**
* 纯文本文件提取器(支持 UTF-8、GBK、GB2312 编码自动检测)
@@ -33,6 +47,9 @@ import java.util.Set;
*/
public class PlainTextExtractor implements FileExtractor {
+ private static final int LINES_PER_SEGMENT = 40;
+ private static final Charset GB18030 = Charset.forName("GB18030");
+ private static final Pattern MARKDOWN_HEADING = Pattern.compile("^(#{1,6})\\s+(.+?)\\s*$");
private static final Set SUPPORTED_MIME_TYPES;
private static final Set SUPPORTED_EXTENSIONS;
@@ -80,21 +97,160 @@ public class PlainTextExtractor implements FileExtractor {
@Override
public String extractText(DocumentSource source) throws IOException {
- try (InputStream is = source.openStream()) {
- try (BufferedReader reader = new BufferedReader(new InputStreamReader(is, "utf-8"))) {
- StringBuilder text = new StringBuilder();
- char[] buffer = new char[8192];
- int read;
- while ((read = reader.read(buffer)) != -1) {
- text.append(buffer, 0, read);
- }
- return text.toString().trim();
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+
+ /**
+ * 按文本行区间读取 TXT 或 Markdown。
+ *
+ * @param request 读取请求
+ * @return 行区间结构化结果
+ * @throws IOException 文本 I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
+ CharsetDetection detection = detectCharset(source);
+ try {
+ return readWithCharset(request, detection.charset(), detection.bomLength());
+ } catch (CharacterCodingException error) {
+ if (!StandardCharsets.UTF_8.equals(detection.charset()) || detection.bomLength() > 0) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Text encoding is invalid", error);
}
- } catch (Exception e) {
- throw new RuntimeException(e);
+ // 严格 UTF-8 解码失败时仅回退到受控 GB18030。
+ return readWithCharset(request, GB18030, 0);
+ } catch (DocumentReadException error) {
+ throw error;
+ } catch (Exception error) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_READ_FAILED,
+ "Failed to read text document", error);
}
}
+ /**
+ * 使用指定字符集流式读取文本。
+ *
+ * @param request 读取请求
+ * @param charset 字符集
+ * @param bomLength BOM 长度
+ * @return 结构化结果
+ * @throws Exception 打开或读取失败
+ */
+ private LightweightDocumentReadResult readWithCharset(LightweightDocumentReadRequest request,
+ Charset charset,
+ int bomLength) throws IOException {
+ DocumentSource source = request.getSource();
+ List segments = new ArrayList<>();
+ List headingPath = new ArrayList<>();
+ boolean markdown = isMarkdown(source.getFileName(), source.getMimeType());
+ try (InputStream input = openStream(source)) {
+ input.skipNBytes(bomLength);
+ InputStreamReader streamReader = new InputStreamReader(input,
+ charset.newDecoder()
+ .onMalformedInput(CodingErrorAction.REPORT)
+ .onUnmappableCharacter(CodingErrorAction.REPORT));
+ try (BufferedReader reader = new BufferedReader(streamReader, 8192)) {
+ StringBuilder block = new StringBuilder();
+ int line = 0;
+ int blockStart = 1;
+ int offset = 0;
+ String value;
+ while ((value = reader.readLine()) != null) {
+ request.checkCancelled();
+ line++;
+ if (markdown) {
+ updateHeadingPath(headingPath, value);
+ }
+ if (!block.isEmpty()) {
+ block.append('\n');
+ }
+ block.append(value);
+ if (line - blockStart + 1 >= LINES_PER_SEGMENT) {
+ DocumentTextSegment segment = addLineSegment(
+ segments, block, blockStart, line, offset, headingPath);
+ offset = segment == null ? offset : segment.getEndIndex() + 1;
+ block.setLength(0);
+ blockStart = line + 1;
+ }
+ if ((long) offset + block.length() > request.getMaxExpandedChars()) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_STRUCTURE_LIMIT_EXCEEDED,
+ "Expanded text exceeds the configured limit");
+ }
+ }
+ if (!block.isEmpty()) {
+ addLineSegment(segments, block, blockStart, line, offset, headingPath);
+ }
+ }
+ }
+ return DocumentReadSupport.result(source, segments, request);
+ }
+
+ private InputStream openStream(DocumentSource source) throws IOException {
+ try {
+ return source.openStream();
+ } catch (IOException error) {
+ throw error;
+ } catch (Exception error) {
+ throw new IOException("Failed to open text document", error);
+ }
+ }
+
+ private DocumentTextSegment addLineSegment(List segments,
+ StringBuilder block,
+ int startLine,
+ int endLine,
+ int offset,
+ List headingPath) {
+ if (block.toString().isBlank()) {
+ return null;
+ }
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "lines-" + startLine + "-" + endLine, block.toString(), "LINE_RANGE",
+ "第 " + startLine + "-" + endLine + " 行", offset, headingPath);
+ segments.add(segment);
+ return segment;
+ }
+
+ private void updateHeadingPath(List headingPath, String line) {
+ Matcher matcher = MARKDOWN_HEADING.matcher(line);
+ if (!matcher.matches()) {
+ return;
+ }
+ int level = matcher.group(1).length();
+ while (headingPath.size() >= level) {
+ headingPath.remove(headingPath.size() - 1);
+ }
+ headingPath.add(matcher.group(2).trim());
+ }
+
+ private CharsetDetection detectCharset(DocumentSource source) throws IOException {
+ try (InputStream input = source.openStream()) {
+ byte[] prefix = input.readNBytes(3);
+ if (prefix.length >= 3 && (prefix[0] & 0xff) == 0xef
+ && (prefix[1] & 0xff) == 0xbb && (prefix[2] & 0xff) == 0xbf) {
+ return new CharsetDetection(StandardCharsets.UTF_8, 3);
+ }
+ if (prefix.length >= 2 && (prefix[0] & 0xff) == 0xff && (prefix[1] & 0xff) == 0xfe) {
+ return new CharsetDetection(StandardCharsets.UTF_16LE, 2);
+ }
+ if (prefix.length >= 2 && (prefix[0] & 0xff) == 0xfe && (prefix[1] & 0xff) == 0xff) {
+ return new CharsetDetection(StandardCharsets.UTF_16BE, 2);
+ }
+ return new CharsetDetection(StandardCharsets.UTF_8, 0);
+ } catch (IOException error) {
+ throw error;
+ } catch (Exception error) {
+ throw new IOException("Failed to inspect text encoding", error);
+ }
+ }
+
+ private boolean isMarkdown(String fileName, String mimeType) {
+ return "text/markdown".equalsIgnoreCase(mimeType)
+ || (fileName != null && (fileName.toLowerCase().endsWith(".md")
+ || fileName.toLowerCase().endsWith(".markdown")));
+ }
+
@Override
public int getOrder() {
@@ -106,4 +262,7 @@ public class PlainTextExtractor implements FileExtractor {
int lastDot = fileName.lastIndexOf('.');
return fileName.substring(lastDot + 1).toLowerCase();
}
+
+ private record CharsetDetection(Charset charset, int bomLength) {
+ }
}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptExtractor.java
new file mode 100644
index 0000000..5047489
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptExtractor.java
@@ -0,0 +1,122 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text.extractor.impl;
+
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.DocumentTextSegment;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
+import com.easyagents.core.file2text.extractor.FileExtractor;
+import com.easyagents.core.file2text.source.DocumentSource;
+import org.apache.poi.hslf.usermodel.HSLFShape;
+import org.apache.poi.hslf.usermodel.HSLFSlide;
+import org.apache.poi.hslf.usermodel.HSLFSlideShow;
+import org.apache.poi.hslf.usermodel.HSLFTextShape;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
+import java.util.Set;
+
+/**
+ * PowerPoint 97-2003 文档提取器。
+ */
+public class PptExtractor implements FileExtractor {
+
+ private static final Set MIME_TYPES = Set.of(
+ "application/vnd.ms-powerpoint",
+ "application/mspowerpoint",
+ "application/powerpoint");
+
+ /**
+ * 判断是否支持 PPT。
+ *
+ * @param source 文档来源
+ * @return 是否支持
+ */
+ @Override
+ public boolean supports(DocumentSource source) {
+ if (source.getMimeType() != null && MIME_TYPES.contains(source.getMimeType().toLowerCase(Locale.ROOT))) {
+ return true;
+ }
+ String fileName = source.getFileName();
+ return fileName != null && (fileName.toLowerCase(Locale.ROOT).endsWith(".ppt")
+ || fileName.toLowerCase(Locale.ROOT).endsWith(".pps"));
+ }
+
+ /**
+ * 提取兼容纯文本。
+ *
+ * @param source 文档来源
+ * @return 文本
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public String extractText(DocumentSource source) throws IOException {
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+
+ /**
+ * 按幻灯片读取 PPT 文本。
+ *
+ * @param request 读取请求
+ * @return 幻灯片结构化结果
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
+ try (InputStream input = source.openStream();
+ HSLFSlideShow slideShow = new HSLFSlideShow(input)) {
+ List slides = slideShow.getSlides();
+ if (slides.size() > request.getMaxSlides()) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_STRUCTURE_LIMIT_EXCEEDED,
+ "Slide count exceeds " + request.getMaxSlides());
+ }
+ List segments = new ArrayList<>();
+ int offset = 0;
+ for (int index = 0; index < slides.size(); index++) {
+ request.checkCancelled();
+ StringBuilder text = new StringBuilder();
+ for (HSLFShape shape : slides.get(index).getShapes()) {
+ if (shape instanceof HSLFTextShape textShape) {
+ String value = textShape.getText();
+ if (value != null && !value.isBlank()) {
+ text.append(value.trim()).append('\n');
+ }
+ }
+ }
+ if (!text.toString().isBlank()) {
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "slide-" + (index + 1), text.toString(), "SLIDE",
+ "第 " + (index + 1) + " 张幻灯片", offset, List.of());
+ segments.add(segment);
+ offset = segment.getEndIndex() + 1;
+ }
+ }
+ return DocumentReadSupport.result(source, segments, request);
+ } catch (DocumentReadException error) {
+ throw error;
+ } catch (Exception error) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Failed to extract PPT", error);
+ }
+ }
+
+ /**
+ * 获取读取器优先级。
+ *
+ * @return 优先级
+ */
+ @Override
+ public int getOrder() {
+ return 12;
+ }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptxExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptxExtractor.java
index bc2f085..8b73cfd 100644
--- a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptxExtractor.java
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/PptxExtractor.java
@@ -15,6 +15,12 @@
*/
package com.easyagents.core.file2text.extractor.impl;
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.DocumentTextSegment;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
import com.easyagents.core.file2text.extractor.FileExtractor;
import com.easyagents.core.file2text.source.DocumentSource;
import org.apache.poi.xslf.usermodel.*;
@@ -77,21 +83,37 @@ public class PptxExtractor implements FileExtractor {
@Override
public String extractText(DocumentSource source) throws IOException {
- StringBuilder text = new StringBuilder();
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+ /**
+ * 按幻灯片读取 PPTX 文本。
+ *
+ * @param request 读取请求
+ * @return 幻灯片结构化结果
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
try (InputStream is = source.openStream();
XMLSlideShow slideShow = new XMLSlideShow(is)) {
List slides = slideShow.getSlides();
-
+ if (slides.size() > request.getMaxSlides()) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_STRUCTURE_LIMIT_EXCEEDED,
+ "Slide count exceeds " + request.getMaxSlides());
+ }
+ List segments = new ArrayList<>();
+ int offset = 0;
for (int i = 0; i < slides.size(); i++) {
+ request.checkCancelled();
XSLFSlide slide = slides.get(i);
- text.append("\n--- Slide ").append(i + 1).append(" ---\n");
+ StringBuilder text = new StringBuilder();
// 提取所有形状中的文本
for (XSLFShape shape : slide.getShapes()) {
- if (shape instanceof XSLFTextShape) {
- XSLFTextShape textShape = (XSLFTextShape) shape;
+ if (shape instanceof XSLFTextShape textShape && !(shape instanceof XSLFTable)) {
String shapeText = textShape.getText();
if (shapeText != null && !shapeText.trim().isEmpty()) {
text.append(shapeText).append("\n");
@@ -101,15 +123,24 @@ public class PptxExtractor implements FileExtractor {
// 可选:提取表格
extractTablesFromSlide(slide, text);
+ if (!text.toString().isBlank()) {
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "slide-" + (i + 1), text.toString(), "SLIDE",
+ "第 " + (i + 1) + " 张幻灯片", offset, List.of());
+ segments.add(segment);
+ offset = segment.getEndIndex() + 1;
+ }
}
-
+ return DocumentReadSupport.result(source, segments, request);
+ } catch (DocumentReadException e) {
+ throw e;
} catch (XmlException e) {
- throw new IOException("Invalid PPTX structure: " + e.getMessage(), e);
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Invalid PPTX structure", e);
} catch (Exception e) {
- throw new IOException("Failed to extract PPTX: " + e.getMessage(), e);
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Failed to extract PPTX", e);
}
-
- return text.toString().trim();
}
/**
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/SpreadsheetReadSupport.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/SpreadsheetReadSupport.java
new file mode 100644
index 0000000..5ce3d8b
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/SpreadsheetReadSupport.java
@@ -0,0 +1,76 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text.extractor.impl;
+
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.DocumentTextSegment;
+
+import java.util.ArrayList;
+import java.util.List;
+
+/**
+ * 表格行到稳定片段的内部转换工具。
+ */
+final class SpreadsheetReadSupport {
+
+ private static final int ROWS_PER_SEGMENT = 25;
+
+ private SpreadsheetReadSupport() {
+ }
+
+ /**
+ * 将工作表行按固定区间组成片段。
+ *
+ * @param sheets 工作表数据
+ * @return 文本片段
+ */
+ static List toSegments(List sheets) {
+ List segments = new ArrayList<>();
+ int offset = 0;
+ for (SheetRows sheet : sheets) {
+ for (int start = 0; start < sheet.rows().size(); start += ROWS_PER_SEGMENT) {
+ int end = Math.min(sheet.rows().size(), start + ROWS_PER_SEGMENT);
+ List rows = sheet.rows().subList(start, end);
+ StringBuilder text = new StringBuilder();
+ for (RowText row : rows) {
+ if (!text.isEmpty()) {
+ text.append('\n');
+ }
+ text.append("第 ").append(row.rowNumber()).append(" 行: ").append(row.text());
+ }
+ int startRow = rows.get(0).rowNumber();
+ int endRow = rows.get(rows.size() - 1).rowNumber();
+ DocumentTextSegment segment = DocumentReadSupport.segment(
+ "sheet-" + sheet.sheetIndex() + "-rows-" + startRow + "-" + endRow,
+ text.toString(), "SHEET_ROW_RANGE",
+ sheet.sheetName() + " 第 " + startRow + "-" + endRow + " 行",
+ offset, List.of(sheet.sheetName()));
+ segments.add(segment);
+ offset = segment.getEndIndex() + 1;
+ }
+ }
+ return segments;
+ }
+
+ /**
+ * 单个工作表的有效行。
+ *
+ * @param sheetIndex 工作表序号
+ * @param sheetName 工作表名
+ * @param rows 有效行
+ */
+ record SheetRows(int sheetIndex, String sheetName, List rows) {
+ }
+
+ /**
+ * 单行格式化文本。
+ *
+ * @param rowNumber 行号
+ * @param text 文本
+ */
+ record RowText(int rowNumber, String text) {
+ }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsExtractor.java
new file mode 100644
index 0000000..3b00eb9
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsExtractor.java
@@ -0,0 +1,256 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text.extractor.impl;
+
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
+import com.easyagents.core.file2text.extractor.FileExtractor;
+import com.easyagents.core.file2text.source.DocumentSource;
+import org.apache.poi.hssf.eventusermodel.FormatTrackingHSSFListener;
+import org.apache.poi.hssf.eventusermodel.HSSFEventFactory;
+import org.apache.poi.hssf.eventusermodel.HSSFListener;
+import org.apache.poi.hssf.eventusermodel.HSSFRequest;
+import org.apache.poi.hssf.record.BOFRecord;
+import org.apache.poi.hssf.record.BoolErrRecord;
+import org.apache.poi.hssf.record.BoundSheetRecord;
+import org.apache.poi.hssf.record.FormulaRecord;
+import org.apache.poi.hssf.record.LabelRecord;
+import org.apache.poi.hssf.record.LabelSSTRecord;
+import org.apache.poi.hssf.record.NumberRecord;
+import org.apache.poi.hssf.record.Record;
+import org.apache.poi.hssf.record.SSTRecord;
+import org.apache.poi.hssf.record.StringRecord;
+import org.apache.poi.poifs.filesystem.POIFSFileSystem;
+import org.apache.poi.ss.util.CellReference;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.ArrayList;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Locale;
+import java.util.Map;
+import java.util.Set;
+
+/**
+ * 基于 HSSF Event API 的 XLS 文档提取器。
+ */
+public class XlsExtractor implements FileExtractor {
+
+ private static final Set MIME_TYPES = Set.of(
+ "application/vnd.ms-excel",
+ "application/msexcel",
+ "application/x-msexcel");
+
+ /**
+ * 判断是否支持 XLS。
+ *
+ * @param source 文档来源
+ * @return 是否支持
+ */
+ @Override
+ public boolean supports(DocumentSource source) {
+ if (source.getMimeType() != null && MIME_TYPES.contains(source.getMimeType().toLowerCase(Locale.ROOT))) {
+ return true;
+ }
+ String name = source.getFileName();
+ return name != null && (name.toLowerCase(Locale.ROOT).endsWith(".xls")
+ || name.toLowerCase(Locale.ROOT).endsWith(".xlt"));
+ }
+
+ /**
+ * 提取兼容纯文本。
+ *
+ * @param source 文档来源
+ * @return 文本
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public String extractText(DocumentSource source) throws IOException {
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+
+ /**
+ * 使用 HSSF 事件模型按工作表和行读取 XLS。
+ *
+ * @param request 读取请求
+ * @return 表格结构化结果
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
+ try (InputStream input = source.openStream();
+ POIFSFileSystem fileSystem = new POIFSFileSystem(input)) {
+ XlsListener listener = new XlsListener(request);
+ FormatTrackingHSSFListener formatter = new FormatTrackingHSSFListener(listener);
+ listener.setFormatter(formatter);
+ HSSFRequest hssfRequest = new HSSFRequest();
+ hssfRequest.addListenerForAllRecords(formatter);
+ new HSSFEventFactory().processWorkbookEvents(hssfRequest, fileSystem);
+ listener.finish();
+ return DocumentReadSupport.result(source,
+ SpreadsheetReadSupport.toSegments(listener.sheets()), request);
+ } catch (StructureLimitRuntimeException error) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_STRUCTURE_LIMIT_EXCEEDED,
+ error.getMessage(), error);
+ } catch (DocumentReadException error) {
+ throw error;
+ } catch (Exception error) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Failed to extract XLS", error);
+ }
+ }
+
+ /**
+ * 获取读取器优先级。
+ *
+ * @return 优先级
+ */
+ @Override
+ public int getOrder() {
+ return 10;
+ }
+
+ /**
+ * HSSF 二进制记录监听器。
+ */
+ private static final class XlsListener implements HSSFListener {
+
+ private final LightweightDocumentReadRequest request;
+ private final List boundSheets = new ArrayList<>();
+ private final List sheets = new ArrayList<>();
+ private final List currentRows = new ArrayList<>();
+ private final Map currentCells = new LinkedHashMap<>();
+ private FormatTrackingHSSFListener formatter;
+ private SSTRecord sharedStrings;
+ private int sheetIndex;
+ private int currentRow = -1;
+ private int nonEmptyCells;
+ private int pendingFormulaRow = -1;
+ private int pendingFormulaColumn = -1;
+
+ private XlsListener(LightweightDocumentReadRequest request) {
+ this.request = request;
+ }
+
+ private void setFormatter(FormatTrackingHSSFListener formatter) {
+ this.formatter = formatter;
+ }
+
+ /**
+ * 处理一个 HSSF 记录。
+ *
+ * @param record 工作簿记录
+ */
+ @Override
+ public void processRecord(Record record) {
+ checkCancelled();
+ if (record instanceof BoundSheetRecord boundSheet) {
+ boundSheets.add(boundSheet);
+ } else if (record instanceof SSTRecord sstRecord) {
+ sharedStrings = sstRecord;
+ } else if (record instanceof BOFRecord bofRecord
+ && bofRecord.getType() == BOFRecord.TYPE_WORKSHEET) {
+ startSheet();
+ } else if (record instanceof LabelSSTRecord label) {
+ String value = sharedStrings == null ? "" : sharedStrings.getString(label.getSSTIndex()).toString();
+ putCell(label.getRow(), label.getColumn(), value);
+ } else if (record instanceof LabelRecord label) {
+ putCell(label.getRow(), label.getColumn(), label.getValue());
+ } else if (record instanceof NumberRecord number) {
+ putCell(number.getRow(), number.getColumn(), formatter.formatNumberDateCell(number));
+ } else if (record instanceof FormulaRecord formula) {
+ if (formula.hasCachedResultString()) {
+ pendingFormulaRow = formula.getRow();
+ pendingFormulaColumn = formula.getColumn();
+ } else {
+ putCell(formula.getRow(), formula.getColumn(), formatter.formatNumberDateCell(formula));
+ }
+ } else if (record instanceof StringRecord string && pendingFormulaRow >= 0) {
+ putCell(pendingFormulaRow, pendingFormulaColumn, string.getString());
+ pendingFormulaRow = -1;
+ pendingFormulaColumn = -1;
+ } else if (record instanceof BoolErrRecord boolError && boolError.isBoolean()) {
+ putCell(boolError.getRow(), boolError.getColumn(),
+ Boolean.toString(boolError.getBooleanValue()));
+ }
+ }
+
+ private void startSheet() {
+ flushSheet();
+ sheetIndex++;
+ if (sheetIndex > request.getMaxSheets()) {
+ throw new StructureLimitRuntimeException(
+ "Sheet count exceeds " + request.getMaxSheets());
+ }
+ }
+
+ private void putCell(int row, int column, String value) {
+ if (value == null || value.isBlank()) {
+ return;
+ }
+ if (currentRow >= 0 && row != currentRow) {
+ flushRow();
+ }
+ currentRow = row;
+ nonEmptyCells++;
+ if (nonEmptyCells > request.getMaxNonEmptyCells()) {
+ throw new StructureLimitRuntimeException(
+ "Non-empty cell count exceeds " + request.getMaxNonEmptyCells());
+ }
+ currentCells.put(column,
+ CellReference.convertNumToColString(column) + (row + 1) + "=" + value.trim());
+ }
+
+ private void flushRow() {
+ if (!currentCells.isEmpty() && currentRow >= 0) {
+ String text = String.join(" | ", currentCells.values());
+ currentRows.add(new SpreadsheetReadSupport.RowText(currentRow + 1, text));
+ }
+ currentCells.clear();
+ currentRow = -1;
+ }
+
+ private void flushSheet() {
+ flushRow();
+ if (sheetIndex <= 0) {
+ return;
+ }
+ String name = sheetIndex <= boundSheets.size()
+ ? boundSheets.get(sheetIndex - 1).getSheetname()
+ : "Sheet " + sheetIndex;
+ sheets.add(new SpreadsheetReadSupport.SheetRows(
+ sheetIndex, name, new ArrayList<>(currentRows)));
+ currentRows.clear();
+ }
+
+ private void finish() {
+ flushSheet();
+ }
+
+ private void checkCancelled() {
+ request.checkCancelled();
+ }
+
+ private List sheets() {
+ return sheets;
+ }
+ }
+
+ /**
+ * HSSF 回调中传递读取中止或结构上限异常。
+ */
+ private static final class StructureLimitRuntimeException extends RuntimeException {
+
+ private StructureLimitRuntimeException(String message) {
+ super(message);
+ }
+ }
+}
diff --git a/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsxExtractor.java b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsxExtractor.java
new file mode 100644
index 0000000..65e727d
--- /dev/null
+++ b/easy-agents-core/src/main/java/com/easyagents/core/file2text/extractor/impl/XlsxExtractor.java
@@ -0,0 +1,228 @@
+/*
+ * Copyright (c) 2023-2026, Easy-Agents (fuhai999@gmail.com).
+ *
+ * Licensed under the Apache License, Version 2.0.
+ */
+package com.easyagents.core.file2text.extractor.impl;
+
+import com.easyagents.core.file2text.DocumentReadErrorCode;
+import com.easyagents.core.file2text.DocumentReadException;
+import com.easyagents.core.file2text.DocumentReadSupport;
+import com.easyagents.core.file2text.LightweightDocumentReadRequest;
+import com.easyagents.core.file2text.LightweightDocumentReadResult;
+import com.easyagents.core.file2text.extractor.FileExtractor;
+import com.easyagents.core.file2text.source.DocumentSource;
+import org.apache.poi.openxml4j.opc.OPCPackage;
+import org.apache.poi.ss.usermodel.DataFormatter;
+import org.apache.poi.ss.util.CellReference;
+import org.apache.poi.util.XMLHelper;
+import org.apache.poi.xssf.eventusermodel.ReadOnlySharedStringsTable;
+import org.apache.poi.xssf.eventusermodel.XSSFReader;
+import org.apache.poi.xssf.eventusermodel.XSSFSheetXMLHandler;
+import org.apache.poi.xssf.model.SharedStrings;
+import org.apache.poi.xssf.model.Styles;
+import org.apache.poi.xssf.usermodel.XSSFComment;
+import org.xml.sax.InputSource;
+import org.xml.sax.XMLReader;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.ArrayList;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Locale;
+import java.util.Map;
+import java.util.Set;
+
+/**
+ * 基于 XSSF SAX 的 XLSX 文档提取器。
+ */
+public class XlsxExtractor implements FileExtractor {
+
+ private static final Set MIME_TYPES = Set.of(
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.template");
+
+ /**
+ * 判断是否支持 XLSX。
+ *
+ * @param source 文档来源
+ * @return 是否支持
+ */
+ @Override
+ public boolean supports(DocumentSource source) {
+ if (source.getMimeType() != null && MIME_TYPES.contains(source.getMimeType().toLowerCase(Locale.ROOT))) {
+ return true;
+ }
+ String name = source.getFileName();
+ return name != null && (name.toLowerCase(Locale.ROOT).endsWith(".xlsx")
+ || name.toLowerCase(Locale.ROOT).endsWith(".xltx"));
+ }
+
+ /**
+ * 提取兼容纯文本。
+ *
+ * @param source 文档来源
+ * @return 文本
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public String extractText(DocumentSource source) throws IOException {
+ return read(new LightweightDocumentReadRequest(source)).getText();
+ }
+
+ /**
+ * 使用 SAX 按工作表和行读取 XLSX。
+ *
+ * @param request 读取请求
+ * @return 表格结构化结果
+ * @throws IOException 文档 I/O 失败
+ */
+ @Override
+ public LightweightDocumentReadResult read(LightweightDocumentReadRequest request) throws IOException {
+ DocumentSource source = request.getSource();
+ try (InputStream input = source.openStream();
+ OPCPackage opcPackage = OPCPackage.open(input)) {
+ XSSFReader reader = new XSSFReader(opcPackage);
+ Styles styles = reader.getStylesTable();
+ SharedStrings sharedStrings = new ReadOnlySharedStringsTable(opcPackage);
+ XSSFReader.SheetIterator sheets = (XSSFReader.SheetIterator) reader.getSheetsData();
+ List resultSheets = new ArrayList<>();
+ int[] nonEmptyCells = {0};
+ int sheetIndex = 0;
+ while (sheets.hasNext()) {
+ request.checkCancelled();
+ sheetIndex++;
+ if (sheetIndex > request.getMaxSheets()) {
+ throw limit("Sheet count exceeds " + request.getMaxSheets());
+ }
+ try (InputStream sheetInput = sheets.next()) {
+ String sheetName = sheets.getSheetName();
+ SheetHandler handler = new SheetHandler(request, nonEmptyCells);
+ XMLReader parser = XMLHelper.newXMLReader();
+ parser.setContentHandler(new XSSFSheetXMLHandler(
+ styles, null, sharedStrings, handler, new DataFormatter(), false));
+ parser.parse(new InputSource(sheetInput));
+ resultSheets.add(new SpreadsheetReadSupport.SheetRows(
+ sheetIndex, sheetName, handler.rows()));
+ }
+ }
+ return DocumentReadSupport.result(source,
+ SpreadsheetReadSupport.toSegments(resultSheets), request);
+ } catch (StructureLimitRuntimeException error) {
+ throw limit(error.getMessage());
+ } catch (DocumentReadException error) {
+ throw error;
+ } catch (Exception error) {
+ throw new DocumentReadException(DocumentReadErrorCode.DOCUMENT_CORRUPTED,
+ "Failed to extract XLSX", error);
+ }
+ }
+
+ /**
+ * 获取读取器优先级。
+ *
+ * @return 优先级
+ */
+ @Override
+ public int getOrder() {
+ return 10;
+ }
+
+ private DocumentReadException limit(String message) {
+ return new DocumentReadException(DocumentReadErrorCode.DOCUMENT_STRUCTURE_LIMIT_EXCEEDED, message);
+ }
+
+ /**
+ * SAX 工作表内容处理器。
+ */
+ private static final class SheetHandler implements XSSFSheetXMLHandler.SheetContentsHandler {
+
+ private final LightweightDocumentReadRequest request;
+ private final int[] nonEmptyCells;
+ private final List rows = new ArrayList<>();
+ private final Map currentCells = new LinkedHashMap<>();
+ private int currentRow;
+
+ private SheetHandler(LightweightDocumentReadRequest request, int[] nonEmptyCells) {
+ this.request = request;
+ this.nonEmptyCells = nonEmptyCells;
+ }
+
+ /**
+ * 开始读取一行。
+ *
+ * @param rowNum 零基行号
+ */
+ @Override
+ public void startRow(int rowNum) {
+ currentRow = rowNum + 1;
+ currentCells.clear();
+ }
+
+ /**
+ * 完成一行并保存有效单元格。
+ *
+ * @param rowNum 零基行号
+ */
+ @Override
+ public void endRow(int rowNum) {
+ if (currentCells.isEmpty()) {
+ return;
+ }
+ String text = currentCells.entrySet().stream()
+ .map(item -> item.getKey() + "=" + item.getValue())
+ .reduce((left, right) -> left + " | " + right)
+ .orElse("");
+ rows.add(new SpreadsheetReadSupport.RowText(currentRow, text));
+ }
+
+ /**
+ * 接收一个格式化单元格值。
+ *
+ * @param cellReference 单元格引用
+ * @param formattedValue 格式化显示值
+ * @param comment 批注
+ */
+ @Override
+ public void cell(String cellReference, String formattedValue, XSSFComment comment) {
+ if (formattedValue == null || formattedValue.isBlank()) {
+ return;
+ }
+ request.checkCancelled();
+ nonEmptyCells[0]++;
+ if (nonEmptyCells[0] > request.getMaxNonEmptyCells()) {
+ throw new StructureLimitRuntimeException(
+ "Non-empty cell count exceeds " + request.getMaxNonEmptyCells());
+ }
+ String column = cellReference == null ? "?" : CellReference.convertNumToColString(
+ new CellReference(cellReference).getCol());
+ currentCells.put(column + currentRow, formattedValue.trim());
+ }
+
+ /**
+ * 接收页眉页脚;轻量读取不纳入正文。
+ *
+ * @param text 文本
+ * @param isHeader 是否页眉
+ * @param tagName 标签名
+ */
+ @Override
+ public void headerFooter(String text, boolean isHeader, String tagName) {
+ }
+
+ private List rows() {
+ return rows;
+ }
+ }
+
+ /**
+ * SAX 回调跨层传递结构上限异常。
+ */
+ private static final class StructureLimitRuntimeException extends RuntimeException {
+
+ private StructureLimitRuntimeException(String message) {
+ super(message);
+ }
+ }
+}
diff --git a/easy-agents-core/src/test/java/com/easyagents/core/file2text/File2TextServiceLightweightReadTest.java b/easy-agents-core/src/test/java/com/easyagents/core/file2text/File2TextServiceLightweightReadTest.java
new file mode 100644
index 0000000..2f7dee5
--- /dev/null
+++ b/easy-agents-core/src/test/java/com/easyagents/core/file2text/File2TextServiceLightweightReadTest.java
@@ -0,0 +1,162 @@
+package com.easyagents.core.file2text;
+
+import com.easyagents.core.file2text.extractor.impl.DocExtractor;
+import com.easyagents.core.file2text.extractor.impl.PdfTextExtractor;
+import com.easyagents.core.file2text.extractor.impl.XlsExtractor;
+import com.easyagents.core.file2text.extractor.impl.XlsxExtractor;
+import com.easyagents.core.file2text.source.ByteArrayDocumentSource;
+import org.apache.poi.hslf.usermodel.HSLFSlide;
+import org.apache.poi.hslf.usermodel.HSLFSlideShow;
+import org.apache.poi.hslf.usermodel.HSLFTextBox;
+import org.apache.poi.hssf.usermodel.HSSFWorkbook;
+import org.apache.poi.xslf.usermodel.XMLSlideShow;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.apache.poi.xssf.usermodel.XSSFWorkbook;
+import org.junit.Assert;
+import org.junit.Test;
+
+import java.io.ByteArrayInputStream;
+import java.io.ByteArrayOutputStream;
+import java.nio.charset.StandardCharsets;
+
+/**
+ * 常规文档格式轻量读取回归测试。
+ */
+public class File2TextServiceLightweightReadTest {
+
+ private final File2TextService service = new File2TextService();
+
+ /**
+ * 验证 TXT 与 Markdown 会保留行区间和标题结构。
+ */
+ @Test
+ public void shouldReadTextAndMarkdownWithStableSegments() {
+ LightweightDocumentReadResult text = read(
+ "sample.txt", "text/plain", "first line\nsecond line".getBytes(StandardCharsets.UTF_8));
+ LightweightDocumentReadResult markdown = read(
+ "sample.md", "text/markdown", "# Chapter\nbody".getBytes(StandardCharsets.UTF_8));
+
+ Assert.assertTrue(text.getText().contains("second line"));
+ Assert.assertEquals("LINE_RANGE", text.getSegments().get(0).getLocatorType());
+ Assert.assertTrue(markdown.getText().contains("Chapter"));
+ Assert.assertEquals("Chapter", markdown.getSegments().get(0).getHeadingPath().get(0));
+ }
+
+ /**
+ * 验证 DOCX、PPTX、PPT、XLSX 与 XLS 均可直接读取正文。
+ *
+ * @throws Exception 测试文档生成失败时抛出
+ */
+ @Test
+ public void shouldReadGeneratedOfficeAndPdfDocuments() throws Exception {
+ assertReadable("sample.docx",
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
+ docxBytes(), "DOCX sample");
+ assertReadable("sample.pptx",
+ "application/vnd.openxmlformats-officedocument.presentationml.presentation",
+ pptxBytes(), "PPTX sample");
+ assertReadable("sample.ppt", "application/vnd.ms-powerpoint", pptBytes(), "PPT sample");
+ assertReadable("sample.xlsx",
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
+ xlsxBytes(), "XLSX sample");
+ assertReadable("sample.xls", "application/vnd.ms-excel", xlsBytes(), "XLS sample");
+ }
+
+ /**
+ * 验证 PDF 与旧版 DOC 扩展名仍路由到专用读取器。
+ */
+ @Test
+ public void shouldRegisterPdfAndLegacyDocReaders() {
+ DocExtractor docExtractor = new DocExtractor();
+ PdfTextExtractor pdfExtractor = new PdfTextExtractor();
+
+ Assert.assertTrue(docExtractor.supports(new ByteArrayDocumentSource(
+ new byte[0], "legacy.doc", "application/msword")));
+ Assert.assertTrue(pdfExtractor.supports(new ByteArrayDocumentSource(
+ new byte[0], "sample.pdf", "application/pdf")));
+ }
+
+ /**
+ * 验证表格读取取消会保留明确错误码。
+ *
+ * @throws Exception 测试文档生成失败时抛出
+ */
+ @Test
+ public void shouldPreserveCancellationErrorForSpreadsheets() throws Exception {
+ assertCancelled(new XlsxExtractor(), "sample.xlsx",
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", xlsxBytes());
+ assertCancelled(new XlsExtractor(), "sample.xls",
+ "application/vnd.ms-excel", xlsBytes());
+ }
+
+ private void assertCancelled(com.easyagents.core.file2text.extractor.FileExtractor extractor,
+ String fileName,
+ String mimeType,
+ byte[] bytes) throws Exception {
+ LightweightDocumentReadRequest request = new LightweightDocumentReadRequest(
+ new ByteArrayDocumentSource(bytes, fileName, mimeType));
+ request.setCancelled(() -> true);
+ try {
+ extractor.read(request);
+ Assert.fail("Expected document read cancellation");
+ } catch (DocumentReadException error) {
+ Assert.assertEquals(DocumentReadErrorCode.DOCUMENT_READ_CANCELLED, error.getErrorCode());
+ }
+ }
+
+ private void assertReadable(String fileName, String mimeType, byte[] bytes, String expected) {
+ Assert.assertTrue(read(fileName, mimeType, bytes).getText().contains(expected));
+ }
+
+ private LightweightDocumentReadResult read(String fileName, String mimeType, byte[] bytes) {
+ return service.readFromStream(new ByteArrayInputStream(bytes), fileName, mimeType);
+ }
+
+ private byte[] docxBytes() throws Exception {
+ try (XWPFDocument document = new XWPFDocument();
+ ByteArrayOutputStream output = new ByteArrayOutputStream()) {
+ document.createParagraph().createRun().setText("DOCX sample");
+ document.write(output);
+ return output.toByteArray();
+ }
+ }
+
+ private byte[] pptxBytes() throws Exception {
+ try (XMLSlideShow presentation = new XMLSlideShow();
+ ByteArrayOutputStream output = new ByteArrayOutputStream()) {
+ presentation.createSlide().createTextBox().setText("PPTX sample");
+ presentation.write(output);
+ return output.toByteArray();
+ }
+ }
+
+ private byte[] pptBytes() throws Exception {
+ try (HSLFSlideShow presentation = new HSLFSlideShow();
+ ByteArrayOutputStream output = new ByteArrayOutputStream()) {
+ HSLFSlide slide = presentation.createSlide();
+ HSLFTextBox textBox = new HSLFTextBox();
+ textBox.setText("PPT sample");
+ slide.addShape(textBox);
+ presentation.write(output);
+ return output.toByteArray();
+ }
+ }
+
+ private byte[] xlsxBytes() throws Exception {
+ try (XSSFWorkbook workbook = new XSSFWorkbook();
+ ByteArrayOutputStream output = new ByteArrayOutputStream()) {
+ workbook.createSheet("Sheet1").createRow(0).createCell(0).setCellValue("XLSX sample");
+ workbook.write(output);
+ return output.toByteArray();
+ }
+ }
+
+ private byte[] xlsBytes() throws Exception {
+ try (HSSFWorkbook workbook = new HSSFWorkbook();
+ ByteArrayOutputStream output = new ByteArrayOutputStream()) {
+ workbook.createSheet("Sheet1").createRow(0).createCell(0).setCellValue("XLS sample");
+ workbook.write(output);
+ return output.toByteArray();
+ }
+ }
+}