package cn.iocoder.yudao.module.qcreport.service.aiimport.document; import org.apache.pdfbox.pdmodel.PDDocument; import org.apache.pdfbox.pdmodel.PDPage; import org.apache.pdfbox.pdmodel.PDPageContentStream; import org.apache.pdfbox.pdmodel.common.PDRectangle; import org.apache.pdfbox.pdmodel.font.PDType1Font; import org.apache.pdfbox.pdmodel.font.Standard14Fonts; import org.apache.poi.ss.usermodel.Row; import org.apache.poi.ss.usermodel.Sheet; import org.apache.poi.xssf.usermodel.XSSFWorkbook; import org.apache.poi.xwpf.model.XWPFHeaderFooterPolicy; import org.apache.poi.xwpf.usermodel.XWPFDocument; import org.apache.poi.xwpf.usermodel.XWPFTable; import org.apache.poi.xwpf.usermodel.XWPFTableCell; import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTTcPr; import org.openxmlformats.schemas.wordprocessingml.x2006.main.STMerge; import java.io.ByteArrayOutputStream; import java.io.IOException; import java.io.UncheckedIOException; import java.math.BigInteger; /** * 测试用的二进制样本,在内存里现场生成,不落地成资源文件。 *
* 刻意不往 {@code src/test/resources} 放 .pdf / .docx 二进制:那种文件在仓库里既 review 不了也改不动, * 想知道它里面写了什么还得专门打开看。现场生成则让「样本里有什么内容」直接写在测试代码里, * 一眼就能看出断言为什么成立。 *
* 代价是样本用了与生产同源的库(PDFBox / POI)来写——这确实是自证。 * 但这里要验证的本来就不是「PDFBox 能不能解析 PDF」,而是适配器拿到一份这样的文件会走哪条通道、 * 抽出什么结构,那正是这些样本能回答的问题。 */ final class AiImportFixtures { private AiImportFixtures() { } /** * 有文本层的电子版 PDF,每页写一段足够长的 ASCII 文本(超过默认阈值 30 字符)。 *
* 用 ASCII 而不是中文:PDF 的 14 号标准字体不含中文字形,写中文要么报错要么变成一串问号,
* 而本用例只关心「有没有文本层」,不关心中文渲染。
*/
static byte[] textPdf(int pageCount) {
try (PDDocument document = new PDDocument()) {
for (int i = 1; i <= pageCount; i++) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
try (PDPageContentStream stream = new PDPageContentStream(document, page)) {
stream.beginText();
stream.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
stream.newLineAtOffset(50, 700);
stream.showText("Incoming Inspection Report page " + i);
stream.newLineAtOffset(0, -20);
stream.showText("Item Size Tolerance Result Qty Remark Inspector");
stream.endText();
}
}
return save(document);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
/**
* 只有页面、没有文字的 PDF——等价于一份扫描件:抽不出文本层,只能渲染成图。
*/
static byte[] imageOnlyPdf(int pageCount) {
try (PDDocument document = new PDDocument()) {
for (int i = 0; i < pageCount; i++) {
document.addPage(new PDPage(PDRectangle.A4));
}
return save(document);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
/**
* 段落与表格交替排列的 .docx,用于验证抽取保留版面顺序而不是「所有段落 → 所有表格」。
*/
static byte[] docx(String beforeTable, String[][] table, String afterTable) {
try (XWPFDocument document = new XWPFDocument()) {
document.createParagraph().createRun().setText(beforeTable);
XWPFTable xwpfTable = document.createTable(table.length, table[0].length);
for (int r = 0; r < table.length; r++) {
for (int c = 0; c < table[r].length; c++) {
xwpfTable.getRow(r).getCell(c).setText(table[r][c]);
}
}
document.createParagraph().createRun().setText(afterTable);
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.write(out);
return out.toByteArray();
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
/**
* 带纵向合并的 .docx:同一检验项目下的子项逐行展开,项目名只在第一行出现,
* 下面几行是 {@code
* 这正是「一个检验项目下挂若干子项」的真实报告形态:合并信息不还原的话, * 抽出来的是一堆看不出父子关系的平铺行。 */ static byte[] docxWithVerticalMerge() { String[][] cells = { {"检验项目", "子项", "检测方法", "标准值", "实测值", "判定"}, {"粒度-筛上物比例(%)", "20目上", "ANA.MTH-000034", "0-5", "", "合格"}, {"", "40目上", "ANA.MTH-000034", "20-40", "", "合格"}, {"水分", "常压干燥法", "ANA.MTH-000035", "≤14.0", "", "合格"}, }; try (XWPFDocument document = new XWPFDocument()) { XWPFTable table = createFilledTable(document, cells); mergeDown(table, 1, 2, 0); return write(document); } catch (IOException e) { throw new UncheckedIOException(e); } } /** * 带横向合并的 .docx:表头第一格跨两列({@code w:gridSpan="2"})。 *
* 只关心列位对齐这个结构问题,列名不追求与真实报告一一对应。表头行因此比数据行多占一格, * 数据行少的那格由适配器补齐——不补的话表头与数据行的列从此错开,模型会张冠李戴。 */ static byte[] docxWithGridSpanHeader() { String[][] cells = { {"检验项目", "标准值", "实测值"}, {"外观", "无划痕", "合格"}, }; try (XWPFDocument document = new XWPFDocument()) { XWPFTable table = createFilledTable(document, cells); cellPropertiesOf(table.getRow(0).getCell(0)).addNewGridSpan().setVal(BigInteger.valueOf(2)); return write(document); } catch (IOException e) { throw new UncheckedIOException(e); } } /** * 带页眉页脚的 .docx:抬头与落款住在正文之外,{@code getBodyElements()} 看不到它们。 */ static byte[] docxWithHeaderFooter(String bodyText, String headerText, String footerText) { try (XWPFDocument document = new XWPFDocument()) { document.createParagraph().createRun().setText(bodyText); XWPFHeaderFooterPolicy policy = document.createHeaderFooterPolicy(); policy.createHeader(XWPFHeaderFooterPolicy.DEFAULT).createParagraph().createRun().setText(headerText); policy.createFooter(XWPFHeaderFooterPolicy.DEFAULT).createParagraph().createRun().setText(footerText); ByteArrayOutputStream out = new ByteArrayOutputStream(); document.write(out); return out.toByteArray(); } catch (IOException e) { throw new UncheckedIOException(e); } } /** * 含列头的 .xlsx。 */ static byte[] xlsx(String sheetName, String[][] rows) { try (XSSFWorkbook workbook = new XSSFWorkbook()) { Sheet sheet = workbook.createSheet(sheetName); for (int r = 0; r < rows.length; r++) { Row row = sheet.createRow(r); for (int c = 0; c < rows[r].length; c++) { row.createCell(c).setCellValue(rows[r][c]); } } ByteArrayOutputStream out = new ByteArrayOutputStream(); workbook.write(out); return out.toByteArray(); } catch (IOException e) { throw new UncheckedIOException(e); } } /** * 带 UTF-8 BOM 的文本,用于验证 BOM 被剥掉(否则第一个单元格会带一个看不见的字符)。 */ static byte[] textWithBom(String text) { byte[] body = text.getBytes(java.nio.charset.StandardCharsets.UTF_8); byte[] result = new byte[body.length + 3]; result[0] = (byte) 0xEF; result[1] = (byte) 0xBB; result[2] = (byte) 0xBF; System.arraycopy(body, 0, result, 3, body.length); return result; } private static byte[] save(PDDocument document) throws IOException { ByteArrayOutputStream out = new ByteArrayOutputStream(); document.save(out); return out.toByteArray(); } /** 建一张每格都填好文字的规则表格,后续再按需打上合并标记 */ private static XWPFTable createFilledTable(XWPFDocument document, String[][] cells) { XWPFTable table = document.createTable(cells.length, cells[0].length); for (int r = 0; r < cells.length; r++) { for (int c = 0; c < cells[r].length; c++) { table.getRow(r).getCell(c).setText(cells[r][c]); } } return table; } /** * 把第 {@code fromRow} 行到第 {@code toRow} 行的第 {@code column} 列纵向合并成一个单元格。 *
* 起始格打 {@code w:val="restart"}、后续格打不带 val 的 {@code