diff --git a/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparer.java b/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparer.java index f062dea..d408c4c 100644 --- a/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparer.java +++ b/src/main/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparer.java @@ -167,28 +167,35 @@ public class LLMAttachmentPreparer { return null; } + /** + * The document's text, said to be exactly that. Headed as an attachment, a model that takes + * only text answers that it cannot open files - it reads the heading as a file it was not + * given, and never gets to the text right under it. + */ private String documentBlock(String name, ReadPdf read) { + String nl = System.lineSeparator(); StringBuilder block = new StringBuilder(); - block.append(System.lineSeparator()).append(System.lineSeparator()) - .append("--- Attached document: ").append(name).append(" (").append(read.pageCount()) - .append(read.pageCount() == 1 ? " page" : " pages").append(") ---").append(System.lineSeparator()); + block.append(nl).append(nl) + .append("You were given the document ").append(name).append(" (").append(read.pageCount()) + .append(read.pageCount() == 1 ? " page" : " pages").append("). "); if (read.hasText()) { - block.append(read.text()).append(System.lineSeparator()); + block.append("Its full text is included here, between the two markers, so you can read it directly:") + .append(nl).append("<<>>").append(nl) + .append(read.text()).append(nl); if (read.textTruncated()) { - block.append("[The text is cut here: the document is longer than can be sent.]") - .append(System.lineSeparator()); + block.append("[The text is cut here: the document is longer than can be sent.]").append(nl); } + block.append("<<>>"); } else { - block.append("[This document has no text layer; read it from its page images.]").append(System.lineSeparator()); + block.append("It has no text layer, so read it from its page images."); } if (!read.pageImages().isEmpty()) { - block.append("[Its first ").append(read.pageImages().size()) + block.append(nl).append("Its first ").append(read.pageImages().size()) .append(read.pageImages().size() == 1 ? " page is" : " pages are") .append(" also attached as images") .append(read.pageImages().size() < read.pageCount() ? ", of " + read.pageCount() : "") - .append(".]").append(System.lineSeparator()); + .append("."); } - block.append("--- End of ").append(name).append(" ---"); return block.toString(); } diff --git a/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparerTest.java b/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparerTest.java index 479a5c6..1babff4 100644 --- a/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparerTest.java +++ b/src/test/java/it/cnr/isti/workflow/manager/executions/executors/blocks/LLMAttachmentPreparerTest.java @@ -120,7 +120,8 @@ class LLMAttachmentPreparerTest { LLMAttachmentPreparer.Prepared prepared = preparer.prepare(new FileAttachments(List.of(), List.of(report)), "summarise", provider(Set.of(ChatAttachment.Kind.IMAGE), true), "m", null); - assertTrue(prepared.documentText().contains("--- Attached document: report.pdf (3 pages) ---"), prepared.documentText()); + assertTrue(prepared.documentText().contains("You were given the document report.pdf (3 pages)."), prepared.documentText()); + assertTrue(prepared.documentText().contains("<<>>"), prepared.documentText()); assertTrue(prepared.documentText().contains("Quarterly revenue grew"), prepared.documentText()); assertTrue(prepared.documentText().contains("first 2 pages are also attached as images, of 3"), prepared.documentText()); assertEquals(List.of("report.pdf-page-1.jpg", "report.pdf-page-2.jpg"),