Merge pull request #2771 from arc53/fix/durable-non-text-uploads

fix: handle non text uploads more carefully
This commit is contained in:
Alex authored and GitHub committed 2026-09-14 21:23:22 +01:00
commit d121eec2f5
21 files changed
+779 -40

No files matched your search

+9
View File
@@ -150,6 +150,15 @@ takes over as the fallback parser. If that fallback also extracts almost
nothing — OCR off, or no engine available — the upload fails with a clear
message instead of silently indexing an empty document.
Chat attachments are the exception: a scanned PDF attached from the message
box is kept with no text (`extraction.status: no_text`), because the file
itself is what a model reads. Models that take PDFs get the document; models
that take only images get its pages as images; a text-only model is told it
cannot read the file, and the message box warns before sending. Images work
the same way with OCR off, and TIFF and BMP attachments are stored as PNG,
since model providers do not accept those formats. A TIFF or BMP larger than
40 million pixels is refused instead of converted.
Mixed documents — text pages with scanned pages among them — convert
through anydoc, which reads the text pages and skips the scanned ones. With
OCR on, DocsGPT probes every page's text layer, OCRs the pages that have
+31
View File
@@ -324,6 +324,25 @@ class LLMHandler(ABC):
return images_data
@staticmethod
def _is_unreadable_visual(attachment: Dict) -> bool:
"""Whether an unsupported attachment is an image or PDF with no text to use.
Images and scanned PDFs (``extraction.status == "no_text"``) reach this
path only on a model that takes neither format natively. They carry no
text, so dropping them would let the model answer as if nothing had
been attached.
"""
mime_type = attachment.get("mime_type") or ""
if not (mime_type.startswith("image/") or mime_type == "application/pdf"):
return False
extraction = (attachment.get("metadata") or {}).get("extraction") or {}
status = extraction.get("status")
if status == "no_text":
return True
content = attachment.get("content")
return status in (None, "ok") and content is not None and not str(content).strip()
def _append_unsupported_attachments(
self, messages: List[Dict], attachments: List[Dict]
) -> List[Dict]:
@@ -339,8 +358,12 @@ class LLMHandler(ABC):
"""
prepared_messages = messages.copy()
attachment_texts = []
unreadable_names = []
for attachment in attachments:
if self._is_unreadable_visual(attachment):
unreadable_names.append(attachment.get("filename") or "an attached file")
continue
# ``metadata.extraction`` records what parsing actually did.
# Rows predating it have no ``extraction`` key and pass; rows
# whose extraction failed must not reach the prompt (a PG row
@@ -374,6 +397,14 @@ class LLMHandler(ABC):
"Scope any whole-document claims to this portion.]\n\n"
)
attachment_texts.append(f"Attached file content:\n\n{note}{content}")
if unreadable_names:
listed = ", ".join(f'"{name}"' for name in unreadable_names)
attachment_texts.append(
f"[NOTE: The user attached {listed}, but the current model cannot read "
"images or scanned PDFs, so the file contents are not available to you. "
"Tell the user you cannot see the file and suggest switching to a model "
"that supports images or PDFs. Do not guess what it contains.]"
)
if attachment_texts:
combined_text = "\n\n".join(attachment_texts)
+2 -1
View File
@@ -23,6 +23,7 @@ from docsgpt.core.settings import settings
from docsgpt.parser.file.base_parser import (
BaseParser,
DocumentParseError,
NoTextLayerError,
delegate_parse,
module_available,
)
@@ -424,7 +425,7 @@ class AnydocParser(BaseParser):
"tesseract binary or a DeepSeek-OCR endpoint (OCR_ENGINE), or "
"through the optional docling extra when installed."
)
raise DocumentParseError(
raise NoTextLayerError(
f"{path.name} appears to be a scanned PDF (no text layer), and "
f"{type(fallback).__name__} extracted almost nothing{hint}"
)
+11
View File
@@ -26,6 +26,17 @@ class DocumentParseError(Exception):
"""
class NoTextLayerError(DocumentParseError):
"""A readable document with no text to extract, such as a scanned PDF.
Still a failed parse wherever text is the point (source ingestion needs
something to embed), so it subclasses ``DocumentParseError`` and callers
that do not know about it keep failing loudly. The file itself is intact,
though: a chat attachment can still go to a model that reads the format
natively, which is why the attachment worker catches this type.
"""
class BaseParser:
"""Base class for all parsers."""
+10 -8
View File
@@ -105,9 +105,6 @@ def _image_entries(factory: Callable[[], BaseParser], suffixes) -> Dict[str, Bas
return {suffix: factory() for suffix in suffixes}
_LEGACY_IMAGE_SUFFIXES = (".png", ".jpg", ".jpeg")
def _legacy_file_extractor(pdf_text_fast_path: bool = False, ocr_enabled: bool = False) -> Dict[str, BaseParser]:
"""Parser map that needs neither docling nor anydoc.
@@ -124,7 +121,10 @@ def _legacy_file_extractor(pdf_text_fast_path: bool = False, ocr_enabled: bool =
from docsgpt.parser.file.ocr_parser import IMAGE_SUFFIXES
pdf_parser: BaseParser = PDFParser()
images: Dict[str, BaseParser] = _image_entries(ImageParser, _LEGACY_IMAGE_SUFFIXES)
# Every image suffix gets an entry even without OCR: the image itself is
# what a vision model is sent, so a missing entry only means the upload is
# refused before any model sees it (.webp was, in production).
images: Dict[str, BaseParser] = _image_entries(ImageParser, sorted(IMAGE_SUFFIXES))
if ocr_enabled:
native = _native_ocr_parsers()
if native is not None:
@@ -490,6 +490,7 @@ class SimpleDirectoryReader(BaseReader):
data: Union[str, List[str]] = ""
data_list: List[str] = []
metadata_list = []
first_error: Optional[DocumentParseError] = None
self.file_token_counts = {}
self.failed_files = []
@@ -529,6 +530,7 @@ class SimpleDirectoryReader(BaseReader):
except DocumentParseError as e:
logging.warning(f"Skipping unreadable file {input_file.name}: {e}")
self.failed_files.append((input_file, str(e)))
first_error = first_error or e
report_progress(file_index + 1)
continue
@@ -579,10 +581,10 @@ class SimpleDirectoryReader(BaseReader):
# ingest tasks) treat it as terminal and tell the user.
if self.failed_files and not data_list:
if len(self.failed_files) == 1:
# Single-file read (the attachment path): the parser's own
# message reaches the user verbatim, so don't wrap it in
# "None of the 1 file(s)…".
raise DocumentParseError(self.failed_files[0][1])
# Single-file read (the attachment path): re-raise the parser's
# own exception, so its message reaches the user verbatim and
# its type (``NoTextLayerError`` for a scan) reaches the caller.
raise first_error
names = ", ".join(p.name for p, _ in self.failed_files[:5])
raise DocumentParseError(
f"None of the {len(self.failed_files)} files could be parsed: "
+56 -5
View File
@@ -1,14 +1,65 @@
"""Image parser.
Contains parser for .png, .jpg, .jpeg files.
Contains the parser for image files (.png, .jpg, .jpeg, .tiff, .tif, .bmp,
.webp) and the PNG re-encoding for the formats model providers reject.
"""
import io
from pathlib import Path
from typing import BinaryIO, Dict, Tuple, Union
import requests
from typing import Dict, Union
from docsgpt.parser.file.base_parser import BaseParser
from docsgpt.core.settings import settings
from docsgpt.parser.file.base_parser import BaseParser, DocumentParseError
# Image types the vision APIs refuse: OpenAI and Anthropic take png, jpeg,
# webp and gif only. A chat attachment in one of these formats is re-encoded
# to PNG before it is stored for the model.
VISION_CONVERTIBLE_MIME_TYPES = frozenset({"image/tiff", "image/bmp", "image/x-ms-bmp"})
# Largest image re-encoded to PNG. A deflate TIFF under 1 MB can declare
# 144 million pixels and take 1.2 GB to convert, while Pillow only warns below
# 179 million. Vision models downscale far below this cap (Anthropic refuses
# more than 8000 px per side), so a larger image gains nothing.
MAX_CONVERTIBLE_PIXELS = 40_000_000
def convert_image_to_png(file_obj: BinaryIO) -> Tuple[bytes, int]:
"""Re-encode the first frame of an image as PNG.
Args:
file_obj: Readable binary stream of the source image.
Returns:
Tuple[bytes, int]: The PNG bytes, and how many frames (pages) the
source had. Only the first is kept.
Raises:
DocumentParseError: If the bytes are not an image Pillow can decode,
or the image is larger than ``MAX_CONVERTIBLE_PIXELS``.
"""
from PIL import Image, UnidentifiedImageError
try:
with Image.open(file_obj) as image:
# Opening reads only the header; refuse before any pixels decode.
width, height = image.size
if width * height > MAX_CONVERTIBLE_PIXELS:
raise DocumentParseError(
f"The image is too large to convert ({width}×{height} pixels; the limit is "
f"{MAX_CONVERTIBLE_PIXELS // 1_000_000} million pixels). Resize it and upload it again."
)
frames = getattr(image, "n_frames", 1)
image.seek(0)
frame = image
if frame.mode not in ("RGB", "RGBA", "L", "LA"):
frame = frame.convert("RGBA" if "A" in frame.getbands() else "RGB")
out = io.BytesIO()
frame.save(out, format="PNG")
except (UnidentifiedImageError, OSError, Image.DecompressionBombError) as exc:
raise DocumentParseError(f"Could not read the image: {exc}") from exc
return out.getvalue(), frames
class ImageParser(BaseParser):
@@ -24,8 +75,8 @@ class ImageParser(BaseParser):
# alternatively you can use local vision capable LLM
with open(file, "rb") as file_loaded:
files = {'file': file_loaded}
response = requests.post(doc2md_service, files=files, timeout=100)
data = response.json()["markdown"]
response = requests.post(doc2md_service, files=files, timeout=100)
data = response.json()["markdown"]
else:
data = ""
return data
+69 -2
View File
@@ -1,4 +1,5 @@
import datetime
import io
import json
import logging
import mimetypes
@@ -23,8 +24,13 @@ from docsgpt.parser.embedding_pipeline import (
assert_index_complete,
embed_and_store_documents,
)
from docsgpt.parser.file.base_parser import NoTextLayerError
from docsgpt.parser.file.bulk import SimpleDirectoryReader, get_default_file_extractor
from docsgpt.parser.file.constants import SUPPORTED_SOURCE_EXTENSIONS
from docsgpt.parser.file.image_parser import (
VISION_CONVERTIBLE_MIME_TYPES,
convert_image_to_png,
)
from docsgpt.parser.remote.remote_creator import (
RemoteCreator,
normalize_remote_data,
@@ -1598,6 +1604,44 @@ def _reject_attachment_zip_bomb(local_path: str) -> None:
raise AttachmentRejectedError(reason)
def _readable_without_text(filename: str) -> bool:
"""Whether a model can read an attachment from its original file alone.
PDFs and images are sent to models as the file itself (a PDF as page
images on image-only models); every other format needs extracted text.
Args:
filename: The upload's original filename.
Returns:
bool: True for PDFs and images.
"""
mime_type = mimetypes.guess_type(filename)[0] or ""
return mime_type == "application/pdf" or mime_type.startswith("image/")
def _store_png_copy(storage, relative_path: str, mime_type: str) -> tuple[str, dict]:
"""Store a PNG copy, beside the original, of an image the providers reject.
Args:
storage: The storage backend holding the upload.
relative_path: Storage path of the original image.
mime_type: The original's mime type, e.g. ``image/tiff``.
Returns:
tuple[str, dict]: The PNG's storage path, and the conversion record
kept in the attachment's metadata.
Raises:
DocumentParseError: If the image cannot be decoded.
"""
with storage.get_file(relative_path) as source:
png_bytes, frames = convert_image_to_png(source)
png_path = f"{os.path.splitext(relative_path)[0]}.png"
storage.save_file(io.BytesIO(png_bytes), png_path)
return png_path, {"from": mime_type, "to": "image/png", "frames": frames}
def _bounded_attachment_copy(local_path: str) -> tuple[str, bool]:
"""Bound how much of a text attachment reaches the parser.
@@ -1794,7 +1838,24 @@ def attachment_worker(self, file_info, user):
except OSError:
pass
attachment_document = storage.process_file(relative_path, _parse_local_file)
extraction_status = "ok"
no_text_reason = None
try:
attachment_document = storage.process_file(relative_path, _parse_local_file)
except NoTextLayerError as exc:
# A PDF with no text layer is still a usable attachment: models
# that read PDFs natively, or as page images, are sent the stored
# file rather than its text. Keep it with nothing to inline; the
# LLM handler names it to any model that cannot read it.
if not _readable_without_text(filename):
raise
logging.info(
f"Attachment {filename} has no text layer; keeping it for native reading",
extra={"user": user},
)
attachment_document = Document(text="", extra_info={})
extraction_status = "no_text"
no_text_reason = str(exc)[:1024]
content = attachment_document.text
# A fast-path parser may have delegated to its fallback for this file,
# so record the engine that actually ran rather than the one selected.
@@ -1823,11 +1884,12 @@ def attachment_worker(self, file_info, user):
metadata = {
**metadata,
"extraction": {
"status": "ok",
"status": extraction_status,
"parser": parser_name,
"truncated": truncated,
"original_tokens": original_tokens,
"stored_tokens": token_count,
**({"reason": no_text_reason} if no_text_reason else {}),
},
}
@@ -1847,6 +1909,10 @@ def attachment_worker(self, file_info, user):
)
mime_type = mimetypes.guess_type(filename)[0] or "application/octet-stream"
if mime_type in VISION_CONVERTIBLE_MIME_TYPES:
relative_path, conversion = _store_png_copy(storage, relative_path, mime_type)
mime_type = conversion["to"]
metadata = {**metadata, "image_conversion": conversion}
_upsert_attachment_row(
user,
@@ -1873,6 +1939,7 @@ def attachment_worker(self, file_info, user):
"filename": filename,
"token_count": token_count,
"mime_type": mime_type,
"extraction_status": extraction_status,
},
scope={"kind": "attachment", "id": str(attachment_id)},
)
+25
View File
@@ -28,6 +28,7 @@ import {
import { ActiveState, Doc } from '../models/misc';
import {
selectSelectedDocs,
selectSelectedModel,
selectSourceDocs,
selectToken,
setSelectedDocs,
@@ -47,6 +48,7 @@ import {
ToolsTrigger,
} from './message-input';
import { useArmedSend } from './message-input/armedSend';
import { cannotReadAttachment } from './message-input/attachmentReadability';
import { handleAbort } from '../conversation/conversationSlice';
import {
AUDIO_FILE_ACCEPT_ATTR,
@@ -345,6 +347,21 @@ export default function MessageInput({
const sourceDocs = useSelector(selectSourceDocs);
const token = useSelector(selectToken);
const attachments = useSelector(selectAttachments);
const selectedModel = useSelector(selectSelectedModel);
const unreadableAttachmentIds = useMemo(
() =>
new Set(
attachments
.filter((attachment) =>
cannotReadAttachment(
attachment,
selectedModel?.supported_attachment_types,
),
)
.map((attachment) => attachment.id),
),
[attachments, selectedModel],
);
const dispatch = useDispatch();
const store = useStore<RootState>();
@@ -453,6 +470,12 @@ export default function MessageInput({
...(Number.isFinite(tokenCount)
? { token_count: tokenCount }
: {}),
...(typeof payload.mime_type === 'string'
? { mimeType: payload.mime_type }
: {}),
...(typeof payload.extraction_status === 'string'
? { extractionStatus: payload.extraction_status }
: {}),
},
}),
);
@@ -1662,6 +1685,8 @@ export default function MessageInput({
onDragStart={handleDragStart}
onDragOver={handleDragOver}
onDropOn={handleDropOn}
unreadableIds={unreadableAttachmentIds}
modelName={selectedModel?.display_name}
/>
{sendArmed && sendReadiness.state === 'waiting' && (
@@ -13,6 +13,10 @@ type AttachmentChipListProps = {
onDragStart: (e: React.DragEvent, id: string) => void;
onDragOver: (e: React.DragEvent) => void;
onDropOn: (e: React.DragEvent, targetId: string) => void;
/** Completed attachments the selected model has no way to read. */
unreadableIds?: Set<string>;
/** Display name of the selected model, for the unreadable-file warning. */
modelName?: string;
};
export default function AttachmentChipList({
@@ -22,6 +26,8 @@ export default function AttachmentChipList({
onDragStart,
onDragOver,
onDropOn,
unreadableIds,
modelName,
}: AttachmentChipListProps) {
const { t } = useTranslation();
@@ -31,6 +37,11 @@ export default function AttachmentChipList({
const failures = attachments.filter(
(attachment) => attachment.status === 'failed' && attachment.errorMessage,
);
// Not a failure: the file is kept and still sends. It only warns that the
// model picked right now would receive nothing it can read.
const unreadable = modelName
? attachments.filter((attachment) => unreadableIds?.has(attachment.id))
: [];
return (
<>
@@ -141,6 +152,22 @@ export default function AttachmentChipList({
))}
</div>
)}
{unreadable.length > 0 && (
<div
className="flex flex-col gap-0.5 px-2 pb-1 text-xs text-[#B54708] sm:px-3 dark:text-[#FDB022]"
role="status"
>
{unreadable.map((attachment) => (
<span key={attachment.id}>
{t('conversation.attachments.unreadableByModel', {
name: attachment.fileName,
model: modelName,
})}
</span>
))}
</div>
)}
</>
);
}
@@ -0,0 +1,76 @@
import type { Attachment } from '../../upload/uploadSlice';
import { cannotReadAttachment } from './attachmentReadability';
const att = (over: Partial<Attachment> = {}): Attachment => ({
id: 'a1',
fileName: 'scan.pdf',
progress: 100,
status: 'completed',
taskId: 't1',
...over,
});
const PDF_AND_IMAGES = [
'application/pdf',
'image/png',
'image/jpeg',
'image/webp',
];
const IMAGES_ONLY = ['image/png', 'image/jpeg', 'image/webp'];
const TEXT_ONLY: string[] = [];
const scannedPdf = att({
mimeType: 'application/pdf',
extractionStatus: 'no_text',
token_count: 0,
});
const photo = att({
fileName: 'photo.webp',
mimeType: 'image/webp',
extractionStatus: 'ok',
token_count: 0,
});
describe('cannotReadAttachment', () => {
it('flags a scanned PDF on a text-only model', () => {
expect(cannotReadAttachment(scannedPdf, TEXT_ONLY)).toBe(true);
});
it('lets a model that reads PDFs natively take a scanned PDF', () => {
expect(cannotReadAttachment(scannedPdf, PDF_AND_IMAGES)).toBe(false);
});
it('lets an image-only model take a scanned PDF as page images', () => {
expect(cannotReadAttachment(scannedPdf, IMAGES_ONLY)).toBe(false);
});
it('flags an image on a text-only model', () => {
expect(cannotReadAttachment(photo, TEXT_ONLY)).toBe(true);
});
it('lets a vision model take the image', () => {
expect(cannotReadAttachment(photo, IMAGES_ONLY)).toBe(false);
});
it('does not flag a PDF whose text was extracted', () => {
const textPdf = att({
mimeType: 'application/pdf',
extractionStatus: 'ok',
token_count: 1200,
});
expect(cannotReadAttachment(textPdf, TEXT_ONLY)).toBe(false);
});
it('stays quiet when the model or the file type is unknown', () => {
expect(cannotReadAttachment(scannedPdf, undefined)).toBe(false);
expect(
cannotReadAttachment(att({ extractionStatus: 'no_text' }), TEXT_ONLY),
).toBe(false);
});
it('ignores attachments that have not finished processing', () => {
expect(
cannotReadAttachment({ ...scannedPdf, status: 'processing' }, TEXT_ONLY),
).toBe(false);
});
});
@@ -0,0 +1,33 @@
import type { Attachment } from '../../upload/uploadSlice';
/**
* Whether the selected model has no way to read a completed attachment.
*
* A file with extracted text reaches any model as text. A file without any
* (an image, or a scanned PDF stored with `extraction_status: 'no_text'`)
* only reaches a model that takes that format natively; a PDF also reaches a
* model that takes images, which is sent its pages as images. When the model
* or the file type is unknown, stay quiet rather than warn on a guess.
*/
export function cannotReadAttachment(
attachment: Attachment,
supportedTypes: string[] | undefined,
): boolean {
if (attachment.status !== 'completed') return false;
const mime = attachment.mimeType;
if (!mime || !supportedTypes) return false;
const isPdf = mime === 'application/pdf';
if (!isPdf && !mime.startsWith('image/')) return false;
const hasText =
attachment.extractionStatus !== 'no_text' &&
(attachment.token_count ?? 0) > 0;
if (hasText) return false;
if (supportedTypes.includes(mime)) return false;
if (isPdf && supportedTypes.some((type) => type.startsWith('image/'))) {
return false;
}
return true;
}
+2 -1
View File
@@ -1105,7 +1105,8 @@
"waitingToSend_other": "Will send when {{count}} files finish processing…",
"cancelQueuedSend": "Cancel",
"sendBlockedByFailed": "{{names}} couldn't be processed — remove to send",
"unsupportedType": "Not a supported file type ({{extension}})"
"unsupportedType": "Not a supported file type ({{extension}})",
"unreadableByModel": "{{model}} can't read {{name}} — switch to a model that supports images or PDFs"
},
"retry": "Retry",
"reasoning": "Reasoning",
+18
View File
@@ -401,6 +401,24 @@ describe('attachment race recovery', () => {
...overrides,
});
it('keeps the mime type and extraction status a completed attachment reports', () => {
let state = reducer(undefined, addAttachment(makeAttachment()));
state = reducer(
state,
sseEventReceived(
attEvent('attachment.completed', {
token_count: 0,
mime_type: 'application/pdf',
extraction_status: 'no_text',
}),
),
);
expect(state.attachments[0].status).toBe('completed');
expect(state.attachments[0].mimeType).toBe('application/pdf');
expect(state.attachments[0].extractionStatus).toBe('no_text');
});
it('drops attachment.completed silently when no row matches attachmentId', () => {
const state = reducer(
undefined,
+13
View File
@@ -46,6 +46,13 @@ export interface Attachment {
*/
attachmentId?: string;
token_count?: number;
/** Mime type the server stored (a converted TIFF/BMP reports ``image/png``). */
mimeType?: string;
/**
* ``metadata.extraction.status`` of a completed attachment: ``ok``, or
* ``no_text`` for a scanned PDF kept for models that read it natively.
*/
extractionStatus?: string;
/** Why a ``failed`` attachment failed, when known (server or client gate). */
errorMessage?: string;
}
@@ -299,6 +306,12 @@ export const uploadSlice = createSlice({
if (Number.isFinite(tokenCount)) {
attachment.token_count = tokenCount;
}
if (typeof payload.mime_type === 'string') {
attachment.mimeType = payload.mime_type;
}
if (typeof payload.extraction_status === 'string') {
attachment.extractionStatus = payload.extraction_status;
}
break;
}
case 'attachment.failed': {
+68
View File
@@ -2394,3 +2394,71 @@ class TestAttachmentExtractionGate:
assert "truncated" in prompt
assert "100,000" in prompt
assert "250,000" in prompt
def test_scanned_pdf_row_tells_the_model_it_cannot_read_the_file(self):
"""A no-text PDF reaches this path only on a model that reads neither
PDFs nor images. Dropping it silently lets the model answer as if
nothing were attached; naming it lets the model tell the user."""
handler = ConcreteHandler()
messages = [{"role": "system", "content": "sys"}]
attachments = [{
"id": "a1",
"filename": "bylaws.pdf",
"mime_type": "application/pdf",
"content": "",
"metadata": {"extraction": {"status": "no_text"}},
}]
prompt = handler._append_unsupported_attachments(messages, attachments)[0]["content"]
assert "bylaws.pdf" in prompt
assert "cannot read" in prompt
assert "Attached file content" not in prompt
def test_image_without_text_is_named_instead_of_an_empty_block(self):
handler = ConcreteHandler()
messages = [{"role": "system", "content": "sys"}]
attachments = [{
"id": "a1",
"filename": "photo.webp",
"mime_type": "image/webp",
"content": "",
"metadata": {"extraction": {"status": "ok"}},
}]
prompt = handler._append_unsupported_attachments(messages, attachments)[0]["content"]
assert "photo.webp" in prompt
assert "cannot read" in prompt
assert "Attached file content" not in prompt
def test_image_with_extracted_text_is_still_appended(self):
handler = ConcreteHandler()
messages = [{"role": "system", "content": "sys"}]
attachments = [{
"id": "a1",
"filename": "receipt.png",
"mime_type": "image/png",
"content": "Total due: 42 EUR",
"metadata": {"extraction": {"status": "ok"}},
}]
prompt = handler._append_unsupported_attachments(messages, attachments)[0]["content"]
assert "Total due: 42 EUR" in prompt
assert "cannot read" not in prompt
def test_failed_scan_row_is_still_skipped_silently(self):
handler = ConcreteHandler()
messages = [{"role": "system", "content": "sys"}]
attachments = [{
"id": "a1",
"filename": "broken.pdf",
"mime_type": "application/pdf",
"content": None,
"metadata": {"extraction": {"status": "failed", "error": "boom"}},
}]
result = handler._append_unsupported_attachments(messages, attachments)
assert result[0]["content"] == "sys"
+8 -2
View File
@@ -527,22 +527,28 @@ def test_tableize_never_touches_docling_reroute(monkeypatch):
def test_scanned_pdf_with_near_empty_fallback_fails_loudly_when_ocr_off(tmp_path):
"""A scan whose fallback (OCR off) extracts almost nothing must fail with
an actionable message, not be stored as an empty document."""
from docsgpt.parser.file.base_parser import NoTextLayerError
path = _scanned_pdf(tmp_path / "scan.pdf")
fallback = _RecordingFallback(result=" ")
fallback.ocr_enabled = False
parser = AnydocParser(fallback_parser=fallback)
with pytest.raises(DocumentParseError, match="OCR_ENABLED"):
# Still a DocumentParseError for source ingestion, but typed, so the
# attachment worker can keep the file for models that read it natively.
with pytest.raises(NoTextLayerError, match="OCR_ENABLED"):
parser.parse_file(path)
assert parser.last_engine is None
def test_scanned_pdf_with_near_empty_fallback_and_ocr_on_reports_it(tmp_path):
from docsgpt.parser.file.base_parser import NoTextLayerError
path = _scanned_pdf(tmp_path / "scan.pdf")
fallback = _RecordingFallback(result="x")
fallback.ocr_enabled = True
with pytest.raises(DocumentParseError, match="even with OCR enabled"):
with pytest.raises(NoTextLayerError, match="even with OCR enabled"):
AnydocParser(fallback_parser=fallback).parse_file(path)
+40
View File
@@ -222,6 +222,31 @@ class TestSimpleDirectoryReaderLoadData:
with pytest.raises(DocumentParseError, match="InvalidCxxCompiler"):
reader.load_data()
def test_single_unparseable_file_keeps_the_parsers_error_type(self, tmp_path):
"""The attachment worker tells a scan from a broken file by the error's type.
Re-wrapping the parser's exception in a plain ``DocumentParseError``
would erase ``NoTextLayerError`` and turn every scanned PDF back into
a failed upload.
"""
from docsgpt.parser.file.base_parser import NoTextLayerError
from docsgpt.parser.file.bulk import SimpleDirectoryReader
(tmp_path / "scan.pdf").write_bytes(b"%PDF-1.7")
mock_parser = MagicMock()
mock_parser.parser_config_set = True
mock_parser.parse_file.side_effect = NoTextLayerError(
"scan.pdf appears to be a scanned PDF (no text layer)"
)
reader = SimpleDirectoryReader(
input_files=[str(tmp_path / "scan.pdf")],
file_extractor={".pdf": mock_parser},
)
with pytest.raises(NoTextLayerError, match="scanned PDF"):
reader.load_data()
def test_all_files_unparseable_raises(self, tmp_path):
"""Nothing parsed is a failed ingest, not an empty success."""
from docsgpt.parser.file.base_parser import DocumentParseError
@@ -588,6 +613,21 @@ class TestParserEngineSwitch:
assert type(extractor[".xlsx"].fallback_parser).__name__ == "ExcelParser"
assert type(extractor[".png"]).__name__ == "ImageParser"
def test_anydoc_without_docling_admits_every_image_format(self, monkeypatch):
"""The production image ships neither docling nor OCR. An image is
sent to the model as an image, so .webp/.tiff/.bmp need a parser
entry or the worker refuses them before any model sees them."""
pytest.importorskip("anydoc")
import sys
from docsgpt.parser.file.bulk import get_default_file_extractor
monkeypatch.setitem(sys.modules, "docling", None)
extractor = get_default_file_extractor(engine="anydoc")
for suffix in (".webp", ".tiff", ".tif", ".bmp"):
assert type(extractor[suffix]).__name__ == "ImageParser", suffix
def test_docling_engine_keeps_docling_map(self):
pytest.importorskip("docling")
from docsgpt.parser.file.bulk import get_default_file_extractor
@@ -0,0 +1,99 @@
"""Re-encoding images the model providers reject (TIFF, BMP) as PNG.
Chat attachments reach vision models as the stored file itself. OpenAI and
Anthropic accept png/jpeg/webp/gif only, so a TIFF fax or a BMP screenshot has
to be re-encoded before it can be sent at all.
"""
import io
import pytest
from PIL import Image
from docsgpt.parser.file.base_parser import DocumentParseError
from docsgpt.parser.file.image_parser import (
VISION_CONVERTIBLE_MIME_TYPES,
convert_image_to_png,
)
def _encoded(fmt, frames=1, size=(40, 24)):
"""An in-memory image of ``frames`` pages, each a different red level."""
images = [Image.new("RGB", size, color=(i * 60, 20, 200)) for i in range(frames)]
buf = io.BytesIO()
if frames > 1:
images[0].save(buf, format=fmt, save_all=True, append_images=images[1:])
else:
images[0].save(buf, format=fmt)
buf.seek(0)
return buf
@pytest.mark.unit
class TestConvertImageToPng:
def test_tiff_becomes_png(self):
png, frames = convert_image_to_png(_encoded("TIFF"))
decoded = Image.open(io.BytesIO(png))
assert decoded.format == "PNG"
assert decoded.size == (40, 24)
assert frames == 1
def test_multi_page_tiff_keeps_the_first_page_and_reports_the_count(self):
png, frames = convert_image_to_png(_encoded("TIFF", frames=3))
decoded = Image.open(io.BytesIO(png)).convert("RGB")
assert frames == 3
assert decoded.getpixel((0, 0))[0] == 0 # page one's red level
def test_bmp_becomes_png(self):
png, frames = convert_image_to_png(_encoded("BMP"))
assert Image.open(io.BytesIO(png)).format == "PNG"
assert frames == 1
def test_palette_tiff_is_stored_as_rgb_png(self):
buf = io.BytesIO()
Image.new("RGB", (40, 24), color=(0, 204, 51)).convert("P").save(buf, format="TIFF")
buf.seek(0)
png, _ = convert_image_to_png(buf)
decoded = Image.open(io.BytesIO(png))
assert decoded.mode == "RGB"
assert decoded.getpixel((0, 0)) == (0, 204, 51)
def test_unreadable_bytes_raise_a_parse_error(self):
with pytest.raises(DocumentParseError):
convert_image_to_png(io.BytesIO(b"not an image"))
def test_image_over_the_pixel_limit_is_refused_before_decoding(self, monkeypatch):
# A deflate TIFF under 1 MB can decode to over a gigabyte, so the
# dimensions in the header are checked before any pixel data is read.
from PIL import TiffImagePlugin
from docsgpt.parser.file import image_parser
def decode(*args, **kwargs):
raise AssertionError("pixel data was decoded")
monkeypatch.setattr(image_parser, "MAX_CONVERTIBLE_PIXELS", 500)
monkeypatch.setattr(TiffImagePlugin.TiffImageFile, "load", decode)
with pytest.raises(DocumentParseError, match="40×24"):
convert_image_to_png(_encoded("TIFF"))
def test_decompression_bomb_refusal_is_a_parse_error(self, monkeypatch):
# Pillow refuses images over twice MAX_IMAGE_PIXELS with an error that
# is not an OSError; it must fail the upload like any unreadable image.
monkeypatch.setattr(Image, "MAX_IMAGE_PIXELS", 100)
with pytest.raises(DocumentParseError):
convert_image_to_png(_encoded("TIFF"))
def test_only_formats_the_providers_reject_are_converted(self):
assert {"image/tiff", "image/bmp"} <= VISION_CONVERTIBLE_MIME_TYPES
assert not (
{"image/png", "image/jpeg", "image/webp", "image/gif"}
& VISION_CONVERTIBLE_MIME_TYPES
)
+6 -3
View File
@@ -536,13 +536,16 @@ class TestNativeOcrImageParser:
@pytest.mark.unit
class TestExtractorWiring:
def test_legacy_map_without_ocr_is_unchanged(self):
def test_legacy_map_without_ocr_hands_every_image_to_image_parser(self):
"""Without OCR an image still reaches the model as an image, so every
image suffix the upload gate admits needs an entry. Only png/jpg/jpeg
had one, and a .webp was refused outright (prod 2026-09-10)."""
from docsgpt.parser.file.bulk import _legacy_file_extractor
extractor = _legacy_file_extractor()
assert type(extractor[".pdf"]).__name__ == "PDFParser"
assert type(extractor[".png"]).__name__ == "ImageParser"
assert ".tiff" not in extractor
for suffix in op.IMAGE_SUFFIXES:
assert type(extractor[suffix]).__name__ == "ImageParser", suffix
def test_legacy_map_with_ocr_uses_native_parsers(self):
from docsgpt.parser.file.bulk import _legacy_file_extractor
+9 -8
View File
@@ -145,23 +145,24 @@ def test_enforce_parseable_attachment_rejects_binary_behind_a_bom(bom, tmp_path)
def test_enforce_parseable_attachment_uses_the_extractor_it_is_given(tmp_path):
"""The worker holds the live parser table; a trimmed install must not admit on trust.
Without docling the fallback extractor has no .webp handler, so a .webp
would otherwise skip the content check and be read as plain text.
Without docling the fallback extractor has no .vtt handler, so a binary
file named .vtt would otherwise skip the content check and be read as
plain text.
"""
from docsgpt.upload_limits import (
enforce_parseable_attachment,
UnsupportedUploadTypeError,
)
path = tmp_path / "scan.webp"
path.write_bytes(b"RIFF\x00\x00\x00\x00WEBPVP8 " + bytes(range(256)))
path = tmp_path / "subs.vtt"
path.write_bytes(b"\x00\x01\x02\x03" + bytes(range(256)))
# Default list: .webp is parser-backed, admitted on its name.
enforce_parseable_attachment(path, "scan.webp")
# Default list: .vtt is parser-backed, admitted on its name.
enforce_parseable_attachment(path, "subs.vtt")
# The extractor actually loaded has no .webp parser.
# The extractor actually loaded has no .vtt parser.
with pytest.raises(UnsupportedUploadTypeError):
enforce_parseable_attachment(path, "scan.webp", {".pdf", ".docx"})
enforce_parseable_attachment(path, "subs.vtt", {".pdf", ".docx"})
@pytest.mark.parametrize(
+167 -10
View File
@@ -384,14 +384,14 @@ class TestAttachmentTypeGuard:
):
"""The guard judges against the parser table actually loaded.
Without docling the fallback extractor has no .webp parser, so a
.webp the route admitted on its name would otherwise be opened as
plain text here.
The route admits ``.vtt`` by name, but a docling-less install has no
VTT parser, so a binary file named ``.vtt`` would otherwise be opened
as plain text here.
"""
from docsgpt import worker
local_path = tmp_path / "scan.webp"
local_path.write_bytes(b"RIFF\x00\x00\x00\x00WEBPVP8 " + bytes(range(256)))
local_path = tmp_path / "subs.vtt"
local_path.write_bytes(b"\x00\x01\x02\x03" + bytes(range(256)))
events = []
fake_storage = MagicMock(name="storage")
@@ -399,7 +399,7 @@ class TestAttachmentTypeGuard:
str(local_path)
)
monkeypatch.setattr(worker.StorageCreator, "get_storage", lambda: fake_storage)
# The docling-less fallback table: images are handled, .webp is not.
# A trimmed parser table: .png is handled, .vtt is not.
monkeypatch.setattr(
worker,
"get_default_file_extractor",
@@ -412,19 +412,176 @@ class TestAttachmentTypeGuard:
)
file_info = {
"filename": "scan.webp",
"filename": "subs.vtt",
"attachment_id": "507f1f77bcf86cd799439014",
"path": "uploads/user1/attachments/scan.webp",
"path": "uploads/user1/attachments/subs.vtt",
"metadata": {"source": "chat"},
}
with pytest.raises(
worker.AttachmentRejectedError, match=r"Unsupported file type: \.webp"
worker.AttachmentRejectedError, match=r"Unsupported file type: \.vtt"
):
worker.attachment_worker(task_self, file_info, "user1")
failed = [payload for name, payload in events if name == "attachment.failed"]
assert failed and failed[0]["error"] == "Unsupported file type: .webp"
assert failed and failed[0]["error"] == "Unsupported file type: .vtt"
def test_scanned_pdf_completes_without_text_for_models_that_read_it_natively(
self, pg_conn, patch_worker_db, task_self, monkeypatch
):
"""A PDF with no text layer is still a usable attachment.
Models that read PDFs natively (or as page images) are sent the
stored file, not its extracted text, so failing the upload only kept
the file from a model that could read it (prod 2026-09-12: a paying
user's scanned club bylaws). The row is kept with ``status: no_text``
— text inlining still skips it — and the user is told it completed.
"""
from docsgpt import worker
from docsgpt.parser.file.base_parser import NoTextLayerError
published: list[tuple[str, dict]] = []
monkeypatch.setattr(
worker,
"publish_user_event",
lambda user, event, payload, **kw: published.append((event, payload)),
)
fake_storage = MagicMock(name="storage")
fake_storage.process_file.side_effect = NoTextLayerError(
"bylaws.pdf appears to be a scanned PDF (no text layer), and "
"PDFParser extracted almost nothing"
)
monkeypatch.setattr(worker.StorageCreator, "get_storage", lambda: fake_storage)
monkeypatch.setattr(
worker, "get_default_file_extractor", lambda ocr_enabled=False, pdf_text_fast_path=False: {}
)
file_info = {
"filename": "bylaws.pdf",
"attachment_id": "507f1f77bcf86cd799439021",
"path": "uploads/user1/attachments/bylaws.pdf",
"metadata": {"source": "chat"},
}
result = worker.attachment_worker(task_self, file_info, "user1")
assert result["token_count"] == 0
assert result["mime_type"] == "application/pdf"
extraction = result["metadata"]["extraction"]
assert extraction["status"] == "no_text"
assert "scanned PDF" in extraction["reason"]
row = AttachmentsRepository(pg_conn).get_by_legacy_id(
file_info["attachment_id"], "user1"
)
assert row["upload_path"] == file_info["path"]
assert row["content"] == ""
assert row["metadata"]["extraction"]["status"] == "no_text"
events = [event for event, _ in published]
assert "attachment.failed" not in events
completed = [payload for event, payload in published if event == "attachment.completed"]
assert completed[0]["extraction_status"] == "no_text"
assert completed[0]["mime_type"] == "application/pdf"
def test_no_text_on_a_type_models_cannot_read_natively_still_fails(
self, pg_conn, patch_worker_db, task_self, monkeypatch
):
"""Only PDFs and images have a native path; anything else without text stays a failure."""
from docsgpt import worker
from docsgpt.parser.file.base_parser import NoTextLayerError
published: list[tuple[str, dict]] = []
monkeypatch.setattr(
worker,
"publish_user_event",
lambda user, event, payload, **kw: published.append((event, payload)),
)
fake_storage = MagicMock(name="storage")
fake_storage.process_file.side_effect = NoTextLayerError("notes.docx has no text")
monkeypatch.setattr(worker.StorageCreator, "get_storage", lambda: fake_storage)
monkeypatch.setattr(
worker, "get_default_file_extractor", lambda ocr_enabled=False, pdf_text_fast_path=False: {}
)
file_info = {
"filename": "notes.docx",
"attachment_id": "507f1f77bcf86cd799439022",
"path": "uploads/user1/attachments/notes.docx",
"metadata": {"source": "chat"},
}
with pytest.raises(NoTextLayerError):
worker.attachment_worker(task_self, file_info, "user1")
events = [event for event, _ in published]
assert "attachment.failed" in events
assert "attachment.completed" not in events
def test_tiff_is_stored_as_png_so_providers_accept_it(
self, pg_conn, patch_worker_db, task_self, monkeypatch
):
"""OpenAI and Anthropic reject TIFF, so the row must point at a PNG copy."""
import io
from PIL import Image
from docsgpt import worker
buf = io.BytesIO()
Image.new("RGB", (32, 20), color=(10, 120, 200)).save(buf, format="TIFF")
tiff_bytes = buf.getvalue()
published: list[tuple[str, dict]] = []
saved: dict[str, bytes] = {}
def _save(data, path, **kwargs):
saved[path] = data.read()
return {"storage_type": "local"}
fake_storage = MagicMock(name="storage")
fake_storage.process_file.return_value = Document(text="", extra_info={})
fake_storage.get_file.side_effect = lambda path: io.BytesIO(tiff_bytes)
fake_storage.save_file.side_effect = _save
monkeypatch.setattr(worker.StorageCreator, "get_storage", lambda: fake_storage)
monkeypatch.setattr(
worker, "get_default_file_extractor", lambda ocr_enabled=False, pdf_text_fast_path=False: {}
)
monkeypatch.setattr(
worker,
"publish_user_event",
lambda user, event, payload, **kw: published.append((event, payload)),
)
file_info = {
"filename": "fax.tiff",
"attachment_id": "507f1f77bcf86cd799439023",
"path": "uploads/user1/attachments/abc/fax.tiff",
"metadata": {"source": "chat"},
}
result = worker.attachment_worker(task_self, file_info, "user1")
png_path = "uploads/user1/attachments/abc/fax.png"
assert list(saved) == [png_path]
assert Image.open(io.BytesIO(saved[png_path])).format == "PNG"
assert result["path"] == png_path
assert result["mime_type"] == "image/png"
assert result["metadata"]["image_conversion"] == {
"from": "image/tiff",
"to": "image/png",
"frames": 1,
}
row = AttachmentsRepository(pg_conn).get_by_legacy_id(
file_info["attachment_id"], "user1"
)
assert row["filename"] == "fax.tiff"
assert row["upload_path"] == png_path
assert row["mime_type"] == "image/png"
completed = [payload for event, payload in published if event == "attachment.completed"]
assert completed[0]["mime_type"] == "image/png"
def test_text_without_a_parser_is_parsed(
self, pg_conn, patch_worker_db, task_self, monkeypatch, tmp_path