// lib/services/pdfrx_page_text_source.dart // // Production wiring for [PdfTextIndexer]'s two injected text sources, backed by // pdfrx (the same engine the editor renders with). Kept SEPARATE from // PdfTextIndexer so the indexer's logic (the threshold decision, idempotency, // sidecar persistence) is unit-testable without the native pdfium/OCR stack — // only this file touches pdfrx, dart:ui, and the OCR engine, and it is exercised // on-device, not in CI. import 'dart:async'; import 'dart:typed_data'; import 'dart:ui' as ui; import 'package:pdfrx/pdfrx.dart'; import 'ocr_engine.dart'; /// pdfrx-backed loaders for [PdfTextIndexer]. class PdfrxPageTextSource { const PdfrxPageTextSource._(); /// Load the embedded text layer of every page (page order). Each entry is a /// page's raw text (possibly empty). Returns an empty list on any failure, so /// the indexer treats the PDF as having no text layer (→ OCR fallback). static Future> loadEmbeddedText(String pdfPath) async { PdfDocument? doc; try { doc = await PdfDocument.openFile(pdfPath); final out = []; for (final page in doc.pages) { final raw = await page.loadText(); out.add(raw?.fullText ?? ''); } return out; } catch (_) { return const []; } finally { await doc?.dispose(); } } /// Render each page and OCR it (page order). Returns one entry per page /// (empty where nothing was recognized), or an empty list when the PDF can't /// be opened. Honours the OCR engine's own graceful no-op: when no backend is /// available every page comes back empty. /// /// Rendering is done at [renderScale]× the page's native 72-dpi size to give /// the recognizer enough resolution on scanned scans without exploding memory. static Future> ocrPages( String pdfPath, { double renderScale = 2.0, }) async { PdfDocument? doc; try { doc = await PdfDocument.openFile(pdfPath); final out = []; for (final page in doc.pages) { final text = await _ocrOnePage(page, renderScale); out.add(text ?? ''); } return out; } catch (_) { return const []; } finally { await doc?.dispose(); } } static Future _ocrOnePage(PdfPage page, double renderScale) async { PdfImage? image; try { final fullWidth = page.width * renderScale; final fullHeight = page.height * renderScale; image = await page.render( fullWidth: fullWidth, fullHeight: fullHeight, ); if (image == null) return null; final png = await _bgraToPng(image.pixels, image.width, image.height); if (png == null) return null; return OcrEngine.recognizeImage(png); } catch (_) { return null; } finally { image?.dispose(); } } /// Encode pdfrx's BGRA8888 raw pixels as PNG (the format [OcrEngine] expects). static Future _bgraToPng( Uint8List bgra, int width, int height, ) async { final completer = Completer(); ui.decodeImageFromPixels( bgra, width, height, ui.PixelFormat.bgra8888, completer.complete, ); final image = await completer.future; try { final data = await image.toByteData(format: ui.ImageByteFormat.png); return data?.buffer.asUint8List(); } finally { image.dispose(); } } }