diff --git a/locale/bigocrpdf.pot b/locale/bigocrpdf.pot index 37713e75..a0835f65 100644 --- a/locale/bigocrpdf.pot +++ b/locale/bigocrpdf.pot @@ -8,7 +8,7 @@ msgid "" msgstr "" "Project-Id-Version: bigocrpdf 3.0.0\n" "Report-Msgid-Bugs-To: biglinux@biglinux.com.br\n" -"POT-Creation-Date: 2026-08-18 02:44-0300\n" +"POT-Creation-Date: 2026-08-19 13:16-0300\n" "PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" "Last-Translator: FULL NAME \n" "Language-Team: LANGUAGE \n" @@ -560,11 +560,11 @@ msgid "Results" msgstr "" #: src/bigocrpdf/processing_controller.py:91 -#: src/bigocrpdf/services/screen_capture.py:428 -#: src/bigocrpdf/services/screen_capture.py:1006 -#: src/bigocrpdf/services/screen_capture.py:1013 -#: src/bigocrpdf/services/screen_capture.py:1037 -#: src/bigocrpdf/ui/image_ocr_window.py:883 +#: src/bigocrpdf/services/screen_capture.py:431 +#: src/bigocrpdf/services/screen_capture.py:1041 +#: src/bigocrpdf/services/screen_capture.py:1048 +#: src/bigocrpdf/services/screen_capture.py:1073 +#: src/bigocrpdf/ui/image_ocr_window.py:906 msgid "OCR processing failed." msgstr "" @@ -813,75 +813,73 @@ msgstr "" msgid "OCR page {0} (rendered)..." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:480 -#: src/bigocrpdf/services/screen_capture.py:489 -#: src/bigocrpdf/services/screen_capture.py:549 -msgid "Screen capture failed." +#: src/bigocrpdf/services/screen_capture.py:503 +#: src/bigocrpdf/services/screen_capture.py:585 +#: src/bigocrpdf/ui/image_ocr_window.py:1008 +msgid "Screen capture failed. Please try again or open an image file." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:500 -#: src/bigocrpdf/services/screen_capture.py:536 -#: src/bigocrpdf/ui/image_ocr_window.py:985 -msgid "Screen capture failed. Please try again or open an image file." +#: src/bigocrpdf/services/screen_capture.py:516 +msgid "Screen capture failed." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:506 +#: src/bigocrpdf/services/screen_capture.py:579 msgid "" "No screenshot tool available. Please install spectacle, gnome-screenshot, or " "flameshot." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:979 +#: src/bigocrpdf/services/screen_capture.py:1014 msgid "Could not prepare the image for OCR." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:999 +#: src/bigocrpdf/services/screen_capture.py:1034 msgid "OCR processing timed out." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1030 +#: src/bigocrpdf/services/screen_capture.py:1066 msgid "OCR engine not available. Please check your installation." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1089 -#: src/bigocrpdf/services/screen_capture.py:1095 -#: src/bigocrpdf/services/screen_capture.py:1101 -#: src/bigocrpdf/services/screen_capture.py:1175 -#: src/bigocrpdf/services/screen_capture.py:1198 -#: src/bigocrpdf/services/screen_capture.py:1206 +#: src/bigocrpdf/services/screen_capture.py:1125 +#: src/bigocrpdf/services/screen_capture.py:1131 +#: src/bigocrpdf/services/screen_capture.py:1137 +#: src/bigocrpdf/services/screen_capture.py:1211 +#: src/bigocrpdf/services/screen_capture.py:1234 +#: src/bigocrpdf/services/screen_capture.py:1242 msgid "Could not load image file." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1103 -#: src/bigocrpdf/services/screen_capture.py:1170 -#: src/bigocrpdf/services/screen_capture.py:1277 +#: src/bigocrpdf/services/screen_capture.py:1139 +#: src/bigocrpdf/services/screen_capture.py:1206 +#: src/bigocrpdf/services/screen_capture.py:1313 msgid "Unsupported or corrupted image file." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1105 -#: src/bigocrpdf/services/screen_capture.py:1208 -#: src/bigocrpdf/services/screen_capture.py:1235 +#: src/bigocrpdf/services/screen_capture.py:1141 +#: src/bigocrpdf/services/screen_capture.py:1244 +#: src/bigocrpdf/services/screen_capture.py:1271 msgid "Image file is too large to process safely." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1121 +#: src/bigocrpdf/services/screen_capture.py:1157 msgid "No supported image decoder is available." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1274 +#: src/bigocrpdf/services/screen_capture.py:1310 msgid "Unsupported image format." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1279 +#: src/bigocrpdf/services/screen_capture.py:1315 #: src/bigocrpdf/ui/image_ocr_window.py:415 msgid "Image dimensions exceed the configured safety limit." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1285 +#: src/bigocrpdf/services/screen_capture.py:1321 msgid "Animated or multi-page images are not supported." msgstr "" -#: src/bigocrpdf/services/screen_capture.py:1306 +#: src/bigocrpdf/services/screen_capture.py:1342 msgid "Image file changed while it was being read. Please try again." msgstr "" @@ -1052,7 +1050,7 @@ msgstr "" #: src/bigocrpdf/ui/conclusion_export_mixin.py:304 #: src/bigocrpdf/ui/conclusion_export_mixin.py:309 #: src/bigocrpdf/ui/file_save_controller.py:84 -#: src/bigocrpdf/ui/header_bar.py:164 src/bigocrpdf/ui/image_ocr_window.py:737 +#: src/bigocrpdf/ui/header_bar.py:164 src/bigocrpdf/ui/image_ocr_window.py:757 #: src/bigocrpdf/ui/pdf_editor/editor_tools_controller.py:87 #: src/bigocrpdf/ui/pdf_editor/editor_tools_controller.py:189 #: src/bigocrpdf/ui/pdf_editor/editor_tools_controller.py:221 @@ -1381,7 +1379,7 @@ msgid "Text saved to {filename}" msgstr "" #: src/bigocrpdf/ui/file_save_controller.py:117 -#: src/bigocrpdf/ui/image_ocr_window.py:1187 +#: src/bigocrpdf/ui/image_ocr_window.py:1210 #: src/bigocrpdf/ui/pdf_editor/editor_window.py:681 msgid "OK" msgstr "" @@ -1493,7 +1491,7 @@ msgstr "" msgid "Start OCR processing" msgstr "" -#: src/bigocrpdf/ui/header_bar.py:226 src/bigocrpdf/ui/image_ocr_window.py:1139 +#: src/bigocrpdf/ui/header_bar.py:226 src/bigocrpdf/ui/image_ocr_window.py:1162 msgid "" "OCR is unavailable. Install the required engine and restart the application." msgstr "" @@ -1514,91 +1512,97 @@ msgstr "" msgid "Clipboard file list is too large" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:693 +#: src/bigocrpdf/ui/image_ocr_window.py:700 +#: src/bigocrpdf/ui/image_ocr_window.py:704 +#: src/bigocrpdf/ui/image_ocr_window.py:826 +msgid "Open Image" +msgstr "" + +#: src/bigocrpdf/ui/image_ocr_window.py:711 +#: src/bigocrpdf/ui/image_ocr_window.py:716 +msgid "Screen Capture" +msgstr "" + +#: src/bigocrpdf/ui/image_ocr_window.py:727 msgid "Image OCR" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:694 +#: src/bigocrpdf/ui/image_ocr_window.py:728 msgid "Extract text from images or screen captures using OCR." msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:702 -#: src/bigocrpdf/ui/image_ocr_window.py:706 -#: src/bigocrpdf/ui/image_ocr_window.py:803 -msgid "Open Image" +#: src/bigocrpdf/ui/image_ocr_window.py:736 +msgid "No text found" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:713 -#: src/bigocrpdf/ui/image_ocr_window.py:718 -msgid "Screen Capture" +#: src/bigocrpdf/ui/image_ocr_window.py:739 +msgid "" +"OCR finished but found no readable text. Capture a tighter region, or try a " +"sharper image." msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:730 +#: src/bigocrpdf/ui/image_ocr_window.py:750 msgid "Extracting text…" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:740 -#: src/bigocrpdf/ui/image_ocr_window.py:741 +#: src/bigocrpdf/ui/image_ocr_window.py:760 +#: src/bigocrpdf/ui/image_ocr_window.py:761 msgid "Cancel text extraction" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:768 +#: src/bigocrpdf/ui/image_ocr_window.py:791 msgid "Extracted text" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:780 +#: src/bigocrpdf/ui/image_ocr_window.py:803 msgid "Copy" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:782 -#: src/bigocrpdf/ui/image_ocr_window.py:784 +#: src/bigocrpdf/ui/image_ocr_window.py:805 +#: src/bigocrpdf/ui/image_ocr_window.py:807 #: src/bigocrpdf/ui/text_viewer_controller.py:208 #: src/bigocrpdf/ui/text_viewer_controller.py:209 msgid "Copy text to clipboard" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:792 +#: src/bigocrpdf/ui/image_ocr_window.py:815 msgid "New Capture" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:795 -#: src/bigocrpdf/ui/image_ocr_window.py:796 +#: src/bigocrpdf/ui/image_ocr_window.py:818 +#: src/bigocrpdf/ui/image_ocr_window.py:819 msgid "Capture a screen region" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:805 -#: src/bigocrpdf/ui/image_ocr_window.py:806 +#: src/bigocrpdf/ui/image_ocr_window.py:828 +#: src/bigocrpdf/ui/image_ocr_window.py:829 msgid "Open an image file" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:893 -msgid "No text extracted." -msgstr "" - -#: src/bigocrpdf/ui/image_ocr_window.py:1049 +#: src/bigocrpdf/ui/image_ocr_window.py:1072 msgid "Open Image to OCR" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:1052 +#: src/bigocrpdf/ui/image_ocr_window.py:1075 #: src/bigocrpdf/ui/settings_queue_mixin.py:892 #: src/bigocrpdf/ui/settings_queue_mixin.py:969 #: src/bigocrpdf/ui/welcome_dialog_controller.py:109 msgid "Images" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:1102 +#: src/bigocrpdf/ui/image_ocr_window.py:1125 msgid "Could not open the selected image" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:1127 +#: src/bigocrpdf/ui/image_ocr_window.py:1150 msgid "Could not copy text to the clipboard" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:1131 +#: src/bigocrpdf/ui/image_ocr_window.py:1154 msgid "Copied to clipboard" msgstr "" -#: src/bigocrpdf/ui/image_ocr_window.py:1185 +#: src/bigocrpdf/ui/image_ocr_window.py:1208 #: src/bigocrpdf/ui/pdf_editor/editor_window.py:679 msgid "Error" msgstr "" diff --git a/locale/pt.po b/locale/pt.po index 1a1fff3f..299f9057 100644 --- a/locale/pt.po +++ b/locale/pt.po @@ -2128,10 +2128,17 @@ msgstr "Capturar uma região do ecrã" msgid "Open an image file" msgstr "Abrir um ficheiro de imagem" -# File: bigocrpdf/src/bigocrpdf/ui/image_ocr_window.py, line: 415 -#: src/bigocrpdf/ui/image_ocr_window.py:893 -msgid "No text extracted." -msgstr "Nenhum texto extraído." +#: src/bigocrpdf/ui/image_ocr_window.py:737 +msgid "No text found" +msgstr "Nenhum texto encontrado" + +#: src/bigocrpdf/ui/image_ocr_window.py:739 +msgid "" +"OCR finished but found no readable text. Capture a tighter region, or try a " +"sharper image." +msgstr "" +"O OCR terminou, mas não encontrou texto legível. Capture uma região mais " +"justa ou tente uma imagem mais nítida." # File: bigocrpdf/src/bigocrpdf/ui/image_ocr_window.py, line: 487 #: src/bigocrpdf/ui/image_ocr_window.py:1049 diff --git a/locale/pt_BR.po b/locale/pt_BR.po index 61092292..06b14e09 100644 --- a/locale/pt_BR.po +++ b/locale/pt_BR.po @@ -2100,10 +2100,17 @@ msgstr "Capturar uma região da tela" msgid "Open an image file" msgstr "Abrir um arquivo de imagem" -# File: bigocrpdf/src/bigocrpdf/ui/image_ocr_window.py, line: 432 -#: src/bigocrpdf/ui/image_ocr_window.py:893 -msgid "No text extracted." -msgstr "Nenhum texto extraído." +#: src/bigocrpdf/ui/image_ocr_window.py:737 +msgid "No text found" +msgstr "Nenhum texto encontrado" + +#: src/bigocrpdf/ui/image_ocr_window.py:739 +msgid "" +"OCR finished but found no readable text. Capture a tighter region, or try a " +"sharper image." +msgstr "" +"O OCR terminou, mas não encontrou texto legível. Capture uma região mais " +"justa ou use uma imagem mais nítida." # File: bigocrpdf/src/bigocrpdf/ui/image_ocr_window.py, line: 486 #: src/bigocrpdf/ui/image_ocr_window.py:1049 diff --git a/src/bigocrpdf/services/screen_capture.py b/src/bigocrpdf/services/screen_capture.py index 4c8e8479..d0abf70e 100644 --- a/src/bigocrpdf/services/screen_capture.py +++ b/src/bigocrpdf/services/screen_capture.py @@ -33,6 +33,9 @@ ) from bigocrpdf.services.rapidocr_service.ocr_worker_engine import build_ocr_worker_command from bigocrpdf.services.rapidocr_service.preprocessor import ImagePreprocessor +from bigocrpdf.services.rapidocr_service.text_formatting_controller import ( + TextFormattingController, +) from bigocrpdf.utils.i18n import _ from bigocrpdf.utils.logger import logger from bigocrpdf.utils.temp_manager import mkstemp as tm_mkstemp @@ -468,45 +471,9 @@ def _run_capture_thread( fd, temp_path = tm_mkstemp(suffix=".png", prefix="bigocrpdf_capture_") os.close(fd) - # Try XDG Portal first (works in Flatpak/sandboxed environments) request.raise_if_cancelled() - portal_result = self._capture_via_portal(request) + outcome = self._capture_into_owned_file(temp_path, request) request.raise_if_cancelled() - if portal_result.status == PortalCaptureStatus.SUCCESS: - portal_path = portal_result.path - if not portal_path: - outcome = ImageOcrOutcome( - ImageOcrStatus.ERROR, - message=_("Screen capture failed."), - ) - else: - self._copy_owned_image_file(portal_path, temp_path) - elif portal_result.status == PortalCaptureStatus.CANCELLED: - outcome = ImageOcrOutcome(ImageOcrStatus.CANCELLED) - elif portal_result.status == PortalCaptureStatus.FAILED: - outcome = ImageOcrOutcome( - ImageOcrStatus.ERROR, - message=_("Screen capture failed."), - ) - else: - # Fallback to CLI tools - request.raise_if_cancelled() - cli_status = self._capture_with_cli_tools(temp_path, request) - if cli_status == CliCaptureStatus.CANCELLED: - outcome = ImageOcrOutcome(ImageOcrStatus.CANCELLED) - elif cli_status == CliCaptureStatus.FAILED: - outcome = ImageOcrOutcome( - ImageOcrStatus.ERROR, - message=_("Screen capture failed. Please try again or open an image file."), - ) - elif cli_status == CliCaptureStatus.UNAVAILABLE: - outcome = ImageOcrOutcome( - ImageOcrStatus.ERROR, - message=_( - "No screenshot tool available. Please install spectacle, " - "gnome-screenshot, or flameshot." - ), - ) # Check if file has content (screenshot was taken, not cancelled) if outcome is None and os.path.exists(temp_path) and os.path.getsize(temp_path) > 0: @@ -552,6 +519,72 @@ def _run_capture_thread( # ── Screenshot Capture ────────────────────────────────────────────── + @staticmethod + def _prefers_native_capture_tool() -> bool: + """Whether this desktop's own screenshot tool beats the portal for a region.""" + return "kde" in os.environ.get("XDG_CURRENT_DESKTOP", "").lower() + + def _capture_into_owned_file( + self, + temp_path: str, + request: ImageOcrRequest, + ) -> ImageOcrOutcome | None: + """Capture a screen region into an owned file, reporting None on success. + + On KDE, `spectacle --region --background` opens the region selector directly + and writes exactly the file we ask for. The portal's interactive mode there + opens a dialog that defaults to a full-screen grab with the pointer drawn in, + which costs several clicks and burns the cursor into the OCR input, so the + portal runs second. Everywhere else the portal leads, because it is the only + backend that works from inside a sandbox. + """ + backends = [self._capture_with_cli_tools, self._capture_via_portal_into] + if not self._prefers_native_capture_tool(): + backends.reverse() + + status = CliCaptureStatus.UNAVAILABLE + for backend in backends: + request.raise_if_cancelled() + status = backend(temp_path, request) + if status != CliCaptureStatus.UNAVAILABLE: + break + return self._outcome_for_capture_status(status) + + def _capture_via_portal_into( + self, + temp_path: str, + request: ImageOcrRequest, + ) -> CliCaptureStatus: + """Run the portal backend and land its borrowed image in our owned file.""" + result = self._capture_via_portal(request) + if result.status != PortalCaptureStatus.SUCCESS: + # Both enums are StrEnums over the same terminal states. + return CliCaptureStatus(result.status) + if not result.path: + return CliCaptureStatus.FAILED + self._copy_owned_image_file(result.path, temp_path) + return CliCaptureStatus.SUCCESS + + @staticmethod + def _outcome_for_capture_status(status: CliCaptureStatus) -> ImageOcrOutcome | None: + """Map a terminal capture status to the outcome to report, None when captured.""" + if status == CliCaptureStatus.SUCCESS: + return None + if status == CliCaptureStatus.CANCELLED: + return ImageOcrOutcome(ImageOcrStatus.CANCELLED) + if status == CliCaptureStatus.UNAVAILABLE: + return ImageOcrOutcome( + ImageOcrStatus.ERROR, + message=_( + "No screenshot tool available. Please install spectacle, " + "gnome-screenshot, or flameshot." + ), + ) + return ImageOcrOutcome( + ImageOcrStatus.ERROR, + message=_("Screen capture failed. Please try again or open an image file."), + ) + @staticmethod def _is_portal_unavailable_error(error: Exception) -> bool: """Return whether D-Bus reports that the screenshot portal is absent.""" @@ -912,13 +945,15 @@ def _run_standard_tool( return CliCaptureStatus.FAILED finally: request.unbind_process(proc) + # Spectacle has been observed to crash on exit after writing a complete PNG, + # so the bytes on disk decide the outcome before the exit status does. + if os.path.getsize(temp_path) > 0: + return CliCaptureStatus.SUCCESS if proc.returncode != 0: logger.debug( f"{tool_name} exited with code {proc.returncode}: {stderr.decode().strip()}" ) return CliCaptureStatus.FAILED - if os.path.getsize(temp_path) > 0: - return CliCaptureStatus.SUCCESS return CliCaptureStatus.CANCELLED # ── RapidOCR Image Processing ─────────────────────────────────────── @@ -1014,8 +1049,9 @@ def _extract_text_result( if not results: return None, None - # Format text with reading order and paragraph detection - text = self._format_text(results) + # Reuse the pipeline formatter so a capture and a page of the same + # layout produce the same lines, columns, and paragraphs. + text = TextFormattingController(config).format(results, float(img.shape[1])) return (text if text.strip() else None), None finally: @@ -1385,62 +1421,6 @@ def _parse_ocr_results(cls, stdout: str) -> list[OCRResult]: logger.error(f"Failed to parse OCR result: {e}") return [] - @staticmethod - def _format_text(results: list[OCRResult]) -> str: - """Format OCR results into readable text with line breaks and paragraphs. - - Sorts text boxes by reading order (top-to-bottom, left-to-right) - and inserts appropriate line/paragraph breaks based on vertical spacing. - - Args: - results: List of OCR results with text and bounding boxes - - Returns: - Formatted text string - """ - if not results: - return "" - - # Sort by reading order: top-to-bottom, left-to-right - def sort_key(r: OCRResult) -> tuple[float, float]: - ys = [p[1] for p in r.box] - xs = [p[0] for p in r.box] - return (min(ys), min(xs)) - - sorted_results = sorted(results, key=sort_key) - - text = "" - prev_y = -1.0 - prev_bottom = -1.0 - - for r in sorted_results: - ys = [p[1] for p in r.box] - curr_top = min(ys) - curr_bottom = max(ys) - curr_h = curr_bottom - curr_top - center_y = (curr_top + curr_bottom) / 2 - - if prev_y != -1: - # Column break (moved UP significantly) - if center_y < prev_y - (curr_h * 2): - text += "\n\n" - # Paragraph break (vertical gap > 60% of line height) - elif (curr_top - prev_bottom) > (curr_h * 0.6): - text += "\n\n" - # Line break (moved DOWN past previous bottom) - elif center_y > prev_bottom: - text += "\n" - elif center_y > prev_y + (curr_h * 0.5): - text += "\n" - else: - text += " " - - text += r.text - prev_y = center_y - prev_bottom = curr_bottom - - return text - # ── Callback Helpers ──────────────────────────────────────────────── def _cleanup_temp_file(self, path: str) -> None: diff --git a/src/bigocrpdf/ui/image_ocr_window.py b/src/bigocrpdf/ui/image_ocr_window.py index 52d07e4b..e6bf0423 100644 --- a/src/bigocrpdf/ui/image_ocr_window.py +++ b/src/bigocrpdf/ui/image_ocr_window.py @@ -98,7 +98,6 @@ def __init__( self._input_cancellable: Gio.Cancellable | None = None self._input_generation = 0 self._stable_page_name = "welcome" - self._result_copyable = False self._clipboard_encode_lock = threading.Lock() self._clipboard_encode_pending: _ClipboardEncodeJob | None = None self._clipboard_encode_active: _ClipboardEncodeJob | None = None @@ -284,6 +283,7 @@ def _setup_ui(self) -> None: self._build_welcome_page() self._build_loading_page() self._build_results_page() + self._build_empty_page() # Enable drag-and-drop for image files self._setup_drop_target() @@ -686,14 +686,12 @@ def on_closed(source: Gio.InputStream, result: Gio.AsyncResult) -> None: except (AttributeError, TypeError): stream.close(None) - def _build_welcome_page(self) -> None: - """Build the welcome page with Adw.StatusPage and action buttons.""" - status = Adw.StatusPage() - status.set_icon_name("camera-photo-symbolic") - status.set_title(_("Image OCR")) - status.set_description(_("Extract text from images or screen captures using OCR.")) + def _build_source_buttons(self) -> Gtk.Box: + """Build a fresh pair of source actions for one status page. - # Action buttons embedded in the status page + Every page that has no text to show offers the same two ways to get some, + and a widget belongs to a single parent, so each page needs its own pair. + """ btn_box = Gtk.Box(orientation=Gtk.Orientation.HORIZONTAL, spacing=12) btn_box.set_halign(Gtk.Align.CENTER) @@ -720,9 +718,31 @@ def _build_welcome_page(self) -> None: self._apply_ocr_availability_to_button(capture_btn) btn_box.append(capture_btn) - status.set_child(btn_box) + return btn_box + + def _build_welcome_page(self) -> None: + """Build the welcome page with Adw.StatusPage and action buttons.""" + status = Adw.StatusPage() + status.set_icon_name("camera-photo-symbolic") + status.set_title(_("Image OCR")) + status.set_description(_("Extract text from images or screen captures using OCR.")) + status.set_child(self._build_source_buttons()) self._stack.add_named(status, "welcome") + def _build_empty_page(self) -> None: + """Build the page shown when OCR ran and found no readable text.""" + status = Adw.StatusPage() + status.set_icon_name("x-office-document-symbolic") + status.set_title(_("No text found")) + status.set_description( + _( + "OCR finished but found no readable text. Capture a tighter region, " + "or try a sharper image." + ) + ) + status.set_child(self._build_source_buttons()) + self._stack.add_named(status, "empty") + def _build_loading_page(self) -> None: """Build the loading page with Adw.StatusPage and spinner.""" status = Adw.StatusPage() @@ -761,6 +781,9 @@ def _build_results_page(self) -> None: self._text_view = Gtk.TextView() self._text_view.set_editable(True) self._text_view.set_wrap_mode(Gtk.WrapMode.WORD_CHAR) + # The formatter aligns columns and indents with spaces, which only line up + # under a fixed-width font. + self._text_view.add_css_class("monospace") self._text_view.set_left_margin(18) self._text_view.set_right_margin(18) self._text_view.set_top_margin(12) @@ -887,23 +910,23 @@ def _on_processing_complete(self, outcome: ImageOcrOutcome) -> None: if outcome.status == ImageOcrStatus.SUCCESS and outcome.text: self._text_buffer.set_text(outcome.text) - self._result_copyable = True self._copy_button.set_sensitive(True) + self._stable_page_name = "results" else: - self._text_buffer.set_text(_("No text extracted.")) - self._result_copyable = False + # An empty result is its own state. Putting the explanation in the + # editable buffer would offer prose the OCR never read as if it were + # the extracted text. + self._text_buffer.set_text("") self._copy_button.set_sensitive(False) + self._stable_page_name = "empty" - self._stack.set_visible_child_name("results") - self._stable_page_name = "results" + self._stack.set_visible_child_name(self._stable_page_name) def _sync_copy_button_state(self) -> None: """Enable Copy exactly when the current result buffer has text.""" start_iter, end_iter = self._text_buffer.get_bounds() text = self._text_buffer.get_text(start_iter, end_iter, True) - self._copy_button.set_sensitive( - bool(text) and bool(getattr(self, "_result_copyable", False)) - ) + self._copy_button.set_sensitive(bool(text)) # ── Capture & Open ────────────────────────────────────────────────── diff --git a/src/bigocrpdf/utils/durable_writes.py b/src/bigocrpdf/utils/durable_writes.py index c1fff242..c485576e 100644 --- a/src/bigocrpdf/utils/durable_writes.py +++ b/src/bigocrpdf/utils/durable_writes.py @@ -331,7 +331,7 @@ def publish_files_transactionally( target_candidates, ) ) - source_identities = _validate_sources(sources, targets, directory) + source_identities = _validate_sources(sources, targets) resolved_retire_requests = retire_requests if retire_candidates is not None: resolved_retire_requests = [ @@ -1174,18 +1174,16 @@ def _remove_staged_sources(prepared: list[_PreparedPublication]) -> None: def _validate_sources( sources: list[Path], targets: list[Path], - directory: Path, ) -> list[_FileIdentity]: if set(sources) & set(targets): raise ValueError("A staged source cannot also be a publication target") - directory_device = directory.stat().st_dev + # Snapshots are created beside targets before atomic installation. Source + # devices need not match; OverlayFS may even differ for files and directories. identities: list[_FileIdentity] = [] for source in sources: source_stat = source.lstat() if not stat.S_ISREG(source_stat.st_mode): raise ValueError(f"Staged output is not a regular file: {source}") - if source_stat.st_dev != directory_device: - raise OSError("Staged output and destination must be on the same filesystem") flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_CLOEXEC", 0) descriptor = os.open(source, flags) try: @@ -1614,7 +1612,6 @@ def _read_journal(path: Path) -> _Journal: overwrite, entries = _parse_journal_payload( payload, transaction_id, - path.parent, ) return _Journal( path=path, @@ -1629,7 +1626,6 @@ def _read_journal(path: Path) -> _Journal: def _parse_journal_payload( payload: object, transaction_id: str, - directory: Path, ) -> tuple[bool, tuple[_JournalEntry, ...]]: if not isinstance(payload, dict) or set(payload) != { "version", @@ -1654,7 +1650,6 @@ def _parse_journal_payload( ): raise ValueError("Invalid publication journal schema") - directory_device = directory.stat().st_dev entries: list[_JournalEntry] = [] target_names: set[str] = set() for raw_entry in raw_entries: @@ -1685,21 +1680,12 @@ def _parse_journal_payload( or target_name in target_names ): raise ValueError("Invalid publication journal target") - new_identity = _parse_identity( - raw_entry["new_identity"], - directory_device, - ) + new_identity = _parse_identity(raw_entry["new_identity"]) new_mode = raw_entry["new_mode"] if new_mode is not None and (type(new_mode) is not int or not 0 <= new_mode <= 0o777): raise ValueError("Invalid publication output mode") - original_identity = _parse_identity( - raw_entry["original_identity"], - directory_device, - ) - backup_identity = _parse_identity( - raw_entry["backup_identity"], - directory_device, - ) + original_identity = _parse_identity(raw_entry["original_identity"]) + backup_identity = _parse_identity(raw_entry["backup_identity"]) backup_mode = raw_entry["backup_mode"] if backup_mode is not None and ( type(backup_mode) is not int or not 0 <= backup_mode <= 0o7777 @@ -1748,14 +1734,14 @@ def _parse_journal_payload( return overwrite, tuple(entries) -def _parse_identity(value: object, directory_device: int) -> _FileIdentity | None: +def _parse_identity(value: object) -> _FileIdentity | None: + # Validate identities against files during recovery, not the parent device. if value is None: return None if ( not isinstance(value, list) or len(value) != 2 or any(type(part) is not int or part < 0 for part in value) - or value[0] != directory_device ): raise ValueError("Invalid publication journal identity") return _FileIdentity(value[0], value[1]) diff --git a/tests/test_durable_writes.py b/tests/test_durable_writes.py index 40751b4f..6fe2e6b4 100644 --- a/tests/test_durable_writes.py +++ b/tests/test_durable_writes.py @@ -45,6 +45,32 @@ def fail_after_partial_write(stream) -> None: assert list(tmp_path.glob(".settings.json.*.tmp")) == [] +@pytest.mark.parametrize("overwrite", [False, True]) +def test_atomic_write_accepts_overlay_directory_device( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, overwrite: bool +) -> None: + target = tmp_path / "settings.json" + if overwrite: + target.write_text("original", encoding="utf-8") + real_stat = Path.stat + + def overlay_stat(path, *args, **kwargs): + result = real_stat(path, *args, **kwargs) + if path == tmp_path: + fields = list(result) + fields[2] += 1 # OverlayFS can report distinct directory and file devices. + return os.stat_result(fields) + return result + + monkeypatch.setattr(Path, "stat", overlay_stat) + + published = write_text_atomically(target, '{"language": "pt"}', overwrite=overwrite) + + assert published == target + assert json.loads(target.read_text(encoding="utf-8")) == {"language": "pt"} + assert list(tmp_path.iterdir()) == [target] + + def test_atomic_copy_publishes_complete_file_and_preserves_source(tmp_path: Path) -> None: source = tmp_path / "source.pdf" source.write_bytes(b"complete pdf payload") diff --git a/tests/test_screen_capture.py b/tests/test_screen_capture.py index 29bb079a..d038d110 100644 --- a/tests/test_screen_capture.py +++ b/tests/test_screen_capture.py @@ -32,6 +32,9 @@ from bigocrpdf.services import screen_capture as screen_capture_module # noqa: E402 from bigocrpdf.services.rapidocr_service.config import OCRConfig, OCRResult # noqa: E402 +from bigocrpdf.services.rapidocr_service.text_formatting_controller import ( # noqa: E402 + TextFormattingController, +) from bigocrpdf.services.screen_capture import ScreenCaptureService # noqa: E402 for mod, original in _saved.items(): @@ -58,6 +61,7 @@ def fake_mkstemp(*, suffix, prefix): return fd, str(owned_path) service = ScreenCaptureService() + monkeypatch.setenv("XDG_CURRENT_DESKTOP", "GNOME") monkeypatch.setattr("bigocrpdf.services.screen_capture.tm_mkstemp", fake_mkstemp) monkeypatch.setattr( service, @@ -193,6 +197,7 @@ def fake_mkstemp(*, suffix, prefix): service = ScreenCaptureService() cli_capture = MagicMock() + monkeypatch.setenv("XDG_CURRENT_DESKTOP", "GNOME") monkeypatch.setattr("bigocrpdf.services.screen_capture._", lambda message: message) monkeypatch.setattr("bigocrpdf.services.screen_capture.tm_mkstemp", fake_mkstemp) monkeypatch.setattr( @@ -219,6 +224,79 @@ def fake_mkstemp(*, suffix, prefix): assert not owned_path.exists() +def test_kde_captures_with_spectacle_before_asking_the_portal(monkeypatch, tmp_path): + # The portal's interactive mode on KDE opens a dialog defaulting to a full-screen + # grab with the pointer drawn in; Spectacle's region mode is one drag. + service = ScreenCaptureService() + monkeypatch.setenv("XDG_CURRENT_DESKTOP", "KDE") + portal = MagicMock() + monkeypatch.setattr(service, "_capture_via_portal_into", portal) + monkeypatch.setattr( + service, + "_capture_with_cli_tools", + lambda _path, _request: screen_capture_module.CliCaptureStatus.SUCCESS, + ) + + outcome = service._capture_into_owned_file( + str(tmp_path / "owned.png"), + screen_capture_module.ImageOcrRequest(), + ) + + assert outcome is None + portal.assert_not_called() + + +def test_capture_falls_back_to_the_portal_when_no_native_tool_answers(monkeypatch, tmp_path): + service = ScreenCaptureService() + monkeypatch.setenv("XDG_CURRENT_DESKTOP", "KDE") + monkeypatch.setattr( + service, + "_capture_with_cli_tools", + lambda _path, _request: screen_capture_module.CliCaptureStatus.UNAVAILABLE, + ) + monkeypatch.setattr( + service, + "_capture_via_portal_into", + lambda _path, _request: screen_capture_module.CliCaptureStatus.SUCCESS, + ) + + outcome = service._capture_into_owned_file( + str(tmp_path / "owned.png"), + screen_capture_module.ImageOcrRequest(), + ) + + assert outcome is None + + +def test_screenshot_tool_that_crashes_after_writing_still_yields_the_capture( + monkeypatch, + tmp_path, +): + # Observed on Plasma: spectacle wrote a complete PNG and then exited on SIGSEGV. + capture_path = tmp_path / "shot.png" + capture_path.write_bytes(b"complete png bytes") + service = ScreenCaptureService() + monkeypatch.setattr(screen_capture_module.shutil, "which", lambda _tool: "/usr/bin/spectacle") + monkeypatch.setattr( + screen_capture_module.subprocess, + "Popen", + lambda *_args, **_kwargs: SimpleNamespace( + args=["spectacle"], + returncode=-11, + poll=lambda: -11, + communicate=lambda timeout=None: (b"", b""), + ), + ) + + status = service._run_standard_tool( + ["spectacle", "-r", "-b", "-n", "-o", str(capture_path)], + str(capture_path), + screen_capture_module.ImageOcrRequest(), + ) + + assert status == screen_capture_module.CliCaptureStatus.SUCCESS + + @pytest.mark.parametrize( ("is_unavailable", "expected_status"), ((True, "unavailable"), (False, "failed")), @@ -943,15 +1021,18 @@ def test_multiple_results(self): class TestFormatText: - """Tests for ScreenCaptureService._format_text.""" + """Image captures are formatted by the shared pipeline formatter.""" + + @staticmethod + def _format(results): + return TextFormattingController(OCRConfig(language="latin")).format(results, 400.0) def test_empty_results(self): - assert ScreenCaptureService._format_text([]) == "" + assert self._format([]) == "" def test_single_line(self): results = [OCRResult(text="Hello World", box=[[0, 10], [200, 10], [200, 30], [0, 30]])] - text = ScreenCaptureService._format_text(results) - assert "Hello World" in text + assert "Hello World" in self._format(results) def test_reading_order(self): # Second box is above first box — should appear first in output @@ -959,15 +1040,24 @@ def test_reading_order(self): OCRResult(text="Line 2", box=[[0, 50], [200, 50], [200, 70], [0, 70]]), OCRResult(text="Line 1", box=[[0, 10], [200, 10], [200, 30], [0, 30]]), ] - text = ScreenCaptureService._format_text(results) - pos1 = text.find("Line 1") - pos2 = text.find("Line 2") - assert pos1 < pos2 + text = self._format(results) + assert text.find("Line 1") < text.find("Line 2") def test_paragraph_break(self): results = [ OCRResult(text="Para 1", box=[[0, 10], [200, 10], [200, 30], [0, 30]]), OCRResult(text="Para 2", box=[[0, 60], [200, 60], [200, 80], [0, 80]]), ] - text = ScreenCaptureService._format_text(results) - assert "\n\n" in text + assert "\n\n" in self._format(results) + + def test_boxes_on_one_visual_line_keep_left_to_right_order(self): + # Real detections jitter vertically by a few pixels on the same line. Sorting + # by box top alone reorders them, which scrambled every multi-box capture. + results = [ + OCRResult(text="Nome:", box=[[10, 100], [90, 100], [90, 130], [10, 130]]), + OCRResult(text="Joao", box=[[100, 98], [200, 98], [200, 128], [100, 128]]), + OCRResult(text="Silva", box=[[210, 102], [320, 102], [320, 132], [210, 132]]), + ] + text = self._format(results) + assert text.split() == ["Nome:", "Joao", "Silva"] + assert "\n" not in text.strip() diff --git a/tests/test_ui_action_state.py b/tests/test_ui_action_state.py index c3dca937..830edc40 100644 --- a/tests/test_ui_action_state.py +++ b/tests/test_ui_action_state.py @@ -177,7 +177,6 @@ def test_cancelled_capture_restores_stable_page_without_showing_error(): _stable_page_name="welcome", _text_buffer=text_buffer, _copy_button=MagicMock(), - _result_copyable=True, ) window._sync_copy_button_state = ImageOcrWindow._sync_copy_button_state.__get__(window) @@ -201,7 +200,6 @@ def test_failed_image_ocr_restores_copy_for_previous_result(): _stable_page_name="results", _text_buffer=text_buffer, _copy_button=MagicMock(), - _result_copyable=True, ) window._sync_copy_button_state = ImageOcrWindow._sync_copy_button_state.__get__(window) @@ -215,18 +213,36 @@ def test_failed_image_ocr_restores_copy_for_previous_result(): window._copy_button.set_sensitive.assert_called_once_with(True) +def test_empty_result_shows_the_empty_page_instead_of_placeholder_prose(): + text_buffer = MagicMock() + text_buffer.get_bounds.return_value = (object(), object()) + window = SimpleNamespace( + _show_error=MagicMock(), + _stack=MagicMock(), + _stable_page_name="welcome", + _text_buffer=text_buffer, + _copy_button=MagicMock(), + ) + + ImageOcrWindow._on_processing_complete(window, ImageOcrOutcome(ImageOcrStatus.EMPTY)) + + window._stack.set_visible_child_name.assert_called_once_with("empty") + assert window._stable_page_name == "empty" + text_buffer.set_text.assert_called_once_with("") + window._copy_button.set_sensitive.assert_called_once_with(False) + + @pytest.mark.parametrize("status", (ImageOcrStatus.CANCELLED, ImageOcrStatus.ERROR)) -def test_empty_result_message_never_becomes_copyable_after_later_failure(status): +def test_failure_after_an_empty_result_keeps_copy_disabled(status): text_buffer = MagicMock() text_buffer.get_bounds.return_value = (object(), object()) - text_buffer.get_text.return_value = "No text extracted." + text_buffer.get_text.return_value = "" window = SimpleNamespace( _show_error=MagicMock(), _stack=MagicMock(), - _stable_page_name="results", + _stable_page_name="empty", _text_buffer=text_buffer, _copy_button=MagicMock(), - _result_copyable=False, ) window._sync_copy_button_state = ImageOcrWindow._sync_copy_button_state.__get__(window) @@ -235,6 +251,7 @@ def test_empty_result_message_never_becomes_copyable_after_later_failure(status) ImageOcrOutcome(status, message="failed" if status == ImageOcrStatus.ERROR else None), ) + window._stack.set_visible_child_name.assert_called_once_with("empty") window._copy_button.set_sensitive.assert_called_once_with(False) diff --git a/usr/share/icons/hicolor/scalable/apps/bigocrimage.svg b/usr/share/icons/hicolor/scalable/apps/bigocrimage.svg index ccfdf154..6edd7944 100644 --- a/usr/share/icons/hicolor/scalable/apps/bigocrimage.svg +++ b/usr/share/icons/hicolor/scalable/apps/bigocrimage.svg @@ -4,8 +4,8 @@ version="1.1" id="svg11414" sodipodi:docname="bigocrimage.svg" - width="24" - height="24" + width="256" + height="256" xml:space="preserve" inkscape:version="1.4.3 (0d15f75042, 2025-12-25)" inkscape:export-filename="restaurar.png" diff --git a/usr/share/icons/hicolor/scalable/apps/bigocrpdf.svg b/usr/share/icons/hicolor/scalable/apps/bigocrpdf.svg index 582ad5ea..31acf27d 100644 --- a/usr/share/icons/hicolor/scalable/apps/bigocrpdf.svg +++ b/usr/share/icons/hicolor/scalable/apps/bigocrpdf.svg @@ -4,8 +4,8 @@ version="1.1" id="svg11414" sodipodi:docname="big-pdf-ocr-converter.svg" - width="24" - height="24" + width="256" + height="256" xml:space="preserve" inkscape:version="1.4.2 (ebf0e940d0, 2025-05-08)" inkscape:export-filename="restaurar.png" diff --git a/usr/share/locale/pt/LC_MESSAGES/bigocrpdf.mo b/usr/share/locale/pt/LC_MESSAGES/bigocrpdf.mo index f2109198..55769d5e 100644 Binary files a/usr/share/locale/pt/LC_MESSAGES/bigocrpdf.mo and b/usr/share/locale/pt/LC_MESSAGES/bigocrpdf.mo differ diff --git a/usr/share/locale/pt_BR/LC_MESSAGES/bigocrpdf.mo b/usr/share/locale/pt_BR/LC_MESSAGES/bigocrpdf.mo index af73424c..e4ad7135 100644 Binary files a/usr/share/locale/pt_BR/LC_MESSAGES/bigocrpdf.mo and b/usr/share/locale/pt_BR/LC_MESSAGES/bigocrpdf.mo differ