Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Prev Previous commit
Next Next commit
Add config option and sanity fixes
  • Loading branch information
LeonarddeR committed Aug 25, 2023
commit 04223ec46611683b8a7b5b3049629481b6993b27
6 changes: 4 additions & 2 deletions source/config/configSpec.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,9 +9,9 @@
from io import StringIO
from configobj import ConfigObj

#: The version of the schema outlined in this file. Increment this when modifying the schema and
#: The version of the schema outlined in this file. Increment this when modifying the schema and
#: provide an upgrade step (@see profileUpgradeSteps.py). An upgrade step does not need to be added when
#: just adding a new element to (or removing from) the schema, only when old versions of the config
#: just adding a new element to (or removing from) the schema, only when old versions of the config
#: (conforming to old schema versions) will not work correctly with the new schema.
latestSchemaVersion = 10

Expand Down Expand Up @@ -297,6 +297,8 @@

[uwpOcr]
language = string(default="")
autoRefresh = boolean(default=false)
autoRefreshIntervalMs = integer(default=1500,min=500)
Comment thread
LeonarddeR marked this conversation as resolved.
Outdated
Comment thread
LeonarddeR marked this conversation as resolved.
Outdated

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It is not common to have the unit in the config param name:

Suggested change
autoRefreshIntervalMs = integer(default=1500,min=500)
autoRefreshInterval = integer(default=1500,min=500)


[upgrade]
newLaptopKeyboardLayout = boolean(default=false)
Expand Down
87 changes: 48 additions & 39 deletions source/contentRecog/__init__.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
#contentRecog/__init__.py
#A part of NonVisual Desktop Access (NVDA)
#Copyright (C) 2017 NV Access Limited
#This file is covered by the GNU General Public License.
#See the file COPYING for more details.
# A part of NonVisual Desktop Access (NVDA)
# Copyright (C) 2017-2023 NV Access Limited, James Teh, Leonard de Ruijter
# This file is covered by the GNU General Public License.
# See the file COPYING for more details.

"""Framework for recognition of content; OCR, image recognition, etc.
When authors don't provide sufficient information for a screen reader user to determine the content of something,
Expand All @@ -14,11 +13,16 @@
"""

from collections import namedtuple
import ctypes
from typing import Callable, Dict, List, Union
import garbageHandler
import cursorManager
import textInfos.offsets
from abc import ABCMeta, abstractmethod
from locationHelper import RectLTWH
from NVDAObjects import NVDAObject

onRecognizeResultCallbackT = Callable[[Union["RecognitionResult", Exception]], None]


class BaseContentRecogTextInfo(cursorManager._ReviewCursorManagerTextInfo):
Expand All @@ -31,26 +35,25 @@ class ContentRecognizer(garbageHandler.TrackedObject, metaclass=ABCMeta):
"""Implementation of a content recognizer.
"""

#: Whether to allow automatic, periodic refresh when using this recognizer.
#: This allows the user to see live changes as they occur. However, if a
#: recognizer uses an internet service or is very resource intensive, this
#: may be undesirable.
allowAutoRefresh = False
allowAutoRefresh: bool = False
"""
Whether to allow automatic, periodic refresh when using this recognizer.
This allows the user to see live changes as they occur. However, if a
recognizer uses an internet service or is very resource intensive, this
may be undesirable.
"""

def getResizeFactor(self, width, height):
def getResizeFactor(self, width: int, height: int) -> Union[int, float]:
"""Return the factor by which an image must be resized
before it is passed to this recognizer.
@param width: The width of the image in pixels.
@type width: int
@param height: The height of the image in pixels.
@type height: int
@return: The resize factor, C{1} for no resizing.
@rtype: int or float
"""
return 1

@abstractmethod
def recognize(self, pixels, imageInfo, onResult):
def recognize(self, pixels: ctypes.Array, imageInfo: "RecogImageInfo", onResult: onRecognizeResultCallbackT):
"""Asynchronously recognize content from an image.
This method should not block.
Only one recognition can be performed at a time.
Expand All @@ -62,9 +65,8 @@ def recognize(self, pixels, imageInfo, onResult):
However, the alpha channel should be ignored.
@type pixels: Two dimensional array (y then x) of L{winGDI.RGBQUAD}
@param imageInfo: Information about the image for recognition.
@type imageInfo: L{RecogImageInfo}
@param onResult: A callable which takes a L{RecognitionResult} (or an exception on failure) as its only argument.
@type onResult: callable
@param onResult: A callable which takes a L{RecognitionResult} (or an exception on failure)
as its only argument.
"""
raise NotImplementedError

Expand All @@ -79,17 +81,16 @@ def validateCaptureBounds(self, location: RectLTWH) -> bool:
"""
return True

def validateObject(self, nav):
def validateObject(self, nav: NVDAObject) -> bool:
"""Validation to be performed on the navigator object before content recognition
@param nav: The navigator object to be validated
@type nav: L{NVDAObjects.NVDAObject}
@return: C{True} or C{False}, depending on whether the navigator object is valid or not.
C{True} for no validation.
@rtype: bool
"""
return True

class RecogImageInfo(object):

class RecogImageInfo:
"""Encapsulates information about a recognized image and
provides functionality to convert coordinates.
An image captured for recognition can begin at any point on the screen.
Expand All @@ -103,18 +104,20 @@ class RecogImageInfo(object):
This is done using the L{convertXToScreen} and L{convertYToScreen} methods.
"""

def __init__(self, screenLeft, screenTop, screenWidth, screenHeight, resizeFactor):
def __init__(
self,
screenLeft: int,
screenTop: int,
screenWidth: int,
screenHeight: int,
resizeFactor: Union[int, float]
):
"""
@param screenLeft: The x screen coordinate of the upper-left corner of the image.
@type screenLeft: int
@param screenTop: The y screen coordinate of the upper-left corner of the image.
@type screenTop: int
@param screenWidth: The width of the image on the screen.
@type screenWidth: int
@param screenHeight: The height of the image on the screen.
@type screenHeight: int
@param resizeFactor: The factor by which the image must be resized for recognition.
@type resizeFactor: int or float
@raise ValueError: If the supplied screen coordinates indicate that
the image is not visible; e.g. width or height of 0.
"""
Expand All @@ -131,7 +134,14 @@ def __init__(self, screenLeft, screenTop, screenWidth, screenHeight, resizeFacto
self.recogHeight = int(screenHeight * resizeFactor)

@classmethod
def createFromRecognizer(cls, screenLeft, screenTop, screenWidth, screenHeight, recognizer):
def createFromRecognizer(
cls,
screenLeft: int,
screenTop: int,
screenWidth: int,
screenHeight: int,
recognizer: ContentRecognizer
):
"""Convenience method to construct an instance using a L{ContentRecognizer}.
The resize factor is obtained by calling L{ContentRecognizer.getResizeFactor}.
"""
Expand Down Expand Up @@ -178,18 +188,20 @@ def makeTextInfo(self, obj, position) -> BaseContentRecogTextInfo:
"""
raise NotImplementedError


# Used internally by LinesWordsResult.
# (Lwr is short for LinesWordsResult.)
LwrWord = namedtuple("LwrWord", ("offset", "left", "top", "width", "height"))


class LinesWordsResult(RecognitionResult):
"""A L{RecognizerResult} which can create TextInfos based on a simple lines/words data structure.
The data structure is a list of lines, wherein each line is a list of words,
wherein each word is a dict containing the keys x, y, width, height and text.
Several OCR engines produce output in a format which can be easily converted to this.
"""

def __init__(self, data, imageInfo):
def __init__(self, data: List[List[Dict[str, Union[str, int]]]], imageInfo: RecogImageInfo):
"""Constructor.
@param data: The lines/words data structure. For example:
[
Expand All @@ -202,11 +214,9 @@ def __init__(self, data, imageInfo):
{"x": 117, "y": 105, "width": 11, "height": 9, "text": "Word4"}
]
]
@type data: list of lists of dicts
@param imageInfo: Information about the recognized image.
This is used to convert coordinates in the recognized image
to screen coordinates.
@type imageInfo: L{RecogImageInfo}
"""
self.data = data
self.imageInfo = imageInfo
Expand All @@ -229,11 +239,13 @@ def _parseData(self):
# Separate with a space.
self._textList.append(" ")
self.textLen += 1
self.words.append(LwrWord(self.textLen,
self.words.append(LwrWord(
self.textLen,
self.imageInfo.convertXToScreen(word["x"]),
self.imageInfo.convertYToScreen(word["y"]),
self.imageInfo.convertWidthToScreen(word["width"]),
self.imageInfo.convertHeightToScreen(word["height"])))
self.imageInfo.convertHeightToScreen(word["height"]))
)
text = word["text"]
self._textList.append(text)
self.textLen += len(text)
Expand All @@ -255,7 +267,7 @@ class LwrTextInfo(BaseContentRecogTextInfo, textInfos.offsets.OffsetsTextInfo):

def __init__(self, obj, position, result):
self.result = result
super(LwrTextInfo, self).__init__(obj, position)
super().__init__(obj, position)

def copy(self):
return self.__class__(self.obj, self.bookmark, self.result)
Expand Down Expand Up @@ -321,7 +333,7 @@ class SimpleResultTextInfo(BaseContentRecogTextInfo, textInfos.offsets.OffsetsTe

def __init__(self, obj, position, result):
self.result = result
super(SimpleResultTextInfo, self).__init__(obj, position)
super().__init__(obj, position)

def copy(self):
return self.__class__(self.obj, self.bookmark, self.result)
Expand All @@ -331,6 +343,3 @@ def _getStoryText(self):

def _getStoryLength(self):
return len(self.result.text)

def _getStoryText(self):
return self.result.text
55 changes: 29 additions & 26 deletions source/contentRecog/recogUi.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
#contentRecog/recogUi.py
#A part of NonVisual Desktop Access (NVDA)
#Copyright (C) 2017 NV Access Limited
#This file is covered by the GNU General Public License.
#See the file COPYING for more details.
# A part of NonVisual Desktop Access (NVDA)
# Copyright (C) 2017-2023 NV Access Limited, James Teh, Leonard de RUijter
# This file is covered by the GNU General Public License.
# See the file COPYING for more details.

"""User interface for content recognition.
This module provides functionality to capture an image from the screen
Expand All @@ -12,6 +11,7 @@
"""

import api
import config
import ui
import screenBitmap
import NVDAObjects.window
Expand All @@ -24,7 +24,7 @@
from logHandler import log
import queueHandler
import core
from . import RecogImageInfo, BaseContentRecogTextInfo
from . import RecogImageInfo, ContentRecognizer


class RecogResultNVDAObject(cursorManager.CursorManager, LiveText, NVDAObjects.window.Window):
Expand All @@ -33,28 +33,25 @@ class RecogResultNVDAObject(cursorManager.CursorManager, LiveText, NVDAObjects.w
Pressing enter will activate (e.g. click) the text at the cursor.
Pressing escape dismisses the recognition result.
"""
#: How often (in ms) to perform recognition.
REFRESH_INTERVAL = 1500

role = controlTypes.Role.DOCUMENT
# Translators: The title of the document used to present the result of content recognition.
name = _("Result")
treeInterceptor = None

def __init__(self, recognizer=None, imageInfo=None, obj=None):
def __init__(
self,
recognizer: ContentRecognizer,
imageInfo: RecogImageInfo,
obj: NVDAObjects.NVDAObject
):
self.parent = parent = api.getFocusObject()
self.recognizer = recognizer
self.imageInfo = imageInfo
self.result = None
super(RecogResultNVDAObject, self).__init__(windowHandle=parent.windowHandle)
super().__init__(windowHandle=parent.windowHandle)
LiveText.initOverlayClass(self)

def start(self):
self._recognize(self._onFirstResult)

def _get_hasFocus(self):
return self is api.getFocusObject()

def _recognize(self, onResult):
Comment thread
LeonarddeR marked this conversation as resolved.
Outdated
if self.result and not self.hasFocus:
# We've already recognized once, so we did have focus, but we don't any
Expand All @@ -74,7 +71,7 @@ def _onFirstResult(self, result):
_activeRecog = None
# This might get called from a background thread, so any UI calls must be queued to the main thread.
if isinstance(result, Exception):
log.error("Recognition failed: %s" % result)
log.error(f"Recognition failed: {result}")
# Translators: Reported when recognition (e.g. OCR) fails.
queueHandler.queueFunction(
queueHandler.eventQueue, ui.message, _("Recognition failed")
Expand All @@ -88,11 +85,9 @@ def _onFirstResult(self, result):
self._scheduleRecognize()

def _scheduleRecognize(self):
core.callLater(self.REFRESH_INTERVAL, self._recognize, self._onResult)
core.callLater(config.conf['uwpOcr']['autoRefreshIntervalMs'], self._recognize, self._onResult)

def _onResult(self, result):
Comment thread
LeonarddeR marked this conversation as resolved.
Outdated
import tones # jtd
tones.beep(1660, 10) # jtd
if not self.hasFocus:
# The user has dismissed the recognition result.
return
Expand Down Expand Up @@ -127,14 +122,20 @@ def setFocus(self):

def event_gainFocus(self):
super().event_gainFocus()
if self.recognizer.allowAutoRefresh:
if config.conf['uwpOcr']['autoRefresh'] and self.recognizer.allowAutoRefresh:
# Make LiveText watch for and report new text.
self.startMonitoring()

def event_loseFocus(self):
# note: If monitoring has not been started, this will have no effect.
self.stopMonitoring()
super().event_loseFocus()
if self.recognizer.allowAutoRefresh:
self.stopMonitoring()

def start(self):
self._recognize(self._onFirstResult)

def _get_hasFocus(self) -> bool:
return self is api.getFocusObject()

def script_activatePosition(self, gesture):
try:
Expand Down Expand Up @@ -171,13 +172,15 @@ def script_findPrevious(self, gesture):
"kb:escape": "exit",
}


#: Keeps track of the recognition in progress, if any.
_activeRecog = None
def recognizeNavigatorObject(recognizer):


def recognizeNavigatorObject(recognizer: ContentRecognizer):
"""User interface function to recognize content in the navigator object.
This should be called from a script or in response to a GUI action.
@param recognizer: The content recognizer to use.
@type recognizer: L{contentRecog.ContentRecognizer}
"""
global _activeRecog
if isinstance(api.getFocusObject(), RecogResultNVDAObject):
Expand Down Expand Up @@ -208,5 +211,5 @@ def recognizeNavigatorObject(recognizer):
_activeRecog.recognizer.cancel()
# Translators: Reporting when content recognition (e.g. OCR) begins.
ui.message(_("Recognizing"))
_activeRecog = RecogResultNVDAObject(recognizer=recognizer, imageInfo=imgInfo)
_activeRecog = RecogResultNVDAObject(recognizer=recognizer, imageInfo=imgInfo, obj=nav)
_activeRecog.start()
Loading