OCR: parse ad-hoc 'Enemy X Y' installations as a new target type
'Enemy Signal Station:' and its ilk are named in plain English rather
than the usual Type#N shape, and get referenced the same way
elsewhere ('Bearing 034 from Enemy Signal Station'), breaking two
assumptions everywhere else in this module: headers/references are a
single whitespace-free token, and an id is digit-shaped.
squash_enemy_names() collapses 'Enemy' + up to 4 Title Case words that
follow it into one token in our own id shape ('Enemy#SignalStation')
before anything else parses the text, so every existing from-<ref>
clue pattern and the named-header matcher keep working unmodified.
Wired into both parse_text() (OCR/clipboard) and
parse_clues_from_text() (manual description tab).
Two follow-on fixes this surfaced: _clean_reference() previously
assumed a named reference's id is always digit-shaped and would
truncate 'Enemy#SignalStation' down to 'Enemy#Sig' via
_fix_id_digits's letter-to-digit mapping; and the letter-id header
check needed to run *before* _NAMED_HEADER_RE, whose digit class
overlaps plain letters (S/B/Z/G/O/I/L) and would otherwise
partial-match and mangle the id first.
New TargetType.ENEMY carries these. Verified end to end (including
the solver resolving the cross-references between them) through both
parse_text() and the real _merge_targets() app flow.
This commit is contained in:
+65
-2
@@ -22,6 +22,7 @@ characters don't drop an otherwise-good line.
|
||||
from __future__ import annotations
|
||||
|
||||
import difflib
|
||||
import html
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
@@ -205,6 +206,31 @@ _RP_HEADER_RE = re.compile(r"Reference\s+Point\s+([A-Za-z][\w-]*)\s*:?", re.IGNO
|
||||
# every observed header: '#', a misread substitute, ...) rules that out,
|
||||
# and also stops a bare "Word 094" clue line (space only) from matching.
|
||||
_NAMED_HEADER_RE = re.compile(rf"^([A-Za-z]+)[^A-Za-z0-9\s]{{1,2}}({_DIGIT_CLASS}+)\s*:?\s*(.*)$")
|
||||
|
||||
# Ad-hoc enemy installations are named in plain English rather than given a
|
||||
# Type#N id ("Enemy Signal Station:", "Bearing 034 from Enemy Signal
|
||||
# Station"), which breaks two assumptions everywhere else in this module:
|
||||
# headers/references are always a single whitespace-free token, and an id
|
||||
# is always digit-shaped. squash_enemy_names() fixes both up front by
|
||||
# collapsing 'Enemy' + however many Title Case words follow it into one
|
||||
# token in our own 'Type#id' shape ('Enemy#SignalStation'), before
|
||||
# anything else tries to parse the text, so every from-<ref> clue pattern
|
||||
# (which just captures \S+) and _NAMED_HEADER_RE both work unmodified
|
||||
# except NAMED_HEADER_RE's id group is digit-only, so a second header
|
||||
# regex below handles the now-squashed, letter-only id.
|
||||
_ENEMY_NAME_RE = re.compile(r"\bEnemy(?:[ \t]+[A-Z][a-zA-Z]*){1,4}\b")
|
||||
_ENEMY_HEADER_RE = re.compile(r"^(Enemy)#([A-Za-z]+)\s*:?\s*(.*)$", re.IGNORECASE)
|
||||
|
||||
|
||||
def squash_enemy_names(text: str) -> str:
|
||||
"""'Enemy Signal Station' -> 'Enemy#SignalStation', anywhere it
|
||||
appears, header or reference alike. The word-separator inside a name
|
||||
is deliberately [ \\t]+, not \\s+: it must not cross a newline, or a
|
||||
header right at the end of a line ('...Enemy Signal Station') would
|
||||
swallow the next line's 'Bearing ...' clue into the same 'name'."""
|
||||
return _ENEMY_NAME_RE.sub(lambda m: "Enemy#" + "".join(m.group(0).split()[1:]), text)
|
||||
|
||||
|
||||
_BLOCK_END_RE = re.compile(r"^[.\s]{1,6}$")
|
||||
_LEADING_NOISE_RE = re.compile(r"^[^A-Za-z]{1,3}(?=[A-Za-z])")
|
||||
|
||||
@@ -244,10 +270,18 @@ def _clean_reference(raw: str) -> str:
|
||||
references ('Spotter#1', 'AmmoCache#2') get their digit part fixed up
|
||||
and their separator normalized to '#'; plain word references ('Alpha')
|
||||
are left untouched, don't run digit-fixing over them or real letters
|
||||
like the 'l' in 'Alpha' get corrupted into '1'."""
|
||||
like the 'l' in 'Alpha' get corrupted into '1'. 'Enemy#SignalStation'
|
||||
(squash_enemy_names() already normalized the separator, and its id is
|
||||
letters, not digits) is passed through as-is rather than falling into
|
||||
the digit-fixing path below, which would otherwise stop at the first
|
||||
letter outside _DIGIT_CLASS and truncate it (e.g. down to 'Enemy#Sig')."""
|
||||
token = re.match(r"\S+", raw.strip())
|
||||
token = token.group(0) if token else raw.strip()
|
||||
|
||||
enemy = _ENEMY_HEADER_RE.match(token)
|
||||
if enemy:
|
||||
return f"{enemy.group(1)}#{enemy.group(2)}"
|
||||
|
||||
named = _REF_NAMED_RE.match(token)
|
||||
if named:
|
||||
word, num = named.groups()
|
||||
@@ -296,7 +330,7 @@ def parse_clues_from_text(text: str) -> list[Clue]:
|
||||
"""Parse every Bearing/Distance clue found in free-form text, used
|
||||
for manually-typed descriptions in the coord dialog, sharing the
|
||||
exact same clue grammar as the OCR'd intel blocks."""
|
||||
return _parse_all_clues(text)
|
||||
return _parse_all_clues(squash_enemy_names(text))
|
||||
|
||||
|
||||
def _resolve_target_type(type_word: str) -> TargetType | None:
|
||||
@@ -372,6 +406,19 @@ def parse_intel_blocks(text: str) -> list[dict]:
|
||||
"raw": [line], "clues": []}
|
||||
continue
|
||||
|
||||
# Tried before _NAMED_HEADER_RE: its digit class overlaps plain
|
||||
# letters (S/B/Z/G/O/I/L), so on a squashed 'Enemy#SignalStation'
|
||||
# header it would otherwise partial-match ('Sig' -> digit-fixed
|
||||
# into the bogus id '516') before this more specific check ever
|
||||
# gets a look.
|
||||
enemy_m = next((m for c in candidates if (m := _ENEMY_HEADER_RE.match(c))), None)
|
||||
if enemy_m:
|
||||
flush()
|
||||
type_word, letter_id, inline = enemy_m.groups()
|
||||
current = {"kind": "named", "name": f"{type_word}#{letter_id}", "type_word": type_word,
|
||||
"id": letter_id, "raw": [line], "clues": []}
|
||||
continue
|
||||
|
||||
named_m = next((m for c in candidates if (m := _NAMED_HEADER_RE.match(c))), None)
|
||||
if named_m:
|
||||
flush()
|
||||
@@ -502,6 +549,21 @@ def parse_destroyed(text: str) -> set[tuple[TargetType, str]]:
|
||||
return destroyed
|
||||
|
||||
|
||||
_HTML_BREAK_RE = re.compile(r"(?i)<br\s*/?>|</(?:p|div|li|tr)>")
|
||||
_HTML_TAG_RE = re.compile(r"<[^>]+>")
|
||||
|
||||
|
||||
def strip_html(text: str) -> str:
|
||||
"""Drop markup from rich-text clipboard pastes (e.g. '<b>145</b>°'),
|
||||
leaving plain text the rest of the pipeline can parse. Block-ish
|
||||
closing tags become newlines first so pastes that rely on markup
|
||||
rather than real line breaks don't get run together; harmless no-op
|
||||
on plain OCR text with no '<' in it."""
|
||||
text = _HTML_BREAK_RE.sub("\n", text)
|
||||
text = _HTML_TAG_RE.sub("", text)
|
||||
return html.unescape(text)
|
||||
|
||||
|
||||
@dataclass
|
||||
class ParsedInfo:
|
||||
nest_coord: Coord | None = None
|
||||
@@ -519,6 +581,7 @@ class ParsedInfo:
|
||||
|
||||
|
||||
def parse_text(text: str) -> ParsedInfo:
|
||||
text = squash_enemy_names(strip_html(text))
|
||||
info = ParsedInfo()
|
||||
|
||||
for raw_line in text.splitlines():
|
||||
|
||||
Reference in New Issue
Block a user