OCR: parse fire-support requests ('Marine Garrison#1 pinned!')

New shape: '<Name>#<id> pinned!' followed by '<Shell> Shells requested
on <coord>' and 'Requested before - <T-time> -'. Different from every
existing coord shape (no 'Grid' keyword), so it needed its own
extractor (_extract_requested_on_coord), plus new ones for the shell
code and the deadline string. New TargetType.MARINE_GARRISON, its
multi-word name + real digit id already works through the existing
squash_multiword_ids() pipeline (same as Coastal Battery/Listening
Post) with no new header regex needed.

Target gained a requested_time field (raw string, this app doesn't
track a game clock to compare it against), persisted through
save/load and shown on the map below the coord label. The requested
shell sets target.shell directly rather than staying a suggestion,
matching how a manually-picked shell already works.

info.targets' value tuple grew from 3 to 5 elements (raw, clues,
coord, shell, requested_time); updated both call sites in app.py
that unpack it. Verified end to end: parse -> merge -> save/load
round trip -> map draw.
This commit is contained in:
2026-08-09 16:20:55 +02:00
parent eedea3ea19
commit 202219abf2
4 changed files with 96 additions and 14 deletions
+63 -9
View File
@@ -31,6 +31,7 @@ import pytesseract
from PIL import Image, ImageFilter
from .models import Clue, Coord, TargetType
from .shells import Shell
# --- preprocessing -----------------------------------------------------------
@@ -152,6 +153,47 @@ def _extract_grid_coord(text: str) -> Coord | None:
return None
# A fire-support request gives its coord differently again: "<Shell>
# Shells requested on <coord>", no "Grid" keyword. Same shape otherwise.
_REQUESTED_ON_COORD_RE = re.compile(
rf"requested\s+on\s+([A-T])\s*({_DIGIT_CLASS}{{1,2}})\s+({_DIGIT_CLASS})\s*[:;.,]\s*({_DIGIT_CLASS})",
re.IGNORECASE,
)
# "SMK Shells requested on ..." names the shell by its short code, one of
# the Shell enum's own member names.
_SHELL_REQUEST_RE = re.compile(r"([A-Za-z]+)\s+Shells?\s+requested", re.IGNORECASE)
# "Requested before - T10:31:41 -": an in-game clock deadline, kept as
# the raw string, this app doesn't track a game clock to compare it
# against.
_REQUESTED_BEFORE_RE = re.compile(r"Requested\s+before\s*-\s*(T?\d{1,2}:\d{2}:\d{2})\s*-", re.IGNORECASE)
def _extract_requested_on_coord(text: str) -> Coord | None:
m = _REQUESTED_ON_COORD_RE.search(text)
if not m:
return None
letter, y, x, yy = m.groups()
try:
return Coord(X=letter.upper(), Y=int(_fix_digits(y)), x=int(_fix_digits(x)), y=int(_fix_digits(yy)))
except ValueError:
return None
def _extract_shell_request(text: str) -> Shell | None:
m = _SHELL_REQUEST_RE.search(text)
if not m:
return None
try:
return Shell[m.group(1).upper()]
except KeyError:
return None
def _extract_requested_time(text: str) -> str | None:
m = _REQUESTED_BEFORE_RE.search(text)
return m.group(1) if m else None
def _extract_coord(line: str) -> Coord | None:
m = _COORD_RE.search(line)
if not m:
@@ -422,8 +464,9 @@ def _resolve_target_type(type_word: str) -> TargetType | None:
def parse_intel_blocks(text: str) -> list[dict]:
"""Parse 'field intelligence' blocks into a list of dicts with keys
kind ('rp' | 'named'), name, type_word, id, raw, clues, coord. Entries
with neither a clue nor a grid coord are dropped (nothing to store).
kind ('rp' | 'named'), name, type_word, id, raw, clues, coord, shell,
requested_time. Entries with none of clue/coord/shell/requested_time
are dropped (nothing to store).
Clues are extracted once per block, from the whole joined block text,
at flush time, not accumulated line-by-line while scanning. That's
@@ -431,7 +474,9 @@ def parse_intel_blocks(text: str) -> list[dict]:
"from Spotter#1" on separate lines) or two clues on one line
("Bearing X from A & Bearing Y from B") both resolve correctly. A
target can also carry an absolute grid ref directly ("Grid Q3 9:0")
instead of/alongside clues."""
instead of/alongside clues, or a fire-support request's own coord
shape ("SMK Shells requested on J6 8:3"), plus that request's shell
and deadline if given."""
entries: list[dict] = []
current: dict | None = None
@@ -440,8 +485,11 @@ def parse_intel_blocks(text: str) -> list[dict]:
if current is not None:
joined = "\n".join(current["raw"])
current["clues"] = _parse_all_clues(joined)
current["coord"] = _extract_grid_coord(joined)
if current["clues"] or current["coord"] is not None:
current["coord"] = _extract_grid_coord(joined) or _extract_requested_on_coord(joined)
current["shell"] = _extract_shell_request(joined)
current["requested_time"] = _extract_requested_time(joined)
if (current["clues"] or current["coord"] is not None
or current["shell"] is not None or current["requested_time"] is not None):
current["raw"] = joined
entries.append(current)
current = None
@@ -690,8 +738,12 @@ class ParsedInfo:
spotters: dict[int, Coord] = field(default_factory=dict)
# name -> (raw description, clues, absolute coord if given directly)
reference_points: dict[str, tuple[str, list[Clue], Coord | None]] = field(default_factory=dict)
# (type, id) -> (raw description, clues, absolute coord if given directly)
targets: dict[tuple[TargetType, str], tuple[str, list[Clue], Coord | None]] = field(default_factory=dict)
# (type, id) -> (raw description, clues, absolute coord if given
# directly, requested shell if a fire-support request named one,
# requested-before deadline string if given)
targets: dict[
tuple[TargetType, str], tuple[str, list[Clue], Coord | None, Shell | None, str | None]
] = field(default_factory=dict)
# (type, id) of targets reported destroyed
destroyed: set[tuple[TargetType, str]] = field(default_factory=set)
@@ -730,7 +782,7 @@ def parse_text(text: str) -> ParsedInfo:
if (TargetType.UNKNOWN, "1") not in info.targets and _fuzzy_contains(line, TARGET_IS_AT_KEYWORD):
coord = _extract_coord(line)
if coord is not None:
info.targets[(TargetType.UNKNOWN, "1")] = (line, [], coord)
info.targets[(TargetType.UNKNOWN, "1")] = (line, [], coord, None, None)
continue
for entry in parse_intel_blocks(text):
@@ -740,7 +792,9 @@ def parse_text(text: str) -> ParsedInfo:
target_type = _resolve_target_type(entry["type_word"])
if target_type is None:
continue
info.targets[(target_type, entry["id"])] = (entry["raw"], entry["clues"], entry["coord"])
info.targets[(target_type, entry["id"])] = (
entry["raw"], entry["clues"], entry["coord"], entry["shell"], entry["requested_time"]
)
for entry in parse_train_intel(text):
info.reference_points[entry["name"]] = (entry["raw"], entry["clues"], entry["coord"])