Skip to content

Commit 7edb926

Browse files
committed
correct: Interpret mixed-case damage
Apply the established majority-case interpretation to standalone correction while preserving immutable context, entered edit semantics, and disclosure accounting. Account the normalized retry frontier even when the first optional search reaches its deadline after finding a candidate, so cumulative capture mass remains fail-closed. Fixes #37.
1 parent f19d462 commit 7edb926

10 files changed

Lines changed: 530 additions & 53 deletions

File tree

‎src/codex32/_cli_input.py‎

Lines changed: 11 additions & 15 deletions
Original file line numberDiff line numberDiff line change
@@ -6,10 +6,11 @@
66
import difflib
77
import os
88
import sys
9-
from collections.abc import Callable, Iterator
9+
from collections.abc import Callable, Iterator, Sequence
1010
from time import monotonic
1111
from typing import Any, Literal, cast
1212

13+
from codex32.bech32 import interpret_mixed_case
1314
from codex32.bip93 import (
1415
Secret,
1516
Share,
@@ -377,15 +378,14 @@ def _case_interpretation(
377378
profiles: tuple[Profile, ...] | None,
378379
allowed: Callable[[CorrectionCandidate], bool] | None,
379380
) -> tuple[CorrectionCandidate | None, str, str, str] | None:
380-
"""Normalize likely casing and mark contrary-case data as erasures."""
381-
if value.upper() == value or value.lower() == value:
382-
return None
381+
# Normalize likely casing and mark contrary-case data as erasures.
383382
separator = value.find("1")
384383
base_length = separator + 1 if separator >= 0 else 0
385384
immutable_length = len(prefix) if prefix and value.lower().startswith(prefix.lower()) else base_length
386-
letters = [character for character in value[immutable_length:] if character.lower() != character.upper()]
387-
uppercase = sum(character.isupper() for character in letters) > len(letters) / 2
388-
corrected = value.upper() if uppercase else value.lower()
385+
interpretation = interpret_mixed_case(value, immutable_length)
386+
if interpretation is None:
387+
return None
388+
corrected, erased, uppercase = interpretation
389389
corrected_prefix = prefix.upper() if uppercase else prefix.lower()
390390
try:
391391
artifact = _parse(corrected, profiles)
@@ -397,14 +397,6 @@ def _case_interpretation(
397397
)
398398
proposed = CorrectionCandidate(artifact, (), 1, 0, 0, None, capture_space_bits=bits)
399399
candidate = proposed if allowed is None or allowed(proposed) else None
400-
erased = "".join(
401-
corrected[index]
402-
if index < immutable_length
403-
or character.lower() == character.upper()
404-
or character.isupper() == uppercase
405-
else "?"
406-
for index, character in enumerate(value)
407-
)
408400
return candidate, corrected, erased, corrected_prefix
409401

410402

@@ -460,6 +452,8 @@ def _correction_candidates(
460452
deadline: float | None = None,
461453
capture_layers: list[tuple[int, int]] | None = None,
462454
fingerprint_match: Callable[[CorrectionCandidate], bool | None] | None = None,
455+
seed_candidates: Sequence[CorrectionCandidate] = (),
456+
required_only: bool = False,
463457
) -> tuple[tuple[CorrectionCandidate, ...], bool, float | None, bool]:
464458
count = len(value.replace(" ", ""))
465459
targets, primary, reduced, _timed = _correction_plan(profile, byte_length, count, target)
@@ -476,6 +470,8 @@ def _correction_candidates(
476470
competitors=True,
477471
allowed=allowed,
478472
capture_layers=capture_layers,
473+
seed_candidates=seed_candidates,
474+
required_only=required_only,
479475
)
480476
if allowed is not None:
481477
candidates = tuple(candidate for candidate in candidates if allowed(candidate))

‎src/codex32/_competitors.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -173,8 +173,10 @@ def _search_competitors(
173173
frontier: dict[_Layer, int],
174174
deadline: float,
175175
allowed: Callable[[CorrectionCandidate], bool] | None,
176+
*,
177+
seed_candidates: Sequence[CorrectionCandidate] = (),
176178
) -> tuple[tuple[CorrectionCandidate, ...], bool]:
177-
results: dict[str, CorrectionCandidate] = {}
179+
results = {candidate.artifact.text.lower(): candidate for candidate in seed_candidates}
178180
fixed: dict[int, CorrectionCandidate | None] = {}
179181
completed: set[_Layer] = set()
180182
try:

‎src/codex32/bech32.py‎

Lines changed: 16 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -53,6 +53,22 @@ def _validate_single_case_ascii(value: str) -> bool:
5353
return value.isupper()
5454

5555

56+
def interpret_mixed_case(value: str, immutable_length: int) -> tuple[str, str, bool] | None:
57+
# Return majority-cased and minority-erased interpretations of mixed-case text.
58+
if not value.isascii() or value.upper() == value or value.lower() == value:
59+
return None
60+
letters = [character for character in value[immutable_length:] if character.isalpha()]
61+
uppercase = sum(character.isupper() for character in letters) > len(letters) / 2
62+
normalized = value.upper() if uppercase else value.lower()
63+
erased = "".join(
64+
normalized[index]
65+
if index < immutable_length or not character.isalpha() or character.isupper() == uppercase
66+
else "?"
67+
for index, character in enumerate(value)
68+
)
69+
return normalized, erased, uppercase
70+
71+
5672
def bech32_encode(hrp: str, data: list[int], spec: _Checksum) -> str:
5773
"""Compute a Bech32 string given HRP and data values."""
5874
checksum = spec.create(bech32_hrp_expand(hrp) + list(data))

‎src/codex32/cli.py‎

Lines changed: 78 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -6,13 +6,15 @@
66
import json
77
import sys
88
from collections.abc import Callable, Sequence
9+
from dataclasses import replace
910
from typing import Literal, NamedTuple, cast
1011

1112
from codex32._bitcoin_core import BitcoinCore, BitcoinCoreError
1213
from codex32._cli_input import (
1314
CorrectionDeclined,
1415
InteractiveConfirmationRequired,
1516
_card_text,
17+
_case_interpretation,
1618
_confirm_correction,
1719
_correction_candidates,
1820
_entered_groups,
@@ -35,7 +37,13 @@
3537
parse_codex32,
3638
recover_secret,
3739
)
38-
from codex32.correction import _best, _residue_low_discrimination, correct_worksheet_residue
40+
from codex32.correction import (
41+
CorrectionCandidate,
42+
_best,
43+
_capture_mass,
44+
_residue_low_discrimination,
45+
correct_worksheet_residue,
46+
)
3947
from codex32.errors import CodexError, HeaderCollision, InvalidCorrectionInput
4048
from codex32.generation import (
4149
ConfirmationResult,
@@ -550,12 +558,75 @@ def _correct(
550558
raise _UsageError("--bytes does not match the valid master-seed backup length.")
551559
_print("The codex32 string is already valid.")
552560
return 0
553-
candidates, complete, _deadline, ambiguous = _correction_candidates(
554-
value,
555-
hrp,
556-
byte_length,
557-
value[: separator + 1],
558-
)
561+
search_value, erased, immutable = normalized, normalized, normalized[: separator + 1]
562+
interpreted = _case_interpretation(normalized, immutable, context.profiles, None)
563+
if interpreted is not None:
564+
candidate, search_value, erased, immutable = interpreted
565+
if (
566+
candidate is not None
567+
and isinstance(byte_length, int)
568+
and len(candidate.artifact.text) != _ms_text_length(byte_length)
569+
):
570+
raise _UsageError("--bytes does not match the corrected master-seed backup length.")
571+
else:
572+
candidate = None
573+
capture_layers: list[tuple[int, int]] = []
574+
if candidate is not None:
575+
candidates: tuple[CorrectionCandidate, ...] = (candidate,)
576+
complete, deadline, ambiguous = True, None, False
577+
else:
578+
first_search = erased if erased != search_value else search_value
579+
retry_search = search_value if erased != search_value else None
580+
seeded: tuple[CorrectionCandidate, ...] = ()
581+
deadline = None
582+
if retry_search is not None:
583+
# Discover required candidates for both case interpretations before
584+
# either full search can spend the shared deadline on optional
585+
# alignment work. Full searches below own capture accounting.
586+
for required_value in (first_search, retry_search):
587+
required, required_complete, deadline, _ = _correction_candidates(
588+
required_value,
589+
hrp,
590+
byte_length,
591+
immutable,
592+
deadline=deadline,
593+
seed_candidates=seeded,
594+
required_only=True,
595+
)
596+
if not required_complete:
597+
raise _CommandError("The correction search did not complete within ten seconds.")
598+
seeded = required
599+
candidates, complete, deadline, ambiguous = _correction_candidates(
600+
first_search,
601+
hrp,
602+
byte_length,
603+
immutable,
604+
deadline=deadline,
605+
capture_layers=capture_layers,
606+
seed_candidates=seeded,
607+
)
608+
if retry_search is not None:
609+
retry_candidates, complete, deadline, retry_ambiguous = _correction_candidates(
610+
retry_search,
611+
hrp,
612+
byte_length,
613+
immutable,
614+
deadline=deadline,
615+
capture_layers=capture_layers,
616+
seed_candidates=(*seeded, *candidates),
617+
)
618+
ambiguous = ambiguous or retry_ambiguous
619+
combined = (*candidates, *retry_candidates)
620+
if combined:
621+
annotated = []
622+
for item in combined:
623+
volume, bits = _capture_mass(capture_layers, item.capture_volume)
624+
annotated.append(replace(item, cumulative_capture_volume=volume, capture_space_bits=bits))
625+
ranked = _best(annotated, prefer_common=byte_length == "?")
626+
unique: dict[str, CorrectionCandidate] = {}
627+
for item in ranked:
628+
unique.setdefault(item.artifact.text.lower(), item)
629+
candidates = tuple(unique.values())
559630
if not complete and not candidates:
560631
raise _CommandError("The correction search did not complete within ten seconds.")
561632
if ambiguous:

‎src/codex32/correction.py‎

Lines changed: 97 additions & 22 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,7 @@
3434
_u5_to_chars,
3535
_validate_single_case_ascii,
3636
bech32_hrp_expand,
37+
interpret_mixed_case,
3738
)
3839
from codex32.bip93 import (
3940
IDX_SORT,
@@ -839,6 +840,17 @@ def _primary(
839840
)
840841

841842

843+
def _restore_case_edits(candidates: tuple[CorrectionCandidate, ...]) -> tuple[CorrectionCandidate, ...]:
844+
# Hide erasures synthesized only to search minority-case symbols.
845+
def restore(edit: CorrectionEdit) -> CorrectionEdit:
846+
kind = "substitution" if edit.kind == "erasure" and edit.observed.lower() in CHARSET else edit.kind
847+
return replace(edit, kind=kind) if kind != edit.kind else edit
848+
849+
return tuple(
850+
replace(candidate, edits=tuple(restore(edit) for edit in candidate.edits)) for candidate in candidates
851+
)
852+
853+
842854
def _best(
843855
candidates: Sequence[CorrectionCandidate],
844856
*,
@@ -874,30 +886,93 @@ def _correct_complete(
874886
# displayed strings are no longer than the largest expanded codeword.
875887
if len(damaged_text) > 2 * (_LONG_SPEC.period + 8):
876888
return (), True
877-
from codex32.indel import _search_many
878-
879889
deadline = monotonic() + 10 if deadline is None else deadline
880-
contexts: tuple[CorrectionContext, ...]
881-
if context.expected_length is not None:
882-
contexts = (context,)
890+
base = f"{context.hrp}1"
891+
locked = context.immutable_prefix or base
892+
immutable_length = len(locked) if damaged_text.lower().startswith(locked.lower()) else len(base)
893+
interpretation = interpret_mixed_case(damaged_text, immutable_length)
894+
inputs: tuple[tuple[CorrectionContext, str], ...]
895+
if interpretation is None:
896+
inputs = ((context, damaged_text),)
883897
else:
884-
# Only lengths reachable by either disjoint family are eligible.
885-
observed = len(damaged_text.replace(" ", ""))
886-
contexts_list = []
887-
for target in sorted({observed + delta for delta in (*range(-4, 5), -8, 8)}):
888-
candidate_context = replace(context, expected_length=target)
889-
try:
890-
_validate_context(candidate_context)
891-
except InvalidCorrectionInput:
892-
continue
893-
contexts_list.append(candidate_context)
894-
contexts = tuple(contexts_list)
895-
return _search_many(
896-
contexts,
897-
damaged_text,
898-
primary=frozenset(c.expected_length for c in contexts if c.expected_length is not None),
899-
deadline=deadline,
900-
)
898+
normalized, erased, uppercase = interpretation
899+
normalized_prefix = locked.upper() if uppercase else locked.lower()
900+
normalized_context = replace(
901+
context, immutable_prefix=normalized_prefix if context.immutable_prefix is not None else None
902+
)
903+
# Minority-case symbols are explicit erasures, so search that stronger
904+
# interpretation before optional alignment work on the normalized text
905+
# can consume the shared correction deadline.
906+
inputs = ((normalized_context, erased), (normalized_context, normalized))
907+
908+
from codex32.indel import _search_many
909+
910+
capture_layers: list[tuple[int, int]] = []
911+
candidates: tuple[CorrectionCandidate, ...] = ()
912+
complete = True
913+
if interpretation is not None:
914+
# Establish both interpretations' fixed/required candidates before
915+
# either interpretation can spend the shared deadline on optional
916+
# alignment work. These discovery passes use a private accounting
917+
# ledger; the full searches below account every admitted layer once.
918+
for input_context, value in inputs:
919+
preflight_contexts: tuple[CorrectionContext, ...]
920+
if input_context.expected_length is not None:
921+
preflight_contexts = (input_context,)
922+
else:
923+
observed = len(value.replace(" ", ""))
924+
contexts_list = []
925+
for target in sorted({observed + delta for delta in (*range(-4, 5), -8, 8)}):
926+
candidate_context = replace(input_context, expected_length=target)
927+
try:
928+
_validate_context(candidate_context)
929+
except InvalidCorrectionInput:
930+
continue
931+
contexts_list.append(candidate_context)
932+
preflight_contexts = tuple(contexts_list)
933+
candidates, current_complete = _search_many(
934+
preflight_contexts,
935+
value,
936+
primary=frozenset(
937+
c.expected_length for c in preflight_contexts if c.expected_length is not None
938+
),
939+
deadline=deadline,
940+
observed_text=damaged_text,
941+
seed_candidates=candidates,
942+
required_only=True,
943+
)
944+
if not current_complete:
945+
return (), False
946+
for input_context, value in inputs:
947+
contexts: tuple[CorrectionContext, ...]
948+
if input_context.expected_length is not None:
949+
contexts = (input_context,)
950+
else:
951+
# Only lengths reachable by either disjoint family are eligible.
952+
observed = len(value.replace(" ", ""))
953+
contexts_list = []
954+
for target in sorted({observed + delta for delta in (*range(-4, 5), -8, 8)}):
955+
candidate_context = replace(input_context, expected_length=target)
956+
try:
957+
_validate_context(candidate_context)
958+
except InvalidCorrectionInput:
959+
continue
960+
contexts_list.append(candidate_context)
961+
contexts = tuple(contexts_list)
962+
candidates, current_complete = _search_many(
963+
contexts,
964+
value,
965+
primary=frozenset(c.expected_length for c in contexts if c.expected_length is not None),
966+
deadline=deadline,
967+
capture_layers=capture_layers,
968+
observed_text=damaged_text,
969+
seed_candidates=candidates,
970+
)
971+
complete &= current_complete
972+
if not current_complete and not candidates:
973+
return (), False
974+
candidates = _restore_case_edits(candidates) if interpretation is not None else candidates
975+
return candidates, complete
901976

902977

903978
def correct(context: CorrectionContext, damaged_text: str) -> tuple[CorrectionCandidate, ...]:

0 commit comments

Comments
 (0)