|
34 | 34 | _u5_to_chars, |
35 | 35 | _validate_single_case_ascii, |
36 | 36 | bech32_hrp_expand, |
| 37 | + interpret_mixed_case, |
37 | 38 | ) |
38 | 39 | from codex32.bip93 import ( |
39 | 40 | IDX_SORT, |
@@ -839,6 +840,17 @@ def _primary( |
839 | 840 | ) |
840 | 841 |
|
841 | 842 |
|
| 843 | +def _restore_case_edits(candidates: tuple[CorrectionCandidate, ...]) -> tuple[CorrectionCandidate, ...]: |
| 844 | + # Hide erasures synthesized only to search minority-case symbols. |
| 845 | + def restore(edit: CorrectionEdit) -> CorrectionEdit: |
| 846 | + kind = "substitution" if edit.kind == "erasure" and edit.observed.lower() in CHARSET else edit.kind |
| 847 | + return replace(edit, kind=kind) if kind != edit.kind else edit |
| 848 | + |
| 849 | + return tuple( |
| 850 | + replace(candidate, edits=tuple(restore(edit) for edit in candidate.edits)) for candidate in candidates |
| 851 | + ) |
| 852 | + |
| 853 | + |
842 | 854 | def _best( |
843 | 855 | candidates: Sequence[CorrectionCandidate], |
844 | 856 | *, |
@@ -874,30 +886,93 @@ def _correct_complete( |
874 | 886 | # displayed strings are no longer than the largest expanded codeword. |
875 | 887 | if len(damaged_text) > 2 * (_LONG_SPEC.period + 8): |
876 | 888 | return (), True |
877 | | - from codex32.indel import _search_many |
878 | | - |
879 | 889 | deadline = monotonic() + 10 if deadline is None else deadline |
880 | | - contexts: tuple[CorrectionContext, ...] |
881 | | - if context.expected_length is not None: |
882 | | - contexts = (context,) |
| 890 | + base = f"{context.hrp}1" |
| 891 | + locked = context.immutable_prefix or base |
| 892 | + immutable_length = len(locked) if damaged_text.lower().startswith(locked.lower()) else len(base) |
| 893 | + interpretation = interpret_mixed_case(damaged_text, immutable_length) |
| 894 | + inputs: tuple[tuple[CorrectionContext, str], ...] |
| 895 | + if interpretation is None: |
| 896 | + inputs = ((context, damaged_text),) |
883 | 897 | else: |
884 | | - # Only lengths reachable by either disjoint family are eligible. |
885 | | - observed = len(damaged_text.replace(" ", "")) |
886 | | - contexts_list = [] |
887 | | - for target in sorted({observed + delta for delta in (*range(-4, 5), -8, 8)}): |
888 | | - candidate_context = replace(context, expected_length=target) |
889 | | - try: |
890 | | - _validate_context(candidate_context) |
891 | | - except InvalidCorrectionInput: |
892 | | - continue |
893 | | - contexts_list.append(candidate_context) |
894 | | - contexts = tuple(contexts_list) |
895 | | - return _search_many( |
896 | | - contexts, |
897 | | - damaged_text, |
898 | | - primary=frozenset(c.expected_length for c in contexts if c.expected_length is not None), |
899 | | - deadline=deadline, |
900 | | - ) |
| 898 | + normalized, erased, uppercase = interpretation |
| 899 | + normalized_prefix = locked.upper() if uppercase else locked.lower() |
| 900 | + normalized_context = replace( |
| 901 | + context, immutable_prefix=normalized_prefix if context.immutable_prefix is not None else None |
| 902 | + ) |
| 903 | + # Minority-case symbols are explicit erasures, so search that stronger |
| 904 | + # interpretation before optional alignment work on the normalized text |
| 905 | + # can consume the shared correction deadline. |
| 906 | + inputs = ((normalized_context, erased), (normalized_context, normalized)) |
| 907 | + |
| 908 | + from codex32.indel import _search_many |
| 909 | + |
| 910 | + capture_layers: list[tuple[int, int]] = [] |
| 911 | + candidates: tuple[CorrectionCandidate, ...] = () |
| 912 | + complete = True |
| 913 | + if interpretation is not None: |
| 914 | + # Establish both interpretations' fixed/required candidates before |
| 915 | + # either interpretation can spend the shared deadline on optional |
| 916 | + # alignment work. These discovery passes use a private accounting |
| 917 | + # ledger; the full searches below account every admitted layer once. |
| 918 | + for input_context, value in inputs: |
| 919 | + preflight_contexts: tuple[CorrectionContext, ...] |
| 920 | + if input_context.expected_length is not None: |
| 921 | + preflight_contexts = (input_context,) |
| 922 | + else: |
| 923 | + observed = len(value.replace(" ", "")) |
| 924 | + contexts_list = [] |
| 925 | + for target in sorted({observed + delta for delta in (*range(-4, 5), -8, 8)}): |
| 926 | + candidate_context = replace(input_context, expected_length=target) |
| 927 | + try: |
| 928 | + _validate_context(candidate_context) |
| 929 | + except InvalidCorrectionInput: |
| 930 | + continue |
| 931 | + contexts_list.append(candidate_context) |
| 932 | + preflight_contexts = tuple(contexts_list) |
| 933 | + candidates, current_complete = _search_many( |
| 934 | + preflight_contexts, |
| 935 | + value, |
| 936 | + primary=frozenset( |
| 937 | + c.expected_length for c in preflight_contexts if c.expected_length is not None |
| 938 | + ), |
| 939 | + deadline=deadline, |
| 940 | + observed_text=damaged_text, |
| 941 | + seed_candidates=candidates, |
| 942 | + required_only=True, |
| 943 | + ) |
| 944 | + if not current_complete: |
| 945 | + return (), False |
| 946 | + for input_context, value in inputs: |
| 947 | + contexts: tuple[CorrectionContext, ...] |
| 948 | + if input_context.expected_length is not None: |
| 949 | + contexts = (input_context,) |
| 950 | + else: |
| 951 | + # Only lengths reachable by either disjoint family are eligible. |
| 952 | + observed = len(value.replace(" ", "")) |
| 953 | + contexts_list = [] |
| 954 | + for target in sorted({observed + delta for delta in (*range(-4, 5), -8, 8)}): |
| 955 | + candidate_context = replace(input_context, expected_length=target) |
| 956 | + try: |
| 957 | + _validate_context(candidate_context) |
| 958 | + except InvalidCorrectionInput: |
| 959 | + continue |
| 960 | + contexts_list.append(candidate_context) |
| 961 | + contexts = tuple(contexts_list) |
| 962 | + candidates, current_complete = _search_many( |
| 963 | + contexts, |
| 964 | + value, |
| 965 | + primary=frozenset(c.expected_length for c in contexts if c.expected_length is not None), |
| 966 | + deadline=deadline, |
| 967 | + capture_layers=capture_layers, |
| 968 | + observed_text=damaged_text, |
| 969 | + seed_candidates=candidates, |
| 970 | + ) |
| 971 | + complete &= current_complete |
| 972 | + if not current_complete and not candidates: |
| 973 | + return (), False |
| 974 | + candidates = _restore_case_edits(candidates) if interpretation is not None else candidates |
| 975 | + return candidates, complete |
901 | 976 |
|
902 | 977 |
|
903 | 978 | def correct(context: CorrectionContext, damaged_text: str) -> tuple[CorrectionCandidate, ...]: |
|
0 commit comments