Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/ARCHITECTURE.md
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ There is no frontend build system, external queue, cloud database, or required A

`ingestion.py` owns file decoding and structural validation.

It accepts CSV and XLSX input, normalizes the data into `CsvTable`, and rejects malformed files before a run is created. Upload reads are capped at the configured size limit plus one byte so oversized requests can be rejected early.
It accepts CSV and XLSX input, normalizes the data into `InputTable`, and rejects malformed files before a run is created. Upload reads are capped at the configured size limit plus one byte so oversized requests can be rejected early.

Staged upload filenames are generated by the application rather than taken from user-provided paths. Staging copies are removed after successful run creation, and abandoned stage directories expire automatically.

Expand Down
6 changes: 3 additions & 3 deletions src/rowbridge/candidates.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@
from rapidfuzz.fuzz import WRatio

from rowbridge.matching_utils import normalize_text, parse_amount, parse_date
from rowbridge.models import CandidateGeneration, CsvTable, FieldMapping, MatchSettings
from rowbridge.models import CandidateGeneration, FieldMapping, InputTable, MatchSettings


def _primary_block_keys(value: str) -> tuple[str, ...]:
Expand Down Expand Up @@ -55,8 +55,8 @@ def _add_to_index(index: dict[str, set[int]], keys: Iterable[str], row_index: in


def generate_candidates(
table_a: CsvTable,
table_b: CsvTable,
table_a: InputTable,
table_b: InputTable,
mapping: FieldMapping,
settings: MatchSettings,
) -> CandidateGeneration:
Expand Down
20 changes: 8 additions & 12 deletions src/rowbridge/ingestion.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@

from openpyxl import load_workbook

from rowbridge.models import CsvTable
from rowbridge.models import InputTable

MAX_UPLOAD_BYTES = 5 * 1024 * 1024
MAX_XLSX_UNCOMPRESSED_BYTES = 50 * 1024 * 1024
Expand All @@ -26,10 +26,6 @@ class InputFileError(ValueError):
pass


# Backward-compatible name for older imports while the input layer becomes format-neutral.
CsvInputError = InputFileError


def _validate_size(content: bytes, filename: str) -> None:
if not content:
raise InputFileError(f"{filename} is empty")
Expand Down Expand Up @@ -85,7 +81,7 @@ def _normalize_headers(values: list[str], filename: str) -> tuple[str, ...]:
return headers


def parse_csv_bytes(content: bytes, filename: str) -> CsvTable:
def parse_csv_bytes(content: bytes, filename: str) -> InputTable:
_validate_size(content, filename)
text, encoding = _decode_csv(content, filename)
delimiter = _detect_delimiter(text)
Expand All @@ -110,7 +106,7 @@ def parse_csv_bytes(content: bytes, filename: str) -> CsvTable:

if not rows:
raise InputFileError(f"{filename} has no data rows")
return CsvTable(
return InputTable(
filename=filename,
headers=headers,
rows=tuple(rows),
Expand Down Expand Up @@ -150,7 +146,7 @@ def _validate_xlsx_archive(content: bytes, filename: str) -> None:
raise InputFileError(f"{filename} is not a valid XLSX workbook") from exc


def parse_xlsx_bytes(content: bytes, filename: str) -> CsvTable:
def parse_xlsx_bytes(content: bytes, filename: str) -> InputTable:
_validate_size(content, filename)
_validate_xlsx_archive(content, filename)

Expand Down Expand Up @@ -197,7 +193,7 @@ def parse_xlsx_bytes(content: bytes, filename: str) -> CsvTable:

if not rows:
raise InputFileError(f"{filename} has no data rows")
return CsvTable(
return InputTable(
filename=filename,
headers=headers,
rows=tuple(rows),
Expand All @@ -208,7 +204,7 @@ def parse_xlsx_bytes(content: bytes, filename: str) -> CsvTable:
workbook.close()


def parse_input_bytes(content: bytes, filename: str) -> CsvTable:
def parse_input_bytes(content: bytes, filename: str) -> InputTable:
suffix = Path(filename).suffix.lower()
if suffix not in SUPPORTED_SUFFIXES:
raise InputFileError(f"{filename} is not supported. Use a .csv or .xlsx file.")
Expand All @@ -231,7 +227,7 @@ def write_staged_input(path: Path, content: bytes) -> None:
path.write_bytes(content)


def read_staged_input(path: Path, original_filename: str) -> CsvTable:
def read_staged_input(path: Path, original_filename: str) -> InputTable:
if not path.is_file():
raise InputFileError("Staged upload was not found. Upload the files again.")
return parse_input_bytes(path.read_bytes(), original_filename)
Expand Down Expand Up @@ -272,5 +268,5 @@ def cleanup_staged_uploads(
return removed


def preview_rows(table: CsvTable) -> tuple[dict[str, str], ...]:
def preview_rows(table: InputTable) -> tuple[dict[str, str], ...]:
return table.rows[:PREVIEW_ROWS]
14 changes: 7 additions & 7 deletions src/rowbridge/matching.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,9 +11,9 @@
CandidateGeneration,
CandidateScore,
ComparisonRule,
CsvTable,
Evidence,
FieldMapping,
InputTable,
MatchDecision,
MatchSettings,
MatchStatus,
Expand Down Expand Up @@ -175,8 +175,8 @@ def score_pair(


def _score_candidates(
table_a: CsvTable,
table_b: CsvTable,
table_a: InputTable,
table_b: InputTable,
mapping: FieldMapping,
settings: MatchSettings,
generation: CandidateGeneration,
Expand Down Expand Up @@ -236,8 +236,8 @@ def _is_ambiguous(


def reconcile_with_diagnostics(
table_a: CsvTable,
table_b: CsvTable,
table_a: InputTable,
table_b: InputTable,
mapping: FieldMapping,
settings: MatchSettings,
) -> tuple[tuple[MatchDecision, ...], CandidateGeneration]:
Expand Down Expand Up @@ -313,8 +313,8 @@ def reconcile_with_diagnostics(


def reconcile(
table_a: CsvTable,
table_b: CsvTable,
table_a: InputTable,
table_b: InputTable,
mapping: FieldMapping,
settings: MatchSettings,
) -> tuple[MatchDecision, ...]:
Expand Down
5 changes: 4 additions & 1 deletion src/rowbridge/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,7 @@ class RuleKind(StrEnum):


@dataclass(frozen=True, slots=True)
class CsvTable:
class InputTable:
filename: str
headers: tuple[str, ...]
rows: tuple[RowPayload, ...]
Expand All @@ -37,6 +37,9 @@ class CsvTable:
sheet_name: str | None = None


CsvTable = InputTable


@dataclass(frozen=True, slots=True)
class FieldMapping:
primary_a: str
Expand Down
10 changes: 5 additions & 5 deletions src/rowbridge/service.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,7 @@

from rowbridge.matching import reconcile
from rowbridge.matching_utils import parse_amount, parse_date
from rowbridge.models import CsvTable, FieldMapping, MatchSettings
from rowbridge.models import FieldMapping, InputTable, MatchSettings
from rowbridge.storage import Repository


Expand All @@ -30,7 +30,7 @@ def _validate_distinct_roles(mapping: FieldMapping) -> None:


def _validate_parseable_column(
table: CsvTable,
table: InputTable,
column: str,
parser: Callable[[str], object | None],
role: str,
Expand All @@ -48,7 +48,7 @@ def _validate_parseable_column(
)


def validate_mapping(table_a: CsvTable, table_b: CsvTable, mapping: FieldMapping) -> None:
def validate_mapping(table_a: InputTable, table_b: InputTable, mapping: FieldMapping) -> None:
selections = (
("primary_a", mapping.primary_a, table_a.headers),
("primary_b", mapping.primary_b, table_b.headers),
Expand Down Expand Up @@ -82,8 +82,8 @@ def validate_mapping(table_a: CsvTable, table_b: CsvTable, mapping: FieldMapping

def create_reconciliation_run(
repository: Repository,
table_a: CsvTable,
table_b: CsvTable,
table_a: InputTable,
table_b: InputTable,
mapping: FieldMapping,
settings: MatchSettings,
) -> str:
Expand Down
4 changes: 2 additions & 2 deletions src/rowbridge/web.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@
staged_filename,
write_staged_input,
)
from rowbridge.models import CsvTable, FieldMapping, MatchSettings, MatchStatus, RowPayload
from rowbridge.models import FieldMapping, InputTable, MatchSettings, MatchStatus, RowPayload
from rowbridge.service import create_reconciliation_run
from rowbridge.storage import Repository, StoredMatch

Expand Down Expand Up @@ -61,7 +61,7 @@ def _display_value(match: StoredMatch, side: str, column: str) -> str:
return _payload_value(payload, column)


def _table_details(table: CsvTable) -> str:
def _table_details(table: InputTable) -> str:
if table.source_format == "xlsx":
return f"Excel workbook · sheet {table.sheet_name or 'first data sheet'}"
delimiter_names = {",": "comma", ";": "semicolon", "\t": "tab", "|": "pipe"}
Expand Down
Loading
Loading