Files
mywhoosh2garmin/tests/fit/test_rewriter_preservation.py
Bastian Wagner ef9ca9eeca fix: harden FIT patcher error boundary and verify patched metadata
Final-review fixes for Plan 2 (fit-rewriter). Every failure mode below now
surfaces as FitFormatError so Plan 3 can classify invalid FIT input as a
non-retryable activity error (spec 10.4).

- Range-check numeric values against the field's declared size before
  struct.pack, so an oversized serial number or a 1-byte product field
  raises FitFormatError instead of leaking a raw struct.error.
- Reject zero-size field definitions during parsing. A zero-size
  device_info field 0 read back as device_index == 0 via
  int.from_bytes(b"", ...), which could have let a paired sensor be
  rewritten as an Edge 1030 Plus (spec 10.2).
- Add DeviceFieldValue.is_creator so callers can tell the creator
  device_info record from sensor records instead of silently keeping
  whichever record appeared last.
- Implement the missing spec 10.4 post-patch step: read the patched
  buffer back and verify file_id 1/2/8 and creator device_info 2/4/27
  hold the target values. A field that could not be written (e.g. a
  product_name field too small for the target string) now fails the whole
  conversion rather than producing a silent partial patch. Verification
  runs before the output is written, so a half-rewritten file never lands
  on disk.
- Use the field's actual endianness in _read_field_value's fallback path.
- Add curated re-exports in app/fit/__init__.py for Plan 3.
- Document _iter_data_fields' caller invariant (validate the container
  first; end_offset is not clamped).
- Extend the preservation fixture with a product_name string field so the
  zero-filling string write path is covered by the byte-preservation
  proof, and test convert_fit_device against a 12-byte header.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-15 14:29:40 +02:00

262 lines
11 KiB
Python

import struct
from dataclasses import dataclass
from pathlib import Path
import pytest
from app.fit.rewriter import FitFormatError, convert_fit_device, is_fit_file, read_device_field_values
from tests.fit.builders import compressed_timestamp_data, data, definition, make_fit
FILE_ID_MESG_NUM = 0
DEVICE_INFO_MESG_NUM = 23
RECORD_MESG_NUM = 20
HEADER_SIZE = 14
PRODUCT_NAME_SIZE = 24
@dataclass(frozen=True)
class ComplexFixture:
fit_bytes: bytes
metadata_offsets: set[int]
file_id_manufacturer_offset: int
file_id_product_offset: int
file_id_product_name_offset: int
file_id_product_name_size: int
creator_manufacturer_offset: int
creator_product_offset: int
preserved_ranges: tuple[tuple[int, int], ...]
def _build_complex_fixture() -> ComplexFixture:
"""Builds one synthetic FIT data section covering all of the advanced record
shapes the parser must handle end-to-end through convert_fit_device():
1. a normal file_id definition/data pair (local 0);
2. a device_info definition with a creator record (local 1, device_index=0);
3. a record definition with one native field and one developer field, plus one
data record using it (local 2);
4. a compressed-timestamp data header referring to that same local 2 definition;
5. a later replacement definition for local message number 2 (redefined as a
non-creator device_info record, device_index=2), plus a data record using it.
All byte offsets are derived here from the construction arithmetic itself
(running length of the accumulated byte stream) -- this function never calls
into app.fit.rewriter's parser.
"""
records = bytearray()
def pos() -> int:
return HEADER_SIZE + len(records)
# 1. normal file_id definition/data pair, including the product_name string
# field (8) so the string write path -- which zero-fills the whole declared
# field, the riskiest byte-preservation behavior in the patcher -- is covered
# by the byte-preservation proof and not only by the patching tests.
records.extend(
definition(
0,
FILE_ID_MESG_NUM,
[(1, 2, 0x84), (2, 2, 0x84), (8, PRODUCT_NAME_SIZE, 0x07)],
)
)
file_id_data_start = pos()
original_product_name = b"MyWhoosh Simulator\x00".ljust(PRODUCT_NAME_SIZE, b"\x2A")
records.extend(data(0, struct.pack("<HH", 255, 999) + original_product_name))
file_id_manufacturer_offset = file_id_data_start + 1 # +1 for the record header byte
file_id_product_offset = file_id_manufacturer_offset + 2 # manufacturer is a u16
file_id_product_name_offset = file_id_product_offset + 2 # product is a u16
# 2. device_info definition with a creator record (device_index == 0).
records.extend(definition(1, DEVICE_INFO_MESG_NUM, [(0, 1, 0x02), (2, 2, 0x84), (4, 2, 0x84)]))
creator_data_start = pos()
records.extend(data(1, struct.pack("<BHH", 0, 255, 999)))
creator_manufacturer_offset = creator_data_start + 1 + 1 # header byte + device_index(u8)
creator_product_offset = creator_manufacturer_offset + 2 # manufacturer is a u16
# 3. record definition with one native field (power) and one developer field,
# plus one data record using it.
records.extend(definition(2, RECORD_MESG_NUM, [(7, 2, 0x84)], developer_fields=[(0, 4, 0)]))
record_data_start = pos()
records.extend(data(2, struct.pack("<H", 1234) + b"\xDE\xAD\xBE\xEF"))
record_data_end = pos()
# 4. compressed-timestamp data header referring to the still-active local
# definition 2 (the record definition above). Same field layout, so the
# payload is a power u16 followed by the 4-byte developer field.
compressed_start = pos()
records.extend(compressed_timestamp_data(2, 17, struct.pack("<H", 5678) + b"\xCA\xFE\xBA\xBE"))
compressed_end = pos()
# 5. later replacement definition for the same local message number (2): local 2
# is redefined as a *non-creator* device_info record (device_index=2), so the
# patcher must skip it entirely and its bytes must stay identical.
records.extend(definition(2, DEVICE_INFO_MESG_NUM, [(0, 1, 0x02), (2, 2, 0x84), (4, 2, 0x84)]))
sensor_data_start = pos()
records.extend(data(2, struct.pack("<BHH", 2, 77, 555)))
sensor_data_end = pos()
metadata_offsets: set[int] = set()
metadata_offsets.update(range(file_id_manufacturer_offset, file_id_manufacturer_offset + 2))
metadata_offsets.update(range(file_id_product_offset, file_id_product_offset + 2))
metadata_offsets.update(
range(file_id_product_name_offset, file_id_product_name_offset + PRODUCT_NAME_SIZE)
)
metadata_offsets.update(range(creator_manufacturer_offset, creator_manufacturer_offset + 2))
metadata_offsets.update(range(creator_product_offset, creator_product_offset + 2))
return ComplexFixture(
fit_bytes=make_fit(bytes(records)),
metadata_offsets=metadata_offsets,
file_id_manufacturer_offset=file_id_manufacturer_offset,
file_id_product_offset=file_id_product_offset,
file_id_product_name_offset=file_id_product_name_offset,
file_id_product_name_size=PRODUCT_NAME_SIZE,
creator_manufacturer_offset=creator_manufacturer_offset,
creator_product_offset=creator_product_offset,
preserved_ranges=(
(record_data_start, record_data_end),
(compressed_start, compressed_end),
(sensor_data_start, sensor_data_end),
),
)
# Built once at import time so the fixture bytes and the expected-offsets bookkeeping
# can never drift apart from each other.
_FIXTURE = _build_complex_fixture()
@pytest.fixture
def complex_fit_bytes() -> bytes:
return _FIXTURE.fit_bytes
def find_expected_device_metadata_offsets(before: bytes) -> set[int]:
"""Byte offsets convert_fit_device() is expected to touch, taken from the
fixture's own construction-time bookkeeping (see _build_complex_fixture above).
Deliberately does not parse `before` or call any part of app.fit.rewriter, so
this test cannot become circular (the parser being tested against itself)."""
del before
return set(_FIXTURE.metadata_offsets)
def test_only_target_fields_and_crcs_change(tmp_path: Path, complex_fit_bytes: bytes) -> None:
source = tmp_path / "source.fit"
output = tmp_path / "output.fit"
source.write_bytes(complex_fit_bytes)
convert_fit_device(source, output)
before = source.read_bytes()
after = output.read_bytes()
assert len(before) == len(after)
changed = {index for index, (a, b) in enumerate(zip(before, after)) if a != b}
expected_metadata_offsets = set(find_expected_device_metadata_offsets(before))
crc_offsets = {12, 13, len(before) - 2, len(before) - 1}
assert changed <= expected_metadata_offsets | crc_offsets
def test_complex_fixture_patches_targets_and_preserves_advanced_records(
tmp_path: Path, complex_fit_bytes: bytes
) -> None:
source = tmp_path / "source.fit"
output = tmp_path / "output.fit"
source.write_bytes(complex_fit_bytes)
result = convert_fit_device(source, output)
assert is_fit_file(output) is True
assert result.patched_field_count == 5
after = output.read_bytes()
assert struct.unpack_from("<H", after, _FIXTURE.file_id_manufacturer_offset)[0] == 1
assert struct.unpack_from("<H", after, _FIXTURE.file_id_product_offset)[0] == 3570
assert struct.unpack_from("<H", after, _FIXTURE.creator_manufacturer_offset)[0] == 1
assert struct.unpack_from("<H", after, _FIXTURE.creator_product_offset)[0] == 3570
# The record message (developer field), the compressed-timestamp record reusing
# its definition, and the redefined local-2 sensor device_info record are all
# untouched by patching -- verify their raw payload bytes are byte-identical.
for start, end in _FIXTURE.preserved_ranges:
assert complex_fit_bytes[start:end] == after[start:end]
values = read_device_field_values(output)
assert any(
v.global_message_num == FILE_ID_MESG_NUM and v.field_num == 2 and v.value == 3570 for v in values
)
def test_product_name_string_write_stays_inside_its_declared_field(
tmp_path: Path, complex_fit_bytes: bytes
) -> None:
"""The string write path zero-fills the *entire* declared field. Prove that the
rewrite is confined to the product_name field's own bytes: the target string plus
a null terminator plus zero padding, with the surrounding record bytes untouched
(the enclosing preservation test already asserts the global changed-byte set)."""
source = tmp_path / "source.fit"
output = tmp_path / "output.fit"
source.write_bytes(complex_fit_bytes)
convert_fit_device(source, output)
start = _FIXTURE.file_id_product_name_offset
end = start + _FIXTURE.file_id_product_name_size
after = output.read_bytes()
expected = b"Edge 1030 Plus\x00".ljust(_FIXTURE.file_id_product_name_size, b"\x00")
assert after[start:end] == expected
# The source deliberately padded past its null terminator with 0x2A bytes, so a
# write that overran (or under-cleared) the field would be visible here.
assert complex_fit_bytes[start:end] != expected
values = read_device_field_values(output)
assert any(
v.global_message_num == FILE_ID_MESG_NUM
and v.field_num == 8
and v.value == "Edge 1030 Plus"
and v.is_creator
for v in values
)
def test_convert_fit_device_supports_12_byte_header(tmp_path: Path) -> None:
"""12-byte headers carry no header CRC field, so conversion must succeed and
report header_crc=None while still rewriting the file CRC."""
file_def = definition(0, FILE_ID_MESG_NUM, [(1, 2, 0x84), (2, 2, 0x84)])
file_data = data(0, struct.pack("<HH", 255, 999))
source = tmp_path / "source12.fit"
source.write_bytes(make_fit(file_def + file_data, header_size=12))
output = tmp_path / "output12.fit"
result = convert_fit_device(source, output)
assert result.header_crc is None
assert result.file_crc is not None
assert result.patched_field_count == 2
assert is_fit_file(output) is True
assert output.read_bytes()[0] == 12
values = {(v.global_message_num, v.field_num): v.value for v in read_device_field_values(output)}
assert values[(FILE_ID_MESG_NUM, 1)] == 1
assert values[(FILE_ID_MESG_NUM, 2)] == 3570
def test_truncated_definition_is_non_recoverable(tmp_path: Path) -> None:
path = tmp_path / "truncated.fit"
path.write_bytes(make_fit(bytes([0x40, 0x00, 0x00])))
assert is_fit_file(path) is False
def test_convert_fit_device_rejects_truncated_definition(tmp_path: Path) -> None:
source = tmp_path / "truncated.fit"
source.write_bytes(make_fit(bytes([0x40, 0x00, 0x00])))
output = tmp_path / "output.fit"
with pytest.raises(FitFormatError):
convert_fit_device(source, output)