''' Read-only NativeDB audit regressions. ''' from hashlib import sha256 import json from pathlib import Path import numpy as np import polars as pl import pytest from typer.testing import CliRunner from piker import config from piker.data import def_iohlcv_fields from piker.storage import _audit as audit_mod from piker.storage._audit import ( audit_ohlcv_frame, audit_ohlcv_parquet, ) from piker.storage.cli import store from piker.storage.nativedb import ( mk_ohlcv_shm_keyed_filepath, ) def mk_frame(times: tuple[int, ...]) -> pl.DataFrame: ''' Build one canonical persisted frame with finite values. ''' size: int = len(times) return pl.DataFrame({ 'index': pl.Series(np.arange(size), dtype=pl.Int64), 'time': pl.Series(times, dtype=pl.Int64), 'open': pl.Series(np.arange(size) + 1, dtype=pl.Float64), 'high': pl.Series(np.arange(size) + 2, dtype=pl.Float64), 'low': pl.Series(np.arange(size), dtype=pl.Float64), 'close': pl.Series(np.arange(size) + 1, dtype=pl.Float64), 'volume': pl.Series(np.arange(size) + 10, dtype=pl.Float64), }) def test_audit_separates_structure_from_positive_gaps() -> None: ''' Expected closures must not masquerade as structural corruption. Persisted CME and session-market history can be canonical while containing positive timestamp gaps. Build a structurally valid frame with one aligned interval and prove validity remains green while exact endpoints and unresolved coverage remain visible. ''' report = audit_ohlcv_frame( mk_frame((60, 120, 300)), fqme='mnq.cme.20260918.ib', period_s=60, ) assert report['result'] == { 'structural_ok': True, 'gap_free': False, 'qualification_ok': False, 'violations': [], 'warnings': ['positive_time_gaps_unclassified'], } assert report['gaps']['count'] == 1 assert report['gaps']['missing_samples_total'] == 2 assert report['gaps']['intervals'][0] == { 'left_timestamp': 120, 'right_timestamp': 300, 'left_utc': '1970-01-01T00:02:00Z', 'right_utc': '1970-01-01T00:05:00Z', 'delta_s': 180, 'period_multiple': True, 'missing_samples': 2, 'classification': 'unclassified', } def test_audit_preserves_raw_defect_evidence() -> None: ''' Audit must count malformed rows without repairing or sorting. A baseline can contain extra provider columns, duplicated and reversed timestamps, a zero epoch, broken indexes, and non-finite values simultaneously. Construct all defects in physical file order and prove stable violations report each layer instead of hiding them through NativeDB canonicalization or dedupe. ''' frame = mk_frame((60, 60, 30, 0)).with_columns( pl.Series('index', [0, 2, 2, 4]), pl.Series('open', [1, 2, 3, 4], dtype=pl.Int64), pl.Series( 'close', [1, float('nan'), 3, 4], dtype=pl.Float64, ), pl.Series( 'volume', [1, 2, float('inf'), 4], dtype=pl.Float64, ), pl.Series('count', [1, 1, 1, 1]), ) report = audit_ohlcv_frame( frame, fqme='x.test', period_s=60, ) assert set(report['result']['violations']) == { 'extra_columns', 'column_order', 'canonical_dtypes', 'nonfinite_ohlcv', 'nonpositive_timestamps', 'duplicate_timestamps', 'timestamp_order', 'index_not_canonical', 'subperiod_time_step', } assert report['timestamps']['duplicate_excess'] == 1 assert report['timestamps']['adjacent_zero_delta'] == 1 assert report['timestamps']['negative_delta'] == 2 assert report['index']['row_position_mismatch'] == 2 assert report['values']['nan_by_column']['close'] == 1 assert report['values']['infinity_by_column']['volume'] == 1 def test_gap_aggregates_are_not_truncated_with_details() -> None: ''' Limiting JSON detail must not undercount total missing coverage. Long-lived session markets can contain thousands of expected closure intervals. Arrange two gaps but request one detail, record, then prove aggregate counts still describe the complete frame while the bounded detail list is marked truncated. ''' report = audit_ohlcv_frame( mk_frame((60, 180, 300)), fqme='x.test', period_s=60, max_gaps=1, ) gaps: dict = report['gaps'] assert gaps['count'] == 2 assert gaps['aligned_count'] == 2 assert gaps['missing_samples_total'] == 2 assert gaps['details_count'] == 1 assert gaps['details_truncated'] is True def test_subperiod_steps_fail_cadence_qualification() -> None: ''' Short positive deltas must not bypass expected-period validation. Gap detection alone considers deltas larger than the expected period. That let a 30-second step in 60-second data qualify. Arrange that defect and prove it remains separate from positive gaps while making structural qualification fail. ''' report = audit_ohlcv_frame( mk_frame((60, 90)), fqme='x.test', period_s=60, ) assert report['gaps']['count'] == 0 assert report['gaps']['subperiod_count'] == 1 assert report['gaps']['subperiod_intervals'][0]['delta_s'] == 30 assert 'subperiod_time_step' in report['result']['violations'] assert report['result']['qualification_ok'] is False def test_integer_evidence_preserves_values_above_float_precision( ) -> None: ''' Int64 timestamp and index evidence must not round through float. Adjacent integers above ``2**53`` collapse when coerced to Float64, inventing duplicate times and index mismatches. Audit a canonical two-row frame at that boundary and prove exact endpoints, uniqueness, and indexes remain JSON-native integers. ''' start: int = 2**53 frame = mk_frame((start, start + 1)).with_columns( pl.Series('index', [start, start + 1], dtype=pl.Int64) ) report = audit_ohlcv_frame( frame, fqme='x.test', period_s=1, ) assert report['timestamps']['minimum'] == start assert report['timestamps']['maximum'] == start + 1 assert report['timestamps']['duplicate_excess'] == 0 assert report['timestamps']['strictly_increasing'] is True assert report['index']['first'] == start assert report['index']['last'] == start + 1 assert report['index']['duplicate_excess'] == 0 assert report['index']['contiguous'] is True assert report['index']['canonical_from_zero'] is False def test_large_integer_gap_counts_remain_exact() -> None: ''' Missing-sample totals must not divide Int64 deltas through float. A near-Int64 interval exceeds Float64's exact range. Audit it at one-second cadence and prove all counts agree with exact integer division rather than rounded evidence. ''' right: int = 2**63 - 1 report = audit_ohlcv_frame( mk_frame((1, right)), fqme='x.test', period_s=1, ) expected: int = right - 2 assert report['gaps']['missing_samples_total'] == expected assert ( report['gaps']['intervals'][0]['missing_samples'] == expected ) def test_decimal_columns_remain_auditable_defect_evidence() -> None: ''' Numeric but noncanonical Polars dtypes must produce a report. Decimal Parquet columns are readable numeric evidence but do not support Polars ``is_nan()``. Cast timestamps and an OHLC field to Decimal and prove audit reports the dtype violation instead of crashing before malformed baseline evidence can be saved. ''' frame = mk_frame((60, 120)).with_columns( pl.col('time').cast(pl.Decimal(scale=0)), pl.col('close').cast(pl.Decimal(scale=2)), ) report = audit_ohlcv_frame( frame, fqme='x.test', period_s=60, ) assert report['timestamps']['numeric'] is True assert report['timestamps']['strictly_increasing'] is True assert report['values']['all_finite'] is True assert 'canonical_dtypes' in report['result']['violations'] def test_empty_frame_has_unverifiable_gap_coverage() -> None: ''' Empty storage must not claim valid cadence or index results. An empty canonical schema is structurally invalid and has no pair of timestamps from which gap coverage can be inferred. Prove the report retains its stable gap shape while gap-free status remains unknown and canonical index qualification is rejected. ''' report = audit_ohlcv_frame( mk_frame(()), fqme='x.test', period_s=60, ) assert report['gaps']['verifiable'] is False assert report['gaps']['aligned_count'] == 0 assert report['result']['gap_free'] is None assert report['index']['canonical_from_zero'] is False assert report['result']['qualification_ok'] is False def test_detail_limit_bounds_all_cadence_deviations() -> None: ''' One detail budget must cover gaps and short steps together. Human output combines both anomaly types. Slicing each list can exceed ``--max-gaps`` and misreport the detail count. Arrange one short step and one long gap with a budget of one and prove the combined detail output remains bounded and marked truncated. ''' report = audit_ohlcv_frame( mk_frame((60, 90, 210)), fqme='x.test', period_s=60, max_gaps=1, ) gaps: dict = report['gaps'] assert gaps['count'] == 1 assert gaps['subperiod_count'] == 1 assert gaps['details_count'] == 1 assert ( len(gaps['intervals']) + len(gaps['subperiod_intervals']) ) == 1 assert gaps['details_truncated'] is True def test_store_audit_json_is_read_only( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: ''' The CLI must inspect exact bytes without storage side effects. Operational storage openers can create config directories and mutate caches. Point config at an existing disposable NativeDB, invoke the real Typer command, and prove its JSON references the exact file while bytes and modification time stay unchanged. ''' monkeypatch.setattr(config, '_config_dir', tmp_path) datadir: Path = tmp_path / 'nativedb' datadir.mkdir() fqme: str = 'x.test' path: Path = mk_ohlcv_shm_keyed_filepath( fqme, 60, datadir, ) mk_frame((60, 120)).write_parquet(path) before_bytes: bytes = path.read_bytes() before_mtime: int = path.stat().st_mtime_ns output: Path = tmp_path / 'audit.json' snapshot: Path = tmp_path / 'audit.parquet' result = CliRunner().invoke( store, [ 'audit', fqme, '--period', '60', '--output', str(output), '--snapshot', str(snapshot), '--json', ], ) report: dict = json.loads(result.stdout) assert result.exit_code == 0 assert report['source']['path'] == str(path) assert report['source']['sha256'] == sha256( before_bytes ).hexdigest() assert report['result']['qualification_ok'] is True assert path.read_bytes() == before_bytes assert path.stat().st_mtime_ns == before_mtime assert snapshot.read_bytes() == before_bytes assert json.loads(output.read_text()) == report assert report['schema']['columns'] == [ name for name, _ in def_iohlcv_fields ] def test_store_audit_refuses_output_collisions( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: ''' Evidence output must not replace source or existing files. The first CLI accepted any ``--output`` path and could overwrite audited Parquet with JSON. Point output at the source, a report, and a directory; prove each exits with original bytes unchanged. ''' monkeypatch.setattr(config, '_config_dir', tmp_path) datadir: Path = tmp_path / 'nativedb' datadir.mkdir() fqme: str = 'x.test' path: Path = mk_ohlcv_shm_keyed_filepath( fqme, 60, datadir, ) mk_frame((60, 120)).write_parquet(path) before: bytes = path.read_bytes() runner = CliRunner() for output in (path, tmp_path / 'existing.json', tmp_path): if output.name == 'existing.json': output.write_text('keep\n') result = runner.invoke( store, ['audit', fqme, '--output', str(output)], ) assert result.exit_code == 2 assert path.read_bytes() == before assert (tmp_path / 'existing.json').read_text() == 'keep\n' def test_direct_snapshot_collision_preserves_existing_file( tmp_path: Path, ) -> None: ''' Exclusive snapshot races must never unlink another writer's file. The helper opens snapshots with ``xb``. Its first cleanup path unlinked the destination when exclusive open failed. Pre-create a destination and prove the collision preserves its existing bytes. ''' source: Path = tmp_path / 'source.parquet' snapshot: Path = tmp_path / 'snapshot.parquet' mk_frame((60, 120)).write_parquet(source) snapshot.write_bytes(b'existing evidence') with pytest.raises(FileExistsError): audit_ohlcv_parquet( source, fqme='x.test', period_s=60, snapshot=snapshot, ) assert snapshot.read_bytes() == b'existing evidence' def test_invalid_parquet_is_snapshotted_before_parse( tmp_path: Path, ) -> None: ''' Unreadable phase bytes must survive a failed structural audit. A truncated or non-Parquet file previously failed during parsing before ``--snapshot`` captured anything, discarding exact failure evidence from repair and restart. Supply invalid bytes to the low-level audit and prove it raises only after creating an exact, read-only snapshot for later diagnosis. ''' source: Path = tmp_path / 'broken.parquet' snapshot: Path = tmp_path / 'snapshot.parquet' evidence: bytes = b'not a parquet file\n' source.write_bytes(evidence) with pytest.raises(pl.exceptions.PolarsError): audit_ohlcv_parquet( source, fqme='x.test', period_s=60, snapshot=snapshot, ) assert snapshot.read_bytes() == evidence assert snapshot.stat().st_mode & 0o222 == 0 def test_snapshot_copy_failure_removes_partial_output( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: ''' Copy failure must not leave partial authoritative evidence. Snapshot bytes are captured before parsing so malformed evidence survives. Failed copies leave partial output that blocks retry. Raise after a prefix. Prove partial output is removed while the source stays whole. ''' source: Path = tmp_path / 'source.parquet' snapshot: Path = tmp_path / 'snapshot.parquet' mk_frame((60, 120)).write_parquet(source) before: bytes = source.read_bytes() def fail_copy(source_file, snapshot_file) -> None: snapshot_file.write(source_file.read(10)) raise OSError('simulated full filesystem') monkeypatch.setattr(audit_mod.shutil, 'copyfileobj', fail_copy) with pytest.raises(OSError, match='full filesystem'): audit_ohlcv_parquet( source, fqme='x.test', period_s=60, snapshot=snapshot, ) assert not snapshot.exists() assert source.read_bytes() == before def test_store_audit_rejects_unsafe_or_symlinked_sources( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ) -> None: ''' User-controlled FQME text must stay inside NativeDB storage. Absolute and parent-relative market names previously flowed into path construction, while a normal market could be a symlink to external bytes. Point config at a disposable NativeDB and prove traversal and source links fail before external bytes change. ''' monkeypatch.setattr(config, '_config_dir', tmp_path) datadir: Path = tmp_path / 'nativedb' datadir.mkdir() outside: Path = tmp_path / 'outside.parquet' mk_frame((60, 120)).write_parquet(outside) before: bytes = outside.read_bytes() source: Path = mk_ohlcv_shm_keyed_filepath( 'x.test', 60, datadir, ) source.symlink_to(outside) runner = CliRunner() for fqme in ('../../outside', str(outside), 'x.test'): result = runner.invoke(store, ['audit', fqme]) assert result.exit_code == 2 assert outside.read_bytes() == before def test_direct_audit_rejects_nonregular_source( tmp_path: Path, ) -> None: ''' Audit must not read devices, directories, or blocking pipes. ``O_NOFOLLOW`` rejects links but opens other node types. Pass a a directory to the reader and prove descriptor metadata rejects it before hashing or handing it to the Parquet parser. ''' with pytest.raises(ValueError, match='regular file'): audit_ohlcv_parquet( tmp_path, fqme='x.test', period_s=60, )