#!env xonsh ''' Sourceable helpers for disposable NativeDB backfill qualification. Usage: source snippets/nativedb_backfill_audit.xsh Every mutating helper requires a marker created by ``bfq_init()`` and refuses to operate outside that disposable root. Keep the private run root free from concurrent manual changes. ''' from datetime import ( UTC, datetime, ) from hashlib import sha256 import json import os from pathlib import Path import shutil import subprocess from typing import BinaryIO import numpy as np import polars as pl _bfq_marker: str = '.piker-backfill-qualification' _bfq_schema: str = 'piker.backfill-qualification/v1' _bfq_audit_schema: str = 'piker.nativedb.audit/v1' def _bfq_process_env() -> dict[str, str]: ''' Return the effective environment inherited by xonsh children. ''' env: dict[str, str] = os.environ.copy() try: from xonsh.built_ins import XSH except ImportError: return env if XSH.env is not None: env.update(XSH.env.detype()) return env def _bfq_setenv(name: str, value: str) -> None: ''' Update both Python and xonsh subprocess environments. ''' os.environ[name] = value try: from xonsh.built_ins import XSH except ImportError: return if XSH.env is not None: XSH.env[name] = value def _bfq_metadata(root: Path) -> dict: marker: Path = root / _bfq_marker if ( not marker.is_file() or marker.is_symlink() ): raise RuntimeError( f'Qualification marker missing or unsafe: {marker}' ) metadata: dict = json.loads(marker.read_text()) if metadata.get('schema') != _bfq_schema: raise RuntimeError(f'Unknown qualification marker: {marker}') if Path(metadata['root']).resolve() != root: raise RuntimeError( f'Qualification marker root mismatch: {marker}' ) return metadata def _bfq_root(run_root: str|Path) -> Path: root = Path(run_root).expanduser().resolve() _bfq_metadata(root) return root def _bfq_fqme(fqme: str) -> str: if ( fqme in {'.', '..'} or Path(fqme).name != fqme or '/' in fqme or '\\' in fqme ): raise ValueError(f'Unsafe FQME path component: {fqme!r}') return fqme def _bfq_paths( run_root: str|Path, fqme: str, period_s: int, ) -> tuple[Path, Path, Path, Path]: root: Path = _bfq_root(run_root) fqme = _bfq_fqme(fqme) xdg: Path = root / 'xdg' evidence: Path = root / 'evidence' nativedb: Path = xdg / 'piker' / 'nativedb' parquet: Path = nativedb / f'{fqme}.ohlcv{period_s}s.parquet' for path in (xdg, evidence, nativedb, parquet): resolved: Path = path.resolve() if not resolved.is_relative_to(root): raise RuntimeError( f'Qualification path escaped run root: {resolved}' ) if path.is_symlink(): raise RuntimeError(f'Refusing symlinked path: {path}') return root, xdg, evidence, parquet def _bfq_hash_file(file: BinaryIO) -> str: digest = sha256() file.seek(0) for chunk in iter( lambda: file.read(1024 * 1024), b'', ): digest.update(chunk) return digest.hexdigest() def _bfq_hash(path: Path) -> str: with path.open('rb') as file: return _bfq_hash_file(file) def _bfq_read_bytes(path: Path) -> bytes: descriptor: int = os.open( path, os.O_RDONLY | os.O_CLOEXEC | os.O_NOFOLLOW, ) with os.fdopen(descriptor, 'rb') as file: return file.read() def _bfq_copy_exclusive( source: Path, destination: Path, mode: int|None = None, ) -> None: ''' Copy bytes to a new non-symlink path without replacement. ''' if destination.is_symlink(): raise RuntimeError( f'Refusing symlinked output: {destination}' ) created: bool = False try: with source.open('rb') as source_file: with destination.open('xb') as destination_file: created = True if mode is not None: os.fchmod(destination_file.fileno(), mode) shutil.copyfileobj(source_file, destination_file) except BaseException: if created: destination.unlink(missing_ok=True) raise def _bfq_phase(phase: str) -> str: safe: str = phase.replace('-', '').replace('_', '') if not safe.isalnum(): raise ValueError( 'Phase names may contain letters, numbers, - and _ only' ) return phase def bfq_init( run_root: str|Path, source_config: str|Path|None = None, validate_piker: bool = True, ) -> Path: ''' Create a disposable XDG root and optionally copy provider config. ''' root = Path(run_root).expanduser().resolve() marker: Path = root / _bfq_marker marker_exists: bool = ( marker.exists() or marker.is_symlink() ) if ( root.exists() and not marker_exists and any(root.iterdir()) ): raise RuntimeError( f'Refusing non-empty unmarked directory: {root}' ) if marker_exists: metadata: dict = _bfq_metadata(root) else: metadata = {} xdg: Path = root / 'xdg' piker_conf: Path = xdg / 'piker' nativedb: Path = piker_conf / 'nativedb' evidence: Path = root / 'evidence' logs: Path = root / 'logs' for path in (root, xdg, piker_conf, nativedb, evidence, logs): if path.is_symlink(): raise RuntimeError(f'Refusing symlinked path: {path}') if not path.resolve().is_relative_to(root): raise RuntimeError( f'Path escaped qualification root: {path}' ) path.mkdir(parents=True, exist_ok=True, mode=0o700) path.chmod(0o700) if not marker_exists: process_env: dict[str, str] = _bfq_process_env() repo_root = subprocess.run( ['git', 'rev-parse', '--show-toplevel'], check=True, capture_output=True, env=process_env, text=True, ).stdout.strip() piker_executable: str|None = shutil.which( 'piker', path=process_env.get('PATH'), ) if piker_executable is None: raise RuntimeError('No `piker` executable found in PATH') marker_data: dict = { 'schema': _bfq_schema, 'root': str(root), 'repo_root': str(Path(repo_root).resolve()), 'piker_executable': str( Path(piker_executable).resolve() ), } metadata = marker_data if validate_piker: process_env = _bfq_process_env() repo_root = metadata['repo_root'] pythonpath: list[str] = [repo_root] pythonpath.extend( path for path in process_env.get('PYTHONPATH', '').split( os.pathsep ) if path and path != repo_root ) process_env['PYTHONPATH'] = os.pathsep.join(pythonpath) proc = subprocess.run( [ metadata['piker_executable'], 'store', 'audit', '--help', ], check=False, capture_output=True, cwd=metadata['repo_root'], env=process_env, text=True, ) if proc.returncode: raise RuntimeError( f'Bound `piker` lacks `store audit`:\n' f'{metadata["piker_executable"]}\n' f'{proc.stderr}' ) if not marker_exists: with marker.open('x') as marker_file: os.fchmod(marker_file.fileno(), 0o600) marker_file.write( json.dumps(metadata, indent=2, sort_keys=True) +'\n' ) if source_config is not None: source = Path(source_config).expanduser().resolve() for name in ('conf.toml', 'brokers.toml'): src: Path = source / name dst: Path = piker_conf / name if src.is_file() and not dst.exists(): shutil.copy2(src, dst) if name == 'brokers.toml': with dst.open('rb') as config_file: os.fchmod(config_file.fileno(), 0o600) bfq_use(root) print( f'Disposable qualification root ready:\n' f' root: {root}\n' f' XDG_CONFIG_HOME: {root / "xdg"}\n' ) return root def bfq_use(run_root: str|Path) -> Path: ''' Point xonsh and future children at the disposable root. ''' root: Path = _bfq_root(run_root) metadata: dict = _bfq_metadata(root) xdg: Path = root / 'xdg' repo_root: str = metadata['repo_root'] process_env: dict[str, str] = _bfq_process_env() pythonpath: list[str] = [repo_root] pythonpath.extend( path for path in process_env.get('PYTHONPATH', '').split( os.pathsep ) if path and path != repo_root ) _bfq_setenv('XDG_CONFIG_HOME', str(xdg)) _bfq_setenv('PYTHONPATH', os.pathsep.join(pythonpath)) print(f'XDG_CONFIG_HOME={xdg}') return xdg def bfq_record_case( run_root: str|Path, fqme: str, period_s: int, provider: str, symptom: str, bad_start_utc: str|None = None, bad_end_utc: str|None = None, ) -> Path: ''' Save read-only operator inputs used to identify one case. ''' root: Path = _bfq_root(run_root) fqme = _bfq_fqme(fqme) path: Path = root / 'case.json' if path.is_symlink(): raise RuntimeError(f'Refusing symlinked case record: {path}') if path.exists(): raise FileExistsError(f'Case record already exists: {path}') case: dict = { 'fqme': fqme, 'period_s': period_s, 'provider': provider, 'symptom': symptom, 'bad_start_utc': bad_start_utc, 'bad_end_utc': bad_end_utc, 'created_at_utc': datetime.now(tz=UTC).isoformat(), } with path.open('x') as case_file: os.fchmod(case_file.fileno(), 0o444) case_file.write( json.dumps(case, indent=2, sort_keys=True) + '\n' ) print(f'Wrote case record: {path}') return path def bfq_seed( run_root: str|Path, fqme: str, period_s: int, source_parquet: str|Path, ) -> Path: ''' Copy a known-bad Parquet into disposable storage and evidence. ''' ( root, _, evidence, parquet, ) = _bfq_paths(run_root, fqme, period_s) source = Path(source_parquet).expanduser().resolve() if not source.is_file(): raise FileNotFoundError(source) if parquet.exists(): raise FileExistsError(parquet) evidence_seed: Path = evidence / 'seed.parquet' if evidence_seed.exists(): raise FileExistsError(evidence_seed) _bfq_copy_exclusive(source, evidence_seed, mode=0o444) _bfq_copy_exclusive(evidence_seed, parquet) print( f'Seeded disposable NativeDB:\n' f' source: {source}\n' f' evidence: {evidence_seed}\n' f' active: {parquet}\n' f' root: {root}\n' ) return parquet def bfq_clear_for_fresh( run_root: str|Path, fqme: str, period_s: int, ) -> Path|None: ''' Remove only disposable active history after archiving it. ''' ( _, _, evidence, parquet, ) = _bfq_paths(run_root, fqme, period_s) if not parquet.exists(): print(f'Already fresh; no active Parquet: {parquet}') return None stamp: str = datetime.now(tz=UTC).strftime( '%Y%m%dT%H%M%S%fZ' ) archived: Path = evidence / f'pre-fresh-{stamp}.parquet' if ( archived.exists() or archived.is_symlink() ): raise FileExistsError(archived) os.link(parquet, archived) with archived.open('rb') as archived_file: os.fchmod(archived_file.fileno(), 0o444) parquet.unlink() print( f'Archived disposable history for fresh backfill:\n' f' archived: {archived}\n' f' active path now absent: {parquet}\n' ) return archived def bfq_audit( run_root: str|Path, phase: str, fqme: str, period_s: int, max_gaps: int = 200, require_structural: bool = False, ) -> dict: ''' Save JSON plus a checksum-matched Parquet snapshot for one phase. ''' phase = _bfq_phase(phase) ( root, _, evidence, parquet, ) = _bfq_paths(run_root, fqme, period_s) bfq_use(root) if not parquet.is_file(): raise FileNotFoundError(parquet) report_path: Path = evidence / f'{phase}.json' snapshot_path: Path = evidence / f'{phase}.parquet' manual_path: Path = evidence / f'{phase}.manual.json' if any( path.exists() for path in (report_path, snapshot_path, manual_path) ): raise FileExistsError( f'Phase evidence already exists: {phase}' ) metadata: dict = _bfq_metadata(root) piker_cmd: str = metadata['piker_executable'] proc = subprocess.run( [ piker_cmd, 'store', 'audit', fqme, '--period', str(period_s), '--max-gaps', str(max_gaps), '--output', str(report_path), '--snapshot', str(snapshot_path), '--json', ], check=False, capture_output=True, env=_bfq_process_env(), text=True, ) if proc.returncode: raise RuntimeError( f'`piker store audit` failed ({proc.returncode}):\n' f'{proc.stderr}' ) report: dict = json.loads(report_path.read_text()) report_hash: str = _bfq_hash(report_path) snapshot_hash: str = _bfq_hash(snapshot_path) if snapshot_hash != report['source']['sha256']: raise RuntimeError('Audit snapshot checksum mismatch') git_head = subprocess.run( ['git', 'rev-parse', 'HEAD'], cwd=metadata['repo_root'], check=False, capture_output=True, text=True, ).stdout.strip() piker_file = Path(report['runtime']['piker_file']).resolve() repo_root = Path(metadata['repo_root']).resolve() if not piker_file.is_relative_to(repo_root): raise RuntimeError( f'`piker` imported outside reviewed repo: {piker_file}' ) manual: dict = { 'phase': phase, 'run_root': str(root), 'snapshot_path': snapshot_path.name, 'report_sha256': report_hash, 'git_head': git_head, 'piker_file': str(piker_file), 'piker_executable': piker_cmd, 'xdg_config_home': os.environ['XDG_CONFIG_HOME'], } with manual_path.open('x') as manual_file: os.fchmod(manual_file.fileno(), 0o444) manual_file.write( json.dumps(manual, indent=2, sort_keys=True) + '\n' ) with report_path.open('rb') as report_file: os.fchmod(report_file.fileno(), 0o444) result: dict = report['result'] print( f'Audit saved: {report_path}\n' f' snapshot: {snapshot_path}\n' f' rows: {report["rows"]["count"]}\n' f' structural_ok: {result["structural_ok"]}\n' f' gaps: {report["gaps"]["count"]}\n' ) if ( require_structural and not result['structural_ok'] ): raise AssertionError( f'Phase is structurally invalid: {report_path}' ) return report def bfq_compare( run_root: str|Path, before_phase: str, after_phase: str, require_newer: bool = False, ) -> dict: ''' Compare evidence and fail on core preservation regressions. ''' before_phase = _bfq_phase(before_phase) after_phase = _bfq_phase(after_phase) if before_phase == after_phase: raise ValueError('Comparison phases must differ') root: Path = _bfq_root(run_root) evidence: Path = root / 'evidence' if ( evidence.is_symlink() or not evidence.is_dir() or not evidence.resolve().is_relative_to(root) ): raise RuntimeError(f'Unsafe evidence directory: {evidence}') case_path: Path = root / 'case.json' if ( case_path.is_symlink() or not case_path.is_file() ): raise RuntimeError( f'Qualification case missing: {case_path}' ) case: dict = json.loads(case_path.read_text()) def load_phase(phase: str) -> tuple[dict, pl.DataFrame]: report_path: Path = evidence / f'{phase}.json' snapshot_path: Path = evidence / f'{phase}.parquet' manual_path: Path = evidence / f'{phase}.manual.json' for path in (report_path, snapshot_path, manual_path): if ( path.is_symlink() or not path.is_file() or not path.resolve().is_relative_to(root) ): raise RuntimeError(f'Unsafe phase evidence: {path}') report_bytes: bytes = _bfq_read_bytes(report_path) manual_bytes: bytes = _bfq_read_bytes(manual_path) report: dict = json.loads(report_bytes) manual: dict = json.loads(manual_bytes) if report.get('audit_schema') != _bfq_audit_schema: raise RuntimeError( f'Unknown audit schema: {report_path}' ) if ( report.get('fqme') != case.get('fqme') or report.get('period_s') != case.get('period_s') ): raise RuntimeError( f'Phase does not match qualification case: {phase}' ) if ( manual.get('phase') != phase or manual.get('run_root') != str(root) or manual.get('snapshot_path') != snapshot_path.name ): raise RuntimeError(f'Manual metadata mismatch: {phase}') report_hash: str = sha256(report_bytes).hexdigest() if report_hash != manual.get('report_sha256'): raise RuntimeError( f'Audit report checksum mismatch: {phase}' ) descriptor: int = os.open( snapshot_path, os.O_RDONLY | os.O_CLOEXEC | os.O_NOFOLLOW, ) with os.fdopen(descriptor, 'rb') as snapshot_file: snapshot_stat = os.fstat(snapshot_file.fileno()) snapshot_hash: str = _bfq_hash_file(snapshot_file) if snapshot_hash != report['source']['sha256']: raise RuntimeError( f'Audit snapshot checksum mismatch: {phase}' ) source_size: int = report['source']['size_bytes'] if snapshot_stat.st_size != source_size: raise RuntimeError( f'Audit snapshot size mismatch: {phase}' ) snapshot_file.seek(0) frame: pl.DataFrame = pl.read_parquet(snapshot_file) return report, frame before, before_frame = load_phase(before_phase) after, after_frame = load_phase(after_phase) limitations: list[str] = [] def valid_timestamps( frame: pl.DataFrame, label: str, ) -> set|None: if 'time' not in frame.columns: limitations.append(f'{label}_time_missing') return None time_dtype = frame['time'].dtype if not ( time_dtype.is_integer() or time_dtype.is_float() ): limitations.append(f'{label}_time_nonnumeric') return None timestamps: set = set() for timestamp in frame['time'].drop_nulls().to_list(): if ( isinstance(timestamp, float) and not np.isfinite(timestamp) ): continue if timestamp <= 0: continue timestamps.add(timestamp) return timestamps before_times: set|None = valid_timestamps( before_frame, 'before', ) after_times: set|None = valid_timestamps( after_frame, 'after', ) missing_times: list = [] if ( before_times is not None and after_times is not None ): missing_times = sorted(before_times - after_times) value_fields: list[str] = [ 'open', 'high', 'low', 'close', 'volume', ] changed_times: list|None = None seam_columns: set[str] = set(value_fields) seam_columns.add('time') seam_ready: bool = bool( seam_columns.issubset(before_frame.columns) and seam_columns.issubset(after_frame.columns) and before_times is not None and after_times is not None ) if seam_ready: before_latest = before_frame.unique( subset='time', keep='last', ).filter( pl.col('time').is_not_null() & pl.col('time').is_finite() & (pl.col('time') > 0) ).select([ 'time', pl.struct(value_fields).hash().alias('value_hash'), ]) after_latest = after_frame.unique( subset='time', keep='last', ).filter( pl.col('time').is_not_null() & pl.col('time').is_finite() & (pl.col('time') > 0) ).select([ 'time', pl.struct(value_fields).hash().alias('value_hash'), ]) common = before_latest.join( after_latest, on='time', how='inner', suffix='_after', ) changed_times = sorted( common .filter( pl.col('value_hash') != pl.col('value_hash_after') ) ['time'] .to_list() ) else: limitations.append('common_value_comparison_unavailable') before_gaps: set[tuple] = { ( gap['left_timestamp'], gap['right_timestamp'], ) for gap in before['gaps']['intervals'] } after_gaps: set[tuple] = { ( gap['left_timestamp'], gap['right_timestamp'], ) for gap in after['gaps']['intervals'] } before_min = min(before_times) if before_times else None after_min = min(after_times) if after_times else None before_max = max(before_times) if before_times else None after_max = max(after_times) if after_times else None checks: dict[str, bool] = { 'after_structural_ok': after['result']['structural_ok'], 'no_timestamps_lost': bool( before_times is not None and after_times is not None and not missing_times ), 'minimum_not_later': bool( before_min is not None and after_min is not None and after_min <= before_min ), 'maximum_not_earlier': bool( before_max is not None and after_max is not None and after_max >= before_max ), 'timestamps_unique': ( after['timestamps'].get('duplicate_excess') == 0 ), 'timestamps_ordered': ( after['timestamps'].get('non_positive_delta') == 0 ), 'index_canonical': after['index']['canonical_from_zero'], 'ohlcv_finite': after['values']['all_finite'], } if require_newer: checks['maximum_advanced'] = bool( before_max is not None and after_max is not None and after_max > before_max ) gap_details_truncated: bool = bool( before['gaps']['details_truncated'] or after['gaps']['details_truncated'] ) gaps_added: list|None = None gaps_removed: list|None = None if not gap_details_truncated: gaps_added = sorted(after_gaps - before_gaps) gaps_removed = sorted(before_gaps - after_gaps) comparison: dict = { 'before_phase': before_phase, 'after_phase': after_phase, 'checks': checks, 'passed': all(checks.values()), 'before_rows': before['rows']['count'], 'after_rows': after['rows']['count'], 'missing_timestamp_count': len(missing_times), 'missing_timestamps': missing_times[:200], 'missing_timestamps_truncated': len(missing_times) > 200, 'before_duplicate_excess': ( before['timestamps'].get('duplicate_excess') ), 'after_duplicate_excess': ( after['timestamps'].get('duplicate_excess') ), 'changed_common_row_count': ( len(changed_times) if changed_times is not None else None ), 'changed_common_timestamps': ( changed_times[:200] if changed_times is not None else None ), 'changed_common_rows_truncated': bool( changed_times is not None and len(changed_times) > 200 ), 'gaps_added': gaps_added, 'gaps_removed': gaps_removed, 'gap_comparison_truncated': gap_details_truncated, 'limitations': limitations, } output: Path = ( evidence /f'compare-{before_phase}-to-{after_phase}.json' ) if output.is_symlink(): raise RuntimeError(f'Refusing symlinked output: {output}') if output.exists(): raise FileExistsError(output) with output.open('x') as output_file: os.fchmod(output_file.fileno(), 0o444) output_file.write( json.dumps( comparison, allow_nan=False, indent=2, sort_keys=True, ) +'\n' ) print(f'Comparison saved: {output}') for check, passed in checks.items(): print(f' {"PASS" if passed else "FAIL"}: {check}') print( f' changed common rows: ' f'{None if changed_times is None else len(changed_times)}\n' f' gaps added: ' f'{None if gaps_added is None else len(gaps_added)}\n' f' gaps removed: ' f'{None if gaps_removed is None else len(gaps_removed)}\n' ) if not comparison['passed']: raise AssertionError( f'Qualification comparison failed: {output}' ) return comparison