Repository navigation
Expand file tree
/
Copy pathmain.py
More file actions
2266 lines (2103 loc) · 86.6 KB
/
Copy pathmain.py
File metadata and controls
2266 lines (2103 loc) · 86.6 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""
CLI entry point: load config (YAML/JSON), run audit and report (optionally tagged with tenant/customer and technician/operator), or start API (--web) on --host/--port (defaults: loopback, 8088; see resolve_api_host).
"""
import argparse
import json
import os
import ssl
import sys
from pathlib import Path
from typing import Any
# Ensure project root on path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from datetime import UTC
from config.loader import load_config
from core.database import LocalDBManager
from core.engine import AuditEngine
from core.licensing import LicenseBlockedError
from core.runtime_trust import get_runtime_trust_snapshot
def _cli_public_version_line() -> str:
"""Public CLI --version string (no maturity_build octet; see ADR-0073)."""
from core.about import _package_version
from core.integrity_anchor import alpha_version_suffix
return f"Data Boar {_package_version()}{alpha_version_suffix()}"
def _run_startup_integrity_check(config: dict[str, Any] | None) -> dict[str, Any]:
"""First-run validate / startup re-verify (#856); stderr banner when tampered."""
from core.integrity_anchor import ALPHA_LABEL, ALPHA_NOTE, ensure_integrity_anchor
snap = ensure_integrity_anchor(config)
if snap.get("integrity_state") == "tampered":
print(
"*** INTEGRITY: behaviour-critical modules diverge from the "
f"validated anchor — runtime self-marked {ALPHA_LABEL} ({ALPHA_NOTE}). "
f"Mismatched: {', '.join(snap.get('mismatched_files', []))} ***",
file=sys.stderr,
)
return snap
def _install_scan_interrupt_signal_handlers() -> None:
"""Map SIGTERM to KeyboardInterrupt so engine finally marks ``interrupted`` (#1251)."""
import signal
def _raise_keyboard_interrupt(signum, frame): # type: ignore[no-untyped-def]
raise KeyboardInterrupt()
if hasattr(signal, "SIGTERM"):
try:
signal.signal(signal.SIGTERM, _raise_keyboard_interrupt)
except (ValueError, OSError):
# Not the main thread / platform rejects — best-effort only.
pass
def _finish_session_interrupted_if_running(engine: AuditEngine) -> None:
"""Mark current session interrupted when still ``running`` (idempotent vs engine finally)."""
sid = engine.db_manager.current_session_id
if sid:
engine.db_manager.finish_session(sid, "interrupted")
def _emit_runtime_trust_info(
snapshot: dict[str, Any], *, to_stdout: bool = True, to_stderr: bool = True
) -> None:
info_line = (
"[INFO] runtime-trust: "
f"{snapshot['trust_level'].upper()} "
f"(state={snapshot.get('trust_state', 'degraded')}, "
f"license_state={snapshot['license_state']}, "
f"mode={snapshot['license_mode']})"
)
license_detail = str(snapshot.get("license_detail") or "")
if license_detail in {
"hybrid_mldsa65_verified",
"mldsa_signature_invalid",
"untrusted_key_override",
} or license_detail.startswith(
("rotation_epoch_", "rotation_rejected:", "hybrid_mldsa65_verified")
):
info_line += f" license_detail={license_detail}"
if to_stdout:
print(info_line)
if to_stderr:
print(info_line, file=sys.stderr)
if not snapshot["is_unexpected"]:
return
attention_line = (
"[INFO] runtime-trust attention: "
"THERE IS SOMETHING DIFFERENT AND UNEXPECTED IN THIS RUNTIME. "
"Review license/integrity state before trusting scan or report outputs."
)
if to_stdout:
print(attention_line)
if to_stderr:
print(attention_line, file=sys.stderr)
def _maybe_init_otel_for_cli() -> None:
"""Opt-in OTel for CLI / oneshot / exports before FastAPI import (#1535).
Fail-soft: never block the CLI. Reuses ``maybe_setup_otel(app=None)``;
``api.routes`` may call again with ``app`` (idempotent).
"""
try:
from core.otel_setup import maybe_setup_otel
maybe_setup_otel(app=None)
except Exception: # noqa: BLE001 — never block CLI
pass
_ENV_FIELDS_TARGET = (
"pass_from_env",
"user_from_env",
"token_from_env",
"api_key_from_env",
)
_ENV_FIELDS_AUTH = ("client_secret_from_env",)
_SENSITIVE_FIELDS = frozenset(
{
"pass_from_env",
"token_from_env",
"api_key_from_env",
"client_secret_from_env",
}
)
def _mask_env_name(field: str, env_name: str) -> str:
"""Return env var name for logs, masking credential field references."""
if field in _SENSITIVE_FIELDS:
return "***"
return env_name
def _validate_config_and_exit(config: dict[str, Any], config_path: str) -> None:
"""Pre-flight: connector recognition, required keys, env hints (no network/DB)."""
# Connector registration runs via top-level ``from core.engine import AuditEngine``.
from core.connector_registry import connector_for_target
errors: list[str] = []
warnings: list[str] = []
targets = config.get("targets", [])
print(f"Validating config: {config_path}")
if not targets:
warnings.append("config: no targets defined")
for i, target in enumerate(targets):
name = target.get("name", f"target[{i}]")
result = connector_for_target(target)
if result is None:
t = target.get("type", "?")
d = target.get("driver", "")
errors.append(
f"target \"{name}\": unknown type/driver '{t}'"
+ (f" driver={d!r}" if d else "")
+ " — no connector registered"
)
continue
_, required_keys = result
for key in required_keys:
if key not in target:
errors.append(f'target "{name}": required key "{key}" missing')
for field in _ENV_FIELDS_TARGET:
env_name = target.get(field)
if env_name and not os.environ.get(env_name):
warnings.append(
f'target "{name}": {field}={_mask_env_name(field, env_name)!r} — env var not set'
)
auth = target.get("auth") or {}
for field in _ENV_FIELDS_AUTH:
env_name = auth.get(field)
if env_name and not os.environ.get(env_name):
warnings.append(
f'target "{name}": auth.{field}={_mask_env_name(field, env_name)!r} — env var not set'
)
# Offline optional SQL driver probe (reuse sql_driver_deps; no connect) — #1246
kind = (target.get("type") or "").strip().lower()
if kind == "database":
from connectors.sql_driver_deps import ensure_sql_driver_available
driver = target.get("driver") or "postgresql"
try:
ensure_sql_driver_available(driver)
except ImportError as exc:
warnings.append(f'target "{name}": {exc}')
driver = target.get("driver", "")
label = f"type={kind or '?'}" + (f" driver={driver}" if driver else "")
print(f' OK target[{i}] "{name}" {label}')
# #1411 — Pro accelerator observability (same class of WARN as missing SQL driver).
from core.pro_scan_path import (
resolve_pro_scan_path,
rust_accelerator_installed,
)
_, pf_status = resolve_pro_scan_path(config)
paid_prefilter_tier = pf_status.get("tier") in (
"pro_plus",
"enterprise",
"partner",
)
# Same class as optional SQL-driver WARN, but only when the tier could use
# the accelerator (OPEN/Community never activate paid accel — avoid noise).
if paid_prefilter_tier and not rust_accelerator_installed():
warnings.append(
"boar_fast_filter (Rust accelerator) is not installed — "
"PyPI installs use the pure-Python regex-stage fallback; "
"wheelhouse is the only distribution channel (see TROUBLESHOOTING.md)"
)
if pf_status.get("active"):
print(
" OK rust-regex-stage readiness ACTIVE "
f"backend={pf_status.get('backend')} tier={pf_status.get('tier')}"
)
else:
# Status JSON may still carry legacy reason codes from WIP routing;
# product framing (#1414): observability only — no skip/latch narrative.
print(
" OK rust-regex-stage readiness inactive "
f"(reason={pf_status.get('reason') or 'n/a'} tier={pf_status.get('tier')})"
)
from core.output_paths import OutputPathError, ensure_config_output_directories
# #538: connector/key errors abort before mkdir, integrity-anchor sqlite, or trust emit.
if errors:
for w in warnings:
print(f" WARN {w}")
for e in errors:
print(f" ERROR {e}")
print(f"\n[INVALID] {len(errors)} error(s), {len(warnings)} warning(s).")
sys.exit(1)
try:
for msg in ensure_config_output_directories(config):
print(f" OK {msg}")
except OutputPathError as e:
for w in warnings:
print(f" WARN {w}")
print(f" ERROR {e}")
print(f"\n[INVALID] 1 error(s), {len(warnings)} warning(s).")
sys.exit(1)
_run_startup_integrity_check(config)
runtime_trust = get_runtime_trust_snapshot(config)
_emit_runtime_trust_info(runtime_trust, to_stdout=True, to_stderr=True)
for w in warnings:
print(f" WARN {w}")
print(f"\n[OK] {len(targets)} target(s) valid. {len(warnings)} warning(s).")
sys.exit(0)
def _print_session_diff(result: dict[str, Any]) -> None:
"""Human-readable summary for --diff (stdout)."""
session_a = result["session_a"]
session_b = result["session_b"]
print(f"\nDiff: {session_a} -> {session_b}\n")
db = result["database"]
fs = result["filesystem"]
db_target_names: set[str] = set()
for bucket in (db["new"], db["resolved"]):
for f in bucket.values():
db_target_names.add(f.target_name or "")
for _k, (fa, _fb) in db["changed"].items():
db_target_names.add(fa.target_name or "")
fs_target_names: set[str] = set()
for bucket in (fs["new"], fs["resolved"]):
for f in bucket.values():
fs_target_names.add(f.target_name or "")
for _k, (fa, _fb) in fs["changed"].items():
fs_target_names.add(fa.target_name or "")
n_db_targets = len(db_target_names) or (
1 if db["new"] or db["resolved"] or db["changed"] else 0
)
n_fs_targets = len(fs_target_names) or (
1 if fs["new"] or fs["resolved"] or fs["changed"] else 0
)
print(f"DATABASE ({n_db_targets} target(s) with delta):")
for f in db["new"].values():
schema = f.schema_name or ""
table = f.table_name or ""
col = f.column_name or ""
loc = ".".join(p for p in (schema, table, col) if p)
print(
f" NEW {f.target_name} {loc} "
f"{f.pattern_detected} / {f.sensitivity_level}"
)
for f in db["resolved"].values():
schema = f.schema_name or ""
table = f.table_name or ""
col = f.column_name or ""
loc = ".".join(p for p in (schema, table, col) if p)
print(f" RESOLVED {f.target_name} {loc} (was {f.sensitivity_level})")
for _k, (fa, fb) in db["changed"].items():
schema = fa.schema_name or ""
table = fa.table_name or ""
col = fa.column_name or ""
loc = ".".join(p for p in (schema, table, col) if p)
print(
f" CHANGED {fa.target_name} {loc} "
f"{fa.sensitivity_level} -> {fb.sensitivity_level}"
)
print(f"\nFILESYSTEM ({n_fs_targets} target(s) with delta):")
for f in fs["new"].values():
path = f.path or ""
fname = f.file_name or ""
label = f"{path} {fname}".strip()
print(
f" NEW {f.target_name} {label} "
f"{f.pattern_detected} / {f.sensitivity_level}"
)
for f in fs["resolved"].values():
path = f.path or ""
fname = f.file_name or ""
label = f"{path} {fname}".strip()
print(f" RESOLVED {f.target_name} {label} (was {f.sensitivity_level})")
for _k, (fa, fb) in fs["changed"].items():
path = fa.path or ""
fname = fa.file_name or ""
label = f"{path} {fname}".strip()
print(
f" CHANGED {fa.target_name} {label} "
f"{fa.sensitivity_level} -> {fb.sensitivity_level}"
)
n_new = len(db["new"]) + len(fs["new"])
n_resolved = len(db["resolved"]) + len(fs["resolved"])
n_changed = len(db["changed"]) + len(fs["changed"])
n_new_high = result["new_high_count"]
print(
f"\nSummary: {n_new} new ({n_new_high} HIGH), "
f"{n_resolved} resolved, {n_changed} severity change(s)."
)
def _run_session_diff_cli(
config: dict[str, Any],
session_a: str,
session_b: str,
*,
fail_on_new_high: bool,
) -> None:
from core.database import LocalDBManager
db_path = config.get("sqlite_path", "audit_results.db")
mgr = LocalDBManager(db_path)
try:
result = mgr.diff_sessions(session_a, session_b)
_print_session_diff(result)
if fail_on_new_high and result["new_high_count"] > 0:
print(
f"\n[FAIL] --fail-on-new-high: {result['new_high_count']} "
"new HIGH finding(s). Exit 1."
)
sys.exit(1)
except ValueError as e:
print(f"Session error: {e}", file=sys.stderr)
sys.exit(2)
finally:
mgr.dispose()
def _run_regenerate_report_cli(
config: dict[str, Any], config_path: str, session_id: str
) -> None:
"""Regenerate Excel + heatmap (+ learned patterns) from SQLite without re-scan."""
from core.engine import AuditEngine
from core.output_paths import OutputPathError, ensure_config_output_directories
sid = (session_id or "").strip()
if not sid:
print("Session error: empty session id", file=sys.stderr)
sys.exit(2)
try:
ensure_config_output_directories(config)
except OutputPathError as e:
print(f"Error: {e}", file=sys.stderr)
sys.exit(1)
engine = AuditEngine(config, config_path=config_path)
try:
known = {row["session_id"] for row in engine.db_manager.list_sessions()}
if sid not in known:
print(f"Session error: Unknown session: {sid}", file=sys.stderr)
sys.exit(2)
report_path = engine.generate_final_reports(sid)
if report_path:
print(f"Report written: {report_path}")
else:
print("No findings to report.")
from core.plugins.hook import maybe_run_remediation_hook
maybe_run_remediation_hook(config, sid, db_manager=engine.db_manager)
finally:
engine.db_manager.dispose()
def _run_governance_report_cli(
config: dict[str, Any],
config_path: str,
output_path: str | None,
session_id: str | None,
) -> None:
"""Render Governance Lens Markdown from SQLite (no re-scan)."""
from core.engine import AuditEngine
from report.governance_lens import governance_lens_feature_allowed
from report.governance_report import (
GovernanceReportError,
GovernanceReportSessionError,
default_governance_report_path,
resolve_governance_session_id,
write_governance_report,
)
if not governance_lens_feature_allowed(config):
print(
"Governance Lens report requires governance.enabled: true and a Pro+ "
"license tier (governance_lens_pro).",
file=sys.stderr,
)
sys.exit(2)
engine = AuditEngine(config, config_path=config_path)
try:
try:
sid = resolve_governance_session_id(engine.db_manager, session_id)
dest = (
Path(output_path)
if output_path
else default_governance_report_path(config, sid)
)
written = write_governance_report(dest, config, engine.db_manager, sid)
except GovernanceReportSessionError as e:
print(str(e), file=sys.stderr)
sys.exit(1)
except GovernanceReportError as e:
print(f"Error: {e}", file=sys.stderr)
sys.exit(1)
except OSError as e:
print(f"Error: cannot write governance report: {e}", file=sys.stderr)
sys.exit(1)
print(written)
finally:
engine.db_manager.dispose()
def _display_prog(argv0: str | None = None) -> str:
"""Return the operator-facing command form for this runtime."""
name = Path(argv0 or sys.argv[0] or "").name.lower()
if name in {"data-boar", "data-boar.exe"}:
return "data-boar"
return "python main.py"
def main() -> None:
prog = _display_prog()
parser = argparse.ArgumentParser(
prog=prog,
description=(
"Data Boar — enterprise data discovery and risk governance engine. "
"Loads YAML/JSON config, scans configured databases/filesystems/APIs/shares, "
"stores finding metadata in local SQLite, and generates Excel reports with heatmaps. "
"Run once from the CLI or start a REST API dashboard (LGPD/GDPR/CCPA-aware patterns; "
"additional frameworks via config)."
),
epilog=(
"Configuration:\n"
" - Main config file (YAML or JSON) defines targets (databases, filesystems, APIs, shares),\n"
" detection options and report settings. Default is 'config.yaml' in the current directory.\n"
" - See docs/USAGE.md for a full schema and examples.\n"
"\n"
"CLI examples:\n"
" # One-shot audit with the default config.yaml\n"
f" {prog} --config config.yaml\n"
"\n"
" # One-shot audit tagging tenant/customer and technician/operator\n"
f' {prog} --config config.yaml --tenant "ACME Corp" --technician "Alice"\n'
"\n"
" # One-shot with archive scan + content-type detection (this run only)\n"
f" {prog} --config config.yaml --scan-compressed --content-type-check\n"
f" {prog} --config config.yaml --progress\n"
"\n"
" # Validate config only (loader checks; no scan or API startup)\n"
f" {prog} --config config.yaml --validate-config\n"
"\n"
" # Scan plan: catalog scope + TCP RTT floor (no sampling)\n"
f" {prog} --config config.yaml --plan\n"
"\n"
" # Show rust-regex-stage readiness (paid-tier accelerator; observability)\n"
f" {prog} --config config.yaml --prefilter-status\n"
"\n"
" # Compare two scan sessions (CI: add --fail-on-new-high)\n"
f" {prog} --config config.yaml --diff <session_a> <session_b>\n"
"\n"
" # DSAR-oriented JSON export for one session (stdout or --dsar-output)\n"
f" {prog} --config config.yaml --export-dsar <session_id>\n"
"\n"
" # L1 metadata_manifest for sidecars (stdout or --l1-output; no raw values)\n"
f" {prog} --config config.yaml --export-l1 <session_id>\n"
"\n"
" # Echo findings to a corporate SQL/Mongo sink (Pro/Enterprise; #552)\n"
f" {prog} --config config.yaml --export-findings-sink <session_id>\n"
"\n"
" # L3 transformed_rows (grant-scoped raw values; stdout unless --l3-persist)\n"
f" {prog} --config config.yaml --export-l3 <session_id> --l3-grant grant.json\n"
"\n"
" # Remediation manifest JSON for a third-party plugin (#649)\n"
f" {prog} --config config.yaml --session <session_id> "
f"--export-remediation-manifest remediation.json\n"
"\n"
" # Regenerate Excel + heatmap for an existing session (SQLite only; no re-scan)\n"
f" {prog} --config config.yaml --regenerate-report <session_id>\n"
"\n"
" # Governance Lens GRC Markdown (SQLite only; optional --session)\n"
f" {prog} --config config.yaml --governance-report ./relatorio_grc.md\n"
"\n"
" # Wipe all collected data and generated reports (dangerous, see SECURITY.md)\n"
f" {prog} --config config.yaml --reset-data\n"
"\n"
" # After a pip/pipx upgrade: re-baseline integrity hashes (operator confirm)\n"
f" {prog} --config config.yaml --reconcile-integrity-anchor "
"--confirm-upgrade-to=1.8.0-rc\n"
"\n"
"Web/API examples:\n"
" # HTTPS: PEM cert + key (TLS >= 1.2)\n"
f" {prog} --config config.yaml --web --https-cert-file server.crt --https-key-file server.key\n"
"\n"
" # Plaintext HTTP (explicit risk acceptance; required when not using TLS)\n"
f" {prog} --config config.yaml --web --allow-insecure-http\n"
"\n"
" # Explicit port or bind (same flags as before, still need TLS or --allow-insecure-http)\n"
f" {prog} --config config.yaml --web --allow-insecure-http --port 9090\n"
f" {prog} --config config.yaml --web --allow-insecure-http --host 0.0.0.0\n"
"\n"
" # Zero-config demo (synthetic corpus, loopback dashboard — no config.yaml)\n"
f" {prog} --demo\n"
"\n"
"Once a one-shot scan finishes, an Excel report and heatmap PNG are written under\n"
"the configured report.output_dir (default: current directory). When the API is\n"
"running, you can navigate to the documented endpoints (see README.md) to trigger\n"
"scans, list sessions and download the latest reports through the browser."
),
formatter_class=argparse.RawDescriptionHelpFormatter,
)
parser.add_argument(
"--version",
action="store_true",
help="Show the public product version and exit (no scan or API startup).",
)
parser.add_argument(
"--check-extras",
action="store_true",
help=(
"List optional extras × status × origin (image vs /extras mount) and exit. "
"First step when a connector fails for missing dependencies "
"(see docs/DOCKER_SETUP.md, docs/USAGE.md)."
),
)
parser.add_argument(
"--demo",
action="store_true",
help=(
"Zero-config demo: generate a synthetic filesystem corpus in a temp directory, "
"run an initial scan, and start the dashboard on loopback (127.0.0.1) with "
"plaintext HTTP (--allow-insecure-http). Does not require --config. "
"Temp files are removed when the process exits."
),
)
parser.add_argument(
"--config",
default="config.yaml",
help=(
"Path to the main YAML or JSON configuration file. "
"Defines targets (databases, filesystems, APIs/shares), detection settings and report.output_dir. "
"Default: config.yaml in the current working directory."
),
)
parser.add_argument(
"--web",
action="store_true",
help=(
"Start the REST API/dashboard instead of running a single audit. "
"Uses api.port from the config when present, otherwise falls back to --port (default 8088)."
),
)
parser.add_argument(
"--port",
type=int,
default=8088,
help=(
"API port when --web is enabled. "
"If api.port is set in the config file it takes precedence, unless you explicitly pass --port here. "
"Default: 8088."
),
)
parser.add_argument(
"--host",
default=None,
metavar="ADDR",
help=(
"Bind address when --web is enabled (e.g. 127.0.0.1 or 0.0.0.0). "
"Takes precedence over api.host in config and over the API_HOST environment variable. "
"If omitted, resolution follows config api.host, then API_HOST, then safe default 127.0.0.1. "
"Ignored in one-shot CLI mode."
),
)
parser.add_argument(
"--https-cert-file",
default=None,
metavar="PATH",
help=(
"PEM certificate file for HTTPS when --web is set. "
"Requires --https-key-file (or api.https_cert_file / api.https_key_file in config). "
"TLS >= 1.2. Without cert+key, you must pass --allow-insecure-http for plaintext."
),
)
parser.add_argument(
"--https-key-file",
default=None,
metavar="PATH",
help=(
"PEM private key for HTTPS when --web is set. "
"Requires --https-cert-file (or matching api.* keys in config)."
),
)
parser.add_argument(
"--allow-insecure-http",
action="store_true",
help=(
"EXPLICIT RISK ACCEPTANCE: serve the dashboard over plaintext HTTP. "
"Use only on trusted loopback or lab networks. "
"For production use TLS (cert+key) or terminate TLS on a reverse proxy. "
"Can be set via api.allow_insecure_http in config instead of this flag."
),
)
parser.add_argument(
"--reset-data",
action="store_true",
help=(
"DANGER: wipe all scan sessions, findings and failures from the SQLite database, "
"delete generated Excel reports and heatmap PNGs under report.output_dir, "
"and record an immutable data_wipe_log entry with the reason. "
"Intended for lab/demo environments; review SECURITY.md before using in production."
),
)
parser.add_argument(
"--reconcile-integrity-anchor",
action="store_true",
help=(
"Re-baseline the SQLite integrity anchor after an official package upgrade "
"(#1262). Must be paired with --confirm-upgrade-to=<installed-version> "
"(must match the running package version). Does not auto-reconcile on "
"semver change. Incompatible with --web, --reset-data, scans, and exports."
),
)
parser.add_argument(
"--confirm-upgrade-to",
metavar="VERSION",
default=None,
help=(
"With --reconcile-integrity-anchor: installed version string the operator "
"confirms (must equal the running package version). Required together."
),
)
parser.add_argument(
"--export-audit-trail",
metavar="PATH",
nargs="?",
const="-",
default=None,
help=(
"Export a JSON audit trail from SQLite (data_wipe_log, session summary, "
"maturity_assessment_integrity when applicable; future: integrity anchor). "
"PATH optional: omit or '-' for stdout; "
"otherwise write to PATH. Does not modify the database. "
"Incompatible with --web and --reset-data."
),
)
parser.add_argument(
"--validate-config",
action="store_true",
help=(
"Validate config structure, connector types, and required keys per target; "
"warn on unset *_from_env vars and missing optional SQL driver packages "
"(offline import probe). Also reports rust-regex-stage / accelerator "
"readiness (#1411 / #1414; observability only). No connections, scan, or --web. "
"On errors, does not create output directories or the sqlite file. "
"Exit 0 when valid, 1 on errors. Incompatible with --web, --reset-data, "
"and --export-audit-trail."
),
)
parser.add_argument(
"--plan",
action="store_true",
help=(
"Print a scan plan and exit: enumerate SQL/Snowflake catalog "
"(tables/columns, no sampling), TCP-connect RTT to each target peer "
"(same SSRF / allow_private_networks guard as a live scan; TCP "
"connect uses the guard-pinned IP, not a second DNS lookup; private "
"peers are skipped and reported, not probed), "
"and an RTT-floor time estimate (1 sample query/column + 1 row-estimate/"
"table + catalog get_columns/table). Latency: local = loopback or RTT "
"< 5 ms; lan = 5–20 ms; remote = RTT ≥ 20 ms. Warns (does not abort) "
"when the floor is ≥ 10 min, or remote RTT ≥ 50 ms with ≥ 200 columns. "
"Incompatible with --web, --validate-config, --reset-data, and exports."
),
)
parser.add_argument(
"--resume",
metavar="SESSION",
dest="resume_session",
default=None,
help=(
"Resume an interrupted SQL/Snowflake scan session (UUID): skip tables "
"already completed in that session, continue the rest (#1330). "
"A completed session is a no-op (does not re-scan). "
"Incompatible with --plan, --web, --reset-data, --diff, and --validate-config."
),
)
parser.add_argument(
"--prefilter-status",
action="store_true",
help=(
"Print rust-regex-stage / prefilter readiness for this config as JSON "
"(active, name, backend rust|python, tier, reason, engine) and exit. "
"Observability only — does not change findings (#1411 / #1412)."
),
)
parser.add_argument(
"--diff",
nargs=2,
metavar=("SESSION_A", "SESSION_B"),
dest="diff_sessions",
help=(
"Compare findings between two scan sessions by UUID. "
"Prints new, resolved, and severity-changed rows. "
"Use --fail-on-new-high for CI exit 1 when new HIGH findings appear."
),
)
parser.add_argument(
"--fail-on-new-high",
action="store_true",
dest="fail_on_new_high",
help=(
"With --diff: exit 1 when SESSION_B has new HIGH-sensitivity findings "
"vs SESSION_A (CI regression gate)."
),
)
parser.add_argument(
"--export-dsar",
metavar="SESSION_ID",
dest="export_dsar",
default=None,
help=(
"Export findings for SESSION_ID as DSAR-ready JSON (LGPD Art. 18 / "
"GDPR Art. 15). Metadata-first by default; use --dsar-include-samples "
"only when stored sample fields must be included. Print to stdout or "
"--dsar-output PATH. Incompatible with --web and --reset-data."
),
)
parser.add_argument(
"--dsar-output",
metavar="PATH",
dest="dsar_output",
default=None,
help="Write DSAR export to PATH instead of stdout. Requires --export-dsar.",
)
parser.add_argument(
"--dsar-include-samples",
action="store_true",
dest="dsar_include_samples",
help=(
"With --export-dsar: include raw sample fields from finding rows when "
"present (increases disclosure risk; SQLite stores metadata only by default)."
),
)
parser.add_argument(
"--export-l1",
metavar="SESSION_ID",
dest="export_l1",
default=None,
help=(
"Export findings for SESSION_ID as an L1 metadata_manifest JSON "
"(SDK contract pin; metadata only — never raw samples). Print to "
"stdout or --l1-output PATH. Unknown or empty session → empty "
"findings, exit 0. Fail-closed: contract violation aborts (exit 1). "
"Incompatible with --web and --reset-data."
),
)
parser.add_argument(
"--l1-output",
metavar="PATH",
dest="l1_output",
default=None,
help="Write L1 metadata_manifest JSON to PATH instead of stdout. Requires --export-l1.",
)
parser.add_argument(
"--export-findings-sink",
metavar="SESSION_ID",
dest="export_findings_sink",
default=None,
help=(
"Echo SESSION_ID findings from local SQLite to the configured "
"findings_sink (SQL Pro / MongoDB Enterprise). Metadata-only by "
"default. If findings_sink.include_sample_content is true, also pass "
"--allow-sample-export (LGPD Art. 46) or the command exits 1. "
"Incompatible with --web and --reset-data."
),
)
parser.add_argument(
"--allow-sample-export",
action="store_true",
dest="allow_sample_export",
help=(
"With --export-findings-sink: acknowledge sample_content export when "
"findings_sink.include_sample_content is true (LGPD Art. 46). "
"Required for that YAML flag; never implied."
),
)
parser.add_argument(
"--export-l3",
metavar="SESSION_ID",
dest="export_l3",
default=None,
help=(
"Export grant-scoped L3 transformed_rows JSON for SESSION_ID "
"(SDK contract pin; INPUT rows carry raw value). Requires --l3-grant. "
"Default is ephemeral stdout/pipe. Disk write only with --l3-persist PATH. "
"Paid tier (not Community). Out-of-grant column → exit 4; missing grant "
"or Community → exit 3; persist containment not proven → exit 5. "
"Incompatible with --web and --reset-data."
),
)
parser.add_argument(
"--l3-grant",
metavar="PATH",
dest="l3_grant",
default=None,
help=(
"JSON grant for --export-l3 (grant_id, target, table, columns). "
"Never projects the whole table. Required with --export-l3."
),
)
parser.add_argument(
"--l3-persist",
metavar="PATH",
dest="l3_persist",
default=None,
help=(
"Explicitly write L3 JSON (raw values) to PATH. POSIX: owner-only "
"mode 0600. Windows: apply with icacls; verify by well-known SID "
"(not localized names). SYSTEM/Administrators may remain; "
"OWNER RIGHTS is owner-equivalent. Unproven containment → exit 5 "
"unless --l3-allow-unprotected. Omitted = stdout only; never a "
"default. Requires --export-l3."
),
)
parser.add_argument(
"--l3-column",
metavar="NAME",
dest="l3_column",
action="append",
default=None,
help=(
"With --export-l3: project only this grant column (repeatable). "
"A name outside the grant is refused (exit 4)."
),
)
parser.add_argument(
"--l3-max-rows",
metavar="N",
dest="l3_max_rows",
type=int,
default=100,
help=(
"With --export-l3: per-column row cap (default 100, hard max 10000). "
"Confirmation-window bound — not a full-table dump."
),
)
parser.add_argument(
"--l3-allow-unprotected",
dest="l3_allow_unprotected",
action="store_true",
help=(
"With --l3-persist: if owner containment cannot be proven, still "
"keep the file and set audit containment=not_enforced. Never silent. "
"Requires --export-l3 and --l3-persist. Default is fail-closed (exit 5)."
),
)
parser.add_argument(
"--session",
metavar="SESSION_ID",
dest="session_id",
default=None,
help=(
"Scan session UUID for session-scoped exports. Required with "
"--export-remediation-manifest; optional with --governance-report."
),
)
parser.add_argument(
"--export-remediation-manifest",
metavar="PATH",
dest="export_remediation_manifest",
default=None,
help=(
"Write a remediation-plugin JSON manifest (schema v1) for --session to PATH. "
"Metadata only (connection_ref, locations, pii_type) — no raw PII and no "
"credentials. Enterprise-tier feature (open in licensing.mode off / OPEN). "
"Incompatible with --web, --reset-data, --export-audit-trail, --export-dsar, "
"--validate-config, --diff, and --regenerate-report."
),
)
parser.add_argument(
"--regenerate-report",
metavar="SESSION_ID",
dest="regenerate_report",
default=None,
help=(
"Regenerate Excel workbook and heatmap PNG for SESSION_ID from the "
"configured SQLite database (also writes learned_patterns when enabled). "
"No live target scan and no --web. Incompatible with --web, --reset-data, "
"--validate-config, --diff, --export-dsar, --export-remediation-manifest, "
"--export-audit-trail, and --governance-report."
),
)
parser.add_argument(
"--governance-report",
nargs="?",
const="",
default=None,
metavar="PATH",
dest="governance_report",
help=(
"Write a Governance Lens GRC Markdown report for an existing SQLite session "
"(pandoc-ready; see config/pandoc_governance.yaml). PATH is optional — "
"defaults under report.output_dir. Use --session to pick a session; "
"otherwise the latest session is used. Requires governance.enabled and "
"Pro+ tier. Incompatible with --web, --reset-data, and other export modes."
),
)
parser.add_argument(
"--tenant",
default=None,
help=(
"Optional customer/tenant name for this scan. "
"Stored in the session metadata and included in the Excel report header for traceability."
),
)
parser.add_argument(
"--technician",
default=None,
help=(
"Optional name of the technician/operator responsible for this scan. "
"Also stored in session metadata and shown in the report header."
),
)
parser.add_argument(
"--scan-compressed",
action="store_true",
help=(
"When set, act as if file_scan.scan_compressed is true for this run: "
"scan inside supported archives (zip, tar, 7z, etc.). May increase run time and I/O."
),
)
parser.add_argument(
"--content-type-check",
action="store_true",
dest="content_type_check",
help=(
"When set, act as if file_scan.use_content_type is true for this run: "
"infer file format from magic bytes (first bytes of each file), not only extension—"
"helps find renamed or cloaked files. Does not dispatch compressed archives: "