Skip to content

Commit 63b47c1

Browse files
committed
Finish llm module unification: canonical paths, no compat shims
1 parent fcc9807 commit 63b47c1

103 files changed

Lines changed: 9436 additions & 1889 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎CHANGELOG.md‎

Lines changed: 38 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -13,22 +13,52 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
1313
### Changed
1414
- Threat-modeling phase disabled by default (`enable_threat_modeling = false`) — it generated a static STRIDE template rather than code-derived analysis
1515

16-
---
16+
## [1.1.0] - 2026-09-15
1717

18-
## [1.1.0] - 2026-08-12
18+
First public release.
1919

2020
### Added
21-
- 24-phase scanner pipeline with Validate phase
22-
- P1-P5 paper integration tracks (VulTriage, VulnLLM-R, MoCQ, PacVD, AgentFlow) behind `enabled = false` defaults
21+
- `baco doctor` pre-flight checks: config parse, per-phase LLM slot validation (warns on phases without `api_key` that will be skipped), semgrep/python3 presence, output dir writability, disk space
22+
- `baco eval` detection-regression suite: 10 labeled targets with ground-truth oracles, precision/recall/F1, CI gate on pass-rate (`BACO_EVAL_FLOOR`, default 0.70)
23+
- `baco init [PATH]` config scaffolding with language detection and preset suggestions
24+
- Scan-health report (console + JSON section): per-phase run/skipped-with-reason, file counters (indexed/analyzed/dropped/chunked), LLM call outcomes by error class, token and cost totals per phase, blind-scan warning when all LLM phases are skipped
25+
- Pipeline profiles: `[scanner] profile = "core"` (default, 23 phases) or `"all"` (experimental phases included)
26+
- Presets: `django`, `laravel`, `cpp`; inline `custom_rules` Semgrep YAML in presets (self-contained detection packages)
27+
- Per-language default Semgrep rulesets derived from `project.languages`
28+
- Environment-variable bridges for all six LLM phase slots (`LLM_DISCOVERY_KEY`, `LLM_VERIFICATION_KEY`, `LLM_AGGREGATION_KEY`, `LLM_STATIC_ANALYSIS_KEY`, `LLM_SECURITY_AGENT_VERIFICATION_KEY`, `LLM_THREAT_MODELING_KEY`)
29+
- LLM cost transparency: optional `[llm.pricing]` table, token counts per phase and model surfaced in the health report
30+
- Configurable never-submit confidence filter (`never_submit_enabled`, `never_submit_multiplier`)
2331
- `max_reasoning_tokens` field for LLM config
2432
- Agent scaffold modules: `call_graph_paths`, `fn_lookup`
25-
- `docs/README.md` documentation index
33+
- Chunked analysis of oversized files via tree-sitter (previously dropped or truncated at 8 KB)
34+
- Config-driven per-language hook registry (`[knowledge.hook_registry.<language>]`) and `required_security_primitives` verification prompts
2635

2736
### Changed
28-
- Internal refactoring: split oversized modules into directory modules, extracted shared tree-sitter parser, consolidated tests under `tests/`
29-
- MSRV-compatible clippy fixes (replaced `is_none_or`, `is_multiple_of`)
37+
- Pipeline defined once in a declarative PhaseSpec table (checkpoint transitions, profile filtering, progress messages and docs all derive from it)
38+
- Semgrep severity read from `extra.severity` first (ERROR/WARNING/INFO mapping); `check_id` keywords can only raise, never lower
39+
- HTML report rendered via embedded minijinja templates
40+
- LLM internals consolidated under `src/llm/` (client, cache, metrics, traits) with canonical import paths
41+
- `max_file_size_kb` defaults to 512; early-termination threshold counts medium+ findings only
42+
- Verification batch parser accepts responses without `index` (positional fallback with warning)
43+
- Multi-model round-robin preserved per phase (`models` list no longer collapsed)
44+
- Internal refactoring: modules split into directory modules, shared tree-sitter parser, tests consolidated under `tests/`
3045

3146
### Fixed
47+
- LLM endpoint URL doubling (`/v1/v1/`) causing silent 404s
48+
- Discovery phase dropping findings that already had LLM evidence
49+
- API key printed in log output during static analysis
50+
- HTML report broken by unclosed `<style>`/`<div>` tags
51+
- Semgrep multi-hit findings attributed to placeholder `multiple_files` path
52+
- Custom-rule IDs prefixed with the materialized temp-file stem
53+
- Discovery enrichment silently overwriting detector severity (enrichment is raise-only)
54+
- LLM findings without a `line` field silently dropped (now salvaged with a warning)
55+
- JSON schema fields serialized as `type_` instead of `type`
3256
- C function-name extraction handles `function_declarator` tree-sitter node
3357
- Call-graph builder treats uncalled functions as entry points
34-
- Phase count references updated from 20 to 24 throughout
58+
59+
### Removed
60+
- MultiVerifier stub phase (fabricated hash-based verdicts)
61+
- Dead modules: report aggregation, scan diff, worktree staging, severity rubric, phase scaffolding (~2,500 lines)
62+
- Phantom `[llm.phases.*]` config slots and unused configuration keys
63+
64+

‎README.md‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -60,6 +60,15 @@ cp config.toml my-config.toml
6060
# Verify findings
6161
./target/release/baco verify --input findings.json
6262

63+
# Scaffold a starter config (detects languages, suggests a preset)
64+
./target/release/baco init /path/to/project
65+
66+
# List built-in presets
67+
./target/release/baco preset list
68+
69+
# Resume an interrupted scan from its checkpoint
70+
./target/release/baco resume --checkpoint baco-output/checkpoint.json
71+
6372
# Scan options
6473
./target/release/baco scan --config my.toml --dry-run # Print estimate and exit
6574
./target/release/baco scan --config my.toml --target /path # Override target path

‎config.example.toml‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -329,3 +329,8 @@ max_runs = 5
329329
# secret_storage = "vault" # "vault" → ${VAULT_TOKEN} refs are placeholders, NOT leaks
330330
# risk_tolerance = "low" # does NOT mean only report criticals — prioritize ruthlessly, report everything real
331331
# severity_rules = { "XSS" = "Critical", "SQLi" = "Critical" } # OVERRIDE lines for specific rules
332+
333+
[eval]
334+
# Minimum aggregate pass-rate for `baco eval` (0.0-1.0).
335+
# The BACO_EVAL_FLOOR environment variable overrides this value.
336+
floor = 0.70

‎docs/configuration.md‎

Lines changed: 89 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -775,4 +775,92 @@ baco scan --config my.toml --preset my-project
775775
| `agent` | `enabled`, `max_turns`, `tool_timeout_secs` |
776776
| `knowledge` | `fp_patterns` (map of CWE → list of false-positive indicator strings), `required_security_primitives` (map of language → list of required primitives), `hook_registry` (map of language → HookRegistryLanguageConfig with `hook_label`, `registrations` regexes with optional `(?P<hook>)` capture, `handler_patterns` override) |
777777

778-
Unset fields keep the base `ScannerConfig` default; CLI flags still override the preset.
778+
Unset fields keep the base `ScannerConfig` default; CLI flags still override the preset.
779+
780+
## Eval suite
781+
782+
The `[eval]` section configures the detection-regression suite run by `baco eval`.
783+
784+
| Key | Type | Default | Description |
785+
| ------ | ----- | ------- | ----------------------------------------------------------------------------------------------- |
786+
| `floor` | f32 | `0.70` | Minimum aggregate pass-rate (0.0-1.0). The suite fails when the aggregate does not strictly exceed it. `BACO_EVAL_FLOOR` overrides it per invocation. |
787+
788+
```toml
789+
[eval]
790+
floor = 0.9
791+
```
792+
793+
794+
## Agent scaffold
795+
796+
The `[agent_scaffold]` section configures the call-graph-guided agent scaffold used during security-agent verification.
797+
798+
| Key | Type | Default | Description |
799+
| ------------------- | ---- | ------- | ---------------------------------------------------- |
800+
| `enabled` | bool | `false` | Enables the agent scaffold modules |
801+
| `max_rounds` | u8 | built-in | Maximum interaction rounds per target function |
802+
| `paths_per_target` | u8 | built-in | Number of call-graph paths to sample per target |
803+
804+
```toml
805+
[agent_scaffold]
806+
enabled = false
807+
```
808+
809+
## Citation verification
810+
811+
The `[citation_verification]` section controls verification that finding citations (file + line) actually resolve in the target source, downgrading unverifiable findings.
812+
813+
| Key | Type | Default | Description |
814+
| --------- | ---- | ------- | ---------------------------------------------------- |
815+
| `enabled` | bool | `false` | Verify finding citations against source files |
816+
817+
```toml
818+
[citation_verification]
819+
enabled = false
820+
```
821+
822+
## Prior runs
823+
824+
The `[prior_runs]` section configures the cross-run findings history used for skip directives and coverage-gap targeting on subsequent scans.
825+
826+
| Key | Type | Default | Description |
827+
| --------- | ----- | ------- | -------------------------------------------------------- |
828+
| `enabled` | bool | `false` | Enable the prior-runs store |
829+
| `max_runs`| usize | built-in | Maximum number of prior runs retained |
830+
831+
```toml
832+
[prior_runs]
833+
enabled = false
834+
```
835+
836+
## Policy sampling
837+
838+
The `[policy_sampling]` section configures policy-based sampling rounds (VulnLLM-R style generation).
839+
840+
| Key | Type | Default | Description |
841+
| --------- | ---- | ------- | ------------------------------------------------- |
842+
| `enabled` | bool | `false` | Enable policy-based generation |
843+
| `samples` | u8 | `4` | Number of sampling rounds used to build the policy |
844+
845+
```toml
846+
[policy_sampling]
847+
enabled = false
848+
```
849+
850+
## Tickets
851+
852+
The `[tickets]` section configures cross-referencing findings against external ticket systems.
853+
854+
Each entry in `systems` describes one ticket backend:
855+
856+
| Key | Type | Description |
857+
| ------------- | --------------- | -------------------------------------------------- |
858+
| `system_type` | string | `"github"` or `"gitlab"` |
859+
| `url` | string | Base URL of the system |
860+
| `api_key` | string (option) | API token; `TICKET_GITHUB_KEY` / `TICKET_GITLAB_KEY` environment variables override it |
861+
862+
```toml
863+
[[tickets.systems]]
864+
system_type = "github"
865+
url = "https://github.com/org/repo"
866+
```

‎eval/README.md‎

Lines changed: 12 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -43,14 +43,20 @@ findings fixture (`eval/findings/<target>.json`) against its oracle, prints a
4343
per-target pass-rate table plus the aggregate, and exits non-zero when the
4444
aggregate does not exceed the floor.
4545

46-
**`BACO_EVAL_FLOOR` is the single knob.** It is a fraction in `0.0..=1.0`
47-
(default `0.70`), read from the environment. The suite passes only when the
48-
aggregate pass-rate (total matched / total expected across all targets) is
49-
*strictly greater* than the floor; an unset or empty value falls back to the
50-
default. **The default of 0.70 is provisional pending maintainer sign-off.**
46+
**`[eval] floor` in the config is the knob.** Set a fraction in `0.0..=1.0`
47+
(default `0.70`) in your `baco.toml`; the suite passes only when the aggregate
48+
pass-rate (total matched / total expected across all targets) is *strictly
49+
greater* than the floor. The `BACO_EVAL_FLOOR` environment variable overrides
50+
the config value (unset or empty falls back to config).
51+
52+
```toml
53+
[eval]
54+
# Fail unless the suite exceeds a stricter floor
55+
floor = 0.9
56+
```
5157

5258
```bash
53-
# Fail unless the suite exceeds a stricter floor
59+
# Or override per invocation via the environment
5460
BACO_EVAL_FLOOR=0.9 baco eval
5561
```
5662

‎presets/django.toml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -182,7 +182,7 @@ max_llm_calls = 250
182182
]
183183

184184
[knowledge.hook_registry.python]
185-
hook_label = ""
185+
hook_label = "urlpatterns"
186186
registrations = [
187187
'''(?si)\bpath\s*\(\s*[\x27\x22](?P<hook>[^\x27\x22]*)[\x27\x22]\s*,\s*''',
188188
'''(?si)\bre_path\s*\(\s*[\x27\x22](?P<hook>[^\x27\x22]*)[\x27\x22]\s*,\s*''',

‎presets/laravel.toml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -198,7 +198,7 @@ sink_patterns = [
198198
]
199199

200200
[knowledge.hook_registry.php]
201-
hook_label = ""
201+
hook_label = "rest_route"
202202
registrations = [
203203
'''(?si)\bRoute::\s*(?:get|post|put|patch|delete|options|any)\s*\(\s*[\x27\x22](?P<hook>[^\x27\x22]*)[\x27\x22]\s*,\s*''',
204204
]

‎src/agent/session.rs‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -121,6 +121,7 @@ impl AgentSession {
121121
loop {
122122
turn += 1;
123123
if turn > self.max_turns {
124+
turn -= 1;
124125
tracing::warn!("Max turns ({}) reached", self.max_turns);
125126
break;
126127
}
@@ -370,6 +371,7 @@ impl AgentSession {
370371
loop {
371372
turn += 1;
372373
if turn > self.max_turns {
374+
turn -= 1;
373375
tracing::warn!("Max turns ({}) during verification", self.max_turns);
374376
break;
375377
}

‎src/citation_verification.rs‎

Lines changed: 24 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -47,6 +47,18 @@ pub fn verify_citations(
4747

4848
let file_path = project_path.join(&finding.file_path);
4949

50+
// Reject absolute paths and path traversal attempts
51+
if finding.file_path.starts_with('/') || finding.file_path.contains("..") {
52+
finding.confidence_score *= 0.5;
53+
let note = format!(
54+
"citation verification failed: path traversal rejected: {}",
55+
finding.file_path
56+
);
57+
append_verification_note(&mut finding.verification_notes, &note);
58+
report.failed += 1;
59+
continue;
60+
}
61+
5062
// Check if file exists and is readable
5163
let file_content = match fs::read_to_string(&file_path) {
5264
Ok(content) => content,
@@ -64,6 +76,18 @@ pub fn verify_citations(
6476

6577
// Check line number if present
6678
if let Some(line_num) = finding.line_number {
79+
// Line 0 is invalid (lines are 1-indexed)
80+
if line_num == 0 {
81+
finding.confidence_score *= 0.5;
82+
let note = format!(
83+
"citation verification failed: line 0 is invalid (1-indexed): {}",
84+
finding.file_path
85+
);
86+
append_verification_note(&mut finding.verification_notes, &note);
87+
report.failed += 1;
88+
continue;
89+
}
90+
6791
let line_count = file_content.lines().count();
6892

6993
if line_num as usize > line_count {

‎src/config/mod.rs‎

Lines changed: 22 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,26 @@ use std::fmt;
2121
use std::fs;
2222
use std::path::PathBuf;
2323

24+
/// Eval-suite settings (`[eval]` section).
25+
#[derive(Debug, Clone, Serialize, Deserialize)]
26+
pub struct EvalConfig {
27+
/// Minimum aggregate pass-rate for `baco eval` (0.0..=1.0).
28+
#[serde(default = "default_eval_floor")]
29+
pub floor: f32,
30+
}
31+
32+
fn default_eval_floor() -> f32 {
33+
0.70
34+
}
35+
36+
impl Default for EvalConfig {
37+
fn default() -> Self {
38+
Self {
39+
floor: default_eval_floor(),
40+
}
41+
}
42+
}
43+
2444
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
2545
pub struct ScannerConfig {
2646
#[serde(default)]
@@ -64,6 +84,8 @@ pub struct ScannerConfig {
6484
#[serde(default)]
6585
pub pacvd: PacvdConfig,
6686
#[serde(default)]
87+
pub eval: EvalConfig,
88+
#[serde(default)]
6789
pub agent_flow: AgentFlowConfig,
6890
#[serde(default)]
6991
pub vuln_spec: VulnSpecConfig,

0 commit comments

Comments
 (0)