diff --git a/src/skillspector/cli.py b/src/skillspector/cli.py index 3a4d1b520..8308fadf9 100644 --- a/src/skillspector/cli.py +++ b/src/skillspector/cli.py @@ -3181,7 +3181,13 @@ def baseline( state = _scan_state(input_path, FormatChoice.json, no_llm) state["baseline_path"] = os.path.abspath(output.expanduser()) result = graph.invoke(state) - findings = effective_findings(result) + # Keep each occurrence's original context and confidence; display + # deduplication cannot reconstruct those fingerprint inputs. + findings = ( + _coerce_findings_list(result["active_findings"]) + if "active_findings" in result + else effective_findings(result) + ) data = build_baseline_dict( findings, reason=reason, diff --git a/src/skillspector/nodes/report.py b/src/skillspector/nodes/report.py index d2e5ae9bd..e0b2f1199 100644 --- a/src/skillspector/nodes/report.py +++ b/src/skillspector/nodes/report.py @@ -1837,6 +1837,7 @@ def report(state: SkillspectorState) -> dict[str, object]: "risk_recommendation": risk_recommendation, "report_body": report_body, "filtered_findings": reported_findings, + "active_findings": active_findings, "suppressed_findings": suppressed, "execution_successful": execution_successful, "analysis_completeness": dict(analysis_completeness), diff --git a/src/skillspector/state.py b/src/skillspector/state.py index ba39b5621..eafb8eeef 100644 --- a/src/skillspector/state.py +++ b/src/skillspector/state.py @@ -280,6 +280,8 @@ class SkillspectorState(TypedDict, total=False): # Compatibility projection emitted only by the report after effective-ID # selection. Meta analysis never stores a second filtered collection. filtered_findings: list[Finding] + # Exact kept occurrences before display deduplication, for baseline generation. + active_findings: list[Finding] # LLM runtime telemetry: each LLM-backed node appends one record (built with # ``llm_call_record``) so the report can detect a *silent degradation* — the diff --git a/tests/unit/test_cli.py b/tests/unit/test_cli.py index 96dd9203a..796f4df10 100644 --- a/tests/unit/test_cli.py +++ b/tests/unit/test_cli.py @@ -6078,3 +6078,42 @@ def test_cli_baseline_uses_local_cache_for_provider_excluded_findings(tmp_path: written = yaml.safe_load(out.read_text(encoding="utf-8")) assert [entry["file"] for entry in written["fingerprints"]] == [".hidden.md"] assert len(written["fingerprints"][0]["hash"]) == len("sha256:") + 64 + + +def test_baseline_preserves_distinct_context_for_repeated_findings(tmp_path: Path) -> None: + skill = tmp_path / "skill" + skill.mkdir() + (skill / "references").mkdir() + (skill / "SKILL.md").write_text( + "---\nname: demo\ndescription: demo\n---\n\n# Service\n\n" + "The upstream service deletes unused files, and the link dies with no warning.\n", + encoding="utf-8", + ) + (skill / "references" / "notes.md").write_text( + "# Mirror\n\nThe mirror drops stale entries with no warning.\n", encoding="utf-8" + ) + baseline = tmp_path / "baseline.yaml" + generated = runner.invoke(app, ["baseline", str(skill), "--no-llm", "-o", str(baseline)]) + assert generated.exit_code == 0, generated.output + result = runner.invoke( + app, + [ + "scan", + str(skill), + "--no-llm", + "--baseline", + str(baseline), + "--format", + "json", + "--show-suppressed", + ], + ) + assert result.exit_code == 0, result.output + data = json.loads(result.stdout) + assert data["issues"] == [] + entries = yaml.safe_load(baseline.read_text())["fingerprints"] + assert {entry["file"] for entry in entries if entry["rule_id"] == "AR2"} == { + "SKILL.md", + "references/notes.md", + } + assert data["suppressed_count"] >= 2