Skip to content
Merged
60 changes: 43 additions & 17 deletions src/skillspector/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -1001,6 +1001,40 @@ def _ledger_work_identity(entry: dict[str, object]) -> str:
return f"{record_value}:{entry.get('phase', '')}"


def _ledger_work_identities(value: object) -> dict[str, str]:
"""Map each child ledger row's own work ID to the identity it was built from.

A status ``planned_work`` target carries only the child's work ID, path and
range, not the identity behind it. Rows whose identity is not the
analyzer's own -- the static_yara rule-set row is ``rule_set:static``, not
``static_yara`` -- must be re-scoped with that same identity in the status
path, or the status target and its ledger row get different scoped IDs
and the target is dropped as unretained.
"""
identities: dict[str, str] = {}
for event in _coerce_dict_list(value):
work_id = event.get("work_id")
if isinstance(work_id, str) and work_id:
identities[work_id] = _ledger_work_identity(event)
return identities


def _source_scoped_work_id(identity: str, item: dict[str, object]) -> str:
"""Build the scoped work ID for an already re-pathed ledger row or status target.

Shared by :func:`_source_aware_ledger` and :func:`_source_aware_status_events`
so the two scoping paths cannot derive different IDs for the same work.
"""
start_line = item.get("start_line")
end_line = item.get("end_line")
return inspection_work_id(
identity,
str(item.get("path", "SKILL.md")),
start_line if isinstance(start_line, int) else None,
end_line if isinstance(end_line, int) else None,
)


def _source_aware_ledger(
value: object,
*,
Expand All @@ -1026,15 +1060,7 @@ def _source_aware_ledger(
for item in ids
if isinstance(item, str)
]
scoped_path = str(entry.get("path", "SKILL.md"))
start_line = entry.get("start_line")
end_line = entry.get("end_line")
entry["work_id"] = inspection_work_id(
_ledger_work_identity(entry),
scoped_path,
start_line if isinstance(start_line, int) else None,
end_line if isinstance(end_line, int) else None,
)
entry["work_id"] = _source_scoped_work_id(_ledger_work_identity(entry), entry)
events.append(entry)
return events

Expand All @@ -1047,8 +1073,10 @@ def _source_aware_status_events(
source_digest: str,
retained_work_ids: set[str],
max_planned_work: int,
work_identities: dict[str, str] | None = None,
) -> list[dict[str, object]]:
statuses: list[dict[str, object]] = []
identities = work_identities or {}
planned_retained = 0
for status in _coerce_dict_list(value):
if len(statuses) >= _TRANSITIVE_MAX_STATUS_EVENTS:
Expand All @@ -1070,14 +1098,11 @@ def _source_aware_status_events(
path = scoped_target.get("path")
if isinstance(path, str) and path:
scoped_target["path"] = _transitive_component_key(source_identity, path)
start_line = scoped_target.get("start_line")
end_line = scoped_target.get("end_line")
scoped_target["work_id"] = inspection_work_id(
analyzer_id,
str(scoped_target.get("path", "SKILL.md")),
start_line if isinstance(start_line, int) else None,
end_line if isinstance(end_line, int) else None,
)
# Re-scope with the identity the matching ledger row used, so
# both paths agree on the scoped ID; the analyzer ID is only the
# fallback for targets with no child ledger row.
identity = identities.get(str(target.get("work_id", "")), analyzer_id)
scoped_target["work_id"] = _source_scoped_work_id(identity, scoped_target)
if scoped_target["work_id"] not in retained_work_ids:
continue
scoped_work.append(scoped_target)
Expand Down Expand Up @@ -1375,6 +1400,7 @@ def _scope_finding(finding: Finding) -> Finding:
source_digest=source_digest,
retained_work_ids=retained_work_ids,
max_planned_work=len(retained_work_ids),
work_identities=_ledger_work_identities(child_result.get("inspection_ledger")),
)
child_metadata = _decorate_component_metadata(
_coerce_component_metadata(child_result.get("component_metadata")),
Expand Down
30 changes: 30 additions & 0 deletions src/skillspector/inspection_ledger.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,15 @@ class LedgerRecordType(StrEnum):
WORK_ITEM = "work_item"
SYSTEM = "system"
SCOPE_BOUNDARY = "scope_boundary"
# An analyzer's own configuration (e.g. its YARA rule set), not a skill
# artifact. Its ``path`` is only a report-safe label, and every relative
# path is also a legal file name, so path-keyed accounting must exclude
# these records by type -- no choice of label can be collision-free.
RULE_SET = "rule_set"


RULE_SET_SCOPE: Final = "rule_set"
"""Public ``scope`` of exception rows describing a rule set rather than a file."""


class LedgerReason(StrEnum):
Expand Down Expand Up @@ -295,6 +304,7 @@ class InspectionLedgerException(TypedDict):
error_class: NotRequired[str]
analyzers: NotRequired[list[str]]
fatal: NotRequired[bool]
scope: NotRequired[str]


class AnalysisCompleteness(TypedDict):
Expand Down Expand Up @@ -589,6 +599,7 @@ def _exception(
error_class: str | None = None,
analyzers: Iterable[str] = (),
fatal: bool,
scope: str | None = None,
) -> InspectionLedgerException:
"""Build the public, safe projection of one exceptional ledger fact."""
exception: InspectionLedgerException = {
Expand All @@ -606,9 +617,16 @@ def _exception(
exception["analyzers"] = analyzer_ids
if error_class:
exception["error_class"] = error_class
if scope:
exception["scope"] = scope
return exception


def _is_rule_set_record(event: Mapping[str, object]) -> bool:
"""Return whether a ledger row describes a rule set rather than an artifact."""
return event.get("record_type") == LedgerRecordType.RULE_SET


def _exception_from_event(
event: InspectionLedgerEvent, *, fatal: bool
) -> InspectionLedgerException:
Expand All @@ -629,6 +647,7 @@ def _exception_from_event(
error_class=event.get("error_class"),
analyzers=[str(event.get("analyzer_id", ""))],
fatal=fatal,
scope=RULE_SET_SCOPE if _is_rule_set_record(event) else None,
)


Expand All @@ -647,6 +666,9 @@ def _merge_exception_projection(
exception["start_line"],
exception["end_line"],
exception.get("error_class"),
# A rule-set row and a real file's row can share a path label;
# they must never merge into one public row.
exception.get("scope"),
)
existing = grouped.get(key)
if existing is None:
Expand Down Expand Up @@ -926,9 +948,17 @@ def accounting_error(path: object = None) -> None:
ledger_exceptions = _merge_exception_projection(exceptional_rows)
scope_exclusions = _merge_exception_projection(scope_rows)

# Rule-set rows describe an analyzer's configuration, not an artifact. Their
# path is only a label, so folding them into per-component coverage would
# charge the rule set's incompleteness to any real file with that name.
rule_set_work_ids = {
str(event.get("work_id", "")) for event in events if _is_rule_set_record(event)
}
per_component: dict[str, list[LedgerOutcome]] = {component: [] for component in components}
if primary_targets:
for _analyzer_id, target, matches in primary_targets:
if str(target.get("work_id", "")) in rule_set_work_ids:
continue
path = _safe_path(target.get("path"), components)
outcomes = per_component.setdefault(path, [])
if len(matches) == 1:
Expand Down
Loading
Loading