Compare commits

...

16 Commits

Author SHA1 Message Date
Ahmed Allam
774c888ca8 fix(tui): fingerprint report revisions by history length instead of second-resolution timestamp 2026-09-09 14:43:40 +00:00
Ahmed Allam
725516b474 fix(tui): detect content-only report revisions in the sync fingerprint
The update callback now fires before the in-memory report is replaced, so
a snapshot taken on that notification can still show the old content. The
periodic sync fingerprint tracks each report's updated_at as well as its
id, so the next tick picks the revision up.
2026-09-09 14:35:08 +00:00
Ahmed Allam
9805d1a67d fix(reporting): keep the proxy-outage retry message on evidence-only revisions
An update that only carries http_exchange_ids used to fail with
'No fields to update' when the proxy could not be reached, hiding the
retry guidance. It now returns the outage warning as the error.
2026-09-09 09:18:14 +00:00
Ahmed Allam
618261aa6a fix(reporting): drop unverified HTTP exchange IDs instead of storing them
When the proxy lookup itself fails, the finding is still filed but the
http_exchange_ids are omitted and the result carries a warning telling
the agent to attach them with update_vulnerability_report once the proxy
responds. Unverified IDs are never recorded as evidence.
2026-09-09 09:12:30 +00:00
Ahmed Allam
265df84048 fix(reporting): surface persistence failures to the agent and tolerate proxy outages
- existing_request_ids: use functools.partial so mypy can type the lookup
- _do_create/_do_update/_do_create_dependency: catch any exception raised
  while committing a report and return a structured success:false result
  instead of leaking a generic tool error
- _verify_http_exchange_ids: unknown IDs are still rejected, but a proxy
  lookup failure now keeps the IDs and attaches a warning to the result
  rather than blocking the finding
2026-09-09 09:07:35 +00:00
bearsyankees
8ca05a9d5f Merge remote-tracking branch 'origin/main' into codex/http-evidence 2026-09-07 11:15:13 -04:00
bearsyankees
3d632c3a69 fix(reporting): resolve captured requests and preserve failed revisions 2026-09-07 11:04:41 -04:00
devin-ai-integration[bot]
52b1923347 fix(models): frontier model check matches the model name only, never the provider route (#1280)
Co-authored-by: Ahmed Allam <ahmed39652003@gmail.com>
2026-09-06 11:09:24 -07:00
Ahmed Allam
ff5c8cc8e4 chore: release v1.6.2 2026-09-05 04:22:29 +03:00
Ahmed Allam
afce7d95e8 fix(telemetry): classify setup-mode TUI preflight and preparation failures 2026-09-05 04:08:09 +03:00
Ahmed Allam
2e1db25786 feat(telemetry): classify error beacons by phase and exception class
error events now carry phase (startup/preflight/sandbox_init/agent_setup/
agent_loop) and the exception class name (plus its cause), never the message
or trace. Startup and preflight failures that exit(1) before the scan starts
are beaconed with a stable error_type instead of vanishing. scan_ended
distinguishes budget_exceeded, rate_limited, and headless agent_stopped
from user_exit.
2026-09-05 04:08:09 +03:00
devin-ai-integration[bot]
f4b0416b71 docs: update README and CLI links (#1272)
Co-authored-by: Ahmed Allam <ahmed39652003@gmail.com>
2026-09-04 17:56:08 -07:00
Alex Schapiro
d40497747f fix(reporting): verify HTTP evidence references 2026-09-04 20:29:37 +00:00
Alex Schapiro
9eb3199d10 fix(reporting): reject unpersisted findings 2026-09-04 20:24:25 +00:00
Alex Schapiro
cdbb2ff4cf docs(reporting): clarify HTTP request IDs 2026-09-04 18:49:28 +00:00
Alex Schapiro
677ae1effa feat(reporting): link HTTP exchange evidence 2026-09-04 18:45:53 +00:00
33 changed files with 1094 additions and 275 deletions

View File

@@ -17,6 +17,9 @@
<a href="https://strix.ai"><img src="https://img.shields.io/badge/Website-strix.ai-f0f0f0?style=for-the-badge&logoColor=000000" alt="Website"></a>
[![](https://dcbadge.limes.pink/api/server/strix-ai)](https://discord.gg/strix-ai)
<a href="https://app.strix.ai?utm_source=github&utm_medium=readme&utm_content=badge_cloud"><img src="https://img.shields.io/badge/Strix%20Cloud-app.strix.ai-2b9246?style=for-the-badge&logoColor=white" alt="Strix Cloud"></a>
<a href="https://strix.ai/demo?utm_source=github&utm_medium=readme&utm_content=badge_demo"><img src="https://img.shields.io/badge/Try%20Strix%20Enterprise-555555?style=for-the-badge&logoColor=white" alt="Try Strix Enterprise"></a>
<a href="https://deepwiki.com/usestrix/strix"><img src="https://deepwiki.com/badge.svg" alt="Ask DeepWiki"></a>
<a href="https://github.com/usestrix/strix"><img src="https://img.shields.io/github/stars/usestrix/strix?style=flat-square" alt="GitHub Stars"></a>
<a href="LICENSE"><img src="https://img.shields.io/badge/License-Apache%202.0-3b82f6?style=flat-square" alt="License"></a>
@@ -34,7 +37,7 @@
> [!TIP]
> **New!** Strix integrates seamlessly with GitHub Actions and CI/CD pipelines. Automatically scan for vulnerabilities on every pull request and block insecure code before it reaches production - [Get started with no setup required](https://app.strix.ai).
> **New!** Strix integrates seamlessly with GitHub Actions and CI/CD pipelines. Automatically scan for vulnerabilities on every pull request and block insecure code before it reaches production - [Get started with no setup required](https://app.strix.ai?utm_source=github&utm_medium=readme&utm_content=tip_ci).
---
@@ -94,9 +97,17 @@ strix --target ./app-directory
---
## ☁️ Strix Platform
## Ways to Run Strix
Try the Strix full-stack penetration testing platform at **[app.strix.ai](https://app.strix.ai)** - sign up for free, connect your repos and domains, and launch a pentest in minutes.
- **Open Source** - free, runs locally with Docker and your own LLM key. [Quick Start](https://docs.strix.ai/quickstart)
- **Strix Cloud** - no setup, validated findings, one-click autofix, and PR reviews. [Run a pentest →](https://app.strix.ai?intent=pentest&utm_source=github&utm_medium=readme&utm_content=table_cloud)
- **Enterprise** - SSO, compliance-ready reports, VPC or self-hosted deployment. [Try Strix Enterprise →](https://strix.ai/demo?utm_source=github&utm_medium=readme&utm_content=table_demo)
---
## ☁️ Strix Cloud
Try the Strix full-stack penetration testing platform at **[app.strix.ai](https://app.strix.ai?utm_source=github&utm_medium=readme&utm_content=cloud_heading)** - sign up for free, connect your repos and domains, and launch a pentest in minutes.
- **Validated findings with PoCs** - every vulnerability includes a working proof-of-concept exploit and reproduction steps
- **One-click autofix** - AI-generated security patches as ready-to-merge pull requests
@@ -104,7 +115,13 @@ Try the Strix full-stack penetration testing platform at **[app.strix.ai](https:
- **DevSecOps integrations** - GitHub, GitLab, Bitbucket, Slack, Jira, Linear, and CI/CD pipelines
- **Continuous learning** - AI that builds on past findings, adapts to your codebase, and reduces false positives over time
[**Start your first pentest →**](https://app.strix.ai)
[**Run a pentest →**](https://app.strix.ai?intent=pentest&utm_source=github&utm_medium=readme&utm_content=cloud_cta)
## 🏢 Enterprise
Get the same Strix experience with enterprise-grade controls: SSO (SAML/OIDC), custom compliance-ready penetration testing reports (SOC 2, ISO 27001, PCI DSS), dedicated support and SLA, custom deployment options (VPC or self-hosted), BYOK model support, and tailored AI pentesting agents optimized for your environment.
[**Try Strix Enterprise →**](https://strix.ai/demo?utm_source=github&utm_medium=readme&utm_content=enterprise_cta)
---
@@ -333,10 +350,6 @@ Each server's tools are namespaced by `name`, for example `github_list_issues`.
See the [LLM Providers documentation](https://docs.strix.ai/llm-providers/overview) for all supported providers including Vertex AI, Bedrock, Azure, and local models.
## Enterprise Pentesting
Get the same Strix experience with [enterprise-grade](https://strix.ai/demo) controls: SSO (SAML/OIDC), custom compliance-ready penetration testing reports (SOC 2, ISO 27001, PCI DSS), dedicated support & SLA, custom deployment options (VPC/self-hosted), BYOK model support, and tailored AI pentesting agents optimized for your environment. [Learn more](https://strix.ai/demo).
## Documentation
Full documentation is available at **[docs.strix.ai](https://docs.strix.ai)** - including detailed guides for usage, CI/CD integrations, skills, and advanced configuration.

View File

@@ -1,6 +1,6 @@
[project]
name = "strix-agent"
version = "1.6.1"
version = "1.6.2"
description = "Open-source AI Hackers for your apps"
readme = "README.md"
license = "Apache-2.0"

View File

@@ -346,6 +346,9 @@ echo -e "${MUTED}For more information visit ${NC}https://strix.ai"
echo -e "${MUTED}Supported models ${NC}https://docs.strix.ai/llm-providers/overview"
echo -e "${MUTED}Join our community ${NC}https://discord.gg/strix-ai"
echo ""
echo -e "${MUTED}Run a pentest in Strix Cloud ${NC}https://app.strix.ai"
echo -e "${MUTED}Enterprise ${NC}https://strix.ai/demo"
echo ""
echo -e "${YELLOW}${NC} Run ${MUTED}source ~/.$(basename $SHELL)rc${NC} or open a new terminal"
echo ""

View File

@@ -593,17 +593,27 @@ RECOMMENDED_MODEL_NAMES = (
_RECOMMENDED_MODEL_NAME_SET = frozenset(name.lower() for name in RECOMMENDED_MODEL_NAMES)
FRONTIER_MODEL_FAMILIES = (
(("azure", "azure_ai", "bedrock_mantle", "chatgpt", "openai"), ("gpt-5",)),
(
("anthropic", "azure_ai", "bedrock", "claude", "databricks", "snowflake", "vertex_ai"),
("claude-fable-5", "claude-opus-5", "claude-opus-4", "claude-sonnet-5", "claude-sonnet-4"),
),
(("google", "gemini", "vertex_ai"), ("gemini-3",)),
(("deepseek",), ("deepseek-v4", "deepseek-r1", "deepseek-reasoner")),
(("alibaba", "dashscope", "qwen"), ("qwen3.8", "qwen3.7", "qwen3-max")),
(("moonshot", "moonshotai", "kimi"), ("kimi-k3", "kimi-k2.7", "kimi-k2.6")),
(("zai", "z-ai", "zai-org", "zhipuai"), ("glm-5.3", "glm-5.2")),
# Matched against the bare model name only: the route (``openai/``, ``openrouter/``,
# a local gateway, ...) says nothing about the model's quality.
FRONTIER_MODEL_PREFIXES = (
"gpt-5",
"claude-fable-5",
"claude-opus-5",
"claude-opus-4",
"claude-sonnet-5",
"claude-sonnet-4",
"gemini-3",
"deepseek-v4",
"deepseek-r1",
"deepseek-reasoner",
"qwen3.8",
"qwen3.7",
"qwen3-max",
"kimi-k3",
"kimi-k2.7",
"kimi-k2.6",
"glm-5.3",
"glm-5.2",
)
@@ -837,11 +847,8 @@ def is_recommended_or_frontier_model(model_name: str) -> bool:
return False
if name in _RECOMMENDED_MODEL_NAME_SET:
return True
provider_name, bare_model_name = _split_model_provider(name)
return any(
_matches_frontier_family(provider_name, bare_model_name, provider_markers, prefixes)
for provider_markers, prefixes in FRONTIER_MODEL_FAMILIES
)
bare_model_name = name.rsplit("/", 1)[-1]
return _matches_model_prefix(bare_model_name, FRONTIER_MODEL_PREFIXES)
def _normalized_model_name(model_name: str) -> str:
@@ -853,28 +860,6 @@ def _normalized_model_name(model_name: str) -> str:
return name
def _split_model_provider(model_name: str) -> tuple[str | None, str]:
if "/" not in model_name:
return None, model_name
provider_name, bare_model_name = model_name.rsplit("/", 1)
return provider_name, bare_model_name
def _matches_frontier_family(
provider_name: str | None,
model_name: str,
provider_markers: tuple[str, ...],
model_prefixes: tuple[str, ...],
) -> bool:
if not _matches_model_prefix(model_name, model_prefixes):
return False
if provider_name is None:
return True
return _contains_provider_marker(
provider_name, provider_markers, split_compound_names=True
) or _contains_provider_marker(model_name, provider_markers)
def _matches_model_prefix(model_name: str, model_prefixes: tuple[str, ...]) -> bool:
return any(
candidate.startswith(prefix)
@@ -892,16 +877,6 @@ def _model_name_candidates(model_name: str) -> tuple[str, ...]:
return (model_name, *suffixes)
def _contains_provider_marker(
value: str, provider_markers: tuple[str, ...], *, split_compound_names: bool = False
) -> bool:
parts = set(value.replace(".", "/").split("/"))
if split_compound_names:
for separator in ("_", "-"):
parts.update(piece for part in tuple(parts) for piece in part.split(separator))
return any(marker in parts for marker in provider_markers)
def is_known_openai_bare_model(model_name: str) -> bool:
import litellm

View File

@@ -45,6 +45,7 @@ from strix.core.paths import run_dir_for, runtime_state_dir
from strix.core.sessions import open_agent_session
from strix.report.state import get_global_report_state
from strix.runtime import session_manager
from strix.telemetry import set_scan_phase
from strix.telemetry.logging import set_scan_id, setup_scan_logging
from strix.tools.output_store import (
WORKSPACE_SPILL_DIR,
@@ -116,6 +117,13 @@ def _record_mcp_connections(connections: list[ConnectedMcpServer]) -> None:
report_state.record_mcp_connections([connection.name for connection in connections])
def _note_exit_reason(reason: str) -> None:
"""Record why the scan stopped so the end-of-scan beacon reports it."""
report_state = get_global_report_state()
if report_state is not None and report_state.scan_ended_exit_reason is None:
report_state.scan_ended_exit_reason = reason
def _persist_mcp_status(roster: list[dict[str, Any]]) -> None:
"""Write the run's non-secret MCP connection status roster to run.json.
@@ -313,6 +321,7 @@ async def run_strix_scan(
root_id = uuid.uuid4().hex[:8]
logger.info("Bringing up sandbox session for scan %s", scan_id)
set_scan_phase("sandbox_init")
bundle = await session_manager.create_or_reuse(
scan_id,
image=image,
@@ -322,6 +331,7 @@ async def run_strix_scan(
)
report("Waiting for the first model response")
logger.info("Sandbox ready for scan %s", scan_id)
set_scan_phase("agent_setup")
sandbox_session = bundle["session"]
@@ -573,6 +583,7 @@ async def run_strix_scan(
async with coordinator._lock:
root_status = coordinator.statuses.get(root_id)
set_scan_phase("agent_loop")
result = await run_agent_loop(
agent=root_agent,
initial_input=initial_input,
@@ -610,6 +621,7 @@ async def run_strix_scan(
return result # noqa: TRY300
except BudgetExceededError as exc:
logger.info("Scan %s stopped: %s", scan_id, exc)
_note_exit_reason("budget_exceeded")
if root_id is not None:
with contextlib.suppress(Exception):
await coordinator.set_status(root_id, "stopped")
@@ -622,6 +634,7 @@ async def run_strix_scan(
exc,
scan_id,
)
_note_exit_reason("rate_limited")
if root_id is not None:
with contextlib.suppress(Exception):
await coordinator.set_status(root_id, "stopped")

View File

@@ -98,6 +98,14 @@ Examples:
# Extra files placed in the sandbox workspace
strix --target ./my-project --workspace-file ./wordlist.txt
strix --target https://app.com --workspace-file ./openapi.yaml:specs/openapi.yaml
Strix Cloud:
strix cloud login
strix cloud scans start --source . --yes --wait
strix cloud # list every cloud resource
Run a pentest in Strix Cloud https://app.strix.ai
Try Strix Enterprise https://strix.ai/demo
""",
)

View File

@@ -14,6 +14,7 @@ from strix.interface.utils import (
image_exists,
process_pull_line,
)
from strix.telemetry import report_error
logger = logging.getLogger(__name__)
@@ -44,6 +45,7 @@ def validate_environment() -> None:
f"[red]STRIX_LLM={settings.llm.model} uses your ChatGPT subscription, "
"but you're not signed in.[/] Run [cyan]strix auth login chatgpt[/] first."
)
report_error("subscription_not_signed_in")
sys.exit(1)
logger.info("Environment OK (ChatGPT subscription)")
return
@@ -153,6 +155,7 @@ def validate_environment() -> None:
console.print("\n")
console.print(panel)
console.print()
report_error("missing_required_config")
sys.exit(1)
logger.info(
"Environment OK (optional missing: %s)",
@@ -180,6 +183,7 @@ def check_docker_installed() -> None:
padding=(1, 2),
)
console.print("\n", panel, "\n")
report_error("docker_not_installed")
sys.exit(1)
logger.debug("Docker CLI present")
@@ -227,6 +231,7 @@ def pull_docker_image() -> None:
padding=(1, 2),
)
console.print(panel, "\n")
report_error("image_pull_failed", e)
sys.exit(1)
logger.info("Docker image %s ready", image)

View File

@@ -42,7 +42,7 @@ from strix.interface.utils import (
build_final_stats_text,
)
from strix.llm.warmup import start_import_warmup, wait_for_import_warmup
from strix.telemetry import posthog, scarf
from strix.telemetry import posthog, report_error, scarf, set_scan_phase
from strix.telemetry.logging import configure_dependency_logging
@@ -333,6 +333,11 @@ def display_completion_message(args: argparse.Namespace, results_path: Path) ->
"[#60a5fa]docs.strix.ai[/] [dim]·[/] "
"[#60a5fa]discord.gg/strix-ai[/]"
)
if not args.non_interactive:
console.print(
"[dim]Run a pentest in Strix Cloud[/] [#60a5fa]app.strix.ai[/] [dim]·[/] "
"[dim]Enterprise[/] [#60a5fa]strix.ai/demo[/]"
)
console.print()
if not args.non_interactive:
notify_update(console)
@@ -396,15 +401,18 @@ def _bootstrap_scan(args: argparse.Namespace) -> None:
happen inside the TUI so the interface paints immediately instead of
waiting on a model round trip.
"""
set_scan_phase("preflight")
try:
asyncio.run(warm_up_llm(show_model_warning=True))
except ModelConnectionError as exc:
report_error("model_connection_failed", exc)
_print_model_connection_error(exc, exc.model_name)
sys.exit(1)
persist_current()
try:
prepare_run(args)
except ValueError as e:
report_error("scan_preparation_failed", e)
_print_error_panel("SCAN PREPARATION FAILED", str(e))
sys.exit(1)
telemetry_start(args)
@@ -479,18 +487,21 @@ def main() -> None:
from strix.interface.cli import run_cli
asyncio.run(run_cli(args))
# Headless runs have no user to quit: the agent either finished
# (already beaconed as finished_by_tool) or stopped on its own.
exit_reason = "agent_stopped"
else:
asyncio.run(run_tui(args))
except InteractiveSetupUnavailableError as exc:
exit_reason = "error"
report_error("interactive_setup_unavailable", exc)
_print_error_panel("INTERACTIVE SETUP UNAVAILABLE", str(exc))
sys.exit(1)
except KeyboardInterrupt:
exit_reason = "interrupted"
except Exception:
except Exception as exc:
exit_reason = "error"
posthog.error("unhandled_exception")
scarf.error("unhandled_exception")
report_error("unhandled_exception", exc)
raise
finally:
report_state = get_global_report_state()

View File

@@ -192,7 +192,7 @@ class TuiController:
model_warning = ""
if model and not is_recommended_or_frontier_model(model):
model_warning = (
f"{model} is not a recommended frontier model; pentest quality could be degraded"
f"{model} is not a recommended frontier model. Pentest quality could be degraded."
)
state = {
"setup_mode": self.setup_mode,

View File

@@ -370,6 +370,17 @@ func TestStartedSnapshotTransitionsToLiveView(t *testing.T) {
}
}
func TestSplashModelWarningRendersTheBackendSentenceOnce(t *testing.T) {
warning := "openai/glm-5.3 is not a recommended frontier model. Pentest quality could be degraded."
got := ansi.Strip(splashModelWarning("openai/glm-5.3", warning))
if got != "⚠ "+warning {
t.Fatalf("splash warning = %q, want %q", got, "⚠ "+warning)
}
if got := ansi.Strip(splashModelWarning("other/model", warning)); got != "⚠ "+warning {
t.Fatalf("splash warning with unrelated model = %q", got)
}
}
func TestSetupStartScreenFitsNarrowTerminal(t *testing.T) {
model := New(nil)
model.width, model.height = 40, 18

View File

@@ -411,7 +411,7 @@ func (m Model) splashView() string {
welcome + "\n" + version + "\n" + tagline + "\n\n" +
start.String() + "\n\n" + url
if warn := m.snapshot.ModelWarning; warn != "" {
content += "\n\n" + splashModelWarning(warn)
content += "\n\n" + splashModelWarning(m.snapshot.Model, warn)
}
panel := lipgloss.NewStyle().Border(lipgloss.RoundedBorder()).BorderForeground(green).Padding(1, 6).Align(lipgloss.Center).Render(content)
// #splash_screen background is solid black.
@@ -419,12 +419,16 @@ func (m Model) splashView() string {
lipgloss.WithWhitespaceBackground(black))
}
// splashModelWarning ports SplashScreen._build_model_warning_text.
func splashModelWarning(model string) string {
// splashModelWarning renders the backend's full warning sentence, with the
// model name highlighted when the sentence leads with it.
func splashModelWarning(model, warning string) string {
yellow := lipgloss.Color("#eab308")
return lipgloss.NewStyle().Bold(true).Foreground(yellow).Render("⚠ ") +
lipgloss.NewStyle().Bold(true).Foreground(render.Cyan).Render(model) +
lipgloss.NewStyle().Foreground(yellow).Render(" is not a recommended frontier model - pentest quality could be degraded")
out := lipgloss.NewStyle().Bold(true).Foreground(yellow).Render("⚠ ")
if model != "" && strings.HasPrefix(warning, model) {
out += lipgloss.NewStyle().Bold(true).Foreground(render.Cyan).Render(model)
warning = strings.TrimPrefix(warning, model)
}
return out + lipgloss.NewStyle().Foreground(yellow).Render(warning)
}
// chatPaneKey identifies everything the bordered trace depends on.

View File

@@ -37,6 +37,7 @@ from strix.interface.tui.sidecar import (
)
from strix.interface.utils import read_workspace_files
from strix.report.state import ReportState, set_global_report_state
from strix.telemetry import report_error, set_scan_phase
from strix.utils.resource_paths import get_strix_resource_path
@@ -48,6 +49,11 @@ if TYPE_CHECKING:
logger = logging.getLogger(__name__)
def _revision_count(report: dict[str, Any]) -> int:
history = report.get("update_history")
return len(history) if isinstance(history, list) else 0
class GoTuiPreActivationError(RuntimeError):
"""A sidecar failure raised before the Go TUI activates."""
@@ -138,11 +144,13 @@ class GoTuiRuntime:
await self._preflight_model()
except Exception as exc:
logger.exception("Go TUI setup model preflight failed")
report_error("model_connection_failed", exc)
raise RuntimeError(f"Model connection failed: {exc}") from exc
async def _preflight_model(self) -> None:
model = (load_settings().llm.model or "").strip()
self.controller.add_message("Verifying model connection...")
set_scan_phase("preflight")
await preflight_model_connection(model)
self.model_verified = True
@@ -181,7 +189,11 @@ class GoTuiRuntime:
candidate.target = list(self.controller.targets)
candidate.target_list = []
build_targets_info(candidate)
prepare_run(candidate)
try:
prepare_run(candidate)
except Exception as exc:
report_error("scan_preparation_failed", exc)
raise
telemetry_start(candidate)
vars(self.args).update(vars(candidate))
@@ -195,13 +207,21 @@ class GoTuiRuntime:
launch so the interface appears immediately.
"""
model = (load_settings().llm.model or "").strip()
set_scan_phase("preflight")
try:
await preflight_model_connection(model)
except Exception as exc:
logger.exception("Go TUI scan preparation failed")
report_error("model_connection_failed", exc)
self.controller.fail_preparation(str(exc))
return
try:
persist_current()
prepare_run(self.args)
telemetry_start(self.args)
except Exception as exc:
logger.exception("Go TUI scan preparation failed")
report_error("scan_preparation_failed", exc)
self.controller.fail_preparation(str(exc))
return
self.controller.scan_state = "running"
@@ -240,6 +260,9 @@ class GoTuiRuntime:
self.controller.scan_state = "completed" if report_status == "completed" else "stopped"
except Exception as exc:
logger.exception("Go TUI scan failed")
report_error("unhandled_exception", exc)
if self.report_state is not None and self.report_state.scan_ended_exit_reason is None:
self.report_state.scan_ended_exit_reason = "error"
self.scan_error = exc
self.controller.error = str(exc)
self.controller.scan_state = "failed"
@@ -321,7 +344,9 @@ class GoTuiRuntime:
if self.report_state is not None:
usage = dict(self.report_state.get_total_llm_usage())
vulnerabilities = [
report.get("id", index) if isinstance(report, dict) else index
(report.get("id", index), _revision_count(report))
if isinstance(report, dict)
else index
for index, report in enumerate(self.report_state.vulnerability_reports)
]
return json.dumps(

View File

@@ -19,6 +19,7 @@ from rich.panel import Panel
from rich.text import Text
from strix.config import load_settings
from strix.telemetry import report_error
from strix.utils.api_spec import detect_spec_format
@@ -1602,7 +1603,8 @@ def check_docker_connection() -> Any:
try:
return docker.from_env()
except DockerException:
except DockerException as exc:
report_error("docker_unavailable", exc)
console = Console()
error_text = Text()
error_text.append("DOCKER NOT AVAILABLE", style="bold red")

View File

@@ -8,6 +8,7 @@ import {
Radar,
Rocket,
ArrowUpRight,
Building2,
History,
} from "lucide-react";
import type { Vulnerability, VulnerabilitySeverity } from "@/types/issues";
@@ -35,7 +36,7 @@ import {
type LoadedRun,
type RunsPayload,
} from "@/data/serverSource";
import { SIGNUP_URL, ctaUrl, trackCta } from "@/lib/cta";
import { SIGNUP_URL, DEMO_URL, ctaUrl, trackCta } from "@/lib/cta";
import { runTitle } from "@/lib/target-utils";
import Sidebar from "@/components/Sidebar";
import PastRunsView from "@/components/PastRunsView";
@@ -706,6 +707,31 @@ function OverviewTab({
</div>
)}
{finished && (
<div className="animate-card-in rounded-xl border border-[#222] bg-[rgba(255,255,255,0.02)] p-5">
<p className="text-sm font-semibold text-white">Strix Cloud</p>
<p className="mt-0.5 text-xs text-[#666]">Run your next pentest in Strix Cloud.</p>
<div className="mt-3 flex flex-wrap gap-2.5">
<ProInlineCta
label="Run a pentest in Strix Cloud"
desc="Validated findings, autofix, and PR reviews."
slug="overview_cloud"
surface="overview"
icon={Rocket}
primary
/>
<ProInlineCta
label="Try Strix Enterprise"
desc="SSO, compliance-ready reports, VPC or self-hosted deployment."
slug="book_demo"
surface="overview"
icon={Building2}
href={DEMO_URL}
/>
</div>
</div>
)}
{sections.length > 0 ? (
<div className="animate-card-in rounded-xl border border-[#222] bg-[rgba(255,255,255,0.02)] p-5 space-y-8">
{sections.map((s) => (
@@ -789,14 +815,23 @@ function AgentsTab({ run, canSteer }: { run: LoadedRun; canSteer: boolean }) {
{/* Re-run always routes to Strix Cloud. */}
<div className="rounded-xl border border-[#222] bg-[rgba(255,255,255,0.02)] p-5">
<p className="text-sm font-semibold text-white">Run this pentest with more depth</p>
<p className="mt-0.5 text-xs text-[#666]">Re-run this pentest on managed infra in the cloud.</p>
<p className="mt-0.5 text-xs text-[#666]">Run this pentest again in Strix Cloud.</p>
<div className="mt-3 flex flex-wrap gap-2.5">
<ProInlineCta
label="Re-run in Strix Pro with more depth"
desc="Run this pentest on managed infra with more depth."
desc="More depth, validated findings, and autofix."
slug="live_scan"
surface="agents"
icon={Rocket}
primary
/>
<ProInlineCta
label="Try Strix Enterprise"
desc="SSO, compliance-ready reports, VPC or self-hosted deployment."
slug="book_demo"
surface="agents"
icon={Building2}
href={DEMO_URL}
/>
</div>
</div>

View File

@@ -48,23 +48,40 @@ export function ProInlineCta({
slug,
icon: Icon,
surface,
href = SIGNUP_URL,
primary = false,
}: {
label: string;
desc: string;
slug: string;
icon: React.ElementType;
surface?: string;
/** Destination before attribution params. Defaults to cloud sign-up. */
href?: string;
/** Solid white button instead of the outlined default. */
primary?: boolean;
}) {
return (
<Tooltip text={desc}>
<a
href={ctaUrl(SIGNUP_URL, slug)}
href={ctaUrl(href, slug)}
target="_blank"
rel="noopener noreferrer"
onClick={() => trackCta(slug, surface)}
className="group inline-flex items-center gap-2 rounded-lg border border-[#222] bg-[rgba(255,255,255,0.02)] px-3 py-2 text-sm text-[#aaa] transition-colors hover:border-[#444] hover:text-white"
className={
primary
? "group inline-flex items-center gap-2 rounded-lg border border-white bg-white px-3 py-2 text-sm font-semibold text-black transition-colors hover:bg-[#e5e5e5]"
: "group inline-flex items-center gap-2 rounded-lg border border-[#222] bg-[rgba(255,255,255,0.02)] px-3 py-2 text-sm text-[#aaa] transition-colors hover:border-[#444] hover:text-white"
}
>
<Icon className="h-4 w-4 text-[#888] transition-colors group-hover:text-white" aria-hidden="true" />
<Icon
className={
primary
? "h-4 w-4 text-black"
: "h-4 w-4 text-[#888] transition-colors group-hover:text-white"
}
aria-hidden="true"
/>
<span>{label}</span>
</a>
</Tooltip>

View File

@@ -10,7 +10,7 @@ import {
WandSparkles,
Plug,
} from "lucide-react";
import { SIGNUP_URL, PRICING_URL, ctaUrl, trackCta } from "@/lib/cta";
import { SIGNUP_URL, PRICING_URL, DEMO_URL, ctaUrl, trackCta } from "@/lib/cta";
/**
* Dialog shown when a platform feature is clicked in the sidebar: a short
@@ -141,6 +141,19 @@ export function UpgradeModal({
<ExternalLink className="h-3 w-3" />
</a>
</div>
<p className="text-center text-xs text-[#666]">
SSO, compliance reports, or a private deployment?{" "}
<a
href={ctaUrl(DEMO_URL, "upgrade_book_demo")}
target="_blank"
rel="noopener noreferrer"
onClick={() => trackCta("upgrade_book_demo", source)}
className="whitespace-nowrap text-[#aaa] underline underline-offset-2 transition-colors hover:text-white"
>
Try Strix Enterprise
</a>
</p>
</div>
</div>
</div>

View File

@@ -6,8 +6,8 @@
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<meta name="color-scheme" content="dark" />
<title>Strix Results</title>
<script type="module" crossorigin src="./assets/index-Bpn8GiSb.js"></script>
<link rel="stylesheet" crossorigin href="./assets/index-qwPOPAGC.css">
<script type="module" crossorigin src="./assets/index-B94ANU8d.js"></script>
<link rel="stylesheet" crossorigin href="./assets/index-DN__rVv3.css">
</head>
<body>
<div id="root"></div>

View File

@@ -73,6 +73,7 @@ UPDATABLE_REPORT_FIELDS = frozenset(
"cve",
"cwe",
"code_locations",
"http_exchange_ids",
"fix_verification",
"fix_pr_body",
}
@@ -333,6 +334,7 @@ class ReportState:
cve: str | None = None,
cwe: str | None = None,
code_locations: list[dict[str, Any]] | None = None,
http_exchange_ids: list[str] | None = None,
fix_verification: str | None = None,
fix_pr_body: str | None = None,
finding_class: str | None = None,
@@ -391,6 +393,8 @@ class ReportState:
report["cwe"] = cwe.strip()
if code_locations:
report["code_locations"] = code_locations
if http_exchange_ids:
report["http_exchange_ids"] = http_exchange_ids
if fix_verification:
report["fix_verification"] = fix_verification.strip()
if fix_pr_body:
@@ -403,14 +407,14 @@ class ReportState:
if agent_name:
report["agent_name"] = agent_name
if self.vulnerability_found_callback:
self.vulnerability_found_callback(report)
self.vulnerability_reports.append(report)
logger.info(f"Added vulnerability report: {report_id} - {title}")
posthog.finding(severity, cwe=cwe, is_cve=bool(cve))
scarf.finding(severity, cwe=cwe, is_cve=bool(cve))
if self.vulnerability_found_callback:
self.vulnerability_found_callback(report)
self.save_run_data()
return report_id
@@ -486,11 +490,18 @@ class ReportState:
)
history.append(entry)
report.update(changed)
revised = {**report, **changed}
for dependent in superseded:
report.pop(dependent, None)
report["update_history"] = history
report["updated_at"] = entry["timestamp"]
revised.pop(dependent, None)
revised["update_history"] = history
revised["updated_at"] = entry["timestamp"]
# Persistence must accept the revision before local state changes. A
# failed callback leaves the old evidence intact and the update retryable.
if self.vulnerability_updated_callback:
self.vulnerability_updated_callback(revised)
report.clear()
report.update(revised)
# The markdown on disk still shows the superseded evidence, so let the
# writer re-render it.
@@ -502,9 +513,6 @@ class ReportState:
", ".join(entry["fields"]) or "no field replaced",
)
if self.vulnerability_updated_callback:
self.vulnerability_updated_callback(report)
self.save_run_data()
return report

View File

@@ -12,7 +12,7 @@ Privacy is our priority. All collected data is anonymized by default. Each sessi
We collect only very **basic** usage data including:
**Session Errors:** Duration and error types (not messages or stack traces)\
**Session Errors:** Duration, the failure category, the scan phase, and the exception class name (not messages or stack traces)\
**System Context:** OS type, architecture, Strix version\
**Scan Context:** Scan mode (quick/standard/deep), scan type (whitebox/blackbox)\
**Model Usage:** Which LLM model is being used and whether it runs via an API key or a model subscription (not prompts or responses)\

View File

@@ -1,7 +1,19 @@
from . import posthog, scarf
from ._common import set_scan_phase
def report_error(error_type: str, exc: BaseException | None = None) -> None:
"""Beacon a failure category, plus the exception class when one is given.
Only class names travel: never the message, arguments, or traceback.
"""
posthog.error(error_type, exc)
scarf.error(error_type, exc)
__all__ = [
"posthog",
"report_error",
"scarf",
"set_scan_phase",
]

View File

@@ -5,7 +5,7 @@ import platform
import sys
from importlib.metadata import PackageNotFoundError, version
from pathlib import Path
from typing import Any
from typing import Any, cast
from uuid import uuid4
@@ -54,3 +54,43 @@ def base_props() -> dict[str, Any]:
"python": f"{sys.version_info.major}.{sys.version_info.minor}",
"strix_version": get_version(),
}
# Coarse stage of the current run, attached to ``error`` beacons so a failure
# can be placed without a message or trace. Process-local, like the rest of the
# CLI telemetry: one process runs one scan.
_scan_phase = "startup"
def set_scan_phase(phase: str) -> None:
global _scan_phase # noqa: PLW0603
_scan_phase = phase
def get_scan_phase() -> str:
return _scan_phase
def _exception_name(exc: BaseException) -> str:
cls = type(exc)
package = cls.__module__.split(".")[0]
return cls.__name__ if package == "builtins" else f"{package}.{cls.__name__}"
def _unwrap_group(exc: BaseException) -> BaseException:
if not isinstance(exc, BaseExceptionGroup):
return exc
group = cast("BaseExceptionGroup[BaseException]", exc)
return group.exceptions[0] if group.exceptions else group
def exception_props(exc: BaseException) -> dict[str, str]:
"""Class names only. Messages, arguments, and tracebacks never leave the machine."""
exc = _unwrap_group(exc)
props = {"exception_type": _exception_name(exc)}
cause = exc.__cause__
if cause is None and not exc.__suppress_context__:
cause = exc.__context__
if cause is not None:
props["exception_cause"] = _exception_name(cause)
return props

View File

@@ -9,6 +9,8 @@ from strix.telemetry._common import (
SEND_TIMEOUT,
SESSION_ID,
base_props,
exception_props,
get_scan_phase,
get_version,
is_first_run,
)
@@ -178,6 +180,12 @@ def viewer_agent_steered() -> None:
_send("viewer_agent_steered", {**base_props()})
def error(error_type: str) -> None:
props = {**base_props(), "error_type": error_type}
def error(error_type: str, exc: BaseException | None = None) -> None:
props: dict[str, Any] = {
**base_props(),
"error_type": error_type,
"phase": get_scan_phase(),
}
if exc is not None:
props.update(exception_props(exc))
_send("error", props)

View File

@@ -12,6 +12,8 @@ from strix.telemetry._common import (
SEND_TIMEOUT,
SESSION_ID,
base_props,
exception_props,
get_scan_phase,
get_version,
is_first_run,
)
@@ -135,10 +137,13 @@ def end(report_state: ReportState, exit_reason: str = "completed") -> None:
)
def error(error_type: str) -> None:
def error(error_type: str, exc: BaseException | None = None) -> None:
props: dict[str, Any] = {
**base_props(),
"session": SESSION_ID,
"error_type": error_type,
"phase": get_scan_phase(),
}
if exc is not None:
props.update(exception_props(exc))
_send("error", props)

View File

@@ -4,6 +4,7 @@ from __future__ import annotations
import asyncio
import dataclasses
import functools
import json
import logging
import re
@@ -67,6 +68,31 @@ async def _call[T](client: Client, fn: Callable[[Client], Awaitable[T]]) -> T:
return await fn(client)
async def existing_request_ids(
ctx: RunContextWrapper,
request_ids: list[str],
) -> set[str]:
"""Return request IDs that exist in the current Caido project."""
if not request_ids:
return set()
client = await _ctx_client(ctx)
if client is None:
raise RuntimeError("Caido client is not available")
# Request IDs are not an HTTPQL field. Resolve each ID through the same
# project-bound lookup as view_request rather than constructing a filter.
existing: set[str] = set()
for request_id in request_ids:
result = await _call(
client,
functools.partial(caido_api.get_request_with_client, request_id=request_id),
)
if result is not None:
existing.add(str(result.request.id))
return existing
def _to_tool_json(value: Any) -> Any:
"""Recursively convert SDK dataclasses/Pydantic objects to tool JSON values."""
if value is None or isinstance(value, str | int | float | bool):

View File

@@ -17,6 +17,7 @@ from typing import TYPE_CHECKING, Any
from agents import RunContextWrapper, function_tool
from strix.tools.nullish import clean_optional
from strix.tools.proxy.tools import existing_request_ids
if TYPE_CHECKING:
@@ -168,6 +169,8 @@ _REQUIRED_FIELDS = {
_VALID_FIX_EFFORT = frozenset({"trivial", "low", "medium", "high"})
_VALID_CONFIDENCE = frozenset({"high", "medium", "low"})
_MAX_HTTP_EXCHANGE_IDS = 10
_MAX_HTTP_EXCHANGE_ID_CHARS = 128
def _validate_required_text(fields: dict[str, str]) -> list[str]:
@@ -177,6 +180,97 @@ def _validate_required_text(fields: dict[str, str]) -> list[str]:
]
def _normalize_http_exchange_ids(raw: Any) -> tuple[list[str] | None, list[str]]:
"""Return distinct proxy exchange ids in their original order."""
if raw is None:
return None, []
if not isinstance(raw, list):
return None, ["http_exchange_ids must be a list of proxy request ids"]
normalized: list[str] = []
errors: list[str] = []
seen: set[str] = set()
for index, value in enumerate(raw):
if not isinstance(value, str):
errors.append(f"http_exchange_ids[{index}] must be a string")
continue
request_id = value.strip()
if not request_id:
errors.append(f"http_exchange_ids[{index}] cannot be empty")
continue
if len(request_id) > _MAX_HTTP_EXCHANGE_ID_CHARS:
errors.append(
f"http_exchange_ids[{index}] must be {_MAX_HTTP_EXCHANGE_ID_CHARS} "
"characters or fewer"
)
continue
if any(ord(char) < 0x21 or ord(char) > 0x7E for char in request_id):
errors.append(f"http_exchange_ids[{index}] must contain only visible ASCII characters")
continue
if not request_id.isdigit():
errors.append(f"http_exchange_ids[{index}] must be a numeric proxy request id")
continue
if request_id not in seen:
seen.add(request_id)
normalized.append(request_id)
if len(normalized) > _MAX_HTTP_EXCHANGE_IDS:
errors.append(
f"http_exchange_ids can contain at most "
f"{_MAX_HTTP_EXCHANGE_IDS} distinct request ids"
)
break
return normalized, errors
_HTTP_EXCHANGE_DROPPED_WARNING = (
"http_exchange_ids were not stored: the proxy project could not be reached to verify "
"them. Attach them with update_vulnerability_report when the proxy responds again."
)
async def _verify_http_exchange_ids(
ctx: RunContextWrapper,
raw: Any,
) -> tuple[list[str] | None, list[str], str | None]:
"""Verify proxy exchange IDs against the current Caido project.
IDs the project does not know are rejected. When the proxy itself cannot be
queried the IDs are dropped and a warning is returned instead, so a proxy
outage never blocks a finding and unverified IDs are never recorded as
evidence.
"""
request_ids, errors = _normalize_http_exchange_ids(raw)
if request_ids is None or errors or not request_ids:
return request_ids, errors, None
try:
existing_ids = await existing_request_ids(ctx, request_ids)
except Exception: # noqa: BLE001
logger.warning(
"Could not verify HTTP exchange IDs against the current Caido project",
exc_info=True,
)
return None, [], _HTTP_EXCHANGE_DROPPED_WARNING
missing_ids = [request_id for request_id in request_ids if request_id not in existing_ids]
if missing_ids:
return (
None,
[
"http_exchange_ids do not exist in the current proxy project: "
+ ", ".join(missing_ids)
],
None,
)
return request_ids, [], None
def _with_warning(result: dict[str, Any], warning: str | None) -> dict[str, Any]:
if warning and result.get("success"):
result["warning"] = warning
return result
def _validate_cvss_breakdown(breakdown: Any) -> list[str]:
"""Check the 8 CVSS metrics are all present with legal values."""
if not isinstance(breakdown, dict) or not breakdown:
@@ -299,7 +393,7 @@ _UPDATE_TEXT_FIELDS = (
)
def _collect_update_changes( # noqa: PLR0912
def _collect_update_changes( # noqa: PLR0912, PLR0915
fields: dict[str, Any],
) -> tuple[dict[str, Any], list[str]]:
"""Validate the fields a revision replaces and return them with any errors."""
@@ -368,6 +462,12 @@ def _collect_update_changes( # noqa: PLR0912
if cwe:
changes["cwe"] = cwe
raw_http_exchange_ids = fields.get("http_exchange_ids")
http_exchange_ids, http_exchange_errors = _normalize_http_exchange_ids(raw_http_exchange_ids)
errors.extend(http_exchange_errors)
if raw_http_exchange_ids is not None and not http_exchange_errors:
changes["http_exchange_ids"] = http_exchange_ids or []
return changes, errors
@@ -378,6 +478,7 @@ _DYNAMIC_ONLY_UPDATE_FIELDS = (
"method",
"poc_description",
"poc_script_code",
"http_exchange_ids",
)
# A dependency finding is rated in the context of the codebase that pins it, and
@@ -551,13 +652,24 @@ def _do_update(
if class_error is not None:
return class_error
updated = report_state.update_vulnerability_report(
report_id,
changes,
update_reason=update_reason,
updated_by_agent_id=agent_id,
updated_by_agent_name=agent_name,
)
try:
updated = report_state.update_vulnerability_report(
report_id,
changes,
update_reason=update_reason,
updated_by_agent_id=agent_id,
updated_by_agent_name=agent_name,
)
except Exception as e:
logger.exception("update_vulnerability_report persistence failed")
return {
"success": False,
"error": (
f"Failed to revise report '{report_id}': {e!s}. "
"The report still carries its previous content; retry the update."
),
"report_id": report_id,
}
if updated is None:
known = [r.get("id") for r in report_state.get_existing_vulnerabilities()]
if report_id not in known:
@@ -606,6 +718,7 @@ async def _do_create(
cve: str | None,
cwe: str | None,
code_locations: list[dict[str, Any]] | None,
http_exchange_ids: list[str] | None = None,
confidence_rationale: str | None = None,
fix_verification: str | None = None,
fix_pr_body: str | None = None,
@@ -651,6 +764,10 @@ async def _do_create(
errors.extend(_validate_fix_verification(parsed_locations, fix_verification))
cve, cwe, identifier_errors = _validate_identifiers(cve, cwe)
errors.extend(identifier_errors)
normalized_http_exchange_ids, http_exchange_errors = _normalize_http_exchange_ids(
http_exchange_ids
)
errors.extend(http_exchange_errors)
if errors:
return {"success": False, "error": "Validation failed", "errors": errors}
@@ -712,6 +829,7 @@ async def _do_create(
"code_locations": parsed_locations,
"fix_verification": fix_verification,
"fix_pr_body": fix_pr_body,
"http_exchange_ids": normalized_http_exchange_ids,
}
dedupe = await check_duplicate(candidate, existing)
@@ -738,9 +856,15 @@ async def _do_create(
agent_id=agent_id if isinstance(agent_id, str) else None,
agent_name=agent_name if isinstance(agent_name, str) else None,
)
except (ImportError, AttributeError) as e:
except Exception as e:
logger.exception("create_vulnerability_report persistence failed")
return {"success": False, "error": f"Failed to create vulnerability report: {e!s}"}
return {
"success": False,
"error": (
f"Failed to create vulnerability report: {e!s}. "
"The finding was not stored; file it again."
),
}
else:
logger.info(
"Vulnerability report created: id=%s severity=%s cvss=%.1f title=%s",
@@ -796,6 +920,7 @@ async def create_vulnerability_report(
cve: str | None = None,
cwe: str | None = None,
code_locations: list[dict[str, Any]] | None = None,
http_exchange_ids: list[str] | None = None,
confidence_rationale: str | None = None,
fix_verification: str | None = None,
fix_pr_body: str | None = None,
@@ -1040,6 +1165,18 @@ async def create_vulnerability_report(
cve: ``CVE-YYYY-NNNNN`` if certain, else omit.
cwe: ``CWE-NNN`` (most specific child) if certain, else omit.
code_locations: White-box findings — list of location objects.
http_exchange_ids: Proxy request IDs that prove this finding.
Copy these IDs from ``list_requests`` or ``view_request``.
For a finding validated over HTTP, capture and inspect the
supporting exchanges and include their IDs here before filing.
Include relevant baseline/control requests as well as the exploit.
Omit only when the finding has no captured HTTP evidence (for
example a static-only code finding). Never invent IDs or drop
them to bypass a verification error; retry the capture instead.
If the result carries a ``warning`` that the IDs were not
stored, the finding is filed without them: attach them with
``update_vulnerability_report`` once the proxy responds.
Keep IDs out of ``evidence`` and all other report text.
**How ``fix_before`` / ``fix_after`` work**: they're used as
literal GitHub/GitLab PR suggestion blocks. When a reviewer
@@ -1187,6 +1324,22 @@ async def create_vulnerability_report(
reduce impact and lower the severity.
fix_effort: "low"
"""
(
http_exchange_ids,
http_exchange_errors,
http_exchange_warning,
) = await _verify_http_exchange_ids(ctx, http_exchange_ids)
if http_exchange_errors:
return json.dumps(
{
"success": False,
"error": "Validation failed",
"errors": http_exchange_errors,
},
ensure_ascii=False,
default=str,
)
agent_id, agent_name = _caller_identity(ctx)
result = await _do_create(
@@ -1211,12 +1364,13 @@ async def create_vulnerability_report(
cve=cve,
cwe=cwe,
code_locations=code_locations,
http_exchange_ids=http_exchange_ids,
fix_verification=fix_verification,
fix_pr_body=fix_pr_body,
agent_id=agent_id,
agent_name=agent_name,
)
return json.dumps(result, ensure_ascii=False, default=str)
return json.dumps(_with_warning(result, http_exchange_warning), ensure_ascii=False, default=str)
@function_tool(timeout=60, strict_mode=False)
@@ -1245,6 +1399,7 @@ async def update_vulnerability_report(
cve: str | None = None,
cwe: str | None = None,
code_locations: list[dict[str, Any]] | None = None,
http_exchange_ids: list[str] | None = None,
fix_verification: str | None = None,
fix_pr_body: str | None = None,
contextual_cvss_reasoning: str | None = None,
@@ -1317,47 +1472,74 @@ async def update_vulnerability_report(
cve: Replacement CVE id.
cwe: Replacement CWE id.
code_locations: Replacement code locations.
http_exchange_ids: Replacement proxy request ids. Pass an empty
list to remove all linked exchanges.
fix_verification: Verification statement for an applyable fix.
fix_pr_body: Replacement fix PR body.
contextual_cvss_reasoning: Dependency findings only. What you
observed in this codebase that justifies the contextual
``cvss_breakdown``.
"""
(
http_exchange_ids,
http_exchange_errors,
http_exchange_warning,
) = await _verify_http_exchange_ids(ctx, http_exchange_ids)
if http_exchange_errors:
return json.dumps(
{
"success": False,
"error": "Validation failed",
"errors": http_exchange_errors,
},
ensure_ascii=False,
default=str,
)
fields = {
"title": title,
"description": description,
"impact": impact,
"target": target,
"technical_analysis": technical_analysis,
"poc_description": poc_description,
"poc_script_code": poc_script_code,
"remediation_steps": remediation_steps,
"evidence": evidence,
"assumptions": assumptions,
"counterevidence": counterevidence,
"confidence": confidence,
"confidence_rationale": confidence_rationale,
"severity_change_conditions": severity_change_conditions,
"fix_effort": fix_effort,
"cvss_breakdown": cvss_breakdown,
"endpoint": endpoint,
"method": method,
"cve": cve,
"cwe": cwe,
"code_locations": code_locations,
"http_exchange_ids": http_exchange_ids,
"fix_verification": fix_verification,
"fix_pr_body": fix_pr_body,
"contextual_cvss_reasoning": contextual_cvss_reasoning,
}
if http_exchange_warning and all(value is None for value in fields.values()):
return json.dumps(
{"success": False, "error": http_exchange_warning, "report_id": report_id},
ensure_ascii=False,
default=str,
)
agent_id, agent_name = _caller_identity(ctx)
result = await asyncio.to_thread(
_do_update,
report_id=report_id,
update_reason=update_reason,
fields={
"title": title,
"description": description,
"impact": impact,
"target": target,
"technical_analysis": technical_analysis,
"poc_description": poc_description,
"poc_script_code": poc_script_code,
"remediation_steps": remediation_steps,
"evidence": evidence,
"assumptions": assumptions,
"counterevidence": counterevidence,
"confidence": confidence,
"confidence_rationale": confidence_rationale,
"severity_change_conditions": severity_change_conditions,
"fix_effort": fix_effort,
"cvss_breakdown": cvss_breakdown,
"endpoint": endpoint,
"method": method,
"cve": cve,
"cwe": cwe,
"code_locations": code_locations,
"fix_verification": fix_verification,
"fix_pr_body": fix_pr_body,
"contextual_cvss_reasoning": contextual_cvss_reasoning,
},
fields=fields,
agent_id=agent_id,
agent_name=agent_name,
)
return json.dumps(result, ensure_ascii=False, default=str)
return json.dumps(_with_warning(result, http_exchange_warning), ensure_ascii=False, default=str)
_DEP_SEVERITY_FROM_CVSS = {
@@ -1747,9 +1929,15 @@ async def _do_create_dependency( # noqa: PLR0912
agent_id=agent_id if isinstance(agent_id, str) else None,
agent_name=agent_name if isinstance(agent_name, str) else None,
)
except (ImportError, AttributeError) as e:
except Exception as e:
logger.exception("create_dependency_report persistence failed")
return {"success": False, "error": f"Failed to create dependency report: {e!s}"}
return {
"success": False,
"error": (
f"Failed to create dependency report: {e!s}. "
"The finding was not stored; file it again."
),
}
else:
logger.info(
"Dependency report created: id=%s cve=%s package=%s severity=%s",

View File

@@ -19,6 +19,7 @@ from strix.config.settings import DEFAULT_MAX_TURNS
from strix.interface.tui import runtime as go_tui
from strix.interface.tui import sidecar
from strix.interface.tui.runtime import GoTuiRuntime
from strix.report.state import ReportState
def args() -> argparse.Namespace:
@@ -1027,3 +1028,29 @@ async def test_prepare_and_start_runs_the_scan_after_preparation(
assert order == ["preflight", "persist", "prepare", "telemetry", "state", "scan"]
assert runtime.controller.scan_state == "running"
def test_sync_fingerprint_tracks_report_revisions(tmp_path: Path) -> None:
runtime = GoTuiRuntime(args())
runtime.report_state = ReportState(run_name="test-run")
runtime.report_state.vulnerability_reports = [{"id": "vuln-0001", "title": "Old title"}]
runtime.report_state.get_run_dir = lambda: tmp_path # type: ignore[method-assign]
report = runtime.report_state.vulnerability_reports[0]
timestamp = "2026-09-09 10:00:00 UTC"
before = runtime._runtime_sync_fingerprint()
report.update(
{
"title": "New title",
"updated_at": timestamp,
"update_history": [{"timestamp": timestamp, "fields": ["title"]}],
}
)
first_revision = runtime._runtime_sync_fingerprint()
assert first_revision != before
report["title"] = "Newer title"
report["update_history"].append({"timestamp": timestamp, "fields": ["title"]})
assert runtime._runtime_sync_fingerprint() != first_revision

View File

@@ -79,6 +79,14 @@ def test_recommended_models_are_matched_case_insensitively() -> None:
"zai/glm-5.3-flash",
"openrouter/z-ai/glm-5.3",
"novita/zai-org/glm-5.2",
"openai/glm-5.3",
"openai/zai-org/glm-5.3",
"hosted_vllm/glm-5.3",
"openai/claude-opus-4-8",
"openai/deepseek-v4-pro",
"custom-ollama/gpt-5-mini-local",
"custom-provider/claude-opus-4-local",
"custom-provider/glm-5.3-local",
],
)
def test_frontier_model_families_are_accepted(model_name: str) -> None:
@@ -93,15 +101,13 @@ def test_frontier_model_families_are_accepted(model_name: str) -> None:
"anthropic/claude-3-5-sonnet-latest",
"ollama/llama3.1",
"deepseek/deepseek-chat",
"custom-ollama/gpt-5-mini-local",
"custom-provider/claude-opus-4-local",
"xai/grok-4.5",
"openrouter/x-ai/grok-4",
"mistral/mistral-medium-3-5",
"mistral/magistral-medium-latest",
"zai/glm-4.7",
"openai/glm-4.7",
"openrouter/z-ai/glm-5",
"custom-provider/glm-5.3-local",
],
)
def test_non_frontier_models_are_rejected(model_name: str) -> None:

View File

@@ -9,6 +9,7 @@ call at a time against the shared client.
from __future__ import annotations
import asyncio
from types import SimpleNamespace
from typing import TYPE_CHECKING, Any, cast
import pytest
@@ -227,3 +228,27 @@ async def test_ctx_client_degrades_when_bootstrap_failed() -> None:
handle = CaidoBootstrapHandle(asyncio.ensure_future(_bootstrap()))
assert await tools._ctx_client(cast("Any", _Ctx({"caido_client": handle}))) is None
async def test_existing_request_ids_queries_current_project(
monkeypatch: pytest.MonkeyPatch,
) -> None:
client = _FakeClient("host")
looked_up: list[str] = []
async def get_request_with_client(passed_client: Any, request_id: str) -> Any:
assert passed_client is client
looked_up.append(request_id)
if request_id == "1042":
return SimpleNamespace(request=SimpleNamespace(id="1042"))
return None
monkeypatch.setattr(caido_api, "get_request_with_client", get_request_with_client)
existing = await tools.existing_request_ids(
cast("Any", _Ctx({"caido_client": client})),
["1042", "1088"],
)
assert existing == {"1042"}
assert looked_up == ["1042", "1088"]

View File

@@ -2,9 +2,11 @@
from __future__ import annotations
from typing import TYPE_CHECKING, Any
import json
from typing import TYPE_CHECKING, Any, cast
import pytest
from agents.tool_context import ToolContext
from strix.report.dedupe import (
_check_dependency_duplicate,
@@ -13,10 +15,13 @@ from strix.report.dedupe import (
)
from strix.report.state import ReportState, set_global_report_state
from strix.tools.finish.tool import finish_scan
from strix.tools.reporting import tool as reporting_tool
from strix.tools.reporting.tool import (
_do_create,
_do_create_dependency,
_do_update,
_normalize_http_exchange_ids,
_verify_http_exchange_ids,
create_dependency_report,
create_vulnerability_report,
update_vulnerability_report,
@@ -115,6 +120,7 @@ async def test_create_report_persists_new_fields(report_state: ReportState) -> N
cve=None,
cwe="CWE-79",
code_locations=None,
http_exchange_ids=["1042", "1042", "1088"],
fix_pr_body="## Fix\nEncode output.",
)
assert result["success"] is True
@@ -127,6 +133,51 @@ async def test_create_report_persists_new_fields(report_state: ReportState) -> N
assert report["counterevidence"] == "No output encoding or CSP observed on this response."
assert report["confidence"] == "high"
assert report["severity_change_conditions"] == "A strict CSP would lower the severity."
assert report["http_exchange_ids"] == ["1042", "1088"]
def test_create_report_does_not_commit_when_callback_fails(
report_state: ReportState,
) -> None:
def fail_persistence(_report: dict[str, Any]) -> None:
raise RuntimeError("persistence failed")
report_state.vulnerability_found_callback = fail_persistence
with pytest.raises(RuntimeError, match="persistence failed"):
report_state.add_vulnerability_report(
title="Unstored finding",
severity="high",
http_exchange_ids=["1042"],
)
assert report_state.vulnerability_reports == []
def test_failed_revision_keeps_old_evidence_and_can_be_retried(
report_state: ReportState,
) -> None:
report_id = report_state.add_vulnerability_report(
title="Original finding", severity="high", http_exchange_ids=["1042"]
)
original = dict(report_state.vulnerability_reports[0])
def fail_persistence(revised: dict[str, Any]) -> None:
assert revised["http_exchange_ids"] == ["1088"]
assert report_state.vulnerability_reports[0] == original
raise RuntimeError("persistence failed")
report_state.vulnerability_updated_callback = fail_persistence
changes = {"title": "Revised finding", "http_exchange_ids": ["1088"]}
with pytest.raises(RuntimeError, match="persistence failed"):
report_state.update_vulnerability_report(report_id, changes)
assert report_state.vulnerability_reports[0] == original
report_state.vulnerability_updated_callback = None
revised = report_state.update_vulnerability_report(report_id, changes)
assert revised is not None
assert revised["http_exchange_ids"] == ["1088"]
assert len(revised["update_history"]) == 1
async def test_create_report_requires_evidence_and_assumptions(
@@ -1040,7 +1091,13 @@ def test_tool_descriptions_include_formatting_guidance() -> None:
def test_vuln_tool_exposes_new_params() -> None:
props = create_vulnerability_report.params_json_schema["properties"]
for field in ("evidence", "assumptions", "fix_effort", "fix_pr_body"):
for field in (
"evidence",
"assumptions",
"fix_effort",
"fix_pr_body",
"http_exchange_ids",
):
assert field in props
dep_props = create_dependency_report.params_json_schema["properties"]
@@ -1355,6 +1412,158 @@ def test_update_vulnerability_report_records_chained_impact(report_state: Report
assert report_state.update_vulnerability_report("vuln-0404", {"severity": "high"}) is None
def test_update_replaces_http_exchange_ids(report_state: ReportState) -> None:
_seed_weak_report(report_state)
result = _do_update(
report_id="vuln-0009",
update_reason="A replay produced a clearer proving exchange.",
fields={"http_exchange_ids": ["204", "204", "205"]},
)
assert result["success"] is True
assert report_state.vulnerability_reports[0]["http_exchange_ids"] == ["204", "205"]
async def test_create_rejects_invalid_http_exchange_ids(report_state: ReportState) -> None:
result = await _do_create(
**_CONFIRMED_KWARGS,
http_exchange_ids=["ok", "contains space"],
)
assert result["success"] is False
assert any("visible ASCII" in error for error in result["errors"])
assert report_state.vulnerability_reports == []
def test_http_exchange_id_limit_applies_after_deduplication() -> None:
request_ids, errors = _normalize_http_exchange_ids(["1042"] * 11)
assert errors == []
assert request_ids == ["1042"]
async def test_http_exchange_ids_must_exist_in_current_proxy_project(
monkeypatch: pytest.MonkeyPatch,
) -> None:
async def existing_request_ids(
_ctx: Any,
_request_ids: list[str],
) -> set[str]:
return {"1042"}
monkeypatch.setattr(reporting_tool, "existing_request_ids", existing_request_ids)
request_ids, errors, warning = await _verify_http_exchange_ids(
cast("Any", object()),
["1042", "1088"],
)
assert request_ids is None
assert errors == ["http_exchange_ids do not exist in the current proxy project: 1088"]
assert warning is None
async def test_http_exchange_ids_are_dropped_when_proxy_cannot_be_queried(
monkeypatch: pytest.MonkeyPatch,
) -> None:
async def existing_request_ids(
_ctx: Any,
_request_ids: list[str],
) -> set[str]:
raise RuntimeError("Caido client is not available")
monkeypatch.setattr(reporting_tool, "existing_request_ids", existing_request_ids)
request_ids, errors, warning = await _verify_http_exchange_ids(
cast("Any", object()),
["1042", "1042", "1088"],
)
assert request_ids is None
assert errors == []
assert warning is not None
assert "not stored" in warning
assert "update_vulnerability_report" in warning
async def test_create_reports_persistence_failure_as_tool_error(
report_state: ReportState,
) -> None:
def fail_persistence(_report: dict[str, Any]) -> None:
raise RuntimeError("persistence failed")
report_state.vulnerability_found_callback = fail_persistence
result = await _do_create(**_CONFIRMED_KWARGS, http_exchange_ids=["1042"])
assert result["success"] is False
assert "persistence failed" in result["error"]
assert "file it again" in result["error"]
assert report_state.vulnerability_reports == []
async def test_evidence_only_update_reports_proxy_outage_as_retryable(
report_state: ReportState,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_seed_weak_report(report_state)
original = dict(report_state.vulnerability_reports[0])
async def existing_request_ids(
_ctx: Any,
_request_ids: list[str],
) -> set[str]:
raise RuntimeError("Caido client is not available")
monkeypatch.setattr(reporting_tool, "existing_request_ids", existing_request_ids)
ctx = ToolContext(
context={"agent_id": "root"},
tool_name="update_vulnerability_report",
tool_call_id="call-1",
tool_arguments="{}",
)
raw = await update_vulnerability_report.on_invoke_tool(
ctx,
json.dumps(
{
"report_id": "vuln-0009",
"update_reason": "A replay produced a clearer proving exchange.",
"http_exchange_ids": ["204"],
}
),
)
result = json.loads(raw)
assert result["success"] is False
assert "No fields to update" not in result["error"]
assert "update_vulnerability_report" in result["error"]
assert result["report_id"] == "vuln-0009"
assert report_state.vulnerability_reports[0] == original
def test_update_reports_persistence_failure_as_tool_error(report_state: ReportState) -> None:
_seed_weak_report(report_state)
original = dict(report_state.vulnerability_reports[0])
def fail_persistence(_report: dict[str, Any]) -> None:
raise RuntimeError("persistence failed")
report_state.vulnerability_updated_callback = fail_persistence
result = _do_update(
report_id="vuln-0009",
update_reason="A replay produced a clearer proving exchange.",
fields={"http_exchange_ids": ["204"]},
)
assert result["success"] is False
assert "persistence failed" in result["error"]
assert result["report_id"] == "vuln-0009"
assert report_state.vulnerability_reports[0] == original
def test_update_vulnerability_report_ignores_identical_content(report_state: ReportState) -> None:
_seed_weak_report(report_state)
assert report_state.update_vulnerability_report("vuln-0009", {"severity": "medium"}) is None

View File

@@ -0,0 +1,125 @@
"""Error beacons carry a category, phase, and exception class — never a message."""
from __future__ import annotations
from typing import Any
import pytest
import requests
from strix.report.state import ReportState
from strix.telemetry import posthog, report_error, scarf, set_scan_phase
from strix.telemetry._common import exception_props
PRIVATE_MESSAGE = "private message that must stay on the machine"
def _capture(sent: list[dict[str, Any]], event: str, props: dict[str, Any]) -> bool:
sent.append({"event": event, **props})
return True
def test_exception_props_uses_bare_name_for_builtins() -> None:
assert exception_props(ValueError(PRIVATE_MESSAGE)) == {"exception_type": "ValueError"}
def test_exception_props_prefixes_third_party_top_level_package() -> None:
props = exception_props(requests.exceptions.ConnectTimeout(PRIVATE_MESSAGE))
assert props == {"exception_type": "requests.ConnectTimeout"}
def _chained(cause: BaseException | None, *, explicit: bool) -> RuntimeError:
exc = RuntimeError("wrapped")
if explicit:
exc.__cause__ = cause
exc.__suppress_context__ = True
else:
exc.__context__ = cause
return exc
def test_exception_props_reports_explicit_cause() -> None:
props = exception_props(_chained(ConnectionError(PRIVATE_MESSAGE), explicit=True))
assert props == {"exception_type": "RuntimeError", "exception_cause": "ConnectionError"}
def test_exception_props_reports_implicit_context() -> None:
props = exception_props(_chained(KeyError("k"), explicit=False))
assert props["exception_cause"] == "KeyError"
def test_exception_props_ignores_suppressed_context() -> None:
exc = _chained(None, explicit=True)
exc.__context__ = KeyError("k")
assert exception_props(exc) == {"exception_type": "RuntimeError"}
def test_exception_props_unwraps_exception_group() -> None:
group = ExceptionGroup("tasks", [TimeoutError("t"), ValueError("v")])
assert exception_props(group) == {"exception_type": "TimeoutError"}
@pytest.mark.parametrize("telemetry", [posthog, scarf])
def test_error_event_carries_phase_and_class_but_no_message(
telemetry: Any,
monkeypatch: pytest.MonkeyPatch,
) -> None:
sent: list[dict[str, Any]] = []
monkeypatch.setattr(telemetry, "_send", lambda event, props: _capture(sent, event, props))
set_scan_phase("sandbox_init")
telemetry.error("scan_failed", RuntimeError(PRIVATE_MESSAGE))
assert len(sent) == 1
event = sent[0]
assert event["event"] == "error"
assert event["error_type"] == "scan_failed"
assert event["phase"] == "sandbox_init"
assert event["exception_type"] == "RuntimeError"
assert PRIVATE_MESSAGE not in repr(event)
@pytest.mark.parametrize("telemetry", [posthog, scarf])
def test_error_event_without_exception_omits_exception_fields(
telemetry: Any,
monkeypatch: pytest.MonkeyPatch,
) -> None:
sent: list[dict[str, Any]] = []
monkeypatch.setattr(telemetry, "_send", lambda event, props: _capture(sent, event, props))
set_scan_phase("startup")
telemetry.error("docker_not_installed")
assert sent[0]["error_type"] == "docker_not_installed"
assert sent[0]["phase"] == "startup"
assert "exception_type" not in sent[0]
assert "exception_cause" not in sent[0]
def test_report_error_fans_out_to_both_backends(monkeypatch: pytest.MonkeyPatch) -> None:
sent: list[dict[str, Any]] = []
monkeypatch.setattr(posthog, "_send", lambda event, props: _capture(sent, event, props))
monkeypatch.setattr(scarf, "_send", lambda event, props: _capture(sent, event, props))
report_error("model_connection_failed", TimeoutError(PRIVATE_MESSAGE))
assert len(sent) == 2
assert {e["error_type"] for e in sent} == {"model_connection_failed"}
assert {e["exception_type"] for e in sent} == {"TimeoutError"}
@pytest.mark.parametrize("telemetry", [posthog, scarf])
def test_scan_ended_prefers_recorded_exit_reason(
telemetry: Any,
monkeypatch: pytest.MonkeyPatch,
) -> None:
state = ReportState()
state.scan_ended_exit_reason = "budget_exceeded"
sent: list[dict[str, Any]] = []
monkeypatch.setattr(telemetry, "_send", lambda event, props: _capture(sent, event, props))
telemetry.end(state, exit_reason="user_exit")
assert sent[0]["event"] == "scan_ended"
assert sent[0]["exit_reason"] == "budget_exceeded"

2
uv.lock generated
View File

@@ -2378,7 +2378,7 @@ wheels = [
[[package]]
name = "strix-agent"
version = "1.6.1"
version = "1.6.2"
source = { editable = "." }
dependencies = [
{ name = "caido-sdk-client" },