-
Notifications
You must be signed in to change notification settings - Fork 171
Expand file tree
/
Copy pathconfigure.py
More file actions
executable file
·1447 lines (1292 loc) · 59.3 KB
/
Copy pathconfigure.py
File metadata and controls
executable file
·1447 lines (1292 loc) · 59.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""Mantis Configuration Manager.
Configures workflow.json with sandboxes (static-only, gvisor, microsandbox, gce),
model selection (Gemini, Claude, GLM, OpenAI-compatible), and fast preflight testing.
Supports local configuration overlays via workflow.local.json.
"""
import argparse
import asyncio
import json
import os
import platform
import shutil
import subprocess
import sys
import tempfile
import time
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple, Union
# Ensure reference root is in sys.path when script is run directly
_REF_ROOT = str(Path(__file__).resolve().parent.parent)
if _REF_ROOT not in sys.path:
sys.path.insert(0, _REF_ROOT)
from core.config import (
DEFAULT_MODEL,
PLACEHOLDER_STRINGS,
RECOMMENDED_MODELS,
SUPPORTED_SANDBOXES,
get_llm_kwargs,
is_placeholder,
normalize_model_id,
)
# Guest images the microsandbox tier can run. The first is the operator-built
# image from ./install.sh (sandbox/Dockerfile); the fallbacks are the pinned
# base images install.sh pulls directly when no container builder is present
# (mirror.gcr.io first because some networks block Docker Hub). Keep this list
# in sync with the FALLBACK_IMAGES list in install.sh.
MICROSANDBOX_DEFAULT_IMAGE = "mantis-sandbox:latest"
MICROSANDBOX_FALLBACK_IMAGES = (
"mirror.gcr.io/library/python:3-slim",
"docker.io/library/python:3-slim",
)
def get_local_workflow_path(base_workflow_path: str) -> str:
"""Returns path to local workflow overlay adjacent to base workflow path."""
base_dir = os.path.dirname(os.path.abspath(base_workflow_path))
base_name = os.path.basename(base_workflow_path)
stem = base_name[:-5] if base_name.endswith(".json") else base_name
return os.path.join(base_dir, f"{stem}.local.json")
def merge_dicts(base: dict, overlay: dict) -> dict:
"""Deeply merges overlay dictionary into base dictionary.
Cleanly replaces sandbox configurations when switching sandbox mechanisms
or resetting options to avoid inheriting incompatible base options.
"""
merged = dict(base)
for k, v in overlay.items():
if k == "sandbox" and isinstance(v, dict):
base_sb = merged.get("sandbox", {}) if isinstance(merged.get("sandbox"), dict) else {}
base_type = base_sb.get("type")
new_type = v.get("type", base_type)
if new_type in ("static-only", "static"):
merged["sandbox"] = {"type": new_type, "options": {}}
sb_proj = v.get("options", {}).get("project") if isinstance(v.get("options"), dict) else None
if sb_proj and not merged.get("project"):
merged["project"] = sb_proj
elif new_type != base_type or ("options" in v and v["options"] == {}):
merged["sandbox"] = dict(v)
if "options" not in merged["sandbox"] or not isinstance(merged["sandbox"]["options"], dict):
merged["sandbox"]["options"] = {}
else:
merged["sandbox"] = {
"type": new_type,
"options": merge_dicts(
base_sb.get("options", {}) if isinstance(base_sb.get("options"), dict) else {},
v.get("options", {}) if isinstance(v.get("options"), dict) else {},
),
}
elif k in merged and isinstance(merged[k], dict) and isinstance(v, dict):
merged[k] = merge_dicts(merged[k], v)
else:
merged[k] = v
return merged
def find_workflow_json(custom_path: str = "") -> str:
"""Discovers workflow.json across standard reference package locations.
Never probes untrusted $CWD/workflow.json by default to prevent repository
graph hijacking. Explicit custom paths must be supplied via CLI/argument.
"""
if custom_path:
return os.path.abspath(custom_path)
# SECURITY (INV-4): every candidate below is derived from __file__ (the installed
# package location). $CWD is never probed: launch.py resolves the operator's
# configured sandbox through this function, so a workflow.json planted in an
# untrusted checkout would otherwise dictate the execution graph and sandbox policy.
package_wf = os.path.join(Path(__file__).resolve().parent.parent, "workflow.json")
parent_ref_wf = os.path.join(Path(__file__).resolve().parent.parent.parent, "reference", "workflow.json")
for candidate in [package_wf, parent_ref_wf]:
if os.path.exists(candidate):
return os.path.abspath(candidate)
return os.path.abspath(package_wf)
def load_workflow_dict(workflow_path: str, load_local: bool = True) -> dict:
"""Loads workflow JSON dictionary from path or returns a default template.
If load_local is True, checks for workflow.local.json (or .workflow.local.json)
adjacent to workflow_path and merges its configuration on top.
"""
data = None
if os.path.exists(workflow_path):
try:
with open(workflow_path, "r", encoding="utf-8") as f:
loaded = json.load(f)
if isinstance(loaded, dict):
data = loaded
except Exception:
pass
if data is None:
data = {
"name": "mantis_vulnerability_pipeline",
"config": {
"db_path": "knowledge.db",
"retry_attempts": 3,
"default_model": DEFAULT_MODEL,
"reasoning_effort": "medium",
"seed_prompt": "Initial Task Input: Evaluate {filepath}",
"sandbox": {"type": "static-only", "options": {}},
},
"nodes": [],
"edges": [],
}
if load_local:
base_dir = os.path.dirname(os.path.abspath(workflow_path))
base_name = os.path.basename(workflow_path)
stem = base_name[:-5] if base_name.endswith(".json") else base_name
local_candidates = [
os.path.join(base_dir, f"{stem}.local.json"),
os.path.join(base_dir, f".{stem}.local.json"),
]
abs_wf = os.path.abspath(workflow_path)
for cand in local_candidates:
if os.path.exists(cand) and os.path.abspath(cand) != abs_wf:
try:
with open(cand, "r", encoding="utf-8") as lf:
local_data = json.load(lf)
if isinstance(local_data, dict):
if "config" in local_data and isinstance(local_data["config"], dict):
data["config"] = merge_dicts(data.get("config", {}), local_data["config"])
for top_k in (
"sandbox",
"default_model",
"api_base",
"timeout",
"reasoning_effort",
"db_path",
"retry_attempts",
"seed_prompt",
):
if top_k in local_data and (
"config" not in local_data
or top_k not in local_data.get("config", {})
):
if top_k == "sandbox" and isinstance(local_data[top_k], dict):
data.setdefault("config", {})["sandbox"] = merge_dicts(
{"sandbox": data.get("config", {}).get("sandbox", {})},
{"sandbox": local_data["sandbox"]},
)["sandbox"]
elif isinstance(local_data[top_k], dict) and isinstance(data.get("config", {}).get(top_k), dict):
data.setdefault("config", {})[top_k] = merge_dicts(
data.get("config", {}).get(top_k, {}), local_data[top_k]
)
else:
data.setdefault("config", {})[top_k] = local_data[top_k]
for k in ("name", "nodes", "edges"):
if k in local_data:
data[k] = local_data[k]
break
except Exception as e:
print(f"[CONFIG WARNING] Could not load local overlay {cand}: {e}")
return data
def check_microsandbox_virtualization() -> Tuple[bool, str]:
"""Platform-aware check that the microsandbox tier can boot microVMs here.
Linux: requires a readable/writable /dev/kvm. macOS (Apple Silicon):
libkrun rides Hypervisor.framework -- no /dev/kvm exists, the bundled
libkrunfw in the microsandbox wheel is the requirement. Mirrors the
platform guard in MicrosandboxEnvironment.__init__.
"""
if sys.platform.startswith("linux"):
if not os.path.exists("/dev/kvm"):
return False, "/dev/kvm device does not exist."
if not os.access("/dev/kvm", os.R_OK | os.W_OK):
return False, "Current user lacks read/write permissions on /dev/kvm."
return True, "KVM virtualization available (/dev/kvm accessible)."
if sys.platform == "darwin":
if platform.machine() != "arm64":
return False, "microsandbox on macOS requires Apple Silicon (arm64)."
try:
import microsandbox as _msb # noqa: F401
except Exception as e:
return False, f"microsandbox package not importable: {e}"
return True, "Hypervisor.framework virtualization available (Apple Silicon + microsandbox wheel)."
return False, f"microsandbox is not supported on platform '{sys.platform}'."
def find_cached_microsandbox_image(preferred: str = "") -> Optional[str]:
"""Returns the first locally cached guest image usable by the microsandbox tier.
The runtime boots with PullPolicy.NEVER, so an image missing from the local
cache means every reproducer campaign fails at first execute. Checks the
operator's configured image first, then the install.sh-built image, then
the pinned fallback base images. Returns None when nothing is cached.
"""
candidates = []
if preferred:
candidates.append(preferred)
candidates.append(MICROSANDBOX_DEFAULT_IMAGE)
candidates.extend(MICROSANDBOX_FALLBACK_IMAGES)
async def _probe() -> Optional[str]:
from microsandbox import Image, ImageNotFoundError
for cand in candidates:
try:
await Image.get(cand)
return cand
except ImageNotFoundError:
continue
except Exception:
# Cache DB unavailable (locked/permissions): treat as no image
# rather than crashing configuration.
return None
return None
try:
try:
loop = asyncio.get_running_loop()
except RuntimeError:
loop = None
if loop and loop.is_running():
import concurrent.futures
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
return executor.submit(asyncio.run, _probe()).result()
return asyncio.run(_probe())
except Exception:
return None
def detect_capabilities() -> dict:
"""Inspects the local host environment to detect available sandboxes, tools, and credentials."""
caps: dict[str, Any] = {
"kvm": False,
"docker": False,
"podman": False,
"container_tool": None,
"runsc": False,
"gcloud": False,
"gcp_auth": False,
"gcp_account": None,
"gcp_project": None,
"vertex_project": os.environ.get("VERTEXAI_PROJECT")
or os.environ.get("GOOGLE_CLOUD_PROJECT"),
"gemini_api_key": bool(
os.environ.get("GEMINI_API_KEY") or os.environ.get("GOOGLE_API_KEY")
),
"anthropic_api_key": bool(os.environ.get("ANTHROPIC_API_KEY")),
"openai_api_key": bool(os.environ.get("OPENAI_API_KEY")),
"llm_api_base": os.environ.get("LLM_API_BASE"),
"microsandbox_image": None,
"recommended_sandbox": "static-only",
"available_sandboxes": ["static-only"],
}
# 1. Check virtualization for microsandbox (Linux KVM or macOS
# Hypervisor.framework -- see check_microsandbox_virtualization).
virt_ok, _ = check_microsandbox_virtualization()
if virt_ok:
caps["kvm"] = os.path.exists("/dev/kvm") and os.access("/dev/kvm", os.R_OK | os.W_OK)
caps["available_sandboxes"].append("microsandbox")
caps["microsandbox_image"] = find_cached_microsandbox_image()
# 2. Check container engines and gVisor
for tool in ("docker", "podman"):
if shutil.which(tool):
caps[tool] = True
if not caps["container_tool"]:
caps["container_tool"] = tool
# Check runsc runtime
try:
out = subprocess.run(
[tool, "info", "--format", "{{json .Runtimes}}"],
capture_output=True,
text=True,
timeout=3,
)
if out.returncode == 0 and "runsc" in out.stdout:
caps["runsc"] = True
if "gvisor" not in caps["available_sandboxes"]:
caps["available_sandboxes"].append("gvisor")
except Exception:
pass
# 3. Check gcloud CLI and GCP Auth
gcloud_bin = shutil.which("gcloud")
if gcloud_bin:
caps["gcloud"] = True
try:
p_auth = subprocess.run(
[
gcloud_bin,
"auth",
"list",
"--filter=status:ACTIVE",
"--format=value(account)",
],
capture_output=True,
text=True,
timeout=4,
)
if p_auth.returncode == 0 and p_auth.stdout.strip():
caps["gcp_auth"] = True
caps["gcp_account"] = p_auth.stdout.strip().splitlines()[0].strip()
p_proj = subprocess.run(
[gcloud_bin, "config", "get-value", "project"],
capture_output=True,
text=True,
timeout=3,
)
if p_proj.returncode == 0:
val = p_proj.stdout.strip()
if val and "unset" not in val and not is_placeholder(val):
caps["gcp_project"] = val
except Exception:
pass
if caps["gcloud"] and caps["gcp_auth"] and (caps["gcp_project"] or caps["vertex_project"]):
caps["available_sandboxes"].append("gce")
# Select recommendation hierarchy. Microsandbox ranks first when a guest
# image is cached: it is the only tier whose runnability is fully verified
# client-side (hardware microVM + image present), whereas 'gce' merely
# having gcloud auth says nothing about the pre-provisioned VPC/subnet/VM
# image it needs, and gvisor shares the host kernel. Without a cached
# image microsandbox is NOT recommended: the runtime never pulls
# (PullPolicy.NEVER), so every campaign would fail at first execute.
if "microsandbox" in caps["available_sandboxes"] and caps["microsandbox_image"]:
caps["recommended_sandbox"] = "microsandbox"
elif "gce" in caps["available_sandboxes"] and caps["gcp_project"]:
caps["recommended_sandbox"] = "gce"
elif "gvisor" in caps["available_sandboxes"]:
caps["recommended_sandbox"] = "gvisor"
else:
caps["recommended_sandbox"] = "static-only"
return caps
def is_default_or_unconfigured(config: dict) -> Tuple[bool, List[str]]:
"""Evaluates if workflow config contains default placeholders or unconfigured settings."""
issues = []
if not isinstance(config, dict):
return True, ["Config is not a valid dictionary."]
sb = config.get("sandbox", {})
sb_type = sb.get("type", "static-only") if isinstance(sb, dict) else "static-only"
sb_opts = sb.get("options", {}) if isinstance(sb, dict) else {}
# Sandbox checks
if sb_type == "gce":
proj = sb_opts.get("project")
if is_placeholder(proj):
env_proj = os.environ.get("GOOGLE_CLOUD_PROJECT") or os.environ.get("VERTEXAI_PROJECT")
if is_placeholder(env_proj):
issues.append(
f"GCE Sandbox 'options.project' contains default placeholder ('{proj}')."
)
if not shutil.which("gcloud"):
issues.append("GCE Sandbox requires 'gcloud' CLI on PATH.")
elif sb_type == "gvisor":
if not shutil.which("docker") and not shutil.which("podman"):
issues.append("gVisor sandbox requires 'docker' or 'podman' on PATH.")
elif sb_type == "microsandbox":
virt_ok, virt_msg = check_microsandbox_virtualization()
if not virt_ok:
issues.append(f"Microsandbox virtualization unavailable: {virt_msg}")
# Model checks
model = config.get("default_model", DEFAULT_MODEL)
if is_placeholder(model):
issues.append(f"Model '{model}' contains placeholder string.")
if str(model).startswith("vertex_ai/"):
proj = (
sb_opts.get("project")
or os.environ.get("VERTEXAI_PROJECT")
or os.environ.get("GOOGLE_CLOUD_PROJECT")
or config.get("project")
)
if is_placeholder(proj):
caps = detect_capabilities()
if not caps.get("gcp_project") and not caps.get("vertex_project"):
issues.append(
f"Vertex AI Model '{model}' requires VERTEXAI_PROJECT / GOOGLE_CLOUD_PROJECT or active gcloud project."
)
return bool(issues), issues
async def _check_sandbox_preflight(sandbox_cfg: dict, target_path: str = "") -> Tuple[bool, str]:
"""Runs the asynchronous preflight check on a sandbox configuration."""
sb_type = sandbox_cfg.get("type", "static-only")
if sb_type in ("static-only", "static"):
return True, "Static-only sandbox ready (dynamic execution disabled)."
if sb_type == "gce":
opts = sandbox_cfg.get("options", {})
proj = opts.get("project") or os.environ.get("GOOGLE_CLOUD_PROJECT") or os.environ.get("VERTEXAI_PROJECT")
if not proj or is_placeholder(proj):
return False, f"GCE Project is unconfigured placeholder '{proj}'."
if not shutil.which("gcloud"):
return False, "'gcloud' CLI tool not found on PATH."
# Fast gcloud auth test
try:
p = subprocess.run(
["gcloud", "auth", "list", "--filter=status:ACTIVE", "--format=value(account)"],
capture_output=True,
text=True,
timeout=5,
cwd=tempfile.gettempdir(),
)
if p.returncode != 0 or not p.stdout.strip():
return False, "No active GCP credentials found in gcloud auth list."
return (
True,
f"GCE credentials & project verified (Project: {proj}). "
f"(Note: Ephemeral VM creation requires pre-provisioned VPC/Subnet/Image per docs/gce_sandbox_setup.md).",
)
except Exception as e:
return False, f"GCE gcloud check failed: {e}"
if sb_type == "gvisor":
raw_tool = sandbox_cfg.get("options", {}).get("container_tool")
if raw_tool and raw_tool not in ("docker", "podman"):
return False, f"Invalid container_tool '{raw_tool}'. Only 'docker' and 'podman' are allowed."
tool = raw_tool or ("docker" if shutil.which("docker") else "podman")
if not tool or not shutil.which(tool):
return False, "Docker or Podman not installed for gVisor sandbox."
try:
p = subprocess.run(
[tool, "info", "--format", "{{json .Runtimes}}"],
capture_output=True,
text=True,
timeout=5,
cwd=tempfile.gettempdir(),
)
if p.returncode != 0:
return False, f"Cannot connect to {tool} daemon."
if "runsc" not in p.stdout:
return False, f"Runtime 'runsc' (gVisor) is not registered in {tool} runtimes."
return True, f"gVisor Sandbox ready ({tool} + runsc)."
except Exception as e:
return False, f"gVisor check failed: {e}"
if sb_type == "microsandbox":
virt_ok, virt_msg = check_microsandbox_virtualization()
if not virt_ok:
return False, virt_msg
configured_image = str(sandbox_cfg.get("options", {}).get("image", "") or "")
cached = find_cached_microsandbox_image(preferred=configured_image)
if configured_image and cached != configured_image:
return False, (
f"Configured guest image '{configured_image}' is not in the local cache "
f"(the sandbox never pulls at run time). Run ./install.sh to provision it."
)
if not cached:
return False, (
"No microsandbox guest image found in the local cache "
"(the sandbox never pulls at run time). Run ./install.sh to provision one."
)
return True, f"Microsandbox ready ({virt_msg} Guest image: {cached})."
return False, f"Unknown sandbox type '{sb_type}'."
def _probe_llm_reachability(
resolved_model: str,
kwargs: dict,
prompt: str = "test",
max_tokens: int = 256,
timeout: float = 15.0,
) -> Tuple[bool, str]:
"""Actively probes LLM endpoint reachability, credentials, and dependencies with a minimal test prompt."""
try:
import litellm
except ImportError:
return False, "LiteLLM is not installed in the current environment."
# For Vertex AI partner models (e.g. vertex_ai/claude-*, vertex_ai/zai_org/*),
# verify that required client dependencies are installed
if resolved_model.startswith("vertex_ai/"):
if "claude" in resolved_model:
try:
import anthropic # noqa: F401
import vertexai # noqa: F401
except ImportError as ie:
return (
False,
f"Missing dependency for Vertex AI partner model '{resolved_model}': {ie}. "
f"Run 'pip install google-cloud-aiplatform anthropic' or re-run './install.sh'."
)
from core.config import (
is_rate_limit_error,
extract_retry_after,
extract_rate_limit_detail,
compute_full_jitter_delay,
is_auth_error,
format_auth_error_message,
)
call_kwargs = dict(kwargs)
call_kwargs["max_tokens"] = max_tokens
call_kwargs["timeout"] = timeout
call_kwargs["model"] = resolved_model
max_probe_attempts = 3
for attempt in range(max_probe_attempts):
try:
response = litellm.completion(
messages=[{"role": "user", "content": prompt}],
**call_kwargs,
)
if response and getattr(response, "choices", None) and len(response.choices) > 0:
return True, f"LLM reachability verified for '{resolved_model}'."
return True, f"LLM probe received response for '{resolved_model}'."
except Exception as e:
err_msg = str(e)
err_type = type(e).__name__
if is_auth_error(e):
return False, format_auth_error_message(e, model=resolved_model)
if "No module named 'vertexai'" in err_msg or "No module named 'anthropic'" in err_msg:
return (
False,
f"Missing dependency for Vertex AI partner models: {err_msg}. "
f"Run 'pip install google-cloud-aiplatform anthropic' or re-run './install.sh'."
)
if is_rate_limit_error(e):
if attempt + 1 < max_probe_attempts:
retry_after = extract_retry_after(e)
delay = compute_full_jitter_delay(
attempt=attempt,
initial_delay=5.0,
max_delay=30.0,
min_offset=5.0,
retry_after=retry_after,
)
time.sleep(delay)
continue
detail = extract_rate_limit_detail(e)
return (
True,
f"LLM reachability verified for '{resolved_model}' (Endpoint & credentials verified; currently rate-limited: {detail}).",
)
return False, f"LLM reachability probe failed ({err_type}): {err_msg}"
return False, f"LLM reachability probe failed after {max_probe_attempts} attempts."
def _check_llm_preflight(config: dict, probe: bool = False) -> Tuple[bool, str]:
"""Fast validation of LLM configuration and credentials in ~1s.
If probe is True (or MANTIS_PROBE_LLM=1), actively tests model reachability
and client dependencies by sending a test prompt with max_tokens=256.
"""
model = config.get("default_model", DEFAULT_MODEL)
api_base = config.get("api_base")
timeout = config.get("timeout")
effort = config.get("reasoning_effort")
try:
resolved_model, kwargs = get_llm_kwargs(
model_id=model,
api_base=api_base,
timeout=timeout,
reasoning_effort=effort,
config=config,
)
except Exception as e:
return False, f"LLM Configuration Error: {e}"
if resolved_model.startswith("vertex_ai/openai/"):
proj = kwargs.get("vertex_project")
if not proj or is_placeholder(proj):
if not api_base:
return False, "Vertex AI OpenAI model requires a valid GCP Project ID or --api-base endpoint."
endpoint_info = f" @ {api_base}" if api_base else ""
static_msg = f"Vertex AI OpenAI LLM configured (Model: {resolved_model}{endpoint_info})."
elif resolved_model.startswith("vertex_ai/"):
proj = kwargs.get("vertex_project")
if not proj or is_placeholder(proj):
return False, "Vertex AI requires a valid GCP Project ID."
static_msg = f"Vertex AI LLM configured (Model: {resolved_model}, Project: {proj})."
elif resolved_model.startswith("anthropic/"):
if not os.environ.get("ANTHROPIC_API_KEY"):
return False, "Anthropic model requires ANTHROPIC_API_KEY environment variable."
static_msg = f"Anthropic LLM configured (Model: {resolved_model})."
elif resolved_model.startswith("openai/") or api_base:
if not os.environ.get("OPENAI_API_KEY") and not api_base:
return False, "OpenAI model requires OPENAI_API_KEY or --api-base endpoint."
endpoint_info = f" @ {api_base}" if api_base else ""
static_msg = f"OpenAI-compatible LLM configured (Model: {resolved_model}{endpoint_info})."
elif resolved_model.startswith("gemini-"):
if not os.environ.get("GEMINI_API_KEY") and not os.environ.get("GOOGLE_API_KEY"):
# Check if vertex credentials are available
if not os.environ.get("GOOGLE_CLOUD_PROJECT") and not os.environ.get("VERTEXAI_PROJECT") and not kwargs.get("vertex_project"):
return False, "Gemini model requires GEMINI_API_KEY or GCP Project ID for Vertex AI."
static_msg = f"Gemini LLM configured (Model: {resolved_model})."
else:
static_msg = f"LLM configured (Model: {resolved_model})."
should_probe = probe or os.environ.get("MANTIS_PROBE_LLM") in ("1", "true", "True")
if should_probe:
probe_timeout = timeout if timeout is not None else 15.0
probe_ok, probe_msg = _probe_llm_reachability(
resolved_model, kwargs, prompt="test", max_tokens=256, timeout=probe_timeout
)
if not probe_ok:
return False, probe_msg
return True, f"{static_msg} [Live probe: OK]"
return True, static_msg
async def run_preflight_checks_async(
config: dict,
test_llm: bool = True,
test_sandbox: bool = True,
target_path: str = "",
probe_llm: bool = False,
) -> Tuple[bool, List[str]]:
"""Runs combined LLM and Sandbox preflight testing asynchronously in ~1-2s."""
messages = []
all_ok = True
if test_llm:
ok, msg = _check_llm_preflight(config, probe=probe_llm)
messages.append(f"[LLM PREFLIGHT] {'✅ PASSED' if ok else '❌ FAILED'}: {msg}")
if not ok:
all_ok = False
if test_sandbox:
sb_cfg = config.get("sandbox", {}) if isinstance(config, dict) else {}
ok, msg = await _check_sandbox_preflight(sb_cfg, target_path=target_path)
messages.append(f"[SANDBOX PREFLIGHT] {'✅ PASSED' if ok else '❌ FAILED'}: {msg}")
if not ok:
all_ok = False
return all_ok, messages
def run_preflight_checks(
config: dict,
test_llm: bool = True,
test_sandbox: bool = True,
target_path: str = "",
probe_llm: bool = False,
) -> Tuple[bool, List[str]]:
"""Runs combined LLM and Sandbox preflight testing safely in sync contexts."""
try:
loop = asyncio.get_running_loop()
except RuntimeError:
loop = None
if loop and loop.is_running():
import concurrent.futures
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
return executor.submit(
asyncio.run,
run_preflight_checks_async(
config,
test_llm=test_llm,
test_sandbox=test_sandbox,
target_path=target_path,
probe_llm=probe_llm,
),
).result()
else:
return asyncio.run(
run_preflight_checks_async(
config,
test_llm=test_llm,
test_sandbox=test_sandbox,
target_path=target_path,
probe_llm=probe_llm,
)
)
def update_workflow_config(
workflow_path: str,
updates: dict,
save: bool = True,
update_all_nodes: bool = False,
save_tracked: bool = False,
) -> dict:
"""Updates workflow configuration.
If save_tracked is True, modifies base workflow.json directly.
If save_tracked is False, applies updates to workflow data and saves local config
to workflow.local.json, preserving tracked workflow.json.
"""
wf_data = load_workflow_dict(workflow_path, load_local=not save_tracked)
cfg = wf_data.setdefault("config", {})
if "default_model" in updates:
cfg["default_model"] = updates["default_model"]
if "api_base" in updates:
cfg["api_base"] = updates["api_base"]
if "timeout" in updates:
cfg["timeout"] = updates["timeout"]
if "reasoning_effort" in updates:
cfg["reasoning_effort"] = updates["reasoning_effort"]
if "db_path" in updates:
cfg["db_path"] = updates["db_path"]
if "project" in updates:
cfg["project"] = updates["project"]
if "sandbox" in updates:
sb_update = updates["sandbox"]
if isinstance(sb_update, str):
cfg["sandbox"] = {"type": sb_update, "options": {}}
elif isinstance(sb_update, dict):
current_sb = cfg.setdefault("sandbox", {})
if "type" in sb_update:
new_type = sb_update["type"]
current_sb["type"] = new_type
if new_type in ("static-only", "static"):
current_sb["options"] = {}
elif "options" in sb_update and isinstance(sb_update["options"], dict):
current_sb["options"] = dict(sb_update["options"])
elif "options" in sb_update and isinstance(sb_update["options"], dict):
current_opts = current_sb.setdefault("options", {})
current_opts.update(sb_update["options"])
for k, v in sb_update.items():
if k not in ("type", "options"):
current_sb.setdefault("options", {})[k] = v
if update_all_nodes and "default_model" in updates:
for node in wf_data.get("nodes", []):
if node.get("type") == "agent":
node["model"] = updates["default_model"]
if save:
if save_tracked:
save_path = workflow_path
os.makedirs(os.path.dirname(os.path.abspath(save_path)), exist_ok=True)
with open(save_path, "w", encoding="utf-8") as f:
json.dump(wf_data, f, indent=2)
f.write("\n")
else:
save_path = get_local_workflow_path(workflow_path)
os.makedirs(os.path.dirname(os.path.abspath(save_path)), exist_ok=True)
local_payload = {"config": cfg}
if update_all_nodes and "nodes" in wf_data:
local_payload["nodes"] = wf_data["nodes"]
with open(save_path, "w", encoding="utf-8") as f:
json.dump(local_payload, f, indent=2)
f.write("\n")
return wf_data
def _refuse_silent_downgrade(reason: str) -> None:
"""Fails closed instead of silently downgrading a configured sandbox to static-only.
In static-only mode dynamic isolation is absent. Downgrading must be an
explicit operator decision via MANTIS_ALLOW_SANDBOX_DOWNGRADE=1.
"""
if os.environ.get("MANTIS_ALLOW_SANDBOX_DOWNGRADE") == "1":
return
print(f"ERROR: {reason}", file=sys.stderr)
print(
" Refusing to silently downgrade sandbox to 'static-only' "
"(dynamic exploit reproduction and patch verification would be skipped).",
file=sys.stderr,
)
print(
" To proceed with a degraded, session-only static scan, re-run with "
"MANTIS_ALLOW_SANDBOX_DOWNGRADE=1. Downgrades are never persisted.",
file=sys.stderr,
)
raise SystemExit(2)
def _sandbox_choice_is_explicit(workflow_path: str) -> bool:
"""True when the operator explicitly chose a sandbox tier.
An explicit choice is a 'sandbox' key in the local overlay
(workflow.local.json -- written by the wizard, --sandbox, or a previous
auto-promotion) or a non-static sandbox committed in the base
workflow.json. The tracked base file keeps a 'static-only' floor by
design, so a bare static-only there is a default, not a decision.
"""
try:
if os.path.exists(workflow_path):
with open(workflow_path, "r", encoding="utf-8") as f:
base = json.load(f)
if isinstance(base, dict):
base_sb = (base.get("config", {}) or {}).get("sandbox", {})
if isinstance(base_sb, dict) and base_sb.get("type") not in (
None,
"",
"static-only",
"static",
):
return True
except Exception:
pass
base_dir = os.path.dirname(os.path.abspath(workflow_path))
base_name = os.path.basename(workflow_path)
stem = base_name[:-5] if base_name.endswith(".json") else base_name
for cand in (
os.path.join(base_dir, f"{stem}.local.json"),
os.path.join(base_dir, f".{stem}.local.json"),
):
if not os.path.exists(cand):
continue
try:
with open(cand, "r", encoding="utf-8") as f:
data = json.load(f)
except Exception:
# Unreadable overlay: treat as explicit so we never overwrite it.
return True
if isinstance(data, dict) and (
"sandbox" in data or "sandbox" in (data.get("config", {}) or {})
):
return True
return False
def _print_static_floor_hint(caps: dict) -> None:
"""One-line, every-launch hint that dynamic reproduction is disabled."""
if "microsandbox" in caps.get("available_sandboxes", []):
print(
"⚠️ Sandbox is 'static-only': this host supports microsandbox but no guest "
"image is cached. Run ./install.sh to enable dynamic exploit reproduction."
)
else:
print(
"⚠️ Sandbox is 'static-only' (no supported isolation tier detected on this host). "
"Dynamic exploit reproduction is DISABLED."
)
async def ensure_configured_async(
workflow_path: str = "",
auto: bool = True,
interactive: bool = False,
overrides: Optional[dict] = None,
save: bool = True,
save_tracked: bool = False,
probe_llm: bool = False,
) -> dict:
"""Ensures workflow configuration is valid asynchronously. Auto-resolves defaults or prompts if needed."""
target_wf = find_workflow_json(workflow_path)
wf_data = load_workflow_dict(target_wf, load_local=not save_tracked)
cfg = wf_data.get("config", {})
overrides = overrides or {}
if overrides:
wf_data = update_workflow_config(
target_wf, overrides, save=False, save_tracked=save_tracked
)
cfg = wf_data.get("config", {})
is_unconf, issues = is_default_or_unconfigured(cfg)
caps = detect_capabilities()
sb_type = cfg.get("sandbox", {}).get("type", "static-only")
sb_opts = dict(cfg.get("sandbox", {}).get("options", {}))
sb_available = sb_type in caps.get("available_sandboxes", ["static-only"])
# Auto-promotion: a 'static-only' floor inherited from the tracked
# workflow.json is a default, not an operator decision. When this host can
# actually run microsandbox (virtualization + cached guest image verified),
# promote to it and persist, instead of silently skipping dynamic
# reproduction forever. An explicit operator choice (any sandbox key in
# workflow.local.json, a non-static tracked type, or a --sandbox override
# this invocation) is never second-guessed.
promote_to_microsandbox = (
auto
and not interactive
and sb_type in ("static-only", "static")
and "sandbox" not in overrides
and caps.get("recommended_sandbox") == "microsandbox"
and bool(caps.get("microsandbox_image"))
and not _sandbox_choice_is_explicit(target_wf)
)
if not is_unconf and not overrides and sb_available and not promote_to_microsandbox:
ok, _ = await run_preflight_checks_async(cfg, probe_llm=probe_llm)
if ok:
if sb_type in ("static-only", "static"):
_print_static_floor_hint(caps)
return cfg
updates: dict[str, Any] = {}
if interactive:
return run_interactive_wizard(target_wf)
if auto:
if promote_to_microsandbox:
msb_image = caps["microsandbox_image"]
updates["sandbox"] = {
"type": "microsandbox",
"options": {"image": msb_image},
}
sb_type = "microsandbox"
sb_opts = {"image": msb_image}
cfg = {**cfg, "sandbox": {"type": "microsandbox", "options": dict(sb_opts)}}
print(
f"🔒 Auto-configured sandbox: 'microsandbox' (networkless hardware microVM, "
f"guest image '{msb_image}'). Persisting to workflow.local.json; "
f"override with --sandbox or configure.py."
)
elif sb_type in ("static-only", "static") and "sandbox" not in overrides:
_print_static_floor_hint(caps)
# Auto-resolve GCE project if unconfigured
if sb_type == "gce":
cur_proj = sb_opts.get("project", "")
if is_placeholder(cur_proj):
resolved_proj = caps.get("gcp_project") or caps.get("vertex_project")
if resolved_proj and "gce" in caps.get("available_sandboxes", []):
sb_opts["project"] = resolved_proj
updates["sandbox"] = {"type": "gce", "options": sb_opts}
os.environ.setdefault("GOOGLE_CLOUD_PROJECT", resolved_proj)
os.environ.setdefault("VERTEXAI_PROJECT", resolved_proj)
else:
_refuse_silent_downgrade(
"GCE sandbox credentials/project not configured or unavailable."
)
print(
"⚠️ [REPRO DISABLED] Operator-approved downgrade 'gce' -> 'static-only' for THIS SESSION ONLY."
)
# Session-only: not added to updates, so never persisted to workflow.local.json
cfg = {**cfg, "sandbox": {"type": "static-only", "options": {}}}
elif "gce" not in caps.get("available_sandboxes", []):
_refuse_silent_downgrade(
"Host lacks requirements for 'gce' sandbox (gcloud/auth missing)."
)
print(
"⚠️ [REPRO DISABLED] Operator-approved downgrade 'gce' -> 'static-only' for THIS SESSION ONLY."
)
cfg = {**cfg, "sandbox": {"type": "static-only", "options": {}}}
elif sb_type not in caps.get("available_sandboxes", []):
_refuse_silent_downgrade(
f"Host lacks requirements for '{sb_type}' sandbox."
)
print(
f"⚠️ [REPRO DISABLED] Operator-approved downgrade '{sb_type}' -> 'static-only' for THIS SESSION ONLY."
)
cfg = {**cfg, "sandbox": {"type": "static-only", "options": {}}}
# Auto-resolve Model
model = cfg.get("default_model", DEFAULT_MODEL)
if is_placeholder(model):
updates["default_model"] = DEFAULT_MODEL
elif str(model).startswith("vertex_ai/"):
if not os.environ.get("VERTEXAI_PROJECT") and not os.environ.get("GOOGLE_CLOUD_PROJECT"):
resolved_proj = (
caps.get("gcp_project")
or caps.get("vertex_project")