Skip to content

Commit de3c9ed

Browse files
committed
Add extremely minimalistic ADK reference implementation of Mantis with safe sandboxing
TAG=agy CONV=3fecc2dd-b200-4a0c-b4c0-3a0868b5cc2a Change-Id: I24fdaa9f757358e1eae7cf5582f37726c9900531
1 parent 876a0c8 commit de3c9ed

29 files changed

Lines changed: 1999 additions & 0 deletions

‎.gitignore‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,2 +1,7 @@
11
# Exclude local execution workspace
22
workspace/
3+
4+
# Exclude python cache
5+
__pycache__/
6+
*.pyc
7+
.venv/

‎reference/.gitignore‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,2 @@
1+
.venv/
2+
findings.db

‎reference/README.md‎

Lines changed: 46 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,46 @@
1+
# MVP ADK Reference Implementation
2+
3+
This directory contains a minimalistic reference implementation of Mantis. It is
4+
not production-ready and does not implement the full Mantis pipeline, but is a
5+
working minimal harness that demonstrates how to build your own with the ADK.
6+
7+
## Core Principles
8+
9+
1. We use the ADK workflow graph orchestrator, defined in `workflow.json`.
10+
2. We store results in sqlite.
11+
3. We use minimalistic prompts with only a few agents, rather than a more
12+
complex agent architecture.
13+
4. We demonstrate a plugin mechanism to hook up new sandboxes.
14+
5. We **DO NOT** protect against prompt injection or malicious code. If you want
15+
such a thing you will require significantly more effort to handle that.
16+
17+
## Additional Documentation
18+
19+
The core is the documentation. There may be bugs and sharp edges to be ironed
20+
out, but hopefully the demo at least works out of the box.
21+
22+
## Suggested Integration with Mantis Skills
23+
24+
If you'd like to use the Mantis skills directly in this harness, you could use
25+
something like this:
26+
27+
```python
28+
import pathlib
29+
30+
from google.adk.skills import load_skill_from_dir
31+
from google.adk.tools import skill_toolset
32+
33+
researcher_skill = load_skill_from_dir(
34+
pathlib.path(__file__).parent / "mantis-researcher"
35+
)
36+
37+
mantis_skill_toolset = skill_toolset.SkillToolset(
38+
skills = [researcher_skill]
39+
)
40+
```
41+
42+
And load the skill toolset as a tool for the relevant agents. This is likely to
43+
also be a very useful pattern to customizing different stages of your pipeline
44+
for your own codebases. This will give them better grounding and also more
45+
efficient since they won't have to rediscover some esoteric properties of your
46+
codebase or deployment that makes something a false positive.

‎reference/core/__init__.py‎

Whitespace-only changes.

‎reference/core/config.py‎

Lines changed: 36 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,36 @@
1+
import os
2+
3+
DEFAULT_MODEL = "vertex_ai/gemini-3.6-flash"
4+
5+
def get_llm_kwargs(model_id: str = None, default_model: str = DEFAULT_MODEL) -> tuple[str, dict]:
6+
"""Resolves the LLM mapping details cleanly with precedence: node > MODEL_ID env > default."""
7+
model_id = model_id or os.environ.get("MODEL_ID") or default_model
8+
9+
llm_kwargs = {"model": model_id}
10+
11+
if model_id.startswith("vertex_ai/"):
12+
project = os.environ.get("VERTEXAI_PROJECT", os.environ.get("GOOGLE_CLOUD_PROJECT"))
13+
location = os.environ.get("VERTEXAI_LOCATION", os.environ.get("GOOGLE_CLOUD_LOCATION", "us-central1"))
14+
15+
if not project:
16+
try:
17+
import google.auth
18+
_, project = google.auth.default()
19+
except Exception:
20+
pass
21+
22+
if not project:
23+
raise ValueError("ERROR: You must set VERTEXAI_PROJECT or GOOGLE_CLOUD_PROJECT env variables.")
24+
25+
llm_kwargs["vertex_project"] = project
26+
llm_kwargs["vertex_location"] = location
27+
# Relax safety filters that sometimes trigger erroneously on defensive security
28+
# analysis and vulnerability remediation workflows.
29+
llm_kwargs["safety_settings"] = [
30+
{"category": "HARM_CATEGORY_DANGEROUS_CONTENT", "threshold": "BLOCK_NONE"},
31+
{"category": "HARM_CATEGORY_HATE_SPEECH", "threshold": "BLOCK_NONE"},
32+
{"category": "HARM_CATEGORY_HARASSMENT", "threshold": "BLOCK_NONE"},
33+
]
34+
35+
return model_id, llm_kwargs
36+

‎reference/core/context.py‎

Lines changed: 16 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,16 @@
1+
import contextvars
2+
from dataclasses import dataclass
3+
from typing import Optional
4+
5+
@dataclass
6+
class RunContext:
7+
jail_dir: str
8+
db_path: str
9+
target_file: str = ""
10+
sandbox: object = None
11+
run_id: str = ""
12+
13+
current_run_context: contextvars.ContextVar[Optional[RunContext]] = contextvars.ContextVar(
14+
"current_run_context", default=None
15+
)
16+

‎reference/core/database.py‎

Lines changed: 149 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,149 @@
1+
import sqlite3
2+
import json
3+
from contextlib import contextmanager
4+
from typing import Optional, List, Dict, Any
5+
6+
@contextmanager
7+
def _db(db_path: str):
8+
conn = sqlite3.connect(db_path)
9+
conn.row_factory = sqlite3.Row
10+
try:
11+
yield conn
12+
conn.commit()
13+
finally:
14+
conn.close()
15+
16+
def init_db(db_path: str):
17+
"""Initialize the SQLite database with findings and risk_scores tables and unique indexes."""
18+
with _db(db_path) as conn:
19+
cursor = conn.cursor()
20+
cursor.execute("""
21+
CREATE TABLE IF NOT EXISTS findings (
22+
id INTEGER PRIMARY KEY AUTOINCREMENT,
23+
run_id TEXT,
24+
timestamp DATETIME DEFAULT CURRENT_TIMESTAMP,
25+
filepath TEXT,
26+
title TEXT,
27+
severity TEXT,
28+
description TEXT,
29+
line_numbers TEXT NOT NULL DEFAULT '[]',
30+
remediation TEXT,
31+
status TEXT NOT NULL DEFAULT 'reported',
32+
UNIQUE(filepath, title, description, line_numbers, run_id)
33+
)
34+
""")
35+
cursor.execute("""
36+
CREATE TABLE IF NOT EXISTS risk_scores (
37+
id INTEGER PRIMARY KEY AUTOINCREMENT,
38+
run_id TEXT,
39+
timestamp DATETIME DEFAULT CURRENT_TIMESTAMP,
40+
filepath TEXT,
41+
score INTEGER,
42+
reasoning TEXT,
43+
UNIQUE(filepath, run_id)
44+
)
45+
""")
46+
47+
def write_findings(db_path: str, filepath: str, findings: list, run_id: str = "", status: str = "reported"):
48+
"""Write structured findings to the database contextually associated with real path."""
49+
with _db(db_path) as conn:
50+
cursor = conn.cursor()
51+
for obj in findings:
52+
finding = obj.model_dump() if hasattr(obj, "model_dump") else (obj if isinstance(obj, dict) else dict(obj))
53+
raw_lines = finding.get("line_numbers")
54+
if raw_lines and isinstance(raw_lines, (list, tuple, set)):
55+
try:
56+
line_numbers = json.dumps(sorted(list(raw_lines)))
57+
except Exception:
58+
line_numbers = json.dumps(list(raw_lines))
59+
else:
60+
line_numbers = "[]"
61+
62+
finding_status = finding.get("status") or status
63+
64+
cursor.execute("""
65+
INSERT OR REPLACE INTO findings (run_id, filepath, title, severity, description, line_numbers, remediation, status)
66+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
67+
""", (
68+
run_id,
69+
filepath,
70+
finding.get("title"),
71+
finding.get("severity"),
72+
finding.get("description"),
73+
line_numbers,
74+
finding.get("remediation"),
75+
finding_status,
76+
))
77+
78+
def update_status(db_path: str, filepath: str, run_id: str, status: str):
79+
"""Update status for all findings of a target file in a given run."""
80+
with _db(db_path) as conn:
81+
cursor = conn.cursor()
82+
cursor.execute("""
83+
UPDATE findings
84+
SET status = ?
85+
WHERE filepath = ? AND run_id = ?
86+
""", (status, filepath, run_id))
87+
88+
89+
def read_findings(
90+
db_path: str,
91+
filepath: Optional[str] = None,
92+
run_id: Optional[str] = None,
93+
status: Optional[str] = None,
94+
) -> List[Dict[str, Any]]:
95+
"""Read findings from the database, optionally filtered by filepath, run_id, or status."""
96+
with _db(db_path) as conn:
97+
cursor = conn.cursor()
98+
query = "SELECT * FROM findings WHERE 1=1"
99+
params = []
100+
if filepath:
101+
query += " AND filepath = ?"
102+
params.append(filepath)
103+
if run_id:
104+
query += " AND run_id = ?"
105+
params.append(run_id)
106+
if status:
107+
query += " AND status = ?"
108+
params.append(status)
109+
query += " ORDER BY id ASC"
110+
cursor.execute(query, params)
111+
rows = []
112+
for r in cursor.fetchall():
113+
row_dict = dict(r)
114+
if row_dict.get("line_numbers"):
115+
try:
116+
parsed = json.loads(row_dict["line_numbers"])
117+
row_dict["line_numbers"] = parsed if parsed else None
118+
except Exception:
119+
row_dict["line_numbers"] = None
120+
else:
121+
row_dict["line_numbers"] = None
122+
rows.append(row_dict)
123+
return rows
124+
125+
def record_calibration(db_path: str, filepath: str, score: int, reasoning: str, run_id: str = ""):
126+
"""Record final risk calibration score into the database with idempotent overwrite on retry."""
127+
with _db(db_path) as conn:
128+
cursor = conn.cursor()
129+
cursor.execute("""
130+
INSERT OR REPLACE INTO risk_scores (run_id, filepath, score, reasoning)
131+
VALUES (?, ?, ?, ?)
132+
""", (run_id, filepath, score, reasoning))
133+
134+
def read_risk_scores(db_path: str, filepath: Optional[str] = None, run_id: Optional[str] = None) -> List[Dict[str, Any]]:
135+
"""Read risk scores from the database, optionally filtered by filepath and run_id."""
136+
with _db(db_path) as conn:
137+
cursor = conn.cursor()
138+
query = "SELECT * FROM risk_scores WHERE 1=1"
139+
params = []
140+
if filepath:
141+
query += " AND filepath = ?"
142+
params.append(filepath)
143+
if run_id:
144+
query += " AND run_id = ?"
145+
params.append(run_id)
146+
query += " ORDER BY id ASC"
147+
cursor.execute(query, params)
148+
return [dict(r) for r in cursor.fetchall()]
149+

0 commit comments

Comments
 (0)