Repository navigation
Expand file tree
/
Copy pathtool-add
More file actions
executable file
·306 lines (270 loc) · 12.5 KB
/
Copy pathtool-add
File metadata and controls
executable file
·306 lines (270 loc) · 12.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
#!/usr/bin/env python3
"""tool-add — add a git repo to the Security Toolkit index.
bin/tool-add <git-url> [<git-url> ...] [--dry-run] [--no-clone]
For each URL: dedupe by URL, pick a non-colliding name (repo, else repo-owner),
clone into repos/, auto-tag via AI Agent against tags.json, detect languages
offline, append to index.jsonl (kept sorted by name), and rebuild web/data.js.
Requires ANTHROPIC_API_KEY in the environment for tagging (falls back to
README/language-only tags if the API is unavailable)."""
import json, os, re, sys, time, subprocess, urllib.request, urllib.error
from pathlib import Path
BASE = Path(__file__).resolve().parent.parent
REPOS = BASE / "repos"
INDEX = BASE / "index.jsonl"
TAGS_JSON = BASE / "tags.json"
BUILD = BASE / "web" / "build.py"
# load Tools/.env (KEY=VALUE lines) into the environment if present.
# real exported env vars take precedence (setdefault), so `.env` is a fallback.
def _load_env(path):
try:
for line in open(path, encoding="utf-8"):
line = line.strip()
if line.startswith("export "):
line = line[7:].strip()
if line and not line.startswith("#") and "=" in line:
k, v = line.split("=", 1)
os.environ.setdefault(k.strip(), v.strip().strip('"').strip("'"))
except FileNotFoundError:
pass
_load_env(BASE / ".env")
MODEL = os.environ.get("STK_MODEL", "claude-sonnet-5")
API_KEY = os.environ.get("ANTHROPIC_API_KEY", "")
# ---- vocabulary + resolver -------------------------------------------------
tags = json.load(open(TAGS_JSON))
resolver = {} # alias/canonical -> (canonical, axis)
for axis, entries in tags.items():
for canon, meta in entries.items():
resolver[canon] = (canon, axis)
for al in meta.get("aliases", []):
resolver[al] = (canon, axis)
SEM_AXES = ["tactic", "target", "function", "reference"]
AXIS_HINT = {
"tactic": "kill-chain phase (pick 1-3)",
"target": "platform/environment (pick 0-3, only if clearly applicable)",
"function": "kind of tool / capability (pick 1-3)",
"reference": "ONLY if it is docs/data, not a runnable tool (pick 0-2)",
}
VOCAB = "\n".join(f"{ax} — {AXIS_HINT[ax]}:\n {', '.join(tags[ax].keys())}" for ax in SEM_AXES)
# file-extension -> language, derived from tags.json (single source of truth)
EXT2LANG = { ext.lower(): lang
for lang, meta in tags.get("language", {}).items()
for ext in meta.get("ext", []) }
def detect_langs(repo_dir: Path):
counts = {}
if not repo_dir.is_dir():
return []
for root, dirs, files in os.walk(repo_dir):
if ".git" in root:
continue
if len(Path(root).relative_to(repo_dir).parts) > 2:
dirs[:] = []; continue
for fn in files:
lang = EXT2LANG.get(Path(fn).suffix.lower())
if lang:
counts[lang] = counts.get(lang, 0) + 1
langs = [l for l, c in counts.items() if c >= 3]
if not langs and counts:
langs = [max(counts, key=counts.get)]
return langs
def read_readme(repo_dir: Path):
for name in ("README.md","README","readme.md","README.txt","README.rst","Readme.md"):
p = repo_dir / name
if p.is_file() and p.stat().st_size < 60000:
try: return p.read_text(errors="ignore")
except Exception: pass
return ""
def first_line(text):
for line in text.splitlines():
s = re.sub(r"^[#>*\-\s]+", "", line).strip() # strip md heading/bullet markers
if not s or s.startswith("!["): # skip blanks / images-badges
continue
if s.startswith("<") or "align=" in s or s.startswith("["):
continue # skip raw HTML / badge lines
s = re.sub(r"<[^>]+>", "", s) # strip inline HTML tags
s = re.sub(r"!?\[([^\]]*)\]\([^)]*\)", r"\1", s) # markdown links/images -> text
s = s.strip()
if len(s) >= 15 and re.search(r"[A-Za-z]\s+[A-Za-z]", s): # needs real prose
return s[:150]
return ""
# json_schema constrains the reply to exactly this shape, so the model cannot
# wrap it in prose or emit a second, revised object.
SCHEMA = {
"type": "object",
"properties": {
"tags": {"type": "array", "items": {"type": "string"}},
"desc": {"type": "string"},
},
"required": ["tags", "desc"],
"additionalProperties": False,
}
def call_api(prompt, retries=3):
# thinking is off and effort low: tagging is a short classification, and on
# models with adaptive thinking the reply would otherwise lead with a
# thinking block and burn the token budget before reaching the JSON.
body = json.dumps({"model": MODEL, "max_tokens": 1024,
"thinking": {"type": "disabled"},
"output_config": {"effort": "low",
"format": {"type": "json_schema", "schema": SCHEMA}},
"messages": [{"role": "user", "content": prompt}]}).encode()
for attempt in range(retries):
try:
req = urllib.request.Request("https://api.anthropic.com/v1/messages", data=body,
headers={"x-api-key": API_KEY, "anthropic-version": "2023-06-01",
"content-type": "application/json"})
with urllib.request.urlopen(req, timeout=90) as r:
reply = json.load(r)
blocks = reply["content"]
# never index content[0] — a thinking block can precede the text
text = next((b["text"] for b in blocks if b["type"] == "text"), None)
if text is None:
types = [b.get("type") for b in blocks]
raise RuntimeError(
f"no text block in reply (stop_reason={reply.get('stop_reason')!r}, "
f"block types={types!r})")
return text
except urllib.error.HTTPError as e:
# retry transient errors (rate limit / overloaded / 5xx); re-raise the rest
if e.code in (429, 500, 502, 503, 529) and attempt < retries - 1:
print(f" … API {e.code}, retrying ({attempt+1}/{retries-1})")
time.sleep(2 * (attempt + 1)); continue
raise
except Exception as e: # timeout / connection reset
if attempt < retries - 1:
print(f" … API {type(e).__name__}, retrying ({attempt+1}/{retries-1})")
time.sleep(2 * (attempt + 1)); continue
raise
def parse_json(text):
try: return json.loads(text) # structured outputs: already bare JSON
except Exception: pass
# fallback for un-constrained replies: decode at each '{' and keep the last
# object that parses — a model that revises itself emits the good one last.
dec, best = json.JSONDecoder(), None
for i, ch in enumerate(text):
if ch != "{":
continue
try: obj, _ = dec.raw_decode(text, i)
except ValueError: continue
if isinstance(obj, dict) and obj.get("tags"):
best = obj
return best
def normalize(raw_tags, langs):
out = []
for t in (raw_tags or []):
s = str(t).strip().lower().strip("#").strip()
if ":" in s: # tolerate "axis: tag" prefixes from the model
s = s.split(":")[-1].strip()
hit = resolver.get(s)
if hit: out.append(hit[0])
out.extend(langs)
return sorted(set(out))
PROMPT = """You are tagging a cybersecurity tool for a searchable index.
Tool name: {name}
Repo URL: {url}
README excerpt:
{readme}
Choose tags ONLY from this controlled vocabulary, using the EXACT spellings shown.
Do not invent tags. Omit an axis if nothing fits.
{vocab}
Also write a concise one-line description (max 150 chars) of what the tool does — concrete, no marketing.
Respond with ONLY this JSON, nothing else:
{{"tags": ["..."], "desc": "..."}}"""
def tag_tool(name, url, repo_dir):
langs = detect_langs(repo_dir)
readme = read_readme(repo_dir)
if not API_KEY:
print(" ! no ANTHROPIC_API_KEY — falling back to offline tags")
else:
try:
raw = call_api(
PROMPT.format(name=name, url=url, readme=(readme[:1600] or "(no README)"), vocab=VOCAB))
parsed = parse_json(raw)
if parsed and parsed.get("tags"):
tags_ = normalize(parsed["tags"], langs)
desc = str(parsed.get("desc", "")).strip()[:200] or first_line(readme)
return tags_, desc, MODEL
print(f" ! could not parse tags from reply — falling back to offline tags"
f"\n reply began: {raw[:120]!r}")
except urllib.error.HTTPError as e:
print(f" ! API error {e.code} — falling back to offline tags")
except Exception as e:
print(f" ! API call failed ({e}) — falling back to offline tags")
# offline fallback
return sorted(set(langs)), first_line(readme), "offline"
# ---- git helpers -----------------------------------------------------------
def parse_git(url):
u = url.strip().rstrip("/")
ssh = re.match(r"git@[^:]+:(.+)", u)
path = ssh.group(1) if ssh else re.sub(r"^https?://[^/]+/", "", u)
path = re.sub(r"\.git$", "", path)
parts = [p for p in path.split("/") if p]
if len(parts) >= 2: return parts[-2], parts[-1]
return "", (parts[-1] if parts else "")
def norm_url(u):
return re.sub(r"\.git$", "", (u or "").strip().rstrip("/")).lower()
# ---- index io --------------------------------------------------------------
def load_index():
return [json.loads(l) for l in open(INDEX, encoding="utf-8") if l.strip()]
def save_index(records):
records.sort(key=lambda r: r["name"].casefold())
with open(INDEX, "w", encoding="utf-8") as f:
for r in records:
f.write(json.dumps(r, ensure_ascii=False, separators=(",", ":")) + "\n")
def rebuild_web():
if BUILD.exists():
subprocess.run([sys.executable, str(BUILD)], check=False)
# ---- main ------------------------------------------------------------------
def main(argv):
dry = "--dry-run" in argv
noclone = "--no-clone" in argv or dry
urls = [a for a in argv if not a.startswith("--")]
if not urls:
print(__doc__); return 1
records = load_index()
names = {r["name"] for r in records}
urlset = {norm_url(r.get("url", "")) for r in records}
added = 0
for url in urls:
owner, repo = parse_git(url)
if not repo:
print(f"✗ {url}: could not parse repo name"); continue
nu = norm_url(url)
if nu in urlset:
existing = next((r["name"] for r in records if norm_url(r.get("url","")) == nu), "?")
print(f"• {url}: already indexed as '{existing}' — skipping"); continue
# resolve a non-colliding name
name = repo
if name in names:
name = f"{repo}-{owner}"
n = 2
while name in names:
name = f"{repo}-{owner}-{n}"; n += 1
print(f" name '{repo}' taken → using '{name}'")
dest = REPOS / name
print(f"→ {name} ({url})")
if dry:
print(f" [dry-run] would clone into repos/{name} and tag");
names.add(name); urlset.add(nu); continue
if dest.exists():
print(f" ✗ repos/{name} already exists on disk — skipping"); continue
if not noclone:
r = subprocess.run(["git", "clone", "--depth", "1", url, str(dest)],
capture_output=True, text=True)
if r.returncode != 0:
print(f" ✗ git clone failed: {r.stderr.strip().splitlines()[-1] if r.stderr.strip() else '?'}")
continue
tags_, desc, how = tag_tool(name, url, dest)
rec = {"name": name, "url": url, "tags": tags_, "desc": desc}
records.append(rec); names.add(name); urlset.add(nu); added += 1
print(f" ✓ tagged ({how}): {tags_}")
print(f" {desc or '(no description)'}")
if added and not dry:
save_index(records)
rebuild_web()
print(f"\n✓ added {added} tool(s) → index.jsonl re-sorted, web/data.js rebuilt")
elif dry:
print("\n[dry-run] no changes written")
else:
print("\nnothing added")
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))