Repository navigation
Expand file tree
/
Copy pathclaudit_scan.py
More file actions
executable file
·1724 lines (1534 loc) · 84 KB
/
Copy pathclaudit_scan.py
File metadata and controls
executable file
·1724 lines (1534 loc) · 84 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""claudit_scan — watch all Claude Code sessions for server-side BLOCKS, dedup them,
and file them as GitHub issues (new issue) or comment when one recurs (update).
Files issues for: cybersecurity safety-filter blocks, AUP/Usage-Policy blocks.
Logs but NEVER sends: overloaded/529, rate-limit, usage-limit, any other API error.
Findings dedup by the triggering prompt (retries collapse into one finding with all
Request IDs). State maps each finding -> its issue, so a recurrence with new Request
IDs becomes a comment ("update"), not a duplicate issue.
Usage:
claudit_scan.py # dry-run: list new findings, file nothing
claudit_scan.py --baseline # mark ALL current findings as seen, file nothing
claudit_scan.py --watch # poll forever; file new blocks + comment recurrences
claudit_scan.py --post # one-shot: review backlog in $EDITOR, then file
claudit_scan.py --post --no-review
Flags: --interval N (watch poll secs, default 30), --delay N (secs between posts,
default 3), --limit N (0=all), -R owner/repo.
"""
import argparse
import atexit
import collections
import hashlib
import json
import os
import re
import platform
import shutil
import subprocess
import sys
import tempfile
import threading
import time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import claudit # noqa: E402 (LLM_SCRUB flag + llm_redact)
from claudit import scrub # noqa: E402 (reuse the PII scrubber)
# Launched from a desktop icon, PATH is often minimal and `gh`/`claude` aren't found.
# Make sure the interpreter's own bin dir (where gh usually lives) + common bins are on PATH.
for _p in (os.path.dirname(sys.executable), "/usr/local/bin", "/opt/homebrew/bin",
os.path.expanduser("~/.local/bin")):
if _p and _p not in os.environ.get("PATH", "").split(os.pathsep):
os.environ["PATH"] = _p + os.pathsep + os.environ.get("PATH", "")
PROJECTS = os.path.expanduser("~/.claude/projects")
STATE_DIR = os.path.expanduser("~/.claude/claudit")
STATE_FILE = os.path.join(STATE_DIR, "filed.json")
ERROR_LOG = os.path.join(STATE_DIR, "error-log.jsonl")
LOCK_FILE = os.path.join(STATE_DIR, "watcher.lock")
ISSUES_DB = os.path.join(STATE_DIR, "issues.jsonl") # local record of every filed issue
__version__ = "2.0.95"
DEFAULT_REPO = "anthropics/claude-code"
REPORT_HARNESS = False # harness (auto-mode-classifier) denials are LOG-ONLY by default.
# They are local permission decisions, not server-side API false positives,
# and often fire correctly (an agent re-enabling a disabled admin flag, etc.).
# ClAudit files only real API blocks (cyber/aup). Opt in with --report-harness.
GATE = False # opt-in: pre-judge "correct block vs false positive" and drop the former.
# OFF by default — that classification is the unreliable thing ClAudit exists to
# surface, so the filer shouldn't pre-judge it. Enable with --gate / config gate:true.
DWELL_SECONDS = 300 # dwell-auto-file: hold a new Request ID this long (5 min; repeats accrete as
# their own linked issues) before the LLM judges + composes + files. Config: dwell_seconds.
# ClAudit's OWN `claude -p` calls (compose / scrub / gate / dup-defense). When one is blocked it lands
# in a transcript; scan() must skip it or ClAudit reports its own prompts — a feedback loop.
INTERNAL_PROMPTS = (
"Write ONE specific GitHub issue title",
"Write a tight, factual 2-3 sentence explanation",
"strict PII redactor",
"duplicate-detection bot",
"false_positive=TRUE",
)
PROJECT_URL = "https://github.com/sworrl/ClAudit" # issues link back here for transparency
ICON = os.path.join(os.path.dirname(os.path.abspath(__file__)), "claudit_icon.png")
DEFAULT_NOTE = ("False positive — in-scope, authorized security work; not out of scope. "
"Filed automatically by claudit.")
REQ_ID = re.compile(r"req_[A-Za-z0-9]+")
TOKEN = re.compile(r"token=[A-Za-z0-9_\-]+")
FILE_KINDS = {
"cyber": "Cybersecurity safety-filter false positive",
"aup": "AUP / Usage-Policy block (false positive)",
"harness": "Claude Code harness / auto-mode classifier denial",
}
WHY = {
"cyber": ("Legitimate, in-scope security / defensive / administration work was flagged by the "
"cybersecurity-topic safety classifier — pattern-matched on terminology, not on any "
"harmful intent. Securing or administering one's own systems is the opposite of attacking them."),
"aup": ("A benign, in-scope request was flagged as a Usage-Policy violation — a false positive on "
"ordinary, authorized work."),
"harness": ("The Claude Code auto-mode classifier denied an action during legitimate, authorized, "
"in-scope work. See the block reason below."),
}
def classify(text):
t = text.lower()
# cyber = the cybersecurity-topic classifier flagged it. Match the flagging signature, not one
# exact sentence — Anthropic reworded it from "safety measures" to "Opus 4.8's safeguards" and
# the old fixed strings stopped matching, silently dropping real cyber blocks to 'other'.
# error_text() only ever feeds real API-error messages here, so this can't catch model refusals.
if (("cybersecurity" in t and "flag" in t)
or "safety filter detected cybersecurity" in t):
return "cyber"
if ("violate our usage policy" in t or "unable to respond to this request" in t
or "against our usage policy" in t or "usage policy violation" in t
or "content policy violation" in t):
return "aup"
if "overloaded" in t or "temporarily limiting" in t or "529" in t:
return "overloaded"
if "hit your limit" in t or "rate limit" in t or "429" in t or "· resets" in t:
return "limit"
return "other"
def human_text(entry):
if entry.get("type") != "user":
return None
content = (entry.get("message") or {}).get("content")
if isinstance(content, str):
return content.strip() or None
if isinstance(content, list):
if any(isinstance(b, dict) and b.get("type") == "tool_result" for b in content):
return None
text = "\n".join(b.get("text", "") for b in content
if isinstance(b, dict) and b.get("type") == "text").strip()
return text or None
return None
def error_text(entry):
if not entry.get("isApiErrorMessage"):
return None
content = (entry.get("message") or {}).get("content")
if isinstance(content, str):
return content
if isinstance(content, list):
return "\n".join(b.get("text", "") for b in content
if isinstance(b, dict) and b.get("type") == "text")
return ""
def sig(kind, prompt):
norm = re.sub(r"\s+", " ", prompt.lower()).strip()[:500]
return hashlib.sha1((kind + "|" + norm).encode()).hexdigest()[:12]
def reqs_of(f):
seen, out = set(), []
for o in f["occ"]:
if o["req"] and o["req"] not in seen:
seen.add(o["req"])
out.append(o)
return out
def should_file(f):
"""ClAudit only auto-publishes server-side API false positives (cyber/aup) that carry a Request
ID Anthropic can look up. Harness is log-only. A cyber/aup block with NO Request ID is not
referenceable server-side, so filing it is wasted; skip it."""
return f.get("kind") in ("cyber", "aup") and bool(reqs_of(f))
def harness_denial(entry):
"""Text of a Claude Code auto-mode-classifier denial (a tool_result error), or None."""
if entry.get("type") != "user":
return None
content = (entry.get("message") or {}).get("content")
if not isinstance(content, list):
return None
for b in content:
if not (isinstance(b, dict) and b.get("type") == "tool_result"):
continue
tc = b.get("content")
if isinstance(tc, str):
text = tc
elif isinstance(tc, list):
text = "\n".join(x.get("text", "") for x in tc
if isinstance(x, dict) and x.get("type") == "text")
else:
text = ""
low = text.lower()
if ("auto mode classifier" in low or "permission for this action was denied" in low
or "denied by the auto mode" in low or "action denied by auto mode" in low):
return text.strip()
return None
def assistant_text(entry):
"""Text of a normal (non-error) assistant turn, for capturing conversation leadup."""
if entry.get("type") != "assistant" or entry.get("isApiErrorMessage"):
return None
content = (entry.get("message") or {}).get("content")
if isinstance(content, str):
return content.strip() or None
if isinstance(content, list):
t = "\n".join(b.get("text", "") for b in content
if isinstance(b, dict) and b.get("type") == "text").strip()
return t or None
return None
_SCAN_CACHE = {"t": -1e9, "val": None}
_FILE_CACHE = {} # path -> (mtime, [findings], [log_lines], counts) so unchanged files aren't re-parsed
def _parse_file(path, name, root):
"""Parse ONE session transcript -> (findings list, log lines, logged-only counts). No shared
state, so the result can be cached by mtime and reused until the file changes."""
findings, log_lines = {}, []
counts = {"overloaded": 0, "limit": 0, "other": 0, "harness": 0}
last_prompt, recent = None, collections.deque(maxlen=6)
try:
with open(path, encoding="utf-8") as fh:
for line in fh:
try:
entry = json.loads(line)
except ValueError:
continue
hp = human_text(entry)
if hp:
last_prompt = hp
recent.append(("user", hp[:300]))
continue
hd = harness_denial(entry)
if hd:
ts = entry.get("timestamp", "")
log_lines.append(json.dumps({"kind": "harness", "ts": ts, "session": name, "req": None}))
if not REPORT_HARNESS: # log-only: local classifier decision, not an API FP
counts["harness"] += 1
continue
s = sig("harness", hd[:300])
f = findings.setdefault(s, {"sig": s, "kind": "harness",
"prompt": last_prompt or "(no preceding prompt)",
"occ": [], "block_text": hd, "leadup": list(recent)})
f["occ"].append({"req": None, "ts": ts, "session": name, "proj": os.path.basename(root)})
continue
err = error_text(entry)
if err is None:
at = assistant_text(entry)
if at:
recent.append(("assistant", at[:300]))
continue
if last_prompt and any(mk in last_prompt for mk in INTERNAL_PROMPTS):
continue # ClAudit's own LLM prompt got blocked — never self-report
kind = classify(err)
ts = entry.get("timestamp", "")
m = REQ_ID.search(err)
req = m.group(0) if m else None
log_lines.append(json.dumps({"kind": kind, "ts": ts, "session": name, "req": req}))
if kind not in FILE_KINDS:
counts[kind] = counts.get(kind, 0) + 1
continue
prompt = last_prompt or "(triggering prompt not recoverable)"
s = sig(kind, prompt)
f = findings.setdefault(s, {"sig": s, "kind": kind, "prompt": prompt,
"occ": [], "block_text": err, "leadup": list(recent)})
f["occ"].append({"req": req, "ts": ts, "session": name, "proj": os.path.basename(root)})
except OSError:
pass
return list(findings.values()), log_lines, counts
def scan(ttl=0.0):
"""Walk all sessions -> (findings dict[sig]->finding, logged-only counts). Incremental: each
file is parsed once and cached by mtime, so a re-scan only re-parses the files that changed
(usually just the active session). With ttl>0, reuse the whole result if it's younger than ttl."""
if ttl and _SCAN_CACHE["val"] is not None and time.monotonic() - _SCAN_CACHE["t"] < ttl:
return _SCAN_CACHE["val"]
findings, logged, log_lines = {}, {"overloaded": 0, "limit": 0, "other": 0, "harness": 0}, []
seen = set()
for root, _, files in os.walk(PROJECTS):
for name in files:
if not name.endswith(".jsonl"):
continue
path = os.path.join(root, name)
seen.add(path)
try:
mtime = os.path.getmtime(path)
except OSError:
continue
cached = _FILE_CACHE.get(path)
if cached and cached[0] == mtime:
ff, fl, fc = cached[1], cached[2], cached[3]
else:
ff, fl, fc = _parse_file(path, name, root)
_FILE_CACHE[path] = (mtime, ff, fl, fc)
log_lines.extend(fl)
for k, v in fc.items():
logged[k] = logged.get(k, 0) + v
for f in ff: # merge by signature across files
ex = findings.get(f["sig"])
if ex:
ex["occ"].extend(f["occ"])
else:
findings[f["sig"]] = {**f, "occ": list(f["occ"])} # copy so cache isn't mutated
for p in [p for p in _FILE_CACHE if p not in seen]: # evict deleted files
del _FILE_CACHE[p]
os.makedirs(STATE_DIR, exist_ok=True)
with open(ERROR_LOG, "w", encoding="utf-8") as fh:
fh.write("\n".join(log_lines) + ("\n" if log_lines else ""))
result = (findings, logged)
_SCAN_CACHE["t"], _SCAN_CACHE["val"] = time.monotonic(), result
return result
def _pid_alive(pid):
"""Best-effort cross-platform liveness check."""
if pid <= 0:
return False
if platform.system() == "Windows":
out = subprocess.run(["tasklist", "/FI", f"PID eq {pid}"],
capture_output=True, text=True).stdout
return str(pid) in out
try:
os.kill(pid, 0)
except ProcessLookupError:
return False
except PermissionError:
return True
except OSError:
return False
return True
def _release_singleton():
try:
if os.path.exists(LOCK_FILE) and open(LOCK_FILE).read().strip() == str(os.getpid()):
os.remove(LOCK_FILE)
except OSError:
pass
def acquire_singleton():
"""Self-awareness: ensure only ONE watcher process runs at a time. Returns True if we
hold the lock, False if a live watcher already does (so the caller should exit)."""
os.makedirs(STATE_DIR, exist_ok=True)
if os.path.exists(LOCK_FILE):
try:
other = int((open(LOCK_FILE).read().strip() or "0"))
except (OSError, ValueError):
other = 0
if other and other != os.getpid() and _pid_alive(other):
return False
with open(LOCK_FILE, "w") as fh:
fh.write(str(os.getpid()))
atexit.register(_release_singleton)
return True
CONFIG_FILE = os.path.join(STATE_DIR, "config.json")
def load_config():
try:
with open(CONFIG_FILE, encoding="utf-8") as fh:
return json.load(fh)
except (OSError, ValueError):
return {}
def save_config(cfg):
os.makedirs(STATE_DIR, exist_ok=True)
with open(CONFIG_FILE, "w", encoding="utf-8") as fh:
json.dump(cfg, fh, indent=1)
def load_state():
try:
with open(STATE_FILE, encoding="utf-8") as fh:
data = json.load(fh)
except (OSError, ValueError):
return {}
if isinstance(data, list): # migrate old set-of-sigs format
return {s: {"issue": None, "url": None, "reqs": []} for s in data}
return data
def save_state(state):
os.makedirs(STATE_DIR, exist_ok=True)
with open(STATE_FILE, "w", encoding="utf-8") as fh:
json.dump(state, fh, indent=1, sort_keys=True)
DOMAIN_PATTERNS = [
("cloud-iam", re.compile(r"\b(entra|azure ad|tenant|conditional access|app role|app registration|service principal|okta|oauth|iam)\b", re.I)),
("defensive-hardening", re.compile(r"\b(harden|hardening|mfa|firewall|edr|blue team|patch|cis benchmark|least privilege|lockdown)\b", re.I)),
("reverse-engineering", re.compile(r"\b(disassemble|decompile|ghidra|ida pro|opcode|unpack|reverse engineer)\b", re.I)),
("infra-devops", re.compile(r"\b(kubernetes|docker|terraform|nginx|ansible|systemd|hypervisor|reverse proxy)\b", re.I)),
("web-security", re.compile(r"\b(xss|sql injection|sqli|csrf|ssrf|burp|owasp)\b", re.I)),
("offensive-pentest", re.compile(r"\b(exploit|payload|msfvenom|metasploit|shellcode|reverse shell|\bc2\b|privilege escalation|lateral movement)\b", re.I)),
("crypto-secrets", re.compile(r"\b(encrypt|decrypt|private key|certificate|tls|keystore)\b", re.I)),
("malware-forensics", re.compile(r"\b(malware|forensic|memory dump|yara|incident response|ioc)\b", re.I)),
]
def categorize(f):
"""Heuristic work-domain tag for the report (helps triage which classifier over-fires)."""
text = (f.get("prompt", "") + " " + f.get("block_text", "") + " "
+ " ".join(t for _, t in (f.get("leadup") or []))).lower()
for cat, pat in DOMAIN_PATTERNS:
if pat.search(text):
return cat
return "general"
def project_label(encoded):
"""Turn a ~/.claude/projects dir name into a readable path with the username scrubbed."""
return scrub("/" + encoded.strip("-").replace("-", "/"))[0]
_REFUSAL_MARKERS = (
"i can't write", "i cannot write", "i won't write", "i will not write", "i refuse to",
"i won't do that", "i won't file", "i can't file", "i'm not able to write",
"not a false positive", "is a true positive", "was a true positive", "block is accurate",
"block is correct", "block was accurate", "block was correct", "policy block is accurate",
"would be dishonest", "i don't feel comfortable", "i do not feel comfortable")
def _is_refusal(text):
"""True if LLM output is a refusal / editorializes that the block was CORRECT — never post it;
the model second-guessing the false-positive premise is exactly what must not reach a report."""
low = (text or "").lower()
return any(m in low for m in _REFUSAL_MARKERS)
def _is_meta_reply(text):
"""True if the model is asking for / complaining about missing context instead of writing the
reply (e.g. 'the bot's comment appears to be missing — could you paste it?'). Posting that to a
GitHub issue is the 'not getting the context' garbage; reject it and keep the deterministic note."""
low = (text or "").lower()
return any(m in low for m in (
"appears to be missing", "is blank", "field is blank", "could you paste", "please paste",
"i don't have the specific", "the context you provided", "missing from the context",
"without the bot", "no bot comment", "can't reference", "cannot reference",
"wasn't provided", "was not provided", "you provided"))
def build_issue(f, note, crossref=""):
reqs = reqs_of(f)
first_ts = min((o["ts"] for o in f["occ"] if o["ts"]), default="")
sessions = len({o["session"] for o in f["occ"]})
projects = sorted({project_label(o["proj"]) for o in f["occ"] if o.get("proj")})
proj = projects[0].rstrip("/").split("/")[-1] if projects else "unknown"
# [Bug] first (issue template), then kind; ClAudit-tagged + a distinct discriminator.
if f["kind"] == "harness":
m = re.search(r"Reason:\s*(.+?)(?:\.\s|\n|$)", f["block_text"])
reason = (m.group(1).strip()[:70] if m else f"#{f['sig']}")
title = f"[Bug][harness] ClAudit: auto-mode classifier denied — {reason}"
else:
lead = reqs[0]["req"] if reqs else f"#{f['sig']}"
title = f"[Bug][{f['kind']}] ClAudit false-positive in {proj} — {lead}"
req_lines = "\n".join(f"- `{o['req']}` ({o['ts']})" for o in reqs) or "- (no Request ID captured)"
note_clean, _ = scrub(note or DEFAULT_NOTE)
# Full PII scrub on the block message (was token-only — leaked IPs/hosts/paths).
block_clean = scrub(TOKEN.sub("token=[SCRUBBED]", f["block_text"]))[0].strip()[:500]
if f["kind"] == "harness":
# Show ONLY the classifier's stated reason — never the quoted command (infra/sketch risk).
m = re.search(r"Reason:\s*(.+?)(?:\.\s|\n|If you have|$)", f["block_text"])
block_clean = (scrub(m.group(1).strip())[0][:300] if m
else "(auto-mode classifier denial — see Request IDs)")
leadup = f.get("leadup") or []
leadup_md = "\n".join(f"**{role}:** {scrub(re.sub(chr(10), ' ', txt))[0]}"
for role, txt in leadup) or "_(not captured)_"
why_text = WHY.get(f["kind"], WHY["cyber"])
neutral = False # when the LLM refuses, file FACTS ONLY (type + Request IDs)
if claudit.BURN_TOKENS: # spend tokens to craft a bespoke, specific title + explanation
ctx = f"work domain: {categorize(f)}\nblock message: {block_clean}\nconversation leadup:\n{leadup_md}"
bt = claudit.llm_compose(
f"Write ONE specific GitHub issue title (max ~95 chars) that starts EXACTLY with "
f"'[Bug][{f['kind']}]' and describes the concrete legitimate work this Claude Code safety "
f"block wrongly stopped. Output ONLY that title line — no preamble, no 'here is', no quotes, "
f"no explanation.", ctx)
if bt:
# the model sometimes adds a chatty preamble line — take the line that's actually a title
lines = [ln.strip().strip('"').strip("`") for ln in bt.splitlines() if ln.strip()]
cand = next((ln for ln in lines if ln.lower().startswith("[bug]")), "")
if cand: # only trust a real title; else keep the deterministic one
cand = scrub(cand)[0][:110]
lead_req = reqs[0]["req"] if reqs else ""
title = cand if (not lead_req or lead_req in cand) else f"{cand} ({lead_req})"
bw = claudit.llm_compose(
"Write a tight, factual 2-3 sentence explanation of why this Claude Code safety block is a "
"false positive on legitimate, in-scope work — suitable for a bug report to Anthropic.", ctx)
if bw and not _is_refusal(bw): # never post the model's refusal/editorial
why_text = scrub(bw)[0]
elif bw and _is_refusal(bw): # model wouldn't vouch -> assert NOTHING, file facts only
neutral = True
is_harness = f["kind"] == "harness"
recur = (f"Recurred **{len(f['occ'])}×** across {sessions} session(s); first seen {first_ts}.")
# harness denials are LOCAL auto-mode-classifier blocks — no server-side Request ID exists.
if reqs:
reqs_block = f"### Request IDs (lookup-able server-side)\n{req_lines}"
verify = "the triggering request is verifiable server-side via the Request ID(s) below"
else:
reqs_block = (
"### No Request ID (auto-mode-classifier denial)\n"
"This is a Claude Code **auto-mode-classifier denial** — a local harness block, not a "
"server-side API error — so it carries **no Request ID** to look up. The classifier's "
"stated reason is in the Block message below.")
verify = "the classifier's stated reason is in the Block message below"
blocked_phrase = ("A Claude Code **auto-mode-classifier denial** stopped authorized, in-scope work"
if is_harness else
"A server-side safety/policy block fired during authorized, in-scope work in Claude Code")
if neutral:
kindphrase = ("A Claude Code auto-mode-classifier denial" if is_harness
else f"A server-side **{f['kind']}** block")
why_block = f"{kindphrase} fired in Claude Code. No rationale is asserted — {verify}. {recur}"
note_block = ""
else:
why_block = (f"### Why this is a false positive\n{why_text}\n\n"
f"{blocked_phrase}. Filing as a false positive. {recur}")
note_block = f"\n### In-scope justification\n{note_clean}\n"
crossref_block = (f"\n### Related reports (same work session, linked)\n{scrub(crossref)[0]}\n"
if crossref else "")
# Structured triage line — ClAudit can't add GitHub labels (no triage perm on the repo), so it
# puts the categorization in the body for maintainers to sort on.
repro = ("yes — server-side via the Request ID(s) below" if reqs
else "local auto-mode-classifier denial (no server-side Request ID)")
triage = (f"**Triage:** kind `{f['kind']}` · domain `{categorize(f)}` · "
f"severity **session-halted** (blocked authorized work) · reproducible: {repro}")
body = f"""{triage}
**Type:** {FILE_KINDS[f['kind']]} · **Work domain (heuristic):** `{categorize(f)}`
{why_block}
{reqs_block}
{note_block}
### Block message
> {block_clean}
**Environment:** Claude Code, Linux. · **Work domain:** `{categorize(f)}`
{crossref_block}
---
<sub>🔎 Filed automatically by [ClAudit v{__version__}]({PROJECT_URL}) — a FOSS tool for reporting false-positive Claude Code blocks.</sub>"""
# Title is built from already-scrubbed parts + the Request ID — regex/denylist scrub only
# (never the LLM, so the Request ID survives). Body gets the opt-in LLM pass.
return scrub(title)[0], claudit.llm_redact(body)
def log_issue(f, repo, url):
"""Append a filed issue to the local issues DB (with PII-scrubbed leadup)."""
rec = {
"sig": f["sig"], "kind": f["kind"], "repo": repo, "url": url,
"version": __version__,
"issue": url.rsplit("/", 1)[-1],
"reqs": [o["req"] for o in reqs_of(f)],
"first_ts": min((o["ts"] for o in f["occ"] if o["ts"]), default=""),
"projects": sorted({project_label(o["proj"]) for o in f["occ"] if o.get("proj")}),
"chain": _chain_key(f), # session-precise chain key (no PII: session is a local UUID)
"leadup": [[role, scrub(txt)[0]] for role, txt in (f.get("leadup") or [])],
}
os.makedirs(STATE_DIR, exist_ok=True)
with open(ISSUES_DB, "a", encoding="utf-8") as fh:
fh.write(json.dumps(rec) + "\n")
def gh_create(repo, title, body):
out = subprocess.run(["gh", "issue", "create", "-R", repo, "--title", title, "--body", body],
capture_output=True, text=True, check=True).stdout.strip()
return out # issue URL
def gh_comment(repo, issue, body):
subprocess.run(["gh", "issue", "comment", issue, "-R", repo, "--body", body],
capture_output=True, text=True, check=True)
# ---------------- community poll (reaction-based vote on the pinned issue) ----------------
POLL_REPO = "sworrl/ClAudit"
POLL_ISSUE = 6
# label, GitHub reaction `content`, emoji, short meaning
POLL_OPTS = [("plus", "+1", "👍", "Anthropic will fix it"),
("minus", "-1", "👎", "Claude Code stays broken"),
("eyes", "eyes", "👀", "Too soon to tell")]
def _gh_json(args):
"""Run `gh <args>` and parse stdout as JSON; return None on any failure."""
try:
out = subprocess.run(["gh", *args], capture_output=True, text=True, timeout=30)
return json.loads(out.stdout) if out.returncode == 0 and out.stdout.strip() else None
except Exception:
return None
def gh_login():
"""Current authenticated GitHub login, or '' if unknown."""
d = _gh_json(["api", "user", "--jq", "{login: .login}"])
return (d or {}).get("login", "") if isinstance(d, dict) else ""
def amplify_community(repo, state, me=None, limit=200):
"""Solidarity 👍: react thumbs-up on OTHER ClAudit users' open cyber/aup false-positive issues to
amplify the shared signal. Never touches your own issues. Idempotent via state['__amplified__'].
Returns count newly reacted. A no-op while you are the only reporter."""
me = me or gh_login()
done = state.setdefault("__amplified__", [])
seen = set(done)
issues = _gh_json(["issue", "list", "-R", repo, "--state", "open", "--limit", str(limit),
"--search", '"Filed automatically by ClAudit"',
"--json", "number,author,title"]) or []
n = 0
for it in issues:
num, author = it["number"], (it.get("author") or {}).get("login", "")
title = (it.get("title", "") or "").lower()
if author == me or not author or str(num) in seen:
continue # skip your own + already-reacted
if "[cyber]" not in title and "[aup]" not in title:
continue # real API false positives only
r = subprocess.run(["gh", "api", f"repos/{repo}/issues/{num}/reactions",
"-f", "content=+1"], capture_output=True, text=True)
if r.returncode == 0:
done.append(str(num)); seen.add(str(num)); n += 1
time.sleep(0.5)
if n:
save_state(state)
return n
def poll_counts():
"""{'plus','minus','eyes','total'} from the pinned poll issue's reaction summary."""
d = _gh_json(["api", f"/repos/{POLL_REPO}/issues/{POLL_ISSUE}",
"--jq", '{plus: .reactions."+1", minus: .reactions."-1", eyes: .reactions.eyes}']) or {}
c = {k: int(d.get(k) or 0) for k in ("plus", "minus", "eyes")}
c["total"] = c["plus"] + c["minus"] + c["eyes"]
return c
def poll_vote(choice, me=None):
"""Cast/switch the user's vote. `choice` in {'plus','minus','eyes'}. Enforces one vote per
user: add the chosen reaction, then remove that user's OTHER poll reactions. Returns counts."""
content = dict((o[0], o[1]) for o in POLL_OPTS)[choice]
me = me or gh_login()
subprocess.run(["gh", "api", "-X", "POST",
f"/repos/{POLL_REPO}/issues/{POLL_ISSUE}/reactions", "-f", f"content={content}"],
capture_output=True, text=True)
valid = {o[1] for o in POLL_OPTS}
for r in (_gh_json(["api", "--paginate",
f"/repos/{POLL_REPO}/issues/{POLL_ISSUE}/reactions"]) or []):
if ((r.get("user") or {}).get("login") == me and r.get("content") in valid
and r.get("content") != content):
subprocess.run(["gh", "api", "-X", "DELETE",
f"/repos/{POLL_REPO}/issues/{POLL_ISSUE}/reactions/{r['id']}"],
capture_output=True, text=True)
return poll_counts()
def file_one(f, note, repo, state):
"""Create an issue, or comment if it already exists with fresh Request IDs.
Returns (action, title_or_ref, url) with action in {'new','updated',None}."""
rec = state.get(f["sig"])
cur_reqs = [o["req"] for o in reqs_of(f)]
if rec is None:
# Reserve the signature and persist BEFORE the network call so a concurrent pass
# (or a crash-restart) can never double-file the same finding.
state[f["sig"]] = {"issue": None, "url": None, "kind": f["kind"], "reqs": cur_reqs}
save_state(state)
title, body = build_issue(f, note)
try:
url = gh_create(repo, title, body)
except Exception:
state.pop(f["sig"], None) # release on failure so it can retry later
save_state(state)
raise
state[f["sig"]].update(issue=url.rsplit("/", 1)[-1], url=url)
log_issue(f, repo, url)
return ("new", title, url)
fresh = [r for r in cur_reqs if r not in rec.get("reqs", [])]
if fresh and rec.get("issue"):
gh_comment(repo, rec["issue"], "Recurred again. Additional Request IDs:\n" +
"\n".join(f"- `{r}`" for r in fresh))
rec["reqs"] = rec.get("reqs", []) + fresh
return ("updated", f"#{rec['issue']}", rec.get("url"))
return (None, None, rec.get("url"))
def baseline(state):
findings, _ = scan()
for s, f in findings.items():
state.setdefault(s, {"issue": None, "url": None, "kind": f["kind"],
"reqs": [o["req"] for o in reqs_of(f)]})
state["__baselined__"] = True
save_state(state)
return len(findings)
def ensure_baseline(state, announce=None):
"""Never file the backlog: if we've never baselined, mark all current findings
seen (file nothing) before any watch/cycle can run."""
if state.get("__baselined__"):
return 0
n = baseline(state)
if announce:
announce("baselined", f"{n} existing blocks marked seen (not filed)", "")
return n
def _toast(title, body):
"""Best-effort desktop notification across Linux / macOS / Windows."""
try:
if shutil.which("notify-send"):
subprocess.run(["notify-send", "-a", "claudit", "-i", ICON, title, body], check=False)
elif platform.system() == "Darwin":
subprocess.run(["osascript", "-e",
f'display notification {json.dumps(body)} with title {json.dumps(title)}'],
check=False)
elif platform.system() == "Windows":
ps = (f"$ws=New-Object -ComObject WScript.Shell;"
f"[void]$ws.Popup({json.dumps(body)},5,{json.dumps(title)},64)")
subprocess.run(["powershell", "-NoProfile", "-Command", ps], check=False)
except Exception:
pass
def notify(action, ref, url):
"""Plain desktop toast (cross-platform)."""
_toast(f"claudit: {action}", f"{ref}\n{url or ''}".strip())
print(f"[{action}] {ref} {url or ''}", file=sys.stderr)
def notify_action(title, body, on_report):
"""Toast with a clickable 'Report it' button (Linux/notify-send only); runs on_report()
if clicked. notify-send -A blocks until the user acts, so call this in a thread. On other
platforms it degrades to a plain notification (use the GUI's menu to file)."""
if not shutil.which("notify-send"):
_toast(title, body)
return
res = subprocess.run(
# normal urgency + an explicit 20s expiry so the toast fades on its own if ignored
# (critical urgency made it persist forever); the GUI still queues them for later filing.
["notify-send", "-a", "claudit", "-i", ICON, "-u", "normal", "-t", "20000",
"-A", "report=Report it", "-A", "dismiss=Dismiss", title, body],
capture_output=True, text=True)
if res.stdout.strip() == "report":
on_report()
def announce_pending(state, repo, delay):
"""Pop an actionable toast for whatever is currently queued (in a background thread)."""
n = len(pending_sigs(state))
if not n:
return
threading.Thread(target=notify_action, daemon=True, args=(
f"ClAudit: {n} false-positive block(s) queued",
"Click ‘Report it’ to file these to GitHub now.",
lambda: notify("filed", f"{file_pending(state, repo, False, delay, notify)} reported", repo),
)).start()
def pending_sigs(state):
return list(state.get("__pending__", []))
def monitor_cycle(state, on_detect):
"""Notify-only pass: detect NEW findings (not seen, not already pending), queue them,
and toast. Files NOTHING. Returns the count of newly-detected findings."""
findings, _ = scan()
pend = state.setdefault("__pending__", [])
fresh = [f for s, f in findings.items()
if s not in state and s not in pend and not s.startswith("__") and should_file(f)]
if fresh:
for f in fresh:
pend.append(f["sig"])
save_state(state)
on_detect(fresh)
return len(fresh)
def passes_gate(f, state):
"""True if the finding should be filed. By default ClAudit files EVERY genuine block — the
correct-vs-false-positive call is exactly what it exists to surface, so it isn't pre-judged.
Only when GATE is explicitly opted in does the LLM skip blocks it judges were correct."""
if not GATE:
return True
ctx = " ".join(t for _, t in (f.get("leadup") or []))
ok, reason = claudit.llm_is_false_positive(f["kind"], f.get("block_text", ""), ctx)
if not ok:
state[f["sig"]] = {"issue": None, "skipped": (reason or "judged a correct block")[:140],
"kind": f["kind"], "reqs": [o["req"] for o in reqs_of(f)]}
save_state(state)
print(f" skip {f['sig']} — not a false positive: {reason[:80]}", file=sys.stderr)
return ok
def auto_cycle(state, repo, delay, on_event):
"""Auto-post pass: file every NEW finding, comment recurrences. Files nothing for
baselined/already-filed findings. Returns count of actions taken."""
findings, _ = scan(ttl=8)
acted = 0
for f in findings.values():
if f["sig"] not in state and (not should_file(f) or not passes_gate(f, state)):
continue # only file NEW cyber/aup blocks that carry a Request ID
try:
action, ref, url = file_one(f, "", repo, state)
except Exception as e:
print(f" ! {f['sig']}: {e}", file=sys.stderr)
continue
if action:
save_state(state)
on_event(action, ref, url)
acted += 1
time.sleep(delay)
return acted
def _chain_key(f):
"""Which chain a finding belongs to. Keyed on the actual Claude Code SESSION (one conversation /
one .jsonl) — far tighter than the whole project, so a 20-block session is one chain instead of
every block in the repo over weeks getting lumped together. Falls back to project, then unknown."""
sessions = [o.get("session") for o in f["occ"] if o.get("session")]
if sessions:
return "sess:" + collections.Counter(sessions).most_common(1)[0][0]
projs = [o.get("proj") for o in f["occ"] if o.get("proj")]
return ("proj:" + collections.Counter(projs).most_common(1)[0][0]) if projs else "unknown"
_proj_of = _chain_key # back-compat alias
def _single_req_finding(f, req):
"""A view of finding f narrowed to one Request ID — so each API ID files as its own bespoke
issue, not folded into a shared one."""
g = dict(f)
g["occ"] = [o for o in f["occ"] if o.get("req") == req] or f["occ"][:1]
return g
def _dwell_crossref(chain):
"""Forward cross-reference body line: the prior bespoke issues from the same work session."""
if not chain:
return ""
return ("Distinct false-positive blocks from the same work session, each its own report:\n"
+ ", ".join(f"#{n}" for n in chain[-20:]))
def dwell_cycle(state, repo, delay, on_event, dwell=None):
"""Auto-file on a dwell. Each NEW cyber/aup Request ID is held for `dwell` seconds (repeats in
that window accrue as their OWN linked issues); once ripe, the LLM gate judges it a real false
positive and burn-tokens composes it, then it files as a bespoke issue cross-linked to its
siblings from the same work session. One issue per Request ID; no manual push. Returns count filed.
The pre-existing backlog is baselined (not flooded) on first run — only blocks seen after that
auto-file."""
dwell = DWELL_SECONDS if dwell is None else dwell
findings, _ = scan(ttl=8)
now = time.time()
seen = state.setdefault("__dwell_seen__", [])
hold = state.setdefault("__dwell_hold__", {}) # req -> first_seen_epoch
filed = state.setdefault("__filed_reqs__", {}) # req -> issue number
skipped = state.setdefault("__skipped_reqs__", [])
chains = state.setdefault("__proj_chain__", {}) # proj -> [issue numbers], the linked string
# req -> finding for every fileable (cyber/aup + Request ID) occurrence currently in the corpus
current = {}
for f in findings.values():
if not should_file(f):
continue
for o in f["occ"]:
if o.get("req"):
current[o["req"]] = f
seen_set, filed_set, skip_set = set(seen), set(filed), set(skipped)
# First run on this state: baseline everything already present so we don't flood the backlog.
if not state.get("__dwell_baselined__"):
seen[:] = sorted(set(seen) | set(current) | filed_set)
state["__dwell_baselined__"] = True
save_state(state)
return 0
# Drain the deprecated manual queue (__pending__) into the dwell pipeline so it doesn't hang as
# un-processable QUEUED rows — each pending block's Request IDs get held + filed like any other.
pend = state.get("__pending__", [])
if pend:
for s in pend:
f = findings.get(s)
if not (f and should_file(f)):
continue
for o in f["occ"]:
req = o.get("req")
if req and req not in filed_set and req not in skip_set:
hold.setdefault(req, now)
seen_set.add(req)
if req not in seen:
seen.append(req)
state["__pending__"] = []
save_state(state)
# Register genuinely new Request IDs into the dwell hold.
for req in current:
if req not in seen_set and req not in filed_set and req not in skip_set:
hold.setdefault(req, now)
seen.append(req)
# A create that keeps failing must NOT re-judge+re-compose every 30s tick (that's the only path
# that could burn tokens "like crazy" at idle) — back a failed Request ID off for FAIL_COOLDOWN.
fails = state.setdefault("__dwell_fail__", {}) # req -> last failed-create epoch
FAIL_COOLDOWN = 1800 # 30 min between retries of a stuck block
ripe = [r for r, t0 in hold.items()
if now - t0 >= dwell and r in current
and now - fails.get(r, 0) >= FAIL_COOLDOWN][:8] # cap bursts
acted, gate_cache = 0, {}
for req in ripe:
f = current[req]
sig = f["sig"]
if sig not in gate_cache: # judge once per block, reuse for its other Request IDs
ctx = " ".join(t for _, t in (f.get("leadup") or []))
ok, reason = claudit.llm_is_false_positive(f["kind"], f.get("block_text", ""), ctx)
gate_cache[sig] = (ok, reason)
ok, reason = gate_cache[sig]
if not ok: # LLM judged it a correct block -> never file it
skipped.append(req)
hold.pop(req, None)
print(f" dwell skip {req} — not a false positive: {(reason or '')[:70]}", file=sys.stderr)
continue
proj = _proj_of(f)
chain = chains.setdefault(proj, [])
g = _single_req_finding(f, req)
title, body = build_issue(g, "", _dwell_crossref(chain))
try:
url = gh_create(repo, title, body)
except Exception as e:
fails[req] = now # back off 30 min before re-judging/re-composing this one
print(f" ! dwell file {req}: {e}", file=sys.stderr)
continue # leave in hold; retry after the cooldown
num = url.rsplit("/", 1)[-1]
if chain: # back-link the previous sibling -> this new report
try:
gh_comment(repo, chain[-1],
f"🔗 Related false positive from the same work session: #{num}")
except Exception:
pass
chain.append(num)
filed[req] = num
hold.pop(req, None)
fails.pop(req, None) # cleared on success
log_issue(g, repo, url)
save_state(state)
on_event("dwell-filed", title, url)
acted += 1
time.sleep(delay)
if ripe:
save_state(state)
return acted
def dwell_status(state, dwell=None):
"""(#holding, seconds until the next one ripens) for the UI."""
dwell = DWELL_SECONDS if dwell is None else dwell
hold = state.get("__dwell_hold__", {})
if not hold:
return 0, 0
now = time.time()
nxt = min((dwell - (now - t0)) for t0 in hold.values())
return len(hold), max(0, int(nxt))
def backfill_one(f, repo, state):
"""File ONE baselined-but-unfiled finding (a backlog item). Returns event tuple or None."""
rec = state.get(f["sig"])
if rec is None or rec.get("issue"):
return None # not a backlog item (new, or already filed)
title, body = build_issue(f, "")
url = gh_create(repo, title, body)
rec.update(issue=url.rsplit("/", 1)[-1], url=url)
save_state(state)
log_issue(f, repo, url)
return ("backfilled", title, url)
def backlog_size(state):
return sum(1 for s, r in state.items()
if not s.startswith("__") and isinstance(r, dict)
and r.get("issue") is None and not r.get("skipped"))
def prune_stale_backlog(state):
"""Clear backlog items that can NEVER be filed: those whose finding no longer exists in any
current session scan (the session rotated / was deleted). They'd otherwise sit in the backlog
count forever ('6 to backfill but won't do it'). Marks them done-skipped. Returns count pruned."""
try:
findings, _ = scan(ttl=8)
except Exception:
return 0
live = set(findings)
pruned = 0
for sig, rec in list(state.items()):
if sig.startswith("__") or not isinstance(rec, dict):
continue
if rec.get("issue") is None and not rec.get("skipped") and sig not in live:
rec["skipped"] = "stale — session no longer present; cannot backfill"
pruned += 1
if pruned:
save_state(state)
print(f"prune_stale_backlog: cleared {pruned} unfilable backlog item(s).", file=sys.stderr)
return pruned
def prune_stale_pending(state):
"""Drop queued (__pending__) sigs that have no current finding: the block aged out of the