-
Notifications
You must be signed in to change notification settings - Fork 11
Expand file tree
/
Copy pathcheck-global-state.py
More file actions
3115 lines (2803 loc) · 124 KB
/
Copy pathcheck-global-state.py
File metadata and controls
3115 lines (2803 loc) · 124 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""Audit the plugin/dyncode/linker trees for plugin-related process-global
mutable compilation state.
This is the first-release hard gate for the design requirement that "frontend,
LTO, linker and dyncode carry no plugin-related process-global mutable
compilation state".
Four complementary layers of checking:
1. FORBIDDEN symbols -- a precise, zero-tolerance list of known process-global
escape hatches (removed prototype loader, dyncode current-options/mode
singletons, the three NeverC ``ListRegister*Callbacks`` vectors, linker
``parallel::strategy``/``getThreadIndex`` dependence and the old
``CommonLinkerContext::destroy``/``lctx`` singleton). Any hit in runtime
code (comments and string literals are stripped first) is a hard failure.
2. Exact TLS declaration scan -- reports ``thread_local`` storage declared in
the plugin/dyncode/linker trees. Entries documented in
``global-state-allowlist.json`` (with owner, lifetime and justification)
are accepted; anything else is reported. This is the only layer that judges
thread-local storage, because it can pin a declaration to a file, an owner
and a clearing test -- none of which a symbol name carries. It fails the
build under ``--strict``, which the workflows pass.
3. Source provenance scans -- hard failures for mutable state defined with
internal linkage in a header, and for unqualified mutable storage in audited
implementation files whose final symbol lacks path or namespace provenance.
Exact reviewed manifests cover immutable lookup data and address sentinels;
every manifest entry must match exactly one declaration. The implementation
scan is deliberately lexical: macro-generated declarations remain a known
limitation until the gate uses compile_commands and Clang AST locations.
Header internal-linkage state is especially dangerous. An anonymous
namespace or namespace-scope ``static`` object exists once per *including
translation unit*, so a flag set through one includer reads unset through
the next. This includes ``inline static`` where ``static`` silently wins.
This tree is a header-only port of LLVM, so state upstream keeps in a .cpp
-- where an anonymous namespace is exactly right -- now sits in headers the
plugin path includes, where it means the opposite. The binary layer below
cannot recover the source path for a bare ``(anonymous namespace)::X``.
Constant-initialized data is exempt only through the conservative scalar
rule or an exact reviewed manifest.
4. Optionally, ``--build-dir`` scans writable data symbols of the linked
compiler (``<build-dir>/bin/neverc``). ELF/Mach-O use ``nm``/``llvm-nm``;
PE uses ``llvm-readobj`` section/symbol facts plus ``llvm-undname`` owner
demangling, with the Windows system DbgHelp undecorator as a strict fallback.
This is the artifact-level backstop for owner-qualified NeverC and
linker symbols, and it is a hard failure. A PE must embed a complete COFF static
symbol table: an ordinary MSVC image carrying only PDB/CodeView data is
reported as unscannable, never clean. A missing compiler or required tool is
likewise reported because ``--build-dir`` is a request for this scan, and an
unavailable input that prints "skipping" is indistinguishable from a clean run.
An empty or stripped static symbol table is also unscannable: the scan requires
both an external artifact sentinel and a local nlist sentinel before it trusts
the table as evidence. Thread-local symbols are excluded here and left to layer
2: TLS is per-thread rather than process-global, and its symbol layout differs
enough between platforms that judging it here made the gate reach opposite
verdicts on ELF and Mach-O for the same source.
Without ``--build-dir`` the checker is intentionally a conservative source-
only preflight: it retains every lexical finding that only a linked symbol
table can discharge, so ``--strict`` is not required to be clean in that mode.
Release qualification always supplies ``--build-dir`` and delegates source
findings only after the artifact table and both completeness sentinels were
successfully audited.
"""
from __future__ import annotations
import argparse
import ctypes
import hashlib
import json
import pathlib
import re
import shutil
import subprocess
import sys
from dataclasses import dataclass, field
from enum import Enum
from typing import Iterable
ROOT = pathlib.Path(__file__).resolve().parents[2]
ALLOWLIST_PATH = pathlib.Path(__file__).resolve().parent / "global-state-allowlist.json"
# Directories whose *runtime* code must be free of plugin-related global state.
AUDIT_DIRS = (
"neverc/lib/Plugin",
"neverc/lib/DynCode",
"neverc/lib/Linker",
"neverc/include/neverc/Plugin",
"neverc/include/neverc/DynCode",
"neverc/include/neverc/Linker",
)
# BackendUtil bridges the compiler into the plugin/dyncode pipeline; it must not
# reach global dyncode/plugin state either.
EXTRA_AUDIT_FILES = (
"neverc/lib/Emit/Backend/BackendUtil.cpp",
"neverc/include/neverc/Emit/Backend/BackendUtil.h",
"neverc/include/neverc/Foundation/Core/LLVMTimeTraceRootLease.h",
"neverc/lib/Foundation/Core/LLVMTimeTraceRootLease.cpp",
"neverc/lib/Foundation/Core/ProcessResourceBroker.cpp",
"neverc/lib/Merge/Common/MergerCommon.h",
)
# Zero-tolerance symbols: any hit in runtime code fails the gate.
FORBIDDEN_SYMBOLS = (
# Prototype loader / global plugin state, since removed.
"getGlobalPluginLoader",
"pluginArgStorage",
# dyncode current-options / mode singletons, since removed.
"getCurrentDynCodeOptions",
"currentDynCodeOptionsStorage",
"setDynCodeModeState",
"gDynCodeModeEnabled",
"dyncodeModeStorage",
"machinePassCallbackInstalled",
"passBuilderCallbackInstalled",
# NeverC-added process-global LLVM callback vectors, since removed.
"ListRegisterPassBuilderCallbacks",
"ListRegisterTargetPassConfigCallbacks",
"ListRegisterTargetPassConfigPostPreEmitCallbacks",
# Linker per-process parallel strategy dependence, since removed.
"parallel::strategy",
"parallel::getThreadIndex",
# Old linker context singleton, since removed.
"CommonLinkerContext::destroy",
)
def forbidden_symbol_pattern(symbol: str) -> re.Pattern:
"""Match a qualified C++ name even when ``::`` is whitespace-split."""
body = r"\s*::\s*".join(
re.escape(component) for component in symbol.split("::"))
return re.compile(
rf"(?<![A-Za-z0-9_]){body}(?![A-Za-z0-9_])")
FORBIDDEN_SYMBOL_PATTERNS = tuple(
(forbidden_symbol_pattern(symbol), symbol)
for symbol in FORBIDDEN_SYMBOLS
)
# Forbidden regexes (need word boundaries / composite spellings).
FORBIDDEN_REGEXES = (
(re.compile(r"static\s+CommonLinkerContext\s*\*\s*lctx\b"),
"static CommonLinkerContext *lctx"),
(re.compile(r"\bInputFile\s*::\s*isInGroup\b"),
"InputFile::isInGroup"),
(re.compile(r"\bSharedFile\s*::\s*vernauxNum\b"),
"SharedFile::vernauxNum"),
(re.compile(r"\bInputFile\s*::\s*idCount\b"), "InputFile::idCount"),
(re.compile(r"\bLCDylib\s*::\s*instanceCount\b"),
"LCDylib::instanceCount"),
)
# All supported TLS storage spellings share this matcher. Detection and exact
# allowlisting must agree: a spelling that can be reported but never named in
# the allowlist makes the ownership/lifetime contract impossible to satisfy.
_TLS_STORAGE = re.compile(
r"\b(?:thread_local|_Thread_local|__thread|LLVM_THREAD_LOCAL)\b|"
r"__declspec\s*\(\s*thread\s*\)")
# Heuristic declaration patterns for the advisory layer. Keep these mutually
# exclusive: a line matching two patterns would be reported twice.
DECL_PATTERNS = (
(_TLS_STORAGE, "thread-local"),
)
# Headers scanned for mutable state that has *internal linkage*, which gives
# every including translation unit its own copy. ``llvm/include`` is in scope
# because this tree is a header-only port of LLVM: state that upstream keeps in
# a .cpp now sits in headers the plugin path includes, so a split there is
# plugin-visible process state. See ``scan_headers``.
HEADER_AUDIT_DIRS = (
"llvm/include",
"neverc/include/neverc/Foundation/Core/LLVMTimeTraceRootLease.h",
"neverc/include/neverc/Plugin",
"neverc/include/neverc/DynCode",
"neverc/include/neverc/Linker",
)
@dataclass
class Report:
forbidden: list[str] = field(default_factory=list)
heuristic: list[str] = field(default_factory=list)
headers: list[str] = field(default_factory=list)
source_globals: list[str] = field(default_factory=list)
binary: list[str] = field(default_factory=list)
# Reasons a configured source/header/binary scan could not run. Kept apart
# from findings so "the input was unavailable" never reads as "clean".
unscannable: list[str] = field(default_factory=list)
def failed(self, strict: bool) -> bool:
if (self.forbidden or self.headers or self.source_globals or
self.binary or self.unscannable):
return True
if strict and self.heuristic:
return True
return False
def is_digit_separator(text: str, i: int) -> bool:
"""Report whether ``text[i]`` is a C++14 digit separator, not a quote.
``0x3b00'0000`` is one number, but reading that apostrophe as the start of a
character literal makes the scanner run to the *next* quote and drop
everything in between -- which can hide a forbidden symbol from the gate
entirely. A separator always sits between digits of a literal, whereas an
encoding prefix (``u8'x'``, ``L'x'``) leaves a non-digit at the head of the
token.
"""
if i == 0 or i + 1 >= len(text):
return False
if not text[i + 1].isalnum():
return False
j = i - 1
while j >= 0 and (text[j].isalnum() or text[j] in "'."):
j -= 1
head = text[j + 1:i]
return bool(head) and head[0].isdigit()
# A raw string opener, optionally encoding-prefixed: R"delim( ... )delim". The
# delimiter excludes whitespace, parentheses and backslash, per [lex.string].
_RAW_STRING_OPEN = re.compile(r'(?:u8|u|U|L)?R"([^ ()\\\t\v\f\n]{0,16})\(')
def raw_string_end(text: str, i: int) -> int | None:
"""Return the index just past the raw string starting at *i*, else None.
Raw strings ignore escapes and may embed quotes, so the ordinary literal
scanner would end one early and then read its contents as code.
"""
match = _RAW_STRING_OPEN.match(text, i)
if not match:
return None
closer = ")" + match.group(1) + '"'
end = text.find(closer, match.end())
return len(text) if end == -1 else end + len(closer)
def blank_preserving_newlines(fragment: str) -> str:
"""Replace tokens with whitespace without changing source offsets."""
return "".join("\n" if char == "\n" else " " for char in fragment)
def strip_comments_and_strings(text: str) -> str:
"""Remove // and /* */ comments and the contents of string/char literals so
that a token only counts when it is real code."""
out: list[str] = []
i = 0
n = len(text)
while i < n:
c = text[i]
two = text[i : i + 2]
if two == "//":
j = text.find("\n", i)
if j == -1:
out.append(" " * (n - i))
break
out.append(" " * (j - i))
i = j
continue
if two == "/*":
start = i
j = text.find("*/", i + 2)
end = n if j == -1 else j + 2
# A comment is whitespace in C++. Keep both token separation and
# every newline so matching and diagnostic locations agree with
# the compiler's token stream.
out.append(blank_preserving_newlines(text[start:end]))
i = end
continue
# A raw string may open on its encoding prefix, so only consider one
# when the preceding character cannot be part of an identifier.
if c in "RLuU" and (i == 0 or not (text[i - 1].isalnum()
or text[i - 1] == "_")):
end = raw_string_end(text, i)
if end is not None:
out.append(blank_preserving_newlines(text[i:end]))
i = end
continue
if c == "'" and is_digit_separator(text, i):
out.append(c)
i += 1
continue
if c in ('"', "'"):
quote = c
start = i
i += 1
while i < n:
if text[i] == "\\":
i += 2
continue
if text[i] == quote:
i += 1
break
i += 1
masked = list(blank_preserving_newlines(text[start:i]))
masked[0] = quote
if i > start + 1 and text[i - 1] == quote:
masked[-1] = quote
out.append("".join(masked))
continue
out.append(c)
i += 1
return "".join(out)
def strip_comments_preserve_literals(text: str) -> str:
"""Remove comments while retaining complete literal token spellings."""
out: list[str] = []
i = 0
n = len(text)
while i < n:
two = text[i:i + 2]
if two == "//":
end = text.find("\n", i)
if end == -1:
out.append(" " * (n - i))
break
out.append(" " * (end - i))
i = end
continue
if two == "/*":
end_marker = text.find("*/", i + 2)
end = n if end_marker == -1 else end_marker + 2
out.append(blank_preserving_newlines(text[i:end]))
i = end
continue
c = text[i]
if c in "RLuU" and (i == 0 or not (text[i - 1].isalnum()
or text[i - 1] == "_")):
end = raw_string_end(text, i)
if end is not None:
out.append(text[i:end])
i = end
continue
if c == "'" and is_digit_separator(text, i):
out.append(c)
i += 1
continue
if c in ('"', "'"):
quote = c
start = i
i += 1
while i < n:
if text[i] == "\\":
i = min(i + 2, n)
continue
i += 1
if text[i - 1] == quote:
break
out.append(text[start:i])
continue
out.append(c)
i += 1
return "".join(out)
def normalize_cpp_whitespace_preserve_literals(text: str) -> str:
"""Fold trivia whitespace while preserving every literal byte."""
out: list[str] = []
pending_space = False
i = 0
n = len(text)
while i < n:
if text[i].isspace():
pending_space = bool(out)
i += 1
continue
if pending_space:
out.append(" ")
pending_space = False
c = text[i]
if c in "RLuU" and (i == 0 or not (text[i - 1].isalnum()
or text[i - 1] == "_")):
end = raw_string_end(text, i)
if end is not None:
out.append(text[i:end])
i = end
continue
if c == "'" and is_digit_separator(text, i):
out.append(c)
i += 1
continue
if c in ('"', "'"):
quote = c
start = i
i += 1
while i < n:
if text[i] == "\\":
i = min(i + 2, n)
continue
i += 1
if text[i - 1] == quote:
break
out.append(text[start:i])
continue
out.append(c)
i += 1
return "".join(out).rstrip()
def iter_source_files(paths: Iterable[pathlib.Path]):
for base in paths:
if base.is_file():
yield base
continue
if not base.exists():
continue
for suffix in ("*.h", "*.hpp", "*.c", "*.cc", "*.cpp", "*.inc"):
yield from base.rglob(suffix)
def splice_line_continuations(text: str) -> str:
"""Apply C/C++ phase-2 line splicing before lexical auditing.
Without this, ``thread_\\\nlocal`` and even a forbidden identifier can be
split in the physical file while the compiler sees one token. Diagnostics
after a splice use logical line numbers; the gate favours seeing the token
over preserving a cosmetically exact physical line in this rare form.
"""
return re.sub(r"\\(?:\r\n|\n|\r)", "", text)
def report_missing_audit_paths(report: Report,
paths: Iterable[pathlib.Path],
layer: str) -> None:
"""Make a stale configured audit root a gate failure, not an empty scan."""
for path in paths:
if path.exists():
continue
try:
display = path.relative_to(ROOT).as_posix()
except ValueError:
display = str(path)
finding = f"configured {layer} audit path is missing: {display}"
if finding not in report.unscannable:
report.unscannable.append(finding)
def load_allowlist() -> dict:
if not ALLOWLIST_PATH.exists():
return {"entries": []}
return json.loads(ALLOWLIST_PATH.read_text(encoding="utf-8"))
def relative_key(path: pathlib.Path) -> str:
"""Path relative to ROOT, always '/'-separated.
The allowlist spells its paths with forward slashes, so a Windows run that
compared against ``neverc\\lib\\...`` matched nothing and reported every
documented thread_local as a violation -- a gate that passed on Linux and
macOS and failed on Windows for identical sources.
"""
return path.relative_to(ROOT).as_posix()
def single_declared_identifier(snippet: str) -> str | None:
"""Return the sole simple TLS declarator, or fail closed.
An allowlisted name must be the object being declared, not a token in its
type or initializer. This deliberately small declaration scanner accepts
the forms used by the audited facades, including template types and
call/braced initializers, while rejecting multiple declarators and syntax
it cannot classify safely. It is not intended to parse arbitrary C++.
"""
storage = _TLS_STORAGE.search(snippet)
if not storage:
return None
text = snippet[storage.end():]
identifiers: list[str] = []
paren_depth = 0
square_depth = 0
brace_depth = 0
angle_depth = 0
in_initializer = False
terminated = False
i = 0
def at_declaration_level() -> bool:
return paren_depth == square_depth == brace_depth == 0
while i < len(text):
c = text[i]
if terminated:
if not c.isspace():
return None
i += 1
continue
if c.isalpha() or c == "_":
j = i + 1
while j < len(text) and (text[j].isalnum() or text[j] == "_"):
j += 1
if (not in_initializer and at_declaration_level() and
angle_depth == 0):
identifiers.append(text[i:j])
i = j
continue
if c == "(":
# With at least a type and a candidate object name, a top-level
# parenthesis starts direct initialization. Parentheses earlier
# in the decl-specifier sequence (for example decltype/alignas)
# remain part of that sequence.
if (not in_initializer and at_declaration_level() and
angle_depth == 0 and len(identifiers) >= 2):
in_initializer = True
paren_depth += 1
elif c == ")":
if paren_depth == 0:
return None
paren_depth -= 1
elif c == "[":
square_depth += 1
elif c == "]":
if square_depth == 0:
return None
square_depth -= 1
elif c == "{":
if (not in_initializer and at_declaration_level() and
angle_depth == 0):
in_initializer = True
brace_depth += 1
elif c == "}":
if brace_depth == 0:
return None
brace_depth -= 1
elif c == "<" and not in_initializer and paren_depth == 0:
angle_depth += 1
elif c == ">" and not in_initializer and paren_depth == 0:
if angle_depth == 0:
return None
angle_depth -= 1
elif c == "=" and not in_initializer and at_declaration_level():
if angle_depth == 0:
in_initializer = True
elif c == "," and at_declaration_level():
# Once an initializer starts, angle brackets may be comparison
# operators rather than templates. Treat every ungrouped comma
# there as another declarator; false positives fail the gate.
if in_initializer or angle_depth == 0:
return None
elif c == ";" and at_declaration_level() and angle_depth == 0:
terminated = True
i += 1
if (not terminated or paren_depth != 0 or square_depth != 0 or
brace_depth != 0 or angle_depth != 0 or not identifiers):
return None
return identifiers[-1]
def allowlisted(rel: str, snippet: str, allowlist: dict) -> bool:
rel = rel.replace("\\", "/")
declaration = canonical_object_declaration(snippet)
if declaration is None:
return False
for entry in allowlist.get("entries", []):
files = [path.replace("\\", "/")
for path in entry.get("files", [])]
if rel not in files:
continue
declared = single_declared_identifier(snippet)
if declared is None:
return False
symbol = entry.get("symbol")
if (symbol and declared == symbol and
entry.get("declaration") == declaration and
entry.get("definition_sha256") ==
object_definition_sha256(snippet)):
return True
return False
def canonical_object_declaration(statement: str) -> str | None:
"""Return the normalized declarator used by exact review manifests."""
text = " ".join(statement.split()).rstrip(";").rstrip()
# scope_scan surfaces a braced definition at its opening brace; direct
# callers may provide the full ``Name{...}`` spelling. In both forms the
# part before the top-level brace is the object declarator.
paren = square = 0
for i, char in enumerate(text):
if char == "(":
paren += 1
elif char == ")":
paren = max(paren - 1, 0)
elif char == "[":
square += 1
elif char == "]":
square = max(square - 1, 0)
elif char == "{" and paren == square == 0:
text = text[:i].rstrip()
break
declarator = object_declarator(text)
return " ".join(declarator.split()) if declarator is not None else None
def canonical_object_definition(statement: str) -> str:
"""Return a stable structural spelling of a complete object definition.
Review manifests bind both the declarator and the complete initializer,
including string/character/raw-literal contents. Comments are not semantic
and are erased before whitespace normalization.
"""
code = strip_comments_preserve_literals(splice_line_continuations(
statement))
return normalize_cpp_whitespace_preserve_literals(code).rstrip(
";").rstrip()
def object_definition_sha256(statement: str) -> str:
canonical = canonical_object_definition(statement)
return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
def declared_object_identifier(statement: str) -> str | None:
"""Return the simple identifier of an object declaration we understand."""
declarator = canonical_object_declaration(statement)
if declarator is None:
return None
tail = _ENDS_IN_IDENTIFIER.search(declarator)
return tail.group("name") if tail else None
def header_constant_allowlisted(rel: str, statement: str,
allowlist: dict) -> bool:
"""Match one complex constexpr header object by exact path and identifier.
A complex ``constexpr`` object is not automatically immutable all the way
down: records may contain ``mutable`` subobjects, and pointers expose a
separate object. Each reviewed exception therefore needs an exact manifest
entry with an owner and a concrete justification.
"""
rel = rel.replace("\\", "/")
declaration = canonical_object_declaration(statement)
symbol = declared_object_identifier(statement)
if symbol is None or declaration is None:
return False
for entry in allowlist.get("header_constants", []):
if entry.get("file", "").replace("\\", "/") != rel:
continue
if entry.get("symbol") != symbol:
continue
if entry.get("declaration") != declaration:
continue
if entry.get("definition_sha256") != object_definition_sha256(
statement):
continue
if not all(isinstance(entry.get(field), str) and
bool(entry[field].strip())
for field in ("declaration", "definition_sha256", "owner",
"justification")):
return False
return True
return False
def source_object_allowlisted(rel: str, statement: str,
allowlist: dict) -> bool:
"""Match one path-provenanced source object by exact identifier."""
rel = rel.replace("\\", "/")
declaration = canonical_object_declaration(statement)
symbol = declared_object_identifier(statement)
if symbol is None or declaration is None:
return False
for entry in allowlist.get("source_objects", []):
if entry.get("file", "").replace("\\", "/") != rel:
continue
if entry.get("symbol") != symbol:
continue
if entry.get("declaration") != declaration:
continue
if entry.get("definition_sha256") != object_definition_sha256(
statement):
continue
if not all(isinstance(entry.get(field), str) and
bool(entry[field].strip())
for field in ("declaration", "definition_sha256", "owner", "kind",
"justification")):
return False
return True
return False
def binary_source_object_allowlisted(rel: str, statement: str,
allowlist: dict) -> bool:
"""Match the source definition promised by one binary exception.
A demangled symbol name is not sufficient provenance for a security
exception: an unrelated writable object can be renamed to the marker, and
an initializer can acquire state without changing that marker. Bind every
binary entry to exactly one complete source definition as well as requiring
exactly one linked occurrence.
"""
rel = rel.replace("\\", "/")
declaration = canonical_object_declaration(statement)
symbol = declared_object_identifier(statement)
if symbol is None or declaration is None:
return False
for entry in allowlist.get("binary_symbols", []):
if entry.get("file", "").replace("\\", "/") != rel:
continue
if entry.get("source_symbol") != symbol:
continue
if entry.get("declaration") != declaration:
continue
if entry.get("definition_sha256") != object_definition_sha256(
statement):
continue
return True
return False
def is_canonical_promotion_suffix(suffix: str) -> bool:
if not re.fullmatch(r"(?:\.llvm\.[0-9]+)+", suffix):
return False
for digits in re.findall(r"\.llvm\.([0-9]+)", suffix):
if ((len(digits) > 1 and digits.startswith("0")) or
int(digits) > 0xffffffffffffffff):
return False
return True
def matches_llvm_promoted_symbol(name: str, marker: str) -> bool:
if name == marker:
return True
if not name.startswith(marker):
return False
suffix = name[len(marker):]
if suffix.startswith(" (") and suffix.endswith(")"):
suffix = suffix[2:-1]
elif suffix.startswith(" [clone ") and suffix.endswith("]"):
suffix = suffix[8:-1]
return is_canonical_promotion_suffix(suffix)
def binary_allowlist_entry(name: str, allowlist: dict) -> dict | None:
"""Return the exact binary-symbol manifest entry matching *name*."""
for entry in allowlist.get("binary_symbols", []):
marker = entry.get("symbol")
if marker and matches_llvm_promoted_symbol(name, marker):
return entry
return None
def allowlisted_binary_symbol(name: str, allowlist: dict) -> bool:
"""Match a writable symbol against the ``binary_symbols`` list.
Only address-only pass/type identity tokens, const objects whose vtable
pointer forces a relocation, and sync primitives without compilation state
are eligible; each entry carries an owner and justification. A lookup or
interface table is not: making it ``constexpr`` moves it to read-only
storage, which is a fix rather than a documented exception. Thread-local
symbols never reach here -- see ``symbol_is_thread_local``.
"""
return binary_allowlist_entry(name, allowlist) is not None
def source_audit_bases() -> list[pathlib.Path]:
return ([ROOT / directory for directory in AUDIT_DIRS] +
[ROOT / filename for filename in EXTRA_AUDIT_FILES])
def audited_source_files() -> dict[str, pathlib.Path]:
"""Return the exact source dependency closure covered by this gate."""
return {relative_key(path): path
for path in iter_source_files(source_audit_bases())}
def iter_tls_declarations(code: str, semantic_code: str | None = None):
"""Yield ``(match, kind, declaration)`` for supported TLS spellings."""
if semantic_code is None:
semantic_code = code
if len(semantic_code) != len(code):
raise ValueError("TLS structural/semantic source views are misaligned")
for pattern, name in DECL_PATTERNS:
for match in pattern.finditer(code):
start = max(code.rfind(delimiter, 0, match.start())
for delimiter in ";{}") + 1
end = code.find(";", match.end())
snippet = semantic_code[
start:end + 1 if end >= 0 else len(code)].strip()
yield match, name, snippet
def scan_sources(report: Report, allowlist: dict) -> None:
bases = source_audit_bases()
report_missing_audit_paths(report, bases, "source")
for path in iter_source_files(bases):
rel = relative_key(path)
raw = path.read_text(encoding="utf-8", errors="replace")
spliced = splice_line_continuations(raw)
code = strip_comments_and_strings(spliced)
semantic_code = strip_comments_preserve_literals(spliced)
# Zero-tolerance checks operate on the complete stripped translation
# unit. C++ permits whitespace (including newlines) around ``::`` and
# throughout declarations, so a physical-line scan is not a hard gate.
for pattern, token in FORBIDDEN_SYMBOL_PATTERNS:
for match in pattern.finditer(code):
line = code.count("\n", 0, match.start()) + 1
report.forbidden.append(
f"{rel}:{line}: forbidden `{token}`")
for pattern, name in FORBIDDEN_REGEXES:
for match in pattern.finditer(code):
line = code.count("\n", 0, match.start()) + 1
report.forbidden.append(
f"{rel}:{line}: forbidden `{name}`")
# Match the complete translation unit. ``__declspec(thread)`` permits
# newlines inside its parentheses, and line-by-line matching silently
# missed that legal spelling. Extract the containing declaration for
# exact allowlist parsing; an unfamiliar declaration fails closed.
for match, name, snippet in iter_tls_declarations(
code, semantic_code):
if allowlisted(rel, snippet, allowlist):
continue
line = code.count("\n", 0, match.start()) + 1
report.heuristic.append(
f"{rel}:{line}: {name} storage `{snippet[:80]}`")
_OPEN_NAMESPACE = re.compile(r"\bnamespace\b(?P<name>[\w\s:]*)$")
_OPEN_LINKAGE = re.compile(
r'^\s*extern\s+(?:"\s*"|"C"|"C\+\+")\s*$')
_OPEN_RECORD = re.compile(r"\b(?:struct|class|union|enum)\b[^;=]*$")
_HAS_STATIC = re.compile(r"(?:^|\s)static(?:\s|$)")
_ENDS_IN_IDENTIFIER = re.compile(
r"(?P<name>[A-Za-z_]\w*)\s*(?:\[[^\]]*\]\s*)*$")
_NOT_AN_OBJECT_STATEMENT = re.compile(
r"^\s*(?:using|typedef|friend|return|static_assert|namespace)\b")
# Types defined in an anonymous namespace, which is what makes a variable of
# that type internally linked no matter how the variable itself is spelled.
_ANON_TYPE = re.compile(r"\b(?:struct|class|union|enum)\s+(\w+)\b")
# A pointer-to-function object hides its identifier inside a top-level
# parenthesised declarator: ``void (CALL *Hook[2])(int)``. Keep the grammar
# deliberately narrow so a function parameter or ``(*factory())(int)`` is not
# promoted into an object finding.
_FUNCTION_POINTER_GROUP = re.compile(
r"\s*(?:[A-Za-z_]\w*\s+)*"
r"(?:(?:[A-Za-z_]\w*)\s*::\s*)*\*\s*"
r"(?:(?:const|volatile|restrict)\s+)*"
r"(?P<name>[A-Za-z_]\w*)\s*"
r"(?:\[[^\]]*\]\s*)*")
_READ_ONLY_BUILTINS = {
"bool", "char", "signed char", "unsigned char", "wchar_t", "char8_t",
"char16_t", "char32_t", "short", "short int", "signed short",
"signed short int", "unsigned short", "unsigned short int", "int",
"signed", "signed int", "unsigned", "unsigned int", "long",
"long int", "signed long", "signed long int", "unsigned long",
"unsigned long int", "long long", "long long int", "signed long long",
"signed long long int", "unsigned long long", "unsigned long long int",
"float", "double", "long double", "std::byte",
}
_READ_ONLY_FIXED_WIDTH = re.compile(
r"(?:std\s*::\s*)?(?:u?int(?:8|16|32|64)_t|u?intptr_t|size_t|ptrdiff_t)")
def top_level_initializer_equal(text: str) -> int | None:
"""Return the top-level copy-initialiser ``=``, if one exists."""
paren = square = brace = 0
for i, char in enumerate(text):
if char == "(":
paren += 1
elif char == ")":
paren = max(paren - 1, 0)
elif char == "[":
square += 1
elif char == "]":
square = max(square - 1, 0)
elif char == "{":
brace += 1
elif char == "}":
brace = max(brace - 1, 0)
elif char == "=" and paren == square == brace == 0:
# ``operator=(...)`` names a function; this token is not a
# copy-initializer. A later ``= default``/``= delete`` is still
# visible to the normal function-declaration grammar.
if text[:i].rstrip().endswith("operator"):
continue
before = text[i - 1] if i else ""
after = text[i + 1] if i + 1 < len(text) else ""
if ((not before or before not in "!<>=") and
(not after or after not in "=>")):
return i
return None
def top_level_parentheses(text: str):
"""Yield the ranges/content of balanced top-level parenthesis groups."""
depth = 0
start = None
for i, char in enumerate(text):
if char == "(":
if depth == 0:
start = i
depth += 1
elif char == ")" and depth:
depth -= 1
if depth == 0 and start is not None:
yield start, i + 1, text[start + 1:i]
start = None
def strip_template_heads(statement: str) -> str | None:
"""Remove one or more balanced C++ template-heads.
Variable templates are definitions just like ordinary namespace variables;
treating every statement beginning with ``template`` as a declaration-only
form left a legal way around the header-state gate. An unbalanced head is
unknown syntax and therefore fails closed at the caller.
"""
text = statement.lstrip()
while re.match(r"template\b", text):
match = re.match(r"template\s*<", text)
if not match:
return None
depth = 1
i = match.end()
while i < len(text) and depth:
if text[i] == "<":
depth += 1
elif text[i] == ">":
depth -= 1
i += 1
if depth:
return None
text = text[i:].lstrip()
return text
_CPP_ATTRIBUTE_SUFFIX = re.compile(r"\s*\[\[.*\]\]\s*$")
_GNU_ATTRIBUTE_SUFFIX = re.compile(
r"\s*__attribute__\s*\(\(.*\)\)\s*$")
_ASM_LABEL_SUFFIX = re.compile(
r"\s+(?:asm|__asm__)\s*\(\s*(?:\"\"|''|[^()]*)\s*\)\s*$")
def strip_post_declarator_attributes(declarator: str) -> str:
"""Strip legal attributes that follow an object's identifier.
The input has already had strings/comments removed and whitespace folded.
Repeating handles combinations such as ``x [[maybe_unused]]
__attribute__((used))`` without weakening the identifier match itself.
"""
previous = None
while declarator != previous:
previous = declarator
declarator = _CPP_ATTRIBUTE_SUFFIX.sub("", declarator)
declarator = _GNU_ATTRIBUTE_SUFFIX.sub("", declarator)
return declarator.strip()
def strip_asm_label(declarator: str) -> str:
"""Remove a GNU/MS-compatible object asm-label suffix.
The label controls the emitted symbol spelling but does not turn an object
definition into a function declaration. Missing it is especially unsafe:
the renamed symbol also loses the owner namespace used by the artifact
audit.
"""
return _ASM_LABEL_SUFFIX.sub("", declarator).strip()
def direct_initializer_starts_with_expression(content: str) -> bool:
"""Recognise direct initializers that cannot be parameter declarations."""
return re.match(
r"\s*(?:[-+]?(?:\d|\.\d)|\"|'|true\b|false\b|nullptr\b|\{)",
content) is not None
def parenthesized_object_name(prefix: str, content: str) -> str | None:
"""Recover redundant parentheses around an object's declared name.
``int (State)`` is an object, while ``Registry State(Arg)`` remains a
potentially most-vexing function declaration. We accept the former only
when the prefix is a type-only spelling rather than an existing declarator.
"""
name = content.strip()
if re.fullmatch(r"[A-Za-z_]\w*", name) is None:
return None
reduced = re.sub(
r"\b(?:alignas|consteval|constexpr|constinit|extern|inline|register|"
r"static|thread_local|typedef|volatile)\b", " ", prefix)
reduced = " ".join(reduced.split())
if not reduced:
return None
builtin_words = set(_READ_ONLY_BUILTINS) | {