|
@@ -47,6 +47,38 @@
|
|
|
* Whole-graph pass after base resolution; all edges are `provenance:'heuristic'`
|
|
* Whole-graph pass after base resolution; all edges are `provenance:'heuristic'`
|
|
|
* (`synthesizedBy:'fn-pointer-dispatch'`). High precision via the (type, field)
|
|
* (`synthesizedBy:'fn-pointer-dispatch'`). High precision via the (type, field)
|
|
|
* key + a real-function gate; a project with no fn-pointer dispatch is a no-op.
|
|
* key + a real-function gate; a project with no fn-pointer dispatch is a no-op.
|
|
|
|
|
+ *
|
|
|
|
|
+ * ## Fuse-then-link architecture (§7a.8, task #5 step 1)
|
|
|
|
|
+ *
|
|
|
|
|
+ * The pass used to sweep every file's text FOUR times (typedefs, registrations,
|
|
|
|
|
+ * propagation, dispatch), and on the Linux kernel the all-or-nothing source
|
|
|
|
|
+ * cache declines, so each sweep re-read + re-stripped the whole corpus — 4.4
|
|
|
|
|
+ * strips/file, ~78s of the ~230s kernel-scale wall (§7a.8 calibration). It now
|
|
|
|
|
+ * runs as ONE extraction sweep plus filtered linking stages:
|
|
|
|
|
+ *
|
|
|
|
|
+ * 1. **Extraction sweep** — reads + strips each file ONCE and collects, per
|
|
|
|
|
+ * file: typedef names, each struct node's field declarations (parsed
|
|
|
|
|
+ * structurally, fn-pointer classification deferred — the typedef sets
|
|
|
|
|
+ * aren't complete mid-sweep), the resolved local includes, and cheap
|
|
|
|
|
+ * SURVIVAL FILTERS for the later stages (distinct initializer type
|
|
|
|
|
+ * tokens, array element types, inline-struct summaries, field-assignment
|
|
|
|
|
+ * field pairs, dispatch field / array names — all interned, a few MB even
|
|
|
|
|
+ * on the kernel).
|
|
|
|
|
+ * 2. **Struct-layout linking** — classifies the deferred fields against the
|
|
|
|
|
+ * now-complete typedef sets and registers layouts by replaying the struct
|
|
|
|
|
+ * kind-scan, so registration order (which decides same-name layout
|
|
|
|
|
+ * precedence) is byte-identical to the old dedicated pass.
|
|
|
|
|
+ * 3. **Registration / propagation / dispatch** — the original pass bodies,
|
|
|
|
|
+ * UNCHANGED, but each file is first checked against its survival filter
|
|
|
|
|
+ * and only surviving files are re-stripped (LRU-served). The filters only
|
|
|
|
|
+ * ever over-approximate: a filtered-out file is one where every match
|
|
|
|
|
+ * would have failed the pass's own gates before any side effect, so
|
|
|
|
|
+ * skipping it cannot change the edge set. On the kernel only ~16% of
|
|
|
|
|
+ * files have any dispatch-shaped match at all, so the lazy re-strips are
|
|
|
|
|
+ * a fraction of a sweep and total strip work drops ~4.4× → ~1.5×.
|
|
|
|
|
+ *
|
|
|
|
|
+ * The extraction sweep is also the step-2 boundary: a native per-file extractor
|
|
|
|
|
+ * can replace the sweep's scans without touching the linking stages.
|
|
|
*/
|
|
*/
|
|
|
import * as path from 'node:path';
|
|
import * as path from 'node:path';
|
|
|
import type { Edge, Node } from '../types';
|
|
import type { Edge, Node } from '../types';
|
|
@@ -71,7 +103,19 @@ interface FieldInfo {
|
|
|
type: string;
|
|
type: string;
|
|
|
}
|
|
}
|
|
|
|
|
|
|
|
-/** Slice a node's body from a pre-split line array — the per-file sweeps (B/D/E)
|
|
|
|
|
|
|
+/** A struct field as parsed during the extraction sweep: structure only. The
|
|
|
|
|
+ * `(*name)(…)` pointer syntax is a local fact (`ptr`), but a typedef-typed
|
|
|
|
|
+ * field's fn-pointer-ness depends on the GLOBAL typedef sets, which aren't
|
|
|
|
|
+ * complete until the sweep ends — so classification into `FieldInfo.isFnPtr`
|
|
|
|
|
+ * is deferred to the linking stage. */
|
|
|
|
|
+interface RawFieldDecl {
|
|
|
|
|
+ name: string | null;
|
|
|
|
|
+ index: number;
|
|
|
|
|
+ ptr: boolean;
|
|
|
|
|
+ type: string;
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+/** Slice a node's body from a pre-split line array — the per-file sweeps
|
|
|
* call this once per NODE, and splitting the whole file per node was an
|
|
* call this once per NODE, and splitting the whole file per node was an
|
|
|
* O(nodes × file-size) term (~1.6M full-file splits on the Linux tree,
|
|
* O(nodes × file-size) term (~1.6M full-file splits on the Linux tree,
|
|
|
* §7a.3 cFnPtr round). Split once per file, slice many times. */
|
|
* §7a.3 cFnPtr round). Split once per file, slice many times. */
|
|
@@ -313,6 +357,78 @@ const INCLUDE_RE = /#[ \t]*include[ \t]+"([^"\n]+)"/g;
|
|
|
/** Included files worth scanning for registration tables (e.g. a generated `.def`). */
|
|
/** Included files worth scanning for registration tables (e.g. a generated `.def`). */
|
|
|
const INCLUDABLE_EXT = /\.(def|inc|h|hh|hpp|hxx|c|cc|cpp|cxx|ipp|tcc|tbl)$/i;
|
|
const INCLUDABLE_EXT = /\.(def|inc|h|hh|hpp|hxx|c|cc|cpp|cxx|ipp|tcc|tbl)$/i;
|
|
|
|
|
|
|
|
|
|
+/** `#define NAME single_identifier` (possibly `struct`-prefixed) — an
|
|
|
|
|
+ * object-macro that COULD alias a struct type name (`resolveTypeName`'s exact
|
|
|
|
|
+ * value shape). The extraction sweep collects every such NAME into a global
|
|
|
|
|
+ * set: an initializer type token that direct-misses the struct layouts still
|
|
|
|
|
+ * survives the registration filter when it is alias-SHAPED anywhere, so the
|
|
|
|
|
+ * per-file macro-env alias resolution (redis' `COMMAND_STRUCT`) keeps working
|
|
|
|
|
+ * without retaining per-file object-macro tables (6.1M `#define`s on the
|
|
|
|
|
+ * Linux tree — the amdgpu register headers — rule that out). Numeric values
|
|
|
|
|
+ * are excluded: `resolveTypeName` would rewrite to a dead-end token that can
|
|
|
|
|
+ * never name a struct, so skipping them is exact, and it drops the register
|
|
|
|
|
+ * flood. */
|
|
|
|
|
+const OBJ_ALIAS_RE = /^[ \t]*#[ \t]*define[ \t]+(\w+)[ \t]+(?:struct[ \t]+)*[A-Za-z_]\w*[ \t\r]*$/gm;
|
|
|
|
|
+
|
|
|
|
|
+/** `(?:struct )?TYPE name[opt] = {` initializers, where TYPE is a struct that
|
|
|
|
|
+ * has ≥1 fn-pointer field. Handles both single (`= {…}`) and array
|
|
|
|
|
+ * (`[] = { {…}, {…} }`) forms. Macro calls inside an element are expanded first. */
|
|
|
|
|
+const INIT_RE =
|
|
|
|
|
+ /(?:^|[;{}])\s*(?:(?:static|const|extern|register|volatile)\s+)*(?:struct\s+)?(\w+)\s+(\w+)\s*(\[[^\]]*\])?\s*=\s*\{/g;
|
|
|
|
|
+/** `struct TAG { … } var[opt] [= {…}]` — the struct is defined INLINE with the
|
|
|
|
|
+ * table (vim's `cmdname`/`nv_cmd`); its layout never became a node, so parse it
|
|
|
|
|
+ * here and register it before reading the entries. No leading anchor: a
|
|
|
|
|
+ * `struct TAG {` with a brace body is always a definition (it may be preceded
|
|
|
|
|
+ * by a `#define …` line ending in a digit, as in vim), and the trailing
|
|
|
|
|
+ * `var … = {` check below is what distinguishes a TABLE from a plain type. */
|
|
|
|
|
+const INLINE_STRUCT_RE = /\bstruct\s+(\w+)\s*\{/g;
|
|
|
|
|
+/** `(?:static …)* ELEMTYPE [*] name[…] = { … }` — a bare array of function
|
|
|
|
|
+ * pointers (no struct wrapper). The optional `*` covers a function-TYPE
|
|
|
|
|
+ * typedef element (`opcode_t *opcodes[]`); a function-pointer typedef element
|
|
|
|
|
+ * (`zend_rc_dtor_func_t t[]`) needs none. The typedef-set membership gate
|
|
|
|
|
+ * is what separates this from a plain data/struct array. */
|
|
|
|
|
+const ARRAY_TABLE_RE =
|
|
|
|
|
+ /(?:^|[;{}])\s*(?:(?:static|const|extern|register|volatile)\s+)*(\w+)\s+(\*\s*)?(\w+)\s*\[[^\]]*\]\s*=\s*\{/g;
|
|
|
|
|
+/** Dispatch sites: `base->…->field(` or `base.…field(` where `field` is a known
|
|
|
|
|
+ * fn-pointer field. The base may be a chain (`c->cmd->proc`) or carry array
|
|
|
|
|
+ * subscripts (`cmdnames[i].cmd_func`). An optional `)` before the call covers
|
|
|
|
|
+ * the parenthesized form `(cmdnames[i].cmd_func)(&ea)` vim uses. */
|
|
|
|
|
+const DISPATCH_RE = /((?:\w+(?:\s*\[[^\][]*\])?\s*(?:->|\.)\s*)+)(\w+)\s*\)?\s*\(/g;
|
|
|
|
|
+/** Bare-array dispatch: `tbl[i](…)` or the explicit-deref `(*tbl[i])(…)`. The
|
|
|
|
|
+ * subscript may itself contain a call (`tbl[GC_TYPE(p)](…)`), so the index
|
|
|
|
|
+ * class excludes only brackets. Precision comes from the `arrayReg` gate —
|
|
|
|
|
+ * this fires only when `tbl` is a known fn-pointer array. */
|
|
|
|
|
+const ARRAY_DISPATCH_RE = /(?:\(\s*\*\s*)?\b(\w+)\s*\[[^\][]*\]\s*\)?\s*\(/g;
|
|
|
|
|
+/** Field←field propagation sites: `a->f = b->g`. */
|
|
|
|
|
+const FIELD_ASSIGN_RE = /(\w+)\s*(?:->|\.)\s*(\w+)\s*=\s*(\w+)\s*(?:->|\.)\s*(\w+)/g;
|
|
|
|
|
+
|
|
|
|
|
+/** Per-file facts the extraction sweep leaves behind for the linking stages.
|
|
|
|
|
+ * Everything here is a SURVIVAL FILTER (over-approximate by construction —
|
|
|
|
|
+ * collected with full-file, no-skip scans that match a superset of what the
|
|
|
|
|
+ * original pass bodies can act on) except `includes`, which is exact. */
|
|
|
|
|
+interface FileFacts {
|
|
|
|
|
+ /** Distinct `INIT_RE` type tokens (registration filter). */
|
|
|
|
|
+ initTokens: string[] | null;
|
|
|
|
|
+ /** Distinct `ARRAY_TABLE_RE` element types, `*`-prefixed when the decl has
|
|
|
|
|
+ * the pointer star (registration filter). */
|
|
|
|
|
+ arrayElems: string[] | null;
|
|
|
|
|
+ /** Any inline-struct candidate with a `(*name)(…)` field (registration filter). */
|
|
|
|
|
+ inlinePtr: boolean;
|
|
|
|
|
+ /** Field type tokens across inline-struct candidates (registration filter —
|
|
|
|
|
+ * fn-pointer-ness via typedef is only decidable once the sweep completes). */
|
|
|
|
|
+ inlineTypes: string[] | null;
|
|
|
|
|
+ /** Distinct `FIELD_ASSIGN_RE` `lfield\0rfield` pairs (propagation filter). */
|
|
|
|
|
+ dPairs: string[] | null;
|
|
|
|
|
+ /** Distinct `DISPATCH_RE` field names (dispatch filter). */
|
|
|
|
|
+ dispatchFields: string[] | null;
|
|
|
|
|
+ /** Distinct `ARRAY_DISPATCH_RE` array names (dispatch filter). */
|
|
|
|
|
+ arrayDispatchNames: string[] | null;
|
|
|
|
|
+ /** Resolved local `#include` targets, in source order (exact, from raw text). */
|
|
|
|
|
+ includes: string[];
|
|
|
|
|
+}
|
|
|
|
|
+
|
|
|
|
|
+const NO_INCLUDES: string[] = [];
|
|
|
|
|
+
|
|
|
export async function cFnPointerDispatchEdges(
|
|
export async function cFnPointerDispatchEdges(
|
|
|
_queries: QueryBuilder,
|
|
_queries: QueryBuilder,
|
|
|
ctx: ResolutionContext,
|
|
ctx: ResolutionContext,
|
|
@@ -324,17 +440,19 @@ export async function cFnPointerDispatchEdges(
|
|
|
if (files.length === 0) return [];
|
|
if (files.length === 0) return [];
|
|
|
|
|
|
|
|
// CODEGRAPH_SYNTH_TIMINGS sub-attribution: this pass is 86% of kernel-scale
|
|
// CODEGRAPH_SYNTH_TIMINGS sub-attribution: this pass is 86% of kernel-scale
|
|
|
- // synthesis (306s, §7a.2/§7a.3) — per-sweep walls + read/strip accounting
|
|
|
|
|
- // name which sweep and which cost class owns it.
|
|
|
|
|
|
|
+ // synthesis (306s, §7a.2/§7a.3) — per-stage walls + read/strip accounting
|
|
|
|
|
+ // name which stage and which cost class owns it. Post-refactor mapping:
|
|
|
|
|
+ // A = extraction sweep, B = struct-layout linking, C = registration,
|
|
|
|
|
+ // D = propagation, E = dispatch.
|
|
|
const prof = process.env.CODEGRAPH_SYNTH_TIMINGS
|
|
const prof = process.env.CODEGRAPH_SYNTH_TIMINGS
|
|
|
? { A: 0, B: 0, C: 0, D: 0, E: 0, readMs: 0, readN: 0, stripMs: 0, stripN: 0, nodesMs: 0, nodesN: 0 }
|
|
? { A: 0, B: 0, C: 0, D: 0, E: 0, readMs: 0, readN: 0, stripMs: 0, stripN: 0, nodesMs: 0, nodesN: 0 }
|
|
|
: null;
|
|
: null;
|
|
|
|
|
|
|
|
// Within-pass progress: this is the pass that parks the "Linking dynamic
|
|
// Within-pass progress: this is the pass that parks the "Linking dynamic
|
|
|
// dispatch" bar on C-heavy repos, so it reports a real fraction of its
|
|
// dispatch" bar on C-heavy repos, so it reports a real fraction of its
|
|
|
- // dominant work. `files` is swept once per file loop below (passes A, C, D,
|
|
|
|
|
- // E — pass B is node-bound and comparatively brief), reported at the same
|
|
|
|
|
- // per-16-files cadence as the cooperative yield.
|
|
|
|
|
|
|
+ // dominant work. `files` is swept once per stage loop below (extraction,
|
|
|
|
|
+ // registration, propagation, dispatch), reported at the same per-16-files
|
|
|
|
|
+ // cadence as the cooperative yield.
|
|
|
const FILE_SWEEPS = 4;
|
|
const FILE_SWEEPS = 4;
|
|
|
const tick = async (): Promise<void> => {
|
|
const tick = async (): Promise<void> => {
|
|
|
if ((++scannedFiles & 15) === 0) {
|
|
if ((++scannedFiles & 15) === 0) {
|
|
@@ -346,21 +464,19 @@ export async function cFnPointerDispatchEdges(
|
|
|
// Cache raw + stripped source per file, LRU-BOUNDED. The old unbounded Maps
|
|
// Cache raw + stripped source per file, LRU-BOUNDED. The old unbounded Maps
|
|
|
// retained every C/C++ file's raw AND stripped text for the whole pass —
|
|
// retained every C/C++ file's raw AND stripped text for the whole pass —
|
|
|
// multiple GB on the Linux kernel, one of the two OOM culprits in #1212.
|
|
// multiple GB on the Linux kernel, one of the two OOM culprits in #1212.
|
|
|
- // Every sweep below iterates in `files` order, and node-kind scans return
|
|
|
|
|
- // rows in file-commit order, so access is near-sequential and a small LRU
|
|
|
|
|
- // hits; a miss just re-reads + re-strips.
|
|
|
|
|
|
|
+ // The extraction sweep reads sequentially; the linking stages re-request
|
|
|
|
|
+ // only surviving files (plus include units), so access is near-sequential
|
|
|
|
|
+ // and a small LRU hits; a miss just re-reads + re-strips.
|
|
|
// Cache sizing is memory-budget-aware AND all-or-nothing (§7a.3 cFnPtr
|
|
// Cache sizing is memory-budget-aware AND all-or-nothing (§7a.3 cFnPtr
|
|
|
- // round): the flat 128 caused 4.4 strips/file across the pass's four file
|
|
|
|
|
- // sweeps — 71.8s of the kernel-scale wall — and a partial LRU is WORSE than
|
|
|
|
|
- // useless for cyclic sweeps (a first attempt sized ~61k against 63.8k files
|
|
|
|
|
- // and thrashed to a ~0% cross-sweep hit rate). Hold every stripped file
|
|
|
|
|
- // (~24KB each measured on the Linux tree) only when 40% of the live memory
|
|
|
|
|
- // budget covers it; otherwise keep the old within-sweep-locality 128. The
|
|
|
|
|
- // cache is pass-scoped — freed on return.
|
|
|
|
|
- // Slack over files.length: non-indexed includes (.def/.inc, generated
|
|
|
|
|
- // headers) join the working set mid-pass, and even a handful of keys past
|
|
|
|
|
- // the cap re-triggers cyclic eviction (measured: cap==files.length still
|
|
|
|
|
- // stripped 2.25×/file). Pass-scoped transient, freed on return.
|
|
|
|
|
|
|
+ // round): a partial LRU is WORSE than useless for cyclic sweeps (a first
|
|
|
|
|
+ // attempt sized ~61k against 63.8k files thrashed to a ~0% cross-sweep hit
|
|
|
|
|
+ // rate). Hold every stripped file (~24KB each measured on the Linux tree)
|
|
|
|
|
+ // only when 40% of the live memory budget covers it; otherwise keep the
|
|
|
|
|
+ // within-stage-locality 128. When the big cache declines (the kernel), the
|
|
|
|
|
+ // survival filters keep the linking stages' re-strips to a fraction of a
|
|
|
|
|
+ // sweep. Slack over files.length: non-indexed includes (.def/.inc, generated
|
|
|
|
|
+ // headers) join the working set mid-pass. Pass-scoped transient, freed on
|
|
|
|
|
+ // return.
|
|
|
const fullCacheCap = Math.ceil(files.length * 1.05) + 512;
|
|
const fullCacheCap = Math.ceil(files.length * 1.05) + 512;
|
|
|
const cacheCap = memoryBudgetBytes() * 0.5 >= fullCacheCap * 24_576 ? fullCacheCap : 128;
|
|
const cacheCap = memoryBudgetBytes() * 0.5 >= fullCacheCap * 24_576 ? fullCacheCap : 128;
|
|
|
const rawCache = new LRUCache<string, string | null>(Math.min(cacheCap, 4096));
|
|
const rawCache = new LRUCache<string, string | null>(Math.min(cacheCap, 4096));
|
|
@@ -398,46 +514,43 @@ export async function cFnPointerDispatchEdges(
|
|
|
return null;
|
|
return null;
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
- // ---- Pass A: function-pointer AND function-type typedefs (cross-file) ----
|
|
|
|
|
|
|
+ // Retained strings are interned through here. Regex captures off a big file
|
|
|
|
|
+ // string are V8 sliced strings — retaining one pins the whole parent file
|
|
|
|
|
+ // text, and the facts tables retain captures from EVERY file for the whole
|
|
|
|
|
+ // pass. The Buffer round-trip forces a flat copy on first sight; repeats
|
|
|
|
|
+ // (field names recur heavily) then share the one flat instance.
|
|
|
|
|
+ const interned = new Map<string, string>();
|
|
|
|
|
+ const intern = (x: string): string => {
|
|
|
|
|
+ let f = interned.get(x);
|
|
|
|
|
+ if (f === undefined) {
|
|
|
|
|
+ f = Buffer.from(x, 'utf8').toString('utf8');
|
|
|
|
|
+ interned.set(f, f);
|
|
|
|
|
+ }
|
|
|
|
|
+ return f;
|
|
|
|
|
+ };
|
|
|
|
|
+
|
|
|
|
|
+ // ---- Global tables the extraction sweep fills ----
|
|
|
// fn-pointer: typedef RET (*NAME)(…) → a field `NAME f` is a fn ptr
|
|
// fn-pointer: typedef RET (*NAME)(…) → a field `NAME f` is a fn ptr
|
|
|
// fn-type: typedef RET NAME(params) → a field `NAME *f` is a fn ptr
|
|
// fn-type: typedef RET NAME(params) → a field `NAME *f` is a fn ptr
|
|
|
// The fn-type form is redis' command idiom: `typedef void redisCommandProc(client*)`
|
|
// The fn-type form is redis' command idiom: `typedef void redisCommandProc(client*)`
|
|
|
// declared as `redisCommandProc *proc;`. Without this, `proc` reads as data.
|
|
// declared as `redisCommandProc *proc;`. Without this, `proc` reads as data.
|
|
|
const fnPtrTypedefs = new Set<string>();
|
|
const fnPtrTypedefs = new Set<string>();
|
|
|
const fnTypeTypedefs = new Set<string>();
|
|
const fnTypeTypedefs = new Set<string>();
|
|
|
- let tPass = Date.now();
|
|
|
|
|
- for (const file of files) {
|
|
|
|
|
- await tick();
|
|
|
|
|
- const s = src(file);
|
|
|
|
|
- if (!s || !s.includes('typedef')) continue;
|
|
|
|
|
- FNPTR_TYPEDEF_RE.lastIndex = 0;
|
|
|
|
|
- let m: RegExpExecArray | null;
|
|
|
|
|
- while ((m = FNPTR_TYPEDEF_RE.exec(s))) fnPtrTypedefs.add(m[1]!);
|
|
|
|
|
- FNTYPE_TYPEDEF_STMT_RE.lastIndex = 0;
|
|
|
|
|
- while ((m = FNTYPE_TYPEDEF_STMT_RE.exec(s))) {
|
|
|
|
|
- const guts = m[1]!;
|
|
|
|
|
- if (guts.includes('(*') || guts.includes('( *')) continue; // pointer form — handled above
|
|
|
|
|
- const fm = guts.match(/\b(\w+)\s*\(/); // last identifier before the param list
|
|
|
|
|
- if (fm && !C_TYPE_KEYWORDS.has(fm[1]!)) fnTypeTypedefs.add(fm[1]!);
|
|
|
|
|
- }
|
|
|
|
|
- }
|
|
|
|
|
- if (prof) { prof.A = Date.now() - tPass; tPass = Date.now(); }
|
|
|
|
|
|
|
+ /** Struct node id → its structurally-parsed fields (classified + registered
|
|
|
|
|
+ * in the linking stage, in kind-scan order). */
|
|
|
|
|
+ const rawFieldsByNode = new Map<string, RawFieldDecl[]>();
|
|
|
|
|
+ const factsByFile = new Map<string, FileFacts>();
|
|
|
|
|
+ /** Every inline-struct candidate tag anywhere — an over-approximation of the
|
|
|
|
|
+ * tags the registration stage can add to `structLayout` mid-stage, folded
|
|
|
|
|
+ * into the registration filter's layout check. */
|
|
|
|
|
+ const inlineTags = new Set<string>();
|
|
|
|
|
+ /** Object-macro names with an alias-shaped value anywhere (see OBJ_ALIAS_RE). */
|
|
|
|
|
+ const aliasNames = new Set<string>();
|
|
|
|
|
|
|
|
- // ---- Pass B: struct field layouts ----
|
|
|
|
|
- // structLayout: struct name → ordered fields, for structs with ≥1 fn-pointer
|
|
|
|
|
- // field (drives positional registration + dispatch).
|
|
|
|
|
- // allStructFields: EVERY struct name → ALL its field layouts (a name can be
|
|
|
|
|
- // reused across files — e.g. redis has two unrelated `client` structs), used
|
|
|
|
|
- // to walk a chained receiver's field types (`c->cmd->proc`: client.cmd →
|
|
|
|
|
- // redisCommand). The walk searches every same-named layout for the field.
|
|
|
|
|
- // fieldToStructs: fn-pointer field name → set of struct names that declare it.
|
|
|
|
|
- const structLayout = new Map<string, FieldInfo[]>();
|
|
|
|
|
- const allStructFields = new Map<string, FieldInfo[][]>();
|
|
|
|
|
- const fieldToStructs = new Map<string, Set<string>>();
|
|
|
|
|
-
|
|
|
|
|
- // Parse a struct body (the text between its `{` and `}`) into ordered fields.
|
|
|
|
|
- const parseStructFields = (inner: string): FieldInfo[] => {
|
|
|
|
|
- const fields: FieldInfo[] = [];
|
|
|
|
|
|
|
+ // Parse a struct body (the text between its `{` and `}`) into ordered fields,
|
|
|
|
|
+ // structure only — see RawFieldDecl for why classification is deferred.
|
|
|
|
|
+ const parseStructFieldsRaw = (inner: string): RawFieldDecl[] => {
|
|
|
|
|
+ const fields: RawFieldDecl[] = [];
|
|
|
let idx = 0;
|
|
let idx = 0;
|
|
|
for (const rawDecl of splitTopLevel(inner, ';')) {
|
|
for (const rawDecl of splitTopLevel(inner, ';')) {
|
|
|
const decl = rawDecl.trim();
|
|
const decl = rawDecl.trim();
|
|
@@ -452,11 +565,11 @@ export async function cFnPointerDispatchEdges(
|
|
|
const p = parts[pi]!.trim();
|
|
const p = parts[pi]!.trim();
|
|
|
let name: string | null = null;
|
|
let name: string | null = null;
|
|
|
let type = '';
|
|
let type = '';
|
|
|
- let isFnPtr = false;
|
|
|
|
|
- const ptr = p.match(FNPTR_DECL_RE);
|
|
|
|
|
- if (ptr) {
|
|
|
|
|
- name = ptr[1]!; // `… (*name)(…)` — a function pointer
|
|
|
|
|
- isFnPtr = true;
|
|
|
|
|
|
|
+ let ptr = false;
|
|
|
|
|
+ const pm = p.match(FNPTR_DECL_RE);
|
|
|
|
|
+ if (pm) {
|
|
|
|
|
+ name = pm[1]!; // `… (*name)(…)` — a function pointer
|
|
|
|
|
+ ptr = true;
|
|
|
} else if (pi === 0) {
|
|
} else if (pi === 0) {
|
|
|
if (firstTyped) { name = firstTyped[2]!; type = sharedType; }
|
|
if (firstTyped) { name = firstTyped[2]!; type = sharedType; }
|
|
|
} else {
|
|
} else {
|
|
@@ -464,17 +577,183 @@ export async function cFnPointerDispatchEdges(
|
|
|
const dm = p.match(/^\**\s*(\w+)/);
|
|
const dm = p.match(/^\**\s*(\w+)/);
|
|
|
if (dm) { name = dm[1]!; type = sharedType; }
|
|
if (dm) { name = dm[1]!; type = sharedType; }
|
|
|
}
|
|
}
|
|
|
- if (!ptr && type) isFnPtr = fnPtrTypedefs.has(type) || fnTypeTypedefs.has(type);
|
|
|
|
|
// Always advance the positional index. An unparsed field (anonymous
|
|
// Always advance the positional index. An unparsed field (anonymous
|
|
|
// union, exotic declarator) still occupies one slot, and macro-expanded
|
|
// union, exotic declarator) still occupies one slot, and macro-expanded
|
|
|
// positional tables (redis' MAKE_CMD) only align if every field counts.
|
|
// positional tables (redis' MAKE_CMD) only align if every field counts.
|
|
|
- fields.push({ name: name ?? '', index: idx, isFnPtr: !!name && isFnPtr, type });
|
|
|
|
|
|
|
+ fields.push({ name, index: idx, ptr, type });
|
|
|
idx++;
|
|
idx++;
|
|
|
}
|
|
}
|
|
|
}
|
|
}
|
|
|
return fields;
|
|
return fields;
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
|
|
+ // Classify deferred fields against the (now-complete) typedef sets.
|
|
|
|
|
+ const classifyFields = (rawFields: RawFieldDecl[]): FieldInfo[] =>
|
|
|
|
|
+ rawFields.map((f) => ({
|
|
|
|
|
+ name: f.name ?? '',
|
|
|
|
|
+ index: f.index,
|
|
|
|
|
+ isFnPtr:
|
|
|
|
|
+ !!f.name &&
|
|
|
|
|
+ (f.ptr || (!!f.type && (fnPtrTypedefs.has(f.type) || fnTypeTypedefs.has(f.type)))),
|
|
|
|
|
+ type: f.type,
|
|
|
|
|
+ }));
|
|
|
|
|
+ const parseStructFields = (inner: string): FieldInfo[] => classifyFields(parseStructFieldsRaw(inner));
|
|
|
|
|
+
|
|
|
|
|
+ // Exact per-file include resolution (from RAW source — string contents survive).
|
|
|
|
|
+ const scanIncludes = (file: string): string[] => {
|
|
|
|
|
+ const rawText = raw(file);
|
|
|
|
|
+ if (!rawText || !rawText.includes('include')) return NO_INCLUDES;
|
|
|
|
|
+ const out: string[] = [];
|
|
|
|
|
+ INCLUDE_RE.lastIndex = 0;
|
|
|
|
|
+ let im: RegExpExecArray | null;
|
|
|
|
|
+ while ((im = INCLUDE_RE.exec(rawText))) {
|
|
|
|
|
+ if (!INCLUDABLE_EXT.test(im[1]!)) continue;
|
|
|
|
|
+ const t = resolveInclude(file, im[1]!);
|
|
|
|
|
+ if (t) out.push(intern(t));
|
|
|
|
|
+ }
|
|
|
|
|
+ return out.length ? out : NO_INCLUDES;
|
|
|
|
|
+ };
|
|
|
|
|
+ // Indexed files answer from their facts; non-indexed includes (reached by
|
|
|
|
|
+ // buildEnv's depth-2 recursion) fall back to a bounded lazy scan.
|
|
|
|
|
+ const includeCache = new LRUCache<string, string[]>(1024);
|
|
|
|
|
+ const localIncludesOf = (file: string): string[] => {
|
|
|
|
|
+ const f = factsByFile.get(file);
|
|
|
|
|
+ if (f) return f.includes;
|
|
|
|
|
+ let out = includeCache.get(file);
|
|
|
|
|
+ if (out) return out;
|
|
|
|
|
+ out = scanIncludes(file);
|
|
|
|
|
+ includeCache.set(file, out);
|
|
|
|
|
+ return out;
|
|
|
|
|
+ };
|
|
|
|
|
+
|
|
|
|
|
+ // ---- Stage A: the extraction sweep — ONE read + strip per file ----
|
|
|
|
|
+ let tPass = Date.now();
|
|
|
|
|
+ for (const file of files) {
|
|
|
|
|
+ await tick();
|
|
|
|
|
+ const s = src(file);
|
|
|
|
|
+ if (!s) continue;
|
|
|
|
|
+
|
|
|
|
|
+ // Typedefs (cross-file).
|
|
|
|
|
+ if (s.includes('typedef')) {
|
|
|
|
|
+ FNPTR_TYPEDEF_RE.lastIndex = 0;
|
|
|
|
|
+ let m: RegExpExecArray | null;
|
|
|
|
|
+ while ((m = FNPTR_TYPEDEF_RE.exec(s))) fnPtrTypedefs.add(intern(m[1]!));
|
|
|
|
|
+ FNTYPE_TYPEDEF_STMT_RE.lastIndex = 0;
|
|
|
|
|
+ while ((m = FNTYPE_TYPEDEF_STMT_RE.exec(s))) {
|
|
|
|
|
+ const guts = m[1]!;
|
|
|
|
|
+ if (guts.includes('(*') || guts.includes('( *')) continue; // pointer form — handled above
|
|
|
|
|
+ const fm = guts.match(/\b(\w+)\s*\(/); // last identifier before the param list
|
|
|
|
|
+ if (fm && !C_TYPE_KEYWORDS.has(fm[1]!)) fnTypeTypedefs.add(intern(fm[1]!));
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Struct-node field declarations (registered later in kind-scan order).
|
|
|
|
|
+ const tN = prof ? Date.now() : 0;
|
|
|
|
|
+ const fileNodes = ctx.getNodesInFile(file);
|
|
|
|
|
+ if (prof) { prof.nodesMs += Date.now() - tN; prof.nodesN++; }
|
|
|
|
|
+ let lines: string[] | null = null;
|
|
|
|
|
+ for (const st of fileNodes) {
|
|
|
|
|
+ if (st.kind !== 'struct') continue;
|
|
|
|
|
+ lines ??= s.split('\n');
|
|
|
|
|
+ const body = sliceLinesPre(lines, st.startLine, st.endLine);
|
|
|
|
|
+ const open = body.indexOf('{');
|
|
|
|
|
+ const close = open >= 0 ? matchBrace(body, open) : -1;
|
|
|
|
|
+ if (open < 0 || close < 0) continue;
|
|
|
|
|
+ rawFieldsByNode.set(st.id, parseStructFieldsRaw(body.slice(open + 1, close)));
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Registration filters. These are full-file, NO-SKIP scans: the original
|
|
|
|
|
+ // registration pass jumps its scan cursor past a processed initializer
|
|
|
|
|
+ // body, so a no-skip scan finds a SUPERSET of its matches — exactly the
|
|
|
|
|
+ // over-approximation the filter needs.
|
|
|
|
|
+ const initTokens = new Set<string>();
|
|
|
|
|
+ const arrayElems = new Set<string>();
|
|
|
|
|
+ const inlineTypes = new Set<string>();
|
|
|
|
|
+ let inlinePtr = false;
|
|
|
|
|
+ if (s.includes('{')) {
|
|
|
|
|
+ INLINE_STRUCT_RE.lastIndex = 0;
|
|
|
|
|
+ let im: RegExpExecArray | null;
|
|
|
|
|
+ while ((im = INLINE_STRUCT_RE.exec(s))) {
|
|
|
|
|
+ const sOpen = im.index + im[0].length - 1;
|
|
|
|
|
+ const sClose = matchBrace(s, sOpen);
|
|
|
|
|
+ if (sClose < 0) continue;
|
|
|
|
|
+ // After `}`, expect `var [opt] [= {…}]` to be a table candidate.
|
|
|
|
|
+ const vm = s.slice(sClose + 1).match(/^\s*(\w+)\s*(\[[^\]]*\])?\s*(=\s*\{)?/);
|
|
|
|
|
+ if (!vm || !vm[1]) continue;
|
|
|
|
|
+ inlineTags.add(intern(im[1]!));
|
|
|
|
|
+ for (const f of parseStructFieldsRaw(s.slice(sOpen + 1, sClose))) {
|
|
|
|
|
+ if (!f.name) continue;
|
|
|
|
|
+ if (f.ptr) inlinePtr = true;
|
|
|
|
|
+ else if (f.type) inlineTypes.add(intern(f.type));
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+ if (s.includes('=')) {
|
|
|
|
|
+ INIT_RE.lastIndex = 0;
|
|
|
|
|
+ let m: RegExpExecArray | null;
|
|
|
|
|
+ while ((m = INIT_RE.exec(s))) initTokens.add(intern(m[1]!));
|
|
|
|
|
+ ARRAY_TABLE_RE.lastIndex = 0;
|
|
|
|
|
+ while ((m = ARRAY_TABLE_RE.exec(s))) arrayElems.add(intern((m[2] ? '*' : '') + m[1]!));
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Alias-shaped object macros (registration filter support).
|
|
|
|
|
+ if (s.includes('#define') || s.includes('# define')) {
|
|
|
|
|
+ const joined = s.replace(/\\\r?\n/g, ' ');
|
|
|
|
|
+ OBJ_ALIAS_RE.lastIndex = 0;
|
|
|
|
|
+ let m: RegExpExecArray | null;
|
|
|
|
|
+ while ((m = OBJ_ALIAS_RE.exec(joined))) aliasNames.add(intern(m[1]!));
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ // Propagation + dispatch filters (full-file scans ⊇ the per-function-body
|
|
|
|
|
+ // scans the pass bodies run — a body slice is a substring of the file).
|
|
|
|
|
+ const dPairs = new Set<string>();
|
|
|
|
|
+ if (s.includes('=')) {
|
|
|
|
|
+ FIELD_ASSIGN_RE.lastIndex = 0;
|
|
|
|
|
+ let m: RegExpExecArray | null;
|
|
|
|
|
+ while ((m = FIELD_ASSIGN_RE.exec(s))) dPairs.add(intern(m[2]! + '\0' + m[4]!));
|
|
|
|
|
+ }
|
|
|
|
|
+ const dispatchFields = new Set<string>();
|
|
|
|
|
+ const arrayNames = new Set<string>();
|
|
|
|
|
+ DISPATCH_RE.lastIndex = 0;
|
|
|
|
|
+ let dm: RegExpExecArray | null;
|
|
|
|
|
+ while ((dm = DISPATCH_RE.exec(s))) dispatchFields.add(intern(dm[2]!));
|
|
|
|
|
+ ARRAY_DISPATCH_RE.lastIndex = 0;
|
|
|
|
|
+ while ((dm = ARRAY_DISPATCH_RE.exec(s))) arrayNames.add(intern(dm[1]!));
|
|
|
|
|
+
|
|
|
|
|
+ const includes = scanIncludes(file);
|
|
|
|
|
+ if (
|
|
|
|
|
+ initTokens.size || arrayElems.size || inlinePtr || inlineTypes.size ||
|
|
|
|
|
+ dPairs.size || dispatchFields.size || arrayNames.size || includes.length
|
|
|
|
|
+ ) {
|
|
|
|
|
+ factsByFile.set(file, {
|
|
|
|
|
+ initTokens: initTokens.size ? [...initTokens] : null,
|
|
|
|
|
+ arrayElems: arrayElems.size ? [...arrayElems] : null,
|
|
|
|
|
+ inlinePtr,
|
|
|
|
|
+ inlineTypes: inlineTypes.size ? [...inlineTypes] : null,
|
|
|
|
|
+ dPairs: dPairs.size ? [...dPairs] : null,
|
|
|
|
|
+ dispatchFields: dispatchFields.size ? [...dispatchFields] : null,
|
|
|
|
|
+ arrayDispatchNames: arrayNames.size ? [...arrayNames] : null,
|
|
|
|
|
+ includes,
|
|
|
|
|
+ });
|
|
|
|
|
+ }
|
|
|
|
|
+ }
|
|
|
|
|
+ if (prof) { prof.A = Date.now() - tPass; tPass = Date.now(); }
|
|
|
|
|
+
|
|
|
|
|
+ // ---- Stage B: struct field layouts (linking — text-free) ----
|
|
|
|
|
+ // structLayout: struct name → ordered fields, for structs with ≥1 fn-pointer
|
|
|
|
|
+ // field (drives positional registration + dispatch).
|
|
|
|
|
+ // allStructFields: EVERY struct name → ALL its field layouts (a name can be
|
|
|
|
|
+ // reused across files — e.g. redis has two unrelated `client` structs), used
|
|
|
|
|
+ // to walk a chained receiver's field types (`c->cmd->proc`: client.cmd →
|
|
|
|
|
+ // redisCommand). The walk searches every same-named layout for the field.
|
|
|
|
|
+ // fieldToStructs: fn-pointer field name → set of struct names that declare it.
|
|
|
|
|
+ // Registration REPLAYS the struct kind-scan (rowid order, ≠ the extraction
|
|
|
|
|
+ // sweep's path order): same-name layout precedence — `structLayout.set`
|
|
|
|
|
+ // last-wins, `allStructFields` first-match in the chain walk — depends on it.
|
|
|
|
|
+ const structLayout = new Map<string, FieldInfo[]>();
|
|
|
|
|
+ const allStructFields = new Map<string, FieldInfo[][]>();
|
|
|
|
|
+ const fieldToStructs = new Map<string, Set<string>>();
|
|
|
|
|
+
|
|
|
// Register a parsed struct under `name` into the three indexes.
|
|
// Register a parsed struct under `name` into the three indexes.
|
|
|
const registerStructLayout = (name: string, fields: FieldInfo[]): void => {
|
|
const registerStructLayout = (name: string, fields: FieldInfo[]): void => {
|
|
|
if (!allStructFields.has(name)) allStructFields.set(name, []);
|
|
if (!allStructFields.has(name)) allStructFields.set(name, []);
|
|
@@ -488,20 +767,14 @@ export async function cFnPointerDispatchEdges(
|
|
|
if (fields.some((f) => f.isFnPtr)) structLayout.set(name, fields);
|
|
if (fields.some((f) => f.isFnPtr)) structLayout.set(name, fields);
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
- let linesFile = '';
|
|
|
|
|
- let linesArr: string[] = [];
|
|
|
|
|
for (const st of (ctx.iterateNodesByKind?.('struct') ?? ctx.getNodesByKind('struct'))) {
|
|
for (const st of (ctx.iterateNodesByKind?.('struct') ?? ctx.getNodesByKind('struct'))) {
|
|
|
if ((++scannedFiles & 255) === 0) await onYield();
|
|
if ((++scannedFiles & 255) === 0) await onYield();
|
|
|
if (!C_CPP_EXT.test(st.filePath)) continue;
|
|
if (!C_CPP_EXT.test(st.filePath)) continue;
|
|
|
- const s = src(st.filePath);
|
|
|
|
|
- if (!s) continue;
|
|
|
|
|
- if (linesFile !== st.filePath) { linesFile = st.filePath; linesArr = s.split('\n'); }
|
|
|
|
|
- const body = sliceLinesPre(linesArr, st.startLine, st.endLine);
|
|
|
|
|
- const open = body.indexOf('{');
|
|
|
|
|
- const close = open >= 0 ? matchBrace(body, open) : -1;
|
|
|
|
|
- if (open < 0 || close < 0) continue;
|
|
|
|
|
- registerStructLayout(st.name, parseStructFields(body.slice(open + 1, close)));
|
|
|
|
|
|
|
+ const rawFields = rawFieldsByNode.get(st.id);
|
|
|
|
|
+ if (!rawFields) continue; // file unreadable or body unparsable at sweep time — the old pass skipped it too
|
|
|
|
|
+ registerStructLayout(st.name, classifyFields(rawFields));
|
|
|
}
|
|
}
|
|
|
|
|
+ rawFieldsByNode.clear();
|
|
|
if (prof) { prof.B = Date.now() - tPass; tPass = Date.now(); }
|
|
if (prof) { prof.B = Date.now() - tPass; tPass = Date.now(); }
|
|
|
// NB: no early return on an empty structLayout here — an inline `struct TAG
|
|
// NB: no early return on an empty structLayout here — an inline `struct TAG
|
|
|
// { … } var[]` table whose struct never became a node (vim's `cmdname`, broken
|
|
// { … } var[]` table whose struct never became a node (vim's `cmdname`, broken
|
|
@@ -511,7 +784,7 @@ export async function cFnPointerDispatchEdges(
|
|
|
const fnPtrFieldOf = (struct: string, field: string): boolean =>
|
|
const fnPtrFieldOf = (struct: string, field: string): boolean =>
|
|
|
!!structLayout.get(struct)?.some((f) => f.name === field && f.isFnPtr);
|
|
!!structLayout.get(struct)?.some((f) => f.name === field && f.isFnPtr);
|
|
|
|
|
|
|
|
- // C/C++ function + method nodes are STREAMED per sweep (see passes D/E) —
|
|
|
|
|
|
|
+ // C/C++ function + method nodes are STREAMED per stage (see D/E) —
|
|
|
// the old materialized `cFns` array held every function node on the repo
|
|
// the old materialized `cFns` array held every function node on the repo
|
|
|
// (O(nodes) memory, part of the #1212 kernel OOM).
|
|
// (O(nodes) memory, part of the #1212 kernel OOM).
|
|
|
|
|
|
|
@@ -527,7 +800,7 @@ export async function cFnPointerDispatchEdges(
|
|
|
return cands[0]!;
|
|
return cands[0]!;
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
- // ---- Pass C: registrations — Map<"struct.field", Set<funcNodeId>> ----
|
|
|
|
|
|
|
+ // ---- Stage C: registrations — Map<"struct.field", Set<funcNodeId>> ----
|
|
|
// Ids only — retaining the full Node per registration (the old `idToNode`)
|
|
// Ids only — retaining the full Node per registration (the old `idToNode`)
|
|
|
// was write-only dead weight at O(registrations) memory.
|
|
// was write-only dead weight at O(registrations) memory.
|
|
|
const reg = new Map<string, Set<string>>();
|
|
const reg = new Map<string, Set<string>>();
|
|
@@ -625,6 +898,10 @@ export async function cFnPointerDispatchEdges(
|
|
|
|
|
|
|
|
// Per-file macro + include parsing (any file, indexed or not), cached.
|
|
// Per-file macro + include parsing (any file, indexed or not), cached.
|
|
|
// Derived per-file caches, LRU-bounded like the content caches (#1212).
|
|
// Derived per-file caches, LRU-bounded like the content caches (#1212).
|
|
|
|
|
+ // These stay LAZY (recompute-on-miss through `src`): retaining every file's
|
|
|
|
|
+ // parsed tables is ruled out by the kernel's 6.1M `#define`s, and the
|
|
|
|
|
+ // registration stage below only builds an env for files that survive its
|
|
|
|
|
+ // filter or carry local includes, so most files never need one.
|
|
|
const fnMacroCache = new LRUCache<string, Map<string, MacroDef>>(256);
|
|
const fnMacroCache = new LRUCache<string, Map<string, MacroDef>>(256);
|
|
|
const fileFnMacros = (file: string): Map<string, MacroDef> => {
|
|
const fileFnMacros = (file: string): Map<string, MacroDef> => {
|
|
|
let m = fnMacroCache.get(file);
|
|
let m = fnMacroCache.get(file);
|
|
@@ -643,24 +920,6 @@ export async function cFnPointerDispatchEdges(
|
|
|
if (!d) { d = parseDefinedNames(src(file) ?? ''); definedCache.set(file, d); }
|
|
if (!d) { d = parseDefinedNames(src(file) ?? ''); definedCache.set(file, d); }
|
|
|
return d;
|
|
return d;
|
|
|
};
|
|
};
|
|
|
- const includeCache = new LRUCache<string, string[]>(1024);
|
|
|
|
|
- const localIncludesOf = (file: string): string[] => {
|
|
|
|
|
- let out = includeCache.get(file);
|
|
|
|
|
- if (out) return out;
|
|
|
|
|
- out = [];
|
|
|
|
|
- const rawText = raw(file);
|
|
|
|
|
- if (rawText && rawText.includes('include')) {
|
|
|
|
|
- INCLUDE_RE.lastIndex = 0;
|
|
|
|
|
- let im: RegExpExecArray | null;
|
|
|
|
|
- while ((im = INCLUDE_RE.exec(rawText))) {
|
|
|
|
|
- if (!INCLUDABLE_EXT.test(im[1]!)) continue;
|
|
|
|
|
- const t = resolveInclude(file, im[1]!);
|
|
|
|
|
- if (t) out.push(t);
|
|
|
|
|
- }
|
|
|
|
|
- }
|
|
|
|
|
- includeCache.set(file, out);
|
|
|
|
|
- return out;
|
|
|
|
|
- };
|
|
|
|
|
|
|
|
|
|
// A file's effective macro environment = its own #defines PLUS those of the
|
|
// A file's effective macro environment = its own #defines PLUS those of the
|
|
|
// headers it #includes (redis' `MAKE_CMD` sits beside the table; sqlite's
|
|
// headers it #includes (redis' `MAKE_CMD` sits beside the table; sqlite's
|
|
@@ -730,25 +989,6 @@ export async function cFnPointerDispatchEdges(
|
|
|
}
|
|
}
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
- // `(?:struct )?TYPE name[opt] = {` initializers, where TYPE is a struct that
|
|
|
|
|
- // has ≥1 fn-pointer field. Handles both single (`= {…}`) and array
|
|
|
|
|
- // (`[] = { {…}, {…} }`) forms. Macro calls inside an element are expanded first.
|
|
|
|
|
- const INIT_RE =
|
|
|
|
|
- /(?:^|[;{}])\s*(?:(?:static|const|extern|register|volatile)\s+)*(?:struct\s+)?(\w+)\s+(\w+)\s*(\[[^\]]*\])?\s*=\s*\{/g;
|
|
|
|
|
- // `struct TAG { … } var[opt] [= {…}]` — the struct is defined INLINE with the
|
|
|
|
|
- // table (vim's `cmdname`/`nv_cmd`); its layout never became a node, so parse it
|
|
|
|
|
- // here and register it before reading the entries. No leading anchor: a
|
|
|
|
|
- // `struct TAG {` with a brace body is always a definition (it may be preceded
|
|
|
|
|
- // by a `#define …` line ending in a digit, as in vim), and the trailing
|
|
|
|
|
- // `var … = {` check below is what distinguishes a TABLE from a plain type.
|
|
|
|
|
- const INLINE_STRUCT_RE = /\bstruct\s+(\w+)\s*\{/g;
|
|
|
|
|
- // `(?:static …)* ELEMTYPE [*] name[…] = { … }` — a bare array of function
|
|
|
|
|
- // pointers (no struct wrapper). The optional `*` covers a function-TYPE
|
|
|
|
|
- // typedef element (`opcode_t *opcodes[]`); a function-pointer typedef element
|
|
|
|
|
- // (`zend_rc_dtor_func_t t[]`) needs none. The typedef-set membership gate
|
|
|
|
|
- // (below) is what separates this from a plain data/struct array.
|
|
|
|
|
- const ARRAY_TABLE_RE =
|
|
|
|
|
- /(?:^|[;{}])\s*(?:(?:static|const|extern|register|volatile)\s+)*(\w+)\s+(\*\s*)?(\w+)\s*\[[^\]]*\]\s*=\s*\{/g;
|
|
|
|
|
// Process ONE unit's text and discard it. The old shape built every unit up
|
|
// Process ONE unit's text and discard it. The old shape built every unit up
|
|
|
// front (`const units: Unit[]`) — the full text of every C file plus its
|
|
// front (`const units: Unit[]`) — the full text of every C file plus its
|
|
|
// expanded includes held simultaneously, gigabytes on the kernel (#1212).
|
|
// expanded includes held simultaneously, gigabytes on the kernel (#1212).
|
|
@@ -815,17 +1055,43 @@ export async function cFnPointerDispatchEdges(
|
|
|
}
|
|
}
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
- // ---- Pass C: registrations — stream each file (and its qualifying local
|
|
|
|
|
- // includes) through processUnit, one at a time.
|
|
|
|
|
|
|
+ // Can this file's OWN unit have any side effect? Every check mirrors a gate
|
|
|
|
|
+ // in processUnit, over-approximated to the filter's coarser knowledge:
|
|
|
|
|
+ // • inline structs — the fn-ptr-field gate, with per-candidate field types
|
|
|
|
|
+ // unioned per file;
|
|
|
|
|
+ // • initializers — `structLayout.has` against the layouts' SUPERSET
|
|
|
|
|
+ // (kind-scan layouts ∪ every inline tag — structLayout only grows during
|
|
|
|
|
+ // this stage), with alias-shaped tokens surviving in place of the
|
|
|
|
|
+ // per-file `resolveTypeName` walk;
|
|
|
|
|
+ // • bare arrays — the exact typedef-set gate.
|
|
|
|
|
+ // A filtered-out file is one where every match fails its gate before any
|
|
|
|
|
+ // side effect, so skipping the unit cannot change the outcome.
|
|
|
|
|
+ const typedefHit = (t: string): boolean => fnPtrTypedefs.has(t) || fnTypeTypedefs.has(t);
|
|
|
|
|
+ const regSurvives = (f: FileFacts): boolean =>
|
|
|
|
|
+ f.inlinePtr ||
|
|
|
|
|
+ (f.inlineTypes?.some(typedefHit) ?? false) ||
|
|
|
|
|
+ (f.initTokens?.some((t) => structLayout.has(t) || inlineTags.has(t) || aliasNames.has(t)) ?? false) ||
|
|
|
|
|
+ (f.arrayElems?.some((e) =>
|
|
|
|
|
+ e.charCodeAt(0) === 42 /* '*' */ ? typedefHit(e.slice(1)) : fnPtrTypedefs.has(e)
|
|
|
|
|
+ ) ?? false);
|
|
|
|
|
+
|
|
|
|
|
+ // ---- Stage C: registrations — stream each surviving file (and every file's
|
|
|
|
|
+ // qualifying local includes) through processUnit, one at a time.
|
|
|
for (const file of files) {
|
|
for (const file of files) {
|
|
|
await tick();
|
|
await tick();
|
|
|
|
|
+ const facts = factsByFile.get(file);
|
|
|
|
|
+ if (!facts) continue; // no facts ⇒ nothing matched at sweep time ⇒ the old pass would no-op here
|
|
|
|
|
+ const survives = regSurvives(facts);
|
|
|
|
|
+ if (!survives && facts.includes.length === 0) continue;
|
|
|
const env = new Map<string, MacroDef>();
|
|
const env = new Map<string, MacroDef>();
|
|
|
const objEnv = new Map<string, string>();
|
|
const objEnv = new Map<string, string>();
|
|
|
const defined = new Set<string>();
|
|
const defined = new Set<string>();
|
|
|
buildEnv(file, 2, new Set(), env, objEnv, defined);
|
|
buildEnv(file, 2, new Set(), env, objEnv, defined);
|
|
|
- const s = src(file);
|
|
|
|
|
- if (s) processUnit({ text: s, file, env, objEnv });
|
|
|
|
|
- for (const target of localIncludesOf(file)) {
|
|
|
|
|
|
|
+ if (survives) {
|
|
|
|
|
+ const s = src(file);
|
|
|
|
|
+ if (s) processUnit({ text: s, file, env, objEnv });
|
|
|
|
|
+ }
|
|
|
|
|
+ for (const target of facts.includes) {
|
|
|
if (seenInclude.has(`${file}>${target}`)) continue;
|
|
if (seenInclude.has(`${file}>${target}`)) continue;
|
|
|
const incSrc = src(target);
|
|
const incSrc = src(target);
|
|
|
if (!incSrc) continue;
|
|
if (!incSrc) continue;
|
|
@@ -907,13 +1173,22 @@ export async function cFnPointerDispatchEdges(
|
|
|
return t;
|
|
return t;
|
|
|
};
|
|
};
|
|
|
|
|
|
|
|
- // ---- Pass D: field←field propagation (`a->f = b->g`) ----
|
|
|
|
|
|
|
+ // ---- Stage D: field←field propagation (`a->f = b->g`) ----
|
|
|
// Collected as (targetStruct.field ← sourceStruct.field) pairs, then merged to
|
|
// Collected as (targetStruct.field ← sourceStruct.field) pairs, then merged to
|
|
|
// a fixpoint so a hook slot inherits a registry field's handlers.
|
|
// a fixpoint so a hook slot inherits a registry field's handlers.
|
|
|
- const FIELD_ASSIGN_RE = /(\w+)\s*(?:->|\.)\s*(\w+)\s*=\s*(\w+)\s*(?:->|\.)\s*(\w+)/g;
|
|
|
|
|
|
|
+ // Filter: a file matters only if SOME collected pair has BOTH fields known as
|
|
|
|
|
+ // fn-pointer fields — the loop body's own pre-gate. A skipped file's matches
|
|
|
|
|
+ // would all `continue` there, so skipping is side-effect-free.
|
|
|
const propagations: { to: string; from: string }[] = [];
|
|
const propagations: { to: string; from: string }[] = [];
|
|
|
for (const file of files) {
|
|
for (const file of files) {
|
|
|
await tick();
|
|
await tick();
|
|
|
|
|
+ const facts = factsByFile.get(file);
|
|
|
|
|
+ if (
|
|
|
|
|
+ !facts?.dPairs?.some((p) => {
|
|
|
|
|
+ const i = p.indexOf('\0');
|
|
|
|
|
+ return fieldToStructs.has(p.slice(0, i)) && fieldToStructs.has(p.slice(i + 1));
|
|
|
|
|
+ })
|
|
|
|
|
+ ) continue;
|
|
|
const s = src(file);
|
|
const s = src(file);
|
|
|
if (!s || !s.includes('=')) continue;
|
|
if (!s || !s.includes('=')) continue;
|
|
|
const tN = prof ? Date.now() : 0;
|
|
const tN = prof ? Date.now() : 0;
|
|
@@ -960,21 +1235,21 @@ export async function cFnPointerDispatchEdges(
|
|
|
if (prof) { prof.D = Date.now() - tPass; tPass = Date.now(); }
|
|
if (prof) { prof.D = Date.now() - tPass; tPass = Date.now(); }
|
|
|
if (reg.size === 0 && arrayReg.size === 0) return [];
|
|
if (reg.size === 0 && arrayReg.size === 0) return [];
|
|
|
|
|
|
|
|
- // ---- Pass E: dispatch sites → edges ----
|
|
|
|
|
- // `base->…->field(` or `base.…field(` where `field` is a known fn-pointer field.
|
|
|
|
|
- // The base may be a chain (`c->cmd->proc`) or carry array subscripts
|
|
|
|
|
- // (`cmdnames[i].cmd_func`). An optional `)` before the call covers the
|
|
|
|
|
- // parenthesized form `(cmdnames[i].cmd_func)(&ea)` vim uses.
|
|
|
|
|
- const DISPATCH_RE = /((?:\w+(?:\s*\[[^\][]*\])?\s*(?:->|\.)\s*)+)(\w+)\s*\)?\s*\(/g;
|
|
|
|
|
- // Bare-array dispatch: `tbl[i](…)` or the explicit-deref `(*tbl[i])(…)`. The
|
|
|
|
|
- // subscript may itself contain a call (`tbl[GC_TYPE(p)](…)`), so the index
|
|
|
|
|
- // class excludes only brackets. Precision comes from the `arrayReg` gate below
|
|
|
|
|
- // — this fires only when `tbl` is a known fn-pointer array.
|
|
|
|
|
- const ARRAY_DISPATCH_RE = /(?:\(\s*\*\s*)?\b(\w+)\s*\[[^\][]*\]\s*\)?\s*\(/g;
|
|
|
|
|
|
|
+ // ---- Stage E: dispatch sites → edges ----
|
|
|
|
|
+ // Filter: a file matters only if some dispatch field is a known fn-pointer
|
|
|
|
|
+ // field, or some subscripted name is a registered fn-pointer array — the loop
|
|
|
|
|
+ // body's own first gates (`owners` / `entries`), which a skipped file's
|
|
|
|
|
+ // matches would all fail before touching `seen`/`added`/`edges`.
|
|
|
const edges: Edge[] = [];
|
|
const edges: Edge[] = [];
|
|
|
const seen = new Set<string>();
|
|
const seen = new Set<string>();
|
|
|
for (const file of files) {
|
|
for (const file of files) {
|
|
|
await tick();
|
|
await tick();
|
|
|
|
|
+ const facts = factsByFile.get(file);
|
|
|
|
|
+ if (!facts) continue;
|
|
|
|
|
+ const eSurvives =
|
|
|
|
|
+ (facts.dispatchFields?.some((f) => fieldToStructs.has(f)) ?? false) ||
|
|
|
|
|
+ (arrayReg.size > 0 && (facts.arrayDispatchNames?.some((n) => arrayReg.has(n)) ?? false));
|
|
|
|
|
+ if (!eSurvives) continue;
|
|
|
const s = src(file);
|
|
const s = src(file);
|
|
|
if (!s) continue;
|
|
if (!s) continue;
|
|
|
const tN = prof ? Date.now() : 0;
|
|
const tN = prof ? Date.now() : 0;
|