queries.ts 150 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979198019811982198319841985198619871988198919901991199219931994199519961997199819992000200120022003200420052006200720082009201020112012201320142015201620172018201920202021202220232024202520262027202820292030203120322033203420352036203720382039204020412042204320442045204620472048204920502051205220532054205520562057205820592060206120622063206420652066206720682069207020712072207320742075207620772078207920802081208220832084208520862087208820892090209120922093209420952096209720982099210021012102210321042105210621072108210921102111211221132114211521162117211821192120212121222123212421252126212721282129213021312132213321342135213621372138213921402141214221432144214521462147214821492150215121522153215421552156215721582159216021612162216321642165216621672168216921702171217221732174217521762177217821792180218121822183218421852186218721882189219021912192219321942195219621972198219922002201220222032204220522062207220822092210221122122213221422152216221722182219222022212222222322242225222622272228222922302231223222332234223522362237223822392240224122422243224422452246224722482249225022512252225322542255225622572258225922602261226222632264226522662267226822692270227122722273227422752276227722782279228022812282228322842285228622872288228922902291229222932294229522962297229822992300230123022303230423052306230723082309231023112312231323142315231623172318231923202321232223232324232523262327232823292330233123322333233423352336233723382339234023412342234323442345234623472348234923502351235223532354235523562357235823592360236123622363236423652366236723682369237023712372237323742375237623772378237923802381238223832384238523862387238823892390239123922393239423952396239723982399240024012402240324042405240624072408240924102411241224132414241524162417241824192420242124222423242424252426242724282429243024312432243324342435243624372438243924402441244224432444244524462447244824492450245124522453245424552456245724582459246024612462246324642465246624672468246924702471247224732474247524762477247824792480248124822483248424852486248724882489249024912492249324942495249624972498249925002501250225032504250525062507250825092510251125122513251425152516251725182519252025212522252325242525252625272528252925302531253225332534253525362537253825392540254125422543254425452546254725482549255025512552255325542555255625572558255925602561256225632564256525662567256825692570257125722573257425752576257725782579258025812582258325842585258625872588258925902591259225932594259525962597259825992600260126022603260426052606260726082609261026112612261326142615261626172618261926202621262226232624262526262627262826292630263126322633263426352636263726382639264026412642264326442645264626472648264926502651265226532654265526562657265826592660266126622663266426652666266726682669267026712672267326742675267626772678267926802681268226832684268526862687268826892690269126922693269426952696269726982699270027012702270327042705270627072708270927102711271227132714271527162717271827192720272127222723272427252726272727282729273027312732273327342735273627372738273927402741274227432744274527462747274827492750275127522753275427552756275727582759276027612762276327642765276627672768276927702771277227732774277527762777277827792780278127822783278427852786278727882789279027912792279327942795279627972798279928002801280228032804280528062807280828092810281128122813281428152816281728182819282028212822282328242825282628272828282928302831283228332834283528362837283828392840284128422843284428452846284728482849285028512852285328542855285628572858285928602861286228632864286528662867286828692870287128722873287428752876287728782879288028812882288328842885288628872888288928902891289228932894289528962897289828992900290129022903290429052906290729082909291029112912291329142915291629172918291929202921292229232924292529262927292829292930293129322933293429352936293729382939294029412942294329442945294629472948294929502951295229532954295529562957295829592960296129622963296429652966296729682969297029712972297329742975297629772978297929802981298229832984298529862987298829892990299129922993299429952996299729982999300030013002300330043005300630073008300930103011301230133014301530163017301830193020302130223023302430253026302730283029303030313032303330343035303630373038303930403041304230433044304530463047304830493050305130523053305430553056305730583059306030613062306330643065306630673068306930703071307230733074307530763077307830793080308130823083308430853086308730883089309030913092309330943095309630973098309931003101310231033104310531063107310831093110311131123113311431153116311731183119312031213122312331243125312631273128312931303131313231333134313531363137313831393140314131423143314431453146314731483149315031513152315331543155315631573158315931603161316231633164316531663167316831693170317131723173317431753176317731783179318031813182318331843185318631873188318931903191319231933194319531963197319831993200320132023203320432053206320732083209321032113212321332143215321632173218321932203221322232233224322532263227322832293230323132323233323432353236323732383239324032413242324332443245324632473248324932503251325232533254325532563257325832593260326132623263326432653266326732683269327032713272327332743275327632773278327932803281328232833284328532863287328832893290329132923293329432953296329732983299330033013302330333043305330633073308330933103311331233133314331533163317331833193320332133223323332433253326332733283329333033313332333333343335333633373338333933403341334233433344334533463347334833493350335133523353335433553356335733583359336033613362336333643365336633673368336933703371337233733374337533763377337833793380338133823383338433853386338733883389339033913392339333943395339633973398339934003401340234033404340534063407340834093410341134123413341434153416341734183419342034213422342334243425342634273428342934303431343234333434343534363437343834393440344134423443344434453446344734483449345034513452345334543455345634573458345934603461346234633464346534663467346834693470347134723473347434753476347734783479348034813482348334843485348634873488348934903491349234933494349534963497349834993500350135023503350435053506350735083509351035113512351335143515351635173518351935203521352235233524352535263527352835293530353135323533353435353536353735383539354035413542354335443545354635473548354935503551355235533554355535563557355835593560356135623563356435653566356735683569357035713572357335743575357635773578357935803581358235833584358535863587358835893590359135923593359435953596359735983599360036013602360336043605360636073608360936103611361236133614361536163617361836193620362136223623362436253626362736283629363036313632363336343635363636373638363936403641364236433644364536463647364836493650365136523653365436553656365736583659366036613662366336643665366636673668366936703671367236733674367536763677367836793680368136823683368436853686368736883689369036913692369336943695369636973698369937003701370237033704370537063707370837093710371137123713371437153716371737183719372037213722372337243725372637273728372937303731373237333734373537363737373837393740374137423743374437453746374737483749375037513752375337543755375637573758375937603761376237633764376537663767376837693770377137723773377437753776377737783779378037813782378337843785378637873788378937903791379237933794379537963797379837993800
  1. /**
  2. * Database Queries
  3. *
  4. * Prepared statements for CRUD operations on the knowledge graph.
  5. */
  6. import { SqliteDatabase, SqliteStatement } from './sqlite-adapter';
  7. import {
  8. Node,
  9. Edge,
  10. FileRecord,
  11. UnresolvedReference,
  12. NodeKind,
  13. EdgeKind,
  14. Language,
  15. GraphStats,
  16. SearchOptions,
  17. SearchResult,
  18. } from '../types';
  19. import { safeJsonParse } from '../utils';
  20. import { kindBonus, nameMatchBonus, scorePathRelevance } from '../search/query-utils';
  21. import { parseQuery, boundedEditDistance } from '../search/query-parser';
  22. import { isGeneratedFile } from '../extraction/generated-detection';
  23. import { splitIdentifierSegments } from '../search/identifier-segments';
  24. /**
  25. * Files that should not be candidates for "dominant file" detection: test/spec
  26. * files and tool-generated files. Generated files (`*.pb.go`, `*.pulsar.go`,
  27. * mock outputs, …) often have huge in-file edge counts that dwarf the real
  28. * source — etcd's `rpc.pb.go` has 4× the in-file edges of `server.go`.
  29. *
  30. * Path patterns plus, when the caller passes the indexed set, files whose
  31. * HEADER declares them generated — a `payroll.go` full of generated CRUD has
  32. * exactly the same edge-density problem as `rpc.pb.go` and nothing in its name
  33. * to catch it (#1500).
  34. */
  35. function isLowValueFile(filePath: string, generated?: ReadonlySet<string>): boolean {
  36. if (generated?.has(filePath)) return true;
  37. const lp = filePath.toLowerCase();
  38. return (
  39. /(?:^|\/)(tests?|__tests?__|spec)\//.test(lp) ||
  40. /_test\.go$/.test(lp) ||
  41. /(?:^|\/)test_[^/]+\.py$/.test(lp) ||
  42. /_test\.py$/.test(lp) ||
  43. /_spec\.rb$/.test(lp) ||
  44. /_test\.rb$/.test(lp) ||
  45. /\.(test|spec)\.[jt]sx?$/.test(lp) ||
  46. /(test|spec|tests)\.(java|kt|scala)$/.test(lp) ||
  47. /(tests?|spec)\.cs$/.test(lp) ||
  48. /tests?\.swift$/.test(lp) ||
  49. /_test\.dart$/.test(lp) ||
  50. isGeneratedFile(filePath)
  51. );
  52. }
  53. const SQLITE_PARAM_CHUNK_SIZE = 500;
  54. /**
  55. * A SQL predicate: is the node aliased `alias` a member an INTERFACE declares?
  56. *
  57. * `method_signature` / `property_signature` enter the graph as `method` /
  58. * `property` nodes hung off their interface by a `contains` edge (#1638). They
  59. * have no body and originate no behaviour, so for a structural judgement about
  60. * a FILE they are the interface restated, not an extra thing the file declares.
  61. * See {@link QueryBuilder.getAmbientDeclarationPathsAmong}, the one caller, for
  62. * why treating them as opaque would break that rule in three places at once.
  63. *
  64. * Seeks `idx_edges_target_kind`, so it costs a key lookup per row rather than a
  65. * join over the whole edge table.
  66. */
  67. const IS_INTERFACE_MEMBER = (alias: string): string => `EXISTS (
  68. SELECT 1 FROM edges ce JOIN nodes owner ON owner.id = ce.source
  69. WHERE ce.target = ${alias}.id AND ce.kind = 'contains' AND owner.kind = 'interface'
  70. )`;
  71. /**
  72. * How much of the exact-name bonus a `deprioritize`d path keeps (#982). Damped
  73. * rather than zeroed: a query that genuinely targets that tree must still rank
  74. * it, the same "discount, don't erase" rule the path penalty follows.
  75. *
  76. * Derived rather than picked. `nameMatchBonus`'s prefix arm tops out below
  77. * `10 + 30 = 40`, and a de-prioritized node also takes the -15 path penalty, so
  78. * `80 * SCALE - 15 > 40` is what stops a damped WHOLE-QUERY exact match from
  79. * losing to a mere prefix match. 0.75 clears it (45). Measured on a 62k-node
  80. * django index: at 0.25 that invariant breaks in practice — `child`, `parent`
  81. * and `method` lose rank 1 to `children`, `all_parents` and `method_decorator`
  82. * — while crowd-out removal is almost flat between 0.75 and 0.5 (39 vs 40 of 88
  83. * peripheral top-10 slots cleared), so a deeper discount buys little and costs
  84. * the invariant. Pinned by a test.
  85. */
  86. export const DEPRIORITIZED_NAME_BONUS_SCALE = 0.75;
  87. /**
  88. * Database row types (snake_case from SQLite)
  89. */
  90. interface NodeRow {
  91. id: string;
  92. kind: string;
  93. name: string;
  94. qualified_name: string;
  95. file_path: string;
  96. language: string;
  97. start_line: number;
  98. end_line: number;
  99. start_column: number;
  100. end_column: number;
  101. docstring: string | null;
  102. signature: string | null;
  103. visibility: string | null;
  104. is_exported: number;
  105. is_async: number;
  106. is_static: number;
  107. is_abstract: number;
  108. decorators: string | null;
  109. type_parameters: string | null;
  110. return_type: string | null;
  111. updated_at: number;
  112. }
  113. interface EdgeRow {
  114. id: number;
  115. source: string;
  116. target: string;
  117. kind: string;
  118. metadata: string | null;
  119. line: number | null;
  120. col: number | null;
  121. provenance: string | null;
  122. }
  123. interface FileRow {
  124. path: string;
  125. content_hash: string;
  126. language: string;
  127. size: number;
  128. modified_at: number;
  129. indexed_at: number;
  130. node_count: number;
  131. errors: string | null;
  132. /** Absent on pre-v9 rows read through a stale prepared statement. */
  133. generated?: number | null;
  134. }
  135. interface UnresolvedRefRow {
  136. id: number;
  137. from_node_id: string;
  138. reference_name: string;
  139. reference_kind: string;
  140. line: number;
  141. col: number;
  142. candidates: string | null;
  143. file_path: string;
  144. language: string;
  145. status: string;
  146. name_tail: string;
  147. }
  148. /**
  149. * Last segment of a (possibly dotted/qualified) reference name — the part a
  150. * new symbol's plain node name could match: 'util.greet' → 'greet',
  151. * 'mod::fn' → 'fn', 'greet' → 'greet'. Written to unresolved_refs.name_tail
  152. * when a ref is marked failed, so the #1240 retry lookup can match dotted
  153. * refs against newly-added node names.
  154. */
  155. function referenceNameTail(referenceName: string): string {
  156. // Erlang refs carry a written arity (`f/1`, `mod::fn/2` — #1610); the tail a
  157. // new symbol's plain name could match is the arity-less function name.
  158. const base = referenceName.replace(/\/\d{1,3}$/, '') || referenceName;
  159. const idx = Math.max(base.lastIndexOf('.'), base.lastIndexOf(':'));
  160. return idx >= 0 ? base.slice(idx + 1) : base;
  161. }
  162. /**
  163. * Convert database row to Node object
  164. */
  165. function rowToNode(row: NodeRow): Node {
  166. return {
  167. id: row.id,
  168. kind: row.kind as NodeKind,
  169. name: row.name,
  170. qualifiedName: row.qualified_name,
  171. filePath: row.file_path,
  172. language: row.language as Language,
  173. startLine: row.start_line,
  174. endLine: row.end_line,
  175. startColumn: row.start_column,
  176. endColumn: row.end_column,
  177. docstring: row.docstring ?? undefined,
  178. signature: row.signature ?? undefined,
  179. visibility: row.visibility as Node['visibility'],
  180. isExported: row.is_exported === 1,
  181. isAsync: row.is_async === 1,
  182. isStatic: row.is_static === 1,
  183. isAbstract: row.is_abstract === 1,
  184. decorators: row.decorators ? safeJsonParse(row.decorators, undefined) : undefined,
  185. typeParameters: row.type_parameters ? safeJsonParse(row.type_parameters, undefined) : undefined,
  186. returnType: row.return_type ?? undefined,
  187. updatedAt: row.updated_at,
  188. };
  189. }
  190. /**
  191. * Convert database row to Edge object
  192. */
  193. function rowToEdge(row: EdgeRow): Edge {
  194. return {
  195. source: row.source,
  196. target: row.target,
  197. kind: row.kind as EdgeKind,
  198. metadata: row.metadata ? safeJsonParse(row.metadata, undefined) : undefined,
  199. line: row.line ?? undefined,
  200. column: row.col ?? undefined,
  201. provenance: row.provenance as Edge['provenance'],
  202. };
  203. }
  204. /**
  205. * Convert database row to FileRecord object
  206. */
  207. function rowToFileRecord(row: FileRow): FileRecord {
  208. return {
  209. path: row.path,
  210. contentHash: row.content_hash,
  211. language: row.language as Language,
  212. size: row.size,
  213. modifiedAt: row.modified_at,
  214. indexedAt: row.indexed_at,
  215. nodeCount: row.node_count,
  216. errors: row.errors ? safeJsonParse(row.errors, undefined) : undefined,
  217. generated: row.generated === 1,
  218. };
  219. }
  220. /**
  221. * Query builder for the knowledge graph database
  222. */
  223. export class QueryBuilder {
  224. private db: SqliteDatabase;
  225. // Project-name tokens (go.mod / package.json / repo dir), normalized. A query
  226. // word matching one is dropped from path-relevance scoring — it names the
  227. // whole project, not a symbol, so it carries no discriminative signal (#720).
  228. // Set once by the CodeGraph instance; empty by default (no down-weighting).
  229. private projectNameTokens: Set<string> = new Set();
  230. private isDeprioritizedPath: ((filePath: string) => boolean) | undefined;
  231. // FTS5 availability flag — detected once at construction time (#1532)
  232. private _fts5Available: boolean | undefined;
  233. // Node cache for frequently accessed nodes (LRU-style, max 1000 entries)
  234. private nodeCache: Map<string, Node> = new Map();
  235. private readonly maxCacheSize = 1000;
  236. // Prepared statements (lazily initialized)
  237. private stmts: {
  238. insertNode?: SqliteStatement;
  239. updateNode?: SqliteStatement;
  240. deleteNode?: SqliteStatement;
  241. deleteNodesByFile?: SqliteStatement;
  242. getNodeById?: SqliteStatement;
  243. getNodesByFile?: SqliteStatement;
  244. getNodesByKind?: SqliteStatement;
  245. insertEdge?: SqliteStatement;
  246. upsertFile?: SqliteStatement;
  247. deleteEdgesBySource?: SqliteStatement;
  248. deleteEdgesByTarget?: SqliteStatement;
  249. getEdgesBySource?: SqliteStatement;
  250. getEdgesByTarget?: SqliteStatement;
  251. getUnresolvedFromNode?: SqliteStatement;
  252. getUnresolvedInFile?: SqliteStatement;
  253. insertFile?: SqliteStatement;
  254. updateFile?: SqliteStatement;
  255. deleteFile?: SqliteStatement;
  256. getFileByPath?: SqliteStatement;
  257. getAllFiles?: SqliteStatement;
  258. insertUnresolved?: SqliteStatement;
  259. deleteUnresolvedByNode?: SqliteStatement;
  260. getUnresolvedByName?: SqliteStatement;
  261. getNodesByName?: SqliteStatement;
  262. getNodesByNamePrefix?: SqliteStatement;
  263. getNodesByQualifiedNameExact?: SqliteStatement;
  264. getNodesByLowerName?: SqliteStatement;
  265. getUnresolvedCount?: SqliteStatement;
  266. getUnresolvedBatch?: SqliteStatement;
  267. getUnresolvedBatchAfter?: SqliteStatement;
  268. getUnresolvedPrerequisitesAfter?: SqliteStatement;
  269. getUnresolvedDependentsAfter?: SqliteStatement;
  270. deleteRefsByRowIdsFull?: SqliteStatement;
  271. getAllFilePaths?: SqliteStatement;
  272. getAllNodeNames?: SqliteStatement;
  273. getDominantFile?: SqliteStatement;
  274. getTopRouteFile?: SqliteStatement;
  275. getRoutingManifest?: SqliteStatement;
  276. insertNameSegment?: SqliteStatement;
  277. } = {};
  278. // Names whose segments were already written this session — skips re-splitting
  279. // and re-inserting for the same-named nodes that repeat across files ("get",
  280. // "render", …). Purely a write-path fast path; INSERT OR IGNORE is the
  281. // correctness backstop. Bounded so a pathological repo can't grow it forever.
  282. private segmentedNames: Set<string> = new Set();
  283. private static readonly MAX_SEGMENTED_NAMES = 65536;
  284. // Multi-row INSERT statements, cached per (statement kind × row count). The
  285. // bulk write path decomposes N rows into a few fixed batch sizes so each
  286. // size's statement is prepared once and reused — one .run() binds a whole
  287. // chunk instead of one row, which is where the per-call overhead lives.
  288. // Row order within and across chunks is the input order, so rowid assignment
  289. // (and therefore resolution's insertion-order disambiguation) is identical
  290. // to the one-row-per-run path.
  291. private batchStmts: Map<string, SqliteStatement> = new Map();
  292. private static readonly BATCH_SIZES: readonly number[] = [128, 32, 8, 1];
  293. /**
  294. * Run `rows` through a multi-row `INSERT` built as `head + (tuple,)*n`,
  295. * decomposed greedily into the cached batch sizes. Preserves row order.
  296. */
  297. private runBatched(kind: string, head: string, tuple: string, rows: unknown[][]): void {
  298. if (rows.length === 0) return;
  299. let i = 0;
  300. for (const size of QueryBuilder.BATCH_SIZES) {
  301. while (rows.length - i >= size) {
  302. const key = `${kind}:${size}`;
  303. let stmt = this.batchStmts.get(key);
  304. if (!stmt) {
  305. stmt = this.db.prepare(head + new Array(size).fill(tuple).join(','));
  306. this.batchStmts.set(key, stmt);
  307. }
  308. if (size === 1) {
  309. stmt.run(...rows[i]!);
  310. } else {
  311. const params: unknown[] = [];
  312. for (let r = 0; r < size; r++) {
  313. const row = rows[i + r]!;
  314. for (let c = 0; c < row.length; c++) params.push(row[c]);
  315. }
  316. stmt.run(...params);
  317. }
  318. i += size;
  319. }
  320. }
  321. }
  322. constructor(db: SqliteDatabase) {
  323. this.db = db;
  324. // Detect FTS5 availability once (#1532)
  325. try {
  326. db.prepare("SELECT * FROM nodes_fts LIMIT 0").get();
  327. this._fts5Available = true;
  328. } catch {
  329. this._fts5Available = false;
  330. }
  331. }
  332. /**
  333. * Swap the underlying connection in place. Used by pool workers'
  334. * connection recycling (plan §7a.6, writes-under-readers): a long-lived
  335. * read connection pins WAL checkpoint progress, and the deep WAL that
  336. * accumulates behind it taxes every main-thread B-tree page operation
  337. * (deletes measured 42.6s → 118.8s from 0 to 4 attached readers on
  338. * identical hardware). Workers therefore close and reopen their read-only
  339. * connection at the pool-idle boundary; everything above the connection —
  340. * this QueryBuilder, the resolver and its warm caches — survives, and only
  341. * connection-derived state (prepared statements) resets, re-preparing
  342. * lazily on next use.
  343. */
  344. rebind(db: SqliteDatabase): void {
  345. this.db = db;
  346. this.stmts = {};
  347. this.batchStmts.clear();
  348. }
  349. /** Set the normalized project-name tokens used to down-weight non-discriminative
  350. * query words in path scoring (#720). Called once when the project opens. */
  351. setProjectNameTokens(tokens: Set<string>): void {
  352. this.projectNameTokens = tokens;
  353. }
  354. /** The normalized project-name tokens (#720); empty if none were derived. */
  355. getProjectNameTokens(): Set<string> {
  356. return this.projectNameTokens;
  357. }
  358. /**
  359. * Set the predicate that marks a path as de-prioritized by the project's
  360. * `codegraph.json` `deprioritize` patterns (#982). Ranking-only: those paths
  361. * stay indexed and findable, they just stop outranking first-party code.
  362. * Called once when the project opens; undefined disables the lever.
  363. */
  364. setDeprioritizedPathMatcher(matcher: ((filePath: string) => boolean) | undefined): void {
  365. this.isDeprioritizedPath = matcher;
  366. }
  367. /** The `deprioritize` predicate (#982), so other rankers apply the same lever. */
  368. getDeprioritizedPathMatcher(): ((filePath: string) => boolean) | undefined {
  369. return this.isDeprioritizedPath;
  370. }
  371. // ===========================================================================
  372. // Node Operations
  373. // ===========================================================================
  374. /**
  375. * Insert a new node
  376. */
  377. insertNode(node: Node): void {
  378. if (!this.stmts.insertNode) {
  379. this.stmts.insertNode = this.db.prepare(`
  380. INSERT OR REPLACE INTO nodes (
  381. id, kind, name, qualified_name, file_path, language,
  382. start_line, end_line, start_column, end_column,
  383. docstring, signature, visibility,
  384. is_exported, is_async, is_static, is_abstract,
  385. decorators, type_parameters, return_type, updated_at
  386. ) VALUES (
  387. @id, @kind, @name, @qualifiedName, @filePath, @language,
  388. @startLine, @endLine, @startColumn, @endColumn,
  389. @docstring, @signature, @visibility,
  390. @isExported, @isAsync, @isStatic, @isAbstract,
  391. @decorators, @typeParameters, @returnType, @updatedAt
  392. )
  393. `);
  394. }
  395. // Validate required fields to prevent SQLite bind errors
  396. if (!node.id || !node.kind || !node.name || !node.filePath || !node.language) {
  397. console.error('[CodeGraph] Skipping node with missing required fields:', {
  398. id: node.id,
  399. kind: node.kind,
  400. name: node.name,
  401. filePath: node.filePath,
  402. language: node.language,
  403. });
  404. return;
  405. }
  406. // INSERT OR REPLACE may overwrite a node we have cached. Drop the
  407. // stale entry so the next getNodeById sees the new row, not the old
  408. // one (matches the cache-invalidation pattern used by updateNode and
  409. // deleteNode below).
  410. this.nodeCache.delete(node.id);
  411. this.stmts.insertNode.run({
  412. id: node.id,
  413. kind: node.kind,
  414. name: node.name,
  415. qualifiedName: node.qualifiedName ?? node.name,
  416. filePath: node.filePath,
  417. language: node.language,
  418. startLine: node.startLine ?? 0,
  419. endLine: node.endLine ?? 0,
  420. startColumn: node.startColumn ?? 0,
  421. endColumn: node.endColumn ?? 0,
  422. docstring: node.docstring ?? null,
  423. signature: node.signature ?? null,
  424. visibility: node.visibility ?? null,
  425. isExported: node.isExported ? 1 : 0,
  426. isAsync: node.isAsync ? 1 : 0,
  427. isStatic: node.isStatic ? 1 : 0,
  428. isAbstract: node.isAbstract ? 1 : 0,
  429. decorators: node.decorators ? JSON.stringify(node.decorators) : null,
  430. typeParameters: node.typeParameters ? JSON.stringify(node.typeParameters) : null,
  431. returnType: node.returnType ?? null,
  432. updatedAt: node.updatedAt ?? Date.now(),
  433. });
  434. // Segment vocabulary rides the same write path (and transaction) so it can
  435. // never drift ahead of the nodes it describes. Deletes intentionally leave
  436. // orphans behind — vocab rows are proposals re-verified against nodes
  437. // before use, and a full index clears the table at its start. File nodes
  438. // are excluded: a file's basename duplicates the symbols inside it
  439. // (state-machine.ts / OrderStateMachine), which double-counts every
  440. // concept and defeats the singleton-vs-cluster rarity statistics. Import
  441. // nodes are excluded too (#1144): they're named after module specifiers
  442. // ("external-unindexed-pkg", "./utils/helpers"), not symbols — an
  443. // import-only name can never be surfaced (getSegmentMatches requires a
  444. // real definition), so its rows would only inflate the rarity statistics.
  445. if (this.isSegmentableKind(node.kind)) this.insertNameSegments(node.name);
  446. }
  447. /** Which node kinds contribute their name to the segment vocabulary — the
  448. * single gate shared by insertNode, updateNode, and the rebuild page query
  449. * (getDistinctNodeNames), so the write paths can't drift apart. */
  450. private isSegmentableKind(kind: string): boolean {
  451. return kind !== 'file' && kind !== 'import';
  452. }
  453. /** Write `name`'s segments into name_segment_vocab (idempotent). */
  454. private insertNameSegments(name: string): void {
  455. const rows: unknown[][] = [];
  456. this.collectNameSegmentRows(name, rows);
  457. this.runBatched(
  458. 'insertNameSegments',
  459. 'INSERT OR IGNORE INTO name_segment_vocab (segment, name) VALUES ',
  460. '(?,?)',
  461. rows
  462. );
  463. }
  464. /**
  465. * Insert multiple nodes in a transaction
  466. */
  467. insertNodes(nodes: Node[]): void {
  468. this.db.transaction(() => {
  469. // Bulk path: same semantics as insertNode() per row (validation, cache
  470. // invalidation, segment vocab), but bound as multi-row INSERTs — the
  471. // per-.run() call overhead dominates the store phase on full indexes.
  472. const rows: unknown[][] = [];
  473. const segmentRows: unknown[][] = [];
  474. for (const node of nodes) {
  475. if (!node.id || !node.kind || !node.name || !node.filePath || !node.language) {
  476. console.error('[CodeGraph] Skipping node with missing required fields:', {
  477. id: node.id,
  478. kind: node.kind,
  479. name: node.name,
  480. filePath: node.filePath,
  481. language: node.language,
  482. });
  483. continue;
  484. }
  485. this.nodeCache.delete(node.id);
  486. rows.push([
  487. node.id,
  488. node.kind,
  489. node.name,
  490. node.qualifiedName ?? node.name,
  491. node.filePath,
  492. node.language,
  493. node.startLine ?? 0,
  494. node.endLine ?? 0,
  495. node.startColumn ?? 0,
  496. node.endColumn ?? 0,
  497. node.docstring ?? null,
  498. node.signature ?? null,
  499. node.visibility ?? null,
  500. node.isExported ? 1 : 0,
  501. node.isAsync ? 1 : 0,
  502. node.isStatic ? 1 : 0,
  503. node.isAbstract ? 1 : 0,
  504. node.decorators ? JSON.stringify(node.decorators) : null,
  505. node.typeParameters ? JSON.stringify(node.typeParameters) : null,
  506. node.returnType ?? null,
  507. node.updatedAt ?? Date.now(),
  508. ]);
  509. if (this.isSegmentableKind(node.kind)) this.collectNameSegmentRows(node.name, segmentRows);
  510. }
  511. this.runBatched(
  512. 'insertNodes',
  513. `INSERT OR REPLACE INTO nodes (
  514. id, kind, name, qualified_name, file_path, language,
  515. start_line, end_line, start_column, end_column,
  516. docstring, signature, visibility,
  517. is_exported, is_async, is_static, is_abstract,
  518. decorators, type_parameters, return_type, updated_at
  519. ) VALUES `,
  520. '(?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)',
  521. rows
  522. );
  523. this.runBatched(
  524. 'insertNameSegments',
  525. 'INSERT OR IGNORE INTO name_segment_vocab (segment, name) VALUES ',
  526. '(?,?)',
  527. segmentRows
  528. );
  529. })();
  530. }
  531. /**
  532. * Store one file's whole extraction bundle — nodes, edges, unresolved refs,
  533. * and the file record — in a SINGLE transaction. The bulk-index path calls
  534. * this once per file instead of opening one transaction per table (#1015
  535. * file-order commit discipline is unchanged: callers still invoke it in file
  536. * order, and row order within is input order).
  537. *
  538. * Edges MUST already be endpoint-filtered by the caller (the store path
  539. * filters to the file's own inserted node ids), so the per-file existence
  540. * SELECT that insertEdges() pays is skipped here.
  541. */
  542. storeFileBundle(bundle: {
  543. nodes: Node[];
  544. edges: Edge[];
  545. refs: UnresolvedReference[];
  546. file: FileRecord;
  547. }): void {
  548. this.db.transaction(() => {
  549. this.insertNodes(bundle.nodes);
  550. if (bundle.edges.length > 0) {
  551. const rows: unknown[][] = [];
  552. for (const edge of bundle.edges) {
  553. rows.push([
  554. edge.source,
  555. edge.target,
  556. edge.kind,
  557. edge.metadata ? JSON.stringify(edge.metadata) : null,
  558. edge.line ?? null,
  559. edge.column ?? null,
  560. edge.provenance ?? null,
  561. ]);
  562. }
  563. this.runBatched(
  564. 'insertEdges',
  565. 'INSERT OR IGNORE INTO edges (source, target, kind, metadata, line, col, provenance) VALUES ',
  566. '(?,?,?,?,?,?,?)',
  567. rows
  568. );
  569. }
  570. if (bundle.refs.length > 0) this.insertUnresolvedRefsBatch(bundle.refs);
  571. this.upsertFile(bundle.file);
  572. })();
  573. }
  574. /**
  575. * Collect (segment, name) rows for a name, honouring the same session-dedupe
  576. * semantics as insertNameSegments(). Shared by the bulk write paths.
  577. */
  578. private collectNameSegmentRows(name: string, out: unknown[][]): void {
  579. if (this.segmentedNames.has(name)) return;
  580. if (this.segmentedNames.size >= QueryBuilder.MAX_SEGMENTED_NAMES) this.segmentedNames.clear();
  581. this.segmentedNames.add(name);
  582. for (const segment of splitIdentifierSegments(name)) out.push([segment, name]);
  583. }
  584. /**
  585. * Update an existing node
  586. */
  587. updateNode(node: Node): void {
  588. if (!this.stmts.updateNode) {
  589. this.stmts.updateNode = this.db.prepare(`
  590. UPDATE nodes SET
  591. kind = @kind,
  592. name = @name,
  593. qualified_name = @qualifiedName,
  594. file_path = @filePath,
  595. language = @language,
  596. start_line = @startLine,
  597. end_line = @endLine,
  598. start_column = @startColumn,
  599. end_column = @endColumn,
  600. docstring = @docstring,
  601. signature = @signature,
  602. visibility = @visibility,
  603. is_exported = @isExported,
  604. is_async = @isAsync,
  605. is_static = @isStatic,
  606. is_abstract = @isAbstract,
  607. decorators = @decorators,
  608. type_parameters = @typeParameters,
  609. return_type = @returnType,
  610. updated_at = @updatedAt
  611. WHERE id = @id
  612. `);
  613. }
  614. // Invalidate cache before update
  615. this.nodeCache.delete(node.id);
  616. // Validate required fields
  617. if (!node.id || !node.kind || !node.name || !node.filePath || !node.language) {
  618. console.error('[CodeGraph] Skipping node update with missing required fields:', node.id);
  619. return;
  620. }
  621. this.stmts.updateNode.run({
  622. id: node.id,
  623. kind: node.kind,
  624. name: node.name,
  625. qualifiedName: node.qualifiedName ?? node.name,
  626. filePath: node.filePath,
  627. language: node.language,
  628. startLine: node.startLine ?? 0,
  629. endLine: node.endLine ?? 0,
  630. startColumn: node.startColumn ?? 0,
  631. endColumn: node.endColumn ?? 0,
  632. docstring: node.docstring ?? null,
  633. signature: node.signature ?? null,
  634. visibility: node.visibility ?? null,
  635. isExported: node.isExported ? 1 : 0,
  636. isAsync: node.isAsync ? 1 : 0,
  637. isStatic: node.isStatic ? 1 : 0,
  638. isAbstract: node.isAbstract ? 1 : 0,
  639. decorators: node.decorators ? JSON.stringify(node.decorators) : null,
  640. typeParameters: node.typeParameters ? JSON.stringify(node.typeParameters) : null,
  641. returnType: node.returnType ?? null,
  642. updatedAt: node.updatedAt ?? Date.now(),
  643. });
  644. // updateNode is a second real write path to `nodes` — framework
  645. // post-extract passes rewrite names through it (NestJS route prefixing),
  646. // and a renamed node's new name must reach the segment vocabulary just
  647. // like an inserted one's (#1141). Without this the rename left the new
  648. // name permanently unsearchable: the old name's rows became honest-gate
  649. // orphans and the only backfill is gated on the vocab being EMPTY.
  650. // insertNameSegments is idempotent (in-memory set + INSERT OR IGNORE),
  651. // so no name-changed check is needed.
  652. if (this.isSegmentableKind(node.kind)) this.insertNameSegments(node.name);
  653. }
  654. /**
  655. * Delete a node by ID
  656. */
  657. deleteNode(id: string): void {
  658. if (!this.stmts.deleteNode) {
  659. this.stmts.deleteNode = this.db.prepare('DELETE FROM nodes WHERE id = ?');
  660. }
  661. // Invalidate cache
  662. this.nodeCache.delete(id);
  663. this.stmts.deleteNode.run(id);
  664. }
  665. /**
  666. * Delete all nodes for a file
  667. */
  668. deleteNodesByFile(filePath: string): void {
  669. if (!this.stmts.deleteNodesByFile) {
  670. this.stmts.deleteNodesByFile = this.db.prepare('DELETE FROM nodes WHERE file_path = ?');
  671. }
  672. // Invalidate cache for nodes in this file
  673. for (const [id, node] of this.nodeCache) {
  674. if (node.filePath === filePath) {
  675. this.nodeCache.delete(id);
  676. }
  677. }
  678. this.stmts.deleteNodesByFile.run(filePath);
  679. }
  680. // ===========================================================================
  681. // Name-segment vocabulary (prompt-hook graph-derived gate)
  682. // ===========================================================================
  683. /** Wipe the segment vocabulary. A full index calls this at its start; the
  684. * node write path repopulates it as files (re-)index, so the end state is
  685. * exactly the current names with no orphan rows. */
  686. clearNameSegmentVocab(): void {
  687. this.db.exec('DELETE FROM name_segment_vocab');
  688. this.segmentedNames.clear();
  689. }
  690. /** True when the vocab has no rows — an index built before the table existed.
  691. * `sync` uses this to heal such databases (see rebuildNameSegmentVocabFrom). */
  692. isNameSegmentVocabEmpty(): boolean {
  693. const row = this.db.prepare('SELECT 1 FROM name_segment_vocab LIMIT 1').get();
  694. return row === undefined;
  695. }
  696. /** One page of distinct segmentable node names, for batched vocab rebuilds
  697. * (file basenames and import specifiers are excluded from the vocab — see
  698. * insertNode). */
  699. getDistinctNodeNames(limit: number, offset: number): string[] {
  700. const rows = this.db
  701. .prepare("SELECT DISTINCT name FROM nodes WHERE kind NOT IN ('file', 'import') ORDER BY name LIMIT ? OFFSET ?")
  702. .all(limit, offset) as Array<{ name: string }>;
  703. return rows.map((r) => r.name);
  704. }
  705. /** Insert segments for a batch of names in one transaction (vocab heal path). */
  706. insertNameSegmentsBatch(names: string[]): void {
  707. this.db.transaction(() => {
  708. const rows: unknown[][] = [];
  709. for (const name of names) this.collectNameSegmentRows(name, rows);
  710. this.runBatched(
  711. 'insertNameSegments',
  712. 'INSERT OR IGNORE INTO name_segment_vocab (segment, name) VALUES ',
  713. '(?,?)',
  714. rows
  715. );
  716. })();
  717. }
  718. /**
  719. * Names whose segments cover at least `minWords` distinct PROMPT WORDS —
  720. * the co-occurrence probe behind the prompt hook's medium tier: the words
  721. * "state" and "machine" both being segments of `OrderStateMachine` is strong
  722. * evidence the prompt names that symbol in prose. Ordered by coverage.
  723. *
  724. * Takes (segment variant → original word) pairs and folds variants back to
  725. * their word INSIDE the SQL: a name matching both `service` and `services`
  726. * counts ONE word, not two. Counting raw variants let plural-variant pairs
  727. * of a single word tie with genuine two-word matches and — because ORDER
  728. * BY/LIMIT run here, before any JS-side re-check — crowd a real match past
  729. * the LIMIT on vocab-heavy repos (#1146).
  730. */
  731. getSegmentCoOccurrence(
  732. variants: Array<{ segment: string; word: string }>,
  733. minWords: number,
  734. limit: number,
  735. ): Array<{ name: string; matches: number }> {
  736. if (variants.length === 0) return [];
  737. const placeholders = variants.map(() => '?').join(', ');
  738. const whens = variants.map(() => 'WHEN ? THEN ?').join(' ');
  739. const rows = this.db
  740. .prepare(
  741. `SELECT name, COUNT(DISTINCT CASE segment ${whens} END) AS matches
  742. FROM name_segment_vocab
  743. WHERE segment IN (${placeholders})
  744. GROUP BY name
  745. HAVING matches >= ?
  746. ORDER BY matches DESC, length(name) ASC
  747. LIMIT ?`,
  748. )
  749. .all(
  750. ...variants.flatMap((v) => [v.segment, v.word]),
  751. ...variants.map((v) => v.segment),
  752. minWords,
  753. limit,
  754. ) as Array<{ name: string; matches: number }>;
  755. return rows;
  756. }
  757. /** How many distinct names each segment appears in — the rarity signal that
  758. * separates a discriminative word ("checkout") from a ubiquitous one ("state"). */
  759. getSegmentNameCounts(segments: string[]): Map<string, number> {
  760. if (segments.length === 0) return new Map();
  761. const placeholders = segments.map(() => '?').join(', ');
  762. const rows = this.db
  763. .prepare(
  764. `SELECT segment, COUNT(*) AS n FROM name_segment_vocab
  765. WHERE segment IN (${placeholders}) GROUP BY segment`,
  766. )
  767. .all(...segments) as Array<{ segment: string; n: number }>;
  768. return new Map(rows.map((r) => [r.segment, r.n]));
  769. }
  770. /** Names containing the given segment (rare-single-word tier). */
  771. getNamesForSegment(segment: string, limit: number): string[] {
  772. const rows = this.db
  773. .prepare('SELECT name FROM name_segment_vocab WHERE segment = ? ORDER BY length(name) ASC LIMIT ?')
  774. .all(segment, limit) as Array<{ name: string }>;
  775. return rows.map((r) => r.name);
  776. }
  777. /**
  778. * Get a node by ID
  779. */
  780. getNodeById(id: string): Node | null {
  781. // Check cache first
  782. if (this.nodeCache.has(id)) {
  783. const cached = this.nodeCache.get(id)!;
  784. // Move to end to implement LRU (delete and re-add)
  785. this.nodeCache.delete(id);
  786. this.nodeCache.set(id, cached);
  787. return cached;
  788. }
  789. if (!this.stmts.getNodeById) {
  790. this.stmts.getNodeById = this.db.prepare('SELECT * FROM nodes WHERE id = ?');
  791. }
  792. const row = this.stmts.getNodeById.get(id) as NodeRow | undefined;
  793. if (!row) {
  794. return null;
  795. }
  796. const node = rowToNode(row);
  797. this.cacheNode(node);
  798. return node;
  799. }
  800. /**
  801. * Batch lookup: fetch many nodes by ID in a single SQL round-trip.
  802. *
  803. * Replaces the N+1 pattern in graph traversal where every edge would
  804. * trigger its own `getNodeById` call. For a function with 50 callers
  805. * this collapses 50 point reads into one IN-list query (~10-50x
  806. * faster end-to-end).
  807. *
  808. * Returns a Map keyed by id so callers can preserve their own ordering
  809. * (typically the order edges were returned from the graph). Missing IDs
  810. * are simply absent from the map.
  811. *
  812. * Cache-aware: ids already in the LRU cache are served from memory and
  813. * the SQL query only touches the misses.
  814. */
  815. getNodesByIds(ids: readonly string[]): Map<string, Node> {
  816. const out = new Map<string, Node>();
  817. if (ids.length === 0) return out;
  818. // Serve cache hits first; build the miss list for SQL.
  819. const misses: string[] = [];
  820. for (const id of ids) {
  821. const cached = this.nodeCache.get(id);
  822. if (cached !== undefined) {
  823. // LRU touch
  824. this.nodeCache.delete(id);
  825. this.nodeCache.set(id, cached);
  826. out.set(id, cached);
  827. } else {
  828. misses.push(id);
  829. }
  830. }
  831. if (misses.length === 0) return out;
  832. // Chunk under SQLite's parameter limit (default 999, raised to 32766
  833. // in better-sqlite3 builds — chunk at 500 for safety across both
  834. // backends and to keep the query plan simple).
  835. for (let i = 0; i < misses.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  836. const chunk = misses.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  837. const placeholders = chunk.map(() => '?').join(',');
  838. const rows = this.db
  839. .prepare(`SELECT * FROM nodes WHERE id IN (${placeholders})`)
  840. .all(...chunk) as NodeRow[];
  841. for (const row of rows) {
  842. const node = rowToNode(row);
  843. out.set(node.id, node);
  844. this.cacheNode(node);
  845. }
  846. }
  847. return out;
  848. }
  849. private getExistingNodeIds(ids: readonly string[]): Set<string> {
  850. const out = new Set<string>();
  851. if (ids.length === 0) return out;
  852. const uniqueIds = [...new Set(ids)];
  853. for (let i = 0; i < uniqueIds.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  854. const chunk = uniqueIds.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  855. const placeholders = chunk.map(() => '?').join(',');
  856. const rows = this.db
  857. .prepare(`SELECT id FROM nodes WHERE id IN (${placeholders})`)
  858. .all(...chunk) as { id: string }[];
  859. for (const row of rows) {
  860. out.add(row.id);
  861. }
  862. }
  863. return out;
  864. }
  865. /**
  866. * Add a node to the cache, evicting oldest if needed
  867. */
  868. private cacheNode(node: Node): void {
  869. if (this.nodeCache.size >= this.maxCacheSize) {
  870. // Evict oldest (first) entry
  871. const firstKey = this.nodeCache.keys().next().value;
  872. if (firstKey) {
  873. this.nodeCache.delete(firstKey);
  874. }
  875. }
  876. this.nodeCache.set(node.id, node);
  877. }
  878. /**
  879. * Clear the node cache
  880. */
  881. clearCache(): void {
  882. this.nodeCache.clear();
  883. }
  884. /**
  885. * Get all nodes in a file
  886. */
  887. getNodesByFile(filePath: string): Node[] {
  888. if (!this.stmts.getNodesByFile) {
  889. this.stmts.getNodesByFile = this.db.prepare(
  890. 'SELECT * FROM nodes WHERE file_path = ? ORDER BY start_line'
  891. );
  892. }
  893. const rows = this.stmts.getNodesByFile.all(filePath) as NodeRow[];
  894. return rows.map(rowToNode);
  895. }
  896. /**
  897. * Find the file that holds the densest concentration of the project's
  898. * internal call graph — the "core" file. Used by context-builder to
  899. * boost ranking of symbols in that file's directory (so e.g. sinatra
  900. * queries surface `lib/sinatra/base.rb`'s `route!` instead of
  901. * `sinatra-contrib/lib/sinatra/multi_route.rb`'s `route` extension).
  902. *
  903. * Returns null if no file has a meaningful concentration (e.g. spread
  904. * evenly across many files, or empty index).
  905. *
  906. * "Internal" = source and target are in the same file. Cross-file
  907. * edges aren't useful here — they don't tell us which file is the
  908. * functional center.
  909. *
  910. * Excludes test/spec files from candidacy via path-pattern. The agent's
  911. * typical question is "how does X work", not "how is X tested", so
  912. * boosting a test file's directory would be a misfire.
  913. */
  914. getDominantFile(): { filePath: string; edgeCount: number; nextEdgeCount: number } | null {
  915. if (!this.stmts.getDominantFile) {
  916. // Pull top 20 candidates; we then filter out test/generated files
  917. // in code (regex-grade matching that SQL LIKE can't express). The
  918. // generated-file filter is critical — without it, etcd's
  919. // `api/etcdserverpb/rpc.pb.go` (1916 in-file edges, generated
  920. // protobuf stub) outranks the real `server/etcdserver/server.go`
  921. // (470 edges) by 4×, and the boost would push the agent toward
  922. // generated code.
  923. this.stmts.getDominantFile = this.db.prepare(`
  924. SELECT n.file_path AS file_path, COUNT(*) AS edge_count
  925. FROM edges e
  926. JOIN nodes n ON e.source = n.id
  927. JOIN nodes m ON e.target = m.id
  928. WHERE n.file_path = m.file_path
  929. GROUP BY n.file_path
  930. ORDER BY edge_count DESC
  931. LIMIT 20
  932. `);
  933. }
  934. const rows = this.stmts.getDominantFile.all() as Array<{ file_path: string; edge_count: number }>;
  935. const generated = this.getGeneratedPathsAmong(rows.map(r => r.file_path));
  936. const filtered = rows.filter(r => !isLowValueFile(r.file_path, generated));
  937. if (filtered.length === 0 || filtered[0]!.edge_count < 20) return null;
  938. return {
  939. filePath: filtered[0]!.file_path,
  940. edgeCount: filtered[0]!.edge_count,
  941. nextEdgeCount: filtered[1]?.edge_count ?? 0,
  942. };
  943. }
  944. /**
  945. * Find the file that holds the densest concentration of the project's
  946. * `route` nodes (framework-emitted: Express/Gin/Flask/Rails/Drupal/etc.).
  947. * Used by handleContext on small repos to inline the project's routing
  948. * config when the agent's query is about request flow — eliminating the
  949. * "Glob + Read routes.rb" pattern that beats codegraph on tiny realworld
  950. * template repos.
  951. *
  952. * Excludes test/generated files from candidacy. Returns null if there
  953. * are fewer than 3 non-test routes total, or if no file holds at least
  954. * 30% of them (diffuse routing → no single answer file).
  955. */
  956. getTopRouteFile(): { filePath: string; routeCount: number; totalRoutes: number } | null {
  957. if (!this.stmts.getTopRouteFile) {
  958. this.stmts.getTopRouteFile = this.db.prepare(`
  959. SELECT file_path, COUNT(*) AS cnt
  960. FROM nodes
  961. WHERE kind = 'route'
  962. GROUP BY file_path
  963. ORDER BY cnt DESC
  964. LIMIT 20
  965. `);
  966. }
  967. const rows = this.stmts.getTopRouteFile.all() as Array<{ file_path: string; cnt: number }>;
  968. const generated = this.getGeneratedPathsAmong(rows.map(r => r.file_path));
  969. const filtered = rows.filter(r => !isLowValueFile(r.file_path, generated));
  970. if (filtered.length === 0) return null;
  971. const totalRoutes = filtered.reduce((sum, r) => sum + r.cnt, 0);
  972. const top = filtered[0]!;
  973. if (totalRoutes < 3 || top.cnt < 3) return null;
  974. if (top.cnt / totalRoutes < 0.30) return null;
  975. return { filePath: top.file_path, routeCount: top.cnt, totalRoutes };
  976. }
  977. /**
  978. * Build a URL → handler manifest from the index. Each route node's
  979. * `references` edge points at the function/method that handles the
  980. * request. We join them in one pass; the agent gets the canonical
  981. * routing answer ("POST /users/login → AuthController#login") without
  982. * having to parse the framework's route DSL itself.
  983. *
  984. * Also returns the file with the most handler endpoints — used as the
  985. * "top handler file" to inline source for, so the agent has both the
  986. * mapping AND the handler implementations.
  987. */
  988. getRoutingManifest(limit: number = 40): {
  989. entries: Array<{
  990. url: string;
  991. handler: string;
  992. handlerFile: string;
  993. handlerLine: number;
  994. handlerKind: string;
  995. /** The route node itself: where the URL is REGISTERED, not where it is served. */
  996. routeId: string;
  997. routeFile: string;
  998. routeLine: number;
  999. }>;
  1000. topHandlerFile: string | null;
  1001. topHandlerFileCount: number;
  1002. totalRoutes: number;
  1003. } | null {
  1004. if (!this.stmts.getRoutingManifest) {
  1005. // Edge kind varies across framework resolvers: Spring/Rails/
  1006. // Laravel/Drupal emit `references`, Express emits `calls`. Accept
  1007. // both — the semantic is the same (route → its handler).
  1008. this.stmts.getRoutingManifest = this.db.prepare(`
  1009. SELECT
  1010. r.name AS url,
  1011. r.id AS route_id,
  1012. r.file_path AS route_file,
  1013. r.start_line AS route_line,
  1014. h.name AS handler,
  1015. h.file_path AS handler_file,
  1016. h.start_line AS handler_line,
  1017. h.kind AS handler_kind
  1018. FROM nodes r
  1019. JOIN edges e ON e.source = r.id
  1020. JOIN nodes h ON e.target = h.id
  1021. WHERE r.kind = 'route'
  1022. AND e.kind IN ('references', 'calls')
  1023. AND h.kind IN ('function', 'method', 'class', 'constant', 'variable')
  1024. ORDER BY r.file_path, r.start_line
  1025. LIMIT ?
  1026. `);
  1027. }
  1028. const rows = this.stmts.getRoutingManifest.all(limit) as Array<{
  1029. url: string; route_id: string; route_file: string; route_line: number;
  1030. handler: string; handler_file: string; handler_line: number; handler_kind: string;
  1031. }>;
  1032. // Drop test/generated handlers — same hygiene as elsewhere.
  1033. const generated = this.getGeneratedPathsAmong(rows.map(r => r.handler_file));
  1034. const filtered = rows.filter(r => !isLowValueFile(r.handler_file, generated));
  1035. if (filtered.length < 3) return null;
  1036. // Identify the file holding the most handlers (the "primary handler file").
  1037. const fileCounts = new Map<string, number>();
  1038. for (const r of filtered) {
  1039. fileCounts.set(r.handler_file, (fileCounts.get(r.handler_file) ?? 0) + 1);
  1040. }
  1041. let topHandlerFile: string | null = null;
  1042. let topHandlerFileCount = 0;
  1043. for (const [file, count] of fileCounts) {
  1044. if (count > topHandlerFileCount) {
  1045. topHandlerFile = file;
  1046. topHandlerFileCount = count;
  1047. }
  1048. }
  1049. return {
  1050. entries: filtered.map(r => ({
  1051. url: r.url,
  1052. handler: r.handler,
  1053. handlerFile: r.handler_file,
  1054. handlerLine: r.handler_line,
  1055. handlerKind: r.handler_kind,
  1056. routeId: r.route_id,
  1057. routeFile: r.route_file,
  1058. routeLine: r.route_line,
  1059. })),
  1060. topHandlerFile,
  1061. topHandlerFileCount,
  1062. totalRoutes: filtered.length,
  1063. };
  1064. }
  1065. /**
  1066. * Get all nodes of a specific kind
  1067. */
  1068. getNodesByKind(kind: NodeKind): Node[] {
  1069. if (!this.stmts.getNodesByKind) {
  1070. this.stmts.getNodesByKind = this.db.prepare('SELECT * FROM nodes WHERE kind = ?');
  1071. }
  1072. const rows = this.stmts.getNodesByKind.all(kind) as NodeRow[];
  1073. return rows.map(rowToNode);
  1074. }
  1075. /**
  1076. * Stream every node of a kind one at a time (lazy) instead of materializing
  1077. * them all like {@link getNodesByKind}. For unbounded kinds (`function`,
  1078. * `method`) on a symbol-dense project the full array is gigabytes; the
  1079. * dynamic-edge synthesizers only scan-and-filter, so they iterate to keep
  1080. * memory O(1) in the node count rather than O(nodes) (#610).
  1081. */
  1082. *iterateNodesByKind(kind: NodeKind): IterableIterator<Node> {
  1083. // Fresh statement per call (not a cached one): an iterator holds an open
  1084. // cursor, so a shared statement would conflict across overlapping scans.
  1085. const stmt = this.db.prepare('SELECT * FROM nodes WHERE kind = ?');
  1086. for (const row of stmt.iterate(kind)) {
  1087. yield rowToNode(row as NodeRow);
  1088. }
  1089. }
  1090. /**
  1091. * Get all nodes in the database
  1092. */
  1093. getAllNodes(): Node[] {
  1094. const rows = this.db.prepare('SELECT * FROM nodes').all() as NodeRow[];
  1095. return rows.map(rowToNode);
  1096. }
  1097. /**
  1098. * Stream nodes of one language whose `decorators` JSON array contains
  1099. * `decorator`. The LIKE on the JSON text is a cheap index-free pre-filter
  1100. * (a decorator name can appear as a substring of another), so callers must
  1101. * still exact-check `node.decorators.includes(decorator)`. Exists so the
  1102. * kotlin expect/actual synthesizer never materializes the whole node table
  1103. * the way `getAllNodes().filter(...)` did — that array alone OOM'd Node's
  1104. * default heap on a 2M-node graph (#1212).
  1105. */
  1106. *iterateNodesByLanguageWithDecorator(language: Language, decorator: string): IterableIterator<Node> {
  1107. // Fresh statement per call — an iterator holds an open cursor (see
  1108. // iterateNodesByKind).
  1109. const stmt = this.db.prepare(
  1110. "SELECT * FROM nodes WHERE language = ? AND decorators LIKE '%' || ? || '%'"
  1111. );
  1112. for (const row of stmt.iterate(language, `"${decorator}"`)) {
  1113. yield rowToNode(row as NodeRow);
  1114. }
  1115. }
  1116. /**
  1117. * Distinct languages present in the files table. One indexed aggregate —
  1118. * lets the dynamic-edge synthesizers skip passes for languages the project
  1119. * doesn't contain at all (a Kotlin pass has no work on a pure-C repo), so
  1120. * their cost is zero rather than a full-graph scan that finds nothing (#1212).
  1121. */
  1122. getDistinctFileLanguages(): Set<string> {
  1123. const rows = this.db.prepare('SELECT DISTINCT language FROM files').all() as Array<{ language: string }>;
  1124. return new Set(rows.map((r) => r.language));
  1125. }
  1126. /**
  1127. * Get nodes by exact name match (uses idx_nodes_name index).
  1128. *
  1129. * This is resolution's candidate list, and the ORDER BY is load-bearing for
  1130. * index correctness, not cosmetic (CG-33). When a reference names a symbol
  1131. * that several files define and nothing disambiguates them, resolution binds
  1132. * to the first candidate — so without an ORDER BY the winner was decided by
  1133. * rowid, i.e. by the order files happened to be WRITTEN. A full index writes
  1134. * them in scan order; an incremental sync appends each file as it changes, so
  1135. * the same tree resolved to different edges depending on how the index was
  1136. * built, and a long-lived synced index drifted away from a rebuild of itself
  1137. * (measured at 4.3% of distinct edges, mostly `calls`).
  1138. *
  1139. * `(file_path, start_line)` is a property of the CODE, so both paths now pick
  1140. * the same candidate. The sort is paid once per distinct name per resolution
  1141. * run — ReferenceResolver memoizes this in its nameCache — and the population
  1142. * is capped by AMBIGUOUS_NAME_CEILING (#999).
  1143. */
  1144. getNodesByName(name: string): Node[] {
  1145. if (!this.stmts.getNodesByName) {
  1146. this.stmts.getNodesByName = this.db.prepare(
  1147. 'SELECT * FROM nodes WHERE name = ? ORDER BY file_path, start_line'
  1148. );
  1149. }
  1150. const rows = this.stmts.getNodesByName.all(name) as NodeRow[];
  1151. return rows.map(rowToNode);
  1152. }
  1153. /**
  1154. * Nodes whose name starts with `prefix`, by index range scan (a LIKE would
  1155. * skip idx_nodes_name under SQLite's default case-insensitive LIKE).
  1156. */
  1157. getNodesByNamePrefix(prefix: string, limit = 20): Node[] {
  1158. if (!this.stmts.getNodesByNamePrefix) {
  1159. this.stmts.getNodesByNamePrefix = this.db.prepare(
  1160. 'SELECT * FROM nodes WHERE name >= ? AND name < ? ORDER BY name LIMIT ?'
  1161. );
  1162. }
  1163. const rows = this.stmts.getNodesByNamePrefix.all(prefix, prefix + '￿', limit) as NodeRow[];
  1164. return rows.map(rowToNode);
  1165. }
  1166. /**
  1167. * Get nodes by exact qualified name match (uses idx_nodes_qualified_name index)
  1168. */
  1169. getNodesByQualifiedNameExact(qualifiedName: string): Node[] {
  1170. if (!this.stmts.getNodesByQualifiedNameExact) {
  1171. this.stmts.getNodesByQualifiedNameExact = this.db.prepare(
  1172. 'SELECT * FROM nodes WHERE qualified_name = ?'
  1173. );
  1174. }
  1175. const rows = this.stmts.getNodesByQualifiedNameExact.all(qualifiedName) as NodeRow[];
  1176. return rows.map(rowToNode);
  1177. }
  1178. /**
  1179. * Get nodes by name, case-insensitively (seeks the idx_nodes_lower_name
  1180. * expression index).
  1181. *
  1182. * The parameter is lowered in SQL rather than trusted to arrive lowered, so
  1183. * the lookup means the same thing whatever casing a caller hands it. Written
  1184. * as a bare `lower(name) = ?` it silently returned nothing for any input
  1185. * carrying an uppercase letter, and — because SQLite's `lower()` folds ASCII
  1186. * only while JavaScript's `.toLowerCase()` folds Unicode — a caller that
  1187. * pre-lowered in JavaScript could not match a non-ASCII name at all.
  1188. *
  1189. * Note this hardens the query, not its one caller: `matchFuzzy` still lowers
  1190. * in JavaScript before calling, so the non-ASCII gap remains open there.
  1191. */
  1192. getNodesByLowerName(name: string): Node[] {
  1193. if (!this.stmts.getNodesByLowerName) {
  1194. this.stmts.getNodesByLowerName = this.db.prepare(
  1195. 'SELECT * FROM nodes WHERE lower(name) = lower(?)'
  1196. );
  1197. }
  1198. const rows = this.stmts.getNodesByLowerName.all(name) as NodeRow[];
  1199. return rows.map(rowToNode);
  1200. }
  1201. /**
  1202. * Search nodes by name using FTS with fallback to LIKE for better matching
  1203. *
  1204. * Search strategy:
  1205. * 1. Try FTS5 prefix match (query*) for word-start matching
  1206. * 2. If no results, try LIKE for substring matching (e.g., "signIn" finds "signInWithGoogle")
  1207. * 3. Score results based on match quality
  1208. */
  1209. searchNodes(query: string, options: SearchOptions = {}): SearchResult[] {
  1210. const { limit = 100, offset = 0 } = options;
  1211. // Parse field-qualified bits out of the raw query (kind:, lang:,
  1212. // path:, name:). Anything not recognised stays in `text` and goes
  1213. // to FTS unchanged. Filters compose with the SearchOptions arg —
  1214. // both are applied (intersection-style).
  1215. const parsed = parseQuery(query);
  1216. const mergedKinds =
  1217. parsed.kinds.length > 0
  1218. ? Array.from(new Set([...(options.kinds ?? []), ...parsed.kinds]))
  1219. : options.kinds;
  1220. const mergedLanguages =
  1221. parsed.languages.length > 0
  1222. ? Array.from(new Set([...(options.languages ?? []), ...parsed.languages]))
  1223. : options.languages;
  1224. const pathFilters = parsed.pathFilters;
  1225. const nameFilters = parsed.nameFilters;
  1226. // The text portion drives FTS/LIKE; if all the user typed was
  1227. // filters (`kind:function`), we still need *some* candidate set,
  1228. // so synthesise an empty-text path that returns everything matching
  1229. // the filters.
  1230. const text = parsed.text;
  1231. const kinds = mergedKinds;
  1232. const languages = mergedLanguages;
  1233. // First try FTS5 with prefix matching (skip if FTS5 not available, #1532)
  1234. let results = text
  1235. ? (this._fts5Available !== false ? this.searchNodesFTS(text, { kinds, languages, limit, offset }) : [])
  1236. // Over-fetch by 5× when running filter-only (no text). The
  1237. // post-scoring path: + name: filters can be very selective, so
  1238. // a smaller multiplier risks returning fewer than `limit`
  1239. // results despite the DB having plenty of matches.
  1240. : this.searchAllByFilters({ kinds, languages, limit: limit * 5 });
  1241. // If no FTS results, try LIKE-based substring search
  1242. if (results.length === 0 && text.length >= 2) {
  1243. results = this.searchNodesLike(text, { kinds, languages, limit, offset });
  1244. }
  1245. // Final fuzzy fallback: scan all known names and keep those within
  1246. // a tight Levenshtein distance. Only fires when both FTS and LIKE
  1247. // returned nothing AND there's a text portion long enough to be
  1248. // worth fuzzing (1-char queries would match too much).
  1249. if (results.length === 0 && text.length >= 3) {
  1250. results = this.searchNodesFuzzy(text, { kinds, languages, limit });
  1251. }
  1252. // Supplement: ensure exact name matches are always candidates.
  1253. // BM25 can bury short exact-match names (e.g. "getBean") under hundreds of
  1254. // compound names (e.g. "getBeanDescriptor") in large codebases,
  1255. // pushing them past the FTS fetch limit before post-hoc scoring can help.
  1256. // Use the max BM25 score as the base so the nameMatchBonus (exact=30 vs
  1257. // prefix=20) actually differentiates them after rescoring.
  1258. //
  1259. // Whole-name equality MUST be written as `lower(name) = lower(?)` so it
  1260. // seeks `idx_nodes_lower_name`. The equivalent `name = ? COLLATE NOCASE`
  1261. // matches no index — `idx_nodes_name` is BINARY-collated and the expression
  1262. // index only matches the same expression — and degrades to a full table
  1263. // scan. The `LIMIT 20` does not rescue it: SQLite can only stop early once
  1264. // it has produced 20 rows, and this runs once per query term, most of which
  1265. // name nothing in the corpus. Measured per term on an unmatched term:
  1266. // 0.08ms on gin (2.5k nodes), 0.39ms on excalidraw (11k), 2.4ms on django
  1267. // (62k) — and growing with the corpus, where the seek is flat at ~0.002ms.
  1268. // Lowering the parameter in SQL rather than in JS is deliberate: SQLite's
  1269. // `lower()` and NOCASE both fold ASCII only, while JS `.toLowerCase()`
  1270. // folds Unicode, which would silently stop matching non-ASCII names.
  1271. if (results.length > 0 && query) {
  1272. const existingIds = new Set(results.map(r => r.node.id));
  1273. const maxFtsScore = Math.max(...results.map(r => r.score));
  1274. const terms = query.split(/\s+/).filter(t => t.length >= 2);
  1275. for (const term of terms) {
  1276. let sql = 'SELECT * FROM nodes WHERE lower(name) = lower(?)';
  1277. const params: (string | number)[] = [term];
  1278. if (kinds && kinds.length > 0) {
  1279. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1280. params.push(...kinds);
  1281. }
  1282. if (languages && languages.length > 0) {
  1283. sql += ` AND language IN (${languages.map(() => '?').join(',')})`;
  1284. params.push(...languages);
  1285. }
  1286. sql += ' LIMIT 20';
  1287. const rows = this.db.prepare(sql).all(...params) as NodeRow[];
  1288. for (const row of rows) {
  1289. if (!existingIds.has(row.id)) {
  1290. results.push({ node: rowToNode(row), score: maxFtsScore });
  1291. existingIds.add(row.id);
  1292. }
  1293. }
  1294. }
  1295. }
  1296. // Apply multi-signal scoring
  1297. if (results.length > 0 && (text || query)) {
  1298. const scoringQuery = text || query;
  1299. results = results.map(r => {
  1300. // A path the project de-prioritized is saying its symbol NAMES are not
  1301. // the answer, so the exact-name bonus has to be damped too. The -15 path
  1302. // penalty alone cannot do it: the bonus is additive and larger (measured
  1303. // on #982's repro, a `usage()` helper sat at 74.8 vs 51.2 for the top
  1304. // product symbol — -15 lands at 59.8, still ahead). Damped, not zeroed,
  1305. // so the tree stays findable when it genuinely is what you asked for.
  1306. // Evaluated once and reused: the predicate stats the config file.
  1307. const deprioritized = this.isDeprioritizedPath?.(r.node.filePath) ?? false;
  1308. const nameBonus = nameMatchBonus(r.node.name, scoringQuery);
  1309. return {
  1310. ...r,
  1311. score: r.score
  1312. + kindBonus(r.node.kind)
  1313. + scorePathRelevance(r.node.filePath, scoringQuery, this.projectNameTokens, deprioritized)
  1314. + (deprioritized ? Math.round(nameBonus * DEPRIORITIZED_NAME_BONUS_SCALE) : nameBonus),
  1315. };
  1316. });
  1317. results.sort((a, b) => b.score - a.score);
  1318. // Trim to requested limit after rescoring
  1319. if (results.length > limit) {
  1320. results = results.slice(0, limit);
  1321. }
  1322. }
  1323. // Apply path: + name: filters AFTER scoring. Scoring already uses
  1324. // path/name as a soft signal; the explicit filters here are a hard
  1325. // gate. Done last so the FTS limit fetched plenty of candidates to
  1326. // narrow from.
  1327. if (pathFilters.length > 0) {
  1328. const lowered = pathFilters.map((p) => p.toLowerCase());
  1329. results = results.filter((r) => {
  1330. const fp = r.node.filePath.toLowerCase();
  1331. return lowered.some((p) => fp.includes(p));
  1332. });
  1333. }
  1334. if (nameFilters.length > 0) {
  1335. const lowered = nameFilters.map((n) => n.toLowerCase());
  1336. results = results.filter((r) => {
  1337. const nm = r.node.name.toLowerCase();
  1338. return lowered.some((n) => nm.includes(n));
  1339. });
  1340. }
  1341. return results;
  1342. }
  1343. /**
  1344. * Match-everything path used when the user supplied only field
  1345. * filters (`kind:function lang:typescript`) with no text. Returns
  1346. * candidates ordered by name; the caller's filter pass narrows to
  1347. * what was asked for.
  1348. */
  1349. private searchAllByFilters(options: {
  1350. kinds?: NodeKind[];
  1351. languages?: Language[];
  1352. limit: number;
  1353. }): SearchResult[] {
  1354. const { kinds, languages, limit } = options;
  1355. let sql = 'SELECT * FROM nodes WHERE 1=1';
  1356. const params: (string | number)[] = [];
  1357. if (kinds && kinds.length > 0) {
  1358. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1359. params.push(...kinds);
  1360. }
  1361. if (languages && languages.length > 0) {
  1362. sql += ` AND language IN (${languages.map(() => '?').join(',')})`;
  1363. params.push(...languages);
  1364. }
  1365. sql += ' ORDER BY name LIMIT ?';
  1366. params.push(limit);
  1367. const rows = this.db.prepare(sql).all(...params) as NodeRow[];
  1368. return rows.map((row) => ({ node: rowToNode(row), score: 1 }));
  1369. }
  1370. /**
  1371. * Fuzzy fallback: when zero FTS/LIKE hits, try an edit-distance
  1372. * sweep over the distinct symbol-name set. Caps `maxDist` at 2 so
  1373. * `getUssr` finds `getUser` but `process` doesn't match `prosody`.
  1374. * Bounded edit distance keeps each comparison cheap; the per-query
  1375. * scan is O(distinct-name-count) which is far smaller than total
  1376. * node count on any real codebase.
  1377. */
  1378. private searchNodesFuzzy(
  1379. text: string,
  1380. options: { kinds?: NodeKind[]; languages?: Language[]; limit: number }
  1381. ): SearchResult[] {
  1382. const { kinds, languages, limit } = options;
  1383. const lowered = text.toLowerCase();
  1384. const maxDist = lowered.length <= 4 ? 1 : 2;
  1385. // Pull the distinct name list once. The set is cached on QueryBuilder
  1386. // by getAllNodeNames(); even on a 200k-node project the distinct
  1387. // name set is typically O(10k) because most names repeat. The
  1388. // candidate-cap below bounds memory regardless.
  1389. const allNames = this.getAllNodeNames();
  1390. const candidates: Array<{ name: string; dist: number }> = [];
  1391. for (const name of allNames) {
  1392. const dist = boundedEditDistance(name.toLowerCase(), lowered, maxDist);
  1393. if (dist <= maxDist) candidates.push({ name, dist });
  1394. }
  1395. candidates.sort((a, b) => a.dist - b.dist);
  1396. // Cap the per-name follow-up queries. Each survivor triggers a
  1397. // separate `SELECT * FROM nodes WHERE name = ?`; without this cap
  1398. // a project with many similar names (`getUser1`, `getUser2`...)
  1399. // could fan out far beyond `limit` queries before the inner-loop
  1400. // limit kicks in.
  1401. const FUZZY_FOLLOWUP_CAP = Math.max(limit * 2, 50);
  1402. const cappedCandidates = candidates.slice(0, FUZZY_FOLLOWUP_CAP);
  1403. const results: SearchResult[] = [];
  1404. const seen = new Set<string>();
  1405. for (const c of cappedCandidates) {
  1406. if (results.length >= limit) break;
  1407. let sql = 'SELECT * FROM nodes WHERE name = ?';
  1408. const params: (string | number)[] = [c.name];
  1409. if (kinds && kinds.length > 0) {
  1410. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1411. params.push(...kinds);
  1412. }
  1413. if (languages && languages.length > 0) {
  1414. sql += ` AND language IN (${languages.map(() => '?').join(',')})`;
  1415. params.push(...languages);
  1416. }
  1417. sql += ' LIMIT 5';
  1418. const rows = this.db.prepare(sql).all(...params) as NodeRow[];
  1419. for (const row of rows) {
  1420. if (seen.has(row.id)) continue;
  1421. seen.add(row.id);
  1422. // Lower the score for each edit step away from the query so
  1423. // exact-match fallbacks (dist 0) outrank dist-2 typos.
  1424. results.push({ node: rowToNode(row), score: 1 / (1 + c.dist) });
  1425. if (results.length >= limit) break;
  1426. }
  1427. }
  1428. return results;
  1429. }
  1430. /**
  1431. * FTS5 search with prefix matching
  1432. */
  1433. private searchNodesFTS(query: string, options: SearchOptions): SearchResult[] {
  1434. const { kinds, languages, limit = 100, offset = 0 } = options;
  1435. // Add prefix wildcard for better matching (e.g., "auth" matches "AuthService", "authenticate")
  1436. // Escape special FTS5 characters and add prefix wildcard.
  1437. //
  1438. // `::` is a qualifier separator in Rust/C++/Ruby, not a token char,
  1439. // so treat it as whitespace before the strip step. Otherwise queries
  1440. // like `stage_apply::run` collapse to `stage_applyrun` (the colons
  1441. // are stripped without splitting) and find nothing. See #173.
  1442. const ftsQuery = query
  1443. .replace(/::/g, ' ') // Rust/C++/Ruby qualifier separator
  1444. .replace(/['"*():^]/g, '') // Remove FTS5 special chars
  1445. .split(/\s+/)
  1446. .filter(term => term.length > 0)
  1447. // Strip FTS5 boolean operators to prevent query manipulation
  1448. .filter(term => !/^(AND|OR|NOT|NEAR)$/i.test(term))
  1449. .map(term => `"${term}"*`) // Prefix match each term
  1450. .join(' OR ');
  1451. if (!ftsQuery) {
  1452. return [];
  1453. }
  1454. // BM25 column weights: id=0, name=20, qualified_name=5, docstring=1, signature=2
  1455. // Heavy name weight ensures exact/prefix name matches rank above incidental
  1456. // mentions in long docstrings or qualified names of nested symbols.
  1457. // Fetch 5x requested limit so post-hoc rescoring (kindBonus, pathRelevance,
  1458. // nameMatchBonus) can promote results that BM25 alone undervalues.
  1459. const ftsLimit = Math.max(limit * 5, 100);
  1460. let sql = `
  1461. SELECT nodes.*, bm25(nodes_fts, 0, 20, 5, 1, 2) as score
  1462. FROM nodes_fts
  1463. JOIN nodes ON nodes_fts.id = nodes.id
  1464. WHERE nodes_fts MATCH ?
  1465. `;
  1466. const params: (string | number)[] = [ftsQuery];
  1467. if (kinds && kinds.length > 0) {
  1468. sql += ` AND nodes.kind IN (${kinds.map(() => '?').join(',')})`;
  1469. params.push(...kinds);
  1470. }
  1471. if (languages && languages.length > 0) {
  1472. sql += ` AND nodes.language IN (${languages.map(() => '?').join(',')})`;
  1473. params.push(...languages);
  1474. }
  1475. sql += ' ORDER BY score LIMIT ? OFFSET ?';
  1476. params.push(ftsLimit, offset);
  1477. try {
  1478. const rows = this.db.prepare(sql).all(...params) as (NodeRow & { score: number })[];
  1479. return rows.map((row) => ({
  1480. node: rowToNode(row),
  1481. score: Math.abs(row.score), // bm25 returns negative scores
  1482. }));
  1483. } catch {
  1484. // FTS query failed, return empty
  1485. return [];
  1486. }
  1487. }
  1488. /**
  1489. * LIKE-based substring search for cases where FTS doesn't match
  1490. * Useful for camelCase matching (e.g., "signIn" finds "signInWithGoogle")
  1491. */
  1492. private searchNodesLike(query: string, options: SearchOptions): SearchResult[] {
  1493. const { kinds, languages, limit = 100, offset = 0 } = options;
  1494. let sql = `
  1495. SELECT nodes.*,
  1496. CASE
  1497. WHEN name = ? THEN 1.0
  1498. WHEN name LIKE ? THEN 0.9
  1499. WHEN name LIKE ? THEN 0.8
  1500. WHEN qualified_name LIKE ? THEN 0.7
  1501. ELSE 0.5
  1502. END as score
  1503. FROM nodes
  1504. WHERE (
  1505. name LIKE ? OR
  1506. qualified_name LIKE ? OR
  1507. name LIKE ?
  1508. )
  1509. `;
  1510. // Pattern variants for better matching
  1511. const exactMatch = query;
  1512. const startsWith = `${query}%`;
  1513. const contains = `%${query}%`;
  1514. const params: (string | number)[] = [
  1515. exactMatch, // Exact match score
  1516. startsWith, // Starts with score
  1517. contains, // Contains score
  1518. contains, // Qualified name score
  1519. contains, // WHERE: name contains
  1520. contains, // WHERE: qualified_name contains
  1521. startsWith, // WHERE: name starts with
  1522. ];
  1523. if (kinds && kinds.length > 0) {
  1524. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1525. params.push(...kinds);
  1526. }
  1527. if (languages && languages.length > 0) {
  1528. sql += ` AND language IN (${languages.map(() => '?').join(',')})`;
  1529. params.push(...languages);
  1530. }
  1531. sql += ' ORDER BY score DESC, length(name) ASC LIMIT ? OFFSET ?';
  1532. params.push(limit, offset);
  1533. const rows = this.db.prepare(sql).all(...params) as (NodeRow & { score: number })[];
  1534. return rows.map((row) => ({
  1535. node: rowToNode(row),
  1536. score: row.score,
  1537. }));
  1538. }
  1539. /**
  1540. * Find nodes by exact name match
  1541. *
  1542. * Used for hybrid search - looks up symbols by exact name or case-insensitive match.
  1543. * Returns high-confidence matches for known symbol names extracted from query.
  1544. *
  1545. * @param names - Array of symbol names to look up
  1546. * @param options - Search options (kinds, languages, limit)
  1547. * @returns SearchResult array with exact matches scored at 1.0
  1548. */
  1549. findNodesByExactName(names: string[], options: SearchOptions = {}): SearchResult[] {
  1550. if (names.length === 0) return [];
  1551. const { kinds, languages, limit = 50 } = options;
  1552. // Two-pass approach to handle common names (e.g., "run" has 40+ matches):
  1553. // Pass 1: Find which files contain distinctive (rare) symbols from the query.
  1554. // Pass 2: Query each name, boosting results that co-locate with distinctive symbols.
  1555. // Pass 1: Find files containing each queried name, identify distinctive names
  1556. //
  1557. // Both passes spell whole-name equality as `lower(name) = lower(?)` so they
  1558. // seek `idx_nodes_lower_name` — see the note in `searchNodes` for why the
  1559. // `name = ? COLLATE NOCASE` form full-scans instead. This path is the one
  1560. // that hurts most: it runs both passes for every symbol extracted from the
  1561. // query, and extraction is generous, so most of those names are absent from
  1562. // the corpus and never reach either LIMIT.
  1563. const nameToFiles = new Map<string, Set<string>>();
  1564. for (const name of names) {
  1565. let sql = 'SELECT DISTINCT file_path FROM nodes WHERE lower(name) = lower(?)';
  1566. const params: (string | number)[] = [name];
  1567. if (kinds && kinds.length > 0) {
  1568. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1569. params.push(...kinds);
  1570. }
  1571. sql += ' LIMIT 100';
  1572. const rows = this.db.prepare(sql).all(...params) as { file_path: string }[];
  1573. nameToFiles.set(name.toLowerCase(), new Set(rows.map(r => r.file_path)));
  1574. }
  1575. // Distinctive names are those with fewer than 10 file matches (e.g., "scrapeLoop" = 1 file)
  1576. const distinctiveFiles = new Set<string>();
  1577. for (const [, files] of nameToFiles) {
  1578. if (files.size > 0 && files.size < 10) {
  1579. for (const f of files) distinctiveFiles.add(f);
  1580. }
  1581. }
  1582. // Pass 2: Query each name with per-name limit, scoring by co-location
  1583. const perNameLimit = Math.max(8, Math.ceil(limit / names.length));
  1584. const allResults: SearchResult[] = [];
  1585. const seenIds = new Set<string>();
  1586. for (const name of names) {
  1587. let sql = `
  1588. SELECT nodes.*, 1.0 as score
  1589. FROM nodes
  1590. WHERE lower(name) = lower(?)
  1591. `;
  1592. const params: (string | number)[] = [name];
  1593. if (kinds && kinds.length > 0) {
  1594. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1595. params.push(...kinds);
  1596. }
  1597. if (languages && languages.length > 0) {
  1598. sql += ` AND language IN (${languages.map(() => '?').join(',')})`;
  1599. params.push(...languages);
  1600. }
  1601. // Fetch enough to find co-located results among common names
  1602. sql += ' LIMIT ?';
  1603. params.push(Math.max(perNameLimit * 3, 50));
  1604. const rows = this.db.prepare(sql).all(...params) as (NodeRow & { score: number })[];
  1605. const nameResults: SearchResult[] = [];
  1606. for (const row of rows) {
  1607. const node = rowToNode(row);
  1608. if (seenIds.has(node.id)) continue;
  1609. // Boost results in files that also contain distinctive symbols
  1610. const coLocationBoost = distinctiveFiles.has(node.filePath) ? 20 : 0;
  1611. nameResults.push({ node, score: row.score + coLocationBoost });
  1612. }
  1613. // Sort by score (co-located first), take per-name limit
  1614. nameResults.sort((a, b) => b.score - a.score);
  1615. for (const r of nameResults.slice(0, perNameLimit)) {
  1616. seenIds.add(r.node.id);
  1617. allResults.push(r);
  1618. }
  1619. }
  1620. // Sort all results by score so co-located results bubble up
  1621. allResults.sort((a, b) => b.score - a.score);
  1622. return allResults.slice(0, limit);
  1623. }
  1624. /**
  1625. * Find nodes whose name contains a substring (LIKE-based).
  1626. * Useful for CamelCase-part matching where FTS fails because
  1627. * e.g. "TransportSearchAction" is one FTS token, not matchable by "Search"*.
  1628. *
  1629. * Results are ordered by name length (shorter = more likely to be the core type).
  1630. */
  1631. findNodesByNameSubstring(
  1632. substring: string,
  1633. options: SearchOptions & { excludePrefix?: boolean } = {}
  1634. ): SearchResult[] {
  1635. const { kinds, languages, limit = 30, excludePrefix } = options;
  1636. let sql = `
  1637. SELECT nodes.*, 1.0 as score
  1638. FROM nodes
  1639. WHERE name LIKE ?
  1640. `;
  1641. const params: (string | number)[] = [`%${substring}%`];
  1642. // Exclude prefix matches (handled by FTS-based prefix search in Step 2b)
  1643. if (excludePrefix) {
  1644. sql += ` AND name NOT LIKE ?`;
  1645. params.push(`${substring}%`);
  1646. }
  1647. if (kinds && kinds.length > 0) {
  1648. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1649. params.push(...kinds);
  1650. }
  1651. if (languages && languages.length > 0) {
  1652. sql += ` AND language IN (${languages.map(() => '?').join(',')})`;
  1653. params.push(...languages);
  1654. }
  1655. sql += ' ORDER BY length(name) ASC LIMIT ?';
  1656. params.push(limit);
  1657. const rows = this.db.prepare(sql).all(...params) as (NodeRow & { score: number })[];
  1658. return rows.map((row) => ({
  1659. node: rowToNode(row),
  1660. score: row.score,
  1661. }));
  1662. }
  1663. // ===========================================================================
  1664. // Edge Operations
  1665. // ===========================================================================
  1666. /**
  1667. * Insert a new edge
  1668. */
  1669. insertEdge(edge: Edge): void {
  1670. if (!this.stmts.insertEdge) {
  1671. this.stmts.insertEdge = this.db.prepare(`
  1672. INSERT OR IGNORE INTO edges (source, target, kind, metadata, line, col, provenance)
  1673. VALUES (@source, @target, @kind, @metadata, @line, @col, @provenance)
  1674. `);
  1675. }
  1676. this.stmts.insertEdge.run({
  1677. source: edge.source,
  1678. target: edge.target,
  1679. kind: edge.kind,
  1680. metadata: edge.metadata ? JSON.stringify(edge.metadata) : null,
  1681. line: edge.line ?? null,
  1682. col: edge.column ?? null,
  1683. provenance: edge.provenance ?? null,
  1684. });
  1685. }
  1686. /**
  1687. * Insert multiple edges in a transaction
  1688. */
  1689. insertEdges(edges: Edge[]): void {
  1690. if (edges.length === 0) return;
  1691. this.db.transaction(() => {
  1692. const endpointIds = new Set<string>();
  1693. for (const edge of edges) {
  1694. endpointIds.add(edge.source);
  1695. endpointIds.add(edge.target);
  1696. }
  1697. const existingNodeIds = this.getExistingNodeIds([...endpointIds]);
  1698. const rows: unknown[][] = [];
  1699. for (const edge of edges) {
  1700. if (!existingNodeIds.has(edge.source) || !existingNodeIds.has(edge.target)) {
  1701. continue;
  1702. }
  1703. rows.push([
  1704. edge.source,
  1705. edge.target,
  1706. edge.kind,
  1707. edge.metadata ? JSON.stringify(edge.metadata) : null,
  1708. edge.line ?? null,
  1709. edge.column ?? null,
  1710. edge.provenance ?? null,
  1711. ]);
  1712. }
  1713. this.runBatched(
  1714. 'insertEdges',
  1715. 'INSERT OR IGNORE INTO edges (source, target, kind, metadata, line, col, provenance) VALUES ',
  1716. '(?,?,?,?,?,?,?)',
  1717. rows
  1718. );
  1719. })();
  1720. }
  1721. /**
  1722. * Delete all edges from a source node
  1723. */
  1724. deleteEdgesBySource(sourceId: string): void {
  1725. if (!this.stmts.deleteEdgesBySource) {
  1726. this.stmts.deleteEdgesBySource = this.db.prepare('DELETE FROM edges WHERE source = ?');
  1727. }
  1728. this.stmts.deleteEdgesBySource.run(sourceId);
  1729. }
  1730. /**
  1731. * Get outgoing edges from a node
  1732. */
  1733. getOutgoingEdges(sourceId: string, kinds?: EdgeKind[], provenance?: string): Edge[] {
  1734. if ((kinds && kinds.length > 0) || provenance) {
  1735. let sql = 'SELECT * FROM edges WHERE source = ?';
  1736. const params: (string | number)[] = [sourceId];
  1737. if (kinds && kinds.length > 0) {
  1738. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1739. params.push(...kinds);
  1740. }
  1741. if (provenance) {
  1742. sql += ' AND provenance = ?';
  1743. params.push(provenance);
  1744. }
  1745. const rows = this.db.prepare(sql).all(...params) as EdgeRow[];
  1746. return rows.map(rowToEdge);
  1747. }
  1748. if (!this.stmts.getEdgesBySource) {
  1749. this.stmts.getEdgesBySource = this.db.prepare('SELECT * FROM edges WHERE source = ?');
  1750. }
  1751. const rows = this.stmts.getEdgesBySource.all(sourceId) as EdgeRow[];
  1752. return rows.map(rowToEdge);
  1753. }
  1754. /**
  1755. * Get incoming edges to a node
  1756. */
  1757. getIncomingEdges(targetId: string, kinds?: EdgeKind[]): Edge[] {
  1758. if (kinds && kinds.length > 0) {
  1759. const sql = `SELECT * FROM edges WHERE target = ? AND kind IN (${kinds.map(() => '?').join(',')})`;
  1760. const rows = this.db.prepare(sql).all(targetId, ...kinds) as EdgeRow[];
  1761. return rows.map(rowToEdge);
  1762. }
  1763. if (!this.stmts.getEdgesByTarget) {
  1764. this.stmts.getEdgesByTarget = this.db.prepare('SELECT * FROM edges WHERE target = ?');
  1765. }
  1766. const rows = this.stmts.getEdgesByTarget.all(targetId) as EdgeRow[];
  1767. return rows.map(rowToEdge);
  1768. }
  1769. /**
  1770. * Outgoing edges for MANY source nodes in one query.
  1771. *
  1772. * The batch form of {@link getOutgoingEdges}. Building a nested outline needs
  1773. * the `contains` edges of every container in a file at once; doing that one
  1774. * source at a time is a query per symbol on files that have hundreds.
  1775. */
  1776. getOutgoingEdgesFrom(sourceIds: readonly string[], kinds?: EdgeKind[]): Edge[] {
  1777. if (sourceIds.length === 0) return [];
  1778. const unique = [...new Set(sourceIds)];
  1779. const out: Edge[] = [];
  1780. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  1781. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  1782. const placeholders = chunk.map(() => '?').join(',');
  1783. let sql = `SELECT * FROM edges WHERE source IN (${placeholders})`;
  1784. const params: string[] = [...chunk];
  1785. if (kinds && kinds.length > 0) {
  1786. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1787. params.push(...kinds);
  1788. }
  1789. const rows = this.db.prepare(sql).all(...params) as EdgeRow[];
  1790. for (const row of rows) out.push(rowToEdge(row));
  1791. }
  1792. return out;
  1793. }
  1794. /**
  1795. * Fan-in (total incoming edge count) for MANY nodes in one query.
  1796. *
  1797. * The per-node alternative — `getIncomingEdges(id).length` — is an indexed
  1798. * lookup each, but a symbol screen rendering a couple of hundred callees
  1799. * would issue a couple of hundred of them. Ids with no incoming edges are
  1800. * absent from the map rather than present as 0, so callers can tell "no
  1801. * edges" from "not asked about".
  1802. */
  1803. countIncomingEdges(ids: readonly string[]): Map<string, number> {
  1804. const out = new Map<string, number>();
  1805. if (ids.length === 0) return out;
  1806. const unique = [...new Set(ids)];
  1807. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  1808. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  1809. const placeholders = chunk.map(() => '?').join(',');
  1810. const rows = this.db
  1811. .prepare(
  1812. `SELECT target, COUNT(*) AS count FROM edges WHERE target IN (${placeholders}) GROUP BY target`
  1813. )
  1814. .all(...chunk) as Array<{ target: string; count: number }>;
  1815. for (const row of rows) out.set(row.target, row.count);
  1816. }
  1817. return out;
  1818. }
  1819. /**
  1820. * Incoming edges for MANY target nodes in one query — the mirror of
  1821. * {@link getOutgoingEdgesFrom}. Needed wherever a whole file's inbound edges
  1822. * are wanted at once ("which files import anything in this one?").
  1823. */
  1824. getIncomingEdgesTo(targetIds: readonly string[], kinds?: EdgeKind[]): Edge[] {
  1825. if (targetIds.length === 0) return [];
  1826. const unique = [...new Set(targetIds)];
  1827. const out: Edge[] = [];
  1828. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  1829. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  1830. const placeholders = chunk.map(() => '?').join(',');
  1831. let sql = `SELECT * FROM edges WHERE target IN (${placeholders})`;
  1832. const params: string[] = [...chunk];
  1833. if (kinds && kinds.length > 0) {
  1834. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  1835. params.push(...kinds);
  1836. }
  1837. const rows = this.db.prepare(sql).all(...params) as EdgeRow[];
  1838. for (const row of rows) out.push(rowToEdge(row));
  1839. }
  1840. return out;
  1841. }
  1842. /**
  1843. * Fan-out (total outgoing edge count) for MANY nodes in one query — the
  1844. * mirror of {@link countIncomingEdges}. Ids with no outgoing edges are absent
  1845. * from the map rather than present as 0.
  1846. */
  1847. countOutgoingEdges(ids: readonly string[]): Map<string, number> {
  1848. const out = new Map<string, number>();
  1849. if (ids.length === 0) return out;
  1850. const unique = [...new Set(ids)];
  1851. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  1852. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  1853. const placeholders = chunk.map(() => '?').join(',');
  1854. const rows = this.db
  1855. .prepare(
  1856. `SELECT source, COUNT(*) AS count FROM edges WHERE source IN (${placeholders}) GROUP BY source`
  1857. )
  1858. .all(...chunk) as Array<{ source: string; count: number }>;
  1859. for (const row of rows) out.set(row.source, row.count);
  1860. }
  1861. return out;
  1862. }
  1863. /**
  1864. * Symbols nothing in the index points at — the candidate set behind the dead
  1865. * code list (`src/graph/dead-code.ts`).
  1866. *
  1867. * "Points at" is every edge kind EXCEPT `contains`: a class containing a
  1868. * method is structure, not use, and counting it would make every member look
  1869. * reached by its own container. A self-edge is excluded for the same reason
  1870. * a recursive function is not its own caller.
  1871. *
  1872. * One scan, one index probe per candidate. `NOT EXISTS` over
  1873. * `idx_edges_target_kind` is what keeps it that way — the alternative
  1874. * (`LEFT JOIN edges … GROUP BY`) builds a row per edge for the whole table
  1875. * before discarding all but the empty groups. Ordered by position so the
  1876. * answer is stable across runs and groups by file without a second sort.
  1877. *
  1878. * The result is deliberately NOT called dead code: an unreferenced symbol is
  1879. * a symbol with no STATIC reference, and the caller applies the exclusions
  1880. * (tests, generated files, overrides, unresolved names) that turn the
  1881. * candidate set into a claim worth making.
  1882. */
  1883. getUnreferencedNodes(
  1884. kinds: readonly string[],
  1885. limit: number
  1886. ): Array<{ node: Node; generated: boolean }> {
  1887. if (kinds.length === 0 || limit <= 0) return [];
  1888. const placeholders = kinds.map(() => '?').join(',');
  1889. const rows = this.db
  1890. .prepare(
  1891. `SELECT n.*, COALESCE(f.generated, 0) AS file_generated
  1892. FROM nodes n
  1893. LEFT JOIN files f ON f.path = n.file_path
  1894. WHERE n.kind IN (${placeholders})
  1895. AND NOT EXISTS (
  1896. SELECT 1 FROM edges e
  1897. WHERE e.target = n.id
  1898. AND e.kind != 'contains'
  1899. AND e.source != n.id
  1900. )
  1901. ORDER BY n.file_path, n.start_line, n.name
  1902. LIMIT ?`
  1903. )
  1904. .all(...kinds, limit) as Array<NodeRow & { file_generated: number }>;
  1905. return rows.map((row) => ({ node: rowToNode(row), generated: row.file_generated === 1 }));
  1906. }
  1907. /**
  1908. * Which of `names` the index holds an UNRESOLVED reference to.
  1909. *
  1910. * The point is honesty about our own blind spots. A `failed` row in
  1911. * `unresolved_refs` records that some file referenced a name and the resolver
  1912. * could not decide what it meant — so a symbol with that name cannot be
  1913. * called unreferenced, whatever the edge table says. It is deliberately
  1914. * matched loosely, on the reference name AND on its tail (`util.greet` →
  1915. * `greet`), because the question being asked is "could this name be the one
  1916. * we failed to follow", and a maybe has to count as a yes.
  1917. *
  1918. * Bounded-lookup like {@link getGeneratedPathsAmong}: the caller holds a
  1919. * candidate list, so this is a chunked probe over `idx_unresolved_name`, not
  1920. * a scan of the table.
  1921. */
  1922. getUnresolvedNamesAmong(names: Iterable<string>): Set<string> {
  1923. const unique = [...new Set(names)].filter((name) => name.length > 0);
  1924. const found = new Set<string>();
  1925. if (unique.length === 0) return found;
  1926. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  1927. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  1928. const placeholders = chunk.map(() => '?').join(',');
  1929. const rows = this.db
  1930. .prepare(
  1931. `SELECT DISTINCT reference_name AS name FROM unresolved_refs
  1932. WHERE reference_name IN (${placeholders})
  1933. UNION
  1934. SELECT DISTINCT name_tail AS name FROM unresolved_refs
  1935. WHERE name_tail IN (${placeholders})`
  1936. )
  1937. .all(...chunk, ...chunk) as Array<{ name: string }>;
  1938. for (const row of rows) found.add(row.name);
  1939. }
  1940. return found;
  1941. }
  1942. /**
  1943. * Which of `names` are carried by MORE THAN ONE symbol, at least one of which
  1944. * something points at.
  1945. *
  1946. * The false positive this exists to kill: `CodeGraph.getTopRouteFile` calls
  1947. * `this.queries.getTopRouteFile()`, and the resolver — which prefers a
  1948. * same-name definition in the call site's own file — attaches that edge to
  1949. * the *calling* method. One of the two ends up with a self-edge and the other
  1950. * with nothing at all, and neither is unreferenced. From the edge table the
  1951. * mis-resolution and a genuinely unused twin are the same picture, so the
  1952. * claim is not made about either.
  1953. *
  1954. * Both halves of the condition are load-bearing. **More than one symbol**:
  1955. * a uniquely-named function that only calls itself is genuinely dead, and
  1956. * excluding every recursive function would gut the list. **Self-edges
  1957. * counted**: the self-edge IS the fingerprint of the mis-resolution above, so
  1958. * it has to count as evidence that this name resolves somewhere.
  1959. *
  1960. * Chunked probe over `idx_nodes_name`, bounded by the caller's candidate list.
  1961. */
  1962. getAmbiguousReferencedNames(names: Iterable<string>): Set<string> {
  1963. const unique = [...new Set(names)].filter((name) => name.length > 0);
  1964. const found = new Set<string>();
  1965. if (unique.length === 0) return found;
  1966. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  1967. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  1968. const placeholders = chunk.map(() => '?').join(',');
  1969. const rows = this.db
  1970. .prepare(
  1971. `SELECT name FROM (
  1972. SELECT n.name AS name,
  1973. EXISTS (
  1974. SELECT 1 FROM edges e
  1975. WHERE e.target = n.id AND e.kind != 'contains'
  1976. ) AS referenced
  1977. FROM nodes n
  1978. WHERE n.name IN (${placeholders})
  1979. )
  1980. GROUP BY name
  1981. HAVING COUNT(*) > 1 AND SUM(referenced) > 0`
  1982. )
  1983. .all(...chunk) as Array<{ name: string }>;
  1984. for (const row of rows) found.add(row.name);
  1985. }
  1986. return found;
  1987. }
  1988. /**
  1989. * Which of the given languages the index records an EXPORT marker for.
  1990. *
  1991. * A self-measurement, and the honest basis for a whole class of exclusion.
  1992. * The dead code report's strongest filter is "exported symbols may be reached
  1993. * from outside this repository" — and that filter silently does nothing for a
  1994. * language whose exports are not recorded, either because the extractor does
  1995. * not record them (Rust `pub`) or because the language has no such concept at
  1996. * all (Python, C, Ruby: the header or the module IS the surface). Rather than
  1997. * carry a table of which is which, ask the index: if nothing in this language
  1998. * is marked exported, the filter did not run, and no claim about outside
  1999. * reachability can be made for it.
  2000. *
  2001. * `idx_nodes_language` covers the grouping; the caller passes the handful of
  2002. * languages its candidates are actually in.
  2003. */
  2004. getLanguagesWithExports(languages: Iterable<string>): Set<string> {
  2005. const unique = [...new Set(languages)].filter((language) => language.length > 0);
  2006. const found = new Set<string>();
  2007. if (unique.length === 0) return found;
  2008. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  2009. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  2010. const placeholders = chunk.map(() => '?').join(',');
  2011. const rows = this.db
  2012. .prepare(
  2013. `SELECT language, MAX(is_exported) AS any_exported
  2014. FROM nodes
  2015. WHERE language IN (${placeholders})
  2016. GROUP BY language`
  2017. )
  2018. .all(...chunk) as Array<{ language: string; any_exported: number }>;
  2019. for (const row of rows) if (row.any_exported === 1) found.add(row.language);
  2020. }
  2021. return found;
  2022. }
  2023. /**
  2024. * The nodes with the most DISTINCT dependents, most first.
  2025. *
  2026. * "Distinct" is the difference that matters: a helper called forty times from
  2027. * one function has a fan-in of 40 but exactly one dependent. This counts the
  2028. * second thing — the number a reader means by "N callers" — so the top of
  2029. * this list is the set of symbols a change actually radiates furthest from.
  2030. *
  2031. * `contains` is excluded because it is structure, not dependency: counting it
  2032. * would rank every file and class above the code they hold.
  2033. */
  2034. getTopDependedOn(limit: number): Array<{ nodeId: string; dependents: number }> {
  2035. if (limit <= 0) return [];
  2036. const rows = this.db
  2037. .prepare(
  2038. `SELECT target AS nodeId, COUNT(DISTINCT source) AS dependents
  2039. FROM edges
  2040. WHERE kind != 'contains' AND source != target
  2041. GROUP BY target
  2042. ORDER BY dependents DESC
  2043. LIMIT ?`
  2044. )
  2045. .all(limit) as Array<{ nodeId: string; dependents: number }>;
  2046. return rows;
  2047. }
  2048. /**
  2049. * The graph's executable roots — files that RUN something at module level,
  2050. * ranked by how much of the project they set in motion.
  2051. *
  2052. * The engine records a statement at the top level of a file as an edge from
  2053. * the *file* node, so `src/bin/codegraph.ts` calling `program.parse()` at
  2054. * module scope is a `calls` edge out of a `file`. That set is what makes the
  2055. * roots of a dependency graph visible: a library module holds definitions and
  2056. * runs nothing until someone imports it, while a CLI, a worker entry or a
  2057. * build script does its work on the way down the file. `instantiates` counts
  2058. * the same way — `new Server(...)` at module scope is the same act.
  2059. *
  2060. * A call made while initializing a module-level `variable` / `constant` —
  2061. * `const service = new Service()`, `app = FastAPI()` — is attributed to the
  2062. * declared name (#693), not to the file, so the file's own edges alone would
  2063. * miss most of what a real entry point runs. Those names are the file's
  2064. * top-level code too, so `tops` counts them alongside the file node.
  2065. *
  2066. * Ranking multiplies the two things an entry point does: it runs (calls), and
  2067. * it wires the project together (distinct other files its symbols reach). One
  2068. * alone is misleading — a registration table makes hundreds of module-level
  2069. * calls into itself, and a barrel file imports everything and runs nothing.
  2070. * The product puts the file that does both at the top.
  2071. */
  2072. getTopCallingFiles(
  2073. limit: number
  2074. ): Array<{ nodeId: string; filePath: string; calls: number; reaches: number; score: number }> {
  2075. if (limit <= 0) return [];
  2076. return this.db
  2077. .prepare(
  2078. `WITH tops AS (
  2079. SELECT n.id AS file_id, n.id AS src
  2080. FROM nodes n
  2081. WHERE n.kind = 'file'
  2082. UNION ALL
  2083. SELECT c.source AS file_id, c.target AS src
  2084. FROM edges c
  2085. JOIN nodes f ON f.id = c.source
  2086. JOIN nodes v ON v.id = c.target
  2087. WHERE c.kind = 'contains'
  2088. AND f.kind = 'file'
  2089. AND v.kind IN ('variable', 'constant')
  2090. ),
  2091. runs AS (
  2092. SELECT t.file_id AS id, COUNT(*) AS calls
  2093. FROM tops t
  2094. JOIN edges e ON e.source = t.src
  2095. WHERE e.kind IN ('calls', 'instantiates')
  2096. GROUP BY t.file_id
  2097. ),
  2098. cand AS (
  2099. SELECT r.id AS id, n.file_path AS fp, r.calls AS calls
  2100. FROM runs r JOIN nodes n ON n.id = r.id
  2101. ),
  2102. wires AS (
  2103. SELECT sn.file_path AS fp, COUNT(DISTINCT tn.file_path) AS reaches
  2104. FROM edges e
  2105. JOIN nodes sn ON sn.id = e.source
  2106. JOIN nodes tn ON tn.id = e.target
  2107. WHERE e.kind != 'contains'
  2108. AND sn.file_path <> tn.file_path
  2109. AND sn.file_path IN (SELECT fp FROM cand)
  2110. GROUP BY sn.file_path
  2111. )
  2112. SELECT c.id AS nodeId,
  2113. c.fp AS filePath,
  2114. c.calls AS calls,
  2115. COALESCE(w.reaches, 0) AS reaches,
  2116. c.calls * (1 + COALESCE(w.reaches, 0)) AS score
  2117. FROM cand c LEFT JOIN wires w ON w.fp = c.fp
  2118. ORDER BY score DESC, calls DESC, filePath
  2119. LIMIT ?`
  2120. )
  2121. .all(limit) as Array<{
  2122. nodeId: string;
  2123. filePath: string;
  2124. calls: number;
  2125. reaches: number;
  2126. score: number;
  2127. }>;
  2128. }
  2129. /**
  2130. * How many OTHER files depend on each of the given files.
  2131. *
  2132. * Counted through the symbols, not the file nodes: an `imports` edge points
  2133. * at the imported symbol, so a file node almost never receives one and
  2134. * counting edges into it would report every file as depended on by nobody.
  2135. * Same-file edges are excluded, which is what makes zero mean "nothing else
  2136. * in the index reaches into this file" — the honest reading of a root.
  2137. */
  2138. getFileDependentCounts(filePaths: string[]): Array<{ filePath: string; dependents: number }> {
  2139. if (filePaths.length === 0) return [];
  2140. return this.db
  2141. .prepare(
  2142. `SELECT tn.file_path AS filePath, COUNT(DISTINCT sn.file_path) AS dependents
  2143. FROM edges e
  2144. JOIN nodes tn ON tn.id = e.target
  2145. JOIN nodes sn ON sn.id = e.source
  2146. WHERE e.kind != 'contains'
  2147. AND tn.file_path IN (SELECT value FROM json_each(?))
  2148. AND sn.file_path <> tn.file_path
  2149. GROUP BY tn.file_path`
  2150. )
  2151. .all(JSON.stringify(filePaths)) as Array<{ filePath: string; dependents: number }>;
  2152. }
  2153. /**
  2154. * How far each of the given files reaches OUT: distinct other files its
  2155. * symbols touch, and how many references that is.
  2156. *
  2157. * The mirror of {@link getFileDependentCounts}, and the same reasoning about
  2158. * `contains` and same-file edges applies. It is driven from `nodes` rather
  2159. * than from `edges` so the work is proportional to the files asked about —
  2160. * the entry-points endpoint asks it about every test file in the index, and
  2161. * an edge-first plan would scan the whole table to answer a question about a
  2162. * tenth of it.
  2163. */
  2164. getFileReachCounts(filePaths: string[]): Array<{ filePath: string; reaches: number; refs: number }> {
  2165. if (filePaths.length === 0) return [];
  2166. return this.db
  2167. .prepare(
  2168. `SELECT sn.file_path AS filePath,
  2169. COUNT(DISTINCT tn.file_path) AS reaches,
  2170. COUNT(*) AS refs
  2171. FROM nodes sn
  2172. JOIN edges e ON e.source = sn.id
  2173. JOIN nodes tn ON tn.id = e.target
  2174. WHERE sn.file_path IN (SELECT value FROM json_each(?))
  2175. AND e.kind != 'contains'
  2176. AND tn.file_path <> sn.file_path
  2177. GROUP BY sn.file_path`
  2178. )
  2179. .all(JSON.stringify(filePaths)) as Array<{
  2180. filePath: string;
  2181. reaches: number;
  2182. refs: number;
  2183. }>;
  2184. }
  2185. /**
  2186. * The `file` nodes for the given paths, in one query.
  2187. *
  2188. * A file's own node is what makes a file row navigable, and looking it up
  2189. * with {@link getNodesInFile} means materialising every symbol in the file to
  2190. * throw all but one away.
  2191. */
  2192. getFileNodes(filePaths: string[]): Node[] {
  2193. if (filePaths.length === 0) return [];
  2194. const rows = this.db
  2195. .prepare(
  2196. `SELECT * FROM nodes
  2197. WHERE kind = 'file'
  2198. AND file_path IN (SELECT value FROM json_each(?))`
  2199. )
  2200. .all(JSON.stringify(filePaths)) as NodeRow[];
  2201. return rows.map(rowToNode);
  2202. }
  2203. /**
  2204. * Roll the whole edge table up to module granularity in one pass.
  2205. *
  2206. * The caller decides what a module IS — it hands in a file → module
  2207. * assignment and gets back the cross-module traffic. That split is
  2208. * deliberate: naming modules is a *policy* (top-level directories, a façade
  2209. * file kept separate, a monorepo root) that belongs where the reader lives,
  2210. * while grouping a million edges by it is *mechanics* that must happen in
  2211. * SQLite. Doing the fold in JavaScript instead means materialising every
  2212. * cross-file edge in memory; doing the naming in SQL means a tower of
  2213. * `instr`/`substr` no one can read.
  2214. *
  2215. * The assignment lands in a TEMP table with a primary key, so the join is
  2216. * indexed and the result set is bounded by modules², not by edges. Temp
  2217. * tables live in SQLite's own temp database, so this stays valid against a
  2218. * read-only main.
  2219. *
  2220. * Two result sets, because they need two different groupings over the same
  2221. * join: `links` counts edges per (module, module, kind), and `pairs` names
  2222. * the busiest symbol pairs behind each link (the map's tooltip). `pairs` is
  2223. * ranked and cut inside SQLite — the un-cut grouping is the one thing here
  2224. * that scales with distinct symbol names rather than with modules. Pairs are
  2225. * ranked by `declared` before raw count, so a link's tooltip names the
  2226. * symbols the source actually points at rather than whichever `has`/`get`
  2227. * happened to name-match most often.
  2228. *
  2229. * `declared` is the subset of a link's edges that came from something the
  2230. * source *writes down*: an import, a qualified name, an inheritance clause,
  2231. * or a call through a typed receiver. It exists because bare name matching
  2232. * (`resolvedBy: 'exact-match'`) is what invents cross-module links out of
  2233. * common method names — `run`, `push`, `finish` — and a map that lets those
  2234. * decide the layering puts the storage layer above the CLI.
  2235. */
  2236. aggregateModuleGraph(
  2237. assignments: ReadonlyArray<{ filePath: string; module: string }>,
  2238. options: {
  2239. kinds: readonly EdgeKind[];
  2240. minConfidence: number;
  2241. topPairsPerLink: number;
  2242. pairKinds: readonly EdgeKind[];
  2243. }
  2244. ): {
  2245. links: Array<{
  2246. source: string;
  2247. target: string;
  2248. kind: EdgeKind;
  2249. count: number;
  2250. declared: number;
  2251. uncertain: number;
  2252. }>;
  2253. pairs: Array<{
  2254. source: string;
  2255. target: string;
  2256. from: string;
  2257. to: string;
  2258. count: number;
  2259. declared: number;
  2260. }>;
  2261. } {
  2262. if (assignments.length === 0 || options.kinds.length === 0) return { links: [], pairs: [] };
  2263. const CONFIDENCE = `COALESCE(json_extract(e.metadata, '$.confidence'), 1)`;
  2264. const DECLARED = `(json_extract(e.metadata, '$.resolvedBy') IN ('import', 'qualified-name')
  2265. OR e.kind IN ('extends', 'implements')
  2266. OR (json_extract(e.metadata, '$.resolvedBy') = 'instance-method'
  2267. AND ${CONFIDENCE} >= 0.9))`;
  2268. this.db.exec('DROP TABLE IF EXISTS temp.cg_module_map');
  2269. this.db.exec('CREATE TEMP TABLE cg_module_map (path TEXT PRIMARY KEY, mod TEXT NOT NULL)');
  2270. try {
  2271. const insert = this.db.prepare(
  2272. 'INSERT OR REPLACE INTO cg_module_map (path, mod) VALUES (?, ?)'
  2273. );
  2274. this.db.exec('BEGIN');
  2275. try {
  2276. for (const row of assignments) insert.run(row.filePath, row.module);
  2277. this.db.exec('COMMIT');
  2278. } catch (err) {
  2279. this.db.exec('ROLLBACK');
  2280. throw err;
  2281. }
  2282. // ONE pass over the edge table. Grouping by the symbol names as well as
  2283. // the modules costs nothing extra in scan time — the join is what is
  2284. // expensive — and it buys both results from a single scan. Measured on
  2285. // this index inflated to 1.6M edges: 1.66s for this query against 3.0s
  2286. // for the module-level and name-level queries run separately, which is
  2287. // the difference between meeting and missing the map's cold budget on a
  2288. // ten-thousand-file repository.
  2289. const rows = this.db
  2290. .prepare(
  2291. `SELECT ms.mod AS source, mt.mod AS target, e.kind AS kind,
  2292. sn.name AS "from", tn.name AS "to",
  2293. SUM(CASE WHEN ${CONFIDENCE} >= ? THEN 1 ELSE 0 END) AS count,
  2294. SUM(CASE WHEN ${CONFIDENCE} >= ? AND ${DECLARED} THEN 1 ELSE 0 END) AS declared,
  2295. SUM(CASE WHEN ${CONFIDENCE} < ? THEN 1 ELSE 0 END) AS uncertain
  2296. FROM edges e
  2297. JOIN nodes sn ON sn.id = e.source
  2298. JOIN nodes tn ON tn.id = e.target
  2299. JOIN cg_module_map ms ON ms.path = sn.file_path
  2300. JOIN cg_module_map mt ON mt.path = tn.file_path
  2301. WHERE e.kind IN (SELECT value FROM json_each(?))
  2302. AND ms.mod <> mt.mod
  2303. GROUP BY ms.mod, mt.mod, e.kind, sn.name, tn.name`
  2304. )
  2305. .all(
  2306. options.minConfidence,
  2307. options.minConfidence,
  2308. options.minConfidence,
  2309. JSON.stringify(options.kinds)
  2310. ) as Array<{
  2311. source: string;
  2312. target: string;
  2313. kind: EdgeKind;
  2314. from: string;
  2315. to: string;
  2316. count: number;
  2317. declared: number;
  2318. uncertain: number;
  2319. }>;
  2320. return foldModuleRows(rows, options);
  2321. } finally {
  2322. this.db.exec('DROP TABLE IF EXISTS temp.cg_module_map');
  2323. }
  2324. }
  2325. /**
  2326. * Every ordered pair of files where one reaches into the other, once each.
  2327. *
  2328. * The input a cycle finder wants: file-level circular dependencies are the
  2329. * strongly connected components of this graph. One query instead of the
  2330. * dependency lookup per file that {@link GraphQueryManager.findCircularDependencies}
  2331. * does — which matters because a cycle report is only interesting on a large
  2332. * repo, and that is exactly where a query per file stops being affordable.
  2333. *
  2334. * `contains` is excluded (a file "contains" its own symbols, which is not a
  2335. * dependency), and so are same-file edges and low-confidence name matches:
  2336. * a cycle conjured by a common method name is a false alarm a reader cannot
  2337. * check.
  2338. */
  2339. getCrossFileDependencyPairs(minConfidence: number): Array<{ source: string; target: string }> {
  2340. return this.db
  2341. .prepare(
  2342. `SELECT DISTINCT sn.file_path AS source, tn.file_path AS target
  2343. FROM edges e
  2344. JOIN nodes sn ON sn.id = e.source
  2345. JOIN nodes tn ON tn.id = e.target
  2346. WHERE e.kind <> 'contains'
  2347. AND sn.file_path <> tn.file_path
  2348. AND COALESCE(json_extract(e.metadata, '$.confidence'), 1) >= ?`
  2349. )
  2350. .all(minConfidence) as Array<{ source: string; target: string }>;
  2351. }
  2352. /**
  2353. * Every unresolved reference recorded in one FILE, ordered by line.
  2354. *
  2355. * The per-symbol form above answers "what does this body reach that the
  2356. * index does not hold". A whole-file reader asks the same question of every
  2357. * line at once, and asking it one symbol at a time is a query per symbol —
  2358. * 153 of them on this repo's largest file. `unresolved_refs.file_path` is
  2359. * indexed, so this is one lookup whatever the file holds.
  2360. *
  2361. * `limit` bounds the answer rather than the work: the caller draws a marker
  2362. * per row, and a generated file with fifty thousand of them would ship
  2363. * megabytes to say something a count already says. Rows come back in line
  2364. * order, so a cap trims the END of the file, which is at least legible.
  2365. */
  2366. getUnresolvedReferencesInFile(filePath: string, limit = 5000): UnresolvedReference[] {
  2367. if (!this.stmts.getUnresolvedInFile) {
  2368. this.stmts.getUnresolvedInFile = this.db.prepare(
  2369. 'SELECT * FROM unresolved_refs WHERE file_path = ? ORDER BY line, col LIMIT ?'
  2370. );
  2371. }
  2372. const rows = this.stmts.getUnresolvedInFile.all(filePath, limit) as UnresolvedRefRow[];
  2373. return rows.map((row) => ({
  2374. fromNodeId: row.from_node_id,
  2375. referenceName: row.reference_name,
  2376. referenceKind: row.reference_kind as EdgeKind,
  2377. line: row.line,
  2378. column: row.col,
  2379. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  2380. filePath: row.file_path,
  2381. language: row.language as Language,
  2382. rowId: row.id,
  2383. }));
  2384. }
  2385. /**
  2386. * References recorded against a symbol that never resolved to a node — the
  2387. * calls and type mentions that leave the index (a third-party package, a
  2388. * runtime builtin, a language construct extraction doesn't model).
  2389. *
  2390. * Read-only. It exists so a reader can say "N calls into symbols outside the
  2391. * index" instead of silently showing a callee list shorter than the body's
  2392. * call sites, which reads as "nothing else happens here".
  2393. */
  2394. getUnresolvedReferencesFrom(fromNodeId: string): UnresolvedReference[] {
  2395. if (!this.stmts.getUnresolvedFromNode) {
  2396. this.stmts.getUnresolvedFromNode = this.db.prepare(
  2397. 'SELECT * FROM unresolved_refs WHERE from_node_id = ?'
  2398. );
  2399. }
  2400. const rows = this.stmts.getUnresolvedFromNode.all(fromNodeId) as UnresolvedRefRow[];
  2401. return rows.map((row) => ({
  2402. fromNodeId: row.from_node_id,
  2403. referenceName: row.reference_name,
  2404. referenceKind: row.reference_kind as EdgeKind,
  2405. line: row.line,
  2406. column: row.col,
  2407. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  2408. filePath: row.file_path,
  2409. language: row.language as Language,
  2410. rowId: row.id,
  2411. }));
  2412. }
  2413. /**
  2414. * Find all edges where both source and target are in the given node set.
  2415. * Useful for recovering inter-node connectivity after BFS.
  2416. */
  2417. findEdgesBetweenNodes(nodeIds: string[], kinds?: EdgeKind[]): Edge[] {
  2418. if (nodeIds.length === 0) return [];
  2419. const idsJson = JSON.stringify(nodeIds);
  2420. let sql = `SELECT * FROM edges WHERE source IN (SELECT value FROM json_each(?)) AND target IN (SELECT value FROM json_each(?))`;
  2421. const params: string[] = [idsJson, idsJson];
  2422. if (kinds && kinds.length > 0) {
  2423. sql += ` AND kind IN (${kinds.map(() => '?').join(',')})`;
  2424. params.push(...kinds);
  2425. }
  2426. const rows = this.db.prepare(sql).all(...params) as EdgeRow[];
  2427. return rows.map(rowToEdge);
  2428. }
  2429. /**
  2430. * Distinct file paths that DEPEND ON `filePath`: every file containing a
  2431. * symbol with a cross-file edge (any kind except `contains`) into a symbol
  2432. * of this file. This is the file-level projection of the symbol dependency
  2433. * graph and the basis for blast-radius / `affected` test selection.
  2434. *
  2435. * It deliberately does NOT restrict to `imports` edges. In this graph an
  2436. * `imports` edge connects a file to its own local import declarations
  2437. * (it is always same-file), so an imports-only lookup returns zero
  2438. * cross-file dependents for every file. The real cross-file dependency
  2439. * signal is the resolved call/reference graph — calls, references,
  2440. * instantiates, extends, implements, overrides, type_of, returns,
  2441. * decorates — exactly what {@link GraphTraverser.getImpactRadius} traverses.
  2442. * `contains` is excluded: a parent containing a symbol does not *depend* on
  2443. * it. One indexed query (idx_nodes_file_path + idx_edges_target_kind).
  2444. */
  2445. getDependentFilePaths(filePath: string): string[] {
  2446. const sql = `SELECT DISTINCT src.file_path AS fp
  2447. FROM edges e
  2448. JOIN nodes tgt ON tgt.id = e.target
  2449. JOIN nodes src ON src.id = e.source
  2450. WHERE tgt.file_path = ?
  2451. AND e.kind != 'contains'
  2452. AND src.file_path != ?`;
  2453. const rows = this.db.prepare(sql).all(filePath, filePath) as Array<{ fp: string }>;
  2454. return rows.map((r) => r.fp);
  2455. }
  2456. /**
  2457. * Distinct file paths that `filePath` DEPENDS ON — the inverse of
  2458. * {@link getDependentFilePaths}: every file containing a symbol that a
  2459. * symbol of this file has a cross-file edge into. Same edge-kind rules
  2460. * (all kinds except `contains`); same reason imports-only is insufficient.
  2461. */
  2462. getDependencyFilePaths(filePath: string): string[] {
  2463. const sql = `SELECT DISTINCT tgt.file_path AS fp
  2464. FROM edges e
  2465. JOIN nodes src ON src.id = e.source
  2466. JOIN nodes tgt ON tgt.id = e.target
  2467. WHERE src.file_path = ?
  2468. AND e.kind != 'contains'
  2469. AND tgt.file_path != ?`;
  2470. const rows = this.db.prepare(sql).all(filePath, filePath) as Array<{ fp: string }>;
  2471. return rows.map((r) => r.fp);
  2472. }
  2473. /**
  2474. * Cross-file edges whose TARGET is a node in `filePath` and whose SOURCE is a
  2475. * node in a *different* file, paired with the target node's (name, kind) so a
  2476. * caller can re-resolve the edge to the re-indexed target's new ID (node IDs
  2477. * are `sha256(filePath:kind:name:line)`, so any line shift in the callee file
  2478. * changes target IDs and a naive re-insert by old ID silently drops them).
  2479. * Used by `storeExtractionResult` to preserve incoming edges across a file
  2480. * re-index (issue #899). Same edge-kind rules as
  2481. * {@link getDependentFilePaths}: all kinds except `contains`.
  2482. */
  2483. getCrossFileIncomingEdgesWithTarget(
  2484. filePath: string
  2485. ): Array<Edge & { targetName: string; targetKind: NodeKind; sourceFilePath: string; sourceLanguage: Language }> {
  2486. const sql = `SELECT e.*, tgt.name AS target_name, tgt.kind AS target_kind,
  2487. src.file_path AS source_file_path, src.language AS source_language
  2488. FROM edges e
  2489. JOIN nodes tgt ON tgt.id = e.target
  2490. JOIN nodes src ON src.id = e.source
  2491. WHERE tgt.file_path = ?
  2492. AND e.kind != 'contains'
  2493. AND src.file_path != ?`;
  2494. const rows = this.db.prepare(sql).all(filePath, filePath) as Array<
  2495. EdgeRow & { target_name: string; target_kind: NodeKind; source_file_path: string; source_language: Language }
  2496. >;
  2497. return rows.map(row => ({
  2498. ...rowToEdge(row),
  2499. targetName: row.target_name,
  2500. targetKind: row.target_kind,
  2501. sourceFilePath: row.source_file_path,
  2502. sourceLanguage: row.source_language,
  2503. }));
  2504. }
  2505. // ===========================================================================
  2506. // File Operations
  2507. // ===========================================================================
  2508. /**
  2509. * Insert or update a file record
  2510. */
  2511. upsertFile(file: FileRecord): void {
  2512. if (!this.stmts.upsertFile) {
  2513. this.stmts.upsertFile = this.db.prepare(`
  2514. INSERT INTO files (path, content_hash, language, size, modified_at, indexed_at, node_count, errors, generated)
  2515. VALUES (@path, @contentHash, @language, @size, @modifiedAt, @indexedAt, @nodeCount, @errors, @generated)
  2516. ON CONFLICT(path) DO UPDATE SET
  2517. content_hash = @contentHash,
  2518. language = @language,
  2519. size = @size,
  2520. modified_at = @modifiedAt,
  2521. indexed_at = @indexedAt,
  2522. node_count = @nodeCount,
  2523. errors = @errors,
  2524. generated = @generated
  2525. `);
  2526. }
  2527. this.stmts.upsertFile.run({
  2528. path: file.path,
  2529. contentHash: file.contentHash,
  2530. language: file.language,
  2531. size: file.size,
  2532. modifiedAt: file.modifiedAt,
  2533. indexedAt: file.indexedAt,
  2534. nodeCount: file.nodeCount,
  2535. errors: file.errors ? JSON.stringify(file.errors) : null,
  2536. // The upsert always REWRITES the flag: a file that loses its banner in an
  2537. // edit must lose the flag on the next sync, not keep a stale 1.
  2538. generated: file.generated ? 1 : 0,
  2539. });
  2540. }
  2541. /**
  2542. * Which of `filePaths` the index flagged as tool-generated (schema v9+).
  2543. *
  2544. * Bounded-lookup by design: every consumer already holds a short candidate
  2545. * list (a ranked file group, an FTS result page, a LIMIT-20 aggregate), so
  2546. * this stays a partial-index probe over a handful of paths — no whole-repo
  2547. * set to materialize, and no cache to invalidate, which means a ranking call
  2548. * can never serve a verdict the last sync already replaced.
  2549. *
  2550. * Returns ONLY the content/index signal; callers union it with
  2551. * {@link isGeneratedFile} so pre-v9 databases (column present, all zeros
  2552. * until a re-index) keep the path-only behavior rather than regressing.
  2553. */
  2554. getGeneratedPathsAmong(filePaths: Iterable<string>): Set<string> {
  2555. const unique = [...new Set(filePaths)];
  2556. const found = new Set<string>();
  2557. if (unique.length === 0) return found;
  2558. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  2559. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  2560. const placeholders = chunk.map(() => '?').join(',');
  2561. const rows = this.db
  2562. .prepare(`SELECT path FROM files WHERE generated = 1 AND path IN (${placeholders})`)
  2563. .all(...chunk) as Array<{ path: string }>;
  2564. for (const row of rows) found.add(row.path);
  2565. }
  2566. return found;
  2567. }
  2568. /**
  2569. * A reusable `(path) => boolean` over a bounded candidate list, unioning the
  2570. * indexed flag with the path convention. This is the shape every ranking
  2571. * comparator wants: one query up front, then O(1) per comparison.
  2572. */
  2573. generatedPredicateFor(filePaths: Iterable<string>): (filePath: string) => boolean {
  2574. const flagged = this.getGeneratedPathsAmong(filePaths);
  2575. return (filePath: string) => flagged.has(filePath) || isGeneratedFile(filePath);
  2576. }
  2577. /**
  2578. * Which of `filePaths` are AMBIENT DECLARATION files — they declare nothing
  2579. * but types, and nothing in the index depends on them (CG-28). A hand-written
  2580. * ambient `.d.ts` of global shims, a vendored typings file, module
  2581. * augmentation: reachable only by name, structurally attached to nothing.
  2582. *
  2583. * Structural, not extension-based, so a hand-written `types.ts` and a `.d.ts`
  2584. * are judged by the same rule and a `.d.ts` that does declare a class or a
  2585. * const is (correctly) not caught. Four conditions, all required:
  2586. *
  2587. * 1. it declares at least one symbol — an empty or unparsed file is not a
  2588. * declaration file, it is a file we know nothing about;
  2589. * 2. EVERY declared symbol is a type-level kind (interface / type alias /
  2590. * enum / namespace). The narrowness is deliberate and measured: a rule
  2591. * of "no callables" alone flags 1–18% of a repo, including Kotlin sealed
  2592. * classes, Rust `mod.rs` re-exports and django's locale constant tables —
  2593. * real source that must not be demoted. This rule flags 0–4%;
  2594. * 3. no symbol in it originates a `calls`/`instantiates` edge — the direct
  2595. * evidence that nothing here has a body;
  2596. * 4. NOTHING ELSE IN THE INDEX points at it. This is the condition that
  2597. * separates an ambient shim from a working type module, and it is why
  2598. * the flag is narrow enough to be safe: `displacement-ts`'s pipeline
  2599. * `types.ts` passes 1–3 identically but carries 13 inbound imports and
  2600. * 21 references, so the files that answer a query about the pipeline are
  2601. * typed BY it — it is part of that answer's structure. An ambient
  2602. * `declare global` shim has zero. Deliberately index-wide rather than
  2603. * restricted to the candidate list: the file that imports it is usually
  2604. * not itself a candidate.
  2605. *
  2606. * ### Interface MEMBERS are transparent to all four conditions
  2607. *
  2608. * A `method_signature` / `property_signature` inside an interface enters the
  2609. * graph as a `method` / `property` node (#1638). Read literally that would
  2610. * break every condition here at once: condition 2 sees non-type kinds and
  2611. * stops flagging, and — worse, because it is silent — condition 4 starts
  2612. * seeing inbound `calls` edges the moment a call site through the shim's API
  2613. * finally has a signature to land on. An ambient `.d.ts` would quietly lose
  2614. * its damping precisely BECAUSE the platform API it declares is widely used.
  2615. *
  2616. * So an interface-owned member is treated the way `parameter` already is: it
  2617. * neither qualifies, disqualifies, nor counts as inbound dependency. That is
  2618. * not a new judgement call, it is what keeps the rule measuring what it was
  2619. * measured on — before #1638 these nodes did not exist, so excluding them
  2620. * reproduces the 0–4% flag rate the thresholds above were tuned against. It
  2621. * is also the semantically right answer: a signature with no body is on the
  2622. * same side of the line as the interface that owns it, and a call edge
  2623. * landing on one is still not a file that can answer a flow question.
  2624. *
  2625. * The interface ITSELF is untouched: the `references` edges an importing
  2626. * module aims at `UploadStorage` still disqualify the file under (4), which
  2627. * is what keeps a depended-on `types.ts` out of the flag.
  2628. *
  2629. * Bounded-lookup like {@link getGeneratedPathsAmong}: callers hold a ranked
  2630. * candidate list, so this is a partial-index probe over a handful of paths.
  2631. */
  2632. getAmbientDeclarationPathsAmong(filePaths: Iterable<string>): Set<string> {
  2633. const unique = [...new Set(filePaths)];
  2634. const found = new Set<string>();
  2635. if (unique.length === 0) return found;
  2636. for (let i = 0; i < unique.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  2637. const chunk = unique.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  2638. const placeholders = chunk.map(() => '?').join(',');
  2639. // `file`/`import`/`export`/`parameter` are structural bookkeeping, not
  2640. // things the file declares, so they neither qualify nor disqualify.
  2641. const rows = this.db
  2642. .prepare(`
  2643. SELECT n.file_path AS file_path,
  2644. SUM(CASE WHEN n.kind NOT IN ('file','import','export','parameter')
  2645. AND NOT ${IS_INTERFACE_MEMBER('n')}
  2646. THEN 1 ELSE 0 END) AS declared,
  2647. SUM(CASE WHEN n.kind IN ('interface','type_alias','enum','enum_member','namespace')
  2648. THEN 1 ELSE 0 END) AS typeDeclared
  2649. FROM nodes n
  2650. WHERE n.file_path IN (${placeholders})
  2651. GROUP BY n.file_path
  2652. `)
  2653. .all(...chunk) as Array<{ file_path: string; declared: number; typeDeclared: number }>;
  2654. let candidates = rows
  2655. .filter((r) => r.declared > 0 && r.declared === r.typeDeclared)
  2656. .map((r) => r.file_path);
  2657. if (candidates.length === 0) continue;
  2658. const disqualify = (sql: string): void => {
  2659. if (candidates.length === 0) return;
  2660. const hit = new Set(
  2661. (this.db
  2662. .prepare(sql.replace('$IN$', candidates.map(() => '?').join(',')))
  2663. .all(...candidates) as Array<{ file_path: string }>).map((r) => r.file_path),
  2664. );
  2665. candidates = candidates.filter((p) => !hit.has(p));
  2666. };
  2667. // (3) originates behaviour — a signature has no body to originate from,
  2668. // so an edge attributed to one is not evidence about this file.
  2669. disqualify(`
  2670. SELECT DISTINCT n.file_path AS file_path
  2671. FROM edges e JOIN nodes n ON n.id = e.source
  2672. WHERE e.kind IN ('calls','instantiates') AND n.file_path IN ($IN$)
  2673. AND NOT ${IS_INTERFACE_MEMBER('n')}
  2674. `);
  2675. // (4) something outside the file depends on it — but a call that lands on
  2676. // an interface's own signature is a use of the API, not a dependency on
  2677. // this file's structure. The edges aimed at the interface still count.
  2678. disqualify(`
  2679. SELECT DISTINCT t.file_path AS file_path
  2680. FROM edges e JOIN nodes t ON t.id = e.target JOIN nodes s ON s.id = e.source
  2681. WHERE t.file_path IN ($IN$) AND s.file_path <> t.file_path
  2682. AND NOT ${IS_INTERFACE_MEMBER('t')}
  2683. `);
  2684. for (const path of candidates) found.add(path);
  2685. }
  2686. return found;
  2687. }
  2688. /**
  2689. * A reusable `(path) => boolean` ambient-declaration test over a bounded
  2690. * candidate list — the shape a ranking comparator wants: one query up front,
  2691. * O(1) per comparison.
  2692. */
  2693. ambientDeclarationPredicateFor(filePaths: Iterable<string>): (filePath: string) => boolean {
  2694. const flagged = this.getAmbientDeclarationPathsAmong(filePaths);
  2695. return (filePath: string) => flagged.has(filePath);
  2696. }
  2697. /** How many indexed files carry the generated flag. Surfaced by `status`. */
  2698. countGeneratedFiles(): number {
  2699. const row = this.db
  2700. .prepare('SELECT COUNT(*) AS n FROM files WHERE generated = 1')
  2701. .get() as { n: number } | undefined;
  2702. return row?.n ?? 0;
  2703. }
  2704. /**
  2705. * Delete a file record and its nodes
  2706. */
  2707. deleteFile(filePath: string): void {
  2708. this.db.transaction(() => {
  2709. this.deleteNodesByFile(filePath);
  2710. if (!this.stmts.deleteFile) {
  2711. this.stmts.deleteFile = this.db.prepare('DELETE FROM files WHERE path = ?');
  2712. }
  2713. this.stmts.deleteFile.run(filePath);
  2714. })();
  2715. }
  2716. /**
  2717. * Get a file record by path
  2718. */
  2719. getFileByPath(filePath: string): FileRecord | null {
  2720. if (!this.stmts.getFileByPath) {
  2721. this.stmts.getFileByPath = this.db.prepare('SELECT * FROM files WHERE path = ?');
  2722. }
  2723. const row = this.stmts.getFileByPath.get(filePath) as FileRow | undefined;
  2724. return row ? rowToFileRecord(row) : null;
  2725. }
  2726. /**
  2727. * Get all tracked files
  2728. */
  2729. getAllFiles(): FileRecord[] {
  2730. if (!this.stmts.getAllFiles) {
  2731. this.stmts.getAllFiles = this.db.prepare('SELECT * FROM files ORDER BY path');
  2732. }
  2733. const rows = this.stmts.getAllFiles.all() as FileRow[];
  2734. return rows.map(rowToFileRecord);
  2735. }
  2736. /**
  2737. * Most recent index timestamp (ms since epoch) across all tracked files, or
  2738. * null when nothing is indexed yet. One indexed aggregate, no per-row scan. (#329)
  2739. */
  2740. getLastIndexedAt(): number | null {
  2741. const row = this.db
  2742. .prepare('SELECT MAX(indexed_at) AS last FROM files')
  2743. .get() as { last: number | null } | undefined;
  2744. return row?.last ?? null;
  2745. }
  2746. /**
  2747. * The index's revision marker: how far the last sync got, and how many files
  2748. * it left behind — one query, both numbers.
  2749. *
  2750. * This is the cheapest honest answer to "has the index moved since I last
  2751. * looked". `MAX(indexed_at)` alone is not enough: a sync that only DELETES
  2752. * files (a branch checkout that removed a directory) advances nothing, and
  2753. * the graph the viewer is showing has still changed underneath it. The row
  2754. * count catches exactly that case.
  2755. */
  2756. getIndexRevision(): { lastIndexedAt: number | null; fileCount: number } {
  2757. const row = this.db
  2758. .prepare('SELECT MAX(indexed_at) AS last, COUNT(*) AS files FROM files')
  2759. .get() as { last: number | null; files: number } | undefined;
  2760. return { lastIndexedAt: row?.last ?? null, fileCount: row?.files ?? 0 };
  2761. }
  2762. /**
  2763. * Files re-indexed strictly after `since` (ms since epoch), newest first.
  2764. *
  2765. * `total` is the real count; `paths` is capped at `limit`. Used by the
  2766. * viewer's live channel to name what a sync just picked up. A file the same
  2767. * sync DELETED cannot appear here — it has no row left — which is why the
  2768. * caller compares {@link getIndexRevision} as well rather than treating an
  2769. * empty list as "nothing happened".
  2770. */
  2771. getFilesIndexedSince(since: number, limit: number): { paths: string[]; total: number } {
  2772. const count = this.db
  2773. .prepare('SELECT COUNT(*) AS n FROM files WHERE indexed_at > ?')
  2774. .get(since) as { n: number } | undefined;
  2775. const rows = this.db
  2776. .prepare('SELECT path FROM files WHERE indexed_at > ? ORDER BY indexed_at DESC, path LIMIT ?')
  2777. .all(since, Math.max(0, limit)) as Array<{ path: string }>;
  2778. return { paths: rows.map((r) => r.path), total: count?.n ?? rows.length };
  2779. }
  2780. /**
  2781. * Get files that need re-indexing (hash changed)
  2782. */
  2783. getStaleFiles(currentHashes: Map<string, string>): FileRecord[] {
  2784. const files = this.getAllFiles();
  2785. return files.filter((f) => {
  2786. const currentHash = currentHashes.get(f.path);
  2787. return currentHash && currentHash !== f.contentHash;
  2788. });
  2789. }
  2790. // ===========================================================================
  2791. // Unresolved References
  2792. // ===========================================================================
  2793. /**
  2794. * Insert an unresolved reference
  2795. */
  2796. insertUnresolvedRef(ref: UnresolvedReference): void {
  2797. if (!this.stmts.insertUnresolved) {
  2798. this.stmts.insertUnresolved = this.db.prepare(`
  2799. INSERT INTO unresolved_refs (from_node_id, reference_name, reference_kind, line, col, candidates, file_path, language)
  2800. VALUES (@fromNodeId, @referenceName, @referenceKind, @line, @col, @candidates, @filePath, @language)
  2801. `);
  2802. }
  2803. this.stmts.insertUnresolved.run({
  2804. fromNodeId: ref.fromNodeId,
  2805. referenceName: ref.referenceName,
  2806. referenceKind: ref.referenceKind,
  2807. line: ref.line,
  2808. col: ref.column,
  2809. candidates: ref.candidates ? JSON.stringify(ref.candidates) : null,
  2810. filePath: ref.filePath ?? '',
  2811. language: ref.language ?? 'unknown',
  2812. });
  2813. }
  2814. /**
  2815. * Insert multiple unresolved references in a transaction
  2816. */
  2817. insertUnresolvedRefsBatch(refs: UnresolvedReference[]): void {
  2818. if (refs.length === 0) return;
  2819. const insert = this.db.transaction(() => {
  2820. const rows: unknown[][] = [];
  2821. for (const ref of refs) {
  2822. rows.push([
  2823. ref.fromNodeId,
  2824. ref.referenceName,
  2825. ref.referenceKind,
  2826. ref.line,
  2827. ref.column,
  2828. ref.candidates ? JSON.stringify(ref.candidates) : null,
  2829. ref.filePath ?? '',
  2830. ref.language ?? 'unknown',
  2831. ]);
  2832. }
  2833. this.runBatched(
  2834. 'insertUnresolvedRefs',
  2835. 'INSERT INTO unresolved_refs (from_node_id, reference_name, reference_kind, line, col, candidates, file_path, language) VALUES ',
  2836. '(?,?,?,?,?,?,?,?)',
  2837. rows
  2838. );
  2839. });
  2840. insert();
  2841. }
  2842. /**
  2843. * Delete unresolved references from a node
  2844. */
  2845. deleteUnresolvedByNode(nodeId: string): void {
  2846. if (!this.stmts.deleteUnresolvedByNode) {
  2847. this.stmts.deleteUnresolvedByNode = this.db.prepare(
  2848. 'DELETE FROM unresolved_refs WHERE from_node_id = ?'
  2849. );
  2850. }
  2851. this.stmts.deleteUnresolvedByNode.run(nodeId);
  2852. }
  2853. /**
  2854. * Get unresolved references by name (for resolution)
  2855. */
  2856. getUnresolvedByName(name: string): UnresolvedReference[] {
  2857. if (!this.stmts.getUnresolvedByName) {
  2858. this.stmts.getUnresolvedByName = this.db.prepare(
  2859. 'SELECT * FROM unresolved_refs WHERE reference_name = ?'
  2860. );
  2861. }
  2862. const rows = this.stmts.getUnresolvedByName.all(name) as UnresolvedRefRow[];
  2863. return rows.map((row) => ({
  2864. fromNodeId: row.from_node_id,
  2865. referenceName: row.reference_name,
  2866. referenceKind: row.reference_kind as EdgeKind,
  2867. line: row.line,
  2868. column: row.col,
  2869. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  2870. filePath: row.file_path,
  2871. language: row.language as Language,
  2872. rowId: row.id,
  2873. }));
  2874. }
  2875. /**
  2876. * Get all unresolved references
  2877. */
  2878. getUnresolvedReferences(): UnresolvedReference[] {
  2879. const rows = this.db.prepare('SELECT * FROM unresolved_refs').all() as UnresolvedRefRow[];
  2880. return rows.map((row) => ({
  2881. fromNodeId: row.from_node_id,
  2882. referenceName: row.reference_name,
  2883. referenceKind: row.reference_kind as EdgeKind,
  2884. line: row.line,
  2885. column: row.col,
  2886. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  2887. filePath: row.file_path,
  2888. language: row.language as Language,
  2889. rowId: row.id,
  2890. }));
  2891. }
  2892. /**
  2893. * Get the count of PENDING (never-attempted) references without loading
  2894. * them into memory. Rows marked status='failed' — attempted by a completed
  2895. * pass, no match — are excluded: they are not outstanding work, only retry
  2896. * candidates for the #1240 sweep, so they must not trip the #1187 orphan
  2897. * sweep or the `status` pending-refs warning.
  2898. */
  2899. getUnresolvedReferencesCount(): number {
  2900. if (!this.stmts.getUnresolvedCount) {
  2901. this.stmts.getUnresolvedCount = this.db.prepare(
  2902. "SELECT COUNT(*) as count FROM unresolved_refs WHERE status = 'pending'"
  2903. );
  2904. }
  2905. const row = this.stmts.getUnresolvedCount.get() as { count: number };
  2906. return row.count;
  2907. }
  2908. /**
  2909. * Get a batch of PENDING unresolved references using LIMIT/OFFSET
  2910. * pagination. Used to process references in bounded memory chunks; failed
  2911. * rows are excluded so the batched drain loop terminates once every row
  2912. * has been attempted.
  2913. */
  2914. getUnresolvedReferencesBatch(offset: number, limit: number): UnresolvedReference[] {
  2915. if (!this.stmts.getUnresolvedBatch) {
  2916. // ORDER BY rowid is load-bearing for the pipelined resolution loop: it
  2917. // prefetches batch k+1 at OFFSET batch_k.length while batch k's rows are
  2918. // still pending, which is only exact under a stable enumeration. (A plain
  2919. // scan and the status index both return rowid order anyway — this pins
  2920. // it.)
  2921. this.stmts.getUnresolvedBatch = this.db.prepare(
  2922. "SELECT * FROM unresolved_refs WHERE status = 'pending' ORDER BY rowid LIMIT ? OFFSET ?"
  2923. );
  2924. }
  2925. const rows = this.stmts.getUnresolvedBatch.all(limit, offset) as UnresolvedRefRow[];
  2926. return rows.map((row) => ({
  2927. fromNodeId: row.from_node_id,
  2928. referenceName: row.reference_name,
  2929. referenceKind: row.reference_kind as EdgeKind,
  2930. line: row.line,
  2931. column: row.col,
  2932. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  2933. filePath: row.file_path,
  2934. language: row.language as Language,
  2935. rowId: row.id,
  2936. }));
  2937. }
  2938. /**
  2939. * Keyset variant of {@link getUnresolvedReferencesBatch} for the batched
  2940. * resolution loop: seek past the last-seen row id instead of OFFSET-walking.
  2941. * OFFSET reads re-scan the accumulated failed-row prefix on every batch —
  2942. * O(failed rows) per read, measured at 54.6s of the kernel-scale batch loop
  2943. * (§7a.2) — while the seek is O(batch) forever. `id` is the rowid alias, so
  2944. * the enumeration order is identical to the OFFSET reader's.
  2945. */
  2946. getUnresolvedReferencesBatchAfter(afterRowId: number, limit: number, prerequisites?: boolean): UnresolvedReference[] {
  2947. // Resolution prerequisites must be committed before dependent calls,
  2948. // even when an interrupted sync queued their rows in a different order
  2949. // from a clean index (#1577). Each phase still seeks by row id in bounded
  2950. // memory; the default preserves the public reader's original enumeration.
  2951. const key = prerequisites === undefined ? 'getUnresolvedBatchAfter'
  2952. : prerequisites ? 'getUnresolvedPrerequisitesAfter' : 'getUnresolvedDependentsAfter';
  2953. if (!this.stmts[key]) {
  2954. const filter = prerequisites === undefined ? ''
  2955. : ` AND reference_kind ${prerequisites ? 'IN' : 'NOT IN'} ('imports', 'extends', 'implements')`;
  2956. this.stmts[key] = this.db.prepare(
  2957. `SELECT * FROM unresolved_refs WHERE status = 'pending' AND id > ?${filter} ORDER BY id LIMIT ?`
  2958. );
  2959. }
  2960. const rows = this.stmts[key]!.all(afterRowId, limit) as UnresolvedRefRow[];
  2961. return rows.map((row) => ({
  2962. fromNodeId: row.from_node_id,
  2963. referenceName: row.reference_name,
  2964. referenceKind: row.reference_kind as EdgeKind,
  2965. line: row.line,
  2966. column: row.col,
  2967. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  2968. filePath: row.file_path,
  2969. language: row.language as Language,
  2970. rowId: row.id,
  2971. }));
  2972. }
  2973. /**
  2974. * Get all tracked file paths (lightweight — no full FileRecord objects)
  2975. */
  2976. getAllFilePaths(): string[] {
  2977. if (!this.stmts.getAllFilePaths) {
  2978. this.stmts.getAllFilePaths = this.db.prepare('SELECT path FROM files ORDER BY path');
  2979. }
  2980. const rows = this.stmts.getAllFilePaths.all() as Array<{ path: string }>;
  2981. return rows.map((r) => r.path);
  2982. }
  2983. /**
  2984. * Get all distinct node names (lightweight — just name strings for pre-filtering)
  2985. */
  2986. getAllNodeNames(): string[] {
  2987. if (!this.stmts.getAllNodeNames) {
  2988. this.stmts.getAllNodeNames = this.db.prepare('SELECT DISTINCT name FROM nodes');
  2989. }
  2990. const rows = this.stmts.getAllNodeNames.all() as Array<{ name: string }>;
  2991. return rows.map((r) => r.name);
  2992. }
  2993. /**
  2994. * Stream the distinct node names one row at a time — the incremental
  2995. * counterpart to {@link getAllNodeNames} for callers that need to yield
  2996. * to the event loop mid-scan (resolver cache warm-up on multi-million-node
  2997. * indexes). Fresh statement per call: the iterator holds an open cursor.
  2998. */
  2999. *iterateNodeNames(): IterableIterator<string> {
  3000. const stmt = this.db.prepare('SELECT DISTINCT name FROM nodes');
  3001. for (const row of stmt.iterate()) {
  3002. yield (row as { name: string }).name;
  3003. }
  3004. }
  3005. /**
  3006. * Get unresolved references scoped to specific file paths.
  3007. * Uses the idx_unresolved_file_path index for efficient lookup.
  3008. */
  3009. getUnresolvedReferencesByFiles(filePaths: string[]): UnresolvedReference[] {
  3010. if (filePaths.length === 0) return [];
  3011. // Chunk under SQLite's parameter limit: the first sync of a very large repo
  3012. // passes every changed file here, which an unbounded `IN (...)` would bind
  3013. // as one parameter each — exceeding MAX_VARIABLE_NUMBER and aborting with
  3014. // "too many SQL variables". (#540)
  3015. const rows: UnresolvedRefRow[] = [];
  3016. for (let i = 0; i < filePaths.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3017. const chunk = filePaths.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3018. const placeholders = chunk.map(() => '?').join(',');
  3019. const chunkRows = this.db
  3020. .prepare(`SELECT * FROM unresolved_refs WHERE status = 'pending' AND file_path IN (${placeholders})`)
  3021. .all(...chunk) as UnresolvedRefRow[];
  3022. // Append with a loop, never a spread: the INPUT chunk is bounded, but
  3023. // the RESULT rows per chunk are not — a dense recovery sync (e.g. the
  3024. // #1541 self-heal re-indexing hundreds of files) returns more rows than
  3025. // V8 allows as arguments, and `push(...chunkRows)` dies with "Maximum
  3026. // call stack size exceeded", aborting resolution mid-sync (#1558).
  3027. for (const row of chunkRows) rows.push(row);
  3028. }
  3029. return rows.map((row) => ({
  3030. fromNodeId: row.from_node_id,
  3031. referenceName: row.reference_name,
  3032. referenceKind: row.reference_kind as EdgeKind,
  3033. line: row.line,
  3034. column: row.col,
  3035. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  3036. filePath: row.file_path,
  3037. language: row.language as Language,
  3038. rowId: row.id,
  3039. }));
  3040. }
  3041. /**
  3042. * Delete all unresolved references (after resolution)
  3043. */
  3044. clearUnresolvedReferences(): void {
  3045. this.db.exec('DELETE FROM unresolved_refs');
  3046. }
  3047. /**
  3048. * Delete resolved references by their IDs
  3049. */
  3050. deleteResolvedReferences(fromNodeIds: string[]): void {
  3051. if (fromNodeIds.length === 0) return;
  3052. // Chunk under SQLite's parameter limit, matching every other IN-list in
  3053. // this file. The internal resolution path uses deleteSpecificResolvedReferences
  3054. // instead, but QueryBuilder is part of the public API, so a library consumer
  3055. // passing more ids than SQLITE_MAX_VARIABLE_NUMBER (32766 on the bundled
  3056. // node:sqlite) would otherwise hit "too many SQL variables". (#540, #1001)
  3057. for (let i = 0; i < fromNodeIds.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3058. const chunk = fromNodeIds.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3059. const placeholders = chunk.map(() => '?').join(',');
  3060. this.db.prepare(`DELETE FROM unresolved_refs WHERE from_node_id IN (${placeholders})`).run(...chunk);
  3061. }
  3062. }
  3063. /**
  3064. * Delete specific resolved references by (fromNodeId, referenceName, referenceKind) tuples.
  3065. * More precise than deleteResolvedReferences — only removes refs that were actually resolved.
  3066. */
  3067. deleteSpecificResolvedReferences(refs: Array<{ fromNodeId: string; referenceName: string; referenceKind: string }>): number {
  3068. if (refs.length === 0) return 0;
  3069. const stmt = this.db.prepare(
  3070. 'DELETE FROM unresolved_refs WHERE from_node_id = ? AND reference_name = ? AND reference_kind = ?'
  3071. );
  3072. // Returns rows actually removed (SQLite `changes`, summed): the batched
  3073. // resolution loop's non-progress guard keys on this — zero removals from
  3074. // a batch that claimed work is the direct runaway signal (§7a.2).
  3075. let changed = 0;
  3076. const deleteMany = this.db.transaction((items: typeof refs) => {
  3077. for (const ref of items) {
  3078. changed += stmt.run(ref.fromNodeId, ref.referenceName, ref.referenceKind).changes;
  3079. }
  3080. });
  3081. deleteMany(refs);
  3082. return changed;
  3083. }
  3084. /**
  3085. * Delete unresolved-ref rows by row id — the precise cleanup for refs a
  3086. * resolution pass actually processed. The key-tuple variant above also
  3087. * deletes SIBLING rows (same caller calling the same callee at other lines)
  3088. * that a later batch hasn't attempted yet, so when a batch boundary split a
  3089. * caller's same-named call sites, the later sites' edges were silently never
  3090. * created (#1269).
  3091. */
  3092. deleteReferencesByRowIds(rowIds: number[]): number {
  3093. if (rowIds.length === 0) return 0;
  3094. // One transaction for all chunks (each chunk was previously its own
  3095. // implicit transaction = its own WAL commit — measurable on 100k+-ref
  3096. // resolution persists), and the full-size chunk statement is cached so
  3097. // repeat calls skip the re-prepare; only the final partial chunk (if any)
  3098. // prepares ad hoc. Returns rows actually removed (summed `changes`) for
  3099. // the batched loop's non-progress guard (§7a.2).
  3100. let changed = 0;
  3101. this.db.transaction(() => {
  3102. for (let i = 0; i < rowIds.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3103. const chunk = rowIds.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3104. if (chunk.length === SQLITE_PARAM_CHUNK_SIZE) {
  3105. if (!this.stmts.deleteRefsByRowIdsFull) {
  3106. const placeholders = new Array(SQLITE_PARAM_CHUNK_SIZE).fill('?').join(',');
  3107. this.stmts.deleteRefsByRowIdsFull = this.db.prepare(
  3108. `DELETE FROM unresolved_refs WHERE id IN (${placeholders})`
  3109. );
  3110. }
  3111. changed += this.stmts.deleteRefsByRowIdsFull.run(...chunk).changes;
  3112. } else {
  3113. const placeholders = chunk.map(() => '?').join(',');
  3114. changed += this.db.prepare(`DELETE FROM unresolved_refs WHERE id IN (${placeholders})`).run(...chunk).changes;
  3115. }
  3116. }
  3117. })();
  3118. return changed;
  3119. }
  3120. /**
  3121. * Mark refs a completed resolution pass could not resolve as status='failed'
  3122. * instead of deleting them (#1240). Failed rows are invisible to the pending
  3123. * count/batch readers (so drain loops and the #1187 orphan sweep still
  3124. * terminate) but stay queryable by name_tail so a later sync can retry them
  3125. * when a changed file introduces a symbol that could satisfy them. name_tail
  3126. * is (re)written here so rows inserted before the v8 migration get their
  3127. * tail the first time they're attempted.
  3128. */
  3129. markReferencesFailed(refs: Array<{ fromNodeId: string; referenceName: string; referenceKind: string }>): number {
  3130. if (refs.length === 0) return 0;
  3131. const stmt = this.db.prepare(
  3132. "UPDATE unresolved_refs SET status = 'failed', name_tail = ? WHERE from_node_id = ? AND reference_name = ? AND reference_kind = ?"
  3133. );
  3134. let changed = 0;
  3135. const markMany = this.db.transaction((items: typeof refs) => {
  3136. for (const ref of items) {
  3137. changed += stmt.run(referenceNameTail(ref.referenceName), ref.fromNodeId, ref.referenceName, ref.referenceKind).changes;
  3138. }
  3139. });
  3140. markMany(refs);
  3141. return changed;
  3142. }
  3143. /**
  3144. * Park refs as status='failed' by row id — the precise counterpart of
  3145. * markReferencesFailed, for the same reason as deleteReferencesByRowIds:
  3146. * the key-tuple variant also flips same-key sibling rows in later batches
  3147. * to 'failed' before they were ever attempted (#1269). Resolution outcome
  3148. * can differ per call site (receiver-type inference reads the ref's line),
  3149. * so a sibling must not inherit this row's failure.
  3150. */
  3151. markReferencesFailedByRowIds(refs: Array<{ rowId: number; referenceName: string }>): number {
  3152. if (refs.length === 0) return 0;
  3153. const stmt = this.db.prepare(
  3154. "UPDATE unresolved_refs SET status = 'failed', name_tail = ? WHERE id = ?"
  3155. );
  3156. let changed = 0;
  3157. const markMany = this.db.transaction((items: typeof refs) => {
  3158. for (const ref of items) {
  3159. changed += stmt.run(referenceNameTail(ref.referenceName), ref.rowId).changes;
  3160. }
  3161. });
  3162. markMany(refs);
  3163. return changed;
  3164. }
  3165. /**
  3166. * Failed refs whose name tail matches one of the given symbol names — the
  3167. * candidates a sync should retry after files carrying those names changed
  3168. * (#1240). Names matching more than `perNameCeiling` failed refs are
  3169. * skipped entirely: at that population a name is external/builtin noise
  3170. * (`get`, `map`, …) that one new definition won't resolve — the same
  3171. * rationale as resolution's AMBIGUOUS_NAME_CEILING (#999) — and retrying an
  3172. * arbitrary subset would be both wasted work and incoherent coverage.
  3173. */
  3174. getRetryableFailedReferences(names: string[], perNameCeiling: number = 500): UnresolvedReference[] {
  3175. if (names.length === 0) return [];
  3176. // Pass 1: per-tail counts, chunked under the SQLite parameter limit.
  3177. const retryNames: string[] = [];
  3178. for (let i = 0; i < names.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3179. const chunk = names.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3180. const placeholders = chunk.map(() => '?').join(',');
  3181. const counts = this.db
  3182. .prepare(
  3183. `SELECT name_tail, COUNT(*) as count FROM unresolved_refs WHERE status = 'failed' AND name_tail IN (${placeholders}) GROUP BY name_tail`
  3184. )
  3185. .all(...chunk) as Array<{ name_tail: string; count: number }>;
  3186. for (const row of counts) {
  3187. if (row.count <= perNameCeiling) retryNames.push(row.name_tail);
  3188. }
  3189. }
  3190. if (retryNames.length === 0) return [];
  3191. // Pass 2: load the surviving rows.
  3192. const rows: UnresolvedRefRow[] = [];
  3193. for (let i = 0; i < retryNames.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3194. const chunk = retryNames.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3195. const placeholders = chunk.map(() => '?').join(',');
  3196. const chunkRows = this.db
  3197. .prepare(`SELECT * FROM unresolved_refs WHERE status = 'failed' AND name_tail IN (${placeholders})`)
  3198. .all(...chunk) as UnresolvedRefRow[];
  3199. // Loop, not spread — same V8 argument-limit hazard as
  3200. // getUnresolvedReferencesByFiles (#1558): a large definition delta can
  3201. // select an unbounded number of failed rows per chunk.
  3202. for (const row of chunkRows) rows.push(row);
  3203. }
  3204. return rows.map((row) => ({
  3205. fromNodeId: row.from_node_id,
  3206. referenceName: row.reference_name,
  3207. referenceKind: row.reference_kind as EdgeKind,
  3208. line: row.line,
  3209. column: row.col,
  3210. candidates: row.candidates ? safeJsonParse(row.candidates, undefined) : undefined,
  3211. filePath: row.file_path,
  3212. language: row.language as Language,
  3213. rowId: row.id,
  3214. }));
  3215. }
  3216. /**
  3217. * Resolution edges whose TARGET symbol is named one of `names` — the edges a
  3218. * sync must re-resolve after `names` gained or lost a definition (CG-33).
  3219. *
  3220. * Resolution binds a reference to a node whose name matches the reference's
  3221. * tail, and it picks among ALL same-named definitions project-wide. So adding
  3222. * or removing one definition of `pct` changes the answer for every `pct(...)`
  3223. * reference in the repo — including references in files this sync never
  3224. * touches, whose edges nothing else revisits. Those edges' current target is,
  3225. * by that same rule, a node named `pct`, which is why the target's name is a
  3226. * sufficient (and index-backed, via idx_nodes_name) way to find them without
  3227. * a schema change or a scan of edge metadata.
  3228. *
  3229. * Returns the source file/language alongside each edge so the caller can
  3230. * resurrect it as its original reference. Excludes `provenance='heuristic'`
  3231. * (synthesized dispatch edges are not resolution output and carry no refName
  3232. * stamp to resurrect from — deleting one would be a permanent loss).
  3233. *
  3234. * Names matching more than `perNameCeiling` edges are skipped entirely, same
  3235. * rationale and same default as {@link getRetryableFailedReferences}: at that
  3236. * population the name is generic (`get`, `clear`, …), one definition changing
  3237. * won't flip most of them, and rebinding an arbitrary subset is both wasted
  3238. * work and incoherent coverage.
  3239. */
  3240. getResolutionEdgesByTargetName(
  3241. names: string[],
  3242. perNameCeiling: number = 500
  3243. ): Array<Edge & { edgeId: number; sourceFilePath: string; sourceLanguage: Language }> {
  3244. if (names.length === 0) return [];
  3245. // Pass 1: per-name edge counts, chunked under the SQLite parameter limit.
  3246. const keep: string[] = [];
  3247. for (let i = 0; i < names.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3248. const chunk = names.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3249. const placeholders = chunk.map(() => '?').join(',');
  3250. const counts = this.db
  3251. .prepare(
  3252. `SELECT tgt.name AS name, COUNT(*) AS count
  3253. FROM edges e
  3254. JOIN nodes tgt ON tgt.id = e.target
  3255. WHERE tgt.name IN (${placeholders})
  3256. AND (e.provenance IS NULL OR e.provenance != 'heuristic')
  3257. GROUP BY tgt.name`
  3258. )
  3259. .all(...chunk) as Array<{ name: string; count: number }>;
  3260. for (const row of counts) {
  3261. if (row.count <= perNameCeiling) keep.push(row.name);
  3262. }
  3263. }
  3264. if (keep.length === 0) return [];
  3265. // Pass 2: load the surviving edges with the source file context a
  3266. // resurrection needs.
  3267. const out: Array<Edge & { edgeId: number; sourceFilePath: string; sourceLanguage: Language }> = [];
  3268. for (let i = 0; i < keep.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3269. const chunk = keep.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3270. const placeholders = chunk.map(() => '?').join(',');
  3271. const rows = this.db
  3272. .prepare(
  3273. `SELECT e.*, src.file_path AS source_file_path, src.language AS source_language
  3274. FROM edges e
  3275. JOIN nodes tgt ON tgt.id = e.target
  3276. JOIN nodes src ON src.id = e.source
  3277. WHERE tgt.name IN (${placeholders})
  3278. AND (e.provenance IS NULL OR e.provenance != 'heuristic')`
  3279. )
  3280. .all(...chunk) as Array<EdgeRow & { source_file_path: string; source_language: Language }>;
  3281. for (const row of rows) {
  3282. out.push({
  3283. ...rowToEdge(row),
  3284. edgeId: row.id,
  3285. sourceFilePath: row.source_file_path,
  3286. sourceLanguage: row.source_language,
  3287. });
  3288. }
  3289. }
  3290. return out;
  3291. }
  3292. /** Delete edges by primary key — the rebind pass's half of a re-resolution. */
  3293. deleteEdgesByIds(edgeIds: number[]): number {
  3294. if (edgeIds.length === 0) return 0;
  3295. let changed = 0;
  3296. this.db.transaction(() => {
  3297. for (let i = 0; i < edgeIds.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3298. const chunk = edgeIds.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3299. const placeholders = chunk.map(() => '?').join(',');
  3300. changed += this.db.prepare(`DELETE FROM edges WHERE id IN (${placeholders})`).run(...chunk).changes;
  3301. }
  3302. })();
  3303. return changed;
  3304. }
  3305. /**
  3306. * Distinct node names present in the given files — the symbol names a sync
  3307. * pass uses to look up retryable failed refs after those files changed.
  3308. */
  3309. getNodeNamesByFiles(filePaths: string[]): string[] {
  3310. if (filePaths.length === 0) return [];
  3311. const names = new Set<string>();
  3312. for (let i = 0; i < filePaths.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3313. const chunk = filePaths.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3314. const placeholders = chunk.map(() => '?').join(',');
  3315. const rows = this.db
  3316. .prepare(`SELECT DISTINCT name FROM nodes WHERE file_path IN (${placeholders})`)
  3317. .all(...chunk) as Array<{ name: string }>;
  3318. for (const row of rows) names.add(row.name);
  3319. }
  3320. return [...names];
  3321. }
  3322. /**
  3323. * Distinct `file\0name` pairs defined by the given files — the shape sync's
  3324. * definition delta needs (CG-33).
  3325. *
  3326. * Deliberately NOT `getNodeNamesByFiles`: a bare name set is taken over the
  3327. * WHOLE changed batch, so a name that moves between two files in one commit
  3328. * (or exists in one changed file and is newly added to another) appears on
  3329. * both sides and cancels out of the symmetric difference — even though a
  3330. * definition genuinely appeared or vanished and every reference to that name
  3331. * repo-wide may now bind elsewhere. Keying by file makes each definition its
  3332. * own fact, so the move is seen as one removal plus one addition.
  3333. */
  3334. getNodeNamePairsByFiles(filePaths: string[]): Set<string> {
  3335. const pairs = new Set<string>();
  3336. if (filePaths.length === 0) return pairs;
  3337. for (let i = 0; i < filePaths.length; i += SQLITE_PARAM_CHUNK_SIZE) {
  3338. const chunk = filePaths.slice(i, i + SQLITE_PARAM_CHUNK_SIZE);
  3339. const placeholders = chunk.map(() => '?').join(',');
  3340. const rows = this.db
  3341. .prepare(`SELECT DISTINCT file_path, name FROM nodes WHERE file_path IN (${placeholders})`)
  3342. .all(...chunk) as Array<{ file_path: string; name: string }>;
  3343. // NUL-joined: a path or a symbol name can contain a space, never a NUL.
  3344. for (const row of rows) pairs.add(`${row.file_path}\0${row.name}`);
  3345. }
  3346. return pairs;
  3347. }
  3348. // ===========================================================================
  3349. // Statistics
  3350. // ===========================================================================
  3351. /**
  3352. * Lightweight (nodes, edges) count snapshot. Used around an index/sync
  3353. * run to compute true additions across extraction + resolution +
  3354. * synthesis — the per-phase counter in the orchestrator only sees
  3355. * extraction's contribution, which is why the CLI summary under-reported
  3356. * the edge count (resolution + synthesizer edges were invisible).
  3357. */
  3358. getNodeAndEdgeCount(): { nodes: number; edges: number } {
  3359. return this.db
  3360. .prepare('SELECT (SELECT COUNT(*) FROM nodes) AS nodes, (SELECT COUNT(*) FROM edges) AS edges')
  3361. .get() as { nodes: number; edges: number };
  3362. }
  3363. /**
  3364. * Get graph statistics
  3365. */
  3366. getStats(): GraphStats {
  3367. // Single query for all three aggregate counts
  3368. const counts = this.db.prepare(`
  3369. SELECT
  3370. (SELECT COUNT(*) FROM nodes) AS node_count,
  3371. (SELECT COUNT(*) FROM edges) AS edge_count,
  3372. (SELECT COUNT(*) FROM files) AS file_count
  3373. `).get() as { node_count: number; edge_count: number; file_count: number };
  3374. const nodesByKind = {} as Record<NodeKind, number>;
  3375. const nodeKindRows = this.db
  3376. .prepare('SELECT kind, COUNT(*) as count FROM nodes GROUP BY kind')
  3377. .all() as Array<{ kind: string; count: number }>;
  3378. for (const row of nodeKindRows) {
  3379. nodesByKind[row.kind as NodeKind] = row.count;
  3380. }
  3381. const edgesByKind = {} as Record<EdgeKind, number>;
  3382. const edgeKindRows = this.db
  3383. .prepare('SELECT kind, COUNT(*) as count FROM edges GROUP BY kind')
  3384. .all() as Array<{ kind: string; count: number }>;
  3385. for (const row of edgeKindRows) {
  3386. edgesByKind[row.kind as EdgeKind] = row.count;
  3387. }
  3388. const filesByLanguage = {} as Record<Language, number>;
  3389. const languageRows = this.db
  3390. .prepare('SELECT language, COUNT(*) as count FROM files GROUP BY language')
  3391. .all() as Array<{ language: string; count: number }>;
  3392. for (const row of languageRows) {
  3393. filesByLanguage[row.language as Language] = row.count;
  3394. }
  3395. return {
  3396. nodeCount: counts.node_count,
  3397. edgeCount: counts.edge_count,
  3398. fileCount: counts.file_count,
  3399. nodesByKind,
  3400. edgesByKind,
  3401. filesByLanguage,
  3402. dbSizeBytes: 0, // Set by caller using DatabaseConnection.getSize()
  3403. walSizeBytes: 0, // Set by caller using DatabaseConnection.getWalSizeBytes()
  3404. lastUpdated: Date.now(),
  3405. };
  3406. }
  3407. // ===========================================================================
  3408. // Project Metadata
  3409. // ===========================================================================
  3410. /**
  3411. * Get a metadata value by key
  3412. */
  3413. getMetadata(key: string): string | null {
  3414. const row = this.db.prepare('SELECT value FROM project_metadata WHERE key = ?').get(key) as { value: string } | undefined;
  3415. return row?.value ?? null;
  3416. }
  3417. /**
  3418. * Set a metadata key-value pair (upsert)
  3419. */
  3420. setMetadata(key: string, value: string): void {
  3421. this.db.prepare(
  3422. 'INSERT INTO project_metadata (key, value, updated_at) VALUES (?, ?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value, updated_at = excluded.updated_at'
  3423. ).run(key, value, Date.now());
  3424. }
  3425. /**
  3426. * Get all metadata as a key-value record
  3427. */
  3428. getAllMetadata(): Record<string, string> {
  3429. const rows = this.db.prepare('SELECT key, value FROM project_metadata').all() as { key: string; value: string }[];
  3430. const result: Record<string, string> = {};
  3431. for (const row of rows) {
  3432. result[row.key] = row.value;
  3433. }
  3434. return result;
  3435. }
  3436. /**
  3437. * Clear all data from the database
  3438. */
  3439. clear(): void {
  3440. this.nodeCache.clear();
  3441. this.db.transaction(() => {
  3442. this.db.exec('DELETE FROM unresolved_refs');
  3443. this.db.exec('DELETE FROM edges');
  3444. this.db.exec('DELETE FROM nodes');
  3445. this.db.exec('DELETE FROM files');
  3446. })();
  3447. }
  3448. }
  3449. /**
  3450. * Turn the module aggregation's one result set into its two answers.
  3451. *
  3452. * The query groups by module pair AND kind AND symbol names, because the join
  3453. * is what costs and a finer grouping rides along free. That leaves two folds:
  3454. * counts per (module, module, kind) for the map's link weights, and the busiest
  3455. * symbol pairs per link for its tooltip.
  3456. *
  3457. * Pairs are ranked `declared` first and only then by raw count, so a link's
  3458. * tooltip names the symbols the source actually points at rather than whichever
  3459. * `has`/`get`/`run` happened to name-match most often. Only `pairKinds` are
  3460. * eligible: "Config to Config" is real traffic but not an interesting row.
  3461. */
  3462. interface ModuleGroupRow {
  3463. source: string;
  3464. target: string;
  3465. kind: EdgeKind;
  3466. from: string;
  3467. to: string;
  3468. count: number;
  3469. declared: number;
  3470. uncertain: number;
  3471. }
  3472. interface ModuleLinkTotal {
  3473. source: string;
  3474. target: string;
  3475. kind: EdgeKind;
  3476. count: number;
  3477. declared: number;
  3478. uncertain: number;
  3479. }
  3480. interface ModulePairTotal {
  3481. source: string;
  3482. target: string;
  3483. from: string;
  3484. to: string;
  3485. count: number;
  3486. declared: number;
  3487. }
  3488. function foldModuleRows(
  3489. rows: ReadonlyArray<ModuleGroupRow>,
  3490. options: { topPairsPerLink: number; pairKinds: readonly EdgeKind[] }
  3491. ): { links: ModuleLinkTotal[]; pairs: ModulePairTotal[] } {
  3492. // A module id is a path and may contain anything printable, so the key
  3493. // separator has to be something a path cannot hold.
  3494. const SEP = '\u0000';
  3495. const links = new Map<string, ModuleLinkTotal>();
  3496. const pairKinds = new Set(options.pairKinds);
  3497. const wantPairs = options.topPairsPerLink > 0 && pairKinds.size > 0;
  3498. const pairTotals = new Map<string, ModulePairTotal>();
  3499. for (const row of rows) {
  3500. const linkKey = `${row.source}${SEP}${row.target}${SEP}${row.kind}`;
  3501. const link = links.get(linkKey);
  3502. if (link) {
  3503. link.count += row.count;
  3504. link.declared += row.declared;
  3505. link.uncertain += row.uncertain;
  3506. } else {
  3507. links.set(linkKey, {
  3508. source: row.source,
  3509. target: row.target,
  3510. kind: row.kind,
  3511. count: row.count,
  3512. declared: row.declared,
  3513. uncertain: row.uncertain,
  3514. });
  3515. }
  3516. // Only the confident half of a row can be named: an uncertain edge is a
  3517. // guess, and printing "a to b, 12" for twelve guesses is the map claiming
  3518. // something it does not know.
  3519. if (!wantPairs || row.count === 0 || !pairKinds.has(row.kind)) continue;
  3520. const pairKey = `${row.source}${SEP}${row.target}${SEP}${row.from}${SEP}${row.to}`;
  3521. const pair = pairTotals.get(pairKey);
  3522. if (pair) {
  3523. pair.count += row.count;
  3524. pair.declared += row.declared;
  3525. } else {
  3526. pairTotals.set(pairKey, {
  3527. source: row.source,
  3528. target: row.target,
  3529. from: row.from,
  3530. to: row.to,
  3531. count: row.count,
  3532. declared: row.declared,
  3533. });
  3534. }
  3535. }
  3536. const byLink = new Map<string, ModulePairTotal[]>();
  3537. for (const pair of pairTotals.values()) {
  3538. const key = `${pair.source}${SEP}${pair.target}`;
  3539. let list = byLink.get(key);
  3540. if (!list) byLink.set(key, (list = []));
  3541. list.push(pair);
  3542. }
  3543. const pairs: ModulePairTotal[] = [];
  3544. for (const list of byLink.values()) {
  3545. list.sort(
  3546. (a, b) =>
  3547. b.declared - a.declared ||
  3548. b.count - a.count ||
  3549. a.from.localeCompare(b.from) ||
  3550. a.to.localeCompare(b.to)
  3551. );
  3552. for (const pair of list.slice(0, options.topPairsPerLink)) pairs.push(pair);
  3553. }
  3554. return { links: [...links.values()], pairs };
  3555. }