store.ts 189 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189119011911192119311941195119611971198119912001201120212031204120512061207120812091210121112121213121412151216121712181219122012211222122312241225122612271228122912301231123212331234123512361237123812391240124112421243124412451246124712481249125012511252125312541255125612571258125912601261126212631264126512661267126812691270127112721273127412751276127712781279128012811282128312841285128612871288128912901291129212931294129512961297129812991300130113021303130413051306130713081309131013111312131313141315131613171318131913201321132213231324132513261327132813291330133113321333133413351336133713381339134013411342134313441345134613471348134913501351135213531354135513561357135813591360136113621363136413651366136713681369137013711372137313741375137613771378137913801381138213831384138513861387138813891390139113921393139413951396139713981399140014011402140314041405140614071408140914101411141214131414141514161417141814191420142114221423142414251426142714281429143014311432143314341435143614371438143914401441144214431444144514461447144814491450145114521453145414551456145714581459146014611462146314641465146614671468146914701471147214731474147514761477147814791480148114821483148414851486148714881489149014911492149314941495149614971498149915001501150215031504150515061507150815091510151115121513151415151516151715181519152015211522152315241525152615271528152915301531153215331534153515361537153815391540154115421543154415451546154715481549155015511552155315541555155615571558155915601561156215631564156515661567156815691570157115721573157415751576157715781579158015811582158315841585158615871588158915901591159215931594159515961597159815991600160116021603160416051606160716081609161016111612161316141615161616171618161916201621162216231624162516261627162816291630163116321633163416351636163716381639164016411642164316441645164616471648164916501651165216531654165516561657165816591660166116621663166416651666166716681669167016711672167316741675167616771678167916801681168216831684168516861687168816891690169116921693169416951696169716981699170017011702170317041705170617071708170917101711171217131714171517161717171817191720172117221723172417251726172717281729173017311732173317341735173617371738173917401741174217431744174517461747174817491750175117521753175417551756175717581759176017611762176317641765176617671768176917701771177217731774177517761777177817791780178117821783178417851786178717881789179017911792179317941795179617971798179918001801180218031804180518061807180818091810181118121813181418151816181718181819182018211822182318241825182618271828182918301831183218331834183518361837183818391840184118421843184418451846184718481849185018511852185318541855185618571858185918601861186218631864186518661867186818691870187118721873187418751876187718781879188018811882188318841885188618871888188918901891189218931894189518961897189818991900190119021903190419051906190719081909191019111912191319141915191619171918191919201921192219231924192519261927192819291930193119321933193419351936193719381939194019411942194319441945194619471948194919501951195219531954195519561957195819591960196119621963196419651966196719681969197019711972197319741975197619771978197919801981198219831984198519861987198819891990199119921993199419951996199719981999200020012002200320042005200620072008200920102011201220132014201520162017201820192020202120222023202420252026202720282029203020312032203320342035203620372038203920402041204220432044204520462047204820492050205120522053205420552056205720582059206020612062206320642065206620672068206920702071207220732074207520762077207820792080208120822083208420852086208720882089209020912092209320942095209620972098209921002101210221032104210521062107210821092110211121122113211421152116211721182119212021212122212321242125212621272128212921302131213221332134213521362137213821392140214121422143214421452146214721482149215021512152215321542155215621572158215921602161216221632164216521662167216821692170217121722173217421752176217721782179218021812182218321842185218621872188218921902191219221932194219521962197219821992200220122022203220422052206220722082209221022112212221322142215221622172218221922202221222222232224222522262227222822292230223122322233223422352236223722382239224022412242224322442245224622472248224922502251225222532254225522562257225822592260226122622263226422652266226722682269227022712272227322742275227622772278227922802281228222832284228522862287228822892290229122922293229422952296229722982299230023012302230323042305230623072308230923102311231223132314231523162317231823192320232123222323232423252326232723282329233023312332233323342335233623372338233923402341234223432344234523462347234823492350235123522353235423552356235723582359236023612362236323642365236623672368236923702371237223732374237523762377237823792380238123822383238423852386238723882389239023912392239323942395239623972398239924002401240224032404240524062407240824092410241124122413241424152416241724182419242024212422242324242425242624272428242924302431243224332434243524362437243824392440244124422443244424452446244724482449245024512452245324542455245624572458245924602461246224632464246524662467246824692470247124722473247424752476247724782479248024812482248324842485248624872488248924902491249224932494249524962497249824992500250125022503250425052506250725082509251025112512251325142515251625172518251925202521252225232524252525262527252825292530253125322533253425352536253725382539254025412542254325442545254625472548254925502551255225532554255525562557255825592560256125622563256425652566256725682569257025712572257325742575257625772578257925802581258225832584258525862587258825892590259125922593259425952596259725982599260026012602260326042605260626072608260926102611261226132614261526162617261826192620262126222623262426252626262726282629263026312632263326342635263626372638263926402641264226432644264526462647264826492650265126522653265426552656265726582659266026612662266326642665266626672668266926702671267226732674267526762677267826792680268126822683268426852686268726882689269026912692269326942695269626972698269927002701270227032704270527062707270827092710271127122713271427152716271727182719272027212722272327242725272627272728272927302731273227332734273527362737273827392740274127422743274427452746274727482749275027512752275327542755275627572758275927602761276227632764276527662767276827692770277127722773277427752776277727782779278027812782278327842785278627872788278927902791279227932794279527962797279827992800280128022803280428052806280728082809281028112812281328142815281628172818281928202821282228232824282528262827282828292830283128322833283428352836283728382839284028412842284328442845284628472848284928502851285228532854285528562857285828592860286128622863286428652866286728682869287028712872287328742875287628772878287928802881288228832884288528862887288828892890289128922893289428952896289728982899290029012902290329042905290629072908290929102911291229132914291529162917291829192920292129222923292429252926292729282929293029312932293329342935293629372938293929402941294229432944294529462947294829492950295129522953295429552956295729582959296029612962296329642965296629672968296929702971297229732974297529762977297829792980298129822983298429852986298729882989299029912992299329942995299629972998299930003001300230033004300530063007300830093010301130123013301430153016301730183019302030213022302330243025302630273028302930303031303230333034303530363037303830393040304130423043304430453046304730483049305030513052305330543055305630573058305930603061306230633064306530663067306830693070307130723073307430753076307730783079308030813082308330843085308630873088308930903091309230933094309530963097309830993100310131023103310431053106310731083109311031113112311331143115311631173118311931203121312231233124312531263127312831293130313131323133313431353136313731383139314031413142314331443145314631473148314931503151315231533154315531563157315831593160316131623163316431653166316731683169317031713172317331743175317631773178317931803181318231833184318531863187318831893190319131923193319431953196319731983199320032013202320332043205320632073208320932103211321232133214321532163217321832193220322132223223322432253226322732283229323032313232323332343235323632373238323932403241324232433244324532463247324832493250325132523253325432553256325732583259326032613262326332643265326632673268326932703271327232733274327532763277327832793280328132823283328432853286328732883289329032913292329332943295329632973298329933003301330233033304330533063307330833093310331133123313331433153316331733183319332033213322332333243325332633273328332933303331333233333334333533363337333833393340334133423343334433453346334733483349335033513352335333543355335633573358335933603361336233633364336533663367336833693370337133723373337433753376337733783379338033813382338333843385338633873388338933903391339233933394339533963397339833993400340134023403340434053406340734083409341034113412341334143415341634173418341934203421342234233424342534263427342834293430343134323433343434353436343734383439344034413442344334443445344634473448344934503451345234533454345534563457345834593460346134623463346434653466346734683469347034713472347334743475347634773478347934803481348234833484348534863487348834893490349134923493349434953496349734983499350035013502350335043505350635073508350935103511351235133514351535163517351835193520352135223523352435253526352735283529353035313532353335343535353635373538353935403541354235433544354535463547354835493550355135523553355435553556355735583559356035613562356335643565356635673568356935703571357235733574357535763577357835793580358135823583358435853586358735883589359035913592359335943595359635973598359936003601360236033604360536063607360836093610361136123613361436153616361736183619362036213622362336243625362636273628362936303631363236333634363536363637363836393640364136423643364436453646364736483649365036513652365336543655365636573658365936603661366236633664366536663667366836693670367136723673367436753676367736783679368036813682368336843685368636873688368936903691369236933694369536963697369836993700370137023703370437053706370737083709371037113712371337143715371637173718371937203721372237233724372537263727372837293730373137323733373437353736373737383739374037413742374337443745374637473748374937503751375237533754375537563757375837593760376137623763376437653766376737683769377037713772377337743775377637773778377937803781378237833784378537863787378837893790379137923793379437953796379737983799380038013802380338043805380638073808380938103811381238133814381538163817381838193820382138223823382438253826382738283829383038313832383338343835383638373838383938403841384238433844384538463847384838493850385138523853385438553856385738583859386038613862386338643865386638673868386938703871387238733874387538763877387838793880388138823883388438853886388738883889389038913892389338943895389638973898389939003901390239033904390539063907390839093910391139123913391439153916391739183919392039213922392339243925392639273928392939303931393239333934393539363937393839393940394139423943394439453946394739483949395039513952395339543955395639573958395939603961396239633964396539663967396839693970397139723973397439753976397739783979398039813982398339843985398639873988398939903991399239933994399539963997399839994000400140024003400440054006400740084009401040114012401340144015401640174018401940204021402240234024402540264027402840294030403140324033403440354036403740384039404040414042404340444045404640474048404940504051405240534054405540564057405840594060406140624063406440654066406740684069407040714072407340744075407640774078407940804081408240834084408540864087408840894090409140924093409440954096409740984099410041014102410341044105410641074108410941104111411241134114411541164117411841194120412141224123412441254126412741284129413041314132413341344135413641374138413941404141414241434144414541464147414841494150415141524153415441554156415741584159416041614162416341644165416641674168416941704171417241734174417541764177417841794180418141824183418441854186418741884189419041914192419341944195419641974198419942004201420242034204420542064207420842094210421142124213421442154216421742184219422042214222422342244225422642274228422942304231423242334234423542364237423842394240424142424243424442454246424742484249425042514252425342544255425642574258425942604261426242634264426542664267426842694270427142724273427442754276427742784279428042814282428342844285428642874288428942904291429242934294429542964297429842994300430143024303430443054306430743084309431043114312431343144315431643174318431943204321432243234324432543264327432843294330433143324333433443354336433743384339434043414342434343444345434643474348434943504351435243534354435543564357435843594360436143624363436443654366436743684369437043714372437343744375437643774378437943804381438243834384438543864387438843894390439143924393439443954396439743984399440044014402440344044405440644074408440944104411441244134414441544164417441844194420442144224423442444254426442744284429443044314432443344344435443644374438443944404441444244434444444544464447444844494450445144524453445444554456445744584459446044614462446344644465446644674468446944704471447244734474447544764477447844794480448144824483448444854486448744884489449044914492449344944495449644974498449945004501450245034504450545064507450845094510451145124513451445154516451745184519452045214522452345244525452645274528452945304531453245334534453545364537453845394540454145424543454445454546454745484549455045514552455345544555455645574558455945604561456245634564456545664567456845694570457145724573457445754576457745784579458045814582458345844585458645874588458945904591459245934594459545964597459845994600460146024603460446054606460746084609461046114612461346144615461646174618461946204621462246234624462546264627462846294630463146324633463446354636463746384639464046414642464346444645464646474648464946504651465246534654465546564657465846594660466146624663466446654666466746684669467046714672467346744675467646774678467946804681468246834684468546864687468846894690469146924693469446954696469746984699470047014702470347044705470647074708470947104711471247134714471547164717471847194720472147224723472447254726472747284729473047314732473347344735473647374738473947404741474247434744474547464747474847494750475147524753475447554756475747584759476047614762476347644765476647674768476947704771477247734774477547764777477847794780478147824783478447854786478747884789479047914792479347944795479647974798479948004801480248034804480548064807480848094810481148124813481448154816481748184819482048214822482348244825482648274828482948304831483248334834483548364837483848394840484148424843484448454846484748484849485048514852485348544855485648574858485948604861486248634864486548664867486848694870487148724873487448754876487748784879488048814882488348844885488648874888488948904891489248934894489548964897489848994900490149024903490449054906490749084909491049114912491349144915491649174918491949204921492249234924492549264927492849294930493149324933493449354936493749384939494049414942494349444945494649474948494949504951495249534954495549564957495849594960496149624963496449654966496749684969497049714972497349744975497649774978497949804981498249834984498549864987498849894990499149924993499449954996499749984999500050015002500350045005500650075008500950105011501250135014501550165017501850195020502150225023502450255026502750285029503050315032503350345035503650375038503950405041504250435044504550465047504850495050505150525053505450555056505750585059506050615062506350645065506650675068506950705071507250735074507550765077507850795080508150825083508450855086508750885089509050915092509350945095509650975098509951005101510251035104510551065107510851095110511151125113511451155116511751185119512051215122512351245125512651275128512951305131513251335134513551365137513851395140514151425143514451455146514751485149515051515152515351545155515651575158515951605161516251635164516551665167516851695170517151725173517451755176517751785179518051815182518351845185518651875188518951905191519251935194519551965197519851995200520152025203520452055206520752085209521052115212521352145215521652175218521952205221522252235224522552265227522852295230
  1. /**
  2. * QMD Store - Core data access and retrieval functions
  3. *
  4. * This module provides all database operations, search functions, and document
  5. * retrieval for QMD. It returns raw data structures that can be formatted by
  6. * CLI or MCP consumers.
  7. *
  8. * Usage:
  9. * const store = createStore("/path/to/db.sqlite");
  10. * // or use default path:
  11. * const store = createStore();
  12. */
  13. import { openDatabase, loadSqliteVec } from "./db.js";
  14. import type { Database } from "./db.js";
  15. import picomatch from "picomatch";
  16. import { createHash } from "crypto";
  17. import { readFileSync, realpathSync, statSync, mkdirSync } from "node:fs";
  18. // Note: node:path resolve is not imported — we export our own cross-platform resolve()
  19. import fastGlob from "fast-glob";
  20. import {
  21. LlamaCpp,
  22. getDefaultLlamaCpp,
  23. formatQueryForEmbedding,
  24. formatDocForEmbedding,
  25. withLLMSessionForLlm,
  26. type LLMSessionOptions,
  27. type RerankDocument,
  28. type ILLMSession,
  29. } from "./llm.js";
  30. import type {
  31. NamedCollection,
  32. Collection,
  33. CollectionConfig,
  34. ContextMap,
  35. } from "./collections.js";
  36. import {
  37. type EmbeddingProvider,
  38. assertModelCompatible,
  39. } from "./embedding/provider.js";
  40. // =============================================================================
  41. // Configuration
  42. // =============================================================================
  43. const HOME = process.env.HOME || "/tmp";
  44. export const DEFAULT_EMBED_MODEL = "embeddinggemma";
  45. export const DEFAULT_RERANK_MODEL = "ExpedientFalcon/qwen3-reranker:0.6b-q8_0";
  46. export const DEFAULT_QUERY_MODEL = "Qwen/Qwen3-1.7B";
  47. export const DEFAULT_GLOB = "**/*.md";
  48. export const DEFAULT_MULTI_GET_MAX_BYTES = 10 * 1024; // 10KB
  49. export const DEFAULT_EMBED_MAX_DOCS_PER_BATCH = 64;
  50. export const DEFAULT_EMBED_MAX_BATCH_BYTES = 64 * 1024 * 1024; // 64MB
  51. // Chunking: 900 tokens per chunk with 15% overlap
  52. // Increased from 800 to accommodate smart chunking finding natural break points
  53. export const CHUNK_SIZE_TOKENS = 900;
  54. export const CHUNK_OVERLAP_TOKENS = Math.floor(CHUNK_SIZE_TOKENS * 0.15); // 135 tokens (15% overlap)
  55. // Fallback char-based approximation for sync chunking (~4 chars per token)
  56. export const CHUNK_SIZE_CHARS = CHUNK_SIZE_TOKENS * 4; // 3600 chars
  57. export const CHUNK_OVERLAP_CHARS = CHUNK_OVERLAP_TOKENS * 4; // 540 chars
  58. // Search window for finding optimal break points (in tokens, ~200 tokens)
  59. export const CHUNK_WINDOW_TOKENS = 200;
  60. export const CHUNK_WINDOW_CHARS = CHUNK_WINDOW_TOKENS * 4; // 800 chars
  61. /**
  62. * Get the LlamaCpp instance for a store — prefers the store's own instance,
  63. * falls back to the global singleton.
  64. */
  65. function getLlm(store: Store): LlamaCpp {
  66. return store.llm ?? getDefaultLlamaCpp();
  67. }
  68. // =============================================================================
  69. // Smart Chunking - Break Point Detection
  70. // =============================================================================
  71. /**
  72. * A potential break point in the document with a base score indicating quality.
  73. */
  74. export interface BreakPoint {
  75. pos: number; // character position
  76. score: number; // base score (higher = better break point)
  77. type: string; // for debugging: 'h1', 'h2', 'blank', etc.
  78. }
  79. /**
  80. * A region where a code fence exists (between ``` markers).
  81. * We should never split inside a code fence.
  82. */
  83. export interface CodeFenceRegion {
  84. start: number; // position of opening ```
  85. end: number; // position of closing ``` (or document end if unclosed)
  86. }
  87. /**
  88. * Patterns for detecting break points in markdown documents.
  89. * Higher scores indicate better places to split.
  90. * Scores are spread wide so headings decisively beat lower-quality breaks.
  91. * Order matters for scoring - more specific patterns first.
  92. */
  93. export const BREAK_PATTERNS: [RegExp, number, string][] = [
  94. [/\n#{1}(?!#)/g, 100, 'h1'], // # but not ##
  95. [/\n#{2}(?!#)/g, 90, 'h2'], // ## but not ###
  96. [/\n#{3}(?!#)/g, 80, 'h3'], // ### but not ####
  97. [/\n#{4}(?!#)/g, 70, 'h4'], // #### but not #####
  98. [/\n#{5}(?!#)/g, 60, 'h5'], // ##### but not ######
  99. [/\n#{6}(?!#)/g, 50, 'h6'], // ######
  100. [/\n```/g, 80, 'codeblock'], // code block boundary (same as h3)
  101. [/\n(?:---|\*\*\*|___)\s*\n/g, 60, 'hr'], // horizontal rule
  102. [/\n\n+/g, 20, 'blank'], // paragraph boundary
  103. [/\n[-*]\s/g, 5, 'list'], // unordered list item
  104. [/\n\d+\.\s/g, 5, 'numlist'], // ordered list item
  105. [/\n/g, 1, 'newline'], // minimal break
  106. ];
  107. /**
  108. * Scan text for all potential break points.
  109. * Returns sorted array of break points with higher-scoring patterns taking precedence
  110. * when multiple patterns match the same position.
  111. */
  112. export function scanBreakPoints(text: string): BreakPoint[] {
  113. const points: BreakPoint[] = [];
  114. const seen = new Map<number, BreakPoint>(); // pos -> best break point at that pos
  115. for (const [pattern, score, type] of BREAK_PATTERNS) {
  116. for (const match of text.matchAll(pattern)) {
  117. const pos = match.index!;
  118. const existing = seen.get(pos);
  119. // Keep higher score if position already seen
  120. if (!existing || score > existing.score) {
  121. const bp = { pos, score, type };
  122. seen.set(pos, bp);
  123. }
  124. }
  125. }
  126. // Convert to array and sort by position
  127. for (const bp of seen.values()) {
  128. points.push(bp);
  129. }
  130. return points.sort((a, b) => a.pos - b.pos);
  131. }
  132. /**
  133. * Find all code fence regions in the text.
  134. * Code fences are delimited by ``` and we should never split inside them.
  135. */
  136. export function findCodeFences(text: string): CodeFenceRegion[] {
  137. const regions: CodeFenceRegion[] = [];
  138. const fencePattern = /\n```/g;
  139. let inFence = false;
  140. let fenceStart = 0;
  141. for (const match of text.matchAll(fencePattern)) {
  142. if (!inFence) {
  143. fenceStart = match.index!;
  144. inFence = true;
  145. } else {
  146. regions.push({ start: fenceStart, end: match.index! + match[0].length });
  147. inFence = false;
  148. }
  149. }
  150. // Handle unclosed fence - extends to end of document
  151. if (inFence) {
  152. regions.push({ start: fenceStart, end: text.length });
  153. }
  154. return regions;
  155. }
  156. /**
  157. * Check if a position is inside a code fence region.
  158. */
  159. export function isInsideCodeFence(pos: number, fences: CodeFenceRegion[]): boolean {
  160. return fences.some(f => pos > f.start && pos < f.end);
  161. }
  162. /**
  163. * Find the best cut position using scored break points with distance decay.
  164. *
  165. * Uses squared distance for gentler early decay - headings far back still win
  166. * over low-quality breaks near the target.
  167. *
  168. * @param breakPoints - Pre-scanned break points from scanBreakPoints()
  169. * @param targetCharPos - The ideal cut position (e.g., maxChars boundary)
  170. * @param windowChars - How far back to search for break points (default ~200 tokens)
  171. * @param decayFactor - How much to penalize distance (0.7 = 30% score at window edge)
  172. * @param codeFences - Code fence regions to avoid splitting inside
  173. * @returns The best position to cut at
  174. */
  175. export function findBestCutoff(
  176. breakPoints: BreakPoint[],
  177. targetCharPos: number,
  178. windowChars: number = CHUNK_WINDOW_CHARS,
  179. decayFactor: number = 0.7,
  180. codeFences: CodeFenceRegion[] = []
  181. ): number {
  182. const windowStart = targetCharPos - windowChars;
  183. let bestScore = -1;
  184. let bestPos = targetCharPos;
  185. for (const bp of breakPoints) {
  186. if (bp.pos < windowStart) continue;
  187. if (bp.pos > targetCharPos) break; // sorted, so we can stop
  188. // Skip break points inside code fences
  189. if (isInsideCodeFence(bp.pos, codeFences)) continue;
  190. const distance = targetCharPos - bp.pos;
  191. // Squared distance decay: gentle early, steep late
  192. // At target: multiplier = 1.0
  193. // At 25% back: multiplier = 0.956
  194. // At 50% back: multiplier = 0.825
  195. // At 75% back: multiplier = 0.606
  196. // At window edge: multiplier = 0.3
  197. const normalizedDist = distance / windowChars;
  198. const multiplier = 1.0 - (normalizedDist * normalizedDist) * decayFactor;
  199. const finalScore = bp.score * multiplier;
  200. if (finalScore > bestScore) {
  201. bestScore = finalScore;
  202. bestPos = bp.pos;
  203. }
  204. }
  205. return bestPos;
  206. }
  207. // =============================================================================
  208. // Chunk Strategy
  209. // =============================================================================
  210. export type ChunkStrategy = "auto" | "regex" | "function";
  211. /**
  212. * Merge two sets of break points (e.g. regex + AST), keeping the highest
  213. * score at each position. Result is sorted by position.
  214. */
  215. export function mergeBreakPoints(a: BreakPoint[], b: BreakPoint[]): BreakPoint[] {
  216. const seen = new Map<number, BreakPoint>();
  217. for (const bp of a) {
  218. const existing = seen.get(bp.pos);
  219. if (!existing || bp.score > existing.score) {
  220. seen.set(bp.pos, bp);
  221. }
  222. }
  223. for (const bp of b) {
  224. const existing = seen.get(bp.pos);
  225. if (!existing || bp.score > existing.score) {
  226. seen.set(bp.pos, bp);
  227. }
  228. }
  229. return Array.from(seen.values()).sort((a, b) => a.pos - b.pos);
  230. }
  231. /**
  232. * Core chunk algorithm that operates on precomputed break points and code fences.
  233. * This is the shared implementation used by both regex-only and AST-aware chunking.
  234. */
  235. export function chunkDocumentWithBreakPoints(
  236. content: string,
  237. breakPoints: BreakPoint[],
  238. codeFences: CodeFenceRegion[],
  239. maxChars: number = CHUNK_SIZE_CHARS,
  240. overlapChars: number = CHUNK_OVERLAP_CHARS,
  241. windowChars: number = CHUNK_WINDOW_CHARS
  242. ): { text: string; pos: number }[] {
  243. if (content.length <= maxChars) {
  244. return [{ text: content, pos: 0 }];
  245. }
  246. const chunks: { text: string; pos: number }[] = [];
  247. let charPos = 0;
  248. while (charPos < content.length) {
  249. const targetEndPos = Math.min(charPos + maxChars, content.length);
  250. let endPos = targetEndPos;
  251. if (endPos < content.length) {
  252. const bestCutoff = findBestCutoff(
  253. breakPoints,
  254. targetEndPos,
  255. windowChars,
  256. 0.7,
  257. codeFences
  258. );
  259. if (bestCutoff > charPos && bestCutoff <= targetEndPos) {
  260. endPos = bestCutoff;
  261. }
  262. }
  263. if (endPos <= charPos) {
  264. endPos = Math.min(charPos + maxChars, content.length);
  265. }
  266. chunks.push({ text: content.slice(charPos, endPos), pos: charPos });
  267. if (endPos >= content.length) {
  268. break;
  269. }
  270. charPos = endPos - overlapChars;
  271. const lastChunkPos = chunks.at(-1)!.pos;
  272. if (charPos <= lastChunkPos) {
  273. charPos = endPos;
  274. }
  275. }
  276. return chunks;
  277. }
  278. // Hybrid query: strong BM25 signal detection thresholds
  279. // Skip expensive LLM expansion when top result is strong AND clearly separated from runner-up
  280. export const STRONG_SIGNAL_MIN_SCORE = 0.85;
  281. export const STRONG_SIGNAL_MIN_GAP = 0.15;
  282. // Max candidates to pass to reranker — balances quality vs latency.
  283. // 40 keeps rank 31-40 visible to the reranker (matters for recall on broad queries).
  284. export const RERANK_CANDIDATE_LIMIT = 40;
  285. /**
  286. * A typed query expansion result. Decoupled from llm.ts internal Queryable —
  287. * same shape, but store.ts owns its own public API type.
  288. *
  289. * - lex: keyword variant → routes to FTS only
  290. * - vec: semantic variant → routes to vector only
  291. * - hyde: hypothetical document → routes to vector only
  292. */
  293. export type ExpandedQuery = {
  294. type: 'lex' | 'vec' | 'hyde';
  295. query: string;
  296. /** Optional line number for error reporting (CLI parser) */
  297. line?: number;
  298. };
  299. // =============================================================================
  300. // Path utilities
  301. // =============================================================================
  302. export function homedir(): string {
  303. return HOME;
  304. }
  305. /**
  306. * Check if a path is absolute.
  307. * Supports:
  308. * - Unix paths: /path/to/file
  309. * - Windows native: C:\path or C:/path
  310. * - Git Bash: /c/path or /C/path (C-Z drives, excluding A/B floppy drives)
  311. *
  312. * Note: /c without trailing slash is treated as Unix path (directory named "c"),
  313. * while /c/ or /c/path are treated as Git Bash paths (C: drive).
  314. */
  315. export function isAbsolutePath(path: string): boolean {
  316. if (!path) return false;
  317. // Unix absolute path
  318. if (path.startsWith('/')) {
  319. // Check if it's a Git Bash style path like /c/ or /c/Users (C-Z only, not A or B)
  320. // Requires path[2] === '/' to distinguish from Unix paths like /c or /cache
  321. // Skipped on WSL where /c/ is a valid drvfs mount point, not a drive letter
  322. if (!isWSL() && path.length >= 3 && path[2] === '/') {
  323. const driveLetter = path[1];
  324. if (driveLetter && /[c-zC-Z]/.test(driveLetter)) {
  325. return true;
  326. }
  327. }
  328. // Any other path starting with / is Unix absolute
  329. return true;
  330. }
  331. // Windows native path: C:\ or C:/ (any letter A-Z)
  332. if (path.length >= 2 && /[a-zA-Z]/.test(path[0]!) && path[1] === ':') {
  333. return true;
  334. }
  335. return false;
  336. }
  337. /**
  338. * Normalize path separators to forward slashes.
  339. * Converts Windows backslashes to forward slashes.
  340. */
  341. export function normalizePathSeparators(path: string): string {
  342. return path.replace(/\\/g, '/');
  343. }
  344. /**
  345. * Detect if running inside WSL (Windows Subsystem for Linux).
  346. * On WSL, paths like /c/work/... are valid drvfs mount points, not Git Bash paths.
  347. */
  348. function isWSL(): boolean {
  349. return !!(process.env.WSL_DISTRO_NAME || process.env.WSL_INTEROP);
  350. }
  351. /**
  352. * Get the relative path from a prefix.
  353. * Returns null if path is not under prefix.
  354. * Returns empty string if path equals prefix.
  355. */
  356. export function getRelativePathFromPrefix(path: string, prefix: string): string | null {
  357. // Empty prefix is invalid
  358. if (!prefix) {
  359. return null;
  360. }
  361. const normalizedPath = normalizePathSeparators(path);
  362. const normalizedPrefix = normalizePathSeparators(prefix);
  363. // Ensure prefix ends with / for proper matching
  364. const prefixWithSlash = !normalizedPrefix.endsWith('/')
  365. ? normalizedPrefix + '/'
  366. : normalizedPrefix;
  367. // Exact match
  368. if (normalizedPath === normalizedPrefix) {
  369. return '';
  370. }
  371. // Check if path starts with prefix
  372. if (normalizedPath.startsWith(prefixWithSlash)) {
  373. return normalizedPath.slice(prefixWithSlash.length);
  374. }
  375. return null;
  376. }
  377. export function resolve(...paths: string[]): string {
  378. if (paths.length === 0) {
  379. throw new Error("resolve: at least one path segment is required");
  380. }
  381. // Normalize all paths to use forward slashes
  382. const normalizedPaths = paths.map(normalizePathSeparators);
  383. let result = '';
  384. let windowsDrive = '';
  385. // Check if first path is absolute
  386. const firstPath = normalizedPaths[0]!;
  387. if (isAbsolutePath(firstPath)) {
  388. result = firstPath;
  389. // Extract Windows drive letter if present
  390. if (firstPath.length >= 2 && /[a-zA-Z]/.test(firstPath[0]!) && firstPath[1] === ':') {
  391. windowsDrive = firstPath.slice(0, 2);
  392. result = firstPath.slice(2);
  393. } else if (!isWSL() && firstPath.startsWith('/') && firstPath.length >= 3 && firstPath[2] === '/') {
  394. // Git Bash style: /c/ -> C: (C-Z drives only, not A or B)
  395. // Skipped on WSL where /c/ is a valid drvfs mount point, not a drive letter
  396. const driveLetter = firstPath[1];
  397. if (driveLetter && /[c-zC-Z]/.test(driveLetter)) {
  398. windowsDrive = driveLetter.toUpperCase() + ':';
  399. result = firstPath.slice(2);
  400. }
  401. }
  402. } else {
  403. // Start with PWD or cwd, then append the first relative path
  404. const pwd = normalizePathSeparators(process.env.PWD || process.cwd());
  405. // Extract Windows drive from PWD if present
  406. if (pwd.length >= 2 && /[a-zA-Z]/.test(pwd[0]!) && pwd[1] === ':') {
  407. windowsDrive = pwd.slice(0, 2);
  408. result = pwd.slice(2) + '/' + firstPath;
  409. } else {
  410. result = pwd + '/' + firstPath;
  411. }
  412. }
  413. // Process remaining paths
  414. for (let i = 1; i < normalizedPaths.length; i++) {
  415. const p = normalizedPaths[i]!;
  416. if (isAbsolutePath(p)) {
  417. // Absolute path replaces everything
  418. result = p;
  419. // Update Windows drive if present
  420. if (p.length >= 2 && /[a-zA-Z]/.test(p[0]!) && p[1] === ':') {
  421. windowsDrive = p.slice(0, 2);
  422. result = p.slice(2);
  423. } else if (!isWSL() && p.startsWith('/') && p.length >= 3 && p[2] === '/') {
  424. // Git Bash style (C-Z drives only, not A or B)
  425. // Skipped on WSL where /c/ is a valid drvfs mount point, not a drive letter
  426. const driveLetter = p[1];
  427. if (driveLetter && /[c-zC-Z]/.test(driveLetter)) {
  428. windowsDrive = driveLetter.toUpperCase() + ':';
  429. result = p.slice(2);
  430. } else {
  431. windowsDrive = '';
  432. }
  433. } else {
  434. windowsDrive = '';
  435. }
  436. } else {
  437. // Relative path - append
  438. result = result + '/' + p;
  439. }
  440. }
  441. // Normalize . and .. components
  442. const parts = result.split('/').filter(Boolean);
  443. const normalized: string[] = [];
  444. for (const part of parts) {
  445. if (part === '..') {
  446. normalized.pop();
  447. } else if (part !== '.') {
  448. normalized.push(part);
  449. }
  450. }
  451. // Build final path
  452. const finalPath = '/' + normalized.join('/');
  453. // Prepend Windows drive if present
  454. if (windowsDrive) {
  455. return windowsDrive + finalPath;
  456. }
  457. return finalPath;
  458. }
  459. // Flag to indicate production mode (set by qmd.ts at startup)
  460. let _productionMode = false;
  461. export function enableProductionMode(): void {
  462. _productionMode = true;
  463. }
  464. /** Reset production mode flag — only for testing. */
  465. export function _resetProductionModeForTesting(): void {
  466. _productionMode = false;
  467. }
  468. export function getDefaultDbPath(indexName: string = "index"): string {
  469. // Always allow override via INDEX_PATH (for testing)
  470. if (process.env.INDEX_PATH) {
  471. return process.env.INDEX_PATH;
  472. }
  473. // In non-production mode (tests), require explicit path
  474. if (!_productionMode) {
  475. throw new Error(
  476. "Database path not set. Tests must set INDEX_PATH env var or use createStore() with explicit path. " +
  477. "This prevents tests from accidentally writing to the global index."
  478. );
  479. }
  480. const cacheDir = process.env.XDG_CACHE_HOME || resolve(homedir(), ".cache");
  481. const qmdCacheDir = resolve(cacheDir, "qmd");
  482. try { mkdirSync(qmdCacheDir, { recursive: true }); } catch { }
  483. return resolve(qmdCacheDir, `${indexName}.sqlite`);
  484. }
  485. export function getPwd(): string {
  486. return process.env.PWD || process.cwd();
  487. }
  488. export function getRealPath(path: string): string {
  489. try {
  490. return realpathSync(path);
  491. } catch {
  492. return resolve(path);
  493. }
  494. }
  495. // =============================================================================
  496. // Virtual Path Utilities (qmd://)
  497. // =============================================================================
  498. export type VirtualPath = {
  499. collectionName: string;
  500. path: string; // relative path within collection
  501. };
  502. /**
  503. * Normalize explicit virtual path formats to standard qmd:// format.
  504. * Only handles paths that are already explicitly virtual:
  505. * - qmd://collection/path.md (already normalized)
  506. * - qmd:////collection/path.md (extra slashes - normalize)
  507. * - //collection/path.md (missing qmd: prefix - add it)
  508. *
  509. * Does NOT handle:
  510. * - collection/path.md (bare paths - could be filesystem relative)
  511. * - :linenum suffix (should be parsed separately before calling this)
  512. */
  513. export function normalizeVirtualPath(input: string): string {
  514. let path = input.trim();
  515. // Handle qmd:// with extra slashes: qmd:////collection/path -> qmd://collection/path
  516. if (path.startsWith('qmd:')) {
  517. // Remove qmd: prefix and normalize slashes
  518. path = path.slice(4);
  519. // Remove leading slashes and re-add exactly two
  520. path = path.replace(/^\/+/, '');
  521. return `qmd://${path}`;
  522. }
  523. // Handle //collection/path (missing qmd: prefix)
  524. if (path.startsWith('//')) {
  525. path = path.replace(/^\/+/, '');
  526. return `qmd://${path}`;
  527. }
  528. // Return as-is for other cases (filesystem paths, docids, bare collection/path, etc.)
  529. return path;
  530. }
  531. /**
  532. * Parse a virtual path like "qmd://collection-name/path/to/file.md"
  533. * into its components.
  534. * Also supports collection root: "qmd://collection-name/" or "qmd://collection-name"
  535. */
  536. export function parseVirtualPath(virtualPath: string): VirtualPath | null {
  537. // Normalize the path first
  538. const normalized = normalizeVirtualPath(virtualPath);
  539. // Match: qmd://collection-name[/optional-path]
  540. // Allows: qmd://name, qmd://name/, qmd://name/path
  541. const match = normalized.match(/^qmd:\/\/([^\/]+)\/?(.*)$/);
  542. if (!match?.[1]) return null;
  543. return {
  544. collectionName: match[1],
  545. path: match[2] ?? '', // Empty string for collection root
  546. };
  547. }
  548. /**
  549. * Build a virtual path from collection name and relative path.
  550. */
  551. export function buildVirtualPath(collectionName: string, path: string): string {
  552. return `qmd://${collectionName}/${path}`;
  553. }
  554. /**
  555. * Check if a path is explicitly a virtual path.
  556. * Only recognizes explicit virtual path formats:
  557. * - qmd://collection/path.md
  558. * - //collection/path.md
  559. *
  560. * Does NOT consider bare collection/path.md as virtual - that should be
  561. * handled separately by checking if the first component is a collection name.
  562. */
  563. export function isVirtualPath(path: string): boolean {
  564. const trimmed = path.trim();
  565. // Explicit qmd:// prefix (with any number of slashes)
  566. if (trimmed.startsWith('qmd:')) return true;
  567. // //collection/path format (missing qmd: prefix)
  568. if (trimmed.startsWith('//')) return true;
  569. return false;
  570. }
  571. /**
  572. * Resolve a virtual path to absolute filesystem path.
  573. */
  574. export function resolveVirtualPath(db: Database, virtualPath: string): string | null {
  575. const parsed = parseVirtualPath(virtualPath);
  576. if (!parsed) return null;
  577. const coll = getCollectionByName(db, parsed.collectionName);
  578. if (!coll) return null;
  579. return resolve(coll.pwd, parsed.path);
  580. }
  581. /**
  582. * Convert an absolute filesystem path to a virtual path.
  583. * Returns null if the file is not in any indexed collection.
  584. */
  585. export function toVirtualPath(db: Database, absolutePath: string): string | null {
  586. // Get all collections from DB
  587. const collections = getStoreCollections(db);
  588. // Find which collection this absolute path belongs to
  589. for (const coll of collections) {
  590. if (absolutePath.startsWith(coll.path + '/') || absolutePath === coll.path) {
  591. // Extract relative path
  592. const relativePath = absolutePath.startsWith(coll.path + '/')
  593. ? absolutePath.slice(coll.path.length + 1)
  594. : '';
  595. // Verify this document exists in the database
  596. const doc = db.prepare(`
  597. SELECT d.path
  598. FROM documents d
  599. WHERE d.collection = ? AND d.path = ? AND d.active = 1
  600. LIMIT 1
  601. `).get(coll.name, relativePath) as { path: string } | null;
  602. if (doc) {
  603. return buildVirtualPath(coll.name, relativePath);
  604. }
  605. }
  606. }
  607. return null;
  608. }
  609. // =============================================================================
  610. // Database initialization
  611. // =============================================================================
  612. function createSqliteVecUnavailableError(reason: string): Error {
  613. return new Error(
  614. "sqlite-vec extension is unavailable. " +
  615. `${reason}. ` +
  616. "Install Homebrew SQLite so the sqlite-vec extension can be loaded, " +
  617. "and set BREW_PREFIX if Homebrew is installed in a non-standard location."
  618. );
  619. }
  620. function getErrorMessage(err: unknown): string {
  621. return err instanceof Error ? err.message : String(err);
  622. }
  623. export function verifySqliteVecLoaded(db: Database): void {
  624. try {
  625. const row = db.prepare(`SELECT vec_version() AS version`).get() as { version?: string } | null;
  626. if (!row?.version || typeof row.version !== "string") {
  627. throw new Error("vec_version() returned no version");
  628. }
  629. } catch (err) {
  630. const message = getErrorMessage(err);
  631. throw createSqliteVecUnavailableError(`sqlite-vec probe failed (${message})`);
  632. }
  633. }
  634. let _sqliteVecAvailable: boolean | null = null;
  635. /**
  636. * Concurrency-friendly pragma defaults applied by `initializeDatabase`.
  637. * Each entry is `{ pragma, default, envVar }` so operators can override
  638. * any one knob via env without code changes.
  639. *
  640. * Defaults are tuned for the Oivo fleet shape — many concurrent MCP
  641. * processes (one per agent session) sharing a single ~10 GB index that
  642. * a 30-minute cron runs `qmd embed` against. See issue i-6sw24v09 for
  643. * the failure mode this prevents.
  644. */
  645. const CONCURRENCY_PRAGMAS: Array<{ pragma: string; defaultValue: string | number; envVar: string }> = [
  646. { pragma: "busy_timeout", defaultValue: 30000, envVar: "QMD_SQLITE_BUSY_TIMEOUT_MS" },
  647. { pragma: "synchronous", defaultValue: "NORMAL", envVar: "QMD_SQLITE_SYNCHRONOUS" },
  648. { pragma: "temp_store", defaultValue: "MEMORY", envVar: "QMD_SQLITE_TEMP_STORE" },
  649. { pragma: "cache_size", defaultValue: -65536, envVar: "QMD_SQLITE_CACHE_SIZE" }, // ~64 MiB
  650. { pragma: "mmap_size", defaultValue: 268435456, envVar: "QMD_SQLITE_MMAP_SIZE" }, // 256 MiB
  651. { pragma: "wal_autocheckpoint", defaultValue: 1000, envVar: "QMD_SQLITE_WAL_AUTOCHECKPOINT" },
  652. ];
  653. /**
  654. * Apply concurrency pragmas with env-var override support. Exported for
  655. * unit tests; consumers should rely on `initializeDatabase` instead.
  656. */
  657. export function applyConcurrencyPragmas(db: Database): void {
  658. for (const { pragma, defaultValue, envVar } of CONCURRENCY_PRAGMAS) {
  659. const override = process.env[envVar];
  660. let value: string | number = defaultValue;
  661. if (override !== undefined && override !== "") {
  662. // Numeric overrides parse as base-10 integers (also accepts negatives
  663. // for cache_size). Non-numeric overrides pass through as identifiers
  664. // (e.g. NORMAL, FULL, MEMORY) — SQLite validates them.
  665. const numericPragmas = new Set(["busy_timeout", "cache_size", "mmap_size", "wal_autocheckpoint"]);
  666. if (numericPragmas.has(pragma)) {
  667. const parsed = parseInt(override, 10);
  668. if (Number.isFinite(parsed)) value = parsed;
  669. } else {
  670. value = override;
  671. }
  672. }
  673. try {
  674. db.exec(`PRAGMA ${pragma} = ${value}`);
  675. } catch (err) {
  676. // Don't blow up on pragma failure — log + carry on. SQLite without
  677. // mmap support, for example, simply ignores mmap_size silently on
  678. // some builds, but a strict build can throw.
  679. const msg = err instanceof Error ? err.message : String(err);
  680. console.warn(`[qmd] PRAGMA ${pragma} = ${value} failed: ${msg}`);
  681. }
  682. }
  683. }
  684. function initializeDatabase(db: Database): void {
  685. try {
  686. loadSqliteVec(db);
  687. verifySqliteVecLoaded(db);
  688. _sqliteVecAvailable = true;
  689. } catch (err) {
  690. // sqlite-vec is optional — vector search won't work but FTS is fine
  691. _sqliteVecAvailable = false;
  692. console.warn(getErrorMessage(err));
  693. }
  694. db.exec("PRAGMA journal_mode = WAL");
  695. db.exec("PRAGMA foreign_keys = ON");
  696. // Concurrency tuning — prevents reader timeouts during long writer windows
  697. // such as `qmd embed` (often 6-30 minutes on the Oivo fleet) which would
  698. // otherwise saturate the default 5s busy_timeout from better-sqlite3 and
  699. // surface as MCP transport timeouts in concurrent `qmd_query`/`qmd_status`
  700. // calls. See issue i-6sw24v09 for the empirical trace.
  701. //
  702. // - busy_timeout (default 30000 ms): readers wait through writer-held
  703. // checkpoints instead of failing fast with SQLITE_BUSY.
  704. // - synchronous=NORMAL: WAL-safe (still durable across crashes), avoids
  705. // the FULL fsync per transaction that compounds embed runtime.
  706. // - temp_store=MEMORY: keep FTS5 + vec sort scratch in RAM, not /tmp.
  707. // - cache_size: ~64 MiB per-connection page cache. Negative kibibyte
  708. // form is the canonical SQLite idiom (positive = pages, negative = KiB).
  709. // - mmap_size: 256 MiB memory-mapped reads for the 10 GB index — cheap
  710. // on Linux (lazy paging), no effect on non-mmap'd syscall fallback.
  711. // - wal_autocheckpoint: keep WAL bounded. Default 1000 pages is fine
  712. // but setting it explicitly prevents drift when callers tune globally.
  713. //
  714. // Each pragma is overridable via env so operators can tune without a
  715. // code change; values must parse as base-10 integers or are skipped.
  716. applyConcurrencyPragmas(db);
  717. // Drop legacy tables that are now managed in YAML
  718. db.exec(`DROP TABLE IF EXISTS path_contexts`);
  719. db.exec(`DROP TABLE IF EXISTS collections`);
  720. // Content-addressable storage - the source of truth for document content
  721. db.exec(`
  722. CREATE TABLE IF NOT EXISTS content (
  723. hash TEXT PRIMARY KEY,
  724. doc TEXT NOT NULL,
  725. created_at TEXT NOT NULL
  726. )
  727. `);
  728. // Documents table - file system layer mapping virtual paths to content hashes
  729. // Collections are now managed in ~/.config/qmd/index.yml
  730. db.exec(`
  731. CREATE TABLE IF NOT EXISTS documents (
  732. id INTEGER PRIMARY KEY AUTOINCREMENT,
  733. collection TEXT NOT NULL,
  734. path TEXT NOT NULL,
  735. title TEXT NOT NULL,
  736. hash TEXT NOT NULL,
  737. created_at TEXT NOT NULL,
  738. modified_at TEXT NOT NULL,
  739. active INTEGER NOT NULL DEFAULT 1,
  740. FOREIGN KEY (hash) REFERENCES content(hash) ON DELETE CASCADE,
  741. UNIQUE(collection, path)
  742. )
  743. `);
  744. db.exec(`CREATE INDEX IF NOT EXISTS idx_documents_collection ON documents(collection, active)`);
  745. db.exec(`CREATE INDEX IF NOT EXISTS idx_documents_hash ON documents(hash)`);
  746. db.exec(`CREATE INDEX IF NOT EXISTS idx_documents_path ON documents(path, active)`);
  747. // Cache table for LLM API calls
  748. db.exec(`
  749. CREATE TABLE IF NOT EXISTS llm_cache (
  750. hash TEXT PRIMARY KEY,
  751. result TEXT NOT NULL,
  752. created_at TEXT NOT NULL
  753. )
  754. `);
  755. // Content vectors
  756. const cvInfo = db.prepare(`PRAGMA table_info(content_vectors)`).all() as { name: string }[];
  757. const hasSeqColumn = cvInfo.some(col => col.name === 'seq');
  758. if (cvInfo.length > 0 && !hasSeqColumn) {
  759. db.exec(`DROP TABLE IF EXISTS content_vectors`);
  760. db.exec(`DROP TABLE IF EXISTS vectors_vec`);
  761. }
  762. db.exec(`
  763. CREATE TABLE IF NOT EXISTS content_vectors (
  764. hash TEXT NOT NULL,
  765. seq INTEGER NOT NULL DEFAULT 0,
  766. pos INTEGER NOT NULL DEFAULT 0,
  767. model TEXT NOT NULL,
  768. embedded_at TEXT NOT NULL,
  769. PRIMARY KEY (hash, seq)
  770. )
  771. `);
  772. // How many chunks a document was split into, recorded at chunk time (i-xeekgx6h).
  773. //
  774. // Without it "is this document embedded?" can only be asked as "does chunk 0
  775. // exist?", and chunk inserts are per-chunk best-effort — so a document whose
  776. // chunk 0 embedded and whose chunk 3 did not looked complete forever, and the
  777. // 30-minute embed cron never picked it up again. A partially embedded document
  778. // does not fail loudly: it answers semantic search with plausible-but-incomplete
  779. // results.
  780. //
  781. // Written BEFORE the chunks are embedded, not after, so that a run which dies
  782. // mid-document still leaves the expectation behind for the next run to compare
  783. // against.
  784. db.exec(`
  785. CREATE TABLE IF NOT EXISTS document_chunk_counts (
  786. hash TEXT PRIMARY KEY,
  787. chunks INTEGER NOT NULL,
  788. chunked_at TEXT NOT NULL
  789. )
  790. `);
  791. // Store collections — makes the DB self-contained (no external config needed)
  792. db.exec(`
  793. CREATE TABLE IF NOT EXISTS store_collections (
  794. name TEXT PRIMARY KEY,
  795. path TEXT NOT NULL,
  796. pattern TEXT NOT NULL DEFAULT '**/*.md',
  797. ignore_patterns TEXT,
  798. include_by_default INTEGER DEFAULT 1,
  799. update_command TEXT,
  800. context TEXT
  801. )
  802. `);
  803. // Store config — key-value metadata (e.g. config_hash for sync optimization)
  804. db.exec(`
  805. CREATE TABLE IF NOT EXISTS store_config (
  806. key TEXT PRIMARY KEY,
  807. value TEXT
  808. )
  809. `);
  810. // FTS - index filepath (collection/path), title, and content
  811. db.exec(`
  812. CREATE VIRTUAL TABLE IF NOT EXISTS documents_fts USING fts5(
  813. filepath, title, body,
  814. tokenize='porter unicode61'
  815. )
  816. `);
  817. // Triggers to keep FTS in sync
  818. db.exec(`
  819. CREATE TRIGGER IF NOT EXISTS documents_ai AFTER INSERT ON documents
  820. WHEN new.active = 1
  821. BEGIN
  822. INSERT INTO documents_fts(rowid, filepath, title, body)
  823. SELECT
  824. new.id,
  825. new.collection || '/' || new.path,
  826. new.title,
  827. (SELECT doc FROM content WHERE hash = new.hash)
  828. WHERE new.active = 1;
  829. END
  830. `);
  831. db.exec(`
  832. CREATE TRIGGER IF NOT EXISTS documents_ad AFTER DELETE ON documents BEGIN
  833. DELETE FROM documents_fts WHERE rowid = old.id;
  834. END
  835. `);
  836. db.exec(`
  837. CREATE TRIGGER IF NOT EXISTS documents_au AFTER UPDATE ON documents
  838. BEGIN
  839. -- Delete from FTS if no longer active
  840. DELETE FROM documents_fts WHERE rowid = old.id AND new.active = 0;
  841. -- Update FTS if still/newly active
  842. INSERT OR REPLACE INTO documents_fts(rowid, filepath, title, body)
  843. SELECT
  844. new.id,
  845. new.collection || '/' || new.path,
  846. new.title,
  847. (SELECT doc FROM content WHERE hash = new.hash)
  848. WHERE new.active = 1;
  849. END
  850. `);
  851. }
  852. // =============================================================================
  853. // Store Collections — DB accessor functions
  854. // =============================================================================
  855. type StoreCollectionRow = {
  856. name: string;
  857. path: string;
  858. pattern: string;
  859. ignore_patterns: string | null;
  860. include_by_default: number;
  861. update_command: string | null;
  862. context: string | null;
  863. };
  864. function rowToNamedCollection(row: StoreCollectionRow): NamedCollection {
  865. return {
  866. name: row.name,
  867. path: row.path,
  868. pattern: row.pattern,
  869. ...(row.ignore_patterns ? { ignore: JSON.parse(row.ignore_patterns) as string[] } : {}),
  870. ...(row.include_by_default === 0 ? { includeByDefault: false } : {}),
  871. ...(row.update_command ? { update: row.update_command } : {}),
  872. ...(row.context ? { context: JSON.parse(row.context) as ContextMap } : {}),
  873. };
  874. }
  875. export function getStoreCollections(db: Database): NamedCollection[] {
  876. const rows = db.prepare(`SELECT * FROM store_collections`).all() as StoreCollectionRow[];
  877. return rows.map(rowToNamedCollection);
  878. }
  879. export function getStoreCollection(db: Database, name: string): NamedCollection | null {
  880. const row = db.prepare(`SELECT * FROM store_collections WHERE name = ?`).get(name) as StoreCollectionRow | null | undefined;
  881. if (row == null) return null;
  882. return rowToNamedCollection(row);
  883. }
  884. export function getStoreGlobalContext(db: Database): string | undefined {
  885. const row = db.prepare(`SELECT value FROM store_config WHERE key = 'global_context'`).get() as { value: string } | null | undefined;
  886. if (row == null) return undefined;
  887. return row.value || undefined;
  888. }
  889. export function getStoreContexts(db: Database): Array<{ collection: string; path: string; context: string }> {
  890. const results: Array<{ collection: string; path: string; context: string }> = [];
  891. // Global context
  892. const globalCtx = getStoreGlobalContext(db);
  893. if (globalCtx) {
  894. results.push({ collection: "*", path: "/", context: globalCtx });
  895. }
  896. // Collection contexts
  897. const rows = db.prepare(`SELECT name, context FROM store_collections WHERE context IS NOT NULL`).all() as { name: string; context: string }[];
  898. for (const row of rows) {
  899. const ctxMap = JSON.parse(row.context) as ContextMap;
  900. for (const [path, context] of Object.entries(ctxMap)) {
  901. results.push({ collection: row.name, path, context });
  902. }
  903. }
  904. return results;
  905. }
  906. export function upsertStoreCollection(db: Database, name: string, collection: Omit<Collection, 'pattern'> & { pattern?: string }): void {
  907. db.prepare(`
  908. INSERT INTO store_collections (name, path, pattern, ignore_patterns, include_by_default, update_command, context)
  909. VALUES (?, ?, ?, ?, ?, ?, ?)
  910. ON CONFLICT(name) DO UPDATE SET
  911. path = excluded.path,
  912. pattern = excluded.pattern,
  913. ignore_patterns = excluded.ignore_patterns,
  914. include_by_default = excluded.include_by_default,
  915. update_command = excluded.update_command,
  916. context = excluded.context
  917. `).run(
  918. name,
  919. collection.path,
  920. collection.pattern || '**/*.md',
  921. collection.ignore ? JSON.stringify(collection.ignore) : null,
  922. collection.includeByDefault === false ? 0 : 1,
  923. collection.update || null,
  924. collection.context ? JSON.stringify(collection.context) : null,
  925. );
  926. }
  927. export function deleteStoreCollection(db: Database, name: string): boolean {
  928. const result = db.prepare(`DELETE FROM store_collections WHERE name = ?`).run(name);
  929. return result.changes > 0;
  930. }
  931. export function renameStoreCollection(db: Database, oldName: string, newName: string): boolean {
  932. // Check target doesn't exist
  933. const existing = db.prepare(`SELECT name FROM store_collections WHERE name = ?`).get(newName) as { name: string } | null | undefined;
  934. if (existing != null) {
  935. throw new Error(`Collection '${newName}' already exists`);
  936. }
  937. const result = db.prepare(`UPDATE store_collections SET name = ? WHERE name = ?`).run(newName, oldName);
  938. return result.changes > 0;
  939. }
  940. export function updateStoreContext(db: Database, collectionName: string, path: string, text: string): boolean {
  941. const row = db.prepare(`SELECT context FROM store_collections WHERE name = ?`).get(collectionName) as { context: string | null } | null | undefined;
  942. if (row == null) return false;
  943. const ctxMap: ContextMap = row.context ? JSON.parse(row.context) : {};
  944. ctxMap[path] = text;
  945. db.prepare(`UPDATE store_collections SET context = ? WHERE name = ?`).run(JSON.stringify(ctxMap), collectionName);
  946. return true;
  947. }
  948. export function removeStoreContext(db: Database, collectionName: string, path: string): boolean {
  949. const row = db.prepare(`SELECT context FROM store_collections WHERE name = ?`).get(collectionName) as { context: string | null } | null | undefined;
  950. if (row == null) return false;
  951. if (!row.context) return false;
  952. const ctxMap: ContextMap = JSON.parse(row.context);
  953. if (!(path in ctxMap)) return false;
  954. delete ctxMap[path];
  955. const newCtx = Object.keys(ctxMap).length > 0 ? JSON.stringify(ctxMap) : null;
  956. db.prepare(`UPDATE store_collections SET context = ? WHERE name = ?`).run(newCtx, collectionName);
  957. return true;
  958. }
  959. export function setStoreGlobalContext(db: Database, value: string | undefined): void {
  960. if (value === undefined) {
  961. db.prepare(`DELETE FROM store_config WHERE key = 'global_context'`).run();
  962. } else {
  963. db.prepare(`INSERT INTO store_config (key, value) VALUES ('global_context', ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value`).run(value);
  964. }
  965. }
  966. /**
  967. * Sync external config (YAML/inline) into SQLite store_collections.
  968. * External config always wins. Skips sync if config hash hasn't changed.
  969. */
  970. export function syncConfigToDb(db: Database, config: CollectionConfig): void {
  971. // Check config hash — skip sync if unchanged
  972. const configJson = JSON.stringify(config);
  973. const hash = createHash('sha256').update(configJson).digest('hex');
  974. const existingHash = db.prepare(`SELECT value FROM store_config WHERE key = 'config_hash'`).get() as { value: string } | null | undefined;
  975. if (existingHash != null && existingHash.value === hash) {
  976. return; // Config unchanged, skip sync
  977. }
  978. // Sync collections
  979. const configNames = new Set(Object.keys(config.collections));
  980. for (const [name, coll] of Object.entries(config.collections)) {
  981. upsertStoreCollection(db, name, coll);
  982. }
  983. // Delete collections not in config
  984. const dbCollections = db.prepare(`SELECT name FROM store_collections`).all() as { name: string }[];
  985. for (const row of dbCollections) {
  986. if (!configNames.has(row.name)) {
  987. db.prepare(`DELETE FROM store_collections WHERE name = ?`).run(row.name);
  988. }
  989. }
  990. // Sync global context
  991. if (config.global_context !== undefined) {
  992. setStoreGlobalContext(db, config.global_context);
  993. } else {
  994. setStoreGlobalContext(db, undefined);
  995. }
  996. // Save config hash
  997. db.prepare(`INSERT INTO store_config (key, value) VALUES ('config_hash', ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value`).run(hash);
  998. }
  999. export function isSqliteVecAvailable(): boolean {
  1000. return _sqliteVecAvailable === true;
  1001. }
  1002. function ensureVecTableInternal(db: Database, dimensions: number): void {
  1003. if (!_sqliteVecAvailable) {
  1004. throw new Error("sqlite-vec is not available. Vector operations require a SQLite build with extension loading support.");
  1005. }
  1006. const tableInfo = db.prepare(`SELECT sql FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get() as { sql: string } | null;
  1007. if (tableInfo) {
  1008. const match = tableInfo.sql.match(/float\[(\d+)\]/);
  1009. const hasHashSeq = tableInfo.sql.includes('hash_seq');
  1010. const hasCosine = tableInfo.sql.includes('distance_metric=cosine');
  1011. const existingDims = match?.[1] ? parseInt(match[1], 10) : null;
  1012. if (existingDims === dimensions && hasHashSeq && hasCosine) return;
  1013. if (existingDims !== null && existingDims !== dimensions) {
  1014. throw new Error(
  1015. `Embedding dimension mismatch: existing vectors are ${existingDims}d but the current model produces ${dimensions}d. ` +
  1016. `Run 'qmd embed -f' to re-embed with the new model.`
  1017. );
  1018. }
  1019. db.exec("DROP TABLE IF EXISTS vectors_vec");
  1020. }
  1021. db.exec(`CREATE VIRTUAL TABLE vectors_vec USING vec0(hash_seq TEXT PRIMARY KEY, embedding float[${dimensions}] distance_metric=cosine)`);
  1022. }
  1023. // =============================================================================
  1024. // Store Factory
  1025. // =============================================================================
  1026. export type Store = {
  1027. db: Database;
  1028. dbPath: string;
  1029. /** Optional LlamaCpp instance for this store (overrides the global singleton) */
  1030. llm?: LlamaCpp;
  1031. close: () => void;
  1032. ensureVecTable: (dimensions: number) => void;
  1033. // Index health
  1034. getHashesNeedingEmbedding: () => number;
  1035. getIndexHealth: () => IndexHealthInfo;
  1036. getStatus: () => IndexStatus;
  1037. // Caching
  1038. getCacheKey: typeof getCacheKey;
  1039. getCachedResult: (cacheKey: string) => string | null;
  1040. setCachedResult: (cacheKey: string, result: string) => void;
  1041. clearCache: () => void;
  1042. // Cleanup and maintenance
  1043. deleteLLMCache: () => number;
  1044. deleteInactiveDocuments: () => number;
  1045. cleanupOrphanedContent: () => number;
  1046. cleanupOrphanedVectors: () => number;
  1047. vacuumDatabase: () => void;
  1048. // Context
  1049. getContextForFile: (filepath: string) => string | null;
  1050. getContextForPath: (collectionName: string, path: string) => string | null;
  1051. getCollectionByName: (name: string) => { name: string; pwd: string; glob_pattern: string } | null;
  1052. getCollectionsWithoutContext: () => { name: string; pwd: string; doc_count: number }[];
  1053. getTopLevelPathsWithoutContext: (collectionName: string) => string[];
  1054. // Virtual paths
  1055. parseVirtualPath: typeof parseVirtualPath;
  1056. buildVirtualPath: typeof buildVirtualPath;
  1057. isVirtualPath: typeof isVirtualPath;
  1058. resolveVirtualPath: (virtualPath: string) => string | null;
  1059. toVirtualPath: (absolutePath: string) => string | null;
  1060. // Search
  1061. searchFTS: (query: string, limit?: number, collectionName?: string) => SearchResult[];
  1062. searchVec: (query: string, model: string, limit?: number, collectionName?: string, session?: ILLMSession, precomputedEmbedding?: number[], embedProvider?: EmbeddingProvider) => Promise<SearchResult[]>;
  1063. // Query expansion & reranking
  1064. expandQuery: (query: string, model?: string, intent?: string) => Promise<ExpandedQuery[]>;
  1065. rerank: (query: string, documents: { file: string; text: string }[], model?: string, intent?: string) => Promise<{ file: string; score: number }[]>;
  1066. // Document retrieval
  1067. findDocument: (filename: string, options?: { includeBody?: boolean }) => DocumentResult | DocumentNotFound;
  1068. getDocumentBody: (doc: DocumentResult | { filepath: string }, fromLine?: number, maxLines?: number) => string | null;
  1069. findDocuments: (pattern: string, options?: { includeBody?: boolean; maxBytes?: number }) => { docs: MultiGetResult[]; errors: string[] };
  1070. // Fuzzy matching and docid lookup
  1071. findSimilarFiles: (query: string, maxDistance?: number, limit?: number) => string[];
  1072. matchFilesByGlob: (pattern: string) => { filepath: string; displayPath: string; bodyLength: number }[];
  1073. findDocumentByDocid: (docid: string) => { filepath: string; hash: string } | null;
  1074. // Document indexing operations
  1075. insertContent: (hash: string, content: string, createdAt: string) => void;
  1076. insertDocument: (collectionName: string, path: string, title: string, hash: string, createdAt: string, modifiedAt: string) => void;
  1077. findActiveDocument: (collectionName: string, path: string) => { id: number; hash: string; title: string } | null;
  1078. updateDocumentTitle: (documentId: number, title: string, modifiedAt: string) => void;
  1079. updateDocument: (documentId: number, title: string, hash: string, modifiedAt: string) => void;
  1080. deactivateDocument: (collectionName: string, path: string) => void;
  1081. getActiveDocumentPaths: (collectionName: string) => string[];
  1082. // Vector/embedding operations
  1083. getHashesForEmbedding: () => { hash: string; body: string; path: string }[];
  1084. clearAllEmbeddings: () => void;
  1085. insertEmbedding: (hash: string, seq: number, pos: number, embedding: Float32Array, model: string, embeddedAt: string) => void;
  1086. };
  1087. // =============================================================================
  1088. // Reindex & Embed — pure-logic functions for SDK and CLI
  1089. // =============================================================================
  1090. export type ReindexProgress = {
  1091. file: string;
  1092. current: number;
  1093. total: number;
  1094. };
  1095. export type ReindexResult = {
  1096. indexed: number;
  1097. updated: number;
  1098. unchanged: number;
  1099. removed: number;
  1100. orphanedCleaned: number;
  1101. };
  1102. /**
  1103. * Re-index a single collection by scanning the filesystem and updating the database.
  1104. * Pure function — no console output, no db lifecycle management.
  1105. */
  1106. export async function reindexCollection(
  1107. store: Store,
  1108. collectionPath: string,
  1109. globPattern: string,
  1110. collectionName: string,
  1111. options?: {
  1112. ignorePatterns?: string[];
  1113. onProgress?: (info: ReindexProgress) => void;
  1114. }
  1115. ): Promise<ReindexResult> {
  1116. const db = store.db;
  1117. const now = new Date().toISOString();
  1118. const excludeDirs = ["node_modules", ".git", ".cache", "vendor", "dist", "build"];
  1119. const allIgnore = [
  1120. ...excludeDirs.map(d => `**/${d}/**`),
  1121. ...(options?.ignorePatterns || []),
  1122. ];
  1123. const allFiles: string[] = await fastGlob(globPattern, {
  1124. cwd: collectionPath,
  1125. onlyFiles: true,
  1126. followSymbolicLinks: false,
  1127. dot: false,
  1128. ignore: allIgnore,
  1129. });
  1130. // Filter hidden files/folders
  1131. const files = allFiles.filter(file => {
  1132. const parts = file.split("/");
  1133. return !parts.some(part => part.startsWith("."));
  1134. });
  1135. const total = files.length;
  1136. let indexed = 0, updated = 0, unchanged = 0, processed = 0;
  1137. const seenPaths = new Set<string>();
  1138. for (const relativeFile of files) {
  1139. const filepath = getRealPath(resolve(collectionPath, relativeFile));
  1140. const path = handelize(relativeFile);
  1141. seenPaths.add(path);
  1142. let content: string;
  1143. try {
  1144. content = readFileSync(filepath, "utf-8");
  1145. } catch {
  1146. processed++;
  1147. options?.onProgress?.({ file: relativeFile, current: processed, total });
  1148. continue;
  1149. }
  1150. if (!content.trim()) {
  1151. processed++;
  1152. continue;
  1153. }
  1154. const hash = await hashContent(content);
  1155. const title = extractTitle(content, relativeFile);
  1156. const existing = findActiveDocument(db, collectionName, path);
  1157. if (existing) {
  1158. if (existing.hash === hash) {
  1159. if (existing.title !== title) {
  1160. updateDocumentTitle(db, existing.id, title, now);
  1161. updated++;
  1162. } else {
  1163. unchanged++;
  1164. }
  1165. } else {
  1166. insertContent(db, hash, content, now);
  1167. const stat = statSync(filepath);
  1168. updateDocument(db, existing.id, title, hash,
  1169. stat ? new Date(stat.mtime).toISOString() : now);
  1170. updated++;
  1171. }
  1172. } else {
  1173. indexed++;
  1174. insertContent(db, hash, content, now);
  1175. const stat = statSync(filepath);
  1176. insertDocument(db, collectionName, path, title, hash,
  1177. stat ? new Date(stat.birthtime).toISOString() : now,
  1178. stat ? new Date(stat.mtime).toISOString() : now);
  1179. }
  1180. processed++;
  1181. options?.onProgress?.({ file: relativeFile, current: processed, total });
  1182. }
  1183. // Deactivate documents that no longer exist
  1184. const allActive = getActiveDocumentPaths(db, collectionName);
  1185. let removed = 0;
  1186. for (const path of allActive) {
  1187. if (!seenPaths.has(path)) {
  1188. deactivateDocument(db, collectionName, path);
  1189. removed++;
  1190. }
  1191. }
  1192. const orphanedCleaned = cleanupOrphanedContent(db);
  1193. return { indexed, updated, unchanged, removed, orphanedCleaned };
  1194. }
  1195. export type EmbedProgress = {
  1196. chunksEmbedded: number;
  1197. totalChunks: number;
  1198. bytesProcessed: number;
  1199. totalBytes: number;
  1200. /** Chunks that were sent to the provider and came back without an embedding. */
  1201. errors: number;
  1202. /**
  1203. * Chunks the run gave up on WITHOUT sending them (abort / expired session).
  1204. * Kept separate from `errors` so "we stopped early" can never be reported as
  1205. * "N chunks failed" — see `generateEmbeddings` (i-yghj098h).
  1206. */
  1207. skipped?: number;
  1208. };
  1209. export type EmbedResult = {
  1210. docsProcessed: number;
  1211. chunksEmbedded: number;
  1212. /** Attempted-and-failed chunks only. Never includes un-attempted ones. */
  1213. errors: number;
  1214. /** Un-attempted chunks left behind by an early abort. */
  1215. skipped?: number;
  1216. durationMs: number;
  1217. };
  1218. export type EmbedOptions = {
  1219. force?: boolean;
  1220. model?: string;
  1221. maxDocsPerBatch?: number;
  1222. maxBatchBytes?: number;
  1223. chunkStrategy?: ChunkStrategy;
  1224. onProgress?: (info: EmbedProgress) => void;
  1225. /**
  1226. * Required provider for embedding work. Embeddings are routed through the
  1227. * approved commercial HTTPS API. The provider's `getModelId()` is verified against existing
  1228. * `content_vectors.model` rows; mismatch throws unless `force` is set.
  1229. *
  1230. * When omitted, learned work reaches the fail-closed compatibility adapter
  1231. * and returns typed HOLD.
  1232. */
  1233. embedProvider?: EmbeddingProvider;
  1234. /**
  1235. * Optional collection name filter (i-ofojj7dy). When set, only content
  1236. * hashes that have at least one document in this collection are embedded.
  1237. * `getPendingEmbeddingDocs` filters at the SQL level. Callers are expected
  1238. * to validate the name against `listCollections(db)` first; passing an
  1239. * unknown name yields zero pending docs (no work, no error).
  1240. */
  1241. collection?: string;
  1242. };
  1243. type PendingEmbeddingDoc = {
  1244. hash: string;
  1245. path: string;
  1246. bytes: number;
  1247. collection: string;
  1248. };
  1249. type EmbeddingDoc = PendingEmbeddingDoc & {
  1250. body: string;
  1251. };
  1252. type ChunkItem = {
  1253. hash: string;
  1254. title: string;
  1255. text: string;
  1256. seq: number;
  1257. pos: number;
  1258. tokens: number;
  1259. bytes: number;
  1260. };
  1261. function validatePositiveIntegerOption(name: string, value: number | undefined, fallback: number): number {
  1262. if (value === undefined) return fallback;
  1263. if (!Number.isInteger(value) || value < 1) {
  1264. throw new Error(`${name} must be a positive integer`);
  1265. }
  1266. return value;
  1267. }
  1268. function resolveEmbedOptions(options?: EmbedOptions): Required<Pick<EmbedOptions, "maxDocsPerBatch" | "maxBatchBytes">> {
  1269. return {
  1270. maxDocsPerBatch: validatePositiveIntegerOption("maxDocsPerBatch", options?.maxDocsPerBatch, DEFAULT_EMBED_MAX_DOCS_PER_BATCH),
  1271. maxBatchBytes: validatePositiveIntegerOption("maxBatchBytes", options?.maxBatchBytes, DEFAULT_EMBED_MAX_BATCH_BYTES),
  1272. };
  1273. }
  1274. /**
  1275. * What "still needs embedding" means, in ONE place (i-xeekgx6h).
  1276. *
  1277. * Two conditions, and the second is the one that was missing:
  1278. *
  1279. * 1. no vector at seq 0 — the document was never embedded at all. A run that
  1280. * fails outright leaves this true, which is why whole-document failures
  1281. * always self-healed on the next pass.
  1282. * 2. fewer vectors than the document has chunks — it was embedded PARTIALLY.
  1283. * Chunk inserts are per-chunk best-effort ("so a single bad chunk doesn't
  1284. * drag down the rest"), so an upstream that fails mid-document leaves
  1285. * chunk 0 present and later chunks missing. Under condition 1 alone that
  1286. * document is complete forever and its missing chunks are unreachable.
  1287. *
  1288. * Documents embedded BEFORE this table existed have no `document_chunk_counts`
  1289. * row, so condition 2 cannot fire for them and they are NOT re-embedded — a
  1290. * whole-corpus re-embed on upgrade would cost more than the holes it heals.
  1291. * They are repaired the next time they are re-chunked, or by `--force`.
  1292. *
  1293. * Requires the caller's FROM clause to alias documents as `d`, and to LEFT JOIN
  1294. * both `content_vectors v ... AND v.seq = 0` and `document_chunk_counts dcc`.
  1295. * Keeping the predicate here rather than inline is deliberate: this bug existed
  1296. * because the list query and the count query each carried their own copy.
  1297. */
  1298. const PENDING_EMBEDDING_JOINS = `
  1299. LEFT JOIN content_vectors v ON d.hash = v.hash AND v.seq = 0
  1300. LEFT JOIN document_chunk_counts dcc ON dcc.hash = d.hash`;
  1301. const PENDING_EMBEDDING_PREDICATE = `(
  1302. v.hash IS NULL
  1303. OR (
  1304. dcc.chunks IS NOT NULL
  1305. AND (SELECT COUNT(*) FROM content_vectors cv WHERE cv.hash = d.hash) < dcc.chunks
  1306. )
  1307. )`;
  1308. /**
  1309. * Record how many chunks a document was split into. Called at chunk time, before
  1310. * the embeddings are attempted — see {@link PENDING_EMBEDDING_PREDICATE}.
  1311. * `INSERT OR REPLACE` because a chunkStrategy change legitimately changes the
  1312. * count, and the newest chunking is the one the vectors will match.
  1313. */
  1314. export function recordDocumentChunkCount(db: Database, hash: string, chunks: number, chunkedAt: string): void {
  1315. db.prepare(
  1316. `INSERT OR REPLACE INTO document_chunk_counts (hash, chunks, chunked_at) VALUES (?, ?, ?)`,
  1317. ).run(hash, chunks, chunkedAt);
  1318. }
  1319. function getPendingEmbeddingDocs(db: Database, collection?: string): PendingEmbeddingDoc[] {
  1320. // `MIN(d.collection)` deterministically picks one collection per hash when
  1321. // the same content is indexed in multiple collections (SQLite tie-breaks
  1322. // alphabetically). The identical bytes produce identical chunks regardless
  1323. // of which collection wins; the chunkStrategy lookup still resolves via
  1324. // that collection's YAML config. See Phase 2 design notes (i-bud0h8vu).
  1325. //
  1326. // i-ofojj7dy — when a collection name is supplied, filter rows BEFORE the
  1327. // GROUP BY so we only emit hashes whose documents include that collection.
  1328. // Other collections sharing the same content hash still benefit from any
  1329. // embeddings generated for the canonical owner (content_vectors is keyed
  1330. // by hash, not by collection).
  1331. if (collection !== undefined) {
  1332. return db.prepare(`
  1333. SELECT d.hash, MIN(d.path) as path, MIN(d.collection) as collection, length(CAST(c.doc AS BLOB)) as bytes
  1334. FROM documents d
  1335. JOIN content c ON d.hash = c.hash${PENDING_EMBEDDING_JOINS}
  1336. WHERE d.active = 1 AND ${PENDING_EMBEDDING_PREDICATE} AND d.collection = ?
  1337. GROUP BY d.hash
  1338. ORDER BY MIN(d.path)
  1339. `).all(collection) as PendingEmbeddingDoc[];
  1340. }
  1341. return db.prepare(`
  1342. SELECT d.hash, MIN(d.path) as path, MIN(d.collection) as collection, length(CAST(c.doc AS BLOB)) as bytes
  1343. FROM documents d
  1344. JOIN content c ON d.hash = c.hash${PENDING_EMBEDDING_JOINS}
  1345. WHERE d.active = 1 AND ${PENDING_EMBEDDING_PREDICATE}
  1346. GROUP BY d.hash
  1347. ORDER BY MIN(d.path)
  1348. `).all() as PendingEmbeddingDoc[];
  1349. }
  1350. function buildEmbeddingBatches(
  1351. docs: PendingEmbeddingDoc[],
  1352. maxDocsPerBatch: number,
  1353. maxBatchBytes: number,
  1354. ): PendingEmbeddingDoc[][] {
  1355. const batches: PendingEmbeddingDoc[][] = [];
  1356. let currentBatch: PendingEmbeddingDoc[] = [];
  1357. let currentBytes = 0;
  1358. for (const doc of docs) {
  1359. const docBytes = Math.max(0, doc.bytes);
  1360. const wouldExceedDocs = currentBatch.length >= maxDocsPerBatch;
  1361. const wouldExceedBytes = currentBatch.length > 0 && (currentBytes + docBytes) > maxBatchBytes;
  1362. if (wouldExceedDocs || wouldExceedBytes) {
  1363. batches.push(currentBatch);
  1364. currentBatch = [];
  1365. currentBytes = 0;
  1366. }
  1367. currentBatch.push(doc);
  1368. currentBytes += docBytes;
  1369. }
  1370. if (currentBatch.length > 0) {
  1371. batches.push(currentBatch);
  1372. }
  1373. return batches;
  1374. }
  1375. function getEmbeddingDocsForBatch(db: Database, batch: PendingEmbeddingDoc[]): EmbeddingDoc[] {
  1376. if (batch.length === 0) return [];
  1377. const placeholders = batch.map(() => "?").join(",");
  1378. const rows = db.prepare(`
  1379. SELECT hash, doc as body
  1380. FROM content
  1381. WHERE hash IN (${placeholders})
  1382. `).all(...batch.map(doc => doc.hash)) as { hash: string; body: string }[];
  1383. const bodyByHash = new Map(rows.map(row => [row.hash, row.body]));
  1384. return batch.map((doc) => ({
  1385. ...doc,
  1386. body: bodyByHash.get(doc.hash) ?? "",
  1387. }));
  1388. }
  1389. /**
  1390. * Run `body` with a session-shaped argument that supplies an AbortSignal +
  1391. * isValid flag. When `provider` is supplied, the session is a lightweight
  1392. * AbortController-backed stub; `getLlm(store)` and the fail-closed legacy
  1393. * session wrapper are bypassed entirely.
  1394. *
  1395. * When `provider` is undefined, the compatibility session returns typed HOLD.
  1396. *
  1397. * The fake session's LLM-only methods (embed/embedBatch/expandQuery/rerank)
  1398. * throw if called — they MUST NOT be reached when `provider` is set, since
  1399. * the embed path is supposed to route through the provider instead.
  1400. */
  1401. async function withEmbedSession<T>(
  1402. store: Store,
  1403. provider: EmbeddingProvider | undefined,
  1404. body: (session: ILLMSession) => Promise<T>,
  1405. options?: LLMSessionOptions,
  1406. ): Promise<T> {
  1407. if (provider) {
  1408. const ac = new AbortController();
  1409. const fakeSession: ILLMSession = {
  1410. get signal() { return ac.signal; },
  1411. get isValid() { return !ac.signal.aborted; },
  1412. embed: async () => {
  1413. throw new Error("withEmbedSession: provider supplied — session.embed must not be called");
  1414. },
  1415. embedBatch: async () => {
  1416. throw new Error("withEmbedSession: provider supplied — session.embedBatch must not be called");
  1417. },
  1418. expandQuery: async () => {
  1419. throw new Error("withEmbedSession: provider supplied — session.expandQuery must not be called");
  1420. },
  1421. rerank: async () => {
  1422. throw new Error("withEmbedSession: provider supplied — session.rerank must not be called");
  1423. },
  1424. };
  1425. try {
  1426. return await body(fakeSession);
  1427. } finally {
  1428. ac.abort();
  1429. }
  1430. }
  1431. return withLLMSessionForLlm(getLlm(store), body, options);
  1432. }
  1433. /**
  1434. * Generate vector embeddings for documents that need them.
  1435. * Pure function — no console output, no db lifecycle management.
  1436. * Uses the store's LlamaCpp instance if set, otherwise the global singleton.
  1437. */
  1438. export async function generateEmbeddings(
  1439. store: Store,
  1440. options?: EmbedOptions
  1441. ): Promise<EmbedResult> {
  1442. const db = store.db;
  1443. const model = options?.model ?? DEFAULT_EMBED_MODEL;
  1444. const now = new Date().toISOString();
  1445. const { maxDocsPerBatch, maxBatchBytes } = resolveEmbedOptions(options);
  1446. const encoder = new TextEncoder();
  1447. // Migration safety: if an embedProvider is supplied, verify its model id
  1448. // matches the existing content_vectors rows (unless we're about to clear
  1449. // them via `force`). This must happen BEFORE we clear vectors so users
  1450. // who pass `--force` aren't blocked.
  1451. if (options?.embedProvider && !options.force) {
  1452. const existing = getDistinctEmbeddingModels(db);
  1453. assertModelCompatible(options.embedProvider.getModelId(), existing);
  1454. }
  1455. if (options?.force) {
  1456. clearAllEmbeddings(db);
  1457. }
  1458. // i-ofojj7dy — optional collection filter restricts the pending-doc set.
  1459. const docsToEmbed = getPendingEmbeddingDocs(db, options?.collection);
  1460. if (docsToEmbed.length === 0) {
  1461. return { docsProcessed: 0, chunksEmbedded: 0, errors: 0, durationMs: 0 };
  1462. }
  1463. const totalBytes = docsToEmbed.reduce((sum, doc) => sum + Math.max(0, doc.bytes), 0);
  1464. const totalDocs = docsToEmbed.length;
  1465. const startTime = Date.now();
  1466. // Per-collection chunkStrategy lookup (Phase 2 — i-bud0h8vu). YAML
  1467. // `chunkStrategy` on a collection wins over `options.chunkStrategy`
  1468. // (global CLI flag); falls back to the global option, then to
  1469. // chunkDocumentByTokens' own "regex" default when neither is set.
  1470. // Opt-in per collection — collections without the field are untouched.
  1471. const collectionStrategies = new Map<string, ChunkStrategy>();
  1472. try {
  1473. const { listCollections: listYamlCollections } = await import("./collections.js");
  1474. for (const c of listYamlCollections()) {
  1475. if (c.chunkStrategy) collectionStrategies.set(c.name, c.chunkStrategy);
  1476. }
  1477. } catch {
  1478. // If YAML config is missing/unreadable, fall back silently to the
  1479. // global strategy — no collection overrides. Keeps SDK/inline
  1480. // callers that never touch ~/.config/qmd working.
  1481. }
  1482. // Provider routing — when an EmbeddingProvider is supplied, embed calls go
  1483. // through it. Otherwise, use the fail-closed compatibility session path.
  1484. // The outer session is still created for its abort signal (chunking uses
  1485. // `session.signal` for cooperative cancellation).
  1486. const provider = options?.embedProvider;
  1487. const providerModel = provider?.getModelId() ?? model;
  1488. // Resolve `embedModelUri` (used for formatting prefixes etc.) lazily —
  1489. // when `provider` is set, take it from the provider; otherwise fall back
  1490. // to the disabled compatibility adapter's model name. Accessing `getLlm(store)`
  1491. // is deferred to the non-provider branch.
  1492. const embedModelUri = provider
  1493. ? provider.getModelId()
  1494. : getLlm(store).embedModelName;
  1495. // Run the embedding loop inside a session-scoped wrapper.
  1496. const result = await withEmbedSession(store, provider, async (session) => {
  1497. let chunksEmbedded = 0;
  1498. let errors = 0;
  1499. let skipped = 0;
  1500. let bytesProcessed = 0;
  1501. let totalChunks = 0;
  1502. let vectorTableInitialized = false;
  1503. // Set once the run decides to stop early. The high-error-rate guard used to
  1504. // `break` the INNER batch loop only, so the very next document re-evaluated
  1505. // the same cumulative ratio and broke again — one transient 429 could book
  1506. // ~740k never-attempted chunks as "failures" across ~50 aborts
  1507. // (i-yghj098h). The flag stops the OUTER loop too, exactly once.
  1508. let abortRun = false;
  1509. // Inner batch size — number of chunks fed into each `embedMany` call.
  1510. // Bumped 32 → 256 (i-fkpnar9i) so the openai provider's concurrent
  1511. // dispatcher receives ≥ 4 sub-chunks of size 64 (worker MAX_BATCH) and
  1512. // can saturate the worker's MAX_CONCURRENT_REQUESTS=4 semaphore.
  1513. // Override per-deploy via `QMD_EMBED_INNER_BATCH_SIZE`.
  1514. const BATCH_SIZE = parseInt(process.env.QMD_EMBED_INNER_BATCH_SIZE ?? "256", 10) || 256;
  1515. const batches = buildEmbeddingBatches(docsToEmbed, maxDocsPerBatch, maxBatchBytes);
  1516. // Sliding error-rate window (i-yghj098h). The abort guard used to compare
  1517. // CUMULATIVE `errors` against CUMULATIVE processed, and fed its own
  1518. // un-attempted remainders back into `errors` — so once tripped the ratio
  1519. // could never fall back under the threshold and the run was poisoned for
  1520. // good. Judging only the last few ATTEMPTED batches means a burst of 429s
  1521. // that the provider then rides out (retry + lane cooldown) no longer
  1522. // condemns the rest of the run.
  1523. const ERROR_WINDOW_BATCHES = parseInt(
  1524. process.env.QMD_EMBED_ERROR_WINDOW_BATCHES ?? "8", 10,
  1525. ) || 8;
  1526. const ERROR_RATE_ABORT_THRESHOLD = 0.8;
  1527. const recentBatches: { attempted: number; errors: number }[] = [];
  1528. const noteBatchOutcome = (attempted: number, failed: number): void => {
  1529. if (attempted <= 0) return;
  1530. recentBatches.push({ attempted, errors: failed });
  1531. while (recentBatches.length > ERROR_WINDOW_BATCHES) recentBatches.shift();
  1532. };
  1533. const windowTotals = (): { attempted: number; errors: number } => {
  1534. let attempted = 0;
  1535. let failed = 0;
  1536. for (const b of recentBatches) {
  1537. attempted += b.attempted;
  1538. failed += b.errors;
  1539. }
  1540. return { attempted, errors: failed };
  1541. };
  1542. // Embedding helpers — single point of provider/session selection.
  1543. // Both return the same shape as ILLMSession.embed/embedBatch so the
  1544. // rest of the loop is unchanged.
  1545. const embedOne = async (
  1546. text: string,
  1547. modelArg: string,
  1548. ): Promise<{ embedding: number[]; model: string } | null> => {
  1549. if (provider) {
  1550. const r = await provider.embed(text, { model: modelArg });
  1551. return r ? { embedding: r.embedding, model: r.model } : null;
  1552. }
  1553. return session.embed(text, { model: modelArg });
  1554. };
  1555. const embedMany = async (
  1556. texts: string[],
  1557. modelArg: string,
  1558. ): Promise<({ embedding: number[]; model: string } | null)[]> => {
  1559. if (provider) {
  1560. const r = await provider.embedBatch(texts, { model: modelArg });
  1561. return r.map((x) => (x ? { embedding: x.embedding, model: x.model } : null));
  1562. }
  1563. return session.embedBatch(texts, { model: modelArg });
  1564. };
  1565. // JS-only token estimator for the provider path. Char-based with
  1566. // avgCharsPerToken=3 — matches the heuristic the chunker already
  1567. // uses for its initial char-space pass, so the safety re-split is a
  1568. // near no-op while populating the `tokens` field with a stable
  1569. // estimate without invoking any learned tokenizer.
  1570. const chunkTokenizer: TokenCounter | undefined = provider
  1571. ? (text: string) => Math.ceil(text.length / 3)
  1572. : undefined;
  1573. for (const batchMeta of batches) {
  1574. // Abort early if session has been invalidated
  1575. if (!session.isValid) {
  1576. console.warn(`⚠ Session expired — skipping remaining document batches`);
  1577. break;
  1578. }
  1579. const batchDocs = getEmbeddingDocsForBatch(db, batchMeta);
  1580. const batchChunks: ChunkItem[] = [];
  1581. const batchBytes = batchMeta.reduce((sum, doc) => sum + Math.max(0, doc.bytes), 0);
  1582. for (const doc of batchDocs) {
  1583. if (!doc.body.trim()) continue;
  1584. const title = extractTitle(doc.body, doc.path);
  1585. const perCollectionStrategy = collectionStrategies.get(doc.collection);
  1586. const chunkStrategy = perCollectionStrategy ?? options?.chunkStrategy;
  1587. const chunks = await chunkDocumentByTokens(
  1588. doc.body,
  1589. undefined, undefined, undefined,
  1590. doc.path,
  1591. chunkStrategy,
  1592. session.signal,
  1593. chunkTokenizer,
  1594. );
  1595. // Record the expectation BEFORE embedding anything (i-xeekgx6h). If this
  1596. // run only gets through chunk 0 of 4, the next run compares 1 < 4 and
  1597. // re-selects the document instead of reading a present chunk 0 as "done".
  1598. recordDocumentChunkCount(db, doc.hash, chunks.length, now);
  1599. for (let seq = 0; seq < chunks.length; seq++) {
  1600. batchChunks.push({
  1601. hash: doc.hash,
  1602. title,
  1603. text: chunks[seq]!.text,
  1604. seq,
  1605. pos: chunks[seq]!.pos,
  1606. tokens: chunks[seq]!.tokens,
  1607. bytes: encoder.encode(chunks[seq]!.text).length,
  1608. });
  1609. }
  1610. }
  1611. totalChunks += batchChunks.length;
  1612. if (batchChunks.length === 0) {
  1613. bytesProcessed += batchBytes;
  1614. options?.onProgress?.({ chunksEmbedded, totalChunks, bytesProcessed, totalBytes, errors, skipped });
  1615. continue;
  1616. }
  1617. if (!vectorTableInitialized) {
  1618. const firstChunk = batchChunks[0]!;
  1619. const firstText = formatDocForEmbedding(firstChunk.text, firstChunk.title, embedModelUri);
  1620. // Single retry on transient failure (issue i-vm1lxwry). The provider
  1621. // swallows per-chunk errors per its contract — `getLastError?.()`
  1622. // surfaces the actual cause (HTTP status / abort / parse error) so we
  1623. // can include it in the thrown message instead of the cryptic
  1624. // "Failed to get embedding dimensions from first chunk".
  1625. let firstResult = await embedOne(firstText, providerModel);
  1626. if (!firstResult && session.isValid) {
  1627. const firstErr = provider?.getLastError?.();
  1628. // Brief backoff before retry — embedding worker may be re-warming
  1629. // a model or the GPU host may be transiently busy. 250ms is short
  1630. // enough to be invisible on the happy path and long enough to
  1631. // clear most "thundering-herd" race conditions.
  1632. await new Promise((resolve) => setTimeout(resolve, 250));
  1633. if (process.env.QMD_EMBED_DEBUG) {
  1634. process.stderr.write(
  1635. `qmd embed: first-chunk dimension probe failed, retrying once${firstErr ? ` (last error: ${firstErr})` : ""}\n`,
  1636. );
  1637. }
  1638. firstResult = await embedOne(firstText, providerModel);
  1639. }
  1640. if (!firstResult) {
  1641. const lastErr = provider?.getLastError?.();
  1642. const providerHint = provider ? `provider=${provider.kind}` : "provider=session";
  1643. const errSuffix = lastErr ? ` — underlying: ${lastErr}` : "";
  1644. const debugHint = process.env.QMD_EMBED_DEBUG
  1645. ? ""
  1646. : " (set QMD_EMBED_DEBUG=1 for per-chunk traces)";
  1647. throw new Error(
  1648. `Failed to get embedding dimensions from first chunk after retry [${providerHint}]${errSuffix}${debugHint}`,
  1649. );
  1650. }
  1651. store.ensureVecTable(firstResult.embedding.length);
  1652. vectorTableInitialized = true;
  1653. }
  1654. const totalBatchChunkBytes = batchChunks.reduce((sum, chunk) => sum + chunk.bytes, 0);
  1655. let batchChunkBytesProcessed = 0;
  1656. for (let batchStart = 0; batchStart < batchChunks.length; batchStart += BATCH_SIZE) {
  1657. // Abort early if session has been invalidated (e.g. max duration exceeded)
  1658. if (!session.isValid) {
  1659. const remaining = batchChunks.length - batchStart;
  1660. skipped += remaining;
  1661. abortRun = true;
  1662. console.warn(`⚠ Session expired — skipping ${remaining} remaining chunks`);
  1663. break;
  1664. }
  1665. // Abort early if the RECENT attempted batches are overwhelmingly failing
  1666. // (>80% over the sliding window). Un-attempted chunks are booked as
  1667. // `skipped`, never as `errors`.
  1668. const window = windowTotals();
  1669. if (
  1670. window.attempted >= BATCH_SIZE &&
  1671. window.errors > window.attempted * ERROR_RATE_ABORT_THRESHOLD
  1672. ) {
  1673. const remaining = batchChunks.length - batchStart;
  1674. skipped += remaining;
  1675. abortRun = true;
  1676. console.warn(
  1677. `⚠ Error rate too high (${window.errors}/${window.attempted} over last ` +
  1678. `${recentBatches.length} batches) — aborting embedding ` +
  1679. `(${remaining} chunks in this document left unattempted)`,
  1680. );
  1681. break;
  1682. }
  1683. const batchEnd = Math.min(batchStart + BATCH_SIZE, batchChunks.length);
  1684. const chunkBatch = batchChunks.slice(batchStart, batchEnd);
  1685. const texts = chunkBatch.map(chunk => formatDocForEmbedding(chunk.text, chunk.title, embedModelUri));
  1686. try {
  1687. const embeddings = await embedMany(texts, providerModel);
  1688. // Wrap the per-chunk inserts in a single SQLite transaction
  1689. // (i-fkpnar9i Phase 1 #3): avoids the WAL fsync per-row tax on
  1690. // large `BATCH_SIZE`. better-sqlite3's `db.transaction(fn)` opens
  1691. // BEGIN IMMEDIATE on entry and COMMITs on return; if any insert
  1692. // throws, the wrapper rolls back AND re-throws, falling through
  1693. // to the per-chunk fallback below — preserving the legacy
  1694. // "best-effort survive partial failures" semantics.
  1695. //
  1696. // We DELIBERATELY do not wrap the fallback's per-chunk loop —
  1697. // that path is per-chunk individual auto-commits so a single
  1698. // bad chunk doesn't drag down the rest. (Wrapping would be a
  1699. // step backward.)
  1700. const insertBatchTxn = db.transaction(() => {
  1701. let okCount = 0;
  1702. let errCount = 0;
  1703. for (let i = 0; i < chunkBatch.length; i++) {
  1704. const chunk = chunkBatch[i]!;
  1705. const embedding = embeddings[i];
  1706. if (embedding) {
  1707. insertEmbedding(db, chunk.hash, chunk.seq, chunk.pos, new Float32Array(embedding.embedding), providerModel, now);
  1708. okCount++;
  1709. } else {
  1710. errCount++;
  1711. }
  1712. }
  1713. return { okCount, errCount };
  1714. });
  1715. const { okCount, errCount } = insertBatchTxn();
  1716. chunksEmbedded += okCount;
  1717. errors += errCount;
  1718. noteBatchOutcome(chunkBatch.length, errCount);
  1719. batchChunkBytesProcessed += chunkBatch.reduce((sum, c) => sum + c.bytes, 0);
  1720. } catch {
  1721. // Batch failed — try individual embeddings as fallback
  1722. // But skip if session is already invalid (avoids N doomed retries)
  1723. if (!session.isValid) {
  1724. errors += chunkBatch.length;
  1725. noteBatchOutcome(chunkBatch.length, chunkBatch.length);
  1726. batchChunkBytesProcessed += chunkBatch.reduce((sum, c) => sum + c.bytes, 0);
  1727. } else {
  1728. let fallbackErrors = 0;
  1729. for (const chunk of chunkBatch) {
  1730. try {
  1731. const text = formatDocForEmbedding(chunk.text, chunk.title, embedModelUri);
  1732. const result = await embedOne(text, providerModel);
  1733. if (result) {
  1734. insertEmbedding(db, chunk.hash, chunk.seq, chunk.pos, new Float32Array(result.embedding), providerModel, now);
  1735. chunksEmbedded++;
  1736. } else {
  1737. errors++;
  1738. fallbackErrors++;
  1739. }
  1740. } catch {
  1741. errors++;
  1742. fallbackErrors++;
  1743. }
  1744. batchChunkBytesProcessed += chunk.bytes;
  1745. }
  1746. noteBatchOutcome(chunkBatch.length, fallbackErrors);
  1747. }
  1748. }
  1749. const proportionalBytes = totalBatchChunkBytes === 0
  1750. ? batchBytes
  1751. : Math.min(batchBytes, Math.round((batchChunkBytesProcessed / totalBatchChunkBytes) * batchBytes));
  1752. options?.onProgress?.({
  1753. chunksEmbedded,
  1754. totalChunks,
  1755. bytesProcessed: bytesProcessed + proportionalBytes,
  1756. totalBytes,
  1757. errors,
  1758. skipped,
  1759. });
  1760. }
  1761. bytesProcessed += batchBytes;
  1762. options?.onProgress?.({ chunksEmbedded, totalChunks, bytesProcessed, totalBytes, errors, skipped });
  1763. // One abort ends the run — it must not re-trip on every remaining doc.
  1764. if (abortRun) break;
  1765. }
  1766. return { chunksEmbedded, errors, skipped };
  1767. }, { maxDuration: 30 * 60 * 1000, name: 'generateEmbeddings' });
  1768. return {
  1769. docsProcessed: totalDocs,
  1770. chunksEmbedded: result.chunksEmbedded,
  1771. errors: result.errors,
  1772. skipped: result.skipped,
  1773. durationMs: Date.now() - startTime,
  1774. };
  1775. }
  1776. /**
  1777. * Create a new store instance with the given database path.
  1778. * If no path is provided, uses the default path (~/.cache/qmd/index.sqlite).
  1779. *
  1780. * @param dbPath - Path to the SQLite database file
  1781. * @returns Store instance with all methods bound to the database
  1782. */
  1783. export function createStore(dbPath?: string): Store {
  1784. const resolvedPath = dbPath || getDefaultDbPath();
  1785. const db = openDatabase(resolvedPath);
  1786. initializeDatabase(db);
  1787. const store: Store = {
  1788. db,
  1789. dbPath: resolvedPath,
  1790. close: () => db.close(),
  1791. ensureVecTable: (dimensions: number) => ensureVecTableInternal(db, dimensions),
  1792. // Index health
  1793. getHashesNeedingEmbedding: () => getHashesNeedingEmbedding(db),
  1794. getIndexHealth: () => getIndexHealth(db),
  1795. getStatus: () => getStatus(db),
  1796. // Caching
  1797. getCacheKey,
  1798. getCachedResult: (cacheKey: string) => getCachedResult(db, cacheKey),
  1799. setCachedResult: (cacheKey: string, result: string) => setCachedResult(db, cacheKey, result),
  1800. clearCache: () => clearCache(db),
  1801. // Cleanup and maintenance
  1802. deleteLLMCache: () => deleteLLMCache(db),
  1803. deleteInactiveDocuments: () => deleteInactiveDocuments(db),
  1804. cleanupOrphanedContent: () => cleanupOrphanedContent(db),
  1805. cleanupOrphanedVectors: () => cleanupOrphanedVectors(db),
  1806. vacuumDatabase: () => vacuumDatabase(db),
  1807. // Context
  1808. getContextForFile: (filepath: string) => getContextForFile(db, filepath),
  1809. getContextForPath: (collectionName: string, path: string) => getContextForPath(db, collectionName, path),
  1810. getCollectionByName: (name: string) => getCollectionByName(db, name),
  1811. getCollectionsWithoutContext: () => getCollectionsWithoutContext(db),
  1812. getTopLevelPathsWithoutContext: (collectionName: string) => getTopLevelPathsWithoutContext(db, collectionName),
  1813. // Virtual paths
  1814. parseVirtualPath,
  1815. buildVirtualPath,
  1816. isVirtualPath,
  1817. resolveVirtualPath: (virtualPath: string) => resolveVirtualPath(db, virtualPath),
  1818. toVirtualPath: (absolutePath: string) => toVirtualPath(db, absolutePath),
  1819. // Search
  1820. searchFTS: (query: string, limit?: number, collectionName?: string) => searchFTS(db, query, limit, collectionName),
  1821. searchVec: (query: string, model: string, limit?: number, collectionName?: string, session?: ILLMSession, precomputedEmbedding?: number[], embedProvider?: EmbeddingProvider) => searchVec(db, query, model, limit, collectionName, session, precomputedEmbedding, embedProvider),
  1822. // Query expansion & reranking
  1823. expandQuery: (query: string, model?: string, intent?: string) => expandQuery(query, model, db, intent, store.llm),
  1824. rerank: (query: string, documents: { file: string; text: string }[], model?: string, intent?: string) => rerank(query, documents, model, db, intent, store.llm),
  1825. // Document retrieval
  1826. findDocument: (filename: string, options?: { includeBody?: boolean }) => findDocument(db, filename, options),
  1827. getDocumentBody: (doc: DocumentResult | { filepath: string }, fromLine?: number, maxLines?: number) => getDocumentBody(db, doc, fromLine, maxLines),
  1828. findDocuments: (pattern: string, options?: { includeBody?: boolean; maxBytes?: number }) => findDocuments(db, pattern, options),
  1829. // Fuzzy matching and docid lookup
  1830. findSimilarFiles: (query: string, maxDistance?: number, limit?: number) => findSimilarFiles(db, query, maxDistance, limit),
  1831. matchFilesByGlob: (pattern: string) => matchFilesByGlob(db, pattern),
  1832. findDocumentByDocid: (docid: string) => findDocumentByDocid(db, docid),
  1833. // Document indexing operations
  1834. insertContent: (hash: string, content: string, createdAt: string) => insertContent(db, hash, content, createdAt),
  1835. insertDocument: (collectionName: string, path: string, title: string, hash: string, createdAt: string, modifiedAt: string) => insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt),
  1836. findActiveDocument: (collectionName: string, path: string) => findActiveDocument(db, collectionName, path),
  1837. updateDocumentTitle: (documentId: number, title: string, modifiedAt: string) => updateDocumentTitle(db, documentId, title, modifiedAt),
  1838. updateDocument: (documentId: number, title: string, hash: string, modifiedAt: string) => updateDocument(db, documentId, title, hash, modifiedAt),
  1839. deactivateDocument: (collectionName: string, path: string) => deactivateDocument(db, collectionName, path),
  1840. getActiveDocumentPaths: (collectionName: string) => getActiveDocumentPaths(db, collectionName),
  1841. // Vector/embedding operations
  1842. getHashesForEmbedding: () => getHashesForEmbedding(db),
  1843. clearAllEmbeddings: () => clearAllEmbeddings(db),
  1844. insertEmbedding: (hash: string, seq: number, pos: number, embedding: Float32Array, model: string, embeddedAt: string) => insertEmbedding(db, hash, seq, pos, embedding, model, embeddedAt),
  1845. };
  1846. return store;
  1847. }
  1848. // =============================================================================
  1849. // Core Document Type
  1850. // =============================================================================
  1851. /**
  1852. * Unified document result type with all metadata.
  1853. * Body is optional - use getDocumentBody() to load it separately if needed.
  1854. */
  1855. export type DocumentResult = {
  1856. filepath: string; // Full filesystem path
  1857. displayPath: string; // Short display path (e.g., "docs/readme.md")
  1858. title: string; // Document title (from first heading or filename)
  1859. context: string | null; // Folder context description if configured
  1860. hash: string; // Content hash for caching/change detection
  1861. docid: string; // Short docid (first 6 chars of hash) for quick reference
  1862. collectionName: string; // Parent collection name
  1863. modifiedAt: string; // Last modification timestamp
  1864. bodyLength: number; // Body length in bytes (useful before loading)
  1865. body?: string; // Document body (optional, load with getDocumentBody)
  1866. };
  1867. /**
  1868. * Extract short docid from a full hash (first 6 characters).
  1869. */
  1870. export function getDocid(hash: string): string {
  1871. return hash.slice(0, 6);
  1872. }
  1873. /**
  1874. * Handelize a filename to be more token-friendly.
  1875. * - Convert triple underscore `___` to `/` (folder separator)
  1876. * - Convert to lowercase
  1877. * - Replace sequences of non-word chars (except /) with single dash
  1878. * - Remove leading/trailing dashes from path segments
  1879. * - Preserve folder structure (a/b/c/d.md stays structured)
  1880. * - Preserve file extension
  1881. */
  1882. /** Replace emoji/symbol codepoints with their hex representation (e.g. 🐘 → 1f418) */
  1883. function emojiToHex(str: string): string {
  1884. return str.replace(/(?:\p{So}\p{Mn}?|\p{Sk})+/gu, (run) => {
  1885. // Split the run into individual emoji and convert each to hex, dash-separated
  1886. return [...run].filter(c => /\p{So}|\p{Sk}/u.test(c))
  1887. .map(c => c.codePointAt(0)!.toString(16)).join('-');
  1888. });
  1889. }
  1890. export function handelize(path: string): string {
  1891. if (!path || path.trim() === '') {
  1892. throw new Error('handelize: path cannot be empty');
  1893. }
  1894. // Allow route-style "$" filenames while still rejecting paths with no usable content.
  1895. // Emoji (\p{So}) counts as valid content — they get converted to hex codepoints below.
  1896. const segments = path.split('/').filter(Boolean);
  1897. const lastSegment = segments[segments.length - 1] || '';
  1898. const filenameWithoutExt = lastSegment.replace(/\.[^.]+$/, '');
  1899. const hasValidContent = /[\p{L}\p{N}\p{So}\p{Sk}$]/u.test(filenameWithoutExt);
  1900. if (!hasValidContent) {
  1901. throw new Error(`handelize: path "${path}" has no valid filename content`);
  1902. }
  1903. const result = path
  1904. .replace(/___/g, '/') // Triple underscore becomes folder separator
  1905. .toLowerCase()
  1906. .split('/')
  1907. .map((segment, idx, arr) => {
  1908. const isLastSegment = idx === arr.length - 1;
  1909. // Convert emoji to hex codepoints before cleaning
  1910. segment = emojiToHex(segment);
  1911. if (isLastSegment) {
  1912. // For the filename (last segment), preserve the extension
  1913. const extMatch = segment.match(/(\.[a-z0-9]+)$/i);
  1914. const ext = extMatch ? extMatch[1] : '';
  1915. const nameWithoutExt = ext ? segment.slice(0, -ext.length) : segment;
  1916. const cleanedName = nameWithoutExt
  1917. .replace(/[^\p{L}\p{N}$]+/gu, '-') // Keep letters, numbers, "$"; dash-separate rest (including dots)
  1918. .replace(/^-+|-+$/g, ''); // Remove leading/trailing dashes
  1919. return cleanedName + ext;
  1920. } else {
  1921. // For directories, just clean normally
  1922. return segment
  1923. .replace(/[^\p{L}\p{N}$]+/gu, '-')
  1924. .replace(/^-+|-+$/g, '');
  1925. }
  1926. })
  1927. .filter(Boolean)
  1928. .join('/');
  1929. if (!result) {
  1930. throw new Error(`handelize: path "${path}" resulted in empty string after processing`);
  1931. }
  1932. return result;
  1933. }
  1934. /**
  1935. * Search result extends DocumentResult with score and source info
  1936. */
  1937. export type SearchResult = DocumentResult & {
  1938. score: number; // Relevance score (0-1)
  1939. source: "fts" | "vec"; // Search source (full-text or vector)
  1940. chunkPos?: number; // Character position of matching chunk (for vector search)
  1941. };
  1942. /**
  1943. * Ranked result for RRF fusion (simplified, used internally)
  1944. */
  1945. export type RankedResult = {
  1946. file: string;
  1947. displayPath: string;
  1948. title: string;
  1949. body: string;
  1950. score: number;
  1951. };
  1952. export type RRFContributionTrace = {
  1953. listIndex: number;
  1954. source: "fts" | "vec";
  1955. queryType: "original" | "lex" | "vec" | "hyde";
  1956. query: string;
  1957. rank: number; // 1-indexed rank within list
  1958. weight: number;
  1959. backendScore: number; // Backend-normalized score before fusion
  1960. rrfContribution: number; // weight / (k + rank)
  1961. };
  1962. export type RRFScoreTrace = {
  1963. contributions: RRFContributionTrace[];
  1964. baseScore: number; // Sum of reciprocal-rank contributions
  1965. topRank: number; // Best (lowest) rank seen across lists
  1966. topRankBonus: number; // +0.05 for rank 1, +0.02 for rank 2-3
  1967. totalScore: number; // baseScore + topRankBonus
  1968. };
  1969. export type HybridQueryExplain = {
  1970. ftsScores: number[];
  1971. vectorScores: number[];
  1972. rrf: {
  1973. rank: number; // Rank after RRF fusion (1-indexed)
  1974. positionScore: number; // 1 / rank used in position-aware blending
  1975. weight: number; // Position-aware RRF weight (0.75 / 0.60 / 0.40)
  1976. baseScore: number;
  1977. topRankBonus: number;
  1978. totalScore: number;
  1979. contributions: RRFContributionTrace[];
  1980. };
  1981. rerankScore: number;
  1982. blendedScore: number;
  1983. };
  1984. /**
  1985. * Error result when document is not found
  1986. */
  1987. export type DocumentNotFound = {
  1988. error: "not_found";
  1989. query: string;
  1990. similarFiles: string[];
  1991. };
  1992. /**
  1993. * Result from multi-get operations
  1994. */
  1995. export type MultiGetResult = {
  1996. doc: DocumentResult;
  1997. skipped: false;
  1998. } | {
  1999. doc: Pick<DocumentResult, "filepath" | "displayPath">;
  2000. skipped: true;
  2001. skipReason: string;
  2002. };
  2003. export type CollectionInfo = {
  2004. name: string;
  2005. path: string | null;
  2006. pattern: string | null;
  2007. documents: number;
  2008. lastUpdated: string;
  2009. };
  2010. export type IndexStatus = {
  2011. totalDocuments: number;
  2012. needsEmbedding: number;
  2013. hasVectorIndex: boolean;
  2014. collections: CollectionInfo[];
  2015. };
  2016. // =============================================================================
  2017. // Index health
  2018. // =============================================================================
  2019. export function getHashesNeedingEmbedding(db: Database, collection?: string): number {
  2020. // i-ofojj7dy — optional collection filter. Restricts the count to hashes
  2021. // whose documents are in the named collection.
  2022. // Same predicate as getPendingEmbeddingDocs, from the same constants — the
  2023. // two carrying independent copies is how the partial-embedding hole survived
  2024. // (i-xeekgx6h).
  2025. if (collection !== undefined) {
  2026. const result = db.prepare(`
  2027. SELECT COUNT(DISTINCT d.hash) as count
  2028. FROM documents d${PENDING_EMBEDDING_JOINS}
  2029. WHERE d.active = 1 AND ${PENDING_EMBEDDING_PREDICATE} AND d.collection = ?
  2030. `).get(collection) as { count: number };
  2031. return result.count;
  2032. }
  2033. const result = db.prepare(`
  2034. SELECT COUNT(DISTINCT d.hash) as count
  2035. FROM documents d${PENDING_EMBEDDING_JOINS}
  2036. WHERE d.active = 1 AND ${PENDING_EMBEDDING_PREDICATE}
  2037. `).get() as { count: number };
  2038. return result.count;
  2039. }
  2040. export type IndexHealthInfo = {
  2041. needsEmbedding: number;
  2042. totalDocs: number;
  2043. daysStale: number | null;
  2044. };
  2045. export function getIndexHealth(db: Database): IndexHealthInfo {
  2046. const needsEmbedding = getHashesNeedingEmbedding(db);
  2047. const totalDocs = (db.prepare(`SELECT COUNT(*) as count FROM documents WHERE active = 1`).get() as { count: number }).count;
  2048. const mostRecent = db.prepare(`SELECT MAX(modified_at) as latest FROM documents WHERE active = 1`).get() as { latest: string | null };
  2049. let daysStale: number | null = null;
  2050. if (mostRecent?.latest) {
  2051. const lastUpdate = new Date(mostRecent.latest);
  2052. daysStale = Math.floor((Date.now() - lastUpdate.getTime()) / (24 * 60 * 60 * 1000));
  2053. }
  2054. return { needsEmbedding, totalDocs, daysStale };
  2055. }
  2056. // =============================================================================
  2057. // Caching
  2058. // =============================================================================
  2059. export function getCacheKey(url: string, body: object): string {
  2060. const hash = createHash("sha256");
  2061. hash.update(url);
  2062. hash.update(JSON.stringify(body));
  2063. return hash.digest("hex");
  2064. }
  2065. export function getCachedResult(db: Database, cacheKey: string): string | null {
  2066. const row = db.prepare(`SELECT result FROM llm_cache WHERE hash = ?`).get(cacheKey) as { result: string } | null;
  2067. return row?.result || null;
  2068. }
  2069. export function setCachedResult(db: Database, cacheKey: string, result: string): void {
  2070. const now = new Date().toISOString();
  2071. db.prepare(`INSERT OR REPLACE INTO llm_cache (hash, result, created_at) VALUES (?, ?, ?)`).run(cacheKey, result, now);
  2072. if (Math.random() < 0.01) {
  2073. db.exec(`DELETE FROM llm_cache WHERE hash NOT IN (SELECT hash FROM llm_cache ORDER BY created_at DESC LIMIT 1000)`);
  2074. }
  2075. }
  2076. export function clearCache(db: Database): void {
  2077. db.exec(`DELETE FROM llm_cache`);
  2078. }
  2079. // =============================================================================
  2080. // Cleanup and maintenance operations
  2081. // =============================================================================
  2082. /**
  2083. * Delete cached LLM API responses.
  2084. * Returns the number of cached responses deleted.
  2085. */
  2086. export function deleteLLMCache(db: Database): number {
  2087. const result = db.prepare(`DELETE FROM llm_cache`).run();
  2088. return result.changes;
  2089. }
  2090. /**
  2091. * Remove inactive document records (active = 0).
  2092. * Returns the number of inactive documents deleted.
  2093. */
  2094. export function deleteInactiveDocuments(db: Database): number {
  2095. const result = db.prepare(`DELETE FROM documents WHERE active = 0`).run();
  2096. return result.changes;
  2097. }
  2098. /**
  2099. * Remove orphaned content hashes that are not referenced by any active document.
  2100. * Returns the number of orphaned content hashes deleted.
  2101. */
  2102. export function cleanupOrphanedContent(db: Database): number {
  2103. const result = db.prepare(`
  2104. DELETE FROM content
  2105. WHERE hash NOT IN (SELECT DISTINCT hash FROM documents WHERE active = 1)
  2106. `).run();
  2107. return result.changes;
  2108. }
  2109. /**
  2110. * Remove orphaned vector embeddings that are not referenced by any active document.
  2111. * Returns the number of orphaned embedding chunks deleted.
  2112. */
  2113. export function cleanupOrphanedVectors(db: Database): number {
  2114. // sqlite-vec may not be loaded (e.g. Bun's bun:sqlite lacks loadExtension).
  2115. // The vectors_vec virtual table can appear in sqlite_master from a prior
  2116. // session, but querying it without the vec0 module loaded will crash (#380).
  2117. if (!isSqliteVecAvailable()) {
  2118. return 0;
  2119. }
  2120. // The schema entry can exist even when sqlite-vec itself is unavailable
  2121. // (for example when reopening a DB without vec0 loaded). In that case,
  2122. // touching the virtual table throws "no such module: vec0" and cleanup
  2123. // should degrade gracefully like the rest of the vector features.
  2124. try {
  2125. db.prepare(`SELECT 1 FROM vectors_vec LIMIT 0`).get();
  2126. } catch {
  2127. return 0;
  2128. }
  2129. // Count orphaned vectors first
  2130. const countResult = db.prepare(`
  2131. SELECT COUNT(*) as c FROM content_vectors cv
  2132. WHERE NOT EXISTS (
  2133. SELECT 1 FROM documents d WHERE d.hash = cv.hash AND d.active = 1
  2134. )
  2135. `).get() as { c: number };
  2136. if (countResult.c === 0) {
  2137. return 0;
  2138. }
  2139. // Delete from vectors_vec first
  2140. db.exec(`
  2141. DELETE FROM vectors_vec WHERE hash_seq IN (
  2142. SELECT cv.hash || '_' || cv.seq FROM content_vectors cv
  2143. WHERE NOT EXISTS (
  2144. SELECT 1 FROM documents d WHERE d.hash = cv.hash AND d.active = 1
  2145. )
  2146. )
  2147. `);
  2148. // Delete from content_vectors
  2149. db.exec(`
  2150. DELETE FROM content_vectors WHERE hash NOT IN (
  2151. SELECT hash FROM documents WHERE active = 1
  2152. )
  2153. `);
  2154. return countResult.c;
  2155. }
  2156. /**
  2157. * Run VACUUM to reclaim unused space in the database.
  2158. * This operation rebuilds the database file to eliminate fragmentation.
  2159. */
  2160. export function vacuumDatabase(db: Database): void {
  2161. db.exec(`VACUUM`);
  2162. }
  2163. // =============================================================================
  2164. // Document helpers
  2165. // =============================================================================
  2166. export async function hashContent(content: string): Promise<string> {
  2167. const hash = createHash("sha256");
  2168. hash.update(content);
  2169. return hash.digest("hex");
  2170. }
  2171. const titleExtractors: Record<string, (content: string) => string | null> = {
  2172. '.md': (content) => {
  2173. const match = content.match(/^##?\s+(.+)$/m);
  2174. if (match) {
  2175. const title = (match[1] ?? "").trim();
  2176. if (title === "📝 Notes" || title === "Notes") {
  2177. const nextMatch = content.match(/^##\s+(.+)$/m);
  2178. if (nextMatch?.[1]) return nextMatch[1].trim();
  2179. }
  2180. return title;
  2181. }
  2182. return null;
  2183. },
  2184. '.org': (content) => {
  2185. const titleProp = content.match(/^#\+TITLE:\s*(.+)$/im);
  2186. if (titleProp?.[1]) return titleProp[1].trim();
  2187. const heading = content.match(/^\*+\s+(.+)$/m);
  2188. if (heading?.[1]) return heading[1].trim();
  2189. return null;
  2190. },
  2191. };
  2192. export function extractTitle(content: string, filename: string): string {
  2193. const ext = filename.slice(filename.lastIndexOf('.')).toLowerCase();
  2194. const extractor = titleExtractors[ext];
  2195. if (extractor) {
  2196. const title = extractor(content);
  2197. if (title) return title;
  2198. }
  2199. return filename.replace(/\.[^.]+$/, "").split("/").pop() || filename;
  2200. }
  2201. // =============================================================================
  2202. // Document indexing operations
  2203. // =============================================================================
  2204. /**
  2205. * Insert content into the content table (content-addressable storage).
  2206. * Uses INSERT OR IGNORE so duplicate hashes are skipped.
  2207. */
  2208. export function insertContent(db: Database, hash: string, content: string, createdAt: string): void {
  2209. db.prepare(`INSERT OR IGNORE INTO content (hash, doc, created_at) VALUES (?, ?, ?)`)
  2210. .run(hash, content, createdAt);
  2211. }
  2212. /**
  2213. * Insert a new document into the documents table.
  2214. */
  2215. export function insertDocument(
  2216. db: Database,
  2217. collectionName: string,
  2218. path: string,
  2219. title: string,
  2220. hash: string,
  2221. createdAt: string,
  2222. modifiedAt: string
  2223. ): void {
  2224. db.prepare(`
  2225. INSERT INTO documents (collection, path, title, hash, created_at, modified_at, active)
  2226. VALUES (?, ?, ?, ?, ?, ?, 1)
  2227. ON CONFLICT(collection, path) DO UPDATE SET
  2228. title = excluded.title,
  2229. hash = excluded.hash,
  2230. modified_at = excluded.modified_at,
  2231. active = 1
  2232. `).run(collectionName, path, title, hash, createdAt, modifiedAt);
  2233. }
  2234. /**
  2235. * Find an active document by collection name and path.
  2236. */
  2237. export function findActiveDocument(
  2238. db: Database,
  2239. collectionName: string,
  2240. path: string
  2241. ): { id: number; hash: string; title: string } | null {
  2242. const row = db.prepare(`
  2243. SELECT id, hash, title FROM documents
  2244. WHERE collection = ? AND path = ? AND active = 1
  2245. `).get(collectionName, path) as { id: number; hash: string; title: string } | undefined;
  2246. return row ?? null;
  2247. }
  2248. /**
  2249. * Update the title and modified_at timestamp for a document.
  2250. */
  2251. export function updateDocumentTitle(
  2252. db: Database,
  2253. documentId: number,
  2254. title: string,
  2255. modifiedAt: string
  2256. ): void {
  2257. db.prepare(`UPDATE documents SET title = ?, modified_at = ? WHERE id = ?`)
  2258. .run(title, modifiedAt, documentId);
  2259. }
  2260. /**
  2261. * Update an existing document's hash, title, and modified_at timestamp.
  2262. * Used when content changes but the file path stays the same.
  2263. */
  2264. export function updateDocument(
  2265. db: Database,
  2266. documentId: number,
  2267. title: string,
  2268. hash: string,
  2269. modifiedAt: string
  2270. ): void {
  2271. db.prepare(`UPDATE documents SET title = ?, hash = ?, modified_at = ? WHERE id = ?`)
  2272. .run(title, hash, modifiedAt, documentId);
  2273. }
  2274. /**
  2275. * Deactivate a document (mark as inactive but don't delete).
  2276. */
  2277. export function deactivateDocument(db: Database, collectionName: string, path: string): void {
  2278. db.prepare(`UPDATE documents SET active = 0 WHERE collection = ? AND path = ? AND active = 1`)
  2279. .run(collectionName, path);
  2280. }
  2281. /**
  2282. * Get all active document paths for a collection.
  2283. */
  2284. export function getActiveDocumentPaths(db: Database, collectionName: string): string[] {
  2285. const rows = db.prepare(`
  2286. SELECT path FROM documents WHERE collection = ? AND active = 1
  2287. `).all(collectionName) as { path: string }[];
  2288. return rows.map(r => r.path);
  2289. }
  2290. export { formatQueryForEmbedding, formatDocForEmbedding };
  2291. /**
  2292. * Chunk a document using regex-only break point detection.
  2293. * This is the sync, backward-compatible API used by tests and legacy callers.
  2294. */
  2295. export function chunkDocument(
  2296. content: string,
  2297. maxChars: number = CHUNK_SIZE_CHARS,
  2298. overlapChars: number = CHUNK_OVERLAP_CHARS,
  2299. windowChars: number = CHUNK_WINDOW_CHARS
  2300. ): { text: string; pos: number }[] {
  2301. const breakPoints = scanBreakPoints(content);
  2302. const codeFences = findCodeFences(content);
  2303. return chunkDocumentWithBreakPoints(content, breakPoints, codeFences, maxChars, overlapChars, windowChars);
  2304. }
  2305. /**
  2306. * Async AST-aware chunking. Detects language from filepath, computes AST
  2307. * break points for supported code files, merges with regex break points,
  2308. * and delegates to the shared chunk algorithm.
  2309. *
  2310. * Strategies:
  2311. * - "regex" (default) — char-based chunking with regex break points only.
  2312. * - "auto" — regex break points merged with AST break points (soft hints).
  2313. * - "function" — one chunk per AST function range (Phase 2); inter-range
  2314. * gaps (imports, top-level code) are char-chunked with AST
  2315. * hints. Falls back to "auto" when zero ranges are detected.
  2316. */
  2317. export async function chunkDocumentAsync(
  2318. content: string,
  2319. maxChars: number = CHUNK_SIZE_CHARS,
  2320. overlapChars: number = CHUNK_OVERLAP_CHARS,
  2321. windowChars: number = CHUNK_WINDOW_CHARS,
  2322. filepath?: string,
  2323. chunkStrategy: ChunkStrategy = "regex",
  2324. ): Promise<{ text: string; pos: number }[]> {
  2325. const regexPoints = scanBreakPoints(content);
  2326. const codeFences = findCodeFences(content);
  2327. // "function" strategy: delegate to the function-level chunker. If no
  2328. // ranges are detected (markdown, unsupported lang, parse failure), fall
  2329. // back to "auto" behavior (AST-break-point-assisted char chunking).
  2330. if (chunkStrategy === "function" && filepath) {
  2331. const { getASTFunctionRanges, getASTBreakPoints } = await import("./ast.js");
  2332. const ranges = await getASTFunctionRanges(content, filepath);
  2333. if (ranges.length > 0) {
  2334. return chunkByFunctionRanges(
  2335. content,
  2336. ranges,
  2337. regexPoints,
  2338. codeFences,
  2339. maxChars,
  2340. overlapChars,
  2341. windowChars,
  2342. );
  2343. }
  2344. // Zero ranges — fall through to auto behavior so break points still help.
  2345. const astPoints = await getASTBreakPoints(content, filepath);
  2346. const merged = astPoints.length > 0 ? mergeBreakPoints(regexPoints, astPoints) : regexPoints;
  2347. return chunkDocumentWithBreakPoints(content, merged, codeFences, maxChars, overlapChars, windowChars);
  2348. }
  2349. let breakPoints = regexPoints;
  2350. if (chunkStrategy === "auto" && filepath) {
  2351. const { getASTBreakPoints } = await import("./ast.js");
  2352. const astPoints = await getASTBreakPoints(content, filepath);
  2353. if (astPoints.length > 0) {
  2354. breakPoints = mergeBreakPoints(regexPoints, astPoints);
  2355. }
  2356. }
  2357. return chunkDocumentWithBreakPoints(content, breakPoints, codeFences, maxChars, overlapChars, windowChars);
  2358. }
  2359. /**
  2360. * Produce one chunk per AST function range, plus char-chunks for the gaps
  2361. * between ranges (imports, top-level code). Ranges that exceed `maxChars`
  2362. * are further split using the existing char-based algorithm so we never
  2363. * emit a single oversized chunk.
  2364. *
  2365. * Preconditions: `ranges` is non-empty, sorted by `startIndex`, and the
  2366. * ranges are non-overlapping (as produced by `getASTFunctionRanges`).
  2367. */
  2368. function chunkByFunctionRanges(
  2369. content: string,
  2370. ranges: import("./ast.js").FunctionRange[],
  2371. regexPoints: BreakPoint[],
  2372. codeFences: CodeFenceRegion[],
  2373. maxChars: number,
  2374. overlapChars: number,
  2375. windowChars: number,
  2376. ): { text: string; pos: number }[] {
  2377. const out: { text: string; pos: number }[] = [];
  2378. let cursor = 0;
  2379. const emitGap = (start: number, end: number) => {
  2380. if (start >= end) return;
  2381. const gap = content.slice(start, end);
  2382. // Whitespace-only gaps are dropped — they carry no embeddable signal.
  2383. if (!gap.trim()) return;
  2384. if (gap.length <= maxChars) {
  2385. out.push({ text: gap, pos: start });
  2386. return;
  2387. }
  2388. // Reuse char-based algorithm for oversized gaps. Restrict break
  2389. // points and code fences to the gap window and rebase positions so
  2390. // chunkDocumentWithBreakPoints operates on a standalone slice.
  2391. const subPoints = regexPoints
  2392. .filter(p => p.pos >= start && p.pos < end)
  2393. .map(p => ({ ...p, pos: p.pos - start }));
  2394. const subFences = codeFences
  2395. .filter(f => f.end > start && f.start < end)
  2396. .map(f => ({
  2397. start: Math.max(0, f.start - start),
  2398. end: Math.max(0, Math.min(end, f.end) - start),
  2399. }));
  2400. const sub = chunkDocumentWithBreakPoints(gap, subPoints, subFences, maxChars, overlapChars, windowChars);
  2401. for (const c of sub) out.push({ text: c.text, pos: start + c.pos });
  2402. };
  2403. for (const range of ranges) {
  2404. // Emit any leading / inter-range gap (imports, top-level code).
  2405. emitGap(cursor, range.startIndex);
  2406. const body = content.slice(range.startIndex, range.endIndex);
  2407. if (body.length === 0) {
  2408. cursor = range.endIndex;
  2409. continue;
  2410. }
  2411. if (body.length <= maxChars) {
  2412. out.push({ text: body, pos: range.startIndex });
  2413. } else {
  2414. // Oversized function/class — split with char algorithm so we stay
  2415. // under the embed token budget. Break points inside the range are
  2416. // reused to keep splits at syntactically-sensible positions.
  2417. const subPoints = regexPoints
  2418. .filter(p => p.pos >= range.startIndex && p.pos < range.endIndex)
  2419. .map(p => ({ ...p, pos: p.pos - range.startIndex }));
  2420. const subFences = codeFences
  2421. .filter(f => f.end > range.startIndex && f.start < range.endIndex)
  2422. .map(f => ({
  2423. start: Math.max(0, f.start - range.startIndex),
  2424. end: Math.max(0, Math.min(range.endIndex, f.end) - range.startIndex),
  2425. }));
  2426. const sub = chunkDocumentWithBreakPoints(body, subPoints, subFences, maxChars, overlapChars, windowChars);
  2427. for (const c of sub) out.push({ text: c.text, pos: range.startIndex + c.pos });
  2428. }
  2429. cursor = range.endIndex;
  2430. }
  2431. // Trailing gap after the last range.
  2432. emitGap(cursor, content.length);
  2433. // Edge case: content consisted entirely of whitespace-only gaps (zero
  2434. // emitted chunks). Preserve the invariant that non-empty content yields
  2435. // at least one chunk.
  2436. if (out.length === 0 && content.length > 0) {
  2437. return [{ text: content, pos: 0 }];
  2438. }
  2439. return out;
  2440. }
  2441. /**
  2442. * Counts the tokens in `text`. Used by `chunkDocumentByTokens` for the
  2443. * safety re-split that splits chunks exceeding `maxTokens`.
  2444. *
  2445. * When `chunkDocumentByTokens` is called without a tokenizer, the disabled
  2446. * compatibility adapter returns typed HOLD.
  2447. *
  2448. * Commercial-provider callers pass a deterministic JS-only approximator. A char-based estimate like
  2449. * `Math.ceil(text.length / 3)` is a reasonable default — it matches the
  2450. * `avgCharsPerToken=3` heuristic used for the initial char-space chunk
  2451. * step, so the safety re-split stays a near no-op while populating the
  2452. * `tokens` field with a stable estimate.
  2453. */
  2454. export type TokenCounter = (text: string) => number | Promise<number>;
  2455. /**
  2456. * Chunk a document with an injected token counter.
  2457. *
  2458. * When `tokenizer` is supplied, no compatibility learned adapter is invoked.
  2459. *
  2460. * When `filepath` and `chunkStrategy` are provided, uses AST-aware break
  2461. * points for supported code files.
  2462. */
  2463. export async function chunkDocumentByTokens(
  2464. content: string,
  2465. maxTokens: number = CHUNK_SIZE_TOKENS,
  2466. overlapTokens: number = CHUNK_OVERLAP_TOKENS,
  2467. windowTokens: number = CHUNK_WINDOW_TOKENS,
  2468. filepath?: string,
  2469. chunkStrategy: ChunkStrategy = "regex",
  2470. signal?: AbortSignal,
  2471. tokenizer?: TokenCounter,
  2472. ): Promise<{ text: string; pos: number; tokens: number }[]> {
  2473. // Resolve token counter lazily so callers that supply `tokenizer` never
  2474. // touch the disabled compatibility adapter unless no tokenizer was supplied.
  2475. let llm: ReturnType<typeof getDefaultLlamaCpp> | undefined;
  2476. const countTokens: TokenCounter = tokenizer ?? (async (text: string) => {
  2477. if (!llm) llm = getDefaultLlamaCpp();
  2478. return (await llm.tokenize(text)).length;
  2479. });
  2480. // Use moderate chars/token estimate (prose ~4, code ~2, mixed ~3)
  2481. // If chunks exceed limit, they'll be re-split with actual ratio
  2482. const avgCharsPerToken = 3;
  2483. const maxChars = maxTokens * avgCharsPerToken;
  2484. const overlapChars = overlapTokens * avgCharsPerToken;
  2485. const windowChars = windowTokens * avgCharsPerToken;
  2486. // Chunk in character space with conservative estimate
  2487. // Use AST-aware chunking for the first pass when filepath/strategy provided
  2488. let charChunks = await chunkDocumentAsync(content, maxChars, overlapChars, windowChars, filepath, chunkStrategy);
  2489. // Tokenize and split any chunks that still exceed limit
  2490. const results: { text: string; pos: number; tokens: number }[] = [];
  2491. for (const chunk of charChunks) {
  2492. // Respect abort signal to avoid runaway tokenization
  2493. if (signal?.aborted) break;
  2494. const tokenCount = await countTokens(chunk.text);
  2495. if (tokenCount <= maxTokens) {
  2496. results.push({ text: chunk.text, pos: chunk.pos, tokens: tokenCount });
  2497. } else {
  2498. // Chunk is still too large - split it further
  2499. // Use actual token count to estimate better char limit
  2500. const actualCharsPerToken = chunk.text.length / tokenCount;
  2501. const safeMaxChars = Math.floor(maxTokens * actualCharsPerToken * 0.95); // 5% safety margin
  2502. const subChunks = chunkDocument(chunk.text, safeMaxChars, Math.floor(overlapChars * actualCharsPerToken / 2), Math.floor(windowChars * actualCharsPerToken / 2));
  2503. for (const subChunk of subChunks) {
  2504. if (signal?.aborted) break;
  2505. const subCount = await countTokens(subChunk.text);
  2506. results.push({
  2507. text: subChunk.text,
  2508. pos: chunk.pos + subChunk.pos,
  2509. tokens: subCount,
  2510. });
  2511. }
  2512. }
  2513. }
  2514. return results;
  2515. }
  2516. // =============================================================================
  2517. // Fuzzy matching
  2518. // =============================================================================
  2519. function levenshtein(a: string, b: string): number {
  2520. const m = a.length, n = b.length;
  2521. if (m === 0) return n;
  2522. if (n === 0) return m;
  2523. const dp: number[][] = Array.from({ length: m + 1 }, () => Array(n + 1).fill(0));
  2524. for (let i = 0; i <= m; i++) dp[i]![0] = i;
  2525. for (let j = 0; j <= n; j++) dp[0]![j] = j;
  2526. for (let i = 1; i <= m; i++) {
  2527. for (let j = 1; j <= n; j++) {
  2528. const cost = a[i - 1] === b[j - 1] ? 0 : 1;
  2529. dp[i]![j] = Math.min(
  2530. dp[i - 1]![j]! + 1,
  2531. dp[i]![j - 1]! + 1,
  2532. dp[i - 1]![j - 1]! + cost
  2533. );
  2534. }
  2535. }
  2536. return dp[m]![n]!;
  2537. }
  2538. /**
  2539. * Normalize a docid input by stripping surrounding quotes and leading #.
  2540. * Handles: "#abc123", 'abc123', "abc123", #abc123, abc123
  2541. * Returns the bare hex string.
  2542. */
  2543. export function normalizeDocid(docid: string): string {
  2544. let normalized = docid.trim();
  2545. // Strip surrounding quotes (single or double)
  2546. if ((normalized.startsWith('"') && normalized.endsWith('"')) ||
  2547. (normalized.startsWith("'") && normalized.endsWith("'"))) {
  2548. normalized = normalized.slice(1, -1);
  2549. }
  2550. // Strip leading # if present
  2551. if (normalized.startsWith('#')) {
  2552. normalized = normalized.slice(1);
  2553. }
  2554. return normalized;
  2555. }
  2556. /**
  2557. * Check if a string looks like a docid reference.
  2558. * Accepts: #abc123, abc123, "#abc123", "abc123", '#abc123', 'abc123'
  2559. * Returns true if the normalized form is a valid hex string of 6+ chars.
  2560. */
  2561. export function isDocid(input: string): boolean {
  2562. const normalized = normalizeDocid(input);
  2563. // Must be at least 6 hex characters
  2564. return normalized.length >= 6 && /^[a-f0-9]+$/i.test(normalized);
  2565. }
  2566. /**
  2567. * Find a document by its short docid (first 6 characters of hash).
  2568. * Returns the document's virtual path if found, null otherwise.
  2569. * If multiple documents match the same short hash (collision), returns the first one.
  2570. *
  2571. * Accepts lenient input: #abc123, abc123, "#abc123", "abc123"
  2572. */
  2573. export function findDocumentByDocid(db: Database, docid: string): { filepath: string; hash: string } | null {
  2574. const shortHash = normalizeDocid(docid);
  2575. if (shortHash.length < 1) return null;
  2576. // Look up documents where hash starts with the short hash
  2577. const doc = db.prepare(`
  2578. SELECT 'qmd://' || d.collection || '/' || d.path as filepath, d.hash
  2579. FROM documents d
  2580. WHERE d.hash LIKE ? AND d.active = 1
  2581. LIMIT 1
  2582. `).get(`${shortHash}%`) as { filepath: string; hash: string } | null;
  2583. return doc;
  2584. }
  2585. export function findSimilarFiles(db: Database, query: string, maxDistance: number = 3, limit: number = 5): string[] {
  2586. const allFiles = db.prepare(`
  2587. SELECT d.path
  2588. FROM documents d
  2589. WHERE d.active = 1
  2590. `).all() as { path: string }[];
  2591. const queryLower = query.toLowerCase();
  2592. const scored = allFiles
  2593. .map(f => ({ path: f.path, dist: levenshtein(f.path.toLowerCase(), queryLower) }))
  2594. .filter(f => f.dist <= maxDistance)
  2595. .sort((a, b) => a.dist - b.dist)
  2596. .slice(0, limit);
  2597. return scored.map(f => f.path);
  2598. }
  2599. export function matchFilesByGlob(db: Database, pattern: string): { filepath: string; displayPath: string; bodyLength: number }[] {
  2600. const allFiles = db.prepare(`
  2601. SELECT
  2602. 'qmd://' || d.collection || '/' || d.path as virtual_path,
  2603. LENGTH(content.doc) as body_length,
  2604. d.path,
  2605. d.collection
  2606. FROM documents d
  2607. JOIN content ON content.hash = d.hash
  2608. WHERE d.active = 1
  2609. `).all() as { virtual_path: string; body_length: number; path: string; collection: string }[];
  2610. const isMatch = picomatch(pattern);
  2611. return allFiles
  2612. .filter(f => isMatch(f.virtual_path) || isMatch(f.path) || isMatch(f.collection + '/' + f.path))
  2613. .map(f => ({
  2614. filepath: f.virtual_path, // Virtual path for precise lookup
  2615. displayPath: f.path, // Relative path for display
  2616. bodyLength: f.body_length
  2617. }));
  2618. }
  2619. // =============================================================================
  2620. // Context
  2621. // =============================================================================
  2622. /**
  2623. * Get context for a file path using hierarchical inheritance.
  2624. * Contexts are collection-scoped and inherit from parent directories.
  2625. * For example, context at "/talks" applies to "/talks/2024/keynote.md".
  2626. *
  2627. * @param db Database instance (unused - kept for compatibility)
  2628. * @param collectionName Collection name
  2629. * @param path Relative path within the collection
  2630. * @returns Context string or null if no context is defined
  2631. */
  2632. export function getContextForPath(db: Database, collectionName: string, path: string): string | null {
  2633. const coll = getStoreCollection(db, collectionName);
  2634. if (!coll) return null;
  2635. // Collect ALL matching contexts (global + all path prefixes)
  2636. const contexts: string[] = [];
  2637. // Add global context if present
  2638. const globalCtx = getStoreGlobalContext(db);
  2639. if (globalCtx) {
  2640. contexts.push(globalCtx);
  2641. }
  2642. // Add all matching path contexts (from most general to most specific)
  2643. if (coll.context) {
  2644. const normalizedPath = path.startsWith("/") ? path : `/${path}`;
  2645. // Collect all matching prefixes
  2646. const matchingContexts: { prefix: string; context: string }[] = [];
  2647. for (const [prefix, context] of Object.entries(coll.context)) {
  2648. const normalizedPrefix = prefix.startsWith("/") ? prefix : `/${prefix}`;
  2649. if (normalizedPath.startsWith(normalizedPrefix)) {
  2650. matchingContexts.push({ prefix: normalizedPrefix, context });
  2651. }
  2652. }
  2653. // Sort by prefix length (shortest/most general first)
  2654. matchingContexts.sort((a, b) => a.prefix.length - b.prefix.length);
  2655. // Add all matching contexts
  2656. for (const match of matchingContexts) {
  2657. contexts.push(match.context);
  2658. }
  2659. }
  2660. // Join all contexts with double newline
  2661. return contexts.length > 0 ? contexts.join('\n\n') : null;
  2662. }
  2663. /**
  2664. * Get context for a file path (virtual or filesystem).
  2665. * Resolves the collection and relative path from the DB store_collections table.
  2666. */
  2667. export function getContextForFile(db: Database, filepath: string): string | null {
  2668. // Handle undefined or null filepath
  2669. if (!filepath) return null;
  2670. // Get all collections from DB
  2671. const collections = getStoreCollections(db);
  2672. // Parse virtual path format: qmd://collection/path
  2673. let collectionName: string | null = null;
  2674. let relativePath: string | null = null;
  2675. const parsedVirtual = filepath.startsWith('qmd://') ? parseVirtualPath(filepath) : null;
  2676. if (parsedVirtual) {
  2677. collectionName = parsedVirtual.collectionName;
  2678. relativePath = parsedVirtual.path;
  2679. } else {
  2680. // Filesystem path: find which collection this absolute path belongs to
  2681. for (const coll of collections) {
  2682. // Skip collections with missing paths
  2683. if (!coll || !coll.path) continue;
  2684. if (filepath.startsWith(coll.path + '/') || filepath === coll.path) {
  2685. collectionName = coll.name;
  2686. // Extract relative path
  2687. relativePath = filepath.startsWith(coll.path + '/')
  2688. ? filepath.slice(coll.path.length + 1)
  2689. : '';
  2690. break;
  2691. }
  2692. }
  2693. if (!collectionName || relativePath === null) return null;
  2694. }
  2695. // Get the collection from DB
  2696. const coll = getStoreCollection(db, collectionName);
  2697. if (!coll) return null;
  2698. // Verify this document exists in the database
  2699. const doc = db.prepare(`
  2700. SELECT d.path
  2701. FROM documents d
  2702. WHERE d.collection = ? AND d.path = ? AND d.active = 1
  2703. LIMIT 1
  2704. `).get(collectionName, relativePath) as { path: string } | null;
  2705. if (!doc) return null;
  2706. // Collect ALL matching contexts (global + all path prefixes)
  2707. const contexts: string[] = [];
  2708. // Add global context if present
  2709. const globalCtx = getStoreGlobalContext(db);
  2710. if (globalCtx) {
  2711. contexts.push(globalCtx);
  2712. }
  2713. // Add all matching path contexts (from most general to most specific)
  2714. if (coll.context) {
  2715. const normalizedPath = relativePath.startsWith("/") ? relativePath : `/${relativePath}`;
  2716. // Collect all matching prefixes
  2717. const matchingContexts: { prefix: string; context: string }[] = [];
  2718. for (const [prefix, context] of Object.entries(coll.context)) {
  2719. const normalizedPrefix = prefix.startsWith("/") ? prefix : `/${prefix}`;
  2720. if (normalizedPath.startsWith(normalizedPrefix)) {
  2721. matchingContexts.push({ prefix: normalizedPrefix, context });
  2722. }
  2723. }
  2724. // Sort by prefix length (shortest/most general first)
  2725. matchingContexts.sort((a, b) => a.prefix.length - b.prefix.length);
  2726. // Add all matching contexts
  2727. for (const match of matchingContexts) {
  2728. contexts.push(match.context);
  2729. }
  2730. }
  2731. // Join all contexts with double newline
  2732. return contexts.length > 0 ? contexts.join('\n\n') : null;
  2733. }
  2734. /**
  2735. * Get collection by name from DB store_collections table.
  2736. */
  2737. export function getCollectionByName(db: Database, name: string): { name: string; pwd: string; glob_pattern: string } | null {
  2738. const collection = getStoreCollection(db, name);
  2739. if (!collection) return null;
  2740. return {
  2741. name: collection.name,
  2742. pwd: collection.path,
  2743. glob_pattern: collection.pattern,
  2744. };
  2745. }
  2746. /**
  2747. * List all collections with document counts from database.
  2748. * Merges store_collections config with database statistics.
  2749. */
  2750. export function listCollections(db: Database): { name: string; pwd: string; glob_pattern: string; doc_count: number; active_count: number; last_modified: string | null; includeByDefault: boolean }[] {
  2751. const collections = getStoreCollections(db);
  2752. // Get document counts from database for each collection
  2753. const result = collections.map(coll => {
  2754. const stats = db.prepare(`
  2755. SELECT
  2756. COUNT(d.id) as doc_count,
  2757. SUM(CASE WHEN d.active = 1 THEN 1 ELSE 0 END) as active_count,
  2758. MAX(d.modified_at) as last_modified
  2759. FROM documents d
  2760. WHERE d.collection = ?
  2761. `).get(coll.name) as { doc_count: number; active_count: number; last_modified: string | null } | null;
  2762. return {
  2763. name: coll.name,
  2764. pwd: coll.path,
  2765. glob_pattern: coll.pattern,
  2766. doc_count: stats?.doc_count || 0,
  2767. active_count: stats?.active_count || 0,
  2768. last_modified: stats?.last_modified || null,
  2769. includeByDefault: coll.includeByDefault !== false,
  2770. };
  2771. });
  2772. return result;
  2773. }
  2774. /**
  2775. * Remove a collection and clean up its documents.
  2776. * Uses collections.ts to remove from YAML config and cleans up database.
  2777. */
  2778. export function removeCollection(db: Database, collectionName: string): { deletedDocs: number; cleanedHashes: number } {
  2779. // Delete documents from database
  2780. const docResult = db.prepare(`DELETE FROM documents WHERE collection = ?`).run(collectionName);
  2781. // Clean up orphaned content hashes
  2782. const cleanupResult = db.prepare(`
  2783. DELETE FROM content
  2784. WHERE hash NOT IN (SELECT DISTINCT hash FROM documents WHERE active = 1)
  2785. `).run();
  2786. // Remove from store_collections
  2787. deleteStoreCollection(db, collectionName);
  2788. return {
  2789. deletedDocs: docResult.changes,
  2790. cleanedHashes: cleanupResult.changes
  2791. };
  2792. }
  2793. /**
  2794. * Rename a collection.
  2795. * Updates both YAML config and database documents table.
  2796. */
  2797. export function renameCollection(db: Database, oldName: string, newName: string): void {
  2798. // Update all documents with the new collection name in database
  2799. db.prepare(`UPDATE documents SET collection = ? WHERE collection = ?`)
  2800. .run(newName, oldName);
  2801. // Rename in store_collections
  2802. renameStoreCollection(db, oldName, newName);
  2803. }
  2804. // =============================================================================
  2805. // Context Management Operations
  2806. // =============================================================================
  2807. /**
  2808. * Insert or update a context for a specific collection and path prefix.
  2809. */
  2810. export function insertContext(db: Database, collectionId: number, pathPrefix: string, context: string): void {
  2811. // Get collection name from ID
  2812. const coll = db.prepare(`SELECT name FROM collections WHERE id = ?`).get(collectionId) as { name: string } | null;
  2813. if (!coll) {
  2814. throw new Error(`Collection with id ${collectionId} not found`);
  2815. }
  2816. // Add context to store_collections
  2817. updateStoreContext(db, coll.name, pathPrefix, context);
  2818. }
  2819. /**
  2820. * Delete a context for a specific collection and path prefix.
  2821. * Returns the number of contexts deleted.
  2822. */
  2823. export function deleteContext(db: Database, collectionName: string, pathPrefix: string): number {
  2824. // Remove context from store_collections
  2825. const success = removeStoreContext(db, collectionName, pathPrefix);
  2826. return success ? 1 : 0;
  2827. }
  2828. /**
  2829. * Delete all global contexts (contexts with empty path_prefix).
  2830. * Returns the number of contexts deleted.
  2831. */
  2832. export function deleteGlobalContexts(db: Database): number {
  2833. let deletedCount = 0;
  2834. // Remove global context
  2835. setStoreGlobalContext(db, undefined);
  2836. deletedCount++;
  2837. // Remove root context (empty string) from all collections
  2838. const collections = getStoreCollections(db);
  2839. for (const coll of collections) {
  2840. const success = removeStoreContext(db, coll.name, '');
  2841. if (success) {
  2842. deletedCount++;
  2843. }
  2844. }
  2845. return deletedCount;
  2846. }
  2847. /**
  2848. * List all contexts, grouped by collection.
  2849. * Returns contexts ordered by collection name, then by path prefix length (longest first).
  2850. */
  2851. export function listPathContexts(db: Database): { collection_name: string; path_prefix: string; context: string }[] {
  2852. const allContexts = getStoreContexts(db);
  2853. // Convert to expected format and sort
  2854. return allContexts.map(ctx => ({
  2855. collection_name: ctx.collection,
  2856. path_prefix: ctx.path,
  2857. context: ctx.context,
  2858. })).sort((a, b) => {
  2859. // Sort by collection name first
  2860. if (a.collection_name !== b.collection_name) {
  2861. return a.collection_name.localeCompare(b.collection_name);
  2862. }
  2863. // Then by path prefix length (longest first)
  2864. if (a.path_prefix.length !== b.path_prefix.length) {
  2865. return b.path_prefix.length - a.path_prefix.length;
  2866. }
  2867. // Then alphabetically
  2868. return a.path_prefix.localeCompare(b.path_prefix);
  2869. });
  2870. }
  2871. /**
  2872. * Get all collections (name only - from YAML config).
  2873. */
  2874. export function getAllCollections(db: Database): { name: string }[] {
  2875. const collections = getStoreCollections(db);
  2876. return collections.map(c => ({ name: c.name }));
  2877. }
  2878. /**
  2879. * Check which collections don't have any context defined.
  2880. * Returns collections that have no context entries at all (not even root context).
  2881. */
  2882. export function getCollectionsWithoutContext(db: Database): { name: string; pwd: string; doc_count: number }[] {
  2883. // Get all collections from DB
  2884. const allCollections = getStoreCollections(db);
  2885. // Filter to those without context
  2886. const collectionsWithoutContext: { name: string; pwd: string; doc_count: number }[] = [];
  2887. for (const coll of allCollections) {
  2888. // Check if collection has any context
  2889. if (!coll.context || Object.keys(coll.context).length === 0) {
  2890. // Get doc count from database
  2891. const stats = db.prepare(`
  2892. SELECT COUNT(d.id) as doc_count
  2893. FROM documents d
  2894. WHERE d.collection = ? AND d.active = 1
  2895. `).get(coll.name) as { doc_count: number } | null;
  2896. collectionsWithoutContext.push({
  2897. name: coll.name,
  2898. pwd: coll.path,
  2899. doc_count: stats?.doc_count || 0,
  2900. });
  2901. }
  2902. }
  2903. return collectionsWithoutContext.sort((a, b) => a.name.localeCompare(b.name));
  2904. }
  2905. /**
  2906. * Get top-level directories in a collection that don't have context.
  2907. * Useful for suggesting where context might be needed.
  2908. */
  2909. export function getTopLevelPathsWithoutContext(db: Database, collectionName: string): string[] {
  2910. // Get all paths in the collection from database
  2911. const paths = db.prepare(`
  2912. SELECT DISTINCT path FROM documents
  2913. WHERE collection = ? AND active = 1
  2914. `).all(collectionName) as { path: string }[];
  2915. // Get existing contexts for this collection from DB
  2916. const dbColl = getStoreCollection(db, collectionName);
  2917. if (!dbColl) return [];
  2918. const contextPrefixes = new Set<string>();
  2919. if (dbColl.context) {
  2920. for (const prefix of Object.keys(dbColl.context)) {
  2921. contextPrefixes.add(prefix);
  2922. }
  2923. }
  2924. // Extract top-level directories (first path component)
  2925. const topLevelDirs = new Set<string>();
  2926. for (const { path } of paths) {
  2927. const parts = path.split('/').filter(Boolean);
  2928. if (parts.length > 1) {
  2929. const dir = parts[0];
  2930. if (dir) topLevelDirs.add(dir);
  2931. }
  2932. }
  2933. // Filter out directories that already have context (exact or parent)
  2934. const missing: string[] = [];
  2935. for (const dir of topLevelDirs) {
  2936. let hasContext = false;
  2937. // Check if this dir or any parent has context
  2938. for (const prefix of contextPrefixes) {
  2939. if (prefix === '' || prefix === dir || dir.startsWith(prefix + '/')) {
  2940. hasContext = true;
  2941. break;
  2942. }
  2943. }
  2944. if (!hasContext) {
  2945. missing.push(dir);
  2946. }
  2947. }
  2948. return missing.sort();
  2949. }
  2950. // =============================================================================
  2951. // FTS Search
  2952. // =============================================================================
  2953. export function sanitizeFTS5Term(term: string): string {
  2954. return term.replace(/[^\p{L}\p{N}'_]/gu, '').toLowerCase();
  2955. }
  2956. /**
  2957. * Check if a token is a hyphenated compound word (e.g., multi-agent, DEC-0054, gpt-4).
  2958. * Returns true if the token contains internal hyphens between word/digit characters.
  2959. */
  2960. function isHyphenatedToken(token: string): boolean {
  2961. return /^[\p{L}\p{N}][\p{L}\p{N}'-]*-[\p{L}\p{N}][\p{L}\p{N}'-]*$/u.test(token);
  2962. }
  2963. /**
  2964. * Sanitize a hyphenated term into an FTS5 phrase by splitting on hyphens
  2965. * and sanitizing each part. Returns the parts joined by spaces for use
  2966. * inside FTS5 quotes: "multi agent" matches "multi-agent" in porter tokenizer.
  2967. */
  2968. function sanitizeHyphenatedTerm(term: string): string {
  2969. return term.split('-').map(t => sanitizeFTS5Term(t)).filter(t => t).join(' ');
  2970. }
  2971. /**
  2972. * Parse lex query syntax into FTS5 query.
  2973. *
  2974. * Supports:
  2975. * - Quoted phrases: "exact phrase" → "exact phrase" (exact match)
  2976. * - Negation: -term or -"phrase" → uses FTS5 NOT operator
  2977. * - Hyphenated tokens: multi-agent, DEC-0054, gpt-4 → treated as phrases
  2978. * - Plain terms: term → "term"* (prefix match)
  2979. *
  2980. * FTS5 NOT is a binary operator: `term1 NOT term2` means "match term1 but not term2".
  2981. * So `-term` only works when there are also positive terms.
  2982. *
  2983. * Hyphen disambiguation: `-sports` at a word boundary is negation, but `multi-agent`
  2984. * (where `-` is between word characters) is treated as a hyphenated phrase.
  2985. * When a leading `-` is followed by what looks like a hyphenated compound word
  2986. * (e.g., `-multi-agent`), the entire token is treated as a negated phrase.
  2987. *
  2988. * Examples:
  2989. * performance -sports → "performance"* NOT "sports"*
  2990. * "machine learning" → "machine learning"
  2991. * multi-agent memory → "multi agent" AND "memory"*
  2992. * DEC-0054 → "dec 0054"
  2993. * -multi-agent → NOT "multi agent"
  2994. */
  2995. function buildFTS5Query(query: string): string | null {
  2996. const positive: string[] = [];
  2997. const negative: string[] = [];
  2998. let i = 0;
  2999. const s = query.trim();
  3000. while (i < s.length) {
  3001. // Skip whitespace
  3002. while (i < s.length && /\s/.test(s[i]!)) i++;
  3003. if (i >= s.length) break;
  3004. // Check for negation prefix
  3005. const negated = s[i] === '-';
  3006. if (negated) i++;
  3007. // Check for quoted phrase
  3008. if (s[i] === '"') {
  3009. const start = i + 1;
  3010. i++;
  3011. while (i < s.length && s[i] !== '"') i++;
  3012. const phrase = s.slice(start, i).trim();
  3013. i++; // skip closing quote
  3014. if (phrase.length > 0) {
  3015. const sanitized = phrase.split(/\s+/).map(t => sanitizeFTS5Term(t)).filter(t => t).join(' ');
  3016. if (sanitized) {
  3017. const ftsPhrase = `"${sanitized}"`; // Exact phrase, no prefix match
  3018. if (negated) {
  3019. negative.push(ftsPhrase);
  3020. } else {
  3021. positive.push(ftsPhrase);
  3022. }
  3023. }
  3024. }
  3025. } else {
  3026. // Plain term (until whitespace or quote)
  3027. const start = i;
  3028. while (i < s.length && !/[\s"]/.test(s[i]!)) i++;
  3029. const term = s.slice(start, i);
  3030. // Handle hyphenated tokens: multi-agent, DEC-0054, gpt-4
  3031. // These get split into phrase queries so FTS5 porter tokenizer matches them.
  3032. if (isHyphenatedToken(term)) {
  3033. const sanitized = sanitizeHyphenatedTerm(term);
  3034. if (sanitized) {
  3035. const ftsPhrase = `"${sanitized}"`; // Phrase match (no prefix)
  3036. if (negated) {
  3037. negative.push(ftsPhrase);
  3038. } else {
  3039. positive.push(ftsPhrase);
  3040. }
  3041. }
  3042. } else {
  3043. const sanitized = sanitizeFTS5Term(term);
  3044. if (sanitized) {
  3045. const ftsTerm = `"${sanitized}"*`; // Prefix match
  3046. if (negated) {
  3047. negative.push(ftsTerm);
  3048. } else {
  3049. positive.push(ftsTerm);
  3050. }
  3051. }
  3052. }
  3053. }
  3054. }
  3055. if (positive.length === 0 && negative.length === 0) return null;
  3056. // If only negative terms, we can't search (FTS5 NOT is binary)
  3057. if (positive.length === 0) return null;
  3058. // Join positive terms with AND
  3059. let result = positive.join(' AND ');
  3060. // Add NOT clause for negative terms
  3061. for (const neg of negative) {
  3062. result = `${result} NOT ${neg}`;
  3063. }
  3064. return result;
  3065. }
  3066. /**
  3067. * Validate that a vec/hyde query doesn't use lex-only syntax.
  3068. * Returns error message if invalid, null if valid.
  3069. *
  3070. * Negation is detected ONLY when `-` is preceded by whitespace or sits at
  3071. * the start of the query. Hyphens inside words (e.g. `auto-archived`,
  3072. * `pre-commit`, `multi-session`, `state-of-the-art`) carry no negation
  3073. * semantics in natural English and must pass through unchanged.
  3074. */
  3075. export function validateSemanticQuery(query: string): string | null {
  3076. // `-term` or `-"phrase"` only counts as negation at SOS or after whitespace.
  3077. if (/(?:^|\s)-\w/.test(query) || /(?:^|\s)-"/.test(query)) {
  3078. return 'Negation (-term) is not supported in vec/hyde queries. Use lex for exclusions.';
  3079. }
  3080. return null;
  3081. }
  3082. export function validateLexQuery(query: string): string | null {
  3083. if (/[\r\n]/.test(query)) {
  3084. return 'Lex queries must be a single line. Remove newline characters or split into separate lex: lines.';
  3085. }
  3086. const quoteCount = (query.match(/"/g) ?? []).length;
  3087. if (quoteCount % 2 === 1) {
  3088. return 'Lex query has an unmatched double quote ("). Add the closing quote or remove it.';
  3089. }
  3090. return null;
  3091. }
  3092. export function searchFTS(db: Database, query: string, limit: number = 20, collectionName?: string): SearchResult[] {
  3093. const ftsQuery = buildFTS5Query(query);
  3094. if (!ftsQuery) return [];
  3095. // Use a CTE to force FTS5 to run first, then filter by collection.
  3096. // Without the CTE, SQLite's query planner combines FTS5 MATCH with the
  3097. // collection filter in a single WHERE clause, which can cause it to
  3098. // abandon the FTS5 index and fall back to a full scan — turning an 8ms
  3099. // query into a 17-second query on large collections.
  3100. const params: (string | number)[] = [ftsQuery];
  3101. // When filtering by collection, fetch extra candidates from the FTS index
  3102. // since some will be filtered out. Without a collection filter we can
  3103. // fetch exactly the requested limit.
  3104. const ftsLimit = collectionName ? limit * 10 : limit;
  3105. let sql = `
  3106. WITH fts_matches AS (
  3107. SELECT rowid, bm25(documents_fts, 1.5, 4.0, 1.0) as bm25_score
  3108. FROM documents_fts
  3109. WHERE documents_fts MATCH ?
  3110. ORDER BY bm25_score ASC
  3111. LIMIT ${ftsLimit}
  3112. )
  3113. SELECT
  3114. 'qmd://' || d.collection || '/' || d.path as filepath,
  3115. d.collection || '/' || d.path as display_path,
  3116. d.title,
  3117. content.doc as body,
  3118. d.hash,
  3119. fm.bm25_score
  3120. FROM fts_matches fm
  3121. JOIN documents d ON d.id = fm.rowid
  3122. JOIN content ON content.hash = d.hash
  3123. WHERE d.active = 1
  3124. `;
  3125. if (collectionName) {
  3126. sql += ` AND d.collection = ?`;
  3127. params.push(String(collectionName));
  3128. }
  3129. // bm25 lower is better; sort ascending.
  3130. sql += ` ORDER BY fm.bm25_score ASC LIMIT ?`;
  3131. params.push(limit);
  3132. const rows = db.prepare(sql).all(...params) as { filepath: string; display_path: string; title: string; body: string; hash: string; bm25_score: number }[];
  3133. return rows.map(row => {
  3134. const collectionName = row.filepath.split('//')[1]?.split('/')[0] || "";
  3135. // Convert bm25 (negative, lower is better) into a stable [0..1) score where higher is better.
  3136. // FTS5 BM25 scores are negative (e.g., -10 is strong, -2 is weak).
  3137. // |x| / (1 + |x|) maps: strong(-10)→0.91, medium(-2)→0.67, weak(-0.5)→0.33, none(0)→0.
  3138. // Monotonic and query-independent — no per-query normalization needed.
  3139. const score = Math.abs(row.bm25_score) / (1 + Math.abs(row.bm25_score));
  3140. return {
  3141. filepath: row.filepath,
  3142. displayPath: row.display_path,
  3143. title: row.title,
  3144. hash: row.hash,
  3145. docid: getDocid(row.hash),
  3146. collectionName,
  3147. modifiedAt: "", // Not available in FTS query
  3148. bodyLength: row.body.length,
  3149. body: row.body,
  3150. context: getContextForFile(db, row.filepath),
  3151. score,
  3152. source: "fts" as const,
  3153. };
  3154. });
  3155. }
  3156. // =============================================================================
  3157. // Vector Search
  3158. // =============================================================================
  3159. export async function searchVec(db: Database, query: string, model: string, limit: number = 20, collectionName?: string, session?: ILLMSession, precomputedEmbedding?: number[], embedProvider?: EmbeddingProvider): Promise<SearchResult[]> {
  3160. const tableExists = db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
  3161. if (!tableExists) return [];
  3162. const embedding = precomputedEmbedding ?? await getEmbedding(query, model, true, session, undefined, embedProvider);
  3163. if (!embedding) return [];
  3164. // IMPORTANT: We use a two-step query approach here because sqlite-vec virtual tables
  3165. // hang indefinitely when combined with JOINs in the same query. Do NOT try to
  3166. // "optimize" this by combining into a single query with JOINs - it will break.
  3167. // See: https://github.com/tobi/qmd/pull/23
  3168. // Step 1: Get vector matches from sqlite-vec (no JOINs allowed)
  3169. const vecResults = db.prepare(`
  3170. SELECT hash_seq, distance
  3171. FROM vectors_vec
  3172. WHERE embedding MATCH ? AND k = ?
  3173. `).all(new Float32Array(embedding), limit * 3) as { hash_seq: string; distance: number }[];
  3174. if (vecResults.length === 0) return [];
  3175. // Step 2: Get chunk info and document data
  3176. const hashSeqs = vecResults.map(r => r.hash_seq);
  3177. const distanceMap = new Map(vecResults.map(r => [r.hash_seq, r.distance]));
  3178. // Build query for document lookup
  3179. const placeholders = hashSeqs.map(() => '?').join(',');
  3180. let docSql = `
  3181. SELECT
  3182. cv.hash || '_' || cv.seq as hash_seq,
  3183. cv.hash,
  3184. cv.pos,
  3185. 'qmd://' || d.collection || '/' || d.path as filepath,
  3186. d.collection || '/' || d.path as display_path,
  3187. d.title,
  3188. content.doc as body
  3189. FROM content_vectors cv
  3190. JOIN documents d ON d.hash = cv.hash AND d.active = 1
  3191. JOIN content ON content.hash = d.hash
  3192. WHERE cv.hash || '_' || cv.seq IN (${placeholders})
  3193. `;
  3194. const params: string[] = [...hashSeqs];
  3195. if (collectionName) {
  3196. docSql += ` AND d.collection = ?`;
  3197. params.push(collectionName);
  3198. }
  3199. const docRows = db.prepare(docSql).all(...params) as {
  3200. hash_seq: string; hash: string; pos: number; filepath: string;
  3201. display_path: string; title: string; body: string;
  3202. }[];
  3203. // Combine with distances and dedupe by filepath
  3204. const seen = new Map<string, { row: typeof docRows[0]; bestDist: number }>();
  3205. for (const row of docRows) {
  3206. const distance = distanceMap.get(row.hash_seq) ?? 1;
  3207. const existing = seen.get(row.filepath);
  3208. if (!existing || distance < existing.bestDist) {
  3209. seen.set(row.filepath, { row, bestDist: distance });
  3210. }
  3211. }
  3212. return Array.from(seen.values())
  3213. .sort((a, b) => a.bestDist - b.bestDist)
  3214. .slice(0, limit)
  3215. .map(({ row, bestDist }) => {
  3216. const collectionName = row.filepath.split('//')[1]?.split('/')[0] || "";
  3217. return {
  3218. filepath: row.filepath,
  3219. displayPath: row.display_path,
  3220. title: row.title,
  3221. hash: row.hash,
  3222. docid: getDocid(row.hash),
  3223. collectionName,
  3224. modifiedAt: "", // Not available in vec query
  3225. bodyLength: row.body.length,
  3226. body: row.body,
  3227. context: getContextForFile(db, row.filepath),
  3228. score: 1 - bestDist, // Cosine similarity = 1 - cosine distance
  3229. source: "vec" as const,
  3230. chunkPos: row.pos,
  3231. };
  3232. });
  3233. }
  3234. // =============================================================================
  3235. // Embeddings
  3236. // =============================================================================
  3237. async function getEmbedding(text: string, model: string, isQuery: boolean, session?: ILLMSession, llmOverride?: LlamaCpp, embedProvider?: EmbeddingProvider): Promise<number[] | null> {
  3238. // When an EmbeddingProvider is supplied, route the encoding through it
  3239. // through the approved commercial API. The provider sees the raw text + the desired
  3240. // model id; query-formatting prefixes are still applied via
  3241. // formatQueryForEmbedding so embedding parity with the index is preserved.
  3242. if (embedProvider) {
  3243. const providerModel = embedProvider.getModelId();
  3244. const formattedText = isQuery
  3245. ? formatQueryForEmbedding(text, providerModel)
  3246. : formatDocForEmbedding(text, undefined, providerModel);
  3247. const result = await embedProvider.embed(formattedText, { model: providerModel });
  3248. return result?.embedding ?? null;
  3249. }
  3250. // Format text using the appropriate prompt template
  3251. const formattedText = isQuery ? formatQueryForEmbedding(text, model) : formatDocForEmbedding(text, undefined, model);
  3252. const result = session
  3253. ? await session.embed(formattedText, { model, isQuery })
  3254. : await (llmOverride ?? getDefaultLlamaCpp()).embed(formattedText, { model, isQuery });
  3255. return result?.embedding || null;
  3256. }
  3257. /**
  3258. * Get all unique content hashes that need embeddings (from active documents).
  3259. * Returns hash, document body, and a sample path for display purposes.
  3260. */
  3261. export function getHashesForEmbedding(db: Database): { hash: string; body: string; path: string }[] {
  3262. return db.prepare(`
  3263. SELECT d.hash, c.doc as body, MIN(d.path) as path
  3264. FROM documents d
  3265. JOIN content c ON d.hash = c.hash
  3266. LEFT JOIN content_vectors v ON d.hash = v.hash AND v.seq = 0
  3267. WHERE d.active = 1 AND v.hash IS NULL
  3268. GROUP BY d.hash
  3269. `).all() as { hash: string; body: string; path: string }[];
  3270. }
  3271. /**
  3272. * Clear all embeddings from the database (force re-index).
  3273. * Deletes all rows from content_vectors and drops the vectors_vec table.
  3274. */
  3275. export function clearAllEmbeddings(db: Database): void {
  3276. db.exec(`DELETE FROM content_vectors`);
  3277. db.exec(`DROP TABLE IF EXISTS vectors_vec`);
  3278. }
  3279. /**
  3280. * Get the distinct set of model identifiers present in `content_vectors`.
  3281. *
  3282. * Used by the embedding migration-safety guard: if a configured provider's
  3283. * `getModelId()` does not appear in this list (and the table is non-empty),
  3284. * we refuse to embed and ask the user to run `qmd embed -f` to rebuild.
  3285. *
  3286. * Returns `[]` when the table is empty (fresh DB) — in which case any
  3287. * provider is allowed.
  3288. */
  3289. export function getDistinctEmbeddingModels(db: Database): string[] {
  3290. const rows = db.prepare(
  3291. `SELECT DISTINCT model FROM content_vectors WHERE model IS NOT NULL`,
  3292. ).all() as { model: string }[];
  3293. return rows.map((r) => r.model).filter((m) => typeof m === "string" && m.length > 0);
  3294. }
  3295. /**
  3296. * Insert a single embedding into both content_vectors and vectors_vec tables.
  3297. * The hash_seq key is formatted as "hash_seq" for the vectors_vec table.
  3298. *
  3299. * content_vectors is inserted first so that getHashesForEmbedding (which checks
  3300. * only content_vectors) won't re-select the hash on a crash between the two inserts.
  3301. *
  3302. * vectors_vec uses DELETE + INSERT instead of INSERT OR REPLACE because sqlite-vec's
  3303. * vec0 virtual tables silently ignore the OR REPLACE conflict clause.
  3304. */
  3305. export function insertEmbedding(
  3306. db: Database,
  3307. hash: string,
  3308. seq: number,
  3309. pos: number,
  3310. embedding: Float32Array,
  3311. model: string,
  3312. embeddedAt: string
  3313. ): void {
  3314. const hashSeq = `${hash}_${seq}`;
  3315. // Insert content_vectors first — crash-safe ordering (see getHashesForEmbedding)
  3316. const insertContentVectorStmt = db.prepare(`INSERT OR REPLACE INTO content_vectors (hash, seq, pos, model, embedded_at) VALUES (?, ?, ?, ?, ?)`);
  3317. insertContentVectorStmt.run(hash, seq, pos, model, embeddedAt);
  3318. // vec0 virtual tables don't support OR REPLACE — use DELETE + INSERT
  3319. const deleteVecStmt = db.prepare(`DELETE FROM vectors_vec WHERE hash_seq = ?`);
  3320. const insertVecStmt = db.prepare(`INSERT INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`);
  3321. deleteVecStmt.run(hashSeq);
  3322. insertVecStmt.run(hashSeq, embedding);
  3323. }
  3324. // =============================================================================
  3325. // Query expansion
  3326. // =============================================================================
  3327. export async function expandQuery(query: string, model: string = DEFAULT_QUERY_MODEL, db: Database, intent?: string, llmOverride?: LlamaCpp): Promise<ExpandedQuery[]> {
  3328. // Check cache first — stored as JSON preserving types
  3329. const cacheKey = getCacheKey("expandQuery", { query, model, ...(intent && { intent }) });
  3330. const cached = getCachedResult(db, cacheKey);
  3331. if (cached) {
  3332. try {
  3333. const parsed = JSON.parse(cached) as any[];
  3334. // Migrate old cache format: { type, text } → { type, query }
  3335. if (parsed.length > 0 && parsed[0].query) {
  3336. return parsed as ExpandedQuery[];
  3337. } else if (parsed.length > 0 && parsed[0].text) {
  3338. return parsed.map((r: any) => ({ type: r.type, query: r.text }));
  3339. }
  3340. } catch {
  3341. // Old cache format (pre-typed, newline-separated text) — re-expand
  3342. }
  3343. }
  3344. const llm = llmOverride ?? getDefaultLlamaCpp();
  3345. // Note: LlamaCpp uses hardcoded model, model parameter is ignored
  3346. const results = await llm.expandQuery(query, { intent });
  3347. // Map Queryable[] → ExpandedQuery[] (same shape, decoupled from llm.ts internals).
  3348. // Filter out entries that duplicate the original query text.
  3349. const expanded: ExpandedQuery[] = results
  3350. .filter(r => r.text !== query)
  3351. .map(r => ({ type: r.type, query: r.text }));
  3352. if (expanded.length > 0) {
  3353. setCachedResult(db, cacheKey, JSON.stringify(expanded));
  3354. }
  3355. return expanded;
  3356. }
  3357. // =============================================================================
  3358. // Reranking
  3359. // =============================================================================
  3360. export async function rerank(query: string, documents: { file: string; text: string }[], model: string = DEFAULT_RERANK_MODEL, db: Database, intent?: string, llmOverride?: LlamaCpp): Promise<{ file: string; score: number }[]> {
  3361. // Prepend intent to rerank query so the reranker scores with domain context
  3362. const rerankQuery = intent ? `${intent}\n\n${query}` : query;
  3363. const cachedResults: Map<string, number> = new Map();
  3364. const uncachedDocsByChunk: Map<string, RerankDocument> = new Map();
  3365. // Check cache for each document
  3366. // Cache key includes chunk text — different queries can select different chunks
  3367. // from the same file, and the reranker score depends on which chunk was sent.
  3368. // File path is excluded from the new cache key because the reranker score
  3369. // depends on the chunk content, not where it came from.
  3370. for (const doc of documents) {
  3371. const cacheKey = getCacheKey("rerank", { query: rerankQuery, model, chunk: doc.text });
  3372. const legacyCacheKey = getCacheKey("rerank", { query, file: doc.file, model, chunk: doc.text });
  3373. const cached = getCachedResult(db, cacheKey) ?? getCachedResult(db, legacyCacheKey);
  3374. if (cached !== null) {
  3375. cachedResults.set(doc.text, parseFloat(cached));
  3376. } else {
  3377. uncachedDocsByChunk.set(doc.text, { file: doc.file, text: doc.text });
  3378. }
  3379. }
  3380. // Rerank uncached documents using LlamaCpp
  3381. if (uncachedDocsByChunk.size > 0) {
  3382. const llm = llmOverride ?? getDefaultLlamaCpp();
  3383. const uncachedDocs = [...uncachedDocsByChunk.values()];
  3384. const rerankResult = await llm.rerank(rerankQuery, uncachedDocs, { model });
  3385. // Cache results by chunk text so identical chunks across files are scored once.
  3386. const textByFile = new Map(uncachedDocs.map(d => [d.file, d.text]));
  3387. for (const result of rerankResult.results) {
  3388. const chunk = textByFile.get(result.file) || "";
  3389. const cacheKey = getCacheKey("rerank", { query: rerankQuery, model, chunk });
  3390. setCachedResult(db, cacheKey, result.score.toString());
  3391. cachedResults.set(chunk, result.score);
  3392. }
  3393. }
  3394. // Return all results sorted by score
  3395. return documents
  3396. .map(doc => ({ file: doc.file, score: cachedResults.get(doc.text) || 0 }))
  3397. .sort((a, b) => b.score - a.score);
  3398. }
  3399. // =============================================================================
  3400. // Reciprocal Rank Fusion
  3401. // =============================================================================
  3402. export function reciprocalRankFusion(
  3403. resultLists: RankedResult[][],
  3404. weights: number[] = [],
  3405. k: number = 60
  3406. ): RankedResult[] {
  3407. const scores = new Map<string, { result: RankedResult; rrfScore: number; topRank: number }>();
  3408. for (let listIdx = 0; listIdx < resultLists.length; listIdx++) {
  3409. const list = resultLists[listIdx];
  3410. if (!list) continue;
  3411. const weight = weights[listIdx] ?? 1.0;
  3412. for (let rank = 0; rank < list.length; rank++) {
  3413. const result = list[rank];
  3414. if (!result) continue;
  3415. const rrfContribution = weight / (k + rank + 1);
  3416. const existing = scores.get(result.file);
  3417. if (existing) {
  3418. existing.rrfScore += rrfContribution;
  3419. existing.topRank = Math.min(existing.topRank, rank);
  3420. } else {
  3421. scores.set(result.file, {
  3422. result,
  3423. rrfScore: rrfContribution,
  3424. topRank: rank,
  3425. });
  3426. }
  3427. }
  3428. }
  3429. // Top-rank bonus
  3430. for (const entry of scores.values()) {
  3431. if (entry.topRank === 0) {
  3432. entry.rrfScore += 0.05;
  3433. } else if (entry.topRank <= 2) {
  3434. entry.rrfScore += 0.02;
  3435. }
  3436. }
  3437. return Array.from(scores.values())
  3438. .sort((a, b) => b.rrfScore - a.rrfScore)
  3439. .map(e => ({ ...e.result, score: e.rrfScore }));
  3440. }
  3441. /**
  3442. * Build per-document RRF contribution traces for explain/debug output.
  3443. */
  3444. export function buildRrfTrace(
  3445. resultLists: RankedResult[][],
  3446. weights: number[] = [],
  3447. listMeta: RankedListMeta[] = [],
  3448. k: number = 60
  3449. ): Map<string, RRFScoreTrace> {
  3450. const traces = new Map<string, RRFScoreTrace>();
  3451. for (let listIdx = 0; listIdx < resultLists.length; listIdx++) {
  3452. const list = resultLists[listIdx];
  3453. if (!list) continue;
  3454. const weight = weights[listIdx] ?? 1.0;
  3455. const meta = listMeta[listIdx] ?? {
  3456. source: "fts",
  3457. queryType: "original",
  3458. query: "",
  3459. } as const;
  3460. for (let rank0 = 0; rank0 < list.length; rank0++) {
  3461. const result = list[rank0];
  3462. if (!result) continue;
  3463. const rank = rank0 + 1; // 1-indexed rank for explain output
  3464. const contribution = weight / (k + rank);
  3465. const existing = traces.get(result.file);
  3466. const detail: RRFContributionTrace = {
  3467. listIndex: listIdx,
  3468. source: meta.source,
  3469. queryType: meta.queryType,
  3470. query: meta.query,
  3471. rank,
  3472. weight,
  3473. backendScore: result.score,
  3474. rrfContribution: contribution,
  3475. };
  3476. if (existing) {
  3477. existing.baseScore += contribution;
  3478. existing.topRank = Math.min(existing.topRank, rank);
  3479. existing.contributions.push(detail);
  3480. } else {
  3481. traces.set(result.file, {
  3482. contributions: [detail],
  3483. baseScore: contribution,
  3484. topRank: rank,
  3485. topRankBonus: 0,
  3486. totalScore: 0,
  3487. });
  3488. }
  3489. }
  3490. }
  3491. for (const trace of traces.values()) {
  3492. let bonus = 0;
  3493. if (trace.topRank === 1) bonus = 0.05;
  3494. else if (trace.topRank <= 3) bonus = 0.02;
  3495. trace.topRankBonus = bonus;
  3496. trace.totalScore = trace.baseScore + bonus;
  3497. }
  3498. return traces;
  3499. }
  3500. // =============================================================================
  3501. // Document retrieval
  3502. // =============================================================================
  3503. type DbDocRow = {
  3504. virtual_path: string;
  3505. display_path: string;
  3506. title: string;
  3507. hash: string;
  3508. collection: string;
  3509. path: string;
  3510. modified_at: string;
  3511. body_length: number;
  3512. body?: string;
  3513. };
  3514. /**
  3515. * Find a document by filename/path, docid (#hash), or with fuzzy matching.
  3516. * Returns document metadata without body by default.
  3517. *
  3518. * Supports:
  3519. * - Virtual paths: qmd://collection/path/to/file.md
  3520. * - Absolute paths: /path/to/file.md
  3521. * - Relative paths: path/to/file.md
  3522. * - Short docid: #abc123 (first 6 chars of hash)
  3523. */
  3524. export function findDocument(db: Database, filename: string, options: { includeBody?: boolean } = {}): DocumentResult | DocumentNotFound {
  3525. let filepath = filename;
  3526. const colonMatch = filepath.match(/:(\d+)$/);
  3527. if (colonMatch) {
  3528. filepath = filepath.slice(0, -colonMatch[0].length);
  3529. }
  3530. // Check if this is a docid lookup (#abc123, abc123, "#abc123", "abc123", etc.)
  3531. if (isDocid(filepath)) {
  3532. const docidMatch = findDocumentByDocid(db, filepath);
  3533. if (docidMatch) {
  3534. filepath = docidMatch.filepath;
  3535. } else {
  3536. return { error: "not_found", query: filename, similarFiles: [] };
  3537. }
  3538. }
  3539. if (filepath.startsWith('~/')) {
  3540. filepath = homedir() + filepath.slice(1);
  3541. }
  3542. const bodyCol = options.includeBody ? `, content.doc as body` : ``;
  3543. // Build computed columns
  3544. // Note: absoluteFilepath is computed from YAML collections after query
  3545. const selectCols = `
  3546. 'qmd://' || d.collection || '/' || d.path as virtual_path,
  3547. d.collection || '/' || d.path as display_path,
  3548. d.title,
  3549. d.hash,
  3550. d.collection,
  3551. d.modified_at,
  3552. LENGTH(content.doc) as body_length
  3553. ${bodyCol}
  3554. `;
  3555. // Try to match by virtual path first
  3556. let doc = db.prepare(`
  3557. SELECT ${selectCols}
  3558. FROM documents d
  3559. JOIN content ON content.hash = d.hash
  3560. WHERE 'qmd://' || d.collection || '/' || d.path = ? AND d.active = 1
  3561. `).get(filepath) as DbDocRow | null;
  3562. // Try fuzzy match by virtual path
  3563. if (!doc) {
  3564. doc = db.prepare(`
  3565. SELECT ${selectCols}
  3566. FROM documents d
  3567. JOIN content ON content.hash = d.hash
  3568. WHERE 'qmd://' || d.collection || '/' || d.path LIKE ? AND d.active = 1
  3569. LIMIT 1
  3570. `).get(`%${filepath}`) as DbDocRow | null;
  3571. }
  3572. // Try to match by absolute path (requires looking up collection paths from DB)
  3573. if (!doc && !filepath.startsWith('qmd://')) {
  3574. const collections = getStoreCollections(db);
  3575. for (const coll of collections) {
  3576. let relativePath: string | null = null;
  3577. // If filepath is absolute and starts with collection path, extract relative part
  3578. if (filepath.startsWith(coll.path + '/')) {
  3579. relativePath = filepath.slice(coll.path.length + 1);
  3580. }
  3581. // Otherwise treat filepath as relative to collection
  3582. else if (!filepath.startsWith('/')) {
  3583. relativePath = filepath;
  3584. }
  3585. if (relativePath) {
  3586. doc = db.prepare(`
  3587. SELECT ${selectCols}
  3588. FROM documents d
  3589. JOIN content ON content.hash = d.hash
  3590. WHERE d.collection = ? AND d.path = ? AND d.active = 1
  3591. `).get(coll.name, relativePath) as DbDocRow | null;
  3592. if (doc) break;
  3593. }
  3594. }
  3595. }
  3596. if (!doc) {
  3597. const similar = findSimilarFiles(db, filepath, 5, 5);
  3598. return { error: "not_found", query: filename, similarFiles: similar };
  3599. }
  3600. // Get context using virtual path
  3601. const virtualPath = doc.virtual_path || `qmd://${doc.collection}/${doc.display_path}`;
  3602. const context = getContextForFile(db, virtualPath);
  3603. return {
  3604. filepath: virtualPath,
  3605. displayPath: doc.display_path,
  3606. title: doc.title,
  3607. context,
  3608. hash: doc.hash,
  3609. docid: getDocid(doc.hash),
  3610. collectionName: doc.collection,
  3611. modifiedAt: doc.modified_at,
  3612. bodyLength: doc.body_length,
  3613. ...(options.includeBody && doc.body !== undefined && { body: doc.body }),
  3614. };
  3615. }
  3616. /**
  3617. * Get the body content for a document
  3618. * Optionally slice by line range
  3619. */
  3620. export function getDocumentBody(db: Database, doc: DocumentResult | { filepath: string }, fromLine?: number, maxLines?: number): string | null {
  3621. const filepath = doc.filepath;
  3622. // Try to resolve document by filepath (absolute or virtual)
  3623. let row: { body: string } | null = null;
  3624. // Try virtual path first
  3625. if (filepath.startsWith('qmd://')) {
  3626. row = db.prepare(`
  3627. SELECT content.doc as body
  3628. FROM documents d
  3629. JOIN content ON content.hash = d.hash
  3630. WHERE 'qmd://' || d.collection || '/' || d.path = ? AND d.active = 1
  3631. `).get(filepath) as { body: string } | null;
  3632. }
  3633. // Try absolute path by looking up in DB store_collections
  3634. if (!row) {
  3635. const collections = getStoreCollections(db);
  3636. for (const coll of collections) {
  3637. if (filepath.startsWith(coll.path + '/')) {
  3638. const relativePath = filepath.slice(coll.path.length + 1);
  3639. row = db.prepare(`
  3640. SELECT content.doc as body
  3641. FROM documents d
  3642. JOIN content ON content.hash = d.hash
  3643. WHERE d.collection = ? AND d.path = ? AND d.active = 1
  3644. `).get(coll.name, relativePath) as { body: string } | null;
  3645. if (row) break;
  3646. }
  3647. }
  3648. }
  3649. if (!row) return null;
  3650. let body = row.body;
  3651. if (fromLine !== undefined || maxLines !== undefined) {
  3652. const lines = body.split('\n');
  3653. const start = (fromLine || 1) - 1;
  3654. const end = maxLines !== undefined ? start + maxLines : lines.length;
  3655. body = lines.slice(start, end).join('\n');
  3656. }
  3657. return body;
  3658. }
  3659. /**
  3660. * Find multiple documents by glob pattern or comma-separated list
  3661. * Returns documents without body by default (use getDocumentBody to load)
  3662. */
  3663. export function findDocuments(
  3664. db: Database,
  3665. pattern: string,
  3666. options: { includeBody?: boolean; maxBytes?: number } = {}
  3667. ): { docs: MultiGetResult[]; errors: string[] } {
  3668. const isCommaSeparated = pattern.includes(',') && !pattern.includes('*') && !pattern.includes('?') && !pattern.includes('{');
  3669. const errors: string[] = [];
  3670. const maxBytes = options.maxBytes ?? DEFAULT_MULTI_GET_MAX_BYTES;
  3671. const bodyCol = options.includeBody ? `, content.doc as body` : ``;
  3672. const selectCols = `
  3673. 'qmd://' || d.collection || '/' || d.path as virtual_path,
  3674. d.collection || '/' || d.path as display_path,
  3675. d.title,
  3676. d.hash,
  3677. d.collection,
  3678. d.modified_at,
  3679. LENGTH(content.doc) as body_length
  3680. ${bodyCol}
  3681. `;
  3682. let fileRows: DbDocRow[];
  3683. if (isCommaSeparated) {
  3684. const names = pattern.split(',').map(s => s.trim()).filter(Boolean);
  3685. fileRows = [];
  3686. for (const name of names) {
  3687. let doc = db.prepare(`
  3688. SELECT ${selectCols}
  3689. FROM documents d
  3690. JOIN content ON content.hash = d.hash
  3691. WHERE 'qmd://' || d.collection || '/' || d.path = ? AND d.active = 1
  3692. `).get(name) as DbDocRow | null;
  3693. if (!doc) {
  3694. doc = db.prepare(`
  3695. SELECT ${selectCols}
  3696. FROM documents d
  3697. JOIN content ON content.hash = d.hash
  3698. WHERE 'qmd://' || d.collection || '/' || d.path LIKE ? AND d.active = 1
  3699. LIMIT 1
  3700. `).get(`%${name}`) as DbDocRow | null;
  3701. }
  3702. if (doc) {
  3703. fileRows.push(doc);
  3704. } else {
  3705. const similar = findSimilarFiles(db, name, 5, 3);
  3706. let msg = `File not found: ${name}`;
  3707. if (similar.length > 0) {
  3708. msg += ` (did you mean: ${similar.join(', ')}?)`;
  3709. }
  3710. errors.push(msg);
  3711. }
  3712. }
  3713. } else {
  3714. // Glob pattern match
  3715. const matched = matchFilesByGlob(db, pattern);
  3716. if (matched.length === 0) {
  3717. errors.push(`No files matched pattern: ${pattern}`);
  3718. return { docs: [], errors };
  3719. }
  3720. const virtualPaths = matched.map(m => m.filepath);
  3721. const placeholders = virtualPaths.map(() => '?').join(',');
  3722. fileRows = db.prepare(`
  3723. SELECT ${selectCols}
  3724. FROM documents d
  3725. JOIN content ON content.hash = d.hash
  3726. WHERE 'qmd://' || d.collection || '/' || d.path IN (${placeholders}) AND d.active = 1
  3727. `).all(...virtualPaths) as DbDocRow[];
  3728. }
  3729. const results: MultiGetResult[] = [];
  3730. for (const row of fileRows) {
  3731. // Get context using virtual path
  3732. const virtualPath = row.virtual_path || `qmd://${row.collection}/${row.display_path}`;
  3733. const context = getContextForFile(db, virtualPath);
  3734. if (row.body_length > maxBytes) {
  3735. results.push({
  3736. doc: { filepath: virtualPath, displayPath: row.display_path },
  3737. skipped: true,
  3738. skipReason: `File too large (${Math.round(row.body_length / 1024)}KB > ${Math.round(maxBytes / 1024)}KB)`,
  3739. });
  3740. continue;
  3741. }
  3742. results.push({
  3743. doc: {
  3744. filepath: virtualPath,
  3745. displayPath: row.display_path,
  3746. title: row.title || row.display_path.split('/').pop() || row.display_path,
  3747. context,
  3748. hash: row.hash,
  3749. docid: getDocid(row.hash),
  3750. collectionName: row.collection,
  3751. modifiedAt: row.modified_at,
  3752. bodyLength: row.body_length,
  3753. ...(options.includeBody && row.body !== undefined && { body: row.body }),
  3754. },
  3755. skipped: false,
  3756. });
  3757. }
  3758. return { docs: results, errors };
  3759. }
  3760. // =============================================================================
  3761. // Status
  3762. // =============================================================================
  3763. export function getStatus(db: Database): IndexStatus {
  3764. // DB is source of truth for collections — config provides supplementary metadata
  3765. const dbCollections = db.prepare(`
  3766. SELECT
  3767. collection as name,
  3768. COUNT(*) as active_count,
  3769. MAX(modified_at) as last_doc_update
  3770. FROM documents
  3771. WHERE active = 1
  3772. GROUP BY collection
  3773. `).all() as { name: string; active_count: number; last_doc_update: string | null }[];
  3774. // Build a lookup from store_collections for path/pattern metadata
  3775. const storeCollections = getStoreCollections(db);
  3776. const configLookup = new Map(storeCollections.map(c => [c.name, { path: c.path, pattern: c.pattern }]));
  3777. const collections: CollectionInfo[] = dbCollections.map(row => {
  3778. const config = configLookup.get(row.name);
  3779. return {
  3780. name: row.name,
  3781. path: config?.path ?? null,
  3782. pattern: config?.pattern ?? null,
  3783. documents: row.active_count,
  3784. lastUpdated: row.last_doc_update || new Date().toISOString(),
  3785. };
  3786. });
  3787. // Sort by last update time (most recent first)
  3788. collections.sort((a, b) => {
  3789. if (!a.lastUpdated) return 1;
  3790. if (!b.lastUpdated) return -1;
  3791. return new Date(b.lastUpdated).getTime() - new Date(a.lastUpdated).getTime();
  3792. });
  3793. const totalDocs = (db.prepare(`SELECT COUNT(*) as c FROM documents WHERE active = 1`).get() as { c: number }).c;
  3794. const needsEmbedding = getHashesNeedingEmbedding(db);
  3795. const hasVectors = !!db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
  3796. return {
  3797. totalDocuments: totalDocs,
  3798. needsEmbedding,
  3799. hasVectorIndex: hasVectors,
  3800. collections,
  3801. };
  3802. }
  3803. // =============================================================================
  3804. // Snippet extraction
  3805. // =============================================================================
  3806. export type SnippetResult = {
  3807. line: number; // 1-indexed line number of best match
  3808. snippet: string; // The snippet text with diff-style header
  3809. linesBefore: number; // Lines in document before snippet
  3810. linesAfter: number; // Lines in document after snippet
  3811. snippetLines: number; // Number of lines in snippet
  3812. };
  3813. /** Weight for intent terms relative to query terms (1.0) in snippet scoring */
  3814. export const INTENT_WEIGHT_SNIPPET = 0.3;
  3815. /** Weight for intent terms relative to query terms (1.0) in chunk selection */
  3816. export const INTENT_WEIGHT_CHUNK = 0.5;
  3817. // Common stop words filtered from intent strings before tokenization.
  3818. // Seeded from finetune/reward.py KEY_TERM_STOPWORDS, extended with common
  3819. // 2-3 char function words so the length threshold can drop to >1 and let
  3820. // short domain terms (API, SQL, LLM, CPU, CDN, …) survive.
  3821. const INTENT_STOP_WORDS = new Set([
  3822. // 2-char function words
  3823. "am", "an", "as", "at", "be", "by", "do", "he", "if",
  3824. "in", "is", "it", "me", "my", "no", "of", "on", "or", "so",
  3825. "to", "up", "us", "we",
  3826. // 3-char function words
  3827. "all", "and", "any", "are", "but", "can", "did", "for", "get",
  3828. "has", "her", "him", "his", "how", "its", "let", "may", "not",
  3829. "our", "out", "the", "too", "was", "who", "why", "you",
  3830. // 4+ char common words
  3831. "also", "does", "find", "from", "have", "into", "more", "need",
  3832. "show", "some", "tell", "that", "them", "this", "want", "what",
  3833. "when", "will", "with", "your",
  3834. // Search-context noise
  3835. "about", "looking", "notes", "search", "where", "which",
  3836. ]);
  3837. /**
  3838. * Extract meaningful terms from an intent string, filtering stop words and punctuation.
  3839. * Uses Unicode-aware punctuation stripping so domain terms like "API" survive.
  3840. * Returns lowercase terms suitable for text matching.
  3841. */
  3842. export function extractIntentTerms(intent: string): string[] {
  3843. return intent.toLowerCase().split(/\s+/)
  3844. .map(t => t.replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, ""))
  3845. .filter(t => t.length > 1 && !INTENT_STOP_WORDS.has(t));
  3846. }
  3847. export function extractSnippet(body: string, query: string, maxLen = 500, chunkPos?: number, chunkLen?: number, intent?: string): SnippetResult {
  3848. const totalLines = body.split('\n').length;
  3849. let searchBody = body;
  3850. let lineOffset = 0;
  3851. if (chunkPos && chunkPos > 0) {
  3852. // Search within the chunk region, with some padding for context
  3853. // Use provided chunkLen or fall back to max chunk size (covers variable-length chunks)
  3854. const searchLen = chunkLen || CHUNK_SIZE_CHARS;
  3855. const contextStart = Math.max(0, chunkPos - 100);
  3856. const contextEnd = Math.min(body.length, chunkPos + searchLen + 100);
  3857. searchBody = body.slice(contextStart, contextEnd);
  3858. if (contextStart > 0) {
  3859. lineOffset = body.slice(0, contextStart).split('\n').length - 1;
  3860. }
  3861. }
  3862. const lines = searchBody.split('\n');
  3863. const queryTerms = query.toLowerCase().split(/\s+/).filter(t => t.length > 0);
  3864. const intentTerms = intent ? extractIntentTerms(intent) : [];
  3865. let bestLine = 0, bestScore = -1;
  3866. for (let i = 0; i < lines.length; i++) {
  3867. const lineLower = (lines[i] ?? "").toLowerCase();
  3868. let score = 0;
  3869. for (const term of queryTerms) {
  3870. if (lineLower.includes(term)) score += 1.0;
  3871. }
  3872. for (const term of intentTerms) {
  3873. if (lineLower.includes(term)) score += INTENT_WEIGHT_SNIPPET;
  3874. }
  3875. if (score > bestScore) {
  3876. bestScore = score;
  3877. bestLine = i;
  3878. }
  3879. }
  3880. const start = Math.max(0, bestLine - 1);
  3881. const end = Math.min(lines.length, bestLine + 3);
  3882. const snippetLines = lines.slice(start, end);
  3883. let snippetText = snippetLines.join('\n');
  3884. // If we focused on a chunk window and it produced an empty/whitespace-only snippet,
  3885. // fall back to a full-document snippet so we always show something useful.
  3886. if (chunkPos && chunkPos > 0 && snippetText.trim().length === 0) {
  3887. return extractSnippet(body, query, maxLen, undefined, undefined, intent);
  3888. }
  3889. if (snippetText.length > maxLen) snippetText = snippetText.substring(0, maxLen - 3) + "...";
  3890. const absoluteStart = lineOffset + start + 1; // 1-indexed
  3891. const snippetLineCount = snippetLines.length;
  3892. const linesBefore = absoluteStart - 1;
  3893. const linesAfter = totalLines - (absoluteStart + snippetLineCount - 1);
  3894. // Format with diff-style header: @@ -start,count @@ (linesBefore before, linesAfter after)
  3895. const header = `@@ -${absoluteStart},${snippetLineCount} @@ (${linesBefore} before, ${linesAfter} after)`;
  3896. const snippet = `${header}\n${snippetText}`;
  3897. return {
  3898. line: lineOffset + bestLine + 1,
  3899. snippet,
  3900. linesBefore,
  3901. linesAfter,
  3902. snippetLines: snippetLineCount,
  3903. };
  3904. }
  3905. // =============================================================================
  3906. // Shared helpers (used by both CLI and MCP)
  3907. // =============================================================================
  3908. /**
  3909. * Add line numbers to text content.
  3910. * Each line becomes: "{lineNum}: {content}"
  3911. */
  3912. export function addLineNumbers(text: string, startLine: number = 1): string {
  3913. const lines = text.split('\n');
  3914. return lines.map((line, i) => `${startLine + i}: ${line}`).join('\n');
  3915. }
  3916. // =============================================================================
  3917. // Shared search orchestration
  3918. //
  3919. // hybridQuery() and vectorSearchQuery() are standalone functions (not Store
  3920. // methods) because they are orchestration over primitives — same rationale as
  3921. // reciprocalRankFusion(). They take a Store as first argument so both CLI
  3922. // and MCP can share the identical pipeline.
  3923. // =============================================================================
  3924. /**
  3925. * Optional progress hooks for search orchestration.
  3926. * CLI wires these to stderr for user feedback; MCP leaves them unset.
  3927. */
  3928. export interface SearchHooks {
  3929. /** BM25 probe found strong signal — expansion will be skipped */
  3930. onStrongSignal?: (topScore: number) => void;
  3931. /** Query expansion starting */
  3932. onExpandStart?: () => void;
  3933. /** Query expansion complete. Empty array = strong signal skip. elapsedMs = time taken. */
  3934. onExpand?: (original: string, expanded: ExpandedQuery[], elapsedMs: number) => void;
  3935. /** Embedding starting (vec/hyde queries) */
  3936. onEmbedStart?: (count: number) => void;
  3937. /** Embedding complete */
  3938. onEmbedDone?: (elapsedMs: number) => void;
  3939. /** Reranking is about to start */
  3940. onRerankStart?: (chunkCount: number) => void;
  3941. /** Reranking finished */
  3942. onRerankDone?: (elapsedMs: number) => void;
  3943. }
  3944. export interface HybridQueryOptions {
  3945. collection?: string;
  3946. limit?: number; // default 10
  3947. minScore?: number; // default 0
  3948. candidateLimit?: number; // default RERANK_CANDIDATE_LIMIT
  3949. explain?: boolean; // include backend/RRF/rerank score traces
  3950. intent?: string; // domain intent hint for disambiguation
  3951. skipRerank?: boolean; // skip LLM reranking, use only RRF scores
  3952. chunkStrategy?: ChunkStrategy;
  3953. hooks?: SearchHooks;
  3954. /**
  3955. * Optional embedding provider for query-side encoding (i-loazq6ze).
  3956. * When supplied, the original-query vector AND any vec/hyde expansion
  3957. * variants are encoded through this commercial provider. Without one,
  3958. * learned work returns typed HOLD.
  3959. */
  3960. embedProvider?: EmbeddingProvider;
  3961. }
  3962. export interface HybridQueryResult {
  3963. file: string; // internal filepath (qmd://collection/path)
  3964. displayPath: string;
  3965. title: string;
  3966. body: string; // full document body (for snippet extraction)
  3967. bestChunk: string; // best chunk text
  3968. bestChunkPos: number; // char offset of best chunk in body
  3969. score: number; // blended score (full precision)
  3970. context: string | null; // user-set context
  3971. docid: string; // content hash prefix (6 chars)
  3972. explain?: HybridQueryExplain;
  3973. }
  3974. export type RankedListMeta = {
  3975. source: "fts" | "vec";
  3976. queryType: "original" | "lex" | "vec" | "hyde";
  3977. query: string;
  3978. };
  3979. /**
  3980. * Hybrid search: BM25 + vector + query expansion + RRF + chunked reranking.
  3981. *
  3982. * Pipeline:
  3983. * 1. BM25 probe → skip expansion if strong signal
  3984. * 2. expandQuery() → typed query variants (lex/vec/hyde)
  3985. * 3. Type-routed search: original→vector, lex→FTS, vec/hyde→vector
  3986. * 4. RRF fusion → slice to candidateLimit
  3987. * 5. chunkDocument() + keyword-best-chunk selection
  3988. * 6. rerank on chunks (NOT full bodies — O(tokens) trap)
  3989. * 7. Position-aware score blending (RRF rank × reranker score)
  3990. * 8. Dedup by file, filter by minScore, slice to limit
  3991. */
  3992. export async function hybridQuery(
  3993. store: Store,
  3994. query: string,
  3995. options?: HybridQueryOptions
  3996. ): Promise<HybridQueryResult[]> {
  3997. const limit = options?.limit ?? 10;
  3998. const minScore = options?.minScore ?? 0;
  3999. const candidateLimit = options?.candidateLimit ?? RERANK_CANDIDATE_LIMIT;
  4000. const collection = options?.collection;
  4001. const explain = options?.explain ?? false;
  4002. const intent = options?.intent;
  4003. const skipRerank = options?.skipRerank ?? false;
  4004. const hooks = options?.hooks;
  4005. const embedProvider = options?.embedProvider;
  4006. const rankedLists: RankedResult[][] = [];
  4007. const rankedListMeta: RankedListMeta[] = [];
  4008. const docidMap = new Map<string, string>(); // filepath -> docid
  4009. const hasVectors = !!store.db.prepare(
  4010. `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`
  4011. ).get();
  4012. // Step 1: BM25 probe — strong signal skips expensive LLM expansion
  4013. // When intent is provided, disable strong-signal bypass — the obvious BM25
  4014. // match may not be what the caller wants (e.g. "performance" with intent
  4015. // "web page load times" should NOT shortcut to a sports-performance doc).
  4016. // Pass collection directly into FTS query (filter at SQL level, not post-hoc)
  4017. const initialFts = store.searchFTS(query, 20, collection);
  4018. const topScore = initialFts[0]?.score ?? 0;
  4019. const secondScore = initialFts[1]?.score ?? 0;
  4020. const hasStrongSignal = !intent && initialFts.length > 0
  4021. && topScore >= STRONG_SIGNAL_MIN_SCORE
  4022. && (topScore - secondScore) >= STRONG_SIGNAL_MIN_GAP;
  4023. if (hasStrongSignal) hooks?.onStrongSignal?.(topScore);
  4024. // Step 2: Expand query (or skip if strong signal)
  4025. hooks?.onExpandStart?.();
  4026. const expandStart = Date.now();
  4027. const expanded = hasStrongSignal
  4028. ? []
  4029. : await store.expandQuery(query, undefined, intent);
  4030. hooks?.onExpand?.(query, expanded, Date.now() - expandStart);
  4031. // Seed with initial FTS results (avoid re-running original query FTS)
  4032. if (initialFts.length > 0) {
  4033. for (const r of initialFts) docidMap.set(r.filepath, r.docid);
  4034. rankedLists.push(initialFts.map(r => ({
  4035. file: r.filepath, displayPath: r.displayPath,
  4036. title: r.title, body: r.body || "", score: r.score,
  4037. })));
  4038. rankedListMeta.push({ source: "fts", queryType: "original", query });
  4039. }
  4040. // Step 3: Route searches by query type
  4041. //
  4042. // Strategy: run all FTS queries immediately (they're sync/instant), then
  4043. // batch-embed all vector queries in one embedBatch() call, then run
  4044. // sqlite-vec lookups with pre-computed embeddings.
  4045. // 3a: Run FTS for all lex expansions right away (no LLM needed)
  4046. for (const q of expanded) {
  4047. if (q.type === 'lex') {
  4048. const ftsResults = store.searchFTS(q.query, 20, collection);
  4049. if (ftsResults.length > 0) {
  4050. for (const r of ftsResults) docidMap.set(r.filepath, r.docid);
  4051. rankedLists.push(ftsResults.map(r => ({
  4052. file: r.filepath, displayPath: r.displayPath,
  4053. title: r.title, body: r.body || "", score: r.score,
  4054. })));
  4055. rankedListMeta.push({ source: "fts", queryType: "lex", query: q.query });
  4056. }
  4057. }
  4058. }
  4059. // 3b: Collect all texts that need vector search (original query + vec/hyde expansions)
  4060. if (hasVectors) {
  4061. const vecQueries: { text: string; queryType: "original" | "vec" | "hyde" }[] = [
  4062. { text: query, queryType: "original" },
  4063. ];
  4064. for (const q of expanded) {
  4065. if (q.type === 'vec' || q.type === 'hyde') {
  4066. vecQueries.push({ text: q.query, queryType: q.type });
  4067. }
  4068. }
  4069. // Batch embed all vector queries in a single call.
  4070. // Route query embeddings through the injected commercial provider. The
  4071. // compatibility path below is fail-closed and returns typed HOLD.
  4072. const embedModelName = embedProvider
  4073. ? embedProvider.getModelId()
  4074. : getLlm(store).embedModelName;
  4075. const textsToEmbed = vecQueries.map(q => formatQueryForEmbedding(q.text, embedModelName));
  4076. hooks?.onEmbedStart?.(textsToEmbed.length);
  4077. const embedStart = Date.now();
  4078. const embeddings = embedProvider
  4079. ? await embedProvider.embedBatch(textsToEmbed, { model: embedModelName })
  4080. : await getLlm(store).embedBatch(textsToEmbed);
  4081. hooks?.onEmbedDone?.(Date.now() - embedStart);
  4082. // Run sqlite-vec lookups with pre-computed embeddings
  4083. for (let i = 0; i < vecQueries.length; i++) {
  4084. const embedding = embeddings[i]?.embedding;
  4085. if (!embedding) continue;
  4086. const vecResults = await store.searchVec(
  4087. vecQueries[i]!.text, DEFAULT_EMBED_MODEL, 20, collection,
  4088. undefined, embedding
  4089. );
  4090. if (vecResults.length > 0) {
  4091. for (const r of vecResults) docidMap.set(r.filepath, r.docid);
  4092. rankedLists.push(vecResults.map(r => ({
  4093. file: r.filepath, displayPath: r.displayPath,
  4094. title: r.title, body: r.body || "", score: r.score,
  4095. })));
  4096. rankedListMeta.push({
  4097. source: "vec",
  4098. queryType: vecQueries[i]!.queryType,
  4099. query: vecQueries[i]!.text,
  4100. });
  4101. }
  4102. }
  4103. }
  4104. // Step 4: RRF fusion — first 2 lists (original FTS + first vec) get 2x weight
  4105. const weights = rankedLists.map((_, i) => i < 2 ? 2.0 : 1.0);
  4106. const fused = reciprocalRankFusion(rankedLists, weights);
  4107. const rrfTraceByFile = explain ? buildRrfTrace(rankedLists, weights, rankedListMeta) : null;
  4108. const candidates = fused.slice(0, candidateLimit);
  4109. if (candidates.length === 0) return [];
  4110. // Step 5: Chunk documents, pick best chunk per doc for reranking.
  4111. // Reranking full bodies is O(tokens) — the critical perf lesson that motivated this refactor.
  4112. const queryTerms = query.toLowerCase().split(/\s+/).filter(t => t.length > 2);
  4113. const intentTerms = intent ? extractIntentTerms(intent) : [];
  4114. const docChunkMap = new Map<string, { chunks: { text: string; pos: number }[]; bestIdx: number }>();
  4115. const chunkStrategy = options?.chunkStrategy;
  4116. for (const cand of candidates) {
  4117. const chunks = await chunkDocumentAsync(cand.body, undefined, undefined, undefined, cand.file, chunkStrategy);
  4118. if (chunks.length === 0) continue;
  4119. // Pick chunk with most keyword overlap (fallback: first chunk)
  4120. // Intent terms contribute at INTENT_WEIGHT_CHUNK (0.5) relative to query terms (1.0)
  4121. let bestIdx = 0;
  4122. let bestScore = -1;
  4123. for (let i = 0; i < chunks.length; i++) {
  4124. const chunkLower = chunks[i]!.text.toLowerCase();
  4125. let score = queryTerms.reduce((acc, term) => acc + (chunkLower.includes(term) ? 1 : 0), 0);
  4126. for (const term of intentTerms) {
  4127. if (chunkLower.includes(term)) score += INTENT_WEIGHT_CHUNK;
  4128. }
  4129. if (score > bestScore) { bestScore = score; bestIdx = i; }
  4130. }
  4131. docChunkMap.set(cand.file, { chunks, bestIdx });
  4132. }
  4133. if (skipRerank) {
  4134. // Skip LLM reranking — return candidates scored by RRF only
  4135. const seenFiles = new Set<string>();
  4136. return candidates
  4137. .map((cand, i) => {
  4138. const chunkInfo = docChunkMap.get(cand.file);
  4139. const bestIdx = chunkInfo?.bestIdx ?? 0;
  4140. const bestChunk = chunkInfo?.chunks[bestIdx]?.text || cand.body || "";
  4141. const bestChunkPos = chunkInfo?.chunks[bestIdx]?.pos || 0;
  4142. const rrfRank = i + 1;
  4143. const rrfScore = 1 / rrfRank;
  4144. const trace = rrfTraceByFile?.get(cand.file);
  4145. const explainData: HybridQueryExplain | undefined = explain ? {
  4146. ftsScores: trace?.contributions.filter(c => c.source === "fts").map(c => c.backendScore) ?? [],
  4147. vectorScores: trace?.contributions.filter(c => c.source === "vec").map(c => c.backendScore) ?? [],
  4148. rrf: {
  4149. rank: rrfRank,
  4150. positionScore: rrfScore,
  4151. weight: 1.0,
  4152. baseScore: trace?.baseScore ?? 0,
  4153. topRankBonus: trace?.topRankBonus ?? 0,
  4154. totalScore: trace?.totalScore ?? 0,
  4155. contributions: trace?.contributions ?? [],
  4156. },
  4157. rerankScore: 0,
  4158. blendedScore: rrfScore,
  4159. } : undefined;
  4160. return {
  4161. file: cand.file,
  4162. displayPath: cand.displayPath,
  4163. title: cand.title,
  4164. body: cand.body,
  4165. bestChunk,
  4166. bestChunkPos,
  4167. score: rrfScore,
  4168. context: store.getContextForFile(cand.file),
  4169. docid: docidMap.get(cand.file) || "",
  4170. ...(explainData ? { explain: explainData } : {}),
  4171. };
  4172. })
  4173. .filter(r => {
  4174. if (seenFiles.has(r.file)) return false;
  4175. seenFiles.add(r.file);
  4176. return true;
  4177. })
  4178. .filter(r => r.score >= minScore)
  4179. .slice(0, limit);
  4180. }
  4181. // Step 6: Rerank chunks (NOT full bodies)
  4182. const chunksToRerank: { file: string; text: string }[] = [];
  4183. for (const cand of candidates) {
  4184. const chunkInfo = docChunkMap.get(cand.file);
  4185. if (chunkInfo) {
  4186. chunksToRerank.push({ file: cand.file, text: chunkInfo.chunks[chunkInfo.bestIdx]!.text });
  4187. }
  4188. }
  4189. hooks?.onRerankStart?.(chunksToRerank.length);
  4190. const rerankStart = Date.now();
  4191. const reranked = await store.rerank(query, chunksToRerank, undefined, intent);
  4192. hooks?.onRerankDone?.(Date.now() - rerankStart);
  4193. // Step 7: Blend RRF position score with reranker score
  4194. // Position-aware weights: top retrieval results get more protection from reranker disagreement
  4195. const candidateMap = new Map(candidates.map(c => [c.file, {
  4196. displayPath: c.displayPath, title: c.title, body: c.body,
  4197. }]));
  4198. const rrfRankMap = new Map(candidates.map((c, i) => [c.file, i + 1]));
  4199. const blended = reranked.map(r => {
  4200. const rrfRank = rrfRankMap.get(r.file) || candidateLimit;
  4201. let rrfWeight: number;
  4202. if (rrfRank <= 3) rrfWeight = 0.75;
  4203. else if (rrfRank <= 10) rrfWeight = 0.60;
  4204. else rrfWeight = 0.40;
  4205. const rrfScore = 1 / rrfRank;
  4206. const blendedScore = rrfWeight * rrfScore + (1 - rrfWeight) * r.score;
  4207. const candidate = candidateMap.get(r.file);
  4208. const chunkInfo = docChunkMap.get(r.file);
  4209. const bestIdx = chunkInfo?.bestIdx ?? 0;
  4210. const bestChunk = chunkInfo?.chunks[bestIdx]?.text || candidate?.body || "";
  4211. const bestChunkPos = chunkInfo?.chunks[bestIdx]?.pos || 0;
  4212. const trace = rrfTraceByFile?.get(r.file);
  4213. const explainData: HybridQueryExplain | undefined = explain ? {
  4214. ftsScores: trace?.contributions.filter(c => c.source === "fts").map(c => c.backendScore) ?? [],
  4215. vectorScores: trace?.contributions.filter(c => c.source === "vec").map(c => c.backendScore) ?? [],
  4216. rrf: {
  4217. rank: rrfRank,
  4218. positionScore: rrfScore,
  4219. weight: rrfWeight,
  4220. baseScore: trace?.baseScore ?? 0,
  4221. topRankBonus: trace?.topRankBonus ?? 0,
  4222. totalScore: trace?.totalScore ?? 0,
  4223. contributions: trace?.contributions ?? [],
  4224. },
  4225. rerankScore: r.score,
  4226. blendedScore,
  4227. } : undefined;
  4228. return {
  4229. file: r.file,
  4230. displayPath: candidate?.displayPath || "",
  4231. title: candidate?.title || "",
  4232. body: candidate?.body || "",
  4233. bestChunk,
  4234. bestChunkPos,
  4235. score: blendedScore,
  4236. context: store.getContextForFile(r.file),
  4237. docid: docidMap.get(r.file) || "",
  4238. ...(explainData ? { explain: explainData } : {}),
  4239. };
  4240. }).sort((a, b) => b.score - a.score);
  4241. // Step 8: Dedup by file (safety net — prevents duplicate output)
  4242. const seenFiles = new Set<string>();
  4243. return blended
  4244. .filter(r => {
  4245. if (seenFiles.has(r.file)) return false;
  4246. seenFiles.add(r.file);
  4247. return true;
  4248. })
  4249. .filter(r => r.score >= minScore)
  4250. .slice(0, limit);
  4251. }
  4252. export interface VectorSearchOptions {
  4253. collection?: string;
  4254. limit?: number; // default 10
  4255. minScore?: number; // default 0.3
  4256. intent?: string; // domain intent hint for disambiguation
  4257. hooks?: Pick<SearchHooks, 'onExpand'>;
  4258. /**
  4259. * Optional embedding provider for query-side encoding (i-loazq6ze).
  4260. * When supplied, query vectors are encoded through the commercial API.
  4261. * Without one, learned work returns typed HOLD.
  4262. */
  4263. embedProvider?: EmbeddingProvider;
  4264. }
  4265. export interface VectorSearchResult {
  4266. file: string;
  4267. displayPath: string;
  4268. title: string;
  4269. body: string;
  4270. score: number;
  4271. context: string | null;
  4272. docid: string;
  4273. }
  4274. /**
  4275. * Vector-only semantic search with query expansion.
  4276. *
  4277. * Pipeline:
  4278. * 1. expandQuery() → typed variants, filter to vec/hyde only (lex irrelevant here)
  4279. * 2. searchVec() for original + vec/hyde variants through the commercial provider
  4280. * 3. Dedup by filepath (keep max score)
  4281. * 4. Sort by score descending, filter by minScore, slice to limit
  4282. */
  4283. export async function vectorSearchQuery(
  4284. store: Store,
  4285. query: string,
  4286. options?: VectorSearchOptions
  4287. ): Promise<VectorSearchResult[]> {
  4288. const limit = options?.limit ?? 10;
  4289. const minScore = options?.minScore ?? 0.3;
  4290. const collection = options?.collection;
  4291. const intent = options?.intent;
  4292. const embedProvider = options?.embedProvider;
  4293. const hasVectors = !!store.db.prepare(
  4294. `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`
  4295. ).get();
  4296. if (!hasVectors) return [];
  4297. // Expand query — filter to vec/hyde only (lex queries target FTS, not vector)
  4298. const expandStart = Date.now();
  4299. const allExpanded = await store.expandQuery(query, undefined, intent);
  4300. const vecExpanded = allExpanded.filter(q => q.type !== 'lex');
  4301. options?.hooks?.onExpand?.(query, vecExpanded, Date.now() - expandStart);
  4302. // Run original + vec/hyde expanded through vector, sequentially — concurrent embed() hangs.
  4303. // When `embedProvider` is supplied (i-loazq6ze), query encoding is routed
  4304. // through it; the per-call signature `searchVec(...)` accepts the provider
  4305. // as the trailing argument so existing tests / callers stay untouched.
  4306. const queryTexts = [query, ...vecExpanded.map(q => q.query)];
  4307. const allResults = new Map<string, VectorSearchResult>();
  4308. for (const q of queryTexts) {
  4309. const vecResults = await store.searchVec(
  4310. q, DEFAULT_EMBED_MODEL, limit, collection,
  4311. undefined, undefined, embedProvider,
  4312. );
  4313. for (const r of vecResults) {
  4314. const existing = allResults.get(r.filepath);
  4315. if (!existing || r.score > existing.score) {
  4316. allResults.set(r.filepath, {
  4317. file: r.filepath,
  4318. displayPath: r.displayPath,
  4319. title: r.title,
  4320. body: r.body || "",
  4321. score: r.score,
  4322. context: store.getContextForFile(r.filepath),
  4323. docid: r.docid,
  4324. });
  4325. }
  4326. }
  4327. }
  4328. return Array.from(allResults.values())
  4329. .sort((a, b) => b.score - a.score)
  4330. .filter(r => r.score >= minScore)
  4331. .slice(0, limit);
  4332. }
  4333. // =============================================================================
  4334. // Structured search — pre-expanded queries from LLM
  4335. // =============================================================================
  4336. /**
  4337. * A single sub-search in a structured search request.
  4338. * Matches the format used in QMD training data.
  4339. */
  4340. export interface StructuredSearchOptions {
  4341. collections?: string[]; // Filter to specific collections (OR match)
  4342. limit?: number; // default 10
  4343. minScore?: number; // default 0
  4344. candidateLimit?: number; // default RERANK_CANDIDATE_LIMIT
  4345. explain?: boolean; // include backend/RRF/rerank score traces
  4346. /** Domain intent hint for disambiguation — steers reranking and chunk selection */
  4347. intent?: string;
  4348. /** Skip LLM reranking, use only RRF scores */
  4349. skipRerank?: boolean;
  4350. chunkStrategy?: ChunkStrategy;
  4351. hooks?: SearchHooks;
  4352. /**
  4353. * Optional embedding provider for query-side encoding (i-loazq6ze).
  4354. * When supplied, vec/hyde sub-queries are batch-encoded via the provider
  4355. * (HTTP / GPU worker / fallback chain) instead of `getLlm(store).embedBatch`.
  4356. */
  4357. embedProvider?: EmbeddingProvider;
  4358. }
  4359. /**
  4360. * Structured search: execute pre-expanded queries without LLM query expansion.
  4361. *
  4362. * Designed for LLM callers (MCP/HTTP) that generate their own query expansions.
  4363. * Skips the internal expandQuery() step — goes directly to:
  4364. *
  4365. * Pipeline:
  4366. * 1. Route searches: lex→FTS, vec/hyde→vector (batch embed)
  4367. * 2. RRF fusion across all result lists
  4368. * 3. Chunk documents + keyword-best-chunk selection
  4369. * 4. Rerank on chunks
  4370. * 5. Position-aware score blending
  4371. * 6. Dedup, filter, slice
  4372. *
  4373. * This is the recommended endpoint when the caller supplies domain-specific
  4374. * query variants and a commercial provider contract is active.
  4375. */
  4376. export async function structuredSearch(
  4377. store: Store,
  4378. searches: ExpandedQuery[],
  4379. options?: StructuredSearchOptions
  4380. ): Promise<HybridQueryResult[]> {
  4381. const limit = options?.limit ?? 10;
  4382. const minScore = options?.minScore ?? 0;
  4383. const candidateLimit = options?.candidateLimit ?? RERANK_CANDIDATE_LIMIT;
  4384. const explain = options?.explain ?? false;
  4385. const intent = options?.intent;
  4386. const skipRerank = options?.skipRerank ?? false;
  4387. const hooks = options?.hooks;
  4388. const embedProvider = options?.embedProvider;
  4389. const collections = options?.collections;
  4390. if (searches.length === 0) return [];
  4391. // Validate queries before executing
  4392. for (const search of searches) {
  4393. const location = search.line ? `Line ${search.line}` : 'Structured search';
  4394. if (/[\r\n]/.test(search.query)) {
  4395. throw new Error(`${location} (${search.type}): queries must be single-line. Remove newline characters.`);
  4396. }
  4397. if (search.type === 'lex') {
  4398. const error = validateLexQuery(search.query);
  4399. if (error) {
  4400. throw new Error(`${location} (lex): ${error}`);
  4401. }
  4402. } else if (search.type === 'vec' || search.type === 'hyde') {
  4403. const error = validateSemanticQuery(search.query);
  4404. if (error) {
  4405. throw new Error(`${location} (${search.type}): ${error}`);
  4406. }
  4407. }
  4408. }
  4409. const rankedLists: RankedResult[][] = [];
  4410. const rankedListMeta: RankedListMeta[] = [];
  4411. const docidMap = new Map<string, string>(); // filepath -> docid
  4412. const hasVectors = !!store.db.prepare(
  4413. `SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`
  4414. ).get();
  4415. // Helper to run search across collections (or all if undefined)
  4416. const collectionList = collections ?? [undefined]; // undefined = all collections
  4417. // Step 1: Run FTS for all lex searches (sync, instant)
  4418. for (const search of searches) {
  4419. if (search.type === 'lex') {
  4420. for (const coll of collectionList) {
  4421. const ftsResults = store.searchFTS(search.query, 20, coll);
  4422. if (ftsResults.length > 0) {
  4423. for (const r of ftsResults) docidMap.set(r.filepath, r.docid);
  4424. rankedLists.push(ftsResults.map(r => ({
  4425. file: r.filepath, displayPath: r.displayPath,
  4426. title: r.title, body: r.body || "", score: r.score,
  4427. })));
  4428. rankedListMeta.push({
  4429. source: "fts",
  4430. queryType: "lex",
  4431. query: search.query,
  4432. });
  4433. }
  4434. }
  4435. }
  4436. }
  4437. // Step 2: Batch embed and run vector searches for vec/hyde
  4438. if (hasVectors) {
  4439. const vecSearches = searches.filter(
  4440. (s): s is ExpandedQuery & { type: 'vec' | 'hyde' } =>
  4441. s.type === 'vec' || s.type === 'hyde'
  4442. );
  4443. if (vecSearches.length > 0) {
  4444. // Route batch encoding through the supplied commercial provider. The
  4445. // compatibility branch returns typed HOLD when no provider is configured.
  4446. const embedModelName = embedProvider
  4447. ? embedProvider.getModelId()
  4448. : getLlm(store).embedModelName;
  4449. const textsToEmbed = vecSearches.map(s => formatQueryForEmbedding(s.query, embedModelName));
  4450. hooks?.onEmbedStart?.(textsToEmbed.length);
  4451. const embedStart = Date.now();
  4452. const embeddings = embedProvider
  4453. ? await embedProvider.embedBatch(textsToEmbed, { model: embedModelName })
  4454. : await getLlm(store).embedBatch(textsToEmbed);
  4455. hooks?.onEmbedDone?.(Date.now() - embedStart);
  4456. for (let i = 0; i < vecSearches.length; i++) {
  4457. const embedding = embeddings[i]?.embedding;
  4458. if (!embedding) continue;
  4459. for (const coll of collectionList) {
  4460. const vecResults = await store.searchVec(
  4461. vecSearches[i]!.query, DEFAULT_EMBED_MODEL, 20, coll,
  4462. undefined, embedding
  4463. );
  4464. if (vecResults.length > 0) {
  4465. for (const r of vecResults) docidMap.set(r.filepath, r.docid);
  4466. rankedLists.push(vecResults.map(r => ({
  4467. file: r.filepath, displayPath: r.displayPath,
  4468. title: r.title, body: r.body || "", score: r.score,
  4469. })));
  4470. rankedListMeta.push({
  4471. source: "vec",
  4472. queryType: vecSearches[i]!.type,
  4473. query: vecSearches[i]!.query,
  4474. });
  4475. }
  4476. }
  4477. }
  4478. }
  4479. }
  4480. if (rankedLists.length === 0) return [];
  4481. // Step 3: RRF fusion — first list gets 2x weight (assume caller ordered by importance)
  4482. const weights = rankedLists.map((_, i) => i === 0 ? 2.0 : 1.0);
  4483. const fused = reciprocalRankFusion(rankedLists, weights);
  4484. const rrfTraceByFile = explain ? buildRrfTrace(rankedLists, weights, rankedListMeta) : null;
  4485. const candidates = fused.slice(0, candidateLimit);
  4486. if (candidates.length === 0) return [];
  4487. hooks?.onExpand?.("", [], 0); // Signal no expansion (pre-expanded)
  4488. // Step 4: Chunk documents, pick best chunk per doc for reranking
  4489. // Use first lex query as the "query" for keyword matching, or first vec if no lex
  4490. const primaryQuery = searches.find(s => s.type === 'lex')?.query
  4491. || searches.find(s => s.type === 'vec')?.query
  4492. || searches[0]?.query || "";
  4493. const queryTerms = primaryQuery.toLowerCase().split(/\s+/).filter(t => t.length > 2);
  4494. const intentTerms = intent ? extractIntentTerms(intent) : [];
  4495. const docChunkMap = new Map<string, { chunks: { text: string; pos: number }[]; bestIdx: number }>();
  4496. const ssChunkStrategy = options?.chunkStrategy;
  4497. for (const cand of candidates) {
  4498. const chunks = await chunkDocumentAsync(cand.body, undefined, undefined, undefined, cand.file, ssChunkStrategy);
  4499. if (chunks.length === 0) continue;
  4500. // Pick chunk with most keyword overlap
  4501. // Intent terms contribute at INTENT_WEIGHT_CHUNK (0.5) relative to query terms (1.0)
  4502. let bestIdx = 0;
  4503. let bestScore = -1;
  4504. for (let i = 0; i < chunks.length; i++) {
  4505. const chunkLower = chunks[i]!.text.toLowerCase();
  4506. let score = queryTerms.reduce((acc, term) => acc + (chunkLower.includes(term) ? 1 : 0), 0);
  4507. for (const term of intentTerms) {
  4508. if (chunkLower.includes(term)) score += INTENT_WEIGHT_CHUNK;
  4509. }
  4510. if (score > bestScore) { bestScore = score; bestIdx = i; }
  4511. }
  4512. docChunkMap.set(cand.file, { chunks, bestIdx });
  4513. }
  4514. if (skipRerank) {
  4515. // Skip LLM reranking — return candidates scored by RRF only
  4516. const seenFiles = new Set<string>();
  4517. return candidates
  4518. .map((cand, i) => {
  4519. const chunkInfo = docChunkMap.get(cand.file);
  4520. const bestIdx = chunkInfo?.bestIdx ?? 0;
  4521. const bestChunk = chunkInfo?.chunks[bestIdx]?.text || cand.body || "";
  4522. const bestChunkPos = chunkInfo?.chunks[bestIdx]?.pos || 0;
  4523. const rrfRank = i + 1;
  4524. const rrfScore = 1 / rrfRank;
  4525. const trace = rrfTraceByFile?.get(cand.file);
  4526. const explainData: HybridQueryExplain | undefined = explain ? {
  4527. ftsScores: trace?.contributions.filter(c => c.source === "fts").map(c => c.backendScore) ?? [],
  4528. vectorScores: trace?.contributions.filter(c => c.source === "vec").map(c => c.backendScore) ?? [],
  4529. rrf: {
  4530. rank: rrfRank,
  4531. positionScore: rrfScore,
  4532. weight: 1.0,
  4533. baseScore: trace?.baseScore ?? 0,
  4534. topRankBonus: trace?.topRankBonus ?? 0,
  4535. totalScore: trace?.totalScore ?? 0,
  4536. contributions: trace?.contributions ?? [],
  4537. },
  4538. rerankScore: 0,
  4539. blendedScore: rrfScore,
  4540. } : undefined;
  4541. return {
  4542. file: cand.file,
  4543. displayPath: cand.displayPath,
  4544. title: cand.title,
  4545. body: cand.body,
  4546. bestChunk,
  4547. bestChunkPos,
  4548. score: rrfScore,
  4549. context: store.getContextForFile(cand.file),
  4550. docid: docidMap.get(cand.file) || "",
  4551. ...(explainData ? { explain: explainData } : {}),
  4552. };
  4553. })
  4554. .filter(r => {
  4555. if (seenFiles.has(r.file)) return false;
  4556. seenFiles.add(r.file);
  4557. return true;
  4558. })
  4559. .filter(r => r.score >= minScore)
  4560. .slice(0, limit);
  4561. }
  4562. // Step 5: Rerank chunks
  4563. const chunksToRerank: { file: string; text: string }[] = [];
  4564. for (const cand of candidates) {
  4565. const chunkInfo = docChunkMap.get(cand.file);
  4566. if (chunkInfo) {
  4567. chunksToRerank.push({ file: cand.file, text: chunkInfo.chunks[chunkInfo.bestIdx]!.text });
  4568. }
  4569. }
  4570. hooks?.onRerankStart?.(chunksToRerank.length);
  4571. const rerankStart2 = Date.now();
  4572. const reranked = await store.rerank(primaryQuery, chunksToRerank, undefined, intent);
  4573. hooks?.onRerankDone?.(Date.now() - rerankStart2);
  4574. // Step 6: Blend RRF position score with reranker score
  4575. const candidateMap = new Map(candidates.map(c => [c.file, {
  4576. displayPath: c.displayPath, title: c.title, body: c.body,
  4577. }]));
  4578. const rrfRankMap = new Map(candidates.map((c, i) => [c.file, i + 1]));
  4579. const blended = reranked.map(r => {
  4580. const rrfRank = rrfRankMap.get(r.file) || candidateLimit;
  4581. let rrfWeight: number;
  4582. if (rrfRank <= 3) rrfWeight = 0.75;
  4583. else if (rrfRank <= 10) rrfWeight = 0.60;
  4584. else rrfWeight = 0.40;
  4585. const rrfScore = 1 / rrfRank;
  4586. const blendedScore = rrfWeight * rrfScore + (1 - rrfWeight) * r.score;
  4587. const candidate = candidateMap.get(r.file);
  4588. const chunkInfo = docChunkMap.get(r.file);
  4589. const bestIdx = chunkInfo?.bestIdx ?? 0;
  4590. const bestChunk = chunkInfo?.chunks[bestIdx]?.text || candidate?.body || "";
  4591. const bestChunkPos = chunkInfo?.chunks[bestIdx]?.pos || 0;
  4592. const trace = rrfTraceByFile?.get(r.file);
  4593. const explainData: HybridQueryExplain | undefined = explain ? {
  4594. ftsScores: trace?.contributions.filter(c => c.source === "fts").map(c => c.backendScore) ?? [],
  4595. vectorScores: trace?.contributions.filter(c => c.source === "vec").map(c => c.backendScore) ?? [],
  4596. rrf: {
  4597. rank: rrfRank,
  4598. positionScore: rrfScore,
  4599. weight: rrfWeight,
  4600. baseScore: trace?.baseScore ?? 0,
  4601. topRankBonus: trace?.topRankBonus ?? 0,
  4602. totalScore: trace?.totalScore ?? 0,
  4603. contributions: trace?.contributions ?? [],
  4604. },
  4605. rerankScore: r.score,
  4606. blendedScore,
  4607. } : undefined;
  4608. return {
  4609. file: r.file,
  4610. displayPath: candidate?.displayPath || "",
  4611. title: candidate?.title || "",
  4612. body: candidate?.body || "",
  4613. bestChunk,
  4614. bestChunkPos,
  4615. score: blendedScore,
  4616. context: store.getContextForFile(r.file),
  4617. docid: docidMap.get(r.file) || "",
  4618. ...(explainData ? { explain: explainData } : {}),
  4619. };
  4620. }).sort((a, b) => b.score - a.score);
  4621. // Step 7: Dedup by file
  4622. const seenFiles = new Set<string>();
  4623. return blended
  4624. .filter(r => {
  4625. if (seenFiles.has(r.file)) return false;
  4626. seenFiles.add(r.file);
  4627. return true;
  4628. })
  4629. .filter(r => r.score >= minScore)
  4630. .slice(0, limit);
  4631. }