From ce6ad3d3bf0b1ee11416be92497aa6aa8d595974 Mon Sep 17 00:00:00 2001 From: drewstone Date: Sun, 20 Sep 2026 04:36:57 -0700 Subject: [PATCH 1/9] chore(audit): transport verified retrieval repair on isolated branch --- .audit/simplification.patch.gz | Bin 0 -> 7284 bytes .github/workflows/apply-simplification.yml | 44 +++++++++++++++++++++ 2 files changed, 44 insertions(+) create mode 100644 .audit/simplification.patch.gz create mode 100644 .github/workflows/apply-simplification.yml diff --git a/.audit/simplification.patch.gz b/.audit/simplification.patch.gz new file mode 100644 index 0000000000000000000000000000000000000000..549011ac21a4ecab2a50a63104f8100578ff64b5 GIT binary patch literal 7284 zcmV-)9E;;0iwFP!00002|Ft~ra@$6d|M?WNNh$^LVYxvJB$ zERYzAh)IA8fRbgcs_r4~3GYd+KjsqvDJSRiZbTB8>FMd|@9CK&oz28xaFtdf-Z}Z{ z*)KnyzIyXxv`oar``JmeM1By5$-(Sk_9TjuCwob1= zDYHtX3H(i~TT#u6e0?<+g`B3VBA>>KK@n%~#cW-sdG=J4@lp=bM6BW~S&CVaFGT`V zl?6h{i}WhZN->Qyk>z5M0|o_vS}o$K6wA1p&eQBlOl~V#Mu^Dsb-GAIvCanNG+)U? zOw%f^U?2h5Ckq$>KEzcKujWD~IHJa{noA)+q-BKwO)r>NzPOe}B!0`%f379WtAvNk zIL*X5n-WQqry^cX(yMj8E)B+Uf#6n)bczcGV@iN>3tg*qkpW(ej)o6t~u7qfIx$$~&$zR&WT1y1x`DdS=~ zzZ5rVHP6?TAlwO3RfTk=2cqnPm`ci^C(@ICDcyv06`nY@)0Rct3bAZ^c|LR=x7Z!-?hl5CgV8>)r8{3hD=(5uGhj)hOmjpFX87Xet5Y#qXf-c5E7XR(Z;I5~_DaaqbGtTC+qKY`KHSqiJp z%UuHNiFk!_!+bN%6JR^mIB$uC)jGwASOb!{sHj4W!a*jvoS+&o2t6-SIa9bc1%jA1 zA+efP;yNwU$wE48qXeeYU&?|6hBXz;q!J~`LM$fuktUHipUbqcQhuo^{xT^q72>e8 z1-ydkzb`v|JqN95x}`%CB&@iZ%M6tVHJhy$3xgvF3{WdAt@;T&eP{~LW+0%TKu}8a z8T3Tm!FZF>lq?OP&_E{vW0&jVIz=@UEA656U8fI1lx8tBW6f9LkQL2xRmkfUbzV8o zi;9qq#F?HtkH%AmrInzrI(-mInXZ7SD{U|k4c3|Qg7rA{QED^B zgk+S|0rIJgr}KtlO;Q-E0utcz`FbHvE<^tbk^T}z9C(L>;1YDYBF_w!AebIda6CR3 zBgAR~jJKWu+f?fc#-(v$W6shH*c~8&C}O8bJabHhRWAZK>ncaCB7PvVfxyW^4YZB~ zIwvSdPdj~Mnc{T4R+IJLJF_ksEYGk;ll%PhmPB14#~Vk#tM2T#7@di#T1D zg5V+9nK<~e(+82LjNPL}#?_Rt=3txvvqg!_Ozf(4ODfS5SplOdy(1UEW71gMNSUWI zWT-Wgk8INAWmWp7VHHo`BUAiQqGj}cD%JmFHrqc)k|=sS+Iut|9{8%?>%gksd&KVB zKDuxFgb-h3z?)wji7&pdfD8*c$iOD&#rqxTmsJD93SWes0kj5{gN7VCu)T%n z{l;@}0pU49ht-)DDRB+HBTp<8dUa9$*RZC{G2k2&IY>Wq>0koL0~rklpifJ1=oH|2 zzLdzv2z6dntMX`P2k1CoPa-hqJKHDoCCd9!Dp24#=4A@JcMHJo{6#5?>5l0(Aij-& zGbS5eXdAtIFxlS)KYQ;m*_%y={z7f&%1gH4B`w}C96g5R8-}|>T0rrQ_!Zdh2)*Db zdM&6Kk?&-ATUK&OKNSfgY`)V$BBNfuL)p(Wl(rL~JxSINDJbVEBo7W9y4}5fekAVM z1@E@=2>6Avn1S!z1(8(Xktxuk7flbKJY&_G5a&GPCQzdXd=6@(6n3K(ywvDNJ6cf= z3L}kOLlc_!op{5B2SnJwVjlzW2DKn8G`mhc@9*0@$S{j*Due-vDhOqJ9rZAjzs_P zQMmsY{a*!Q24<9lmZlpSAlU>(jW-~0H{!G?@*=o&bQfW~j6XoY3{lu}U6QX51G*WA z(-m;#HAPqwV%bn!fyw&nPB*{5?BM|SKnc*nJQwzbp5GG5j%m=>@#^pF2Dv=;ZP3GB zXv+fb3Nnih-b}lUMA)7OF)C#hP`hYFJj>s?3Ba>>a5M@Sv~f*<1W$H(9Vfmu({7yM4#-3w6k z7$28M!fScoYlT9t%qB4O(!$}k2GBiy_E1CZ>Q28*Vt7wflGdq0Y`&c zfvE770-GHU!fk*bDVRS?xk4Wk+{XoKbyns;E?6B1kiky@i>lP4^(l>K1TWAHG<1Yh zwvm`kg^!gYG56+B>UgVgd8Eb;M>So0teX(OI|8qtrZ6u z;_5ZxgA8}1LU|W0{0IE|%n5w3DaNkIn6b$QA-YeUFp~oqH>~jh)}kTgZjOjdr~^-R z#tQ-i9KwX705-mk(?tw4jVxH?WCv}d{WKc@B_B$R;M{s`Ahag}Fg7_cfsqa~V#I8x z1SGVmcw=w`>Mq#)1pVJ3`hy`1R{ght>@DtovG(w#7b}zrxFDJ z91Qp=H&%$a@U+3CZq6|J>%HbQ^~A^viMnH#)Qr^-_<{u95ti*%ibN^2hDXm#S)BeR zd(;06)~5e?Yz@tq#ebewP2p!HVAA*4U6OYT4`HN`YaDcCF}Wo4}3} zEo62D<_~8*5?_C9VQd#EA1AG2J=$5mLL%BJRkqC<8(1qHd|O0ho$Pt1>d6&m`qXC?}V4zDVh~x6fXmDh!?E62z); zTG-5zmf%8xd?@B}h#rayvv6Zygj}OAHzFXE328ACZIm=GZMyU5VRLHLn7a%h%Z~4h zBs^+~UM|MKfUIHMhC#EX`1+wRucv_-E??k6Ydhw=T}XJR_*V=&!t;Ru1N4Ac|DejG zKz?^GJcL00$>Z?JpNlY%Iq3_}>SeA?4wsN{$0h*vK9BTuuS`;V;4 zFOjA0c{v^(aAM_PFT{OKXCaXE;LQ-!Hje9Qe&DzpmCWTpTzqxslk|Qp_z|h9DysLxOX6<=pf$Thws}met!S9w4dJsd31Mo2)xLbXZv{d8R$tAJUTYnwaAXMWB2FV!ECYfF36NEi87AaCxf&P%xWm zn=t!fU>4S{!5i)iUIA}NDbXKFJ6OlwxHmu6GJX-Qh8`|qkjBByzR zTfMf$tN;pvp#zhz-8Sd|T#xO#m=sVAIoC~E8k~4D0Qrg+T}7qtxFzIrn!DTnN3L6b zEsG?bBCfQeG5!F$08yEUp>3K^XzML`hOPDf`R?$1I6R{N|7X`}3>;VDP~DS!jmcm6 zA;ukahXD7!6U9r$xyrXNC2}oYFQcS~ru!m~lM@zcmjN#BlnY`P-&&XM91yW{6Zhj* z9GJt8MNMqZQRD)H6615^p=_`#*5u1{G5;FdIja~_7Qgdm2(&;E)(yZvM+OK2lx~_1 zeu*$^NqVtc-*)TzUN)?9{vKUyMUgaHtzog&8TQL;#zKf7zKK&k)rj{#3vLgw4Om(v z(w*-k7`NtL$2%yqfda*2gGtvh63k7>M;A@EW4g*-ROP4Wmj)NwtyI7aewLGb@NR6_ zGE(Ti-?%a8wc0F6_)mpX7e+GO*Y_rn_NJ#=II}eRX47;ZreU~cI%W|D0%FKqFVy~62P?R=$;EL%~BhTJdaC&ip3jAZ^MxE(uMy#&ftHb zCgw6(!}JXNnzms`xKR-_Epu{VKRYPeHW>a1HLzqt7BD1N5RFv@ZZC}>(dM86HfY zMA80GK6$b~`UH=1t24=dpR zND1w)>VaZDgjBeT#9_x}6yVkTDUM*=M=|&-))M_MQcxQ8X|*rPg(hJ%1l(ns32i=_ zq2-GIU|yvi8i8dTT3*^xc@`msYvKAkf{rb`4C4BIX+8+ zRTMx3G!!&uKEV05wX_9kg#~v$V{k6dx_e3v})xrtC`MV{ZQuJ$1yPZK;4oR ztV;`2sD?(@xfw;8EqljS*?IMc6*}a7QcG)AlPLvp8(Nrdn*Blrn5#zHJ5fOf(x9fQ z@Hi_BKITq6W7226glJ=@wm0b|I*0w5$eDZI^IL2?@m|PksnVb-qOAi6HbFTvwf)*$ zm)k_b>yi+}b^k<&?)m{ET-3z!)R{tQLsSs`DMFPF$0Wfp>7)?e`*GZC$OWxY`E>1UuAUc@oFhiiJ#;a?F6 z@ohPsZ>+a+YzMOG+t>hqs*a;Aw7Ge|om02`)vcu<1zYFe$pbA=8l3NHn8<0pDm>lN zJ6DTl>fZKHIRDPePGFB;YF*Jg_Tq_+{j)cMJCtR-3fd+yfh_Pea068t$&O~_RRmmmW#vK z?0*oek_mMQ{2c_{1W{uuP&A2v5|7Cf#uj36+)tD$o23QF1A2$($Qdh#V=jo<+M$Kl zmn%CHjW3B{NA|W_MxCU{X{JgEYR`aXUW$K(tLi4!kb&8>(&%nG`;{uJ4`alKGyRn2I1XRY?Y*^bM#p_A^{lF#nm*|qgdVrkverfhQ^ zw};LRI8>U_C}o!VpCJFgvn5(#`%nY+t%Y{V*GI%ac-gqL!y3V7&{~46V^zJ>0f|xa z9@>NG`Zb_k{^>nZdLwe{#TBrcrv^%+bA>MRX+kULr8y&Rc1l98Iy*TBM{EhmWX)&G zWlf_v_e$v`_Yc;TZBaJV@v1uT|3f?L4ejj^e0KNlfr)cU{OrEQ=6S@zql77lC9(=GH@S8T`%wt)t*j=jcN*=JpC z&2%Sr9NlmkjXM%XCY>${2-hlAfmacO^2Gsl*?EN3G-VQ?@TW|CwTeS$+LkSn9DJ;1&bfK4}<=`_3qRI#&y2Md1HyuaYib?b_pgxRE8aL(*B z_MXz21ZdF4({?ym%6Q}nmd|uy*%PdLs^$!J2HJEMN-qy&vt)}2%d?w|1!>yQd1T8n znfO2KwymeHC2o_bQy$y79jW#T@Xf5suoZ8LHFqp6>YRc@$?idgmr3>#-B!!f0P;75 zt$3KtQ4uMK8) z3U;O*dA9zG55qoriFzx{;W?GLmaDusM$6+k4D9OC*>{9Q*XWLtsT8+g&QB_o1l6|- zkqy`TCE`bbD>Tnm5ev@b0V7A(Z<%9jEZQ4rgy%mXRKbHPw6*{ZGVbT9{6H!pbgA{< zKGa)*z3_L7_Lo{BsN;b3d##LK;MI!3$=8gUfyV`KnkBk@S1&zUy4N{hZlPJPHTj`( z>6@XNbqRP3RWmPPUfo>%G)`5+i6ArNUD_V!sFszvBWoL- zzhJNhNnBF_!_gnqJE=+==vySzMxkK01EC+E# z(ufw?hQLen%|3ni1aucrkB3IVCM;d)ChJ!qe%+I&^Cc1Jj(cG1SxjPd&$M*%AI-jbaJKiiI&0JFcIqNi^ z26lGEXo&Vp_EhXeAyZBa-97HiN~a4M;39XPLQ9IEbruEphjqK1ap{>pEcO2W=v{(s zI+u&?Xq!P&KAFz*YPdQO?BGJj=GEZhTxWoUvqK@DoHE`!;8N=g#WSSj=s*!TeO@)d2D*I5~%F>*^re*<(%5!SHMx6B9aMdB~UFBwr$eks}`i%O|vvJgH z6(GI2pDBb-rRxg5RDQbTg zKs?#P?rh>_77$a3%i2kvu4~>ry73y!+?F0 z-)l>xtfWk`j_$9G*wd~4y;;Fm+=bh*jPrwZ0d{3Wmy5Sx@;q8YMc7VBYt?5E5Fx0W zexRkzZ}6d@a4T}E#b>posZ-lqkfzg4kD?^zbg zwjqeipBNp>R0G?~Ao1GXJBJrmDz>rJ$tki*1UIiV%d!SxU7%`Yu(rwnW>Poqw_nAw ztfb|+d=7=3>JkHy?$}x~5RW1;*5}6Qn?71Y++0c>X=Sa~fa#qRipv||UN8Iem3jQN zovPyc*>4sTRZcojY(-e@Z1{tG(Bf=jM2S3Yvqc!>j?*X$cg};q46&tOnH3!# Date: Sun, 20 Sep 2026 11:37:15 +0000 Subject: [PATCH 2/9] fix(knowledge): preserve retrieval identity across origins and expose historical search --- .audit/simplification.patch.gz | Bin 7284 -> 0 bytes .github/workflows/apply-simplification.yml | 44 ------ CHANGELOG.md | 8 + docs/run-scoped-citations.md | 19 +++ package.json | 2 +- src/knowledge-brief.ts | 96 ++++++++---- src/knowledge-tools.test.ts | 63 ++++++++ src/knowledge-tools.ts | 13 +- src/search-origins.test.ts | 165 +++++++++++++++++++++ src/search.ts | 51 +++++-- 10 files changed, 369 insertions(+), 92 deletions(-) delete mode 100644 .audit/simplification.patch.gz delete mode 100644 .github/workflows/apply-simplification.yml create mode 100644 src/search-origins.test.ts diff --git a/.audit/simplification.patch.gz b/.audit/simplification.patch.gz deleted file mode 100644 index 549011ac21a4ecab2a50a63104f8100578ff64b5..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 7284 zcmV-)9E;;0iwFP!00002|Ft~ra@$6d|M?WNNh$^LVYxvJB$ zERYzAh)IA8fRbgcs_r4~3GYd+KjsqvDJSRiZbTB8>FMd|@9CK&oz28xaFtdf-Z}Z{ z*)KnyzIyXxv`oar``JmeM1By5$-(Sk_9TjuCwob1= zDYHtX3H(i~TT#u6e0?<+g`B3VBA>>KK@n%~#cW-sdG=J4@lp=bM6BW~S&CVaFGT`V zl?6h{i}WhZN->Qyk>z5M0|o_vS}o$K6wA1p&eQBlOl~V#Mu^Dsb-GAIvCanNG+)U? zOw%f^U?2h5Ckq$>KEzcKujWD~IHJa{noA)+q-BKwO)r>NzPOe}B!0`%f379WtAvNk zIL*X5n-WQqry^cX(yMj8E)B+Uf#6n)bczcGV@iN>3tg*qkpW(ej)o6t~u7qfIx$$~&$zR&WT1y1x`DdS=~ zzZ5rVHP6?TAlwO3RfTk=2cqnPm`ci^C(@ICDcyv06`nY@)0Rct3bAZ^c|LR=x7Z!-?hl5CgV8>)r8{3hD=(5uGhj)hOmjpFX87Xet5Y#qXf-c5E7XR(Z;I5~_DaaqbGtTC+qKY`KHSqiJp z%UuHNiFk!_!+bN%6JR^mIB$uC)jGwASOb!{sHj4W!a*jvoS+&o2t6-SIa9bc1%jA1 zA+efP;yNwU$wE48qXeeYU&?|6hBXz;q!J~`LM$fuktUHipUbqcQhuo^{xT^q72>e8 z1-ydkzb`v|JqN95x}`%CB&@iZ%M6tVHJhy$3xgvF3{WdAt@;T&eP{~LW+0%TKu}8a z8T3Tm!FZF>lq?OP&_E{vW0&jVIz=@UEA656U8fI1lx8tBW6f9LkQL2xRmkfUbzV8o zi;9qq#F?HtkH%AmrInzrI(-mInXZ7SD{U|k4c3|Qg7rA{QED^B zgk+S|0rIJgr}KtlO;Q-E0utcz`FbHvE<^tbk^T}z9C(L>;1YDYBF_w!AebIda6CR3 zBgAR~jJKWu+f?fc#-(v$W6shH*c~8&C}O8bJabHhRWAZK>ncaCB7PvVfxyW^4YZB~ zIwvSdPdj~Mnc{T4R+IJLJF_ksEYGk;ll%PhmPB14#~Vk#tM2T#7@di#T1D zg5V+9nK<~e(+82LjNPL}#?_Rt=3txvvqg!_Ozf(4ODfS5SplOdy(1UEW71gMNSUWI zWT-Wgk8INAWmWp7VHHo`BUAiQqGj}cD%JmFHrqc)k|=sS+Iut|9{8%?>%gksd&KVB zKDuxFgb-h3z?)wji7&pdfD8*c$iOD&#rqxTmsJD93SWes0kj5{gN7VCu)T%n z{l;@}0pU49ht-)DDRB+HBTp<8dUa9$*RZC{G2k2&IY>Wq>0koL0~rklpifJ1=oH|2 zzLdzv2z6dntMX`P2k1CoPa-hqJKHDoCCd9!Dp24#=4A@JcMHJo{6#5?>5l0(Aij-& zGbS5eXdAtIFxlS)KYQ;m*_%y={z7f&%1gH4B`w}C96g5R8-}|>T0rrQ_!Zdh2)*Db zdM&6Kk?&-ATUK&OKNSfgY`)V$BBNfuL)p(Wl(rL~JxSINDJbVEBo7W9y4}5fekAVM z1@E@=2>6Avn1S!z1(8(Xktxuk7flbKJY&_G5a&GPCQzdXd=6@(6n3K(ywvDNJ6cf= z3L}kOLlc_!op{5B2SnJwVjlzW2DKn8G`mhc@9*0@$S{j*Due-vDhOqJ9rZAjzs_P zQMmsY{a*!Q24<9lmZlpSAlU>(jW-~0H{!G?@*=o&bQfW~j6XoY3{lu}U6QX51G*WA z(-m;#HAPqwV%bn!fyw&nPB*{5?BM|SKnc*nJQwzbp5GG5j%m=>@#^pF2Dv=;ZP3GB zXv+fb3Nnih-b}lUMA)7OF)C#hP`hYFJj>s?3Ba>>a5M@Sv~f*<1W$H(9Vfmu({7yM4#-3w6k z7$28M!fScoYlT9t%qB4O(!$}k2GBiy_E1CZ>Q28*Vt7wflGdq0Y`&c zfvE770-GHU!fk*bDVRS?xk4Wk+{XoKbyns;E?6B1kiky@i>lP4^(l>K1TWAHG<1Yh zwvm`kg^!gYG56+B>UgVgd8Eb;M>So0teX(OI|8qtrZ6u z;_5ZxgA8}1LU|W0{0IE|%n5w3DaNkIn6b$QA-YeUFp~oqH>~jh)}kTgZjOjdr~^-R z#tQ-i9KwX705-mk(?tw4jVxH?WCv}d{WKc@B_B$R;M{s`Ahag}Fg7_cfsqa~V#I8x z1SGVmcw=w`>Mq#)1pVJ3`hy`1R{ght>@DtovG(w#7b}zrxFDJ z91Qp=H&%$a@U+3CZq6|J>%HbQ^~A^viMnH#)Qr^-_<{u95ti*%ibN^2hDXm#S)BeR zd(;06)~5e?Yz@tq#ebewP2p!HVAA*4U6OYT4`HN`YaDcCF}Wo4}3} zEo62D<_~8*5?_C9VQd#EA1AG2J=$5mLL%BJRkqC<8(1qHd|O0ho$Pt1>d6&m`qXC?}V4zDVh~x6fXmDh!?E62z); zTG-5zmf%8xd?@B}h#rayvv6Zygj}OAHzFXE328ACZIm=GZMyU5VRLHLn7a%h%Z~4h zBs^+~UM|MKfUIHMhC#EX`1+wRucv_-E??k6Ydhw=T}XJR_*V=&!t;Ru1N4Ac|DejG zKz?^GJcL00$>Z?JpNlY%Iq3_}>SeA?4wsN{$0h*vK9BTuuS`;V;4 zFOjA0c{v^(aAM_PFT{OKXCaXE;LQ-!Hje9Qe&DzpmCWTpTzqxslk|Qp_z|h9DysLxOX6<=pf$Thws}met!S9w4dJsd31Mo2)xLbXZv{d8R$tAJUTYnwaAXMWB2FV!ECYfF36NEi87AaCxf&P%xWm zn=t!fU>4S{!5i)iUIA}NDbXKFJ6OlwxHmu6GJX-Qh8`|qkjBByzR zTfMf$tN;pvp#zhz-8Sd|T#xO#m=sVAIoC~E8k~4D0Qrg+T}7qtxFzIrn!DTnN3L6b zEsG?bBCfQeG5!F$08yEUp>3K^XzML`hOPDf`R?$1I6R{N|7X`}3>;VDP~DS!jmcm6 zA;ukahXD7!6U9r$xyrXNC2}oYFQcS~ru!m~lM@zcmjN#BlnY`P-&&XM91yW{6Zhj* z9GJt8MNMqZQRD)H6615^p=_`#*5u1{G5;FdIja~_7Qgdm2(&;E)(yZvM+OK2lx~_1 zeu*$^NqVtc-*)TzUN)?9{vKUyMUgaHtzog&8TQL;#zKf7zKK&k)rj{#3vLgw4Om(v z(w*-k7`NtL$2%yqfda*2gGtvh63k7>M;A@EW4g*-ROP4Wmj)NwtyI7aewLGb@NR6_ zGE(Ti-?%a8wc0F6_)mpX7e+GO*Y_rn_NJ#=II}eRX47;ZreU~cI%W|D0%FKqFVy~62P?R=$;EL%~BhTJdaC&ip3jAZ^MxE(uMy#&ftHb zCgw6(!}JXNnzms`xKR-_Epu{VKRYPeHW>a1HLzqt7BD1N5RFv@ZZC}>(dM86HfY zMA80GK6$b~`UH=1t24=dpR zND1w)>VaZDgjBeT#9_x}6yVkTDUM*=M=|&-))M_MQcxQ8X|*rPg(hJ%1l(ns32i=_ zq2-GIU|yvi8i8dTT3*^xc@`msYvKAkf{rb`4C4BIX+8+ zRTMx3G!!&uKEV05wX_9kg#~v$V{k6dx_e3v})xrtC`MV{ZQuJ$1yPZK;4oR ztV;`2sD?(@xfw;8EqljS*?IMc6*}a7QcG)AlPLvp8(Nrdn*Blrn5#zHJ5fOf(x9fQ z@Hi_BKITq6W7226glJ=@wm0b|I*0w5$eDZI^IL2?@m|PksnVb-qOAi6HbFTvwf)*$ zm)k_b>yi+}b^k<&?)m{ET-3z!)R{tQLsSs`DMFPF$0Wfp>7)?e`*GZC$OWxY`E>1UuAUc@oFhiiJ#;a?F6 z@ohPsZ>+a+YzMOG+t>hqs*a;Aw7Ge|om02`)vcu<1zYFe$pbA=8l3NHn8<0pDm>lN zJ6DTl>fZKHIRDPePGFB;YF*Jg_Tq_+{j)cMJCtR-3fd+yfh_Pea068t$&O~_RRmmmW#vK z?0*oek_mMQ{2c_{1W{uuP&A2v5|7Cf#uj36+)tD$o23QF1A2$($Qdh#V=jo<+M$Kl zmn%CHjW3B{NA|W_MxCU{X{JgEYR`aXUW$K(tLi4!kb&8>(&%nG`;{uJ4`alKGyRn2I1XRY?Y*^bM#p_A^{lF#nm*|qgdVrkverfhQ^ zw};LRI8>U_C}o!VpCJFgvn5(#`%nY+t%Y{V*GI%ac-gqL!y3V7&{~46V^zJ>0f|xa z9@>NG`Zb_k{^>nZdLwe{#TBrcrv^%+bA>MRX+kULr8y&Rc1l98Iy*TBM{EhmWX)&G zWlf_v_e$v`_Yc;TZBaJV@v1uT|3f?L4ejj^e0KNlfr)cU{OrEQ=6S@zql77lC9(=GH@S8T`%wt)t*j=jcN*=JpC z&2%Sr9NlmkjXM%XCY>${2-hlAfmacO^2Gsl*?EN3G-VQ?@TW|CwTeS$+LkSn9DJ;1&bfK4}<=`_3qRI#&y2Md1HyuaYib?b_pgxRE8aL(*B z_MXz21ZdF4({?ym%6Q}nmd|uy*%PdLs^$!J2HJEMN-qy&vt)}2%d?w|1!>yQd1T8n znfO2KwymeHC2o_bQy$y79jW#T@Xf5suoZ8LHFqp6>YRc@$?idgmr3>#-B!!f0P;75 zt$3KtQ4uMK8) z3U;O*dA9zG55qoriFzx{;W?GLmaDusM$6+k4D9OC*>{9Q*XWLtsT8+g&QB_o1l6|- zkqy`TCE`bbD>Tnm5ev@b0V7A(Z<%9jEZQ4rgy%mXRKbHPw6*{ZGVbT9{6H!pbgA{< zKGa)*z3_L7_Lo{BsN;b3d##LK;MI!3$=8gUfyV`KnkBk@S1&zUy4N{hZlPJPHTj`( z>6@XNbqRP3RWmPPUfo>%G)`5+i6ArNUD_V!sFszvBWoL- zzhJNhNnBF_!_gnqJE=+==vySzMxkK01EC+E# z(ufw?hQLen%|3ni1aucrkB3IVCM;d)ChJ!qe%+I&^Cc1Jj(cG1SxjPd&$M*%AI-jbaJKiiI&0JFcIqNi^ z26lGEXo&Vp_EhXeAyZBa-97HiN~a4M;39XPLQ9IEbruEphjqK1ap{>pEcO2W=v{(s zI+u&?Xq!P&KAFz*YPdQO?BGJj=GEZhTxWoUvqK@DoHE`!;8N=g#WSSj=s*!TeO@)d2D*I5~%F>*^re*<(%5!SHMx6B9aMdB~UFBwr$eks}`i%O|vvJgH z6(GI2pDBb-rRxg5RDQbTg zKs?#P?rh>_77$a3%i2kvu4~>ry73y!+?F0 z-)l>xtfWk`j_$9G*wd~4y;;Fm+=bh*jPrwZ0d{3Wmy5Sx@;q8YMc7VBYt?5E5Fx0W zexRkzZ}6d@a4T}E#b>posZ-lqkfzg4kD?^zbg zwjqeipBNp>R0G?~Ao1GXJBJrmDz>rJ$tki*1UIiV%d!SxU7%`Yu(rwnW>Poqw_nAw ztfb|+d=7=3>JkHy?$}x~5RW1;*5}6Qn?71Y++0c>X=Sa~fa#qRipv||UN8Iem3jQN zovPyc*>4sTRZcojY(-e@Z1{tG(Bf=jM2S3Yvqc!>j?*X$cg};q46&tOnH3!#() - for (const entry of visiblePages) originByPage.set(entry.page, entry.origin) - - const ranked = searchKnowledgePages( - visiblePages.map((entry) => entry.page), - question, - { - limit, - excludeInvalidated, - ...(options.tags === undefined ? {} : { tags: options.tags }), - ...(options.kinds === undefined ? {} : { kinds: options.kinds }), - }, - ) + // Resolve citation addresses with the same owner used by read/write validation. + // Keep a small candidate list per id so resolving links does not rescan the corpus. + const byId = new Map() + for (const entry of visiblePages) { + const matches = byId.get(entry.page.id) ?? [] + matches.push(entry) + byId.set(entry.page.id, matches) + } + const qualified = (entry: OriginatedPage) => + formatKnowledgeCitationReference({ pageId: entry.page.id, origin: entry.origin }) + const originals = new Map() + const pages = visiblePages.map((entry) => { + const page: KnowledgePage = { + ...entry.page, + id: qualified(entry), + outLinks: entry.page.outLinks.flatMap((link: string) => { + // A malformed link is not a ranking edge; retain the page unchanged so + // citation audit can still diagnose it without making retrieval unavailable. + try { + const reference = parseKnowledgeCitationReference(link) + const resolution = resolveKnowledgeCitation(byId.get(reference.pageId) ?? [], reference) + return resolution.resolved ? [qualified(resolution.resolved)] : [] + } catch (error) { + if (error instanceof TypeError) return [] + throw error + } + }), + } + originals.set(page, entry) + return page + }) + const ranked = searchKnowledgePages(pages, question, { + limit, + excludeInvalidated, + ...(options.tags === undefined ? {} : { tags: options.tags }), + ...(options.kinds === undefined ? {} : { kinds: options.kinds }), + }).map((hit) => { + const entry = originals.get(hit.page)! + const candidates = byId.get(entry.page.id)! + const reference = { + pageId: entry.page.id, + ...(candidates.length === 1 && + parseKnowledgeCitationReference(entry.page.id).origin === undefined + ? {} + : { origin: entry.origin }), + } + // The whole visible chain determines ambiguity, not just this query's filtered hits. + // Reusing one id within the SAME origin cannot be repaired with a qualifier. + assertKnowledgeCitationsResolved(candidates, [reference]) + return { + ...hit, + page: entry.page, + origin: entry.origin, + citationId: formatKnowledgeCitationReference(reference), + } + }) - const hits: KnowledgeSearchHit[] = [] + const hits: (KnowledgeSearchHit & OriginatedKnowledgeSearchResult)[] = [] const lines: string[] = [] let length = 0 for (const hit of ranked) { @@ -125,9 +174,7 @@ export function buildKnowledgeBrief( }), hits: Object.freeze(hits), citationIds: Object.freeze(hits.map((hit) => hit.citationId)), - results: Object.freeze( - hits.map((hit) => Object.freeze({ ...hit, origin: originOf(originByPage, hit) })), - ), + results: Object.freeze(hits.map((hit) => Object.freeze(hit))), text: lines.join('\n'), }) } @@ -138,14 +185,3 @@ function briefLine(hit: KnowledgeSearchHit): string { ? `- [${hit.citationId}] ${hit.page.title}` : `- [${hit.citationId}] ${hit.page.title} — ${snippet}` } - -function originOf( - originByPage: ReadonlyMap, - hit: KnowledgeSearchHit, -): PageOrigin { - const origin = originByPage.get(hit.page) - if (origin === undefined) { - throw new Error(`knowledge brief ranked a page outside the visible chain: ${hit.page.path}`) - } - return origin -} diff --git a/src/knowledge-tools.test.ts b/src/knowledge-tools.test.ts index 88e958e..8a668e9 100644 --- a/src/knowledge-tools.test.ts +++ b/src/knowledge-tools.test.ts @@ -220,3 +220,66 @@ describe('createKnowledgeRetrievalDisposition', () => { ) }) }) + +describe('search controls use the existing brief semantics', () => { + it('allows an agent to inspect invalidated history without changing host defaults', async () => { + const historical = { + id: 'refuted', + path: 'knowledge/refuted.md', + title: 'Refuted quantum approach', + text: 'Quantum decoding.', + frontmatter: { kind: 'finding' }, + sourceIds: [], + tags: ['history'], + outLinks: [], + invalidation: { + verdict: 'contradicted' as const, + observedAt: '2026-09-20T00:00:00Z', + reason: 'Counterexample.', + }, + } + const scopedStores = { + ...stores, + loadChain: async () => [{ page: historical, origin: 'here' as const }], + } + const search = createKnowledgeTools({ + stores: scopedStores, + runId: 'run-a', + retrieverVersion: 'test', + }).find((tool) => tool.name === 'knowledge_search')! + const invoke = async (input: unknown) => + (await search.handler(input, {})) as { citationIds: string[] } + expect((await invoke({ question: 'quantum' })).citationIds).toEqual([]) + expect( + ( + await invoke({ + question: 'quantum', + excludeInvalidated: false, + tags: ['history'], + kinds: ['finding'], + }) + ).citationIds, + ).toEqual(['refuted']) + expect( + (await invoke({ question: 'quantum', excludeInvalidated: false, tags: ['other'] })) + .citationIds, + ).toEqual([]) + expect((await invoke({ question: 'quantum' })).citationIds).toEqual([]) + }) + + it('search-to-read round trips origin-qualified handles through the actual tools and stores', async () => { + await writePage(stores.storePath('run-a'), 'state', 'Quantum decoding.') + await initKnowledgeBase(shared) + await writePage(shared, 'state', 'Banana schedules.') + const search = await call('knowledge_search', { question: 'quantum' }) + expect(search.citationIds).toEqual(['here::state']) + const found = await call('knowledge_read', { pageId: (search.citationIds as string[])[0] }) + expect(found.status).toBe('resolved') + expect(found.page).toMatchObject({ + origin: 'here', + pageId: 'state', + text: expect.stringContaining('Quantum'), + }) + expect(recorded[0]?.results[0]).toMatchObject({ origin: 'here', pageId: 'state' }) + }) +}) diff --git a/src/knowledge-tools.ts b/src/knowledge-tools.ts index 29708b7..60e7761 100644 --- a/src/knowledge-tools.ts +++ b/src/knowledge-tools.ts @@ -60,6 +60,12 @@ export interface CreateKnowledgeToolsOptions { const searchInput = z.object({ question: z.string().min(1), limit: z.int().min(1).max(50).optional(), + excludeInvalidated: z + .boolean() + .optional() + .describe('False includes refuted pages for historical research.'), + tags: z.array(z.string()).optional(), + kinds: z.array(z.string()).optional(), }) const readInput = z.object({ pageId: z.string().min(1) }) const recordInput = z.object({ @@ -94,13 +100,18 @@ export function createKnowledgeTools(options: CreateKnowledgeToolsOptions): Tool return [ tool( 'knowledge_search', - 'Search the knowledge this run can see and return a brief with the ids to cite.', + 'Search visible knowledge with unambiguous citation handles. Optionally include refuted history or filter tags and kinds.', searchInput, async (input) => { const chain = await stores.loadChain(runId) const brief = buildKnowledgeBrief(chain, input.question, { ...options.brief, ...(input.limit === undefined ? {} : { limit: input.limit }), + ...(input.excludeInvalidated === undefined + ? {} + : { excludeInvalidated: input.excludeInvalidated }), + ...(input.tags === undefined ? {} : { tags: input.tags }), + ...(input.kinds === undefined ? {} : { kinds: input.kinds }), }) const visibility = createKnowledgeVisibilitySnapshot(chain) const visibilityArtifact = options.recordRetrieval diff --git a/src/search-origins.test.ts b/src/search-origins.test.ts new file mode 100644 index 0000000..aa36eca --- /dev/null +++ b/src/search-origins.test.ts @@ -0,0 +1,165 @@ +import { describe, expect, it } from 'vitest' +import { parseKnowledgeCitationReference, resolveKnowledgeCitation } from './citation-resolution' +import { buildKnowledgeBrief } from './knowledge-brief' +import { + assertKnowledgeRetrievalMatchesVisibility, + createKnowledgeRetrievalReceipt, + createKnowledgeVisibilitySnapshot, +} from './knowledge-use-receipts' +import type { OriginatedPage } from './run-scoped' +import { searchKnowledgePages } from './search' +import type { KnowledgePage } from './types' + +const page = (id: string, title: string, text: string, outLinks: string[] = []): KnowledgePage => ({ + id, + title, + path: `knowledge/${id}.md`, + text, + outLinks, + tags: [], + sourceIds: [], + frontmatter: { id }, +}) +const quantum = page('state', 'Quantum error correction', 'Quantum syndrome decoding.') +const banana = page('state', 'Banana logistics', 'Banana shipment schedules.') +const visible: OriginatedPage[] = [ + { page: quantum, origin: 'here' }, + { page: banana, origin: 'inherited:prior' }, +] + +function assertRoundTrips(chain: OriginatedPage[], question: string) { + const brief = buildKnowledgeBrief(chain, question, { limit: 50 }) + for (const hit of brief.results) { + const reference = brief.citationIds[hit.rank - 1]! + const resolved = resolveKnowledgeCitation(chain, parseKnowledgeCitationReference(reference)) + expect(resolved.status).toBe('resolved') + expect(resolved.resolved?.page).toBe(hit.page) + expect(resolved.resolved?.origin).toBe(hit.origin) + } + const visibility = createKnowledgeVisibilitySnapshot(chain) + const receipt = createKnowledgeRetrievalReceipt({ + runId: 'origin-test', + query: brief.question, + retriever: { + id: brief.retrieverId, + version: 'test', + configDigest: brief.retrieverConfigDigest, + }, + visibility, + results: brief.results, + }) + assertKnowledgeRetrievalMatchesVisibility(receipt, visibility) + return brief +} + +describe('retrieval preserves document identity across origins', () => { + it('does not replace a matching document with an unrelated same-id document, in either order', () => { + for (const chain of [visible, [...visible].reverse()]) { + const raw = searchKnowledgePages( + chain.map((entry) => entry.page), + 'quantum', + ) + expect(raw).toHaveLength(1) + expect(raw[0]!.page).toBe(quantum) + const brief = assertRoundTrips(chain, 'quantum') + expect(brief.citationIds).toEqual(['here::state']) + expect(brief.text).toContain('Quantum error correction') + expect(brief.text).not.toContain('Banana') + } + }) + + it('keeps both independently matching same-id pages rather than fusing them into one', () => { + const a = page('state', 'Quantum A', 'quantum') + const b = page('state', 'Quantum B', 'quantum') + expect(searchKnowledgePages([a, b], 'quantum').map((hit) => hit.page)).toEqual([a, b]) + }) + + it('can distinguish the same page object exposed at several origins', () => { + const chain: OriginatedPage[] = ['here', 'shared', 'inherited:prior'].map((origin) => ({ + page: quantum, + origin: origin as OriginatedPage['origin'], + })) + const before = JSON.stringify(chain) + const result = assertRoundTrips(chain, 'quantum') + expect(new Set(result.citationIds).size).toBe(3) + expect(result.results.every((hit) => hit.page === quantum)).toBe(true) + expect(assertRoundTrips([...chain].reverse(), 'quantum').citationIds).toEqual( + result.citationIds, + ) + expect(JSON.stringify(chain)).toBe(before) + }) + + it('qualifies against all visible pages, even when only one duplicate survives the filters', () => { + const chain: OriginatedPage[] = [ + { origin: 'here', page: { ...quantum, tags: ['selected'] } }, + { origin: 'shared', page: { ...banana, tags: ['excluded'] } }, + ] + const brief = buildKnowledgeBrief(chain, 'quantum', { tags: ['selected'], limit: 1 }) + expect(brief.citationIds).toEqual(['here::state']) + expect( + resolveKnowledgeCitation(chain, parseKnowledgeCitationReference(brief.citationIds[0]!)) + .resolved?.page, + ).toBe(chain[0]!.page) + }) + + it('preserves unqualified handles for unique ordinary page ids', () => { + expect(assertRoundTrips([visible[0]!], 'quantum').citationIds).toEqual(['state']) + }) + + it('keeps text, hits and receipt results aligned after the rendering bound', () => { + const brief = buildKnowledgeBrief(visible, 'quantum', { maxChars: 1 }) + expect(brief.text).toBe('') + expect(brief.hits).toEqual([]) + expect(brief.results).toEqual([]) + expect(brief.citationIds).toEqual([]) + }) + + it('refuses a genuinely ambiguous same-origin citation rather than choosing a page', () => { + expect(() => + buildKnowledgeBrief( + [ + { origin: 'here', page: quantum }, + { origin: 'here', page: { ...banana, path: 'knowledge/other.md' } }, + ], + 'quantum', + ), + ).toThrow(/ambiguous/) + }) + + it('uses the existing citation resolver for graph edges, not bare id coincidence', () => { + const chain: OriginatedPage[] = [ + ...visible, + { + origin: 'here', + page: page('right-link', 'Correct neighbor', 'A useful neighbor.', ['here::state']), + }, + { + origin: 'here', + page: page('wrong-link', 'Unrelated neighbor', 'Another neighbor.', [ + 'inherited:prior::state', + ]), + }, + { + origin: 'here', + page: page('ambiguous-link', 'Ambiguous neighbor', 'Ambiguous.', ['state']), + }, + ] + const brief = assertRoundTrips(chain, 'quantum') + expect([...brief.citationIds].sort()).toEqual(['here::state', 'right-link']) + }) + + it('does not infer an unscoped graph edge from an ambiguous bare id', () => { + const linked = page('link', 'Neighbor', 'Neighbor.', ['state']) + expect( + searchKnowledgePages([quantum, banana, linked], 'quantum').map((hit) => hit.page), + ).toEqual([quantum]) + }) + + it('keeps searchable evidence with malformed outgoing links without inventing an edge', () => { + const malformed = { ...quantum, outLinks: ['inherited:::state'] } + const chain: OriginatedPage[] = [{ origin: 'here', page: malformed }] + const result = assertRoundTrips(chain, 'quantum') + expect(result.results[0]!.page).toBe(malformed) + expect(result.results[0]!.page.outLinks).toEqual(['inherited:::state']) + }) +}) diff --git a/src/search.ts b/src/search.ts index 22117a3..3411404 100644 --- a/src/search.ts +++ b/src/search.ts @@ -39,8 +39,8 @@ export interface SearchKnowledgeOptions { /** * A retrieval result with an explicit citation handle. * - * `citationId` is exactly `page.id`; later writes should persist this value - * when they cite the page. Keeping it at the result's top level prevents tool + * Unscoped search returns `page.id`; a run-scoped brief qualifies ambiguous ids + * through the citation resolver. Later writes should persist the returned value. Keeping it at the result's top level prevents tool * renderers from accidentally hiding the only stable handle a model can copy. */ export interface KnowledgeSearchHit extends KnowledgeSearchResult { @@ -90,17 +90,14 @@ export function searchKnowledgePages( ? assertLexicalIndexMatches(options.lexicalIndex, pages) : buildKnowledgeLexicalIndex(pages) const lexicalRanked = rankLexical(matched, trimmed, lexicalIndex) - const graphRanked = rankByGraph(matched, lexicalRanked) - const scores = reciprocalRankFusion([ - lexicalRanked.map((p) => p.id), - graphRanked.map((p) => p.id), - ]) - const byId = new Map(matched.map((page) => [page.id, page])) + const graphRanked = rankByGraph(matched, lexicalRanked, pages) + // Stable ids are citation addresses, not unique document identities across stores. + // Fuse the same page objects used by BM25, then return those exact pages. + const scores = fuseRanks([lexicalRanked, graphRanked]) const ranked = [...scores.entries()] - .map(([id, score]) => ({ page: byId.get(id), score })) - .filter((item): item is { page: KnowledgePage; score: number } => Boolean(item.page)) - .sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)) + .map(([page, score]) => ({ page, score })) + .sort((a, b) => b.score - a.score || comparePages(a.page, b.page)) .slice(0, limit) // Normalize against the top hit so callers can compare against natural @@ -122,7 +119,12 @@ export function searchKnowledgePages( } export function reciprocalRankFusion(rankLists: string[][], k = RRF_K): Map { - const scores = new Map() + return fuseRanks(rankLists, k) +} + +// The public string-key helper and document retrieval share one ranking implementation. +function fuseRanks(rankLists: readonly (readonly T[])[], k = RRF_K): Map { + const scores = new Map() for (const list of rankLists) { list.forEach((id, idx) => { scores.set(id, (scores.get(id) ?? 0) + 1 / (k + idx + 1)) @@ -190,7 +192,7 @@ function rankLexical( if (score === 0 && tier === 0) return [] return [{ page, tier, score }] }) - .sort((a, b) => b.tier - a.tier || b.score - a.score || a.page.path.localeCompare(b.page.path)) + .sort((a, b) => b.tier - a.tier || b.score - a.score || comparePages(a.page, b.page)) .map((item) => item.page) } @@ -202,9 +204,22 @@ function phraseTier(page: KnowledgePage, phrase: string): number { return 0 } -function rankByGraph(pages: KnowledgePage[], lexicalRanked: KnowledgePage[]): KnowledgePage[] { +function rankByGraph( + pages: KnowledgePage[], + lexicalRanked: KnowledgePage[], + visiblePages: readonly KnowledgePage[], +): KnowledgePage[] { if (lexicalRanked.length === 0) return [] - const seeds = new Set(lexicalRanked.slice(0, 5).map((page) => page.id)) + // A bare link to a duplicate id has no unambiguous target. Scoped callers + // qualify links before ranking; unscoped callers must not invent that edge. + const counts = new Map() + for (const page of visiblePages) counts.set(page.id, (counts.get(page.id) ?? 0) + 1) + const seeds = new Set( + lexicalRanked + .slice(0, 5) + .filter((page) => counts.get(page.id) === 1) + .map((page) => page.id), + ) return pages .map((page) => ({ page, @@ -215,10 +230,14 @@ function rankByGraph(pages: KnowledgePage[], lexicalRanked: KnowledgePage[]): Kn ).length, })) .filter((item) => item.score > 0) - .sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)) + .sort((a, b) => b.score - a.score || comparePages(a.page, b.page)) .map((item) => item.page) } +function comparePages(a: KnowledgePage, b: KnowledgePage): number { + return a.path.localeCompare(b.path) || a.id.localeCompare(b.id) +} + function buildSnippet(text: string, query: string): string { const compact = text.replace(/\s+/g, ' ').trim() const idx = compact.toLowerCase().indexOf(query.toLowerCase()) From 9501e0f66c8c33e1082b686031197a556f4f3dc5 Mon Sep 17 00:00:00 2001 From: drewstone Date: Sun, 20 Sep 2026 07:55:27 -0700 Subject: [PATCH 3/9] fix(knowledge): make origin citation handles round-trip without delimiter collisions --- src/citation-encoding.test.ts | 107 ++++++++++++++++++++++++++++++++++ src/citation-resolution.ts | 42 +++++++++++-- src/knowledge-brief.ts | 2 +- 3 files changed, 146 insertions(+), 5 deletions(-) create mode 100644 src/citation-encoding.test.ts diff --git a/src/citation-encoding.test.ts b/src/citation-encoding.test.ts new file mode 100644 index 0000000..7f27e5c --- /dev/null +++ b/src/citation-encoding.test.ts @@ -0,0 +1,107 @@ +import { describe, expect, it } from 'vitest' +import { + formatKnowledgeCitationReference, + parseKnowledgeCitationReference, + resolveKnowledgeCitation, +} from './citation-resolution' +import { buildKnowledgeBrief } from './knowledge-brief' +import type { PageOrigin } from './run-scoped' +import type { KnowledgePage } from './types' + +const page = (id: string, text: string): KnowledgePage => ({ + id, + text, + title: text, + path: `${id}.md`, + outLinks: [], + tags: [], + sourceIds: [], + frontmatter: { id }, +}) + +describe('citation serialization is injective, including legacy delimiter collisions', () => { + it('round trips every supported origin/page pair without banning existing names', () => { + const origins: (PageOrigin | undefined)[] = [ + undefined, + 'here', + 'shared', + 'inherited:a', + 'inherited:a::b', + 'inherited:a:', + 'inherited:a%3A%3Ab', + 'inherited:∑:研究', + 'inherited:%/::x', + ] + const ids = [ + 'c', + 'b::c', + 'here::literal', + 'shared::', + 'knowledge-ref:v1:literal', + '%3A%3A', + 'quantum-∑', + 'a::b::c', + ] + const handles = new Set() + for (const origin of origins) { + for (const pageId of ids) { + const reference = { pageId, ...(origin === undefined ? {} : { origin }) } + const handle = formatKnowledgeCitationReference(reference) + expect(parseKnowledgeCitationReference(handle)).toEqual(reference) + expect(handles.has(handle)).toBe(false) + handles.add(handle) + } + } + }) + + it('leaves normal handles and legacy literal percent sequences unchanged', () => { + expect(formatKnowledgeCitationReference({ origin: 'inherited:a', pageId: 'c' })).toBe( + 'inherited:a::c', + ) + expect(parseKnowledgeCitationReference('inherited:a%3A%3Ab::c')).toEqual({ + origin: 'inherited:a%3A%3Ab', + pageId: 'c', + }) + expect(formatKnowledgeCitationReference({ pageId: 'plain' })).toBe('plain') + }) + + it('keeps the two former colliding origins distinct through search and resolution', () => { + const first = page('c', 'Quantum first') + const second = page('b::c', 'Quantum second') + const visible = [ + { page: first, origin: 'inherited:a::b' as const }, + { page: second, origin: 'inherited:a' as const }, + { page: page('c', 'Banana local'), origin: 'here' as const }, + ] + const result = buildKnowledgeBrief(visible, 'quantum') + expect(result.results).toHaveLength(2) + expect(new Set(result.citationIds).size).toBe(2) + for (const hit of result.results) { + const resolved = resolveKnowledgeCitation( + visible, + parseKnowledgeCitationReference(hit.citationId), + ) + expect(resolved.resolved?.page).toBe(hit.page) + expect(resolved.resolved?.origin).toBe(hit.origin) + } + }) + + it('retrieves literal reserved-prefix page names through the same formatter', () => { + const literal = page('knowledge-ref:v1:literal', 'Quantum reserved name') + const visible = [{ page: literal, origin: 'here' as const }] + const brief = buildKnowledgeBrief(visible, 'quantum') + expect(brief.results).toHaveLength(1) + expect( + resolveKnowledgeCitation(visible, parseKnowledgeCitationReference(brief.citationIds[0]!)) + .resolved?.page, + ).toBe(literal) + }) + + it('rejects malformed encoded references instead of silently changing their identity', () => { + for (const value of ['%', '%5B%5D', '%5Bnull%5D', '%5Btrue%2C%22p%22%5D']) { + expect(() => parseKnowledgeCitationReference(`knowledge-ref:v1:${value}`)).toThrow( + /invalid encoded knowledge citation/, + ) + } + }) +}) diff --git a/src/citation-resolution.ts b/src/citation-resolution.ts index 66c45c0..545afa9 100644 --- a/src/citation-resolution.ts +++ b/src/citation-resolution.ts @@ -91,17 +91,40 @@ export class KnowledgeCitationAuditError extends Error { } } +// Legacy origin::page handles stay readable and byte-compatible. Use an explicit, +// versioned tuple only when delimiters or a reserved prefix would lose identity. +const ENCODED_REFERENCE_PREFIX = 'knowledge-ref:v1:' + /** * Parse the persisted citation form. * * `page-id` is unqualified. `here::page-id`, `shared::page-id`, and * `inherited:::page-id` bind an intentional duplicate to one origin. + * Ambiguous names use `knowledge-ref:v1:` plus a URI-encoded JSON [origin, pageId] + * tuple (null origin means unqualified). Always use the formatter to mint handles. + * Legacy percent sequences are literal; old records are not silently reinterpreted. */ export function parseKnowledgeCitationReference(value: string): KnowledgeCitationReference { if (typeof value !== 'string' || value.trim().length === 0) { throw new TypeError('persisted knowledge citation must be a non-empty string') } const normalized = value.trim() + if (normalized.startsWith(ENCODED_REFERENCE_PREFIX)) { + try { + const tuple: unknown = JSON.parse( + decodeURIComponent(normalized.slice(ENCODED_REFERENCE_PREFIX.length)), + ) + if (!Array.isArray(tuple) || tuple.length !== 2) { + throw new TypeError('encoded citation must contain an origin/page tuple') + } + return normalizeReference({ + pageId: tuple[1], + ...(tuple[0] === null ? {} : { origin: tuple[0] }), + }) + } catch (cause) { + throw new TypeError('invalid encoded knowledge citation', { cause }) + } + } const separator = normalized.indexOf('::') if (separator < 0) return Object.freeze({ pageId: normalized }) const possibleOrigin = normalized.slice(0, separator) @@ -115,9 +138,20 @@ export function parseKnowledgeCitationReference(value: string): KnowledgeCitatio /** Serialize one reference into the canonical frontmatter representation. */ export function formatKnowledgeCitationReference(reference: KnowledgeCitationReference): string { const normalized = normalizeReference(reference) - return normalized.origin === undefined - ? normalized.pageId - : `${normalized.origin}::${normalized.pageId}` + const legacy = + normalized.origin === undefined + ? normalized.pageId + : `${normalized.origin}::${normalized.pageId}` + try { + const parsed = parseKnowledgeCitationReference(legacy) + if (parsed.pageId === normalized.pageId && parsed.origin === normalized.origin) return legacy + } catch (error) { + // A literal page id can itself look like a malformed reserved handle. + if (!(error instanceof TypeError)) throw error + } + return `${ENCODED_REFERENCE_PREFIX}${encodeURIComponent( + JSON.stringify([normalized.origin ?? null, normalized.pageId]), + )}` } /** Resolve one reference against an already materialized visibility chain. */ @@ -220,7 +254,7 @@ export function auditKnowledgeCitations( return Object.freeze({ ok: issues.length === 0, - checkedPages: sources.length, + checkedPages: visiblePages.filter((entry) => origins === null || origins.has(entry.origin)).length, checkedCitations, issues: Object.freeze(issues), }) diff --git a/src/knowledge-brief.ts b/src/knowledge-brief.ts index 59d5fc0..5310f88 100644 --- a/src/knowledge-brief.ts +++ b/src/knowledge-brief.ts @@ -135,7 +135,7 @@ export function buildKnowledgeBrief( const reference = { pageId: entry.page.id, ...(candidates.length === 1 && - parseKnowledgeCitationReference(entry.page.id).origin === undefined + formatKnowledgeCitationReference({ pageId: entry.page.id }) === entry.page.id ? {} : { origin: entry.origin }), } From 3fe9377f4b21521e0ac83dc23c73bd1dde68e477 Mon Sep 17 00:00:00 2001 From: drewstone Date: Sun, 20 Sep 2026 08:03:27 -0700 Subject: [PATCH 4/9] chore(audit): apply exact citation review cleanup on existing repair branch --- .github/workflows/citation-review-cleanup.yml | 32 +++++++++++++++++++ 1 file changed, 32 insertions(+) create mode 100644 .github/workflows/citation-review-cleanup.yml diff --git a/.github/workflows/citation-review-cleanup.yml b/.github/workflows/citation-review-cleanup.yml new file mode 100644 index 0000000..c79d1e1 --- /dev/null +++ b/.github/workflows/citation-review-cleanup.yml @@ -0,0 +1,32 @@ +name: Citation review cleanup +on: + push: + branches: [fix/origin-aware-retrieval] + paths: [.github/workflows/citation-review-cleanup.yml] +permissions: + contents: write +jobs: + apply: + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 + with: + ref: fix/origin-aware-retrieval + - name: Remove redundant rescan and this temporary transport + run: | + set -euo pipefail + python3 - <<'PY' + from pathlib import Path + p = Path('src/citation-resolution.ts') + text = p.read_text() + old = 'checkedPages: visiblePages.filter((entry) => origins === null || origins.has(entry.origin)).length,' + assert text.count(old) == 1 + p.write_text(text.replace(old, 'checkedPages: sources.length,')) + PY + git rm .github/workflows/citation-review-cleanup.yml + git add src/citation-resolution.ts + git config user.name 'Tangle automation' + git config user.email 'automation@tangle.tools' + git commit -m 'fix(knowledge): reuse the already selected citation audit sources' + git push origin HEAD:fix/origin-aware-retrieval From 679d589bf65d476ee3b588f6714129815757547a Mon Sep 17 00:00:00 2001 From: Tangle automation Date: Sun, 20 Sep 2026 15:03:37 +0000 Subject: [PATCH 5/9] fix(knowledge): reuse the already selected citation audit sources --- .github/workflows/citation-review-cleanup.yml | 32 ------------------- src/citation-resolution.ts | 2 +- 2 files changed, 1 insertion(+), 33 deletions(-) delete mode 100644 .github/workflows/citation-review-cleanup.yml diff --git a/.github/workflows/citation-review-cleanup.yml b/.github/workflows/citation-review-cleanup.yml deleted file mode 100644 index c79d1e1..0000000 --- a/.github/workflows/citation-review-cleanup.yml +++ /dev/null @@ -1,32 +0,0 @@ -name: Citation review cleanup -on: - push: - branches: [fix/origin-aware-retrieval] - paths: [.github/workflows/citation-review-cleanup.yml] -permissions: - contents: write -jobs: - apply: - runs-on: ubuntu-latest - timeout-minutes: 5 - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 - with: - ref: fix/origin-aware-retrieval - - name: Remove redundant rescan and this temporary transport - run: | - set -euo pipefail - python3 - <<'PY' - from pathlib import Path - p = Path('src/citation-resolution.ts') - text = p.read_text() - old = 'checkedPages: visiblePages.filter((entry) => origins === null || origins.has(entry.origin)).length,' - assert text.count(old) == 1 - p.write_text(text.replace(old, 'checkedPages: sources.length,')) - PY - git rm .github/workflows/citation-review-cleanup.yml - git add src/citation-resolution.ts - git config user.name 'Tangle automation' - git config user.email 'automation@tangle.tools' - git commit -m 'fix(knowledge): reuse the already selected citation audit sources' - git push origin HEAD:fix/origin-aware-retrieval diff --git a/src/citation-resolution.ts b/src/citation-resolution.ts index 545afa9..f3097d3 100644 --- a/src/citation-resolution.ts +++ b/src/citation-resolution.ts @@ -254,7 +254,7 @@ export function auditKnowledgeCitations( return Object.freeze({ ok: issues.length === 0, - checkedPages: visiblePages.filter((entry) => origins === null || origins.has(entry.origin)).length, + checkedPages: sources.length, checkedCitations, issues: Object.freeze(issues), }) From 1d2de3b7646b5f5b2321670947d7075832379c22 Mon Sep 17 00:00:00 2001 From: drewstone Date: Sun, 20 Sep 2026 08:08:08 -0700 Subject: [PATCH 6/9] docs(knowledge): document collision-free citation handles and legacy compatibility --- docs/run-scoped-citations.md | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/docs/run-scoped-citations.md b/docs/run-scoped-citations.md index ad9a9ab..05df394 100644 --- a/docs/run-scoped-citations.md +++ b/docs/run-scoped-citations.md @@ -31,6 +31,17 @@ cites: Use `parseKnowledgeCitationReference()` and `formatKnowledgeCitationReference()` rather than assembling qualified strings in application code. +Names containing the delimiter are supported, not banned. When the legacy spelling would +lose identity, the formatter emits `knowledge-ref:v1:` followed by a URI-encoded JSON +`[origin, pageId]` tuple; a null origin means unqualified. For example, origin +`inherited:a::b` with page `c` must not alias origin `inherited:a` with page `b::c`. +Ordinary handles retain their previous bytes, and percent sequences in legacy handles +remain literal. The new prefix is reserved for encoded references; use the formatter +(or the structured reference API) for a literal page id beginning with that prefix. +Malformed encoded handles fail explicitly. Historical ambiguous handles are not guessed +or silently rewritten; retain the original evidence and qualify a new reference from +its known origin. + ## Search and read use the same identity `buildKnowledgeBrief` and `knowledge_search` rank distinct visible documents without From 43cf11b190c2f8ba60d206bc404808f89cecec46 Mon Sep 17 00:00:00 2001 From: drewstone Date: Sun, 20 Sep 2026 08:17:51 -0700 Subject: [PATCH 7/9] test(knowledge): use the public citationIds contract for search round trips --- src/citation-encoding.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/citation-encoding.test.ts b/src/citation-encoding.test.ts index 7f27e5c..2867a99 100644 --- a/src/citation-encoding.test.ts +++ b/src/citation-encoding.test.ts @@ -76,10 +76,10 @@ describe('citation serialization is injective, including legacy delimiter collis const result = buildKnowledgeBrief(visible, 'quantum') expect(result.results).toHaveLength(2) expect(new Set(result.citationIds).size).toBe(2) - for (const hit of result.results) { + for (const [index, hit] of result.results.entries()) { const resolved = resolveKnowledgeCitation( visible, - parseKnowledgeCitationReference(hit.citationId), + parseKnowledgeCitationReference(result.citationIds[index]!), ) expect(resolved.resolved?.page).toBe(hit.page) expect(resolved.resolved?.origin).toBe(hit.origin) From af1c930e46d5569a8153c78bc2bc58b045a99001 Mon Sep 17 00:00:00 2001 From: drewstone Date: Sun, 20 Sep 2026 10:29:44 -0600 Subject: [PATCH 8/9] chore(audit): reconcile pinned retrieval repair with merged Eval compatibility --- .../reconcile-retrieval-compatibility.yml | 82 +++++++++++++++++++ 1 file changed, 82 insertions(+) create mode 100644 .github/workflows/reconcile-retrieval-compatibility.yml diff --git a/.github/workflows/reconcile-retrieval-compatibility.yml b/.github/workflows/reconcile-retrieval-compatibility.yml new file mode 100644 index 0000000..dd19c9b --- /dev/null +++ b/.github/workflows/reconcile-retrieval-compatibility.yml @@ -0,0 +1,82 @@ +name: Reconcile pinned retrieval compatibility +on: + push: + branches: [fix/origin-aware-retrieval] + paths: [.github/workflows/reconcile-retrieval-compatibility.yml] +permissions: + contents: write +jobs: + reconcile: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 + with: + ref: ${{ github.sha }} + fetch-depth: 1 + - name: Preserve both changes and remove temporary reconciliation transport + env: + TARGET: fix/origin-aware-retrieval + run: | + set -euo pipefail + out="$RUNNER_TEMP/retrieval-reconciliation" + mkdir -p "$out" + base=ee172b1ed30d507652f1132eda382f46aadc0f95 + ours=43cf11b190c2f8ba60d206bc404808f89cecec46 + theirs=f1072251361ef78aee7d985d32516524bec8cb9a + git fetch --no-tags --depth=1 origin "$base" "$ours" "$theirs" + test "$(git ls-remote origin "refs/heads/$TARGET" | cut -f1)" = "$GITHUB_SHA" + set +e + git merge-tree --write-tree --merge-base="$base" "$ours" "$theirs" > "$out/merge-tree.txt" + result=$? + set -e + test "$result" -le 1 + tree=$(head -n1 "$out/merge-tree.txt") + git read-tree --reset -u "$tree" + python3 - <<'PY' + import json, subprocess + from pathlib import Path + ours='43cf11b190c2f8ba60d206bc404808f89cecec46' + theirs='f1072251361ef78aee7d985d32516524bec8cb9a' + def show(ref,path): return subprocess.check_output(['git','show',f'{ref}:{path}']).decode() + package=show(theirs,'package.json') + assert json.loads(package)['version']=='17.0.3' + assert json.loads(package)['peerDependencies']['@tangle-network/agent-eval']=='>=0.182.0 <0.184.0' + assert package.count('"version": "17.0.3"')==1 + Path('package.json').write_text(package.replace('"version": "17.0.3"','"version": "17.1.0"')) + old=show(ours,'CHANGELOG.md'); incoming=show(theirs,'CHANGELOG.md') + marker='## 17.0.2 — 2026-09-16' + assert old.count(marker)==1 and incoming.count(marker)==1 + assert old.startswith('# Changelog\n\n## 17.1.0\n') + addition=incoming.split(marker)[0].removeprefix('# Changelog\n\n') + assert addition.startswith('## 17.0.3 — 2026-09-20\n') + Path('CHANGELOG.md').write_text(old.replace(marker,addition+marker)) + readme=show(theirs,'README.md') + assert readme.count('@tangle-network/agent-knowledge@17.0.3')==1 + Path('README.md').write_text(readme.replace('@tangle-network/agent-knowledge@17.0.3','@tangle-network/agent-knowledge@17.1.0')) + # No retrieval implementation or test changes during compatibility reconciliation. + for line in subprocess.check_output(['git','ls-tree','-r','--name-only',ours,'src','docs/run-scoped-citations.md']).decode().splitlines(): + assert Path(line).read_bytes()==subprocess.check_output(['git','show',f'{ours}:{line}']),line + # Preserve main's new qualification jobs and scripts byte-for-byte. + for path in ['.github/workflows/ci.yml','.github/workflows/publish.yml','scripts/lib/peer-range.mjs','scripts/lib/peer-range.test.mjs','scripts/verify-package.mjs','scripts/verify-official-optimizers.mjs']: + assert Path(path).read_bytes()==subprocess.check_output(['git','show',f'{theirs}:{path}']),path + PY + git add package.json CHANGELOG.md README.md + git diff --cached --check "$theirs" + test ! -e .github/workflows/reconcile-retrieval-compatibility.yml + git diff --cached --binary "$theirs" > "$out/candidate.patch" + git diff --cached --stat "$theirs" > "$out/stat.txt" + git config user.name 'github-actions[bot]' + git config user.email '41898282+github-actions[bot]@users.noreply.github.com' + final_tree=$(git write-tree) + printf '%s\n' "$final_tree" > "$out/tree.txt" + commit=$(printf '%s\n' 'Merge Eval 0.183 compatibility into origin-safe retrieval repair' '' 'Preserve #215 peer range, package/optimizer qualification, and 17.0.3 history. Keep candidate 17.1.0 and retrieval implementations unchanged. Remove temporary reconciliation workflow.' | git commit-tree "$final_tree" -p "$GITHUB_SHA" -p "$theirs") + printf '%s\n' "$commit" > "$out/commit.txt" + git push origin "$commit:refs/heads/$TARGET" + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a + if: always() + with: + name: retrieval-compatibility-reconciliation + path: ${{ runner.temp }}/retrieval-reconciliation/ + retention-days: 2 + if-no-files-found: error From a89c128a6a5502cb9800abcd9764cbc3b9859a96 Mon Sep 17 00:00:00 2001 From: drewstone Date: Sun, 20 Sep 2026 10:30:42 -0600 Subject: [PATCH 9/9] test(knowledge): qualify reconciled retrieval against both supported Eval minors