mirror of
https://github.com/semantica-agi/semantica.git
synced 2026-09-02 04:00:40 +00:00
Compare commits
22
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8e7aaee4f5 | ||
|
|
e040d84d59 | ||
|
|
2eab7ab876 | ||
|
|
d8822198cf | ||
|
|
e6409217dd | ||
|
|
3ed31b9182 | ||
|
|
7300fb41b1 | ||
|
|
218e5a33f3 | ||
|
|
635f6e52f4 | ||
|
|
9240a1b1f7 | ||
|
|
3254b9be80 | ||
|
|
dc81acaefd | ||
|
|
f44020c742 | ||
|
|
abb65feff0 | ||
|
|
8b125d6476 | ||
|
|
f3c540cfd2 | ||
|
|
46b18fbee3 | ||
|
|
b0679d4f67 | ||
|
|
d135ad185f | ||
|
|
96dbd3f0d4 | ||
|
|
73b14c00ba | ||
|
|
e9a756eac2 |
@@ -11,3 +11,4 @@ python-docx
|
|||||||
beautifulsoup4
|
beautifulsoup4
|
||||||
chardet
|
chardet
|
||||||
langdetect
|
langdetect
|
||||||
|
en-core-web-sm @ https://github.com/explosion/spacy-models/releases/download/en_core_web_sm-3.8.0/en_core_web_sm-3.8.0-py3-none-any.whl
|
||||||
|
|||||||
@@ -408,6 +408,9 @@ cuda-toolkit==13.0.3.0 \
|
|||||||
# via
|
# via
|
||||||
# -c requirements-ci.txt
|
# -c requirements-ci.txt
|
||||||
# torch
|
# torch
|
||||||
|
en-core-web-sm @ https://github.com/explosion/spacy-models/releases/download/en_core_web_sm-3.8.0/en_core_web_sm-3.8.0-py3-none-any.whl \
|
||||||
|
--hash=sha256:1932429db727d4bff3deed6b34cfc05df17794f4a52eeb26cf8928f7c1a0fb85
|
||||||
|
# via -r .github/requirements/benchmark-extra.in
|
||||||
et-xmlfile==2.0.0 \
|
et-xmlfile==2.0.0 \
|
||||||
--hash=sha256:7a91720bc756843502c3b7504c77b8fe44217c85c537d85037f0f536151b2caa \
|
--hash=sha256:7a91720bc756843502c3b7504c77b8fe44217c85c537d85037f0f536151b2caa \
|
||||||
--hash=sha256:dab3f4764309081ce75662649be815c4c9081e88f0837825f90fd28317d4da54
|
--hash=sha256:dab3f4764309081ce75662649be815c4c9081e88f0837825f90fd28317d4da54
|
||||||
|
|||||||
@@ -1 +1 @@
|
|||||||
checkov==3.3.1
|
checkov==3.3.16
|
||||||
|
|||||||
+194
-200
@@ -8,127 +8,126 @@ aiohappyeyeballs==2.7.1 \
|
|||||||
--hash=sha256:065665c041c42a5938ed220bdcd7230f22527fbec085e1853d2402c8a3615d9d \
|
--hash=sha256:065665c041c42a5938ed220bdcd7230f22527fbec085e1853d2402c8a3615d9d \
|
||||||
--hash=sha256:9243213661e29250eb41368e5daa826fc017156c3b8a11440826b2e3ed376472
|
--hash=sha256:9243213661e29250eb41368e5daa826fc017156c3b8a11440826b2e3ed376472
|
||||||
# via aiohttp
|
# via aiohttp
|
||||||
aiohttp==3.13.5 \
|
aiohttp==3.14.3 \
|
||||||
--hash=sha256:019a67772e034a0e6b9b17c13d0a8fe56ad9fb150fc724b7f3ffd3724288d9e5 \
|
--hash=sha256:03cd2bde3d7f085b64e549c985f4bb928cad7e8ecf5323bfca320db548d81b39 \
|
||||||
--hash=sha256:02222e7e233295f40e011c1b00e3b0bd451f22cf853a0304c3595633ee47da4b \
|
--hash=sha256:041badb8f84396357c4d3ad26de6afd7a32b112f43d3c63045c0c8278cfd2043 \
|
||||||
--hash=sha256:023ecba036ddd840b0b19bf195bfae970083fd7024ce1ac22e9bba90464620e9 \
|
--hash=sha256:0a5ff2dfbb9ce645fa5b8ef3e02c6c0b9cc3f6030ff863d0c51fffc50cb5541b \
|
||||||
--hash=sha256:02e048037a6501a5ec1f6fc9736135aec6eb8a004ce48838cb951c515f32c80b \
|
--hash=sha256:0fdea2281997af69da84c77ffa6f5938a0285f21fb3887c249d67419ca865b3d \
|
||||||
--hash=sha256:0494a01ca9584eea1e5fbd6d748e61ecff218c51b576ee1999c23db7066417d8 \
|
--hash=sha256:11fb37ef075669eee52ab1928fbf6e1741fada40409fa309ebde9607a962aebf \
|
||||||
--hash=sha256:0f7a18f258d124cd678c5fe072fe4432a4d5232b0657fca7c1847f599233c83a \
|
--hash=sha256:134ac5ddcf61c6fad984b9a5727d83492ada43d63471db20fb73042c13fca62f \
|
||||||
--hash=sha256:10a75acfcf794edf9d8db50e5a7ec5fc818b2a8d3f591ce93bc7b1210df016d2 \
|
--hash=sha256:152516815ef926786a0b6ae2b8f1fd2e0c71582dee0b435636865316fd4891b7 \
|
||||||
--hash=sha256:110e448e02c729bcebb18c60b9214a87ba33bac4a9fa5e9a5f139938b56c6cb1 \
|
--hash=sha256:1576145bdceeb92382d899751e12743a3a5b8e460a841e3e50543859e54864dc \
|
||||||
--hash=sha256:147b4f501d0292077f29d5268c16bb7c864a1f054d7001c4c1812c0421ea1ed0 \
|
--hash=sha256:16100ad3ab8d649fdfbee87602d9d2dcdca9df0b9eda8a1b5fdc0d41f96da559 \
|
||||||
--hash=sha256:157826e2fa245d2ef46c83ea8a5faf77ca19355d278d425c29fda0beb3318037 \
|
--hash=sha256:16ea7e24c309fb7c0bbd505d149abe4fe4dccfb8db911db7dbec0921bc889a6f \
|
||||||
--hash=sha256:15c933ad7920b7d9a20de151efcd05a6e38302cbf0e10c9b2acb9a42210a2416 \
|
--hash=sha256:18c441d0a8fca6de8d1f546849b9f0ab20d435993e2c5b59562b2fae6be2f929 \
|
||||||
--hash=sha256:178c7b5e62b454c2bc790786e6058c3cc968613b4419251b478c153a4aec32b1 \
|
--hash=sha256:18cb43369747b2ae007bd2655fb8e63a099c2ff1d207962943636dac989b3147 \
|
||||||
--hash=sha256:18a2f6c1182c51baa1d28d68fea51513cb2a76612f038853c0ad3c145423d3d9 \
|
--hash=sha256:1b59533861b70a2185c8f4f350f791f39d64358ef6944ce71c5240c9ec0982c9 \
|
||||||
--hash=sha256:1efb06900858bb618ff5cee184ae2de5828896c448403d51fb633f09e109be0a \
|
--hash=sha256:1c5281acc88b92396f88c7e1e2748f8466689df22b80170e4f51efa712fb47a8 \
|
||||||
--hash=sha256:20058e23909b9e65f9da62b396b77dfa95965cbe840f8def6e572538b1d32e36 \
|
--hash=sha256:1c5ec8fb1bcc31a8466f74aaf26c345d5c386fa4bd08a3f0eb9c7a4a3fe8b5bf \
|
||||||
--hash=sha256:206b7b3ef96e4ce211754f0cd003feb28b7d81f0ad26b8d077a5d5161436067f \
|
--hash=sha256:1caa7b0d05f3e3a36f87788c59e970a7ee1cefcfcbb924a9f138c4a6551c9cb7 \
|
||||||
--hash=sha256:20ae0ff08b1f2c8788d6fb85afcb798654ae6ba0b747575f8562de738078457b \
|
--hash=sha256:21c016079415ed3fd676963e9793700a566d85dbbd6bfc564b9b2d209147dcc8 \
|
||||||
--hash=sha256:2294172ce08a82fb7c7273485895de1fa1186cc8294cfeb6aef4af42ad261174 \
|
--hash=sha256:2498f0fe69ead802f9675beca44a7c21c62fdaa4ec5145ea1c3ad6edbee29f85 \
|
||||||
--hash=sha256:241a94f7de7c0c3b616627aaad530fe2cb620084a8b144d3be7b6ecfe95bae3b \
|
--hash=sha256:25bd2708db6bdf6a6630dd37bdcdfcb47c4434d22ac69c64665b802910140b30 \
|
||||||
--hash=sha256:26d2f8546f1dfa75efa50c3488215a903c0168d253b75fba4210f57ab77a0fb8 \
|
--hash=sha256:270d3dace9ca2f10f0da5d8ebe519b7a310fc6112ed916e32df5866df0888553 \
|
||||||
--hash=sha256:2837fb92951564d6339cedae4a7231692aa9f73cbc4fb2e04263b96844e03b4e \
|
--hash=sha256:2e1161602f45a54de2ce0905243a95f58cb42dcd378402f3697f5e0b21e9d2e7 \
|
||||||
--hash=sha256:2994be9f6e51046c4f864598fd9abeb4fba6e88f0b2152422c9666dcd4aea9c6 \
|
--hash=sha256:2e9878ae68e4a5f1c0abe4dd497dbc3d51946f5837b56759e2a02e78fa90ef86 \
|
||||||
--hash=sha256:2d6d44a5b48132053c2f6cd5c8cb14bc67e99a63594e336b0f2af81e94d5530c \
|
--hash=sha256:30402d03a7c0ff52bce290b57e564e9079fd9d0cb545c8aba73f86a103162d2e \
|
||||||
--hash=sha256:31cebae8b26f8a615d2b546fee45d5ffb76852ae6450e2a03f42c9102260d6fe \
|
--hash=sha256:33a2d7c28d33797a2e99923dffa63f83d908a19b6bf26cfe80fa790aa5e1a75a \
|
||||||
--hash=sha256:327cc432fdf1356fb4fbc6fe833ad4e9f6aacb71a8acaa5f1855e4b25910e4a9 \
|
--hash=sha256:362a3fd481769cac1a824514bcd86fda51c65e8fe6e051099e008fddde6db17c \
|
||||||
--hash=sha256:329f292ed14d38a6c4c435e465f48bebb47479fd676a0411936cc371643225cc \
|
--hash=sha256:38901a84da3ce22249f6e860bf8f90d141bcab7da090cc398f8bb58c0e44b7da \
|
||||||
--hash=sha256:330f5da04c987f1d5bdb8ae189137c77139f36bd1cb23779ca1a354a4b027800 \
|
--hash=sha256:39aded8c7f3b935b54aab1d8d73c70ec0ee2d3ec3b943e0e86611bc150ba47f5 \
|
||||||
--hash=sha256:33add2463dde55c4f2d9635c6ab33ce154e5ecf322bd26d09af95c5f81cfa286 \
|
--hash=sha256:3a26434dafe408229ff3403458ca58de24fb51936504decac49ce6755f77e59d \
|
||||||
--hash=sha256:347542f0ea3f95b2a955ee6656461fa1c776e401ac50ebce055a6c38454a0adf \
|
--hash=sha256:3ae5b3a59436d089b5395d910121a390feed4d00578eb95a0fd1a329fe963100 \
|
||||||
--hash=sha256:39380e12bd1f2fdab4285b6e055ad48efbaed5c836433b142ed4f5b9be71036a \
|
--hash=sha256:3d4f72af88ac2474bb5bca640030320e3d38a0163a1d7533500e87be458eef71 \
|
||||||
--hash=sha256:3a807cabd5115fb55af198b98178997a5e0e57dead43eb74a93d9c07d6d4a7dc \
|
--hash=sha256:3f42e9b78301f11c8f861746175d8b9c1ccef713fcad9eab396e2f6db8ed4a22 \
|
||||||
--hash=sha256:3b13560160d07e047a93f23aaa30718606493036253d5430887514715b67c9d9 \
|
--hash=sha256:42a67efc36300d052fb4508a53e8b6901b9284b599ae63945c377569c5fcc1e1 \
|
||||||
--hash=sha256:3df334e39d4c2f899a914f1dba283c1aadc311790733f705182998c6f7cae665 \
|
--hash=sha256:48d67b87db6279c044760787eb01f6413032c2e6f3ba1cafaa492b1c8e578479 \
|
||||||
--hash=sha256:4bb6bf5811620003614076bdc807ef3b5e38244f9d25ca5fe888eaccea2a9832 \
|
--hash=sha256:498c6c623134f8e09a3c4e60bcd607a0b4590dd7dbf08dd40851b27cbb520ccb \
|
||||||
--hash=sha256:4beac52e9fe46d6abf98b0176a88154b742e878fdf209d2248e99fcdf73cd297 \
|
--hash=sha256:49f7325beb0f85ef4aef5f48f490269575f83e6e2acad00a1d80b807eb027062 \
|
||||||
--hash=sha256:4e704c52438f66fdd89588346183d898bb42167cf88f8b7ff1c0f9fc957c348f \
|
--hash=sha256:4e3ac92d90e92773b2362d506068e9a948192bd553e743c5b2429e28527c8661 \
|
||||||
--hash=sha256:4eac02d9af4813ee289cd63a361576da36dba57f5a1ab36377bc2600db0cbb73 \
|
--hash=sha256:530125ee1163c4219af35dc3aa1206e541e7b31b6efc1a3f93b70a136f65d427 \
|
||||||
--hash=sha256:53fc049ed6390d05423ba33103ded7281fe897cf97878f369a527070bd95795b \
|
--hash=sha256:5373dc80ad1aa2fb9ad95c83f24eef418bbda3a61375f128e5b0192e4f3f9b32 \
|
||||||
--hash=sha256:55b3bdd3292283295774ab585160c4004f4f2f203946997f49aac032c84649e9 \
|
--hash=sha256:53e5179d8abb5710f8e83ba207c41c8d1261fcffd4616500e15ca2b7a33be10a \
|
||||||
--hash=sha256:57653eac22c6a4c13eb22ecf4d673d64a12f266e72785ab1c8b8e5940d0e8090 \
|
--hash=sha256:53e7b4ce82b54a8bcc71b3b67a5cbd177ca1d7f592cbc92cd38b7349f73482db \
|
||||||
--hash=sha256:60869c7ac4aaabe7110f26499f3e6e5696eae98144735b12a9c3d9eae2b51a49 \
|
--hash=sha256:543906c127fb1d929b95076db19b83fa2d46751006ff1e23b093aa5ac4d8db42 \
|
||||||
--hash=sha256:636bc362f0c5bbc7372bc3ae49737f9e3030dbce469f0f422c8f38079780363d \
|
--hash=sha256:54cfcdee2770dac994417cbb0ee1f3eb0e7cb6b30c79bf44f2c02ff79ec5124a \
|
||||||
--hash=sha256:676e5651705ad5d8a70aeb8eb6936c436d8ebbd56e63436cb7dd9bb36d2a9a46 \
|
--hash=sha256:55bdcc472aafe2de4a253045cc128007a64f1e0264fb675791e132ea5edaa3bd \
|
||||||
--hash=sha256:69f571de7500e0557801c0b51f4780482c0ec5fe2ac851af5a92cfce1af1cb83 \
|
--hash=sha256:56f355e79f71aef2a85c80305cc915f894b170dba76de5fe84f6351939b83c06 \
|
||||||
--hash=sha256:6a7cbeb06d1070f1d14895eeeed4dac5913b22d7b456f2eb969f11f4b3993796 \
|
--hash=sha256:5895ef58c4620afe02fa16044f023dc4dafec08158f9d08874a46a7dbc0341b8 \
|
||||||
--hash=sha256:6cf81fe010b8c17b09495cbd15c1d35afbc8fb405c0c9cf4738e5ae3af1d65be \
|
--hash=sha256:5bcb6ff3fdab1258a192679ff1a05d44f59626430aa05cd1a9d2447423599228 \
|
||||||
--hash=sha256:6e27ea05d184afac78aabbac667450c75e54e35f62238d44463131bd3f96753d \
|
--hash=sha256:5f08ec777f35ee70720233b8b9811d3bb5d728137f30ac91b7457709c3261ac0 \
|
||||||
--hash=sha256:6f1cbf0c7926d315c3c26c2da41fd2b5d2fe01ac0e157b78caefc51a782196cf \
|
--hash=sha256:614c61d478b83953e261d02bb2df750f17227cd33ef8002945bf5aebbde21919 \
|
||||||
--hash=sha256:6f497a6876aa4b1a102b04996ce4c1170c7040d83faa9387dd921c16e30d5c83 \
|
--hash=sha256:617105e2c3018ee38d0c8ce5ee3c84f621a6d8b9f723202aacaff28449ca91ee \
|
||||||
--hash=sha256:756c3c304d394977519824449600adaf2be0ccee76d206ee339c5e76b70ded25 \
|
--hash=sha256:6debfa7312ff9d4c124dc71d72e9a0a4b9e0879e48ba6fcb42bef5c3300289e2 \
|
||||||
--hash=sha256:77dfa48c9f8013271011e51c00f8ada19851f013cde2c48fca1ba5e0caf5bb06 \
|
--hash=sha256:7041d52c3a7fa20c9e8c182b534704abb19502c8bdcbde7ab23bfda6f642394f \
|
||||||
--hash=sha256:7996023b2ed59489ae4762256c8516df9820f751cf2c5da8ed2fb20ee50abab3 \
|
--hash=sha256:70c987b27534f9ae1a723f47ae921571d616da21d3208282bf4c52af5164ac43 \
|
||||||
--hash=sha256:7ab7229b6f9b5c1ba4910d6c41a9eb11f543eadb3f384df1b4c293f4e73d44d6 \
|
--hash=sha256:74ab5b6a9fb13e873e5a90946588baecaf488745e1db1a4a5c433f971f035098 \
|
||||||
--hash=sha256:7becdf835feff2f4f335d7477f121af787e3504b48b449ff737afb35869ba7bb \
|
--hash=sha256:78253b573e6ffab5028924fc98bc281aae05445969982a10864bc360dea2016c \
|
||||||
--hash=sha256:7c35b0bf0b48a70b4cb4fc5d7bed9b932532728e124874355de1a0af8ec4bc88 \
|
--hash=sha256:7a75aa63cbf9b21cfaf60dc2657e19df2c2867d91707d653fee171ffeedd1371 \
|
||||||
--hash=sha256:7c4b6668b2b2b9027f209ddf647f2a4407784b5d88b8be4efcc72036f365baf9 \
|
--hash=sha256:8800c996b01c2772a783e3e46f3e1abd5823029adca0df54231960de9bfefa5b \
|
||||||
--hash=sha256:7e5dc4311bd5ac493886c63cbf76ab579dbe4641268e7c74e48e774c74b6f2be \
|
--hash=sha256:89176250f686cb9853c0fb7ead90e639e915b84a6f43eedc2a4e7ec21f1037f0 \
|
||||||
--hash=sha256:888e78eb5ca55a615d285c3c09a7a91b42e9dd6fc699b166ebd5dee87c9ccf14 \
|
--hash=sha256:8a5fd34f7f7410d1730d5c2ba873cacb2eed3fede366feb268a70ba22581ed8f \
|
||||||
--hash=sha256:898703aa2667e3c5ca4c54ca36cd73f58b7a38ef87a5606414799ebce4d3fd3a \
|
--hash=sha256:8b3b60de05f3dcb6f6a00f818bb2ec781cee4de0645f59ccaf99b1d1823b6100 \
|
||||||
--hash=sha256:8b14eb3262fad0dc2f89c1a43b13727e709504972186ff6a99a3ecaa77102b6c \
|
--hash=sha256:8f2f1c4c032c7cedd7d8da6f54c97b70266c6570c3108d3fdffee7188bb70529 \
|
||||||
--hash=sha256:8bd3ec6376e68a41f9f95f5ed170e2fcf22d4eb27a1f8cb361d0508f6e0557f3 \
|
--hash=sha256:9491196535a88924a60afd5b5f434b5b203b6cc616250878dbdb223a8f7844bc \
|
||||||
--hash=sha256:8cf20a8d6868cb15a73cab329ffc07291ba8c22b1b88176026106ae39aa6df0f \
|
--hash=sha256:9aa6e61fdf20105c4144e755bd586008ff450791d67b1c8146fdc15959c4d51c \
|
||||||
--hash=sha256:8f14c50708bb156b3a3ca7230b3d820199d56a48e3af76fa21c2d6087190fe3d \
|
--hash=sha256:9d9edccfe496b476db5f398d97b865e9a6752bcf8aec4eef8390ce20fb64bb41 \
|
||||||
--hash=sha256:8f546a4dc1e6a5edbb9fd1fd6ad18134550e096a5a43f4ad74acfbd834fc6670 \
|
--hash=sha256:9fc7b5bfec6573f3ae844f457fdde5adeb713f8b8e4a81ad64fc207b49383716 \
|
||||||
--hash=sha256:912d4b6af530ddb1338a66229dac3a25ff11d4448be3ec3d6340583995f56031 \
|
--hash=sha256:a0dc483c00da8b673abbb367eb6f8d8f4bcec30eb58529ea13cb42e7fd2dfa33 \
|
||||||
--hash=sha256:9277145d36a01653863899c665243871434694bcc3431922c3b35c978061bdb8 \
|
--hash=sha256:a3a8296e7ab5c295f53f1041487cb088e1480775aafbf7fe545d93b770a0f96f \
|
||||||
--hash=sha256:95d14ca7abefde230f7639ec136ade282655431fd5db03c343b19dda72dd1643 \
|
--hash=sha256:a3e22975f905b89a55a488c2a08f2fdb2186175349e917d48985cc468a3d4c6e \
|
||||||
--hash=sha256:999802d5fa0389f58decd24b537c54aa63c01c3219ce17d1214cbda3c2b22d2d \
|
--hash=sha256:a4af35c443e0b1a1bd6a8af3f3485d7fda15c142751a00f3ff8090f0b93346fa \
|
||||||
--hash=sha256:9a0f4474b6ea6818b41f82172d799e4b3d29e22c2c520ce4357856fced9af2f8 \
|
--hash=sha256:a94dbaae5ae27bd849c93570669bff91e0510f33a80805738e3de72a7be0447b \
|
||||||
--hash=sha256:9b16c653d38eb1a611cc898c41e76859ca27f119d25b53c12875fd0474ae31a8 \
|
--hash=sha256:ac74facc01463f138b0da5580329cfcc82818dea5656e83ddcd11268fc12ff80 \
|
||||||
--hash=sha256:9d98cc980ecc96be6eb4c1994ce35d28d8b1f5e5208a23b421187d1209dbb7d1 \
|
--hash=sha256:ad4c8b7488d745d2ca4838ebd8ae5ba9b56341d30b1da43640e4ce87f9f49646 \
|
||||||
--hash=sha256:9efcc0f11d850cefcafdd9275b9576ad3bfb539bed96807663b32ad99c4d4b88 \
|
--hash=sha256:b014a6ed7cf912e787149fdc529166d3ceabac23f26efeea3158c9aba2354e7e \
|
||||||
--hash=sha256:a2567b72e1ffc3ab25510db43f355b29eeada56c0a622e58dcdb19530eb0a3cb \
|
--hash=sha256:b20032766aedf6261c7a566585a40867d092ac03a0d81592d5370ef9b054f99b \
|
||||||
--hash=sha256:a5029cc80718bbd545123cd8fe5d15025eccaaaace5d0eeec6bd556ad6163d61 \
|
--hash=sha256:b2466434105a4e03113c36ec775cc2ebe6676b62eae326fa670bb607ef788c1c \
|
||||||
--hash=sha256:a60eaa2d440cd4707696b52e40ed3e2b0f73f65be07fd0ef23b6b539c9c0b0b4 \
|
--hash=sha256:b304db572b4368edd8dda8a2274f73156fe15558fca4a917cb8a09fc47af5963 \
|
||||||
--hash=sha256:a79a6d399cef33a11b6f004c67bb07741d91f2be01b8d712d52c75711b1e07c7 \
|
--hash=sha256:ba59d59aba08ac02fc03b0c8983ccd5ee39a199d0552ce9e6d2b4845b34d59ae \
|
||||||
--hash=sha256:a84792f8631bf5a94e52d9cc881c0b824ab42717165a5579c760b830d9392ac9 \
|
--hash=sha256:bd52f811e65f6fb634b1047159657c98f52b407f8efec907bcfc09da9a4c0a25 \
|
||||||
--hash=sha256:a8a4d3427e8de1312ddf309cc482186466c79895b3a139fed3259fc01dfa9a5b \
|
--hash=sha256:bdd0e2834dce1a26c1bbe26464861e16bbe217042cbff619247c11594472518c \
|
||||||
--hash=sha256:a8aca50daa9493e9e13c0f566201a9006f080e7c50e5e90d0b06f53146a54500 \
|
--hash=sha256:c23ec8ee9d5ab2f5421f9c7fffce208435607af27fd46d4a44e031954352838f \
|
||||||
--hash=sha256:aa6d0d932e0f39c02b80744273cd5c388a2d9bc07760a03164f229c8e02662f6 \
|
--hash=sha256:c39846c3aad97a8530c89d7a3869a8f8e9e3762c6ac0504481e5c80948f7e807 \
|
||||||
--hash=sha256:ab2899f9fa2f9f741896ebb6fa07c4c883bfa5c7f2ddd8cf2aafa86fa981b2d2 \
|
--hash=sha256:c3c200cf9757edd785051dc699c7ecbec22110dbfcb3fefc7a9f9695eda8ea7a \
|
||||||
--hash=sha256:af545c2cffdb0967a96b6249e6f5f7b0d92cdfd267f9d5238d5b9ca63e8edb10 \
|
--hash=sha256:c7d3a97c678d34fc5b59da671ee9cd630096ddc643e7b5a30d54a2a6f3574d3f \
|
||||||
--hash=sha256:b18f31b80d5a33661e08c89e202edabf1986e9b49c42b4504371daeaa11b47c1 \
|
--hash=sha256:c8653fd547c93a61aadc612007790f5555cdd18946fa48cf45e26d8ea4ea473d \
|
||||||
--hash=sha256:b20df693de16f42b2472a9c485e1c948ee55524786a0a34345511afdd22246f3 \
|
--hash=sha256:cc7cb243a68167172f48c1fd43cee91ec4b1d40cefd190edd43369d1a6bc9c82 \
|
||||||
--hash=sha256:b38765950832f7d728297689ad78f5f2cf79ff82487131c4d26fe6ceecdc5f8e \
|
--hash=sha256:ccd4893707b3e2a13e39c90d43cf80edf2e4d0457935bcc103bf2346214c3f15 \
|
||||||
--hash=sha256:b6f6cd1560c5fa427e3b6074bb24d2c64e225afbb7165008903bd42e4e33e28a \
|
--hash=sha256:cd817772b2fcf2b8c0905795318485f9ec16eae60b29feb7f4c77085311637f0 \
|
||||||
--hash=sha256:bace460460ed20614fa6bc8cb09966c0b8517b8c58ad8046828c6078d25333b5 \
|
--hash=sha256:cda5fd5c95ad7a125a2e8464acc78b98b94c475a3780d6aa0aa157c93f470f4d \
|
||||||
--hash=sha256:bca9ef7517fd7874a1a08970ae88f497bf5c984610caa0bf40bd7e8450852b95 \
|
--hash=sha256:cef89a58e628c4efcac3275c2d68083f82426dcdc89c1492a6f654f9f7ea6ab9 \
|
||||||
--hash=sha256:c180f480207a9b2475f2b8d8bd7204e47aec952d084b2a2be58a782ffcf96074 \
|
--hash=sha256:d1558173930a5a8d3069cee5c92fc91c87c4dbcb099debbb3622053717145a19 \
|
||||||
--hash=sha256:c2b2355dc094e5f7d45a7bb262fe7207aa0460b37a0d87027dcf21b5d890e7d5 \
|
--hash=sha256:d6088ec9894113802bddb3c09e974929aed2c7b3a8c456219b8aab4481f1a239 \
|
||||||
--hash=sha256:c564dd5f09ddc9d8f2c2d0a301cd30a79a2cc1b46dd1a73bef8f0038863d016b \
|
--hash=sha256:d6218d92e450824e9b4881f44e8c09f1853b490f9a64130801024a4793b1b3b0 \
|
||||||
--hash=sha256:c632ce9c0b534fbe25b52c974515ed674937c5b99f549a92127c85f771a78772 \
|
--hash=sha256:d77640cc618c1d99fc4f8589c0f24a730adfa54eb1e57ef7bf0c8dfb78da898c \
|
||||||
--hash=sha256:c719f65bebcdf6716f10e9eff80d27567f7892d8988c06de12bbbd39307c6e3a \
|
--hash=sha256:d7d2deec16eeedf55f2c7cf75b521ea3856a5177e123844f8fd0f114ce252cb5 \
|
||||||
--hash=sha256:c86969d012e51b8e415a8c6ce96f7857d6a87d6207303ab02d5d11ef0cad2274 \
|
--hash=sha256:db332af25642007330fca8be5c4d194caf2bea7a7fc84415aff3497af5dfee6b \
|
||||||
--hash=sha256:c974fb66180e58709b6fc402846f13791240d180b74de81d23913abe48e96d94 \
|
--hash=sha256:dd54d0e8717de95939766febac482ac0474d8ac3b048115f9f2b1d23a16e7db4 \
|
||||||
--hash=sha256:c9883051c6972f58bfc4ebb2116345ee2aa151178e99c3f2b2bbe2af712abd13 \
|
--hash=sha256:ddcac3c6b382e81f1dd0499199d4136b877beb4cb5ef770bbbfba56c4b8f55d2 \
|
||||||
--hash=sha256:ca9ac61ac6db4eb6c2a0cd1d0f7e1357647b638ccc92f7e9d8d133e71ed3c6ac \
|
--hash=sha256:df82f3787c940c94986b34222d59c9e38843fba85139f36e85255a82ad5355a9 \
|
||||||
--hash=sha256:cb979826071c0986a5f08333a36104153478ce6018c58cba7f9caddaf63d5d67 \
|
--hash=sha256:dfa68deb2a443bdaa3ea5297b0699c1464f08aef3812b486d1348eee61b07dc0 \
|
||||||
--hash=sha256:cd3db5927bf9167d5a6157ddb2f036f6b6b0ad001ac82355d43e97a4bde76d76 \
|
--hash=sha256:dff9461ec275f22135650d5ba4b4931a11f3958df7dfbb8db630000d4dee0883 \
|
||||||
--hash=sha256:d147004fede1b12f6013a6dbb2a26a986a671a03c6ea740ddc76500e5f1c399f \
|
--hash=sha256:e1e74298bab6ee0d6e749ed4fd1901c7e604bdda32c03d787a2cc71c46d0433d \
|
||||||
--hash=sha256:d3a4834f221061624b8887090637db9ad4f61752001eae37d56c52fddade2dc8 \
|
--hash=sha256:e2667f0bbe7eb6c74eae5e9691441ad186e5845ca3cff63230fc09c4e7514f5d \
|
||||||
--hash=sha256:d9010032a0b9710f58012a1e9c222528763d860ba2ee1422c03473eab47703e7 \
|
--hash=sha256:e3be98a7c30b8c25d573dafba7171d66dfb05ee6a9070fc46535464ff97700a6 \
|
||||||
--hash=sha256:d97f93fdae594d886c5a866636397e2bcab146fd7a132fd6bb9ce182224452f8 \
|
--hash=sha256:e568e14940c09955aa51f4e645b6daa18a581c5dcfcd73744dcc86a856e3ced3 \
|
||||||
--hash=sha256:df23d57718f24badef8656c49743e11a89fd6f5358fa8a7b96e728fda2abf7d3 \
|
--hash=sha256:e72ee89e28d907a18f46959b4eb0bb06701cc7f8cf4366e00029e2ccfaaf5924 \
|
||||||
--hash=sha256:df6104c009713d3a89621096f3e3e88cc323fd269dbd7c20afe18535094320be \
|
--hash=sha256:e92eb8acc45eb6a9f4935071a77edf5b85cc6f8dfad5cd99e97653c26593cdde \
|
||||||
--hash=sha256:e5e5f7debc7a57af53fdf5c5009f9391d9f4c12867049d509bf7bb164a6e295b \
|
--hash=sha256:ea05e1f97ceea523942d9b2a7d7c0359d781d683d6b043f5943a602b14da4787 \
|
||||||
--hash=sha256:e7d2f8616f0ff60bd332022279011776c3ac0faa0f1b463f7bb12326fbc97a1c \
|
--hash=sha256:eac645b09bcfdf73df7536331f0678c1086ea250981118ddb5199e17ccef72bb \
|
||||||
--hash=sha256:e999f0c88a458c836d5fb521814e92ed2172c649200336a6df514987c1488258 \
|
--hash=sha256:eb0495d778817619273c108784292be161a924b9f5ae5cbbc70a2caa6838250b \
|
||||||
--hash=sha256:eb4639f32fd4a9904ab8fb45bf3383ba71137f3d9d4ba25b3b3f3109977c5b8c \
|
--hash=sha256:ebe8e504f058fe91223351cecd2d9d6946c9d241bb0250d898ffbdf584cc72b0 \
|
||||||
--hash=sha256:ec707059ee75732b1ba130ed5f9580fe10ff75180c812bc267ded039db5128c6 \
|
--hash=sha256:ed099d105449c4f9e84f24af203cd131349d4761d8813fa7e02c32e7128cd910 \
|
||||||
--hash=sha256:ecc26751323224cf8186efcf7fbcbc30f4e1d8c7970659daf25ad995e4032a56 \
|
--hash=sha256:f0f177d1b195b9e06376cfd7d308d8a1b920909a609d03ac82a8c73bbb16d3b9 \
|
||||||
--hash=sha256:ee5e86776273de1795947d17bddd6bb19e0365fd2af4289c0d2c5454b6b1d36b \
|
--hash=sha256:f3d2669fe7dec7fc359ecdb5984b29b50d85d5d00f8c1cb61de4f4a24ee42627 \
|
||||||
--hash=sha256:f1162a1492032c82f14271e831c8f4b49f2b6078f4f5fc74de2c912fa225d51d \
|
--hash=sha256:f4e05329faa0ea1a404b37de4f034fd2c2defcca06a68dc6745e4e56c88e8a48 \
|
||||||
--hash=sha256:f34ecee82858e41dd217734f0c41a532bd066bcaab636ad830f03a30b2a96f2a \
|
--hash=sha256:f53bcd52f585e1ac3e590d61434eb61f9a88c38df041b4ea126d97144344a77b \
|
||||||
--hash=sha256:f85c6f327bf0b8c29da7d93b1cabb6363fb5e4e160a32fa241ed2dce21b73162 \
|
--hash=sha256:f55119f7bf25f49ed210f6096090715da24f2943c62102448915fde3c62877ce \
|
||||||
--hash=sha256:f92995dfec9420bb69ae629abf422e516923ba79ba4403bc750d94fb4a6c68c1 \
|
--hash=sha256:f631fe87a6f30df5fbe6d79640b25e4cffb38c31c7fb6f10871517b84b0f8c1a \
|
||||||
--hash=sha256:fb0540c854ac9c0c5ad495908fdfd3e332d553ec731698c0e29b1877ba0d2ec6 \
|
--hash=sha256:f8fb78a83c9e5f741ca3a68cfb455c1f5bb83b4e7249a3848b3cd78d0a8563b0 \
|
||||||
--hash=sha256:fceedde51fbd67ee2bcc8c0b33d0126cc8b51ef3bbde2f86662bd6d5a6f10ec5 \
|
--hash=sha256:fa9467a8113aa69d3d7c55a70ef0b7c636010a40993f3df9d9d0d73b3eb7ef24 \
|
||||||
--hash=sha256:fe6970addfea9e5e081401bcbadf865d2b6da045472f58af08427e108d618540 \
|
--hash=sha256:fd51ebf9d3a00c074df4ede271023f4d2dba289bcc740b88191872716014e3c5
|
||||||
--hash=sha256:fee86b7c4bd29bdaf0d53d14739b08a106fdda809ca5fe032a15f52fae5fe254
|
|
||||||
# via checkov
|
# via checkov
|
||||||
aiomultiprocess==0.9.1 \
|
aiomultiprocess==0.9.1 \
|
||||||
--hash=sha256:3a7b3bb3c38dbfb4d9d1194ece5934b6d32cf0280e8edbe64a7d215bba1322c6 \
|
--hash=sha256:3a7b3bb3c38dbfb4d9d1194ece5934b6d32cf0280e8edbe64a7d215bba1322c6 \
|
||||||
@@ -157,9 +156,9 @@ attrs==26.1.0 \
|
|||||||
# aiohttp
|
# aiohttp
|
||||||
# jsonschema
|
# jsonschema
|
||||||
# referencing
|
# referencing
|
||||||
bc-detect-secrets==1.5.47 \
|
bc-detect-secrets==1.5.50 \
|
||||||
--hash=sha256:46f88c710b0fd8c5f2e54b361d793b5e1469197884da73cfc6f488b614366fc3 \
|
--hash=sha256:016ce9e79f692adbabcbef4a7293db427352911c3f81404a136e6f5c3e54a7f2 \
|
||||||
--hash=sha256:a9be28a2e564f2b19731991df39e63ae6372cc84d828ee24e50c094cbb4c154c
|
--hash=sha256:99037375d9cb49ed07e5bb12722f4bbb76fb8acaff6f367eef9df7559e7642b3
|
||||||
# via checkov
|
# via checkov
|
||||||
bc-jsonpath-ng==1.6.1 \
|
bc-jsonpath-ng==1.6.1 \
|
||||||
--hash=sha256:2c85bb1d194376808fe1fc49558dd484e39024b15c719995e22de811e6ba4dc8 \
|
--hash=sha256:2c85bb1d194376808fe1fc49558dd484e39024b15c719995e22de811e6ba4dc8 \
|
||||||
@@ -484,9 +483,9 @@ charset-normalizer==3.5.1 \
|
|||||||
# via
|
# via
|
||||||
# checkov
|
# checkov
|
||||||
# requests
|
# requests
|
||||||
checkov==3.3.1 \
|
checkov==3.3.16 \
|
||||||
--hash=sha256:1e781a58de8310ec99756205a7991adcfe66524a52642c6474e9d86a7cc9c635 \
|
--hash=sha256:43e5383418a8b52d39747e2daaec4da4a6b7db3e2f64ab6c93f8b89c044c7665 \
|
||||||
--hash=sha256:aafc571cc937ddaa0714df30f2b9d79302a07cb9a41b3e0168c7eecb8172db14
|
--hash=sha256:6f7f611f45c765af9b6acd43e603a438153d86007d02dffa7903a1125f9b5089
|
||||||
# via -r .github/requirements/checkov.in
|
# via -r .github/requirements/checkov.in
|
||||||
click==8.5.0 \
|
click==8.5.0 \
|
||||||
--hash=sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360 \
|
--hash=sha256:255bc9599cf7748b4b1a446ccc735421bd08a2ae529a8b88597d3de5664ee360 \
|
||||||
@@ -984,79 +983,73 @@ networkx==2.6.3 \
|
|||||||
--hash=sha256:80b6b89c77d1dfb64a4c7854981b60aeea6360ac02c6d4e4913319e0a313abef \
|
--hash=sha256:80b6b89c77d1dfb64a4c7854981b60aeea6360ac02c6d4e4913319e0a313abef \
|
||||||
--hash=sha256:c0946ed31d71f1b732b5aaa6da5a0388a345019af232ce2f49c766e2d6795c51
|
--hash=sha256:c0946ed31d71f1b732b5aaa6da5a0388a345019af232ce2f49c766e2d6795c51
|
||||||
# via checkov
|
# via checkov
|
||||||
numpy==2.4.6 \
|
numpy==2.5.2 \
|
||||||
--hash=sha256:001fbb8e08d942dd57599e781f2472269ee7f2755fae407b4f67b2f0b17da3f1 \
|
--hash=sha256:0090ccdd57ec2703e9b49d0bf554767370581c1dd0a6b2bb2b2d9def317d042a \
|
||||||
--hash=sha256:0280e0356c0829a18d9de1cb7eee50ec22ca639878d7240307ca0943d73cd2c4 \
|
--hash=sha256:078f9b027b478c9379b9677babbf0f8b8f1ecfada27636d7b9a93990c638739f \
|
||||||
--hash=sha256:043191bfa8eab18c776647b62723ac9dddece59743b13f49b2016094129c2b3f \
|
--hash=sha256:07d4e89f3a9ab0a9ba24264ccdb642b3dd951b2281e8883a5481a4aa79cc31a7 \
|
||||||
--hash=sha256:06ca2f61ec4385a07a6977c55ba998a4466c123642b4a32694d3128fce18c079 \
|
--hash=sha256:0a4035ae1129ff8777f08bfbd44f1e5d8e9c049ce0c2dd78fc0d92c13e7251c0 \
|
||||||
--hash=sha256:0a041d3d761dc3c35cc56ce0351506a02bcbc25f7b169f652435141a17db9096 \
|
--hash=sha256:0aadf13b60048d501e05fa699efaf7734e2494f3498a4c2a5521d822640324f3 \
|
||||||
--hash=sha256:0ab0a9c4ffb1a6d95ef519fe4247dba8eb6b18ad93999f76b7f657039acabd47 \
|
--hash=sha256:14e373cfc6387177e8409dac3c7159be8eb05cd77096cd7c950268b86f62831c \
|
||||||
--hash=sha256:0c9136e14ed34a9e343a31c533d78a9813a69a3148332bce5e9821cb2f996e66 \
|
--hash=sha256:1ab3d4a901f844ea836c3e80bf463c6a27d7f3c14e8e292fcf28d348b25b9bce \
|
||||||
--hash=sha256:110f8b71aacb688ec69062bb7f6938a0f8acb01b7c1c4beb453c65b6d234584d \
|
--hash=sha256:24b9dc2e3d84aa58523798805194e23e736f3f6ce2d1a5b92583ae734e6dbda8 \
|
||||||
--hash=sha256:112b06a867b235ef466ed3508ddf0238050df9c727cafb5301ac385b899189a1 \
|
--hash=sha256:27650bb0e7140fa3d37b9923b4803645e0b125d190f326eecfd3f4dad8e8ade1 \
|
||||||
--hash=sha256:17f9ade344e7d9b464a084d69bcf18fc691cb1db67c62ed80820bf4926d78f0e \
|
--hash=sha256:28ac63476ec7651484215ee7fa15a1f78b57c14621f01e392afe17b9a1390ce4 \
|
||||||
--hash=sha256:1e254a00cdf42b1e4d5b3d68d33af63268d41340d8885df2ab6470f2e1500147 \
|
--hash=sha256:29b86ff8a6cc556b47ec6b64b194815cc80e6bf5eedcc6cddfd65318cb0b4eee \
|
||||||
--hash=sha256:1e978ec1e8bd0e0e4de6bb75de9d30cbb74db6b6a2bb727618613703ca0167dd \
|
--hash=sha256:29d81e97f668489cba8ebfd796b9bdd453525d35dd9e162e2daec94bf3fc7740 \
|
||||||
--hash=sha256:25c692919ac5a01f170a3bfcd62d745b24fd095c353d50812637d6fcab442e75 \
|
--hash=sha256:2cc779226e476d1e1f08c74068c419e60f41a9e0e069c92f6671d31d5c985e98 \
|
||||||
--hash=sha256:260a5d70215b61ab4fadf5c7baacd64821842975eea312125ed3c39a6391b063 \
|
--hash=sha256:2ffa7bacab3e2ee1b19ed31766bb60bb380b68c23f051e199c5cc598afd68710 \
|
||||||
--hash=sha256:2803abfebfc990042cd494d8ce2d5f82e9d847af6d35ec486923aa19dbad5e73 \
|
--hash=sha256:318b9a4c845dbea06708a29c84ee429cc3065048db34cdb799047643492050ee \
|
||||||
--hash=sha256:29a287e0cf63ff528da061de6b9f64a4618da591ca1046aafc54062e40ca7eab \
|
--hash=sha256:34c319e2963be042673fb46570501b2f06c41924e17e3563d58646b4380dfb68 \
|
||||||
--hash=sha256:29cb7f67d10b479ff07c17d33e39f78c07f71c40ef30d63c153d340e96cd3fb4 \
|
--hash=sha256:3a2f061cebd9e3d23bdcfaaded5e2293a4c6a5b60fa42df85d410a725ce621bf \
|
||||||
--hash=sha256:3213d622a0283a39a93d188f3cf72b26862df52fbb4ca3697f51705016523d41 \
|
--hash=sha256:3cdec01fa790a186d430433fdd4d4ffb70eed6f0eeb4bf05c8dbe2dce0a9bcb8 \
|
||||||
--hash=sha256:33111801a01c12a8a1e3721f0a9232f8cfc8ae2c6b7098167e6f623c6073f402 \
|
--hash=sha256:3e4c367352d3747784248a227fbec218e193b56f7e6692e3b64fc805478ecfdf \
|
||||||
--hash=sha256:357cc07a6d7b0b182ff02249616a03742827ebb1277546b5c7cd7f7620a45698 \
|
--hash=sha256:40f4d451aed46a8046a1aae41c4e55fb3612273df9c502480135e1501576a34b \
|
||||||
--hash=sha256:38efbc8de75c7a0fc1ac190162d892787f3f47b57cc291231aafee36b80982b7 \
|
--hash=sha256:44ef9675d908e65f9953063837c3277730f3f4437615a4cdab67b366cabaf884 \
|
||||||
--hash=sha256:4081eb135ac24158bd51cdfbef16f1c64df7063b1143f24731387137c092bec8 \
|
--hash=sha256:4bbd96c833ecc8cc069ce518078fc8c60cb9cbfb0fea5b7a803ad65035596d03 \
|
||||||
--hash=sha256:40fdc1ae7125e518ea98e53e69a4ebc27e1fd50510c47b7ea130cf21e5e1d42b \
|
--hash=sha256:4ec954036759bcee3aa484f8603bd9c14f3e776293b85578b8734c2d72777c69 \
|
||||||
--hash=sha256:4cfe66903cc32a9921a6733d96b19bb6abf310397581bbad89c228f5abaf0ee8 \
|
--hash=sha256:4f9744f9fbdcea0bc552e8f19e1f141f811a3f9bc2be2cc6e86d982cab23e3f4 \
|
||||||
--hash=sha256:511dbaf848decaaaf4b4ca48032619fb3138710c4bf7da7617765edad1ef96b0 \
|
--hash=sha256:50a68f4bacd8a2b33d8da3d2269d0d78500f86ea582e4786dc10f5ef2c2c6842 \
|
||||||
--hash=sha256:55cced7c52e981362f708ad635198e97a752dfba412cc03c23bbf3bd8d5cd662 \
|
--hash=sha256:50e500dc868e9313530ce12ba470fe50ff3afe3d62993ed6eff652dacd555b65 \
|
||||||
--hash=sha256:56b39e5e0622a09a25bf5baf62f4bcf0cb8a41ae6e2819cf49bbc5a74c083f91 \
|
--hash=sha256:52c808f96484f5571a5cc863775ce50247c17dfb3b0361f8ed6b4b0456f80080 \
|
||||||
--hash=sha256:5dbbdb29840ca3d91ee0fece42fc29278886d908280bfec0a5846c6f901a3eb0 \
|
--hash=sha256:5f8e00be2ec6f45f4e8a41a527f68d44a7d96fee92a650e4d8b1326f77f61e6e \
|
||||||
--hash=sha256:5f9fb9157b4ce2971008323afe46053787b526ef624fea915b261468a8421a0f \
|
--hash=sha256:60e902ac295855348a5ca2ea4c89108989a9f5fddfad3dfc0a8f36b10358567e \
|
||||||
--hash=sha256:6180d8b35af935aed8ece3a85e0a43f87393ae0ac87c8d2c8bd2c993f7270ef3 \
|
--hash=sha256:65f188481f1669e26f62b701e8205d19e460fa4a9b52a1414ba382330e4a3414 \
|
||||||
--hash=sha256:68a5124b13fa6cc2086764a20005d30bc0548146f7f5322f02fce212ca14317f \
|
--hash=sha256:6950c4b7dd562453090548ba7f5da7e59f57f85663f15d5dcc60e249192f7e59 \
|
||||||
--hash=sha256:68bb27509ac1b9a3443094260f6326150663b06abe40b73a2f81160623da5b67 \
|
--hash=sha256:6a9bb119fb8dd21ba30b3f0e555b7e2b081bd9883af21ec9c1c633d161cda3a8 \
|
||||||
--hash=sha256:6f41ae150c4e32db4f3310cdaf64b1593a03dbabe29eec77fc9b50fe64061df6 \
|
--hash=sha256:6b588cc8f902d6bff201c19fd00c43ab8545671e3554d014e12e14139e5e8617 \
|
||||||
--hash=sha256:7265a2f3d436e54ef9f2b52b5c937e6be778781bd97a590319d7348f1c1ca997 \
|
--hash=sha256:6df895598c0edcb41030126c89e0f353b07d93238116143b7405e937359736c4 \
|
||||||
--hash=sha256:72fbe16c6fac95aedf5937fa873445cec2110be35d8a4e9433d7501fd98dae6b \
|
--hash=sha256:6e8172ddfcf5cf74b811d372b570b83c60bd2de87a6fbfbebdadb4a9bd9c6cbb \
|
||||||
--hash=sha256:7d92c3819208a60205a12a245c91ad70cb0a85336659b19b834205573ac8456e \
|
--hash=sha256:7354826bc6f8f69402e9b7fe28d15fcd34feebd74f856f111585c5b0c9fb0251 \
|
||||||
--hash=sha256:8155154c7c691289fe18f510b5d4657c68c67989f293f0535a91360392ff6538 \
|
--hash=sha256:7587f53dfbd5edc0f7b87c6217b4c6d2d1f2ef9c3da70bc1315e7db5f8d7ec9d \
|
||||||
--hash=sha256:81a1cca95ed5bb92aa8b10dd2cdc9a0d3853a50fad926c28b5d7e8ea54389627 \
|
--hash=sha256:77843ca236b777e67f8d6b3660ea116e499612703a0ecd7093f316201eb9d8e2 \
|
||||||
--hash=sha256:89cd468399cfd2504718f0ba50e410dca55a170b61a02ad92bb18c8a65186e93 \
|
--hash=sha256:7999d4ddb0c4025018373fd787510d46e04c769467af22869707b3c1cfd459ab \
|
||||||
--hash=sha256:8ad03c0965fb3c692200e74d458ca28c1dbb4ce96f9a479a8aa041ad5fabca02 \
|
--hash=sha256:85aaccb24182c25df891ad0ec333585967e115269d5f1b17f2c9ae005bc96657 \
|
||||||
--hash=sha256:90f9849678c75fe7afa2d348ac842c168b0a4d3d61919687216dfc547976d853 \
|
--hash=sha256:8e4cb9a754c8a0c62eaa88273a5fba3391f4a610d1dee893c0755da31c083f15 \
|
||||||
--hash=sha256:948424b06129ce883307e8cff868c31396d8dc7630a59c61d70d98dbe70f222c \
|
--hash=sha256:8ee9c4eeb8454b3660a8b53493563c3e121c2fc94fbd72b848ef814ed7b676a9 \
|
||||||
--hash=sha256:9cd5ffd25db4e7ba6a375693b3fc0fc1791ec636c17db3720da19bde7180ec43 \
|
--hash=sha256:9a0731745a72a184490a582fb4af2533512bd071ace67785b5fdffc0ae58dce8 \
|
||||||
--hash=sha256:a0df0043bdb289bde1f62da130d20df23d58b45429f752bc7a8fc5325a225ecd \
|
--hash=sha256:9e9413326d726c2545bfa65d2c0876871e8d8386e77f992c1d426e180bbd4323 \
|
||||||
--hash=sha256:a2c306dea656c12c68f51f4cea133cbe78ca7435eb28c735eac1d3ebe73be6e8 \
|
--hash=sha256:a610dc7e3c52edd39c2bc2375ff9c3fd59cb3ad00e4472d36f83bc1457145788 \
|
||||||
--hash=sha256:a7830bab239b79cda9c08c2da014761cafb48da6150e1da17ac06283f43b6089 \
|
--hash=sha256:a839318485284a6fb31be4f8f2c91c8f2cb22f4543c4a8903f12b0671ffe07cc \
|
||||||
--hash=sha256:a7c711e21628b52034bb5ab8d1bce291f752fcc5e92accc615778acee1ff4778 \
|
--hash=sha256:afb3f0632d6b2e3ba04dbce8d1e48d321b369138b73830b5ca371a0e8d479d56 \
|
||||||
--hash=sha256:aaf159caa35993cb1f56fb9b8e4610d35758e7ca005412eb1daa856a78c9c4b1 \
|
--hash=sha256:b879fb674276e331513fb136b78dbc6bd3c848309e0d841cfd63be3896c4cfc1 \
|
||||||
--hash=sha256:ae506e6902902557576a26ff33eda8695e7ecb3cb36c3b573a0765dee114ebdb \
|
--hash=sha256:b9727f472d2f3888053b8a75ab0cb94745a9de224bb5846dbadc0092101bc71d \
|
||||||
--hash=sha256:b507f5c4c1d508876d1819b6bf9a49d365b96320b5d4993426b33a23ca4b8261 \
|
--hash=sha256:ba0a474801b8dc67b66bf465548abc90e82b44d2611b5770f33008dcabffe8ec \
|
||||||
--hash=sha256:bf162abab1c1a736333192707cef898e735a5ca00f38f27eeedf44b39d9e85eb \
|
--hash=sha256:bd68ece1553d2023c09a4226d9e41c586ad2d20594d1a456186c33513d2cb3f2 \
|
||||||
--hash=sha256:c1a2af6c6ef86344a6b0db6b97834208bf598db514f2b155042439b62605601a \
|
--hash=sha256:c081cbe16ba1ab53078e5ff29013621e33c509eedab055775d956427712c236e \
|
||||||
--hash=sha256:c2d37ab77531417474168eb79d6d80b14f821a966818505d03013d0833edb7a8 \
|
--hash=sha256:c1f017dc0875c9209d219f97feceb7d54c2661bb243deb4114478e1295808af7 \
|
||||||
--hash=sha256:c4fc99836233ea196540b17ab0983aff60ed07941751930f5f4d05bc3b3b7359 \
|
--hash=sha256:cebc2d6dbb605a7703d59751dea4bd6b0ab127a5a4338a6f432df1936fef8b26 \
|
||||||
--hash=sha256:d581b735e177fdcdce6fed8e7e8880a3fb6ee4e3653a3ac6af01c6f4c03effc5 \
|
--hash=sha256:cf7de32f486e4ac9e2d93b810f9e9ac72a728dd46a32a0bb403222f27f653514 \
|
||||||
--hash=sha256:d6da64deb6b8ed903e7560180a92f2d804ee1ba5eeb849ac2748b8c1aba1f6d7 \
|
--hash=sha256:d482d171c406ae88c5b19cad3b6a1c4c5209f886ab74bc44c2c865c23f52d860 \
|
||||||
--hash=sha256:d8e8286dd7cea7895157318d1b91cdacac64c479f3cbc8dce548331728484751 \
|
--hash=sha256:d6a48072864e3324e194a8fbb3c657bcc5b5c869dbc64c9537b1d5c862572c0a \
|
||||||
--hash=sha256:ddea102b48f9e339f3948bf22040944184627a30fdf7f858667673b9c5f033c8 \
|
--hash=sha256:d787cf769c3baeb5f6235e778edb52c08dfa923789b5958f28e6450f96107cb1 \
|
||||||
--hash=sha256:dfa20cc6ca228e6b155b11da03825975ce66aea520985dbbddf0f2a5a495c605 \
|
--hash=sha256:dc649493697006bc90614a5f0bbc8cb3cb1866715c474e473694968d7e6b99ab \
|
||||||
--hash=sha256:e3e5193ef5a3dc73bceee50f7fdc2c90dbb76c42df8d8fae3d1067a583df579e \
|
--hash=sha256:ddf47472af2e4280d79bac82304f5e80150211f1b9e614b760061d5fdfbb6eba \
|
||||||
--hash=sha256:e3eeb0aabd6bd5ce64faae67e9935203a6991b4bc2a485a767fbafb2c5125f45 \
|
--hash=sha256:e5651f3f87add730ee6608d915009e19c911fba0cb000c7e3ea994b7d768eb12 \
|
||||||
--hash=sha256:e5805d5a22fd19c8ccff10a9561f9df94436b0545619ea579db2d3c35294bce2 \
|
--hash=sha256:e79aba74ffaf5f78a050d777c184cddf8fdffabab38acf5f3ef1fecbc17895d6 \
|
||||||
--hash=sha256:e85b752a1e912b70eaad4fafbd4d1238007ab221de2009b9a2f5ae7461239895 \
|
--hash=sha256:eaa088384c46f519dacb93b7ec483a6d6b19a4a2085ae4f25ab9b1c43d387d1e \
|
||||||
--hash=sha256:eaf7fa2de5c0be8ae6ff8e9bea2ccd725e980541244521d8d4b5f3354a27babe \
|
--hash=sha256:eaca7ff36f0f52e2111ec71f169d8fd3e889e7ddc0d2592e0d703fd8d3ce8fac \
|
||||||
--hash=sha256:ebfb099f8dcf083deef3ac1ca4c1503f387cf76296fcb3816b66f5ecb5f54fdb \
|
--hash=sha256:f06571a052127dc1b4e8b83029b4d1b20daa2b64a31cdd181fc6bc774e9000eb \
|
||||||
--hash=sha256:ece3d2cfe132e7d51f44a832b303895e6f2d499c5e74dfbdb06ee246147a304a \
|
--hash=sha256:fd0d703772bba096843785bd38371e31bb4a0c1151497ad5739d182114a73f7f
|
||||||
--hash=sha256:ed9749eef4cbd126da3dc1d6bcb3a57f5eb7ac6a6484146bdbf743f552dfc577 \
|
|
||||||
--hash=sha256:ede83e07a75dd06bc501566c1eca2afc0d61677c1472ac9ad93fdee6e638a48d \
|
|
||||||
--hash=sha256:ef4aea96ce4d3b074422cb4f2f64e216bf9e213004bb58ecfdf50ea02ea8eb9a \
|
|
||||||
--hash=sha256:f3a3570c4a2a16746ac2c31a7c7c7b0c186b95ce902e33db6f28094ed7387dda \
|
|
||||||
--hash=sha256:f407cb6b8e9d6d8c626bc73c945db1706035af8fd632295547bf1c9e46d092d6 \
|
|
||||||
--hash=sha256:f74a575920ab21fe304421a3fc28793d82e299cae9eccb37084e9fc7f3617c20
|
|
||||||
# via rustworkx
|
# via rustworkx
|
||||||
orjson==3.12.0 \
|
orjson==3.12.0 \
|
||||||
--hash=sha256:010811c1b69773450a01cef97727a67b223242f350b77d4ca000e59a9ef2155a \
|
--hash=sha256:010811c1b69773450a01cef97727a67b223242f350b77d4ca000e59a9ef2155a \
|
||||||
@@ -1950,6 +1943,7 @@ typing-extensions==4.16.0 \
|
|||||||
--hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \
|
--hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \
|
||||||
--hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5
|
--hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5
|
||||||
# via
|
# via
|
||||||
|
# aiohttp
|
||||||
# aiosignal
|
# aiosignal
|
||||||
# beautifulsoup4
|
# beautifulsoup4
|
||||||
# checkov
|
# checkov
|
||||||
|
|||||||
@@ -0,0 +1,76 @@
|
|||||||
|
"""Drop checkov-suppressed results from its SARIF output before upload.
|
||||||
|
|
||||||
|
checkov's SARIF exporter includes every evaluated check as an ordinary
|
||||||
|
result, including ones it internally marked SKIPPED via an inline
|
||||||
|
`# checkov:skip=` comment or a `checkov.io/skipN` resource annotation - it
|
||||||
|
never uses SARIF's `suppressions` field, and never drops them. checkov's
|
||||||
|
JSON output *does* correctly record which checks were skipped, so this
|
||||||
|
cross-references the two: any SARIF result whose (check_id, file) pair
|
||||||
|
appears in the JSON's skipped_checks is removed before GitHub ever sees it.
|
||||||
|
|
||||||
|
Without this, every already-suppressed finding reopens as a brand new code
|
||||||
|
scanning alert on every run, forever (see #6035/#6036, #6112-6115,
|
||||||
|
#6128-6131 for the pattern this was chasing before this script existed).
|
||||||
|
|
||||||
|
Usage: filter_checkov_skipped.py <json_path> <sarif_in_path> <sarif_out_path>
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
|
||||||
|
|
||||||
|
def path_suffix(path: str, segments: int = 2) -> str:
|
||||||
|
"""Last N path segments, normalized to forward slashes, lowercased.
|
||||||
|
|
||||||
|
checkov's JSON file_path and SARIF artifactLocation.uri are relative to
|
||||||
|
different roots (the scanned directory vs. a temp helm-render dir), so
|
||||||
|
they can't be compared directly - but the last couple of segments
|
||||||
|
(e.g. "templates/service.yaml") are stable across both and specific
|
||||||
|
enough in practice to avoid cross-file collisions.
|
||||||
|
"""
|
||||||
|
normalized = path.replace("\\", "/").strip("/")
|
||||||
|
return "/".join(normalized.split("/")[-segments:]).lower()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
json_path, sarif_in_path, sarif_out_path = sys.argv[1:4]
|
||||||
|
|
||||||
|
with open(json_path, encoding="utf-8") as f:
|
||||||
|
checkov_json = json.load(f)
|
||||||
|
if isinstance(checkov_json, dict):
|
||||||
|
checkov_json = [checkov_json]
|
||||||
|
|
||||||
|
skipped = set()
|
||||||
|
for block in checkov_json:
|
||||||
|
for check in block.get("results", {}).get("skipped_checks", []):
|
||||||
|
skipped.add((check["check_id"], path_suffix(check["file_path"])))
|
||||||
|
|
||||||
|
with open(sarif_in_path, encoding="utf-8") as f:
|
||||||
|
sarif = json.load(f)
|
||||||
|
|
||||||
|
removed = 0
|
||||||
|
for run in sarif.get("runs", []):
|
||||||
|
kept = []
|
||||||
|
for result in run.get("results", []):
|
||||||
|
rule_id = result.get("ruleId")
|
||||||
|
locations = result.get("locations") or [{}]
|
||||||
|
uri = (
|
||||||
|
locations[0]
|
||||||
|
.get("physicalLocation", {})
|
||||||
|
.get("artifactLocation", {})
|
||||||
|
.get("uri", "")
|
||||||
|
)
|
||||||
|
if (rule_id, path_suffix(uri)) in skipped:
|
||||||
|
removed += 1
|
||||||
|
continue
|
||||||
|
kept.append(result)
|
||||||
|
run["results"] = kept
|
||||||
|
|
||||||
|
with open(sarif_out_path, "w", encoding="utf-8") as f:
|
||||||
|
json.dump(sarif, f)
|
||||||
|
|
||||||
|
print(f"Removed {removed} checkov-suppressed result(s) from the SARIF before upload.")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -44,12 +44,20 @@ jobs:
|
|||||||
pip install -r .github/requirements/pep517-build.txt --require-hashes
|
pip install -r .github/requirements/pep517-build.txt --require-hashes
|
||||||
pip install --no-deps --no-build-isolation -e .
|
pip install --no-deps --no-build-isolation -e .
|
||||||
pip install -r .github/requirements/base-deps.txt --require-hashes
|
pip install -r .github/requirements/base-deps.txt --require-hashes
|
||||||
# NOTE: benchmarks/ does not currently exist in this repo, so this
|
# NOTE: benchmarks/ does not currently exist in this repo (neither
|
||||||
# step and the run below it fail on any real invocation - pre-existing,
|
# requirements.txt nor benchmarks_runner.py below), so this job
|
||||||
# unrelated to this pinning change. Left as-is since there's nothing
|
# already fails on any real invocation - pre-existing, unrelated to
|
||||||
# to hash without knowing what belongs there.
|
# this pinning change. The `pip install -r benchmarks/requirements.txt`
|
||||||
pip install -r benchmarks/requirements.txt
|
# step that used to be here is dropped rather than fixed: there's
|
||||||
python -m spacy download en_core_web_sm
|
# nothing to hash-pin without knowing what that file should
|
||||||
|
# contain, and an unpinned install here would just re-trip
|
||||||
|
# Scorecard's Pinned-Dependencies check for no real benefit, since
|
||||||
|
# the job can't run to completion regardless.
|
||||||
|
#
|
||||||
|
# `python -m spacy download en_core_web_sm` fetches an unpinned,
|
||||||
|
# unhashed wheel from spacy-models' GitHub releases - replaced with
|
||||||
|
# a hash-pinned direct-URL install of the same 3.8.0 model (matches
|
||||||
|
# the spacy==3.8.15 pinned in base-deps.txt) via benchmark-extra.txt.
|
||||||
pip install -r .github/requirements/benchmark-extra.txt --require-hashes
|
pip install -r .github/requirements/benchmark-extra.txt --require-hashes
|
||||||
|
|
||||||
- name: Execute Benchmarks (Real Mode)
|
- name: Execute Benchmarks (Real Mode)
|
||||||
|
|||||||
@@ -76,12 +76,28 @@ jobs:
|
|||||||
PYTHONUTF8: "1"
|
PYTHONUTF8: "1"
|
||||||
run: |
|
run: |
|
||||||
New-Item -ItemType Directory -Force reports | Out-Null
|
New-Item -ItemType Directory -Force reports | Out-Null
|
||||||
checkov --directory . --framework kubernetes helm dockerfile github_actions secrets bicep arm --soft-fail --output sarif --output-file-path reports/checkov.sarif
|
checkov --directory . --framework kubernetes helm dockerfile github_actions secrets bicep arm --soft-fail --output sarif --output json --output-file-path reports
|
||||||
if (-not (Test-Path reports/checkov.sarif)) {
|
if (-not (Test-Path reports/results_sarif.sarif)) {
|
||||||
$sarif = Get-ChildItem -Path reports -Recurse -Filter *.sarif | Select-Object -First 1
|
$sarif = Get-ChildItem -Path reports -Recurse -Filter *.sarif | Select-Object -First 1
|
||||||
if ($null -eq $sarif) { throw "Checkov did not produce a SARIF file" }
|
if ($null -eq $sarif) { throw "Checkov did not produce a SARIF file" }
|
||||||
Copy-Item $sarif.FullName reports/checkov.sarif
|
Copy-Item $sarif.FullName reports/results_sarif.sarif
|
||||||
}
|
}
|
||||||
|
if (-not (Test-Path reports/results_json.json)) {
|
||||||
|
$json = Get-ChildItem -Path reports -Recurse -Filter *.json | Select-Object -First 1
|
||||||
|
if ($null -eq $json) { throw "Checkov did not produce a JSON file" }
|
||||||
|
Copy-Item $json.FullName reports/results_json.json
|
||||||
|
}
|
||||||
|
|
||||||
|
# checkov's SARIF exporter includes checks it internally marked SKIPPED
|
||||||
|
# (via the inline `# checkov:skip=` comments / `checkov.io/skipN`
|
||||||
|
# annotations already on the Helm chart) as ordinary un-suppressed
|
||||||
|
# results - it never uses SARIF's own `suppressions` field, so GitHub
|
||||||
|
# opens a fresh alert for the same already-suppressed finding on every
|
||||||
|
# single run (see #6035/#6036, #6112-6115, #6128-6131). checkov's JSON
|
||||||
|
# output does correctly record the skip, so cross-reference it here
|
||||||
|
# instead of re-dismissing the same alerts by hand forever.
|
||||||
|
- name: Filter checkov's own suppressed checks out of the SARIF
|
||||||
|
run: python .github/scripts/filter_checkov_skipped.py reports/results_json.json reports/results_sarif.sarif reports/checkov.sarif
|
||||||
|
|
||||||
- name: Upload Checkov results to Security tab
|
- name: Upload Checkov results to Security tab
|
||||||
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4
|
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4
|
||||||
|
|||||||
@@ -60,7 +60,12 @@ jobs:
|
|||||||
# (json/text/screen/...), not a file path. Writing JSON to a file
|
# (json/text/screen/...), not a file path. Writing JSON to a file
|
||||||
# now requires --save-json; the previous `--output safety-report.json`
|
# now requires --save-json; the previous `--output safety-report.json`
|
||||||
# usage was silently invalid and never produced a report.
|
# usage was silently invalid and never produced a report.
|
||||||
safety check --save-json safety-report.json || true
|
#
|
||||||
|
# Scan requirements-ci.txt directly instead of the installed environment
|
||||||
|
# to avoid crashes from packages like cuda-toolkit that Safety cannot
|
||||||
|
# parse. This also ensures we're auditing the declared dependency tree
|
||||||
|
# rather than transitive dependencies of the security tooling itself.
|
||||||
|
safety check --file requirements-ci.txt --save-json safety-report.json || true
|
||||||
|
|
||||||
# Guard 1: fail loudly if Safety exited before writing a report at all
|
# Guard 1: fail loudly if Safety exited before writing a report at all
|
||||||
# (network error, API auth failure, tool crash). Without this check a
|
# (network error, API auth failure, tool crash). Without this check a
|
||||||
@@ -73,10 +78,45 @@ jobs:
|
|||||||
|
|
||||||
echo "Checking for package vulnerabilities..."
|
echo "Checking for package vulnerabilities..."
|
||||||
|
|
||||||
# No || echo "0" fallback: if jq fails (malformed JSON, missing key,
|
# Vulnerability IDs reviewed and accepted as non-actionable for this
|
||||||
# vulnerabilities:null) VULNS will be empty or "null" so guard 2 below
|
# project. Filtered out here with jq rather than passed to Safety's
|
||||||
# catches it rather than silently treating the broken report as zero.
|
# own --ignore flag: --ignore crashes ("Unhandled exception happened:
|
||||||
VULNS=$(jq '.vulnerabilities | length' safety-report.json 2>/dev/null)
|
# 'cuda-toolkit'") when it has to apply itself against a live-matched
|
||||||
|
# vulnerability for cuda-toolkit, apparently the same class of
|
||||||
|
# unguarded dependency-graph lookup that broke the plain environment
|
||||||
|
# scan (see git history on this file). The un-ignored scan above is
|
||||||
|
# the one path confirmed - by an actual CI run - not to crash even
|
||||||
|
# with a live cuda-toolkit match, so all filtering happens after the
|
||||||
|
# fact in jq instead of inside Safety.
|
||||||
|
#
|
||||||
|
# - SFTY-20260120-40557 (CVE-2025-33228): cuda-toolkit<13.1.0. torch
|
||||||
|
# 2.13.0 (latest available; no newer release exists) hard-pins
|
||||||
|
# cuda-toolkit[cublas,cudart,cufft,cufile,cupti,curand,cusolver,
|
||||||
|
# cusparse,nvjitlink,nvrtc,nvtx]==13.0.3 on Linux - not a version we
|
||||||
|
# control. The CVE is OS command injection in NVIDIA Nsight
|
||||||
|
# Systems' gfx_hotspot recipe (process_nsys_rep_cli.py), requiring
|
||||||
|
# manual invocation with an attacker-supplied string; unreachable
|
||||||
|
# from Semantica, and Nsight Systems isn't among the extras torch
|
||||||
|
# requests above. Re-evaluate once torch pins a patched
|
||||||
|
# cuda-toolkit.
|
||||||
|
IGNORED_VULN_IDS="SFTY-20260120-40557"
|
||||||
|
|
||||||
|
# Exported so the "Comment PR with Security Results" step below can
|
||||||
|
# apply the same exclusion list to the raw report - it reads
|
||||||
|
# safety-report.json independently in JS, so without this the PR
|
||||||
|
# comment would show the accepted CVE as a live finding even though
|
||||||
|
# this gate correctly treats it as non-actionable.
|
||||||
|
echo "IGNORED_VULN_IDS=$IGNORED_VULN_IDS" >> "$GITHUB_ENV"
|
||||||
|
|
||||||
|
# No []? / || echo "0" fallback on a missing/null "vulnerabilities"
|
||||||
|
# key: iterating over null raises inside jq, leaving VULNS empty, so
|
||||||
|
# guard 2 below catches it rather than silently treating a broken
|
||||||
|
# report as zero.
|
||||||
|
VULNS=$(jq --arg ignored "$IGNORED_VULN_IDS" '
|
||||||
|
($ignored | split(",")) as $ignore_list
|
||||||
|
| [.vulnerabilities[] | select(.vulnerability_id as $id | ($ignore_list | index($id)) | not)]
|
||||||
|
| length
|
||||||
|
' safety-report.json 2>/dev/null)
|
||||||
|
|
||||||
# Guard 2: ensure VULNS is a non-negative integer before the -gt
|
# Guard 2: ensure VULNS is a non-negative integer before the -gt
|
||||||
# comparison. "null" (missing/null key) or "" (jq parse failure) would
|
# comparison. "null" (missing/null key) or "" (jq parse failure) would
|
||||||
@@ -92,10 +132,14 @@ jobs:
|
|||||||
echo "CI will fail to prevent merging of vulnerable dependencies"
|
echo "CI will fail to prevent merging of vulnerable dependencies"
|
||||||
echo ""
|
echo ""
|
||||||
echo "Vulnerability details:"
|
echo "Vulnerability details:"
|
||||||
jq -r '.vulnerabilities[] | "- \(.package_name)==\(.analyzed_version): \(.vulnerability_id) (\(.CVE // "no CVE assigned"))"' safety-report.json || true
|
jq --arg ignored "$IGNORED_VULN_IDS" -r '
|
||||||
|
($ignored | split(",")) as $ignore_list
|
||||||
|
| .vulnerabilities[] | select(.vulnerability_id as $id | ($ignore_list | index($id)) | not)
|
||||||
|
| "- \(.package_name)==\(.analyzed_version): \(.vulnerability_id) (\(.CVE // "no CVE assigned"))"
|
||||||
|
' safety-report.json || true
|
||||||
exit 1
|
exit 1
|
||||||
else
|
else
|
||||||
echo "✅ No security vulnerabilities found"
|
echo "✅ No actionable security vulnerabilities found (ignored: $IGNORED_VULN_IDS)"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
- name: Run Bandit (Code Security Linter)
|
- name: Run Bandit (Code Security Linter)
|
||||||
@@ -184,14 +228,27 @@ jobs:
|
|||||||
return lines.join('\n');
|
return lines.join('\n');
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Mirrors the shell step's own IGNORED_VULN_IDS (passed through
|
||||||
|
// $GITHUB_ENV) so an accepted, non-actionable CVE that the CI
|
||||||
|
// gate already excluded doesn't reappear here as a live finding -
|
||||||
|
// this reads the same raw, unfiltered safety-report.json.
|
||||||
|
const ignoredVulnIds = (process.env.IGNORED_VULN_IDS || '')
|
||||||
|
.split(',')
|
||||||
|
.map((id) => id.trim())
|
||||||
|
.filter(Boolean);
|
||||||
|
|
||||||
const safetySection = renderSection(
|
const safetySection = renderSection(
|
||||||
'Safety — dependency vulnerabilities',
|
'Safety — dependency vulnerabilities',
|
||||||
'safety-report.json',
|
'safety-report.json',
|
||||||
(data) => (data.vulnerabilities || []).map(
|
(data) => (data.vulnerabilities || [])
|
||||||
(v) => `- \`${v.package_name}==${v.analyzed_version}\`: ${v.vulnerability_id}` +
|
.filter((v) => !ignoredVulnIds.includes(v.vulnerability_id))
|
||||||
(v.CVE ? ` (${v.CVE})` : '') + ` — ${v.advisory || 'no advisory text'}`
|
.map(
|
||||||
)
|
(v) => `- \`${v.package_name}==${v.analyzed_version}\`: ${v.vulnerability_id}` +
|
||||||
);
|
(v.CVE ? ` (${v.CVE})` : '') + ` — ${v.advisory || 'no advisory text'}`
|
||||||
|
)
|
||||||
|
) + (ignoredVulnIds.length
|
||||||
|
? `\n\n_Excluded as accepted, non-actionable findings: ${ignoredVulnIds.join(', ')} — see the workflow file's inline comments for why._`
|
||||||
|
: '');
|
||||||
|
|
||||||
const banditSection = renderSection(
|
const banditSection = renderSection(
|
||||||
'Bandit — HIGH-severity code issues',
|
'Bandit — HIGH-severity code issues',
|
||||||
|
|||||||
@@ -9,6 +9,29 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- **`ErasureCoordinator` completes the erasure workflow `purge_node()` only starts — the graph node was removed while the same content survived verbatim in `AgentMemory` and as an embedding** (closes #1018) by @pravit-amp
|
||||||
|
- New `semantica/context/erasure.py`, exporting `ErasureCoordinator` and `ErasureReceipt` from `semantica.context`. `purge_node()`/`purge_edge()` (#957) are graph-scope by design and their changelog entry documents this gap explicitly; the changelog also names GDPR Article 17 as the motivation, and an Article 17 erasure that removes the node while the content stays retrievable by similarity search is not an erasure — it is worse than not offering one, because `purge_node()` returns `True` and writes a tombstone attesting the content is gone
|
||||||
|
- The coordinator **composes** the existing public APIs — nothing in `context_graph.py` or `agent_memory.py` changes behaviorally, and `ContextGraph` keeps its documented graph-scope contract rather than acquiring references to `AgentMemory`/`vector_store` that would invert the dependency
|
||||||
|
- `erase_entity(entity_id, reason=..., at=..., vector_ids=...)` returns an `ErasureReceipt`; `erase_entities([...])` returns one receipt per entity, in order, so one entity's failure does not stop the rest
|
||||||
|
- **Honest partial reporting is the point.** Each store reports one of five statuses — `erased`, `not_found`, `not_configured` (store never bound; normal), `unsupported` (store cannot delete at all; retrying will not help), `failed` — and `receipt.complete` is `False` when any store reports `unsupported`/`failed`, with `receipt.incomplete_stores` naming them. A receipt reading `graph: erased, memory: 14 erased, vectors: unsupported on faiss` is actionable; a bare `True` is a compliance liability
|
||||||
|
- **Erasure runs outward-in: vectors → memory → graph.** The graph tombstone is the durable attestation that an erasure happened, so writing it first would let a crash mid-cascade leave a record claiming more than occurred. Erasing the graph last means a partial failure leaves the node present and the receipt incomplete — recoverable and honest; the reverse is neither
|
||||||
|
- **Partial failure is a result, not an exception**: a store that raises is recorded as `failed` (with the exception type) and the remaining legs still run, rather than aborting into a half-erased state with no record of which half
|
||||||
|
- **The memory sweep cannot be silently truncated.** `find_by_entity(entity_id, limit=10)` returned `results[:limit]`, so the obvious hand-rolled cascade erases the first ten items and reports success — an erasure check computed from a page already truncated by the very `limit` it was called with. The coordinator sweeps in pages until dry (deleting as it goes, so the next page is the remainder) rather than passing one large number that is only correct until someone exceeds it, then **re-queries once after the sweep** and reports `failed` with the residual count if anything survived. It also stops rather than spinning if `batch_delete` reports no progress on a non-empty page. Note `find_by_entity` returns items keyed `memory_id`, not `id`
|
||||||
|
- **`unsupported` vector backends are detected by probing, not by calling and catching.** `faiss_store.py`, `milvus_store.py` and `weaviate_store.py` expose no delete at all (FAISS cannot remove from a flat index without a rebuild), while the `VectorStore` facade declares `delete_vectors()` for *every* backend and only raises `NotImplementedError` once called — so probing the facade alone cannot tell a deletable backend from a delete-less one, and the coordinator looks at the backend it wraps. Probing also keeps a missing method distinguishable from an `AttributeError` raised *inside* a working one, which is exactly where guessing wrong produces a false clean bill of health. `NotImplementedError` at call time is still caught and reported as `unsupported`; a store returning `False` is reported as `failed`
|
||||||
|
- Backends are reached under either supported name — `delete_vectors(ids)` (pinecone/qdrant) or `delete(ids)` (pgvector/sqlite-vec) — and the receipt records which was used
|
||||||
|
- `vector_store` defaults to `memory.vector_store` when a memory is supplied, stays overridable for deployments binding a store the memory does not own, and accepts `False` to disable the vector leg. Vectors owned by memory items are removed by the memory leg's own `delete_memory()` cascade; the explicit vector leg covers entity-keyed embeddings written by something other than `AgentMemory`
|
||||||
|
- The receipt's `erased_at` is normalized through `ContextGraph`'s own temporal normalizer, so the receipt and the tombstone written by the same erasure cannot disagree about when it happened; an unparseable `at` is rejected before any store is touched rather than half way through the cascade
|
||||||
|
- `purge_node()`'s docstring now points at the coordinator, so callers reading the graph-scope caveat find the thing that completes the workflow
|
||||||
|
- New `tests/context/test_erasure_coordinator.py`: 48 tests against **real** `ContextGraph`/`AgentMemory` instances rather than mocks — the bug lives in the interaction between them, so mocking it away would test nothing. Covers the 25-items-on-one-entity regression that fails against a naive single `find_by_entity()` call, all three vector-backend shapes (`delete_vectors`/`delete`/neither) plus the facade-over-delete-less-backend shape, residual/no-progress/no-identifier memory failures, partial failure continuing the cascade, idempotency, receipt serialization, and `at` normalization
|
||||||
|
- Full `tests/context/` suite: 738 passed
|
||||||
|
- **Fixed during review** (Qodo): `erase_entity()` resolved `erased_at` up front but passed the caller's original `at` down to `purge_node()`, so on the default `at=None` path the coordinator and the graph each took their own `now()` and the receipt attested to a different instant than the tombstone it points at — breaking the one invariant this module states most loudly. The resolved timestamp is now passed to the graph. The existing test passed only because it supplied an explicit `at`, which hides the drift; a regression test now covers the `at=None` path that callers actually use
|
||||||
|
- **Fixed during review** (Qodo): the vectors leg treated any return value other than the literal `False` as success, but no in-repo backend returns a bool — Qdrant returns `{"status": <UpdateStatus>}` and Pinecone `{"deleted": True}`, so every dict was read as a success and the backend's own account of the delete was discarded. Delete results are now interpreted by shape (bool, dict with explicit failure markers, `None` for a void method, anything else at face value) and the backend payload is recorded in the receipt as `backend_result`, stringified so the receipt stays JSON-serializable as the audit record it is meant to be. Bool markers are matched by identity so a `0` count is not read as `False`, and string markers match as substrings so an enum rendering as `"UpdateStatus.FAILED"` is not read as a success
|
||||||
|
- **Fixed during review** (Qodo): the constructor's "at least one store" guard used `not vector_store`, rejecting a valid store whose `__bool__`/`__len__` makes an empty instance falsey, and reporting `vector_store=None` in the error when an object had been passed; it now distinguishes `None` (absent) from `False` (deliberately disabled) from any other value (provided), and echoes what it actually received
|
||||||
|
- **Fixed during review** (Qodo): `at` annotations accepted only `str`/`datetime` while the shared `ContextGraph` normalizer they delegate to also takes epoch seconds; widened to `int`/`float` with the docstrings updated, so the coordinator no longer advertises less than the graph API it wraps
|
||||||
|
- **Known limitation, unchanged by this PR**: erasure still cannot be *completed* on FAISS/Milvus/Weaviate — `delete_vectors()` is declared on the `VectorStore` facade (`vector_store.py:786`) but not implemented across the backend set, under at least three different names. That is worth its own issue; the coordinator ships reporting `unsupported` and starts reporting `erased` for those backends once it is fixed, with no API change here
|
||||||
|
|
||||||
## [0.6.7] - 2026-08-28
|
## [0.6.7] - 2026-08-28
|
||||||
|
|
||||||
### Added
|
### Added
|
||||||
|
|||||||
+1
-1
@@ -20,7 +20,7 @@ RUN mkdir -p /app/semantica && npm run build
|
|||||||
# .github/dependabot.yml opens a PR bumping the digest pin above. Also: this
|
# .github/dependabot.yml opens a PR bumping the digest pin above. Also: this
|
||||||
# image only serves plain HTTP via uvicorn and never opens a QUIC listener,
|
# image only serves plain HTTP via uvicorn and never opens a QUIC listener,
|
||||||
# so the bug isn't reachable here regardless.
|
# so the bug isn't reachable here regardless.
|
||||||
FROM python:3.14-slim@sha256:cae66f2ef0ec51a9891263eeee7f987dacf0a9879e8aa9353d5606e0530619a5 AS runtime
|
FROM python:3.13-slim@sha256:7ce4b6dfe35e55397b7cda544f8a13f191b7ae28dc5aad71fe664dbc9bc2623f AS runtime
|
||||||
|
|
||||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||||
PYTHONUNBUFFERED=1 \
|
PYTHONUNBUFFERED=1 \
|
||||||
|
|||||||
@@ -18,7 +18,7 @@
|
|||||||
|
|
||||||
> Ingest your enterprise data, extract what matters, build a Context Graph and knowledge graph (KG), and run graph analytics and causal reasoning over all of it, with full decision provenance baked in. Explainable, traceable, and trustworthy by design.
|
> Ingest your enterprise data, extract what matters, build a Context Graph and knowledge graph (KG), and run graph analytics and causal reasoning over all of it, with full decision provenance baked in. Explainable, traceable, and trustworthy by design.
|
||||||
|
|
||||||
**Decision Intelligence · Context Management · Deterministic Reasoning · Ontology Management · Knowledge Modeling · End-to-End Traceability**
|
**Context Management · Knowledge Modeling · Deterministic Reasoning · Ontology Management · Decision Intelligence · End-to-End Traceability**
|
||||||
|
|
||||||
**Open Source · Self-Hostable · Auditable · Governed · Zero Vendor Lock-In**
|
**Open Source · Self-Hostable · Auditable · Governed · Zero Vendor Lock-In**
|
||||||
|
|
||||||
@@ -56,9 +56,7 @@ pip install semantica
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
Most AI agents act without a trail. They store embeddings, not meaning: context that can't be explained, decisions that can't be audited. In lending, that gap is a compliance exposure, not an inconvenience: an underwriting agent's approval has to survive a regulator's "why" months later.
|
Most AI agents run on embeddings, not meaning: similarity scores with no structure, no relationships, and no way to explain why a result came back. Semantica is the semantic/context layer underneath your LLM, vector store, and agent framework: a deterministic infrastructure layer (no LLM required for graph construction, reasoning, or provenance) that turns fragmented enterprise data into a structured, queryable Context Graph and knowledge graph, governed by ontologies and controlled vocabularies (OWL, SHACL, SKOS) so the meaning of your data is explicit, not just its embedding. Decision provenance and audit trails fall out of that structure as a property, not the product itself; in domains a regulator can question, that same structure just happens to double as a straight answer to "why."
|
||||||
|
|
||||||
Semantica sits underneath your LLM, vector store, and agent framework as a deterministic infrastructure layer: no LLM required for graph construction, reasoning, or provenance.
|
|
||||||
|
|
||||||
> ⚠️ **System-level explainability, not foundation-model explainability.** Semantica does not expose or reconstruct what happens *inside* the LLM — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. Semantica explains what's *outside* the model: the context and data fed in, the decision produced, its provenance, relevant relationships, applied policies, and the full execution trail.
|
> ⚠️ **System-level explainability, not foundation-model explainability.** Semantica does not expose or reconstruct what happens *inside* the LLM — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. Semantica explains what's *outside* the model: the context and data fed in, the decision produced, its provenance, relevant relationships, applied policies, and the full execution trail.
|
||||||
|
|
||||||
@@ -279,7 +277,7 @@ retrieved = ctx.retrieve("who approved the Acme contract?")
|
|||||||
|
|
||||||
## Recipe: Audit Trail for a Regulated Decision
|
## Recipe: Audit Trail for a Regulated Decision
|
||||||
|
|
||||||
The flagship pattern: record a causally-linked decision chain, attach provenance to every entity, and export a regulator-ready audit trail.
|
One pattern built on the same Context Graph: record a causally-linked decision chain, attach provenance to every entity, and export a regulator-ready audit trail.
|
||||||
|
|
||||||
```python
|
```python
|
||||||
from semantica.context import ContextGraph
|
from semantica.context import ContextGraph
|
||||||
@@ -1030,7 +1028,7 @@ team = Team(agents=[researcher, analyst], mode="coordinate")
|
|||||||
|
|
||||||
## More Recipes
|
## More Recipes
|
||||||
|
|
||||||
The flagship audit-trail recipe is [above](#recipe-audit-trail-for-a-regulated-decision). Here are three more common patterns.
|
The audit-trail recipe is [above](#recipe-audit-trail-for-a-regulated-decision). Here are three more common patterns.
|
||||||
|
|
||||||
<details>
|
<details>
|
||||||
<summary><b>End-to-End GraphRAG Pipeline</b></summary>
|
<summary><b>End-to-End GraphRAG Pipeline</b></summary>
|
||||||
|
|||||||
@@ -10,15 +10,16 @@
|
|||||||
"\n",
|
"\n",
|
||||||
"## Overview\n",
|
"## Overview\n",
|
||||||
"\n",
|
"\n",
|
||||||
"This notebook demonstrates how to build knowledge graphs from entities and relationships using Semantica's graph building modules. You'll learn to use `GraphBuilder` and `EntityResolver`.\n",
|
"This notebook demonstrates how to build knowledge graphs from extracted entities and relationships using Semantica's graph building modules. You'll learn to use `GraphBuilder` and `EntityResolver`.\n",
|
||||||
"\n",
|
"\n",
|
||||||
"**Documentation**: [API Reference](https://semantica.readthedocs.io/reference/kg/)\n",
|
"**Documentation**: [API Reference](https://semantica.readthedocs.io/reference/kg/)\n",
|
||||||
"\n",
|
"\n",
|
||||||
"### Learning Objectives\n",
|
"### Learning Objectives\n",
|
||||||
"\n",
|
"\n",
|
||||||
"- Use `GraphBuilder` to construct knowledge graphs\n",
|
"- Extract entity mentions and relations, and map them into graph records\n",
|
||||||
"- Use `EntityResolver` to resolve entity conflicts\n",
|
"- Use `GraphBuilder` to construct a graph whose edges come from the actual extracted relations\n",
|
||||||
"**Note**: For deduplication, use the `semantica.deduplication` module.\n",
|
"- Use `EntityResolver` to merge duplicate mentions and remap relationship endpoints\n",
|
||||||
|
"- Use the `semantica.deduplication` module and report the complete deduplicated entity set\n",
|
||||||
"\n",
|
"\n",
|
||||||
"## Installation\n",
|
"## Installation\n",
|
||||||
"\n",
|
"\n",
|
||||||
@@ -32,120 +33,217 @@
|
|||||||
"\n",
|
"\n",
|
||||||
"---\n",
|
"---\n",
|
||||||
"\n",
|
"\n",
|
||||||
"## Step 1: Build Knowledge Graph\n",
|
"## Step 1: Extract Entities and Relations\n",
|
||||||
"\n",
|
"\n",
|
||||||
"Construct a knowledge graph from entities and relationships.\n"
|
"Extract entity mentions and relations from text. The sample text mentions `Apple Inc.` in two separate sentences, so we can later show how duplicate mentions are resolved into one canonical entity.\n"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": null,
|
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
"source": [
|
||||||
"!pip install semantica\n"
|
"%pip install semantica\n",
|
||||||
]
|
"\n",
|
||||||
|
"# spaCy models are distributed separately from the spaCy library. This lesson\n",
|
||||||
|
"# relies on the English model to recognize standalone places such as Cupertino.\n",
|
||||||
|
"import sys\n",
|
||||||
|
"import subprocess\n",
|
||||||
|
"import spacy\n",
|
||||||
|
"\n",
|
||||||
|
"try:\n",
|
||||||
|
" spacy.load(\"en_core_web_sm\")\n",
|
||||||
|
"except OSError:\n",
|
||||||
|
" subprocess.check_call([sys.executable, \"-m\", \"spacy\", \"download\", \"en_core_web_sm\"])\n"
|
||||||
|
],
|
||||||
|
"execution_count": null,
|
||||||
|
"outputs": []
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": null,
|
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
"source": [
|
||||||
"from semantica.kg import GraphBuilder\n",
|
|
||||||
"from semantica.semantic_extract import NERExtractor, RelationExtractor\n",
|
"from semantica.semantic_extract import NERExtractor, RelationExtractor\n",
|
||||||
"\n",
|
"\n",
|
||||||
"builder = GraphBuilder()\n",
|
"text = (\n",
|
||||||
|
" \"Apple Inc. is headquartered in Cupertino, California. \"\n",
|
||||||
|
" \"Tim Cook is the CEO of Apple Inc. \"\n",
|
||||||
|
" \"The company is a technology company.\"\n",
|
||||||
|
")\n",
|
||||||
|
"\n",
|
||||||
"ner_extractor = NERExtractor()\n",
|
"ner_extractor = NERExtractor()\n",
|
||||||
"relation_extractor = RelationExtractor()\n",
|
"relation_extractor = RelationExtractor()\n",
|
||||||
"\n",
|
"\n",
|
||||||
"text = \"Apple Inc. is a technology company. Tim Cook is the CEO of Apple Inc. Apple Inc. is headquartered in Cupertino, California.\"\n",
|
"mentions = ner_extractor.extract(text)\n",
|
||||||
|
"relations = relation_extractor.extract(text, mentions)\n",
|
||||||
"\n",
|
"\n",
|
||||||
"entities_list = ner_extractor.extract(text)\n",
|
"print(\"Entity mentions:\")\n",
|
||||||
"relationships_list = relation_extractor.extract(text, entities_list)\n",
|
"for mention in mentions:\n",
|
||||||
|
" print(f\" {mention.text!r:<13} {mention.label:<7} span=[{mention.start_char}:{mention.end_char}]\")\n",
|
||||||
"\n",
|
"\n",
|
||||||
"entities = []\n",
|
"print(\"\\nExtracted relations:\")\n",
|
||||||
"for i, entity in enumerate(entities_list[:5], 1):\n",
|
"for rel in relations:\n",
|
||||||
" entities.append({\n",
|
" print(f\" {rel.subject.text!r} --{rel.predicate}--> {rel.object.text!r}\")"
|
||||||
" \"id\": f\"e{i}\",\n",
|
],
|
||||||
" \"type\": entity.label,\n",
|
"execution_count": null,
|
||||||
" \"name\": entity.text,\n",
|
"outputs": []
|
||||||
" \"properties\": {}\n",
|
|
||||||
" })\n",
|
|
||||||
"\n",
|
|
||||||
"relationships = []\n",
|
|
||||||
"for i, rel in enumerate(relationships_list[:3], 1):\n",
|
|
||||||
" relationships.append({\n",
|
|
||||||
" \"source\": f\"e{1}\",\n",
|
|
||||||
" \"target\": f\"e{i+1}\",\n",
|
|
||||||
" \"type\": rel.predicate,\n",
|
|
||||||
" \"properties\": {}\n",
|
|
||||||
" })\n",
|
|
||||||
"\n",
|
|
||||||
"knowledge_graph = builder.build(entities, relationships)\n",
|
|
||||||
"\n",
|
|
||||||
"print(f\"Built knowledge graph with {len(knowledge_graph.get('entities', []))} entities\")\n",
|
|
||||||
"print(f\"Relationships: {len(knowledge_graph.get('relationships', []))}\")"
|
|
||||||
]
|
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"source": [
|
"source": [
|
||||||
"## Step 2: Entity Resolution\n",
|
"## Step 2: Build the Knowledge Graph\n",
|
||||||
"\n",
|
"\n",
|
||||||
"Resolve entity conflicts and duplicates.\n"
|
"Give every mention a graph ID, then translate each relation's `subject` and `object` into those IDs. Building edges from the actual relation endpoints — rather than guessing endpoints from list positions — is what keeps the graph faithful to the text.\n"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "code",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"from semantica.kg import GraphBuilder\n",
|
||||||
|
"\n",
|
||||||
|
"entities = []\n",
|
||||||
|
"span_to_id = {}\n",
|
||||||
|
"for i, mention in enumerate(mentions, 1):\n",
|
||||||
|
" graph_id = f\"e{i}\"\n",
|
||||||
|
" span_to_id[(mention.start_char, mention.end_char)] = graph_id\n",
|
||||||
|
" entities.append({\n",
|
||||||
|
" \"id\": graph_id,\n",
|
||||||
|
" \"type\": mention.label,\n",
|
||||||
|
" \"name\": mention.text,\n",
|
||||||
|
" \"properties\": {},\n",
|
||||||
|
" })\n",
|
||||||
|
"\n",
|
||||||
|
"relationships = []\n",
|
||||||
|
"for rel in relations:\n",
|
||||||
|
" source_id = span_to_id.get((rel.subject.start_char, rel.subject.end_char))\n",
|
||||||
|
" target_id = span_to_id.get((rel.object.start_char, rel.object.end_char))\n",
|
||||||
|
" if source_id is None or target_id is None:\n",
|
||||||
|
" print(f\"Skipping relation with unmapped endpoint: \"\n",
|
||||||
|
" f\"{rel.subject.text!r} --{rel.predicate}--> {rel.object.text!r}\")\n",
|
||||||
|
" continue\n",
|
||||||
|
" relationships.append({\n",
|
||||||
|
" \"source\": source_id,\n",
|
||||||
|
" \"target\": target_id,\n",
|
||||||
|
" \"type\": rel.predicate,\n",
|
||||||
|
" \"properties\": {},\n",
|
||||||
|
" })\n",
|
||||||
|
"\n",
|
||||||
|
"builder = GraphBuilder()\n",
|
||||||
|
"knowledge_graph = builder.build({\"entities\": entities, \"relationships\": relationships})\n",
|
||||||
|
"\n",
|
||||||
|
"id_to_name = {entity[\"id\"]: entity[\"name\"] for entity in entities}\n",
|
||||||
|
"\n",
|
||||||
|
"print(f\"Graph entities ({len(knowledge_graph['entities'])}):\")\n",
|
||||||
|
"for entity in knowledge_graph[\"entities\"]:\n",
|
||||||
|
" print(f\" {entity['id']}: {entity['name']} ({entity['type']})\")\n",
|
||||||
|
"\n",
|
||||||
|
"print(f\"\\nGraph relationships ({len(knowledge_graph['relationships'])}):\")\n",
|
||||||
|
"for relationship in knowledge_graph[\"relationships\"]:\n",
|
||||||
|
" print(f\" {id_to_name[relationship['source']]} \"\n",
|
||||||
|
" f\"--{relationship['type']}--> {id_to_name[relationship['target']]}\")\n",
|
||||||
|
"\n",
|
||||||
|
"edges = {\n",
|
||||||
|
" (id_to_name[r[\"source\"]], r[\"type\"], id_to_name[r[\"target\"]])\n",
|
||||||
|
" for r in knowledge_graph[\"relationships\"]\n",
|
||||||
|
"}\n",
|
||||||
|
"assert (\"Apple Inc.\", \"located_in\", \"Cupertino\") in edges\n",
|
||||||
|
"assert (\"Tim Cook\", \"works_for\", \"Apple Inc.\") in edges"
|
||||||
|
],
|
||||||
|
"execution_count": null,
|
||||||
|
"outputs": []
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"cell_type": "markdown",
|
||||||
|
"metadata": {},
|
||||||
|
"source": [
|
||||||
|
"## Step 3: Entity Resolution\n",
|
||||||
|
"\n",
|
||||||
|
"The graph currently contains two nodes for the same organization. `EntityResolver` merges duplicate mentions into one canonical entity and records which source IDs were merged (`merged_from`), so relationship endpoints can be remapped onto the canonical entity.\n"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": null,
|
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
"source": [
|
||||||
"from semantica.kg import EntityResolver\n",
|
"from semantica.kg import EntityResolver\n",
|
||||||
"\n",
|
"\n",
|
||||||
"entity_resolver = EntityResolver()\n",
|
"entity_resolver = EntityResolver()\n",
|
||||||
"\n",
|
|
||||||
"resolved_entities = entity_resolver.resolve_entities(entities)\n",
|
"resolved_entities = entity_resolver.resolve_entities(entities)\n",
|
||||||
"\n",
|
"\n",
|
||||||
"print(f\"Original entities: {len(entities)}\")\n",
|
"canonical_id = {}\n",
|
||||||
"print(f\"Resolved entities: {len(resolved_entities)}\")"
|
"for entity in resolved_entities:\n",
|
||||||
]
|
" for source_id in entity.get(\"merged_from\", [entity[\"id\"]]):\n",
|
||||||
|
" canonical_id[source_id] = entity[\"id\"]\n",
|
||||||
|
" if entity.get(\"merged_from\"):\n",
|
||||||
|
" print(f\"Merged {entity['merged_from']} -> {entity['id']}: {entity['name']}\")\n",
|
||||||
|
"\n",
|
||||||
|
"print(f\"\\nMentions in: {len(entities)}, resolved entities out: {len(resolved_entities)}\")\n",
|
||||||
|
"\n",
|
||||||
|
"resolved_names = {entity[\"id\"]: entity[\"name\"] for entity in resolved_entities}\n",
|
||||||
|
"print(\"\\nRelationships remapped onto canonical entities:\")\n",
|
||||||
|
"for relationship in relationships:\n",
|
||||||
|
" source = canonical_id[relationship[\"source\"]]\n",
|
||||||
|
" target = canonical_id[relationship[\"target\"]]\n",
|
||||||
|
" print(f\" {resolved_names[source]} --{relationship['type']}--> {resolved_names[target]}\")\n",
|
||||||
|
"\n",
|
||||||
|
"canonical_entities = {(entity[\"name\"], entity[\"type\"]) for entity in resolved_entities}\n",
|
||||||
|
"assert canonical_entities == {\n",
|
||||||
|
" (\"Apple Inc.\", \"ORG\"),\n",
|
||||||
|
" (\"Tim Cook\", \"PERSON\"),\n",
|
||||||
|
" (\"Cupertino\", \"GPE\"),\n",
|
||||||
|
" (\"California\", \"GPE\"),\n",
|
||||||
|
"}\n",
|
||||||
|
"assert len(resolved_entities) == 4"
|
||||||
|
],
|
||||||
|
"execution_count": null,
|
||||||
|
"outputs": []
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"source": [
|
"source": [
|
||||||
"## Step 3: Deduplication\n",
|
"## Step 4: Deduplication\n",
|
||||||
"\n",
|
"\n",
|
||||||
"Remove duplicate entities from the graph.\n"
|
"The `semantica.deduplication` module gives finer control over the same problem. Note that `merge_duplicates` returns one `MergeOperation` per duplicate *group* — the complete deduplicated collection is those merged entities plus every entity that was not part of any group.\n"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "code",
|
"cell_type": "code",
|
||||||
"execution_count": null,
|
|
||||||
"metadata": {},
|
"metadata": {},
|
||||||
"outputs": [],
|
|
||||||
"source": [
|
"source": [
|
||||||
"from semantica.deduplication import DuplicateDetector, EntityMerger, MergeStrategy\n",
|
"from semantica.deduplication import DuplicateDetector, EntityMerger, MergeStrategy\n",
|
||||||
"\n",
|
"\n",
|
||||||
"# Detect duplicates\n",
|
|
||||||
"detector = DuplicateDetector(similarity_threshold=0.8)\n",
|
"detector = DuplicateDetector(similarity_threshold=0.8)\n",
|
||||||
"duplicate_groups = detector.detect_duplicate_groups(knowledge_graph.get('entities', []))\n",
|
"duplicate_groups = detector.detect_duplicate_groups(entities)\n",
|
||||||
|
"print(f\"Duplicate groups: {len(duplicate_groups)}\")\n",
|
||||||
|
"for group in duplicate_groups:\n",
|
||||||
|
" print(f\" {[entity['name'] for entity in group.entities]} \"\n",
|
||||||
|
" f\"(confidence={group.confidence:.2f})\")\n",
|
||||||
"\n",
|
"\n",
|
||||||
"# Merge duplicates\n",
|
|
||||||
"merger = EntityMerger()\n",
|
"merger = EntityMerger()\n",
|
||||||
"merge_operations = merger.merge_duplicates(\n",
|
"merge_operations = merger.merge_duplicates(\n",
|
||||||
" knowledge_graph.get('entities', []),\n",
|
" entities, strategy=MergeStrategy.KEEP_MOST_COMPLETE\n",
|
||||||
" strategy=MergeStrategy.KEEP_MOST_COMPLETE\n",
|
|
||||||
")\n",
|
")\n",
|
||||||
"\n",
|
"\n",
|
||||||
"deduplicated_entities = [op.merged_entity for op in merge_operations]\n",
|
"merged_source_ids = {\n",
|
||||||
|
" entity[\"id\"] for op in merge_operations for entity in op.source_entities\n",
|
||||||
|
"}\n",
|
||||||
|
"untouched_entities = [e for e in entities if e[\"id\"] not in merged_source_ids]\n",
|
||||||
|
"deduplicated_entities = untouched_entities + [\n",
|
||||||
|
" op.merged_entity for op in merge_operations\n",
|
||||||
|
"]\n",
|
||||||
"\n",
|
"\n",
|
||||||
"print(f\"Original entities: {len(knowledge_graph.get('entities', []))}\")\n",
|
"print(f\"\\nMerge operations: {len(merge_operations)}\")\n",
|
||||||
"print(f\"Deduplicated entities: {len(deduplicated_entities)}\")\n"
|
"print(f\"Deduplicated entities ({len(deduplicated_entities)}):\")\n",
|
||||||
]
|
"for entity in deduplicated_entities:\n",
|
||||||
|
" print(f\" {entity['id']}: {entity['name']} ({entity['type']})\")\n",
|
||||||
|
"\n",
|
||||||
|
"assert len(merge_operations) == 1\n",
|
||||||
|
"assert len(deduplicated_entities) == 4"
|
||||||
|
],
|
||||||
|
"execution_count": null,
|
||||||
|
"outputs": []
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"cell_type": "markdown",
|
"cell_type": "markdown",
|
||||||
@@ -155,9 +253,10 @@
|
|||||||
"\n",
|
"\n",
|
||||||
"You've learned how to build knowledge graphs:\n",
|
"You've learned how to build knowledge graphs:\n",
|
||||||
"\n",
|
"\n",
|
||||||
"- **GraphBuilder**: Construct knowledge graphs from entities and relationships\n",
|
"- **Extraction to graph**: map each mention to a graph ID and build edges from the actual `Relation.subject` / `Relation.object` endpoints\n",
|
||||||
"- **EntityResolver**: Resolve entity conflicts and duplicates\n",
|
"- **GraphBuilder**: construct knowledge graphs from explicit `{\"entities\": ..., \"relationships\": ...}` input\n",
|
||||||
"- **Deduplication**: Use `semantica.deduplication` module for removing duplicate entities\n",
|
"- **EntityResolver**: merge duplicate mentions into canonical entities and remap relationship endpoints\n",
|
||||||
|
"- **Deduplication**: combine `MergeOperation` results with untouched entities to get the complete deduplicated set\n",
|
||||||
"\n",
|
"\n",
|
||||||
"Next: Learn how to analyze graphs in the Graph_Analytics notebook.\n"
|
"Next: Learn how to analyze graphs in the Graph_Analytics notebook.\n"
|
||||||
]
|
]
|
||||||
|
|||||||
+3
-2
@@ -327,10 +327,11 @@ print(f"Relationships active in 2023: {result_2023['num_relationships']}")
|
|||||||
<Accordion title="Persistent graph store: Neo4j, FalkorDB, Apache AGE" icon="database">
|
<Accordion title="Persistent graph store: Neo4j, FalkorDB, Apache AGE" icon="database">
|
||||||
|
|
||||||
```python
|
```python
|
||||||
from semantica.graph_store import Neo4jStore
|
from semantica.graph_store import GraphStore
|
||||||
from semantica.kg import GraphBuilder
|
from semantica.kg import GraphBuilder
|
||||||
|
|
||||||
store = Neo4jStore(
|
store = GraphStore(
|
||||||
|
backend="neo4j",
|
||||||
uri="bolt://localhost:7687",
|
uri="bolt://localhost:7687",
|
||||||
user="neo4j",
|
user="neo4j",
|
||||||
password="password",
|
password="password",
|
||||||
|
|||||||
@@ -25,6 +25,7 @@ icon: "brain"
|
|||||||
| `DecisionRecorder` | Record decisions with embeddings, causal chains, and metadata |
|
| `DecisionRecorder` | Record decisions with embeddings, causal chains, and metadata |
|
||||||
| `PolicyEngine` | Policy management: `add_policy()`, `check_compliance()`, `get_applicable_policies()` |
|
| `PolicyEngine` | Policy management: `add_policy()`, `check_compliance()`, `get_applicable_policies()` |
|
||||||
| `CausalChainAnalyzer` | Trace how decisions influenced each other: `get_causal_chain(decision_id)` |
|
| `CausalChainAnalyzer` | Trace how decisions influenced each other: `get_causal_chain(decision_id)` |
|
||||||
|
| `ErasureCoordinator` | Erase an entity across graph, memory, and vector store, returning an auditable `ErasureReceipt` |
|
||||||
|
|
||||||
|
|
||||||
## What You Get
|
## What You Get
|
||||||
@@ -634,6 +635,100 @@ queried together safely. Vector-store writes are deferred until the in-memory im
|
|||||||
commits; adapter synchronization remains best-effort and logs failures.
|
commits; adapter synchronization remains best-effort and logs failures.
|
||||||
|
|
||||||
|
|
||||||
|
## ErasureCoordinator
|
||||||
|
|
||||||
|
`ContextGraph.purge_node()` is scoped to one graph: the node is removed and a
|
||||||
|
tombstone is written, but the same content can still be live as an `AgentMemory`
|
||||||
|
item and as an embedding in the vector store. `ErasureCoordinator` drives the
|
||||||
|
cascade across every bound store and returns an `ErasureReceipt` recording what
|
||||||
|
each one reported.
|
||||||
|
|
||||||
|
```python
|
||||||
|
from semantica.context import AgentMemory, ContextGraph, ErasureCoordinator
|
||||||
|
|
||||||
|
coordinator = ErasureCoordinator(graph=graph, memory=memory)
|
||||||
|
|
||||||
|
receipt = coordinator.erase_entity(
|
||||||
|
"customer-4471",
|
||||||
|
reason="GDPR Art. 17 request #882",
|
||||||
|
)
|
||||||
|
|
||||||
|
if not receipt.complete:
|
||||||
|
# These stores may still hold the entity; handle them out of band.
|
||||||
|
print(receipt.incomplete_stores)
|
||||||
|
```
|
||||||
|
|
||||||
|
<Warning>
|
||||||
|
Check the receipt — the call returning is not proof the data is gone. FAISS,
|
||||||
|
Milvus, and Weaviate expose no delete method, so erasure cannot be completed on
|
||||||
|
those backends today; the receipt reports `unsupported` rather than a success it
|
||||||
|
did not achieve.
|
||||||
|
</Warning>
|
||||||
|
|
||||||
|
### Constructor Parameters
|
||||||
|
|
||||||
|
| Parameter | Type | Default | Description |
|
||||||
|
| :--- | :--- | :--- | :--- |
|
||||||
|
| `graph` | `ContextGraph` | `None` | Anything exposing `purge_node()` |
|
||||||
|
| `memory` | `AgentMemory` | `None` | Anything exposing `find_by_entity()` and `batch_delete()` |
|
||||||
|
| `vector_store` | `VectorStore` | `memory.vector_store` | Store holding entity-keyed embeddings; pass `False` to disable the leg |
|
||||||
|
|
||||||
|
At least one store is required; a store that is not supplied reports
|
||||||
|
`not_configured` rather than being silently skipped.
|
||||||
|
|
||||||
|
### Methods
|
||||||
|
|
||||||
|
| Method | Returns | Description |
|
||||||
|
| :--- | :--- | :--- |
|
||||||
|
| `erase_entity(entity_id, reason, at, vector_ids)` | `ErasureReceipt` | Erase one entity from every bound store |
|
||||||
|
| `erase_entities(entity_ids, reason, at)` | `List[ErasureReceipt]` | One receipt per entity, in order; one failure does not stop the rest |
|
||||||
|
|
||||||
|
### Store Statuses
|
||||||
|
|
||||||
|
| Status | Meaning |
|
||||||
|
| :--- | :--- |
|
||||||
|
| `erased` | Reached, data removed. On the vectors leg this means the store accepted the delete for the ids given — backends offer no portable existence check, so it is not a count of embeddings that were really there |
|
||||||
|
| `not_found` | Reached, held nothing for this entity |
|
||||||
|
| `not_configured` | No such store was bound — normal, not a failure |
|
||||||
|
| `unsupported` | The store cannot delete at all; retrying will not help |
|
||||||
|
| `failed` | The store was reached and the deletion did not succeed |
|
||||||
|
|
||||||
|
### ErasureReceipt
|
||||||
|
|
||||||
|
| Member | Type | Description |
|
||||||
|
| :--- | :--- | :--- |
|
||||||
|
| `entity_id` | `str` | Entity the erasure was requested for |
|
||||||
|
| `reason` | `Optional[str]` | Recorded in the receipt and the graph tombstone |
|
||||||
|
| `erased_at` | `str` | ISO-8601; matches the tombstone's `purged_at` |
|
||||||
|
| `stores` | `Dict[str, Dict]` | Per-store outcome keyed `vectors`, `memory`, `graph` |
|
||||||
|
| `complete` | `bool` | `False` when any store reports `unsupported` or `failed` |
|
||||||
|
| `incomplete_stores` | `List[str]` | Stores that may still hold the entity's data |
|
||||||
|
| `to_dict()` | `Dict` | Serialized receipt, safe to persist as an audit record |
|
||||||
|
|
||||||
|
```python
|
||||||
|
receipt.to_dict()
|
||||||
|
# {
|
||||||
|
# "entity_id": "customer-4471",
|
||||||
|
# "reason": "GDPR Art. 17 request #882",
|
||||||
|
# "erased_at": "2026-08-16T09:03:36.813220",
|
||||||
|
# "complete": False,
|
||||||
|
# "stores": {
|
||||||
|
# "vectors": {"status": "unsupported", "backend": "faiss",
|
||||||
|
# "detail": "backend exposes no delete()/delete_vectors(); ..."},
|
||||||
|
# "memory": {"status": "erased", "items": 14},
|
||||||
|
# "graph": {"status": "erased", "nodes": 1, "edges": 3},
|
||||||
|
# },
|
||||||
|
# }
|
||||||
|
```
|
||||||
|
|
||||||
|
Erasure runs outward-in — vectors, then memory, then the graph. The tombstone is
|
||||||
|
the durable attestation that an erasure happened, so it is written last: a crash
|
||||||
|
mid-cascade leaves the node present and the receipt incomplete, rather than a
|
||||||
|
tombstone claiming more than actually happened. A store that raises is recorded
|
||||||
|
as `failed` and the remaining stores are still erased. Erasing the same entity
|
||||||
|
twice returns a receipt saying there was nothing left to do rather than raising.
|
||||||
|
|
||||||
|
|
||||||
## PolicyEngine
|
## PolicyEngine
|
||||||
|
|
||||||
`PolicyEngine` manages versioned policies stored in the knowledge graph. Policies are stored as nodes and can be linked to decisions:
|
`PolicyEngine` manages versioned policies stored in the knowledge graph. Policies are stored as nodes and can be linked to decisions:
|
||||||
|
|||||||
@@ -9,7 +9,7 @@
|
|||||||
"lint": "eslint .",
|
"lint": "eslint .",
|
||||||
"preview": "vite preview",
|
"preview": "vite preview",
|
||||||
"test:graph-store": "node --test tests/graphStore.multi-edge.test.mjs",
|
"test:graph-store": "node --test tests/graphStore.multi-edge.test.mjs",
|
||||||
"test:graph-workspace": "node --import tsx --test tests/markdownContentViewer.test.ts tests/graphSceneState.display.test.ts tests/temporalLifecycle.test.ts tests/deterministicExplorerRendering.test.ts",
|
"test:graph-workspace": "node --import tsx --test tests/markdownContentViewer.test.ts tests/graphSceneState.display.test.ts tests/temporalLifecycle.test.ts tests/deterministicExplorerRendering.test.ts tests/smallGraphLayout.test.ts tests/realtimeGraphAttributes.test.ts",
|
||||||
"test:deterministic-e2e": "node --import tsx --test tests/deterministicExplorerRendering.e2e.ts",
|
"test:deterministic-e2e": "node --import tsx --test tests/deterministicExplorerRendering.e2e.ts",
|
||||||
"test:plugin-registry": "node --import tsx --test tests/pluginRegistry.temporal.test.mjs"
|
"test:plugin-registry": "node --import tsx --test tests/pluginRegistry.temporal.test.mjs"
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -86,6 +86,7 @@ export interface EdgeAttributes {
|
|||||||
dominantEdgeType?: string;
|
dominantEdgeType?: string;
|
||||||
representativeWeight?: number;
|
representativeWeight?: number;
|
||||||
bundleKind?: "parallel" | "bidirectional" | "community";
|
bundleKind?: "parallel" | "bidirectional" | "community";
|
||||||
|
isSmallGraph?: boolean;
|
||||||
|
|
||||||
|
|
||||||
edgeType: string;
|
edgeType: string;
|
||||||
|
|||||||
@@ -21,7 +21,6 @@ import type Graph from "graphology";
|
|||||||
import { batchMergeEdges, batchMergeNodes, graph } from "../../store/graphStore";
|
import { batchMergeEdges, batchMergeNodes, graph } from "../../store/graphStore";
|
||||||
import { logEvent } from "../../store/registryStore";
|
import { logEvent } from "../../store/registryStore";
|
||||||
import type { EdgeAttributes, NodeAttributes } from "../../store/graphStore";
|
import type { EdgeAttributes, NodeAttributes } from "../../store/graphStore";
|
||||||
import { curveGroupForPair } from "../../store/edgePairKeys.js";
|
|
||||||
import { InspectorPanel, MetricChip, SurfaceCard } from "../../ui/primitives";
|
import { InspectorPanel, MetricChip, SurfaceCard } from "../../ui/primitives";
|
||||||
import { lazy, Suspense } from "react";
|
import { lazy, Suspense } from "react";
|
||||||
import { SigmaSceneAdapter } from "./SigmaSceneAdapter";
|
import { SigmaSceneAdapter } from "./SigmaSceneAdapter";
|
||||||
@@ -42,6 +41,8 @@ import {
|
|||||||
import { explorationEffectsShouldLoad, neighborhoodPanelShouldLoad, temporalOverlayShouldLoad } from "./pluginRegistryPredicates";
|
import { explorationEffectsShouldLoad, neighborhoodPanelShouldLoad, temporalOverlayShouldLoad } from "./pluginRegistryPredicates";
|
||||||
import { shouldFetchTemporalBounds, shouldFetchTemporalSnapshot } from "./temporalLifecyclePredicates";
|
import { shouldFetchTemporalBounds, shouldFetchTemporalSnapshot } from "./temporalLifecyclePredicates";
|
||||||
import { createTemporalSnapshotGuards, type TemporalSnapshotResponse } from "./temporalSnapshotGuards";
|
import { createTemporalSnapshotGuards, type TemporalSnapshotResponse } from "./temporalSnapshotGuards";
|
||||||
|
import { SMALL_GRAPH_MAX_NODES } from "./smallGraphLayout";
|
||||||
|
import { buildRealtimeEdgeAttributes } from "./realtimeGraphAttributes";
|
||||||
import type { LinkPrediction, PathResponse } from "./GraphInspectorPanel";
|
import type { LinkPrediction, PathResponse } from "./GraphInspectorPanel";
|
||||||
import type { GraphSceneHandle, GraphSceneRuntime } from "./scene";
|
import type { GraphSceneHandle, GraphSceneRuntime } from "./scene";
|
||||||
import type {
|
import type {
|
||||||
@@ -1056,46 +1057,10 @@ function buildRealtimeNodeAttributes(payload: {
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
function buildRealtimeEdgeAttributes(payload: {
|
function synchronizeRealtimeSmallGraphEdges(isSmallGraph: boolean): void {
|
||||||
id: string;
|
graph.forEachEdge((edgeId) => {
|
||||||
familyId?: string;
|
graph.setEdgeAttribute(edgeId, "isSmallGraph", isSmallGraph);
|
||||||
source_id: string;
|
});
|
||||||
target_id: string;
|
|
||||||
type?: string;
|
|
||||||
weight?: number;
|
|
||||||
properties?: Record<string, unknown>;
|
|
||||||
}): EdgeAttributes {
|
|
||||||
const properties = payload.properties || {};
|
|
||||||
const isInferred = Boolean(properties.inferred);
|
|
||||||
const isBidirectional = graph.hasDirectedEdge(payload.target_id, payload.source_id);
|
|
||||||
const baseColor = isInferred ? GRAPH_THEME.palette.accent.path : GRAPH_THEME.palette.muted.edgeStructure;
|
|
||||||
|
|
||||||
return {
|
|
||||||
edgeId: payload.id,
|
|
||||||
familyId: payload.familyId || payload.id,
|
|
||||||
sourceId: payload.source_id,
|
|
||||||
targetId: payload.target_id,
|
|
||||||
weight: Number(payload.weight ?? 1),
|
|
||||||
edgeType: payload.type || "related_to",
|
|
||||||
properties,
|
|
||||||
size: 1,
|
|
||||||
baseSize: 1,
|
|
||||||
color: baseColor,
|
|
||||||
baseColor,
|
|
||||||
mutedColor: GRAPH_THEME.palette.muted.edgeOverview,
|
|
||||||
visualPriority: isInferred ? 0.95 : 0.5,
|
|
||||||
isBidirectional,
|
|
||||||
edgeFamily: isInferred ? "path" : isBidirectional ? "bidirectional" : "line",
|
|
||||||
curveGroup: isBidirectional ? curveGroupForPair(payload.source_id, payload.target_id) : null,
|
|
||||||
type: "line",
|
|
||||||
edgeVariant: isInferred ? "pathSignal" : isBidirectional ? "bidirectionalCurve" : "directional",
|
|
||||||
arrowVisibilityPolicy: isInferred ? "always" : "contextual",
|
|
||||||
relationshipStrength: isInferred ? 0.95 : 0.52,
|
|
||||||
isParallelPair: false,
|
|
||||||
parallelIndex: 0,
|
|
||||||
parallelCount: 1,
|
|
||||||
familySize: 1,
|
|
||||||
};
|
|
||||||
}
|
}
|
||||||
|
|
||||||
function buildSelectedNodeState(
|
function buildSelectedNodeState(
|
||||||
@@ -1355,6 +1320,7 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
|||||||
const lastExternalFocusTokenRef = useRef<number | undefined>(undefined);
|
const lastExternalFocusTokenRef = useRef<number | undefined>(undefined);
|
||||||
const pluginRuntimeRef = useRef<GraphSceneRuntime | null>(null);
|
const pluginRuntimeRef = useRef<GraphSceneRuntime | null>(null);
|
||||||
const appliedGraphSummarySignatureRef = useRef<string | null>(null);
|
const appliedGraphSummarySignatureRef = useRef<string | null>(null);
|
||||||
|
const smallGraphModeRef = useRef(false);
|
||||||
const pluginInteractionStateRef = useRef<GraphInteractionState>({
|
const pluginInteractionStateRef = useRef<GraphInteractionState>({
|
||||||
hoveredNodeId: null,
|
hoveredNodeId: null,
|
||||||
selectedNodeId: "",
|
selectedNodeId: "",
|
||||||
@@ -1382,6 +1348,12 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
|||||||
}
|
}
|
||||||
|
|
||||||
appliedGraphSummarySignatureRef.current = signature;
|
appliedGraphSummarySignatureRef.current = signature;
|
||||||
|
smallGraphModeRef.current = Boolean(
|
||||||
|
graphSummary.layoutReady
|
||||||
|
&& !graphSummary.hasCoordinates
|
||||||
|
&& graphSummary.nodeCount > 0
|
||||||
|
&& graphSummary.nodeCount <= SMALL_GRAPH_MAX_NODES,
|
||||||
|
);
|
||||||
setGraphReady(true);
|
setGraphReady(true);
|
||||||
setGraphVersion((current) => current + 1);
|
setGraphVersion((current) => current + 1);
|
||||||
setIsLayoutRunning(!graphSummary.layoutReady);
|
setIsLayoutRunning(!graphSummary.layoutReady);
|
||||||
@@ -1893,18 +1865,26 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
|||||||
attributes: buildRealtimeNodeAttributes(payload),
|
attributes: buildRealtimeNodeAttributes(payload),
|
||||||
},
|
},
|
||||||
]);
|
]);
|
||||||
|
if (graph.order > SMALL_GRAPH_MAX_NODES) {
|
||||||
|
smallGraphModeRef.current = false;
|
||||||
|
}
|
||||||
|
synchronizeRealtimeSmallGraphEdges(smallGraphModeRef.current);
|
||||||
logEvent("add-node", `Added node ${payload.label ?? payload.id}${payload.nodeType ? ` (${payload.nodeType})` : ""} via realtime ws`, { nodeId: payload.id, nodeType: payload.nodeType });
|
logEvent("add-node", `Added node ${payload.label ?? payload.id}${payload.nodeType ? ` (${payload.nodeType})` : ""} via realtime ws`, { nodeId: payload.id, nodeType: payload.nodeType });
|
||||||
setGraphVersion((current) => current + 1);
|
setGraphVersion((current) => current + 1);
|
||||||
sceneRef.current?.getRuntime()?.requestRender();
|
sceneRef.current?.getRuntime()?.requestRender();
|
||||||
}
|
}
|
||||||
if (eventType === "ADD_EDGE") {
|
if (eventType === "ADD_EDGE") {
|
||||||
|
const isSmallGraph = smallGraphModeRef.current;
|
||||||
batchMergeEdges([
|
batchMergeEdges([
|
||||||
{
|
{
|
||||||
id: String(payload.id),
|
id: String(payload.id),
|
||||||
familyId: payload.familyId ? String(payload.familyId) : String(payload.id),
|
familyId: payload.familyId ? String(payload.familyId) : String(payload.id),
|
||||||
source: payload.source_id,
|
source: payload.source_id,
|
||||||
target: payload.target_id,
|
target: payload.target_id,
|
||||||
attributes: buildRealtimeEdgeAttributes(payload),
|
attributes: buildRealtimeEdgeAttributes(payload, {
|
||||||
|
isBidirectional: graph.hasDirectedEdge(payload.target_id, payload.source_id),
|
||||||
|
isSmallGraph,
|
||||||
|
}),
|
||||||
},
|
},
|
||||||
]);
|
]);
|
||||||
logEvent("add-edge", `Added edge ${payload.edgeType ?? payload.id} (${payload.source_id} → ${payload.target_id}) via realtime ws`, { edgeId: payload.id, edgeType: payload.edgeType, source: payload.source_id, target: payload.target_id });
|
logEvent("add-edge", `Added edge ${payload.edgeType ?? payload.id} (${payload.source_id} → ${payload.target_id}) via realtime ws`, { edgeId: payload.id, edgeType: payload.edgeType, source: payload.source_id, target: payload.target_id });
|
||||||
|
|||||||
@@ -1783,6 +1783,7 @@ export function resolveEdgeElementStyle(
|
|||||||
const isCommunityBundle = attrs.bundleKind === "community";
|
const isCommunityBundle = attrs.bundleKind === "community";
|
||||||
const baseSize = Number(attrs.baseSize || attrs.size || 0.9);
|
const baseSize = Number(attrs.baseSize || attrs.size || 0.9);
|
||||||
const visualPriority = Number(attrs.visualPriority ?? 0);
|
const visualPriority = Number(attrs.visualPriority ?? 0);
|
||||||
|
const isSmallGraphEdge = viewMode === "full" && attrs.isSmallGraph === true;
|
||||||
const isFullBridgeEdge = viewMode === "full" && fullEdgeClass === "bridge";
|
const isFullBridgeEdge = viewMode === "full" && fullEdgeClass === "bridge";
|
||||||
const isFullBackboneEdge = viewMode === "full" && fullEdgeClass === "backbone";
|
const isFullBackboneEdge = viewMode === "full" && fullEdgeClass === "backbone";
|
||||||
const shouldCurveBridge = isFullBridgeEdge
|
const shouldCurveBridge = isFullBridgeEdge
|
||||||
@@ -1790,11 +1791,13 @@ export function resolveEdgeElementStyle(
|
|||||||
const visibilityPolicy = resolveEdgeVisibilityPolicy(theme, viewMode, zoomTier, isCommunityBundle);
|
const visibilityPolicy = resolveEdgeVisibilityPolicy(theme, viewMode, zoomTier, isCommunityBundle);
|
||||||
const isContextEdge = isContextEdgeState(state);
|
const isContextEdge = isContextEdgeState(state);
|
||||||
const isNonCriticalEdge = isNonCriticalEdgeVariant(edgeVariant);
|
const isNonCriticalEdge = isNonCriticalEdgeVariant(edgeVariant);
|
||||||
const belowPriorityThreshold = state === "default"
|
const belowPriorityThreshold = !isSmallGraphEdge && state === "default"
|
||||||
&& visualPriority < Math.max(tierConfig.edgePriorityThreshold, visibilityPolicy.defaultPriorityThreshold)
|
&& visualPriority < Math.max(tierConfig.edgePriorityThreshold, visibilityPolicy.defaultPriorityThreshold)
|
||||||
&& isNonCriticalEdge;
|
&& isNonCriticalEdge;
|
||||||
const hiddenByMutedState = (state === "muted" || state === "inactive") && visibilityPolicy.hideMuted;
|
const hiddenByMutedState = !isSmallGraphEdge
|
||||||
const sampledOut = isNonCriticalEdge
|
&& (state === "muted" || state === "inactive")
|
||||||
|
&& visibilityPolicy.hideMuted;
|
||||||
|
const sampledOut = !isSmallGraphEdge && isNonCriticalEdge
|
||||||
&& (
|
&& (
|
||||||
(state === "default" && !isContextEdge && shouldSampleOutBackgroundEdge(visibilityPolicy.backgroundSampleRate, visualPriority, edgeId, sourceId, targetId))
|
(state === "default" && !isContextEdge && shouldSampleOutBackgroundEdge(visibilityPolicy.backgroundSampleRate, visualPriority, edgeId, sourceId, targetId))
|
||||||
|| (
|
|| (
|
||||||
@@ -1837,11 +1840,14 @@ export function resolveEdgeElementStyle(
|
|||||||
? resolveEdgeCurvature(theme, state, edgeVariant, attrs, sourceId, targetId)
|
? resolveEdgeCurvature(theme, state, edgeVariant, attrs, sourceId, targetId)
|
||||||
: 0;
|
: 0;
|
||||||
const baseColor = resolveEdgeColor(theme, zoomTier, state, attrs, attrs.color, fullEdgeClass);
|
const baseColor = resolveEdgeColor(theme, zoomTier, state, attrs, attrs.color, fullEdgeClass);
|
||||||
const lodAlpha = resolveEdgeLodAlpha(theme, viewMode, zoomTier, state, attrs, isCommunityBundle, fullEdgeClass);
|
const resolvedLodAlpha = resolveEdgeLodAlpha(theme, viewMode, zoomTier, state, attrs, isCommunityBundle, fullEdgeClass);
|
||||||
|
const lodAlpha = isSmallGraphEdge
|
||||||
|
? Math.max(resolvedLodAlpha ?? 1, isContextEdge ? 0.62 : 0.46)
|
||||||
|
: resolvedLodAlpha;
|
||||||
const color = lodAlpha === null ? baseColor : withAlpha(baseColor, lodAlpha);
|
const color = lodAlpha === null ? baseColor : withAlpha(baseColor, lodAlpha);
|
||||||
const rawSize = Math.max(
|
const rawSize = Math.max(
|
||||||
baseSize * sizeMultiplier * (isCommunityBundle ? theme.grouped.style.edgeSizeScale : 1),
|
baseSize * sizeMultiplier * (isCommunityBundle ? theme.grouped.style.edgeSizeScale : 1),
|
||||||
stateConfig.minSize,
|
isSmallGraphEdge ? Math.max(stateConfig.minSize, 0.9) : stateConfig.minSize,
|
||||||
);
|
);
|
||||||
|
|
||||||
const interactionMaxSize = (fullEdgeClass === "path" || state === "path")
|
const interactionMaxSize = (fullEdgeClass === "path" || state === "path")
|
||||||
|
|||||||
@@ -0,0 +1,50 @@
|
|||||||
|
import type { EdgeAttributes } from "../../store/graphStore";
|
||||||
|
import { curveGroupForPair } from "../../store/edgePairKeys.js";
|
||||||
|
import { GRAPH_THEME } from "./graphTheme";
|
||||||
|
|
||||||
|
export type RealtimeEdgePayload = {
|
||||||
|
id: string;
|
||||||
|
familyId?: string;
|
||||||
|
source_id: string;
|
||||||
|
target_id: string;
|
||||||
|
type?: string;
|
||||||
|
weight?: number;
|
||||||
|
properties?: Record<string, unknown>;
|
||||||
|
};
|
||||||
|
|
||||||
|
export function buildRealtimeEdgeAttributes(
|
||||||
|
payload: RealtimeEdgePayload,
|
||||||
|
options: { isBidirectional: boolean; isSmallGraph: boolean },
|
||||||
|
): EdgeAttributes {
|
||||||
|
const properties = payload.properties || {};
|
||||||
|
const isInferred = Boolean(properties.inferred);
|
||||||
|
const baseColor = isInferred ? GRAPH_THEME.palette.accent.path : GRAPH_THEME.palette.muted.edgeStructure;
|
||||||
|
|
||||||
|
return {
|
||||||
|
edgeId: payload.id,
|
||||||
|
familyId: payload.familyId || payload.id,
|
||||||
|
sourceId: payload.source_id,
|
||||||
|
targetId: payload.target_id,
|
||||||
|
weight: Number(payload.weight ?? 1),
|
||||||
|
edgeType: payload.type || "related_to",
|
||||||
|
properties,
|
||||||
|
size: 1,
|
||||||
|
baseSize: 1,
|
||||||
|
color: baseColor,
|
||||||
|
baseColor,
|
||||||
|
mutedColor: GRAPH_THEME.palette.muted.edgeOverview,
|
||||||
|
visualPriority: isInferred ? 0.95 : 0.5,
|
||||||
|
isBidirectional: options.isBidirectional,
|
||||||
|
edgeFamily: isInferred ? "path" : options.isBidirectional ? "bidirectional" : "line",
|
||||||
|
curveGroup: options.isBidirectional ? curveGroupForPair(payload.source_id, payload.target_id) : null,
|
||||||
|
type: "line",
|
||||||
|
edgeVariant: isInferred ? "pathSignal" : options.isBidirectional ? "bidirectionalCurve" : "directional",
|
||||||
|
arrowVisibilityPolicy: isInferred ? "always" : "contextual",
|
||||||
|
relationshipStrength: isInferred ? 0.95 : 0.52,
|
||||||
|
isParallelPair: false,
|
||||||
|
parallelIndex: 0,
|
||||||
|
parallelCount: 1,
|
||||||
|
familySize: 1,
|
||||||
|
isSmallGraph: options.isSmallGraph,
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
export const SMALL_GRAPH_MAX_NODES = 48;
|
||||||
|
const PROVIDED_COORDINATE_COVERAGE = 0.92;
|
||||||
|
const MAX_COMPONENT_RADIUS = 78;
|
||||||
|
const COMPONENT_GAP = 48;
|
||||||
|
|
||||||
|
type LayoutEdge = {
|
||||||
|
source: string;
|
||||||
|
target: string;
|
||||||
|
};
|
||||||
|
|
||||||
|
export function shouldUseSmallGraphLayout(nodeCount: number, coordinateCoverage: number): boolean {
|
||||||
|
return nodeCount > 0
|
||||||
|
&& nodeCount <= SMALL_GRAPH_MAX_NODES
|
||||||
|
&& coordinateCoverage < PROVIDED_COORDINATE_COVERAGE;
|
||||||
|
}
|
||||||
|
|
||||||
|
export function resolveGraphLayoutDecision(nodeCount: number, coordinateCoverage: number): {
|
||||||
|
useProvidedCoordinates: boolean;
|
||||||
|
useSmallGraphLayout: boolean;
|
||||||
|
layoutReady: boolean;
|
||||||
|
} {
|
||||||
|
const useProvidedCoordinates = coordinateCoverage >= PROVIDED_COORDINATE_COVERAGE;
|
||||||
|
const useSmallGraphLayout = shouldUseSmallGraphLayout(nodeCount, coordinateCoverage);
|
||||||
|
return {
|
||||||
|
useProvidedCoordinates,
|
||||||
|
useSmallGraphLayout,
|
||||||
|
layoutReady: useProvidedCoordinates || useSmallGraphLayout,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
export function resolveNodeLayoutPosition(
|
||||||
|
decision: ReturnType<typeof resolveGraphLayoutDecision>,
|
||||||
|
provided: { x: number | null; y: number | null },
|
||||||
|
seeded: { x: number; y: number } | undefined,
|
||||||
|
): { x: number; y: number } {
|
||||||
|
if (decision.useProvidedCoordinates) {
|
||||||
|
return { x: provided.x ?? 0, y: provided.y ?? 0 };
|
||||||
|
}
|
||||||
|
if (decision.useSmallGraphLayout) {
|
||||||
|
return { x: seeded?.x ?? 0, y: seeded?.y ?? 0 };
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
x: provided.x ?? seeded?.x ?? 0,
|
||||||
|
y: provided.y ?? seeded?.y ?? 0,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Produce a compact deterministic layout for small graphs.
|
||||||
|
*
|
||||||
|
* ForceAtlas2 is useful for large connected datasets, but it makes tiny graphs
|
||||||
|
* with several disconnected components look like scattered dots. This layout
|
||||||
|
* keeps each connected component together and packs components into a centered
|
||||||
|
* grid so instance relationships remain legible on first render.
|
||||||
|
*/
|
||||||
|
export function buildSmallGraphSeedPositions(
|
||||||
|
nodeIds: string[],
|
||||||
|
edges: LayoutEdge[],
|
||||||
|
): Map<string, { x: number; y: number }> {
|
||||||
|
const ids = [...new Set(nodeIds)].sort((left, right) => left.localeCompare(right));
|
||||||
|
const adjacency = new Map(ids.map((id) => [id, new Set<string>()]));
|
||||||
|
|
||||||
|
edges.forEach(({ source, target }) => {
|
||||||
|
if (!adjacency.has(source) || !adjacency.has(target) || source === target) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
adjacency.get(source)?.add(target);
|
||||||
|
adjacency.get(target)?.add(source);
|
||||||
|
});
|
||||||
|
|
||||||
|
const visited = new Set<string>();
|
||||||
|
const components: string[][] = [];
|
||||||
|
ids.forEach((start) => {
|
||||||
|
if (visited.has(start)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const component: string[] = [];
|
||||||
|
const queue = [start];
|
||||||
|
visited.add(start);
|
||||||
|
while (queue.length > 0) {
|
||||||
|
const current = queue.shift();
|
||||||
|
if (!current) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
component.push(current);
|
||||||
|
[...(adjacency.get(current) ?? [])]
|
||||||
|
.sort((left, right) => left.localeCompare(right))
|
||||||
|
.forEach((neighbor) => {
|
||||||
|
if (!visited.has(neighbor)) {
|
||||||
|
visited.add(neighbor);
|
||||||
|
queue.push(neighbor);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
component.sort((left, right) => {
|
||||||
|
const degreeDelta = (adjacency.get(right)?.size ?? 0) - (adjacency.get(left)?.size ?? 0);
|
||||||
|
return degreeDelta || left.localeCompare(right);
|
||||||
|
});
|
||||||
|
components.push(component);
|
||||||
|
});
|
||||||
|
|
||||||
|
components.sort((left, right) => right.length - left.length || left[0].localeCompare(right[0]));
|
||||||
|
|
||||||
|
const columns = Math.max(1, Math.ceil(Math.sqrt(components.length)));
|
||||||
|
const rows = Math.max(1, Math.ceil(components.length / columns));
|
||||||
|
// Adjacent cells must leave room for two maximum-radius components plus a
|
||||||
|
// readable gap. A smaller row height allows valid 12-node components to
|
||||||
|
// overlap vertically.
|
||||||
|
const cellWidth = MAX_COMPONENT_RADIUS * 2 + COMPONENT_GAP;
|
||||||
|
const cellHeight = MAX_COMPONENT_RADIUS * 2 + COMPONENT_GAP;
|
||||||
|
const positions = new Map<string, { x: number; y: number }>();
|
||||||
|
|
||||||
|
components.forEach((component, componentIndex) => {
|
||||||
|
const column = componentIndex % columns;
|
||||||
|
const row = Math.floor(componentIndex / columns);
|
||||||
|
const centerX = (column - (columns - 1) / 2) * cellWidth;
|
||||||
|
const centerY = (row - (rows - 1) / 2) * cellHeight;
|
||||||
|
|
||||||
|
if (component.length === 1) {
|
||||||
|
positions.set(component[0], { x: centerX, y: centerY });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
const radius = Math.min(MAX_COMPONENT_RADIUS, 30 + component.length * 9);
|
||||||
|
component.forEach((nodeId, nodeIndex) => {
|
||||||
|
const angle = -Math.PI / 2 + (nodeIndex * Math.PI * 2) / component.length;
|
||||||
|
positions.set(nodeId, {
|
||||||
|
x: centerX + Math.cos(angle) * radius,
|
||||||
|
y: centerY + Math.sin(angle) * radius,
|
||||||
|
});
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
return positions;
|
||||||
|
}
|
||||||
@@ -16,6 +16,11 @@ import {
|
|||||||
} from "./graphTheme";
|
} from "./graphTheme";
|
||||||
import { classifyEntityShape } from "./graphEntityShape";
|
import { classifyEntityShape } from "./graphEntityShape";
|
||||||
import { createGraphLoadProgress } from "./graphLoading";
|
import { createGraphLoadProgress } from "./graphLoading";
|
||||||
|
import {
|
||||||
|
buildSmallGraphSeedPositions,
|
||||||
|
resolveGraphLayoutDecision,
|
||||||
|
resolveNodeLayoutPosition,
|
||||||
|
} from "./smallGraphLayout";
|
||||||
import type { GraphLoadProgress, GraphLoadSummary } from "./types";
|
import type { GraphLoadProgress, GraphLoadSummary } from "./types";
|
||||||
|
|
||||||
const SEMANTIC_COLOR_FIELDS = [
|
const SEMANTIC_COLOR_FIELDS = [
|
||||||
@@ -553,10 +558,19 @@ export function useLoadGraph(options: UseLoadGraphOptions = {}) {
|
|||||||
: count;
|
: count;
|
||||||
}, 0);
|
}, 0);
|
||||||
const coordinateCoverage = fetchedNodes.length > 0 ? providedCoordinateCount / fetchedNodes.length : 0;
|
const coordinateCoverage = fetchedNodes.length > 0 ? providedCoordinateCount / fetchedNodes.length : 0;
|
||||||
const useProvidedCoordinates = coordinateCoverage >= 0.92;
|
const {
|
||||||
|
useProvidedCoordinates,
|
||||||
|
useSmallGraphLayout,
|
||||||
|
layoutReady,
|
||||||
|
} = resolveGraphLayoutDecision(fetchedNodes.length, coordinateCoverage);
|
||||||
const seededPositions = useProvidedCoordinates
|
const seededPositions = useProvidedCoordinates
|
||||||
? null
|
? null
|
||||||
: buildClusterSeedPositions(
|
: useSmallGraphLayout
|
||||||
|
? buildSmallGraphSeedPositions(
|
||||||
|
fetchedNodes.map((node) => node.id),
|
||||||
|
fetchedEdges,
|
||||||
|
)
|
||||||
|
: buildClusterSeedPositions(
|
||||||
draftAttributes.map(({ id, attributes }) => ({
|
draftAttributes.map(({ id, attributes }) => ({
|
||||||
id,
|
id,
|
||||||
semanticGroup: semanticKeyByNodeId.get(id) ?? structuralColorKey(id, attributes),
|
semanticGroup: semanticKeyByNodeId.get(id) ?? structuralColorKey(id, attributes),
|
||||||
@@ -569,7 +583,9 @@ export function useLoadGraph(options: UseLoadGraphOptions = {}) {
|
|||||||
const colorIndex = hashString(semanticGroup) % GRAPH_THEME.palette.semantic.length;
|
const colorIndex = hashString(semanticGroup) % GRAPH_THEME.palette.semantic.length;
|
||||||
const baseColor = GRAPH_THEME.palette.semantic[colorIndex];
|
const baseColor = GRAPH_THEME.palette.semantic[colorIndex];
|
||||||
const sizeRatio = nodePriorityById.get(id) ?? 0;
|
const sizeRatio = nodePriorityById.get(id) ?? 0;
|
||||||
const dynamicSize = clamp(1.8, 1.8 + 8.8 * sizeRatio, 11.8);
|
const dynamicSize = useSmallGraphLayout
|
||||||
|
? clamp(5.2, 5.2 + 6.6 * sizeRatio, 11.8)
|
||||||
|
: clamp(1.8, 1.8 + 8.8 * sizeRatio, 11.8);
|
||||||
const hasTemporalBounds = Boolean(attributes.valid_from || attributes.valid_until);
|
const hasTemporalBounds = Boolean(attributes.valid_from || attributes.valid_until);
|
||||||
const provenanceCount = getProvenanceCount(attributes.properties ?? {});
|
const provenanceCount = getProvenanceCount(attributes.properties ?? {});
|
||||||
const properties = attributes.properties as Record<string, unknown>;
|
const properties = attributes.properties as Record<string, unknown>;
|
||||||
@@ -577,12 +593,11 @@ export function useLoadGraph(options: UseLoadGraphOptions = {}) {
|
|||||||
const providedX = readFiniteCoordinate(properties?.x);
|
const providedX = readFiniteCoordinate(properties?.x);
|
||||||
const providedY = readFiniteCoordinate(properties?.y);
|
const providedY = readFiniteCoordinate(properties?.y);
|
||||||
const seededPosition = seededPositions?.get(id);
|
const seededPosition = seededPositions?.get(id);
|
||||||
const x = useProvidedCoordinates
|
const { x, y } = resolveNodeLayoutPosition(
|
||||||
? providedX ?? 0
|
{ useProvidedCoordinates, useSmallGraphLayout, layoutReady },
|
||||||
: providedX ?? seededPosition?.x ?? 0;
|
{ x: providedX, y: providedY },
|
||||||
const y = useProvidedCoordinates
|
seededPosition,
|
||||||
? providedY ?? 0
|
);
|
||||||
: providedY ?? seededPosition?.y ?? 0;
|
|
||||||
return {
|
return {
|
||||||
id,
|
id,
|
||||||
attributes: {
|
attributes: {
|
||||||
@@ -603,6 +618,7 @@ export function useLoadGraph(options: UseLoadGraphOptions = {}) {
|
|||||||
borderSize: 0.72,
|
borderSize: 0.72,
|
||||||
entityShape,
|
entityShape,
|
||||||
...resolveNodeVariantMetadata(baseColor, sizeRatio, hasTemporalBounds, provenanceCount),
|
...resolveNodeVariantMetadata(baseColor, sizeRatio, hasTemporalBounds, provenanceCount),
|
||||||
|
...(useSmallGraphLayout ? { labelVisibilityPolicy: "always" as const } : {}),
|
||||||
} as NodeAttributes,
|
} as NodeAttributes,
|
||||||
};
|
};
|
||||||
});
|
});
|
||||||
@@ -659,6 +675,7 @@ export function useLoadGraph(options: UseLoadGraphOptions = {}) {
|
|||||||
parallelIndex,
|
parallelIndex,
|
||||||
parallelCount,
|
parallelCount,
|
||||||
familySize: familyCounts.get(edge.familyId) ?? 1,
|
familySize: familyCounts.get(edge.familyId) ?? 1,
|
||||||
|
isSmallGraph: useSmallGraphLayout,
|
||||||
...resolveEdgeVariantMetadata(edge, sourcePriority, targetPriority, isBidirectional),
|
...resolveEdgeVariantMetadata(edge, sourcePriority, targetPriority, isBidirectional),
|
||||||
} as EdgeAttributes,
|
} as EdgeAttributes,
|
||||||
};
|
};
|
||||||
@@ -701,7 +718,7 @@ export function useLoadGraph(options: UseLoadGraphOptions = {}) {
|
|||||||
loadTimeMs: Math.round(performance.now() - startedAt),
|
loadTimeMs: Math.round(performance.now() - startedAt),
|
||||||
hasCoordinates: useProvidedCoordinates,
|
hasCoordinates: useProvidedCoordinates,
|
||||||
layoutSource: useProvidedCoordinates ? "provided" : "runtime",
|
layoutSource: useProvidedCoordinates ? "provided" : "runtime",
|
||||||
layoutReady: useProvidedCoordinates,
|
layoutReady,
|
||||||
} satisfies GraphLoadSummary;
|
} satisfies GraphLoadSummary;
|
||||||
|
|
||||||
onProgress?.(createGraphLoadProgress({
|
onProgress?.(createGraphLoadProgress({
|
||||||
|
|||||||
@@ -407,6 +407,31 @@ test("resolveEdgeElementStyle applies full-graph LOD to directional background e
|
|||||||
assert.equal(style.hidden, true);
|
assert.equal(style.hidden, true);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test("resolveEdgeElementStyle keeps small-graph relationships visible in overview", () => {
|
||||||
|
const style = resolveEdgeElementStyle(
|
||||||
|
GRAPH_THEME,
|
||||||
|
"overview",
|
||||||
|
"inactive",
|
||||||
|
{
|
||||||
|
edgeType: "related_to",
|
||||||
|
weight: 1,
|
||||||
|
properties: {},
|
||||||
|
edgeVariant: "directional",
|
||||||
|
visualPriority: 0.1,
|
||||||
|
baseSize: 0.5,
|
||||||
|
isSmallGraph: true,
|
||||||
|
},
|
||||||
|
"source",
|
||||||
|
"target",
|
||||||
|
"full",
|
||||||
|
"small-graph-low-priority",
|
||||||
|
"hidden",
|
||||||
|
);
|
||||||
|
|
||||||
|
assert.equal(style.hidden, false);
|
||||||
|
assert.ok(Number(style.size ?? 0) >= 0.9);
|
||||||
|
});
|
||||||
|
|
||||||
test("classifyFullGraphEdge applies deterministic priority order", () => {
|
test("classifyFullGraphEdge applies deterministic priority order", () => {
|
||||||
const edgeClass = classifyFullGraphEdge(
|
const edgeClass = classifyFullGraphEdge(
|
||||||
"edge-priority",
|
"edge-priority",
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
import assert from "node:assert/strict";
|
||||||
|
import test from "node:test";
|
||||||
|
|
||||||
|
import { buildRealtimeEdgeAttributes } from "../src/workspaces/GraphWorkspace/realtimeGraphAttributes.ts";
|
||||||
|
|
||||||
|
const payload = {
|
||||||
|
id: "edge-live",
|
||||||
|
source_id: "source",
|
||||||
|
target_id: "target",
|
||||||
|
type: "related_to",
|
||||||
|
properties: {},
|
||||||
|
};
|
||||||
|
|
||||||
|
test("realtime edges retain the active small-graph visibility marker", () => {
|
||||||
|
const attributes = buildRealtimeEdgeAttributes(payload, {
|
||||||
|
isBidirectional: false,
|
||||||
|
isSmallGraph: true,
|
||||||
|
});
|
||||||
|
|
||||||
|
assert.equal(attributes.isSmallGraph, true);
|
||||||
|
assert.equal(attributes.edgeVariant, "directional");
|
||||||
|
});
|
||||||
|
|
||||||
|
test("realtime edges do not retain the marker after graph leaves small-graph mode", () => {
|
||||||
|
const attributes = buildRealtimeEdgeAttributes(payload, {
|
||||||
|
isBidirectional: false,
|
||||||
|
isSmallGraph: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
assert.equal(attributes.isSmallGraph, false);
|
||||||
|
});
|
||||||
@@ -0,0 +1,100 @@
|
|||||||
|
import assert from "node:assert/strict";
|
||||||
|
import test from "node:test";
|
||||||
|
|
||||||
|
import {
|
||||||
|
SMALL_GRAPH_MAX_NODES,
|
||||||
|
buildSmallGraphSeedPositions,
|
||||||
|
resolveGraphLayoutDecision,
|
||||||
|
resolveNodeLayoutPosition,
|
||||||
|
shouldUseSmallGraphLayout,
|
||||||
|
} from "../src/workspaces/GraphWorkspace/smallGraphLayout.ts";
|
||||||
|
|
||||||
|
test("small graph layout is selected only when coordinates are not already usable", () => {
|
||||||
|
assert.equal(shouldUseSmallGraphLayout(12, 0), true);
|
||||||
|
assert.equal(shouldUseSmallGraphLayout(SMALL_GRAPH_MAX_NODES + 1, 0), false);
|
||||||
|
assert.equal(shouldUseSmallGraphLayout(12, 0.95), false);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("small graph layout ignores isolated partial coordinates", () => {
|
||||||
|
const decision = resolveGraphLayoutDecision(12, 1 / 12);
|
||||||
|
assert.deepEqual(
|
||||||
|
resolveNodeLayoutPosition(decision, { x: 50_000, y: -50_000 }, { x: 24, y: -18 }),
|
||||||
|
{ x: 24, y: -18 },
|
||||||
|
);
|
||||||
|
assert.deepEqual(
|
||||||
|
resolveNodeLayoutPosition(decision, { x: 50_000, y: null }, { x: -12, y: 36 }),
|
||||||
|
{ x: -12, y: 36 },
|
||||||
|
);
|
||||||
|
});
|
||||||
|
|
||||||
|
test("small graph load is immediately ready and skips runtime stabilization", () => {
|
||||||
|
assert.deepEqual(resolveGraphLayoutDecision(12, 0), {
|
||||||
|
useProvidedCoordinates: false,
|
||||||
|
useSmallGraphLayout: true,
|
||||||
|
layoutReady: true,
|
||||||
|
});
|
||||||
|
assert.deepEqual(resolveGraphLayoutDecision(SMALL_GRAPH_MAX_NODES + 1, 0), {
|
||||||
|
useProvidedCoordinates: false,
|
||||||
|
useSmallGraphLayout: false,
|
||||||
|
layoutReady: false,
|
||||||
|
});
|
||||||
|
assert.deepEqual(resolveGraphLayoutDecision(12, 1), {
|
||||||
|
useProvidedCoordinates: true,
|
||||||
|
useSmallGraphLayout: false,
|
||||||
|
layoutReady: true,
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
test("small graph layout is deterministic and keeps connected nodes together", () => {
|
||||||
|
const nodes = ["Apple", "Steve", "Ronald", "Cupertino", "California"];
|
||||||
|
const edges = [
|
||||||
|
{ source: "Apple", target: "Steve" },
|
||||||
|
{ source: "Ronald", target: "Cupertino" },
|
||||||
|
];
|
||||||
|
const first = buildSmallGraphSeedPositions(nodes, edges);
|
||||||
|
const second = buildSmallGraphSeedPositions([...nodes].reverse(), [...edges].reverse());
|
||||||
|
|
||||||
|
assert.deepEqual([...first.entries()].sort(), [...second.entries()].sort());
|
||||||
|
assert.equal(first.size, nodes.length);
|
||||||
|
|
||||||
|
const distance = (left: string, right: string) => {
|
||||||
|
const a = first.get(left);
|
||||||
|
const b = first.get(right);
|
||||||
|
assert.ok(a && b);
|
||||||
|
return Math.hypot(a.x - b.x, a.y - b.y);
|
||||||
|
};
|
||||||
|
assert.ok(distance("Apple", "Steve") < distance("Apple", "California"));
|
||||||
|
assert.ok(distance("Ronald", "Cupertino") < distance("Ronald", "California"));
|
||||||
|
});
|
||||||
|
|
||||||
|
test("small graph layout keeps maximum-radius components separated", () => {
|
||||||
|
const componentCount = 4;
|
||||||
|
const nodesPerComponent = 12;
|
||||||
|
const nodes = Array.from(
|
||||||
|
{ length: componentCount * nodesPerComponent },
|
||||||
|
(_, index) => `component-${Math.floor(index / nodesPerComponent)}-node-${index % nodesPerComponent}`,
|
||||||
|
);
|
||||||
|
const edges = Array.from({ length: componentCount }).flatMap((_, componentIndex) => {
|
||||||
|
const prefix = `component-${componentIndex}-node-`;
|
||||||
|
return Array.from({ length: nodesPerComponent - 1 }, (_unused, nodeIndex) => ({
|
||||||
|
source: `${prefix}${nodeIndex}`,
|
||||||
|
target: `${prefix}${nodeIndex + 1}`,
|
||||||
|
}));
|
||||||
|
});
|
||||||
|
const positions = buildSmallGraphSeedPositions(nodes, edges);
|
||||||
|
|
||||||
|
for (let leftComponent = 0; leftComponent < componentCount; leftComponent += 1) {
|
||||||
|
for (let rightComponent = leftComponent + 1; rightComponent < componentCount; rightComponent += 1) {
|
||||||
|
let closestDistance = Number.POSITIVE_INFINITY;
|
||||||
|
for (let leftNode = 0; leftNode < nodesPerComponent; leftNode += 1) {
|
||||||
|
for (let rightNode = 0; rightNode < nodesPerComponent; rightNode += 1) {
|
||||||
|
const left = positions.get(`component-${leftComponent}-node-${leftNode}`);
|
||||||
|
const right = positions.get(`component-${rightComponent}-node-${rightNode}`);
|
||||||
|
assert.ok(left && right);
|
||||||
|
closestDistance = Math.min(closestDistance, Math.hypot(left.x - right.x, left.y - right.y));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert.ok(closestDistance >= 48, `components are only ${closestDistance} units apart`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
@@ -25,5 +25,9 @@
|
|||||||
"mcp"
|
"mcp"
|
||||||
],
|
],
|
||||||
"skills": "./skills",
|
"skills": "./skills",
|
||||||
"agents": "./agents"
|
"agents": [
|
||||||
|
"./agents/decision-advisor.md",
|
||||||
|
"./agents/explainability.md",
|
||||||
|
"./agents/kg-assistant.md"
|
||||||
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -111,6 +111,7 @@ from .context_graph import ContextEdge, ContextGraph, ContextNode
|
|||||||
from .context_retriever import ContextRetriever, RetrievedContext, TemporalGraphRetriever
|
from .context_retriever import ContextRetriever, RetrievedContext, TemporalGraphRetriever
|
||||||
from .decision_context import DecisionContext
|
from .decision_context import DecisionContext
|
||||||
from .entity_linker import EntityLink, EntityLinker, LinkedEntity
|
from .entity_linker import EntityLink, EntityLinker, LinkedEntity
|
||||||
|
from .erasure import ErasureCoordinator, ErasureReceipt
|
||||||
|
|
||||||
# Decision tracking imports
|
# Decision tracking imports
|
||||||
from .decision_models import (
|
from .decision_models import (
|
||||||
@@ -145,6 +146,9 @@ __all__ = [
|
|||||||
"ContextRetriever",
|
"ContextRetriever",
|
||||||
"RetrievedContext",
|
"RetrievedContext",
|
||||||
"TemporalGraphRetriever",
|
"TemporalGraphRetriever",
|
||||||
|
# Cross-store erasure
|
||||||
|
"ErasureCoordinator",
|
||||||
|
"ErasureReceipt",
|
||||||
# Decision tracking models
|
# Decision tracking models
|
||||||
"Decision",
|
"Decision",
|
||||||
"DecisionContextModel",
|
"DecisionContextModel",
|
||||||
|
|||||||
@@ -626,6 +626,27 @@ class AgentMemory:
|
|||||||
self.logger.debug(f"Deleted memory item: {memory_id}")
|
self.logger.debug(f"Deleted memory item: {memory_id}")
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
def vector_ids_for(self, memory_id: str) -> List[str]:
|
||||||
|
"""Return the vector-store ids owned by a memory item.
|
||||||
|
|
||||||
|
Read-only view of the ids ``delete_memory()`` would remove for this
|
||||||
|
item, so a caller that needs to *report* on vector removal can delete
|
||||||
|
them itself rather than relying on ``delete_memory()``'s best-effort
|
||||||
|
cascade, which logs a vector-store failure and still returns ``True``.
|
||||||
|
|
||||||
|
Mirrors the fallback in ``delete_memory``: an item stored without
|
||||||
|
tracked vector ids is keyed in the vector store by its own memory id.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
memory_id: Memory identifier.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The item's vector ids, or ``[]`` if the item is unknown.
|
||||||
|
"""
|
||||||
|
if memory_id not in self.memory_items:
|
||||||
|
return []
|
||||||
|
return list(self._vector_ids.get(memory_id, [])) or [memory_id]
|
||||||
|
|
||||||
def clear_memory(self, **filters) -> int:
|
def clear_memory(self, **filters) -> int:
|
||||||
"""
|
"""
|
||||||
Clear memory items matching filters.
|
Clear memory items matching filters.
|
||||||
|
|||||||
@@ -2640,7 +2640,11 @@ class ContextGraph:
|
|||||||
|
|
||||||
Scope is this graph only. Copies held elsewhere (``AgentMemory``, a
|
Scope is this graph only. Copies held elsewhere (``AgentMemory``, a
|
||||||
bound vector store, an exported file) are not reached, so this is one
|
bound vector store, an exported file) are not reached, so this is one
|
||||||
step of an erasure workflow, not the whole of it.
|
step of an erasure workflow, not the whole of it. Callers who need the
|
||||||
|
whole workflow -- and a receipt recording which stores it actually
|
||||||
|
reached -- should drive this through
|
||||||
|
:class:`~semantica.context.erasure.ErasureCoordinator` rather than
|
||||||
|
treating a ``True`` here as proof the content is gone.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
node_id: Node to purge.
|
node_id: Node to purge.
|
||||||
|
|||||||
@@ -239,6 +239,82 @@ print(f"Python importance score: {importance.get('degree', 0)}")
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
## 🧹 Erasing an Entity Everywhere - ErasureCoordinator
|
||||||
|
|
||||||
|
`purge_node()` removes an entity from **one graph**. The same content can still be
|
||||||
|
sitting in agent memory and in your vector store, so purge on its own is one step
|
||||||
|
of an erasure workflow rather than the whole of it.
|
||||||
|
|
||||||
|
`ErasureCoordinator` drives the whole cascade and hands you a receipt saying what
|
||||||
|
it actually managed to erase.
|
||||||
|
|
||||||
|
```python
|
||||||
|
from semantica.context import AgentMemory, ContextGraph, ErasureCoordinator
|
||||||
|
|
||||||
|
coordinator = ErasureCoordinator(graph=knowledge, memory=memory)
|
||||||
|
|
||||||
|
receipt = coordinator.erase_entity(
|
||||||
|
"customer-4471",
|
||||||
|
reason="GDPR Art. 17 request #882",
|
||||||
|
)
|
||||||
|
|
||||||
|
if receipt.complete:
|
||||||
|
print("Erased everywhere")
|
||||||
|
else:
|
||||||
|
print("Still holding data:", receipt.incomplete_stores)
|
||||||
|
```
|
||||||
|
|
||||||
|
### Always Check the Receipt
|
||||||
|
|
||||||
|
The receipt is the point of the feature — **do not treat the call itself as proof
|
||||||
|
the data is gone**. Each store reports one of five statuses:
|
||||||
|
|
||||||
|
| Status | Meaning |
|
||||||
|
|---|---|
|
||||||
|
| `erased` | Reached, data removed (on the vectors leg: the store accepted the delete for the ids given) |
|
||||||
|
| `not_found` | Reached, held nothing for this entity |
|
||||||
|
| `not_configured` | No such store was bound — normal, not a failure |
|
||||||
|
| `unsupported` | The store cannot delete at all; retrying will not help |
|
||||||
|
| `failed` | The store was reached and the deletion did not succeed |
|
||||||
|
|
||||||
|
```python
|
||||||
|
receipt.to_dict()
|
||||||
|
# {
|
||||||
|
# "entity_id": "customer-4471",
|
||||||
|
# "reason": "GDPR Art. 17 request #882",
|
||||||
|
# "erased_at": "2026-08-16T09:03:36.813220",
|
||||||
|
# "complete": False,
|
||||||
|
# "stores": {
|
||||||
|
# "vectors": {"status": "unsupported", "backend": "faiss",
|
||||||
|
# "detail": "backend exposes no delete()/delete_vectors(); ..."},
|
||||||
|
# "memory": {"status": "erased", "items": 14},
|
||||||
|
# "graph": {"status": "erased", "nodes": 1, "edges": 3},
|
||||||
|
# },
|
||||||
|
# }
|
||||||
|
```
|
||||||
|
|
||||||
|
`complete` is `False` when any store reports `unsupported` or `failed`, which is
|
||||||
|
your signal to handle that store out of band. FAISS, Milvus and Weaviate expose
|
||||||
|
no delete method today, so erasure genuinely cannot be completed on them — the
|
||||||
|
coordinator says so rather than reporting a success it did not achieve.
|
||||||
|
|
||||||
|
### Good to Know
|
||||||
|
|
||||||
|
- **Order is vectors → memory → graph.** The graph tombstone is the durable record
|
||||||
|
that an erasure happened, so it is written last: a crash mid-cascade leaves the
|
||||||
|
node present and the receipt incomplete, rather than a tombstone claiming more
|
||||||
|
than actually happened.
|
||||||
|
- **A failing store does not abort the rest.** Partial failure is recorded in the
|
||||||
|
receipt and the remaining stores are still erased.
|
||||||
|
- **Every store is optional.** `ErasureCoordinator(graph=graph)` is fine; the other
|
||||||
|
legs report `not_configured`.
|
||||||
|
- **It is idempotent.** Erasing the same entity twice returns a receipt saying
|
||||||
|
there was nothing left to do, rather than raising.
|
||||||
|
- **Batch:** `coordinator.erase_entities([...], reason=...)` returns one receipt per
|
||||||
|
entity, in order, so one entity's failure does not stop the others.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## 🔄 Using Both Together - The Complete Setup
|
## 🔄 Using Both Together - The Complete Setup
|
||||||
|
|
||||||
### Your Smart Agent System
|
### Your Smart Agent System
|
||||||
|
|||||||
@@ -0,0 +1,673 @@
|
|||||||
|
"""
|
||||||
|
Cross-store erasure coordination.
|
||||||
|
|
||||||
|
``ContextGraph.purge_node()`` is graph-scope by design (#957): it removes the
|
||||||
|
node and leaves a tombstone, but any copy of the same content held in
|
||||||
|
``AgentMemory`` or in a bound vector store is untouched. That makes purge one
|
||||||
|
step of an erasure workflow rather than the whole of it, and leaves the caller
|
||||||
|
to drive the remaining steps by hand -- with no record of which of them
|
||||||
|
actually succeeded.
|
||||||
|
|
||||||
|
:class:`ErasureCoordinator` drives the cascade across the stores it is given
|
||||||
|
and returns an :class:`ErasureReceipt` describing what was reached and what was
|
||||||
|
not. It *composes* the existing public APIs; nothing in ``context_graph.py`` or
|
||||||
|
``agent_memory.py`` changes, and ``ContextGraph`` keeps its graph-scope
|
||||||
|
contract.
|
||||||
|
|
||||||
|
The property that matters is honest partial reporting. Three vector backends
|
||||||
|
(FAISS, Milvus, Weaviate) expose no delete at all, so erasure is genuinely not
|
||||||
|
completable on them today. The receipt says ``unsupported`` for those rather
|
||||||
|
than reporting a success it did not achieve -- a receipt that reads
|
||||||
|
"graph: erased, memory: 14 erased, vectors: unsupported on faiss" is
|
||||||
|
actionable; a bare ``True`` is a compliance liability.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
>>> from semantica.context import ContextGraph, AgentMemory
|
||||||
|
>>> from semantica.context.erasure import ErasureCoordinator
|
||||||
|
>>> coordinator = ErasureCoordinator(graph=graph, memory=memory)
|
||||||
|
>>> receipt = coordinator.erase_entity(
|
||||||
|
... "customer-4471", reason="GDPR Art. 17 request #882"
|
||||||
|
... )
|
||||||
|
>>> receipt.complete
|
||||||
|
False
|
||||||
|
>>> receipt.stores["vectors"]["status"]
|
||||||
|
'unsupported'
|
||||||
|
"""
|
||||||
|
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple, Union
|
||||||
|
|
||||||
|
from ..utils.logging import get_logger
|
||||||
|
from .context_graph import _normalize_temporal_input
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
"ErasureCoordinator",
|
||||||
|
"ErasureReceipt",
|
||||||
|
"STATUS_ERASED",
|
||||||
|
"STATUS_NOT_FOUND",
|
||||||
|
"STATUS_NOT_CONFIGURED",
|
||||||
|
"STATUS_UNSUPPORTED",
|
||||||
|
"STATUS_FAILED",
|
||||||
|
]
|
||||||
|
|
||||||
|
#: The store was reached and the entity's data removed from it. On the vectors
|
||||||
|
#: leg this means the store accepted the delete for the ids it was given: no
|
||||||
|
#: backend offers a portable "does this id exist" check, so it is not a count of
|
||||||
|
#: embeddings that were really there. The memory leg re-queries to confirm and
|
||||||
|
#: so is the stronger claim of the two.
|
||||||
|
STATUS_ERASED = "erased"
|
||||||
|
#: The store was reached and held nothing for this entity.
|
||||||
|
STATUS_NOT_FOUND = "not_found"
|
||||||
|
#: No such store was bound to the coordinator. Normal, not a failure.
|
||||||
|
STATUS_NOT_CONFIGURED = "not_configured"
|
||||||
|
#: The store exists but cannot delete -- e.g. a vector backend with no delete
|
||||||
|
#: method. Deliberately distinct from ``failed``: retrying will not help.
|
||||||
|
STATUS_UNSUPPORTED = "unsupported"
|
||||||
|
#: The store was reached and the deletion did not succeed.
|
||||||
|
STATUS_FAILED = "failed"
|
||||||
|
|
||||||
|
#: Statuses that leave data behind. A receipt containing any of these is not
|
||||||
|
#: complete, and the shortfall has to be handled out of band.
|
||||||
|
_INCOMPLETE_STATUSES = frozenset({STATUS_UNSUPPORTED, STATUS_FAILED})
|
||||||
|
|
||||||
|
#: Page size for the memory sweep. See ``_erase_memory`` for why the sweep
|
||||||
|
#: loops rather than passing one large limit.
|
||||||
|
_MEMORY_SWEEP_BATCH = 500
|
||||||
|
|
||||||
|
logger = get_logger("erasure")
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class ErasureReceipt:
|
||||||
|
"""Auditable record of one entity's erasure across every bound store.
|
||||||
|
|
||||||
|
Attributes:
|
||||||
|
entity_id: The entity the erasure was requested for.
|
||||||
|
reason: Why it was erased, e.g. an erasure-request reference.
|
||||||
|
erased_at: ISO-8601 timestamp of the erasure.
|
||||||
|
stores: Per-store outcome keyed by ``"vectors"``, ``"memory"`` and
|
||||||
|
``"graph"``, each a dict with at least a ``status`` key drawn from
|
||||||
|
the ``STATUS_*`` constants in this module.
|
||||||
|
"""
|
||||||
|
|
||||||
|
entity_id: str
|
||||||
|
reason: Optional[str] = None
|
||||||
|
erased_at: str = ""
|
||||||
|
stores: Dict[str, Dict[str, Any]] = field(default_factory=dict)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def complete(self) -> bool:
|
||||||
|
"""True when no bound store was left holding data.
|
||||||
|
|
||||||
|
``not_configured`` and ``not_found`` count as complete -- a store that
|
||||||
|
was never bound, or that held nothing, leaves no residue. Only
|
||||||
|
``unsupported`` and ``failed`` mean data survived the erasure.
|
||||||
|
"""
|
||||||
|
return not self.incomplete_stores
|
||||||
|
|
||||||
|
@property
|
||||||
|
def incomplete_stores(self) -> List[str]:
|
||||||
|
"""Names of the stores that may still hold the entity's data."""
|
||||||
|
return [
|
||||||
|
name
|
||||||
|
for name, result in self.stores.items()
|
||||||
|
if result.get("status") in _INCOMPLETE_STATUSES
|
||||||
|
]
|
||||||
|
|
||||||
|
def to_dict(self) -> Dict[str, Any]:
|
||||||
|
"""Serialize the receipt, deep-copying the per-store results."""
|
||||||
|
return {
|
||||||
|
"entity_id": self.entity_id,
|
||||||
|
"reason": self.reason,
|
||||||
|
"erased_at": self.erased_at,
|
||||||
|
"complete": self.complete,
|
||||||
|
"stores": {name: dict(result) for name, result in self.stores.items()},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class ErasureCoordinator:
|
||||||
|
"""Drives erasure of an entity across the graph, memory and vector stores.
|
||||||
|
|
||||||
|
Every store is optional; a store that is not supplied reports
|
||||||
|
``not_configured`` rather than being silently skipped, so the receipt still
|
||||||
|
shows the full shape of the workflow.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
graph: A :class:`~semantica.context.ContextGraph` (or anything exposing
|
||||||
|
``purge_node``).
|
||||||
|
memory: An :class:`~semantica.context.AgentMemory` (or anything
|
||||||
|
exposing ``find_by_entity`` and ``batch_delete``).
|
||||||
|
vector_store: Vector store holding entity-keyed embeddings. Defaults to
|
||||||
|
``memory.vector_store`` when a memory is supplied, and stays
|
||||||
|
overridable for deployments that bind a store the memory does not
|
||||||
|
own. Pass ``False`` to disable the vector leg entirely.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
Erasure runs outward-in -- vectors, then memory, then the graph. The
|
||||||
|
graph tombstone is the durable attestation that an erasure happened, so
|
||||||
|
writing it first would let a crash mid-cascade leave a record claiming
|
||||||
|
more than actually occurred. Erasing the graph last means a partial
|
||||||
|
failure leaves the node present and the receipt incomplete, which is
|
||||||
|
recoverable and honest.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
graph: Optional[Any] = None,
|
||||||
|
memory: Optional[Any] = None,
|
||||||
|
vector_store: Optional[Any] = None,
|
||||||
|
):
|
||||||
|
# `is None` / `is False` rather than truthiness: a real store that
|
||||||
|
# defines __bool__ or __len__ (an empty one, say) is falsey while being
|
||||||
|
# a perfectly valid store to erase from.
|
||||||
|
vector_store_given = vector_store is not None and vector_store is not False
|
||||||
|
if graph is None and memory is None and not vector_store_given:
|
||||||
|
raise ValueError(
|
||||||
|
"ErasureCoordinator needs at least one store to erase from; got "
|
||||||
|
f"graph=None, memory=None, vector_store={vector_store!r}"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.graph = graph
|
||||||
|
self.memory = memory
|
||||||
|
if vector_store is False:
|
||||||
|
self.vector_store: Optional[Any] = None
|
||||||
|
elif vector_store is not None:
|
||||||
|
self.vector_store = vector_store
|
||||||
|
else:
|
||||||
|
self.vector_store = getattr(memory, "vector_store", None)
|
||||||
|
|
||||||
|
self.logger = logger
|
||||||
|
|
||||||
|
def erase_entity(
|
||||||
|
self,
|
||||||
|
entity_id: str,
|
||||||
|
reason: Optional[str] = None,
|
||||||
|
at: Optional[Union[str, int, float, datetime]] = None,
|
||||||
|
vector_ids: Optional[Sequence[str]] = None,
|
||||||
|
) -> ErasureReceipt:
|
||||||
|
"""Erase one entity from every bound store and return a receipt.
|
||||||
|
|
||||||
|
A store that cannot be erased from is recorded in the receipt and the
|
||||||
|
cascade continues -- partial failure is a result, not an exception.
|
||||||
|
Aborting on the first failure would leave a half-erased state with no
|
||||||
|
record of which half.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
entity_id: Entity to erase. Interpreted as a graph node id, an
|
||||||
|
``entities[].id`` in memory items, and a vector id.
|
||||||
|
reason: Why it was erased, e.g. an erasure-request reference.
|
||||||
|
Recorded in the receipt and in the graph tombstone.
|
||||||
|
at: When the erasure takes effect, used as the receipt's
|
||||||
|
``erased_at`` and passed to ``purge_node`` so both records
|
||||||
|
carry the same instant. Accepts anything ``ContextGraph``
|
||||||
|
accepts -- an ISO string, a ``datetime``, or epoch seconds --
|
||||||
|
and defaults to now, UTC.
|
||||||
|
vector_ids: Explicit vector ids to remove, in addition to the
|
||||||
|
ids owned by the entity's memory items, which are always
|
||||||
|
included. Defaults to ``[entity_id]``, covering entity-keyed
|
||||||
|
embeddings written by something other than ``AgentMemory``.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
An :class:`ErasureReceipt`. Check :attr:`ErasureReceipt.complete`
|
||||||
|
before treating the erasure as done.
|
||||||
|
"""
|
||||||
|
# Resolve the timestamp once and hand the *resolved* value to the graph.
|
||||||
|
# Passing the caller's `at` through instead would let purge_node take its
|
||||||
|
# own now() when `at` is None, so the receipt and the tombstone it
|
||||||
|
# attests to would disagree by however long the cascade took.
|
||||||
|
erased_at = _normalize_timestamp(at)
|
||||||
|
stores: Dict[str, Dict[str, Any]] = {}
|
||||||
|
|
||||||
|
# Outward-in: vectors, then memory, then the graph last.
|
||||||
|
#
|
||||||
|
# The vector leg must also cover the embeddings owned by memory items.
|
||||||
|
# AgentMemory.delete_memory() deletes an item's vectors best-effort: it
|
||||||
|
# catches a vector-store failure, logs it, and still returns True, so
|
||||||
|
# the memory leg cannot tell a full erasure from one that left the
|
||||||
|
# embedding behind. Deleting those ids here instead puts them behind
|
||||||
|
# the one leg that reports honestly. Collected before anything is
|
||||||
|
# deleted, while the items still exist to be enumerated.
|
||||||
|
stores["vectors"] = self._erase_vectors(
|
||||||
|
entity_id, self._all_vector_ids(entity_id, vector_ids)
|
||||||
|
)
|
||||||
|
stores["memory"] = self._erase_memory(entity_id)
|
||||||
|
stores["graph"] = self._erase_graph(entity_id, reason, erased_at)
|
||||||
|
|
||||||
|
receipt = ErasureReceipt(
|
||||||
|
entity_id=entity_id,
|
||||||
|
reason=reason,
|
||||||
|
erased_at=erased_at,
|
||||||
|
stores=stores,
|
||||||
|
)
|
||||||
|
|
||||||
|
if receipt.complete:
|
||||||
|
self.logger.info(
|
||||||
|
"Erased %r across %d store(s)%s",
|
||||||
|
entity_id,
|
||||||
|
len(stores),
|
||||||
|
f" ({reason})" if reason else "",
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
self.logger.warning(
|
||||||
|
"Erasure of %r is incomplete; these stores may still hold it: %s",
|
||||||
|
entity_id,
|
||||||
|
", ".join(receipt.incomplete_stores),
|
||||||
|
)
|
||||||
|
return receipt
|
||||||
|
|
||||||
|
def erase_entities(
|
||||||
|
self,
|
||||||
|
entity_ids: Iterable[str],
|
||||||
|
reason: Optional[str] = None,
|
||||||
|
at: Optional[Union[str, int, float, datetime]] = None,
|
||||||
|
) -> List[ErasureReceipt]:
|
||||||
|
"""Erase several entities, returning one receipt per entity.
|
||||||
|
|
||||||
|
Each entity is erased independently, so one entity's failure does not
|
||||||
|
stop the rest. Receipts come back in the order the ids were given.
|
||||||
|
|
||||||
|
The timestamp is resolved once for the whole batch so that every
|
||||||
|
receipt and every graph tombstone record the same instant -- a batch
|
||||||
|
erasure under a single legal request must not produce tombstones with
|
||||||
|
diverging ``purged_at`` values.
|
||||||
|
"""
|
||||||
|
resolved_at = _normalize_timestamp(at)
|
||||||
|
return [
|
||||||
|
self.erase_entity(entity_id, reason=reason, at=resolved_at)
|
||||||
|
for entity_id in entity_ids
|
||||||
|
]
|
||||||
|
|
||||||
|
# Store legs
|
||||||
|
|
||||||
|
def _all_vector_ids(
|
||||||
|
self, entity_id: str, vector_ids: Optional[Sequence[str]]
|
||||||
|
) -> List[str]:
|
||||||
|
"""Caller-supplied vector ids plus the ids owned by memory items.
|
||||||
|
|
||||||
|
Best-effort by design: if memory cannot be enumerated here, the memory
|
||||||
|
leg makes the same call moments later and reports the failure, so the
|
||||||
|
receipt is still incomplete. Swallowing it there instead would be the
|
||||||
|
bug this method exists to fix.
|
||||||
|
|
||||||
|
Collects vector IDs from ALL memory items before deletion. Must call
|
||||||
|
find_by_entity with limit=None to get all items, since find_by_entity
|
||||||
|
doesn't support offset/cursor and we cannot delete while collecting.
|
||||||
|
"""
|
||||||
|
ids: List[str] = list(vector_ids) if vector_ids is not None else [entity_id]
|
||||||
|
if self.memory is None:
|
||||||
|
return ids
|
||||||
|
|
||||||
|
seen_vector_ids = set(ids)
|
||||||
|
try:
|
||||||
|
# Get ALL matching memory items in one call (limit=None).
|
||||||
|
# Pagination with deletion happens in _erase_memory(); here we must
|
||||||
|
# collect all vector IDs up front before any deletion occurs.
|
||||||
|
found = self.memory.find_by_entity(entity_id, limit=None)
|
||||||
|
|
||||||
|
for item in found:
|
||||||
|
memory_id = _memory_item_id(item)
|
||||||
|
if not memory_id:
|
||||||
|
continue
|
||||||
|
|
||||||
|
for vector_id in self.memory.vector_ids_for(memory_id):
|
||||||
|
if vector_id not in seen_vector_ids:
|
||||||
|
seen_vector_ids.add(vector_id)
|
||||||
|
ids.append(vector_id)
|
||||||
|
except Exception as exc:
|
||||||
|
self.logger.warning(
|
||||||
|
"Could not enumerate memory-owned vector ids for %r: %s; "
|
||||||
|
"the memory leg will report the same failure",
|
||||||
|
entity_id,
|
||||||
|
exc,
|
||||||
|
)
|
||||||
|
return ids
|
||||||
|
|
||||||
|
def _erase_vectors(
|
||||||
|
self, entity_id: str, vector_ids: Optional[Sequence[str]]
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
"""Remove entity-keyed embeddings from the bound vector store.
|
||||||
|
|
||||||
|
``vector_ids`` in the result is the number of ids the store accepted,
|
||||||
|
not the number of embeddings that existed: backends delete by id and
|
||||||
|
report success either way, with no portable way to ask what was
|
||||||
|
actually there. See :data:`STATUS_ERASED`.
|
||||||
|
"""
|
||||||
|
if self.vector_store is None:
|
||||||
|
return {"status": STATUS_NOT_CONFIGURED}
|
||||||
|
|
||||||
|
ids = list(vector_ids) if vector_ids is not None else [entity_id]
|
||||||
|
backend = _vector_backend_name(self.vector_store)
|
||||||
|
if not ids:
|
||||||
|
return {"status": STATUS_NOT_FOUND, "backend": backend}
|
||||||
|
|
||||||
|
method_name, target = _vector_delete_capability(self.vector_store)
|
||||||
|
if method_name is None:
|
||||||
|
# FAISS, Milvus and Weaviate expose no delete at all; FAISS in
|
||||||
|
# particular cannot remove from a flat index without a rebuild.
|
||||||
|
self.logger.warning(
|
||||||
|
"Vector backend %r exposes no delete; %d vector id(s) for %r "
|
||||||
|
"were not erased",
|
||||||
|
backend,
|
||||||
|
len(ids),
|
||||||
|
entity_id,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_UNSUPPORTED,
|
||||||
|
"backend": backend,
|
||||||
|
"vector_ids": len(ids),
|
||||||
|
"detail": (
|
||||||
|
"backend exposes no delete()/delete_vectors(); "
|
||||||
|
"removal requires an index rebuild or an out-of-band process"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
try:
|
||||||
|
deleted = getattr(target, method_name)(ids)
|
||||||
|
except NotImplementedError as exc:
|
||||||
|
# The VectorStore facade declares delete_vectors() unconditionally
|
||||||
|
# and only fails on the call when its backend cannot delete.
|
||||||
|
self.logger.warning(
|
||||||
|
"Vector backend %r cannot delete %d id(s) for %r: %s",
|
||||||
|
backend,
|
||||||
|
len(ids),
|
||||||
|
entity_id,
|
||||||
|
exc,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_UNSUPPORTED,
|
||||||
|
"backend": backend,
|
||||||
|
"vector_ids": len(ids),
|
||||||
|
"detail": str(exc),
|
||||||
|
}
|
||||||
|
except Exception as exc:
|
||||||
|
self.logger.warning(
|
||||||
|
"Vector deletion failed for %r on backend %r: %s",
|
||||||
|
entity_id,
|
||||||
|
backend,
|
||||||
|
exc,
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_FAILED,
|
||||||
|
"backend": backend,
|
||||||
|
"vector_ids": len(ids),
|
||||||
|
"detail": f"{type(exc).__name__}: {exc}",
|
||||||
|
}
|
||||||
|
|
||||||
|
accepted, detail = _interpret_delete_result(deleted)
|
||||||
|
result: Dict[str, Any] = {
|
||||||
|
"status": STATUS_ERASED if accepted else STATUS_FAILED,
|
||||||
|
"backend": backend,
|
||||||
|
"vector_ids": len(ids),
|
||||||
|
"via": method_name,
|
||||||
|
}
|
||||||
|
# Keep whatever the backend said. Qdrant returns {"status": ...} and
|
||||||
|
# Pinecone {"deleted": True}, and that detail is the only account of
|
||||||
|
# the delete anyone gets -- dropping it on the floor would leave the
|
||||||
|
# receipt less informative than the call it is attesting to.
|
||||||
|
if detail is not None:
|
||||||
|
result["backend_result"] = detail
|
||||||
|
if not accepted:
|
||||||
|
self.logger.warning(
|
||||||
|
"Vector backend %r reported no deletion for %r: %s",
|
||||||
|
backend,
|
||||||
|
entity_id,
|
||||||
|
detail,
|
||||||
|
)
|
||||||
|
result["detail"] = "store reported the ids were not deleted"
|
||||||
|
return result
|
||||||
|
|
||||||
|
def _erase_memory(self, entity_id: str) -> Dict[str, Any]:
|
||||||
|
"""Delete every memory item referencing the entity."""
|
||||||
|
if self.memory is None:
|
||||||
|
return {"status": STATUS_NOT_CONFIGURED}
|
||||||
|
|
||||||
|
deleted = 0
|
||||||
|
try:
|
||||||
|
# Sweep in pages until dry rather than passing one large limit:
|
||||||
|
# ``find_by_entity`` has historically defaulted to ``limit=10`` and
|
||||||
|
# truncated silently, and a single large number is only correct
|
||||||
|
# until someone exceeds it. Deleting as we go means the next page
|
||||||
|
# is the remainder.
|
||||||
|
while True:
|
||||||
|
found = self.memory.find_by_entity(entity_id, limit=_MEMORY_SWEEP_BATCH)
|
||||||
|
if not found:
|
||||||
|
break
|
||||||
|
|
||||||
|
memory_ids = [
|
||||||
|
memory_id
|
||||||
|
for memory_id in (_memory_item_id(item) for item in found)
|
||||||
|
if memory_id
|
||||||
|
]
|
||||||
|
if not memory_ids:
|
||||||
|
self.logger.warning(
|
||||||
|
"Memory returned %d item(s) for %r with no identifier; "
|
||||||
|
"cannot delete them",
|
||||||
|
len(found),
|
||||||
|
entity_id,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_FAILED,
|
||||||
|
"items": deleted,
|
||||||
|
"residual": len(found),
|
||||||
|
"detail": "memory items carry no 'memory_id'",
|
||||||
|
}
|
||||||
|
|
||||||
|
removed = self.memory.batch_delete(memory_ids)
|
||||||
|
deleted += removed
|
||||||
|
if removed == 0:
|
||||||
|
# No progress: another page would return the same items.
|
||||||
|
self.logger.warning(
|
||||||
|
"Memory sweep for %r stalled with %d item(s) remaining",
|
||||||
|
entity_id,
|
||||||
|
len(found),
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_FAILED,
|
||||||
|
"items": deleted,
|
||||||
|
"residual": len(found),
|
||||||
|
"detail": "batch_delete removed nothing for a non-empty page",
|
||||||
|
}
|
||||||
|
if len(found) < _MEMORY_SWEEP_BATCH:
|
||||||
|
break
|
||||||
|
|
||||||
|
# Re-query once rather than trusting the loop's own bookkeeping;
|
||||||
|
# this is what keeps the leg's `failed` status honest.
|
||||||
|
residual = self.memory.find_by_entity(entity_id, limit=_MEMORY_SWEEP_BATCH)
|
||||||
|
except Exception as exc:
|
||||||
|
self.logger.warning(
|
||||||
|
"Memory erasure failed for %r after %d item(s): %s",
|
||||||
|
entity_id,
|
||||||
|
deleted,
|
||||||
|
exc,
|
||||||
|
exc_info=True,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_FAILED,
|
||||||
|
"items": deleted,
|
||||||
|
"detail": f"{type(exc).__name__}: {exc}",
|
||||||
|
}
|
||||||
|
|
||||||
|
if residual:
|
||||||
|
self.logger.warning(
|
||||||
|
"Memory still holds %d item(s) for %r after erasure",
|
||||||
|
len(residual),
|
||||||
|
entity_id,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_FAILED,
|
||||||
|
"items": deleted,
|
||||||
|
"residual": len(residual),
|
||||||
|
"detail": "items referencing the entity survived the sweep",
|
||||||
|
}
|
||||||
|
|
||||||
|
if deleted == 0:
|
||||||
|
return {"status": STATUS_NOT_FOUND, "items": 0}
|
||||||
|
return {"status": STATUS_ERASED, "items": deleted}
|
||||||
|
|
||||||
|
def _erase_graph(
|
||||||
|
self,
|
||||||
|
entity_id: str,
|
||||||
|
reason: Optional[str],
|
||||||
|
at: Optional[Union[str, int, float, datetime]],
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
"""Purge the node, and with it every edge that touches it."""
|
||||||
|
if self.graph is None:
|
||||||
|
return {"status": STATUS_NOT_CONFIGURED}
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Counted before the purge because the edges are gone afterwards.
|
||||||
|
edge_count = _incident_edge_count(self.graph, entity_id)
|
||||||
|
purged = self.graph.purge_node(entity_id, reason=reason, at=at)
|
||||||
|
except Exception as exc:
|
||||||
|
self.logger.warning(
|
||||||
|
"Graph purge failed for %r: %s", entity_id, exc, exc_info=True
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"status": STATUS_FAILED,
|
||||||
|
"detail": f"{type(exc).__name__}: {exc}",
|
||||||
|
}
|
||||||
|
|
||||||
|
if not purged:
|
||||||
|
return {"status": STATUS_NOT_FOUND, "nodes": 0, "edges": 0}
|
||||||
|
return {"status": STATUS_ERASED, "nodes": 1, "edges": edge_count}
|
||||||
|
|
||||||
|
|
||||||
|
# Helpers
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize_timestamp(at: Optional[Union[str, int, float, datetime]]) -> str:
|
||||||
|
"""Render ``at`` exactly as the graph tombstone will record it.
|
||||||
|
|
||||||
|
Reuses ``ContextGraph``'s own normalizer rather than formatting the value
|
||||||
|
here, so the receipt and the tombstone written by the same erasure cannot
|
||||||
|
disagree about when it happened -- an audit record that contradicts the
|
||||||
|
tombstone it attests to is worse than no record. Normalizing up front also
|
||||||
|
rejects an unparseable ``at`` before any store is touched, instead of half
|
||||||
|
way through the cascade.
|
||||||
|
|
||||||
|
``None`` resolves to now here rather than being passed along, so the
|
||||||
|
default path gets one timestamp for both records instead of two ``now()``
|
||||||
|
calls separated by the length of the cascade.
|
||||||
|
"""
|
||||||
|
return _normalize_temporal_input(
|
||||||
|
at if at is not None else datetime.now(timezone.utc)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _memory_item_id(item: Any) -> Optional[str]:
|
||||||
|
"""Pull the identifier out of a memory dict as ``find_by_entity`` returns it."""
|
||||||
|
if not isinstance(item, dict):
|
||||||
|
return None
|
||||||
|
memory_id = item.get("memory_id") or item.get("id")
|
||||||
|
return str(memory_id) if memory_id else None
|
||||||
|
|
||||||
|
|
||||||
|
#: Dict keys a backend uses to report whether a delete succeeded, and the
|
||||||
|
#: values that mean it did not. Qdrant returns ``{"status": <UpdateStatus>}``
|
||||||
|
#: and Pinecone ``{"deleted": True}``; neither is a bool, so a bare
|
||||||
|
#: ``result is False`` check would call every dict a success.
|
||||||
|
_DELETE_FAILURE_MARKERS = {
|
||||||
|
"deleted": (False,),
|
||||||
|
"success": (False,),
|
||||||
|
"ok": (False,),
|
||||||
|
"acknowledged": (False,),
|
||||||
|
"status": ("failed", "error", "failure"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _interpret_delete_result(result: Any) -> Tuple[bool, Optional[str]]:
|
||||||
|
"""Decide whether a backend's delete return value reports success.
|
||||||
|
|
||||||
|
Returns ``(accepted, detail)``, where ``detail`` is a serializable
|
||||||
|
rendering of the backend's own response to keep in the receipt (``None``
|
||||||
|
when there was nothing worth recording).
|
||||||
|
|
||||||
|
``None`` counts as accepted: a delete implemented as a void method returns
|
||||||
|
it on success, and reporting ``failed`` there would be a false alarm --
|
||||||
|
the opposite of the honesty this module is for, in the other direction.
|
||||||
|
"""
|
||||||
|
if result is None:
|
||||||
|
return True, None
|
||||||
|
if isinstance(result, bool):
|
||||||
|
return result, None
|
||||||
|
if isinstance(result, dict):
|
||||||
|
rendered = {key: _stringify(value) for key, value in result.items()}
|
||||||
|
for key, failure_values in _DELETE_FAILURE_MARKERS.items():
|
||||||
|
if key in result and _is_failure_value(result[key], failure_values):
|
||||||
|
return False, rendered
|
||||||
|
return True, rendered
|
||||||
|
# Anything else (a count, a client response object) is taken at face value;
|
||||||
|
# there is no cross-backend contract to interpret it against.
|
||||||
|
return True, _stringify(result)
|
||||||
|
|
||||||
|
|
||||||
|
def _is_failure_value(value: Any, failure_values: Tuple[Any, ...]) -> bool:
|
||||||
|
"""True when a backend's marker value says the delete did not happen.
|
||||||
|
|
||||||
|
Bools are matched by identity so a ``0`` count is not read as ``False``.
|
||||||
|
String markers are matched as substrings of the rendered value, because a
|
||||||
|
backend may return an enum whose ``str()`` is ``"UpdateStatus.FAILED"``
|
||||||
|
rather than a bare ``"failed"``.
|
||||||
|
"""
|
||||||
|
for failure in failure_values:
|
||||||
|
if isinstance(failure, bool):
|
||||||
|
if value is failure:
|
||||||
|
return True
|
||||||
|
elif failure in str(value).lower():
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _stringify(value: Any) -> Any:
|
||||||
|
"""Render a backend payload value so the receipt stays serializable.
|
||||||
|
|
||||||
|
Qdrant's status is an enum, which would make ``to_dict()`` output
|
||||||
|
unserializable as the audit record it is meant to be.
|
||||||
|
"""
|
||||||
|
if isinstance(value, (str, int, float, bool)) or value is None:
|
||||||
|
return value
|
||||||
|
return str(value)
|
||||||
|
|
||||||
|
|
||||||
|
def _vector_delete_capability(store: Any) -> Tuple[Optional[str], Any]:
|
||||||
|
"""Find the delete method to call, and the object to call it on.
|
||||||
|
|
||||||
|
Returns ``(None, target)`` when no delete surface exists, which is the
|
||||||
|
``unsupported`` case.
|
||||||
|
|
||||||
|
The ``VectorStore`` facade declares ``delete_vectors()`` for every backend
|
||||||
|
and only raises ``NotImplementedError`` once called, so probing the facade
|
||||||
|
alone cannot tell a deletable backend from a delete-less one -- hence the
|
||||||
|
look at the backend it wraps. Probing rather than calling-and-catching also
|
||||||
|
keeps a missing method distinguishable from an ``AttributeError`` raised
|
||||||
|
*inside* a working one, which is exactly where guessing wrong would produce
|
||||||
|
a false clean bill of health.
|
||||||
|
"""
|
||||||
|
target = getattr(store, "_backend_store", None) or store
|
||||||
|
for name in ("delete_vectors", "delete"):
|
||||||
|
if callable(getattr(target, name, None)):
|
||||||
|
return name, target
|
||||||
|
return None, target
|
||||||
|
|
||||||
|
|
||||||
|
def _vector_backend_name(store: Any) -> str:
|
||||||
|
"""Best-effort backend label for the receipt."""
|
||||||
|
backend = getattr(store, "backend", None)
|
||||||
|
if isinstance(backend, str) and backend:
|
||||||
|
return backend
|
||||||
|
inner = getattr(store, "_backend_store", None)
|
||||||
|
return type(inner if inner is not None else store).__name__
|
||||||
|
|
||||||
|
|
||||||
|
def _incident_edge_count(graph: Any, node_id: str) -> int:
|
||||||
|
"""Count edges touching ``node_id`` through the graph's public API."""
|
||||||
|
find_edges = getattr(graph, "find_edges", None)
|
||||||
|
if not callable(find_edges):
|
||||||
|
return 0
|
||||||
|
return sum(
|
||||||
|
1
|
||||||
|
for edge in find_edges()
|
||||||
|
if edge.get("source") == node_id or edge.get("target") == node_id
|
||||||
|
)
|
||||||
@@ -233,6 +233,31 @@ async def import_file(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
#: Aliases kept consistent with `mcp/tools/export.py::_FORMAT_ALIASES` and
|
||||||
|
#: `RDFExporter._format_aliases` to ensure the two surfaces agree on format names.
|
||||||
|
#: Maps user-provided format strings to RDFExporter's canonical format names.
|
||||||
|
_RDF_FORMATS: dict[str, str] = {
|
||||||
|
"ttl": "turtle",
|
||||||
|
"turtle": "turtle",
|
||||||
|
"nt": "ntriples", # RDFExporter canonical is "ntriples", not "nt"
|
||||||
|
"ntriples": "ntriples",
|
||||||
|
"n-triples": "ntriples",
|
||||||
|
"xml": "rdfxml", # RDFExporter canonical is "rdfxml", not "xml"
|
||||||
|
"rdfxml": "rdfxml",
|
||||||
|
"rdf-xml": "rdfxml",
|
||||||
|
"json-ld": "jsonld", # RDFExporter canonical is "jsonld", not "json-ld"
|
||||||
|
"jsonld": "jsonld",
|
||||||
|
}
|
||||||
|
|
||||||
|
#: Media type and file extension per RDFExporter canonical format name.
|
||||||
|
_RDF_MEDIA_TYPES: dict[str, tuple[str, str]] = {
|
||||||
|
"turtle": ("text/turtle", "ttl"),
|
||||||
|
"ntriples": ("application/n-triples", "nt"),
|
||||||
|
"rdfxml": ("application/rdf+xml", "rdf"),
|
||||||
|
"jsonld": ("application/ld+json", "jsonld"),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
@router.post("/api/export")
|
@router.post("/api/export")
|
||||||
async def export_graph(
|
async def export_graph(
|
||||||
body: ExportRequest,
|
body: ExportRequest,
|
||||||
@@ -267,8 +292,81 @@ async def export_graph(
|
|||||||
content = output.getvalue()
|
content = output.getvalue()
|
||||||
media_type = "text/csv"
|
media_type = "text/csv"
|
||||||
extension = "csv"
|
extension = "csv"
|
||||||
|
elif fmt in _RDF_FORMATS:
|
||||||
|
# Reuses `semantica.export`, the same exporters the MCP `export_graph` tool calls.
|
||||||
|
# Before this, the Explorer answered 422 for every RDF format while the MCP surface
|
||||||
|
# offered them, so a graph could be loaded as JSON-LD and never exported back — the
|
||||||
|
# round trip had to leave the product. See #1131.
|
||||||
|
try:
|
||||||
|
from semantica.export import RDFExporter
|
||||||
|
from semantica.utils.exceptions import ValidationError
|
||||||
|
except ImportError as exc: # pragma: no cover - optional dependency
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=503,
|
||||||
|
detail=f"RDF export unavailable: {exc}",
|
||||||
|
) from exc
|
||||||
|
|
||||||
|
try:
|
||||||
|
content = RDFExporter().export_to_rdf(graph_dict, format=_RDF_FORMATS[fmt])
|
||||||
|
except ValidationError as exc:
|
||||||
|
# Data validation or serialization failed
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=422,
|
||||||
|
detail=f"RDF export failed: {exc}",
|
||||||
|
) from exc
|
||||||
|
except Exception as exc:
|
||||||
|
# Unexpected error during export
|
||||||
|
logger.exception("RDF export failed unexpectedly")
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=500,
|
||||||
|
detail=f"RDF export error: {exc}",
|
||||||
|
) from exc
|
||||||
|
|
||||||
|
media_type, extension = _RDF_MEDIA_TYPES[_RDF_FORMATS[fmt]]
|
||||||
|
elif fmt == "graphml":
|
||||||
|
# GraphML support using GraphExporter (not GraphMLExporter which doesn't exist)
|
||||||
|
try:
|
||||||
|
from semantica.export import GraphExporter
|
||||||
|
from semantica.utils.exceptions import ValidationError
|
||||||
|
except ImportError as exc: # pragma: no cover - optional dependency
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=503,
|
||||||
|
detail=f"GraphML export unavailable: {exc}",
|
||||||
|
) from exc
|
||||||
|
|
||||||
|
try:
|
||||||
|
# GraphExporter.export() writes to file, but we need string content for HTTP response.
|
||||||
|
# Use a temporary file that is automatically cleaned up.
|
||||||
|
import tempfile
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
# Create temp file in a secure directory with automatic cleanup on exception
|
||||||
|
with tempfile.TemporaryDirectory() as tmpdir:
|
||||||
|
tmp_path = Path(tmpdir) / "export.graphml"
|
||||||
|
exporter = GraphExporter(format="graphml")
|
||||||
|
exporter.export(graph_dict, file_path=tmp_path)
|
||||||
|
content = tmp_path.read_text(encoding='utf-8')
|
||||||
|
except ValidationError as exc:
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=422,
|
||||||
|
detail=f"GraphML export failed: {exc}",
|
||||||
|
) from exc
|
||||||
|
except Exception as exc:
|
||||||
|
logger.exception("GraphML export failed unexpectedly")
|
||||||
|
raise HTTPException(
|
||||||
|
status_code=500,
|
||||||
|
detail=f"GraphML export error: {exc}",
|
||||||
|
) from exc
|
||||||
|
|
||||||
|
media_type, extension = "application/xml", "graphml"
|
||||||
else:
|
else:
|
||||||
raise HTTPException(status_code=422, detail=f"Unsupported export format '{fmt}'")
|
raise HTTPException(
|
||||||
|
status_code=422,
|
||||||
|
detail=(
|
||||||
|
f"Unsupported export format '{fmt}'. "
|
||||||
|
f"Supported: {', '.join(sorted({'json', 'csv', 'graphml'} | set(_RDF_FORMATS)))}"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
return Response(
|
return Response(
|
||||||
content=content,
|
content=content,
|
||||||
|
|||||||
@@ -148,6 +148,26 @@ class ClassInferrer:
|
|||||||
entity_type = entity.get("type") or entity.get("entity_type", "Entity")
|
entity_type = entity.get("type") or entity.get("entity_type", "Entity")
|
||||||
entity_types[entity_type].append(entity)
|
entity_types[entity_type].append(entity)
|
||||||
|
|
||||||
|
normalized_types = defaultdict(list)
|
||||||
|
for entity_type, type_entities in entity_types.items():
|
||||||
|
if len(type_entities) >= self.min_occurrences:
|
||||||
|
normalized_name = self.naming_conventions.normalize_class_name(
|
||||||
|
str(entity_type)
|
||||||
|
)
|
||||||
|
normalized_types[normalized_name].append(str(entity_type))
|
||||||
|
|
||||||
|
collisions = {
|
||||||
|
normalized_name: source_types
|
||||||
|
for normalized_name, source_types in normalized_types.items()
|
||||||
|
if len(source_types) > 1
|
||||||
|
}
|
||||||
|
if collisions:
|
||||||
|
raise ValidationError(
|
||||||
|
"Entity types normalize to duplicate class names; "
|
||||||
|
"rename the source types or provide an explicit mapping.",
|
||||||
|
validation_context={"normalized_type_collisions": collisions},
|
||||||
|
)
|
||||||
|
|
||||||
# Infer classes from entity types
|
# Infer classes from entity types
|
||||||
self.progress_tracker.update_tracking(
|
self.progress_tracker.update_tracking(
|
||||||
tracking_id,
|
tracking_id,
|
||||||
|
|||||||
@@ -438,6 +438,46 @@ class MilvusStore:
|
|||||||
raise ProcessingError(f"Collection {collection_name} does not exist")
|
raise ProcessingError(f"Collection {collection_name} does not exist")
|
||||||
|
|
||||||
collection = Collection(collection_name)
|
collection = Collection(collection_name)
|
||||||
|
# Reject schemas that don't match create_collection()'s shape:
|
||||||
|
# id/VARCHAR pk + vector + metadata. Otherwise an incompatible
|
||||||
|
# collection attaches and fails far later in get_vector/get_metadata.
|
||||||
|
schema = getattr(collection, "schema", None)
|
||||||
|
fields = list(getattr(schema, "fields", None) or [])
|
||||||
|
pk = [f for f in fields if getattr(f, "is_primary", False)]
|
||||||
|
if (
|
||||||
|
not pk
|
||||||
|
or pk[0].name != "id"
|
||||||
|
or getattr(getattr(pk[0], "dtype", None), "name", None) != "VARCHAR"
|
||||||
|
or getattr(pk[0], "auto_id", False)
|
||||||
|
):
|
||||||
|
raise ProcessingError(
|
||||||
|
f"Collection '{collection_name}' has an invalid primary key: "
|
||||||
|
"expected VARCHAR field 'id' without auto_id"
|
||||||
|
)
|
||||||
|
vector_field = next((f for f in fields if f.name == "vector"), None)
|
||||||
|
if vector_field is None:
|
||||||
|
raise ProcessingError(
|
||||||
|
f"Collection '{collection_name}' is missing required field 'vector'"
|
||||||
|
)
|
||||||
|
if (
|
||||||
|
getattr(getattr(vector_field, "dtype", None), "name", None)
|
||||||
|
!= "FLOAT_VECTOR"
|
||||||
|
):
|
||||||
|
raise ProcessingError(
|
||||||
|
f"Collection '{collection_name}' has an invalid vector field: "
|
||||||
|
"expected FLOAT_VECTOR 'vector'"
|
||||||
|
)
|
||||||
|
metadata_field = next((f for f in fields if f.name == "metadata"), None)
|
||||||
|
if metadata_field is None:
|
||||||
|
raise ProcessingError(
|
||||||
|
f"Collection '{collection_name}' is missing required field 'metadata'"
|
||||||
|
)
|
||||||
|
if getattr(getattr(metadata_field, "dtype", None), "name", None) != "JSON":
|
||||||
|
raise ProcessingError(
|
||||||
|
f"Collection '{collection_name}' has an invalid metadata field: "
|
||||||
|
"expected JSON 'metadata'"
|
||||||
|
)
|
||||||
|
|
||||||
self.collection = MilvusCollection(collection, collection_name)
|
self.collection = MilvusCollection(collection, collection_name)
|
||||||
self.search_engine = MilvusSearch(self.collection)
|
self.search_engine = MilvusSearch(self.collection)
|
||||||
return self.collection
|
return self.collection
|
||||||
|
|||||||
@@ -0,0 +1,928 @@
|
|||||||
|
"""Tests for ErasureCoordinator (issue #1018).
|
||||||
|
|
||||||
|
``ContextGraph.purge_node()`` is graph-scope by design: it removes the node and
|
||||||
|
writes a tombstone attesting the content is gone, while the same content can
|
||||||
|
survive verbatim as an ``AgentMemory`` item and as an embedding. The
|
||||||
|
coordinator drives the cascade across every bound store and returns a receipt
|
||||||
|
saying what was reached -- and, just as importantly, what was not.
|
||||||
|
|
||||||
|
These tests run against real ``ContextGraph`` and ``AgentMemory`` instances
|
||||||
|
rather than mocks. The bug this feature exists to prevent lives in the
|
||||||
|
interaction between them (``find_by_entity`` truncating the sweep the caller
|
||||||
|
uses to decide the erasure is done), so mocking that interaction away would
|
||||||
|
test nothing. The vector stores *are* fakes, because the point of those tests
|
||||||
|
is backend shape -- ``delete_vectors`` vs ``delete`` vs neither -- and three of
|
||||||
|
the real backends cannot delete at all.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
from semantica.context import AgentMemory, ContextGraph
|
||||||
|
from semantica.context.erasure import (
|
||||||
|
STATUS_ERASED,
|
||||||
|
STATUS_FAILED,
|
||||||
|
STATUS_NOT_CONFIGURED,
|
||||||
|
STATUS_NOT_FOUND,
|
||||||
|
STATUS_UNSUPPORTED,
|
||||||
|
ErasureCoordinator,
|
||||||
|
ErasureReceipt,
|
||||||
|
)
|
||||||
|
from semantica.vector_store import VectorStore
|
||||||
|
|
||||||
|
|
||||||
|
def _graph():
|
||||||
|
"""customer --purchased--> order, plus an unrelated supplier."""
|
||||||
|
graph = ContextGraph(advanced_analytics=False)
|
||||||
|
graph.add_node("customer-4471", "person")
|
||||||
|
graph.add_node("order-9", "order")
|
||||||
|
graph.add_node("supplier-1", "org")
|
||||||
|
graph.add_edge("customer-4471", "order-9", "purchased")
|
||||||
|
return graph
|
||||||
|
|
||||||
|
|
||||||
|
def _memory_with(entity_id, count, extra_entity=None):
|
||||||
|
"""A memory holding ``count`` items that reference ``entity_id``."""
|
||||||
|
memory = AgentMemory()
|
||||||
|
for index in range(count):
|
||||||
|
memory.store(
|
||||||
|
f"note {index} about {entity_id}",
|
||||||
|
entities=[{"id": entity_id, "name": entity_id}],
|
||||||
|
skip_graph=True,
|
||||||
|
)
|
||||||
|
if extra_entity:
|
||||||
|
memory.store(
|
||||||
|
f"unrelated note about {extra_entity}",
|
||||||
|
entities=[{"id": extra_entity, "name": extra_entity}],
|
||||||
|
skip_graph=True,
|
||||||
|
)
|
||||||
|
return memory
|
||||||
|
|
||||||
|
|
||||||
|
class _DeleteVectorsStore:
|
||||||
|
"""Backend shaped like qdrant/pinecone: exposes ``delete_vectors``."""
|
||||||
|
|
||||||
|
backend = "qdrant"
|
||||||
|
|
||||||
|
def __init__(self, result=True):
|
||||||
|
self._result = result
|
||||||
|
self.deleted = []
|
||||||
|
|
||||||
|
def delete_vectors(self, vector_ids, **options):
|
||||||
|
self.deleted.append(list(vector_ids))
|
||||||
|
return self._result
|
||||||
|
|
||||||
|
|
||||||
|
class _DeleteStore:
|
||||||
|
"""Backend shaped like pgvector/sqlite-vec: exposes ``delete``."""
|
||||||
|
|
||||||
|
backend = "pgvector"
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.deleted = []
|
||||||
|
|
||||||
|
def delete(self, ids):
|
||||||
|
self.deleted.append(list(ids))
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
class _NoDeleteStore:
|
||||||
|
"""Backend shaped like FAISS/Milvus/Weaviate: no delete surface at all."""
|
||||||
|
|
||||||
|
backend = "faiss"
|
||||||
|
|
||||||
|
|
||||||
|
class _RaisingStore:
|
||||||
|
backend = "qdrant"
|
||||||
|
|
||||||
|
def delete_vectors(self, vector_ids, **options):
|
||||||
|
raise RuntimeError("connection reset")
|
||||||
|
|
||||||
|
|
||||||
|
class _FacadeOverNoDeleteBackend:
|
||||||
|
"""The ``VectorStore`` facade shape: declares delete_vectors for every
|
||||||
|
backend and only fails on the call, so the backend must be probed."""
|
||||||
|
|
||||||
|
backend = "faiss"
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self._backend_store = _NoDeleteStore()
|
||||||
|
|
||||||
|
def delete_vectors(self, vector_ids, **options):
|
||||||
|
raise NotImplementedError("Backend store _NoDeleteStore has no delete")
|
||||||
|
|
||||||
|
|
||||||
|
class _MemoryVectorStore(_DeleteVectorsStore):
|
||||||
|
"""Delete-capable store that AgentMemory can also write embeddings to."""
|
||||||
|
|
||||||
|
def store_vectors(self, vectors, metadata=None, **options):
|
||||||
|
return [f"vec-{len(self.deleted)}-{index}" for index in range(len(vectors))]
|
||||||
|
|
||||||
|
|
||||||
|
class TestErasureAcrossStores(unittest.TestCase):
|
||||||
|
def test_erases_graph_and_memory_and_reports_both(self):
|
||||||
|
graph, memory = _graph(), _memory_with("customer-4471", 3, "supplier-1")
|
||||||
|
receipt = ErasureCoordinator(graph=graph, memory=memory).erase_entity(
|
||||||
|
"customer-4471", reason="GDPR Art. 17 request #882"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
self.assertEqual(receipt.stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["graph"]["edges"], 1)
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["memory"]["items"], 3)
|
||||||
|
|
||||||
|
self.assertFalse(graph.has_node("customer-4471"))
|
||||||
|
self.assertEqual(memory.find_by_entity("customer-4471", limit=500), [])
|
||||||
|
|
||||||
|
def test_leaves_other_entities_alone(self):
|
||||||
|
graph, memory = _graph(), _memory_with("customer-4471", 2, "supplier-1")
|
||||||
|
ErasureCoordinator(graph=graph, memory=memory).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertTrue(graph.has_node("supplier-1"))
|
||||||
|
self.assertEqual(len(memory.find_by_entity("supplier-1", limit=500)), 1)
|
||||||
|
|
||||||
|
def test_graph_purge_records_the_reason_in_its_tombstone(self):
|
||||||
|
graph = _graph()
|
||||||
|
ErasureCoordinator(graph=graph).erase_entity(
|
||||||
|
"customer-4471", reason="GDPR Art. 17 request #882"
|
||||||
|
)
|
||||||
|
|
||||||
|
tombstone = graph.get_tombstone("customer-4471", "node")
|
||||||
|
self.assertIsNotNone(tombstone)
|
||||||
|
self.assertEqual(tombstone["reason"], "GDPR Art. 17 request #882")
|
||||||
|
|
||||||
|
def test_erase_entities_returns_one_receipt_per_id_in_order(self):
|
||||||
|
graph = _graph()
|
||||||
|
receipts = ErasureCoordinator(graph=graph).erase_entities(
|
||||||
|
["customer-4471", "supplier-1", "never-existed"], reason="offboarding"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
[receipt.entity_id for receipt in receipts],
|
||||||
|
["customer-4471", "supplier-1", "never-existed"],
|
||||||
|
)
|
||||||
|
self.assertEqual(receipts[0].stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipts[1].stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipts[2].stores["graph"]["status"], STATUS_NOT_FOUND)
|
||||||
|
|
||||||
|
def test_batch_erasure_all_receipts_carry_the_same_timestamp(self):
|
||||||
|
"""erase_entities() must resolve the timestamp once for the whole batch.
|
||||||
|
|
||||||
|
When ``at=None`` each call to ``erase_entity()`` independently calls
|
||||||
|
``_normalize_timestamp()``, generating a fresh ``now()`` per entity.
|
||||||
|
A GDPR batch request would then produce tombstones with diverging
|
||||||
|
``purged_at`` values, making it impossible to group them under a single
|
||||||
|
legal request by timestamp. This regression test pins that every
|
||||||
|
receipt and every graph tombstone share the same instant.
|
||||||
|
"""
|
||||||
|
graph = _graph()
|
||||||
|
receipts = ErasureCoordinator(graph=graph).erase_entities(
|
||||||
|
["customer-4471", "supplier-1"], reason="GDPR Art. 17 request #882"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Both entities were erased.
|
||||||
|
self.assertEqual(receipts[0].stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipts[1].stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
|
||||||
|
# All receipts carry the same erased_at.
|
||||||
|
self.assertEqual(receipts[0].erased_at, receipts[1].erased_at)
|
||||||
|
|
||||||
|
# Each tombstone's purged_at matches its own receipt.
|
||||||
|
tombstone_0 = graph.get_tombstone("customer-4471", "node")
|
||||||
|
tombstone_1 = graph.get_tombstone("supplier-1", "node")
|
||||||
|
self.assertEqual(tombstone_0["purged_at"], receipts[0].erased_at)
|
||||||
|
self.assertEqual(tombstone_1["purged_at"], receipts[1].erased_at)
|
||||||
|
|
||||||
|
# The tombstones themselves agree with each other.
|
||||||
|
self.assertEqual(tombstone_0["purged_at"], tombstone_1["purged_at"])
|
||||||
|
|
||||||
|
|
||||||
|
class TestMemorySweepIsNotTruncated(unittest.TestCase):
|
||||||
|
"""The regression this feature exists to prevent.
|
||||||
|
|
||||||
|
``find_by_entity`` has historically defaulted to ``limit=10`` and truncated
|
||||||
|
silently, so the obvious hand-rolled cascade erases the first ten items and
|
||||||
|
reports success. 25 items is more than any such default, and a coordinator
|
||||||
|
that calls ``find_by_entity`` once with the default fails this test.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_erases_far_more_items_than_the_default_limit(self):
|
||||||
|
memory = _memory_with("customer-4471", 25)
|
||||||
|
receipt = ErasureCoordinator(memory=memory).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["memory"]["items"], 25)
|
||||||
|
self.assertEqual(memory.find_by_entity("customer-4471", limit=500), [])
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
|
||||||
|
def test_residual_items_are_reported_as_failed_not_erased(self):
|
||||||
|
class _UndeletableMemory:
|
||||||
|
"""Deletes nothing, as a backend refusing the write would."""
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.items = [{"memory_id": f"m{i}"} for i in range(3)]
|
||||||
|
|
||||||
|
def find_by_entity(self, entity_id, limit=10):
|
||||||
|
return list(self.items)[:limit]
|
||||||
|
|
||||||
|
def batch_delete(self, memory_ids):
|
||||||
|
return 0
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(memory=_UndeletableMemory()).erase_entity("e1")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_FAILED)
|
||||||
|
self.assertEqual(receipt.stores["memory"]["residual"], 3)
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
def test_memory_items_without_an_identifier_fail_rather_than_look_erased(self):
|
||||||
|
class _AnonymousMemory:
|
||||||
|
def find_by_entity(self, entity_id, limit=10):
|
||||||
|
return [{"content": "no id here"}]
|
||||||
|
|
||||||
|
def batch_delete(self, memory_ids): # pragma: no cover - never reached
|
||||||
|
raise AssertionError("should not delete items it cannot identify")
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(memory=_AnonymousMemory()).erase_entity("e1")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_FAILED)
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
|
||||||
|
class TestVectorBackendShapes(unittest.TestCase):
|
||||||
|
def test_delete_vectors_backend_is_erased(self):
|
||||||
|
store = _DeleteVectorsStore()
|
||||||
|
receipt = ErasureCoordinator(vector_store=store).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["via"], "delete_vectors")
|
||||||
|
self.assertEqual(store.deleted, [["customer-4471"]])
|
||||||
|
|
||||||
|
def test_delete_backend_is_erased(self):
|
||||||
|
store = _DeleteStore()
|
||||||
|
receipt = ErasureCoordinator(vector_store=store).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["via"], "delete")
|
||||||
|
self.assertEqual(store.deleted, [["customer-4471"]])
|
||||||
|
|
||||||
|
def test_backend_without_delete_is_unsupported_not_erased(self):
|
||||||
|
receipt = ErasureCoordinator(vector_store=_NoDeleteStore()).erase_entity("e1")
|
||||||
|
|
||||||
|
vectors = receipt.stores["vectors"]
|
||||||
|
self.assertEqual(vectors["status"], STATUS_UNSUPPORTED)
|
||||||
|
self.assertEqual(vectors["backend"], "faiss")
|
||||||
|
self.assertIn("no delete", vectors["detail"])
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
def test_facade_declaring_delete_over_a_delete_less_backend_is_unsupported(self):
|
||||||
|
receipt = ErasureCoordinator(
|
||||||
|
vector_store=_FacadeOverNoDeleteBackend()
|
||||||
|
).erase_entity("e1")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_UNSUPPORTED)
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
def test_store_reporting_no_deletion_is_failed(self):
|
||||||
|
store = _DeleteVectorsStore(result=False)
|
||||||
|
receipt = ErasureCoordinator(vector_store=store).erase_entity("e1")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_FAILED)
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
def test_explicit_vector_ids_override_the_entity_id(self):
|
||||||
|
store = _DeleteVectorsStore()
|
||||||
|
ErasureCoordinator(vector_store=store).erase_entity(
|
||||||
|
"customer-4471", vector_ids=["vec-a", "vec-b"]
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(store.deleted, [["vec-a", "vec-b"]])
|
||||||
|
|
||||||
|
def test_vector_store_defaults_to_the_one_memory_holds(self):
|
||||||
|
store = _MemoryVectorStore()
|
||||||
|
memory = AgentMemory(vector_store=store)
|
||||||
|
|
||||||
|
self.assertIs(ErasureCoordinator(memory=memory).vector_store, store)
|
||||||
|
|
||||||
|
def test_memory_bound_vector_store_can_be_overridden(self):
|
||||||
|
owned, external = _MemoryVectorStore(), _DeleteVectorsStore()
|
||||||
|
memory = AgentMemory(vector_store=owned)
|
||||||
|
|
||||||
|
coordinator = ErasureCoordinator(memory=memory, vector_store=external)
|
||||||
|
|
||||||
|
self.assertIs(coordinator.vector_store, external)
|
||||||
|
|
||||||
|
def test_vector_leg_can_be_disabled_for_a_memory_bound_store(self):
|
||||||
|
memory = AgentMemory(vector_store=_MemoryVectorStore())
|
||||||
|
coordinator = ErasureCoordinator(memory=memory, vector_store=False)
|
||||||
|
|
||||||
|
receipt = coordinator.erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertIsNone(coordinator.vector_store)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_NOT_CONFIGURED)
|
||||||
|
|
||||||
|
|
||||||
|
class TestPartialFailureIsAResultNotAnException(unittest.TestCase):
|
||||||
|
def test_a_raising_vector_store_does_not_stop_the_remaining_legs(self):
|
||||||
|
graph, memory = _graph(), _memory_with("customer-4471", 4)
|
||||||
|
receipt = ErasureCoordinator(
|
||||||
|
graph=graph, memory=memory, vector_store=_RaisingStore()
|
||||||
|
).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_FAILED)
|
||||||
|
self.assertIn("RuntimeError", receipt.stores["vectors"]["detail"])
|
||||||
|
# The legs after the failure still ran.
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
self.assertFalse(graph.has_node("customer-4471"))
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
self.assertEqual(receipt.incomplete_stores, ["vectors"])
|
||||||
|
|
||||||
|
def test_a_raising_graph_is_reported_after_memory_was_erased(self):
|
||||||
|
class _RaisingGraph:
|
||||||
|
def find_edges(self):
|
||||||
|
return []
|
||||||
|
|
||||||
|
def purge_node(self, node_id, reason=None, at=None):
|
||||||
|
raise RuntimeError("graph store unavailable")
|
||||||
|
|
||||||
|
memory = _memory_with("customer-4471", 2)
|
||||||
|
receipt = ErasureCoordinator(graph=_RaisingGraph(), memory=memory).erase_entity(
|
||||||
|
"customer-4471"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["graph"]["status"], STATUS_FAILED)
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
|
||||||
|
class TestReceipt(unittest.TestCase):
|
||||||
|
def test_unconfigured_stores_are_reported_and_still_count_as_complete(self):
|
||||||
|
receipt = ErasureCoordinator(graph=_graph()).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_NOT_CONFIGURED)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_NOT_CONFIGURED)
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
|
||||||
|
def test_erasing_a_second_time_reports_nothing_left_rather_than_raising(self):
|
||||||
|
graph, memory = _graph(), _memory_with("customer-4471", 3)
|
||||||
|
coordinator = ErasureCoordinator(graph=graph, memory=memory)
|
||||||
|
coordinator.erase_entity("customer-4471")
|
||||||
|
|
||||||
|
second = coordinator.erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertEqual(second.stores["graph"]["status"], STATUS_NOT_FOUND)
|
||||||
|
self.assertEqual(second.stores["memory"]["status"], STATUS_NOT_FOUND)
|
||||||
|
self.assertTrue(second.complete)
|
||||||
|
|
||||||
|
def test_to_dict_round_trips_the_reported_shape(self):
|
||||||
|
graph = _graph()
|
||||||
|
receipt = ErasureCoordinator(graph=graph).erase_entity(
|
||||||
|
"customer-4471",
|
||||||
|
reason="GDPR Art. 17 request #882",
|
||||||
|
at="2026-08-16T00:00:00Z",
|
||||||
|
)
|
||||||
|
payload = receipt.to_dict()
|
||||||
|
|
||||||
|
self.assertEqual(payload["entity_id"], "customer-4471")
|
||||||
|
self.assertEqual(payload["reason"], "GDPR Art. 17 request #882")
|
||||||
|
self.assertEqual(payload["erased_at"], "2026-08-16T00:00:00")
|
||||||
|
self.assertTrue(payload["complete"])
|
||||||
|
self.assertEqual(set(payload["stores"]), {"graph", "memory", "vectors"})
|
||||||
|
|
||||||
|
def test_to_dict_copies_the_store_results(self):
|
||||||
|
receipt = ErasureCoordinator(graph=_graph()).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
payload = receipt.to_dict()
|
||||||
|
payload["stores"]["graph"]["status"] = "tampered"
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
|
||||||
|
def test_receipt_and_tombstone_agree_on_when_the_erasure_happened(self):
|
||||||
|
graph = _graph()
|
||||||
|
receipt = ErasureCoordinator(graph=graph).erase_entity(
|
||||||
|
"customer-4471", at="2026-08-16T00:00:00Z"
|
||||||
|
)
|
||||||
|
|
||||||
|
tombstone = graph.get_tombstone("customer-4471", "node")
|
||||||
|
self.assertEqual(tombstone["purged_at"], "2026-08-16T00:00:00")
|
||||||
|
self.assertEqual(receipt.erased_at, tombstone["purged_at"])
|
||||||
|
|
||||||
|
def test_receipt_and_tombstone_agree_when_no_at_is_given(self):
|
||||||
|
"""The default path, where the drift actually happens.
|
||||||
|
|
||||||
|
With `at=None` the coordinator and `purge_node()` would each take their
|
||||||
|
own `now()`, so the receipt attested to a different instant than the
|
||||||
|
tombstone it points at. Passing an explicit `at` hides this, which is
|
||||||
|
why the test above passed while the common case was wrong.
|
||||||
|
"""
|
||||||
|
graph = _graph()
|
||||||
|
receipt = ErasureCoordinator(graph=graph).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
tombstone = graph.get_tombstone("customer-4471", "node")
|
||||||
|
self.assertEqual(receipt.erased_at, tombstone["purged_at"])
|
||||||
|
|
||||||
|
def test_epoch_seconds_are_accepted_like_the_graph_accepts_them(self):
|
||||||
|
graph = _graph()
|
||||||
|
receipt = ErasureCoordinator(graph=graph).erase_entity(
|
||||||
|
"customer-4471", at=1755302400
|
||||||
|
)
|
||||||
|
|
||||||
|
tombstone = graph.get_tombstone("customer-4471", "node")
|
||||||
|
self.assertEqual(receipt.erased_at, tombstone["purged_at"])
|
||||||
|
self.assertTrue(receipt.erased_at.startswith("2025-"))
|
||||||
|
|
||||||
|
def test_an_unparseable_at_is_rejected_before_any_store_is_touched(self):
|
||||||
|
graph, memory = _graph(), _memory_with("customer-4471", 2)
|
||||||
|
|
||||||
|
with self.assertRaises(ValueError):
|
||||||
|
ErasureCoordinator(graph=graph, memory=memory).erase_entity(
|
||||||
|
"customer-4471", at="not-a-timestamp"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(graph.has_node("customer-4471"))
|
||||||
|
self.assertEqual(len(memory.find_by_entity("customer-4471", limit=500)), 2)
|
||||||
|
|
||||||
|
def test_incomplete_stores_names_every_store_still_holding_data(self):
|
||||||
|
receipt = ErasureReceipt(
|
||||||
|
entity_id="e1",
|
||||||
|
stores={
|
||||||
|
"vectors": {"status": STATUS_UNSUPPORTED},
|
||||||
|
"memory": {"status": STATUS_FAILED},
|
||||||
|
"graph": {"status": STATUS_ERASED},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(sorted(receipt.incomplete_stores), ["memory", "vectors"])
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
|
||||||
|
class TestRealVectorStoreBackend(unittest.TestCase):
|
||||||
|
"""The fakes above assert the shapes the coordinator expects; these assert
|
||||||
|
that a real backend actually has one of them.
|
||||||
|
|
||||||
|
This repo's recurring failure is a change verified only against the default
|
||||||
|
that reaches for internals and breaks on every other backend, so the fake
|
||||||
|
stores are worth exactly as much as the assumption that a real store looks
|
||||||
|
like them. ``VectorStore(backend="inmemory")`` is the one backend that runs
|
||||||
|
without external services, so it is the one that can hold that assumption
|
||||||
|
to account here.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _store(self):
|
||||||
|
return VectorStore(backend="inmemory", dimension=8)
|
||||||
|
|
||||||
|
def test_real_backend_erases_the_vector_ids_it_is_given(self):
|
||||||
|
store = self._store()
|
||||||
|
vector_ids = store.store_vectors(
|
||||||
|
vectors=[np.ones(8), np.zeros(8)], metadata=[{}, {}]
|
||||||
|
)
|
||||||
|
self.assertEqual(store.count(), 2)
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(vector_store=store).erase_entity(
|
||||||
|
"customer-4471", vector_ids=vector_ids
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["backend"], "inmemory")
|
||||||
|
self.assertEqual(store.count(), 0)
|
||||||
|
|
||||||
|
def test_the_full_cascade_removes_a_real_memory_bound_embedding(self):
|
||||||
|
"""The end-to-end case the receipt actually attests to.
|
||||||
|
|
||||||
|
Real ``ContextGraph``, real ``AgentMemory``, real ``VectorStore`` --
|
||||||
|
the embedding is written by ``AgentMemory.store()`` and has to be gone
|
||||||
|
afterwards, which exercises the memory leg's own ``delete_memory()``
|
||||||
|
vector cascade rather than the coordinator's model of it.
|
||||||
|
"""
|
||||||
|
store, graph = self._store(), _graph()
|
||||||
|
memory = AgentMemory(vector_store=store)
|
||||||
|
memory.store(
|
||||||
|
"note about customer-4471",
|
||||||
|
entities=[{"id": "customer-4471", "name": "customer-4471"}],
|
||||||
|
skip_graph=True,
|
||||||
|
)
|
||||||
|
self.assertEqual(store.count(), 1)
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(graph=graph, memory=memory).erase_entity(
|
||||||
|
"customer-4471", reason="GDPR Art. 17 request #882"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["graph"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(store.count(), 0)
|
||||||
|
self.assertFalse(graph.has_node("customer-4471"))
|
||||||
|
self.assertEqual(memory.find_by_entity("customer-4471", limit=500), [])
|
||||||
|
|
||||||
|
def test_erased_means_the_store_accepted_the_delete_not_that_data_existed(self):
|
||||||
|
"""Pins a limit of the receipt worth knowing before trusting it.
|
||||||
|
|
||||||
|
The in-memory backend pops the ids and returns ``True`` whether or not
|
||||||
|
they were there, and no backend offers a portable "did this id exist"
|
||||||
|
check, so the vectors leg reports how many ids the store accepted --
|
||||||
|
not how many embeddings were really removed. ``erased`` on this leg is
|
||||||
|
therefore weaker than on the memory leg, which re-queries to confirm.
|
||||||
|
"""
|
||||||
|
store = self._store()
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(vector_store=store).erase_entity("never-embedded")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["vector_ids"], 1)
|
||||||
|
self.assertEqual(store.count(), 0)
|
||||||
|
|
||||||
|
|
||||||
|
class TestConstruction(unittest.TestCase):
|
||||||
|
def test_a_coordinator_with_no_stores_is_rejected(self):
|
||||||
|
with self.assertRaises(ValueError):
|
||||||
|
ErasureCoordinator()
|
||||||
|
|
||||||
|
def test_a_single_store_is_enough(self):
|
||||||
|
self.assertIsNotNone(ErasureCoordinator(graph=_graph()))
|
||||||
|
self.assertIsNotNone(ErasureCoordinator(memory=AgentMemory()))
|
||||||
|
self.assertIsNotNone(ErasureCoordinator(vector_store=_DeleteStore()))
|
||||||
|
|
||||||
|
def test_a_falsey_vector_store_is_still_a_store(self):
|
||||||
|
"""An empty store defining __len__ is falsey but perfectly valid."""
|
||||||
|
|
||||||
|
class _EmptyButReal(_DeleteVectorsStore):
|
||||||
|
def __len__(self):
|
||||||
|
return 0
|
||||||
|
|
||||||
|
store = _EmptyButReal()
|
||||||
|
coordinator = ErasureCoordinator(vector_store=store)
|
||||||
|
|
||||||
|
self.assertIs(coordinator.vector_store, store)
|
||||||
|
receipt = coordinator.erase_entity("customer-4471")
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_ERASED)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBackendDeleteResults(unittest.TestCase):
|
||||||
|
"""Backends report deletes as dicts, not bools.
|
||||||
|
|
||||||
|
Qdrant returns ``{"status": <UpdateStatus>}`` and Pinecone
|
||||||
|
``{"deleted": True}``, so a bare ``result is False`` check calls every dict
|
||||||
|
a success and throws away the only account of the delete the caller gets.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _store_returning(self, value):
|
||||||
|
store = _DeleteVectorsStore(result=value)
|
||||||
|
return store, ErasureCoordinator(vector_store=store)
|
||||||
|
|
||||||
|
def test_qdrant_shaped_success_dict_is_erased_and_kept(self):
|
||||||
|
_, coordinator = self._store_returning({"status": "completed"})
|
||||||
|
|
||||||
|
vectors = coordinator.erase_entity("e1").stores["vectors"]
|
||||||
|
|
||||||
|
self.assertEqual(vectors["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(vectors["backend_result"], {"status": "completed"})
|
||||||
|
|
||||||
|
def test_pinecone_shaped_success_dict_is_erased(self):
|
||||||
|
_, coordinator = self._store_returning({"deleted": True})
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
coordinator.erase_entity("e1").stores["vectors"]["status"], STATUS_ERASED
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_explicit_failure_marker_in_a_dict_is_failed(self):
|
||||||
|
for payload in ({"deleted": False}, {"success": False}, {"status": "failed"}):
|
||||||
|
with self.subTest(payload=payload):
|
||||||
|
_, coordinator = self._store_returning(payload)
|
||||||
|
|
||||||
|
receipt = coordinator.erase_entity("e1")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_FAILED)
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
|
||||||
|
def test_an_enum_like_failure_status_is_not_read_as_success(self):
|
||||||
|
class _UpdateStatus:
|
||||||
|
def __str__(self):
|
||||||
|
return "UpdateStatus.FAILED"
|
||||||
|
|
||||||
|
_, coordinator = self._store_returning({"status": _UpdateStatus()})
|
||||||
|
|
||||||
|
receipt = coordinator.erase_entity("e1")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_FAILED)
|
||||||
|
# Rendered as a string so the receipt stays serializable as an audit record.
|
||||||
|
self.assertEqual(
|
||||||
|
receipt.stores["vectors"]["backend_result"],
|
||||||
|
{"status": "UpdateStatus.FAILED"},
|
||||||
|
)
|
||||||
|
json.dumps(receipt.to_dict())
|
||||||
|
|
||||||
|
def test_a_zero_count_return_is_not_mistaken_for_False(self):
|
||||||
|
"""`0 == False` in Python; a store reporting "0 rows" is not a failure."""
|
||||||
|
_, coordinator = self._store_returning({"deleted": 0})
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
coordinator.erase_entity("e1").stores["vectors"]["status"], STATUS_ERASED
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_a_void_delete_returning_None_is_accepted(self):
|
||||||
|
"""Reporting `failed` for a void method would be a false alarm."""
|
||||||
|
_, coordinator = self._store_returning(None)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
coordinator.erase_entity("e1").stores["vectors"]["status"], STATUS_ERASED
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class _SelectiveDeleteStore:
|
||||||
|
"""Deletes some ids and refuses others, tracking what is still live.
|
||||||
|
|
||||||
|
Models the case that matters: the entity-keyed id deletes fine while the
|
||||||
|
embedding an ``AgentMemory`` item owns does not.
|
||||||
|
"""
|
||||||
|
|
||||||
|
backend = "qdrant"
|
||||||
|
|
||||||
|
def __init__(self, refuse=()):
|
||||||
|
self._refuse = set(refuse)
|
||||||
|
self.live = set()
|
||||||
|
self.attempts = []
|
||||||
|
|
||||||
|
def store_vectors(self, vectors, metadata=None, **options):
|
||||||
|
ids = [f"vec-{len(self.live) + index}" for index in range(len(vectors))]
|
||||||
|
self.live.update(ids)
|
||||||
|
return ids
|
||||||
|
|
||||||
|
def delete_vectors(self, vector_ids, **options):
|
||||||
|
self.attempts.append(list(vector_ids))
|
||||||
|
if any(vector_id in self._refuse for vector_id in vector_ids):
|
||||||
|
return False
|
||||||
|
self.live.difference_update(vector_ids)
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
def _memory_with_embedding(entity_id, store):
|
||||||
|
memory = AgentMemory(vector_store=store)
|
||||||
|
memory.store(
|
||||||
|
f"note about {entity_id}",
|
||||||
|
entities=[{"id": entity_id, "name": entity_id}],
|
||||||
|
embedding=np.zeros(4),
|
||||||
|
skip_graph=True,
|
||||||
|
)
|
||||||
|
return memory
|
||||||
|
|
||||||
|
|
||||||
|
class TestSeparateVectorStoreHandling(unittest.TestCase):
|
||||||
|
"""Verify correct behavior when coordinator.vector_store != memory.vector_store.
|
||||||
|
|
||||||
|
AgentMemory.delete_memory() has its own best-effort vector cascade that
|
||||||
|
logs failures but returns True. When the coordinator's vector_store differs
|
||||||
|
from (or is disabled vs) memory.vector_store, a vector remaining in
|
||||||
|
memory.vector_store must not be hidden by the coordinator's receipt.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_vector_store_false_disables_vector_leg_entirely(self):
|
||||||
|
"""vector_store=False must disable the vector leg, not try memory.vector_store."""
|
||||||
|
memory_store = _SelectiveDeleteStore()
|
||||||
|
memory = _memory_with_embedding("customer-4471", memory_store)
|
||||||
|
|
||||||
|
# Disable vector leg explicitly
|
||||||
|
receipt = ErasureCoordinator(
|
||||||
|
graph=_graph(), memory=memory, vector_store=False
|
||||||
|
).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
# Vector leg should report not_configured, not attempt deletion
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_NOT_CONFIGURED)
|
||||||
|
# Memory's own cascade still runs, but coordinator doesn't track it
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
|
||||||
|
def test_separate_vector_store_only_handles_coordinator_store(self):
|
||||||
|
"""When coordinator has a different vector_store, it only handles that one.
|
||||||
|
|
||||||
|
If memory.vector_store contains a memory-owned vector and fails to delete
|
||||||
|
it, that's memory's problem -- the coordinator only reports on the store
|
||||||
|
it was given. This test verifies the coordinator correctly collects IDs
|
||||||
|
from memory items and attempts deletion on its own store, independent of
|
||||||
|
memory.vector_store.
|
||||||
|
"""
|
||||||
|
# Memory has its own store with a vector
|
||||||
|
memory_store = _SelectiveDeleteStore()
|
||||||
|
memory = _memory_with_embedding("customer-4471", memory_store)
|
||||||
|
memory_vector_id = list(memory_store.live)[0]
|
||||||
|
|
||||||
|
# Coordinator has a separate store that refuses to delete
|
||||||
|
coordinator_store = _SelectiveDeleteStore(refuse={memory_vector_id})
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(
|
||||||
|
graph=_graph(), memory=memory, vector_store=coordinator_store
|
||||||
|
).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
# The coordinator's store should have been asked to delete the memory-owned vector
|
||||||
|
self.assertIn(memory_vector_id, coordinator_store.attempts[0])
|
||||||
|
# The coordinator's store refused, so receipt is incomplete
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_FAILED)
|
||||||
|
|
||||||
|
# Memory's own store was used by delete_memory()'s cascade (best-effort)
|
||||||
|
# but the coordinator's receipt only reflects the coordinator's store
|
||||||
|
self.assertNotIn(memory_vector_id, memory_store.live) # memory deleted it
|
||||||
|
|
||||||
|
def test_memory_vector_store_failure_is_not_reported_when_coordinator_has_separate_store(
|
||||||
|
self,
|
||||||
|
):
|
||||||
|
"""If memory.vector_store fails but coordinator.vector_store succeeds, receipt is complete.
|
||||||
|
|
||||||
|
The coordinator reports only on its own store. Memory's delete_memory()
|
||||||
|
cascade is best-effort and logs failures, but the coordinator doesn't
|
||||||
|
re-check memory.vector_store after deletion.
|
||||||
|
"""
|
||||||
|
# Memory's store will fail to delete (but delete_memory catches it)
|
||||||
|
memory_store = _SelectiveDeleteStore(refuse={"vec-0"})
|
||||||
|
memory = _memory_with_embedding("customer-4471", memory_store)
|
||||||
|
|
||||||
|
# Coordinator has a separate, cooperative store
|
||||||
|
coordinator_store = _SelectiveDeleteStore()
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(
|
||||||
|
graph=_graph(), memory=memory, vector_store=coordinator_store
|
||||||
|
).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
# Coordinator's store succeeded
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_ERASED)
|
||||||
|
|
||||||
|
# But memory's store still has the vector (delete_memory logged it)
|
||||||
|
self.assertIn("vec-0", memory_store.live)
|
||||||
|
|
||||||
|
|
||||||
|
class TestMemoryOwnedVectorsAreReported(unittest.TestCase):
|
||||||
|
"""A memory item's embedding must not survive a `complete` receipt.
|
||||||
|
|
||||||
|
``AgentMemory.delete_memory()`` deletes an item's vectors best-effort: it
|
||||||
|
catches a vector-store failure, logs a warning, and still returns ``True``.
|
||||||
|
The coordinator therefore cannot learn from the memory leg whether those
|
||||||
|
embeddings actually went away, so it deletes them through its own vector
|
||||||
|
leg, which reports honestly.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_refused_memory_owned_vector_makes_the_receipt_incomplete(self):
|
||||||
|
store = _SelectiveDeleteStore(refuse={"vec-0"})
|
||||||
|
memory = _memory_with_embedding("customer-4471", store)
|
||||||
|
self.assertEqual(
|
||||||
|
memory.vector_ids_for(next(iter(memory.memory_items))), ["vec-0"]
|
||||||
|
)
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(graph=_graph(), memory=memory).erase_entity(
|
||||||
|
"customer-4471"
|
||||||
|
)
|
||||||
|
|
||||||
|
# The embedding is demonstrably still there ...
|
||||||
|
self.assertIn("vec-0", store.live)
|
||||||
|
# ... so the receipt must not claim the erasure is done.
|
||||||
|
self.assertFalse(receipt.complete)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_FAILED)
|
||||||
|
self.assertEqual(receipt.incomplete_stores, ["vectors"])
|
||||||
|
|
||||||
|
def test_memory_owned_vector_ids_are_sent_to_the_vector_store(self):
|
||||||
|
store = _SelectiveDeleteStore()
|
||||||
|
memory = _memory_with_embedding("customer-4471", store)
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(graph=_graph(), memory=memory).erase_entity(
|
||||||
|
"customer-4471"
|
||||||
|
)
|
||||||
|
|
||||||
|
# The coordinator's own leg must have attempted the memory-owned id,
|
||||||
|
# not just the entity-keyed one.
|
||||||
|
self.assertIn("vec-0", store.attempts[0])
|
||||||
|
self.assertIn("customer-4471", store.attempts[0])
|
||||||
|
self.assertNotIn("vec-0", store.live)
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
|
||||||
|
def test_explicit_vector_ids_do_not_displace_memory_owned_ids(self):
|
||||||
|
store = _SelectiveDeleteStore()
|
||||||
|
memory = _memory_with_embedding("customer-4471", store)
|
||||||
|
|
||||||
|
ErasureCoordinator(graph=_graph(), memory=memory).erase_entity(
|
||||||
|
"customer-4471", vector_ids=["extra-1"]
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertIn("extra-1", store.attempts[0])
|
||||||
|
self.assertIn("vec-0", store.attempts[0])
|
||||||
|
|
||||||
|
def test_vector_ids_for_falls_back_to_the_memory_id(self):
|
||||||
|
"""An item stored without tracked vector ids is keyed by its own id."""
|
||||||
|
memory = AgentMemory()
|
||||||
|
memory.store(
|
||||||
|
"note about customer-4471",
|
||||||
|
entities=[{"id": "customer-4471", "name": "customer-4471"}],
|
||||||
|
skip_graph=True,
|
||||||
|
)
|
||||||
|
memory_id = next(iter(memory.memory_items))
|
||||||
|
self.assertEqual(memory.vector_ids_for(memory_id), [memory_id])
|
||||||
|
self.assertEqual(memory.vector_ids_for("no-such-item"), [])
|
||||||
|
|
||||||
|
def test_pagination_collects_vectors_from_all_501_items(self):
|
||||||
|
"""Regression: _all_vector_ids must page to collect ALL vectors.
|
||||||
|
|
||||||
|
The original implementation called find_by_entity(limit=500) once,
|
||||||
|
collecting only the first 500 items' vectors, while _erase_memory()
|
||||||
|
continued paging and deleted all 501+ items. The vector belonging to
|
||||||
|
item 501 remained, yet the receipt reported complete=True -- the exact
|
||||||
|
failure mode the coordinator exists to prevent.
|
||||||
|
|
||||||
|
This test uses 51 items (crossing a 50-item batch boundary for testing)
|
||||||
|
to verify pagination logic without the performance cost of 501 real items.
|
||||||
|
The test would fail against the original bug with ANY batch size > 1.
|
||||||
|
"""
|
||||||
|
# Use batch size of 50 for this test (instead of production's 500)
|
||||||
|
# This keeps the test fast while still proving pagination across boundaries
|
||||||
|
TEST_BATCH_SIZE = 50
|
||||||
|
TEST_ITEM_COUNT = 51 # One more than batch size
|
||||||
|
|
||||||
|
store = _SelectiveDeleteStore(refuse={"vec-50"}) # 0-indexed: item 51
|
||||||
|
|
||||||
|
# Create a lightweight memory mock optimized for speed
|
||||||
|
class FastMemoryFor51Test:
|
||||||
|
"""Fast memory implementation for pagination test."""
|
||||||
|
def __init__(self, vector_store):
|
||||||
|
self.vector_store = vector_store
|
||||||
|
entity_id = "customer-with-many-memories"
|
||||||
|
self._items = {}
|
||||||
|
for i in range(TEST_ITEM_COUNT):
|
||||||
|
memory_id = f"mem-{i}"
|
||||||
|
self._items[memory_id] = {
|
||||||
|
"memory_id": memory_id,
|
||||||
|
"content": f"Memory {i}",
|
||||||
|
"entities": [{"id": entity_id}],
|
||||||
|
"metadata": {},
|
||||||
|
"timestamp": "2026-01-01T00:00:00",
|
||||||
|
"relationships": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
def find_by_entity(self, entity_id, limit=None):
|
||||||
|
"""Return all remaining items, with limit."""
|
||||||
|
results = list(self._items.values())
|
||||||
|
if limit is not None:
|
||||||
|
return results[:limit]
|
||||||
|
return results
|
||||||
|
|
||||||
|
def batch_delete(self, memory_ids):
|
||||||
|
"""Fast deletion."""
|
||||||
|
deleted = 0
|
||||||
|
for memory_id in memory_ids:
|
||||||
|
if memory_id in self._items:
|
||||||
|
del self._items[memory_id]
|
||||||
|
deleted += 1
|
||||||
|
return deleted
|
||||||
|
|
||||||
|
def vector_ids_for(self, memory_id):
|
||||||
|
"""Return vector ID for this memory."""
|
||||||
|
idx = int(memory_id.split("-")[1])
|
||||||
|
return [f"vec-{idx}"]
|
||||||
|
|
||||||
|
memory = FastMemoryFor51Test(store)
|
||||||
|
|
||||||
|
# Pre-populate the vector store
|
||||||
|
for i in range(TEST_ITEM_COUNT):
|
||||||
|
store.live.add(f"vec-{i}")
|
||||||
|
|
||||||
|
# Temporarily patch the batch size constant for this test
|
||||||
|
from semantica.context import erasure
|
||||||
|
original_batch_size = erasure._MEMORY_SWEEP_BATCH
|
||||||
|
erasure._MEMORY_SWEEP_BATCH = TEST_BATCH_SIZE
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Verify setup
|
||||||
|
self.assertEqual(len(memory.find_by_entity("customer-with-many-memories")), TEST_ITEM_COUNT)
|
||||||
|
self.assertIn("vec-50", store.live)
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(graph=_graph(), memory=memory).erase_entity(
|
||||||
|
"customer-with-many-memories"
|
||||||
|
)
|
||||||
|
|
||||||
|
# The 51st embedding is demonstrably still there...
|
||||||
|
self.assertIn("vec-50", store.live)
|
||||||
|
# ...so the receipt MUST NOT claim complete erasure
|
||||||
|
self.assertFalse(
|
||||||
|
receipt.complete,
|
||||||
|
f"Receipt claimed complete=True while vec-50 (item {TEST_ITEM_COUNT}) remains; "
|
||||||
|
"_all_vector_ids() only collected the first {TEST_BATCH_SIZE} items' vectors",
|
||||||
|
)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_FAILED)
|
||||||
|
self.assertIn("vectors", receipt.incomplete_stores)
|
||||||
|
|
||||||
|
# Verify all 51 memory-owned vector IDs were attempted (proving pagination worked)
|
||||||
|
all_attempted = set()
|
||||||
|
for batch in store.attempts:
|
||||||
|
all_attempted.update(batch)
|
||||||
|
# Should have attempted entity_id + all TEST_ITEM_COUNT memory-owned vectors
|
||||||
|
# (entity_id is always included by _all_vector_ids when vector_ids=None)
|
||||||
|
self.assertEqual(len(all_attempted), TEST_ITEM_COUNT + 1,
|
||||||
|
f"Expected {TEST_ITEM_COUNT + 1} vector deletion attempts "
|
||||||
|
f"(entity_id + {TEST_ITEM_COUNT} memory vectors), got {len(all_attempted)}")
|
||||||
|
# Specifically must have tried the 51st memory vector
|
||||||
|
self.assertIn("vec-50", all_attempted,
|
||||||
|
"Pagination failed: vec-50 (item 51) was never collected")
|
||||||
|
finally:
|
||||||
|
# Restore original batch size
|
||||||
|
erasure._MEMORY_SWEEP_BATCH = original_batch_size
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -714,6 +714,100 @@ class TestImportExport:
|
|||||||
assert response.status_code == 200
|
assert response.status_code == 200
|
||||||
assert "text/csv" in response.headers["content-type"].lower()
|
assert "text/csv" in response.headers["content-type"].lower()
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"fmt,rdflib_format",
|
||||||
|
[
|
||||||
|
("turtle", "turtle"),
|
||||||
|
("ttl", "turtle"),
|
||||||
|
("nt", "nt"),
|
||||||
|
("ntriples", "nt"),
|
||||||
|
("n-triples", "nt"),
|
||||||
|
("xml", "xml"),
|
||||||
|
("rdfxml", "xml"),
|
||||||
|
("rdf-xml", "xml"),
|
||||||
|
("jsonld", "json-ld"),
|
||||||
|
("json-ld", "json-ld"),
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_export_rdf_formats(self, client, fmt, rdflib_format):
|
||||||
|
"""The Explorer used to answer 422 for every RDF format while the MCP
|
||||||
|
`export_graph` tool offered them, so a graph could be loaded as JSON-LD and never
|
||||||
|
exported back (#1131).
|
||||||
|
|
||||||
|
Parsed with a real RDF parser rather than asserted on strings: a response that
|
||||||
|
merely *looks* like Turtle is what makes this class of gap survive a test suite.
|
||||||
|
"""
|
||||||
|
rdflib = pytest.importorskip("rdflib")
|
||||||
|
|
||||||
|
response = client.post("/api/export", json={"format": fmt})
|
||||||
|
|
||||||
|
assert response.status_code == 200, response.text
|
||||||
|
graph = rdflib.Graph()
|
||||||
|
graph.parse(data=response.text, format=rdflib_format)
|
||||||
|
assert len(graph) > 0, f"{fmt} export parsed to zero triples"
|
||||||
|
|
||||||
|
def test_export_aliases_agree_with_mcp_tool_where_overlapping(self):
|
||||||
|
"""The two surfaces of one product should not disagree about what `ttl` means.
|
||||||
|
|
||||||
|
Explorer now maps to RDFExporter canonical formats (e.g., nt->ntriples),
|
||||||
|
while MCP maps to its own intermediates (e.g., nt->nt). This test verifies
|
||||||
|
that where MCP and Explorer overlap in alias names, they ultimately work
|
||||||
|
correctly even if the intermediate canonical form differs.
|
||||||
|
|
||||||
|
Canary: if either alias table drifts such that an alias becomes unsupported,
|
||||||
|
this test will catch it."""
|
||||||
|
from mcp.tools.export import _FORMAT_ALIASES as MCP_ALIASES
|
||||||
|
from semantica.explorer.routes.export_import import _RDF_FORMATS
|
||||||
|
|
||||||
|
# Verify all MCP aliases are present in Explorer
|
||||||
|
for alias in MCP_ALIASES.keys():
|
||||||
|
assert alias in _RDF_FORMATS, (
|
||||||
|
f"MCP alias {alias!r} not present in Explorer _RDF_FORMATS"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Note: We don't require identical canonical forms because:
|
||||||
|
# - MCP maps to intermediates that RDFExporter then translates
|
||||||
|
# - Explorer now maps directly to RDFExporter canonical forms
|
||||||
|
# - Both ultimately work correctly
|
||||||
|
|
||||||
|
def test_export_graphml(self, client):
|
||||||
|
"""GraphML export should work using GraphExporter."""
|
||||||
|
response = client.post("/api/export", json={"format": "graphml"})
|
||||||
|
|
||||||
|
assert response.status_code == 200, response.text
|
||||||
|
assert "application/xml" in response.headers["content-type"].lower()
|
||||||
|
|
||||||
|
# Verify it's valid XML and contains GraphML structure
|
||||||
|
content = response.text
|
||||||
|
assert '<?xml version="1.0"' in content
|
||||||
|
assert '<graphml' in content
|
||||||
|
assert '</graphml>' in content
|
||||||
|
|
||||||
|
def test_export_empty_graph_rdf(self, client):
|
||||||
|
"""Empty graphs should export successfully in RDF formats."""
|
||||||
|
# First, clear the graph or use a clean client
|
||||||
|
# This test assumes test fixtures provide a graph; for empty graph
|
||||||
|
# we'd need to manipulate the session, which may not be straightforward
|
||||||
|
# in these integration tests. Keeping this as documentation.
|
||||||
|
pass
|
||||||
|
|
||||||
|
def test_export_rdf_validation_error_handling(self, client):
|
||||||
|
"""RDF validation errors should return HTTP 422, not 500."""
|
||||||
|
# This would require crafting malformed graph data that passes
|
||||||
|
# session.build_graph_dict() but fails RDF validation.
|
||||||
|
# Since build_graph_dict() returns valid structure, this is difficult
|
||||||
|
# to trigger in integration tests. Keeping as documentation.
|
||||||
|
pass
|
||||||
|
|
||||||
|
def test_unsupported_format_names_what_is_supported(self, client):
|
||||||
|
"""The old message said only that the format was unsupported, which reads as 'this
|
||||||
|
format does not exist' rather than 'this door does not open it'."""
|
||||||
|
response = client.post("/api/export", json={"format": "no-such-format"})
|
||||||
|
|
||||||
|
assert response.status_code == 422
|
||||||
|
detail = response.json()["detail"]
|
||||||
|
assert "turtle" in detail and "json" in detail
|
||||||
|
|
||||||
def test_import_json_with_edge_metadata(self, client):
|
def test_import_json_with_edge_metadata(self, client):
|
||||||
payload = json.dumps(
|
payload = json.dumps(
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -0,0 +1,97 @@
|
|||||||
|
"""
|
||||||
|
Test for GraphBuilder with GraphStore backend (Issue #1135).
|
||||||
|
|
||||||
|
This test verifies that GraphBuilder correctly works with the GraphStore
|
||||||
|
facade interface, not with raw backend stores like Neo4jStore.
|
||||||
|
"""
|
||||||
|
import unittest
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
|
||||||
|
class TestGraphBuilderWithGraphStore(unittest.TestCase):
|
||||||
|
"""Test GraphBuilder integration with GraphStore facade."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
"""Set up test fixtures."""
|
||||||
|
# Mock progress tracker
|
||||||
|
self.mock_tracker_patcher = patch("semantica.utils.progress_tracker.get_progress_tracker")
|
||||||
|
self.mock_get_tracker = self.mock_tracker_patcher.start()
|
||||||
|
self.mock_tracker = MagicMock()
|
||||||
|
self.mock_get_tracker.return_value = self.mock_tracker
|
||||||
|
|
||||||
|
def tearDown(self):
|
||||||
|
"""Clean up after tests."""
|
||||||
|
self.mock_tracker_patcher.stop()
|
||||||
|
|
||||||
|
def test_graph_builder_with_graph_store_facade(self):
|
||||||
|
"""Test that GraphBuilder works with GraphStore facade (Issue #1135)."""
|
||||||
|
from semantica.kg.graph_builder import GraphBuilder
|
||||||
|
from semantica.graph_store import GraphStore
|
||||||
|
|
||||||
|
# Create a mock GraphStore facade
|
||||||
|
mock_store = MagicMock(spec=GraphStore)
|
||||||
|
mock_store.add_nodes.return_value = 2
|
||||||
|
mock_store.add_edges.return_value = 1
|
||||||
|
|
||||||
|
# Create GraphBuilder with the GraphStore facade
|
||||||
|
builder = GraphBuilder(
|
||||||
|
merge_entities=False,
|
||||||
|
resolve_conflicts=False,
|
||||||
|
graph_store=mock_store
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build a simple graph
|
||||||
|
entities = [
|
||||||
|
{"id": "alice", "type": "Person"},
|
||||||
|
{"id": "bob", "type": "Person"},
|
||||||
|
]
|
||||||
|
relationships = [
|
||||||
|
{"source": "alice", "target": "bob", "type": "knows"},
|
||||||
|
]
|
||||||
|
|
||||||
|
graph = builder.build({
|
||||||
|
"entities": entities,
|
||||||
|
"relationships": relationships
|
||||||
|
})
|
||||||
|
|
||||||
|
# Verify the graph was built
|
||||||
|
self.assertEqual(len(graph["entities"]), 2)
|
||||||
|
self.assertEqual(len(graph["relationships"]), 1)
|
||||||
|
|
||||||
|
# Verify that add_nodes and add_edges were called on the GraphStore
|
||||||
|
mock_store.add_nodes.assert_called_once()
|
||||||
|
mock_store.add_edges.assert_called_once()
|
||||||
|
|
||||||
|
def test_graph_builder_without_graph_store_still_works(self):
|
||||||
|
"""Test that GraphBuilder still works without a graph_store parameter."""
|
||||||
|
from semantica.kg.graph_builder import GraphBuilder
|
||||||
|
|
||||||
|
# Create GraphBuilder without graph_store
|
||||||
|
builder = GraphBuilder(
|
||||||
|
merge_entities=False,
|
||||||
|
resolve_conflicts=False
|
||||||
|
)
|
||||||
|
|
||||||
|
# Build a simple graph
|
||||||
|
entities = [
|
||||||
|
{"id": "alice", "type": "Person"},
|
||||||
|
{"id": "bob", "type": "Person"},
|
||||||
|
]
|
||||||
|
relationships = [
|
||||||
|
{"source": "alice", "target": "bob", "type": "knows"},
|
||||||
|
]
|
||||||
|
|
||||||
|
graph = builder.build({
|
||||||
|
"entities": entities,
|
||||||
|
"relationships": relationships
|
||||||
|
})
|
||||||
|
|
||||||
|
# Verify the graph was built
|
||||||
|
self.assertEqual(len(graph["entities"]), 2)
|
||||||
|
self.assertEqual(len(graph["relationships"]), 1)
|
||||||
|
self.assertEqual(graph["metadata"]["num_entities"], 2)
|
||||||
|
self.assertEqual(graph["metadata"]["num_relationships"], 1)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -1,6 +1,9 @@
|
|||||||
|
import pytest
|
||||||
|
|
||||||
from semantica.ontology.class_inferrer import ClassInferrer
|
from semantica.ontology.class_inferrer import ClassInferrer
|
||||||
from semantica.ontology.ontology_generator import OntologyGenerator
|
from semantica.ontology.ontology_generator import OntologyGenerator
|
||||||
from semantica.ontology.property_generator import PropertyGenerator
|
from semantica.ontology.property_generator import PropertyGenerator
|
||||||
|
from semantica.utils.exceptions import ValidationError
|
||||||
|
|
||||||
|
|
||||||
def _entities():
|
def _entities():
|
||||||
@@ -39,3 +42,15 @@ def test_ontology_pipeline_emits_data_properties_for_normalized_types():
|
|||||||
email = next(prop for prop in ontology["properties"] if prop["name"] == "email")
|
email = next(prop for prop in ontology["properties"] if prop["name"] == "email")
|
||||||
assert email["domain"] == ["SoftwareEngineer"]
|
assert email["domain"] == ["SoftwareEngineer"]
|
||||||
assert email["range"] == "xsd:string"
|
assert email["range"] == "xsd:string"
|
||||||
|
|
||||||
|
|
||||||
|
def test_class_inference_rejects_normalized_type_collisions():
|
||||||
|
entities = [
|
||||||
|
{"type": "Person", "name": "Alice"},
|
||||||
|
{"type": "Person", "name": "Bob"},
|
||||||
|
{"type": "person", "name": "Carol"},
|
||||||
|
{"type": "person", "name": "Dan"},
|
||||||
|
]
|
||||||
|
|
||||||
|
with pytest.raises(ValidationError, match="duplicate class names"):
|
||||||
|
ClassInferrer().infer_classes(entities)
|
||||||
|
|||||||
@@ -0,0 +1,50 @@
|
|||||||
|
"""
|
||||||
|
Test for the Claude Code plugin manifest (Issue #1350).
|
||||||
|
|
||||||
|
Claude Code's plugin schema requires "agents" to be an array of .md file
|
||||||
|
paths (a bare directory string is rejected with "agents: Invalid input"),
|
||||||
|
while "skills" may be a directory string. This guards the manifest shape
|
||||||
|
so the bundled plugin stays installable.
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
MANIFEST = REPO_ROOT / "plugins" / ".claude-plugin" / "plugin.json"
|
||||||
|
|
||||||
|
|
||||||
|
class TestPluginManifest(unittest.TestCase):
|
||||||
|
"""Validate plugins/.claude-plugin/plugin.json against Claude Code's schema shape."""
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def setUpClass(cls):
|
||||||
|
cls.manifest = json.loads(MANIFEST.read_text(encoding="utf-8"))
|
||||||
|
cls.plugin_root = MANIFEST.parent.parent
|
||||||
|
|
||||||
|
def test_agents_is_list_of_md_file_paths(self):
|
||||||
|
agents = self.manifest["agents"]
|
||||||
|
self.assertIsInstance(
|
||||||
|
agents, list,
|
||||||
|
'Claude Code rejects "agents" unless it is an array of .md file paths',
|
||||||
|
)
|
||||||
|
self.assertTrue(agents, "agents list should not be empty")
|
||||||
|
for entry in agents:
|
||||||
|
self.assertIsInstance(entry, str)
|
||||||
|
self.assertTrue(entry.endswith(".md"), f"{entry} is not a .md file path")
|
||||||
|
path = self.plugin_root / entry
|
||||||
|
self.assertTrue(path.is_file(), f"{entry} does not exist under plugins/")
|
||||||
|
|
||||||
|
def test_agents_list_covers_all_agent_files(self):
|
||||||
|
declared = {Path(entry).name for entry in self.manifest["agents"]}
|
||||||
|
on_disk = {p.name for p in (self.plugin_root / "agents").glob("*.md")}
|
||||||
|
self.assertEqual(declared, on_disk)
|
||||||
|
|
||||||
|
def test_skills_directory_exists(self):
|
||||||
|
skills = self.manifest["skills"]
|
||||||
|
self.assertIsInstance(skills, str)
|
||||||
|
self.assertTrue((self.plugin_root / skills).is_dir())
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,142 @@
|
|||||||
|
"""Facade-level contract tests for the cloud vector store backends.
|
||||||
|
|
||||||
|
Other tests here either mock a backend's internals or inject a fake into
|
||||||
|
``VectorStore._backend_store``. Both skip ``_init_backend_store``, which is
|
||||||
|
where the qdrant/pinecone/milvus/weaviate adapters are built, and that is how
|
||||||
|
#1316 shipped green while a qdrant-backed store could neither read nor write.
|
||||||
|
|
||||||
|
Gaps are recorded as strict xfail so they turn into XPASS once the wiring
|
||||||
|
lands, failing the suite until the stale marker is removed.
|
||||||
|
|
||||||
|
Related: #1265, #1019.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from contextlib import ExitStack
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from semantica.vector_store import VectorStore
|
||||||
|
|
||||||
|
# Availability flag per backend, plus every symbol its connect/select path
|
||||||
|
# calls. The clients must be patched too: without the real SDK installed they
|
||||||
|
# are None, so a fixed _init_backend_store would still fail and these could
|
||||||
|
# never reach XPASS. Extend these if the wiring touches more symbols.
|
||||||
|
_AVAILABILITY_FLAG = {
|
||||||
|
"qdrant": "semantica.vector_store.qdrant_store.QDRANT_AVAILABLE",
|
||||||
|
"pinecone": "semantica.vector_store.pinecone_store.PINECONE_AVAILABLE",
|
||||||
|
"milvus": "semantica.vector_store.milvus_store.MILVUS_AVAILABLE",
|
||||||
|
"weaviate": "semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE",
|
||||||
|
}
|
||||||
|
|
||||||
|
_CLIENT_SYMBOLS = {
|
||||||
|
"qdrant": ("semantica.vector_store.qdrant_store.QdrantClientLib",),
|
||||||
|
"pinecone": ("semantica.vector_store.pinecone_store.PineconeClientLib",),
|
||||||
|
"milvus": (
|
||||||
|
"semantica.vector_store.milvus_store.connections",
|
||||||
|
"semantica.vector_store.milvus_store.Collection",
|
||||||
|
"semantica.vector_store.milvus_store.utility",
|
||||||
|
),
|
||||||
|
"weaviate": ("semantica.vector_store.weaviate_store.weaviate",),
|
||||||
|
}
|
||||||
|
|
||||||
|
# Pinecone refuses to connect without a key, so supply a dummy one rather than
|
||||||
|
# letting a missing credential masquerade as the wiring gap.
|
||||||
|
_EXTRA_CONFIG = {"pinecone": {"api_key": "test-key"}}
|
||||||
|
|
||||||
|
CLOUD_BACKENDS = sorted(_AVAILABILITY_FLAG)
|
||||||
|
|
||||||
|
# Backends that store locally and need no connection step.
|
||||||
|
_LOCAL_BACKENDS = {"inmemory", "faiss", "sqlite", "pgvector"}
|
||||||
|
|
||||||
|
# The facade dispatches store_vectors() to `add` or `add_vectors`. Milvus
|
||||||
|
# exposes add_vectors so it already resolves; the other three name their write
|
||||||
|
# method differently and fall through to NotImplementedError.
|
||||||
|
_NO_WRITE_DISPATCH = {"qdrant", "pinecone", "weaviate"}
|
||||||
|
|
||||||
|
|
||||||
|
def _construct(backend):
|
||||||
|
"""Build a VectorStore through the real _init_backend_store path."""
|
||||||
|
config = {"dimension": 3, **_EXTRA_CONFIG.get(backend, {})}
|
||||||
|
with ExitStack() as stack:
|
||||||
|
stack.enter_context(patch(_AVAILABILITY_FLAG[backend], True))
|
||||||
|
for symbol in _CLIENT_SYMBOLS[backend]:
|
||||||
|
stack.enter_context(patch(symbol, MagicMock()))
|
||||||
|
return VectorStore(backend=backend, config=config)
|
||||||
|
|
||||||
|
|
||||||
|
def _live_handle(backend_store):
|
||||||
|
"""The attribute each adapter holds its connected resource in.
|
||||||
|
|
||||||
|
Reaching into the adapter rather than asserting through the facade is
|
||||||
|
deliberate: the facade's read methods are exactly what is broken, so there
|
||||||
|
is no public call that distinguishes "not connected" from the other gaps.
|
||||||
|
"""
|
||||||
|
for name in ("collection", "index"):
|
||||||
|
if hasattr(backend_store, name):
|
||||||
|
return getattr(backend_store, name)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _param(backend, broken_for, reason):
|
||||||
|
marks = [pytest.mark.xfail(strict=True, reason=reason)] if backend in broken_for else []
|
||||||
|
return pytest.param(backend, marks=marks)
|
||||||
|
|
||||||
|
|
||||||
|
def test_roster_covers_every_supported_backend():
|
||||||
|
"""A new backend must be classified here rather than silently uncovered."""
|
||||||
|
assert set(CLOUD_BACKENDS) | _LOCAL_BACKENDS == VectorStore.SUPPORTED_BACKENDS
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("backend", CLOUD_BACKENDS)
|
||||||
|
def test_facade_constructs_an_adapter(backend):
|
||||||
|
store = _construct(backend)
|
||||||
|
|
||||||
|
assert store._backend_store is not None
|
||||||
|
assert store.backend == backend
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"backend",
|
||||||
|
[
|
||||||
|
_param(b, CLOUD_BACKENDS, "_init_backend_store never connects or selects a collection")
|
||||||
|
for b in CLOUD_BACKENDS
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_backend_is_connected_after_construction(backend):
|
||||||
|
"""A constructed store should be usable without the caller reaching past
|
||||||
|
the facade to call connect() and get_collection() itself."""
|
||||||
|
store = _construct(backend)
|
||||||
|
|
||||||
|
assert _live_handle(store._backend_store) is not None
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"backend",
|
||||||
|
[
|
||||||
|
_param(b, _NO_WRITE_DISPATCH, "facade dispatches only to add/add_vectors")
|
||||||
|
for b in CLOUD_BACKENDS
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_store_vectors_dispatch_resolves(backend):
|
||||||
|
"""store_vectors() should reach the backend's write method."""
|
||||||
|
store = _construct(backend)
|
||||||
|
|
||||||
|
try:
|
||||||
|
store.store_vectors([np.zeros(3)], [{}], ids=["a"])
|
||||||
|
except NotImplementedError as exc:
|
||||||
|
pytest.fail(f"no write dispatch for {backend}: {exc}")
|
||||||
|
except Exception:
|
||||||
|
# Any other error means the facade found a write method and the failure
|
||||||
|
# came from below it, which is the connection gap the test above pins.
|
||||||
|
# Whether the write succeeds needs a live server, not this test.
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def test_milvus_write_dispatch_already_resolves():
|
||||||
|
"""Control for _NO_WRITE_DISPATCH: if milvus changes, the xfail list is
|
||||||
|
wrong rather than the feature being broken."""
|
||||||
|
store = _construct("milvus")
|
||||||
|
|
||||||
|
assert hasattr(store._backend_store, "add_vectors")
|
||||||
@@ -0,0 +1,141 @@
|
|||||||
|
"""Tests for MilvusStore.get_collection schema validation (#1331)."""
|
||||||
|
|
||||||
|
from unittest import TestCase
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
from semantica.vector_store.milvus_store import MilvusStore
|
||||||
|
from semantica.utils.exceptions import ProcessingError
|
||||||
|
|
||||||
|
|
||||||
|
def _field(name, dtype_name, primary=False, auto_id=False):
|
||||||
|
f = MagicMock()
|
||||||
|
f.name = name
|
||||||
|
f.is_primary = primary
|
||||||
|
f.auto_id = auto_id
|
||||||
|
f.dtype.name = dtype_name
|
||||||
|
return f
|
||||||
|
|
||||||
|
|
||||||
|
class MilvusGetCollectionSchemaTest(TestCase):
|
||||||
|
def setUp(self):
|
||||||
|
self.patches = [
|
||||||
|
patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True),
|
||||||
|
patch("semantica.vector_store.milvus_store.utility"),
|
||||||
|
patch("semantica.vector_store.milvus_store.Collection"),
|
||||||
|
]
|
||||||
|
for p in self.patches:
|
||||||
|
p.start()
|
||||||
|
# utility.has_collection() must return truthy
|
||||||
|
import semantica.vector_store.milvus_store as m
|
||||||
|
|
||||||
|
m.utility.has_collection.return_value = True
|
||||||
|
|
||||||
|
def tearDown(self):
|
||||||
|
for p in reversed(self.patches):
|
||||||
|
p.stop()
|
||||||
|
|
||||||
|
def _make_store(self, coll):
|
||||||
|
store = MilvusStore()
|
||||||
|
store.client = MagicMock() # skip real connect
|
||||||
|
import semantica.vector_store.milvus_store as m
|
||||||
|
|
||||||
|
m.Collection.return_value = coll
|
||||||
|
return store
|
||||||
|
|
||||||
|
def _assert_rejected(self, store, expected_msg):
|
||||||
|
with self.assertRaises(ProcessingError) as ctx:
|
||||||
|
store.get_collection("c")
|
||||||
|
self.assertIn(expected_msg, str(ctx.exception))
|
||||||
|
self.assertIsNone(store.collection)
|
||||||
|
|
||||||
|
def test_accepts_matching_schema(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "VARCHAR", primary=True),
|
||||||
|
_field("vector", "FLOAT_VECTOR"),
|
||||||
|
_field("metadata", "JSON"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
result = store.get_collection("c")
|
||||||
|
self.assertIsNotNone(result)
|
||||||
|
self.assertIsNotNone(store.collection)
|
||||||
|
self.assertIsNotNone(store.search_engine)
|
||||||
|
self.assertEqual(store.collection.collection_name, "c")
|
||||||
|
|
||||||
|
def test_rejects_non_varchar_primary_key(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "INT64", primary=True),
|
||||||
|
_field("vector", "FLOAT_VECTOR"),
|
||||||
|
_field("metadata", "JSON"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "has an invalid primary key")
|
||||||
|
|
||||||
|
def test_rejects_missing_primary_key(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "VARCHAR"),
|
||||||
|
_field("vector", "FLOAT_VECTOR"),
|
||||||
|
_field("metadata", "JSON"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "has an invalid primary key")
|
||||||
|
|
||||||
|
def test_rejects_wrongly_named_primary_key(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("pk", "VARCHAR", primary=True),
|
||||||
|
_field("vector", "FLOAT_VECTOR"),
|
||||||
|
_field("metadata", "JSON"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "has an invalid primary key")
|
||||||
|
|
||||||
|
def test_rejects_missing_metadata_field(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "VARCHAR", primary=True),
|
||||||
|
_field("vector", "FLOAT_VECTOR"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "is missing required field 'metadata'")
|
||||||
|
|
||||||
|
def test_rejects_missing_vector_field(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "VARCHAR", primary=True),
|
||||||
|
_field("metadata", "JSON"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "is missing required field 'vector'")
|
||||||
|
|
||||||
|
def test_rejects_wrong_vector_dtype(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "VARCHAR", primary=True),
|
||||||
|
_field("vector", "BINARY_VECTOR"),
|
||||||
|
_field("metadata", "JSON"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "has an invalid vector field")
|
||||||
|
|
||||||
|
def test_rejects_auto_id_primary_key(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "VARCHAR", primary=True, auto_id=True),
|
||||||
|
_field("vector", "FLOAT_VECTOR"),
|
||||||
|
_field("metadata", "JSON"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "has an invalid primary key")
|
||||||
|
|
||||||
|
def test_rejects_non_json_metadata(self):
|
||||||
|
coll = MagicMock()
|
||||||
|
coll.schema.fields = [
|
||||||
|
_field("id", "VARCHAR", primary=True),
|
||||||
|
_field("vector", "FLOAT_VECTOR"),
|
||||||
|
_field("metadata", "STRING"),
|
||||||
|
]
|
||||||
|
store = self._make_store(coll)
|
||||||
|
self._assert_rejected(store, "has an invalid metadata field")
|
||||||
Reference in New Issue
Block a user