diff --git a/.sqlx-postgres/query-09d9c924a02b20f667212d907d7297433d8b2137b8922f3c9ac3f0d66d6adfaa.json b/.sqlx-postgres/query-09d9c924a02b20f667212d907d7297433d8b2137b8922f3c9ac3f0d66d6adfaa.json new file mode 100644 index 00000000..bf9b6cb5 --- /dev/null +++ b/.sqlx-postgres/query-09d9c924a02b20f667212d907d7297433d8b2137b8922f3c9ac3f0d66d6adfaa.json @@ -0,0 +1,18 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"content_revisions\" (\"content_type\", \"record_id\", \"revision_number\", \"snapshot\", \"created_by\") VALUES ($1, $2, $3, $4, $5)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Varchar", + "Int8", + "Int8", + "Jsonb", + "Int8" + ] + }, + "nullable": [] + }, + "hash": "09d9c924a02b20f667212d907d7297433d8b2137b8922f3c9ac3f0d66d6adfaa" +} diff --git a/.sqlx-postgres/query-0bec1bf43b85130c239be6fdc11bc25871bf9a1cd4fb25ad19fe76cef09d54ed.json b/.sqlx-postgres/query-0bec1bf43b85130c239be6fdc11bc25871bf9a1cd4fb25ad19fe76cef09d54ed.json new file mode 100644 index 00000000..51fe5949 --- /dev/null +++ b/.sqlx-postgres/query-0bec1bf43b85130c239be6fdc11bc25871bf9a1cd4fb25ad19fe76cef09d54ed.json @@ -0,0 +1,23 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_documents\" (\"id\", \"kb_id\", \"title\", \"source\", \"storage_key\", \"mime_type\", \"size\", \"created_by\", \"created_at\", \"tenant_id\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Text", + "Text", + "Text", + "Text", + "Int8", + "Int8", + "Timestamptz", + "Text" + ] + }, + "nullable": [] + }, + "hash": "0bec1bf43b85130c239be6fdc11bc25871bf9a1cd4fb25ad19fe76cef09d54ed" +} diff --git a/.sqlx-postgres/query-6c6a16d47c7b266ae3b076676dde9f161f4c1b551848a4c0c684e52ec67caf68.json b/.sqlx-postgres/query-6c6a16d47c7b266ae3b076676dde9f161f4c1b551848a4c0c684e52ec67caf68.json new file mode 100644 index 00000000..93c04124 --- /dev/null +++ b/.sqlx-postgres/query-6c6a16d47c7b266ae3b076676dde9f161f4c1b551848a4c0c684e52ec67caf68.json @@ -0,0 +1,22 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_faqs\" (\"id\", \"kb_id\", \"standard_question\", \"similar_questions\", \"answers\", \"enabled\", \"created_by\", \"created_at\", \"tenant_id\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Text", + "Jsonb", + "Jsonb", + "Bool", + "Int8", + "Timestamptz", + "Text" + ] + }, + "nullable": [] + }, + "hash": "6c6a16d47c7b266ae3b076676dde9f161f4c1b551848a4c0c684e52ec67caf68" +} diff --git a/.sqlx-postgres/query-78aa9d70db29561c1cd08c4cad69df67c84676f2a4506bd2d5391da344cb16d9.json b/.sqlx-postgres/query-78aa9d70db29561c1cd08c4cad69df67c84676f2a4506bd2d5391da344cb16d9.json new file mode 100644 index 00000000..2d3522d8 --- /dev/null +++ b/.sqlx-postgres/query-78aa9d70db29561c1cd08c4cad69df67c84676f2a4506bd2d5391da344cb16d9.json @@ -0,0 +1,23 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_knowledge_bases\" (\"id\", \"name\", \"description\", \"slug\", \"kind\", \"indexing_strategy\", \"embedding_model\", \"embedding_dim\", \"created_at\", \"tenant_id\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Text", + "Text", + "Text", + "Text", + "Jsonb", + "Text", + "Int8", + "Timestamptz", + "Text" + ] + }, + "nullable": [] + }, + "hash": "78aa9d70db29561c1cd08c4cad69df67c84676f2a4506bd2d5391da344cb16d9" +} diff --git a/.sqlx-postgres/query-813e13d36e81fe85e93989d63a060884347aa2fe0f5071483faae846c99d1810.json b/.sqlx-postgres/query-813e13d36e81fe85e93989d63a060884347aa2fe0f5071483faae846c99d1810.json new file mode 100644 index 00000000..e8841a4c --- /dev/null +++ b/.sqlx-postgres/query-813e13d36e81fe85e93989d63a060884347aa2fe0f5071483faae846c99d1810.json @@ -0,0 +1,24 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_wiki_pages\" (\"id\", \"kb_id\", \"title\", \"slug\", \"status\", \"content\", \"summary\", \"linked_page_ids\", \"created_by\", \"created_at\", \"tenant_id\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Text", + "Text", + "Text", + "Text", + "Text", + "Jsonb", + "Int8", + "Timestamptz", + "Text" + ] + }, + "nullable": [] + }, + "hash": "813e13d36e81fe85e93989d63a060884347aa2fe0f5071483faae846c99d1810" +} diff --git a/.sqlx-postgres/query-8706e0d7ce25b6cd52161b1cf09ebd43c77d6c45eda7a692d0329492af00b04e.json b/.sqlx-postgres/query-8706e0d7ce25b6cd52161b1cf09ebd43c77d6c45eda7a692d0329492af00b04e.json new file mode 100644 index 00000000..363f9ca9 --- /dev/null +++ b/.sqlx-postgres/query-8706e0d7ce25b6cd52161b1cf09ebd43c77d6c45eda7a692d0329492af00b04e.json @@ -0,0 +1,22 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_query_logs\" (\"id\", \"kb_id\", \"question\", \"answer\", \"cited_units\", \"status\", \"top_score\", \"user_id\", \"created_at\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Text", + "Text", + "Jsonb", + "Text", + "Float8", + "Int8", + "Timestamptz" + ] + }, + "nullable": [] + }, + "hash": "8706e0d7ce25b6cd52161b1cf09ebd43c77d6c45eda7a692d0329492af00b04e" +} diff --git a/.sqlx-postgres/query-ad4c536899f56911752a92c69c8e8d44508245405b1ba58b6ec2725e3cee0748.json b/.sqlx-postgres/query-ad4c536899f56911752a92c69c8e8d44508245405b1ba58b6ec2725e3cee0748.json new file mode 100644 index 00000000..e46a642b --- /dev/null +++ b/.sqlx-postgres/query-ad4c536899f56911752a92c69c8e8d44508245405b1ba58b6ec2725e3cee0748.json @@ -0,0 +1,22 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_documents\" (\"id\", \"kb_id\", \"title\", \"source\", \"storage_key\", \"mime_type\", \"size\", \"created_by\", \"created_at\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Text", + "Text", + "Text", + "Text", + "Int8", + "Int8", + "Timestamptz" + ] + }, + "nullable": [] + }, + "hash": "ad4c536899f56911752a92c69c8e8d44508245405b1ba58b6ec2725e3cee0748" +} diff --git a/.sqlx-postgres/query-d288d14d7a964ffcac9b98b36863cec6aafc50ceb50f878c5c2251cb43ebda39.json b/.sqlx-postgres/query-d288d14d7a964ffcac9b98b36863cec6aafc50ceb50f878c5c2251cb43ebda39.json new file mode 100644 index 00000000..149d3f60 --- /dev/null +++ b/.sqlx-postgres/query-d288d14d7a964ffcac9b98b36863cec6aafc50ceb50f878c5c2251cb43ebda39.json @@ -0,0 +1,20 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_wiki_sources\" (\"id\", \"page_id\", \"page_revision\", \"doc_id\", \"span_start\", \"span_end\", \"created_at\") VALUES ($1, $2, $3, $4, $5, $6, $7)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Int8", + "Int8", + "Int8", + "Int8", + "Timestamptz" + ] + }, + "nullable": [] + }, + "hash": "d288d14d7a964ffcac9b98b36863cec6aafc50ceb50f878c5c2251cb43ebda39" +} diff --git a/.sqlx-postgres/query-dc1c6ca616d472d5493e98d95d307e50e34d9ec95c7b0bedbed67dd79a249b23.json b/.sqlx-postgres/query-dc1c6ca616d472d5493e98d95d307e50e34d9ec95c7b0bedbed67dd79a249b23.json new file mode 100644 index 00000000..8425030a --- /dev/null +++ b/.sqlx-postgres/query-dc1c6ca616d472d5493e98d95d307e50e34d9ec95c7b0bedbed67dd79a249b23.json @@ -0,0 +1,23 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_wiki_pages\" (\"id\", \"kb_id\", \"title\", \"slug\", \"status\", \"content\", \"summary\", \"linked_page_ids\", \"created_by\", \"created_at\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Text", + "Text", + "Text", + "Text", + "Text", + "Jsonb", + "Int8", + "Timestamptz" + ] + }, + "nullable": [] + }, + "hash": "dc1c6ca616d472d5493e98d95d307e50e34d9ec95c7b0bedbed67dd79a249b23" +} diff --git a/.sqlx-postgres/query-e0455a4e23c38dfe4c81da49afb33e9b98a2ba81a8654188a7d6a3bbd95dbeff.json b/.sqlx-postgres/query-e0455a4e23c38dfe4c81da49afb33e9b98a2ba81a8654188a7d6a3bbd95dbeff.json new file mode 100644 index 00000000..bd7fd4ad --- /dev/null +++ b/.sqlx-postgres/query-e0455a4e23c38dfe4c81da49afb33e9b98a2ba81a8654188a7d6a3bbd95dbeff.json @@ -0,0 +1,21 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_faqs\" (\"id\", \"kb_id\", \"standard_question\", \"similar_questions\", \"answers\", \"enabled\", \"created_by\", \"created_at\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Text", + "Jsonb", + "Jsonb", + "Bool", + "Int8", + "Timestamptz" + ] + }, + "nullable": [] + }, + "hash": "e0455a4e23c38dfe4c81da49afb33e9b98a2ba81a8654188a7d6a3bbd95dbeff" +} diff --git a/.sqlx-postgres/query-eb7318e16b7f5f95b5982ec568aac909ae00b720e36223af36a67182275d44ec.json b/.sqlx-postgres/query-eb7318e16b7f5f95b5982ec568aac909ae00b720e36223af36a67182275d44ec.json new file mode 100644 index 00000000..6b2d75f4 --- /dev/null +++ b/.sqlx-postgres/query-eb7318e16b7f5f95b5982ec568aac909ae00b720e36223af36a67182275d44ec.json @@ -0,0 +1,22 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_knowledge_bases\" (\"id\", \"name\", \"description\", \"slug\", \"kind\", \"indexing_strategy\", \"embedding_model\", \"embedding_dim\", \"created_at\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Text", + "Text", + "Text", + "Text", + "Jsonb", + "Text", + "Int8", + "Timestamptz" + ] + }, + "nullable": [] + }, + "hash": "eb7318e16b7f5f95b5982ec568aac909ae00b720e36223af36a67182275d44ec" +} diff --git a/.sqlx-postgres/query-f0e284d8bfa766b901b7d73555026398263d83aa1de304610982f2fe00ab86b9.json b/.sqlx-postgres/query-f0e284d8bfa766b901b7d73555026398263d83aa1de304610982f2fe00ab86b9.json new file mode 100644 index 00000000..17bf097a --- /dev/null +++ b/.sqlx-postgres/query-f0e284d8bfa766b901b7d73555026398263d83aa1de304610982f2fe00ab86b9.json @@ -0,0 +1,29 @@ +{ + "db_name": "PostgreSQL", + "query": "INSERT INTO \"kb_chunks\" (\"id\", \"kb_id\", \"doc_id\", \"faq_id\", \"wiki_page_id\", \"kind\", \"parent_id\", \"seq\", \"content\", \"breadcrumb\", \"byte_start\", \"byte_end\", \"questions\", \"embedding\", \"embedding_model\", \"created_at\") VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16)", + "describe": { + "columns": [], + "parameters": { + "Left": [ + "Int8", + "Int8", + "Int8", + "Int8", + "Int8", + "Text", + "Int8", + "Int8", + "Text", + "Text", + "Int8", + "Int8", + "Jsonb", + "Bytea", + "Text", + "Timestamptz" + ] + }, + "nullable": [] + }, + "hash": "f0e284d8bfa766b901b7d73555026398263d83aa1de304610982f2fe00ab86b9" +} diff --git a/.sqlx-sqlite/query-0ac15c4493172c78f0baf9cdc0b6aa9d022dd2ab156f20d7db42c20cc2708154.json b/.sqlx-sqlite/query-0ac15c4493172c78f0baf9cdc0b6aa9d022dd2ab156f20d7db42c20cc2708154.json new file mode 100644 index 00000000..08b7edf9 --- /dev/null +++ b/.sqlx-sqlite/query-0ac15c4493172c78f0baf9cdc0b6aa9d022dd2ab156f20d7db42c20cc2708154.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_faqs (id, kb_id, standard_question, similar_questions, answers, enabled, created_by, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9)", + "describe": { + "columns": [], + "parameters": { + "Right": 9 + }, + "nullable": [] + }, + "hash": "0ac15c4493172c78f0baf9cdc0b6aa9d022dd2ab156f20d7db42c20cc2708154" +} diff --git a/.sqlx-sqlite/query-0ad861ce29102168279ce940a3e4354f1ef510dedad97bee3194cbe2a1da672d.json b/.sqlx-sqlite/query-0ad861ce29102168279ce940a3e4354f1ef510dedad97bee3194cbe2a1da672d.json new file mode 100644 index 00000000..da6a5d9d --- /dev/null +++ b/.sqlx-sqlite/query-0ad861ce29102168279ce940a3e4354f1ef510dedad97bee3194cbe2a1da672d.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_documents (id, kb_id, title, source, storage_key, mime_type, size, created_by, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)", + "describe": { + "columns": [], + "parameters": { + "Right": 10 + }, + "nullable": [] + }, + "hash": "0ad861ce29102168279ce940a3e4354f1ef510dedad97bee3194cbe2a1da672d" +} diff --git a/.sqlx-sqlite/query-379e82955374a059ef82417d2217f191e13959f2c2ff5334c4772d519e3ca84d.json b/.sqlx-sqlite/query-379e82955374a059ef82417d2217f191e13959f2c2ff5334c4772d519e3ca84d.json deleted file mode 100644 index bf9e6e37..00000000 --- a/.sqlx-sqlite/query-379e82955374a059ef82417d2217f191e13959f2c2ff5334c4772d519e3ca84d.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_faqs (id, kb_id, tenant_id, standard_question, similar_questions, answers, enabled, created_by, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9)", - "describe": { - "columns": [], - "parameters": { - "Right": 9 - }, - "nullable": [] - }, - "hash": "379e82955374a059ef82417d2217f191e13959f2c2ff5334c4772d519e3ca84d" -} diff --git a/.sqlx-sqlite/query-42c239c023850cd8c0b7be15683e71b209eb382669358ab79762ed23198fbdfb.json b/.sqlx-sqlite/query-42c239c023850cd8c0b7be15683e71b209eb382669358ab79762ed23198fbdfb.json deleted file mode 100644 index 5021a00c..00000000 --- a/.sqlx-sqlite/query-42c239c023850cd8c0b7be15683e71b209eb382669358ab79762ed23198fbdfb.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_documents (id, kb_id, tenant_id, title, source, storage_key, mime_type, size, created_by, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11)", - "describe": { - "columns": [], - "parameters": { - "Right": 11 - }, - "nullable": [] - }, - "hash": "42c239c023850cd8c0b7be15683e71b209eb382669358ab79762ed23198fbdfb" -} diff --git a/.sqlx-sqlite/query-5163b808b5a6443f0d2315e16c3a61bff034b5a0a94ca64aca46b53e2bc84f81.json b/.sqlx-sqlite/query-5163b808b5a6443f0d2315e16c3a61bff034b5a0a94ca64aca46b53e2bc84f81.json new file mode 100644 index 00000000..8d4b0fff --- /dev/null +++ b/.sqlx-sqlite/query-5163b808b5a6443f0d2315e16c3a61bff034b5a0a94ca64aca46b53e2bc84f81.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_knowledge_bases (id, name, description, slug, kind, indexing_strategy, embedding_model, embedding_dim, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)", + "describe": { + "columns": [], + "parameters": { + "Right": 10 + }, + "nullable": [] + }, + "hash": "5163b808b5a6443f0d2315e16c3a61bff034b5a0a94ca64aca46b53e2bc84f81" +} diff --git a/.sqlx-sqlite/query-6ba570ed6d4bbae11a5894b7386319790e0d3defe85133146e7918acd25914cc.json b/.sqlx-sqlite/query-6ba570ed6d4bbae11a5894b7386319790e0d3defe85133146e7918acd25914cc.json new file mode 100644 index 00000000..bc8dd90d --- /dev/null +++ b/.sqlx-sqlite/query-6ba570ed6d4bbae11a5894b7386319790e0d3defe85133146e7918acd25914cc.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_wiki_pages (id, kb_id, title, slug, status, content, summary, linked_page_ids, created_by, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11)", + "describe": { + "columns": [], + "parameters": { + "Right": 11 + }, + "nullable": [] + }, + "hash": "6ba570ed6d4bbae11a5894b7386319790e0d3defe85133146e7918acd25914cc" +} diff --git a/.sqlx-sqlite/query-8be3d98a3d29aedbb29f88844455df45bb39a847b0858d0271d964d9d09b5d09.json b/.sqlx-sqlite/query-8be3d98a3d29aedbb29f88844455df45bb39a847b0858d0271d964d9d09b5d09.json deleted file mode 100644 index a70149d4..00000000 --- a/.sqlx-sqlite/query-8be3d98a3d29aedbb29f88844455df45bb39a847b0858d0271d964d9d09b5d09.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_knowledge_bases (id, tenant_id, name, description, slug, kind, indexing_strategy, embedding_model, embedding_dim, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)", - "describe": { - "columns": [], - "parameters": { - "Right": 10 - }, - "nullable": [] - }, - "hash": "8be3d98a3d29aedbb29f88844455df45bb39a847b0858d0271d964d9d09b5d09" -} diff --git a/.sqlx-sqlite/query-a8c45c9dceb37dc3309422cba040b7eb7618b189c6fd8d02fff3713650f9ee8a.json b/.sqlx-sqlite/query-a8c45c9dceb37dc3309422cba040b7eb7618b189c6fd8d02fff3713650f9ee8a.json deleted file mode 100644 index 8ab683e2..00000000 --- a/.sqlx-sqlite/query-a8c45c9dceb37dc3309422cba040b7eb7618b189c6fd8d02fff3713650f9ee8a.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_faqs (id, kb_id, tenant_id, standard_question, similar_questions, answers, enabled, created_by, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)", - "describe": { - "columns": [], - "parameters": { - "Right": 10 - }, - "nullable": [] - }, - "hash": "a8c45c9dceb37dc3309422cba040b7eb7618b189c6fd8d02fff3713650f9ee8a" -} diff --git a/.sqlx-sqlite/query-acefe408a18f22704c36d14371d41b22ba7c219e42a69127c6e9e0b7ca6a6c82.json b/.sqlx-sqlite/query-acefe408a18f22704c36d14371d41b22ba7c219e42a69127c6e9e0b7ca6a6c82.json deleted file mode 100644 index 47ea3c42..00000000 --- a/.sqlx-sqlite/query-acefe408a18f22704c36d14371d41b22ba7c219e42a69127c6e9e0b7ca6a6c82.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_wiki_pages (id, kb_id, tenant_id, title, slug, status, content, summary, linked_page_ids, created_by, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11, ?12)", - "describe": { - "columns": [], - "parameters": { - "Right": 12 - }, - "nullable": [] - }, - "hash": "acefe408a18f22704c36d14371d41b22ba7c219e42a69127c6e9e0b7ca6a6c82" -} diff --git a/.sqlx-sqlite/query-c2e3e023f8511ff4a61f71ea3aa60f49842416e78d9a59a306eb7dd1b362e76b.json b/.sqlx-sqlite/query-c2e3e023f8511ff4a61f71ea3aa60f49842416e78d9a59a306eb7dd1b362e76b.json deleted file mode 100644 index 7c9df30f..00000000 --- a/.sqlx-sqlite/query-c2e3e023f8511ff4a61f71ea3aa60f49842416e78d9a59a306eb7dd1b362e76b.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_wiki_pages (id, kb_id, tenant_id, title, slug, status, content, summary, linked_page_ids, created_by, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11)", - "describe": { - "columns": [], - "parameters": { - "Right": 11 - }, - "nullable": [] - }, - "hash": "c2e3e023f8511ff4a61f71ea3aa60f49842416e78d9a59a306eb7dd1b362e76b" -} diff --git a/.sqlx-sqlite/query-cc1b2323157789cddf2730539d8a3483e43ef98341a9993fb5a41ad2886ec4e2.json b/.sqlx-sqlite/query-cc1b2323157789cddf2730539d8a3483e43ef98341a9993fb5a41ad2886ec4e2.json new file mode 100644 index 00000000..a09bbfc6 --- /dev/null +++ b/.sqlx-sqlite/query-cc1b2323157789cddf2730539d8a3483e43ef98341a9993fb5a41ad2886ec4e2.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_faqs (id, kb_id, standard_question, similar_questions, answers, enabled, created_by, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8)", + "describe": { + "columns": [], + "parameters": { + "Right": 8 + }, + "nullable": [] + }, + "hash": "cc1b2323157789cddf2730539d8a3483e43ef98341a9993fb5a41ad2886ec4e2" +} diff --git a/.sqlx-sqlite/query-cc5af75f667fffc113e4369c75b0010c49269abff921e35e8bb52f08a3fb2871.json b/.sqlx-sqlite/query-cc5af75f667fffc113e4369c75b0010c49269abff921e35e8bb52f08a3fb2871.json new file mode 100644 index 00000000..c48206b6 --- /dev/null +++ b/.sqlx-sqlite/query-cc5af75f667fffc113e4369c75b0010c49269abff921e35e8bb52f08a3fb2871.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_knowledge_bases (id, name, description, slug, kind, indexing_strategy, embedding_model, embedding_dim, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9)", + "describe": { + "columns": [], + "parameters": { + "Right": 9 + }, + "nullable": [] + }, + "hash": "cc5af75f667fffc113e4369c75b0010c49269abff921e35e8bb52f08a3fb2871" +} diff --git a/.sqlx-sqlite/query-d9adc734a69b50095744da9c8ed1c0bfbfb54f90659b24c8a6ec0d300043e8cd.json b/.sqlx-sqlite/query-d9adc734a69b50095744da9c8ed1c0bfbfb54f90659b24c8a6ec0d300043e8cd.json deleted file mode 100644 index 6e5ed526..00000000 --- a/.sqlx-sqlite/query-d9adc734a69b50095744da9c8ed1c0bfbfb54f90659b24c8a6ec0d300043e8cd.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_knowledge_bases (id, tenant_id, name, description, slug, kind, indexing_strategy, embedding_model, embedding_dim, created_at, tenant_id) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10, ?11)", - "describe": { - "columns": [], - "parameters": { - "Right": 11 - }, - "nullable": [] - }, - "hash": "d9adc734a69b50095744da9c8ed1c0bfbfb54f90659b24c8a6ec0d300043e8cd" -} diff --git a/.sqlx-sqlite/query-dec62a4238e2bda5b1874497a8ca4828ff692de5c6e2dbf71c615558f82c1c26.json b/.sqlx-sqlite/query-dec62a4238e2bda5b1874497a8ca4828ff692de5c6e2dbf71c615558f82c1c26.json new file mode 100644 index 00000000..d2d6f34f --- /dev/null +++ b/.sqlx-sqlite/query-dec62a4238e2bda5b1874497a8ca4828ff692de5c6e2dbf71c615558f82c1c26.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_wiki_pages (id, kb_id, title, slug, status, content, summary, linked_page_ids, created_by, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)", + "describe": { + "columns": [], + "parameters": { + "Right": 10 + }, + "nullable": [] + }, + "hash": "dec62a4238e2bda5b1874497a8ca4828ff692de5c6e2dbf71c615558f82c1c26" +} diff --git a/.sqlx-sqlite/query-fa9a097fc07d66663a39713a0bd019014092e486166af2ef907b0c7e14b6efa9.json b/.sqlx-sqlite/query-fa9a097fc07d66663a39713a0bd019014092e486166af2ef907b0c7e14b6efa9.json deleted file mode 100644 index e11ae6fd..00000000 --- a/.sqlx-sqlite/query-fa9a097fc07d66663a39713a0bd019014092e486166af2ef907b0c7e14b6efa9.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "db_name": "SQLite", - "query": "INSERT INTO kb_documents (id, kb_id, tenant_id, title, source, storage_key, mime_type, size, created_by, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9, ?10)", - "describe": { - "columns": [], - "parameters": { - "Right": 10 - }, - "nullable": [] - }, - "hash": "fa9a097fc07d66663a39713a0bd019014092e486166af2ef907b0c7e14b6efa9" -} diff --git a/.sqlx-sqlite/query-fc0f0aaa5a1b82c40861fbf502a944b27c7ed04ef10258828f866880e634b1e0.json b/.sqlx-sqlite/query-fc0f0aaa5a1b82c40861fbf502a944b27c7ed04ef10258828f866880e634b1e0.json new file mode 100644 index 00000000..8b1a68d8 --- /dev/null +++ b/.sqlx-sqlite/query-fc0f0aaa5a1b82c40861fbf502a944b27c7ed04ef10258828f866880e634b1e0.json @@ -0,0 +1,12 @@ +{ + "db_name": "SQLite", + "query": "INSERT INTO kb_documents (id, kb_id, title, source, storage_key, mime_type, size, created_by, created_at) VALUES (?1, ?2, ?3, ?4, ?5, ?6, ?7, ?8, ?9)", + "describe": { + "columns": [], + "parameters": { + "Right": 9 + }, + "nullable": [] + }, + "hash": "fc0f0aaa5a1b82c40861fbf502a944b27c7ed04ef10258828f866880e634b1e0" +} diff --git a/crates/agent/src/tool.rs b/crates/agent/src/tool.rs index f3397a54..8d051a8d 100644 --- a/crates/agent/src/tool.rs +++ b/crates/agent/src/tool.rs @@ -11,6 +11,9 @@ pub struct ToolSpec { pub name: String, pub description: String, pub parameters: Value, + /// Presentation category (e.g. "content", "files", "mcp") — used by + /// admin UIs to group the tool catalog; "other" when unset. + pub category: &'static str, } impl ToolSpec { @@ -19,6 +22,7 @@ impl ToolSpec { name: name.into(), description: description.into(), parameters, + category: "other", } } } @@ -29,6 +33,11 @@ pub trait Tool: Send + Sync { fn name(&self) -> &str; fn description(&self) -> &str; fn parameters_schema(&self) -> Value; + /// Presentation category for admin tool catalogs (grouping only; no + /// behavioral meaning). Default `"other"`. + fn category(&self) -> &'static str { + "other" + } async fn execute(&self, args: Value) -> ToolExecution; } @@ -60,7 +69,11 @@ impl ToolRegistry { pub fn specs(&self) -> Vec { self.tools .iter() - .map(|t| ToolSpec::new(t.name(), t.description(), t.parameters_schema())) + .map(|t| { + let mut spec = ToolSpec::new(t.name(), t.description(), t.parameters_schema()); + spec.category = t.category(); + spec + }) .collect() } diff --git a/crates/core/src/agent/handler.rs b/crates/core/src/agent/handler.rs index 2f95fe33..4480625a 100644 --- a/crates/core/src/agent/handler.rs +++ b/crates/core/src/agent/handler.rs @@ -32,6 +32,29 @@ pub fn routes( _config: &crate::config::app::AppConfig, ) -> axum::Router { let r = axum::Router::new(); + let r = reg_route!( + r, + registry, + _config.api_restful, + "/admin/ai/tools", + get, + admin_list_domain_tools, + "system", + "admin/ai/tools", + "admin" + ); + #[cfg(feature = "mcp")] + let r = reg_route!( + r, + registry, + _config.api_restful, + "/admin/ai/tools/mcp", + get, + admin_list_mcp_tools, + "system", + "admin/ai/tools", + "admin" + ); let r = reg_route!( r, registry, @@ -142,6 +165,17 @@ pub fn routes( "ai/sessions/compact", "authed" ); + let r = reg_route!( + r, + registry, + _config.api_restful, + "/ai/sessions/{id}", + delete, + delete_session, + "system", + "ai/sessions", + "authed" + ); let r = reg_route!( r, registry, @@ -296,6 +330,71 @@ fn default_true() -> bool { true } +/// `GET /admin/ai/tools` — the domain-tool catalog for the admin agents +/// form and the tools page. +/// +/// Built-in/conditional tools only — pure construction, zero IO, no MCP: +/// this endpoint never touches the network and answers instantly. +/// Code/env-level tool changes surface after the mandatory +/// compile/restart with no cache to invalidate. MCP tools live on +/// [`admin_list_mcp_tools`]. +/// +/// `read_skill` is registered per-turn in compact mode only, but is listed +/// here so admins can allowlist it up front. +pub async fn admin_list_domain_tools( + auth: AuthUser, + State(state): State, +) -> AppResult> { + auth.ensure_admin()?; + // Tool *specs* (name/description/category) are identical for every + // actor; the admin caller's `auth` only matters for execution, which + // never happens here. + let mut registry = crate::agent::tools::build_static_tools(&state, &auth, None).await; + // knowledge_search: listed whenever the KB subsystem is enabled + // (mounting is a per-agent binding configured separately). + crate::agent::tools::kb::register_catalog(&mut registry, &state); + // read_skill: per-turn tool (compact mode), listed for allowlisting. + registry.register(crate::agent::tools::skills::ReadSkillTool::new( + crate::agent::skills::skills_root(), + auth.tenant_id().map(str::to_string), + Vec::new(), + )); + let items: Vec = registry + .specs() + .into_iter() + .map(|s| { + json!({ + "name": s.name, + "description": s.description, + "category": s.category, + }) + }) + .collect(); + Ok(ApiResponse::success(json!({ "items": items }))) +} + +/// `GET /admin/ai/tools/mcp` — MCP tools only. This is the only catalog +/// path that talks to MCP servers; specs come from a TTL cache +/// ([`mcp::cached_catalog_specs`]) so a dead/hanging server costs at most +/// one bounded attempt per TTL, and server-side tool changes surface +/// within one TTL. +#[cfg(feature = "mcp")] +pub async fn admin_list_mcp_tools( + auth: AuthUser, + State(state): State, +) -> AppResult> { + auth.ensure_admin()?; + let items: Vec = + crate::agent::tools::mcp::cached_catalog_specs(&state.config.ai.mcp_servers) + .await + .iter() + .map(|(name, description)| { + json!({ "name": name, "description": description, "category": "mcp" }) + }) + .collect(); + Ok(ApiResponse::success(json!({ "items": items }))) +} + pub async fn admin_create_agent( auth: AuthUser, State(state): State, @@ -858,6 +957,23 @@ pub async fn compact_session( }))) } +/// `DELETE /api/v1/ai/sessions/{id}` — owner-scoped session deletion +/// (cascade: session + its messages). +pub async fn delete_session( + auth: AuthUser, + State(state): State, + Path(id): Path, +) -> AppResult> { + let owner = current_owner(&auth)?; + let id = crate::types::snowflake_id::parse_id(&id)?; + let session = ai_service::find_session(&state.pool, id, auth.tenant_id()).await?; + if session.user_id != owner { + return Err(AppError::ForbiddenOwnership); + } + ai_service::delete_session(&state.pool, auth.tenant_id(), session.id).await?; + Ok(ApiResponse::success(json!({ "deleted": true }))) +} + /// `POST /api/v1/ai/sessions/{id}/turns` — streamed SSE of one turn. pub async fn run_turn( auth: AuthUser, @@ -872,7 +988,7 @@ pub async fn run_turn( return Err(AppError::ForbiddenOwnership); } let agent = ai_service::find_agent(&state.pool, session.agent_id, auth.tenant_id()).await?; - let extra_tools = crate::agent::tools::build_domain_tools(&state, &auth).await; + let extra_tools = crate::agent::tools::build_domain_tools(&state, &auth, Some(&agent)).await; let pool = state.pool.clone(); let ai_cfg = state.config.ai.clone(); diff --git a/crates/core/src/agent/tools/files.rs b/crates/core/src/agent/tools/files.rs index 82e04512..74f74467 100644 --- a/crates/core/src/agent/tools/files.rs +++ b/crates/core/src/agent/tools/files.rs @@ -62,6 +62,10 @@ impl Tool for ManagedFileTool { &self.description } + fn category(&self) -> &'static str { + "files" + } + fn parameters_schema(&self) -> Value { serde_json::json!({ "type": "object", diff --git a/crates/core/src/agent/tools/kb.rs b/crates/core/src/agent/tools/kb.rs new file mode 100644 index 00000000..ad94a746 --- /dev/null +++ b/crates/core/src/agent/tools/kb.rs @@ -0,0 +1,570 @@ +//! Knowledge-base retrieval tool (`knowledge_search`) — the agent-side +//! consumption seam of the KB subsystem (kb-technical-design §10: +//! "Agent 工具化消费的入口"). +//! +//! Behavior mirrors WeKnora's agent knowledge_search tool +//! [抄WK:internal/agent/tools/knowledge_search.go]: +//! - 1–5 short semantic queries per call (the agent formulates them); +//! - `kb_ids` can only **narrow** the pre-bound scope, never expand it +//! (WeKnora `validateKnowledgeBaseIDsInSearchTargets` semantics); +//! - XML `` output with per-unit ids for citations; +//! - empty results return anti-fabrication guidance instead of nothing; +//! - units already returned earlier in the same turn render compactly +//! (`already_seen="true"`, content omitted) so repeat calls don't burn +//! tokens (WeKnora seenChunks). +//! +//! Retrieval goes through `kb::pipeline::search_units` — S2–S7 without S1 +//! (query rewriting is the agent's job) and without S9 generation (the +//! agent's own turn is the generator). + +use async_trait::async_trait; +use raisfast_agent::tool::ToolExecution; +use raisfast_agent::{Tool, ToolRegistry}; +use serde_json::Value; +use std::collections::{HashMap, HashSet}; +use std::sync::Mutex; + +use crate::AppState; +use crate::agent::models::ai_agent::AiAgent; +use crate::constants::DEFAULT_TENANT; +use crate::kb::pipeline::{self, ContextUnit}; +use crate::kb::service::KbDeps; +use crate::middleware::auth::AuthUser; + +/// Max queries per call [抄WK schema maxItems=5]. +const MAX_QUERIES: usize = 5; +/// Default / max kept units per call (output-size guard: agent context +/// windowing exists, but a single tool result must stay bounded). +const DEFAULT_TOP_K: usize = 5; +const MAX_TOP_K: usize = 10; +/// Per-unit content cap in chars [自造+理由: WK emits full chunk content; +/// our parent-expanded units can exceed it — cap with an explicit marker]. +const CONTENT_CAP: usize = 4000; + +/// Register the `knowledge_search` tool when the KB subsystem is enabled +/// and the agent has a valid KB scope. Scope resolution is fail-closed: +/// any error (KB disabled, invalid `params.kb_ids`, no active KBs) skips +/// registration with a warning instead of failing the turn. +/// Register the `knowledge_search` tool when the KB subsystem is enabled +/// and the agent carries an **explicit** KB binding (`params.kb_ids`). +/// Binding is select-only: no binding (or an empty/invalid list) means no +/// tool — there is deliberately no "default all tenant KBs" fallback +/// (least privilege; mounting must be an explicit per-KB choice). +pub async fn register( + registry: &mut ToolRegistry, + state: &AppState, + auth: &AuthUser, + agent: &AiAgent, +) { + if state.kb_runtime.is_none() { + return; // KB subsystem disabled — tool simply absent. + } + let Ok(deps) = state.kb_deps() else { + return; + }; + let tenant = auth.tenant_id().unwrap_or(DEFAULT_TENANT).to_string(); + let requested = match bound_kb_ids(agent) { + Ok(Some(ids)) if !ids.is_empty() => ids, + Ok(_) => return, // unbound → tool not mounted (explicit-only) + Err(e) => { + tracing::warn!( + agent = agent.id.0, + error = %e, + "knowledge_search: invalid params.kb_ids, tool skipped" + ); + return; + } + }; + let scope = match pipeline::resolve_kbs(&deps, &requested, &tenant).await { + Ok(scope) => scope, + Err(e) => { + tracing::warn!( + agent = agent.id.0, + error = %e, + "knowledge_search: bound kb ids failed tenant/active validation, tool skipped" + ); + return; + } + }; + registry.register(KnowledgeSearchTool { + deps, + tenant, + scope, + seen: Mutex::new(HashSet::new()), + }); +} + +/// Catalog registration for the admin tool listing (`GET /admin/ai/tools`): +/// exposes the `knowledge_search` spec whenever the KB subsystem is enabled. +/// Mounting is a separate per-agent binding, so no scope is bound here and +/// this instance is never executed — only its name/description are read. +pub fn register_catalog(registry: &mut ToolRegistry, state: &AppState) { + if let Ok(deps) = state.kb_deps() { + registry.register(KnowledgeSearchTool { + deps, + tenant: String::new(), + scope: Vec::new(), + seen: Mutex::new(HashSet::new()), + }); + } +} + +/// Parse `ai_agents.params.kb_ids` — `Some(ids)` = explicit binding, +/// `None` = unbound (tool stays unmounted). Accepts JSON numbers or +/// numeric strings (admin forms may submit strings); any garbage entry +/// rejects the whole list (fail-closed). +fn bound_kb_ids(agent: &AiAgent) -> Result>, String> { + let Some(params) = agent.params.as_ref() else { + return Ok(None); + }; + let Some(raw) = params.get("kb_ids") else { + return Ok(None); + }; + if raw.is_null() { + return Ok(None); + } + let Value::Array(items) = raw else { + return Err("params.kb_ids must be an array of kb ids".into()); + }; + let mut ids = Vec::with_capacity(items.len()); + for item in items { + let id = match item { + Value::Number(n) => n.as_i64(), + Value::String(s) => s.trim().parse::().ok(), + _ => None, + }; + ids.push(id.ok_or("params.kb_ids contains a non-id entry")?); + } + Ok(Some(ids)) +} + +/// The `knowledge_search` tool instance. One per turn (built by +/// `build_domain_tools`), so the seen-set dedups repeat calls within a +/// single agent turn, matching WeKnora's per-session instance semantics +/// at our turn granularity. +struct KnowledgeSearchTool { + deps: KbDeps, + tenant: String, + /// Tenant-validated KB ids bound at registration (search targets). + scope: Vec, + /// Unit ids already returned earlier this turn. + seen: Mutex>, +} + +#[async_trait] +impl Tool for KnowledgeSearchTool { + fn name(&self) -> &str { + "knowledge_search" + } + + fn description(&self) -> &str { + "Semantic search over knowledge bases. Retrieves chunks by meaning, \ +intent, and conceptual relevance.\n\ +Use for: conceptual explanations, topic overviews, how/why questions, \ +definitions, comparisons.\n\ +Do NOT use for: exact keyword or entity lookup, error-code search.\n\ +Input: 1-5 short, well-formed semantic questions (not keyword lists, not \ +raw user text). Optionally narrow the KB scope with kb_ids, and set top_k \ +(1-10, default 5).\n\ +Output: XML with ranked units (unit_id/title/score/content). \ +Cite unit_id when using retrieved content. When nothing is retrieved, state \ +that the knowledge base does not cover the question — never fabricate." + } + + fn category(&self) -> &'static str { + "kb" + } + + fn parameters_schema(&self) -> Value { + serde_json::json!({ + "type": "object", + "properties": { + "queries": { + "type": "array", + "description": "REQUIRED: 1-5 semantic questions/topics (e.g. [\"What is RAG?\", \"RAG benefits\"])", + "items": { "type": "string" }, + "minItems": 1, + "maxItems": 5 + }, + "kb_ids": { + "type": "array", + "description": "Optional: narrow the search to these knowledge-base ids (must be within the bound scope)", + "items": { "type": "integer" } + }, + "top_k": { + "type": "integer", + "description": "Max units to return (1-10, default 5)", + "minimum": 1, + "maximum": 10 + } + }, + "required": ["queries"] + }) + } + + async fn execute(&self, args: Value) -> ToolExecution { + let queries = parse_queries(&args)?; + if queries.is_empty() { + return Err("queries must contain 1-5 non-empty strings".into()); + } + + // kb_ids may only narrow the bound scope [抄WK narrowing semantics]. + let effective_scope: Vec = match args.get("kb_ids") { + None | Some(Value::Null) => self.scope.clone(), + Some(Value::Array(items)) => { + let mut narrowed = Vec::with_capacity(items.len()); + for item in items { + let Some(id) = item.as_i64() else { + return Err("kb_ids entries must be integers".into()); + }; + if !self.scope.contains(&id) { + return Err(format!( + "kb_id {id} is outside this agent's bound knowledge-base scope" + )); + } + narrowed.push(id); + } + if narrowed.is_empty() { + self.scope.clone() + } else { + narrowed + } + } + Some(_) => return Err("kb_ids must be an array of integers".into()), + }; + + let top_k = args + .get("top_k") + .and_then(Value::as_i64) + .map_or(DEFAULT_TOP_K, |k| k.clamp(1, MAX_TOP_K as i64) as usize); + + // Per-query retrieval (S2–S7, no S1 rewrite, no generation), merged + // across queries keeping each unit's best score and first source. + let mut merged: HashMap = HashMap::new(); + for q in &queries { + let (_top, units) = + pipeline::search_units(&self.deps, &self.tenant, &effective_scope, q) + .await + .map_err(|e| format!("knowledge_search failed: {e}"))?; + for u in units { + match merged.get(&u.unit_id) { + Some((best, _, _)) if *best >= u.score => {} + _ => { + merged.insert(u.unit_id, (u.score, q.clone(), u)); + } + } + } + } + let mut ranked: Vec<(f32, String, ContextUnit)> = merged.into_values().collect(); + ranked.sort_by(|a, b| b.0.partial_cmp(&a.0).unwrap_or(std::cmp::Ordering::Equal)); + ranked.truncate(top_k); + + if ranked.is_empty() { + // [抄WK formatOutput empty-result guidance] — anti-fabrication. + return Ok("No relevant content found in the knowledge base.\n\ + - DO NOT answer from your own knowledge or invent facts.\n\ + - Tell the user the knowledge base does not cover this question." + .to_string()); + } + + let mut out = String::new(); + out.push_str(&format!("\n", ranked.len())); + for q in &queries { + out.push_str(&format!("{}\n", xml_escape(q))); + } + // Mark all outgoing units as seen (compact on repeat calls). + let previously_seen: HashSet = { + let mut seen = self.seen.lock().unwrap_or_else(|p| p.into_inner()); + let prev: HashSet = ranked + .iter() + .map(|(_, _, u)| u.unit_id) + .filter(|id| seen.contains(id)) + .collect(); + for (_, _, u) in &ranked { + seen.insert(u.unit_id); + } + prev + }; + for (i, (score, source, u)) in ranked.iter().enumerate() { + let seen_attr = if previously_seen.contains(&u.unit_id) { + " already_seen=\"true\"" + } else { + "" + }; + out.push_str(&format!( + "\n", + i + 1, + u.unit_id, + xml_escape(&u.kind), + xml_escape(&u.title), + score, + xml_escape(source), + seen_attr + )); + if previously_seen.contains(&u.unit_id) { + out.push_str( + "(content omitted, already returned in an earlier knowledge_search call this turn)\n", + ); + } else { + let content = truncate_chars(&u.content, CONTENT_CAP); + out.push_str(&format!("{}\n", xml_escape(content))); + } + out.push_str("\n"); + } + out.push_str(""); + Ok(out) + } +} + +/// Parse the `queries` arg: non-empty strings, deduped, capped at MAX_QUERIES. +fn parse_queries(args: &Value) -> Result, String> { + let Some(Value::Array(items)) = args.get("queries") else { + return Err("queries (array of 1-5 strings) is required".into()); + }; + let mut queries: Vec = items + .iter() + .filter_map(|v| v.as_str().map(str::to_string)) + .map(|q| q.trim().to_string()) + .filter(|q| !q.is_empty()) + .collect(); + queries.dedup(); + queries.truncate(MAX_QUERIES); + Ok(queries) +} + +/// XML-escape attribute values and text nodes [抄WK xmlEscape]. +fn xml_escape(s: &str) -> String { + let mut out = String::with_capacity(s.len()); + for c in s.chars() { + match c { + '&' => out.push_str("&"), + '<' => out.push_str("<"), + '>' => out.push_str(">"), + '"' => out.push_str("""), + '\'' => out.push_str("'"), + _ => out.push(c), + } + } + out +} + +/// Char-boundary-safe truncation with an explicit marker. +fn truncate_chars(s: &str, cap: usize) -> &str { + if s.chars().count() <= cap { + return s; + } + let end = s.char_indices().nth(cap).map_or(s.len(), |(i, _)| i); + &s[..end] +} + +// ─────────────────────────── tests ──────────────────────────────────────── + +#[cfg(test)] +mod tests { + use super::*; + use crate::errors::app_error::AppResult; + use std::sync::Arc; + + struct MockEmbedder; + + #[async_trait::async_trait] + impl crate::kb::service::KbEmbedder for MockEmbedder { + async fn embed(&self, texts: &[&str]) -> AppResult>> { + Ok(texts + .iter() + .map(|t| { + let mut v = vec![0.0_f32; 4]; + let seed = t.bytes().map(|b| b as usize).sum::(); + v[seed % 4] = 1.0; + v + }) + .collect()) + } + } + + async fn deps() -> KbDeps { + let pool = crate::test_pool!(); + let mut config = crate::config::app::AppConfig::test_defaults(); + config.kb.enabled = true; + let bus = crate::eventbus::EventBus::new(16); + KbDeps { + pool, + config: Arc::new(config), + storage: Arc::new( + crate::storage::local::LocalStorage::new("/tmp/kb-tool-test", "/uploads").unwrap(), + ), + vector: Arc::new(crate::kb::vectors::BruteForceIndex::new()), + kbsearch: Arc::new(crate::kb::kbsearch::KbSearchEngine::open_in_memory().unwrap()), + embedder: Arc::new(MockEmbedder), + provider: None, // search_units must never need a chat provider + emitter: crate::event::EventEmitter::eventbus_only(bus), + } + } + + async fn seeded_kb(deps: &KbDeps, tenant: &str, slug: &str, body: &str) -> i64 { + let kb = crate::kb::models::knowledge_base::create_kb( + &deps.pool, + &crate::kb::models::knowledge_base::CreateKbCmd { + name: slug.into(), + description: None, + slug: slug.into(), + kind: "document".into(), + indexing_strategy: None, + embedding_model: Some("m".into()), + embedding_dim: Some(4), + }, + tenant, + ) + .await + .unwrap(); + let mut markdown = "# 文档\n\n".to_string(); + markdown.push_str(&body.repeat(60)); + let doc = + crate::kb::service::create_online_document(deps, kb.id, slug, &markdown, None, tenant) + .await + .unwrap(); + crate::kb::service::process_document(deps, doc.id, tenant) + .await + .unwrap(); + i64::from(kb.id) + } + + fn tool(deps: KbDeps, tenant: &str, scope: Vec) -> KnowledgeSearchTool { + KnowledgeSearchTool { + deps, + tenant: tenant.to_string(), + scope, + seen: Mutex::new(HashSet::new()), + } + } + + #[tokio::test] + async fn returns_xml_results() { + let deps = deps().await; + let kb_id = seeded_kb( + &deps, + "default", + "db-doc", + "raisfast 支持 SQLite PostgreSQL MySQL 数据库后端。", + ) + .await; + let t = tool(deps, "default", vec![kb_id]); + let out = t + .execute(serde_json::json!({ "queries": ["支持哪些数据库?"] })) + .await + .unwrap(); + assert!(out.contains(""), + "seen unit omits content: {second}" + ); + } + + #[test] + fn bound_kb_ids_parses_params() { + let agent = AiAgent { + id: crate::types::snowflake_id::SnowflakeId(1), + tenant_id: Some("default".into()), + user_id: None, + name: "a".into(), + system_prompt: String::new(), + provider: "openai".into(), + model: "m".into(), + temperature: None, + max_iterations: 10, + tools: serde_json::json!(["*"]), + memory_enabled: false, + params: Some(serde_json::json!({ "kb_ids": ["123", 456] })), + created_at: crate::utils::tz::now_utc(), + updated_at: crate::utils::tz::now_utc(), + }; + assert_eq!(bound_kb_ids(&agent).unwrap(), Some(vec![123, 456])); + let unbound = AiAgent { + params: None, + ..agent + }; + assert_eq!(bound_kb_ids(&unbound).unwrap(), None); + } +} diff --git a/crates/core/src/agent/tools/mcp.rs b/crates/core/src/agent/tools/mcp.rs index 85b88f52..ccbc2a1c 100644 --- a/crates/core/src/agent/tools/mcp.rs +++ b/crates/core/src/agent/tools/mcp.rs @@ -119,6 +119,9 @@ impl Tool for McpTool { fn description(&self) -> &str { &self.description } + fn category(&self) -> &'static str { + "mcp" + } fn parameters_schema(&self) -> Value { self.schema.clone() } @@ -173,6 +176,65 @@ impl Tool for McpTool { } } +/// Catalog metadata cache for `GET /admin/ai/tools`: MCP tool specs are +/// the only live-derived part of the catalog, so they are cached with a +/// TTL instead of re-handshaking on every request. On refresh failure the +/// last-known list keeps being served (stale beats blocking); the refresh +/// attempt itself is capped at [`CATALOG_ATTEMPT`] so a hanging server +/// can only ever delay one request per TTL by that much. +const CATALOG_TTL: std::time::Duration = std::time::Duration::from_secs(5 * 60); +const CATALOG_ATTEMPT: std::time::Duration = std::time::Duration::from_secs(3); + +/// `(fetched_at, specs)` pairs keyed fresh for one TTL. +type CatalogCache = Option<(std::time::Instant, Arc>)>; + +static CATALOG_SPECS: std::sync::OnceLock> = + std::sync::OnceLock::new(); + +/// Cached `(name, description)` list of all MCP tools across servers. +/// Server-side tool changes surface within one TTL; server config changes +/// require a restart (env-derived, same as the turn path). +pub async fn cached_catalog_specs(servers: &[Value]) -> Arc> { + let cell = CATALOG_SPECS.get_or_init(|| std::sync::Mutex::new(None)); + if let Some((at, specs)) = cell.lock().unwrap_or_else(|p| p.into_inner()).clone() + && at.elapsed() < CATALOG_TTL + { + return specs; + } + // Stale or absent: try one bounded refresh; on failure keep serving + // whatever we had (possibly none). + let refreshed = tokio::time::timeout(CATALOG_ATTEMPT, async { + let mut registry = raisfast_agent::ToolRegistry::new(); + register_mcp_tools(&mut registry, servers).await; + registry + .specs() + .into_iter() + .map(|s| (s.name, s.description)) + .collect::>() + }) + .await; + let mut guard = cell.lock().unwrap_or_else(|p| p.into_inner()); + match refreshed { + Ok(list) => { + let specs = Arc::new(list); + *guard = Some((std::time::Instant::now(), Arc::clone(&specs))); + specs + } + Err(_) => { + // Attempt timed out (dead/hanging server): keep the previous + // entry but bump its timestamp so we do not retry on every + // request (negative cache until the next TTL). + if let Some((_at, specs)) = guard.as_ref() { + let cloned = Arc::clone(specs); + *guard = Some((std::time::Instant::now(), Arc::clone(&cloned))); + cloned + } else { + Arc::new(Vec::new()) + } + } + } +} + /// Discover and register all tools of every configured server. pub async fn register_mcp_tools( registry: &mut raisfast_agent::ToolRegistry, diff --git a/crates/core/src/agent/tools/mod.rs b/crates/core/src/agent/tools/mod.rs index 67df1730..df928358 100644 --- a/crates/core/src/agent/tools/mod.rs +++ b/crates/core/src/agent/tools/mod.rs @@ -8,6 +8,7 @@ //! `architecture.md §3`, `prompt-engineering.md §5`). pub mod files; +pub mod kb; #[cfg(feature = "mcp")] pub mod mcp; pub mod posts; @@ -21,24 +22,49 @@ use raisfast_agent::ToolRegistry; use crate::AppState; use crate::middleware::auth::AuthUser; +use super::models::ai_agent::AiAgent; + /// Build the domain tool registry for one turn from the agent's actor. /// Every available domain tool is registered here; the per-agent allowlist -/// (`ai_agents.tools`) is applied later by `AgentService`. -pub async fn build_domain_tools(state: &AppState, auth: &AuthUser) -> ToolRegistry { - let mut registry = ToolRegistry::new(); - posts::register(&mut registry, state, auth); - system::register(&mut registry, state, auth); - script::register(&mut registry, &state.plugins); - files::register(&mut registry, state, auth); - // `run_shell` is default closed: only registered when an operator enabled - // `[ai].allow_shell` (RAISFAST_AI_ALLOW_SHELL=true), then gated per agent - // by the `tools` allowlist like every other domain tool. - if state.config.ai.allow_shell { - shell::register(&mut registry, auth); - } +/// (`ai_agents.tools`) is applied later by `AgentService`. `agent` carries +/// the turn's agent row (KB binding etc.); `None` on paths without one. +pub async fn build_domain_tools( + state: &AppState, + auth: &AuthUser, + agent: Option<&AiAgent>, +) -> ToolRegistry { + let mut registry = build_static_tools(state, auth, agent).await; #[cfg(feature = "mcp")] { mcp::register_mcp_tools(&mut registry, &state.config.ai.mcp_servers).await; } registry } + +/// The non-MCP part of the domain registry: pure construction (Arc clones, +/// path joins — zero IO), so the admin tool catalog rebuilds it fresh on +/// every request and code/env-level tool changes surface after the +/// mandatory compile/restart without any cache to invalidate. MCP tools +/// (live connections) are added by [`build_domain_tools`] for the turn +/// path, and via [`mcp::cached_catalog_specs`] for the catalog path. +pub async fn build_static_tools( + state: &AppState, + auth: &AuthUser, + agent: Option<&AiAgent>, +) -> ToolRegistry { + let mut registry = ToolRegistry::new(); + posts::register(&mut registry, state, auth); + system::register(&mut registry, state, auth); + script::register(&mut registry, &state.plugins); + files::register(&mut registry, state, auth); + if let Some(agent) = agent { + kb::register(&mut registry, state, auth, agent).await; + } + // `run_shell` is default closed: only registered when an operator enabled + // `[ai].allow_shell` (RAISFAST_AI_ALLOW_SHELL=true), then gated per agent + // by the `tools` allowlist like every other domain tool. + if state.config.ai.allow_shell { + shell::register(&mut registry, auth); + } + registry +} diff --git a/crates/core/src/agent/tools/posts.rs b/crates/core/src/agent/tools/posts.rs index 77869d4e..c1656cb0 100644 --- a/crates/core/src/agent/tools/posts.rs +++ b/crates/core/src/agent/tools/posts.rs @@ -35,6 +35,10 @@ impl Tool for SearchPostsTool { "Full-text search over blog posts. Use when the user wants to find posts by keywords." } + fn category(&self) -> &'static str { + "content" + } + fn parameters_schema(&self) -> Value { serde_json::json!({ "type": "object", @@ -111,6 +115,10 @@ impl Tool for ListPostsTool { "List blog posts (published). Supports keyword search and pagination." } + fn category(&self) -> &'static str { + "content" + } + fn parameters_schema(&self) -> Value { serde_json::json!({ "type": "object", diff --git a/crates/core/src/agent/tools/script.rs b/crates/core/src/agent/tools/script.rs index 44ef0c92..095fadd0 100644 --- a/crates/core/src/agent/tools/script.rs +++ b/crates/core/src/agent/tools/script.rs @@ -60,6 +60,10 @@ impl Tool for RunCodeTool { &self.description } + fn category(&self) -> &'static str { + "script" + } + fn parameters_schema(&self) -> Value { serde_json::json!({ "type": "object", diff --git a/crates/core/src/agent/tools/shell.rs b/crates/core/src/agent/tools/shell.rs index 14e0db4d..ff8a11e4 100644 --- a/crates/core/src/agent/tools/shell.rs +++ b/crates/core/src/agent/tools/shell.rs @@ -149,6 +149,10 @@ impl Tool for RunShellTool { Only registered when the operator enabled RAISFAST_AI_ALLOW_SHELL." } + fn category(&self) -> &'static str { + "shell" + } + fn parameters_schema(&self) -> Value { serde_json::json!({ "type": "object", diff --git a/crates/core/src/agent/tools/skills.rs b/crates/core/src/agent/tools/skills.rs index 334e132b..4e7cec71 100644 --- a/crates/core/src/agent/tools/skills.rs +++ b/crates/core/src/agent/tools/skills.rs @@ -135,6 +135,10 @@ impl Tool for ReadSkillTool { "Load the full SKILL.md instructions for an available skill by name." } + fn category(&self) -> &'static str { + "skills" + } + fn parameters_schema(&self) -> Value { serde_json::json!({ "type": "object", diff --git a/crates/core/src/agent/tools/system.rs b/crates/core/src/agent/tools/system.rs index d0853aec..b9f38bdf 100644 --- a/crates/core/src/agent/tools/system.rs +++ b/crates/core/src/agent/tools/system.rs @@ -28,6 +28,10 @@ impl Tool for TodayTool { "Return today's UTC date as YYYY-MM-DD." } + fn category(&self) -> &'static str { + "system" + } + fn parameters_schema(&self) -> Value { serde_json::json!({ "type": "object", "properties": {}, "additionalProperties": false }) } diff --git a/crates/core/src/flows/model/flow_node_run.rs b/crates/core/src/flows/model/flow_node_run.rs index 18e37b59..33745245 100644 --- a/crates/core/src/flows/model/flow_node_run.rs +++ b/crates/core/src/flows/model/flow_node_run.rs @@ -3,6 +3,7 @@ //! UI lists them under an instance. use serde::Serialize; +use serde_json::Value; use crate::db::{DbDriver, Driver}; use crate::errors::app_error::AppResult; @@ -26,12 +27,16 @@ pub struct FlowNodeRun { pub finished_at: Option, #[cfg_attr(feature = "export-types", ts(type = "number"))] pub latency_ms: Option, - pub input_summary: Option, - pub output_summary: Option, + #[cfg_attr(feature = "export-types", ts(type = "unknown"))] + pub input_summary: Option, + #[cfg_attr(feature = "export-types", ts(type = "unknown"))] + pub output_summary: Option, /// LLM usage payload ({"prompt_tokens","completion_tokens","total_tokens"}) - /// for billing (llm-node.md §6); serialized JSON text. - pub usage_json: Option, - pub error: Option, + /// for billing (llm-node.md §6). + #[cfg_attr(feature = "export-types", ts(type = "unknown"))] + pub usage_json: Option, + #[cfg_attr(feature = "export-types", ts(type = "unknown"))] + pub error: Option, pub created_at: Timestamp, } @@ -57,10 +62,10 @@ pub async fn record_node_run( node_type: &str, status: &str, attempt: i64, - input: Option<&str>, - output: Option<&str>, - error: Option<&str>, - usage: Option<&str>, + input: Option<&Value>, + output: Option<&Value>, + error: Option<&Value>, + usage: Option<&Value>, latency_ms: Option, ) -> AppResult<()> { let found: Option = sqlx::query_scalar(crate::db::safe_sql(&format!( diff --git a/crates/core/src/flows/model/flow_resume.rs b/crates/core/src/flows/model/flow_resume.rs index 955fa6b8..cbd51458 100644 --- a/crates/core/src/flows/model/flow_resume.rs +++ b/crates/core/src/flows/model/flow_resume.rs @@ -232,9 +232,17 @@ mod tests { .await .unwrap(); let rows = find_expired_open(&pool, now).await.unwrap(); + // Shared-database discipline: only assert on this test's instance + // (the global sweep legitimately surfaces overdue rows from other + // tests' instances). + let mine: Vec<_> = rows.iter().filter(|r| r.instance_id == iid).collect(); assert!( - rows.iter().all(|r| r.node_id == past), - "only the overdue row surfaces: {rows:?}" + mine.iter().any(|r| r.node_id == past), + "the overdue row must surface: {mine:?}" + ); + assert!( + mine.iter().all(|r| r.node_id != future), + "the future row must not surface: {mine:?}" ); } } diff --git a/crates/core/src/flows/run.rs b/crates/core/src/flows/run.rs index e1d67957..56d3c287 100644 --- a/crates/core/src/flows/run.rs +++ b/crates/core/src/flows/run.rs @@ -450,16 +450,6 @@ async fn record_node_runs( let Some(node) = graph.nodes.get(node_id) else { continue; }; - let error = st.error.as_ref().map(|v| { - if let Some(s) = v.as_str() { - s.to_string() - } else { - v.to_string() - } - }); - let input = st.input.as_ref().map(|v| v.to_string()); - let output = st.output.as_ref().map(|v| v.to_string()); - let usage = st.usage.as_ref().map(|v| v.to_string()); model::record_node_run( pool, instance_id, @@ -467,10 +457,10 @@ async fn record_node_runs( node.data.kind.as_str(), status, st.attempt, - input.as_deref(), - output.as_deref(), - error.as_deref(), - usage.as_deref(), + st.input.as_ref(), + st.output.as_ref(), + st.error.as_ref(), + st.usage.as_ref(), st.latency_ms, ) .await?; diff --git a/crates/core/src/kb/eval.rs b/crates/core/src/kb/eval.rs index 2d55cb7c..d388db95 100644 --- a/crates/core/src/kb/eval.rs +++ b/crates/core/src/kb/eval.rs @@ -120,10 +120,16 @@ pub fn rouge_l(reference: &str, candidate: &str) -> f64 { /// Run the dataset against a prepared KB: retrieval hit = every expected /// keyword appears in some recalled context unit. -pub async fn run_eval(deps: &KbDeps, kb_ids: &[i64], cases: &[EvalCase]) -> AppResult { +pub async fn run_eval( + deps: &KbDeps, + tenant_id: &str, + kb_ids: &[i64], + cases: &[EvalCase], +) -> AppResult { let mut reports = Vec::with_capacity(cases.len()); for case in cases { let ask = AskRequest { + tenant_id: tenant_id.to_string(), kb_ids: kb_ids.to_vec(), doc_ids: Vec::new(), question: case.question.clone(), @@ -289,6 +295,7 @@ mod tests { let report = run_eval( &deps, + "default", &[i64::from(kb.id)], &[ EvalCase { diff --git a/crates/core/src/kb/handler.rs b/crates/core/src/kb/handler.rs index 7fa7ee34..bfff9dc1 100644 --- a/crates/core/src/kb/handler.rs +++ b/crates/core/src/kb/handler.rs @@ -418,7 +418,7 @@ pub fn routes( impl AppState { /// Build the KB dependency set from app state. - fn kb_deps(&self) -> AppResult { + pub(crate) fn kb_deps(&self) -> AppResult { let kb = self .kb_runtime .as_ref() @@ -513,10 +513,14 @@ async fn admin_update_kb( auth.ensure_admin()?; state.kb_deps()?; if req.name.trim().is_empty() || req.slug.trim().is_empty() { - return Err(AppError::BadRequest("name and slug must not be empty".into())); + return Err(AppError::BadRequest( + "name and slug must not be empty".into(), + )); } if !matches!(req.status.as_str(), "active" | "archived") { - return Err(AppError::BadRequest("status must be 'active' or 'archived'".into())); + return Err(AppError::BadRequest( + "status must be 'active' or 'archived'".into(), + )); } let id = parse_snowflake(&id)?; let tenant = tenant_of(&auth); @@ -763,7 +767,9 @@ async fn admin_document_jobs( }) }) .collect(); - Ok(ApiResponse::success(json!({ "items": items, "total": items.len() }))) + Ok(ApiResponse::success( + json!({ "items": items, "total": items.len() }), + )) } async fn admin_get_document( @@ -935,8 +941,14 @@ async fn public_ask( auth.ensure_authenticated()?; let deps = state.kb_deps()?; let ask = crate::kb::pipeline::AskRequest { + tenant_id: tenant_of(&auth), kb_ids: req.kb_ids.unwrap_or_default().iter().map(|i| i.0).collect(), - doc_ids: req.doc_ids.unwrap_or_default().iter().map(|i| i.0).collect(), + doc_ids: req + .doc_ids + .unwrap_or_default() + .iter() + .map(|i| i.0) + .collect(), question: req.question, }; let user_id = auth.user_id().map(crate::types::snowflake_id::SnowflakeId); @@ -1012,7 +1024,7 @@ async fn log_ask( Some(&outcome.answer), Some(&cited), outcome.status, - Some(outcome.top_score), + Some(f64::from(outcome.top_score)), user_id, ) .await @@ -1038,6 +1050,7 @@ async fn public_search( auth.ensure_authenticated()?; let deps = state.kb_deps()?; let ask = crate::kb::pipeline::AskRequest { + tenant_id: tenant_of(&auth), kb_ids: req.kb_ids.unwrap_or_default().iter().map(|i| i.0).collect(), doc_ids: Vec::new(), question: req.query, @@ -1468,27 +1481,30 @@ async fn admin_delete_chunk( return Err(AppError::NotFound("kb_chunk".into())); }; // Children of a parent chunk are part of its content — cascade. - let children = - crate::kb::models::chunk::find_children_by_parent(&state.pool, id).await?; + let children = crate::kb::models::chunk::find_children_by_parent(&state.pool, id).await?; let mut gone_ids: Vec = children.iter().map(|c| i64::from(c.id)).collect(); gone_ids.push(i64::from(id)); - deps.vector.delete(i64::from(chunk.kb_id), &gone_ids).await?; + deps.vector + .delete(i64::from(chunk.kb_id), &gone_ids) + .await?; crate::kb::models::chunk::delete_chunks_by_ids(&state.pool, &gone_ids).await?; // FTS: rebuild the remaining units of the source facet per kind // (document → doc_id facet, faq → faq_id facet, wiki → self-keyed). if let Some(doc_id) = chunk.doc_id { - let remaining = - crate::kb::models::chunk::find_chunks_by_doc(&state.pool, doc_id).await?; + let remaining = crate::kb::models::chunk::find_chunks_by_doc(&state.pool, doc_id).await?; rebuild_fts_facet( &deps, i64::from(doc_id), - &remaining.iter() - .map(|c| (i64::from(c.id), i64::from(c.kb_id), c.kind.clone(), { - match &c.breadcrumb { - Some(b) => format!("{b}\n{}", c.content), - None => c.content.clone(), - } - })) + &remaining + .iter() + .map(|c| { + (i64::from(c.id), i64::from(c.kb_id), c.kind.clone(), { + match &c.breadcrumb { + Some(b) => format!("{b}\n{}", c.content), + None => c.content.clone(), + } + }) + }) .collect::>(), ) .await?; @@ -1502,13 +1518,20 @@ async fn admin_delete_chunk( ); } } else if let Some(faq_id) = chunk.faq_id { - let remaining = - crate::kb::models::chunk::find_chunks_by_faq(&state.pool, faq_id).await?; + let remaining = crate::kb::models::chunk::find_chunks_by_faq(&state.pool, faq_id).await?; rebuild_fts_facet( &deps, i64::from(faq_id), - &remaining.iter() - .map(|c| (i64::from(c.id), i64::from(c.kb_id), c.kind.clone(), c.content.clone())) + &remaining + .iter() + .map(|c| { + ( + i64::from(c.id), + i64::from(c.kb_id), + c.kind.clone(), + c.content.clone(), + ) + }) .collect::>(), ) .await?; @@ -1536,13 +1559,15 @@ async fn rebuild_fts_facet( } let units: Vec = remaining .iter() - .map(|(unit_id, kb_id, kind, text)| crate::kb::kbsearch::KbIndexUnit { - unit_id: *unit_id, - kb_id: *kb_id, - doc_id: facet_id, - kind: kind.clone(), - text: text.clone(), - }) + .map( + |(unit_id, kb_id, kind, text)| crate::kb::kbsearch::KbIndexUnit { + unit_id: *unit_id, + kb_id: *kb_id, + doc_id: facet_id, + kind: kind.clone(), + text: text.clone(), + }, + ) .collect(); deps.kbsearch.reindex_document(&units).await?; Ok(()) @@ -1826,14 +1851,9 @@ async fn admin_list_faqs( auth.ensure_admin()?; let page = q.page.unwrap_or(1); let page_size = q.page_size.unwrap_or(20); - let (faqs, total) = crate::kb::models::faq::list_faqs( - &state.pool, - q.kb_id, - page, - page_size, - &tenant_of(&auth), - ) - .await?; + let (faqs, total) = + crate::kb::models::faq::list_faqs(&state.pool, q.kb_id, page, page_size, &tenant_of(&auth)) + .await?; Ok(ApiResponse::success( json!({ "items": faqs, "total": total, "page": page, "page_size": page_size }), )) @@ -1918,7 +1938,10 @@ async fn admin_list_chunks( ids }; let doc_ids: Vec = { - let mut ids: Vec = chunks.iter().filter_map(|c| c.doc_id.map(i64::from)).collect(); + let mut ids: Vec = chunks + .iter() + .filter_map(|c| c.doc_id.map(i64::from)) + .collect(); ids.sort_unstable(); ids.dedup(); ids diff --git a/crates/core/src/kb/models/chunk.rs b/crates/core/src/kb/models/chunk.rs index 1a30b17d..f6e4e951 100644 --- a/crates/core/src/kb/models/chunk.rs +++ b/crates/core/src/kb/models/chunk.rs @@ -165,10 +165,9 @@ pub async fn delete_chunks_by_ids(pool: &crate::db::Pool, ids: &[i64]) -> AppRes for id in ids { query = query.bind(id); } - query - .execute(pool) - .await - .map_err(|e| crate::errors::app_error::AppError::Internal(anyhow::anyhow!(e.to_string())))?; + query.execute(pool).await.map_err(|e| { + crate::errors::app_error::AppError::Internal(anyhow::anyhow!(e.to_string())) + })?; Ok(()) } diff --git a/crates/core/src/kb/models/document.rs b/crates/core/src/kb/models/document.rs index 4a05e0c5..2fab77b0 100644 --- a/crates/core/src/kb/models/document.rs +++ b/crates/core/src/kb/models/document.rs @@ -66,7 +66,6 @@ pub async fn create_document( [ "id" => id, "kb_id" => cmd.kb_id, - "tenant_id" => tenant_id, "title" => cmd.title.as_str(), "source" => cmd.source.as_str(), "storage_key" => cmd.storage_key.as_deref(), @@ -266,9 +265,11 @@ pub async fn find_doc_titles_by_ids( for id in ids { query = query.bind(id); } - for (id, title) in query.fetch_all(pool).await.map_err(|e| { - AppError::Internal(anyhow::anyhow!(e.to_string())) - })? { + for (id, title) in query + .fetch_all(pool) + .await + .map_err(|e| AppError::Internal(anyhow::anyhow!(e.to_string())))? + { out.insert(id, title); } Ok(out) diff --git a/crates/core/src/kb/models/faq.rs b/crates/core/src/kb/models/faq.rs index 99fc6f58..3f0058a9 100644 --- a/crates/core/src/kb/models/faq.rs +++ b/crates/core/src/kb/models/faq.rs @@ -100,7 +100,6 @@ pub async fn create_faq( [ "id" => id, "kb_id" => cmd.kb_id, - "tenant_id" => tenant_id, "standard_question" => cmd.standard_question.as_str(), "similar_questions" => serde_json::json!(cmd.similar_questions), "answers" => serde_json::json!(cmd.answers), diff --git a/crates/core/src/kb/models/knowledge_base.rs b/crates/core/src/kb/models/knowledge_base.rs index 5052f7ac..d6129202 100644 --- a/crates/core/src/kb/models/knowledge_base.rs +++ b/crates/core/src/kb/models/knowledge_base.rs @@ -56,7 +56,6 @@ pub async fn create_kb( "kb_knowledge_bases", [ "id" => id, - "tenant_id" => tenant_id, "name" => cmd.name.as_str(), "description" => cmd.description.as_deref(), "slug" => cmd.slug.as_str(), @@ -121,11 +120,7 @@ pub async fn find_kb_by_id( )?) } -pub async fn delete_kb( - pool: &crate::db::Pool, - id: SnowflakeId, - tenant_id: &str, -) -> AppResult<()> { +pub async fn delete_kb(pool: &crate::db::Pool, id: SnowflakeId, tenant_id: &str) -> AppResult<()> { raisfast_derive::crud_delete!( pool, "kb_knowledge_bases", @@ -171,9 +166,11 @@ pub async fn find_kb_names_by_ids( for id in ids { query = query.bind(id); } - for (id, name) in query.fetch_all(pool).await.map_err(|e| { - AppError::Internal(anyhow::anyhow!(e.to_string())) - })? { + for (id, name) in query + .fetch_all(pool) + .await + .map_err(|e| AppError::Internal(anyhow::anyhow!(e.to_string())))? + { out.insert(id, name); } Ok(out) diff --git a/crates/core/src/kb/models/query_log.rs b/crates/core/src/kb/models/query_log.rs index 08e69a55..a6387bd8 100644 --- a/crates/core/src/kb/models/query_log.rs +++ b/crates/core/src/kb/models/query_log.rs @@ -35,7 +35,7 @@ pub async fn insert_log( answer: Option<&str>, cited_units: Option<&Value>, status: &str, - top_score: Option, + top_score: Option, user_id: Option, ) -> AppResult { let (id, now) = ( diff --git a/crates/core/src/kb/models/wiki_page.rs b/crates/core/src/kb/models/wiki_page.rs index 454becef..98f9be91 100644 --- a/crates/core/src/kb/models/wiki_page.rs +++ b/crates/core/src/kb/models/wiki_page.rs @@ -62,7 +62,6 @@ pub async fn create_page( [ "id" => id, "kb_id" => cmd.kb_id, - "tenant_id" => tenant_id, "title" => cmd.title.as_str(), "slug" => cmd.slug.as_str(), "status" => "draft", diff --git a/crates/core/src/kb/models/wiki_source.rs b/crates/core/src/kb/models/wiki_source.rs index 9c9e1085..6aac5877 100644 --- a/crates/core/src/kb/models/wiki_source.rs +++ b/crates/core/src/kb/models/wiki_source.rs @@ -39,8 +39,8 @@ pub async fn link_source( "page_id" => page_id, "page_revision" => page_revision, "doc_id" => doc_id, - "span_start" => 0, - "span_end" => 0, + "span_start" => 0_i64, + "span_end" => 0_i64, "created_at" => now ] )?; @@ -63,10 +63,7 @@ pub async fn find_page_ids_by_doc( /// Drop provenance links of a deleted document (its pages are marked /// stale separately — the links themselves must not outlive the doc). -pub async fn delete_sources_by_doc( - pool: &crate::db::Pool, - doc_id: SnowflakeId, -) -> AppResult<()> { +pub async fn delete_sources_by_doc(pool: &crate::db::Pool, doc_id: SnowflakeId) -> AppResult<()> { raisfast_derive::crud_delete!( pool, "kb_wiki_sources", @@ -77,10 +74,7 @@ pub async fn delete_sources_by_doc( /// Drop every provenance link of a KB (the table has no kb_id column; /// resolve via the KB's pages and documents before those rows vanish). -pub async fn delete_sources_by_kb( - pool: &crate::db::Pool, - kb_id: SnowflakeId, -) -> AppResult<()> { +pub async fn delete_sources_by_kb(pool: &crate::db::Pool, kb_id: SnowflakeId) -> AppResult<()> { let sql = format!( "DELETE FROM kb_wiki_sources WHERE page_id IN (SELECT id FROM kb_wiki_pages WHERE kb_id = {}) \ OR doc_id IN (SELECT id FROM kb_documents WHERE kb_id = {})", @@ -92,6 +86,8 @@ pub async fn delete_sources_by_kb( .bind(i64::from(kb_id)) .execute(pool) .await - .map_err(|e| crate::errors::app_error::AppError::Internal(anyhow::anyhow!(e.to_string())))?; + .map_err(|e| { + crate::errors::app_error::AppError::Internal(anyhow::anyhow!(e.to_string())) + })?; Ok(()) } diff --git a/crates/core/src/kb/pipeline/mod.rs b/crates/core/src/kb/pipeline/mod.rs index 573031e3..2978646b 100644 --- a/crates/core/src/kb/pipeline/mod.rs +++ b/crates/core/src/kb/pipeline/mod.rs @@ -60,7 +60,11 @@ pub struct Reference { /// The ask request (single-turn, D5). #[derive(Debug, Clone)] pub struct AskRequest { - /// KB scope; empty = all enabled KBs (D4 [抄WK:SearchTargets 语义]). + /// Calling tenant — every KB scope resolution is tenant-scoped + /// (`resolve_kbs` validates both the default and the explicit list). + pub tenant_id: String, + /// KB scope; empty = all enabled KBs of the tenant (D4 [抄WK:SearchTargets + /// 语义]). pub kb_ids: Vec, pub question: String, /// Document scope for per-doc testing; empty = whole KB scope. @@ -70,6 +74,7 @@ pub struct AskRequest { } /// Pipeline outcome consumed by the HTTP layer (stream + non-stream). +#[derive(Debug)] pub struct AskOutcome { pub status: &'static str, /// The (possibly rewritten) question — carried for S9 and query logging. @@ -88,18 +93,58 @@ pub async fn prepare_answer(deps: &KbDeps, req: &AskRequest) -> AppResult AppResult<(f32, Vec)> { + if query.trim().is_empty() { + return Err(AppError::BadRequest("query must not be empty".into())); + } + let kbs = resolve_kbs(deps, kb_ids, tenant_id).await?; + let understood = understand::UnderstoodQuery::raw(query); + recall_and_merge(deps, &kbs, &[], &understood).await +} + +/// S2–S7 over a validated KB scope: recall → fuse → top-k → hydrate → +/// wiki boost → merge. Returns `(top_score, context_units)`. +async fn recall_and_merge( + deps: &KbDeps, + kbs: &[i64], + doc_ids: &[i64], + understood: &understand::UnderstoodQuery, +) -> AppResult<(f32, Vec)> { // S2+S3+S4 recall → fuse → top-k, per KB scope. let mut candidates = Vec::new(); - for kb_id in &kbs { - let mut recalled = search::recall(deps, *kb_id, &understood).await?; + for kb_id in kbs { + let mut recalled = search::recall(deps, *kb_id, understood).await?; // Document scope (playground per-doc testing): drop units that do // not belong to the selected documents, before fusion cuts top-k. - if !req.doc_ids.is_empty() { + if !doc_ids.is_empty() { let mut ids: Vec = recalled .bm25 .iter() @@ -113,12 +158,16 @@ pub async fn prepare_answer(deps: &KbDeps, req: &AskRequest) -> AppResult AppResult AppResult> { - if !requested.is_empty() { - return Ok(requested.to_vec()); +/// Resolve the KB scope, tenant-scoped on both branches (also used by the +/// agent tool registration to pre-bind its search targets): +/// - empty request = all enabled KBs **of the tenant**; +/// - explicit ids = validated against the tenant and `status='active'`; +/// any id that fails validation rejects the request (fail-closed — +/// object-level authz on the retrieval path, mirroring the write path's +/// `ensure_kb` gate). +pub(crate) async fn resolve_kbs( + deps: &KbDeps, + requested: &[i64], + tenant_id: &str, +) -> AppResult> { + // Dedup first so the row-count comparison below is sound. + let mut requested = requested.to_vec(); + requested.sort_unstable(); + requested.dedup(); + let requested = requested.as_slice(); + if requested.is_empty() { + let sql = format!( + "SELECT {} FROM kb_knowledge_bases WHERE status = 'active' AND tenant_id = {}", + crate::db::Driver::cast_int("id"), + crate::db::Driver::ph(1) + ); + let rows: Vec = sqlx::query_scalar(crate::db::safe_sql(&sql)) + .bind(tenant_id) + .fetch_all(&deps.pool) + .await + .map_err(|e| AppError::Internal(anyhow::anyhow!(e.to_string())))?; + if rows.is_empty() { + return Err(AppError::NotFound("kb_knowledge_base".into())); + } + return Ok(rows); } + let placeholders: Vec = (1..=requested.len()).map(crate::db::Driver::ph).collect(); let sql = format!( - "SELECT {} FROM kb_knowledge_bases WHERE status = 'active'", - crate::db::Driver::cast_int("id") + "SELECT {} FROM kb_knowledge_bases WHERE id IN ({}) AND status = 'active' AND tenant_id = {}", + crate::db::Driver::cast_int("id"), + placeholders.join(", "), + crate::db::Driver::ph(requested.len() + 1) ); - let rows: Vec = sqlx::query_scalar(crate::db::safe_sql(&sql)) + let mut query = sqlx::query_scalar::<_, i64>(crate::db::safe_sql(&sql)); + for id in requested { + query = query.bind(id); + } + let rows: Vec = query + .bind(tenant_id) .fetch_all(&deps.pool) .await .map_err(|e| AppError::Internal(anyhow::anyhow!(e.to_string())))?; - if rows.is_empty() { + if rows.len() != requested.len() { + // Some requested ids are unknown, inactive, or belong to another + // tenant — do not reveal which (uniform NotFound). return Err(AppError::NotFound("kb_knowledge_base".into())); } Ok(rows) @@ -337,6 +417,7 @@ mod tests { .unwrap(); let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb.id)], doc_ids: Vec::new(), question: "支持哪些数据库?".into(), @@ -379,6 +460,7 @@ mod tests { .unwrap(); let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb.id)], doc_ids: Vec::new(), question: "量子力学的诠释有哪些".into(), @@ -420,6 +502,7 @@ mod tests { .unwrap(); let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb.id)], doc_ids: Vec::new(), question: "知识库怎么开关".into(), @@ -439,4 +522,104 @@ mod tests { ); assert!(!outcome.references.is_empty()); } + + async fn seeded_tenant_kb(deps: &KbDeps, tenant: &str) -> i64 { + let kb = crate::kb::models::knowledge_base::create_kb( + &deps.pool, + &crate::kb::models::knowledge_base::CreateKbCmd { + name: "iso".into(), + description: None, + slug: format!("iso-{tenant}"), + kind: "document".into(), + indexing_strategy: None, + embedding_model: Some("m".into()), + embedding_dim: Some(4), + }, + tenant, + ) + .await + .unwrap(); + let mut markdown = "# 隔离\n\n".to_string(); + markdown.push_str(&format!("租户 {tenant} 的私有部署步骤说明。").repeat(60)); + let doc = + crate::kb::service::create_online_document(deps, kb.id, "iso", &markdown, None, tenant) + .await + .unwrap(); + crate::kb::service::process_document(deps, doc.id, tenant) + .await + .unwrap(); + i64::from(kb.id) + } + + #[tokio::test] + async fn tenant_scope_is_isolated() { + let deps = deps().await; + let kb_a = seeded_tenant_kb(&deps, "tenant-a").await; + + // Explicit foreign kb id → rejected (object-level authz). + let ask = AskRequest { + tenant_id: "tenant-b".into(), + kb_ids: vec![kb_a], + doc_ids: Vec::new(), + question: "私有部署步骤".into(), + }; + let err = prepare_answer(&deps, &ask).await.unwrap_err(); + assert!(matches!( + err, + crate::errors::app_error::AppError::NotFound(_) + )); + + // Default scope from tenant-b → no active KBs there → NotFound. + let ask = AskRequest { + tenant_id: "tenant-b".into(), + kb_ids: Vec::new(), + doc_ids: Vec::new(), + question: "私有部署步骤".into(), + }; + let err = prepare_answer(&deps, &ask).await.unwrap_err(); + assert!(matches!( + err, + crate::errors::app_error::AppError::NotFound(_) + )); + + // Owner tenant still retrieves. + let ask = AskRequest { + tenant_id: "tenant-a".into(), + kb_ids: vec![kb_a], + doc_ids: Vec::new(), + question: "私有部署步骤".into(), + }; + let outcome = prepare_answer(&deps, &ask).await.unwrap(); + assert!(!outcome.context_units.is_empty(), "owner must retrieve"); + } + + #[tokio::test] + async fn inactive_kb_rejected() { + let deps = deps().await; + let kb_id = seeded_tenant_kb(&deps, "default").await; + crate::kb::models::knowledge_base::update_kb( + &deps.pool, + crate::types::snowflake_id::SnowflakeId(kb_id), + &crate::kb::models::knowledge_base::UpdateKbCmd { + name: "iso".into(), + description: None, + slug: "iso-default".into(), + status: "disabled".into(), + }, + "default", + ) + .await + .unwrap(); + let ask = AskRequest { + tenant_id: "default".into(), + kb_ids: vec![kb_id], + doc_ids: Vec::new(), + question: "私有部署步骤".into(), + }; + let err = prepare_answer(&deps, &ask).await.unwrap_err(); + assert!( + matches!(err, crate::errors::app_error::AppError::NotFound(_)), + "inactive kb must fail closed" + ); + } } diff --git a/crates/core/src/kb/service.rs b/crates/core/src/kb/service.rs index 9415f675..52e6e834 100644 --- a/crates/core/src/kb/service.rs +++ b/crates/core/src/kb/service.rs @@ -124,7 +124,7 @@ impl ProviderEmbedder { return Err(AppError::ServiceUnavailable(format!( "embedding: {}", last_err.unwrap_or_else(|| "unknown error".into()) - ))) + ))); } } } @@ -1026,6 +1026,7 @@ mod faq_tests { // pipeline: ask the FAQ → answered, FAQ unit pinned first let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb_id)], doc_ids: Vec::new(), question: "如何重置密码".into(), diff --git a/crates/core/src/worker/handlers/kb.rs b/crates/core/src/worker/handlers/kb.rs index 92f2ec55..93489515 100644 --- a/crates/core/src/worker/handlers/kb.rs +++ b/crates/core/src/worker/handlers/kb.rs @@ -47,9 +47,8 @@ impl JobHandler for KbProcessDocumentHandler { /// the SAME doc — one idempotent run suffices /// [抄RF:worker/handlers/search_index.rs coalesce 配对实现]. fn coalesce(&self, jobs: Vec) -> Option { - jobs.into_iter().find(|j| { - matches!(j, Job::KbProcessDocument { .. }) - }) + jobs.into_iter() + .find(|j| matches!(j, Job::KbProcessDocument { .. })) } async fn handle(&self, job: &Job) -> AppResult<()> { diff --git a/crates/core/tests/kb.rs b/crates/core/tests/kb.rs index 63bb12c2..cbf66985 100644 --- a/crates/core/tests/kb.rs +++ b/crates/core/tests/kb.rs @@ -258,14 +258,18 @@ async fn s03_delete_cleans_all_three_stores() { .unwrap(); let key = row.storage_key.clone().expect("online doc has storage key"); assert!( - tokio::fs::try_exists(format!("/tmp/kb-it-uploads/{key}")).await.unwrap(), + tokio::fs::try_exists(format!("/tmp/kb-it-uploads/{key}")) + .await + .unwrap(), "storage file must exist before delete" ); service::delete_document_everywhere(&deps, doc, "default") .await .unwrap(); assert!( - !tokio::fs::try_exists(format!("/tmp/kb-it-uploads/{key}")).await.unwrap(), + !tokio::fs::try_exists(format!("/tmp/kb-it-uploads/{key}")) + .await + .unwrap(), "storage file must be removed with the document" ); assert!( @@ -307,6 +311,7 @@ async fn s04_ask_answered_with_aligned_citations() { .await; let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb)], doc_ids: Vec::new(), question: "支持什么数据库".into(), @@ -327,6 +332,7 @@ async fn s05_uncovered_kb_never_generates() { let deps = deps().await; let kb = seed_kb(&deps, "空库").await; let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb)], doc_ids: Vec::new(), question: "任意问题".into(), @@ -359,6 +365,7 @@ async fn s06_multi_kb_isolation() { .await; let ask_a = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb_a)], doc_ids: Vec::new(), question: "pineapple 是什么".into(), @@ -373,6 +380,7 @@ async fn s06_multi_kb_isolation() { ); let ask_all = AskRequest { + tenant_id: "default".into(), kb_ids: vec![], doc_ids: Vec::new(), question: "durian 是什么".into(), @@ -405,6 +413,7 @@ async fn s07_understanding_degrades_to_raw_question() { .await; let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb)], doc_ids: Vec::new(), question: "支持什么数据库".into(), @@ -448,6 +457,7 @@ async fn s08_faq_lifecycle_and_pinned_context() { service::index_faq(&deps, &faq).await.unwrap(); let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb)], doc_ids: Vec::new(), question: "如何重置密码".into(), @@ -461,6 +471,7 @@ async fn s08_faq_lifecycle_and_pinned_context() { // 撤池:disable 后不再进入检索 service::deindex_faq(&deps, faq.id, kb).await.unwrap(); let ask2 = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb)], doc_ids: Vec::new(), question: "如何重置密码".into(), @@ -725,6 +736,7 @@ async fn s14_child_hit_expands_to_parent_content() { ingest_md(&deps, kb, "长文", &md).await; // 用子块里的原文提问 → BM25 命中子块 → 装配上下文必须是父块(远大于 384 子块上限) let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb)], doc_ids: Vec::new(), question: "PARENT_SENTINEL_甲乙丙丁".into(), @@ -781,6 +793,7 @@ async fn s15_wiki_boost_reorders_over_document() { // 双路召回后同分并列时,wiki ×1.3 必须把 wiki 单元顶到最前 let ask = AskRequest { + tenant_id: "default".into(), kb_ids: vec![i64::from(kb)], doc_ids: Vec::new(), question: "zephyr".into(), diff --git a/crates/derive/src/schema.rs b/crates/derive/src/schema.rs index 210fa4e3..373599ce 100644 --- a/crates/derive/src/schema.rs +++ b/crates/derive/src/schema.rs @@ -376,6 +376,7 @@ fn parse_column_line(line: &str) -> Option { && !rest.starts_with("LONGBLOB") && !rest.starts_with("MEDIUMBLOB") && !rest.starts_with("TINYBLOB") + && !rest.starts_with("BYTEA") && !rest.starts_with("JSON") && !rest.starts_with("BOOLEAN") && !rest.starts_with("BOOL") @@ -395,7 +396,7 @@ fn parse_column_line(line: &str) -> Option { SqlType::Integer } else if rest.starts_with("REAL") || rest.starts_with("FLOAT") || rest.starts_with("DOUBLE") { SqlType::Real - } else if rest.starts_with("BLOB") { + } else if rest.starts_with("BLOB") || rest.starts_with("BYTEA") { SqlType::Blob } else { SqlType::Text diff --git a/justfile b/justfile index 02cf016f..7644fe82 100644 --- a/justfile +++ b/justfile @@ -179,6 +179,46 @@ db-reset: db-migrate: DATABASE_URL={{db_url}} cargo run --no-default-features --features "{{features}}" -- db migrate +# Salvage a corrupted SQLite database (SQLITE_CORRUPT) via .recover — +# keeps a timestamped backup, then integrity-checks + VACUUMs the result. +# STOP the dev server first. Damaged rows/objects may be dropped; follow +# up with `just db-init` to restore missing schema objects. +# Usage: just db-recover (sqlite backend only) +db-recover: + #!/usr/bin/env bash + set -euo pipefail + DB="{{db}}" + if [ "$DB" != "sqlite" ]; then + echo ">> db-recover only applies to the sqlite backend (current: $DB)" >&2 + exit 1 + fi + DB_FILE="$(echo '{{db_url}}' | sed 's/sqlite://;s/?.*//')" + if [ ! -f "$DB_FILE" ]; then + echo ">> database file not found: $DB_FILE" >&2 + exit 1 + fi + if command -v lsof >/dev/null 2>&1; then + if [ -n "$(lsof -t "$DB_FILE" 2>/dev/null)" ]; then + echo ">> a process still holds $DB_FILE open — stop the dev server first:" >&2 + lsof "$DB_FILE" >&2 || true + exit 1 + fi + elif [ -f "$DB_FILE-wal" ]; then + echo ">> cannot verify (lsof missing) and a -wal file exists — stop the dev server first." >&2 + exit 1 + fi + STAMP="$(date +%Y%m%d%H%M%S)" + echo ">> recovering $DB_FILE ..." + sqlite3 "$DB_FILE" ".recover" | sqlite3 "${DB_FILE}.recovered" + mv "$DB_FILE" "${DB_FILE}.corrupt.${STAMP}.bak" + rm -f "$DB_FILE-shm" + mv "${DB_FILE}.recovered" "$DB_FILE" + sqlite3 "$DB_FILE" "VACUUM;" + echo ">> integrity check:" + sqlite3 "$DB_FILE" "PRAGMA integrity_check;" + echo ">> done. corrupt backup: ${DB_FILE}.corrupt.${STAMP}.bak" + echo ">> if schema objects are missing, run: just db-init" + # Backup database db-backup: DATABASE_URL={{db_url}} cargo run --no-default-features --features "{{features}}" -- db backup ./backups diff --git a/migrations/mysql/schema.mysql.sql b/migrations/mysql/schema.mysql.sql index 5a0be879..57cff159 100644 --- a/migrations/mysql/schema.mysql.sql +++ b/migrations/mysql/schema.mysql.sql @@ -1553,7 +1553,7 @@ CREATE TABLE IF NOT EXISTS kb_knowledge_bases ( indexing_strategy JSON, chunking_config JSON, embedding_model VARCHAR(100), - embedding_dim INTEGER, + embedding_dim BIGINT, status VARCHAR(20) NOT NULL DEFAULT 'active', created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, updated_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP, @@ -1573,7 +1573,7 @@ CREATE TABLE IF NOT EXISTS kb_documents ( parse_format VARCHAR(20) NOT NULL DEFAULT 'markdown', status VARCHAR(20) NOT NULL DEFAULT 'pending', error TEXT, - chunk_count INTEGER NOT NULL DEFAULT 0, + chunk_count BIGINT NOT NULL DEFAULT 0, steps JSON, created_by BIGINT, created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, @@ -1589,11 +1589,11 @@ CREATE TABLE IF NOT EXISTS kb_chunks ( wiki_page_id BIGINT NULL, kind VARCHAR(20) NOT NULL DEFAULT 'document', parent_id BIGINT NULL, - seq INTEGER NOT NULL DEFAULT 0, + seq BIGINT NOT NULL DEFAULT 0, content TEXT NOT NULL, breadcrumb VARCHAR(512), - byte_start INTEGER NOT NULL DEFAULT 0, - byte_end INTEGER NOT NULL DEFAULT 0, + byte_start BIGINT NOT NULL DEFAULT 0, + byte_end BIGINT NOT NULL DEFAULT 0, questions JSON, embedding BLOB, embedding_model VARCHAR(100), @@ -1630,8 +1630,8 @@ CREATE TABLE IF NOT EXISTS kb_wiki_sources ( page_revision BIGINT NOT NULL DEFAULT 1, doc_id BIGINT NOT NULL, chunk_id BIGINT NULL, - span_start INTEGER NOT NULL DEFAULT 0, - span_end INTEGER NOT NULL DEFAULT 0, + span_start BIGINT NOT NULL DEFAULT 0, + span_end BIGINT NOT NULL DEFAULT 0, created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, INDEX idx_kb_wiki_sources_page (page_id, page_revision), INDEX idx_kb_wiki_sources_doc (doc_id) @@ -1661,7 +1661,7 @@ CREATE TABLE IF NOT EXISTS kb_query_logs ( cited_units JSON, status VARCHAR(20) NOT NULL DEFAULT 'answered', top_score DOUBLE, - feedback INTEGER, + feedback BIGINT, user_id BIGINT, created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP, INDEX idx_kb_query_logs_status (status, created_at) diff --git a/migrations/postgres/schema.postgres.sql b/migrations/postgres/schema.postgres.sql index 2f3cb631..06e04af9 100644 --- a/migrations/postgres/schema.postgres.sql +++ b/migrations/postgres/schema.postgres.sql @@ -1620,7 +1620,7 @@ CREATE TABLE IF NOT EXISTS kb_knowledge_bases ( indexing_strategy JSONB, chunking_config JSONB, embedding_model TEXT, - embedding_dim INTEGER, + embedding_dim BIGINT, status TEXT NOT NULL DEFAULT 'active', created_at TIMESTAMPTZ(0) NOT NULL DEFAULT NOW(), updated_at TIMESTAMPTZ(0) NOT NULL DEFAULT NOW() @@ -1640,7 +1640,7 @@ CREATE TABLE IF NOT EXISTS kb_documents ( parse_format TEXT NOT NULL DEFAULT 'markdown', status TEXT NOT NULL DEFAULT 'pending', error TEXT, - chunk_count INTEGER NOT NULL DEFAULT 0, + chunk_count BIGINT NOT NULL DEFAULT 0, steps JSONB, created_by BIGINT, created_at TIMESTAMPTZ(0) NOT NULL DEFAULT NOW(), @@ -1656,11 +1656,11 @@ CREATE TABLE IF NOT EXISTS kb_chunks ( wiki_page_id BIGINT, kind TEXT NOT NULL DEFAULT 'document', parent_id BIGINT, - seq INTEGER NOT NULL DEFAULT 0, + seq BIGINT NOT NULL DEFAULT 0, content TEXT NOT NULL, breadcrumb TEXT, - byte_start INTEGER NOT NULL DEFAULT 0, - byte_end INTEGER NOT NULL DEFAULT 0, + byte_start BIGINT NOT NULL DEFAULT 0, + byte_end BIGINT NOT NULL DEFAULT 0, questions JSONB, embedding BYTEA, embedding_model TEXT, @@ -1697,8 +1697,8 @@ CREATE TABLE IF NOT EXISTS kb_wiki_sources ( page_revision BIGINT NOT NULL DEFAULT 1, doc_id BIGINT NOT NULL, chunk_id BIGINT, - span_start INTEGER NOT NULL DEFAULT 0, - span_end INTEGER NOT NULL DEFAULT 0, + span_start BIGINT NOT NULL DEFAULT 0, + span_end BIGINT NOT NULL DEFAULT 0, created_at TIMESTAMPTZ(0) NOT NULL DEFAULT NOW() ); CREATE INDEX IF NOT EXISTS idx_kb_wiki_sources_page ON kb_wiki_sources(page_id, page_revision); @@ -1728,7 +1728,7 @@ CREATE TABLE IF NOT EXISTS kb_query_logs ( cited_units JSONB, status TEXT NOT NULL DEFAULT 'answered', top_score DOUBLE PRECISION, - feedback INTEGER, + feedback BIGINT, user_id BIGINT, created_at TIMESTAMPTZ(0) NOT NULL DEFAULT NOW() ); diff --git a/migrations/sqlite/schema.sqlite.sql b/migrations/sqlite/schema.sqlite.sql index f9c516f5..17a29cec 100644 --- a/migrations/sqlite/schema.sqlite.sql +++ b/migrations/sqlite/schema.sqlite.sql @@ -1612,7 +1612,7 @@ CREATE TABLE IF NOT EXISTS kb_knowledge_bases ( indexing_strategy TEXT, chunking_config TEXT, embedding_model TEXT, - embedding_dim INTEGER, + embedding_dim BIGINT, status TEXT NOT NULL DEFAULT 'active', created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')), updated_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')) @@ -1632,7 +1632,7 @@ CREATE TABLE IF NOT EXISTS kb_documents ( parse_format TEXT NOT NULL DEFAULT 'markdown', status TEXT NOT NULL DEFAULT 'pending', error TEXT, - chunk_count INTEGER NOT NULL DEFAULT 0, + chunk_count BIGINT NOT NULL DEFAULT 0, steps TEXT, created_by BIGINT, created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')), @@ -1648,11 +1648,11 @@ CREATE TABLE IF NOT EXISTS kb_chunks ( wiki_page_id INTEGER, kind TEXT NOT NULL DEFAULT 'document', parent_id INTEGER, - seq INTEGER NOT NULL DEFAULT 0, + seq BIGINT NOT NULL DEFAULT 0, content TEXT NOT NULL, breadcrumb TEXT, - byte_start INTEGER NOT NULL DEFAULT 0, - byte_end INTEGER NOT NULL DEFAULT 0, + byte_start BIGINT NOT NULL DEFAULT 0, + byte_end BIGINT NOT NULL DEFAULT 0, questions TEXT, embedding BLOB, embedding_model TEXT, @@ -1689,8 +1689,8 @@ CREATE TABLE IF NOT EXISTS kb_wiki_sources ( page_revision INTEGER NOT NULL DEFAULT 1, doc_id INTEGER NOT NULL, chunk_id INTEGER, - span_start INTEGER NOT NULL DEFAULT 0, - span_end INTEGER NOT NULL DEFAULT 0, + span_start BIGINT NOT NULL DEFAULT 0, + span_end BIGINT NOT NULL DEFAULT 0, created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')) ); CREATE INDEX IF NOT EXISTS idx_kb_wiki_sources_page ON kb_wiki_sources(page_id, page_revision); @@ -1720,7 +1720,7 @@ CREATE TABLE IF NOT EXISTS kb_query_logs ( cited_units TEXT, status TEXT NOT NULL DEFAULT 'answered', top_score REAL, - feedback INTEGER, + feedback BIGINT, user_id BIGINT, created_at TEXT NOT NULL DEFAULT (strftime('%Y-%m-%dT%H:%M:%SZ','now')) );